@amaster.ai/pi-lark 0.1.8 → 0.1.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -3
- package/package.json +2 -2
- package/skills/lark-apps/SKILL.md +3 -1
- package/skills/lark-apps/references/lark-apps-cache.md +38 -5
- package/skills/lark-apps/references/lark-apps-db.md +130 -2
- package/skills/lark-apps/references/lark-apps-user-id-convert.md +63 -0
- package/skills/lark-base/SKILL.md +172 -167
- package/skills/lark-base/references/{lark-base-role-guide.md → lark-base-advanced-permission-and-role.md} +5 -5
- package/skills/lark-base/references/lark-base-app-block-data-config.md +122 -0
- package/skills/lark-base/references/lark-base-app.md +243 -0
- package/skills/lark-base/references/lark-base-cell-value.md +26 -19
- package/skills/lark-base/references/{dashboard-block-data-config.md → lark-base-dashboard-block-config.md} +65 -6
- package/skills/lark-base/references/lark-base-dashboard-block-get-data.md +1 -1
- package/skills/lark-base/references/lark-base-dashboard.md +38 -20
- package/skills/lark-base/references/lark-base-data-analysis-pandas.md +93 -0
- package/skills/lark-base/references/lark-base-data-analysis-python-stdlib.md +120 -0
- package/skills/lark-base/references/lark-base-data-query.md +8 -11
- package/skills/lark-base/references/lark-base-field-create.md +7 -50
- package/skills/lark-base/references/{formula-field-guide.md → lark-base-field-formula.md} +1 -1
- package/skills/lark-base/references/{lookup-field-guide.md → lark-base-field-lookup.md} +1 -1
- package/skills/lark-base/references/{lark-base-field-json.md → lark-base-field-schema.md} +25 -101
- package/skills/lark-base/references/lark-base-field-update.md +13 -51
- package/skills/lark-base/references/lark-base-filter-condition.md +19 -31
- package/skills/lark-base/references/lark-base-form-questions-create.md +36 -5
- package/skills/lark-base/references/lark-base-record-batch-create.md +5 -1
- package/skills/lark-base/references/lark-base-record-batch-update.md +5 -2
- package/skills/lark-base/references/lark-base-record-history-list.md +19 -2
- package/skills/lark-base/references/lark-base-record-query-and-analysis-cloud-sop.md +145 -0
- package/skills/lark-base/references/lark-base-record-query-and-analysis-sop.md +233 -0
- package/skills/lark-base/references/{role-config.md → lark-base-role-config.md} +2 -2
- package/skills/lark-base/references/lark-base-template-center.md +195 -0
- package/skills/lark-base/references/lark-base-view-set-filter.md +1 -1
- package/skills/lark-base/references/lark-base-workflow-schema.md +2 -2
- package/skills/lark-base/references/{lark-base-workflow-guide.md → lark-base-workflow.md} +1 -1
- package/skills/lark-calendar/SKILL.md +11 -6
- package/skills/lark-calendar/references/lark-calendar-create.md +4 -3
- package/skills/lark-calendar/references/lark-calendar-transfer.md +89 -0
- package/skills/lark-doc/SKILL.md +3 -3
- package/skills/lark-doc/references/lark-doc-fetch.md +9 -4
- package/skills/lark-doc/references/lark-doc-update.md +12 -8
- package/skills/lark-drive/SKILL.md +5 -3
- package/skills/lark-drive/references/lark-drive-add-comment.md +2 -2
- package/skills/lark-drive/references/lark-drive-download.md +27 -1
- package/skills/lark-drive/references/lark-drive-export.md +1 -0
- package/skills/lark-drive/references/lark-drive-member-remove.md +59 -0
- package/skills/lark-drive/references/lark-drive-preview.md +21 -2
- package/skills/lark-drive/references/lark-drive-push.md +5 -1
- package/skills/lark-drive/references/lark-drive-search.md +2 -0
- package/skills/lark-im/SKILL.md +14 -3
- package/skills/lark-im/references/lark-im-message-read-status.md +96 -0
- package/skills/lark-mail/references/lark-mail-draft-create.md +12 -12
- package/skills/lark-mail/references/lark-mail-forward.md +17 -17
- package/skills/lark-mail/references/lark-mail-reply-all.md +8 -8
- package/skills/lark-mail/references/lark-mail-reply.md +6 -6
- package/skills/lark-mail/references/lark-mail-send.md +20 -20
- package/skills/lark-mail/references/lark-mail-template-create.md +7 -6
- package/skills/lark-mail/references/lark-mail-template-update.md +7 -6
- package/skills/lark-meeting/SKILL.md +146 -0
- package/skills/lark-meeting/references/lark-minutes-apply-permission.md +92 -0
- package/skills/lark-meeting/references/lark-minutes-detail.md +52 -0
- package/skills/{lark-minutes → lark-meeting}/references/lark-minutes-download.md +7 -7
- package/skills/{lark-minutes → lark-meeting}/references/lark-minutes-search.md +4 -34
- package/skills/{lark-minutes → lark-meeting}/references/lark-minutes-speaker-replace.md +3 -4
- package/skills/{lark-minutes → lark-meeting}/references/lark-minutes-summary.md +2 -5
- package/skills/{lark-minutes → lark-meeting}/references/lark-minutes-todo.md +5 -15
- package/skills/{lark-minutes → lark-meeting}/references/lark-minutes-update.md +2 -3
- package/skills/lark-meeting/references/lark-minutes-upload.md +65 -0
- package/skills/lark-meeting/references/lark-note-detail.md +15 -0
- package/skills/{lark-note → lark-meeting}/references/lark-note-transcript.md +5 -9
- package/skills/{lark-vc-agent → lark-meeting}/references/lark-vc-agent-meeting-join.md +4 -55
- package/skills/{lark-vc-agent → lark-meeting}/references/lark-vc-agent-meeting-leave.md +2 -41
- package/skills/lark-meeting/references/lark-vc-detail.md +31 -0
- package/skills/{lark-vc-agent/references/lark-vc-agent-meeting-events.md → lark-meeting/references/lark-vc-meeting-events.md} +120 -109
- package/skills/{lark-vc-agent/references/lark-vc-agent-meeting-list-active.md → lark-meeting/references/lark-vc-meeting-list-active.md} +4 -29
- package/skills/{lark-vc-agent/references/lark-vc-agent-meeting-message-send.md → lark-meeting/references/lark-vc-meeting-message-send.md} +3 -5
- package/skills/{lark-vc → lark-meeting}/references/lark-vc-recording.md +5 -64
- package/skills/{lark-vc → lark-meeting}/references/lark-vc-search.md +9 -28
- package/skills/lark-meeting/scenes/create-and-edit-minutes.md +125 -0
- package/skills/lark-meeting/scenes/live-meeting-attend.md +107 -0
- package/skills/lark-meeting/scenes/live-meeting-interact.md +72 -0
- package/skills/lark-meeting/scenes/query-meeting-and-artifacts.md +90 -0
- package/skills/lark-meeting/scenes/query-minutes-and-artifacts.md +70 -0
- package/skills/lark-meeting/scenes/query-note-and-artifacts.md +127 -0
- package/skills/lark-minutes/SKILL.md +5 -197
- package/skills/lark-note/SKILL.md +5 -84
- package/skills/lark-shared/SKILL.md +25 -188
- package/skills/lark-shared/references/lark-shared-config-init.md +12 -0
- package/skills/lark-shared/references/lark-shared-high-risk-approval.md +38 -0
- package/skills/lark-shared/references/lark-shared-identity-and-permissions.md +105 -0
- package/skills/lark-shared/references/lark-shared-output-contract.md +17 -0
- package/skills/lark-shared/references/lark-shared-update-notice.md +23 -0
- package/skills/lark-slides/SKILL.md +56 -54
- package/skills/lark-slides/references/cli/lark-slides-add-slide.md +92 -0
- package/skills/lark-slides/references/cli/lark-slides-create.md +176 -0
- package/skills/lark-slides/references/cli/lark-slides-delete-slide.md +65 -0
- package/skills/lark-slides/references/cli/lark-slides-history.md +132 -0
- package/skills/lark-slides/references/cli/lark-slides-media-upload.md +103 -0
- package/skills/lark-slides/references/cli/lark-slides-replace-slide.md +259 -0
- package/skills/lark-slides/references/cli/lark-slides-screenshot.md +115 -0
- package/skills/lark-slides/references/{lark-slides-update-slide.md → cli/lark-slides-update-slide.md} +21 -4
- package/skills/lark-slides/references/cli/lark-slides-xml-presentation-slide-get.md +110 -0
- package/skills/lark-slides/references/cli/lark-slides-xml-presentation-slide-replace.md +188 -0
- package/skills/lark-slides/references/cli/lark-slides-xml-presentations-get.md +157 -0
- package/skills/lark-slides/references/iconpark-index.json +5 -41901
- package/skills/lark-slides/references/iconpark.md +3 -44
- package/skills/lark-slides/references/lark-slides-add-slide.md +3 -90
- package/skills/lark-slides/references/lark-slides-create.md +3 -174
- package/skills/lark-slides/references/lark-slides-delete-slide.md +3 -63
- package/skills/lark-slides/references/lark-slides-edit-workflows.md +3 -141
- package/skills/lark-slides/references/lark-slides-history.md +3 -130
- package/skills/lark-slides/references/lark-slides-media-upload.md +3 -102
- package/skills/lark-slides/references/lark-slides-pptx-template-workflows.md +3 -83
- package/skills/lark-slides/references/lark-slides-replace-slide.md +3 -256
- package/skills/lark-slides/references/lark-slides-screenshot.md +3 -113
- package/skills/lark-slides/references/lark-slides-xml-presentation-slide-get.md +3 -108
- package/skills/lark-slides/references/lark-slides-xml-presentation-slide-replace.md +3 -186
- package/skills/lark-slides/references/lark-slides-xml-presentations-get.md +3 -155
- package/skills/lark-slides/references/planning-layer.md +1 -1
- package/skills/lark-slides/references/slides_chart_demo.xml +5 -1415
- package/skills/lark-slides/references/slides_xml_schema_definition.xml +3 -3512
- package/skills/lark-slides/references/troubleshooting.md +3 -60
- package/skills/lark-slides/references/validation-checklist.md +3 -154
- package/skills/lark-slides/references/workflow/error-handling.md +62 -0
- package/skills/lark-slides/references/workflow/slides-editing.md +143 -0
- package/skills/lark-slides/references/workflow/template-editing.md +85 -0
- package/skills/lark-slides/references/workflow/validation-xml.md +156 -0
- package/skills/lark-slides/references/xml/iconpark-index.json +37458 -0
- package/skills/lark-slides/references/xml/iconpark.md +46 -0
- package/skills/lark-slides/references/xml/slides_chart_demo.xml +1415 -0
- package/skills/lark-slides/references/xml/slides_xml_schema_definition.xml +3601 -0
- package/skills/lark-slides/references/xml/xml-schema-quick-ref.md +497 -0
- package/skills/lark-slides/references/xml-schema-quick-ref.md +3 -495
- package/skills/lark-slides/scripts/iconpark_tool.py +1 -1
- package/skills/lark-slides/scripts/xml_lint.py +2989 -0
- package/skills/lark-slides/scripts/xml_lint_test.py +4720 -0
- package/skills/lark-slides/scripts/xml_text_overlap_lint.py +3 -2975
- package/skills/lark-slides/scripts/xml_text_overlap_lint_test.py +5 -4712
- package/skills/lark-task/SKILL.md +13 -1
- package/skills/lark-task/references/lark-task-create.md +3 -1
- package/skills/lark-vc/SKILL.md +5 -195
- package/skills/lark-vc-agent/SKILL.md +5 -191
- package/skills/lark-wiki/SKILL.md +3 -1
- package/skills/lark-wiki/references/lark-wiki-node-copy.md +5 -19
- package/skills/lark-wiki/references/lark-wiki-node-create.md +19 -2
- package/skills/lark-wiki/references/lark-wiki-node-get.md +15 -0
- package/skills/lark-wiki/references/lark-wiki-node-list.md +1 -1
- package/skills/lark-workflow-meeting-summary/SKILL.md +20 -13
- package/skills/lark-base/references/lark-base-data-analysis-sop.md +0 -210
- package/skills/lark-base/references/lark-base-data-query-guide.md +0 -69
- package/skills/lark-base/references/lark-base-record-upsert.md +0 -63
- package/skills/lark-minutes/references/lark-minutes-detail.md +0 -62
- package/skills/lark-minutes/references/lark-minutes-upload.md +0 -104
- package/skills/lark-note/references/lark-note-detail.md +0 -26
- package/skills/lark-vc/references/lark-vc-detail.md +0 -44
- package/skills/lark-vc/references/vc-domain-boundaries.md +0 -196
|
@@ -1,2984 +1,12 @@
|
|
|
1
1
|
#!/usr/bin/env python3
|
|
2
2
|
# Copyright (c) 2026 Lark Technologies Pte. Ltd.
|
|
3
3
|
# SPDX-License-Identifier: MIT
|
|
4
|
-
"""
|
|
4
|
+
"""Compatibility entry point for the renamed xml_lint module."""
|
|
5
5
|
|
|
6
|
-
from __future__ import annotations
|
|
7
|
-
|
|
8
|
-
import copy
|
|
9
|
-
import json
|
|
10
|
-
import math
|
|
11
|
-
import re
|
|
12
6
|
import sys
|
|
13
|
-
import unicodedata
|
|
14
|
-
import xml.parsers.expat as expat
|
|
15
|
-
import xml.etree.ElementTree as ET
|
|
16
|
-
from difflib import SequenceMatcher, get_close_matches
|
|
17
|
-
from pathlib import Path
|
|
18
|
-
from typing import Any
|
|
19
|
-
|
|
20
|
-
import sxsd_validator
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
XS_NS = "{http://www.w3.org/2001/XMLSchema}"
|
|
24
|
-
XML_NS = "{http://www.w3.org/XML/1998/namespace}"
|
|
25
|
-
SVG_NS = "{http://www.w3.org/2000/svg}"
|
|
26
|
-
SML_NAMESPACE = "https://www.larkoffice.com/sml/2.0"
|
|
27
|
-
SXSD_SCHEMA_PATH = Path(__file__).resolve().parents[1] / "references" / "slides_xml_schema_definition.xml"
|
|
28
|
-
ICONPARK_INDEX_PATH = Path(__file__).resolve().parents[1] / "references" / "iconpark-index.json"
|
|
29
|
-
SXSD_TAG_ALIASES = {
|
|
30
|
-
"textbox": "<shape type=\"text\">",
|
|
31
|
-
"textBox": "<shape type=\"text\">",
|
|
32
|
-
"image": "<img>",
|
|
33
|
-
"picture": "<img>",
|
|
34
|
-
}
|
|
35
|
-
SXSD_ATTR_ALIASES = {
|
|
36
|
-
"x": "topLeftX",
|
|
37
|
-
"left": "topLeftX",
|
|
38
|
-
"y": "topLeftY",
|
|
39
|
-
"top": "topLeftY",
|
|
40
|
-
"w": "width",
|
|
41
|
-
"h": "height",
|
|
42
|
-
"fontColor": "color",
|
|
43
|
-
}
|
|
44
|
-
SERVER_FILLED_SXSD_ATTRS = {"id"}
|
|
45
|
-
ROUNDTRIP_SXSD_ATTRS = {
|
|
46
|
-
("chart", "updated"),
|
|
47
|
-
("chartData", "isStaticData"),
|
|
48
|
-
}
|
|
49
|
-
# Slides readback echoes each chartField's CSV text as per-value <chartParsedValues> children;
|
|
50
|
-
# it is server-emitted and absent from the write schema, so it must not block page linting.
|
|
51
|
-
ROUNDTRIP_SXSD_TAGS = {("chartField", "chartParsedValues")}
|
|
52
|
-
DEFAULT_TABLE_COLUMN_WIDTH = 110
|
|
53
|
-
DEFAULT_TABLE_ROW_HEIGHT = 37
|
|
54
|
-
DEFAULT_TEXT_LINE_SPACING_MULTIPLE = 1.5
|
|
55
|
-
TEXT_WRAP_WIDTH_TOLERANCE_PX = 1.0
|
|
56
|
-
TEXT_HEIGHT_OVERFLOW_TOLERANCE_PX = 0.5
|
|
57
|
-
SINGLE_LINE_METRIC_WIDTH_RATIO = 1.18
|
|
58
|
-
CENTERED_SHORT_LABEL_WIDTH_RATIO = 1.12
|
|
59
|
-
HEADLINE_NEAR_FIT_WIDTH_RATIO = 1.04
|
|
60
|
-
DENSE_BODY_LINE_SPACING_MAX_MULTIPLE = 1.6
|
|
61
|
-
GHOST_TEXT_MIN_FONT_SIZE = 96
|
|
62
|
-
GHOST_TEXT_MAX_ALPHA = 0.5
|
|
63
|
-
GHOST_TEXT_FAINT_MIN_FONT_SIZE = 36
|
|
64
|
-
GHOST_TEXT_FAINT_MAX_ALPHA = 0.35
|
|
65
|
-
# A <line> crossing text glyphs is a legibility defect (see line_crosses_text_glyphs). We erode the
|
|
66
|
-
# glyph box by this margin before testing intersection so a line that only skims a glyph edge or the
|
|
67
|
-
# padding-only text frame -- but does not actually cut through the letterforms -- is not flagged.
|
|
68
|
-
LINE_TEXT_GRAZE_MIN_PX = 2.0
|
|
69
|
-
LINE_TEXT_GRAZE_FONT_RATIO = 0.12
|
|
70
|
-
# A line whose effective stroke alpha is below this is not visibly rendered, so it cannot occlude text.
|
|
71
|
-
LINE_MIN_VISIBLE_ALPHA = 0.08
|
|
72
|
-
# Sub-pixel canvas overflow is floating-point rounding noise (e.g. rotated-bbox math), not a
|
|
73
|
-
# visible defect; keep this well under 1px so real overflow is still always caught.
|
|
74
|
-
CANVAS_OVERFLOW_TOLERANCE = 0.5
|
|
75
|
-
XML_PATH_HINT_PREFIX = "Locate via related_objects[].xml_path."
|
|
76
|
-
_SXSD_TAG_ATTRIBUTES_CACHE: dict[str, set[str]] | None = None
|
|
77
|
-
_ICONPARK_ICON_TYPES_CACHE: set[str] | None = None
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
class XmlLayoutLintError(Exception):
|
|
81
|
-
pass
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
def fail(message: str) -> None:
|
|
85
|
-
raise XmlLayoutLintError(message)
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
def read_file(file_path: str | Path) -> str:
|
|
89
|
-
return Path(file_path).read_text(encoding="utf-8")
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
def parse_args(argv: list[str]) -> dict[str, Any]:
|
|
93
|
-
options: dict[str, Any] = {}
|
|
94
|
-
index = 0
|
|
95
|
-
while index < len(argv):
|
|
96
|
-
token = argv[index]
|
|
97
|
-
if not token.startswith("--"):
|
|
98
|
-
fail(f"unexpected argument: {token}, need --input")
|
|
99
|
-
key = token[2:]
|
|
100
|
-
next_token = argv[index + 1] if index + 1 < len(argv) else None
|
|
101
|
-
if next_token is None or next_token.startswith("--"):
|
|
102
|
-
options[key] = True
|
|
103
|
-
index += 1
|
|
104
|
-
continue
|
|
105
|
-
options[key] = next_token
|
|
106
|
-
index += 2
|
|
107
|
-
return options
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
def extract_attribute(tag_source: str, name: str) -> str | None:
|
|
111
|
-
match = re.search(
|
|
112
|
-
fr"(?:^|\s){re.escape(name)}\s*=\s*(?:\"([^\"]+)\"|'([^']+)')", tag_source
|
|
113
|
-
)
|
|
114
|
-
if not match:
|
|
115
|
-
return None
|
|
116
|
-
return match.group(1) if match.group(1) is not None else match.group(2)
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
def extract_numeric_attribute(tag_source: str, name: str) -> int | float | None:
|
|
120
|
-
raw = extract_attribute(tag_source, name)
|
|
121
|
-
if raw is None:
|
|
122
|
-
return None
|
|
123
|
-
try:
|
|
124
|
-
value = float(raw)
|
|
125
|
-
except ValueError:
|
|
126
|
-
return None
|
|
127
|
-
return int(value) if value.is_integer() else value
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
def extract_bool_attribute(tag_source: str, name: str) -> bool:
|
|
131
|
-
value = extract_attribute(tag_source, name)
|
|
132
|
-
return value in {"true", "1", "yes"}
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
def extract_color_alpha(color: str | None) -> int | float | None:
|
|
136
|
-
if color is None:
|
|
137
|
-
return None
|
|
138
|
-
normalized = re.sub(r"\s+", "", color).lower()
|
|
139
|
-
if normalized == "transparent":
|
|
140
|
-
return 0
|
|
141
|
-
rgba_match = re.fullmatch(
|
|
142
|
-
r"rgba\([^,]+,[^,]+,[^,]+,([+-]?(?:[0-9]+(?:\.[0-9]*)?|\.[0-9]+))\)",
|
|
143
|
-
normalized,
|
|
144
|
-
)
|
|
145
|
-
if rgba_match is None:
|
|
146
|
-
return None
|
|
147
|
-
try:
|
|
148
|
-
alpha = float(rgba_match.group(1))
|
|
149
|
-
except ValueError:
|
|
150
|
-
return None
|
|
151
|
-
return int(alpha) if alpha.is_integer() else alpha
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
def effective_text_alpha(shape_alpha: int | float | None, text_color: str | None) -> int | float:
|
|
155
|
-
base_alpha = shape_alpha if isinstance(shape_alpha, (int, float)) else 1
|
|
156
|
-
color_alpha = extract_color_alpha(text_color)
|
|
157
|
-
if not isinstance(color_alpha, (int, float)):
|
|
158
|
-
return base_alpha
|
|
159
|
-
return base_alpha * color_alpha
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
def detect_inline_style_presence(content_xml: str, style_tags: set[str]) -> bool:
|
|
163
|
-
for tag_name in style_tags:
|
|
164
|
-
if re.search(fr"<{re.escape(tag_name)}\b[\s>]", content_xml) is not None:
|
|
165
|
-
return True
|
|
166
|
-
return False
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
def detect_any_span_bool_attribute(content_xml: str, attr_name: str) -> bool:
|
|
170
|
-
for attrs in re.findall(r"<span\b([^>]*)>", content_xml):
|
|
171
|
-
if extract_bool_attribute(attrs, attr_name):
|
|
172
|
-
return True
|
|
173
|
-
return False
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
def sum_sizes(sizes: list[int | float]) -> int | float:
|
|
177
|
-
return sum(sizes)
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
def is_filled_size(size: int | float | None) -> bool:
|
|
181
|
-
return isinstance(size, (int, float)) and math.isfinite(size) and size > 0
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
def fill_last_size_gap(sizes: list[int | float], target_size: int | float) -> list[int | float]:
|
|
185
|
-
if not sizes:
|
|
186
|
-
return sizes
|
|
187
|
-
final_sizes = [
|
|
188
|
-
size if index == len(sizes) - 1 else max(1, math.floor(size + 0.5))
|
|
189
|
-
for index, size in enumerate(sizes)
|
|
190
|
-
]
|
|
191
|
-
remaining_size = target_size - sum_sizes(final_sizes[:-1])
|
|
192
|
-
if remaining_size >= 1:
|
|
193
|
-
final_sizes[-1] = remaining_size
|
|
194
|
-
return final_sizes
|
|
195
|
-
|
|
196
|
-
size_to_redistribute = 1 - remaining_size
|
|
197
|
-
for index in range(len(final_sizes) - 2, -1, -1):
|
|
198
|
-
reduction = min(final_sizes[index] - 1, size_to_redistribute)
|
|
199
|
-
final_sizes[index] -= reduction
|
|
200
|
-
size_to_redistribute -= reduction
|
|
201
|
-
if size_to_redistribute == 0:
|
|
202
|
-
final_sizes[-1] = 1
|
|
203
|
-
return final_sizes
|
|
204
|
-
|
|
205
|
-
final_sizes[-1] = 1
|
|
206
|
-
return final_sizes
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
def solve_weighted_min_layout(
|
|
210
|
-
input_sizes: list[int | float | None], default_size: int | float, target_min_size: int | float | None
|
|
211
|
-
) -> dict[str, Any]:
|
|
212
|
-
filled_indexes: list[int] = []
|
|
213
|
-
empty_indexes: list[int] = []
|
|
214
|
-
base_sizes: list[int | float] = []
|
|
215
|
-
for index, size in enumerate(input_sizes):
|
|
216
|
-
if is_filled_size(size):
|
|
217
|
-
filled_indexes.append(index)
|
|
218
|
-
base_sizes.append(size)
|
|
219
|
-
else:
|
|
220
|
-
empty_indexes.append(index)
|
|
221
|
-
base_sizes.append(0)
|
|
222
|
-
filled_sum = sum_sizes(base_sizes)
|
|
223
|
-
|
|
224
|
-
if target_min_size is None:
|
|
225
|
-
final_sizes = [default_size if index in empty_indexes else size for index, size in enumerate(base_sizes)]
|
|
226
|
-
return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": 1}
|
|
227
|
-
|
|
228
|
-
if not filled_indexes:
|
|
229
|
-
average_size = target_min_size / len(input_sizes)
|
|
230
|
-
final_sizes = fill_last_size_gap([average_size] * len(input_sizes), target_min_size)
|
|
231
|
-
return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": 1}
|
|
232
|
-
|
|
233
|
-
if empty_indexes:
|
|
234
|
-
remaining_size = target_min_size - filled_sum
|
|
235
|
-
final_sizes = [*base_sizes]
|
|
236
|
-
if remaining_size > 0:
|
|
237
|
-
average_size = remaining_size / len(empty_indexes)
|
|
238
|
-
empty_sizes = fill_last_size_gap([average_size] * len(empty_indexes), remaining_size)
|
|
239
|
-
for index, empty_size in zip(empty_indexes, empty_sizes):
|
|
240
|
-
final_sizes[index] = empty_size
|
|
241
|
-
else:
|
|
242
|
-
for index in empty_indexes:
|
|
243
|
-
final_sizes[index] = default_size
|
|
244
|
-
return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": 1}
|
|
245
|
-
|
|
246
|
-
ratio = max(1, target_min_size / filled_sum)
|
|
247
|
-
actual_size = max(target_min_size, filled_sum)
|
|
248
|
-
if ratio == 1:
|
|
249
|
-
return {"final_sizes": [*base_sizes], "actual_size": actual_size, "ratio": ratio}
|
|
250
|
-
final_sizes = fill_last_size_gap([size * ratio for size in base_sizes], actual_size)
|
|
251
|
-
return {"final_sizes": final_sizes, "actual_size": sum_sizes(final_sizes), "ratio": ratio}
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
def strip_xml(value: str, preserve_line_breaks: bool = False) -> str:
|
|
255
|
-
stripped = re.sub(r"<!\[CDATA\[([\s\S]*?)\]\]>", r"\1", value)
|
|
256
|
-
if preserve_line_breaks:
|
|
257
|
-
stripped = re.sub(r"<br\b[^>]*>", "\n", stripped)
|
|
258
|
-
stripped = re.sub(r"<[^>]+>", " ", stripped)
|
|
259
|
-
stripped = stripped.replace(" ", " ")
|
|
260
|
-
stripped = stripped.replace("&", "&")
|
|
261
|
-
stripped = stripped.replace("<", "<")
|
|
262
|
-
stripped = stripped.replace(">", ">")
|
|
263
|
-
stripped = stripped.replace(""", '"')
|
|
264
|
-
stripped = stripped.replace("'", "'")
|
|
265
|
-
if preserve_line_breaks:
|
|
266
|
-
return "\n".join(re.sub(r"\s+", " ", line).strip() for line in stripped.split("\n"))
|
|
267
|
-
return re.sub(r"\s+", " ", stripped).strip()
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
def strip_xml_paragraphs(value: str) -> str:
|
|
271
|
-
paragraphs = re.findall(r"<p\b[^>]*>([\s\S]*?)</p\s*>", value)
|
|
272
|
-
if paragraphs:
|
|
273
|
-
return "\n".join(strip_xml(paragraph, preserve_line_breaks=True) for paragraph in paragraphs)
|
|
274
|
-
return strip_xml(value, preserve_line_breaks=True)
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
def extract_text_paragraphs(value: str, default_font_size: int | float) -> list[dict[str, Any]]:
|
|
278
|
-
paragraphs = []
|
|
279
|
-
for attrs, body in re.findall(r"<p\b([^>]*)>([\s\S]*?)</p\s*>", value):
|
|
280
|
-
paragraphs.append(
|
|
281
|
-
{
|
|
282
|
-
"text": strip_xml(body, preserve_line_breaks=True),
|
|
283
|
-
"fontSize": extract_max_span_font_size(body, default_font_size),
|
|
284
|
-
"textAlign": extract_attribute(attrs, "textAlign"),
|
|
285
|
-
"lineSpacing": extract_attribute(attrs, "lineSpacing"),
|
|
286
|
-
"beforeLineSpacing": extract_attribute(attrs, "beforeLineSpacing"),
|
|
287
|
-
"afterLineSpacing": extract_attribute(attrs, "afterLineSpacing"),
|
|
288
|
-
"letterSpacing": extract_numeric_attribute(attrs, "letterSpacing"),
|
|
289
|
-
}
|
|
290
|
-
)
|
|
291
|
-
return paragraphs
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
def extract_max_span_font_size(value: str, default_font_size: int | float) -> int | float:
|
|
295
|
-
font_sizes = [
|
|
296
|
-
font_size
|
|
297
|
-
for attrs in re.findall(r"<span\b([^>]*)>", value)
|
|
298
|
-
if (font_size := extract_numeric_attribute(attrs, "fontSize")) is not None
|
|
299
|
-
]
|
|
300
|
-
return max([default_font_size, *font_sizes])
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
def extract_tag_attributes(value: str, tag: str) -> str:
|
|
304
|
-
match = re.search(fr"<{re.escape(tag)}\b([^>]*)>", value)
|
|
305
|
-
return match.group(1) if match else ""
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
def xml_local_name(tag: str) -> str:
|
|
309
|
-
return tag.rsplit("}", 1)[-1] if tag.startswith("{") else tag
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
def xml_namespace(tag: str) -> str | None:
|
|
313
|
-
return tag.split("}", 1)[0] + "}" if tag.startswith("{") else None
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
def load_sxsd_tag_attributes() -> dict[str, set[str]]:
|
|
317
|
-
global _SXSD_TAG_ATTRIBUTES_CACHE
|
|
318
|
-
if _SXSD_TAG_ATTRIBUTES_CACHE is not None:
|
|
319
|
-
return _SXSD_TAG_ATTRIBUTES_CACHE
|
|
320
|
-
|
|
321
|
-
_SXSD_TAG_ATTRIBUTES_CACHE = sxsd_validator.load_tag_attributes(SXSD_SCHEMA_PATH)
|
|
322
|
-
return _SXSD_TAG_ATTRIBUTES_CACHE
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
def load_iconpark_icon_types() -> set[str]:
|
|
326
|
-
global _ICONPARK_ICON_TYPES_CACHE
|
|
327
|
-
if _ICONPARK_ICON_TYPES_CACHE is not None:
|
|
328
|
-
return _ICONPARK_ICON_TYPES_CACHE
|
|
329
|
-
|
|
330
|
-
try:
|
|
331
|
-
index_data = json.loads(ICONPARK_INDEX_PATH.read_text(encoding="utf-8"))
|
|
332
|
-
except json.JSONDecodeError as error:
|
|
333
|
-
fail(f"invalid iconpark index JSON: {error}")
|
|
334
|
-
icons = index_data.get("icons")
|
|
335
|
-
if not isinstance(icons, list):
|
|
336
|
-
fail("iconpark index must contain an icons array")
|
|
337
|
-
|
|
338
|
-
icon_types = {
|
|
339
|
-
icon["iconType"]
|
|
340
|
-
for icon in icons
|
|
341
|
-
if isinstance(icon, dict) and isinstance(icon.get("iconType"), str) and icon["iconType"]
|
|
342
|
-
}
|
|
343
|
-
_ICONPARK_ICON_TYPES_CACHE = icon_types
|
|
344
|
-
return icon_types
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
def build_sxsd_tag_hint(tag_name: str, supported_tags: set[str]) -> str:
|
|
348
|
-
alias = SXSD_TAG_ALIASES.get(tag_name)
|
|
349
|
-
if alias:
|
|
350
|
-
return f"Use {alias} instead of <{tag_name}>."
|
|
351
|
-
if tag_name == "svg":
|
|
352
|
-
return 'Inside <embed> or <whiteboard>, write SVG as <svg xmlns="http://www.w3.org/2000/svg">...</svg>.'
|
|
353
|
-
close_matches = get_close_matches(tag_name, sorted(supported_tags), n=3, cutoff=0.72)
|
|
354
|
-
if close_matches:
|
|
355
|
-
return "Unsupported SXSD tag. Did you mean " + ", ".join(f"<{match}>" for match in close_matches) + "?"
|
|
356
|
-
return "Unsupported SXSD tag. Use only tags defined in slides_xml_schema_definition.xml."
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
def suggest_sxsd_attrs(attr_name: str, allowed_attrs: set[str]) -> list[str]:
|
|
360
|
-
alias = SXSD_ATTR_ALIASES.get(attr_name)
|
|
361
|
-
if alias and alias in allowed_attrs:
|
|
362
|
-
return [alias]
|
|
363
|
-
return get_close_matches(attr_name, sorted(allowed_attrs), n=3, cutoff=0.68)
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
def build_sxsd_attr_hint(tag_name: str, attr_name: str, allowed_attrs: set[str]) -> str:
|
|
367
|
-
suggestions = suggest_sxsd_attrs(attr_name, allowed_attrs)
|
|
368
|
-
if suggestions:
|
|
369
|
-
if SXSD_ATTR_ALIASES.get(attr_name) == suggestions[0]:
|
|
370
|
-
return f'Use "{suggestions[0]}" on <{tag_name}> instead of "{attr_name}".'
|
|
371
|
-
return "Unsupported SXSD attribute. Did you mean " + ", ".join(f'"{match}"' for match in suggestions) + "?"
|
|
372
|
-
allowed_summary = ", ".join(sorted(allowed_attrs)[:8])
|
|
373
|
-
if len(allowed_attrs) > 8:
|
|
374
|
-
allowed_summary += ", ..."
|
|
375
|
-
return f"Unsupported SXSD attribute for <{tag_name}>. Allowed attributes include: {allowed_summary}."
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
def should_skip_sxsd_subtree(element: ET.Element, ancestors: list[str]) -> bool:
|
|
379
|
-
return ("whiteboard" in ancestors or "embed" in ancestors) and xml_namespace(element.tag) == SVG_NS
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
def should_skip_sxsd_attribute(tag_name: str, attr_name: str) -> bool:
|
|
383
|
-
return attr_name in SERVER_FILLED_SXSD_ATTRS or (tag_name, attr_name) in ROUNDTRIP_SXSD_ATTRS
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
def should_skip_sxsd_tag(parent_name: str | None, tag_name: str) -> bool:
|
|
387
|
-
return (parent_name, tag_name) in ROUNDTRIP_SXSD_TAGS
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
def without_server_filled_sxsd_fields(root: ET.Element) -> ET.Element:
|
|
391
|
-
sanitized_root = copy.deepcopy(root)
|
|
392
|
-
|
|
393
|
-
def sanitize(element: ET.Element) -> None:
|
|
394
|
-
tag_name = xml_local_name(element.tag)
|
|
395
|
-
for raw_attr_name in list(element.attrib):
|
|
396
|
-
if should_skip_sxsd_attribute(tag_name, xml_local_name(raw_attr_name)):
|
|
397
|
-
del element.attrib[raw_attr_name]
|
|
398
|
-
for child in list(element):
|
|
399
|
-
if should_skip_sxsd_tag(tag_name, xml_local_name(child.tag)):
|
|
400
|
-
element.remove(child)
|
|
401
|
-
continue
|
|
402
|
-
sanitize(child)
|
|
403
|
-
|
|
404
|
-
sanitize(sanitized_root)
|
|
405
|
-
return sanitized_root
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
def validate_sxsd_document(xml: str, root: ET.Element) -> list[dict[str, Any]]:
|
|
409
|
-
tag_attributes = load_sxsd_tag_attributes()
|
|
410
|
-
supported_tags = set(tag_attributes)
|
|
411
|
-
issues: list[dict[str, Any]] = []
|
|
412
|
-
suggested_attr_candidates: dict[tuple[str, str], list[set[str]]] = {}
|
|
413
|
-
|
|
414
|
-
def visit(element: ET.Element, ancestors: list[str], path: str) -> None:
|
|
415
|
-
if should_skip_sxsd_subtree(element, ancestors):
|
|
416
|
-
return
|
|
417
|
-
|
|
418
|
-
tag_name = xml_local_name(element.tag)
|
|
419
|
-
current_path = f"{path}/{tag_name}" if path else tag_name
|
|
420
|
-
parent_name = ancestors[-1] if ancestors else None
|
|
421
|
-
if should_skip_sxsd_tag(parent_name, tag_name):
|
|
422
|
-
return
|
|
423
|
-
if tag_name not in supported_tags:
|
|
424
|
-
issues.append(
|
|
425
|
-
{
|
|
426
|
-
"level": "error",
|
|
427
|
-
"code": "sxsd_unsupported_tag",
|
|
428
|
-
"tag": tag_name,
|
|
429
|
-
"path": current_path,
|
|
430
|
-
"message": f"unsupported SXSD tag <{tag_name}> at {current_path}",
|
|
431
|
-
"hint": build_sxsd_tag_hint(tag_name, supported_tags),
|
|
432
|
-
}
|
|
433
|
-
)
|
|
434
|
-
return
|
|
435
|
-
else:
|
|
436
|
-
allowed_attrs = tag_attributes[tag_name]
|
|
437
|
-
for raw_attr_name in element.attrib:
|
|
438
|
-
if raw_attr_name.startswith(XML_NS):
|
|
439
|
-
continue
|
|
440
|
-
attr_name = xml_local_name(raw_attr_name)
|
|
441
|
-
if should_skip_sxsd_attribute(tag_name, attr_name):
|
|
442
|
-
continue
|
|
443
|
-
if attr_name in allowed_attrs:
|
|
444
|
-
continue
|
|
445
|
-
suggestions = suggest_sxsd_attrs(attr_name, allowed_attrs)
|
|
446
|
-
if suggestions:
|
|
447
|
-
suggested_attr_candidates.setdefault((current_path, tag_name), []).append(
|
|
448
|
-
set(suggestions)
|
|
449
|
-
)
|
|
450
|
-
issues.append(
|
|
451
|
-
{
|
|
452
|
-
"level": "error",
|
|
453
|
-
"code": "sxsd_unsupported_attr",
|
|
454
|
-
"tag": tag_name,
|
|
455
|
-
"attr": attr_name,
|
|
456
|
-
"path": current_path,
|
|
457
|
-
"message": f'unsupported SXSD attribute "{attr_name}" on <{tag_name}> at {current_path}',
|
|
458
|
-
"hint": build_sxsd_attr_hint(tag_name, attr_name, allowed_attrs),
|
|
459
|
-
}
|
|
460
|
-
)
|
|
461
|
-
|
|
462
|
-
for child in element:
|
|
463
|
-
visit(child, [*ancestors, tag_name], current_path)
|
|
464
|
-
|
|
465
|
-
visit(root, [], "")
|
|
466
|
-
existing = {
|
|
467
|
-
(issue.get("code"), issue.get("path"), issue.get("tag"), issue.get("attr"))
|
|
468
|
-
for issue in issues
|
|
469
|
-
}
|
|
470
|
-
unsupported_tag_locations = {
|
|
471
|
-
(issue.get("path"), issue.get("tag"))
|
|
472
|
-
for issue in issues
|
|
473
|
-
if issue.get("code") == "sxsd_unsupported_tag"
|
|
474
|
-
}
|
|
475
|
-
schema_issues = _validate_sxsd_schema_constraints(xml, root)
|
|
476
|
-
missing_attrs_by_location: dict[tuple[str, str], set[str]] = {}
|
|
477
|
-
for schema_issue in schema_issues:
|
|
478
|
-
if schema_issue.get("code") != "sxsd_missing_required_attr":
|
|
479
|
-
continue
|
|
480
|
-
location = (schema_issue.get("path"), schema_issue.get("tag"))
|
|
481
|
-
missing_attrs_by_location.setdefault(location, set()).add(schema_issue.get("attr"))
|
|
482
|
-
|
|
483
|
-
suggested_attrs: set[tuple[str, str, str]] = set()
|
|
484
|
-
for location, candidate_groups in suggested_attr_candidates.items():
|
|
485
|
-
missing_attrs = missing_attrs_by_location.get(location, set())
|
|
486
|
-
for candidates in candidate_groups:
|
|
487
|
-
matching_missing_attrs = candidates & missing_attrs
|
|
488
|
-
if len(matching_missing_attrs) == 1:
|
|
489
|
-
suggested_attrs.add((*location, next(iter(matching_missing_attrs))))
|
|
490
|
-
|
|
491
|
-
for schema_issue in schema_issues:
|
|
492
|
-
if schema_issue.get("code") == "sxsd_unexpected_child" and (
|
|
493
|
-
schema_issue.get("path"),
|
|
494
|
-
schema_issue.get("tag"),
|
|
495
|
-
) in unsupported_tag_locations:
|
|
496
|
-
continue
|
|
497
|
-
if schema_issue.get("code") == "sxsd_missing_required_attr" and (
|
|
498
|
-
schema_issue.get("path"),
|
|
499
|
-
schema_issue.get("tag"),
|
|
500
|
-
schema_issue.get("attr"),
|
|
501
|
-
) in suggested_attrs:
|
|
502
|
-
continue
|
|
503
|
-
key = (
|
|
504
|
-
schema_issue.get("code"),
|
|
505
|
-
schema_issue.get("path"),
|
|
506
|
-
schema_issue.get("tag"),
|
|
507
|
-
schema_issue.get("attr"),
|
|
508
|
-
)
|
|
509
|
-
if key not in existing:
|
|
510
|
-
issues.append(schema_issue)
|
|
511
|
-
return issues
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
def _validate_sxsd_schema_constraints(xml: str, root: ET.Element) -> list[dict[str, Any]]:
|
|
515
|
-
issues: list[dict[str, Any]] = []
|
|
516
|
-
if re.match(r"^\s*<\?xml\b", xml):
|
|
517
|
-
issues.append(
|
|
518
|
-
{
|
|
519
|
-
"level": "error",
|
|
520
|
-
"code": "sxsd_unsupported_declaration",
|
|
521
|
-
"path": xml_local_name(root.tag),
|
|
522
|
-
"tag": xml_local_name(root.tag),
|
|
523
|
-
"expected": "SXSD document without an XML declaration",
|
|
524
|
-
"actual": "<?xml ...?>",
|
|
525
|
-
"message": "XML declarations are not supported by the Slides SXSD write format",
|
|
526
|
-
"hint": "Remove the <?xml ...?> declaration and keep the SXSD root element.",
|
|
527
|
-
}
|
|
528
|
-
)
|
|
529
|
-
|
|
530
|
-
issues.extend(
|
|
531
|
-
sxsd_validator.validate_sxsd(
|
|
532
|
-
without_server_filled_sxsd_fields(root),
|
|
533
|
-
SXSD_SCHEMA_PATH,
|
|
534
|
-
)
|
|
535
|
-
)
|
|
536
|
-
issues.extend(validate_embed_svg_roots(root))
|
|
537
|
-
return issues
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
def validate_embed_svg_roots(root: ET.Element) -> list[dict[str, Any]]:
|
|
541
|
-
issues: list[dict[str, Any]] = []
|
|
542
|
-
document_namespace = sxsd_validator.element_namespace(root.tag)
|
|
543
|
-
is_bare_slide_fragment = (
|
|
544
|
-
xml_local_name(root.tag) == "slide" and document_namespace is None
|
|
545
|
-
)
|
|
546
|
-
if (
|
|
547
|
-
document_namespace not in sxsd_validator.ACCEPTED_SML_NAMESPACES
|
|
548
|
-
and not is_bare_slide_fragment
|
|
549
|
-
):
|
|
550
|
-
return issues
|
|
551
|
-
|
|
552
|
-
def visit(element: ET.Element, ancestors: list[str], parent_path: str) -> None:
|
|
553
|
-
if should_skip_sxsd_subtree(element, ancestors):
|
|
554
|
-
return
|
|
555
|
-
|
|
556
|
-
tag_name = xml_local_name(element.tag)
|
|
557
|
-
path = f"{parent_path}/{tag_name}" if parent_path else tag_name
|
|
558
|
-
if (
|
|
559
|
-
tag_name == "embed"
|
|
560
|
-
and sxsd_validator.element_namespace(element.tag) == document_namespace
|
|
561
|
-
):
|
|
562
|
-
for child in element:
|
|
563
|
-
if xml_namespace(child.tag) != SVG_NS or xml_local_name(child.tag) == "svg":
|
|
564
|
-
continue
|
|
565
|
-
child_name = xml_local_name(child.tag)
|
|
566
|
-
child_path = f"{path}/{child_name}"
|
|
567
|
-
issues.append(
|
|
568
|
-
{
|
|
569
|
-
"level": "error",
|
|
570
|
-
"code": "sxsd_unexpected_child",
|
|
571
|
-
"path": child_path,
|
|
572
|
-
"tag": child_name,
|
|
573
|
-
"expected": '<svg xmlns="http://www.w3.org/2000/svg">',
|
|
574
|
-
"actual": child_name,
|
|
575
|
-
"message": f"embedded SVG content must use an <svg> root at {child_path}",
|
|
576
|
-
"hint": 'Wrap the SVG content in <svg xmlns="http://www.w3.org/2000/svg">...</svg>.',
|
|
577
|
-
}
|
|
578
|
-
)
|
|
579
|
-
for child in element:
|
|
580
|
-
visit(child, [*ancestors, tag_name], path)
|
|
581
|
-
|
|
582
|
-
visit(root, [], "")
|
|
583
|
-
return issues
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
def build_iconpark_icon_type_hint(icon_type: str, supported_icon_types: set[str]) -> str:
|
|
587
|
-
close_matches = get_close_matches(icon_type, sorted(supported_icon_types), n=3, cutoff=0.58)
|
|
588
|
-
if close_matches:
|
|
589
|
-
return (
|
|
590
|
-
"iconType must exist in iconpark-index.json. Did you mean "
|
|
591
|
-
+ ", ".join(f'"{match}"' for match in close_matches)
|
|
592
|
-
+ "?"
|
|
593
|
-
)
|
|
594
|
-
return "iconType must exist in iconpark-index.json. Use scripts/iconpark_tool.py to search supported icons."
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
def validate_iconpark_icon_types(root: ET.Element) -> list[dict[str, Any]]:
|
|
598
|
-
supported_icon_types: set[str] | None = None
|
|
599
|
-
issues: list[dict[str, Any]] = []
|
|
600
|
-
|
|
601
|
-
def direct_child(element: ET.Element, local_name: str) -> ET.Element | None:
|
|
602
|
-
return next((child for child in element if xml_local_name(child.tag) == local_name), None)
|
|
603
|
-
|
|
604
|
-
def is_transparent_color(color: str) -> bool:
|
|
605
|
-
normalized = re.sub(r"\s+", "", color).lower()
|
|
606
|
-
if normalized == "transparent":
|
|
607
|
-
return True
|
|
608
|
-
rgba_match = re.fullmatch(r"rgba\([^,]+,[^,]+,[^,]+,([0-9.]+)\)", normalized)
|
|
609
|
-
if not rgba_match:
|
|
610
|
-
return False
|
|
611
|
-
try:
|
|
612
|
-
return float(rgba_match.group(1)) <= 0
|
|
613
|
-
except ValueError:
|
|
614
|
-
return False
|
|
615
|
-
|
|
616
|
-
def append_missing_fill_color_issue(current_path: str) -> None:
|
|
617
|
-
issues.append(
|
|
618
|
-
{
|
|
619
|
-
"level": "error",
|
|
620
|
-
"code": "icon_missing_fill_color",
|
|
621
|
-
"tag": "icon",
|
|
622
|
-
"path": current_path,
|
|
623
|
-
"message": f"<icon> must set explicit non-transparent fillColor for visual visibility at {current_path}",
|
|
624
|
-
"hint": 'Add <fill><fillColor color="rgba(R, G, B, 1)"/></fill> inside <icon>. This is a visual lint rule, not an SXSD required field.',
|
|
625
|
-
}
|
|
626
|
-
)
|
|
627
|
-
|
|
628
|
-
def visit(element: ET.Element, ancestors: list[str], path: str) -> None:
|
|
629
|
-
nonlocal supported_icon_types
|
|
630
|
-
if should_skip_sxsd_subtree(element, ancestors):
|
|
631
|
-
return
|
|
632
|
-
|
|
633
|
-
tag_name = xml_local_name(element.tag)
|
|
634
|
-
current_path = f"{path}/{tag_name}" if path else tag_name
|
|
635
|
-
if tag_name == "icon":
|
|
636
|
-
icon_type = element.attrib.get("iconType")
|
|
637
|
-
if icon_type is not None:
|
|
638
|
-
if supported_icon_types is None:
|
|
639
|
-
supported_icon_types = load_iconpark_icon_types()
|
|
640
|
-
if icon_type not in supported_icon_types:
|
|
641
|
-
issues.append(
|
|
642
|
-
{
|
|
643
|
-
"level": "error",
|
|
644
|
-
"code": "iconpark_unsupported_icon_type",
|
|
645
|
-
"tag": "icon",
|
|
646
|
-
"attr": "iconType",
|
|
647
|
-
"iconType": icon_type,
|
|
648
|
-
"path": current_path,
|
|
649
|
-
"message": f'unsupported iconpark iconType "{icon_type}" at {current_path}',
|
|
650
|
-
"hint": build_iconpark_icon_type_hint(icon_type, supported_icon_types),
|
|
651
|
-
}
|
|
652
|
-
)
|
|
653
|
-
fill = direct_child(element, "fill")
|
|
654
|
-
fill_color = direct_child(fill, "fillColor") if fill is not None else None
|
|
655
|
-
color = fill_color.attrib.get("color") if fill_color is not None else None
|
|
656
|
-
if not color:
|
|
657
|
-
append_missing_fill_color_issue(current_path)
|
|
658
|
-
elif is_transparent_color(color):
|
|
659
|
-
issues.append(
|
|
660
|
-
{
|
|
661
|
-
"level": "error",
|
|
662
|
-
"code": "icon_transparent_fill_color",
|
|
663
|
-
"tag": "icon",
|
|
664
|
-
"attr": "fillColor",
|
|
665
|
-
"path": current_path,
|
|
666
|
-
"color": color,
|
|
667
|
-
"message": f'<icon> fillColor must not be transparent for visual visibility at {current_path}: "{color}"',
|
|
668
|
-
"hint": 'Use an opaque visible color, for example <fillColor color="rgba(37, 99, 235, 1)"/>.',
|
|
669
|
-
}
|
|
670
|
-
)
|
|
671
|
-
for child in element:
|
|
672
|
-
visit(child, [*ancestors, tag_name], current_path)
|
|
673
|
-
|
|
674
|
-
visit(root, [], "")
|
|
675
|
-
return issues
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
def extract_error_context(xml: str, line: int | None, column: int | None, radius: int = 40) -> str | None:
|
|
679
|
-
if line is None or column is None:
|
|
680
|
-
return None
|
|
681
|
-
lines = xml.splitlines()
|
|
682
|
-
if line < 1 or line > len(lines):
|
|
683
|
-
return None
|
|
684
|
-
source_line = lines[line - 1]
|
|
685
|
-
start = max(column - radius, 0)
|
|
686
|
-
end = min(column + radius, len(source_line))
|
|
687
|
-
return source_line[start:end].strip()
|
|
688
|
-
|
|
689
|
-
|
|
690
|
-
def build_xml_error_issue(error: ET.ParseError, xml: str) -> dict[str, Any]:
|
|
691
|
-
line, column = getattr(error, "position", (None, None))
|
|
692
|
-
return {
|
|
693
|
-
"level": "error",
|
|
694
|
-
"code": "xml_not_well_formed",
|
|
695
|
-
"message": f"XML is not well-formed: {error}",
|
|
696
|
-
"line": line,
|
|
697
|
-
"column": column,
|
|
698
|
-
"context": extract_error_context(xml, line, column),
|
|
699
|
-
"hint": (
|
|
700
|
-
"Escape raw user text before placing it in XML. In text nodes and attribute values, bare & must be "
|
|
701
|
-
"written as &. In text nodes, write < as < and > as >. For attribute URLs, use a=1&b=2."
|
|
702
|
-
),
|
|
703
|
-
}
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
def validate_sml_tag_prefixes(xml: str) -> list[dict[str, Any]]:
|
|
707
|
-
namespace_map: dict[str, str] = {}
|
|
708
|
-
pending_declarations: list[tuple[str, str | None]] = []
|
|
709
|
-
declarations_by_element: list[list[tuple[str, str | None]]] = []
|
|
710
|
-
element_stack: list[str] = []
|
|
711
|
-
issues: list[dict[str, Any]] = []
|
|
712
|
-
|
|
713
|
-
parser = expat.ParserCreate(namespace_separator="|")
|
|
714
|
-
parser.namespace_prefixes = True
|
|
715
|
-
|
|
716
|
-
def handle_namespace_decl(prefix: str | None, namespace: str) -> None:
|
|
717
|
-
normalized_prefix = prefix or ""
|
|
718
|
-
previous_namespace = namespace_map.get(normalized_prefix)
|
|
719
|
-
namespace_map[normalized_prefix] = namespace
|
|
720
|
-
pending_declarations.append((normalized_prefix, previous_namespace))
|
|
721
|
-
|
|
722
|
-
def handle_start_element(name: str, _attrs: dict[str, str]) -> None:
|
|
723
|
-
declarations_by_element.append(pending_declarations.copy())
|
|
724
|
-
pending_declarations.clear()
|
|
725
|
-
name_parts = name.rsplit("|", 2)
|
|
726
|
-
if len(name_parts) == 3:
|
|
727
|
-
_namespace, local_name, prefix = name_parts
|
|
728
|
-
element_name = f"{prefix}:{local_name}"
|
|
729
|
-
else:
|
|
730
|
-
prefix = ""
|
|
731
|
-
local_name = name_parts[-1]
|
|
732
|
-
element_name = local_name
|
|
733
|
-
element_stack.append(element_name)
|
|
734
|
-
if not prefix:
|
|
735
|
-
return
|
|
736
|
-
|
|
737
|
-
actual_namespace = namespace_map.get(prefix)
|
|
738
|
-
if actual_namespace not in sxsd_validator.ACCEPTED_SML_NAMESPACES:
|
|
739
|
-
return
|
|
740
|
-
path = "/".join(element_stack)
|
|
741
|
-
issues.append(
|
|
742
|
-
{
|
|
743
|
-
"level": "error",
|
|
744
|
-
"code": "sml_prefixed_tag",
|
|
745
|
-
"tag": element_name,
|
|
746
|
-
"namespace": actual_namespace,
|
|
747
|
-
"path": path,
|
|
748
|
-
"line": parser.CurrentLineNumber,
|
|
749
|
-
"column": parser.CurrentColumnNumber,
|
|
750
|
-
"message": f"SML tag <{element_name}> must not use a namespace prefix at {path}",
|
|
751
|
-
"hint": (
|
|
752
|
-
f'Use <{local_name}> under the default namespace '
|
|
753
|
-
f'<{local_name} xmlns="{SML_NAMESPACE}">, or use an unprefixed SML tag.'
|
|
754
|
-
),
|
|
755
|
-
}
|
|
756
|
-
)
|
|
757
|
-
|
|
758
|
-
def handle_end_element(_name: str) -> None:
|
|
759
|
-
for prefix, previous_namespace in reversed(declarations_by_element.pop()):
|
|
760
|
-
if previous_namespace is None:
|
|
761
|
-
namespace_map.pop(prefix, None)
|
|
762
|
-
else:
|
|
763
|
-
namespace_map[prefix] = previous_namespace
|
|
764
|
-
element_stack.pop()
|
|
765
|
-
|
|
766
|
-
parser.StartNamespaceDeclHandler = handle_namespace_decl
|
|
767
|
-
parser.StartElementHandler = handle_start_element
|
|
768
|
-
parser.EndElementHandler = handle_end_element
|
|
769
|
-
parser.Parse(xml, True)
|
|
770
|
-
return issues
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
def parse_xml_root(xml: str) -> tuple[ET.Element | None, dict[str, Any] | None]:
|
|
774
|
-
try:
|
|
775
|
-
root = ET.fromstring(xml)
|
|
776
|
-
except ET.ParseError as error:
|
|
777
|
-
return None, build_xml_error_issue(error, xml)
|
|
778
|
-
|
|
779
|
-
root_name = xml_local_name(root.tag)
|
|
780
|
-
if root_name not in {"presentation", "slide"}:
|
|
781
|
-
fail("input must contain a <presentation> or <slide> root")
|
|
782
|
-
return root, None
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
def validate_xml_well_formed(xml: str) -> dict[str, Any] | None:
|
|
786
|
-
_, xml_error = parse_xml_root(xml)
|
|
787
|
-
return xml_error
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
def serialize_slide_for_layout(slide_root: ET.Element) -> str:
|
|
791
|
-
slide_copy = copy.deepcopy(slide_root)
|
|
792
|
-
for element in slide_copy.iter():
|
|
793
|
-
if not isinstance(element.tag, str):
|
|
794
|
-
continue
|
|
795
|
-
element.tag = xml_local_name(element.tag)
|
|
796
|
-
attributes = {
|
|
797
|
-
xml_local_name(attribute_name): value
|
|
798
|
-
for attribute_name, value in element.attrib.items()
|
|
799
|
-
}
|
|
800
|
-
element.attrib.clear()
|
|
801
|
-
element.attrib.update(attributes)
|
|
802
|
-
return ET.tostring(slide_copy, encoding="unicode")
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
def parse_presentation(root: ET.Element) -> dict[str, Any]:
|
|
806
|
-
root_name = xml_local_name(root.tag)
|
|
807
|
-
if root_name == "slide":
|
|
808
|
-
slide_roots = [root]
|
|
809
|
-
width = 960
|
|
810
|
-
height = 540
|
|
811
|
-
elif root_name == "presentation":
|
|
812
|
-
slide_roots = [child for child in root if xml_local_name(child.tag) == "slide"]
|
|
813
|
-
width = int(float(root.attrib.get("width", 960)))
|
|
814
|
-
height = int(float(root.attrib.get("height", 540)))
|
|
815
|
-
else:
|
|
816
|
-
fail("input must contain a <presentation> or <slide> root")
|
|
817
|
-
return {
|
|
818
|
-
"width": width,
|
|
819
|
-
"height": height,
|
|
820
|
-
"slides": [serialize_slide_for_layout(slide_root) for slide_root in slide_roots],
|
|
821
|
-
"slide_roots": slide_roots,
|
|
822
|
-
}
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
def build_source_xml_paths(slide_xml: str, slide_number: int) -> dict[str, list[str]]:
|
|
826
|
-
root = ET.fromstring(slide_xml)
|
|
827
|
-
data = next((child for child in root if xml_local_name(child.tag) == "data"), None)
|
|
828
|
-
paths: dict[str, list[str]] = {}
|
|
829
|
-
if data is None:
|
|
830
|
-
return paths
|
|
831
|
-
counts: dict[str, int] = {}
|
|
832
|
-
for child in data:
|
|
833
|
-
kind = xml_local_name(child.tag)
|
|
834
|
-
counts[kind] = counts.get(kind, 0) + 1
|
|
835
|
-
paths.setdefault(kind, []).append(
|
|
836
|
-
f"slide[{slide_number}]/data/{kind}[{counts[kind]}]"
|
|
837
|
-
)
|
|
838
|
-
return paths
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
def attach_source_xml_paths(
|
|
842
|
-
elements: list[dict[str, Any]], source_paths: dict[str, list[str]]
|
|
843
|
-
) -> None:
|
|
844
|
-
offsets: dict[str, int] = {}
|
|
845
|
-
for element in elements:
|
|
846
|
-
kind = element["kind"]
|
|
847
|
-
source_kind_index = element.get("_source_kind_index")
|
|
848
|
-
offset = (
|
|
849
|
-
source_kind_index - 1
|
|
850
|
-
if isinstance(source_kind_index, int) and source_kind_index > 0
|
|
851
|
-
else offsets.get(kind, 0)
|
|
852
|
-
)
|
|
853
|
-
kind_paths = source_paths.get(kind, [])
|
|
854
|
-
if offset < len(kind_paths):
|
|
855
|
-
element["xml_path"] = kind_paths[offset]
|
|
856
|
-
element["_ref"] = kind_paths[offset]
|
|
857
|
-
offsets[kind] = max(offsets.get(kind, 0), offset + 1)
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
def extract_source_id_elements(slide_xml: str, slide_number: int) -> list[dict[str, Any]]:
|
|
861
|
-
root = ET.fromstring(slide_xml)
|
|
862
|
-
elements: list[dict[str, Any]] = []
|
|
863
|
-
root_path = f"slide[{slide_number}]"
|
|
864
|
-
|
|
865
|
-
def walk(parent: ET.Element, parent_path: str) -> None:
|
|
866
|
-
child_counts: dict[str, int] = {}
|
|
867
|
-
for child in parent:
|
|
868
|
-
kind = xml_local_name(child.tag)
|
|
869
|
-
child_counts[kind] = child_counts.get(kind, 0) + 1
|
|
870
|
-
xml_path = (
|
|
871
|
-
f"{parent_path}/data"
|
|
872
|
-
if parent is root and kind == "data"
|
|
873
|
-
else f"{parent_path}/{kind}[{child_counts[kind]}]"
|
|
874
|
-
)
|
|
875
|
-
source_id = extract_attribute(
|
|
876
|
-
ET.tostring(child, encoding="unicode").split(">", 1)[0], "id"
|
|
877
|
-
)
|
|
878
|
-
if source_id:
|
|
879
|
-
elements.append(
|
|
880
|
-
{
|
|
881
|
-
"id": source_id,
|
|
882
|
-
"_source_id": source_id,
|
|
883
|
-
"kind": kind,
|
|
884
|
-
"type": child.attrib.get("type") or kind,
|
|
885
|
-
"xml_path": xml_path,
|
|
886
|
-
"_ref": xml_path,
|
|
887
|
-
"_slide_number": slide_number,
|
|
888
|
-
}
|
|
889
|
-
)
|
|
890
|
-
walk(child, xml_path)
|
|
891
|
-
|
|
892
|
-
walk(root, root_path)
|
|
893
|
-
return elements
|
|
894
|
-
|
|
895
|
-
|
|
896
|
-
def element_ref(element: dict[str, Any]) -> str:
|
|
897
|
-
ref = element.get("_ref") or element.get("xml_path")
|
|
898
|
-
if isinstance(ref, str) and ref:
|
|
899
|
-
return ref
|
|
900
|
-
# Low-level detector tests and external callers may pass already-extracted objects.
|
|
901
|
-
# The lint_xml pipeline always attaches the source path before issue detection.
|
|
902
|
-
fallback = element.get("id")
|
|
903
|
-
if isinstance(fallback, str) and fallback:
|
|
904
|
-
return fallback
|
|
905
|
-
raise AssertionError("lint element must have a source xml path or fallback id")
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
def source_element_id(element: dict[str, Any]) -> str | None:
|
|
909
|
-
value = element.get("_source_id")
|
|
910
|
-
return value if isinstance(value, str) and value else None
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
def element_label(element: dict[str, Any]) -> str:
|
|
914
|
-
return source_element_id(element) or element_ref(element)
|
|
915
|
-
|
|
916
|
-
|
|
917
|
-
def extract_elements(slide_xml: str) -> list[dict[str, Any]]:
|
|
918
|
-
elements: list[dict[str, Any]] = []
|
|
919
|
-
source_kind_counts: dict[str, int] = {}
|
|
920
|
-
|
|
921
|
-
for match in re.finditer(r"<(shape|img|table|chart|whiteboard|embed)\b([^>]*)>", slide_xml):
|
|
922
|
-
kind, attrs = match.group(1), match.group(2)
|
|
923
|
-
source_kind_counts[kind] = source_kind_counts.get(kind, 0) + 1
|
|
924
|
-
source_kind_index = source_kind_counts[kind]
|
|
925
|
-
is_self_closing = attrs.rstrip().endswith("/")
|
|
926
|
-
content = ""
|
|
927
|
-
if kind in {"shape", "table"} and not is_self_closing:
|
|
928
|
-
close_index = slide_xml.find(f"</{kind}>", match.end())
|
|
929
|
-
if close_index != -1:
|
|
930
|
-
content = slide_xml[match.end() : close_index]
|
|
931
|
-
|
|
932
|
-
source_id = extract_attribute(attrs, "id") or None
|
|
933
|
-
element_id = source_id or f"{kind}-{len(elements) + 1}"
|
|
934
|
-
x = extract_numeric_attribute(attrs, "topLeftX")
|
|
935
|
-
y = extract_numeric_attribute(attrs, "topLeftY")
|
|
936
|
-
width = extract_numeric_attribute(attrs, "width")
|
|
937
|
-
height = extract_numeric_attribute(attrs, "height")
|
|
938
|
-
rotation = extract_numeric_attribute(attrs, "rotation") or 0
|
|
939
|
-
alpha = extract_numeric_attribute(attrs, "alpha")
|
|
940
|
-
table_layouts: dict[str, dict[str, Any] | None] = {}
|
|
941
|
-
if kind == "table":
|
|
942
|
-
width, table_layouts["width"] = resolve_table_dimension(
|
|
943
|
-
content, width, extract_table_column_sizes, DEFAULT_TABLE_COLUMN_WIDTH
|
|
944
|
-
)
|
|
945
|
-
height, table_layouts["height"] = resolve_table_dimension(
|
|
946
|
-
content, height, extract_table_row_sizes, DEFAULT_TABLE_ROW_HEIGHT
|
|
947
|
-
)
|
|
948
|
-
if all(value is not None for value in [x, y, width, height]):
|
|
949
|
-
element = {
|
|
950
|
-
"id": element_id,
|
|
951
|
-
"_source_id": source_id,
|
|
952
|
-
"kind": kind,
|
|
953
|
-
"type": extract_attribute(attrs, "type") or kind,
|
|
954
|
-
"x": x,
|
|
955
|
-
"y": y,
|
|
956
|
-
"width": width,
|
|
957
|
-
"height": height,
|
|
958
|
-
"rotation": rotation,
|
|
959
|
-
"alpha": alpha if alpha is not None else 1,
|
|
960
|
-
"order": len(elements),
|
|
961
|
-
"_source_kind_index": source_kind_index,
|
|
962
|
-
}
|
|
963
|
-
if kind == "table":
|
|
964
|
-
element.update(
|
|
965
|
-
{
|
|
966
|
-
"declared_width": extract_numeric_attribute(attrs, "width"),
|
|
967
|
-
"declared_height": extract_numeric_attribute(attrs, "height"),
|
|
968
|
-
"table_layouts": table_layouts,
|
|
969
|
-
}
|
|
970
|
-
)
|
|
971
|
-
if kind == "shape":
|
|
972
|
-
content_attrs = extract_tag_attributes(content, "content")
|
|
973
|
-
font_size = extract_numeric_attribute(content_attrs, "fontSize")
|
|
974
|
-
if font_size is None:
|
|
975
|
-
font_size = extract_numeric_attribute(attrs, "fontSize")
|
|
976
|
-
font_family = extract_attribute(content_attrs, "fontFamily") or extract_attribute(attrs, "fontFamily")
|
|
977
|
-
text_color = extract_attribute(content_attrs, "color") or extract_attribute(attrs, "color")
|
|
978
|
-
bold = (
|
|
979
|
-
extract_bool_attribute(content_attrs, "bold")
|
|
980
|
-
or extract_bool_attribute(attrs, "bold")
|
|
981
|
-
or detect_inline_style_presence(content, {"strong", "b"})
|
|
982
|
-
or detect_any_span_bool_attribute(content, "bold")
|
|
983
|
-
)
|
|
984
|
-
italic = (
|
|
985
|
-
extract_bool_attribute(content_attrs, "italic")
|
|
986
|
-
or extract_bool_attribute(attrs, "italic")
|
|
987
|
-
or detect_inline_style_presence(content, {"i", "em"})
|
|
988
|
-
or detect_any_span_bool_attribute(content, "italic")
|
|
989
|
-
)
|
|
990
|
-
element.update(
|
|
991
|
-
{
|
|
992
|
-
"textType": extract_attribute(content_attrs, "textType"),
|
|
993
|
-
"textAlign": extract_attribute(content_attrs, "textAlign"),
|
|
994
|
-
"verticalAlign": extract_attribute(content_attrs, "verticalAlign") or "middle",
|
|
995
|
-
"vert": extract_attribute(attrs, "vert") or "horz",
|
|
996
|
-
"autoFit": extract_attribute(content_attrs, "autoFit"),
|
|
997
|
-
"wrap": extract_attribute(content_attrs, "wrap"),
|
|
998
|
-
"lineSpacing": extract_attribute(content_attrs, "lineSpacing"),
|
|
999
|
-
"beforeLineSpacing": extract_attribute(content_attrs, "beforeLineSpacing"),
|
|
1000
|
-
"afterLineSpacing": extract_attribute(content_attrs, "afterLineSpacing"),
|
|
1001
|
-
"letterSpacing": extract_numeric_attribute(content_attrs, "letterSpacing"),
|
|
1002
|
-
"paddingTop": extract_numeric_attribute(content_attrs, "paddingTop") or 0,
|
|
1003
|
-
"paddingRight": extract_numeric_attribute(content_attrs, "paddingRight") or 0,
|
|
1004
|
-
"paddingBottom": extract_numeric_attribute(content_attrs, "paddingBottom") or 0,
|
|
1005
|
-
"paddingLeft": extract_numeric_attribute(content_attrs, "paddingLeft") or 0,
|
|
1006
|
-
"fontSize": font_size if font_size is not None else 16,
|
|
1007
|
-
"fontFamily": font_family or "",
|
|
1008
|
-
"color": text_color,
|
|
1009
|
-
"textAlpha": effective_text_alpha(alpha, text_color),
|
|
1010
|
-
"bold": bold,
|
|
1011
|
-
"italic": italic,
|
|
1012
|
-
"text": strip_xml_paragraphs(content),
|
|
1013
|
-
"paragraphs": extract_text_paragraphs(content, font_size if font_size is not None else 16),
|
|
1014
|
-
}
|
|
1015
|
-
)
|
|
1016
|
-
elements.append(element)
|
|
1017
|
-
return elements
|
|
1018
|
-
|
|
1019
|
-
|
|
1020
|
-
def intersects(left: dict[str, Any], right: dict[str, Any]) -> bool:
|
|
1021
|
-
return (
|
|
1022
|
-
left["x"] < right["x"] + right["width"]
|
|
1023
|
-
and left["x"] + left["width"] > right["x"]
|
|
1024
|
-
and left["y"] < right["y"] + right["height"]
|
|
1025
|
-
and left["y"] + left["height"] > right["y"]
|
|
1026
|
-
)
|
|
1027
|
-
|
|
1028
|
-
|
|
1029
|
-
def is_text_element(element: dict[str, Any]) -> bool:
|
|
1030
|
-
return element["kind"] == "shape" and element["type"] == "text"
|
|
1031
|
-
|
|
1032
|
-
|
|
1033
|
-
def is_whiteboard_element(element: dict[str, Any]) -> bool:
|
|
1034
|
-
return element["kind"] == "whiteboard"
|
|
1035
|
-
|
|
1036
|
-
|
|
1037
|
-
def has_text_content(element: dict[str, Any]) -> bool:
|
|
1038
|
-
return bool(element.get("text"))
|
|
1039
|
-
|
|
1040
|
-
|
|
1041
|
-
def is_vertical_text(element: dict[str, Any]) -> bool:
|
|
1042
|
-
return element.get("vert") in {"vert", "vert270", "word-art-vert", "word-art-vert-rtl", "ea-vert"}
|
|
1043
|
-
|
|
1044
|
-
|
|
1045
|
-
def detect_image_text_occlusions(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
1046
|
-
issues: list[dict[str, Any]] = []
|
|
1047
|
-
text_elements = [
|
|
1048
|
-
element
|
|
1049
|
-
for element in elements
|
|
1050
|
-
if is_text_element(element) and has_text_content(element) and not is_ghost_text(element)
|
|
1051
|
-
]
|
|
1052
|
-
image_elements = [element for element in elements if element["kind"] == "img" and element["alpha"] > 0]
|
|
1053
|
-
for text_element in text_elements:
|
|
1054
|
-
for image_element in image_elements:
|
|
1055
|
-
if image_element["order"] <= text_element["order"]:
|
|
1056
|
-
continue
|
|
1057
|
-
if is_vertical_text(text_element):
|
|
1058
|
-
if intersects(image_element, text_element):
|
|
1059
|
-
issues.append({
|
|
1060
|
-
"level": "info",
|
|
1061
|
-
"code": "image_may_cover_vertical_text",
|
|
1062
|
-
"elements": [element_ref(image_element), element_ref(text_element)],
|
|
1063
|
-
"message": (
|
|
1064
|
-
f"image {element_label(image_element)} may cover vertical text shape "
|
|
1065
|
-
f"{element_label(text_element)}"
|
|
1066
|
-
),
|
|
1067
|
-
"hint": "Inspect the rendered slide because vertical text layout is not statically modeled.",
|
|
1068
|
-
})
|
|
1069
|
-
continue
|
|
1070
|
-
text_visual_bbox = estimate_text_visual_bbox(text_element)
|
|
1071
|
-
if text_visual_bbox is not None and intersects(image_element, text_visual_bbox):
|
|
1072
|
-
issues.append({
|
|
1073
|
-
"level": "error",
|
|
1074
|
-
"code": "image_covers_text",
|
|
1075
|
-
"elements": [element_ref(image_element), element_ref(text_element)],
|
|
1076
|
-
"message": (
|
|
1077
|
-
f"image {element_label(image_element)} covers text shape "
|
|
1078
|
-
f"{element_label(text_element)}"
|
|
1079
|
-
),
|
|
1080
|
-
"hint": "Move the image before the text shape in XML order, or adjust the image and text shape coordinates or dimensions.",
|
|
1081
|
-
})
|
|
1082
|
-
return issues
|
|
1083
|
-
|
|
1084
|
-
|
|
1085
|
-
def is_decorative_text(element: dict[str, Any]) -> bool:
|
|
1086
|
-
text = element.get("text") or ""
|
|
1087
|
-
return bool(text) and re.search(r"[A-Za-z0-9\u4e00-\u9fff]", text) is None
|
|
1088
|
-
|
|
1089
|
-
|
|
1090
|
-
def normalize_text_for_overlap(text: str) -> str:
|
|
1091
|
-
return re.sub(r"\s+", "", text)
|
|
1092
|
-
|
|
1093
|
-
|
|
1094
|
-
SERIF_FONT_PATTERNS = {
|
|
1095
|
-
"song", "songti", "simsun", "ming", "mincho",
|
|
1096
|
-
"georgia", "times", "caslon", "garamond", "sourcehan-serif",
|
|
1097
|
-
"source han serif", "思源宋体", "宋体", "明体",
|
|
1098
|
-
}
|
|
1099
|
-
|
|
1100
|
-
SANS_EXPLICIT_MARKERS = {"sans", "sans-serif", "sans serif", "sourcehan-sans", "source han sans", "思源黑体", "黑体",
|
|
1101
|
-
"helvetica", "arial", "inter", "roboto", "verdana", "tahoma", "calibri", "open sans"}
|
|
1102
|
-
|
|
1103
|
-
|
|
1104
|
-
def classify_font_family(font_family: str | None) -> str:
|
|
1105
|
-
if not font_family:
|
|
1106
|
-
return "sans"
|
|
1107
|
-
family_lower = font_family.lower()
|
|
1108
|
-
for marker in SANS_EXPLICIT_MARKERS:
|
|
1109
|
-
if marker in family_lower:
|
|
1110
|
-
return "sans"
|
|
1111
|
-
serif_keywords = SERIF_FONT_PATTERNS | {"serif"}
|
|
1112
|
-
for pattern in serif_keywords:
|
|
1113
|
-
if pattern in family_lower:
|
|
1114
|
-
return "serif"
|
|
1115
|
-
return "sans"
|
|
1116
|
-
|
|
1117
|
-
|
|
1118
|
-
_FONT_CATEGORY_MULTIPLIERS: dict[str, dict[str, float]] = {
|
|
1119
|
-
"sans": {"upper": 0.57, "lower": 0.51, "digit": 0.58, "punct": 0.50},
|
|
1120
|
-
"serif": {"upper": 0.57, "lower": 0.53, "digit": 0.58, "punct": 0.50},
|
|
1121
|
-
}
|
|
1122
|
-
|
|
1123
|
-
|
|
1124
|
-
def estimate_character_width(
|
|
1125
|
-
character: str,
|
|
1126
|
-
font_size: int | float,
|
|
1127
|
-
bold: bool = False,
|
|
1128
|
-
font_family: str | None = None,
|
|
1129
|
-
) -> int | float:
|
|
1130
|
-
bold_multiplier = 1.05 if bold else 1.0
|
|
1131
|
-
if character.isspace():
|
|
1132
|
-
return font_size * 0.33 * bold_multiplier
|
|
1133
|
-
ea_width = unicodedata.east_asian_width(character)
|
|
1134
|
-
if ea_width in {"F", "W"}:
|
|
1135
|
-
return font_size * bold_multiplier
|
|
1136
|
-
category = classify_font_family(font_family)
|
|
1137
|
-
coeffs = _FONT_CATEGORY_MULTIPLIERS[category]
|
|
1138
|
-
if character.isupper():
|
|
1139
|
-
return font_size * coeffs["upper"] * bold_multiplier
|
|
1140
|
-
if character.islower():
|
|
1141
|
-
return font_size * coeffs["lower"] * bold_multiplier
|
|
1142
|
-
if character.isdigit():
|
|
1143
|
-
return font_size * coeffs["digit"] * bold_multiplier
|
|
1144
|
-
return font_size * coeffs["punct"] * bold_multiplier
|
|
1145
|
-
|
|
1146
|
-
|
|
1147
|
-
def estimate_text_width(
|
|
1148
|
-
text: str,
|
|
1149
|
-
font_size: int | float,
|
|
1150
|
-
letter_spacing: int | float = 0,
|
|
1151
|
-
bold: bool = False,
|
|
1152
|
-
font_family: str | None = None,
|
|
1153
|
-
) -> int | float:
|
|
1154
|
-
base = sum(estimate_character_width(character, font_size, bold, font_family) for character in text)
|
|
1155
|
-
return base + max(len(text) - 1, 0) * letter_spacing
|
|
1156
|
-
|
|
1157
|
-
|
|
1158
|
-
def resolve_letter_spacing(element: dict[str, Any], paragraph: dict[str, Any] | None = None) -> int | float:
|
|
1159
|
-
if paragraph is not None:
|
|
1160
|
-
value = paragraph.get("letterSpacing")
|
|
1161
|
-
if isinstance(value, (int, float)):
|
|
1162
|
-
return value
|
|
1163
|
-
value = element.get("letterSpacing")
|
|
1164
|
-
return value if isinstance(value, (int, float)) else 0
|
|
1165
|
-
|
|
1166
|
-
|
|
1167
|
-
def text_wrap_width_tolerance() -> int | float:
|
|
1168
|
-
return TEXT_WRAP_WIDTH_TOLERANCE_PX
|
|
1169
|
-
|
|
1170
|
-
|
|
1171
|
-
def text_height_overflow_tolerance() -> int | float:
|
|
1172
|
-
return TEXT_HEIGHT_OVERFLOW_TOLERANCE_PX
|
|
1173
|
-
|
|
1174
|
-
|
|
1175
|
-
def has_explicit_height_auto_fit(element: dict[str, Any]) -> bool:
|
|
1176
|
-
return element.get("autoFit") in {"normal-auto-fit", "shape-auto-fit"}
|
|
1177
|
-
|
|
1178
|
-
|
|
1179
|
-
def is_short_metric_text(text: str) -> bool:
|
|
1180
|
-
compact = re.sub(r"\s+", "", text)
|
|
1181
|
-
if not compact or len(compact) > 16 or re.search(r"\d", compact) is None:
|
|
1182
|
-
return False
|
|
1183
|
-
if re.fullmatch(r"[+\-–—]?[0-9,.,]+[\u4e00-\u9fffA-Za-z]{1,4}", compact):
|
|
1184
|
-
return True
|
|
1185
|
-
if re.search(r"[,.,+\-–—/%%]", compact) is None:
|
|
1186
|
-
return False
|
|
1187
|
-
return re.fullmatch(r"[+\-–—]?[0-9A-Za-z,.,/%%\-–—\u4e00-\u9fff]+", compact) is not None
|
|
1188
|
-
|
|
1189
|
-
|
|
1190
|
-
def is_single_line_visual_candidate(
|
|
1191
|
-
element: dict[str, Any],
|
|
1192
|
-
paragraph: dict[str, Any] | None,
|
|
1193
|
-
text: str,
|
|
1194
|
-
logical_width: int | float,
|
|
1195
|
-
effective_width: int | float,
|
|
1196
|
-
) -> bool:
|
|
1197
|
-
if "\n" in text or logical_width <= effective_width:
|
|
1198
|
-
return False
|
|
1199
|
-
if is_short_metric_text(text):
|
|
1200
|
-
return logical_width <= effective_width * SINGLE_LINE_METRIC_WIDTH_RATIO
|
|
1201
|
-
|
|
1202
|
-
text_align = (paragraph or {}).get("textAlign") or element.get("textAlign")
|
|
1203
|
-
compact_len = len(re.sub(r"\s+", "", text))
|
|
1204
|
-
if text_align == "center" and compact_len <= 32:
|
|
1205
|
-
return logical_width <= effective_width * CENTERED_SHORT_LABEL_WIDTH_RATIO
|
|
1206
|
-
|
|
1207
|
-
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
1208
|
-
if element.get("textType") in {"headline", "title"} and font_size <= 30 and compact_len <= 40:
|
|
1209
|
-
return logical_width <= effective_width * HEADLINE_NEAR_FIT_WIDTH_RATIO
|
|
1210
|
-
return False
|
|
1211
|
-
|
|
1212
|
-
|
|
1213
|
-
def estimate_text_max_line_width(element: dict[str, Any]) -> int | float:
|
|
1214
|
-
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
1215
|
-
bold = element.get("bold", False)
|
|
1216
|
-
font_family = element.get("fontFamily", "")
|
|
1217
|
-
letter_spacing = resolve_letter_spacing(element)
|
|
1218
|
-
paragraphs = [paragraph for paragraph in re.split(r"\n+", element["text"]) if paragraph]
|
|
1219
|
-
return max(
|
|
1220
|
-
[estimate_text_width(paragraph, font_size, letter_spacing, bold, font_family) for paragraph in paragraphs]
|
|
1221
|
-
or [1]
|
|
1222
|
-
)
|
|
1223
|
-
|
|
1224
|
-
|
|
1225
|
-
def is_similar_text_overlay(left: dict[str, Any], right: dict[str, Any]) -> bool:
|
|
1226
|
-
left_text = normalize_text_for_overlap(left.get("text") or "")
|
|
1227
|
-
right_text = normalize_text_for_overlap(right.get("text") or "")
|
|
1228
|
-
if not left_text or not right_text:
|
|
1229
|
-
return False
|
|
1230
|
-
if left_text == right_text or left_text in right_text or right_text in left_text:
|
|
1231
|
-
return True
|
|
1232
|
-
return SequenceMatcher(None, left_text, right_text).ratio() >= 0.75
|
|
1233
|
-
|
|
1234
|
-
|
|
1235
|
-
def estimate_text_line_count_for_text(
|
|
1236
|
-
element: dict[str, Any], text: str, paragraph: dict[str, Any] | None = None
|
|
1237
|
-
) -> int:
|
|
1238
|
-
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
1239
|
-
bold = element.get("bold", False)
|
|
1240
|
-
font_family = element.get("fontFamily", "")
|
|
1241
|
-
letter_spacing = resolve_letter_spacing(element, paragraph)
|
|
1242
|
-
available_width = max(element["width"] - element.get("paddingLeft", 0) - element.get("paddingRight", 0), 1)
|
|
1243
|
-
hard_lines = text.split("\n")
|
|
1244
|
-
if not text:
|
|
1245
|
-
return 0
|
|
1246
|
-
line_count = 0
|
|
1247
|
-
for hard_line in hard_lines:
|
|
1248
|
-
if element.get("wrap") in {"false", "0"}:
|
|
1249
|
-
line_count += 1
|
|
1250
|
-
continue
|
|
1251
|
-
logical_width = max(estimate_text_width(hard_line, font_size, letter_spacing, bold, font_family), 1)
|
|
1252
|
-
effective_width = available_width + text_wrap_width_tolerance()
|
|
1253
|
-
if is_single_line_visual_candidate(element, paragraph, hard_line, logical_width, effective_width):
|
|
1254
|
-
line_count += 1
|
|
1255
|
-
continue
|
|
1256
|
-
line_count += max(1, math.ceil(logical_width / effective_width))
|
|
1257
|
-
return line_count
|
|
1258
|
-
|
|
1259
|
-
|
|
1260
|
-
def estimate_text_line_count(element: dict[str, Any]) -> int:
|
|
1261
|
-
return max(estimate_text_line_count_for_text(element, element["text"]), 1)
|
|
1262
|
-
|
|
1263
|
-
|
|
1264
|
-
def estimate_text_line_height(element: dict[str, Any], line_spacing: str | None = None) -> int | float | None:
|
|
1265
|
-
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
1266
|
-
if line_spacing is None:
|
|
1267
|
-
return font_size * DEFAULT_TEXT_LINE_SPACING_MULTIPLE
|
|
1268
|
-
match = re.fullmatch(r"(multiple|fixed):([0-9]+(?:\.[0-9]+)?)", line_spacing)
|
|
1269
|
-
if match is None:
|
|
1270
|
-
return None
|
|
1271
|
-
spacing_type, value = match.groups()
|
|
1272
|
-
return font_size * float(value) if spacing_type == "multiple" else float(value)
|
|
1273
|
-
|
|
1274
|
-
|
|
1275
|
-
def adjust_dense_body_line_height(
|
|
1276
|
-
element: dict[str, Any],
|
|
1277
|
-
line_spacing: str | None,
|
|
1278
|
-
line_height: int | float,
|
|
1279
|
-
paragraph_count: int,
|
|
1280
|
-
) -> int | float:
|
|
1281
|
-
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
1282
|
-
if paragraph_count < 4 or font_size > 14 or not line_spacing:
|
|
1283
|
-
return line_height
|
|
1284
|
-
match = re.fullmatch(r"multiple:([0-9]+(?:\.[0-9]+)?)", line_spacing)
|
|
1285
|
-
if match is None:
|
|
1286
|
-
return line_height
|
|
1287
|
-
return min(line_height, font_size * min(float(match.group(1)), DENSE_BODY_LINE_SPACING_MAX_MULTIPLE))
|
|
1288
|
-
|
|
1289
|
-
|
|
1290
|
-
def detect_text_may_overflow_shapes(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
1291
|
-
issues: list[dict[str, Any]] = []
|
|
1292
|
-
for element in elements:
|
|
1293
|
-
if not is_text_element(element) or not has_text_content(element):
|
|
1294
|
-
continue
|
|
1295
|
-
if has_explicit_height_auto_fit(element):
|
|
1296
|
-
continue
|
|
1297
|
-
|
|
1298
|
-
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
1299
|
-
paragraphs = element.get("paragraphs") or [
|
|
1300
|
-
{
|
|
1301
|
-
"text": element["text"],
|
|
1302
|
-
"lineSpacing": None,
|
|
1303
|
-
"beforeLineSpacing": None,
|
|
1304
|
-
"afterLineSpacing": None,
|
|
1305
|
-
}
|
|
1306
|
-
]
|
|
1307
|
-
line_count = 0
|
|
1308
|
-
estimated_height = 0.0
|
|
1309
|
-
line_heights: list[int | float] = []
|
|
1310
|
-
for paragraph in paragraphs:
|
|
1311
|
-
paragraph_line_count = estimate_text_line_count_for_text(element, paragraph["text"], paragraph)
|
|
1312
|
-
if paragraph_line_count == 0:
|
|
1313
|
-
continue
|
|
1314
|
-
resolved_line_spacing = paragraph["lineSpacing"] or element["lineSpacing"]
|
|
1315
|
-
line_height = estimate_text_line_height(element, resolved_line_spacing)
|
|
1316
|
-
before_spacing = estimate_text_line_height(
|
|
1317
|
-
element, paragraph["beforeLineSpacing"] or element["beforeLineSpacing"] or "fixed:0"
|
|
1318
|
-
)
|
|
1319
|
-
after_spacing = estimate_text_line_height(
|
|
1320
|
-
element, paragraph["afterLineSpacing"] or element["afterLineSpacing"] or "fixed:0"
|
|
1321
|
-
)
|
|
1322
|
-
if line_height is None or before_spacing is None or after_spacing is None:
|
|
1323
|
-
line_count = 0
|
|
1324
|
-
break
|
|
1325
|
-
line_height = adjust_dense_body_line_height(element, resolved_line_spacing, line_height, len(paragraphs))
|
|
1326
|
-
first_line_height = font_size if line_count == 0 else line_height
|
|
1327
|
-
line_count += paragraph_line_count
|
|
1328
|
-
line_heights.append(line_height)
|
|
1329
|
-
estimated_height += (
|
|
1330
|
-
before_spacing + first_line_height + max(paragraph_line_count - 1, 0) * line_height + after_spacing
|
|
1331
|
-
)
|
|
1332
|
-
if line_count == 0:
|
|
1333
|
-
continue
|
|
1334
|
-
available_height = max(element["height"] - element["paddingTop"] - element["paddingBottom"], 0)
|
|
1335
|
-
overflow = estimated_height - available_height
|
|
1336
|
-
if overflow <= text_height_overflow_tolerance():
|
|
1337
|
-
continue
|
|
1338
|
-
|
|
1339
|
-
is_background = is_background_decorative_text(element, elements)
|
|
1340
|
-
if is_background:
|
|
1341
|
-
level = "info"
|
|
1342
|
-
else:
|
|
1343
|
-
level = "error" if overflow > 10 else "warning"
|
|
1344
|
-
message = (
|
|
1345
|
-
f"text shape {element_label(element)} may overflow its own content box "
|
|
1346
|
-
f'(estimated {estimated_height:g}px, available {available_height:g}px); '
|
|
1347
|
-
'consider setting content wrap="true" autoFit="normal-auto-fit"'
|
|
1348
|
-
)
|
|
1349
|
-
if is_background:
|
|
1350
|
-
message += " (likely background decoration: large font, low alpha, underneath other text)"
|
|
1351
|
-
issues.append(
|
|
1352
|
-
{
|
|
1353
|
-
"level": level,
|
|
1354
|
-
"code": "text_may_overflow_shape",
|
|
1355
|
-
"elements": [element_ref(element)],
|
|
1356
|
-
"line_count": line_count,
|
|
1357
|
-
"line_height": max(line_heights),
|
|
1358
|
-
"estimated_height": estimated_height,
|
|
1359
|
-
"available_height": available_height,
|
|
1360
|
-
"overflow": overflow,
|
|
1361
|
-
"message": message,
|
|
1362
|
-
"hint": (
|
|
1363
|
-
"Increase shape.height, reduce the text, or set content wrap=\"true\" "
|
|
1364
|
-
"autoFit=\"normal-auto-fit\". "
|
|
1365
|
-
"This is an estimate based on font size, line spacing, and wrapped line count."
|
|
1366
|
-
),
|
|
1367
|
-
}
|
|
1368
|
-
)
|
|
1369
|
-
return issues
|
|
1370
|
-
|
|
1371
|
-
|
|
1372
|
-
def is_background_decorative_text(
|
|
1373
|
-
element: dict[str, Any], elements: list[dict[str, Any]]
|
|
1374
|
-
) -> bool:
|
|
1375
|
-
if not is_ghost_text(element):
|
|
1376
|
-
return False
|
|
1377
|
-
for other in elements:
|
|
1378
|
-
if other is element:
|
|
1379
|
-
continue
|
|
1380
|
-
if not is_text_element(other) or not has_text_content(other):
|
|
1381
|
-
continue
|
|
1382
|
-
foreground_alpha = other.get("textAlpha", other.get("alpha", 1))
|
|
1383
|
-
if not isinstance(foreground_alpha, (int, float)) or foreground_alpha <= 0:
|
|
1384
|
-
continue
|
|
1385
|
-
if other["order"] <= element["order"]:
|
|
1386
|
-
continue
|
|
1387
|
-
if intersects(element, other):
|
|
1388
|
-
return True
|
|
1389
|
-
return False
|
|
1390
|
-
|
|
1391
|
-
|
|
1392
|
-
def is_ghost_text(element: dict[str, Any]) -> bool:
|
|
1393
|
-
if not is_text_element(element) or not has_text_content(element):
|
|
1394
|
-
return False
|
|
1395
|
-
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
1396
|
-
text_alpha = element.get("textAlpha", element.get("alpha", 1))
|
|
1397
|
-
if not isinstance(text_alpha, (int, float)):
|
|
1398
|
-
return False
|
|
1399
|
-
if font_size > GHOST_TEXT_MIN_FONT_SIZE and text_alpha < GHOST_TEXT_MAX_ALPHA:
|
|
1400
|
-
return True
|
|
1401
|
-
return font_size >= GHOST_TEXT_FAINT_MIN_FONT_SIZE and text_alpha < GHOST_TEXT_FAINT_MAX_ALPHA
|
|
1402
|
-
|
|
1403
|
-
|
|
1404
|
-
def estimate_text_visual_bbox(element: dict[str, Any]) -> dict[str, int | float] | None:
|
|
1405
|
-
if not is_text_element(element) or not has_text_content(element) or is_decorative_text(element):
|
|
1406
|
-
return None
|
|
1407
|
-
|
|
1408
|
-
padding_left = element.get("paddingLeft", 0)
|
|
1409
|
-
padding_right = element.get("paddingRight", 0)
|
|
1410
|
-
padding_top = element.get("paddingTop", 0)
|
|
1411
|
-
padding_bottom = element.get("paddingBottom", 0)
|
|
1412
|
-
content_width = max(element["width"] - padding_left - padding_right, 0)
|
|
1413
|
-
content_height = max(element["height"] - padding_top - padding_bottom, 0)
|
|
1414
|
-
font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
|
|
1415
|
-
line_count = estimate_text_line_count(element)
|
|
1416
|
-
estimated_width = max(1, estimate_text_max_line_width(element))
|
|
1417
|
-
visual_width = estimated_width if element.get("wrap") in {"false", "0"} else min(content_width, estimated_width)
|
|
1418
|
-
visual_height = min(content_height, max(1, line_count * font_size * 1.2))
|
|
1419
|
-
x = element["x"] + padding_left
|
|
1420
|
-
if element.get("textAlign") == "center":
|
|
1421
|
-
x += (content_width - visual_width) / 2
|
|
1422
|
-
elif element.get("textAlign") == "right":
|
|
1423
|
-
x += content_width - visual_width
|
|
1424
|
-
y = element["y"] + padding_top
|
|
1425
|
-
if element.get("verticalAlign") == "middle":
|
|
1426
|
-
y += (content_height - visual_height) / 2
|
|
1427
|
-
elif element.get("verticalAlign") == "bottom":
|
|
1428
|
-
y += content_height - visual_height
|
|
1429
|
-
return {
|
|
1430
|
-
"x": x,
|
|
1431
|
-
"y": y,
|
|
1432
|
-
"width": visual_width,
|
|
1433
|
-
"height": visual_height,
|
|
1434
|
-
}
|
|
1435
|
-
|
|
1436
|
-
|
|
1437
|
-
def intersection_area(left: dict[str, Any], right: dict[str, Any]) -> int | float:
|
|
1438
|
-
width = min(left["x"] + left["width"], right["x"] + right["width"]) - max(left["x"], right["x"])
|
|
1439
|
-
height = min(left["y"] + left["height"], right["y"] + right["height"]) - max(left["y"], right["y"])
|
|
1440
|
-
if width <= 0 or height <= 0:
|
|
1441
|
-
return 0
|
|
1442
|
-
return width * height
|
|
1443
|
-
|
|
1444
|
-
|
|
1445
|
-
def intersection_height(left: dict[str, Any], right: dict[str, Any]) -> int | float:
|
|
1446
|
-
height = min(left["y"] + left["height"], right["y"] + right["height"]) - max(left["y"], right["y"])
|
|
1447
|
-
return max(height, 0)
|
|
1448
|
-
|
|
1449
|
-
|
|
1450
|
-
def intersection_width(left: dict[str, Any], right: dict[str, Any]) -> int | float:
|
|
1451
|
-
width = min(left["x"] + left["width"], right["x"] + right["width"]) - max(left["x"], right["x"])
|
|
1452
|
-
return max(width, 0)
|
|
1453
|
-
|
|
1454
|
-
|
|
1455
|
-
def element_area(element: dict[str, Any]) -> int | float:
|
|
1456
|
-
return max(element["width"], 0) * max(element["height"], 0)
|
|
1457
|
-
|
|
1458
|
-
|
|
1459
|
-
def contains(outer: dict[str, Any], inner: dict[str, Any], tolerance: int | float = 2) -> bool:
|
|
1460
|
-
return (
|
|
1461
|
-
inner["x"] >= outer["x"] - tolerance
|
|
1462
|
-
and inner["y"] >= outer["y"] - tolerance
|
|
1463
|
-
and inner["x"] + inner["width"] <= outer["x"] + outer["width"] + tolerance
|
|
1464
|
-
and inner["y"] + inner["height"] <= outer["y"] + outer["height"] + tolerance
|
|
1465
|
-
)
|
|
1466
|
-
|
|
1467
|
-
|
|
1468
|
-
def is_bottom_layer_full_slide_whiteboard(
|
|
1469
|
-
whiteboard: dict[str, Any], other: dict[str, Any], slide_width: int | float, slide_height: int | float
|
|
1470
|
-
) -> bool:
|
|
1471
|
-
return (
|
|
1472
|
-
whiteboard["order"] < other["order"]
|
|
1473
|
-
and whiteboard["x"] <= 2
|
|
1474
|
-
and whiteboard["y"] <= 2
|
|
1475
|
-
and whiteboard["width"] >= slide_width - 4
|
|
1476
|
-
and whiteboard["height"] >= slide_height - 4
|
|
1477
|
-
)
|
|
1478
|
-
|
|
1479
|
-
|
|
1480
|
-
def is_background_container_for_whiteboard(container: dict[str, Any], whiteboard: dict[str, Any]) -> bool:
|
|
1481
|
-
if container["order"] > whiteboard["order"]:
|
|
1482
|
-
return False
|
|
1483
|
-
if is_text_element(container):
|
|
1484
|
-
return False
|
|
1485
|
-
return contains(container, whiteboard)
|
|
1486
|
-
|
|
1487
|
-
|
|
1488
|
-
def is_template_text_stack(left: dict[str, Any], right: dict[str, Any]) -> bool:
|
|
1489
|
-
if not (is_text_element(left) and is_text_element(right)):
|
|
1490
|
-
return False
|
|
1491
|
-
if not (has_text_content(left) and has_text_content(right)):
|
|
1492
|
-
return True
|
|
1493
|
-
top, bottom = sorted([left, right], key=lambda element: element["y"])
|
|
1494
|
-
top_type = top.get("textType")
|
|
1495
|
-
bottom_type = bottom.get("textType")
|
|
1496
|
-
allowed_pairs = {
|
|
1497
|
-
("title", "sub-headline"),
|
|
1498
|
-
("title", None),
|
|
1499
|
-
("headline", "headline"),
|
|
1500
|
-
("headline", None),
|
|
1501
|
-
}
|
|
1502
|
-
if (top_type, bottom_type) not in allowed_pairs:
|
|
1503
|
-
return False
|
|
1504
|
-
same_column = abs(top["x"] - bottom["x"]) <= 4
|
|
1505
|
-
vertical_offset = bottom["y"] - top["y"]
|
|
1506
|
-
top_font_size = float(top.get("fontSize", 16))
|
|
1507
|
-
return same_column and vertical_offset >= top_font_size * 0.75
|
|
1508
|
-
|
|
1509
|
-
|
|
1510
|
-
def should_flag_horizontal_text_overflow(left: dict[str, Any], right: dict[str, Any]) -> bool:
|
|
1511
|
-
if not (is_text_element(left) and is_text_element(right)):
|
|
1512
|
-
return False
|
|
1513
|
-
if not (has_text_content(left) and has_text_content(right)):
|
|
1514
|
-
return False
|
|
1515
|
-
if is_ghost_text(left) or is_ghost_text(right):
|
|
1516
|
-
return False
|
|
1517
|
-
if is_template_text_stack(left, right) or is_similar_text_overlay(left, right):
|
|
1518
|
-
return False
|
|
1519
|
-
|
|
1520
|
-
source, target = sorted([left, right], key=lambda element: element["x"])
|
|
1521
|
-
if source["x"] == target["x"]:
|
|
1522
|
-
return False
|
|
1523
|
-
wrap_enabled = source.get("wrap") not in {"false", "0"}
|
|
1524
|
-
has_horizontal_gap = source["x"] + source["width"] <= target["x"]
|
|
1525
|
-
if wrap_enabled and has_horizontal_gap:
|
|
1526
|
-
return False
|
|
1527
|
-
if source.get("autoFit") == "normal-auto-fit":
|
|
1528
|
-
return False
|
|
1529
|
-
if source.get("textAlign") in {"center", "right"}:
|
|
1530
|
-
return False
|
|
1531
|
-
|
|
1532
|
-
font_size = source["fontSize"] if isinstance(source["fontSize"], (int, float)) else 16
|
|
1533
|
-
padding_left = source.get("paddingLeft", 0)
|
|
1534
|
-
padding_right = source.get("paddingRight", 0)
|
|
1535
|
-
available_width = max(source["width"] - padding_left - padding_right, 1)
|
|
1536
|
-
visual_width = estimate_text_max_line_width(source)
|
|
1537
|
-
overflow_width = visual_width - available_width
|
|
1538
|
-
min_overflow = max(font_size * 1.5, available_width * 0.08)
|
|
1539
|
-
if overflow_width < min_overflow:
|
|
1540
|
-
return False
|
|
1541
|
-
|
|
1542
|
-
intrusion_width = source["x"] + padding_left + visual_width - target["x"]
|
|
1543
|
-
min_intrusion = max(font_size * 1.5, target["width"] * 0.08)
|
|
1544
|
-
if intrusion_width < min_intrusion:
|
|
1545
|
-
return False
|
|
1546
|
-
|
|
1547
|
-
vertical_overlap = intersection_height(source, target)
|
|
1548
|
-
min_vertical_overlap = min(source["height"], target["height"]) * 0.40
|
|
1549
|
-
return vertical_overlap >= min_vertical_overlap
|
|
1550
|
-
|
|
1551
|
-
|
|
1552
|
-
def horizontal_text_overflow_measurement(left: dict[str, Any], right: dict[str, Any]) -> dict[str, int | float]:
|
|
1553
|
-
source, target = sorted([left, right], key=lambda element: element["x"])
|
|
1554
|
-
padding_left = source.get("paddingLeft", 0)
|
|
1555
|
-
visual_width = estimate_text_max_line_width(source)
|
|
1556
|
-
source_visual_bbox = {"x": source["x"] + padding_left, "y": source["y"], "width": visual_width, "height": source["height"]}
|
|
1557
|
-
width = intersection_width(source_visual_bbox, target)
|
|
1558
|
-
height = intersection_height(source_visual_bbox, target)
|
|
1559
|
-
return {
|
|
1560
|
-
"intersection_width": round(width, 3),
|
|
1561
|
-
"intersection_height": round(height, 3),
|
|
1562
|
-
"intersection_area": round(width * height, 3),
|
|
1563
|
-
}
|
|
1564
|
-
|
|
1565
|
-
|
|
1566
|
-
def should_flag_overlap(left: dict[str, Any], right: dict[str, Any]) -> bool:
|
|
1567
|
-
if is_text_element(left) and not has_text_content(left):
|
|
1568
|
-
return False
|
|
1569
|
-
if is_text_element(right) and not has_text_content(right):
|
|
1570
|
-
return False
|
|
1571
|
-
if is_ghost_text(left) or is_ghost_text(right):
|
|
1572
|
-
return False
|
|
1573
|
-
if is_template_text_stack(left, right):
|
|
1574
|
-
return False
|
|
1575
|
-
if is_text_element(left) and is_text_element(right):
|
|
1576
|
-
if is_similar_text_overlay(left, right):
|
|
1577
|
-
return False
|
|
1578
|
-
left_visual = estimate_text_visual_bbox(left)
|
|
1579
|
-
right_visual = estimate_text_visual_bbox(right)
|
|
1580
|
-
if left_visual is None or right_visual is None:
|
|
1581
|
-
return False
|
|
1582
|
-
overlap_area = intersection_area(left_visual, right_visual)
|
|
1583
|
-
if overlap_area <= 0:
|
|
1584
|
-
return False
|
|
1585
|
-
smaller_area = min(
|
|
1586
|
-
left_visual["width"] * left_visual["height"],
|
|
1587
|
-
right_visual["width"] * right_visual["height"],
|
|
1588
|
-
)
|
|
1589
|
-
return smaller_area > 0 and overlap_area / smaller_area >= 0.30
|
|
1590
|
-
return False
|
|
1591
|
-
|
|
1592
|
-
|
|
1593
|
-
def build_whiteboard_external_overlap_issue(
|
|
1594
|
-
whiteboard: dict[str, Any], overlap_details: list[dict[str, Any]]
|
|
1595
|
-
) -> dict[str, Any]:
|
|
1596
|
-
element_refs = [detail["element"] for detail in overlap_details]
|
|
1597
|
-
return {
|
|
1598
|
-
"level": "warning",
|
|
1599
|
-
"code": "whiteboard_external_overlap",
|
|
1600
|
-
"elements": [element_ref(whiteboard), *element_refs],
|
|
1601
|
-
"message": (
|
|
1602
|
-
f"whiteboard {element_label(whiteboard)} overlaps {len(element_refs)} "
|
|
1603
|
-
"sibling elements across its boundary"
|
|
1604
|
-
),
|
|
1605
|
-
"hint": (
|
|
1606
|
-
"Treat this as a static whiteboard container-bbox risk, not final visual proof. "
|
|
1607
|
-
"After moving or accepting the overlap, use screenshot QA or equivalent rendered visual inspection as "
|
|
1608
|
-
"the final authority because XML readback does not include whiteboard SVG/Mermaid internals."
|
|
1609
|
-
),
|
|
1610
|
-
"overlaps": overlap_details,
|
|
1611
|
-
}
|
|
1612
|
-
|
|
1613
|
-
|
|
1614
|
-
def should_report_whiteboard_overlap(
|
|
1615
|
-
whiteboard: dict[str, Any],
|
|
1616
|
-
other: dict[str, Any],
|
|
1617
|
-
slide_width: int | float,
|
|
1618
|
-
slide_height: int | float,
|
|
1619
|
-
) -> dict[str, Any] | None:
|
|
1620
|
-
if other is whiteboard or not intersects(whiteboard, other):
|
|
1621
|
-
return None
|
|
1622
|
-
if is_ghost_text(other):
|
|
1623
|
-
return None
|
|
1624
|
-
if contains(whiteboard, other):
|
|
1625
|
-
return None
|
|
1626
|
-
if is_bottom_layer_full_slide_whiteboard(whiteboard, other, slide_width, slide_height):
|
|
1627
|
-
return None
|
|
1628
|
-
if is_background_container_for_whiteboard(other, whiteboard):
|
|
1629
|
-
return None
|
|
1630
|
-
|
|
1631
|
-
overlap_width = intersection_width(whiteboard, other)
|
|
1632
|
-
overlap_height = intersection_height(whiteboard, other)
|
|
1633
|
-
if overlap_width < 8 or overlap_height < 8:
|
|
1634
|
-
return None
|
|
1635
|
-
|
|
1636
|
-
other_area = element_area(other)
|
|
1637
|
-
if other_area <= 0:
|
|
1638
|
-
return None
|
|
1639
|
-
overlap_area = overlap_width * overlap_height
|
|
1640
|
-
overlap_ratio = overlap_area / other_area
|
|
1641
|
-
if overlap_ratio < 0.15:
|
|
1642
|
-
return None
|
|
1643
|
-
|
|
1644
|
-
return {
|
|
1645
|
-
"element": element_ref(other),
|
|
1646
|
-
"kind": other["kind"],
|
|
1647
|
-
"type": other.get("type"),
|
|
1648
|
-
"overlap_width": overlap_width,
|
|
1649
|
-
"overlap_height": overlap_height,
|
|
1650
|
-
"target_overlap_ratio": round(overlap_ratio, 3),
|
|
1651
|
-
}
|
|
1652
|
-
|
|
1653
|
-
|
|
1654
|
-
def prune_contained_text_overlap_details(
|
|
1655
|
-
overlap_details: list[dict[str, Any]], elements_by_ref: dict[str, dict[str, Any]]
|
|
1656
|
-
) -> list[dict[str, Any]]:
|
|
1657
|
-
pruned: list[dict[str, Any]] = []
|
|
1658
|
-
for detail in overlap_details:
|
|
1659
|
-
element = elements_by_ref[detail["element"]]
|
|
1660
|
-
if is_text_element(element):
|
|
1661
|
-
has_reported_container = any(
|
|
1662
|
-
detail["element"] != other_detail["element"]
|
|
1663
|
-
and not is_text_element(elements_by_ref[other_detail["element"]])
|
|
1664
|
-
and contains(elements_by_ref[other_detail["element"]], element)
|
|
1665
|
-
for other_detail in overlap_details
|
|
1666
|
-
)
|
|
1667
|
-
if has_reported_container:
|
|
1668
|
-
continue
|
|
1669
|
-
pruned.append(detail)
|
|
1670
|
-
return pruned
|
|
1671
|
-
|
|
1672
|
-
|
|
1673
|
-
def detect_whiteboard_external_overlaps(
|
|
1674
|
-
elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
|
|
1675
|
-
) -> list[dict[str, Any]]:
|
|
1676
|
-
issues: list[dict[str, Any]] = []
|
|
1677
|
-
elements_by_ref = {element_ref(element): element for element in elements}
|
|
1678
|
-
for whiteboard in [element for element in elements if is_whiteboard_element(element)]:
|
|
1679
|
-
overlap_details = [
|
|
1680
|
-
detail
|
|
1681
|
-
for element in elements
|
|
1682
|
-
if (
|
|
1683
|
-
detail := should_report_whiteboard_overlap(
|
|
1684
|
-
whiteboard,
|
|
1685
|
-
element,
|
|
1686
|
-
slide_width,
|
|
1687
|
-
slide_height,
|
|
1688
|
-
)
|
|
1689
|
-
)
|
|
1690
|
-
is not None
|
|
1691
|
-
]
|
|
1692
|
-
overlap_details = prune_contained_text_overlap_details(overlap_details, elements_by_ref)
|
|
1693
|
-
if overlap_details:
|
|
1694
|
-
issues.append(build_whiteboard_external_overlap_issue(whiteboard, overlap_details))
|
|
1695
|
-
return issues
|
|
1696
|
-
|
|
1697
|
-
|
|
1698
|
-
def element_canvas_bbox(element: dict[str, Any]) -> dict[str, int | float]:
|
|
1699
|
-
bbox = {key: element[key] for key in ("x", "y", "width", "height")}
|
|
1700
|
-
if element["kind"] != "chart" and not (element["kind"] == "shape" and element["type"] == "text"):
|
|
1701
|
-
return bbox
|
|
1702
|
-
rotation = element["rotation"]
|
|
1703
|
-
if not isinstance(rotation, (int, float)) or not math.isfinite(rotation):
|
|
1704
|
-
rotation = 0
|
|
1705
|
-
rotation %= 360
|
|
1706
|
-
if math.isclose(rotation, 0, abs_tol=1e-9):
|
|
1707
|
-
return bbox
|
|
1708
|
-
radians = math.radians(rotation)
|
|
1709
|
-
sine = abs(math.sin(radians))
|
|
1710
|
-
cosine = abs(math.cos(radians))
|
|
1711
|
-
sine = 0 if math.isclose(sine, 0, abs_tol=1e-12) else sine
|
|
1712
|
-
cosine = 0 if math.isclose(cosine, 0, abs_tol=1e-12) else cosine
|
|
1713
|
-
rotated_width = element["width"] * cosine + element["height"] * sine
|
|
1714
|
-
rotated_height = element["width"] * sine + element["height"] * cosine
|
|
1715
|
-
return {
|
|
1716
|
-
"x": element["x"] - (rotated_width - element["width"]) / 2,
|
|
1717
|
-
"y": element["y"] - (rotated_height - element["height"]) / 2,
|
|
1718
|
-
"width": rotated_width,
|
|
1719
|
-
"height": rotated_height,
|
|
1720
|
-
}
|
|
1721
|
-
|
|
1722
|
-
|
|
1723
|
-
def detect_elements_out_of_canvas(
|
|
1724
|
-
elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
|
|
1725
|
-
) -> list[dict[str, Any]]:
|
|
1726
|
-
issues: list[dict[str, Any]] = []
|
|
1727
|
-
for element in (
|
|
1728
|
-
element
|
|
1729
|
-
for element in elements
|
|
1730
|
-
if element["kind"] in {"table", "chart"}
|
|
1731
|
-
or (element["kind"] == "shape" and element["type"] in {"rect", "text"})
|
|
1732
|
-
):
|
|
1733
|
-
bbox = element_canvas_bbox(element)
|
|
1734
|
-
overflow = {
|
|
1735
|
-
"left": max(-bbox["x"], 0),
|
|
1736
|
-
"top": max(-bbox["y"], 0),
|
|
1737
|
-
"right": max(bbox["x"] + bbox["width"] - slide_width, 0),
|
|
1738
|
-
"bottom": max(bbox["y"] + bbox["height"] - slide_height, 0),
|
|
1739
|
-
}
|
|
1740
|
-
overflow_details = [
|
|
1741
|
-
f"{side} by {amount:g}px"
|
|
1742
|
-
for side, amount in overflow.items()
|
|
1743
|
-
if amount > CANVAS_OVERFLOW_TOLERANCE
|
|
1744
|
-
]
|
|
1745
|
-
if not overflow_details:
|
|
1746
|
-
continue
|
|
1747
|
-
issues.append(
|
|
1748
|
-
{
|
|
1749
|
-
"level": "error",
|
|
1750
|
-
"code": f'{element["kind"]}_out_of_canvas',
|
|
1751
|
-
"elements": [element_ref(element)],
|
|
1752
|
-
"canvas": {"width": slide_width, "height": slide_height},
|
|
1753
|
-
"bbox": bbox,
|
|
1754
|
-
"overflow": overflow,
|
|
1755
|
-
"message": (
|
|
1756
|
-
f'{element["kind"]} {element_label(element)} exceeds the {slide_width:g}x{slide_height:g} canvas '
|
|
1757
|
-
f'({", ".join(overflow_details)})'
|
|
1758
|
-
),
|
|
1759
|
-
"hint": (
|
|
1760
|
-
"Move the table inside the canvas, reduce table.width/table.height, or split the table across "
|
|
1761
|
-
"slides."
|
|
1762
|
-
if element["kind"] == "table"
|
|
1763
|
-
else f'Move the {element["kind"]} inside the canvas or reduce its width/height.'
|
|
1764
|
-
),
|
|
1765
|
-
}
|
|
1766
|
-
)
|
|
1767
|
-
return issues
|
|
1768
|
-
|
|
1769
|
-
|
|
1770
|
-
def extract_table_column_sizes(table_xml: str) -> list[int | float | None]:
|
|
1771
|
-
sizes: list[int | float | None] = []
|
|
1772
|
-
for match in re.finditer(r"<col\b([^>]*)/?>", table_xml):
|
|
1773
|
-
attrs = match.group(1)
|
|
1774
|
-
span = extract_numeric_attribute(attrs, "span") or 1
|
|
1775
|
-
span_count = int(span) if math.isfinite(span) and span > 0 and float(span).is_integer() else 1
|
|
1776
|
-
sizes.extend([extract_numeric_attribute(attrs, "width")] * span_count)
|
|
1777
|
-
return sizes
|
|
1778
|
-
|
|
1779
|
-
|
|
1780
|
-
def extract_table_row_sizes(table_xml: str) -> list[int | float | None]:
|
|
1781
|
-
return [extract_numeric_attribute(match.group(1), "height") for match in re.finditer(r"<tr\b([^>]*)>", table_xml)]
|
|
1782
|
-
|
|
1783
|
-
|
|
1784
|
-
def resolve_table_dimension(
|
|
1785
|
-
table_xml: str,
|
|
1786
|
-
declared_size: int | float | None,
|
|
1787
|
-
extract_sizes: Any,
|
|
1788
|
-
default_size: int | float,
|
|
1789
|
-
) -> tuple[int | float | None, dict[str, Any] | None]:
|
|
1790
|
-
input_sizes = extract_sizes(table_xml)
|
|
1791
|
-
if not input_sizes:
|
|
1792
|
-
return declared_size, None
|
|
1793
|
-
layout = solve_weighted_min_layout(
|
|
1794
|
-
input_sizes, default_size, declared_size if is_filled_size(declared_size) else None
|
|
1795
|
-
)
|
|
1796
|
-
return layout["actual_size"], layout
|
|
1797
|
-
|
|
1798
|
-
|
|
1799
|
-
def format_size(size: int | float) -> str:
|
|
1800
|
-
return f"{size:g}"
|
|
1801
|
-
|
|
1802
|
-
|
|
1803
|
-
def detect_table_layout_size_mismatches(elements: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
1804
|
-
issues: list[dict[str, Any]] = []
|
|
1805
|
-
dimensions = {
|
|
1806
|
-
"width": ("col", "column widths"),
|
|
1807
|
-
"height": ("tr", "row heights"),
|
|
1808
|
-
}
|
|
1809
|
-
for table in (element for element in elements if element["kind"] == "table"):
|
|
1810
|
-
for dimension, (child_tag, child_description) in dimensions.items():
|
|
1811
|
-
target_size = table[f"declared_{dimension}"]
|
|
1812
|
-
if not is_filled_size(target_size):
|
|
1813
|
-
continue
|
|
1814
|
-
layout = table["table_layouts"][dimension]
|
|
1815
|
-
if layout is None:
|
|
1816
|
-
continue
|
|
1817
|
-
actual_size = layout["actual_size"]
|
|
1818
|
-
if math.isclose(actual_size, target_size, rel_tol=1e-9, abs_tol=1e-9):
|
|
1819
|
-
continue
|
|
1820
|
-
issues.append(
|
|
1821
|
-
{
|
|
1822
|
-
"level": "info",
|
|
1823
|
-
"code": "table_resolved_size_mismatch",
|
|
1824
|
-
"elements": [element_ref(table)],
|
|
1825
|
-
"dimension": dimension,
|
|
1826
|
-
"declared_size": target_size,
|
|
1827
|
-
"resolved_size": actual_size,
|
|
1828
|
-
"resolved_sizes": layout["final_sizes"],
|
|
1829
|
-
"message": (
|
|
1830
|
-
f'table {element_label(table)} declares {dimension}={format_size(target_size)}px, but its '
|
|
1831
|
-
f"{child_description} resolve to {format_size(actual_size)}px"
|
|
1832
|
-
),
|
|
1833
|
-
"hint": (
|
|
1834
|
-
f"Set table.{dimension} to {format_size(actual_size)}px, or adjust <{child_tag}> sizes "
|
|
1835
|
-
f"so their resolved total matches {format_size(target_size)}px."
|
|
1836
|
-
),
|
|
1837
|
-
}
|
|
1838
|
-
)
|
|
1839
|
-
return issues
|
|
1840
|
-
|
|
1841
|
-
|
|
1842
|
-
def segment_intersects_rect(
|
|
1843
|
-
x1: float, y1: float, x2: float, y2: float, rect: dict[str, int | float]
|
|
1844
|
-
) -> bool:
|
|
1845
|
-
"""True when segment (x1,y1)-(x2,y2) enters the axis-aligned rect (Liang-Barsky clip)."""
|
|
1846
|
-
left = rect["x"]
|
|
1847
|
-
top = rect["y"]
|
|
1848
|
-
right = rect["x"] + rect["width"]
|
|
1849
|
-
bottom = rect["y"] + rect["height"]
|
|
1850
|
-
if right <= left or bottom <= top:
|
|
1851
|
-
return False
|
|
1852
|
-
dx = x2 - x1
|
|
1853
|
-
dy = y2 - y1
|
|
1854
|
-
if dx == 0 and dy == 0:
|
|
1855
|
-
return left <= x1 <= right and top <= y1 <= bottom
|
|
1856
|
-
t_enter, t_exit = 0.0, 1.0
|
|
1857
|
-
for delta, distance in ((-dx, x1 - left), (dx, right - x1), (-dy, y1 - top), (dy, bottom - y1)):
|
|
1858
|
-
if delta == 0:
|
|
1859
|
-
if distance < 0:
|
|
1860
|
-
return False
|
|
1861
|
-
continue
|
|
1862
|
-
t = distance / delta
|
|
1863
|
-
if delta < 0:
|
|
1864
|
-
t_enter = max(t_enter, t)
|
|
1865
|
-
else:
|
|
1866
|
-
t_exit = min(t_exit, t)
|
|
1867
|
-
if t_enter > t_exit:
|
|
1868
|
-
return False
|
|
1869
|
-
return True
|
|
1870
|
-
|
|
1871
|
-
|
|
1872
|
-
def line_text_graze_margin(text_element: dict[str, Any]) -> float:
|
|
1873
|
-
font_size = text_element["fontSize"] if isinstance(text_element.get("fontSize"), (int, float)) else 16
|
|
1874
|
-
return max(font_size * LINE_TEXT_GRAZE_FONT_RATIO, LINE_TEXT_GRAZE_MIN_PX)
|
|
1875
|
-
|
|
1876
|
-
|
|
1877
|
-
def erode_rect(rect: dict[str, int | float], margin: float) -> dict[str, int | float] | None:
|
|
1878
|
-
width = rect["width"] - 2 * margin
|
|
1879
|
-
height = rect["height"] - 2 * margin
|
|
1880
|
-
if width <= 0 or height <= 0:
|
|
1881
|
-
return None
|
|
1882
|
-
return {"x": rect["x"] + margin, "y": rect["y"] + margin, "width": width, "height": height}
|
|
1883
|
-
|
|
1884
|
-
|
|
1885
|
-
def line_crosses_text(line: dict[str, Any], text_element: dict[str, Any]) -> bool:
|
|
1886
|
-
if not is_visually_rendered(line) or line.get("alpha", 1) < LINE_MIN_VISIBLE_ALPHA:
|
|
1887
|
-
return False
|
|
1888
|
-
if not is_text_element(text_element) or not has_text_content(text_element):
|
|
1889
|
-
return False
|
|
1890
|
-
if is_ghost_text(text_element) or is_decorative_text(text_element):
|
|
1891
|
-
return False
|
|
1892
|
-
glyph_bbox = estimate_text_visual_bbox(text_element)
|
|
1893
|
-
if glyph_bbox is None:
|
|
1894
|
-
return False
|
|
1895
|
-
# Erode the glyph box so a line skimming the letter edge or only clipping the padding-only text
|
|
1896
|
-
# frame is exempt; only a line that actually cuts through the letterforms is a crossing.
|
|
1897
|
-
target = erode_rect(glyph_bbox, line_text_graze_margin(text_element))
|
|
1898
|
-
if target is None:
|
|
1899
|
-
return False
|
|
1900
|
-
return segment_intersects_rect(
|
|
1901
|
-
line["startX"], line["startY"], line["endX"], line["endY"], target
|
|
1902
|
-
)
|
|
1903
|
-
|
|
1904
|
-
|
|
1905
|
-
def detect_line_text_crossings(
|
|
1906
|
-
slide_xml: str, elements: list[dict[str, Any]], slide_number: int
|
|
1907
|
-
) -> list[dict[str, Any]]:
|
|
1908
|
-
lines = extract_line_elements(slide_xml)
|
|
1909
|
-
if not lines:
|
|
1910
|
-
return []
|
|
1911
|
-
attach_source_xml_paths(lines, build_source_xml_paths(slide_xml, slide_number))
|
|
1912
|
-
text_elements = [element for element in elements if is_text_element(element)]
|
|
1913
|
-
issues: list[dict[str, Any]] = []
|
|
1914
|
-
for line in lines:
|
|
1915
|
-
for text_element in text_elements:
|
|
1916
|
-
if not line_crosses_text(line, text_element):
|
|
1917
|
-
continue
|
|
1918
|
-
issues.append(
|
|
1919
|
-
{
|
|
1920
|
-
"level": "error",
|
|
1921
|
-
"code": "bbox_overlap",
|
|
1922
|
-
"elements": [element_ref(line), element_ref(text_element)],
|
|
1923
|
-
"message": (
|
|
1924
|
-
f"line {element_label(line)} crosses text {element_label(text_element)}"
|
|
1925
|
-
),
|
|
1926
|
-
"hint": "Move the line off the text glyphs so it no longer cuts through the letterforms.",
|
|
1927
|
-
}
|
|
1928
|
-
)
|
|
1929
|
-
return issues
|
|
1930
|
-
|
|
1931
|
-
|
|
1932
|
-
def lint_slide(
|
|
1933
|
-
slide_xml: str, slide_number: int, slide_width: int | float = 960, slide_height: int | float = 540
|
|
1934
|
-
) -> dict[str, Any]:
|
|
1935
|
-
elements = extract_elements(slide_xml)
|
|
1936
|
-
attach_source_xml_paths(elements, build_source_xml_paths(slide_xml, slide_number))
|
|
1937
|
-
issues: list[dict[str, Any]] = [
|
|
1938
|
-
*detect_whiteboard_external_overlaps(elements, slide_width, slide_height),
|
|
1939
|
-
*detect_elements_out_of_canvas(elements, slide_width, slide_height),
|
|
1940
|
-
*detect_table_layout_size_mismatches(elements),
|
|
1941
|
-
*detect_text_may_overflow_shapes(elements),
|
|
1942
|
-
*detect_image_text_occlusions(elements),
|
|
1943
|
-
*detect_line_text_crossings(slide_xml, elements, slide_number),
|
|
1944
|
-
]
|
|
1945
|
-
|
|
1946
|
-
for index, left in enumerate(elements):
|
|
1947
|
-
for right in elements[index + 1 :]:
|
|
1948
|
-
horizontal_overflow = should_flag_horizontal_text_overflow(left, right)
|
|
1949
|
-
if not horizontal_overflow and (not intersects(left, right) or not should_flag_overlap(left, right)):
|
|
1950
|
-
continue
|
|
1951
|
-
issues.append(
|
|
1952
|
-
{
|
|
1953
|
-
"level": "error",
|
|
1954
|
-
"code": "bbox_overlap",
|
|
1955
|
-
"elements": [element_ref(left), element_ref(right)],
|
|
1956
|
-
"message": f"{element_label(left)} overlaps {element_label(right)}",
|
|
1957
|
-
"hint": "Move or resize the elements so their visual bounds no longer intersect.",
|
|
1958
|
-
**(
|
|
1959
|
-
{"measurement": horizontal_text_overflow_measurement(left, right)}
|
|
1960
|
-
if horizontal_overflow
|
|
1961
|
-
else {}
|
|
1962
|
-
),
|
|
1963
|
-
}
|
|
1964
|
-
)
|
|
1965
|
-
|
|
1966
|
-
return {
|
|
1967
|
-
"slide_number": slide_number,
|
|
1968
|
-
"element_count": len(elements),
|
|
1969
|
-
"elements": elements,
|
|
1970
|
-
"issues": issues,
|
|
1971
|
-
}
|
|
1972
|
-
|
|
1973
|
-
|
|
1974
|
-
|
|
1975
|
-
MIN_CONTAINER_WIDTH = 140
|
|
1976
|
-
MIN_CONTAINER_HEIGHT = 160
|
|
1977
|
-
MIN_SHORT_CARD_HEIGHT = 80
|
|
1978
|
-
MIN_CONTAINER_AREA = 20_000
|
|
1979
|
-
MIN_CONTENT_COVERAGE_RATIO = 0.15
|
|
1980
|
-
MIN_SLIDE_CONTENT_COVERAGE_RATIO = 0.035
|
|
1981
|
-
MIN_SLIDE_CONTENT_ELEMENT_COUNT = 4
|
|
1982
|
-
SHORT_CARD_SIZE_TOLERANCE_RATIO = 0.10
|
|
1983
|
-
MIN_SIMILAR_SHORT_CARD_COUNT = 2
|
|
1984
|
-
LARGE_VISUAL_CHILD_RATIO = 0.35
|
|
1985
|
-
LAYOUT_PANEL_SPAN_RATIO = 0.90
|
|
1986
|
-
IMAGE_OVERLAY_MATCH_RATIO = 0.90
|
|
1987
|
-
DENSITY_CONTAINMENT_TOLERANCE = 8
|
|
1988
|
-
|
|
1989
|
-
|
|
1990
|
-
def clipped_bbox(element: dict[str, Any], container: dict[str, Any]) -> dict[str, int | float] | None:
|
|
1991
|
-
left = max(element["x"], container["x"])
|
|
1992
|
-
top = max(element["y"], container["y"])
|
|
1993
|
-
right = min(element["x"] + element["width"], container["x"] + container["width"])
|
|
1994
|
-
bottom = min(element["y"] + element["height"], container["y"] + container["height"])
|
|
1995
|
-
if right <= left or bottom <= top:
|
|
1996
|
-
return None
|
|
1997
|
-
return {"x": left, "y": top, "width": right - left, "height": bottom - top}
|
|
1998
|
-
|
|
1999
|
-
|
|
2000
|
-
def rectangle_union_area(rectangles: list[dict[str, int | float]]) -> int | float:
|
|
2001
|
-
x_coordinates = sorted({coordinate for rect in rectangles for coordinate in (rect["x"], rect["x"] + rect["width"])})
|
|
2002
|
-
area = 0
|
|
2003
|
-
for left, right in zip(x_coordinates, x_coordinates[1:]):
|
|
2004
|
-
intervals = sorted(
|
|
2005
|
-
(rect["y"], rect["y"] + rect["height"])
|
|
2006
|
-
for rect in rectangles
|
|
2007
|
-
if rect["x"] < right and rect["x"] + rect["width"] > left
|
|
2008
|
-
)
|
|
2009
|
-
covered_height = 0
|
|
2010
|
-
interval_end: int | float | None = None
|
|
2011
|
-
for top, bottom in intervals:
|
|
2012
|
-
if interval_end is None:
|
|
2013
|
-
covered_height += bottom - top
|
|
2014
|
-
interval_end = bottom
|
|
2015
|
-
elif bottom > interval_end:
|
|
2016
|
-
covered_height += bottom - max(top, interval_end)
|
|
2017
|
-
interval_end = bottom
|
|
2018
|
-
area += (right - left) * covered_height
|
|
2019
|
-
return area
|
|
2020
|
-
|
|
2021
|
-
|
|
2022
|
-
def has_similar_short_card_peer(element: dict[str, Any], elements: list[dict[str, Any]]) -> bool:
|
|
2023
|
-
return sum(
|
|
2024
|
-
other is not element
|
|
2025
|
-
and is_visually_rendered(other)
|
|
2026
|
-
and other["kind"] == "shape"
|
|
2027
|
-
and other["type"] == "rect"
|
|
2028
|
-
and other["width"] >= MIN_CONTAINER_WIDTH
|
|
2029
|
-
and other["height"] >= MIN_SHORT_CARD_HEIGHT
|
|
2030
|
-
and element_area(other) >= MIN_CONTAINER_AREA
|
|
2031
|
-
and abs(other["width"] - element["width"]) / max(other["width"], element["width"])
|
|
2032
|
-
<= SHORT_CARD_SIZE_TOLERANCE_RATIO
|
|
2033
|
-
and abs(other["height"] - element["height"]) / max(other["height"], element["height"])
|
|
2034
|
-
<= SHORT_CARD_SIZE_TOLERANCE_RATIO
|
|
2035
|
-
for other in elements
|
|
2036
|
-
) >= MIN_SIMILAR_SHORT_CARD_COUNT
|
|
2037
|
-
|
|
2038
|
-
|
|
2039
|
-
def is_layout_container(
|
|
2040
|
-
element: dict[str, Any],
|
|
2041
|
-
slide_width: int | float,
|
|
2042
|
-
slide_height: int | float,
|
|
2043
|
-
elements: list[dict[str, Any]] | None = None,
|
|
2044
|
-
) -> bool:
|
|
2045
|
-
has_supported_height = element["height"] >= MIN_CONTAINER_HEIGHT or (
|
|
2046
|
-
elements is not None
|
|
2047
|
-
and element["height"] >= MIN_SHORT_CARD_HEIGHT
|
|
2048
|
-
and has_similar_short_card_peer(element, elements)
|
|
2049
|
-
)
|
|
2050
|
-
return (
|
|
2051
|
-
element["kind"] == "shape"
|
|
2052
|
-
and element["type"] == "rect"
|
|
2053
|
-
and is_visually_rendered(element)
|
|
2054
|
-
and element["width"] >= MIN_CONTAINER_WIDTH
|
|
2055
|
-
and has_supported_height
|
|
2056
|
-
and element_area(element) >= MIN_CONTAINER_AREA
|
|
2057
|
-
and not (
|
|
2058
|
-
element["x"] <= 2
|
|
2059
|
-
and element["y"] <= 2
|
|
2060
|
-
and element["width"] >= slide_width - 4
|
|
2061
|
-
and element["height"] >= slide_height - 4
|
|
2062
|
-
)
|
|
2063
|
-
)
|
|
2064
|
-
|
|
2065
|
-
|
|
2066
|
-
def is_edge_spanning_layout_panel(
|
|
2067
|
-
element: dict[str, Any], slide_width: int | float, slide_height: int | float
|
|
2068
|
-
) -> bool:
|
|
2069
|
-
touches_horizontal_edge = element["x"] <= 2 or element["x"] + element["width"] >= slide_width - 2
|
|
2070
|
-
touches_vertical_edge = element["y"] <= 2 or element["y"] + element["height"] >= slide_height - 2
|
|
2071
|
-
return (touches_horizontal_edge and element["height"] >= slide_height * LAYOUT_PANEL_SPAN_RATIO) or (
|
|
2072
|
-
touches_vertical_edge and element["width"] >= slide_width * LAYOUT_PANEL_SPAN_RATIO
|
|
2073
|
-
)
|
|
2074
|
-
|
|
2075
|
-
|
|
2076
|
-
def has_matching_image_overlay(container: dict[str, Any], elements: list[dict[str, Any]]) -> bool:
|
|
2077
|
-
container_area = element_area(container)
|
|
2078
|
-
return any(
|
|
2079
|
-
element["kind"] == "img"
|
|
2080
|
-
and is_visually_rendered(element)
|
|
2081
|
-
and intersection_area(container, element) / max(1, container_area) >= IMAGE_OVERLAY_MATCH_RATIO
|
|
2082
|
-
for element in elements
|
|
2083
|
-
)
|
|
2084
|
-
|
|
2085
|
-
|
|
2086
|
-
def is_nested_in_layout_panel(
|
|
2087
|
-
container: dict[str, Any], elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
|
|
2088
|
-
) -> bool:
|
|
2089
|
-
return any(
|
|
2090
|
-
element is not container
|
|
2091
|
-
and element["kind"] == "shape"
|
|
2092
|
-
and element["type"] == "rect"
|
|
2093
|
-
and is_visually_rendered(element)
|
|
2094
|
-
and is_edge_spanning_layout_panel(element, slide_width, slide_height)
|
|
2095
|
-
and contains(element, container, tolerance=DENSITY_CONTAINMENT_TOLERANCE)
|
|
2096
|
-
for element in elements
|
|
2097
|
-
)
|
|
2098
|
-
|
|
2099
|
-
|
|
2100
|
-
def extract_density_elements(slide_xml: str, slide_number: int = 1) -> list[dict[str, Any]]:
|
|
2101
|
-
elements = extract_elements(slide_xml)
|
|
2102
|
-
source_paths = build_source_xml_paths(slide_xml, slide_number)
|
|
2103
|
-
attach_source_xml_paths(elements, source_paths)
|
|
2104
|
-
shape_elements_by_index = {
|
|
2105
|
-
element["_source_kind_index"]: element
|
|
2106
|
-
for element in elements
|
|
2107
|
-
if element["kind"] == "shape"
|
|
2108
|
-
}
|
|
2109
|
-
root = ET.fromstring(slide_xml)
|
|
2110
|
-
shape_index = 0
|
|
2111
|
-
for node in root.iter():
|
|
2112
|
-
if xml_local_name(node.tag) != "shape":
|
|
2113
|
-
continue
|
|
2114
|
-
shape_index += 1
|
|
2115
|
-
element = shape_elements_by_index.get(shape_index)
|
|
2116
|
-
if element is None:
|
|
2117
|
-
continue
|
|
2118
|
-
content_node = next(
|
|
2119
|
-
(child for child in node if xml_local_name(child.tag) == "content"),
|
|
2120
|
-
None,
|
|
2121
|
-
)
|
|
2122
|
-
paragraphs = (
|
|
2123
|
-
[
|
|
2124
|
-
" ".join("".join(paragraph.itertext()).split())
|
|
2125
|
-
for paragraph in content_node.iter()
|
|
2126
|
-
if xml_local_name(paragraph.tag) == "p"
|
|
2127
|
-
]
|
|
2128
|
-
if content_node is not None
|
|
2129
|
-
else []
|
|
2130
|
-
)
|
|
2131
|
-
raw_font_size = (
|
|
2132
|
-
content_node.attrib.get("fontSize") if content_node is not None else None
|
|
2133
|
-
) or node.attrib.get("fontSize")
|
|
2134
|
-
try:
|
|
2135
|
-
base_font_size = float(raw_font_size or 16)
|
|
2136
|
-
except ValueError:
|
|
2137
|
-
base_font_size = 16.0
|
|
2138
|
-
element.update(
|
|
2139
|
-
{
|
|
2140
|
-
"textType": content_node.attrib.get("textType") if content_node is not None else None,
|
|
2141
|
-
"textAlign": content_node.attrib.get("textAlign") if content_node is not None else None,
|
|
2142
|
-
"autoFit": content_node.attrib.get("autoFit") if content_node is not None else None,
|
|
2143
|
-
"fontSize": base_font_size,
|
|
2144
|
-
"text": "\n".join(paragraph for paragraph in paragraphs if paragraph),
|
|
2145
|
-
}
|
|
2146
|
-
)
|
|
2147
|
-
if not has_text_content(element):
|
|
2148
|
-
continue
|
|
2149
|
-
declared_font_sizes = []
|
|
2150
|
-
for descendant in node.iter():
|
|
2151
|
-
raw_declared_font_size = descendant.attrib.get("fontSize")
|
|
2152
|
-
if raw_declared_font_size is None:
|
|
2153
|
-
continue
|
|
2154
|
-
try:
|
|
2155
|
-
declared_font_sizes.append(float(raw_declared_font_size))
|
|
2156
|
-
except ValueError:
|
|
2157
|
-
continue
|
|
2158
|
-
if declared_font_sizes:
|
|
2159
|
-
element["fontSize"] = max(declared_font_sizes)
|
|
2160
|
-
for source_kind_index, match in enumerate(
|
|
2161
|
-
re.finditer(r"<icon\b([^>]*)>", slide_xml), start=1
|
|
2162
|
-
):
|
|
2163
|
-
attrs = match.group(1)
|
|
2164
|
-
source_id = extract_attribute(attrs, "id") or None
|
|
2165
|
-
x = extract_numeric_attribute(attrs, "topLeftX")
|
|
2166
|
-
y = extract_numeric_attribute(attrs, "topLeftY")
|
|
2167
|
-
width = extract_numeric_attribute(attrs, "width")
|
|
2168
|
-
height = extract_numeric_attribute(attrs, "height")
|
|
2169
|
-
if any(value is None for value in (x, y, width, height)):
|
|
2170
|
-
continue
|
|
2171
|
-
icon_alpha = extract_numeric_attribute(attrs, "alpha")
|
|
2172
|
-
elements.append(
|
|
2173
|
-
{
|
|
2174
|
-
"id": source_id or f"icon-{len(elements) + 1}",
|
|
2175
|
-
"_source_id": source_id,
|
|
2176
|
-
"kind": "icon",
|
|
2177
|
-
"type": "icon",
|
|
2178
|
-
"x": x,
|
|
2179
|
-
"y": y,
|
|
2180
|
-
"width": width,
|
|
2181
|
-
"height": height,
|
|
2182
|
-
"rotation": extract_numeric_attribute(attrs, "rotation") or 0,
|
|
2183
|
-
"alpha": icon_alpha if icon_alpha is not None else 1,
|
|
2184
|
-
"order": len(elements),
|
|
2185
|
-
"_source_kind_index": source_kind_index,
|
|
2186
|
-
}
|
|
2187
|
-
)
|
|
2188
|
-
for source_kind_index, match in enumerate(
|
|
2189
|
-
re.finditer(r"<polyline\b([^>]*)>", slide_xml), start=1
|
|
2190
|
-
):
|
|
2191
|
-
attrs = match.group(1)
|
|
2192
|
-
x = extract_numeric_attribute(attrs, "topLeftX")
|
|
2193
|
-
y = extract_numeric_attribute(attrs, "topLeftY")
|
|
2194
|
-
width = extract_numeric_attribute(attrs, "width")
|
|
2195
|
-
height = extract_numeric_attribute(attrs, "height")
|
|
2196
|
-
if any(value is None for value in (x, y, width, height)):
|
|
2197
|
-
continue
|
|
2198
|
-
polyline_alpha = extract_numeric_attribute(attrs, "alpha")
|
|
2199
|
-
source_id = extract_attribute(attrs, "id") or None
|
|
2200
|
-
elements.append(
|
|
2201
|
-
{
|
|
2202
|
-
"id": source_id or f"polyline-{len(elements) + 1}",
|
|
2203
|
-
"_source_id": source_id,
|
|
2204
|
-
"kind": "polyline",
|
|
2205
|
-
"type": "polyline",
|
|
2206
|
-
"x": x,
|
|
2207
|
-
"y": y,
|
|
2208
|
-
"width": width,
|
|
2209
|
-
"height": height,
|
|
2210
|
-
"rotation": extract_numeric_attribute(attrs, "rotation") or 0,
|
|
2211
|
-
"alpha": polyline_alpha if polyline_alpha is not None else 1,
|
|
2212
|
-
"order": len(elements),
|
|
2213
|
-
"_source_kind_index": source_kind_index,
|
|
2214
|
-
}
|
|
2215
|
-
)
|
|
2216
|
-
for line_element in extract_line_elements(slide_xml):
|
|
2217
|
-
line_element["order"] = len(elements)
|
|
2218
|
-
elements.append(line_element)
|
|
2219
|
-
attach_source_xml_paths(elements, source_paths)
|
|
2220
|
-
for element in elements:
|
|
2221
|
-
element["_slide_number"] = slide_number
|
|
2222
|
-
return elements
|
|
2223
|
-
|
|
2224
|
-
|
|
2225
|
-
def is_visually_rendered(element: dict[str, Any]) -> bool:
|
|
2226
|
-
return element.get("alpha", 1) > 0
|
|
2227
|
-
|
|
2228
|
-
|
|
2229
|
-
def visual_bbox(element: dict[str, Any], container: dict[str, Any]) -> dict[str, int | float] | None:
|
|
2230
|
-
if not is_visually_rendered(element):
|
|
2231
|
-
return None
|
|
2232
|
-
if is_text_element(element):
|
|
2233
|
-
estimated = estimate_text_visual_bbox(element)
|
|
2234
|
-
return clipped_bbox(estimated, container) if estimated else None
|
|
2235
|
-
return clipped_bbox(element, container)
|
|
2236
|
-
|
|
2237
|
-
|
|
2238
|
-
def own_text_visual_bbox(container: dict[str, Any]) -> dict[str, int | float] | None:
|
|
2239
|
-
if container["kind"] != "shape" or not has_text_content(container):
|
|
2240
|
-
return None
|
|
2241
|
-
text_proxy = {**container, "type": "text"}
|
|
2242
|
-
estimated = estimate_text_visual_bbox(text_proxy)
|
|
2243
|
-
return clipped_bbox(estimated, container) if estimated else None
|
|
2244
|
-
|
|
2245
|
-
|
|
2246
|
-
def slide_content_visual_bbox(
|
|
2247
|
-
element: dict[str, Any], slide_bbox: dict[str, int | float]
|
|
2248
|
-
) -> dict[str, int | float] | None:
|
|
2249
|
-
if not is_visually_rendered(element):
|
|
2250
|
-
return None
|
|
2251
|
-
if is_text_element(element):
|
|
2252
|
-
estimated = estimate_text_visual_bbox(element)
|
|
2253
|
-
return clipped_bbox(estimated, slide_bbox) if estimated else None
|
|
2254
|
-
if element["kind"] == "shape" and has_text_content(element):
|
|
2255
|
-
estimated = own_text_visual_bbox(element)
|
|
2256
|
-
return clipped_bbox(estimated, slide_bbox) if estimated else None
|
|
2257
|
-
if element["kind"] == "line":
|
|
2258
|
-
# a straight horizontal/vertical line has zero width or height in one axis; clipped_bbox
|
|
2259
|
-
# treats zero-area rects as invisible, so pad to its rendered stroke thickness instead.
|
|
2260
|
-
return clipped_bbox(line_stroke_bbox(element), slide_bbox)
|
|
2261
|
-
if element["kind"] in {"img", "chart", "table", "whiteboard", "embed", "icon", "polyline"}:
|
|
2262
|
-
return clipped_bbox(element, slide_bbox)
|
|
2263
|
-
return None
|
|
2264
|
-
|
|
2265
|
-
|
|
2266
|
-
def line_stroke_bbox(element: dict[str, Any]) -> dict[str, Any]:
|
|
2267
|
-
return {**element, "width": max(element["width"], 1), "height": max(element["height"], 1)}
|
|
2268
|
-
|
|
2269
|
-
|
|
2270
|
-
def is_slide_content_present(
|
|
2271
|
-
element: dict[str, Any], slide_bbox: dict[str, int | float]
|
|
2272
|
-
) -> bool:
|
|
2273
|
-
# Deliberately permissive, unlike slide_content_visual_bbox: blank_slide is asking "is
|
|
2274
|
-
# *anything* rendered here", not the richer "counts toward meaningful content density" bar
|
|
2275
|
-
# that sparse_slide_content/sparse_container_content apply. A plain shape with no text (a
|
|
2276
|
-
# decorative rect/ellipse/etc.), <undefined>, or any future SXSD data element should all
|
|
2277
|
-
# count here — deny-list only what's actually invisible (alpha<=0 or zero on-canvas area)
|
|
2278
|
-
# instead of maintaining an allow-list that silently treats unlisted kinds as blank.
|
|
2279
|
-
if not is_visually_rendered(element):
|
|
2280
|
-
return False
|
|
2281
|
-
if (
|
|
2282
|
-
element["kind"] == "shape"
|
|
2283
|
-
and element["type"] == "rect"
|
|
2284
|
-
and not has_text_content(element)
|
|
2285
|
-
and element["x"] <= 2
|
|
2286
|
-
and element["y"] <= 2
|
|
2287
|
-
and element["width"] >= slide_bbox["width"] - 4
|
|
2288
|
-
and element["height"] >= slide_bbox["height"] - 4
|
|
2289
|
-
):
|
|
2290
|
-
# A full-canvas plain rect is a background panel, not content -- same reasoning as
|
|
2291
|
-
# is_layout_container's existing background exclusion. A slide with nothing else on it
|
|
2292
|
-
# is still effectively blank.
|
|
2293
|
-
return False
|
|
2294
|
-
bbox = line_stroke_bbox(element) if element["kind"] == "line" else element
|
|
2295
|
-
return clipped_bbox(bbox, slide_bbox) is not None
|
|
2296
|
-
|
|
2297
|
-
|
|
2298
|
-
def is_large_visual_child(element: dict[str, Any], container: dict[str, Any]) -> bool:
|
|
2299
|
-
if element["kind"] not in {"img", "chart", "table", "whiteboard", "embed"}:
|
|
2300
|
-
return False
|
|
2301
|
-
if not is_visually_rendered(element):
|
|
2302
|
-
return False
|
|
2303
|
-
return element_area(element) / element_area(container) >= LARGE_VISUAL_CHILD_RATIO
|
|
2304
|
-
|
|
2305
|
-
|
|
2306
|
-
def detect_sparse_container_content(
|
|
2307
|
-
elements: list[dict[str, Any]], slide_number: int, slide_width: int | float, slide_height: int | float
|
|
2308
|
-
) -> list[dict[str, Any]]:
|
|
2309
|
-
issues: list[dict[str, Any]] = []
|
|
2310
|
-
for container in (
|
|
2311
|
-
element for element in elements if is_layout_container(element, slide_width, slide_height, elements)
|
|
2312
|
-
):
|
|
2313
|
-
if (
|
|
2314
|
-
is_edge_spanning_layout_panel(container, slide_width, slide_height)
|
|
2315
|
-
or is_nested_in_layout_panel(container, elements, slide_width, slide_height)
|
|
2316
|
-
or has_matching_image_overlay(container, elements)
|
|
2317
|
-
):
|
|
2318
|
-
continue
|
|
2319
|
-
children = [
|
|
2320
|
-
element
|
|
2321
|
-
for element in elements
|
|
2322
|
-
if element is not container
|
|
2323
|
-
and contains(container, element, tolerance=DENSITY_CONTAINMENT_TOLERANCE)
|
|
2324
|
-
]
|
|
2325
|
-
if any(is_large_visual_child(child, container) for child in children):
|
|
2326
|
-
continue
|
|
2327
|
-
own_text_bbox = own_text_visual_bbox(container)
|
|
2328
|
-
rectangles = ([own_text_bbox] if own_text_bbox else []) + [
|
|
2329
|
-
bbox for child in children if (bbox := visual_bbox(child, container)) is not None
|
|
2330
|
-
]
|
|
2331
|
-
content_area = rectangle_union_area(rectangles) if rectangles else 0
|
|
2332
|
-
coverage_ratio = content_area / element_area(container)
|
|
2333
|
-
if coverage_ratio >= MIN_CONTENT_COVERAGE_RATIO:
|
|
2334
|
-
continue
|
|
2335
|
-
issues.append(
|
|
2336
|
-
{
|
|
2337
|
-
"level": "warning",
|
|
2338
|
-
"code": "sparse_container_content",
|
|
2339
|
-
"target": {
|
|
2340
|
-
"slide_number": slide_number,
|
|
2341
|
-
**(
|
|
2342
|
-
{"container_id": source_element_id(container)}
|
|
2343
|
-
if source_element_id(container) is not None
|
|
2344
|
-
else {}
|
|
2345
|
-
),
|
|
2346
|
-
"container_xml_path": element_ref(container),
|
|
2347
|
-
"container_type": container["type"],
|
|
2348
|
-
"bbox": {key: container[key] for key in ("x", "y", "width", "height")},
|
|
2349
|
-
},
|
|
2350
|
-
"rule": {
|
|
2351
|
-
"name": "large_container_visible_content_coverage",
|
|
2352
|
-
"threshold": MIN_CONTENT_COVERAGE_RATIO,
|
|
2353
|
-
"comparison": "content_coverage_ratio < threshold",
|
|
2354
|
-
},
|
|
2355
|
-
"measurement": {
|
|
2356
|
-
"container_area": element_area(container),
|
|
2357
|
-
"visible_content_area": round(content_area, 3),
|
|
2358
|
-
"content_coverage_ratio": round(coverage_ratio, 3),
|
|
2359
|
-
"content_element_count": len(children) + (1 if own_text_bbox else 0),
|
|
2360
|
-
},
|
|
2361
|
-
"elements": [
|
|
2362
|
-
element_ref(container),
|
|
2363
|
-
*[element_ref(child) for child in children],
|
|
2364
|
-
],
|
|
2365
|
-
}
|
|
2366
|
-
)
|
|
2367
|
-
return issues
|
|
2368
|
-
|
|
2369
|
-
|
|
2370
|
-
def detect_sparse_slide_content(
|
|
2371
|
-
elements: list[dict[str, Any]], slide_number: int, slide_width: int | float, slide_height: int | float
|
|
2372
|
-
) -> list[dict[str, Any]]:
|
|
2373
|
-
slide_bbox = {"x": 0, "y": 0, "width": slide_width, "height": slide_height}
|
|
2374
|
-
content = [
|
|
2375
|
-
(element, bbox)
|
|
2376
|
-
for element in elements
|
|
2377
|
-
if (bbox := slide_content_visual_bbox(element, slide_bbox)) is not None
|
|
2378
|
-
]
|
|
2379
|
-
if len(content) < MIN_SLIDE_CONTENT_ELEMENT_COUNT:
|
|
2380
|
-
return []
|
|
2381
|
-
content_area = rectangle_union_area([bbox for _, bbox in content])
|
|
2382
|
-
slide_area = slide_width * slide_height
|
|
2383
|
-
coverage_ratio = content_area / slide_area
|
|
2384
|
-
if coverage_ratio >= MIN_SLIDE_CONTENT_COVERAGE_RATIO:
|
|
2385
|
-
return []
|
|
2386
|
-
return [
|
|
2387
|
-
{
|
|
2388
|
-
"level": "warning",
|
|
2389
|
-
"code": "sparse_slide_content",
|
|
2390
|
-
"target": {
|
|
2391
|
-
"slide_number": slide_number,
|
|
2392
|
-
"bbox": slide_bbox,
|
|
2393
|
-
},
|
|
2394
|
-
"rule": {
|
|
2395
|
-
"name": "slide_visible_content_coverage",
|
|
2396
|
-
"threshold": MIN_SLIDE_CONTENT_COVERAGE_RATIO,
|
|
2397
|
-
"comparison": "content_coverage_ratio < threshold",
|
|
2398
|
-
},
|
|
2399
|
-
"measurement": {
|
|
2400
|
-
"slide_area": slide_area,
|
|
2401
|
-
"visible_content_area": round(content_area, 3),
|
|
2402
|
-
"content_coverage_ratio": round(coverage_ratio, 3),
|
|
2403
|
-
"content_element_count": len(content),
|
|
2404
|
-
},
|
|
2405
|
-
"elements": [element_ref(element) for element, _ in content],
|
|
2406
|
-
}
|
|
2407
|
-
]
|
|
2408
|
-
|
|
2409
|
-
|
|
2410
|
-
def detect_blank_slide(
|
|
2411
|
-
elements: list[dict[str, Any]],
|
|
2412
|
-
slide_number: int,
|
|
2413
|
-
slide_width: int | float,
|
|
2414
|
-
slide_height: int | float,
|
|
2415
|
-
) -> list[dict[str, Any]]:
|
|
2416
|
-
slide_bbox = {"x": 0, "y": 0, "width": slide_width, "height": slide_height}
|
|
2417
|
-
visible_elements = [
|
|
2418
|
-
element for element in elements if is_slide_content_present(element, slide_bbox)
|
|
2419
|
-
]
|
|
2420
|
-
if visible_elements:
|
|
2421
|
-
return []
|
|
2422
|
-
return [
|
|
2423
|
-
{
|
|
2424
|
-
"level": "error",
|
|
2425
|
-
"code": "blank_slide",
|
|
2426
|
-
"schema_version": "2.0",
|
|
2427
|
-
"target": {"slide_number": slide_number},
|
|
2428
|
-
"rule": {
|
|
2429
|
-
"name": "slide_has_visible_content",
|
|
2430
|
-
"comparison": "visible_element_count == 0",
|
|
2431
|
-
},
|
|
2432
|
-
"measurement": {
|
|
2433
|
-
"visible_element_count": 0,
|
|
2434
|
-
"declared_element_count": len(elements),
|
|
2435
|
-
},
|
|
2436
|
-
"elements": [element_ref(element) for element in elements],
|
|
2437
|
-
"message": "slide has no visible content beyond empty layout shapes",
|
|
2438
|
-
"hint": "Add visible text, an image, a chart, a table, a whiteboard, or an icon before creating the slide.",
|
|
2439
|
-
}
|
|
2440
|
-
]
|
|
2441
|
-
|
|
2442
|
-
|
|
2443
|
-
def detect_duplicate_element_ids(
|
|
2444
|
-
elements: list[dict[str, Any]], *, cross_slide_only: bool = False
|
|
2445
|
-
) -> list[dict[str, Any]]:
|
|
2446
|
-
elements_by_source_id: dict[str, list[dict[str, Any]]] = {}
|
|
2447
|
-
for element in elements:
|
|
2448
|
-
source_id = source_element_id(element)
|
|
2449
|
-
if source_id is not None:
|
|
2450
|
-
elements_by_source_id.setdefault(source_id, []).append(element)
|
|
2451
|
-
return [
|
|
2452
|
-
{
|
|
2453
|
-
"level": "error",
|
|
2454
|
-
"code": "duplicate_element_id",
|
|
2455
|
-
"elements": [element_ref(element) for element in duplicates],
|
|
2456
|
-
"measurement": {
|
|
2457
|
-
"element_id": source_id,
|
|
2458
|
-
"duplicate_count": len(duplicates),
|
|
2459
|
-
},
|
|
2460
|
-
"message": f'element id "{source_id}" is used by {len(duplicates)} elements',
|
|
2461
|
-
"hint": (
|
|
2462
|
-
"Do not invent replacement IDs. For newly authored elements, remove the id attribute. "
|
|
2463
|
-
"When updating read-back XML, keep the server ID on the original element only and remove it "
|
|
2464
|
-
"from copied or new elements."
|
|
2465
|
-
),
|
|
2466
|
-
}
|
|
2467
|
-
for source_id, duplicates in elements_by_source_id.items()
|
|
2468
|
-
if len(duplicates) > 1
|
|
2469
|
-
and (
|
|
2470
|
-
not cross_slide_only
|
|
2471
|
-
or len({element.get("_slide_number") for element in duplicates}) > 1
|
|
2472
|
-
)
|
|
2473
|
-
]
|
|
2474
|
-
|
|
2475
|
-
|
|
2476
|
-
RULE_METADATA: dict[str, dict[str, Any]] = {
|
|
2477
|
-
"xml_not_well_formed": {
|
|
2478
|
-
"name": "xml_is_well_formed",
|
|
2479
|
-
"comparison": "xml_parse_error == false",
|
|
2480
|
-
},
|
|
2481
|
-
"sml_prefixed_tag": {
|
|
2482
|
-
"name": "sml_uses_default_namespace",
|
|
2483
|
-
"comparison": "prefixed_sml_tag_count == 0",
|
|
2484
|
-
},
|
|
2485
|
-
"sxsd_unsupported_tag": {
|
|
2486
|
-
"name": "tag_is_supported_by_slides_xml_schema",
|
|
2487
|
-
"comparison": "unsupported_tag_count == 0",
|
|
2488
|
-
},
|
|
2489
|
-
"sxsd_unsupported_attr": {
|
|
2490
|
-
"name": "attribute_is_supported_by_slides_xml_schema",
|
|
2491
|
-
"comparison": "unsupported_attribute_count == 0",
|
|
2492
|
-
},
|
|
2493
|
-
"icon_missing_fill_color": {
|
|
2494
|
-
"name": "icon_has_visible_fill_color",
|
|
2495
|
-
"comparison": "fill_color_present == true",
|
|
2496
|
-
},
|
|
2497
|
-
"icon_transparent_fill_color": {
|
|
2498
|
-
"name": "icon_has_visible_fill_color",
|
|
2499
|
-
"comparison": "fill_alpha > 0",
|
|
2500
|
-
},
|
|
2501
|
-
"iconpark_unsupported_icon_type": {
|
|
2502
|
-
"name": "iconpark_type_is_supported",
|
|
2503
|
-
"comparison": "icon_type in iconpark_index",
|
|
2504
|
-
},
|
|
2505
|
-
"bbox_overlap": {
|
|
2506
|
-
"name": "text_visual_bounds_do_not_overlap",
|
|
2507
|
-
"comparison": "intersection_area == 0",
|
|
2508
|
-
},
|
|
2509
|
-
"text_may_overflow_shape": {
|
|
2510
|
-
"name": "estimated_text_fits_declared_shape",
|
|
2511
|
-
"comparison": "estimated_height <= available_height",
|
|
2512
|
-
},
|
|
2513
|
-
"whiteboard_external_overlap": {
|
|
2514
|
-
"name": "whiteboard_does_not_cross_sibling_content",
|
|
2515
|
-
"comparison": "external_overlap_count == 0",
|
|
2516
|
-
},
|
|
2517
|
-
"image_covers_text": {
|
|
2518
|
-
"name": "image_does_not_cover_text",
|
|
2519
|
-
"comparison": "intersection_area == 0",
|
|
2520
|
-
},
|
|
2521
|
-
"image_may_cover_vertical_text": {
|
|
2522
|
-
"name": "image_vertical_text_occlusion_requires_review",
|
|
2523
|
-
"comparison": "intersection_area == 0",
|
|
2524
|
-
},
|
|
2525
|
-
"table_resolved_size_mismatch": {
|
|
2526
|
-
"name": "table_declared_size_matches_resolved_grid",
|
|
2527
|
-
"comparison": "declared_size == resolved_size",
|
|
2528
|
-
},
|
|
2529
|
-
"blank_slide": {
|
|
2530
|
-
"name": "slide_has_visible_content",
|
|
2531
|
-
"comparison": "visible_element_count > 0",
|
|
2532
|
-
},
|
|
2533
|
-
"duplicate_element_id": {
|
|
2534
|
-
"name": "element_ids_are_unique",
|
|
2535
|
-
"comparison": "duplicate_count == 0",
|
|
2536
|
-
},
|
|
2537
|
-
}
|
|
2538
|
-
|
|
2539
|
-
|
|
2540
|
-
def issue_rule(issue: dict[str, Any]) -> dict[str, Any]:
|
|
2541
|
-
if issue.get("rule"):
|
|
2542
|
-
return {**issue["rule"], "id": issue["code"]}
|
|
2543
|
-
if issue["code"].endswith("_out_of_canvas"):
|
|
2544
|
-
return {
|
|
2545
|
-
"id": issue["code"],
|
|
2546
|
-
"name": "element_stays_within_slide_canvas",
|
|
2547
|
-
"comparison": "max(left, top, right, bottom overflow) == 0",
|
|
2548
|
-
}
|
|
2549
|
-
return {
|
|
2550
|
-
"id": issue["code"],
|
|
2551
|
-
**RULE_METADATA.get(
|
|
2552
|
-
issue["code"],
|
|
2553
|
-
{"name": issue["code"], "comparison": "violation_count == 0"},
|
|
2554
|
-
),
|
|
2555
|
-
}
|
|
2556
|
-
|
|
2557
|
-
|
|
2558
|
-
def issue_measurement(
|
|
2559
|
-
issue: dict[str, Any], elements_by_ref: dict[str, dict[str, Any]]
|
|
2560
|
-
) -> dict[str, Any]:
|
|
2561
|
-
if issue.get("measurement") is not None:
|
|
2562
|
-
return issue["measurement"]
|
|
2563
|
-
if issue["code"] == "bbox_overlap" and len(issue.get("elements", [])) == 2:
|
|
2564
|
-
left = elements_by_ref.get(issue["elements"][0])
|
|
2565
|
-
right = elements_by_ref.get(issue["elements"][1])
|
|
2566
|
-
if left and right:
|
|
2567
|
-
left_box = (estimate_text_visual_bbox(left) if is_text_element(left) else None) or left
|
|
2568
|
-
right_box = (estimate_text_visual_bbox(right) if is_text_element(right) else None) or right
|
|
2569
|
-
width = intersection_width(left_box, right_box)
|
|
2570
|
-
height = intersection_height(left_box, right_box)
|
|
2571
|
-
return {
|
|
2572
|
-
"intersection_width": round(width, 3),
|
|
2573
|
-
"intersection_height": round(height, 3),
|
|
2574
|
-
"intersection_area": round(width * height, 3),
|
|
2575
|
-
}
|
|
2576
|
-
if issue["code"].endswith("_out_of_canvas"):
|
|
2577
|
-
return {
|
|
2578
|
-
"canvas": issue.get("canvas"),
|
|
2579
|
-
"bbox": issue.get("bbox"),
|
|
2580
|
-
"overflow": issue.get("overflow"),
|
|
2581
|
-
}
|
|
2582
|
-
measurement_keys = (
|
|
2583
|
-
"line",
|
|
2584
|
-
"column",
|
|
2585
|
-
"tag",
|
|
2586
|
-
"attr",
|
|
2587
|
-
"iconType",
|
|
2588
|
-
"line_count",
|
|
2589
|
-
"line_height",
|
|
2590
|
-
"estimated_height",
|
|
2591
|
-
"available_height",
|
|
2592
|
-
"overflow",
|
|
2593
|
-
"dimension",
|
|
2594
|
-
"declared_size",
|
|
2595
|
-
"resolved_size",
|
|
2596
|
-
"resolved_sizes",
|
|
2597
|
-
"overlaps",
|
|
2598
|
-
)
|
|
2599
|
-
measured = {key: issue[key] for key in measurement_keys if key in issue}
|
|
2600
|
-
return measured or {"violation_count": 1}
|
|
2601
|
-
|
|
2602
|
-
|
|
2603
|
-
def related_object(element: dict[str, Any]) -> dict[str, Any]:
|
|
2604
|
-
related = {
|
|
2605
|
-
"kind": element["kind"],
|
|
2606
|
-
"type": element["type"],
|
|
2607
|
-
}
|
|
2608
|
-
bbox_keys = ("x", "y", "width", "height")
|
|
2609
|
-
if all(key in element for key in bbox_keys):
|
|
2610
|
-
related["bbox"] = {key: element[key] for key in bbox_keys}
|
|
2611
|
-
if source_element_id(element) is not None:
|
|
2612
|
-
related["element_id"] = source_element_id(element)
|
|
2613
|
-
if element.get("xml_path"):
|
|
2614
|
-
related["xml_path"] = element["xml_path"]
|
|
2615
|
-
return related
|
|
2616
|
-
|
|
2617
|
-
|
|
2618
|
-
def extract_line_elements(slide_xml: str) -> list[dict[str, Any]]:
|
|
2619
|
-
elements: list[dict[str, Any]] = []
|
|
2620
|
-
for source_kind_index, match in enumerate(
|
|
2621
|
-
re.finditer(r"<line\b([^>]*?)(/?)>", slide_xml), start=1
|
|
2622
|
-
):
|
|
2623
|
-
attrs = match.group(1)
|
|
2624
|
-
source_id = extract_attribute(attrs, "id") or None
|
|
2625
|
-
start_x = extract_numeric_attribute(attrs, "startX")
|
|
2626
|
-
start_y = extract_numeric_attribute(attrs, "startY")
|
|
2627
|
-
end_x = extract_numeric_attribute(attrs, "endX")
|
|
2628
|
-
end_y = extract_numeric_attribute(attrs, "endY")
|
|
2629
|
-
if any(value is None for value in (start_x, start_y, end_x, end_y)):
|
|
2630
|
-
continue
|
|
2631
|
-
line_alpha = extract_numeric_attribute(attrs, "alpha")
|
|
2632
|
-
base_alpha = line_alpha if line_alpha is not None else 1
|
|
2633
|
-
border_alpha = 1
|
|
2634
|
-
if match.group(2) != "/":
|
|
2635
|
-
close_index = slide_xml.find("</line>", match.end())
|
|
2636
|
-
body = slide_xml[match.end() : close_index] if close_index != -1 else ""
|
|
2637
|
-
border_attrs = extract_tag_attributes(body, "border")
|
|
2638
|
-
color_alpha = extract_color_alpha(extract_attribute(border_attrs, "color"))
|
|
2639
|
-
if isinstance(color_alpha, (int, float)):
|
|
2640
|
-
border_alpha = color_alpha
|
|
2641
|
-
elements.append(
|
|
2642
|
-
{
|
|
2643
|
-
"id": source_id or f"line-{len(elements) + 1}",
|
|
2644
|
-
"_source_id": source_id,
|
|
2645
|
-
"kind": "line",
|
|
2646
|
-
"type": "line",
|
|
2647
|
-
"x": min(start_x, end_x),
|
|
2648
|
-
"y": min(start_y, end_y),
|
|
2649
|
-
"width": abs(end_x - start_x),
|
|
2650
|
-
"height": abs(end_y - start_y),
|
|
2651
|
-
"startX": start_x,
|
|
2652
|
-
"startY": start_y,
|
|
2653
|
-
"endX": end_x,
|
|
2654
|
-
"endY": end_y,
|
|
2655
|
-
"rotation": 0,
|
|
2656
|
-
"alpha": base_alpha * border_alpha,
|
|
2657
|
-
"order": len(elements),
|
|
2658
|
-
"_source_kind_index": source_kind_index,
|
|
2659
|
-
}
|
|
2660
|
-
)
|
|
2661
|
-
return elements
|
|
2662
|
-
|
|
2663
|
-
|
|
2664
|
-
def normalize_issue(
|
|
2665
|
-
issue: dict[str, Any],
|
|
2666
|
-
slide_number: int | None,
|
|
2667
|
-
elements_by_ref: dict[str, dict[str, Any]],
|
|
2668
|
-
) -> dict[str, Any]:
|
|
2669
|
-
normalized = dict(issue)
|
|
2670
|
-
element_refs = list(dict.fromkeys(normalized.get("elements", [])))
|
|
2671
|
-
resolved_elements = [
|
|
2672
|
-
elements_by_ref[element_ref]
|
|
2673
|
-
for element_ref in element_refs
|
|
2674
|
-
if element_ref in elements_by_ref
|
|
2675
|
-
]
|
|
2676
|
-
element_locators = [
|
|
2677
|
-
source_element_id(elements_by_ref[element_ref]) or element_ref
|
|
2678
|
-
if element_ref in elements_by_ref
|
|
2679
|
-
else element_ref
|
|
2680
|
-
for element_ref in element_refs
|
|
2681
|
-
]
|
|
2682
|
-
element_ids = [
|
|
2683
|
-
source_id
|
|
2684
|
-
for element in resolved_elements
|
|
2685
|
-
if (source_id := source_element_id(element)) is not None
|
|
2686
|
-
]
|
|
2687
|
-
normalized["schema_version"] = "2.0"
|
|
2688
|
-
normalized["elements"] = element_locators
|
|
2689
|
-
normalized["element_ids"] = element_ids
|
|
2690
|
-
normalized["target"] = {
|
|
2691
|
-
**({"slide_number": slide_number} if slide_number is not None else {}),
|
|
2692
|
-
**normalized.get("target", {}),
|
|
2693
|
-
}
|
|
2694
|
-
normalized["rule"] = issue_rule(normalized)
|
|
2695
|
-
normalized["measurement"] = issue_measurement(issue, elements_by_ref)
|
|
2696
|
-
normalized["related_objects"] = [related_object(element) for element in resolved_elements]
|
|
2697
|
-
if normalized["code"] == "sparse_container_content":
|
|
2698
|
-
ratio = normalized["measurement"]["content_coverage_ratio"]
|
|
2699
|
-
threshold = normalized["rule"]["threshold"]
|
|
2700
|
-
container_locator = (
|
|
2701
|
-
normalized["target"].get("container_id")
|
|
2702
|
-
or normalized["target"].get("container_xml_path")
|
|
2703
|
-
or "unknown"
|
|
2704
|
-
)
|
|
2705
|
-
normalized.setdefault(
|
|
2706
|
-
"message",
|
|
2707
|
-
f"large card {container_locator} content coverage {ratio:.1%} is below {threshold:.1%}",
|
|
2708
|
-
)
|
|
2709
|
-
normalized.setdefault(
|
|
2710
|
-
"hint",
|
|
2711
|
-
"Review the rendered screenshot; add or enlarge meaningful content if the whitespace is not intentional.",
|
|
2712
|
-
)
|
|
2713
|
-
elif normalized["code"] == "sparse_slide_content":
|
|
2714
|
-
ratio = normalized["measurement"]["content_coverage_ratio"]
|
|
2715
|
-
threshold = normalized["rule"]["threshold"]
|
|
2716
|
-
normalized.setdefault(
|
|
2717
|
-
"message",
|
|
2718
|
-
f"slide visible content coverage {ratio:.1%} is below {threshold:.1%}",
|
|
2719
|
-
)
|
|
2720
|
-
normalized.setdefault(
|
|
2721
|
-
"hint",
|
|
2722
|
-
"Review the rendered screenshot to decide whether the page is intentionally sparse.",
|
|
2723
|
-
)
|
|
2724
|
-
else:
|
|
2725
|
-
normalized.setdefault("message", normalized["code"].replace("_", " "))
|
|
2726
|
-
normalized.setdefault(
|
|
2727
|
-
"hint", "Inspect the reported elements and adjust them to satisfy the rule comparison."
|
|
2728
|
-
)
|
|
2729
|
-
if any(related.get("xml_path") for related in normalized["related_objects"]):
|
|
2730
|
-
hint = normalized["hint"]
|
|
2731
|
-
if not hint.startswith(XML_PATH_HINT_PREFIX):
|
|
2732
|
-
normalized["hint"] = f"{XML_PATH_HINT_PREFIX} {hint}"
|
|
2733
|
-
return normalized
|
|
2734
|
-
|
|
2735
|
-
|
|
2736
|
-
def slide_status(errors: list[dict[str, Any]], warnings: list[dict[str, Any]]) -> str:
|
|
2737
|
-
if errors:
|
|
2738
|
-
return "blocked"
|
|
2739
|
-
if warnings:
|
|
2740
|
-
return "needs_screenshot_review"
|
|
2741
|
-
return "passed"
|
|
2742
|
-
|
|
2743
|
-
|
|
2744
|
-
def is_slide_scoped_sxsd_issue(issue: dict[str, Any], root_name: str) -> bool:
|
|
2745
|
-
if issue.get("code") == "sxsd_unsupported_declaration":
|
|
2746
|
-
return False
|
|
2747
|
-
if root_name == "slide":
|
|
2748
|
-
return True
|
|
2749
|
-
path = issue.get("path")
|
|
2750
|
-
if not isinstance(path, str):
|
|
2751
|
-
return False
|
|
2752
|
-
if path.startswith("presentation/slide/"):
|
|
2753
|
-
return True
|
|
2754
|
-
return path == "presentation/slide" and (
|
|
2755
|
-
issue.get("attr") is not None or issue.get("code") == "sxsd_invalid_namespace"
|
|
2756
|
-
)
|
|
2757
|
-
|
|
2758
|
-
|
|
2759
|
-
def build_result(
|
|
2760
|
-
source_path: str | None,
|
|
2761
|
-
slide_size: dict[str, int | float],
|
|
2762
|
-
top_level_issues: list[dict[str, Any]],
|
|
2763
|
-
slides: list[dict[str, Any]],
|
|
2764
|
-
) -> dict[str, Any]:
|
|
2765
|
-
document_errors = [issue for issue in top_level_issues if issue["level"] == "error"]
|
|
2766
|
-
document_warnings = [issue for issue in top_level_issues if issue["level"] == "warning"]
|
|
2767
|
-
document_infos = [issue for issue in top_level_issues if issue["level"] == "info"]
|
|
2768
|
-
error_count = len(document_errors) + sum(len(slide["errors"]) for slide in slides)
|
|
2769
|
-
warning_count = len(document_warnings) + sum(len(slide["warnings"]) for slide in slides)
|
|
2770
|
-
info_count = len(document_infos) + sum(len(slide["infos"]) for slide in slides)
|
|
2771
|
-
all_errors = document_errors + [issue for slide in slides for issue in slide["errors"]]
|
|
2772
|
-
all_warnings = document_warnings + [issue for slide in slides for issue in slide["warnings"]]
|
|
2773
|
-
status = slide_status(all_errors, all_warnings)
|
|
2774
|
-
result: dict[str, Any] = {
|
|
2775
|
-
"schema_version": "2.0",
|
|
2776
|
-
"tool": "xml_text_overlap_lint",
|
|
2777
|
-
"file": source_path,
|
|
2778
|
-
"slide_size": slide_size,
|
|
2779
|
-
"summary": {
|
|
2780
|
-
"slide_count": len(slides),
|
|
2781
|
-
"error_count": error_count,
|
|
2782
|
-
"warning_count": warning_count,
|
|
2783
|
-
"info_count": info_count,
|
|
2784
|
-
"status": status,
|
|
2785
|
-
"release_ready": error_count == 0,
|
|
2786
|
-
"screenshot_review_required": warning_count > 0,
|
|
2787
|
-
},
|
|
2788
|
-
"document": {
|
|
2789
|
-
"errors": document_errors,
|
|
2790
|
-
"warnings": document_warnings,
|
|
2791
|
-
"infos": document_infos,
|
|
2792
|
-
},
|
|
2793
|
-
"slides": slides,
|
|
2794
|
-
}
|
|
2795
|
-
if top_level_issues:
|
|
2796
|
-
result["issues"] = top_level_issues
|
|
2797
|
-
return result
|
|
2798
|
-
|
|
2799
|
-
|
|
2800
|
-
def lint_xml(xml: str, source_path: str | None = None) -> dict[str, Any]:
|
|
2801
|
-
root, xml_error = parse_xml_root(xml)
|
|
2802
|
-
if xml_error:
|
|
2803
|
-
issue = normalize_issue(xml_error, None, {})
|
|
2804
|
-
return build_result(
|
|
2805
|
-
source_path,
|
|
2806
|
-
{"width": 960, "height": 540},
|
|
2807
|
-
[issue],
|
|
2808
|
-
[],
|
|
2809
|
-
)
|
|
2810
|
-
if root is None:
|
|
2811
|
-
raise AssertionError("parse_xml_root must return a root or error")
|
|
2812
|
-
|
|
2813
|
-
namespace_issues = validate_sml_tag_prefixes(xml)
|
|
2814
|
-
root_name = xml_local_name(root.tag)
|
|
2815
|
-
sxsd_issues = validate_sxsd_document(xml, root)
|
|
2816
|
-
iconpark_issues = validate_iconpark_icon_types(root)
|
|
2817
|
-
top_level_issues = [
|
|
2818
|
-
normalize_issue(issue, None, {})
|
|
2819
|
-
for issue in [
|
|
2820
|
-
*namespace_issues,
|
|
2821
|
-
*[
|
|
2822
|
-
issue
|
|
2823
|
-
for issue in sxsd_issues
|
|
2824
|
-
if not is_slide_scoped_sxsd_issue(issue, root_name)
|
|
2825
|
-
],
|
|
2826
|
-
*iconpark_issues,
|
|
2827
|
-
]
|
|
2828
|
-
]
|
|
2829
|
-
if any(issue["level"] == "error" for issue in top_level_issues):
|
|
2830
|
-
return build_result(
|
|
2831
|
-
source_path,
|
|
2832
|
-
{"width": 960, "height": 540},
|
|
2833
|
-
top_level_issues,
|
|
2834
|
-
[],
|
|
2835
|
-
)
|
|
2836
|
-
|
|
2837
|
-
presentation = parse_presentation(root)
|
|
2838
|
-
slide_roots = presentation["slide_roots"]
|
|
2839
|
-
slides: list[dict[str, Any]] = []
|
|
2840
|
-
presentation_id_elements: list[dict[str, Any]] = []
|
|
2841
|
-
presentation_elements_by_ref: dict[str, dict[str, Any]] = {}
|
|
2842
|
-
for index, slide_xml in enumerate(presentation["slides"]):
|
|
2843
|
-
slide_number = index + 1
|
|
2844
|
-
slide_root = slide_roots[index]
|
|
2845
|
-
slide_sxsd_issues = [
|
|
2846
|
-
normalize_issue(issue, slide_number, {})
|
|
2847
|
-
for issue in validate_sxsd_document(slide_xml, slide_root)
|
|
2848
|
-
]
|
|
2849
|
-
slide_sxsd_errors = [
|
|
2850
|
-
issue for issue in slide_sxsd_issues if issue["level"] == "error"
|
|
2851
|
-
]
|
|
2852
|
-
if slide_sxsd_errors:
|
|
2853
|
-
slide_sxsd_warnings = [
|
|
2854
|
-
issue for issue in slide_sxsd_issues if issue["level"] == "warning"
|
|
2855
|
-
]
|
|
2856
|
-
slides.append(
|
|
2857
|
-
{
|
|
2858
|
-
"slide_number": slide_number,
|
|
2859
|
-
"status": slide_status(slide_sxsd_errors, slide_sxsd_warnings),
|
|
2860
|
-
"element_count": 0,
|
|
2861
|
-
"errors": slide_sxsd_errors,
|
|
2862
|
-
"warnings": slide_sxsd_warnings,
|
|
2863
|
-
"infos": [],
|
|
2864
|
-
"issues": slide_sxsd_issues,
|
|
2865
|
-
}
|
|
2866
|
-
)
|
|
2867
|
-
continue
|
|
2868
|
-
|
|
2869
|
-
geometry = lint_slide(
|
|
2870
|
-
slide_xml,
|
|
2871
|
-
slide_number,
|
|
2872
|
-
presentation["width"],
|
|
2873
|
-
presentation["height"],
|
|
2874
|
-
)
|
|
2875
|
-
density_elements = extract_density_elements(slide_xml, slide_number)
|
|
2876
|
-
id_elements = extract_source_id_elements(slide_xml, slide_number)
|
|
2877
|
-
presentation_id_elements.extend(id_elements)
|
|
2878
|
-
extra_elements = [
|
|
2879
|
-
element for element in density_elements if element["kind"] in {"icon", "polyline", "line"}
|
|
2880
|
-
]
|
|
2881
|
-
elements_by_ref = {
|
|
2882
|
-
element_ref(element): element for element in density_elements
|
|
2883
|
-
}
|
|
2884
|
-
visible_element_count = len(elements_by_ref)
|
|
2885
|
-
for element in id_elements:
|
|
2886
|
-
elements_by_ref.setdefault(element_ref(element), element)
|
|
2887
|
-
# geometry["elements"] are the exact objects should_flag_overlap/detect_elements_out_of_canvas
|
|
2888
|
-
# selected inside lint_slide; prefer them so measurement/related_objects stay consistent
|
|
2889
|
-
# with whatever actually triggered the issue, instead of density_elements' separate re-parse.
|
|
2890
|
-
elements_by_ref.update(
|
|
2891
|
-
{element_ref(element): element for element in geometry["elements"]}
|
|
2892
|
-
)
|
|
2893
|
-
presentation_elements_by_ref.update(
|
|
2894
|
-
{
|
|
2895
|
-
element_ref(element): elements_by_ref[element_ref(element)]
|
|
2896
|
-
for element in id_elements
|
|
2897
|
-
}
|
|
2898
|
-
)
|
|
2899
|
-
extra_overflow_issues = detect_elements_out_of_canvas(
|
|
2900
|
-
extra_elements,
|
|
2901
|
-
presentation["width"],
|
|
2902
|
-
presentation["height"],
|
|
2903
|
-
)
|
|
2904
|
-
raw_issues = [
|
|
2905
|
-
*geometry["issues"],
|
|
2906
|
-
*extra_overflow_issues,
|
|
2907
|
-
*detect_duplicate_element_ids(id_elements),
|
|
2908
|
-
*detect_blank_slide(
|
|
2909
|
-
density_elements,
|
|
2910
|
-
slide_number,
|
|
2911
|
-
presentation["width"],
|
|
2912
|
-
presentation["height"],
|
|
2913
|
-
),
|
|
2914
|
-
*detect_sparse_container_content(
|
|
2915
|
-
density_elements,
|
|
2916
|
-
slide_number,
|
|
2917
|
-
presentation["width"],
|
|
2918
|
-
presentation["height"],
|
|
2919
|
-
),
|
|
2920
|
-
*detect_sparse_slide_content(
|
|
2921
|
-
density_elements,
|
|
2922
|
-
slide_number,
|
|
2923
|
-
presentation["width"],
|
|
2924
|
-
presentation["height"],
|
|
2925
|
-
),
|
|
2926
|
-
]
|
|
2927
|
-
issues = [
|
|
2928
|
-
*slide_sxsd_issues,
|
|
2929
|
-
*[
|
|
2930
|
-
normalize_issue(issue, slide_number, elements_by_ref)
|
|
2931
|
-
for issue in raw_issues
|
|
2932
|
-
],
|
|
2933
|
-
]
|
|
2934
|
-
errors = [issue for issue in issues if issue["level"] == "error"]
|
|
2935
|
-
warnings = [issue for issue in issues if issue["level"] == "warning"]
|
|
2936
|
-
infos = [issue for issue in issues if issue["level"] == "info"]
|
|
2937
|
-
slides.append(
|
|
2938
|
-
{
|
|
2939
|
-
"slide_number": slide_number,
|
|
2940
|
-
"status": slide_status(errors, warnings),
|
|
2941
|
-
"element_count": visible_element_count,
|
|
2942
|
-
"errors": errors,
|
|
2943
|
-
"warnings": warnings,
|
|
2944
|
-
"infos": infos,
|
|
2945
|
-
"issues": issues,
|
|
2946
|
-
}
|
|
2947
|
-
)
|
|
2948
|
-
|
|
2949
|
-
top_level_issues.extend(
|
|
2950
|
-
normalize_issue(issue, None, presentation_elements_by_ref)
|
|
2951
|
-
for issue in detect_duplicate_element_ids(
|
|
2952
|
-
presentation_id_elements, cross_slide_only=True
|
|
2953
|
-
)
|
|
2954
|
-
)
|
|
2955
|
-
|
|
2956
|
-
return build_result(
|
|
2957
|
-
source_path,
|
|
2958
|
-
{"width": presentation["width"], "height": presentation["height"]},
|
|
2959
|
-
top_level_issues,
|
|
2960
|
-
slides,
|
|
2961
|
-
)
|
|
2962
|
-
|
|
2963
|
-
|
|
2964
|
-
def print_usage() -> None:
|
|
2965
|
-
print("Usage:\n python3 xml_text_overlap_lint.py --input <presentation.xml>", file=sys.stderr)
|
|
2966
|
-
|
|
2967
7
|
|
|
2968
|
-
|
|
2969
|
-
|
|
2970
|
-
if options.get("help") or options.get("--help"):
|
|
2971
|
-
print_usage()
|
|
2972
|
-
raise SystemExit(0)
|
|
2973
|
-
if not options.get("input"):
|
|
2974
|
-
print_usage()
|
|
2975
|
-
fail("--input is required")
|
|
2976
|
-
requested_path = options["input"]
|
|
2977
|
-
resolved_path = Path(requested_path).resolve()
|
|
2978
|
-
result = lint_xml(read_file(resolved_path), requested_path)
|
|
2979
|
-
print(json.dumps(result, ensure_ascii=False, indent=2))
|
|
2980
|
-
if result["summary"]["error_count"] > 0:
|
|
2981
|
-
raise SystemExit(1)
|
|
8
|
+
from xml_lint import * # noqa: F401,F403
|
|
9
|
+
from xml_lint import XmlLayoutLintError, run_cli
|
|
2982
10
|
|
|
2983
11
|
|
|
2984
12
|
if __name__ == "__main__":
|