@amaster.ai/pi-lark 0.1.2-beta.41 → 0.1.2-beta.42

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/package.json +2 -2
  2. package/skills/lark-apps/SKILL.md +5 -3
  3. package/skills/lark-apps/references/lark-apps-automation.md +164 -0
  4. package/skills/lark-apps/references/lark-apps-get.md +43 -0
  5. package/skills/lark-apps/references/lark-apps-html-publish.md +7 -2
  6. package/skills/lark-apps/references/lark-apps-init.md +1 -2
  7. package/skills/lark-apps/references/lark-apps-openapi-key.md +1 -1
  8. package/skills/lark-apps/references/lark-apps-release-create.md +3 -1
  9. package/skills/lark-base/SKILL.md +1 -1
  10. package/skills/lark-base/references/lark-base-dashboard-block-get-data.md +7 -7
  11. package/skills/lark-calendar/references/lark-calendar-create.md +1 -0
  12. package/skills/lark-doc/references/lark-doc-xml-extended-blocks.md +2 -1
  13. package/skills/lark-drive/SKILL.md +10 -4
  14. package/skills/lark-drive/references/lark-drive-comment-location.md +16 -4
  15. package/skills/lark-drive/references/lark-drive-comments-guide.md +16 -8
  16. package/skills/lark-drive/references/lark-drive-export.md +39 -10
  17. package/skills/lark-drive/references/lark-drive-list-comments.md +125 -0
  18. package/skills/lark-drive/references/lark-drive-member-add.md +1 -1
  19. package/skills/lark-drive/references/lark-drive-pull.md +3 -3
  20. package/skills/lark-drive/references/lark-drive-push.md +1 -1
  21. package/skills/lark-drive/references/lark-drive-status.md +12 -14
  22. package/skills/lark-im/SKILL.md +5 -4
  23. package/skills/lark-im/references/lark-im-messages-reply.md +1 -1
  24. package/skills/lark-im/references/lark-im-messages-send.md +1 -1
  25. package/skills/lark-minutes/SKILL.md +19 -4
  26. package/skills/lark-minutes/references/lark-minutes-todo.md +2 -2
  27. package/skills/lark-shared/SKILL.md +9 -9
  28. package/skills/lark-sheets/SKILL.md +98 -29
  29. package/skills/lark-sheets/references/lark-sheets-batch-update.md +18 -9
  30. package/skills/lark-sheets/references/lark-sheets-changeset.md +105 -0
  31. package/skills/lark-sheets/references/lark-sheets-chart.md +4 -2
  32. package/skills/lark-sheets/references/lark-sheets-conditional-format.md +2 -0
  33. package/skills/lark-sheets/references/lark-sheets-filter-view.md +1 -1
  34. package/skills/lark-sheets/references/lark-sheets-float-image.md +6 -6
  35. package/skills/lark-sheets/references/lark-sheets-formula-translation.md +12 -3
  36. package/skills/lark-sheets/references/lark-sheets-formula-verify.md +77 -0
  37. package/skills/lark-sheets/references/lark-sheets-history.md +93 -0
  38. package/skills/lark-sheets/references/lark-sheets-pivot-table.md +7 -2
  39. package/skills/lark-sheets/references/lark-sheets-range-operations.md +44 -14
  40. package/skills/lark-sheets/references/lark-sheets-read-data.md +3 -3
  41. package/skills/lark-sheets/references/lark-sheets-sheet-structure.md +4 -4
  42. package/skills/lark-sheets/references/lark-sheets-visual-standards.md +4 -4
  43. package/skills/lark-sheets/references/lark-sheets-workbook.md +29 -4
  44. package/skills/lark-sheets/references/lark-sheets-write-cells.md +21 -11
  45. package/skills/lark-slides/SKILL.md +29 -18
  46. package/skills/lark-slides/references/asset-planning.md +0 -1
  47. package/skills/lark-slides/references/examples.md +57 -227
  48. package/skills/lark-slides/references/iconpark.md +2 -2
  49. package/skills/lark-slides/references/lark-slides-create.md +21 -2
  50. package/skills/lark-slides/references/lark-slides-media-upload.md +0 -1
  51. package/skills/lark-slides/references/lark-slides-pptx-template-workflows.md +89 -0
  52. package/skills/lark-slides/references/lark-slides-replace-pages.md +1 -1
  53. package/skills/lark-slides/references/lark-slides-replace-slide.md +1 -1
  54. package/skills/lark-slides/references/lark-slides-screenshot.md +11 -8
  55. package/skills/lark-slides/references/lark-slides-xml-get.md +100 -0
  56. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-delete.md +9 -7
  57. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-get.md +4 -4
  58. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-replace.md +12 -10
  59. package/skills/lark-slides/references/lark-slides-xml-presentations-get.md +14 -13
  60. package/skills/lark-slides/references/planning-layer.md +1 -1
  61. package/skills/lark-slides/references/troubleshooting.md +7 -25
  62. package/skills/lark-slides/references/validation-checklist.md +18 -9
  63. package/skills/lark-slides/references/visual-planning.md +4 -3
  64. package/skills/lark-slides/references/xml-schema-quick-ref.md +6 -2
  65. package/skills/lark-slides/scripts/xml_text_overlap_lint.py +647 -52
  66. package/skills/lark-slides/scripts/xml_text_overlap_lint_test.py +529 -0
  67. package/skills/lark-task/SKILL.md +1 -0
  68. package/skills/lark-vc-agent/SKILL.md +11 -4
  69. package/skills/lark-vc-agent/references/lark-vc-agent-meeting-events.md +1 -1
  70. package/skills/lark-vc-agent/references/lark-vc-agent-meeting-leave.md +1 -1
  71. package/skills/lark-vc-agent/references/lark-vc-agent-meeting-list-active.md +2 -2
  72. package/skills/lark-sheets/references/lark-sheets-core-operations.md +0 -103
  73. package/skills/lark-slides/references/lark-slides-xml-presentation-slide-create.md +0 -220
@@ -5,14 +5,43 @@
5
5
  from __future__ import annotations
6
6
 
7
7
  import json
8
+ import math
8
9
  import re
9
10
  import sys
11
+ import unicodedata
12
+ import xml.parsers.expat as expat
10
13
  import xml.etree.ElementTree as ET
11
- from difflib import SequenceMatcher
14
+ from difflib import SequenceMatcher, get_close_matches
12
15
  from pathlib import Path
13
16
  from typing import Any
14
17
 
15
18
 
19
+ XS_NS = "{http://www.w3.org/2001/XMLSchema}"
20
+ XML_NS = "{http://www.w3.org/XML/1998/namespace}"
21
+ SVG_NS = "{http://www.w3.org/2000/svg}"
22
+ SML_NAMESPACE = "http://www.larkoffice.com/sml/2.0"
23
+ SXSD_SCHEMA_PATH = Path(__file__).resolve().parents[1] / "references" / "slides_xml_schema_definition.xml"
24
+ ICONPARK_INDEX_PATH = Path(__file__).resolve().parents[1] / "references" / "iconpark-index.json"
25
+ SXSD_TAG_ALIASES = {
26
+ "textbox": "<shape type=\"text\">",
27
+ "textBox": "<shape type=\"text\">",
28
+ "image": "<img>",
29
+ "picture": "<img>",
30
+ }
31
+ SXSD_ATTR_ALIASES = {
32
+ "x": "topLeftX",
33
+ "left": "topLeftX",
34
+ "y": "topLeftY",
35
+ "top": "topLeftY",
36
+ "w": "width",
37
+ "h": "height",
38
+ "fontColor": "color",
39
+ }
40
+ SERVER_FILLED_SXSD_ATTRS = {"id"}
41
+ _SXSD_TAG_ATTRIBUTES_CACHE: dict[str, set[str]] | None = None
42
+ _ICONPARK_ICON_TYPES_CACHE: set[str] | None = None
43
+
44
+
16
45
  class XmlTextOverlapLintError(Exception):
17
46
  pass
18
47
 
@@ -71,10 +100,290 @@ def strip_xml(value: str) -> str:
71
100
  return re.sub(r"\s+", " ", stripped).strip()
72
101
 
73
102
 
103
+ def strip_xml_paragraphs(value: str) -> str:
104
+ paragraphs = re.findall(r"<p\b[^>]*>([\s\S]*?)</p\s*>", value)
105
+ if paragraphs:
106
+ return "\n".join(strip_xml(paragraph) for paragraph in paragraphs)
107
+ return strip_xml(value)
108
+
109
+
74
110
  def xml_local_name(tag: str) -> str:
75
111
  return tag.rsplit("}", 1)[-1] if tag.startswith("{") else tag
76
112
 
77
113
 
114
+ def xml_namespace(tag: str) -> str | None:
115
+ return tag.split("}", 1)[0] + "}" if tag.startswith("{") else None
116
+
117
+
118
+ def strip_xsd_prefix(value: str | None) -> str | None:
119
+ if value is None:
120
+ return None
121
+ return value.rsplit(":", 1)[-1]
122
+
123
+
124
+ def iter_direct_xsd_children(element: ET.Element, local_name: str) -> list[ET.Element]:
125
+ return [child for child in element if child.tag == f"{XS_NS}{local_name}"]
126
+
127
+
128
+ def load_sxsd_tag_attributes() -> dict[str, set[str]]:
129
+ global _SXSD_TAG_ATTRIBUTES_CACHE
130
+ if _SXSD_TAG_ATTRIBUTES_CACHE is not None:
131
+ return _SXSD_TAG_ATTRIBUTES_CACHE
132
+
133
+ schema_root = ET.parse(SXSD_SCHEMA_PATH).getroot()
134
+ named_complex_types = {
135
+ complex_type.attrib["name"]: complex_type
136
+ for complex_type in schema_root.findall(f"{XS_NS}complexType")
137
+ if complex_type.attrib.get("name")
138
+ }
139
+ resolving: set[str] = set()
140
+
141
+ def attributes_for_complex_type(complex_type: ET.Element) -> set[str]:
142
+ attrs: set[str] = {
143
+ attribute.attrib["name"]
144
+ for attribute in iter_direct_xsd_children(complex_type, "attribute")
145
+ if attribute.attrib.get("name")
146
+ }
147
+ for content_name in ("simpleContent", "complexContent"):
148
+ for complex_content in iter_direct_xsd_children(complex_type, content_name):
149
+ for extension in iter_direct_xsd_children(complex_content, "extension"):
150
+ base_type = strip_xsd_prefix(extension.attrib.get("base"))
151
+ if base_type:
152
+ attrs.update(attributes_for_type(base_type))
153
+ attrs.update(
154
+ attribute.attrib["name"]
155
+ for attribute in iter_direct_xsd_children(extension, "attribute")
156
+ if attribute.attrib.get("name")
157
+ )
158
+ return attrs
159
+
160
+ def attributes_for_type(type_name: str) -> set[str]:
161
+ if type_name in resolving:
162
+ return set()
163
+ complex_type = named_complex_types.get(type_name)
164
+ if complex_type is None:
165
+ return set()
166
+ resolving.add(type_name)
167
+ try:
168
+ return attributes_for_complex_type(complex_type)
169
+ finally:
170
+ resolving.remove(type_name)
171
+
172
+ tag_attributes: dict[str, set[str]] = {}
173
+ for element in schema_root.iter(f"{XS_NS}element"):
174
+ tag_name = element.attrib.get("name")
175
+ if not tag_name:
176
+ continue
177
+
178
+ attrs: set[str] = set()
179
+ type_name = strip_xsd_prefix(element.attrib.get("type"))
180
+ if type_name:
181
+ attrs.update(attributes_for_type(type_name))
182
+ for complex_type in iter_direct_xsd_children(element, "complexType"):
183
+ attrs.update(attributes_for_complex_type(complex_type))
184
+
185
+ tag_attributes.setdefault(tag_name, set()).update(attrs)
186
+
187
+ _SXSD_TAG_ATTRIBUTES_CACHE = tag_attributes
188
+ return tag_attributes
189
+
190
+
191
+ def load_iconpark_icon_types() -> set[str]:
192
+ global _ICONPARK_ICON_TYPES_CACHE
193
+ if _ICONPARK_ICON_TYPES_CACHE is not None:
194
+ return _ICONPARK_ICON_TYPES_CACHE
195
+
196
+ try:
197
+ index_data = json.loads(ICONPARK_INDEX_PATH.read_text(encoding="utf-8"))
198
+ except json.JSONDecodeError as error:
199
+ fail(f"invalid iconpark index JSON: {error}")
200
+ icons = index_data.get("icons")
201
+ if not isinstance(icons, list):
202
+ fail("iconpark index must contain an icons array")
203
+
204
+ icon_types = {
205
+ icon["iconType"]
206
+ for icon in icons
207
+ if isinstance(icon, dict) and isinstance(icon.get("iconType"), str) and icon["iconType"]
208
+ }
209
+ _ICONPARK_ICON_TYPES_CACHE = icon_types
210
+ return icon_types
211
+
212
+
213
+ def build_sxsd_tag_hint(tag_name: str, supported_tags: set[str]) -> str:
214
+ alias = SXSD_TAG_ALIASES.get(tag_name)
215
+ if alias:
216
+ return f"Use {alias} instead of <{tag_name}>."
217
+ if tag_name == "svg":
218
+ return 'Inside <whiteboard>, write SVG as <svg xmlns="http://www.w3.org/2000/svg">...</svg>.'
219
+ close_matches = get_close_matches(tag_name, sorted(supported_tags), n=3, cutoff=0.72)
220
+ if close_matches:
221
+ return "Unsupported SXSD tag. Did you mean " + ", ".join(f"<{match}>" for match in close_matches) + "?"
222
+ return "Unsupported SXSD tag. Use only tags defined in slides_xml_schema_definition.xml."
223
+
224
+
225
+ def build_sxsd_attr_hint(tag_name: str, attr_name: str, allowed_attrs: set[str]) -> str:
226
+ alias = SXSD_ATTR_ALIASES.get(attr_name)
227
+ if alias and alias in allowed_attrs:
228
+ return f'Use "{alias}" on <{tag_name}> instead of "{attr_name}".'
229
+ close_matches = get_close_matches(attr_name, sorted(allowed_attrs), n=3, cutoff=0.68)
230
+ if close_matches:
231
+ return "Unsupported SXSD attribute. Did you mean " + ", ".join(f'"{match}"' for match in close_matches) + "?"
232
+ allowed_summary = ", ".join(sorted(allowed_attrs)[:8])
233
+ if len(allowed_attrs) > 8:
234
+ allowed_summary += ", ..."
235
+ return f"Unsupported SXSD attribute for <{tag_name}>. Allowed attributes include: {allowed_summary}."
236
+
237
+
238
+ def should_skip_sxsd_subtree(element: ET.Element, ancestors: list[str]) -> bool:
239
+ return "whiteboard" in ancestors and xml_namespace(element.tag) == SVG_NS
240
+
241
+
242
+ def should_skip_sxsd_attribute(attr_name: str) -> bool:
243
+ return attr_name in SERVER_FILLED_SXSD_ATTRS
244
+
245
+
246
+ def validate_sxsd_tag_attributes(root: ET.Element) -> list[dict[str, Any]]:
247
+ tag_attributes = load_sxsd_tag_attributes()
248
+ supported_tags = set(tag_attributes)
249
+ issues: list[dict[str, Any]] = []
250
+
251
+ def visit(element: ET.Element, ancestors: list[str], path: str) -> None:
252
+ if should_skip_sxsd_subtree(element, ancestors):
253
+ return
254
+
255
+ tag_name = xml_local_name(element.tag)
256
+ current_path = f"{path}/{tag_name}" if path else tag_name
257
+ if tag_name not in supported_tags:
258
+ issues.append(
259
+ {
260
+ "level": "error",
261
+ "code": "sxsd_unsupported_tag",
262
+ "tag": tag_name,
263
+ "path": current_path,
264
+ "message": f"unsupported SXSD tag <{tag_name}> at {current_path}",
265
+ "hint": build_sxsd_tag_hint(tag_name, supported_tags),
266
+ }
267
+ )
268
+ return
269
+ else:
270
+ allowed_attrs = tag_attributes[tag_name]
271
+ for raw_attr_name in element.attrib:
272
+ if raw_attr_name.startswith(XML_NS):
273
+ continue
274
+ attr_name = xml_local_name(raw_attr_name)
275
+ if should_skip_sxsd_attribute(attr_name):
276
+ continue
277
+ if attr_name in allowed_attrs:
278
+ continue
279
+ issues.append(
280
+ {
281
+ "level": "error",
282
+ "code": "sxsd_unsupported_attr",
283
+ "tag": tag_name,
284
+ "attr": attr_name,
285
+ "path": current_path,
286
+ "message": f'unsupported SXSD attribute "{attr_name}" on <{tag_name}> at {current_path}',
287
+ "hint": build_sxsd_attr_hint(tag_name, attr_name, allowed_attrs),
288
+ }
289
+ )
290
+
291
+ for child in element:
292
+ visit(child, [*ancestors, tag_name], current_path)
293
+
294
+ visit(root, [], "")
295
+ return issues
296
+
297
+
298
+ def build_iconpark_icon_type_hint(icon_type: str, supported_icon_types: set[str]) -> str:
299
+ close_matches = get_close_matches(icon_type, sorted(supported_icon_types), n=3, cutoff=0.58)
300
+ if close_matches:
301
+ return (
302
+ "iconType must exist in iconpark-index.json. Did you mean "
303
+ + ", ".join(f'"{match}"' for match in close_matches)
304
+ + "?"
305
+ )
306
+ return "iconType must exist in iconpark-index.json. Use scripts/iconpark_tool.py to search supported icons."
307
+
308
+
309
+ def validate_iconpark_icon_types(root: ET.Element) -> list[dict[str, Any]]:
310
+ supported_icon_types: set[str] | None = None
311
+ issues: list[dict[str, Any]] = []
312
+
313
+ def direct_child(element: ET.Element, local_name: str) -> ET.Element | None:
314
+ return next((child for child in element if xml_local_name(child.tag) == local_name), None)
315
+
316
+ def is_transparent_color(color: str) -> bool:
317
+ normalized = re.sub(r"\s+", "", color).lower()
318
+ if normalized == "transparent":
319
+ return True
320
+ rgba_match = re.fullmatch(r"rgba\([^,]+,[^,]+,[^,]+,([0-9.]+)\)", normalized)
321
+ if not rgba_match:
322
+ return False
323
+ try:
324
+ return float(rgba_match.group(1)) <= 0
325
+ except ValueError:
326
+ return False
327
+
328
+ def append_missing_fill_color_issue(current_path: str) -> None:
329
+ issues.append(
330
+ {
331
+ "level": "error",
332
+ "code": "icon_missing_fill_color",
333
+ "tag": "icon",
334
+ "path": current_path,
335
+ "message": f"<icon> must set explicit non-transparent fillColor for visual visibility at {current_path}",
336
+ "hint": 'Add <fill><fillColor color="rgba(R, G, B, 1)"/></fill> inside <icon>. This is a visual lint rule, not an SXSD required field.',
337
+ }
338
+ )
339
+
340
+ def visit(element: ET.Element, path: str) -> None:
341
+ nonlocal supported_icon_types
342
+ tag_name = xml_local_name(element.tag)
343
+ current_path = f"{path}/{tag_name}" if path else tag_name
344
+ if tag_name == "icon":
345
+ icon_type = element.attrib.get("iconType")
346
+ if icon_type is not None:
347
+ if supported_icon_types is None:
348
+ supported_icon_types = load_iconpark_icon_types()
349
+ if icon_type not in supported_icon_types:
350
+ issues.append(
351
+ {
352
+ "level": "error",
353
+ "code": "iconpark_unsupported_icon_type",
354
+ "tag": "icon",
355
+ "attr": "iconType",
356
+ "iconType": icon_type,
357
+ "path": current_path,
358
+ "message": f'unsupported iconpark iconType "{icon_type}" at {current_path}',
359
+ "hint": build_iconpark_icon_type_hint(icon_type, supported_icon_types),
360
+ }
361
+ )
362
+ fill = direct_child(element, "fill")
363
+ fill_color = direct_child(fill, "fillColor") if fill is not None else None
364
+ color = fill_color.attrib.get("color") if fill_color is not None else None
365
+ if not color:
366
+ append_missing_fill_color_issue(current_path)
367
+ elif is_transparent_color(color):
368
+ issues.append(
369
+ {
370
+ "level": "error",
371
+ "code": "icon_transparent_fill_color",
372
+ "tag": "icon",
373
+ "attr": "fillColor",
374
+ "path": current_path,
375
+ "color": color,
376
+ "message": f'<icon> fillColor must not be transparent for visual visibility at {current_path}: "{color}"',
377
+ "hint": 'Use an opaque visible color, for example <fillColor color="rgba(37, 99, 235, 1)"/>.',
378
+ }
379
+ )
380
+ for child in element:
381
+ visit(child, current_path)
382
+
383
+ visit(root, "")
384
+ return issues
385
+
386
+
78
387
  def extract_error_context(xml: str, line: int | None, column: int | None, radius: int = 40) -> str | None:
79
388
  if line is None or column is None:
80
389
  return None
@@ -103,16 +412,87 @@ def build_xml_error_issue(error: ET.ParseError, xml: str) -> dict[str, Any]:
103
412
  }
104
413
 
105
414
 
106
- def validate_xml_well_formed(xml: str) -> dict[str, Any] | None:
415
+ def validate_sml_tag_prefixes(xml: str) -> list[dict[str, Any]]:
416
+ namespace_map: dict[str, str] = {}
417
+ pending_declarations: list[tuple[str, str | None]] = []
418
+ declarations_by_element: list[list[tuple[str, str | None]]] = []
419
+ element_stack: list[str] = []
420
+ issues: list[dict[str, Any]] = []
421
+
422
+ parser = expat.ParserCreate(namespace_separator="|")
423
+ parser.namespace_prefixes = True
424
+
425
+ def handle_namespace_decl(prefix: str | None, namespace: str) -> None:
426
+ normalized_prefix = prefix or ""
427
+ previous_namespace = namespace_map.get(normalized_prefix)
428
+ namespace_map[normalized_prefix] = namespace
429
+ pending_declarations.append((normalized_prefix, previous_namespace))
430
+
431
+ def handle_start_element(name: str, _attrs: dict[str, str]) -> None:
432
+ declarations_by_element.append(pending_declarations.copy())
433
+ pending_declarations.clear()
434
+ name_parts = name.rsplit("|", 2)
435
+ if len(name_parts) == 3:
436
+ _namespace, local_name, prefix = name_parts
437
+ element_name = f"{prefix}:{local_name}"
438
+ else:
439
+ prefix = ""
440
+ local_name = name_parts[-1]
441
+ element_name = local_name
442
+ element_stack.append(element_name)
443
+ if not prefix:
444
+ return
445
+
446
+ if namespace_map.get(prefix) != SML_NAMESPACE:
447
+ return
448
+ path = "/".join(element_stack)
449
+ issues.append(
450
+ {
451
+ "level": "error",
452
+ "code": "sml_prefixed_tag",
453
+ "tag": element_name,
454
+ "namespace": SML_NAMESPACE,
455
+ "path": path,
456
+ "line": parser.CurrentLineNumber,
457
+ "column": parser.CurrentColumnNumber,
458
+ "message": f"SML tag <{element_name}> must not use a namespace prefix at {path}",
459
+ "hint": (
460
+ f'Use <{local_name}> under the default namespace '
461
+ f'<{local_name} xmlns="{SML_NAMESPACE}">, or use an unprefixed SML tag.'
462
+ ),
463
+ }
464
+ )
465
+
466
+ def handle_end_element(_name: str) -> None:
467
+ for prefix, previous_namespace in reversed(declarations_by_element.pop()):
468
+ if previous_namespace is None:
469
+ namespace_map.pop(prefix, None)
470
+ else:
471
+ namespace_map[prefix] = previous_namespace
472
+ element_stack.pop()
473
+
474
+ parser.StartNamespaceDeclHandler = handle_namespace_decl
475
+ parser.StartElementHandler = handle_start_element
476
+ parser.EndElementHandler = handle_end_element
477
+ parser.Parse(xml, True)
478
+ return issues
479
+
480
+
481
+ def parse_xml_root(xml: str) -> tuple[ET.Element | None, dict[str, Any] | None]:
107
482
  try:
108
483
  root = ET.fromstring(xml)
109
484
  except ET.ParseError as error:
110
- return build_xml_error_issue(error, xml)
485
+ return None, build_xml_error_issue(error, xml)
111
486
 
112
487
  root_name = xml_local_name(root.tag)
113
488
  if root_name not in {"presentation", "slide"}:
114
489
  fail("input must contain a <presentation> or <slide> root")
115
- return None
490
+ return root, None
491
+
492
+
493
+ def validate_xml_well_formed(xml: str) -> dict[str, Any] | None:
494
+ _, xml_error = parse_xml_root(xml)
495
+ return xml_error
116
496
 
117
497
 
118
498
  def parse_presentation(xml: str) -> dict[str, Any]:
@@ -131,47 +511,44 @@ def parse_presentation(xml: str) -> dict[str, Any]:
131
511
 
132
512
  def extract_elements(slide_xml: str) -> list[dict[str, Any]]:
133
513
  elements: list[dict[str, Any]] = []
134
- for match in re.finditer(r"<shape\b([^>]*)>([\s\S]*?)</shape>", slide_xml):
135
- attrs, content = match.group(1), match.group(2)
136
- x = extract_numeric_attribute(attrs, "topLeftX")
137
- y = extract_numeric_attribute(attrs, "topLeftY")
138
- width = extract_numeric_attribute(attrs, "width")
139
- height = extract_numeric_attribute(attrs, "height")
140
- if all(value is not None for value in [x, y, width, height]):
141
- font_size = float(extract_attribute(content, "fontSize") or extract_attribute(attrs, "fontSize") or 16)
142
- elements.append(
143
- {
144
- "id": f"shape-{len(elements) + 1}",
145
- "kind": "shape",
146
- "type": extract_attribute(attrs, "type") or "shape",
147
- "textType": extract_attribute(content, "textType"),
148
- "x": x,
149
- "y": y,
150
- "width": width,
151
- "height": height,
152
- "fontSize": font_size,
153
- "text": strip_xml(content),
154
- }
155
- )
156
514
 
157
- for match in re.finditer(r"<(img|table|chart)\b([^>]*)/?>", slide_xml):
158
- attrs = match.group(2)
515
+ for match in re.finditer(r"<(shape|img|table|chart|whiteboard)\b([^>]*)>", slide_xml):
516
+ kind, attrs = match.group(1), match.group(2)
517
+ content = ""
518
+ if kind == "shape":
519
+ close_index = slide_xml.find("</shape>", match.end())
520
+ if close_index != -1:
521
+ content = slide_xml[match.end() : close_index]
522
+
523
+ element_id = extract_attribute(attrs, "id") or f"{kind}-{len(elements) + 1}"
159
524
  x = extract_numeric_attribute(attrs, "topLeftX")
160
525
  y = extract_numeric_attribute(attrs, "topLeftY")
161
526
  width = extract_numeric_attribute(attrs, "width")
162
527
  height = extract_numeric_attribute(attrs, "height")
163
528
  if all(value is not None for value in [x, y, width, height]):
164
- elements.append(
165
- {
166
- "id": f"{match.group(1)}-{len(elements) + 1}",
167
- "kind": match.group(1),
168
- "type": match.group(1),
169
- "x": x,
170
- "y": y,
171
- "width": width,
172
- "height": height,
173
- }
174
- )
529
+ element = {
530
+ "id": element_id,
531
+ "kind": kind,
532
+ "type": extract_attribute(attrs, "type") or kind,
533
+ "x": x,
534
+ "y": y,
535
+ "width": width,
536
+ "height": height,
537
+ "order": len(elements),
538
+ }
539
+ if kind == "shape":
540
+ element.update(
541
+ {
542
+ "textType": extract_attribute(content, "textType"),
543
+ "textAlign": extract_attribute(content, "textAlign"),
544
+ "autoFit": extract_attribute(content, "autoFit"),
545
+ "fontSize": float(
546
+ extract_attribute(content, "fontSize") or extract_attribute(attrs, "fontSize") or 16
547
+ ),
548
+ "text": strip_xml_paragraphs(content),
549
+ }
550
+ )
551
+ elements.append(element)
175
552
  return elements
176
553
 
177
554
 
@@ -188,6 +565,10 @@ def is_text_element(element: dict[str, Any]) -> bool:
188
565
  return element["kind"] == "shape" and element["type"] == "text"
189
566
 
190
567
 
568
+ def is_whiteboard_element(element: dict[str, Any]) -> bool:
569
+ return element["kind"] == "whiteboard"
570
+
571
+
191
572
  def has_text_content(element: dict[str, Any]) -> bool:
192
573
  return bool(element.get("text"))
193
574
 
@@ -201,6 +582,24 @@ def normalize_text_for_overlap(text: str) -> str:
201
582
  return re.sub(r"\s+", "", text)
202
583
 
203
584
 
585
+ def estimate_character_width(character: str, font_size: int | float) -> int | float:
586
+ if character.isspace():
587
+ return font_size * 0.33
588
+ if unicodedata.east_asian_width(character) in {"F", "W"}:
589
+ return font_size
590
+ return font_size * 0.55
591
+
592
+
593
+ def estimate_text_width(text: str, font_size: int | float) -> int | float:
594
+ return sum(estimate_character_width(character, font_size) for character in text)
595
+
596
+
597
+ def estimate_text_max_line_width(element: dict[str, Any]) -> int | float:
598
+ font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
599
+ paragraphs = [paragraph for paragraph in re.split(r"\n+", element["text"]) if paragraph]
600
+ return max([estimate_text_width(paragraph, font_size) for paragraph in paragraphs] or [1])
601
+
602
+
204
603
  def is_similar_text_overlay(left: dict[str, Any], right: dict[str, Any]) -> bool:
205
604
  left_text = normalize_text_for_overlap(left.get("text") or "")
206
605
  right_text = normalize_text_for_overlap(right.get("text") or "")
@@ -213,12 +612,11 @@ def is_similar_text_overlay(left: dict[str, Any], right: dict[str, Any]) -> bool
213
612
 
214
613
  def estimate_text_line_count(element: dict[str, Any]) -> int:
215
614
  font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
216
- chars_per_line = max(1, int(element["width"] // max(font_size * 0.55, 1)))
217
615
  paragraphs = [paragraph for paragraph in re.split(r"\n+", element["text"]) if paragraph]
218
616
  line_count = 0
219
617
  for paragraph in paragraphs:
220
- logical_length = max(len(paragraph), 1)
221
- line_count += max(1, -(-logical_length // chars_per_line))
618
+ logical_width = max(estimate_text_width(paragraph, font_size), 1)
619
+ line_count += max(1, math.ceil(logical_width / max(element["width"], 1)))
222
620
  return max(line_count, 1)
223
621
 
224
622
 
@@ -227,9 +625,8 @@ def estimate_text_visual_bbox(element: dict[str, Any]) -> dict[str, int | float]
227
625
  return None
228
626
 
229
627
  font_size = element["fontSize"] if isinstance(element["fontSize"], (int, float)) else 16
230
- char_width = max(font_size * 0.55, 1)
231
628
  line_count = estimate_text_line_count(element)
232
- visual_width = min(element["width"], max(1, len(element["text"]) * char_width))
629
+ visual_width = min(element["width"], max(1, estimate_text_max_line_width(element)))
233
630
  visual_height = min(element["height"], max(1, line_count * font_size * 1.2))
234
631
  return {
235
632
  "x": element["x"],
@@ -247,6 +644,49 @@ def intersection_area(left: dict[str, Any], right: dict[str, Any]) -> int | floa
247
644
  return width * height
248
645
 
249
646
 
647
+ def intersection_height(left: dict[str, Any], right: dict[str, Any]) -> int | float:
648
+ height = min(left["y"] + left["height"], right["y"] + right["height"]) - max(left["y"], right["y"])
649
+ return max(height, 0)
650
+
651
+
652
+ def intersection_width(left: dict[str, Any], right: dict[str, Any]) -> int | float:
653
+ width = min(left["x"] + left["width"], right["x"] + right["width"]) - max(left["x"], right["x"])
654
+ return max(width, 0)
655
+
656
+
657
+ def element_area(element: dict[str, Any]) -> int | float:
658
+ return max(element["width"], 0) * max(element["height"], 0)
659
+
660
+
661
+ def contains(outer: dict[str, Any], inner: dict[str, Any], tolerance: int | float = 2) -> bool:
662
+ return (
663
+ inner["x"] >= outer["x"] - tolerance
664
+ and inner["y"] >= outer["y"] - tolerance
665
+ and inner["x"] + inner["width"] <= outer["x"] + outer["width"] + tolerance
666
+ and inner["y"] + inner["height"] <= outer["y"] + outer["height"] + tolerance
667
+ )
668
+
669
+
670
+ def is_bottom_layer_full_slide_whiteboard(
671
+ whiteboard: dict[str, Any], other: dict[str, Any], slide_width: int | float, slide_height: int | float
672
+ ) -> bool:
673
+ return (
674
+ whiteboard["order"] < other["order"]
675
+ and whiteboard["x"] <= 2
676
+ and whiteboard["y"] <= 2
677
+ and whiteboard["width"] >= slide_width - 4
678
+ and whiteboard["height"] >= slide_height - 4
679
+ )
680
+
681
+
682
+ def is_background_container_for_whiteboard(container: dict[str, Any], whiteboard: dict[str, Any]) -> bool:
683
+ if container["order"] > whiteboard["order"]:
684
+ return False
685
+ if is_text_element(container):
686
+ return False
687
+ return contains(container, whiteboard)
688
+
689
+
250
690
  def is_template_text_stack(left: dict[str, Any], right: dict[str, Any]) -> bool:
251
691
  if not (is_text_element(left) and is_text_element(right)):
252
692
  return False
@@ -269,6 +709,39 @@ def is_template_text_stack(left: dict[str, Any], right: dict[str, Any]) -> bool:
269
709
  return same_column and vertical_offset >= top_font_size * 0.75
270
710
 
271
711
 
712
+ def should_flag_horizontal_text_overflow(left: dict[str, Any], right: dict[str, Any]) -> bool:
713
+ if not (is_text_element(left) and is_text_element(right)):
714
+ return False
715
+ if not (has_text_content(left) and has_text_content(right)):
716
+ return False
717
+ if is_template_text_stack(left, right) or is_similar_text_overlay(left, right):
718
+ return False
719
+
720
+ source, target = sorted([left, right], key=lambda element: element["x"])
721
+ if source["x"] == target["x"]:
722
+ return False
723
+ if source.get("autoFit") == "normal-auto-fit":
724
+ return False
725
+ if source.get("textAlign") in {"center", "right"}:
726
+ return False
727
+
728
+ font_size = source["fontSize"] if isinstance(source["fontSize"], (int, float)) else 16
729
+ visual_width = estimate_text_max_line_width(source)
730
+ overflow_width = visual_width - source["width"]
731
+ min_overflow = max(font_size * 1.5, source["width"] * 0.08)
732
+ if overflow_width < min_overflow:
733
+ return False
734
+
735
+ intrusion_width = source["x"] + visual_width - target["x"]
736
+ min_intrusion = max(font_size * 1.5, target["width"] * 0.08)
737
+ if intrusion_width < min_intrusion:
738
+ return False
739
+
740
+ vertical_overlap = intersection_height(source, target)
741
+ min_vertical_overlap = min(source["height"], target["height"]) * 0.40
742
+ return vertical_overlap >= min_vertical_overlap
743
+
744
+
272
745
  def should_flag_overlap(left: dict[str, Any], right: dict[str, Any]) -> bool:
273
746
  if is_text_element(left) and not has_text_content(left):
274
747
  return False
@@ -294,13 +767,116 @@ def should_flag_overlap(left: dict[str, Any], right: dict[str, Any]) -> bool:
294
767
  return False
295
768
 
296
769
 
297
- def lint_slide(slide_xml: str, slide_number: int) -> dict[str, Any]:
298
- elements = extract_elements(slide_xml)
770
+ def build_whiteboard_external_overlap_issue(
771
+ whiteboard: dict[str, Any], overlap_details: list[dict[str, Any]]
772
+ ) -> dict[str, Any]:
773
+ element_ids = [detail["element"] for detail in overlap_details]
774
+ return {
775
+ "level": "warning",
776
+ "code": "whiteboard_external_overlap",
777
+ "elements": [whiteboard["id"], *element_ids],
778
+ "message": f'whiteboard {whiteboard["id"]} overlaps {len(element_ids)} sibling elements across its boundary',
779
+ "hint": (
780
+ "Treat this as a static whiteboard container-bbox risk, not final visual proof. "
781
+ "After moving or accepting the overlap, use screenshot QA or equivalent rendered visual inspection as "
782
+ "the final authority because XML readback does not include whiteboard SVG/Mermaid internals."
783
+ ),
784
+ "overlaps": overlap_details,
785
+ }
786
+
787
+
788
+ def should_report_whiteboard_overlap(
789
+ whiteboard: dict[str, Any],
790
+ other: dict[str, Any],
791
+ slide_width: int | float,
792
+ slide_height: int | float,
793
+ ) -> dict[str, Any] | None:
794
+ if other is whiteboard or not intersects(whiteboard, other):
795
+ return None
796
+ if contains(whiteboard, other):
797
+ return None
798
+ if is_bottom_layer_full_slide_whiteboard(whiteboard, other, slide_width, slide_height):
799
+ return None
800
+ if is_background_container_for_whiteboard(other, whiteboard):
801
+ return None
802
+
803
+ overlap_width = intersection_width(whiteboard, other)
804
+ overlap_height = intersection_height(whiteboard, other)
805
+ if overlap_width < 8 or overlap_height < 8:
806
+ return None
807
+
808
+ other_area = element_area(other)
809
+ if other_area <= 0:
810
+ return None
811
+ overlap_area = overlap_width * overlap_height
812
+ overlap_ratio = overlap_area / other_area
813
+ if overlap_ratio < 0.15:
814
+ return None
815
+
816
+ return {
817
+ "element": other["id"],
818
+ "kind": other["kind"],
819
+ "type": other.get("type"),
820
+ "overlap_width": overlap_width,
821
+ "overlap_height": overlap_height,
822
+ "target_overlap_ratio": round(overlap_ratio, 3),
823
+ }
824
+
825
+
826
+ def prune_contained_text_overlap_details(
827
+ overlap_details: list[dict[str, Any]], elements_by_id: dict[str, dict[str, Any]]
828
+ ) -> list[dict[str, Any]]:
829
+ pruned: list[dict[str, Any]] = []
830
+ for detail in overlap_details:
831
+ element = elements_by_id[detail["element"]]
832
+ if is_text_element(element):
833
+ has_reported_container = any(
834
+ detail["element"] != other_detail["element"]
835
+ and not is_text_element(elements_by_id[other_detail["element"]])
836
+ and contains(elements_by_id[other_detail["element"]], element)
837
+ for other_detail in overlap_details
838
+ )
839
+ if has_reported_container:
840
+ continue
841
+ pruned.append(detail)
842
+ return pruned
843
+
844
+
845
+ def detect_whiteboard_external_overlaps(
846
+ elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
847
+ ) -> list[dict[str, Any]]:
299
848
  issues: list[dict[str, Any]] = []
849
+ elements_by_id = {element["id"]: element for element in elements}
850
+ for whiteboard in [element for element in elements if is_whiteboard_element(element)]:
851
+ overlap_details = [
852
+ detail
853
+ for element in elements
854
+ if (
855
+ detail := should_report_whiteboard_overlap(
856
+ whiteboard,
857
+ element,
858
+ slide_width,
859
+ slide_height,
860
+ )
861
+ )
862
+ is not None
863
+ ]
864
+ overlap_details = prune_contained_text_overlap_details(overlap_details, elements_by_id)
865
+ if overlap_details:
866
+ issues.append(build_whiteboard_external_overlap_issue(whiteboard, overlap_details))
867
+ return issues
868
+
869
+
870
+ def lint_slide(
871
+ slide_xml: str, slide_number: int, slide_width: int | float = 960, slide_height: int | float = 540
872
+ ) -> dict[str, Any]:
873
+ elements = extract_elements(slide_xml)
874
+ issues: list[dict[str, Any]] = detect_whiteboard_external_overlaps(elements, slide_width, slide_height)
300
875
 
301
876
  for index, left in enumerate(elements):
302
877
  for right in elements[index + 1 :]:
303
- if not intersects(left, right) or not should_flag_overlap(left, right):
878
+ horizontal_overflow = should_flag_horizontal_text_overflow(left, right)
879
+ if not horizontal_overflow and (not intersects(left, right) or not should_flag_overlap(left, right)):
304
880
  continue
305
881
  issues.append(
306
882
  {
@@ -315,7 +891,7 @@ def lint_slide(slide_xml: str, slide_number: int) -> dict[str, Any]:
315
891
 
316
892
 
317
893
  def lint_xml(xml: str, source_path: str | None = None) -> dict[str, Any]:
318
- xml_error = validate_xml_well_formed(xml)
894
+ root, xml_error = parse_xml_root(xml)
319
895
  if xml_error:
320
896
  return {
321
897
  "file": source_path,
@@ -325,19 +901,38 @@ def lint_xml(xml: str, source_path: str | None = None) -> dict[str, Any]:
325
901
  "slides": [],
326
902
  }
327
903
 
904
+ namespace_issues = validate_sml_tag_prefixes(xml)
905
+ sxsd_issues = validate_sxsd_tag_attributes(root) if root is not None else []
906
+ iconpark_issues = validate_iconpark_icon_types(root) if root is not None else []
907
+ top_level_issues = [*namespace_issues, *sxsd_issues, *iconpark_issues]
908
+ if namespace_issues:
909
+ error_count = sum(1 for issue in top_level_issues if issue["level"] == "error")
910
+ warning_count = sum(1 for issue in top_level_issues if issue["level"] == "warning")
911
+ return {
912
+ "file": source_path,
913
+ "slide_size": {"width": 960, "height": 540},
914
+ "summary": {"slide_count": 0, "error_count": error_count, "warning_count": warning_count},
915
+ "issues": top_level_issues,
916
+ "slides": [],
917
+ }
328
918
  presentation = parse_presentation(xml)
329
919
  slides = [
330
- lint_slide(slide_xml, index + 1)
920
+ lint_slide(slide_xml, index + 1, presentation["width"], presentation["height"])
331
921
  for index, slide_xml in enumerate(presentation["slides"])
332
922
  ]
333
- error_count = sum(1 for slide in slides for issue in slide["issues"] if issue["level"] == "error")
334
- warning_count = sum(1 for slide in slides for issue in slide["issues"] if issue["level"] == "warning")
335
- return {
923
+ error_count = sum(1 for issue in top_level_issues if issue["level"] == "error")
924
+ error_count += sum(1 for slide in slides for issue in slide["issues"] if issue["level"] == "error")
925
+ warning_count = sum(1 for issue in top_level_issues if issue["level"] == "warning")
926
+ warning_count += sum(1 for slide in slides for issue in slide["issues"] if issue["level"] == "warning")
927
+ result = {
336
928
  "file": source_path,
337
929
  "slide_size": {"width": presentation["width"], "height": presentation["height"]},
338
930
  "summary": {"slide_count": len(slides), "error_count": error_count, "warning_count": warning_count},
339
931
  "slides": slides,
340
932
  }
933
+ if top_level_issues:
934
+ result["issues"] = top_level_issues
935
+ return result
341
936
 
342
937
 
343
938
  def print_usage() -> None: