@amaster.ai/pi-lark 0.1.2-beta.45 → 0.1.2-beta.47

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,7 @@
1
1
  #!/usr/bin/env python3
2
2
  # Copyright (c) 2026 Lark Technologies Pte. Ltd.
3
3
  # SPDX-License-Identifier: MIT
4
+ """Validate Slides XML structure and page layout through one release gate."""
4
5
 
5
6
  from __future__ import annotations
6
7
 
@@ -42,18 +43,25 @@ ROUNDTRIP_SXSD_ATTRS = {
42
43
  ("chart", "updated"),
43
44
  ("chartData", "isStaticData"),
44
45
  }
46
+ # Slides readback echoes each chartField's CSV text as per-value <chartParsedValues> children;
47
+ # it's server-emitted, absent from the write schema, and appears on virtually every chart-bearing
48
+ # deck, so treating it as an unsupported tag would block per-slide linting document-wide.
49
+ ROUNDTRIP_SXSD_TAGS = {"chartParsedValues"}
45
50
  DEFAULT_TABLE_COLUMN_WIDTH = 110
46
51
  DEFAULT_TABLE_ROW_HEIGHT = 37
52
+ # Sub-pixel canvas overflow is floating-point rounding noise (e.g. rotated-bbox math), not a
53
+ # visible defect; keep this well under 1px so real overflow is still always caught.
54
+ CANVAS_OVERFLOW_TOLERANCE = 0.5
47
55
  _SXSD_TAG_ATTRIBUTES_CACHE: dict[str, set[str]] | None = None
48
56
  _ICONPARK_ICON_TYPES_CACHE: set[str] | None = None
49
57
 
50
58
 
51
- class XmlTextOverlapLintError(Exception):
59
+ class XmlLayoutLintError(Exception):
52
60
  pass
53
61
 
54
62
 
55
63
  def fail(message: str) -> None:
56
- raise XmlTextOverlapLintError(message)
64
+ raise XmlLayoutLintError(message)
57
65
 
58
66
 
59
67
  def read_file(file_path: str | Path) -> str:
@@ -79,8 +87,12 @@ def parse_args(argv: list[str]) -> dict[str, Any]:
79
87
 
80
88
 
81
89
  def extract_attribute(tag_source: str, name: str) -> str | None:
82
- match = re.search(fr'{re.escape(name)}="([^"]+)"', tag_source)
83
- return match.group(1) if match else None
90
+ match = re.search(
91
+ fr"(?:^|\s){re.escape(name)}\s*=\s*(?:\"([^\"]+)\"|'([^']+)')", tag_source
92
+ )
93
+ if not match:
94
+ return None
95
+ return match.group(1) if match.group(1) is not None else match.group(2)
84
96
 
85
97
 
86
98
  def extract_numeric_attribute(tag_source: str, name: str) -> int | float | None:
@@ -372,6 +384,8 @@ def validate_sxsd_tag_attributes(root: ET.Element) -> list[dict[str, Any]]:
372
384
 
373
385
  tag_name = xml_local_name(element.tag)
374
386
  current_path = f"{path}/{tag_name}" if path else tag_name
387
+ if tag_name in ROUNDTRIP_SXSD_TAGS:
388
+ return
375
389
  if tag_name not in supported_tags:
376
390
  issues.append(
377
391
  {
@@ -632,8 +646,9 @@ def extract_elements(slide_xml: str) -> list[dict[str, Any]]:
632
646
 
633
647
  for match in re.finditer(r"<(shape|img|table|chart|whiteboard)\b([^>]*)>", slide_xml):
634
648
  kind, attrs = match.group(1), match.group(2)
649
+ is_self_closing = attrs.rstrip().endswith("/")
635
650
  content = ""
636
- if kind in {"shape", "table"}:
651
+ if kind in {"shape", "table"} and not is_self_closing:
637
652
  close_index = slide_xml.find(f"</{kind}>", match.end())
638
653
  if close_index != -1:
639
654
  content = slide_xml[match.end() : close_index]
@@ -1039,6 +1054,19 @@ def should_flag_horizontal_text_overflow(left: dict[str, Any], right: dict[str,
1039
1054
  return vertical_overlap >= min_vertical_overlap
1040
1055
 
1041
1056
 
1057
+ def horizontal_text_overflow_measurement(left: dict[str, Any], right: dict[str, Any]) -> dict[str, int | float]:
1058
+ source, target = sorted([left, right], key=lambda element: element["x"])
1059
+ visual_width = estimate_text_max_line_width(source)
1060
+ source_visual_bbox = {"x": source["x"], "y": source["y"], "width": visual_width, "height": source["height"]}
1061
+ width = intersection_width(source_visual_bbox, target)
1062
+ height = intersection_height(source_visual_bbox, target)
1063
+ return {
1064
+ "intersection_width": round(width, 3),
1065
+ "intersection_height": round(height, 3),
1066
+ "intersection_area": round(width * height, 3),
1067
+ }
1068
+
1069
+
1042
1070
  def should_flag_overlap(left: dict[str, Any], right: dict[str, Any]) -> bool:
1043
1071
  if is_text_element(left) and not has_text_content(left):
1044
1072
  return False
@@ -1166,9 +1194,6 @@ def detect_whiteboard_external_overlaps(
1166
1194
 
1167
1195
  def element_canvas_bbox(element: dict[str, Any]) -> dict[str, int | float]:
1168
1196
  bbox = {key: element[key] for key in ("x", "y", "width", "height")}
1169
- if element["kind"] != "chart" and not (element["kind"] == "shape" and element["type"] == "text"):
1170
- return bbox
1171
-
1172
1197
  rotation = element["rotation"]
1173
1198
  if not isinstance(rotation, (int, float)) or not math.isfinite(rotation):
1174
1199
  rotation = 0
@@ -1194,12 +1219,7 @@ def detect_elements_out_of_canvas(
1194
1219
  elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
1195
1220
  ) -> list[dict[str, Any]]:
1196
1221
  issues: list[dict[str, Any]] = []
1197
- for element in (
1198
- element
1199
- for element in elements
1200
- if element["kind"] in {"table", "chart"}
1201
- or (element["kind"] == "shape" and element["type"] == "text")
1202
- ):
1222
+ for element in elements:
1203
1223
  bbox = element_canvas_bbox(element)
1204
1224
  overflow = {
1205
1225
  "left": max(-bbox["x"], 0),
@@ -1208,7 +1228,9 @@ def detect_elements_out_of_canvas(
1208
1228
  "bottom": max(bbox["y"] + bbox["height"] - slide_height, 0),
1209
1229
  }
1210
1230
  overflow_details = [
1211
- f"{side} by {amount:g}px" for side, amount in overflow.items() if amount > 0
1231
+ f"{side} by {amount:g}px"
1232
+ for side, amount in overflow.items()
1233
+ if amount > CANVAS_OVERFLOW_TOLERANCE
1212
1234
  ]
1213
1235
  if not overflow_details:
1214
1236
  continue
@@ -1330,62 +1352,714 @@ def lint_slide(
1330
1352
  "code": "bbox_overlap",
1331
1353
  "elements": [left["id"], right["id"]],
1332
1354
  "message": f'{left["id"]} overlaps {right["id"]}',
1355
+ "hint": "Move or resize the elements so their visual bounds no longer intersect.",
1356
+ **(
1357
+ {"measurement": horizontal_text_overflow_measurement(left, right)}
1358
+ if horizontal_overflow
1359
+ else {}
1360
+ ),
1333
1361
  }
1334
1362
  )
1335
1363
 
1336
- return {"slide_number": slide_number, "element_count": len(elements), "issues": issues}
1364
+ return {
1365
+ "slide_number": slide_number,
1366
+ "element_count": len(elements),
1367
+ "elements": elements,
1368
+ "issues": issues,
1369
+ }
1337
1370
 
1338
1371
 
1339
- def lint_xml(xml: str, source_path: str | None = None) -> dict[str, Any]:
1340
- root, xml_error = parse_xml_root(xml)
1341
- if xml_error:
1372
+
1373
+ MIN_CONTAINER_WIDTH = 140
1374
+ MIN_CONTAINER_HEIGHT = 160
1375
+ MIN_SHORT_CARD_HEIGHT = 80
1376
+ MIN_CONTAINER_AREA = 20_000
1377
+ MIN_CONTENT_COVERAGE_RATIO = 0.15
1378
+ MIN_SLIDE_CONTENT_COVERAGE_RATIO = 0.035
1379
+ MIN_SLIDE_CONTENT_ELEMENT_COUNT = 4
1380
+ SHORT_CARD_SIZE_TOLERANCE_RATIO = 0.10
1381
+ MIN_SIMILAR_SHORT_CARD_COUNT = 2
1382
+ LARGE_VISUAL_CHILD_RATIO = 0.35
1383
+ LAYOUT_PANEL_SPAN_RATIO = 0.90
1384
+ IMAGE_OVERLAY_MATCH_RATIO = 0.90
1385
+ DENSITY_CONTAINMENT_TOLERANCE = 8
1386
+
1387
+
1388
+ def clipped_bbox(element: dict[str, Any], container: dict[str, Any]) -> dict[str, int | float] | None:
1389
+ left = max(element["x"], container["x"])
1390
+ top = max(element["y"], container["y"])
1391
+ right = min(element["x"] + element["width"], container["x"] + container["width"])
1392
+ bottom = min(element["y"] + element["height"], container["y"] + container["height"])
1393
+ if right <= left or bottom <= top:
1394
+ return None
1395
+ return {"x": left, "y": top, "width": right - left, "height": bottom - top}
1396
+
1397
+
1398
+ def rectangle_union_area(rectangles: list[dict[str, int | float]]) -> int | float:
1399
+ x_coordinates = sorted({coordinate for rect in rectangles for coordinate in (rect["x"], rect["x"] + rect["width"])})
1400
+ area = 0
1401
+ for left, right in zip(x_coordinates, x_coordinates[1:]):
1402
+ intervals = sorted(
1403
+ (rect["y"], rect["y"] + rect["height"])
1404
+ for rect in rectangles
1405
+ if rect["x"] < right and rect["x"] + rect["width"] > left
1406
+ )
1407
+ covered_height = 0
1408
+ interval_end: int | float | None = None
1409
+ for top, bottom in intervals:
1410
+ if interval_end is None:
1411
+ covered_height += bottom - top
1412
+ interval_end = bottom
1413
+ elif bottom > interval_end:
1414
+ covered_height += bottom - max(top, interval_end)
1415
+ interval_end = bottom
1416
+ area += (right - left) * covered_height
1417
+ return area
1418
+
1419
+
1420
+ def has_similar_short_card_peer(element: dict[str, Any], elements: list[dict[str, Any]]) -> bool:
1421
+ return sum(
1422
+ other is not element
1423
+ and is_visually_rendered(other)
1424
+ and other["kind"] == "shape"
1425
+ and other["type"] == "rect"
1426
+ and other["width"] >= MIN_CONTAINER_WIDTH
1427
+ and other["height"] >= MIN_SHORT_CARD_HEIGHT
1428
+ and element_area(other) >= MIN_CONTAINER_AREA
1429
+ and abs(other["width"] - element["width"]) / max(other["width"], element["width"])
1430
+ <= SHORT_CARD_SIZE_TOLERANCE_RATIO
1431
+ and abs(other["height"] - element["height"]) / max(other["height"], element["height"])
1432
+ <= SHORT_CARD_SIZE_TOLERANCE_RATIO
1433
+ for other in elements
1434
+ ) >= MIN_SIMILAR_SHORT_CARD_COUNT
1435
+
1436
+
1437
+ def is_layout_container(
1438
+ element: dict[str, Any],
1439
+ slide_width: int | float,
1440
+ slide_height: int | float,
1441
+ elements: list[dict[str, Any]] | None = None,
1442
+ ) -> bool:
1443
+ has_supported_height = element["height"] >= MIN_CONTAINER_HEIGHT or (
1444
+ elements is not None
1445
+ and element["height"] >= MIN_SHORT_CARD_HEIGHT
1446
+ and has_similar_short_card_peer(element, elements)
1447
+ )
1448
+ return (
1449
+ element["kind"] == "shape"
1450
+ and element["type"] == "rect"
1451
+ and is_visually_rendered(element)
1452
+ and element["width"] >= MIN_CONTAINER_WIDTH
1453
+ and has_supported_height
1454
+ and element_area(element) >= MIN_CONTAINER_AREA
1455
+ and not (
1456
+ element["x"] <= 2
1457
+ and element["y"] <= 2
1458
+ and element["width"] >= slide_width - 4
1459
+ and element["height"] >= slide_height - 4
1460
+ )
1461
+ )
1462
+
1463
+
1464
+ def is_edge_spanning_layout_panel(
1465
+ element: dict[str, Any], slide_width: int | float, slide_height: int | float
1466
+ ) -> bool:
1467
+ touches_horizontal_edge = element["x"] <= 2 or element["x"] + element["width"] >= slide_width - 2
1468
+ touches_vertical_edge = element["y"] <= 2 or element["y"] + element["height"] >= slide_height - 2
1469
+ return (touches_horizontal_edge and element["height"] >= slide_height * LAYOUT_PANEL_SPAN_RATIO) or (
1470
+ touches_vertical_edge and element["width"] >= slide_width * LAYOUT_PANEL_SPAN_RATIO
1471
+ )
1472
+
1473
+
1474
+ def has_matching_image_overlay(container: dict[str, Any], elements: list[dict[str, Any]]) -> bool:
1475
+ container_area = element_area(container)
1476
+ return any(
1477
+ element["kind"] == "img"
1478
+ and is_visually_rendered(element)
1479
+ and intersection_area(container, element) / max(1, container_area) >= IMAGE_OVERLAY_MATCH_RATIO
1480
+ for element in elements
1481
+ )
1482
+
1483
+
1484
+ def is_nested_in_layout_panel(
1485
+ container: dict[str, Any], elements: list[dict[str, Any]], slide_width: int | float, slide_height: int | float
1486
+ ) -> bool:
1487
+ return any(
1488
+ element is not container
1489
+ and element["kind"] == "shape"
1490
+ and element["type"] == "rect"
1491
+ and is_visually_rendered(element)
1492
+ and is_edge_spanning_layout_panel(element, slide_width, slide_height)
1493
+ and contains(element, container, tolerance=DENSITY_CONTAINMENT_TOLERANCE)
1494
+ for element in elements
1495
+ )
1496
+
1497
+
1498
+ def extract_density_elements(slide_xml: str) -> list[dict[str, Any]]:
1499
+ elements = extract_elements(slide_xml)
1500
+ elements_by_id = {element["id"]: element for element in elements}
1501
+ root = ET.fromstring(slide_xml)
1502
+ for node in root.iter():
1503
+ if xml_local_name(node.tag) != "shape":
1504
+ continue
1505
+ element = elements_by_id.get(node.attrib.get("id", ""))
1506
+ if element is None:
1507
+ continue
1508
+ content_node = next(
1509
+ (child for child in node if xml_local_name(child.tag) == "content"),
1510
+ None,
1511
+ )
1512
+ paragraphs = (
1513
+ [
1514
+ " ".join("".join(paragraph.itertext()).split())
1515
+ for paragraph in content_node.iter()
1516
+ if xml_local_name(paragraph.tag) == "p"
1517
+ ]
1518
+ if content_node is not None
1519
+ else []
1520
+ )
1521
+ raw_font_size = (
1522
+ content_node.attrib.get("fontSize") if content_node is not None else None
1523
+ ) or node.attrib.get("fontSize")
1524
+ try:
1525
+ base_font_size = float(raw_font_size or 16)
1526
+ except ValueError:
1527
+ base_font_size = 16.0
1528
+ element.update(
1529
+ {
1530
+ "textType": content_node.attrib.get("textType") if content_node is not None else None,
1531
+ "textAlign": content_node.attrib.get("textAlign") if content_node is not None else None,
1532
+ "autoFit": content_node.attrib.get("autoFit") if content_node is not None else None,
1533
+ "fontSize": base_font_size,
1534
+ "text": "\n".join(paragraph for paragraph in paragraphs if paragraph),
1535
+ }
1536
+ )
1537
+ if not has_text_content(element):
1538
+ continue
1539
+ declared_font_sizes = []
1540
+ for descendant in node.iter():
1541
+ raw_declared_font_size = descendant.attrib.get("fontSize")
1542
+ if raw_declared_font_size is None:
1543
+ continue
1544
+ try:
1545
+ declared_font_sizes.append(float(raw_declared_font_size))
1546
+ except ValueError:
1547
+ continue
1548
+ if declared_font_sizes:
1549
+ element["fontSize"] = max(declared_font_sizes)
1550
+ for match in re.finditer(r"<icon\b([^>]*)>", slide_xml):
1551
+ attrs = match.group(1)
1552
+ x = extract_numeric_attribute(attrs, "topLeftX")
1553
+ y = extract_numeric_attribute(attrs, "topLeftY")
1554
+ width = extract_numeric_attribute(attrs, "width")
1555
+ height = extract_numeric_attribute(attrs, "height")
1556
+ if any(value is None for value in (x, y, width, height)):
1557
+ continue
1558
+ icon_alpha = extract_numeric_attribute(attrs, "alpha")
1559
+ elements.append(
1560
+ {
1561
+ "id": extract_attribute(attrs, "id") or f"icon-{len(elements) + 1}",
1562
+ "kind": "icon",
1563
+ "type": "icon",
1564
+ "x": x,
1565
+ "y": y,
1566
+ "width": width,
1567
+ "height": height,
1568
+ "rotation": extract_numeric_attribute(attrs, "rotation") or 0,
1569
+ "alpha": icon_alpha if icon_alpha is not None else 1,
1570
+ "order": len(elements),
1571
+ }
1572
+ )
1573
+ for match in re.finditer(r"<polyline\b([^>]*)>", slide_xml):
1574
+ attrs = match.group(1)
1575
+ x = extract_numeric_attribute(attrs, "topLeftX")
1576
+ y = extract_numeric_attribute(attrs, "topLeftY")
1577
+ width = extract_numeric_attribute(attrs, "width")
1578
+ height = extract_numeric_attribute(attrs, "height")
1579
+ if any(value is None for value in (x, y, width, height)):
1580
+ continue
1581
+ polyline_alpha = extract_numeric_attribute(attrs, "alpha")
1582
+ elements.append(
1583
+ {
1584
+ "id": extract_attribute(attrs, "id") or f"polyline-{len(elements) + 1}",
1585
+ "kind": "polyline",
1586
+ "type": "polyline",
1587
+ "x": x,
1588
+ "y": y,
1589
+ "width": width,
1590
+ "height": height,
1591
+ "rotation": extract_numeric_attribute(attrs, "rotation") or 0,
1592
+ "alpha": polyline_alpha if polyline_alpha is not None else 1,
1593
+ "order": len(elements),
1594
+ }
1595
+ )
1596
+ for line_element in extract_line_elements(slide_xml):
1597
+ line_element["order"] = len(elements)
1598
+ elements.append(line_element)
1599
+ return elements
1600
+
1601
+
1602
+ def is_visually_rendered(element: dict[str, Any]) -> bool:
1603
+ return element.get("alpha", 1) > 0
1604
+
1605
+
1606
+ def visual_bbox(element: dict[str, Any], container: dict[str, Any]) -> dict[str, int | float] | None:
1607
+ if not is_visually_rendered(element):
1608
+ return None
1609
+ if is_text_element(element):
1610
+ estimated = estimate_text_visual_bbox(element)
1611
+ return clipped_bbox(estimated, container) if estimated else None
1612
+ return clipped_bbox(element, container)
1613
+
1614
+
1615
+ def own_text_visual_bbox(container: dict[str, Any]) -> dict[str, int | float] | None:
1616
+ if container["kind"] != "shape" or not has_text_content(container):
1617
+ return None
1618
+ text_proxy = {**container, "type": "text"}
1619
+ estimated = estimate_text_visual_bbox(text_proxy)
1620
+ return clipped_bbox(estimated, container) if estimated else None
1621
+
1622
+
1623
+ def slide_content_visual_bbox(
1624
+ element: dict[str, Any], slide_bbox: dict[str, int | float]
1625
+ ) -> dict[str, int | float] | None:
1626
+ if not is_visually_rendered(element):
1627
+ return None
1628
+ if is_text_element(element):
1629
+ estimated = estimate_text_visual_bbox(element)
1630
+ return clipped_bbox(estimated, slide_bbox) if estimated else None
1631
+ if element["kind"] == "shape" and has_text_content(element):
1632
+ estimated = own_text_visual_bbox(element)
1633
+ return clipped_bbox(estimated, slide_bbox) if estimated else None
1634
+ if element["kind"] == "line":
1635
+ # a straight horizontal/vertical line has zero width or height in one axis; clipped_bbox
1636
+ # treats zero-area rects as invisible, so pad to its rendered stroke thickness instead.
1637
+ return clipped_bbox(line_stroke_bbox(element), slide_bbox)
1638
+ if element["kind"] in {"img", "chart", "table", "whiteboard", "icon", "polyline"}:
1639
+ return clipped_bbox(element, slide_bbox)
1640
+ return None
1641
+
1642
+
1643
+ def line_stroke_bbox(element: dict[str, Any]) -> dict[str, Any]:
1644
+ return {**element, "width": max(element["width"], 1), "height": max(element["height"], 1)}
1645
+
1646
+
1647
+ def is_slide_content_present(
1648
+ element: dict[str, Any], slide_bbox: dict[str, int | float]
1649
+ ) -> bool:
1650
+ # Deliberately permissive, unlike slide_content_visual_bbox: blank_slide is asking "is
1651
+ # *anything* rendered here", not the richer "counts toward meaningful content density" bar
1652
+ # that sparse_slide_content/sparse_container_content apply. A plain shape with no text (a
1653
+ # decorative rect/ellipse/etc.), <undefined>, or any future SXSD data element should all
1654
+ # count here — deny-list only what's actually invisible (alpha<=0 or zero on-canvas area)
1655
+ # instead of maintaining an allow-list that silently treats unlisted kinds as blank.
1656
+ if not is_visually_rendered(element):
1657
+ return False
1658
+ if (
1659
+ element["kind"] == "shape"
1660
+ and element["type"] == "rect"
1661
+ and not has_text_content(element)
1662
+ and element["x"] <= 2
1663
+ and element["y"] <= 2
1664
+ and element["width"] >= slide_bbox["width"] - 4
1665
+ and element["height"] >= slide_bbox["height"] - 4
1666
+ ):
1667
+ # A full-canvas plain rect is a background panel, not content -- same reasoning as
1668
+ # is_layout_container's existing background exclusion. A slide with nothing else on it
1669
+ # is still effectively blank.
1670
+ return False
1671
+ bbox = line_stroke_bbox(element) if element["kind"] == "line" else element
1672
+ return clipped_bbox(bbox, slide_bbox) is not None
1673
+
1674
+
1675
+ def is_large_visual_child(element: dict[str, Any], container: dict[str, Any]) -> bool:
1676
+ if element["kind"] not in {"img", "chart", "table", "whiteboard"}:
1677
+ return False
1678
+ if not is_visually_rendered(element):
1679
+ return False
1680
+ return element_area(element) / element_area(container) >= LARGE_VISUAL_CHILD_RATIO
1681
+
1682
+
1683
+ def detect_sparse_container_content(
1684
+ elements: list[dict[str, Any]], slide_number: int, slide_width: int | float, slide_height: int | float
1685
+ ) -> list[dict[str, Any]]:
1686
+ issues: list[dict[str, Any]] = []
1687
+ for container in (
1688
+ element for element in elements if is_layout_container(element, slide_width, slide_height, elements)
1689
+ ):
1690
+ if (
1691
+ is_edge_spanning_layout_panel(container, slide_width, slide_height)
1692
+ or is_nested_in_layout_panel(container, elements, slide_width, slide_height)
1693
+ or has_matching_image_overlay(container, elements)
1694
+ ):
1695
+ continue
1696
+ children = [
1697
+ element
1698
+ for element in elements
1699
+ if element is not container
1700
+ and contains(container, element, tolerance=DENSITY_CONTAINMENT_TOLERANCE)
1701
+ ]
1702
+ if any(is_large_visual_child(child, container) for child in children):
1703
+ continue
1704
+ own_text_bbox = own_text_visual_bbox(container)
1705
+ rectangles = ([own_text_bbox] if own_text_bbox else []) + [
1706
+ bbox for child in children if (bbox := visual_bbox(child, container)) is not None
1707
+ ]
1708
+ content_area = rectangle_union_area(rectangles) if rectangles else 0
1709
+ coverage_ratio = content_area / element_area(container)
1710
+ if coverage_ratio >= MIN_CONTENT_COVERAGE_RATIO:
1711
+ continue
1712
+ issues.append(
1713
+ {
1714
+ "level": "warning",
1715
+ "code": "sparse_container_content",
1716
+ "target": {
1717
+ "slide_number": slide_number,
1718
+ "container_id": container["id"],
1719
+ "container_type": container["type"],
1720
+ "bbox": {key: container[key] for key in ("x", "y", "width", "height")},
1721
+ },
1722
+ "rule": {
1723
+ "name": "large_container_visible_content_coverage",
1724
+ "threshold": MIN_CONTENT_COVERAGE_RATIO,
1725
+ "comparison": "content_coverage_ratio < threshold",
1726
+ },
1727
+ "measurement": {
1728
+ "container_area": element_area(container),
1729
+ "visible_content_area": round(content_area, 3),
1730
+ "content_coverage_ratio": round(coverage_ratio, 3),
1731
+ "content_element_count": len(children) + (1 if own_text_bbox else 0),
1732
+ },
1733
+ "elements": [container["id"], *[child["id"] for child in children]],
1734
+ }
1735
+ )
1736
+ return issues
1737
+
1738
+
1739
+ def detect_sparse_slide_content(
1740
+ elements: list[dict[str, Any]], slide_number: int, slide_width: int | float, slide_height: int | float
1741
+ ) -> list[dict[str, Any]]:
1742
+ slide_bbox = {"x": 0, "y": 0, "width": slide_width, "height": slide_height}
1743
+ content = [
1744
+ (element, bbox)
1745
+ for element in elements
1746
+ if (bbox := slide_content_visual_bbox(element, slide_bbox)) is not None
1747
+ ]
1748
+ if len(content) < MIN_SLIDE_CONTENT_ELEMENT_COUNT:
1749
+ return []
1750
+ content_area = rectangle_union_area([bbox for _, bbox in content])
1751
+ slide_area = slide_width * slide_height
1752
+ coverage_ratio = content_area / slide_area
1753
+ if coverage_ratio >= MIN_SLIDE_CONTENT_COVERAGE_RATIO:
1754
+ return []
1755
+ return [
1756
+ {
1757
+ "level": "warning",
1758
+ "code": "sparse_slide_content",
1759
+ "target": {
1760
+ "slide_number": slide_number,
1761
+ "bbox": slide_bbox,
1762
+ },
1763
+ "rule": {
1764
+ "name": "slide_visible_content_coverage",
1765
+ "threshold": MIN_SLIDE_CONTENT_COVERAGE_RATIO,
1766
+ "comparison": "content_coverage_ratio < threshold",
1767
+ },
1768
+ "measurement": {
1769
+ "slide_area": slide_area,
1770
+ "visible_content_area": round(content_area, 3),
1771
+ "content_coverage_ratio": round(coverage_ratio, 3),
1772
+ "content_element_count": len(content),
1773
+ },
1774
+ "elements": [element["id"] for element, _ in content],
1775
+ }
1776
+ ]
1777
+
1778
+
1779
+ def detect_blank_slide(
1780
+ elements: list[dict[str, Any]],
1781
+ slide_number: int,
1782
+ slide_width: int | float,
1783
+ slide_height: int | float,
1784
+ ) -> list[dict[str, Any]]:
1785
+ slide_bbox = {"x": 0, "y": 0, "width": slide_width, "height": slide_height}
1786
+ visible_elements = [
1787
+ element for element in elements if is_slide_content_present(element, slide_bbox)
1788
+ ]
1789
+ if visible_elements:
1790
+ return []
1791
+ return [
1792
+ {
1793
+ "level": "error",
1794
+ "code": "blank_slide",
1795
+ "schema_version": "2.0",
1796
+ "target": {"slide_number": slide_number},
1797
+ "rule": {
1798
+ "name": "slide_has_visible_content",
1799
+ "comparison": "visible_element_count == 0",
1800
+ },
1801
+ "measurement": {
1802
+ "visible_element_count": 0,
1803
+ "declared_element_count": len(elements),
1804
+ },
1805
+ "elements": [element["id"] for element in elements],
1806
+ "message": "slide has no visible content beyond empty layout shapes",
1807
+ "hint": "Add visible text, an image, a chart, a table, a whiteboard, or an icon before creating the slide.",
1808
+ }
1809
+ ]
1810
+
1811
+
1812
+
1813
+ RULE_METADATA: dict[str, dict[str, Any]] = {
1814
+ "xml_not_well_formed": {
1815
+ "name": "xml_is_well_formed",
1816
+ "comparison": "xml_parse_error == false",
1817
+ },
1818
+ "sml_prefixed_tag": {
1819
+ "name": "sml_uses_default_namespace",
1820
+ "comparison": "prefixed_sml_tag_count == 0",
1821
+ },
1822
+ "sxsd_unsupported_tag": {
1823
+ "name": "tag_is_supported_by_slides_xml_schema",
1824
+ "comparison": "unsupported_tag_count == 0",
1825
+ },
1826
+ "sxsd_unsupported_attr": {
1827
+ "name": "attribute_is_supported_by_slides_xml_schema",
1828
+ "comparison": "unsupported_attribute_count == 0",
1829
+ },
1830
+ "icon_missing_fill_color": {
1831
+ "name": "icon_has_visible_fill_color",
1832
+ "comparison": "fill_color_present == true",
1833
+ },
1834
+ "icon_transparent_fill_color": {
1835
+ "name": "icon_has_visible_fill_color",
1836
+ "comparison": "fill_alpha > 0",
1837
+ },
1838
+ "iconpark_unsupported_icon_type": {
1839
+ "name": "iconpark_type_is_supported",
1840
+ "comparison": "icon_type in iconpark_index",
1841
+ },
1842
+ "bbox_overlap": {
1843
+ "name": "text_visual_bounds_do_not_overlap",
1844
+ "comparison": "intersection_area == 0",
1845
+ },
1846
+ "text_may_overflow_shape": {
1847
+ "name": "estimated_text_fits_declared_shape",
1848
+ "comparison": "estimated_height <= available_height",
1849
+ },
1850
+ "whiteboard_external_overlap": {
1851
+ "name": "whiteboard_does_not_cross_sibling_content",
1852
+ "comparison": "external_overlap_count == 0",
1853
+ },
1854
+ "image_covers_text": {
1855
+ "name": "image_does_not_cover_text",
1856
+ "comparison": "intersection_area == 0",
1857
+ },
1858
+ "image_may_cover_vertical_text": {
1859
+ "name": "image_vertical_text_occlusion_requires_review",
1860
+ "comparison": "intersection_area == 0",
1861
+ },
1862
+ "table_resolved_size_mismatch": {
1863
+ "name": "table_declared_size_matches_resolved_grid",
1864
+ "comparison": "declared_size == resolved_size",
1865
+ },
1866
+ "blank_slide": {
1867
+ "name": "slide_has_visible_content",
1868
+ "comparison": "visible_element_count > 0",
1869
+ },
1870
+ }
1871
+
1872
+
1873
+ def issue_rule(issue: dict[str, Any]) -> dict[str, Any]:
1874
+ if issue.get("rule"):
1875
+ return {**issue["rule"], "id": issue["code"]}
1876
+ if issue["code"].endswith("_out_of_canvas"):
1342
1877
  return {
1343
- "file": source_path,
1344
- "slide_size": {"width": 960, "height": 540},
1345
- "summary": {"slide_count": 0, "error_count": 1, "warning_count": 0, "info_count": 0},
1346
- "issues": [xml_error],
1347
- "slides": [],
1878
+ "id": issue["code"],
1879
+ "name": "element_stays_within_slide_canvas",
1880
+ "comparison": "max(left, top, right, bottom overflow) == 0",
1348
1881
  }
1882
+ return {
1883
+ "id": issue["code"],
1884
+ **RULE_METADATA.get(
1885
+ issue["code"],
1886
+ {"name": issue["code"], "comparison": "violation_count == 0"},
1887
+ ),
1888
+ }
1349
1889
 
1350
- namespace_issues = validate_sml_tag_prefixes(xml)
1351
- sxsd_issues = validate_sxsd_tag_attributes(root) if root is not None else []
1352
- iconpark_issues = validate_iconpark_icon_types(root) if root is not None else []
1353
- top_level_issues = [*namespace_issues, *sxsd_issues, *iconpark_issues]
1354
- if namespace_issues:
1355
- error_count = sum(1 for issue in top_level_issues if issue["level"] == "error")
1356
- warning_count = sum(1 for issue in top_level_issues if issue["level"] == "warning")
1357
- info_count = sum(1 for issue in top_level_issues if issue["level"] == "info")
1890
+
1891
+ def issue_measurement(
1892
+ issue: dict[str, Any], elements_by_id: dict[str, dict[str, Any]]
1893
+ ) -> dict[str, Any]:
1894
+ if issue.get("measurement") is not None:
1895
+ return issue["measurement"]
1896
+ if issue["code"] == "bbox_overlap" and len(issue.get("elements", [])) == 2:
1897
+ left = elements_by_id.get(issue["elements"][0])
1898
+ right = elements_by_id.get(issue["elements"][1])
1899
+ if left and right:
1900
+ left_box = (estimate_text_visual_bbox(left) if is_text_element(left) else None) or left
1901
+ right_box = (estimate_text_visual_bbox(right) if is_text_element(right) else None) or right
1902
+ width = intersection_width(left_box, right_box)
1903
+ height = intersection_height(left_box, right_box)
1904
+ return {
1905
+ "intersection_width": round(width, 3),
1906
+ "intersection_height": round(height, 3),
1907
+ "intersection_area": round(width * height, 3),
1908
+ }
1909
+ if issue["code"].endswith("_out_of_canvas"):
1358
1910
  return {
1359
- "file": source_path,
1360
- "slide_size": {"width": 960, "height": 540},
1361
- "summary": {
1362
- "slide_count": 0,
1363
- "error_count": error_count,
1364
- "warning_count": warning_count,
1365
- "info_count": info_count,
1366
- },
1367
- "issues": top_level_issues,
1368
- "slides": [],
1911
+ "canvas": issue.get("canvas"),
1912
+ "bbox": issue.get("bbox"),
1913
+ "overflow": issue.get("overflow"),
1369
1914
  }
1370
- presentation = parse_presentation(xml)
1371
- slides = [
1372
- lint_slide(slide_xml, index + 1, presentation["width"], presentation["height"])
1373
- for index, slide_xml in enumerate(presentation["slides"])
1915
+ measurement_keys = (
1916
+ "line",
1917
+ "column",
1918
+ "tag",
1919
+ "attr",
1920
+ "iconType",
1921
+ "line_count",
1922
+ "line_height",
1923
+ "estimated_height",
1924
+ "available_height",
1925
+ "overflow",
1926
+ "dimension",
1927
+ "declared_size",
1928
+ "resolved_size",
1929
+ "resolved_sizes",
1930
+ "overlaps",
1931
+ )
1932
+ measured = {key: issue[key] for key in measurement_keys if key in issue}
1933
+ return measured or {"violation_count": 1}
1934
+
1935
+
1936
+ def related_object(element: dict[str, Any]) -> dict[str, Any]:
1937
+ return {
1938
+ "element_id": element["id"],
1939
+ "kind": element["kind"],
1940
+ "type": element["type"],
1941
+ "bbox": {key: element[key] for key in ("x", "y", "width", "height")},
1942
+ }
1943
+
1944
+
1945
+ def extract_line_elements(slide_xml: str) -> list[dict[str, Any]]:
1946
+ elements: list[dict[str, Any]] = []
1947
+ for match in re.finditer(r"<line\b([^>]*)>", slide_xml):
1948
+ attrs = match.group(1)
1949
+ start_x = extract_numeric_attribute(attrs, "startX")
1950
+ start_y = extract_numeric_attribute(attrs, "startY")
1951
+ end_x = extract_numeric_attribute(attrs, "endX")
1952
+ end_y = extract_numeric_attribute(attrs, "endY")
1953
+ if any(value is None for value in (start_x, start_y, end_x, end_y)):
1954
+ continue
1955
+ line_alpha = extract_numeric_attribute(attrs, "alpha")
1956
+ elements.append(
1957
+ {
1958
+ "id": extract_attribute(attrs, "id") or f"line-{len(elements) + 1}",
1959
+ "kind": "line",
1960
+ "type": "line",
1961
+ "x": min(start_x, end_x),
1962
+ "y": min(start_y, end_y),
1963
+ "width": abs(end_x - start_x),
1964
+ "height": abs(end_y - start_y),
1965
+ "rotation": 0,
1966
+ "alpha": line_alpha if line_alpha is not None else 1,
1967
+ "order": len(elements),
1968
+ }
1969
+ )
1970
+ return elements
1971
+
1972
+
1973
+ def normalize_issue(
1974
+ issue: dict[str, Any],
1975
+ slide_number: int | None,
1976
+ elements_by_id: dict[str, dict[str, Any]],
1977
+ ) -> dict[str, Any]:
1978
+ normalized = dict(issue)
1979
+ if normalized.get("level") == "info":
1980
+ normalized["level"] = "warning"
1981
+ element_ids = list(dict.fromkeys(normalized.get("elements", [])))
1982
+ normalized["schema_version"] = "2.0"
1983
+ normalized["element_ids"] = element_ids
1984
+ normalized["target"] = {
1985
+ **({"slide_number": slide_number} if slide_number is not None else {}),
1986
+ **normalized.get("target", {}),
1987
+ }
1988
+ normalized["rule"] = issue_rule(normalized)
1989
+ normalized["measurement"] = issue_measurement(normalized, elements_by_id)
1990
+ normalized["related_objects"] = [
1991
+ related_object(elements_by_id[element_id])
1992
+ for element_id in element_ids
1993
+ if element_id in elements_by_id
1374
1994
  ]
1375
- error_count = sum(1 for issue in top_level_issues if issue["level"] == "error")
1376
- error_count += sum(1 for slide in slides for issue in slide["issues"] if issue["level"] == "error")
1377
- warning_count = sum(1 for issue in top_level_issues if issue["level"] == "warning")
1378
- warning_count += sum(1 for slide in slides for issue in slide["issues"] if issue["level"] == "warning")
1379
- info_count = sum(1 for issue in top_level_issues if issue["level"] == "info")
1380
- info_count += sum(1 for slide in slides for issue in slide["issues"] if issue["level"] == "info")
1381
- result = {
1995
+ if normalized["code"] == "sparse_container_content":
1996
+ ratio = normalized["measurement"]["content_coverage_ratio"]
1997
+ threshold = normalized["rule"]["threshold"]
1998
+ container_id = normalized["target"].get("container_id", "unknown")
1999
+ normalized.setdefault(
2000
+ "message",
2001
+ f"large card {container_id} content coverage {ratio:.1%} is below {threshold:.1%}",
2002
+ )
2003
+ normalized.setdefault(
2004
+ "hint",
2005
+ "Review the rendered screenshot; add or enlarge meaningful content if the whitespace is not intentional.",
2006
+ )
2007
+ elif normalized["code"] == "sparse_slide_content":
2008
+ ratio = normalized["measurement"]["content_coverage_ratio"]
2009
+ threshold = normalized["rule"]["threshold"]
2010
+ normalized.setdefault(
2011
+ "message",
2012
+ f"slide visible content coverage {ratio:.1%} is below {threshold:.1%}",
2013
+ )
2014
+ normalized.setdefault(
2015
+ "hint",
2016
+ "Review the rendered screenshot to decide whether the page is intentionally sparse.",
2017
+ )
2018
+ else:
2019
+ normalized.setdefault("message", normalized["code"].replace("_", " "))
2020
+ normalized.setdefault(
2021
+ "hint", "Inspect the reported elements and adjust them to satisfy the rule comparison."
2022
+ )
2023
+ return normalized
2024
+
2025
+
2026
+ def slide_status(errors: list[dict[str, Any]], warnings: list[dict[str, Any]]) -> str:
2027
+ if errors:
2028
+ return "blocked"
2029
+ if warnings:
2030
+ return "needs_screenshot_review"
2031
+ return "passed"
2032
+
2033
+
2034
+ def build_result(
2035
+ source_path: str | None,
2036
+ slide_size: dict[str, int | float],
2037
+ top_level_issues: list[dict[str, Any]],
2038
+ slides: list[dict[str, Any]],
2039
+ ) -> dict[str, Any]:
2040
+ document_errors = [issue for issue in top_level_issues if issue["level"] == "error"]
2041
+ document_warnings = [issue for issue in top_level_issues if issue["level"] == "warning"]
2042
+ error_count = len(document_errors) + sum(len(slide["errors"]) for slide in slides)
2043
+ warning_count = len(document_warnings) + sum(len(slide["warnings"]) for slide in slides)
2044
+ all_errors = document_errors + [issue for slide in slides for issue in slide["errors"]]
2045
+ all_warnings = document_warnings + [issue for slide in slides for issue in slide["warnings"]]
2046
+ status = slide_status(all_errors, all_warnings)
2047
+ result: dict[str, Any] = {
2048
+ "schema_version": "2.0",
2049
+ "tool": "xml_text_overlap_lint",
1382
2050
  "file": source_path,
1383
- "slide_size": {"width": presentation["width"], "height": presentation["height"]},
2051
+ "slide_size": slide_size,
1384
2052
  "summary": {
1385
2053
  "slide_count": len(slides),
1386
2054
  "error_count": error_count,
1387
2055
  "warning_count": warning_count,
1388
- "info_count": info_count,
2056
+ "status": status,
2057
+ "release_ready": error_count == 0,
2058
+ "screenshot_review_required": warning_count > 0,
2059
+ },
2060
+ "document": {
2061
+ "errors": document_errors,
2062
+ "warnings": document_warnings,
1389
2063
  },
1390
2064
  "slides": slides,
1391
2065
  }
@@ -1394,6 +2068,107 @@ def lint_xml(xml: str, source_path: str | None = None) -> dict[str, Any]:
1394
2068
  return result
1395
2069
 
1396
2070
 
2071
+ def lint_xml(xml: str, source_path: str | None = None) -> dict[str, Any]:
2072
+ root, xml_error = parse_xml_root(xml)
2073
+ if xml_error:
2074
+ issue = normalize_issue(xml_error, None, {})
2075
+ return build_result(
2076
+ source_path,
2077
+ {"width": 960, "height": 540},
2078
+ [issue],
2079
+ [],
2080
+ )
2081
+ if root is None:
2082
+ raise AssertionError("parse_xml_root must return a root or error")
2083
+
2084
+ namespace_issues = validate_sml_tag_prefixes(xml)
2085
+ sxsd_issues = validate_sxsd_tag_attributes(root)
2086
+ iconpark_issues = validate_iconpark_icon_types(root)
2087
+ top_level_issues = [
2088
+ normalize_issue(issue, None, {})
2089
+ for issue in [*namespace_issues, *sxsd_issues, *iconpark_issues]
2090
+ ]
2091
+ if any(issue["level"] == "error" for issue in top_level_issues):
2092
+ return build_result(
2093
+ source_path,
2094
+ {"width": 960, "height": 540},
2095
+ top_level_issues,
2096
+ [],
2097
+ )
2098
+
2099
+ presentation = parse_presentation(xml)
2100
+ slides: list[dict[str, Any]] = []
2101
+ for index, slide_xml in enumerate(presentation["slides"]):
2102
+ slide_number = index + 1
2103
+ geometry = lint_slide(
2104
+ slide_xml,
2105
+ slide_number,
2106
+ presentation["width"],
2107
+ presentation["height"],
2108
+ )
2109
+ density_elements = extract_density_elements(slide_xml)
2110
+ extra_elements = [
2111
+ element for element in density_elements if element["kind"] in {"icon", "polyline", "line"}
2112
+ ]
2113
+ elements_by_id = {
2114
+ element["id"]: element for element in [*density_elements, *extra_elements]
2115
+ }
2116
+ # geometry["elements"] are the exact objects should_flag_overlap/detect_elements_out_of_canvas
2117
+ # decided with inside lint_slide; prefer them so measurement/related_objects stay consistent
2118
+ # with whatever actually triggered the issue, instead of density_elements' separate re-parse.
2119
+ elements_by_id.update({element["id"]: element for element in geometry["elements"]})
2120
+ extra_overflow_issues = detect_elements_out_of_canvas(
2121
+ extra_elements,
2122
+ presentation["width"],
2123
+ presentation["height"],
2124
+ )
2125
+ raw_issues = [
2126
+ *geometry["issues"],
2127
+ *extra_overflow_issues,
2128
+ *detect_blank_slide(
2129
+ density_elements,
2130
+ slide_number,
2131
+ presentation["width"],
2132
+ presentation["height"],
2133
+ ),
2134
+ *detect_sparse_container_content(
2135
+ density_elements,
2136
+ slide_number,
2137
+ presentation["width"],
2138
+ presentation["height"],
2139
+ ),
2140
+ *detect_sparse_slide_content(
2141
+ density_elements,
2142
+ slide_number,
2143
+ presentation["width"],
2144
+ presentation["height"],
2145
+ ),
2146
+ ]
2147
+ issues = [
2148
+ normalize_issue(issue, slide_number, elements_by_id)
2149
+ for issue in raw_issues
2150
+ ]
2151
+ errors = [issue for issue in issues if issue["level"] == "error"]
2152
+ warnings = [issue for issue in issues if issue["level"] == "warning"]
2153
+ slides.append(
2154
+ {
2155
+ "slide_number": slide_number,
2156
+ "status": slide_status(errors, warnings),
2157
+ "element_count": len(elements_by_id),
2158
+ "errors": errors,
2159
+ "warnings": warnings,
2160
+ "issues": issues,
2161
+ }
2162
+ )
2163
+
2164
+ return build_result(
2165
+ source_path,
2166
+ {"width": presentation["width"], "height": presentation["height"]},
2167
+ top_level_issues,
2168
+ slides,
2169
+ )
2170
+
2171
+
1397
2172
  def print_usage() -> None:
1398
2173
  print("Usage:\n python3 xml_text_overlap_lint.py --input <presentation.xml>", file=sys.stderr)
1399
2174
 
@@ -1416,6 +2191,6 @@ def run_cli(argv: list[str] | None = None) -> None:
1416
2191
  if __name__ == "__main__":
1417
2192
  try:
1418
2193
  run_cli()
1419
- except XmlTextOverlapLintError as error:
2194
+ except XmlLayoutLintError as error:
1420
2195
  print(f"xml-text-overlap-lint error: {error}", file=sys.stderr)
1421
2196
  raise SystemExit(1) from error