@amaster.ai/pi-lark 0.1.2-beta.70 → 0.1.2-beta.71

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (28) hide show
  1. package/package.json +2 -2
  2. package/skills/lark-base/references/lark-base-app.md +2 -2
  3. package/skills/lark-calendar/SKILL.md +10 -5
  4. package/skills/lark-calendar/references/lark-calendar-meeting-relation.md +99 -0
  5. package/skills/lark-calendar/references/lark-calendar-meeting.md +1 -1
  6. package/skills/lark-calendar/references/lark-calendar-recurring.md +3 -1
  7. package/skills/lark-doc/SKILL.md +1 -1
  8. package/skills/lark-doc/references/lark-doc-create-workflow.md +8 -10
  9. package/skills/lark-doc/references/lark-doc-script.md +11 -17
  10. package/skills/lark-mail/SKILL.md +9 -4
  11. package/skills/lark-mail/references/lark-mail-thread-modify.md +73 -0
  12. package/skills/lark-mail/references/lark-mail-thread-trash.md +62 -0
  13. package/skills/lark-meeting/SKILL.md +2 -2
  14. package/skills/lark-meeting/references/lark-minutes-search.md +2 -2
  15. package/skills/lark-meeting/references/lark-vc-search.md +3 -3
  16. package/skills/lark-meeting/scenes/create-and-edit-minutes.md +4 -0
  17. package/skills/lark-sheets/SKILL.md +3 -1
  18. package/skills/lark-sheets/references/lark-sheets-chart.md +66 -32
  19. package/skills/lark-sheets/references/lark-sheets-visual-standards.md +6 -3
  20. package/skills/lark-sheets/scripts/lark_chart_quality_check.py +1524 -0
  21. package/skills/lark-sheets/scripts/lark_chart_size_advisor.py +408 -0
  22. package/skills/lark-sheets/scripts/lark_chart_size_rules.py +292 -0
  23. package/skills/lark-slides/references/xml/slides_xml_schema_definition.xml +333 -20
  24. package/skills/lark-wiki/references/lark-wiki-move.md +3 -2
  25. package/skills/lark-wiki/references/lark-wiki-node-create.md +3 -2
  26. package/skills/lark-wiki/references/lark-wiki-node-delete.md +8 -4
  27. package/skills/lark-wiki/references/lark-wiki-node-get.md +7 -4
  28. package/skills/lark-sheets/scripts/lark_chart_layout_check.py +0 -472
@@ -0,0 +1,1524 @@
1
+ #!/usr/bin/env python3
2
+ # Copyright (c) 2026 Lark Technologies Pte. Ltd.
3
+ # SPDX-License-Identifier: MIT
4
+ """Check Lark Sheet chart quality, placement, and numeric source-data issues.
5
+
6
+ The single required argument is a spreadsheet URL or spreadsheet token. By
7
+ default every worksheet is checked; pass --worksheet-id to restrict the check
8
+ to one worksheet reference_id.
9
+
10
+ Numeric source checks sample at most 50 data points per series and request at
11
+ most 2000 source cells per chart across the style and typed-value reads,
12
+ including headers and gaps between series.
13
+ Sampled zero/constant values do not establish a whole-series issue.
14
+
15
+ Exit codes:
16
+ 0: check completed and no issue was found
17
+ 1: the check could not be completed (CLI/read/response error)
18
+ 2: check completed and at least one chart-quality issue was found
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import argparse
24
+ import json
25
+ import re
26
+ from typing import Any
27
+
28
+ from lark_sheet_read_cli import (
29
+ LarkCliError,
30
+ emit_error,
31
+ envelope_data,
32
+ resolve_target_sheets,
33
+ run_sheets,
34
+ sheet_identifier,
35
+ sheet_title,
36
+ )
37
+ from lark_chart_size_rules import MAX_ASPECT_RATIO, MAX_CHART_WIDTH, minimum_chart_size
38
+
39
+ ACTION = "chart_quality_check"
40
+ DEFAULT_COLUMN_WIDTH = 105.0
41
+ DEFAULT_ROW_HEIGHT = 27.0
42
+ MAX_CELL_READ_SIZE = 2_000
43
+ MAX_SOURCE_SAMPLE_POINTS = 50
44
+
45
+
46
+ CellBounds = tuple[int, int, int, int]
47
+ CellCache = dict[tuple[str, str, str, bool], dict[str, Any]]
48
+ SeriesProfile = dict[str, Any]
49
+
50
+
51
+ def _parse_a1_bounds(cell_range: str) -> CellBounds:
52
+ value = str(cell_range).rsplit("!", 1)[-1].replace("$", "")
53
+ match = re.fullmatch(r"([A-Za-z]+)(\d+)(?::([A-Za-z]+)(\d+))?", value)
54
+ if not match:
55
+ raise ValueError(f"Invalid A1 range: {cell_range!r}")
56
+ start_column = column_to_index(match.group(1))
57
+ start_row = int(match.group(2))
58
+ end_column = column_to_index(match.group(3) or match.group(1))
59
+ end_row = int(match.group(4) or match.group(2))
60
+ if end_row < start_row or end_column < start_column:
61
+ raise ValueError(f"Invalid A1 range: {cell_range!r}")
62
+ return start_row, end_row, start_column, end_column
63
+
64
+
65
+ def _format_a1_bounds(bounds: CellBounds) -> str:
66
+ start_row, end_row, start_column, end_column = bounds
67
+ return (
68
+ f"{index_to_column(start_column)}{start_row}:"
69
+ f"{index_to_column(end_column)}{end_row}"
70
+ )
71
+
72
+
73
+ def _bounds_area(bounds: CellBounds) -> int:
74
+ start_row, end_row, start_column, end_column = bounds
75
+ return (end_row - start_row + 1) * (end_column - start_column + 1)
76
+
77
+
78
+ def _merge_bounds(first: CellBounds, second: CellBounds) -> CellBounds:
79
+ return (
80
+ min(first[0], second[0]),
81
+ max(first[1], second[1]),
82
+ min(first[2], second[2]),
83
+ max(first[3], second[3]),
84
+ )
85
+
86
+
87
+ def _cluster_cell_reads(
88
+ items: list[tuple[dict[str, Any], CellBounds]],
89
+ ) -> list[dict[str, Any]]:
90
+ clusters: list[dict[str, Any]] = []
91
+ for rectangle, bounds in sorted(items, key=lambda item: (item[1][0], item[1][2])):
92
+ if clusters:
93
+ merged = _merge_bounds(clusters[-1]["bounds"], bounds)
94
+ if _bounds_area(merged) <= MAX_CELL_READ_SIZE:
95
+ clusters[-1]["bounds"] = merged
96
+ clusters[-1]["members"].append((rectangle, bounds))
97
+ continue
98
+ clusters.append({"bounds": bounds, "members": [(rectangle, bounds)]})
99
+ return clusters
100
+
101
+
102
+ def column_to_index(column: str) -> int:
103
+ value = 0
104
+ text = str(column).strip().upper()
105
+ if not text or not text.isalpha():
106
+ raise ValueError(f"Invalid column: {column!r}")
107
+ for char in text:
108
+ value = value * 26 + ord(char) - ord("A") + 1
109
+ return value - 1
110
+
111
+
112
+ def index_to_column(index: int) -> str:
113
+ if index < 0:
114
+ raise ValueError(f"Invalid column index: {index}")
115
+ chars: list[str] = []
116
+ value = index + 1
117
+ while value:
118
+ value, remainder = divmod(value - 1, 26)
119
+ chars.append(chr(ord("A") + remainder))
120
+ return "".join(reversed(chars))
121
+
122
+
123
+ def _span_bounds(span: str, *, columns: bool) -> tuple[int, int]:
124
+ start, separator, end = str(span).partition(":")
125
+ end = end if separator else start
126
+ if columns:
127
+ return column_to_index(start), column_to_index(end)
128
+ return int(start) - 1, int(end) - 1
129
+
130
+
131
+ def _size_edges(
132
+ groups: Any,
133
+ *,
134
+ count: int,
135
+ span_key: str,
136
+ size_key: str,
137
+ columns: bool,
138
+ default_size: float,
139
+ ) -> tuple[list[float], bool]:
140
+ sizes: list[float | None] = [None] * count
141
+ if isinstance(groups, list):
142
+ for group in groups:
143
+ if not isinstance(group, dict) or group.get(span_key) is None:
144
+ continue
145
+ start, end = _span_bounds(str(group[span_key]), columns=columns)
146
+ size = float(group.get(size_key, default_size))
147
+ for index in range(max(0, start), min(count - 1, end) + 1):
148
+ sizes[index] = max(0.0, size)
149
+
150
+ used_default = any(size is None for size in sizes)
151
+ resolved = [default_size if size is None else size for size in sizes]
152
+ edges = [0.0]
153
+ for size in resolved:
154
+ edges.append(edges[-1] + size)
155
+ return edges, used_default
156
+
157
+
158
+ def build_layout(
159
+ structure: dict[str, Any], row_count: int, column_count: int
160
+ ) -> tuple[list[float], list[float], list[str]]:
161
+ row_groups = structure.get("row_heights")
162
+ column_groups = structure.get("col_widths", structure.get("column_widths"))
163
+ row_edges, row_defaulted = _size_edges(
164
+ row_groups,
165
+ count=row_count,
166
+ span_key="rows",
167
+ size_key="height",
168
+ columns=False,
169
+ default_size=DEFAULT_ROW_HEIGHT,
170
+ )
171
+ column_edges, column_defaulted = _size_edges(
172
+ column_groups,
173
+ count=column_count,
174
+ span_key="cols",
175
+ size_key="width",
176
+ columns=True,
177
+ default_size=DEFAULT_COLUMN_WIDTH,
178
+ )
179
+ warnings: list[str] = []
180
+ if row_defaulted:
181
+ warnings.append("部分行缺少高度信息,按 27 px 估算")
182
+ if column_defaulted:
183
+ warnings.append("部分列缺少宽度信息,按 105 px 估算")
184
+ return row_edges, column_edges, warnings
185
+
186
+
187
+ def _first_dict(value: Any) -> dict[str, Any] | None:
188
+ if isinstance(value, dict):
189
+ return value
190
+ if isinstance(value, list):
191
+ return next((item for item in value if isinstance(item, dict)), None)
192
+ return None
193
+
194
+
195
+ def extract_sheet_structure(data: dict[str, Any]) -> dict[str, Any]:
196
+ sheet = _first_dict(data.get("sheets")) or _first_dict(data.get("sheet"))
197
+ return sheet or data
198
+
199
+
200
+ def extract_charts(data: dict[str, Any], sheet_id: str, title: str) -> list[dict[str, Any]]:
201
+ sheets = data.get("sheets")
202
+ if isinstance(sheets, list):
203
+ for sheet in sheets:
204
+ if not isinstance(sheet, dict):
205
+ continue
206
+ if sheet_identifier(sheet) == sheet_id or sheet_title(sheet) == title:
207
+ charts = sheet.get("charts")
208
+ return [chart for chart in charts if isinstance(chart, dict)] if isinstance(charts, list) else []
209
+ charts = data.get("charts")
210
+ return [chart for chart in charts if isinstance(chart, dict)] if isinstance(charts, list) else []
211
+
212
+
213
+ def chart_rectangle(
214
+ chart: dict[str, Any], row_edges: list[float], column_edges: list[float]
215
+ ) -> dict[str, Any]:
216
+ details = chart.get("details") if isinstance(chart.get("details"), dict) else chart
217
+ position = details.get("position") if isinstance(details.get("position"), dict) else {}
218
+ offset = details.get("offset") if isinstance(details.get("offset"), dict) else {}
219
+ size = details.get("size") if isinstance(details.get("size"), dict) else {}
220
+
221
+ row = int(position["row"])
222
+ column = column_to_index(str(position["col"]))
223
+ if row < 0 or column < 0 or row >= len(row_edges) - 1 or column >= len(column_edges) - 1:
224
+ raise ValueError(f"anchor outside sheet: {position!r}")
225
+
226
+ width = float(size["width"])
227
+ height = float(size["height"])
228
+ if width <= 0 or height <= 0:
229
+ raise ValueError(f"invalid chart size: {size!r}")
230
+
231
+ left = column_edges[column] + float(offset.get("col_offset", 0) or 0)
232
+ top = row_edges[row] + float(offset.get("row_offset", 0) or 0)
233
+ return {
234
+ "chart_id": str(chart.get("chart_id") or chart.get("id") or ""),
235
+ "anchor_cell": f"{index_to_column(column)}{row + 1}",
236
+ "left": left,
237
+ "top": top,
238
+ "right": left + width,
239
+ "bottom": top + height,
240
+ "width": width,
241
+ "height": height,
242
+ }
243
+
244
+
245
+ def intersection(first: dict[str, Any], second: dict[str, Any]) -> dict[str, float] | None:
246
+ left = max(float(first["left"]), float(second["left"]))
247
+ top = max(float(first["top"]), float(second["top"]))
248
+ right = min(float(first["right"]), float(second["right"]))
249
+ bottom = min(float(first["bottom"]), float(second["bottom"]))
250
+ if right <= left or bottom <= top:
251
+ return None
252
+ return {
253
+ "width": round(right - left, 2),
254
+ "height": round(bottom - top, 2),
255
+ "area": round((right - left) * (bottom - top), 2),
256
+ }
257
+
258
+
259
+ def chart_context(rectangle: dict[str, Any]) -> dict[str, Any]:
260
+ return {
261
+ "chart_id": rectangle["chart_id"],
262
+ "anchor_cell": rectangle["anchor_cell"],
263
+ "rectangle_px": {
264
+ "left": round(rectangle["left"], 2),
265
+ "top": round(rectangle["top"], 2),
266
+ "right": round(rectangle["right"], 2),
267
+ "bottom": round(rectangle["bottom"], 2),
268
+ "width": round(rectangle["width"], 2),
269
+ "height": round(rectangle["height"], 2),
270
+ },
271
+ }
272
+
273
+
274
+ def _covered_indexes(edges: list[float], start: float, end: float) -> list[int]:
275
+ return [
276
+ index
277
+ for index in range(len(edges) - 1)
278
+ if edges[index + 1] > start and edges[index] < end
279
+ ]
280
+
281
+
282
+ def rectangle_cell_range(
283
+ rectangle: dict[str, Any], row_edges: list[float], column_edges: list[float]
284
+ ) -> str | None:
285
+ rows = _covered_indexes(row_edges, max(0.0, rectangle["top"]), rectangle["bottom"])
286
+ columns = _covered_indexes(column_edges, max(0.0, rectangle["left"]), rectangle["right"])
287
+ if not rows or not columns:
288
+ return None
289
+ return f"{index_to_column(columns[0])}{rows[0] + 1}:{index_to_column(columns[-1])}{rows[-1] + 1}"
290
+
291
+
292
+ def _has_content(cell: Any) -> bool:
293
+ if not isinstance(cell, dict):
294
+ return False
295
+ for key in ("value", "formula", "note"):
296
+ value = cell.get(key)
297
+ if value not in (None, ""):
298
+ return True
299
+ return bool(cell.get("rich_text") or cell.get("multiple_values"))
300
+
301
+
302
+ def _iter_cells(data: dict[str, Any]):
303
+ ranges = data.get("ranges")
304
+ if not isinstance(ranges, list):
305
+ return
306
+ for result_range in ranges:
307
+ if not isinstance(result_range, dict):
308
+ continue
309
+ cells = result_range.get("cells")
310
+ rows = result_range.get("row_indices")
311
+ columns = result_range.get("col_indices")
312
+ if not isinstance(cells, list):
313
+ continue
314
+ for row_offset, row in enumerate(cells):
315
+ if not isinstance(row, list):
316
+ continue
317
+ row_number = int(rows[row_offset]) if isinstance(rows, list) and row_offset < len(rows) else row_offset + 1
318
+ for column_offset, cell in enumerate(row):
319
+ column = columns[column_offset] if isinstance(columns, list) and column_offset < len(columns) else index_to_column(column_offset)
320
+ yield row_number, column_to_index(str(column)), cell
321
+
322
+
323
+ def non_empty_cells(
324
+ data: dict[str, Any], sample_limit: int, bounds: CellBounds | None = None
325
+ ) -> tuple[int, list[str], bool]:
326
+ count = 0
327
+ samples: list[str] = []
328
+ truncated = bool(data.get("has_more"))
329
+ ranges = data.get("ranges")
330
+ if not isinstance(ranges, list):
331
+ return 0, [], truncated
332
+ for result_range in ranges:
333
+ if not isinstance(result_range, dict):
334
+ continue
335
+ truncated = truncated or bool(result_range.get("truncated"))
336
+ for row_number, column_index, cell in _iter_cells(data):
337
+ if bounds and not (
338
+ bounds[0] <= row_number <= bounds[1]
339
+ and bounds[2] <= column_index <= bounds[3]
340
+ ):
341
+ continue
342
+ if not _has_content(cell):
343
+ continue
344
+ count += 1
345
+ if len(samples) < sample_limit:
346
+ samples.append(f"{index_to_column(column_index)}{row_number}")
347
+ return count, samples, truncated
348
+
349
+
350
+ def _read_cells(
351
+ cache: CellCache,
352
+ locator: dict[str, str],
353
+ *,
354
+ sheet_id: str | None,
355
+ sheet_name: str | None,
356
+ cell_range: str,
357
+ include: str,
358
+ skip_hidden: bool = False,
359
+ timeout: int,
360
+ ) -> dict[str, Any]:
361
+ selector = f"id:{sheet_id}" if sheet_id else f"name:{sheet_name}"
362
+ key = (selector, cell_range, include, skip_hidden)
363
+ if key not in cache:
364
+ flags: dict[str, Any] = {"range": cell_range, "include": include}
365
+ if skip_hidden:
366
+ flags["skip-hidden"] = True
367
+ cache[key] = envelope_data(
368
+ run_sheets(
369
+ "+cells-get",
370
+ **locator,
371
+ **({"sheet_id": sheet_id} if sheet_id else {"sheet_name": sheet_name}),
372
+ flags=flags,
373
+ timeout=timeout,
374
+ )
375
+ )
376
+ return cache[key]
377
+
378
+
379
+ def _read_typed_table(
380
+ cache: CellCache,
381
+ locator: dict[str, str],
382
+ *,
383
+ sheet_id: str | None,
384
+ sheet_name: str | None,
385
+ cell_range: str,
386
+ timeout: int,
387
+ ) -> tuple[list[list[Any]], list[str], bool]:
388
+ selector = f"id:{sheet_id}" if sheet_id else f"name:{sheet_name}"
389
+ key = (selector, cell_range, "typed_table", False)
390
+ if key not in cache:
391
+ cache[key] = envelope_data(
392
+ run_sheets(
393
+ "+table-get",
394
+ **locator,
395
+ **({"sheet_id": sheet_id} if sheet_id else {"sheet_name": sheet_name}),
396
+ flags={"range": cell_range, "no-header": True},
397
+ timeout=timeout,
398
+ )
399
+ )
400
+ data = cache[key]
401
+ sheets = data.get("sheets")
402
+ if not isinstance(sheets, list) or len(sheets) != 1 or not isinstance(sheets[0], dict):
403
+ return [], [], True
404
+ table = sheets[0]
405
+ rows = table.get("data")
406
+ columns = table.get("columns")
407
+ dtypes = table.get("dtypes")
408
+ if not isinstance(rows, list) or not isinstance(columns, list) or not isinstance(dtypes, dict):
409
+ return [], [], True
410
+ def value_kind(dtype: Any) -> str:
411
+ value = str(dtype or "").lower()
412
+ if value in {"number", "int64", "float64"}:
413
+ return "number"
414
+ if value in {"bool", "boolean"}:
415
+ return "bool"
416
+ if value.startswith("datetime"):
417
+ return "date"
418
+ return "string"
419
+
420
+ types = [value_kind(dtypes.get(str(column))) for column in columns]
421
+ truncated = bool(data.get("truncated")) or bool(table.get("truncated"))
422
+ return rows, types, truncated
423
+
424
+
425
+ def _typed_cell(
426
+ rows: list[list[Any]],
427
+ column_types: list[str],
428
+ row_offset: int,
429
+ column_offset: int,
430
+ visible_offsets: set[tuple[int, int]],
431
+ ) -> tuple[Any, str]:
432
+ if (
433
+ row_offset < 0
434
+ or row_offset >= len(rows)
435
+ or not isinstance(rows[row_offset], list)
436
+ or column_offset < 0
437
+ or column_offset >= len(rows[row_offset])
438
+ ):
439
+ return None, ""
440
+ value = rows[row_offset][column_offset]
441
+ if isinstance(value, bool):
442
+ return value, "bool"
443
+ if isinstance(value, (int, float)):
444
+ return value, "number"
445
+ kind = column_types[column_offset] if column_offset < len(column_types) else ""
446
+ if kind != "string" or not isinstance(value, str) or not _looks_numeric(value):
447
+ return value, kind
448
+
449
+ # table-get infers one dtype per physical column. A hidden cell or another
450
+ # row-series can widen that dtype and stringify otherwise numeric cells.
451
+ # Treat such cells as unknown instead of reporting a false storage issue.
452
+ numeric_string_count = 0
453
+ for other_row_offset, row in enumerate(rows):
454
+ if not isinstance(row, list) or column_offset >= len(row):
455
+ continue
456
+ other = row[column_offset]
457
+ if other not in (None, "") and (other_row_offset, column_offset) not in visible_offsets:
458
+ return value, "unknown"
459
+ if other in (None, "") or (
460
+ isinstance(other, (int, float)) and not isinstance(other, bool)
461
+ ):
462
+ continue
463
+ if isinstance(other, str) and _looks_numeric(other):
464
+ numeric_string_count += 1
465
+ continue
466
+ return value, "unknown"
467
+ return value, "string" if numeric_string_count == 1 else "ambiguous_string"
468
+
469
+
470
+ def _chart_snapshot(chart: dict[str, Any]) -> dict[str, Any]:
471
+ details = chart.get("details") if isinstance(chart.get("details"), dict) else chart
472
+ snapshot = details.get("snapshot")
473
+ return snapshot if isinstance(snapshot, dict) else {}
474
+
475
+
476
+ def _chart_type(snapshot: dict[str, Any]) -> str:
477
+ plot_area = snapshot.get("plotArea")
478
+ plot = plot_area.get("plot") if isinstance(plot_area, dict) else None
479
+ return str(plot.get("type") or "").lower() if isinstance(plot, dict) else ""
480
+
481
+
482
+ def _plot(snapshot: dict[str, Any]) -> dict[str, Any]:
483
+ plot_area = snapshot.get("plotArea")
484
+ plot = plot_area.get("plot") if isinstance(plot_area, dict) else None
485
+ return plot if isinstance(plot, dict) else {}
486
+
487
+
488
+ def _static_series_profiles(snapshot: dict[str, Any]) -> list[SeriesProfile]:
489
+ data = snapshot.get("data")
490
+ dim2 = data.get("dim2") if isinstance(data, dict) else None
491
+ fields = dim2.get("fields") if isinstance(dim2, dict) else None
492
+ if not isinstance(fields, list):
493
+ return []
494
+ profiles: list[SeriesProfile] = []
495
+ for offset, field in enumerate(fields, start=1):
496
+ if not isinstance(field, dict):
497
+ continue
498
+ values = []
499
+ for value in field.get("parsedValues") or []:
500
+ numeric = (
501
+ float(value)
502
+ if isinstance(value, (int, float)) and not isinstance(value, bool)
503
+ else _numeric_text_value(value) if isinstance(value, str) else None
504
+ )
505
+ if numeric is not None:
506
+ values.append(numeric)
507
+ profiles.append(
508
+ {
509
+ "dimension_index": offset,
510
+ "series_name": str(field.get("name") or f"Series {offset}"),
511
+ "point_count": len(field.get("parsedValues") or []),
512
+ "numeric_value_count": len(values),
513
+ "unique_numeric_values": list(dict.fromkeys(values))[:2],
514
+ "source_sheet": "",
515
+ "source_range": "",
516
+ "series_range": "",
517
+ }
518
+ )
519
+ return profiles
520
+
521
+
522
+ def _labeled_series_indexes(
523
+ snapshot: dict[str, Any], profiles: list[SeriesProfile]
524
+ ) -> set[int]:
525
+ plot = _plot(snapshot)
526
+ available = {
527
+ int(profile["dimension_index"])
528
+ for profile in profiles
529
+ if profile.get("dimension_index") is not None
530
+ }
531
+ labeled = set(available) if isinstance(plot.get("labels"), dict) else set()
532
+ series = plot.get("series")
533
+ if isinstance(series, list):
534
+ labeled.update(
535
+ int(item["index"])
536
+ for item in series
537
+ if isinstance(item, dict)
538
+ and item.get("index") is not None
539
+ and isinstance(item.get("labels"), dict)
540
+ )
541
+ return labeled & available
542
+
543
+
544
+ def _constant_labeled_series(
545
+ chart: dict[str, Any], profiles: list[SeriesProfile]
546
+ ) -> list[dict[str, Any]]:
547
+ chart_id = str(chart.get("chart_id") or chart.get("id") or "")
548
+ snapshot = _chart_snapshot(chart)
549
+ labeled = _labeled_series_indexes(snapshot, profiles)
550
+ return [
551
+ {
552
+ "chart_id": chart_id,
553
+ "dimension_index": profile["dimension_index"],
554
+ "series_name": profile["series_name"],
555
+ "source_sheet": profile["source_sheet"],
556
+ "source_range": profile["source_range"],
557
+ "series_range": profile["series_range"],
558
+ "reason": "constant_labeled_series",
559
+ "data_point_count": profile["point_count"],
560
+ "constant_value": profile["unique_numeric_values"][0],
561
+ "suggested_fix": "remove_series_labels_or_use_one_sparse_marker",
562
+ }
563
+ for profile in profiles
564
+ if int(profile.get("dimension_index", -1)) in labeled
565
+ and profile.get("constant_check_unverifiable") is not True
566
+ and int(profile.get("numeric_value_count", 0)) >= 2
567
+ and len(profile.get("unique_numeric_values") or []) == 1
568
+ ]
569
+
570
+
571
+ def _unbound_secondary_axis(chart: dict[str, Any]) -> dict[str, Any] | None:
572
+ snapshot = _chart_snapshot(chart)
573
+ plot_area = snapshot.get("plotArea")
574
+ if not isinstance(plot_area, dict):
575
+ return None
576
+ plot = plot_area.get("plot")
577
+ if not isinstance(plot, dict) or str(plot.get("type") or "").lower() != "combo":
578
+ return None
579
+
580
+ axes = plot_area.get("axes")
581
+ has_right_axis = isinstance(axes, list) and any(
582
+ isinstance(axis, dict)
583
+ and str(axis.get("type") or "").lower() == "y"
584
+ and str(axis.get("position") or axis.get("axisPosition") or "").lower() == "right"
585
+ for axis in axes
586
+ )
587
+ series = plot.get("series")
588
+ configured_series = (
589
+ [item for item in series if isinstance(item, dict)]
590
+ if isinstance(series, list)
591
+ else []
592
+ )
593
+ if not has_right_axis or not configured_series:
594
+ return None
595
+ if str(plot.get("yAxisPosition") or "").lower() == "right" or any(
596
+ str(item.get("yAxisPosition") or "").lower() == "right"
597
+ for item in configured_series
598
+ ):
599
+ return None
600
+
601
+ return {
602
+ "chart_id": str(chart.get("chart_id") or chart.get("id") or ""),
603
+ "reason": "secondary_axis_has_no_bound_series",
604
+ "series_indexes": [
605
+ int(item["index"])
606
+ for item in configured_series
607
+ if isinstance(item.get("index"), (int, float))
608
+ ],
609
+ "suggested_fix": "bind_the_intended_combo_series_to_the_right_axis",
610
+ }
611
+
612
+
613
+ def _undersized_chart(chart: dict[str, Any]) -> dict[str, Any] | None:
614
+ details = chart.get("details") if isinstance(chart.get("details"), dict) else chart
615
+ snapshot = _chart_snapshot(chart)
616
+ chart_type = _chart_type(snapshot)
617
+ size = details.get("size")
618
+ if not isinstance(size, dict):
619
+ return None
620
+ width = size.get("width")
621
+ height = size.get("height")
622
+ if (
623
+ not isinstance(width, (int, float))
624
+ or isinstance(width, bool)
625
+ or width <= 0
626
+ or not isinstance(height, (int, float))
627
+ or isinstance(height, bool)
628
+ or height <= 0
629
+ ):
630
+ return None
631
+ actual = {
632
+ "width": float(width),
633
+ "height": float(height),
634
+ }
635
+ minimum = minimum_chart_size(chart_type)
636
+ if actual["width"] >= minimum["width"] and actual["height"] >= minimum["height"]:
637
+ return None
638
+ return {
639
+ "chart_id": str(chart.get("chart_id") or chart.get("id") or ""),
640
+ "reason": "chart_below_minimum_size",
641
+ "chart_type": chart_type,
642
+ "actual_size": actual,
643
+ "minimum_size": minimum,
644
+ "suggested_fix": "run_lark_chart_size_advisor_before_resizing",
645
+ }
646
+
647
+
648
+ def _overwide_chart(chart: dict[str, Any]) -> dict[str, Any] | None:
649
+ details = chart.get("details") if isinstance(chart.get("details"), dict) else chart
650
+ snapshot = _chart_snapshot(chart)
651
+ chart_type = _chart_type(snapshot)
652
+ size = details.get("size") if isinstance(details.get("size"), dict) else {}
653
+ width = size.get("width")
654
+ height = size.get("height")
655
+ if (
656
+ not isinstance(width, (int, float))
657
+ or isinstance(width, bool)
658
+ or width <= 0
659
+ or not isinstance(height, (int, float))
660
+ or isinstance(height, bool)
661
+ or height <= 0
662
+ ):
663
+ return None
664
+ width = float(width)
665
+ height = float(height)
666
+ minimum = minimum_chart_size(chart_type)
667
+ aspect_ratio = width / height if height > 0 else 0
668
+ width_exceeded = width > MAX_CHART_WIDTH
669
+ aspect_ratio_exceeded = (
670
+ width >= minimum["width"]
671
+ and height >= minimum["height"]
672
+ and aspect_ratio > MAX_ASPECT_RATIO
673
+ )
674
+ if not width_exceeded and not aspect_ratio_exceeded:
675
+ return None
676
+ return {
677
+ "chart_id": str(chart.get("chart_id") or chart.get("id") or ""),
678
+ "reason": "chart_too_wide",
679
+ "chart_type": chart_type,
680
+ "actual_size": {"width": width, "height": height},
681
+ "actual_aspect_ratio": round(aspect_ratio, 2),
682
+ "maximum_width": MAX_CHART_WIDTH,
683
+ "maximum_aspect_ratio": MAX_ASPECT_RATIO,
684
+ "suggested_fix": "use_size_advisor_create_flags_or_change_chart_structure",
685
+ }
686
+
687
+
688
+ def _numeric_dimensions(snapshot: dict[str, Any]) -> list[tuple[int, str]]:
689
+ data = snapshot.get("data")
690
+ if not isinstance(data, dict):
691
+ return []
692
+ dim2 = data.get("dim2")
693
+ series = dim2.get("series") if isinstance(dim2, dict) else None
694
+ chart_type = _chart_type(snapshot)
695
+ dimensions: list[tuple[int, str]] = []
696
+ if isinstance(series, list):
697
+ for offset, serie in enumerate(series):
698
+ if not isinstance(serie, dict) or serie.get("index") is None:
699
+ continue
700
+ if str(serie.get("aggregateType") or "").lower() == "counta":
701
+ continue
702
+ role = str(serie.get("role") or "").lower()
703
+ if chart_type == "bubble":
704
+ role = role or ("x", "y", "group", "size")[min(offset, 3)]
705
+ if role not in {"x", "y", "size"}:
706
+ continue
707
+ dimensions.append((int(serie["index"]), role or "value"))
708
+
709
+ plot_area = snapshot.get("plotArea")
710
+ axes = plot_area.get("axes") if isinstance(plot_area, dict) else None
711
+ continuous_x = chart_type == "scatter"
712
+ if isinstance(axes, list):
713
+ for axis in axes:
714
+ if not isinstance(axis, dict) or str(axis.get("type") or "").lower() != "x":
715
+ continue
716
+ position = axis.get("position")
717
+ if (
718
+ (position is None or str(position).lower() in {"bottom", "x"})
719
+ and str(axis.get("valueType") or "").lower() == "linear"
720
+ ):
721
+ continuous_x = True
722
+ break
723
+ dim1 = data.get("dim1")
724
+ serie = dim1.get("serie") if isinstance(dim1, dict) else None
725
+ if continuous_x and chart_type != "bubble" and isinstance(serie, dict) and serie.get("index") is not None:
726
+ dimensions.append((int(serie["index"]), "x"))
727
+
728
+ return list(dict.fromkeys(dimensions))
729
+
730
+
731
+ def _series_aggregate_type(data: dict[str, Any], dimension_index: int) -> str:
732
+ dim2 = data.get("dim2")
733
+ value_series = dim2.get("series") if isinstance(dim2, dict) else None
734
+ source_series = next(
735
+ (
736
+ item
737
+ for item in (value_series if isinstance(value_series, list) else [])
738
+ if isinstance(item, dict)
739
+ and item.get("index") is not None
740
+ and int(item["index"]) == dimension_index
741
+ ),
742
+ {},
743
+ )
744
+ return str(source_series.get("aggregateType") or "sum").lower()
745
+
746
+
747
+ def _aggregation_can_change_constant(data: dict[str, Any], dimension_index: int) -> bool:
748
+ dim1 = data.get("dim1")
749
+ category_series = dim1.get("serie") if isinstance(dim1, dict) else None
750
+ if not isinstance(category_series, dict) or category_series.get("aggregate") is False:
751
+ return False
752
+ return _series_aggregate_type(data, dimension_index) in {"sum", "count", "counta"}
753
+
754
+
755
+ def _parse_chart_ref(value: str, default_sheet: str) -> tuple[str, str, CellBounds]:
756
+ raw = str(value).strip()
757
+ sheet_name = default_sheet
758
+ cell_range = raw
759
+ if "!" in raw:
760
+ sheet_name, cell_range = raw.rsplit("!", 1)
761
+ sheet_name = sheet_name.strip()
762
+ if len(sheet_name) >= 2 and sheet_name[0] == sheet_name[-1] == "'":
763
+ sheet_name = sheet_name[1:-1].replace("''", "'")
764
+ return sheet_name, cell_range.replace("$", ""), _parse_a1_bounds(cell_range)
765
+
766
+
767
+ def _looks_numeric(value: str) -> bool:
768
+ return _numeric_text_value(value) is not None
769
+
770
+
771
+ def _numeric_text_value(value: str) -> float | None:
772
+ text = value.strip()
773
+ if not text:
774
+ return None
775
+ text = re.sub(r"^([+-]?)[\$\u00a5\uffe5\u20ac\u00a3]", r"\1", text)
776
+ if text.endswith("%"):
777
+ text = text[:-1]
778
+ # Group separators must be consistent and separate groups of three digits.
779
+ if not re.fullmatch(
780
+ r"[+-]?(?:(?:\d+|\d{1,3}([, ])\d{3}(?:\1\d{3})*)(?:\.\d*)?|\.\d+)"
781
+ r"(?:[eE][+-]?\d+)?",
782
+ text,
783
+ ):
784
+ return None
785
+ return float(text.replace(" ", "").replace(",", ""))
786
+
787
+
788
+ def _zero_state(value: Any) -> str:
789
+ if value in (None, ""):
790
+ return "empty"
791
+ if isinstance(value, (int, float)) and not isinstance(value, bool):
792
+ return "zero" if value == 0 else "nonzero"
793
+ if isinstance(value, str):
794
+ numeric_value = _numeric_text_value(value)
795
+ if numeric_value is not None:
796
+ return "zero" if numeric_value == 0 else "nonzero"
797
+ return "nonzero"
798
+
799
+
800
+ def _update_series_state(state: dict[str, Any], value: Any) -> None:
801
+ state[_zero_state(value)] = True
802
+ numeric = (
803
+ float(value)
804
+ if isinstance(value, (int, float)) and not isinstance(value, bool)
805
+ else _numeric_text_value(value) if isinstance(value, str) else None
806
+ )
807
+ if numeric is None:
808
+ return
809
+ state["numeric_value_count"] += 1
810
+ if len(state["unique_numeric_values"]) < 2:
811
+ state["unique_numeric_values"].add(numeric)
812
+
813
+
814
+ def _cells_truncated(data: dict[str, Any]) -> bool:
815
+ ranges = data.get("ranges")
816
+ return bool(data.get("has_more")) or any(
817
+ isinstance(item, dict) and item.get("truncated")
818
+ for item in (ranges if isinstance(ranges, list) else [])
819
+ )
820
+
821
+
822
+ def _numeric_source_issues(
823
+ chart: dict[str, Any],
824
+ *,
825
+ owner_sheet_id: str,
826
+ owner_sheet_name: str,
827
+ cache: CellCache,
828
+ locator: dict[str, str],
829
+ timeout: int,
830
+ sample_limit: int,
831
+ ) -> tuple[
832
+ list[dict[str, Any]],
833
+ list[dict[str, Any]],
834
+ list[dict[str, str]],
835
+ list[SeriesProfile],
836
+ ]:
837
+ chart_id = str(chart.get("chart_id") or chart.get("id") or "")
838
+ snapshot = _chart_snapshot(chart)
839
+ data = snapshot.get("data")
840
+ if not isinstance(data, dict):
841
+ return [], [], [{"chart_id": chart_id, "reason": "chart snapshot.data is missing"}], []
842
+ if data.get("isStaticData") is True:
843
+ return [], [], [], _static_series_profiles(snapshot)
844
+ dimensions = _numeric_dimensions(snapshot)
845
+ if not dimensions:
846
+ return [], [], [], []
847
+ refs = data.get("refs")
848
+ if not isinstance(refs, list) or not refs:
849
+ return [], [], [{"chart_id": chart_id, "reason": "chart data.refs is missing"}], []
850
+
851
+ parsed_refs: list[tuple[str, str, CellBounds]] = []
852
+ unverifiable: list[dict[str, str]] = []
853
+ for ref in refs:
854
+ raw_ref = ref.get("value") if isinstance(ref, dict) else ref
855
+ try:
856
+ parsed_refs.append(_parse_chart_ref(str(raw_ref), owner_sheet_name))
857
+ except (TypeError, ValueError) as exc:
858
+ unverifiable.append({"chart_id": chart_id, "reason": str(exc)})
859
+ return [], [], unverifiable, []
860
+
861
+ direction = str(data.get("direction") or "column").lower()
862
+ skip_hidden = data.get("includeHiddenOrFilter") is not True
863
+ mapped: list[tuple[int, str, int, str, str, CellBounds]] = []
864
+ for dimension_index, role in dimensions:
865
+ offset = 0
866
+ for source_sheet, source_range, bounds in parsed_refs:
867
+ dimension_count = (
868
+ bounds[3] - bounds[2] + 1 if direction == "column" else bounds[1] - bounds[0] + 1
869
+ )
870
+ if offset < dimension_index <= offset + dimension_count:
871
+ mapped.append(
872
+ (
873
+ dimension_index,
874
+ role,
875
+ dimension_index - offset,
876
+ source_sheet,
877
+ source_range,
878
+ bounds,
879
+ )
880
+ )
881
+ break
882
+ offset += dimension_count
883
+ else:
884
+ unverifiable.append(
885
+ {
886
+ "chart_id": chart_id,
887
+ "reason": f"numeric dimension index {dimension_index} is outside data.refs",
888
+ }
889
+ )
890
+
891
+ detached = str(data.get("headerMode") or "").lower() == "detached"
892
+ mapped_by_ref: dict[
893
+ tuple[str, str, CellBounds], list[tuple[int, str, int]]
894
+ ] = {}
895
+ for dimension_index, role, local_index, source_sheet, source_range, bounds in mapped:
896
+ mapped_by_ref.setdefault((source_sheet, source_range, bounds), []).append(
897
+ (dimension_index, role, local_index)
898
+ )
899
+
900
+ issue_groups: dict[tuple[int, str, str, str, str, str], list[str]] = {}
901
+ issue_counts: dict[tuple[int, str, str, str, str, str], int] = {}
902
+ degenerate_series: list[dict[str, Any]] = []
903
+ series_profiles: list[SeriesProfile] = []
904
+ remaining_source_cells = MAX_CELL_READ_SIZE
905
+ remaining_dimension_spans = sum(
906
+ max(local_index for _, _, local_index in ref_dimensions)
907
+ - min(local_index for _, _, local_index in ref_dimensions)
908
+ + 1
909
+ for ref_dimensions in mapped_by_ref.values()
910
+ )
911
+ for (source_sheet, source_range, bounds), ref_dimensions in mapped_by_ref.items():
912
+ dimension_start = bounds[2] if direction == "column" else bounds[0]
913
+ selected = {
914
+ dimension_start + local_index - 1: (dimension_index, role)
915
+ for dimension_index, role, local_index in ref_dimensions
916
+ }
917
+ dimension_span = max(selected) - min(selected) + 1
918
+ header_points = 0 if detached else 1
919
+ # Each sampled rectangle is read twice: cells-get supplies coordinates,
920
+ # styles, and hidden/filter semantics; table-get supplies typed values.
921
+ cells_per_dimension = remaining_source_cells // remaining_dimension_spans
922
+ remaining_dimension_spans -= dimension_span
923
+ point_axis_size = (
924
+ bounds[1] - bounds[0] + 1
925
+ if direction == "column"
926
+ else bounds[3] - bounds[2] + 1
927
+ )
928
+ if point_axis_size <= header_points:
929
+ unverifiable.append(
930
+ {
931
+ "chart_id": chart_id,
932
+ "reason": f"source has no data points: {source_sheet}!{source_range}",
933
+ }
934
+ )
935
+ continue
936
+ sample_points = min(MAX_SOURCE_SAMPLE_POINTS, max(0, (cells_per_dimension - header_points) // 2))
937
+ if sample_points == 0:
938
+ unverifiable.append({
939
+ "chart_id": chart_id,
940
+ "reason": f"source sampling 2000-cell budget cannot cover {source_sheet}!{source_range}",
941
+ })
942
+ continue
943
+ point_count = sample_points + header_points
944
+ if direction == "column":
945
+ checked_bounds = (
946
+ bounds[0],
947
+ min(bounds[1], bounds[0] + point_count - 1),
948
+ min(selected),
949
+ max(selected),
950
+ )
951
+ else:
952
+ checked_bounds = (
953
+ min(selected),
954
+ max(selected),
955
+ bounds[2],
956
+ min(bounds[3], bounds[2] + point_count - 1),
957
+ )
958
+ typed_bounds = (
959
+ (checked_bounds[0] + header_points, checked_bounds[1], checked_bounds[2], checked_bounds[3])
960
+ if direction == "column"
961
+ else (checked_bounds[0], checked_bounds[1], checked_bounds[2] + header_points, checked_bounds[3])
962
+ )
963
+ remaining_source_cells -= _bounds_area(checked_bounds) + _bounds_area(typed_bounds)
964
+ checked_range = _format_a1_bounds(checked_bounds)
965
+ same_sheet = source_sheet == owner_sheet_name
966
+ cells_data = _read_cells(
967
+ cache,
968
+ locator,
969
+ sheet_id=owner_sheet_id if same_sheet else None,
970
+ sheet_name=None if same_sheet else source_sheet,
971
+ cell_range=checked_range,
972
+ include="value,style",
973
+ skip_hidden=skip_hidden,
974
+ timeout=timeout,
975
+ )
976
+ cells_truncated = _cells_truncated(cells_data)
977
+ typed_rows, typed_columns, table_truncated = _read_typed_table(
978
+ cache,
979
+ locator,
980
+ sheet_id=owner_sheet_id if same_sheet else None,
981
+ sheet_name=None if same_sheet else source_sheet,
982
+ cell_range=_format_a1_bounds(typed_bounds),
983
+ timeout=timeout,
984
+ )
985
+ if cells_truncated:
986
+ unverifiable.append(
987
+ {"chart_id": chart_id, "reason": f"cells-get truncated for {source_sheet}!{checked_range}"}
988
+ )
989
+ if table_truncated:
990
+ unverifiable.append(
991
+ {
992
+ "chart_id": chart_id,
993
+ "reason": (
994
+ "table-get truncated or returned invalid typed data for "
995
+ f"{source_sheet}!{_format_a1_bounds(typed_bounds)}"
996
+ ),
997
+ }
998
+ )
999
+ truncated = cells_truncated or table_truncated
1000
+ states = {
1001
+ coordinate: {
1002
+ "zero": False,
1003
+ "empty": False,
1004
+ "nonzero": False,
1005
+ "sample_point_count": 0,
1006
+ "numeric_value_count": 0,
1007
+ "unique_numeric_values": set(),
1008
+ }
1009
+ for coordinate in selected
1010
+ }
1011
+ type_unverifiable_dimensions: set[int] = set()
1012
+ visible_offsets = {
1013
+ (row_number - typed_bounds[0], column_index - typed_bounds[2])
1014
+ for row_number, column_index, _ in _iter_cells(cells_data)
1015
+ if typed_bounds[0] <= row_number <= typed_bounds[1]
1016
+ and typed_bounds[2] <= column_index <= typed_bounds[3]
1017
+ }
1018
+ for row_number, column_index, cell in _iter_cells(cells_data):
1019
+ coordinate = column_index if direction == "column" else row_number
1020
+ dimension = selected.get(coordinate)
1021
+ if dimension is None:
1022
+ continue
1023
+ if not detached and (
1024
+ (direction == "column" and row_number == bounds[0])
1025
+ or (direction != "column" and column_index == bounds[2])
1026
+ ):
1027
+ continue
1028
+ states[coordinate]["sample_point_count"] += 1
1029
+ row_offset = row_number - typed_bounds[0]
1030
+ column_offset = column_index - typed_bounds[2]
1031
+ value, raw_type = _typed_cell(
1032
+ typed_rows,
1033
+ typed_columns,
1034
+ row_offset,
1035
+ column_offset,
1036
+ visible_offsets,
1037
+ )
1038
+ _update_series_state(states[coordinate], value)
1039
+ number_format = (
1040
+ cell.get("cell_styles", {}).get("number_format")
1041
+ if isinstance(cell, dict) and isinstance(cell.get("cell_styles"), dict)
1042
+ else None
1043
+ )
1044
+ reason = ""
1045
+ if _looks_numeric(str(value or "")):
1046
+ if str(number_format or "").strip() == "@":
1047
+ reason = "numeric_value_uses_text_format"
1048
+ elif raw_type == "string":
1049
+ reason = "numeric_value_stored_as_text"
1050
+ elif raw_type in {"ambiguous_string", "unknown"}:
1051
+ type_unverifiable_dimensions.add(dimension[0])
1052
+ if reason:
1053
+ dimension_index, role = dimension
1054
+ key = (
1055
+ dimension_index,
1056
+ role,
1057
+ source_sheet,
1058
+ source_range,
1059
+ checked_range,
1060
+ reason,
1061
+ )
1062
+ issue_counts[key] = issue_counts.get(key, 0) + 1
1063
+ samples = issue_groups.setdefault(key, [])
1064
+ if len(samples) < sample_limit:
1065
+ samples.append(f"{index_to_column(column_index)}{row_number}")
1066
+
1067
+ if truncated:
1068
+ continue
1069
+ for dimension_index in sorted(type_unverifiable_dimensions):
1070
+ unverifiable.append(
1071
+ {
1072
+ "chart_id": chart_id,
1073
+ "reason": (
1074
+ "numeric storage type cannot be attributed to individual cells after "
1075
+ f"typed column coercion for {source_sheet}!{checked_range}, "
1076
+ f"dimension {dimension_index}"
1077
+ ),
1078
+ }
1079
+ )
1080
+ zero_candidates = {
1081
+ coordinate
1082
+ for coordinate, state in states.items()
1083
+ if not state["nonzero"]
1084
+ and _series_aggregate_type(data, selected[coordinate][0]) != "count"
1085
+ }
1086
+ data_start = bounds[0] + (0 if detached else 1)
1087
+ data_column = bounds[2] + (0 if detached else 1)
1088
+ dim2 = data.get("dim2")
1089
+ value_series = dim2.get("series") if isinstance(dim2, dict) else None
1090
+ for coordinate, state in states.items():
1091
+ dimension_index, role = selected[coordinate]
1092
+ if direction == "column":
1093
+ point_total = max(0, bounds[1] - data_start + 1)
1094
+ series_range = (
1095
+ f"{index_to_column(coordinate)}{data_start}:"
1096
+ f"{index_to_column(coordinate)}{bounds[1]}"
1097
+ if point_total
1098
+ else ""
1099
+ )
1100
+ else:
1101
+ point_total = max(0, bounds[3] - data_column + 1)
1102
+ series_range = (
1103
+ f"{index_to_column(data_column)}{coordinate}:"
1104
+ f"{index_to_column(bounds[3])}{coordinate}"
1105
+ if point_total
1106
+ else ""
1107
+ )
1108
+ source_series = next(
1109
+ (
1110
+ item
1111
+ for item in (value_series if isinstance(value_series, list) else [])
1112
+ if isinstance(item, dict)
1113
+ and item.get("index") is not None
1114
+ and int(item["index"]) == dimension_index
1115
+ ),
1116
+ {},
1117
+ )
1118
+ profile = {
1119
+ "dimension_index": dimension_index,
1120
+ "series_name": str(
1121
+ source_series.get("name")
1122
+ or source_series.get("nameRef")
1123
+ or f"Series {dimension_index}"
1124
+ ),
1125
+ "point_count": point_total,
1126
+ "sample_point_count": state["sample_point_count"],
1127
+ "checked_range": checked_range,
1128
+ "sampled": (
1129
+ checked_bounds[1] < bounds[1]
1130
+ if direction == "column"
1131
+ else checked_bounds[3] < bounds[3]
1132
+ ),
1133
+ "numeric_value_count": state["numeric_value_count"],
1134
+ "unique_numeric_values": list(state["unique_numeric_values"]),
1135
+ "source_sheet": source_sheet,
1136
+ "source_range": source_range,
1137
+ "series_range": series_range,
1138
+ }
1139
+ aggregate_type = _series_aggregate_type(data, dimension_index)
1140
+ if aggregate_type == "count":
1141
+ profile["constant_check_unverifiable"] = True
1142
+ constant_labeled = (
1143
+ aggregate_type != "count"
1144
+ and state["numeric_value_count"] >= 2
1145
+ and len(state["unique_numeric_values"]) == 1
1146
+ and dimension_index
1147
+ in _labeled_series_indexes(snapshot, [{"dimension_index": dimension_index}])
1148
+ )
1149
+ if profile["sampled"]:
1150
+ profile["constant_check_unverifiable"] = True
1151
+ if coordinate in zero_candidates or constant_labeled:
1152
+ unverifiable.append({
1153
+ "chart_id": chart_id,
1154
+ "reason": (
1155
+ "zero/constant-series check is unverifiable outside sampled source "
1156
+ f"{source_sheet}!{checked_range} for dimension {dimension_index}"
1157
+ ),
1158
+ })
1159
+ elif constant_labeled and _aggregation_can_change_constant(data, dimension_index):
1160
+ profile["constant_check_unverifiable"] = True
1161
+ unverifiable.append(
1162
+ {
1163
+ "chart_id": chart_id,
1164
+ "reason": (
1165
+ "constant-series check is unverifiable after category "
1166
+ f"aggregation for dimension {dimension_index}"
1167
+ ),
1168
+ }
1169
+ )
1170
+ series_profiles.append(profile)
1171
+ if coordinate not in zero_candidates or profile["sampled"]:
1172
+ continue
1173
+ degenerate_series.append(
1174
+ {
1175
+ "chart_id": chart_id,
1176
+ "dimension_index": dimension_index,
1177
+ "role": role,
1178
+ "source_sheet": source_sheet,
1179
+ "source_range": source_range,
1180
+ "series_range": series_range,
1181
+ "reason": (
1182
+ "numeric_series_all_zero_or_empty"
1183
+ if states[coordinate]["zero"]
1184
+ else "numeric_series_all_empty"
1185
+ ),
1186
+ "data_point_count": point_total,
1187
+ }
1188
+ )
1189
+
1190
+ issues = [
1191
+ {
1192
+ "chart_id": chart_id,
1193
+ "dimension_index": key[0],
1194
+ "role": key[1],
1195
+ "source_sheet": key[2],
1196
+ "source_range": key[3],
1197
+ "checked_range": key[4],
1198
+ "reason": key[5],
1199
+ "suggested_fix": (
1200
+ "set_numeric_number_format"
1201
+ if key[5] == "numeric_value_uses_text_format"
1202
+ else "rewrite_as_number_and_set_numeric_number_format"
1203
+ ),
1204
+ "affected_sample_cell_count": issue_counts[key],
1205
+ "sample_cells": samples,
1206
+ }
1207
+ for key, samples in issue_groups.items()
1208
+ ]
1209
+ return issues, degenerate_series, unverifiable, series_profiles
1210
+
1211
+
1212
+ def _locator(target: str) -> dict[str, str]:
1213
+ return {"url": target} if target.startswith(("http://", "https://")) else {"spreadsheet_token": target}
1214
+
1215
+
1216
+ def _sheet_counts(sheet: dict[str, Any]) -> tuple[int, int]:
1217
+ row_count = int(sheet.get("row_count") or sheet.get("rowCount") or 0)
1218
+ column_count = int(sheet.get("column_count") or sheet.get("columnCount") or 0)
1219
+ if row_count <= 0 or column_count <= 0:
1220
+ raise LarkCliError(f"Missing row_count/column_count for sheet {sheet_title(sheet)!r}")
1221
+ return row_count, column_count
1222
+
1223
+
1224
+ def check_sheet(
1225
+ locator: dict[str, str],
1226
+ sheet: dict[str, Any],
1227
+ *,
1228
+ timeout: int,
1229
+ sample_limit: int,
1230
+ cell_cache: CellCache | None = None,
1231
+ ) -> dict[str, Any]:
1232
+ cell_cache = cell_cache if cell_cache is not None else {}
1233
+ sheet_id = sheet_identifier(sheet)
1234
+ title = sheet_title(sheet)
1235
+ if not sheet_id:
1236
+ raise LarkCliError(f"Missing sheet_id for sheet {title!r}")
1237
+
1238
+ chart_data = envelope_data(
1239
+ run_sheets("+chart-list", **locator, sheet_id=sheet_id, timeout=timeout)
1240
+ )
1241
+ charts = extract_charts(chart_data, sheet_id, title)
1242
+ unverifiable: list[dict[str, str]] = []
1243
+ expected_chart_count = sheet.get("chart_count")
1244
+ if expected_chart_count is not None and int(expected_chart_count) != len(charts):
1245
+ unverifiable.append(
1246
+ {
1247
+ "chart_id": "",
1248
+ "reason": (
1249
+ f"chart-list returned {len(charts)} charts, "
1250
+ f"but workbook-info reported {int(expected_chart_count)}"
1251
+ ),
1252
+ }
1253
+ )
1254
+ if not charts:
1255
+ return {
1256
+ "sheet_id": sheet_id,
1257
+ "sheet_name": title,
1258
+ "chart_count": 0,
1259
+ "sheet_size_px": None,
1260
+ "chart_overlaps": [],
1261
+ "cell_content_overlaps": [],
1262
+ "numeric_source_format_issues": [],
1263
+ "numeric_source_samples": [],
1264
+ "degenerate_numeric_series": [],
1265
+ "constant_labeled_series": [],
1266
+ "unbound_secondary_axes": [],
1267
+ "undersized_charts": [],
1268
+ "overwide_charts": [],
1269
+ "out_of_visible_range": [],
1270
+ "unverifiable_charts": unverifiable,
1271
+ "issue_count": 0,
1272
+ "unverifiable_count": len(unverifiable),
1273
+ "warnings": [],
1274
+ }
1275
+
1276
+ row_count, column_count = _sheet_counts(sheet)
1277
+ structure_data = envelope_data(
1278
+ run_sheets(
1279
+ "+sheet-info",
1280
+ **locator,
1281
+ sheet_id=sheet_id,
1282
+ flags={"include": "row_heights,col_widths"},
1283
+ timeout=timeout,
1284
+ )
1285
+ )
1286
+ row_edges, column_edges, warnings = build_layout(
1287
+ extract_sheet_structure(structure_data), row_count, column_count
1288
+ )
1289
+
1290
+ rectangles: list[dict[str, Any]] = []
1291
+ for chart in charts:
1292
+ chart_id = str(chart.get("chart_id") or chart.get("id") or "")
1293
+ if not chart_id:
1294
+ unverifiable.append({"chart_id": "", "reason": "chart is missing chart_id"})
1295
+ continue
1296
+ try:
1297
+ rectangles.append(chart_rectangle(chart, row_edges, column_edges))
1298
+ except (KeyError, TypeError, ValueError) as exc:
1299
+ unverifiable.append({"chart_id": chart_id, "reason": str(exc)})
1300
+
1301
+ overlaps: list[dict[str, Any]] = []
1302
+ for index, first in enumerate(rectangles):
1303
+ for second in rectangles[index + 1 :]:
1304
+ overlap = intersection(first, second)
1305
+ if overlap:
1306
+ overlaps.append(
1307
+ {
1308
+ "chart_ids": [first["chart_id"], second["chart_id"]],
1309
+ "charts": [chart_context(first), chart_context(second)],
1310
+ "intersection": overlap,
1311
+ }
1312
+ )
1313
+
1314
+ sheet_width = column_edges[-1]
1315
+ sheet_height = row_edges[-1]
1316
+ out_of_bounds: list[dict[str, Any]] = []
1317
+ content_overlaps: list[dict[str, Any]] = []
1318
+ covered_items: list[tuple[dict[str, Any], CellBounds]] = []
1319
+ for rectangle in rectangles:
1320
+ overflow = {
1321
+ "left": round(max(0.0, -rectangle["left"]), 2),
1322
+ "top": round(max(0.0, -rectangle["top"]), 2),
1323
+ "right": round(max(0.0, rectangle["right"] - sheet_width), 2),
1324
+ "bottom": round(max(0.0, rectangle["bottom"] - sheet_height), 2),
1325
+ }
1326
+ if any(overflow.values()):
1327
+ out_of_bounds.append({**chart_context(rectangle), "overflow_px": overflow})
1328
+
1329
+ covered_range = rectangle_cell_range(rectangle, row_edges, column_edges)
1330
+ if not covered_range:
1331
+ continue
1332
+ covered_items.append((rectangle, _parse_a1_bounds(covered_range)))
1333
+
1334
+ for cluster in _cluster_cell_reads(covered_items):
1335
+ read_range = _format_a1_bounds(cluster["bounds"])
1336
+ cells_data = _read_cells(
1337
+ cell_cache,
1338
+ locator,
1339
+ sheet_id=sheet_id,
1340
+ sheet_name=None,
1341
+ cell_range=read_range,
1342
+ include="value,formula,comment",
1343
+ timeout=timeout,
1344
+ )
1345
+ for rectangle, bounds in cluster["members"]:
1346
+ covered_range = _format_a1_bounds(bounds)
1347
+ count, samples, truncated = non_empty_cells(cells_data, sample_limit, bounds)
1348
+ if truncated:
1349
+ unverifiable.append(
1350
+ {
1351
+ "chart_id": rectangle["chart_id"],
1352
+ "reason": f"cells-get truncated for {read_range}",
1353
+ }
1354
+ )
1355
+ if count:
1356
+ content_overlaps.append(
1357
+ {
1358
+ **chart_context(rectangle),
1359
+ "covered_range": covered_range,
1360
+ "non_empty_cell_count": count,
1361
+ "sample_cells": samples,
1362
+ }
1363
+ )
1364
+
1365
+ numeric_source_issues: list[dict[str, Any]] = []
1366
+ numeric_source_samples: list[dict[str, Any]] = []
1367
+ degenerate_numeric_series: list[dict[str, Any]] = []
1368
+ constant_series_issues: list[dict[str, Any]] = []
1369
+ unbound_secondary_axes: list[dict[str, Any]] = []
1370
+ undersized_charts: list[dict[str, Any]] = []
1371
+ overwide_charts: list[dict[str, Any]] = []
1372
+ for chart in charts:
1373
+ issues, degenerate, source_unverifiable, profiles = _numeric_source_issues(
1374
+ chart,
1375
+ owner_sheet_id=sheet_id,
1376
+ owner_sheet_name=title,
1377
+ cache=cell_cache,
1378
+ locator=locator,
1379
+ timeout=timeout,
1380
+ sample_limit=sample_limit,
1381
+ )
1382
+ numeric_source_issues.extend(issues)
1383
+ numeric_source_samples.extend(
1384
+ {"chart_id": str(chart.get("chart_id") or chart.get("id") or ""), **profile}
1385
+ for profile in profiles
1386
+ if "checked_range" in profile
1387
+ )
1388
+ degenerate_numeric_series.extend(degenerate)
1389
+ unverifiable.extend(source_unverifiable)
1390
+ constant_series_issues.extend(_constant_labeled_series(chart, profiles))
1391
+ unbound_secondary_axis = _unbound_secondary_axis(chart)
1392
+ if unbound_secondary_axis:
1393
+ unbound_secondary_axes.append(unbound_secondary_axis)
1394
+ undersized = _undersized_chart(chart)
1395
+ if undersized:
1396
+ undersized_charts.append(undersized)
1397
+ overwide = _overwide_chart(chart)
1398
+ if overwide:
1399
+ overwide_charts.append(overwide)
1400
+
1401
+ issue_count = (
1402
+ len(overlaps)
1403
+ + len(out_of_bounds)
1404
+ + len(content_overlaps)
1405
+ + len(numeric_source_issues)
1406
+ + len(degenerate_numeric_series)
1407
+ + len(constant_series_issues)
1408
+ + len(unbound_secondary_axes)
1409
+ + len(undersized_charts)
1410
+ + len(overwide_charts)
1411
+ )
1412
+ return {
1413
+ "sheet_id": sheet_id,
1414
+ "sheet_name": title,
1415
+ "chart_count": len(charts),
1416
+ "sheet_size_px": {"width": round(sheet_width, 2), "height": round(sheet_height, 2)},
1417
+ "chart_overlaps": overlaps,
1418
+ "cell_content_overlaps": content_overlaps,
1419
+ "numeric_source_format_issues": numeric_source_issues,
1420
+ "numeric_source_samples": numeric_source_samples,
1421
+ "degenerate_numeric_series": degenerate_numeric_series,
1422
+ "constant_labeled_series": constant_series_issues,
1423
+ "unbound_secondary_axes": unbound_secondary_axes,
1424
+ "undersized_charts": undersized_charts,
1425
+ "overwide_charts": overwide_charts,
1426
+ "out_of_visible_range": out_of_bounds,
1427
+ "unverifiable_charts": unverifiable,
1428
+ "issue_count": issue_count,
1429
+ "unverifiable_count": len(unverifiable),
1430
+ "warnings": warnings,
1431
+ }
1432
+
1433
+
1434
+ def parse_args() -> argparse.Namespace:
1435
+ parser = argparse.ArgumentParser(
1436
+ description=(
1437
+ "Check chart overlap, covered cell content, worksheet boundary overflow, "
1438
+ "minimum size, excessive width, constant labeled series, numeric source-cell "
1439
+ "formats, all-zero/empty numeric series, and unbound combo-chart secondary axes."
1440
+ )
1441
+ )
1442
+ parser.add_argument("sheet_id", help="Spreadsheet URL or spreadsheet token")
1443
+ parser.add_argument("--worksheet-id", help="Only check this worksheet reference_id")
1444
+ parser.add_argument("--timeout", type=int, default=60)
1445
+ parser.add_argument("--sample-limit", type=int, default=10)
1446
+ return parser.parse_args()
1447
+
1448
+
1449
+ def success_envelope(results: list[dict[str, Any]]) -> dict[str, Any]:
1450
+ issue_count = sum(result["issue_count"] for result in results)
1451
+ unverifiable_count = sum(result["unverifiable_count"] for result in results)
1452
+ warnings = [
1453
+ f"{result['sheet_name'] or result['sheet_id']}: {warning}"
1454
+ for result in results
1455
+ for warning in result["warnings"]
1456
+ ]
1457
+ return {
1458
+ "ok": True,
1459
+ "engine": "lark",
1460
+ "action": ACTION,
1461
+ "data": {
1462
+ "passed": issue_count == 0 and unverifiable_count == 0,
1463
+ "scope_note": (
1464
+ "out_of_visible_range checks worksheet drawable bounds, not a device-specific browser viewport; "
1465
+ "numeric source checks sample at most the first 50 data points of each chart value dimension "
1466
+ "for formats, all-zero/empty, and constant-value checks; zero/constant results are conclusive "
1467
+ "only when that sample covers the complete series, otherwise they are marked unverifiable"
1468
+ ),
1469
+ "summary": {
1470
+ "worksheet_count": len(results),
1471
+ "chart_count": sum(result["chart_count"] for result in results),
1472
+ "issue_count": issue_count,
1473
+ "unverifiable_count": unverifiable_count,
1474
+ },
1475
+ "sheets": results,
1476
+ },
1477
+ "warnings": warnings,
1478
+ }
1479
+
1480
+
1481
+ def report_exit_code(report: dict[str, Any]) -> int:
1482
+ if report["data"]["passed"]:
1483
+ return 0
1484
+ if report["data"]["summary"]["issue_count"] > 0:
1485
+ return 2
1486
+ return 1
1487
+
1488
+
1489
+ def main() -> None:
1490
+ args = parse_args()
1491
+ locator = _locator(args.sheet_id)
1492
+ cell_cache: CellCache = {}
1493
+ try:
1494
+ workbook_data = envelope_data(
1495
+ run_sheets("+workbook-info", **locator, timeout=args.timeout)
1496
+ )
1497
+ sheets = resolve_target_sheets(workbook_data, sheet_id=args.worksheet_id)
1498
+ if not args.worksheet_id:
1499
+ sheets = [sheet for sheet in sheets if not bool(sheet.get("is_hidden"))]
1500
+ if not sheets:
1501
+ raise LarkCliError("No visible worksheet matched")
1502
+ results = [
1503
+ check_sheet(
1504
+ locator,
1505
+ sheet,
1506
+ timeout=args.timeout,
1507
+ sample_limit=args.sample_limit,
1508
+ cell_cache=cell_cache,
1509
+ )
1510
+ for sheet in sheets
1511
+ ]
1512
+ except (LarkCliError, KeyError, TypeError, ValueError) as exc:
1513
+ emit_error(ACTION, str(exc))
1514
+ raise SystemExit(1) from exc
1515
+
1516
+ report = success_envelope(results)
1517
+ print(json.dumps(report, ensure_ascii=False, indent=2))
1518
+ exit_code = report_exit_code(report)
1519
+ if exit_code:
1520
+ raise SystemExit(exit_code)
1521
+
1522
+
1523
+ if __name__ == "__main__":
1524
+ main()