@amaster.ai/pi-lark 0.1.13 → 0.1.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +2 -2
- package/skills/lark-apps/references/lark-apps-local-dev.md +1 -1
- package/skills/lark-base/SKILL.md +4 -3
- package/skills/lark-base/references/lark-base-app.md +2 -2
- package/skills/lark-base/references/lark-base-dashboard-block-config.md +1 -1
- package/skills/lark-base/references/lark-base-workflow-schema.md +92 -14
- package/skills/lark-base/references/lark-base-workflow.md +99 -3
- package/skills/lark-calendar/SKILL.md +55 -22
- package/skills/lark-calendar/references/lark-calendar-list-attendees.md +33 -0
- package/skills/lark-calendar/references/lark-calendar-meeting-relation.md +99 -0
- package/skills/lark-calendar/references/lark-calendar-meeting.md +1 -1
- package/skills/lark-calendar/references/lark-calendar-recurring.md +64 -66
- package/skills/lark-calendar/references/lark-calendar-schedule-clear-time.md +7 -1
- package/skills/lark-doc/SKILL.md +1 -1
- package/skills/lark-doc/references/lark-doc-create-workflow.md +8 -10
- package/skills/lark-doc/references/lark-doc-script.md +11 -17
- package/skills/lark-drive/references/lark-drive-comment-location.md +1 -1
- package/skills/lark-drive/references/lark-drive-inspect.md +1 -1
- package/skills/lark-drive/references/lark-drive-permission-guide.md +1 -1
- package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector-resolve-verify.md +1 -1
- package/skills/lark-drive/references/lark-drive-workflow-topic-move-collector.md +1 -1
- package/skills/lark-im/SKILL.md +7 -1
- package/skills/lark-im/references/lark-im-chat-messages-list.md +6 -1
- package/skills/lark-im/references/lark-im-messages-mget.md +19 -2
- package/skills/lark-im/references/lark-im-messages-resources-download.md +3 -1
- package/skills/lark-im/references/lark-im-messages-search.md +1 -1
- package/skills/lark-im/references/lark-im-threads-messages-list.md +5 -1
- package/skills/lark-mail/SKILL.md +19 -8
- package/skills/lark-mail/references/lark-mail-draft-create.md +1 -1
- package/skills/lark-mail/references/lark-mail-draft-edit.md +1 -1
- package/skills/lark-mail/references/lark-mail-forward.md +1 -1
- package/skills/lark-mail/references/lark-mail-reply-all.md +1 -1
- package/skills/lark-mail/references/lark-mail-reply.md +1 -1
- package/skills/lark-mail/references/lark-mail-rules.md +87 -4
- package/skills/lark-mail/references/lark-mail-send.md +1 -1
- package/skills/lark-mail/references/lark-mail-thread-modify.md +73 -0
- package/skills/lark-mail/references/lark-mail-thread-trash.md +62 -0
- package/skills/lark-mail/references/lark-mail-watch.md +1 -1
- package/skills/lark-meeting/SKILL.md +2 -2
- package/skills/lark-meeting/references/lark-minutes-search.md +2 -2
- package/skills/lark-meeting/references/lark-vc-meeting-events.md +3 -2
- package/skills/lark-meeting/references/lark-vc-search.md +10 -7
- package/skills/lark-meeting/scenes/create-and-edit-minutes.md +4 -0
- package/skills/lark-meeting/scenes/query-meeting-and-artifacts.md +3 -3
- package/skills/lark-okr/SKILL.md +38 -29
- package/skills/lark-okr/references/lark-okr-comment-create.md +103 -0
- package/skills/lark-okr/references/lark-okr-comment-delete.md +59 -0
- package/skills/lark-okr/references/lark-okr-comment-detail.md +80 -0
- package/skills/lark-okr/references/lark-okr-comment-get.md +66 -0
- package/skills/lark-okr/references/lark-okr-comment-list.md +79 -0
- package/skills/lark-okr/references/lark-okr-comment-patch.md +73 -0
- package/skills/lark-okr/references/lark-okr-comment-solve-reopen.md +83 -0
- package/skills/lark-okr/references/lark-okr-entities.md +66 -2
- package/skills/lark-shared/references/lark-wiki-token-routing.md +7 -7
- package/skills/lark-sheets/SKILL.md +4 -1
- package/skills/lark-sheets/references/lark-sheets-batch-update.md +3 -3
- package/skills/lark-sheets/references/lark-sheets-chart.md +66 -32
- package/skills/lark-sheets/references/lark-sheets-legacy-command-migration.md +152 -0
- package/skills/lark-sheets/references/lark-sheets-read-data.md +2 -2
- package/skills/lark-sheets/references/lark-sheets-visual-standards.md +6 -3
- package/skills/lark-sheets/references/lark-sheets-write-cells.md +40 -17
- package/skills/lark-sheets/scripts/lark_chart_quality_check.py +1524 -0
- package/skills/lark-sheets/scripts/lark_chart_size_advisor.py +408 -0
- package/skills/lark-sheets/scripts/lark_chart_size_rules.py +292 -0
- package/skills/lark-slides/references/cli/lark-slides-add-slide.md +1 -1
- package/skills/lark-slides/references/cli/lark-slides-media-upload.md +1 -1
- package/skills/lark-slides/references/cli/lark-slides-replace-slide.md +1 -1
- package/skills/lark-slides/references/xml/slides_xml_schema_definition.xml +333 -20
- package/skills/lark-wiki/SKILL.md +1 -2
- package/skills/lark-wiki/references/lark-wiki-move.md +3 -2
- package/skills/lark-wiki/references/lark-wiki-node-create.md +3 -2
- package/skills/lark-wiki/references/lark-wiki-node-delete.md +8 -4
- package/skills/lark-wiki/references/lark-wiki-node-get.md +7 -4
- package/skills/lark-sheets/scripts/lark_chart_layout_check.py +0 -472
|
@@ -0,0 +1,1524 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# Copyright (c) 2026 Lark Technologies Pte. Ltd.
|
|
3
|
+
# SPDX-License-Identifier: MIT
|
|
4
|
+
"""Check Lark Sheet chart quality, placement, and numeric source-data issues.
|
|
5
|
+
|
|
6
|
+
The single required argument is a spreadsheet URL or spreadsheet token. By
|
|
7
|
+
default every worksheet is checked; pass --worksheet-id to restrict the check
|
|
8
|
+
to one worksheet reference_id.
|
|
9
|
+
|
|
10
|
+
Numeric source checks sample at most 50 data points per series and request at
|
|
11
|
+
most 2000 source cells per chart across the style and typed-value reads,
|
|
12
|
+
including headers and gaps between series.
|
|
13
|
+
Sampled zero/constant values do not establish a whole-series issue.
|
|
14
|
+
|
|
15
|
+
Exit codes:
|
|
16
|
+
0: check completed and no issue was found
|
|
17
|
+
1: the check could not be completed (CLI/read/response error)
|
|
18
|
+
2: check completed and at least one chart-quality issue was found
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import argparse
|
|
24
|
+
import json
|
|
25
|
+
import re
|
|
26
|
+
from typing import Any
|
|
27
|
+
|
|
28
|
+
from lark_sheet_read_cli import (
|
|
29
|
+
LarkCliError,
|
|
30
|
+
emit_error,
|
|
31
|
+
envelope_data,
|
|
32
|
+
resolve_target_sheets,
|
|
33
|
+
run_sheets,
|
|
34
|
+
sheet_identifier,
|
|
35
|
+
sheet_title,
|
|
36
|
+
)
|
|
37
|
+
from lark_chart_size_rules import MAX_ASPECT_RATIO, MAX_CHART_WIDTH, minimum_chart_size
|
|
38
|
+
|
|
39
|
+
ACTION = "chart_quality_check"
|
|
40
|
+
DEFAULT_COLUMN_WIDTH = 105.0
|
|
41
|
+
DEFAULT_ROW_HEIGHT = 27.0
|
|
42
|
+
MAX_CELL_READ_SIZE = 2_000
|
|
43
|
+
MAX_SOURCE_SAMPLE_POINTS = 50
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
CellBounds = tuple[int, int, int, int]
|
|
47
|
+
CellCache = dict[tuple[str, str, str, bool], dict[str, Any]]
|
|
48
|
+
SeriesProfile = dict[str, Any]
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _parse_a1_bounds(cell_range: str) -> CellBounds:
|
|
52
|
+
value = str(cell_range).rsplit("!", 1)[-1].replace("$", "")
|
|
53
|
+
match = re.fullmatch(r"([A-Za-z]+)(\d+)(?::([A-Za-z]+)(\d+))?", value)
|
|
54
|
+
if not match:
|
|
55
|
+
raise ValueError(f"Invalid A1 range: {cell_range!r}")
|
|
56
|
+
start_column = column_to_index(match.group(1))
|
|
57
|
+
start_row = int(match.group(2))
|
|
58
|
+
end_column = column_to_index(match.group(3) or match.group(1))
|
|
59
|
+
end_row = int(match.group(4) or match.group(2))
|
|
60
|
+
if end_row < start_row or end_column < start_column:
|
|
61
|
+
raise ValueError(f"Invalid A1 range: {cell_range!r}")
|
|
62
|
+
return start_row, end_row, start_column, end_column
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _format_a1_bounds(bounds: CellBounds) -> str:
|
|
66
|
+
start_row, end_row, start_column, end_column = bounds
|
|
67
|
+
return (
|
|
68
|
+
f"{index_to_column(start_column)}{start_row}:"
|
|
69
|
+
f"{index_to_column(end_column)}{end_row}"
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _bounds_area(bounds: CellBounds) -> int:
|
|
74
|
+
start_row, end_row, start_column, end_column = bounds
|
|
75
|
+
return (end_row - start_row + 1) * (end_column - start_column + 1)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _merge_bounds(first: CellBounds, second: CellBounds) -> CellBounds:
|
|
79
|
+
return (
|
|
80
|
+
min(first[0], second[0]),
|
|
81
|
+
max(first[1], second[1]),
|
|
82
|
+
min(first[2], second[2]),
|
|
83
|
+
max(first[3], second[3]),
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _cluster_cell_reads(
|
|
88
|
+
items: list[tuple[dict[str, Any], CellBounds]],
|
|
89
|
+
) -> list[dict[str, Any]]:
|
|
90
|
+
clusters: list[dict[str, Any]] = []
|
|
91
|
+
for rectangle, bounds in sorted(items, key=lambda item: (item[1][0], item[1][2])):
|
|
92
|
+
if clusters:
|
|
93
|
+
merged = _merge_bounds(clusters[-1]["bounds"], bounds)
|
|
94
|
+
if _bounds_area(merged) <= MAX_CELL_READ_SIZE:
|
|
95
|
+
clusters[-1]["bounds"] = merged
|
|
96
|
+
clusters[-1]["members"].append((rectangle, bounds))
|
|
97
|
+
continue
|
|
98
|
+
clusters.append({"bounds": bounds, "members": [(rectangle, bounds)]})
|
|
99
|
+
return clusters
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def column_to_index(column: str) -> int:
|
|
103
|
+
value = 0
|
|
104
|
+
text = str(column).strip().upper()
|
|
105
|
+
if not text or not text.isalpha():
|
|
106
|
+
raise ValueError(f"Invalid column: {column!r}")
|
|
107
|
+
for char in text:
|
|
108
|
+
value = value * 26 + ord(char) - ord("A") + 1
|
|
109
|
+
return value - 1
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def index_to_column(index: int) -> str:
|
|
113
|
+
if index < 0:
|
|
114
|
+
raise ValueError(f"Invalid column index: {index}")
|
|
115
|
+
chars: list[str] = []
|
|
116
|
+
value = index + 1
|
|
117
|
+
while value:
|
|
118
|
+
value, remainder = divmod(value - 1, 26)
|
|
119
|
+
chars.append(chr(ord("A") + remainder))
|
|
120
|
+
return "".join(reversed(chars))
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _span_bounds(span: str, *, columns: bool) -> tuple[int, int]:
|
|
124
|
+
start, separator, end = str(span).partition(":")
|
|
125
|
+
end = end if separator else start
|
|
126
|
+
if columns:
|
|
127
|
+
return column_to_index(start), column_to_index(end)
|
|
128
|
+
return int(start) - 1, int(end) - 1
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _size_edges(
|
|
132
|
+
groups: Any,
|
|
133
|
+
*,
|
|
134
|
+
count: int,
|
|
135
|
+
span_key: str,
|
|
136
|
+
size_key: str,
|
|
137
|
+
columns: bool,
|
|
138
|
+
default_size: float,
|
|
139
|
+
) -> tuple[list[float], bool]:
|
|
140
|
+
sizes: list[float | None] = [None] * count
|
|
141
|
+
if isinstance(groups, list):
|
|
142
|
+
for group in groups:
|
|
143
|
+
if not isinstance(group, dict) or group.get(span_key) is None:
|
|
144
|
+
continue
|
|
145
|
+
start, end = _span_bounds(str(group[span_key]), columns=columns)
|
|
146
|
+
size = float(group.get(size_key, default_size))
|
|
147
|
+
for index in range(max(0, start), min(count - 1, end) + 1):
|
|
148
|
+
sizes[index] = max(0.0, size)
|
|
149
|
+
|
|
150
|
+
used_default = any(size is None for size in sizes)
|
|
151
|
+
resolved = [default_size if size is None else size for size in sizes]
|
|
152
|
+
edges = [0.0]
|
|
153
|
+
for size in resolved:
|
|
154
|
+
edges.append(edges[-1] + size)
|
|
155
|
+
return edges, used_default
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def build_layout(
|
|
159
|
+
structure: dict[str, Any], row_count: int, column_count: int
|
|
160
|
+
) -> tuple[list[float], list[float], list[str]]:
|
|
161
|
+
row_groups = structure.get("row_heights")
|
|
162
|
+
column_groups = structure.get("col_widths", structure.get("column_widths"))
|
|
163
|
+
row_edges, row_defaulted = _size_edges(
|
|
164
|
+
row_groups,
|
|
165
|
+
count=row_count,
|
|
166
|
+
span_key="rows",
|
|
167
|
+
size_key="height",
|
|
168
|
+
columns=False,
|
|
169
|
+
default_size=DEFAULT_ROW_HEIGHT,
|
|
170
|
+
)
|
|
171
|
+
column_edges, column_defaulted = _size_edges(
|
|
172
|
+
column_groups,
|
|
173
|
+
count=column_count,
|
|
174
|
+
span_key="cols",
|
|
175
|
+
size_key="width",
|
|
176
|
+
columns=True,
|
|
177
|
+
default_size=DEFAULT_COLUMN_WIDTH,
|
|
178
|
+
)
|
|
179
|
+
warnings: list[str] = []
|
|
180
|
+
if row_defaulted:
|
|
181
|
+
warnings.append("部分行缺少高度信息,按 27 px 估算")
|
|
182
|
+
if column_defaulted:
|
|
183
|
+
warnings.append("部分列缺少宽度信息,按 105 px 估算")
|
|
184
|
+
return row_edges, column_edges, warnings
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _first_dict(value: Any) -> dict[str, Any] | None:
|
|
188
|
+
if isinstance(value, dict):
|
|
189
|
+
return value
|
|
190
|
+
if isinstance(value, list):
|
|
191
|
+
return next((item for item in value if isinstance(item, dict)), None)
|
|
192
|
+
return None
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def extract_sheet_structure(data: dict[str, Any]) -> dict[str, Any]:
|
|
196
|
+
sheet = _first_dict(data.get("sheets")) or _first_dict(data.get("sheet"))
|
|
197
|
+
return sheet or data
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def extract_charts(data: dict[str, Any], sheet_id: str, title: str) -> list[dict[str, Any]]:
|
|
201
|
+
sheets = data.get("sheets")
|
|
202
|
+
if isinstance(sheets, list):
|
|
203
|
+
for sheet in sheets:
|
|
204
|
+
if not isinstance(sheet, dict):
|
|
205
|
+
continue
|
|
206
|
+
if sheet_identifier(sheet) == sheet_id or sheet_title(sheet) == title:
|
|
207
|
+
charts = sheet.get("charts")
|
|
208
|
+
return [chart for chart in charts if isinstance(chart, dict)] if isinstance(charts, list) else []
|
|
209
|
+
charts = data.get("charts")
|
|
210
|
+
return [chart for chart in charts if isinstance(chart, dict)] if isinstance(charts, list) else []
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def chart_rectangle(
|
|
214
|
+
chart: dict[str, Any], row_edges: list[float], column_edges: list[float]
|
|
215
|
+
) -> dict[str, Any]:
|
|
216
|
+
details = chart.get("details") if isinstance(chart.get("details"), dict) else chart
|
|
217
|
+
position = details.get("position") if isinstance(details.get("position"), dict) else {}
|
|
218
|
+
offset = details.get("offset") if isinstance(details.get("offset"), dict) else {}
|
|
219
|
+
size = details.get("size") if isinstance(details.get("size"), dict) else {}
|
|
220
|
+
|
|
221
|
+
row = int(position["row"])
|
|
222
|
+
column = column_to_index(str(position["col"]))
|
|
223
|
+
if row < 0 or column < 0 or row >= len(row_edges) - 1 or column >= len(column_edges) - 1:
|
|
224
|
+
raise ValueError(f"anchor outside sheet: {position!r}")
|
|
225
|
+
|
|
226
|
+
width = float(size["width"])
|
|
227
|
+
height = float(size["height"])
|
|
228
|
+
if width <= 0 or height <= 0:
|
|
229
|
+
raise ValueError(f"invalid chart size: {size!r}")
|
|
230
|
+
|
|
231
|
+
left = column_edges[column] + float(offset.get("col_offset", 0) or 0)
|
|
232
|
+
top = row_edges[row] + float(offset.get("row_offset", 0) or 0)
|
|
233
|
+
return {
|
|
234
|
+
"chart_id": str(chart.get("chart_id") or chart.get("id") or ""),
|
|
235
|
+
"anchor_cell": f"{index_to_column(column)}{row + 1}",
|
|
236
|
+
"left": left,
|
|
237
|
+
"top": top,
|
|
238
|
+
"right": left + width,
|
|
239
|
+
"bottom": top + height,
|
|
240
|
+
"width": width,
|
|
241
|
+
"height": height,
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def intersection(first: dict[str, Any], second: dict[str, Any]) -> dict[str, float] | None:
|
|
246
|
+
left = max(float(first["left"]), float(second["left"]))
|
|
247
|
+
top = max(float(first["top"]), float(second["top"]))
|
|
248
|
+
right = min(float(first["right"]), float(second["right"]))
|
|
249
|
+
bottom = min(float(first["bottom"]), float(second["bottom"]))
|
|
250
|
+
if right <= left or bottom <= top:
|
|
251
|
+
return None
|
|
252
|
+
return {
|
|
253
|
+
"width": round(right - left, 2),
|
|
254
|
+
"height": round(bottom - top, 2),
|
|
255
|
+
"area": round((right - left) * (bottom - top), 2),
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def chart_context(rectangle: dict[str, Any]) -> dict[str, Any]:
|
|
260
|
+
return {
|
|
261
|
+
"chart_id": rectangle["chart_id"],
|
|
262
|
+
"anchor_cell": rectangle["anchor_cell"],
|
|
263
|
+
"rectangle_px": {
|
|
264
|
+
"left": round(rectangle["left"], 2),
|
|
265
|
+
"top": round(rectangle["top"], 2),
|
|
266
|
+
"right": round(rectangle["right"], 2),
|
|
267
|
+
"bottom": round(rectangle["bottom"], 2),
|
|
268
|
+
"width": round(rectangle["width"], 2),
|
|
269
|
+
"height": round(rectangle["height"], 2),
|
|
270
|
+
},
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def _covered_indexes(edges: list[float], start: float, end: float) -> list[int]:
|
|
275
|
+
return [
|
|
276
|
+
index
|
|
277
|
+
for index in range(len(edges) - 1)
|
|
278
|
+
if edges[index + 1] > start and edges[index] < end
|
|
279
|
+
]
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def rectangle_cell_range(
|
|
283
|
+
rectangle: dict[str, Any], row_edges: list[float], column_edges: list[float]
|
|
284
|
+
) -> str | None:
|
|
285
|
+
rows = _covered_indexes(row_edges, max(0.0, rectangle["top"]), rectangle["bottom"])
|
|
286
|
+
columns = _covered_indexes(column_edges, max(0.0, rectangle["left"]), rectangle["right"])
|
|
287
|
+
if not rows or not columns:
|
|
288
|
+
return None
|
|
289
|
+
return f"{index_to_column(columns[0])}{rows[0] + 1}:{index_to_column(columns[-1])}{rows[-1] + 1}"
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def _has_content(cell: Any) -> bool:
|
|
293
|
+
if not isinstance(cell, dict):
|
|
294
|
+
return False
|
|
295
|
+
for key in ("value", "formula", "note"):
|
|
296
|
+
value = cell.get(key)
|
|
297
|
+
if value not in (None, ""):
|
|
298
|
+
return True
|
|
299
|
+
return bool(cell.get("rich_text") or cell.get("multiple_values"))
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def _iter_cells(data: dict[str, Any]):
|
|
303
|
+
ranges = data.get("ranges")
|
|
304
|
+
if not isinstance(ranges, list):
|
|
305
|
+
return
|
|
306
|
+
for result_range in ranges:
|
|
307
|
+
if not isinstance(result_range, dict):
|
|
308
|
+
continue
|
|
309
|
+
cells = result_range.get("cells")
|
|
310
|
+
rows = result_range.get("row_indices")
|
|
311
|
+
columns = result_range.get("col_indices")
|
|
312
|
+
if not isinstance(cells, list):
|
|
313
|
+
continue
|
|
314
|
+
for row_offset, row in enumerate(cells):
|
|
315
|
+
if not isinstance(row, list):
|
|
316
|
+
continue
|
|
317
|
+
row_number = int(rows[row_offset]) if isinstance(rows, list) and row_offset < len(rows) else row_offset + 1
|
|
318
|
+
for column_offset, cell in enumerate(row):
|
|
319
|
+
column = columns[column_offset] if isinstance(columns, list) and column_offset < len(columns) else index_to_column(column_offset)
|
|
320
|
+
yield row_number, column_to_index(str(column)), cell
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def non_empty_cells(
|
|
324
|
+
data: dict[str, Any], sample_limit: int, bounds: CellBounds | None = None
|
|
325
|
+
) -> tuple[int, list[str], bool]:
|
|
326
|
+
count = 0
|
|
327
|
+
samples: list[str] = []
|
|
328
|
+
truncated = bool(data.get("has_more"))
|
|
329
|
+
ranges = data.get("ranges")
|
|
330
|
+
if not isinstance(ranges, list):
|
|
331
|
+
return 0, [], truncated
|
|
332
|
+
for result_range in ranges:
|
|
333
|
+
if not isinstance(result_range, dict):
|
|
334
|
+
continue
|
|
335
|
+
truncated = truncated or bool(result_range.get("truncated"))
|
|
336
|
+
for row_number, column_index, cell in _iter_cells(data):
|
|
337
|
+
if bounds and not (
|
|
338
|
+
bounds[0] <= row_number <= bounds[1]
|
|
339
|
+
and bounds[2] <= column_index <= bounds[3]
|
|
340
|
+
):
|
|
341
|
+
continue
|
|
342
|
+
if not _has_content(cell):
|
|
343
|
+
continue
|
|
344
|
+
count += 1
|
|
345
|
+
if len(samples) < sample_limit:
|
|
346
|
+
samples.append(f"{index_to_column(column_index)}{row_number}")
|
|
347
|
+
return count, samples, truncated
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def _read_cells(
|
|
351
|
+
cache: CellCache,
|
|
352
|
+
locator: dict[str, str],
|
|
353
|
+
*,
|
|
354
|
+
sheet_id: str | None,
|
|
355
|
+
sheet_name: str | None,
|
|
356
|
+
cell_range: str,
|
|
357
|
+
include: str,
|
|
358
|
+
skip_hidden: bool = False,
|
|
359
|
+
timeout: int,
|
|
360
|
+
) -> dict[str, Any]:
|
|
361
|
+
selector = f"id:{sheet_id}" if sheet_id else f"name:{sheet_name}"
|
|
362
|
+
key = (selector, cell_range, include, skip_hidden)
|
|
363
|
+
if key not in cache:
|
|
364
|
+
flags: dict[str, Any] = {"range": cell_range, "include": include}
|
|
365
|
+
if skip_hidden:
|
|
366
|
+
flags["skip-hidden"] = True
|
|
367
|
+
cache[key] = envelope_data(
|
|
368
|
+
run_sheets(
|
|
369
|
+
"+cells-get",
|
|
370
|
+
**locator,
|
|
371
|
+
**({"sheet_id": sheet_id} if sheet_id else {"sheet_name": sheet_name}),
|
|
372
|
+
flags=flags,
|
|
373
|
+
timeout=timeout,
|
|
374
|
+
)
|
|
375
|
+
)
|
|
376
|
+
return cache[key]
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def _read_typed_table(
|
|
380
|
+
cache: CellCache,
|
|
381
|
+
locator: dict[str, str],
|
|
382
|
+
*,
|
|
383
|
+
sheet_id: str | None,
|
|
384
|
+
sheet_name: str | None,
|
|
385
|
+
cell_range: str,
|
|
386
|
+
timeout: int,
|
|
387
|
+
) -> tuple[list[list[Any]], list[str], bool]:
|
|
388
|
+
selector = f"id:{sheet_id}" if sheet_id else f"name:{sheet_name}"
|
|
389
|
+
key = (selector, cell_range, "typed_table", False)
|
|
390
|
+
if key not in cache:
|
|
391
|
+
cache[key] = envelope_data(
|
|
392
|
+
run_sheets(
|
|
393
|
+
"+table-get",
|
|
394
|
+
**locator,
|
|
395
|
+
**({"sheet_id": sheet_id} if sheet_id else {"sheet_name": sheet_name}),
|
|
396
|
+
flags={"range": cell_range, "no-header": True},
|
|
397
|
+
timeout=timeout,
|
|
398
|
+
)
|
|
399
|
+
)
|
|
400
|
+
data = cache[key]
|
|
401
|
+
sheets = data.get("sheets")
|
|
402
|
+
if not isinstance(sheets, list) or len(sheets) != 1 or not isinstance(sheets[0], dict):
|
|
403
|
+
return [], [], True
|
|
404
|
+
table = sheets[0]
|
|
405
|
+
rows = table.get("data")
|
|
406
|
+
columns = table.get("columns")
|
|
407
|
+
dtypes = table.get("dtypes")
|
|
408
|
+
if not isinstance(rows, list) or not isinstance(columns, list) or not isinstance(dtypes, dict):
|
|
409
|
+
return [], [], True
|
|
410
|
+
def value_kind(dtype: Any) -> str:
|
|
411
|
+
value = str(dtype or "").lower()
|
|
412
|
+
if value in {"number", "int64", "float64"}:
|
|
413
|
+
return "number"
|
|
414
|
+
if value in {"bool", "boolean"}:
|
|
415
|
+
return "bool"
|
|
416
|
+
if value.startswith("datetime"):
|
|
417
|
+
return "date"
|
|
418
|
+
return "string"
|
|
419
|
+
|
|
420
|
+
types = [value_kind(dtypes.get(str(column))) for column in columns]
|
|
421
|
+
truncated = bool(data.get("truncated")) or bool(table.get("truncated"))
|
|
422
|
+
return rows, types, truncated
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
def _typed_cell(
|
|
426
|
+
rows: list[list[Any]],
|
|
427
|
+
column_types: list[str],
|
|
428
|
+
row_offset: int,
|
|
429
|
+
column_offset: int,
|
|
430
|
+
visible_offsets: set[tuple[int, int]],
|
|
431
|
+
) -> tuple[Any, str]:
|
|
432
|
+
if (
|
|
433
|
+
row_offset < 0
|
|
434
|
+
or row_offset >= len(rows)
|
|
435
|
+
or not isinstance(rows[row_offset], list)
|
|
436
|
+
or column_offset < 0
|
|
437
|
+
or column_offset >= len(rows[row_offset])
|
|
438
|
+
):
|
|
439
|
+
return None, ""
|
|
440
|
+
value = rows[row_offset][column_offset]
|
|
441
|
+
if isinstance(value, bool):
|
|
442
|
+
return value, "bool"
|
|
443
|
+
if isinstance(value, (int, float)):
|
|
444
|
+
return value, "number"
|
|
445
|
+
kind = column_types[column_offset] if column_offset < len(column_types) else ""
|
|
446
|
+
if kind != "string" or not isinstance(value, str) or not _looks_numeric(value):
|
|
447
|
+
return value, kind
|
|
448
|
+
|
|
449
|
+
# table-get infers one dtype per physical column. A hidden cell or another
|
|
450
|
+
# row-series can widen that dtype and stringify otherwise numeric cells.
|
|
451
|
+
# Treat such cells as unknown instead of reporting a false storage issue.
|
|
452
|
+
numeric_string_count = 0
|
|
453
|
+
for other_row_offset, row in enumerate(rows):
|
|
454
|
+
if not isinstance(row, list) or column_offset >= len(row):
|
|
455
|
+
continue
|
|
456
|
+
other = row[column_offset]
|
|
457
|
+
if other not in (None, "") and (other_row_offset, column_offset) not in visible_offsets:
|
|
458
|
+
return value, "unknown"
|
|
459
|
+
if other in (None, "") or (
|
|
460
|
+
isinstance(other, (int, float)) and not isinstance(other, bool)
|
|
461
|
+
):
|
|
462
|
+
continue
|
|
463
|
+
if isinstance(other, str) and _looks_numeric(other):
|
|
464
|
+
numeric_string_count += 1
|
|
465
|
+
continue
|
|
466
|
+
return value, "unknown"
|
|
467
|
+
return value, "string" if numeric_string_count == 1 else "ambiguous_string"
|
|
468
|
+
|
|
469
|
+
|
|
470
|
+
def _chart_snapshot(chart: dict[str, Any]) -> dict[str, Any]:
|
|
471
|
+
details = chart.get("details") if isinstance(chart.get("details"), dict) else chart
|
|
472
|
+
snapshot = details.get("snapshot")
|
|
473
|
+
return snapshot if isinstance(snapshot, dict) else {}
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
def _chart_type(snapshot: dict[str, Any]) -> str:
|
|
477
|
+
plot_area = snapshot.get("plotArea")
|
|
478
|
+
plot = plot_area.get("plot") if isinstance(plot_area, dict) else None
|
|
479
|
+
return str(plot.get("type") or "").lower() if isinstance(plot, dict) else ""
|
|
480
|
+
|
|
481
|
+
|
|
482
|
+
def _plot(snapshot: dict[str, Any]) -> dict[str, Any]:
|
|
483
|
+
plot_area = snapshot.get("plotArea")
|
|
484
|
+
plot = plot_area.get("plot") if isinstance(plot_area, dict) else None
|
|
485
|
+
return plot if isinstance(plot, dict) else {}
|
|
486
|
+
|
|
487
|
+
|
|
488
|
+
def _static_series_profiles(snapshot: dict[str, Any]) -> list[SeriesProfile]:
|
|
489
|
+
data = snapshot.get("data")
|
|
490
|
+
dim2 = data.get("dim2") if isinstance(data, dict) else None
|
|
491
|
+
fields = dim2.get("fields") if isinstance(dim2, dict) else None
|
|
492
|
+
if not isinstance(fields, list):
|
|
493
|
+
return []
|
|
494
|
+
profiles: list[SeriesProfile] = []
|
|
495
|
+
for offset, field in enumerate(fields, start=1):
|
|
496
|
+
if not isinstance(field, dict):
|
|
497
|
+
continue
|
|
498
|
+
values = []
|
|
499
|
+
for value in field.get("parsedValues") or []:
|
|
500
|
+
numeric = (
|
|
501
|
+
float(value)
|
|
502
|
+
if isinstance(value, (int, float)) and not isinstance(value, bool)
|
|
503
|
+
else _numeric_text_value(value) if isinstance(value, str) else None
|
|
504
|
+
)
|
|
505
|
+
if numeric is not None:
|
|
506
|
+
values.append(numeric)
|
|
507
|
+
profiles.append(
|
|
508
|
+
{
|
|
509
|
+
"dimension_index": offset,
|
|
510
|
+
"series_name": str(field.get("name") or f"Series {offset}"),
|
|
511
|
+
"point_count": len(field.get("parsedValues") or []),
|
|
512
|
+
"numeric_value_count": len(values),
|
|
513
|
+
"unique_numeric_values": list(dict.fromkeys(values))[:2],
|
|
514
|
+
"source_sheet": "",
|
|
515
|
+
"source_range": "",
|
|
516
|
+
"series_range": "",
|
|
517
|
+
}
|
|
518
|
+
)
|
|
519
|
+
return profiles
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
def _labeled_series_indexes(
|
|
523
|
+
snapshot: dict[str, Any], profiles: list[SeriesProfile]
|
|
524
|
+
) -> set[int]:
|
|
525
|
+
plot = _plot(snapshot)
|
|
526
|
+
available = {
|
|
527
|
+
int(profile["dimension_index"])
|
|
528
|
+
for profile in profiles
|
|
529
|
+
if profile.get("dimension_index") is not None
|
|
530
|
+
}
|
|
531
|
+
labeled = set(available) if isinstance(plot.get("labels"), dict) else set()
|
|
532
|
+
series = plot.get("series")
|
|
533
|
+
if isinstance(series, list):
|
|
534
|
+
labeled.update(
|
|
535
|
+
int(item["index"])
|
|
536
|
+
for item in series
|
|
537
|
+
if isinstance(item, dict)
|
|
538
|
+
and item.get("index") is not None
|
|
539
|
+
and isinstance(item.get("labels"), dict)
|
|
540
|
+
)
|
|
541
|
+
return labeled & available
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
def _constant_labeled_series(
|
|
545
|
+
chart: dict[str, Any], profiles: list[SeriesProfile]
|
|
546
|
+
) -> list[dict[str, Any]]:
|
|
547
|
+
chart_id = str(chart.get("chart_id") or chart.get("id") or "")
|
|
548
|
+
snapshot = _chart_snapshot(chart)
|
|
549
|
+
labeled = _labeled_series_indexes(snapshot, profiles)
|
|
550
|
+
return [
|
|
551
|
+
{
|
|
552
|
+
"chart_id": chart_id,
|
|
553
|
+
"dimension_index": profile["dimension_index"],
|
|
554
|
+
"series_name": profile["series_name"],
|
|
555
|
+
"source_sheet": profile["source_sheet"],
|
|
556
|
+
"source_range": profile["source_range"],
|
|
557
|
+
"series_range": profile["series_range"],
|
|
558
|
+
"reason": "constant_labeled_series",
|
|
559
|
+
"data_point_count": profile["point_count"],
|
|
560
|
+
"constant_value": profile["unique_numeric_values"][0],
|
|
561
|
+
"suggested_fix": "remove_series_labels_or_use_one_sparse_marker",
|
|
562
|
+
}
|
|
563
|
+
for profile in profiles
|
|
564
|
+
if int(profile.get("dimension_index", -1)) in labeled
|
|
565
|
+
and profile.get("constant_check_unverifiable") is not True
|
|
566
|
+
and int(profile.get("numeric_value_count", 0)) >= 2
|
|
567
|
+
and len(profile.get("unique_numeric_values") or []) == 1
|
|
568
|
+
]
|
|
569
|
+
|
|
570
|
+
|
|
571
|
+
def _unbound_secondary_axis(chart: dict[str, Any]) -> dict[str, Any] | None:
|
|
572
|
+
snapshot = _chart_snapshot(chart)
|
|
573
|
+
plot_area = snapshot.get("plotArea")
|
|
574
|
+
if not isinstance(plot_area, dict):
|
|
575
|
+
return None
|
|
576
|
+
plot = plot_area.get("plot")
|
|
577
|
+
if not isinstance(plot, dict) or str(plot.get("type") or "").lower() != "combo":
|
|
578
|
+
return None
|
|
579
|
+
|
|
580
|
+
axes = plot_area.get("axes")
|
|
581
|
+
has_right_axis = isinstance(axes, list) and any(
|
|
582
|
+
isinstance(axis, dict)
|
|
583
|
+
and str(axis.get("type") or "").lower() == "y"
|
|
584
|
+
and str(axis.get("position") or axis.get("axisPosition") or "").lower() == "right"
|
|
585
|
+
for axis in axes
|
|
586
|
+
)
|
|
587
|
+
series = plot.get("series")
|
|
588
|
+
configured_series = (
|
|
589
|
+
[item for item in series if isinstance(item, dict)]
|
|
590
|
+
if isinstance(series, list)
|
|
591
|
+
else []
|
|
592
|
+
)
|
|
593
|
+
if not has_right_axis or not configured_series:
|
|
594
|
+
return None
|
|
595
|
+
if str(plot.get("yAxisPosition") or "").lower() == "right" or any(
|
|
596
|
+
str(item.get("yAxisPosition") or "").lower() == "right"
|
|
597
|
+
for item in configured_series
|
|
598
|
+
):
|
|
599
|
+
return None
|
|
600
|
+
|
|
601
|
+
return {
|
|
602
|
+
"chart_id": str(chart.get("chart_id") or chart.get("id") or ""),
|
|
603
|
+
"reason": "secondary_axis_has_no_bound_series",
|
|
604
|
+
"series_indexes": [
|
|
605
|
+
int(item["index"])
|
|
606
|
+
for item in configured_series
|
|
607
|
+
if isinstance(item.get("index"), (int, float))
|
|
608
|
+
],
|
|
609
|
+
"suggested_fix": "bind_the_intended_combo_series_to_the_right_axis",
|
|
610
|
+
}
|
|
611
|
+
|
|
612
|
+
|
|
613
|
+
def _undersized_chart(chart: dict[str, Any]) -> dict[str, Any] | None:
|
|
614
|
+
details = chart.get("details") if isinstance(chart.get("details"), dict) else chart
|
|
615
|
+
snapshot = _chart_snapshot(chart)
|
|
616
|
+
chart_type = _chart_type(snapshot)
|
|
617
|
+
size = details.get("size")
|
|
618
|
+
if not isinstance(size, dict):
|
|
619
|
+
return None
|
|
620
|
+
width = size.get("width")
|
|
621
|
+
height = size.get("height")
|
|
622
|
+
if (
|
|
623
|
+
not isinstance(width, (int, float))
|
|
624
|
+
or isinstance(width, bool)
|
|
625
|
+
or width <= 0
|
|
626
|
+
or not isinstance(height, (int, float))
|
|
627
|
+
or isinstance(height, bool)
|
|
628
|
+
or height <= 0
|
|
629
|
+
):
|
|
630
|
+
return None
|
|
631
|
+
actual = {
|
|
632
|
+
"width": float(width),
|
|
633
|
+
"height": float(height),
|
|
634
|
+
}
|
|
635
|
+
minimum = minimum_chart_size(chart_type)
|
|
636
|
+
if actual["width"] >= minimum["width"] and actual["height"] >= minimum["height"]:
|
|
637
|
+
return None
|
|
638
|
+
return {
|
|
639
|
+
"chart_id": str(chart.get("chart_id") or chart.get("id") or ""),
|
|
640
|
+
"reason": "chart_below_minimum_size",
|
|
641
|
+
"chart_type": chart_type,
|
|
642
|
+
"actual_size": actual,
|
|
643
|
+
"minimum_size": minimum,
|
|
644
|
+
"suggested_fix": "run_lark_chart_size_advisor_before_resizing",
|
|
645
|
+
}
|
|
646
|
+
|
|
647
|
+
|
|
648
|
+
def _overwide_chart(chart: dict[str, Any]) -> dict[str, Any] | None:
|
|
649
|
+
details = chart.get("details") if isinstance(chart.get("details"), dict) else chart
|
|
650
|
+
snapshot = _chart_snapshot(chart)
|
|
651
|
+
chart_type = _chart_type(snapshot)
|
|
652
|
+
size = details.get("size") if isinstance(details.get("size"), dict) else {}
|
|
653
|
+
width = size.get("width")
|
|
654
|
+
height = size.get("height")
|
|
655
|
+
if (
|
|
656
|
+
not isinstance(width, (int, float))
|
|
657
|
+
or isinstance(width, bool)
|
|
658
|
+
or width <= 0
|
|
659
|
+
or not isinstance(height, (int, float))
|
|
660
|
+
or isinstance(height, bool)
|
|
661
|
+
or height <= 0
|
|
662
|
+
):
|
|
663
|
+
return None
|
|
664
|
+
width = float(width)
|
|
665
|
+
height = float(height)
|
|
666
|
+
minimum = minimum_chart_size(chart_type)
|
|
667
|
+
aspect_ratio = width / height if height > 0 else 0
|
|
668
|
+
width_exceeded = width > MAX_CHART_WIDTH
|
|
669
|
+
aspect_ratio_exceeded = (
|
|
670
|
+
width >= minimum["width"]
|
|
671
|
+
and height >= minimum["height"]
|
|
672
|
+
and aspect_ratio > MAX_ASPECT_RATIO
|
|
673
|
+
)
|
|
674
|
+
if not width_exceeded and not aspect_ratio_exceeded:
|
|
675
|
+
return None
|
|
676
|
+
return {
|
|
677
|
+
"chart_id": str(chart.get("chart_id") or chart.get("id") or ""),
|
|
678
|
+
"reason": "chart_too_wide",
|
|
679
|
+
"chart_type": chart_type,
|
|
680
|
+
"actual_size": {"width": width, "height": height},
|
|
681
|
+
"actual_aspect_ratio": round(aspect_ratio, 2),
|
|
682
|
+
"maximum_width": MAX_CHART_WIDTH,
|
|
683
|
+
"maximum_aspect_ratio": MAX_ASPECT_RATIO,
|
|
684
|
+
"suggested_fix": "use_size_advisor_create_flags_or_change_chart_structure",
|
|
685
|
+
}
|
|
686
|
+
|
|
687
|
+
|
|
688
|
+
def _numeric_dimensions(snapshot: dict[str, Any]) -> list[tuple[int, str]]:
|
|
689
|
+
data = snapshot.get("data")
|
|
690
|
+
if not isinstance(data, dict):
|
|
691
|
+
return []
|
|
692
|
+
dim2 = data.get("dim2")
|
|
693
|
+
series = dim2.get("series") if isinstance(dim2, dict) else None
|
|
694
|
+
chart_type = _chart_type(snapshot)
|
|
695
|
+
dimensions: list[tuple[int, str]] = []
|
|
696
|
+
if isinstance(series, list):
|
|
697
|
+
for offset, serie in enumerate(series):
|
|
698
|
+
if not isinstance(serie, dict) or serie.get("index") is None:
|
|
699
|
+
continue
|
|
700
|
+
if str(serie.get("aggregateType") or "").lower() == "counta":
|
|
701
|
+
continue
|
|
702
|
+
role = str(serie.get("role") or "").lower()
|
|
703
|
+
if chart_type == "bubble":
|
|
704
|
+
role = role or ("x", "y", "group", "size")[min(offset, 3)]
|
|
705
|
+
if role not in {"x", "y", "size"}:
|
|
706
|
+
continue
|
|
707
|
+
dimensions.append((int(serie["index"]), role or "value"))
|
|
708
|
+
|
|
709
|
+
plot_area = snapshot.get("plotArea")
|
|
710
|
+
axes = plot_area.get("axes") if isinstance(plot_area, dict) else None
|
|
711
|
+
continuous_x = chart_type == "scatter"
|
|
712
|
+
if isinstance(axes, list):
|
|
713
|
+
for axis in axes:
|
|
714
|
+
if not isinstance(axis, dict) or str(axis.get("type") or "").lower() != "x":
|
|
715
|
+
continue
|
|
716
|
+
position = axis.get("position")
|
|
717
|
+
if (
|
|
718
|
+
(position is None or str(position).lower() in {"bottom", "x"})
|
|
719
|
+
and str(axis.get("valueType") or "").lower() == "linear"
|
|
720
|
+
):
|
|
721
|
+
continuous_x = True
|
|
722
|
+
break
|
|
723
|
+
dim1 = data.get("dim1")
|
|
724
|
+
serie = dim1.get("serie") if isinstance(dim1, dict) else None
|
|
725
|
+
if continuous_x and chart_type != "bubble" and isinstance(serie, dict) and serie.get("index") is not None:
|
|
726
|
+
dimensions.append((int(serie["index"]), "x"))
|
|
727
|
+
|
|
728
|
+
return list(dict.fromkeys(dimensions))
|
|
729
|
+
|
|
730
|
+
|
|
731
|
+
def _series_aggregate_type(data: dict[str, Any], dimension_index: int) -> str:
|
|
732
|
+
dim2 = data.get("dim2")
|
|
733
|
+
value_series = dim2.get("series") if isinstance(dim2, dict) else None
|
|
734
|
+
source_series = next(
|
|
735
|
+
(
|
|
736
|
+
item
|
|
737
|
+
for item in (value_series if isinstance(value_series, list) else [])
|
|
738
|
+
if isinstance(item, dict)
|
|
739
|
+
and item.get("index") is not None
|
|
740
|
+
and int(item["index"]) == dimension_index
|
|
741
|
+
),
|
|
742
|
+
{},
|
|
743
|
+
)
|
|
744
|
+
return str(source_series.get("aggregateType") or "sum").lower()
|
|
745
|
+
|
|
746
|
+
|
|
747
|
+
def _aggregation_can_change_constant(data: dict[str, Any], dimension_index: int) -> bool:
|
|
748
|
+
dim1 = data.get("dim1")
|
|
749
|
+
category_series = dim1.get("serie") if isinstance(dim1, dict) else None
|
|
750
|
+
if not isinstance(category_series, dict) or category_series.get("aggregate") is False:
|
|
751
|
+
return False
|
|
752
|
+
return _series_aggregate_type(data, dimension_index) in {"sum", "count", "counta"}
|
|
753
|
+
|
|
754
|
+
|
|
755
|
+
def _parse_chart_ref(value: str, default_sheet: str) -> tuple[str, str, CellBounds]:
|
|
756
|
+
raw = str(value).strip()
|
|
757
|
+
sheet_name = default_sheet
|
|
758
|
+
cell_range = raw
|
|
759
|
+
if "!" in raw:
|
|
760
|
+
sheet_name, cell_range = raw.rsplit("!", 1)
|
|
761
|
+
sheet_name = sheet_name.strip()
|
|
762
|
+
if len(sheet_name) >= 2 and sheet_name[0] == sheet_name[-1] == "'":
|
|
763
|
+
sheet_name = sheet_name[1:-1].replace("''", "'")
|
|
764
|
+
return sheet_name, cell_range.replace("$", ""), _parse_a1_bounds(cell_range)
|
|
765
|
+
|
|
766
|
+
|
|
767
|
+
def _looks_numeric(value: str) -> bool:
|
|
768
|
+
return _numeric_text_value(value) is not None
|
|
769
|
+
|
|
770
|
+
|
|
771
|
+
def _numeric_text_value(value: str) -> float | None:
|
|
772
|
+
text = value.strip()
|
|
773
|
+
if not text:
|
|
774
|
+
return None
|
|
775
|
+
text = re.sub(r"^([+-]?)[\$\u00a5\uffe5\u20ac\u00a3]", r"\1", text)
|
|
776
|
+
if text.endswith("%"):
|
|
777
|
+
text = text[:-1]
|
|
778
|
+
# Group separators must be consistent and separate groups of three digits.
|
|
779
|
+
if not re.fullmatch(
|
|
780
|
+
r"[+-]?(?:(?:\d+|\d{1,3}([, ])\d{3}(?:\1\d{3})*)(?:\.\d*)?|\.\d+)"
|
|
781
|
+
r"(?:[eE][+-]?\d+)?",
|
|
782
|
+
text,
|
|
783
|
+
):
|
|
784
|
+
return None
|
|
785
|
+
return float(text.replace(" ", "").replace(",", ""))
|
|
786
|
+
|
|
787
|
+
|
|
788
|
+
def _zero_state(value: Any) -> str:
|
|
789
|
+
if value in (None, ""):
|
|
790
|
+
return "empty"
|
|
791
|
+
if isinstance(value, (int, float)) and not isinstance(value, bool):
|
|
792
|
+
return "zero" if value == 0 else "nonzero"
|
|
793
|
+
if isinstance(value, str):
|
|
794
|
+
numeric_value = _numeric_text_value(value)
|
|
795
|
+
if numeric_value is not None:
|
|
796
|
+
return "zero" if numeric_value == 0 else "nonzero"
|
|
797
|
+
return "nonzero"
|
|
798
|
+
|
|
799
|
+
|
|
800
|
+
def _update_series_state(state: dict[str, Any], value: Any) -> None:
|
|
801
|
+
state[_zero_state(value)] = True
|
|
802
|
+
numeric = (
|
|
803
|
+
float(value)
|
|
804
|
+
if isinstance(value, (int, float)) and not isinstance(value, bool)
|
|
805
|
+
else _numeric_text_value(value) if isinstance(value, str) else None
|
|
806
|
+
)
|
|
807
|
+
if numeric is None:
|
|
808
|
+
return
|
|
809
|
+
state["numeric_value_count"] += 1
|
|
810
|
+
if len(state["unique_numeric_values"]) < 2:
|
|
811
|
+
state["unique_numeric_values"].add(numeric)
|
|
812
|
+
|
|
813
|
+
|
|
814
|
+
def _cells_truncated(data: dict[str, Any]) -> bool:
|
|
815
|
+
ranges = data.get("ranges")
|
|
816
|
+
return bool(data.get("has_more")) or any(
|
|
817
|
+
isinstance(item, dict) and item.get("truncated")
|
|
818
|
+
for item in (ranges if isinstance(ranges, list) else [])
|
|
819
|
+
)
|
|
820
|
+
|
|
821
|
+
|
|
822
|
+
def _numeric_source_issues(
|
|
823
|
+
chart: dict[str, Any],
|
|
824
|
+
*,
|
|
825
|
+
owner_sheet_id: str,
|
|
826
|
+
owner_sheet_name: str,
|
|
827
|
+
cache: CellCache,
|
|
828
|
+
locator: dict[str, str],
|
|
829
|
+
timeout: int,
|
|
830
|
+
sample_limit: int,
|
|
831
|
+
) -> tuple[
|
|
832
|
+
list[dict[str, Any]],
|
|
833
|
+
list[dict[str, Any]],
|
|
834
|
+
list[dict[str, str]],
|
|
835
|
+
list[SeriesProfile],
|
|
836
|
+
]:
|
|
837
|
+
chart_id = str(chart.get("chart_id") or chart.get("id") or "")
|
|
838
|
+
snapshot = _chart_snapshot(chart)
|
|
839
|
+
data = snapshot.get("data")
|
|
840
|
+
if not isinstance(data, dict):
|
|
841
|
+
return [], [], [{"chart_id": chart_id, "reason": "chart snapshot.data is missing"}], []
|
|
842
|
+
if data.get("isStaticData") is True:
|
|
843
|
+
return [], [], [], _static_series_profiles(snapshot)
|
|
844
|
+
dimensions = _numeric_dimensions(snapshot)
|
|
845
|
+
if not dimensions:
|
|
846
|
+
return [], [], [], []
|
|
847
|
+
refs = data.get("refs")
|
|
848
|
+
if not isinstance(refs, list) or not refs:
|
|
849
|
+
return [], [], [{"chart_id": chart_id, "reason": "chart data.refs is missing"}], []
|
|
850
|
+
|
|
851
|
+
parsed_refs: list[tuple[str, str, CellBounds]] = []
|
|
852
|
+
unverifiable: list[dict[str, str]] = []
|
|
853
|
+
for ref in refs:
|
|
854
|
+
raw_ref = ref.get("value") if isinstance(ref, dict) else ref
|
|
855
|
+
try:
|
|
856
|
+
parsed_refs.append(_parse_chart_ref(str(raw_ref), owner_sheet_name))
|
|
857
|
+
except (TypeError, ValueError) as exc:
|
|
858
|
+
unverifiable.append({"chart_id": chart_id, "reason": str(exc)})
|
|
859
|
+
return [], [], unverifiable, []
|
|
860
|
+
|
|
861
|
+
direction = str(data.get("direction") or "column").lower()
|
|
862
|
+
skip_hidden = data.get("includeHiddenOrFilter") is not True
|
|
863
|
+
mapped: list[tuple[int, str, int, str, str, CellBounds]] = []
|
|
864
|
+
for dimension_index, role in dimensions:
|
|
865
|
+
offset = 0
|
|
866
|
+
for source_sheet, source_range, bounds in parsed_refs:
|
|
867
|
+
dimension_count = (
|
|
868
|
+
bounds[3] - bounds[2] + 1 if direction == "column" else bounds[1] - bounds[0] + 1
|
|
869
|
+
)
|
|
870
|
+
if offset < dimension_index <= offset + dimension_count:
|
|
871
|
+
mapped.append(
|
|
872
|
+
(
|
|
873
|
+
dimension_index,
|
|
874
|
+
role,
|
|
875
|
+
dimension_index - offset,
|
|
876
|
+
source_sheet,
|
|
877
|
+
source_range,
|
|
878
|
+
bounds,
|
|
879
|
+
)
|
|
880
|
+
)
|
|
881
|
+
break
|
|
882
|
+
offset += dimension_count
|
|
883
|
+
else:
|
|
884
|
+
unverifiable.append(
|
|
885
|
+
{
|
|
886
|
+
"chart_id": chart_id,
|
|
887
|
+
"reason": f"numeric dimension index {dimension_index} is outside data.refs",
|
|
888
|
+
}
|
|
889
|
+
)
|
|
890
|
+
|
|
891
|
+
detached = str(data.get("headerMode") or "").lower() == "detached"
|
|
892
|
+
mapped_by_ref: dict[
|
|
893
|
+
tuple[str, str, CellBounds], list[tuple[int, str, int]]
|
|
894
|
+
] = {}
|
|
895
|
+
for dimension_index, role, local_index, source_sheet, source_range, bounds in mapped:
|
|
896
|
+
mapped_by_ref.setdefault((source_sheet, source_range, bounds), []).append(
|
|
897
|
+
(dimension_index, role, local_index)
|
|
898
|
+
)
|
|
899
|
+
|
|
900
|
+
issue_groups: dict[tuple[int, str, str, str, str, str], list[str]] = {}
|
|
901
|
+
issue_counts: dict[tuple[int, str, str, str, str, str], int] = {}
|
|
902
|
+
degenerate_series: list[dict[str, Any]] = []
|
|
903
|
+
series_profiles: list[SeriesProfile] = []
|
|
904
|
+
remaining_source_cells = MAX_CELL_READ_SIZE
|
|
905
|
+
remaining_dimension_spans = sum(
|
|
906
|
+
max(local_index for _, _, local_index in ref_dimensions)
|
|
907
|
+
- min(local_index for _, _, local_index in ref_dimensions)
|
|
908
|
+
+ 1
|
|
909
|
+
for ref_dimensions in mapped_by_ref.values()
|
|
910
|
+
)
|
|
911
|
+
for (source_sheet, source_range, bounds), ref_dimensions in mapped_by_ref.items():
|
|
912
|
+
dimension_start = bounds[2] if direction == "column" else bounds[0]
|
|
913
|
+
selected = {
|
|
914
|
+
dimension_start + local_index - 1: (dimension_index, role)
|
|
915
|
+
for dimension_index, role, local_index in ref_dimensions
|
|
916
|
+
}
|
|
917
|
+
dimension_span = max(selected) - min(selected) + 1
|
|
918
|
+
header_points = 0 if detached else 1
|
|
919
|
+
# Each sampled rectangle is read twice: cells-get supplies coordinates,
|
|
920
|
+
# styles, and hidden/filter semantics; table-get supplies typed values.
|
|
921
|
+
cells_per_dimension = remaining_source_cells // remaining_dimension_spans
|
|
922
|
+
remaining_dimension_spans -= dimension_span
|
|
923
|
+
point_axis_size = (
|
|
924
|
+
bounds[1] - bounds[0] + 1
|
|
925
|
+
if direction == "column"
|
|
926
|
+
else bounds[3] - bounds[2] + 1
|
|
927
|
+
)
|
|
928
|
+
if point_axis_size <= header_points:
|
|
929
|
+
unverifiable.append(
|
|
930
|
+
{
|
|
931
|
+
"chart_id": chart_id,
|
|
932
|
+
"reason": f"source has no data points: {source_sheet}!{source_range}",
|
|
933
|
+
}
|
|
934
|
+
)
|
|
935
|
+
continue
|
|
936
|
+
sample_points = min(MAX_SOURCE_SAMPLE_POINTS, max(0, (cells_per_dimension - header_points) // 2))
|
|
937
|
+
if sample_points == 0:
|
|
938
|
+
unverifiable.append({
|
|
939
|
+
"chart_id": chart_id,
|
|
940
|
+
"reason": f"source sampling 2000-cell budget cannot cover {source_sheet}!{source_range}",
|
|
941
|
+
})
|
|
942
|
+
continue
|
|
943
|
+
point_count = sample_points + header_points
|
|
944
|
+
if direction == "column":
|
|
945
|
+
checked_bounds = (
|
|
946
|
+
bounds[0],
|
|
947
|
+
min(bounds[1], bounds[0] + point_count - 1),
|
|
948
|
+
min(selected),
|
|
949
|
+
max(selected),
|
|
950
|
+
)
|
|
951
|
+
else:
|
|
952
|
+
checked_bounds = (
|
|
953
|
+
min(selected),
|
|
954
|
+
max(selected),
|
|
955
|
+
bounds[2],
|
|
956
|
+
min(bounds[3], bounds[2] + point_count - 1),
|
|
957
|
+
)
|
|
958
|
+
typed_bounds = (
|
|
959
|
+
(checked_bounds[0] + header_points, checked_bounds[1], checked_bounds[2], checked_bounds[3])
|
|
960
|
+
if direction == "column"
|
|
961
|
+
else (checked_bounds[0], checked_bounds[1], checked_bounds[2] + header_points, checked_bounds[3])
|
|
962
|
+
)
|
|
963
|
+
remaining_source_cells -= _bounds_area(checked_bounds) + _bounds_area(typed_bounds)
|
|
964
|
+
checked_range = _format_a1_bounds(checked_bounds)
|
|
965
|
+
same_sheet = source_sheet == owner_sheet_name
|
|
966
|
+
cells_data = _read_cells(
|
|
967
|
+
cache,
|
|
968
|
+
locator,
|
|
969
|
+
sheet_id=owner_sheet_id if same_sheet else None,
|
|
970
|
+
sheet_name=None if same_sheet else source_sheet,
|
|
971
|
+
cell_range=checked_range,
|
|
972
|
+
include="value,style",
|
|
973
|
+
skip_hidden=skip_hidden,
|
|
974
|
+
timeout=timeout,
|
|
975
|
+
)
|
|
976
|
+
cells_truncated = _cells_truncated(cells_data)
|
|
977
|
+
typed_rows, typed_columns, table_truncated = _read_typed_table(
|
|
978
|
+
cache,
|
|
979
|
+
locator,
|
|
980
|
+
sheet_id=owner_sheet_id if same_sheet else None,
|
|
981
|
+
sheet_name=None if same_sheet else source_sheet,
|
|
982
|
+
cell_range=_format_a1_bounds(typed_bounds),
|
|
983
|
+
timeout=timeout,
|
|
984
|
+
)
|
|
985
|
+
if cells_truncated:
|
|
986
|
+
unverifiable.append(
|
|
987
|
+
{"chart_id": chart_id, "reason": f"cells-get truncated for {source_sheet}!{checked_range}"}
|
|
988
|
+
)
|
|
989
|
+
if table_truncated:
|
|
990
|
+
unverifiable.append(
|
|
991
|
+
{
|
|
992
|
+
"chart_id": chart_id,
|
|
993
|
+
"reason": (
|
|
994
|
+
"table-get truncated or returned invalid typed data for "
|
|
995
|
+
f"{source_sheet}!{_format_a1_bounds(typed_bounds)}"
|
|
996
|
+
),
|
|
997
|
+
}
|
|
998
|
+
)
|
|
999
|
+
truncated = cells_truncated or table_truncated
|
|
1000
|
+
states = {
|
|
1001
|
+
coordinate: {
|
|
1002
|
+
"zero": False,
|
|
1003
|
+
"empty": False,
|
|
1004
|
+
"nonzero": False,
|
|
1005
|
+
"sample_point_count": 0,
|
|
1006
|
+
"numeric_value_count": 0,
|
|
1007
|
+
"unique_numeric_values": set(),
|
|
1008
|
+
}
|
|
1009
|
+
for coordinate in selected
|
|
1010
|
+
}
|
|
1011
|
+
type_unverifiable_dimensions: set[int] = set()
|
|
1012
|
+
visible_offsets = {
|
|
1013
|
+
(row_number - typed_bounds[0], column_index - typed_bounds[2])
|
|
1014
|
+
for row_number, column_index, _ in _iter_cells(cells_data)
|
|
1015
|
+
if typed_bounds[0] <= row_number <= typed_bounds[1]
|
|
1016
|
+
and typed_bounds[2] <= column_index <= typed_bounds[3]
|
|
1017
|
+
}
|
|
1018
|
+
for row_number, column_index, cell in _iter_cells(cells_data):
|
|
1019
|
+
coordinate = column_index if direction == "column" else row_number
|
|
1020
|
+
dimension = selected.get(coordinate)
|
|
1021
|
+
if dimension is None:
|
|
1022
|
+
continue
|
|
1023
|
+
if not detached and (
|
|
1024
|
+
(direction == "column" and row_number == bounds[0])
|
|
1025
|
+
or (direction != "column" and column_index == bounds[2])
|
|
1026
|
+
):
|
|
1027
|
+
continue
|
|
1028
|
+
states[coordinate]["sample_point_count"] += 1
|
|
1029
|
+
row_offset = row_number - typed_bounds[0]
|
|
1030
|
+
column_offset = column_index - typed_bounds[2]
|
|
1031
|
+
value, raw_type = _typed_cell(
|
|
1032
|
+
typed_rows,
|
|
1033
|
+
typed_columns,
|
|
1034
|
+
row_offset,
|
|
1035
|
+
column_offset,
|
|
1036
|
+
visible_offsets,
|
|
1037
|
+
)
|
|
1038
|
+
_update_series_state(states[coordinate], value)
|
|
1039
|
+
number_format = (
|
|
1040
|
+
cell.get("cell_styles", {}).get("number_format")
|
|
1041
|
+
if isinstance(cell, dict) and isinstance(cell.get("cell_styles"), dict)
|
|
1042
|
+
else None
|
|
1043
|
+
)
|
|
1044
|
+
reason = ""
|
|
1045
|
+
if _looks_numeric(str(value or "")):
|
|
1046
|
+
if str(number_format or "").strip() == "@":
|
|
1047
|
+
reason = "numeric_value_uses_text_format"
|
|
1048
|
+
elif raw_type == "string":
|
|
1049
|
+
reason = "numeric_value_stored_as_text"
|
|
1050
|
+
elif raw_type in {"ambiguous_string", "unknown"}:
|
|
1051
|
+
type_unverifiable_dimensions.add(dimension[0])
|
|
1052
|
+
if reason:
|
|
1053
|
+
dimension_index, role = dimension
|
|
1054
|
+
key = (
|
|
1055
|
+
dimension_index,
|
|
1056
|
+
role,
|
|
1057
|
+
source_sheet,
|
|
1058
|
+
source_range,
|
|
1059
|
+
checked_range,
|
|
1060
|
+
reason,
|
|
1061
|
+
)
|
|
1062
|
+
issue_counts[key] = issue_counts.get(key, 0) + 1
|
|
1063
|
+
samples = issue_groups.setdefault(key, [])
|
|
1064
|
+
if len(samples) < sample_limit:
|
|
1065
|
+
samples.append(f"{index_to_column(column_index)}{row_number}")
|
|
1066
|
+
|
|
1067
|
+
if truncated:
|
|
1068
|
+
continue
|
|
1069
|
+
for dimension_index in sorted(type_unverifiable_dimensions):
|
|
1070
|
+
unverifiable.append(
|
|
1071
|
+
{
|
|
1072
|
+
"chart_id": chart_id,
|
|
1073
|
+
"reason": (
|
|
1074
|
+
"numeric storage type cannot be attributed to individual cells after "
|
|
1075
|
+
f"typed column coercion for {source_sheet}!{checked_range}, "
|
|
1076
|
+
f"dimension {dimension_index}"
|
|
1077
|
+
),
|
|
1078
|
+
}
|
|
1079
|
+
)
|
|
1080
|
+
zero_candidates = {
|
|
1081
|
+
coordinate
|
|
1082
|
+
for coordinate, state in states.items()
|
|
1083
|
+
if not state["nonzero"]
|
|
1084
|
+
and _series_aggregate_type(data, selected[coordinate][0]) != "count"
|
|
1085
|
+
}
|
|
1086
|
+
data_start = bounds[0] + (0 if detached else 1)
|
|
1087
|
+
data_column = bounds[2] + (0 if detached else 1)
|
|
1088
|
+
dim2 = data.get("dim2")
|
|
1089
|
+
value_series = dim2.get("series") if isinstance(dim2, dict) else None
|
|
1090
|
+
for coordinate, state in states.items():
|
|
1091
|
+
dimension_index, role = selected[coordinate]
|
|
1092
|
+
if direction == "column":
|
|
1093
|
+
point_total = max(0, bounds[1] - data_start + 1)
|
|
1094
|
+
series_range = (
|
|
1095
|
+
f"{index_to_column(coordinate)}{data_start}:"
|
|
1096
|
+
f"{index_to_column(coordinate)}{bounds[1]}"
|
|
1097
|
+
if point_total
|
|
1098
|
+
else ""
|
|
1099
|
+
)
|
|
1100
|
+
else:
|
|
1101
|
+
point_total = max(0, bounds[3] - data_column + 1)
|
|
1102
|
+
series_range = (
|
|
1103
|
+
f"{index_to_column(data_column)}{coordinate}:"
|
|
1104
|
+
f"{index_to_column(bounds[3])}{coordinate}"
|
|
1105
|
+
if point_total
|
|
1106
|
+
else ""
|
|
1107
|
+
)
|
|
1108
|
+
source_series = next(
|
|
1109
|
+
(
|
|
1110
|
+
item
|
|
1111
|
+
for item in (value_series if isinstance(value_series, list) else [])
|
|
1112
|
+
if isinstance(item, dict)
|
|
1113
|
+
and item.get("index") is not None
|
|
1114
|
+
and int(item["index"]) == dimension_index
|
|
1115
|
+
),
|
|
1116
|
+
{},
|
|
1117
|
+
)
|
|
1118
|
+
profile = {
|
|
1119
|
+
"dimension_index": dimension_index,
|
|
1120
|
+
"series_name": str(
|
|
1121
|
+
source_series.get("name")
|
|
1122
|
+
or source_series.get("nameRef")
|
|
1123
|
+
or f"Series {dimension_index}"
|
|
1124
|
+
),
|
|
1125
|
+
"point_count": point_total,
|
|
1126
|
+
"sample_point_count": state["sample_point_count"],
|
|
1127
|
+
"checked_range": checked_range,
|
|
1128
|
+
"sampled": (
|
|
1129
|
+
checked_bounds[1] < bounds[1]
|
|
1130
|
+
if direction == "column"
|
|
1131
|
+
else checked_bounds[3] < bounds[3]
|
|
1132
|
+
),
|
|
1133
|
+
"numeric_value_count": state["numeric_value_count"],
|
|
1134
|
+
"unique_numeric_values": list(state["unique_numeric_values"]),
|
|
1135
|
+
"source_sheet": source_sheet,
|
|
1136
|
+
"source_range": source_range,
|
|
1137
|
+
"series_range": series_range,
|
|
1138
|
+
}
|
|
1139
|
+
aggregate_type = _series_aggregate_type(data, dimension_index)
|
|
1140
|
+
if aggregate_type == "count":
|
|
1141
|
+
profile["constant_check_unverifiable"] = True
|
|
1142
|
+
constant_labeled = (
|
|
1143
|
+
aggregate_type != "count"
|
|
1144
|
+
and state["numeric_value_count"] >= 2
|
|
1145
|
+
and len(state["unique_numeric_values"]) == 1
|
|
1146
|
+
and dimension_index
|
|
1147
|
+
in _labeled_series_indexes(snapshot, [{"dimension_index": dimension_index}])
|
|
1148
|
+
)
|
|
1149
|
+
if profile["sampled"]:
|
|
1150
|
+
profile["constant_check_unverifiable"] = True
|
|
1151
|
+
if coordinate in zero_candidates or constant_labeled:
|
|
1152
|
+
unverifiable.append({
|
|
1153
|
+
"chart_id": chart_id,
|
|
1154
|
+
"reason": (
|
|
1155
|
+
"zero/constant-series check is unverifiable outside sampled source "
|
|
1156
|
+
f"{source_sheet}!{checked_range} for dimension {dimension_index}"
|
|
1157
|
+
),
|
|
1158
|
+
})
|
|
1159
|
+
elif constant_labeled and _aggregation_can_change_constant(data, dimension_index):
|
|
1160
|
+
profile["constant_check_unverifiable"] = True
|
|
1161
|
+
unverifiable.append(
|
|
1162
|
+
{
|
|
1163
|
+
"chart_id": chart_id,
|
|
1164
|
+
"reason": (
|
|
1165
|
+
"constant-series check is unverifiable after category "
|
|
1166
|
+
f"aggregation for dimension {dimension_index}"
|
|
1167
|
+
),
|
|
1168
|
+
}
|
|
1169
|
+
)
|
|
1170
|
+
series_profiles.append(profile)
|
|
1171
|
+
if coordinate not in zero_candidates or profile["sampled"]:
|
|
1172
|
+
continue
|
|
1173
|
+
degenerate_series.append(
|
|
1174
|
+
{
|
|
1175
|
+
"chart_id": chart_id,
|
|
1176
|
+
"dimension_index": dimension_index,
|
|
1177
|
+
"role": role,
|
|
1178
|
+
"source_sheet": source_sheet,
|
|
1179
|
+
"source_range": source_range,
|
|
1180
|
+
"series_range": series_range,
|
|
1181
|
+
"reason": (
|
|
1182
|
+
"numeric_series_all_zero_or_empty"
|
|
1183
|
+
if states[coordinate]["zero"]
|
|
1184
|
+
else "numeric_series_all_empty"
|
|
1185
|
+
),
|
|
1186
|
+
"data_point_count": point_total,
|
|
1187
|
+
}
|
|
1188
|
+
)
|
|
1189
|
+
|
|
1190
|
+
issues = [
|
|
1191
|
+
{
|
|
1192
|
+
"chart_id": chart_id,
|
|
1193
|
+
"dimension_index": key[0],
|
|
1194
|
+
"role": key[1],
|
|
1195
|
+
"source_sheet": key[2],
|
|
1196
|
+
"source_range": key[3],
|
|
1197
|
+
"checked_range": key[4],
|
|
1198
|
+
"reason": key[5],
|
|
1199
|
+
"suggested_fix": (
|
|
1200
|
+
"set_numeric_number_format"
|
|
1201
|
+
if key[5] == "numeric_value_uses_text_format"
|
|
1202
|
+
else "rewrite_as_number_and_set_numeric_number_format"
|
|
1203
|
+
),
|
|
1204
|
+
"affected_sample_cell_count": issue_counts[key],
|
|
1205
|
+
"sample_cells": samples,
|
|
1206
|
+
}
|
|
1207
|
+
for key, samples in issue_groups.items()
|
|
1208
|
+
]
|
|
1209
|
+
return issues, degenerate_series, unverifiable, series_profiles
|
|
1210
|
+
|
|
1211
|
+
|
|
1212
|
+
def _locator(target: str) -> dict[str, str]:
|
|
1213
|
+
return {"url": target} if target.startswith(("http://", "https://")) else {"spreadsheet_token": target}
|
|
1214
|
+
|
|
1215
|
+
|
|
1216
|
+
def _sheet_counts(sheet: dict[str, Any]) -> tuple[int, int]:
|
|
1217
|
+
row_count = int(sheet.get("row_count") or sheet.get("rowCount") or 0)
|
|
1218
|
+
column_count = int(sheet.get("column_count") or sheet.get("columnCount") or 0)
|
|
1219
|
+
if row_count <= 0 or column_count <= 0:
|
|
1220
|
+
raise LarkCliError(f"Missing row_count/column_count for sheet {sheet_title(sheet)!r}")
|
|
1221
|
+
return row_count, column_count
|
|
1222
|
+
|
|
1223
|
+
|
|
1224
|
+
def check_sheet(
|
|
1225
|
+
locator: dict[str, str],
|
|
1226
|
+
sheet: dict[str, Any],
|
|
1227
|
+
*,
|
|
1228
|
+
timeout: int,
|
|
1229
|
+
sample_limit: int,
|
|
1230
|
+
cell_cache: CellCache | None = None,
|
|
1231
|
+
) -> dict[str, Any]:
|
|
1232
|
+
cell_cache = cell_cache if cell_cache is not None else {}
|
|
1233
|
+
sheet_id = sheet_identifier(sheet)
|
|
1234
|
+
title = sheet_title(sheet)
|
|
1235
|
+
if not sheet_id:
|
|
1236
|
+
raise LarkCliError(f"Missing sheet_id for sheet {title!r}")
|
|
1237
|
+
|
|
1238
|
+
chart_data = envelope_data(
|
|
1239
|
+
run_sheets("+chart-list", **locator, sheet_id=sheet_id, timeout=timeout)
|
|
1240
|
+
)
|
|
1241
|
+
charts = extract_charts(chart_data, sheet_id, title)
|
|
1242
|
+
unverifiable: list[dict[str, str]] = []
|
|
1243
|
+
expected_chart_count = sheet.get("chart_count")
|
|
1244
|
+
if expected_chart_count is not None and int(expected_chart_count) != len(charts):
|
|
1245
|
+
unverifiable.append(
|
|
1246
|
+
{
|
|
1247
|
+
"chart_id": "",
|
|
1248
|
+
"reason": (
|
|
1249
|
+
f"chart-list returned {len(charts)} charts, "
|
|
1250
|
+
f"but workbook-info reported {int(expected_chart_count)}"
|
|
1251
|
+
),
|
|
1252
|
+
}
|
|
1253
|
+
)
|
|
1254
|
+
if not charts:
|
|
1255
|
+
return {
|
|
1256
|
+
"sheet_id": sheet_id,
|
|
1257
|
+
"sheet_name": title,
|
|
1258
|
+
"chart_count": 0,
|
|
1259
|
+
"sheet_size_px": None,
|
|
1260
|
+
"chart_overlaps": [],
|
|
1261
|
+
"cell_content_overlaps": [],
|
|
1262
|
+
"numeric_source_format_issues": [],
|
|
1263
|
+
"numeric_source_samples": [],
|
|
1264
|
+
"degenerate_numeric_series": [],
|
|
1265
|
+
"constant_labeled_series": [],
|
|
1266
|
+
"unbound_secondary_axes": [],
|
|
1267
|
+
"undersized_charts": [],
|
|
1268
|
+
"overwide_charts": [],
|
|
1269
|
+
"out_of_visible_range": [],
|
|
1270
|
+
"unverifiable_charts": unverifiable,
|
|
1271
|
+
"issue_count": 0,
|
|
1272
|
+
"unverifiable_count": len(unverifiable),
|
|
1273
|
+
"warnings": [],
|
|
1274
|
+
}
|
|
1275
|
+
|
|
1276
|
+
row_count, column_count = _sheet_counts(sheet)
|
|
1277
|
+
structure_data = envelope_data(
|
|
1278
|
+
run_sheets(
|
|
1279
|
+
"+sheet-info",
|
|
1280
|
+
**locator,
|
|
1281
|
+
sheet_id=sheet_id,
|
|
1282
|
+
flags={"include": "row_heights,col_widths"},
|
|
1283
|
+
timeout=timeout,
|
|
1284
|
+
)
|
|
1285
|
+
)
|
|
1286
|
+
row_edges, column_edges, warnings = build_layout(
|
|
1287
|
+
extract_sheet_structure(structure_data), row_count, column_count
|
|
1288
|
+
)
|
|
1289
|
+
|
|
1290
|
+
rectangles: list[dict[str, Any]] = []
|
|
1291
|
+
for chart in charts:
|
|
1292
|
+
chart_id = str(chart.get("chart_id") or chart.get("id") or "")
|
|
1293
|
+
if not chart_id:
|
|
1294
|
+
unverifiable.append({"chart_id": "", "reason": "chart is missing chart_id"})
|
|
1295
|
+
continue
|
|
1296
|
+
try:
|
|
1297
|
+
rectangles.append(chart_rectangle(chart, row_edges, column_edges))
|
|
1298
|
+
except (KeyError, TypeError, ValueError) as exc:
|
|
1299
|
+
unverifiable.append({"chart_id": chart_id, "reason": str(exc)})
|
|
1300
|
+
|
|
1301
|
+
overlaps: list[dict[str, Any]] = []
|
|
1302
|
+
for index, first in enumerate(rectangles):
|
|
1303
|
+
for second in rectangles[index + 1 :]:
|
|
1304
|
+
overlap = intersection(first, second)
|
|
1305
|
+
if overlap:
|
|
1306
|
+
overlaps.append(
|
|
1307
|
+
{
|
|
1308
|
+
"chart_ids": [first["chart_id"], second["chart_id"]],
|
|
1309
|
+
"charts": [chart_context(first), chart_context(second)],
|
|
1310
|
+
"intersection": overlap,
|
|
1311
|
+
}
|
|
1312
|
+
)
|
|
1313
|
+
|
|
1314
|
+
sheet_width = column_edges[-1]
|
|
1315
|
+
sheet_height = row_edges[-1]
|
|
1316
|
+
out_of_bounds: list[dict[str, Any]] = []
|
|
1317
|
+
content_overlaps: list[dict[str, Any]] = []
|
|
1318
|
+
covered_items: list[tuple[dict[str, Any], CellBounds]] = []
|
|
1319
|
+
for rectangle in rectangles:
|
|
1320
|
+
overflow = {
|
|
1321
|
+
"left": round(max(0.0, -rectangle["left"]), 2),
|
|
1322
|
+
"top": round(max(0.0, -rectangle["top"]), 2),
|
|
1323
|
+
"right": round(max(0.0, rectangle["right"] - sheet_width), 2),
|
|
1324
|
+
"bottom": round(max(0.0, rectangle["bottom"] - sheet_height), 2),
|
|
1325
|
+
}
|
|
1326
|
+
if any(overflow.values()):
|
|
1327
|
+
out_of_bounds.append({**chart_context(rectangle), "overflow_px": overflow})
|
|
1328
|
+
|
|
1329
|
+
covered_range = rectangle_cell_range(rectangle, row_edges, column_edges)
|
|
1330
|
+
if not covered_range:
|
|
1331
|
+
continue
|
|
1332
|
+
covered_items.append((rectangle, _parse_a1_bounds(covered_range)))
|
|
1333
|
+
|
|
1334
|
+
for cluster in _cluster_cell_reads(covered_items):
|
|
1335
|
+
read_range = _format_a1_bounds(cluster["bounds"])
|
|
1336
|
+
cells_data = _read_cells(
|
|
1337
|
+
cell_cache,
|
|
1338
|
+
locator,
|
|
1339
|
+
sheet_id=sheet_id,
|
|
1340
|
+
sheet_name=None,
|
|
1341
|
+
cell_range=read_range,
|
|
1342
|
+
include="value,formula,comment",
|
|
1343
|
+
timeout=timeout,
|
|
1344
|
+
)
|
|
1345
|
+
for rectangle, bounds in cluster["members"]:
|
|
1346
|
+
covered_range = _format_a1_bounds(bounds)
|
|
1347
|
+
count, samples, truncated = non_empty_cells(cells_data, sample_limit, bounds)
|
|
1348
|
+
if truncated:
|
|
1349
|
+
unverifiable.append(
|
|
1350
|
+
{
|
|
1351
|
+
"chart_id": rectangle["chart_id"],
|
|
1352
|
+
"reason": f"cells-get truncated for {read_range}",
|
|
1353
|
+
}
|
|
1354
|
+
)
|
|
1355
|
+
if count:
|
|
1356
|
+
content_overlaps.append(
|
|
1357
|
+
{
|
|
1358
|
+
**chart_context(rectangle),
|
|
1359
|
+
"covered_range": covered_range,
|
|
1360
|
+
"non_empty_cell_count": count,
|
|
1361
|
+
"sample_cells": samples,
|
|
1362
|
+
}
|
|
1363
|
+
)
|
|
1364
|
+
|
|
1365
|
+
numeric_source_issues: list[dict[str, Any]] = []
|
|
1366
|
+
numeric_source_samples: list[dict[str, Any]] = []
|
|
1367
|
+
degenerate_numeric_series: list[dict[str, Any]] = []
|
|
1368
|
+
constant_series_issues: list[dict[str, Any]] = []
|
|
1369
|
+
unbound_secondary_axes: list[dict[str, Any]] = []
|
|
1370
|
+
undersized_charts: list[dict[str, Any]] = []
|
|
1371
|
+
overwide_charts: list[dict[str, Any]] = []
|
|
1372
|
+
for chart in charts:
|
|
1373
|
+
issues, degenerate, source_unverifiable, profiles = _numeric_source_issues(
|
|
1374
|
+
chart,
|
|
1375
|
+
owner_sheet_id=sheet_id,
|
|
1376
|
+
owner_sheet_name=title,
|
|
1377
|
+
cache=cell_cache,
|
|
1378
|
+
locator=locator,
|
|
1379
|
+
timeout=timeout,
|
|
1380
|
+
sample_limit=sample_limit,
|
|
1381
|
+
)
|
|
1382
|
+
numeric_source_issues.extend(issues)
|
|
1383
|
+
numeric_source_samples.extend(
|
|
1384
|
+
{"chart_id": str(chart.get("chart_id") or chart.get("id") or ""), **profile}
|
|
1385
|
+
for profile in profiles
|
|
1386
|
+
if "checked_range" in profile
|
|
1387
|
+
)
|
|
1388
|
+
degenerate_numeric_series.extend(degenerate)
|
|
1389
|
+
unverifiable.extend(source_unverifiable)
|
|
1390
|
+
constant_series_issues.extend(_constant_labeled_series(chart, profiles))
|
|
1391
|
+
unbound_secondary_axis = _unbound_secondary_axis(chart)
|
|
1392
|
+
if unbound_secondary_axis:
|
|
1393
|
+
unbound_secondary_axes.append(unbound_secondary_axis)
|
|
1394
|
+
undersized = _undersized_chart(chart)
|
|
1395
|
+
if undersized:
|
|
1396
|
+
undersized_charts.append(undersized)
|
|
1397
|
+
overwide = _overwide_chart(chart)
|
|
1398
|
+
if overwide:
|
|
1399
|
+
overwide_charts.append(overwide)
|
|
1400
|
+
|
|
1401
|
+
issue_count = (
|
|
1402
|
+
len(overlaps)
|
|
1403
|
+
+ len(out_of_bounds)
|
|
1404
|
+
+ len(content_overlaps)
|
|
1405
|
+
+ len(numeric_source_issues)
|
|
1406
|
+
+ len(degenerate_numeric_series)
|
|
1407
|
+
+ len(constant_series_issues)
|
|
1408
|
+
+ len(unbound_secondary_axes)
|
|
1409
|
+
+ len(undersized_charts)
|
|
1410
|
+
+ len(overwide_charts)
|
|
1411
|
+
)
|
|
1412
|
+
return {
|
|
1413
|
+
"sheet_id": sheet_id,
|
|
1414
|
+
"sheet_name": title,
|
|
1415
|
+
"chart_count": len(charts),
|
|
1416
|
+
"sheet_size_px": {"width": round(sheet_width, 2), "height": round(sheet_height, 2)},
|
|
1417
|
+
"chart_overlaps": overlaps,
|
|
1418
|
+
"cell_content_overlaps": content_overlaps,
|
|
1419
|
+
"numeric_source_format_issues": numeric_source_issues,
|
|
1420
|
+
"numeric_source_samples": numeric_source_samples,
|
|
1421
|
+
"degenerate_numeric_series": degenerate_numeric_series,
|
|
1422
|
+
"constant_labeled_series": constant_series_issues,
|
|
1423
|
+
"unbound_secondary_axes": unbound_secondary_axes,
|
|
1424
|
+
"undersized_charts": undersized_charts,
|
|
1425
|
+
"overwide_charts": overwide_charts,
|
|
1426
|
+
"out_of_visible_range": out_of_bounds,
|
|
1427
|
+
"unverifiable_charts": unverifiable,
|
|
1428
|
+
"issue_count": issue_count,
|
|
1429
|
+
"unverifiable_count": len(unverifiable),
|
|
1430
|
+
"warnings": warnings,
|
|
1431
|
+
}
|
|
1432
|
+
|
|
1433
|
+
|
|
1434
|
+
def parse_args() -> argparse.Namespace:
|
|
1435
|
+
parser = argparse.ArgumentParser(
|
|
1436
|
+
description=(
|
|
1437
|
+
"Check chart overlap, covered cell content, worksheet boundary overflow, "
|
|
1438
|
+
"minimum size, excessive width, constant labeled series, numeric source-cell "
|
|
1439
|
+
"formats, all-zero/empty numeric series, and unbound combo-chart secondary axes."
|
|
1440
|
+
)
|
|
1441
|
+
)
|
|
1442
|
+
parser.add_argument("sheet_id", help="Spreadsheet URL or spreadsheet token")
|
|
1443
|
+
parser.add_argument("--worksheet-id", help="Only check this worksheet reference_id")
|
|
1444
|
+
parser.add_argument("--timeout", type=int, default=60)
|
|
1445
|
+
parser.add_argument("--sample-limit", type=int, default=10)
|
|
1446
|
+
return parser.parse_args()
|
|
1447
|
+
|
|
1448
|
+
|
|
1449
|
+
def success_envelope(results: list[dict[str, Any]]) -> dict[str, Any]:
|
|
1450
|
+
issue_count = sum(result["issue_count"] for result in results)
|
|
1451
|
+
unverifiable_count = sum(result["unverifiable_count"] for result in results)
|
|
1452
|
+
warnings = [
|
|
1453
|
+
f"{result['sheet_name'] or result['sheet_id']}: {warning}"
|
|
1454
|
+
for result in results
|
|
1455
|
+
for warning in result["warnings"]
|
|
1456
|
+
]
|
|
1457
|
+
return {
|
|
1458
|
+
"ok": True,
|
|
1459
|
+
"engine": "lark",
|
|
1460
|
+
"action": ACTION,
|
|
1461
|
+
"data": {
|
|
1462
|
+
"passed": issue_count == 0 and unverifiable_count == 0,
|
|
1463
|
+
"scope_note": (
|
|
1464
|
+
"out_of_visible_range checks worksheet drawable bounds, not a device-specific browser viewport; "
|
|
1465
|
+
"numeric source checks sample at most the first 50 data points of each chart value dimension "
|
|
1466
|
+
"for formats, all-zero/empty, and constant-value checks; zero/constant results are conclusive "
|
|
1467
|
+
"only when that sample covers the complete series, otherwise they are marked unverifiable"
|
|
1468
|
+
),
|
|
1469
|
+
"summary": {
|
|
1470
|
+
"worksheet_count": len(results),
|
|
1471
|
+
"chart_count": sum(result["chart_count"] for result in results),
|
|
1472
|
+
"issue_count": issue_count,
|
|
1473
|
+
"unverifiable_count": unverifiable_count,
|
|
1474
|
+
},
|
|
1475
|
+
"sheets": results,
|
|
1476
|
+
},
|
|
1477
|
+
"warnings": warnings,
|
|
1478
|
+
}
|
|
1479
|
+
|
|
1480
|
+
|
|
1481
|
+
def report_exit_code(report: dict[str, Any]) -> int:
|
|
1482
|
+
if report["data"]["passed"]:
|
|
1483
|
+
return 0
|
|
1484
|
+
if report["data"]["summary"]["issue_count"] > 0:
|
|
1485
|
+
return 2
|
|
1486
|
+
return 1
|
|
1487
|
+
|
|
1488
|
+
|
|
1489
|
+
def main() -> None:
|
|
1490
|
+
args = parse_args()
|
|
1491
|
+
locator = _locator(args.sheet_id)
|
|
1492
|
+
cell_cache: CellCache = {}
|
|
1493
|
+
try:
|
|
1494
|
+
workbook_data = envelope_data(
|
|
1495
|
+
run_sheets("+workbook-info", **locator, timeout=args.timeout)
|
|
1496
|
+
)
|
|
1497
|
+
sheets = resolve_target_sheets(workbook_data, sheet_id=args.worksheet_id)
|
|
1498
|
+
if not args.worksheet_id:
|
|
1499
|
+
sheets = [sheet for sheet in sheets if not bool(sheet.get("is_hidden"))]
|
|
1500
|
+
if not sheets:
|
|
1501
|
+
raise LarkCliError("No visible worksheet matched")
|
|
1502
|
+
results = [
|
|
1503
|
+
check_sheet(
|
|
1504
|
+
locator,
|
|
1505
|
+
sheet,
|
|
1506
|
+
timeout=args.timeout,
|
|
1507
|
+
sample_limit=args.sample_limit,
|
|
1508
|
+
cell_cache=cell_cache,
|
|
1509
|
+
)
|
|
1510
|
+
for sheet in sheets
|
|
1511
|
+
]
|
|
1512
|
+
except (LarkCliError, KeyError, TypeError, ValueError) as exc:
|
|
1513
|
+
emit_error(ACTION, str(exc))
|
|
1514
|
+
raise SystemExit(1) from exc
|
|
1515
|
+
|
|
1516
|
+
report = success_envelope(results)
|
|
1517
|
+
print(json.dumps(report, ensure_ascii=False, indent=2))
|
|
1518
|
+
exit_code = report_exit_code(report)
|
|
1519
|
+
if exit_code:
|
|
1520
|
+
raise SystemExit(exit_code)
|
|
1521
|
+
|
|
1522
|
+
|
|
1523
|
+
if __name__ == "__main__":
|
|
1524
|
+
main()
|