modao-prd-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- modao_prd_cli/__init__.py +1 -0
- modao_prd_cli/modao_prd/__init__.py +4 -0
- modao_prd_cli/modao_prd/__main__.py +7 -0
- modao_prd_cli/modao_prd/browser.py +578 -0
- modao_prd_cli/modao_prd/capture_evidence.py +547 -0
- modao_prd_cli/modao_prd/classifier.py +96 -0
- modao_prd_cli/modao_prd/cli.py +205 -0
- modao_prd_cli/modao_prd/errors.py +31 -0
- modao_prd_cli/modao_prd/evidence.py +207 -0
- modao_prd_cli/modao_prd/explorer.py +300 -0
- modao_prd_cli/modao_prd/extractor.py +465 -0
- modao_prd_cli/modao_prd/models.py +56 -0
- modao_prd_cli/modao_prd/normalizer.py +135 -0
- modao_prd_cli/modao_prd/schemas/coverage-1.0.json +15 -0
- modao_prd_cli/modao_prd/schemas/document-2.0.json +21 -0
- modao_prd_cli/modao_prd/schemas/document-2.1.json +31 -0
- modao_prd_cli/modao_prd/schemas/manifest-1.0.json +28 -0
- modao_prd_cli/modao_prd/tests/__init__.py +1 -0
- modao_prd_cli/modao_prd/tests/fixtures/modao_sample.html +20 -0
- modao_prd_cli/modao_prd/tests/test_browser.py +54 -0
- modao_prd_cli/modao_prd/tests/test_classifier.py +32 -0
- modao_prd_cli/modao_prd/tests/test_cli.py +88 -0
- modao_prd_cli/modao_prd/tests/test_extractor.py +57 -0
- modao_prd_cli/modao_prd/tests/test_full_e2e.py +23 -0
- modao_prd_cli/modao_prd/tests/test_writers.py +79 -0
- modao_prd_cli/modao_prd/writers.py +468 -0
- modao_prd_cli-0.1.0.dist-info/METADATA +108 -0
- modao_prd_cli-0.1.0.dist-info/RECORD +32 -0
- modao_prd_cli-0.1.0.dist-info/WHEEL +5 -0
- modao_prd_cli-0.1.0.dist-info/entry_points.txt +2 -0
- modao_prd_cli-0.1.0.dist-info/licenses/LICENSE +22 -0
- modao_prd_cli-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,547 @@
|
|
|
1
|
+
"""Evidence capture helpers for rendered Modao pages.
|
|
2
|
+
|
|
3
|
+
The browser layer keeps the original DOM records for the existing extractor and
|
|
4
|
+
adds a loss-aware evidence snapshot beside them. The snapshot is deliberately
|
|
5
|
+
mechanical: it records what the browser rendered and does not infer product
|
|
6
|
+
meaning.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import hashlib
|
|
12
|
+
import json
|
|
13
|
+
import mimetypes
|
|
14
|
+
import re
|
|
15
|
+
from dataclasses import asdict
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Any, Callable
|
|
18
|
+
from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit
|
|
19
|
+
|
|
20
|
+
from .evidence import EvidenceBundleWriter
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
SENSITIVE_QUERY_KEYS = {
|
|
24
|
+
"access_token",
|
|
25
|
+
"authorization",
|
|
26
|
+
"code",
|
|
27
|
+
"key",
|
|
28
|
+
"secret",
|
|
29
|
+
"session",
|
|
30
|
+
"sig",
|
|
31
|
+
"signature",
|
|
32
|
+
"token",
|
|
33
|
+
"accesstoken",
|
|
34
|
+
"refreshtoken",
|
|
35
|
+
"clientsecret",
|
|
36
|
+
"jwt",
|
|
37
|
+
"auth",
|
|
38
|
+
"ticket",
|
|
39
|
+
"nonce",
|
|
40
|
+
"sessionid",
|
|
41
|
+
}
|
|
42
|
+
DEBUG_TEXT_CONTENT_TYPES = {
|
|
43
|
+
"application/json",
|
|
44
|
+
"application/manifest+json",
|
|
45
|
+
"application/xml",
|
|
46
|
+
"text/html",
|
|
47
|
+
"text/plain",
|
|
48
|
+
}
|
|
49
|
+
NOISE_ASSET_RE = re.compile(r"(?:favicon|loading|logo|spinner|tracker|tracking|pixel|analytics|beacon|collect)", re.I)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def redact_url(value: str, *, base_url: str | None = None) -> str:
|
|
53
|
+
"""Remove credential-like query values and optionally reduce cross-origin URLs."""
|
|
54
|
+
|
|
55
|
+
try:
|
|
56
|
+
parsed = urlsplit(value)
|
|
57
|
+
except ValueError:
|
|
58
|
+
return "[invalid-url]"
|
|
59
|
+
pairs = [
|
|
60
|
+
(key, val)
|
|
61
|
+
for key, val in parse_qsl(parsed.query, keep_blank_values=True)
|
|
62
|
+
if key.casefold() not in SENSITIVE_QUERY_KEYS
|
|
63
|
+
]
|
|
64
|
+
query = urlencode(pairs)
|
|
65
|
+
sanitized = urlunsplit((parsed.scheme, parsed.netloc, parsed.path, query, ""))
|
|
66
|
+
if base_url and not same_origin(sanitized, base_url):
|
|
67
|
+
return urlunsplit((parsed.scheme, parsed.netloc, parsed.path, "", ""))
|
|
68
|
+
return sanitized
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def redact_text(value: str) -> str:
|
|
72
|
+
"""Remove credential material from debug text without rewriting product copy."""
|
|
73
|
+
|
|
74
|
+
value = re.sub(r"(?i)(authorization\s*[:=]\s*bearer\s+)[^\s,;\"']+", r"\1[REDACTED]", value)
|
|
75
|
+
value = re.sub(r"(?i)((?:cookie|set-cookie)\s*[:=]\s*)[^\n]+", r"\1[REDACTED]", value)
|
|
76
|
+
value = re.sub(
|
|
77
|
+
r"(?i)([?&](?:access[_-]?token|refresh[_-]?token|client[_-]?secret|authorization|signature|token|jwt|ticket)=)[^&#\s\"']+",
|
|
78
|
+
r"\1[REDACTED]",
|
|
79
|
+
value,
|
|
80
|
+
)
|
|
81
|
+
return value
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def same_origin(left: str, right: str) -> bool:
|
|
85
|
+
try:
|
|
86
|
+
a = urlsplit(left)
|
|
87
|
+
b = urlsplit(right)
|
|
88
|
+
except ValueError:
|
|
89
|
+
return False
|
|
90
|
+
return (a.scheme.lower(), a.hostname, a.port or _default_port(a.scheme)) == (
|
|
91
|
+
b.scheme.lower(),
|
|
92
|
+
b.hostname,
|
|
93
|
+
b.port or _default_port(b.scheme),
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def merge_evidence(existing: list[dict[str, Any]], incoming: list[dict[str, Any]]) -> None:
|
|
98
|
+
"""Append evidence records while keeping IDs unique.
|
|
99
|
+
|
|
100
|
+
Network response callbacks can add assets while a state screenshot is
|
|
101
|
+
being captured. The old offset-based IDs were therefore able to collide
|
|
102
|
+
with a newly observed asset, causing a real file to be present in the
|
|
103
|
+
bundle without a reference in ``document.json``.
|
|
104
|
+
"""
|
|
105
|
+
|
|
106
|
+
used = {str(item.get("id")) for item in existing if item.get("id")}
|
|
107
|
+
next_number = 1
|
|
108
|
+
for value in used:
|
|
109
|
+
match = re.fullmatch(r"evidence-(\d+)", value)
|
|
110
|
+
if match:
|
|
111
|
+
next_number = max(next_number, int(match.group(1)) + 1)
|
|
112
|
+
for item in incoming:
|
|
113
|
+
record = dict(item)
|
|
114
|
+
evidence_id = str(record.get("id") or "")
|
|
115
|
+
if not evidence_id or evidence_id in used:
|
|
116
|
+
while f"evidence-{next_number:03d}" in used:
|
|
117
|
+
next_number += 1
|
|
118
|
+
evidence_id = f"evidence-{next_number:03d}"
|
|
119
|
+
next_number += 1
|
|
120
|
+
record["id"] = evidence_id
|
|
121
|
+
used.add(evidence_id)
|
|
122
|
+
existing.append(record)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _default_port(scheme: str) -> int | None:
|
|
126
|
+
return {"http": 80, "https": 443}.get(scheme.lower())
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def capture_state_evidence(
|
|
130
|
+
page: Any,
|
|
131
|
+
context: Any,
|
|
132
|
+
*,
|
|
133
|
+
writer: EvidenceBundleWriter | None,
|
|
134
|
+
state_id: str,
|
|
135
|
+
requested_url: str,
|
|
136
|
+
evidence_mode: str,
|
|
137
|
+
max_item: int,
|
|
138
|
+
max_total: int,
|
|
139
|
+
warnings: list[str],
|
|
140
|
+
network_records: list[dict[str, Any]],
|
|
141
|
+
evidence_id_offset: int = 0,
|
|
142
|
+
) -> tuple[list[dict[str, Any]], dict[str, Any], dict[str, Any]]:
|
|
143
|
+
"""Capture rendered page artifacts and return refs, DOM facts and stats."""
|
|
144
|
+
|
|
145
|
+
evidence: list[dict[str, Any]] = []
|
|
146
|
+
snapshot: dict[str, Any] = {}
|
|
147
|
+
|
|
148
|
+
def add_artifact(relative_path: str, content: Any, media_type: str, kind: str) -> None:
|
|
149
|
+
if writer is None:
|
|
150
|
+
return
|
|
151
|
+
try:
|
|
152
|
+
item = writer.add(relative_path, content, media_type=media_type)
|
|
153
|
+
except Exception as exc:
|
|
154
|
+
warnings.append(f"证据文件未保存({kind}):{_short_error(exc)}")
|
|
155
|
+
return
|
|
156
|
+
evidence.append({"id": f"evidence-{evidence_id_offset + len(evidence) + 1:03d}", "kind": kind, "state_id": state_id, **asdict(item)})
|
|
157
|
+
|
|
158
|
+
if evidence_mode == "full":
|
|
159
|
+
try:
|
|
160
|
+
rendered_html = redact_text(page.content())
|
|
161
|
+
add_artifact(f"evidence/{state_id}.rendered.html", rendered_html, "text/html", "html")
|
|
162
|
+
except Exception as exc:
|
|
163
|
+
warnings.append(f"无法保存渲染后 HTML:{_short_error(exc)}")
|
|
164
|
+
|
|
165
|
+
try:
|
|
166
|
+
snapshot = page.evaluate(DOM_LAYOUT_SCRIPT)
|
|
167
|
+
snapshot = sanitize_snapshot(snapshot, base_url=requested_url)
|
|
168
|
+
warnings.extend(str(item) for item in snapshot.get("warnings", []) if item)
|
|
169
|
+
add_artifact(
|
|
170
|
+
f"evidence/{state_id}.dom.json",
|
|
171
|
+
json.dumps(snapshot, ensure_ascii=False, indent=2) + "\n",
|
|
172
|
+
"application/json",
|
|
173
|
+
"dom",
|
|
174
|
+
)
|
|
175
|
+
except Exception as exc:
|
|
176
|
+
warnings.append(f"无法保存 DOM 布局快照:{_short_error(exc)}")
|
|
177
|
+
|
|
178
|
+
if evidence_mode == "full":
|
|
179
|
+
try:
|
|
180
|
+
try:
|
|
181
|
+
aria = page.aria_snapshot(mode="ai", boxes=True)
|
|
182
|
+
except TypeError:
|
|
183
|
+
aria = page.aria_snapshot()
|
|
184
|
+
add_artifact(f"evidence/{state_id}.aria.yaml", redact_text(aria), "text/yaml", "aria")
|
|
185
|
+
except Exception as exc:
|
|
186
|
+
warnings.append(f"无法保存 ARIA 快照:{_short_error(exc)}")
|
|
187
|
+
if evidence_mode == "full":
|
|
188
|
+
try:
|
|
189
|
+
screenshot = page.screenshot(full_page=True, animations="disabled")
|
|
190
|
+
add_artifact(f"evidence/{state_id}.full.png", screenshot, "image/png", "screenshot")
|
|
191
|
+
except Exception as exc:
|
|
192
|
+
warnings.append(f"无法保存页面截图:{_short_error(exc)}")
|
|
193
|
+
if evidence_mode in {"essential", "full"}:
|
|
194
|
+
try:
|
|
195
|
+
_capture_canvas_images(page, add_artifact, state_id=state_id, warnings=warnings)
|
|
196
|
+
except Exception as exc:
|
|
197
|
+
warnings.append(f"无法枚举画布截图:{_short_error(exc)}")
|
|
198
|
+
|
|
199
|
+
stats = {
|
|
200
|
+
"dom_node_count": len(snapshot.get("nodes", [])),
|
|
201
|
+
"visual_only_count": len(snapshot.get("visual_only", [])),
|
|
202
|
+
"frame_count": snapshot.get("frame_count", 0),
|
|
203
|
+
"evidence_count": len(evidence),
|
|
204
|
+
"network_count": len(network_records),
|
|
205
|
+
"max_item": max_item,
|
|
206
|
+
"max_total": max_total,
|
|
207
|
+
}
|
|
208
|
+
return evidence, snapshot, stats
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def make_response_handler(
|
|
212
|
+
*,
|
|
213
|
+
writer: EvidenceBundleWriter | None,
|
|
214
|
+
requested_url: str,
|
|
215
|
+
state_id: str,
|
|
216
|
+
evidence: list[dict[str, Any]],
|
|
217
|
+
network_records: list[dict[str, Any]],
|
|
218
|
+
warnings: list[str],
|
|
219
|
+
max_item: int,
|
|
220
|
+
max_total: int,
|
|
221
|
+
evidence_mode: str = "full",
|
|
222
|
+
) -> Callable[[Any], None]:
|
|
223
|
+
"""Create a safe Playwright response callback for one capture run."""
|
|
224
|
+
|
|
225
|
+
def handle(response: Any) -> None:
|
|
226
|
+
try:
|
|
227
|
+
url = str(response.url)
|
|
228
|
+
if evidence_mode == "none":
|
|
229
|
+
return
|
|
230
|
+
response_same_origin = same_origin(url, requested_url)
|
|
231
|
+
if evidence_mode == "essential" and not response_same_origin:
|
|
232
|
+
return
|
|
233
|
+
headers = getattr(response, "headers", {}) or {}
|
|
234
|
+
content_type = str(headers.get("content-type", "")).split(";", 1)[0].strip().lower()
|
|
235
|
+
record: dict[str, Any] = {
|
|
236
|
+
"state_id": state_id,
|
|
237
|
+
"url": redact_url(url, base_url=requested_url),
|
|
238
|
+
"same_origin": response_same_origin,
|
|
239
|
+
"status": int(response.status),
|
|
240
|
+
"ok": bool(response.ok),
|
|
241
|
+
"content_type": content_type,
|
|
242
|
+
"headers": {
|
|
243
|
+
key: str(headers[key])
|
|
244
|
+
for key in ("content-type", "content-length", "etag", "last-modified", "cache-control")
|
|
245
|
+
if key in headers
|
|
246
|
+
},
|
|
247
|
+
}
|
|
248
|
+
is_image = content_type.startswith("image/")
|
|
249
|
+
is_debug_document = same_origin(url, requested_url) and content_type in DEBUG_TEXT_CONTENT_TYPES
|
|
250
|
+
should_capture_body = bool(writer and evidence_mode != "none")
|
|
251
|
+
should_capture_body = should_capture_body and (
|
|
252
|
+
(is_image and same_origin(url, requested_url))
|
|
253
|
+
or (evidence_mode == "full" and is_debug_document)
|
|
254
|
+
)
|
|
255
|
+
if should_capture_body:
|
|
256
|
+
try:
|
|
257
|
+
body = response.body()
|
|
258
|
+
if content_type in DEBUG_TEXT_CONTENT_TYPES:
|
|
259
|
+
body = redact_text(body.decode("utf-8", "replace")).encode("utf-8")
|
|
260
|
+
if is_image and not _is_useful_image(body, url, content_type):
|
|
261
|
+
record["skipped_reason"] = "irrelevant_or_tiny_asset"
|
|
262
|
+
elif not body:
|
|
263
|
+
record["skipped_reason"] = "empty_body"
|
|
264
|
+
elif len(body) <= max_item:
|
|
265
|
+
digest = hashlib.sha256(body).hexdigest()
|
|
266
|
+
suffix = _suffix(content_type, url)
|
|
267
|
+
directory = "evidence/assets" if is_image else "evidence/network/bodies"
|
|
268
|
+
relative = f"{directory}/{digest}{suffix}"
|
|
269
|
+
if writer:
|
|
270
|
+
existing = next((item for item in evidence if item.get("path") == relative), None)
|
|
271
|
+
if existing:
|
|
272
|
+
record["evidence_id"] = existing["id"]
|
|
273
|
+
else:
|
|
274
|
+
item = writer.add(relative, body, media_type=content_type or "application/octet-stream")
|
|
275
|
+
evidence.append(
|
|
276
|
+
{
|
|
277
|
+
"id": f"evidence-{len(evidence) + 1:03d}",
|
|
278
|
+
"kind": "asset" if is_image else "network_body",
|
|
279
|
+
"state_id": state_id,
|
|
280
|
+
**asdict(item),
|
|
281
|
+
}
|
|
282
|
+
)
|
|
283
|
+
record["evidence_id"] = evidence[-1]["id"]
|
|
284
|
+
else:
|
|
285
|
+
record["skipped_reason"] = "item_size_limit"
|
|
286
|
+
except Exception as exc:
|
|
287
|
+
record["skipped_reason"] = "body_unavailable"
|
|
288
|
+
warnings.append(f"网络响应正文未保存:{_short_error(exc)}")
|
|
289
|
+
network_records.append(record)
|
|
290
|
+
except Exception as exc:
|
|
291
|
+
warnings.append(f"网络响应记录失败:{_short_error(exc)}")
|
|
292
|
+
|
|
293
|
+
return handle
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def _suffix(content_type: str, url: str) -> str:
|
|
297
|
+
suffix = mimetypes.guess_extension(content_type) or Path(urlsplit(url).path).suffix
|
|
298
|
+
if suffix == ".jpe":
|
|
299
|
+
return ".jpg"
|
|
300
|
+
return suffix or ".bin"
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def _is_useful_image(body: bytes, url: str, content_type: str) -> bool:
|
|
304
|
+
"""Keep page artwork, but discard trackers and browser chrome assets."""
|
|
305
|
+
|
|
306
|
+
if NOISE_ASSET_RE.search(url):
|
|
307
|
+
return False
|
|
308
|
+
# Modao commonly serves small UI glyphs as standalone SVG files. They are
|
|
309
|
+
# already represented by the DOM and screenshots, so keeping them as
|
|
310
|
+
# binary evidence adds noise without improving an Agent's understanding of
|
|
311
|
+
# the prototype. Full/debug mode still retains the rendered HTML and DOM
|
|
312
|
+
# references needed to audit such elements.
|
|
313
|
+
if content_type == "image/svg+xml":
|
|
314
|
+
header = body[:4096].decode("utf-8", "ignore").lower()
|
|
315
|
+
if re.search(r"<svg[^>]*(?:class=[\"'][^\"']*\bicon\b|symbol|viewbox=[\"']0 0 1024 1024)", header):
|
|
316
|
+
return False
|
|
317
|
+
width = height = None
|
|
318
|
+
if content_type == "image/png" and len(body) >= 24 and body.startswith(b"\x89PNG"):
|
|
319
|
+
width = int.from_bytes(body[16:20], "big")
|
|
320
|
+
height = int.from_bytes(body[20:24], "big")
|
|
321
|
+
elif content_type == "image/gif" and len(body) >= 10 and body.startswith((b"GIF87a", b"GIF89a")):
|
|
322
|
+
width = int.from_bytes(body[6:8], "little")
|
|
323
|
+
height = int.from_bytes(body[8:10], "little")
|
|
324
|
+
if width is not None and height is not None and width <= 1 and height <= 1:
|
|
325
|
+
return False
|
|
326
|
+
return True
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def _capture_canvas_images(page: Any, add_artifact: Callable[..., None], *, state_id: str, warnings: list[str]) -> None:
|
|
330
|
+
"""Capture a readable copy of the actual prototype artboard.
|
|
331
|
+
|
|
332
|
+
Modao's inspect view may scale ``#canvas`` to 10% and its ``.mb-screen``
|
|
333
|
+
wrapper can therefore report a zero-sized box. Cloning the artboard into a
|
|
334
|
+
temporary overlay preserves the rendered CSS while allowing a readable,
|
|
335
|
+
bounded screenshot without clicking the UI zoom controls.
|
|
336
|
+
"""
|
|
337
|
+
|
|
338
|
+
target_selector = ".rResCanvas"
|
|
339
|
+
targets = page.locator(target_selector)
|
|
340
|
+
count = targets.count()
|
|
341
|
+
if not count:
|
|
342
|
+
target_selector = ".mb-screen"
|
|
343
|
+
targets = page.locator(target_selector)
|
|
344
|
+
count = targets.count()
|
|
345
|
+
for index in range(count):
|
|
346
|
+
try:
|
|
347
|
+
target = targets.nth(index)
|
|
348
|
+
box = target.bounding_box()
|
|
349
|
+
natural = target.evaluate("element => ({width: element.offsetWidth, height: element.offsetHeight})")
|
|
350
|
+
width = float(natural.get("width") or (box or {}).get("width", 0))
|
|
351
|
+
height = float(natural.get("height") or (box or {}).get("height", 0))
|
|
352
|
+
if width <= 1 or height <= 1:
|
|
353
|
+
warnings.append(f"画布节点不可见,未生成裁剪图(screen-{index + 1:03d})。")
|
|
354
|
+
continue
|
|
355
|
+
# Keep one native-resolution image for the complete artboard. The
|
|
356
|
+
# image can be very large, but it preserves the original layout and
|
|
357
|
+
# lets an Agent zoom into any region without stitching tiles back
|
|
358
|
+
# together. Text facts remain available in document.json/dom.json.
|
|
359
|
+
overview = _canvas_overlay_screenshot(
|
|
360
|
+
page,
|
|
361
|
+
target_selector=target_selector,
|
|
362
|
+
index=index,
|
|
363
|
+
scale=1.0,
|
|
364
|
+
offset_x=0,
|
|
365
|
+
offset_y=0,
|
|
366
|
+
tile_width=width,
|
|
367
|
+
tile_height=height,
|
|
368
|
+
)
|
|
369
|
+
if overview is not None:
|
|
370
|
+
add_artifact(
|
|
371
|
+
f"evidence/{state_id}.screen-{index + 1:03d}.png",
|
|
372
|
+
overview,
|
|
373
|
+
"image/png",
|
|
374
|
+
"screen_screenshot",
|
|
375
|
+
)
|
|
376
|
+
except Exception as exc:
|
|
377
|
+
warnings.append(f"画布截图未保存(screen-{index + 1:03d}):{_short_error(exc)}")
|
|
378
|
+
try:
|
|
379
|
+
page.evaluate("() => document.getElementById('modao-prd-capture-overlay')?.remove()")
|
|
380
|
+
except Exception:
|
|
381
|
+
pass
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def _canvas_overlay_screenshot(
|
|
385
|
+
page: Any,
|
|
386
|
+
*,
|
|
387
|
+
target_selector: str,
|
|
388
|
+
index: int,
|
|
389
|
+
scale: float,
|
|
390
|
+
offset_x: int,
|
|
391
|
+
offset_y: int,
|
|
392
|
+
tile_width: float,
|
|
393
|
+
tile_height: float,
|
|
394
|
+
) -> bytes | None:
|
|
395
|
+
"""Clone one artboard into a scaled viewport and screenshot it."""
|
|
396
|
+
|
|
397
|
+
tile_width = max(1, int(round(tile_width)))
|
|
398
|
+
tile_height = max(1, int(round(tile_height)))
|
|
399
|
+
result = page.evaluate(
|
|
400
|
+
"""({selector, index, scale, offsetX, offsetY, tileWidth, tileHeight}) => {
|
|
401
|
+
const nodes = Array.from(document.querySelectorAll(selector));
|
|
402
|
+
const source = nodes[index];
|
|
403
|
+
if (!source) return null;
|
|
404
|
+
const id = 'modao-prd-capture-overlay';
|
|
405
|
+
document.getElementById(id)?.remove();
|
|
406
|
+
const wrapper = document.createElement('div');
|
|
407
|
+
wrapper.id = id;
|
|
408
|
+
wrapper.style.cssText = `position:fixed;left:0;top:0;z-index:2147483647;overflow:hidden;background:#fff;width:${tileWidth * scale}px;height:${tileHeight * scale}px;`;
|
|
409
|
+
const clone = source.cloneNode(true);
|
|
410
|
+
clone.style.cssText += `;position:absolute;left:0;top:0;transform:translate(-${offsetX * scale}px, -${offsetY * scale}px) scale(${scale})!important;transform-origin:top left!important;width:${source.offsetWidth}px;height:${source.offsetHeight}px;`;
|
|
411
|
+
wrapper.appendChild(clone);
|
|
412
|
+
document.body.appendChild(wrapper);
|
|
413
|
+
return {id, width: wrapper.offsetWidth, height: wrapper.offsetHeight};
|
|
414
|
+
}""",
|
|
415
|
+
{
|
|
416
|
+
"selector": target_selector,
|
|
417
|
+
"index": index,
|
|
418
|
+
"scale": scale,
|
|
419
|
+
"offsetX": offset_x,
|
|
420
|
+
"offsetY": offset_y,
|
|
421
|
+
"tileWidth": tile_width,
|
|
422
|
+
"tileHeight": tile_height,
|
|
423
|
+
},
|
|
424
|
+
)
|
|
425
|
+
if not result:
|
|
426
|
+
return None
|
|
427
|
+
try:
|
|
428
|
+
return page.locator("#modao-prd-capture-overlay").screenshot(animations="disabled")
|
|
429
|
+
finally:
|
|
430
|
+
page.evaluate("() => document.getElementById('modao-prd-capture-overlay')?.remove()")
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
def sanitize_snapshot(value: Any, *, base_url: str) -> Any:
|
|
434
|
+
if isinstance(value, dict):
|
|
435
|
+
result = {}
|
|
436
|
+
for key, item in value.items():
|
|
437
|
+
if key in {"href", "src", "url"} and isinstance(item, str):
|
|
438
|
+
result[key] = redact_url(item, base_url=base_url)
|
|
439
|
+
elif key == "attributes" and isinstance(item, dict):
|
|
440
|
+
result[key] = {
|
|
441
|
+
attr: redact_url(attr_value, base_url=base_url)
|
|
442
|
+
if attr.casefold() in {"href", "src"} and isinstance(attr_value, str)
|
|
443
|
+
else sanitize_snapshot(attr_value, base_url=base_url)
|
|
444
|
+
for attr, attr_value in item.items()
|
|
445
|
+
}
|
|
446
|
+
else:
|
|
447
|
+
result[key] = sanitize_snapshot(item, base_url=base_url)
|
|
448
|
+
return result
|
|
449
|
+
if isinstance(value, list):
|
|
450
|
+
return [sanitize_snapshot(item, base_url=base_url) for item in value]
|
|
451
|
+
return value
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def _short_error(exc: Exception, limit: int = 300) -> str:
|
|
455
|
+
text = " ".join(str(exc).split())
|
|
456
|
+
return text[:limit] + ("…" if len(text) > limit else "")
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
DOM_LAYOUT_SCRIPT = r"""
|
|
460
|
+
() => {
|
|
461
|
+
const root = document.documentElement || document.body;
|
|
462
|
+
const nodes = [];
|
|
463
|
+
const visualOnly = [];
|
|
464
|
+
const warnings = [];
|
|
465
|
+
const screenNodes = Array.from(document.querySelectorAll('.mb-screen'));
|
|
466
|
+
const screens = screenNodes.length ? screenNodes : [document.body];
|
|
467
|
+
const normalize = value => (value || '').replace(/\u00a0/g, ' ').replace(/[ \t]+/g, ' ').replace(/\s*\n\s*/g, '\n').trim();
|
|
468
|
+
const visible = (element, style, rect) => style.display !== 'none' && style.visibility !== 'hidden' && Number(style.opacity || 1) > 0 && rect.width > 0 && rect.height > 0;
|
|
469
|
+
const bounds = element => {
|
|
470
|
+
const rect = element.getBoundingClientRect();
|
|
471
|
+
const owner = element.closest?.('.mb-screen') || screens[0];
|
|
472
|
+
const screenRect = owner?.getBoundingClientRect?.() || {x: 0, y: 0};
|
|
473
|
+
return {
|
|
474
|
+
viewport: {x: Math.round(rect.x), y: Math.round(rect.y), width: Math.round(rect.width), height: Math.round(rect.height)},
|
|
475
|
+
document: {x: Math.round(rect.x + window.scrollX), y: Math.round(rect.y + window.scrollY), width: Math.round(rect.width), height: Math.round(rect.height)},
|
|
476
|
+
screen: {x: Math.round(rect.x - screenRect.x), y: Math.round(rect.y - screenRect.y), width: Math.round(rect.width), height: Math.round(rect.height)}
|
|
477
|
+
};
|
|
478
|
+
};
|
|
479
|
+
const screenId = element => {
|
|
480
|
+
const screen = element.closest?.('.mb-screen');
|
|
481
|
+
const index = screen ? screens.indexOf(screen) : 0;
|
|
482
|
+
return `screen-${String(Math.max(index, 0) + 1).padStart(3, '0')}`;
|
|
483
|
+
};
|
|
484
|
+
const allowedAttrs = new Set(['id', 'class', 'role', 'title', 'alt', 'type', 'name', 'placeholder', 'aria-label', 'aria-selected', 'aria-expanded', 'aria-controls', 'data-testid', 'href', 'src']);
|
|
485
|
+
const ignoredTags = new Set(['script', 'style', 'link', 'meta', 'noscript', 'template']);
|
|
486
|
+
const visit = (element, parentId = null, frameId = null) => {
|
|
487
|
+
if (!element || element.nodeType !== Node.ELEMENT_NODE) return;
|
|
488
|
+
const tag = element.tagName.toLowerCase();
|
|
489
|
+
if (ignoredTags.has(tag)) return;
|
|
490
|
+
const id = `dom-${nodes.length + 1}`;
|
|
491
|
+
const style = getComputedStyle(element);
|
|
492
|
+
const rect = element.getBoundingClientRect();
|
|
493
|
+
const directText = normalize(Array.from(element.childNodes).filter(node => node.nodeType === Node.TEXT_NODE).map(node => node.textContent).join(' '));
|
|
494
|
+
const semanticText = element.matches('.wRichText,button,[role=button],[role=tab],summary,table,th,td,label');
|
|
495
|
+
const node = {
|
|
496
|
+
id,
|
|
497
|
+
parent_id: parentId,
|
|
498
|
+
children: [],
|
|
499
|
+
frame_id: frameId,
|
|
500
|
+
screen_id: screenId(element),
|
|
501
|
+
tag,
|
|
502
|
+
role: element.getAttribute('role') || null,
|
|
503
|
+
attributes: {},
|
|
504
|
+
direct_text: directText,
|
|
505
|
+
text: directText || semanticText || element.children.length === 0 ? normalize(element.innerText || element.textContent || '') : null,
|
|
506
|
+
visibility: {visible: visible(element, style, rect), display: style.display, visibility: style.visibility, opacity: style.opacity},
|
|
507
|
+
bounds: bounds(element),
|
|
508
|
+
style: {position: style.position, z_index: style.zIndex, overflow: style.overflow, color: style.color, background_color: style.backgroundColor, font_size: style.fontSize, font_weight: style.fontWeight, transform: style.transform},
|
|
509
|
+
special: null
|
|
510
|
+
};
|
|
511
|
+
for (const attr of Array.from(element.attributes)) {
|
|
512
|
+
if (allowedAttrs.has(attr.name)) node.attributes[attr.name] = String(attr.value).slice(0, 2048);
|
|
513
|
+
}
|
|
514
|
+
nodes.push(node);
|
|
515
|
+
if (parentId) {
|
|
516
|
+
const parent = nodes.find(item => item.id === parentId);
|
|
517
|
+
if (parent) parent.children.push(id);
|
|
518
|
+
}
|
|
519
|
+
if (tag === 'iframe') {
|
|
520
|
+
node.special = {kind: 'iframe', src: element.getAttribute('src') || ''};
|
|
521
|
+
try {
|
|
522
|
+
if (element.contentDocument?.documentElement) visit(element.contentDocument.documentElement, id, id);
|
|
523
|
+
} catch (error) {
|
|
524
|
+
warnings.push(`跨域 iframe 无法读取:${element.getAttribute('src') || '[unknown]'}`);
|
|
525
|
+
}
|
|
526
|
+
}
|
|
527
|
+
if (tag === 'canvas') {
|
|
528
|
+
node.special = {kind: 'canvas', width: element.width, height: element.height};
|
|
529
|
+
visualOnly.push({node_id: id, kind: 'canvas', bounds: node.bounds});
|
|
530
|
+
}
|
|
531
|
+
if (tag === 'svg') {
|
|
532
|
+
node.special = {kind: 'svg'};
|
|
533
|
+
if (!normalize(element.textContent)) visualOnly.push({node_id: id, kind: 'svg', bounds: node.bounds});
|
|
534
|
+
}
|
|
535
|
+
if (visible(element, style, rect) && style.backgroundImage && style.backgroundImage !== 'none') {
|
|
536
|
+
visualOnly.push({node_id: id, kind: 'background_image', bounds: node.bounds});
|
|
537
|
+
}
|
|
538
|
+
if (element.shadowRoot) {
|
|
539
|
+
node.special = {...(node.special || {}), shadow_root: 'open'};
|
|
540
|
+
for (const child of Array.from(element.shadowRoot.children)) visit(child, id, frameId);
|
|
541
|
+
}
|
|
542
|
+
for (const child of Array.from(element.children)) visit(child, id, frameId);
|
|
543
|
+
};
|
|
544
|
+
visit(root);
|
|
545
|
+
return {nodes, screens: screens.map((screen, index) => ({id: `screen-${String(index + 1).padStart(3, '0')}`, title: normalize(screen.querySelector?.('.canvas-title')?.textContent) || `页面 ${index + 1}`, bounds: bounds(screen)})), visual_only: visualOnly, warnings, frame_count: nodes.filter(node => node.special?.kind === 'iframe').length};
|
|
546
|
+
}
|
|
547
|
+
"""
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""Explainable Chinese text classification for prototype requirements."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
CLASSIFICATION_PATTERNS: list[tuple[str, tuple[str, ...]]] = [
|
|
10
|
+
("warning", ("注意", "谨慎", "禁止", "无关", "不可退款", "仅供娱乐")),
|
|
11
|
+
("validation", ("若", "如果", "低于", "不足", "超过", "达到上限", "不可继续", "提示")),
|
|
12
|
+
("persistence", ("不会重置", "不跟随", "保留", "整个活动期间有效", "退出活动清空")),
|
|
13
|
+
("limit", ("上限", "最多", "至少", "1-3", "超过", "最高")),
|
|
14
|
+
("timing", ("活动时间", "活动期间", "倒计时", "小时", "天", "有效期")),
|
|
15
|
+
("probability", ("概率", "随机", "几率", "%")),
|
|
16
|
+
("reward", ("奖励", "获得", "礼包", "礼物", "赠送", "奖池")),
|
|
17
|
+
("interaction", ("点击", "选择", "切换", "弹出", "关闭", "打开", "输入", "按键")),
|
|
18
|
+
]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def classify_blocks(blocks: list[dict[str, Any]]) -> tuple[list[dict[str, Any]], list[dict[str, Any]], list[str]]:
|
|
22
|
+
requirements: list[dict[str, Any]] = []
|
|
23
|
+
interactions: list[dict[str, Any]] = []
|
|
24
|
+
warnings: list[str] = []
|
|
25
|
+
|
|
26
|
+
for block in blocks:
|
|
27
|
+
if block.get("type") not in {"text", "heading", "label", "tab"}:
|
|
28
|
+
continue
|
|
29
|
+
text = str(block.get("text", "")).strip()
|
|
30
|
+
if not text:
|
|
31
|
+
continue
|
|
32
|
+
categories = [category for category, patterns in CLASSIFICATION_PATTERNS if any(pattern in text for pattern in patterns)]
|
|
33
|
+
if categories:
|
|
34
|
+
category = _select_primary_category(categories)
|
|
35
|
+
requirements.append(
|
|
36
|
+
{
|
|
37
|
+
"id": f"rule-{len(requirements) + 1:03d}",
|
|
38
|
+
"type": category,
|
|
39
|
+
"statement": text,
|
|
40
|
+
"conditions": _extract_conditions(text),
|
|
41
|
+
"feedback": _extract_feedback(text),
|
|
42
|
+
"source_block_ids": [block["id"]],
|
|
43
|
+
}
|
|
44
|
+
)
|
|
45
|
+
if "点击" in text or "选择" in text or "切换" in text or "弹出" in text:
|
|
46
|
+
interactions.append(
|
|
47
|
+
{
|
|
48
|
+
"id": f"interaction-{len(interactions) + 1:03d}",
|
|
49
|
+
"trigger": _extract_trigger(text),
|
|
50
|
+
"action": text,
|
|
51
|
+
"source_block_ids": [block["id"]],
|
|
52
|
+
}
|
|
53
|
+
)
|
|
54
|
+
if re.search(r"\b20\d{2}[./-]\d{1,2}[./-]\d{1,2}", text):
|
|
55
|
+
warnings.append("页面包含带具体日期的示例记录,请确认它们不是最终业务配置。")
|
|
56
|
+
if "谷歌文档" in text or "外部文档" in text:
|
|
57
|
+
warnings.append("页面引用了外部文档,外部文档内容未包含在本次提取结果中。")
|
|
58
|
+
|
|
59
|
+
if any(item["type"] == "probability" for item in requirements):
|
|
60
|
+
warnings.append("页面包含概率或随机规则,提取结果需要人工确认。")
|
|
61
|
+
return requirements, interactions, _unique(warnings)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _select_primary_category(categories: list[str]) -> str:
|
|
65
|
+
# Keep validation and interaction semantics visible; generic rewards come later.
|
|
66
|
+
priority = ["warning", "validation", "persistence", "limit", "timing", "probability", "reward", "interaction"]
|
|
67
|
+
return min(categories, key=priority.index)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _extract_conditions(text: str) -> list[str]:
|
|
71
|
+
conditions = re.findall(r"(?:若|如果|当|超过|低于|不足|达到)[^。;\n]*", text)
|
|
72
|
+
return [item.strip(" ,,::") for item in conditions[:5]]
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _extract_feedback(text: str) -> str | None:
|
|
76
|
+
match = re.search(r"(?:toast|提示|报错)\s*[::]\s*([^。;\n]+)", text, re.I)
|
|
77
|
+
return match.group(1).strip() if match else None
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _extract_trigger(text: str) -> str:
|
|
81
|
+
match = re.search(r"点击([^,。;\n]{0,24})", text)
|
|
82
|
+
if match:
|
|
83
|
+
return f"点击{match.group(1).strip()}".rstrip(",。")
|
|
84
|
+
match = re.search(r"(选择|切换|输入|弹出)[^,。;\n]{0,24}", text)
|
|
85
|
+
return match.group(0).strip() if match else "用户操作"
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _unique(values: list[str]) -> list[str]:
|
|
89
|
+
seen: set[str] = set()
|
|
90
|
+
result: list[str] = []
|
|
91
|
+
for value in values:
|
|
92
|
+
if value not in seen:
|
|
93
|
+
seen.add(value)
|
|
94
|
+
result.append(value)
|
|
95
|
+
return result
|
|
96
|
+
|