modao-prd-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. modao_prd_cli/__init__.py +1 -0
  2. modao_prd_cli/modao_prd/__init__.py +4 -0
  3. modao_prd_cli/modao_prd/__main__.py +7 -0
  4. modao_prd_cli/modao_prd/browser.py +578 -0
  5. modao_prd_cli/modao_prd/capture_evidence.py +547 -0
  6. modao_prd_cli/modao_prd/classifier.py +96 -0
  7. modao_prd_cli/modao_prd/cli.py +205 -0
  8. modao_prd_cli/modao_prd/errors.py +31 -0
  9. modao_prd_cli/modao_prd/evidence.py +207 -0
  10. modao_prd_cli/modao_prd/explorer.py +300 -0
  11. modao_prd_cli/modao_prd/extractor.py +465 -0
  12. modao_prd_cli/modao_prd/models.py +56 -0
  13. modao_prd_cli/modao_prd/normalizer.py +135 -0
  14. modao_prd_cli/modao_prd/schemas/coverage-1.0.json +15 -0
  15. modao_prd_cli/modao_prd/schemas/document-2.0.json +21 -0
  16. modao_prd_cli/modao_prd/schemas/document-2.1.json +31 -0
  17. modao_prd_cli/modao_prd/schemas/manifest-1.0.json +28 -0
  18. modao_prd_cli/modao_prd/tests/__init__.py +1 -0
  19. modao_prd_cli/modao_prd/tests/fixtures/modao_sample.html +20 -0
  20. modao_prd_cli/modao_prd/tests/test_browser.py +54 -0
  21. modao_prd_cli/modao_prd/tests/test_classifier.py +32 -0
  22. modao_prd_cli/modao_prd/tests/test_cli.py +88 -0
  23. modao_prd_cli/modao_prd/tests/test_extractor.py +57 -0
  24. modao_prd_cli/modao_prd/tests/test_full_e2e.py +23 -0
  25. modao_prd_cli/modao_prd/tests/test_writers.py +79 -0
  26. modao_prd_cli/modao_prd/writers.py +468 -0
  27. modao_prd_cli-0.1.0.dist-info/METADATA +108 -0
  28. modao_prd_cli-0.1.0.dist-info/RECORD +32 -0
  29. modao_prd_cli-0.1.0.dist-info/WHEEL +5 -0
  30. modao_prd_cli-0.1.0.dist-info/entry_points.txt +2 -0
  31. modao_prd_cli-0.1.0.dist-info/licenses/LICENSE +22 -0
  32. modao_prd_cli-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,547 @@
1
+ """Evidence capture helpers for rendered Modao pages.
2
+
3
+ The browser layer keeps the original DOM records for the existing extractor and
4
+ adds a loss-aware evidence snapshot beside them. The snapshot is deliberately
5
+ mechanical: it records what the browser rendered and does not infer product
6
+ meaning.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import hashlib
12
+ import json
13
+ import mimetypes
14
+ import re
15
+ from dataclasses import asdict
16
+ from pathlib import Path
17
+ from typing import Any, Callable
18
+ from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit
19
+
20
+ from .evidence import EvidenceBundleWriter
21
+
22
+
23
+ SENSITIVE_QUERY_KEYS = {
24
+ "access_token",
25
+ "authorization",
26
+ "code",
27
+ "key",
28
+ "secret",
29
+ "session",
30
+ "sig",
31
+ "signature",
32
+ "token",
33
+ "accesstoken",
34
+ "refreshtoken",
35
+ "clientsecret",
36
+ "jwt",
37
+ "auth",
38
+ "ticket",
39
+ "nonce",
40
+ "sessionid",
41
+ }
42
+ DEBUG_TEXT_CONTENT_TYPES = {
43
+ "application/json",
44
+ "application/manifest+json",
45
+ "application/xml",
46
+ "text/html",
47
+ "text/plain",
48
+ }
49
+ NOISE_ASSET_RE = re.compile(r"(?:favicon|loading|logo|spinner|tracker|tracking|pixel|analytics|beacon|collect)", re.I)
50
+
51
+
52
+ def redact_url(value: str, *, base_url: str | None = None) -> str:
53
+ """Remove credential-like query values and optionally reduce cross-origin URLs."""
54
+
55
+ try:
56
+ parsed = urlsplit(value)
57
+ except ValueError:
58
+ return "[invalid-url]"
59
+ pairs = [
60
+ (key, val)
61
+ for key, val in parse_qsl(parsed.query, keep_blank_values=True)
62
+ if key.casefold() not in SENSITIVE_QUERY_KEYS
63
+ ]
64
+ query = urlencode(pairs)
65
+ sanitized = urlunsplit((parsed.scheme, parsed.netloc, parsed.path, query, ""))
66
+ if base_url and not same_origin(sanitized, base_url):
67
+ return urlunsplit((parsed.scheme, parsed.netloc, parsed.path, "", ""))
68
+ return sanitized
69
+
70
+
71
+ def redact_text(value: str) -> str:
72
+ """Remove credential material from debug text without rewriting product copy."""
73
+
74
+ value = re.sub(r"(?i)(authorization\s*[:=]\s*bearer\s+)[^\s,;\"']+", r"\1[REDACTED]", value)
75
+ value = re.sub(r"(?i)((?:cookie|set-cookie)\s*[:=]\s*)[^\n]+", r"\1[REDACTED]", value)
76
+ value = re.sub(
77
+ r"(?i)([?&](?:access[_-]?token|refresh[_-]?token|client[_-]?secret|authorization|signature|token|jwt|ticket)=)[^&#\s\"']+",
78
+ r"\1[REDACTED]",
79
+ value,
80
+ )
81
+ return value
82
+
83
+
84
+ def same_origin(left: str, right: str) -> bool:
85
+ try:
86
+ a = urlsplit(left)
87
+ b = urlsplit(right)
88
+ except ValueError:
89
+ return False
90
+ return (a.scheme.lower(), a.hostname, a.port or _default_port(a.scheme)) == (
91
+ b.scheme.lower(),
92
+ b.hostname,
93
+ b.port or _default_port(b.scheme),
94
+ )
95
+
96
+
97
+ def merge_evidence(existing: list[dict[str, Any]], incoming: list[dict[str, Any]]) -> None:
98
+ """Append evidence records while keeping IDs unique.
99
+
100
+ Network response callbacks can add assets while a state screenshot is
101
+ being captured. The old offset-based IDs were therefore able to collide
102
+ with a newly observed asset, causing a real file to be present in the
103
+ bundle without a reference in ``document.json``.
104
+ """
105
+
106
+ used = {str(item.get("id")) for item in existing if item.get("id")}
107
+ next_number = 1
108
+ for value in used:
109
+ match = re.fullmatch(r"evidence-(\d+)", value)
110
+ if match:
111
+ next_number = max(next_number, int(match.group(1)) + 1)
112
+ for item in incoming:
113
+ record = dict(item)
114
+ evidence_id = str(record.get("id") or "")
115
+ if not evidence_id or evidence_id in used:
116
+ while f"evidence-{next_number:03d}" in used:
117
+ next_number += 1
118
+ evidence_id = f"evidence-{next_number:03d}"
119
+ next_number += 1
120
+ record["id"] = evidence_id
121
+ used.add(evidence_id)
122
+ existing.append(record)
123
+
124
+
125
+ def _default_port(scheme: str) -> int | None:
126
+ return {"http": 80, "https": 443}.get(scheme.lower())
127
+
128
+
129
+ def capture_state_evidence(
130
+ page: Any,
131
+ context: Any,
132
+ *,
133
+ writer: EvidenceBundleWriter | None,
134
+ state_id: str,
135
+ requested_url: str,
136
+ evidence_mode: str,
137
+ max_item: int,
138
+ max_total: int,
139
+ warnings: list[str],
140
+ network_records: list[dict[str, Any]],
141
+ evidence_id_offset: int = 0,
142
+ ) -> tuple[list[dict[str, Any]], dict[str, Any], dict[str, Any]]:
143
+ """Capture rendered page artifacts and return refs, DOM facts and stats."""
144
+
145
+ evidence: list[dict[str, Any]] = []
146
+ snapshot: dict[str, Any] = {}
147
+
148
+ def add_artifact(relative_path: str, content: Any, media_type: str, kind: str) -> None:
149
+ if writer is None:
150
+ return
151
+ try:
152
+ item = writer.add(relative_path, content, media_type=media_type)
153
+ except Exception as exc:
154
+ warnings.append(f"证据文件未保存({kind}):{_short_error(exc)}")
155
+ return
156
+ evidence.append({"id": f"evidence-{evidence_id_offset + len(evidence) + 1:03d}", "kind": kind, "state_id": state_id, **asdict(item)})
157
+
158
+ if evidence_mode == "full":
159
+ try:
160
+ rendered_html = redact_text(page.content())
161
+ add_artifact(f"evidence/{state_id}.rendered.html", rendered_html, "text/html", "html")
162
+ except Exception as exc:
163
+ warnings.append(f"无法保存渲染后 HTML:{_short_error(exc)}")
164
+
165
+ try:
166
+ snapshot = page.evaluate(DOM_LAYOUT_SCRIPT)
167
+ snapshot = sanitize_snapshot(snapshot, base_url=requested_url)
168
+ warnings.extend(str(item) for item in snapshot.get("warnings", []) if item)
169
+ add_artifact(
170
+ f"evidence/{state_id}.dom.json",
171
+ json.dumps(snapshot, ensure_ascii=False, indent=2) + "\n",
172
+ "application/json",
173
+ "dom",
174
+ )
175
+ except Exception as exc:
176
+ warnings.append(f"无法保存 DOM 布局快照:{_short_error(exc)}")
177
+
178
+ if evidence_mode == "full":
179
+ try:
180
+ try:
181
+ aria = page.aria_snapshot(mode="ai", boxes=True)
182
+ except TypeError:
183
+ aria = page.aria_snapshot()
184
+ add_artifact(f"evidence/{state_id}.aria.yaml", redact_text(aria), "text/yaml", "aria")
185
+ except Exception as exc:
186
+ warnings.append(f"无法保存 ARIA 快照:{_short_error(exc)}")
187
+ if evidence_mode == "full":
188
+ try:
189
+ screenshot = page.screenshot(full_page=True, animations="disabled")
190
+ add_artifact(f"evidence/{state_id}.full.png", screenshot, "image/png", "screenshot")
191
+ except Exception as exc:
192
+ warnings.append(f"无法保存页面截图:{_short_error(exc)}")
193
+ if evidence_mode in {"essential", "full"}:
194
+ try:
195
+ _capture_canvas_images(page, add_artifact, state_id=state_id, warnings=warnings)
196
+ except Exception as exc:
197
+ warnings.append(f"无法枚举画布截图:{_short_error(exc)}")
198
+
199
+ stats = {
200
+ "dom_node_count": len(snapshot.get("nodes", [])),
201
+ "visual_only_count": len(snapshot.get("visual_only", [])),
202
+ "frame_count": snapshot.get("frame_count", 0),
203
+ "evidence_count": len(evidence),
204
+ "network_count": len(network_records),
205
+ "max_item": max_item,
206
+ "max_total": max_total,
207
+ }
208
+ return evidence, snapshot, stats
209
+
210
+
211
+ def make_response_handler(
212
+ *,
213
+ writer: EvidenceBundleWriter | None,
214
+ requested_url: str,
215
+ state_id: str,
216
+ evidence: list[dict[str, Any]],
217
+ network_records: list[dict[str, Any]],
218
+ warnings: list[str],
219
+ max_item: int,
220
+ max_total: int,
221
+ evidence_mode: str = "full",
222
+ ) -> Callable[[Any], None]:
223
+ """Create a safe Playwright response callback for one capture run."""
224
+
225
+ def handle(response: Any) -> None:
226
+ try:
227
+ url = str(response.url)
228
+ if evidence_mode == "none":
229
+ return
230
+ response_same_origin = same_origin(url, requested_url)
231
+ if evidence_mode == "essential" and not response_same_origin:
232
+ return
233
+ headers = getattr(response, "headers", {}) or {}
234
+ content_type = str(headers.get("content-type", "")).split(";", 1)[0].strip().lower()
235
+ record: dict[str, Any] = {
236
+ "state_id": state_id,
237
+ "url": redact_url(url, base_url=requested_url),
238
+ "same_origin": response_same_origin,
239
+ "status": int(response.status),
240
+ "ok": bool(response.ok),
241
+ "content_type": content_type,
242
+ "headers": {
243
+ key: str(headers[key])
244
+ for key in ("content-type", "content-length", "etag", "last-modified", "cache-control")
245
+ if key in headers
246
+ },
247
+ }
248
+ is_image = content_type.startswith("image/")
249
+ is_debug_document = same_origin(url, requested_url) and content_type in DEBUG_TEXT_CONTENT_TYPES
250
+ should_capture_body = bool(writer and evidence_mode != "none")
251
+ should_capture_body = should_capture_body and (
252
+ (is_image and same_origin(url, requested_url))
253
+ or (evidence_mode == "full" and is_debug_document)
254
+ )
255
+ if should_capture_body:
256
+ try:
257
+ body = response.body()
258
+ if content_type in DEBUG_TEXT_CONTENT_TYPES:
259
+ body = redact_text(body.decode("utf-8", "replace")).encode("utf-8")
260
+ if is_image and not _is_useful_image(body, url, content_type):
261
+ record["skipped_reason"] = "irrelevant_or_tiny_asset"
262
+ elif not body:
263
+ record["skipped_reason"] = "empty_body"
264
+ elif len(body) <= max_item:
265
+ digest = hashlib.sha256(body).hexdigest()
266
+ suffix = _suffix(content_type, url)
267
+ directory = "evidence/assets" if is_image else "evidence/network/bodies"
268
+ relative = f"{directory}/{digest}{suffix}"
269
+ if writer:
270
+ existing = next((item for item in evidence if item.get("path") == relative), None)
271
+ if existing:
272
+ record["evidence_id"] = existing["id"]
273
+ else:
274
+ item = writer.add(relative, body, media_type=content_type or "application/octet-stream")
275
+ evidence.append(
276
+ {
277
+ "id": f"evidence-{len(evidence) + 1:03d}",
278
+ "kind": "asset" if is_image else "network_body",
279
+ "state_id": state_id,
280
+ **asdict(item),
281
+ }
282
+ )
283
+ record["evidence_id"] = evidence[-1]["id"]
284
+ else:
285
+ record["skipped_reason"] = "item_size_limit"
286
+ except Exception as exc:
287
+ record["skipped_reason"] = "body_unavailable"
288
+ warnings.append(f"网络响应正文未保存:{_short_error(exc)}")
289
+ network_records.append(record)
290
+ except Exception as exc:
291
+ warnings.append(f"网络响应记录失败:{_short_error(exc)}")
292
+
293
+ return handle
294
+
295
+
296
+ def _suffix(content_type: str, url: str) -> str:
297
+ suffix = mimetypes.guess_extension(content_type) or Path(urlsplit(url).path).suffix
298
+ if suffix == ".jpe":
299
+ return ".jpg"
300
+ return suffix or ".bin"
301
+
302
+
303
+ def _is_useful_image(body: bytes, url: str, content_type: str) -> bool:
304
+ """Keep page artwork, but discard trackers and browser chrome assets."""
305
+
306
+ if NOISE_ASSET_RE.search(url):
307
+ return False
308
+ # Modao commonly serves small UI glyphs as standalone SVG files. They are
309
+ # already represented by the DOM and screenshots, so keeping them as
310
+ # binary evidence adds noise without improving an Agent's understanding of
311
+ # the prototype. Full/debug mode still retains the rendered HTML and DOM
312
+ # references needed to audit such elements.
313
+ if content_type == "image/svg+xml":
314
+ header = body[:4096].decode("utf-8", "ignore").lower()
315
+ if re.search(r"<svg[^>]*(?:class=[\"'][^\"']*\bicon\b|symbol|viewbox=[\"']0 0 1024 1024)", header):
316
+ return False
317
+ width = height = None
318
+ if content_type == "image/png" and len(body) >= 24 and body.startswith(b"\x89PNG"):
319
+ width = int.from_bytes(body[16:20], "big")
320
+ height = int.from_bytes(body[20:24], "big")
321
+ elif content_type == "image/gif" and len(body) >= 10 and body.startswith((b"GIF87a", b"GIF89a")):
322
+ width = int.from_bytes(body[6:8], "little")
323
+ height = int.from_bytes(body[8:10], "little")
324
+ if width is not None and height is not None and width <= 1 and height <= 1:
325
+ return False
326
+ return True
327
+
328
+
329
+ def _capture_canvas_images(page: Any, add_artifact: Callable[..., None], *, state_id: str, warnings: list[str]) -> None:
330
+ """Capture a readable copy of the actual prototype artboard.
331
+
332
+ Modao's inspect view may scale ``#canvas`` to 10% and its ``.mb-screen``
333
+ wrapper can therefore report a zero-sized box. Cloning the artboard into a
334
+ temporary overlay preserves the rendered CSS while allowing a readable,
335
+ bounded screenshot without clicking the UI zoom controls.
336
+ """
337
+
338
+ target_selector = ".rResCanvas"
339
+ targets = page.locator(target_selector)
340
+ count = targets.count()
341
+ if not count:
342
+ target_selector = ".mb-screen"
343
+ targets = page.locator(target_selector)
344
+ count = targets.count()
345
+ for index in range(count):
346
+ try:
347
+ target = targets.nth(index)
348
+ box = target.bounding_box()
349
+ natural = target.evaluate("element => ({width: element.offsetWidth, height: element.offsetHeight})")
350
+ width = float(natural.get("width") or (box or {}).get("width", 0))
351
+ height = float(natural.get("height") or (box or {}).get("height", 0))
352
+ if width <= 1 or height <= 1:
353
+ warnings.append(f"画布节点不可见,未生成裁剪图(screen-{index + 1:03d})。")
354
+ continue
355
+ # Keep one native-resolution image for the complete artboard. The
356
+ # image can be very large, but it preserves the original layout and
357
+ # lets an Agent zoom into any region without stitching tiles back
358
+ # together. Text facts remain available in document.json/dom.json.
359
+ overview = _canvas_overlay_screenshot(
360
+ page,
361
+ target_selector=target_selector,
362
+ index=index,
363
+ scale=1.0,
364
+ offset_x=0,
365
+ offset_y=0,
366
+ tile_width=width,
367
+ tile_height=height,
368
+ )
369
+ if overview is not None:
370
+ add_artifact(
371
+ f"evidence/{state_id}.screen-{index + 1:03d}.png",
372
+ overview,
373
+ "image/png",
374
+ "screen_screenshot",
375
+ )
376
+ except Exception as exc:
377
+ warnings.append(f"画布截图未保存(screen-{index + 1:03d}):{_short_error(exc)}")
378
+ try:
379
+ page.evaluate("() => document.getElementById('modao-prd-capture-overlay')?.remove()")
380
+ except Exception:
381
+ pass
382
+
383
+
384
+ def _canvas_overlay_screenshot(
385
+ page: Any,
386
+ *,
387
+ target_selector: str,
388
+ index: int,
389
+ scale: float,
390
+ offset_x: int,
391
+ offset_y: int,
392
+ tile_width: float,
393
+ tile_height: float,
394
+ ) -> bytes | None:
395
+ """Clone one artboard into a scaled viewport and screenshot it."""
396
+
397
+ tile_width = max(1, int(round(tile_width)))
398
+ tile_height = max(1, int(round(tile_height)))
399
+ result = page.evaluate(
400
+ """({selector, index, scale, offsetX, offsetY, tileWidth, tileHeight}) => {
401
+ const nodes = Array.from(document.querySelectorAll(selector));
402
+ const source = nodes[index];
403
+ if (!source) return null;
404
+ const id = 'modao-prd-capture-overlay';
405
+ document.getElementById(id)?.remove();
406
+ const wrapper = document.createElement('div');
407
+ wrapper.id = id;
408
+ wrapper.style.cssText = `position:fixed;left:0;top:0;z-index:2147483647;overflow:hidden;background:#fff;width:${tileWidth * scale}px;height:${tileHeight * scale}px;`;
409
+ const clone = source.cloneNode(true);
410
+ clone.style.cssText += `;position:absolute;left:0;top:0;transform:translate(-${offsetX * scale}px, -${offsetY * scale}px) scale(${scale})!important;transform-origin:top left!important;width:${source.offsetWidth}px;height:${source.offsetHeight}px;`;
411
+ wrapper.appendChild(clone);
412
+ document.body.appendChild(wrapper);
413
+ return {id, width: wrapper.offsetWidth, height: wrapper.offsetHeight};
414
+ }""",
415
+ {
416
+ "selector": target_selector,
417
+ "index": index,
418
+ "scale": scale,
419
+ "offsetX": offset_x,
420
+ "offsetY": offset_y,
421
+ "tileWidth": tile_width,
422
+ "tileHeight": tile_height,
423
+ },
424
+ )
425
+ if not result:
426
+ return None
427
+ try:
428
+ return page.locator("#modao-prd-capture-overlay").screenshot(animations="disabled")
429
+ finally:
430
+ page.evaluate("() => document.getElementById('modao-prd-capture-overlay')?.remove()")
431
+
432
+
433
+ def sanitize_snapshot(value: Any, *, base_url: str) -> Any:
434
+ if isinstance(value, dict):
435
+ result = {}
436
+ for key, item in value.items():
437
+ if key in {"href", "src", "url"} and isinstance(item, str):
438
+ result[key] = redact_url(item, base_url=base_url)
439
+ elif key == "attributes" and isinstance(item, dict):
440
+ result[key] = {
441
+ attr: redact_url(attr_value, base_url=base_url)
442
+ if attr.casefold() in {"href", "src"} and isinstance(attr_value, str)
443
+ else sanitize_snapshot(attr_value, base_url=base_url)
444
+ for attr, attr_value in item.items()
445
+ }
446
+ else:
447
+ result[key] = sanitize_snapshot(item, base_url=base_url)
448
+ return result
449
+ if isinstance(value, list):
450
+ return [sanitize_snapshot(item, base_url=base_url) for item in value]
451
+ return value
452
+
453
+
454
+ def _short_error(exc: Exception, limit: int = 300) -> str:
455
+ text = " ".join(str(exc).split())
456
+ return text[:limit] + ("…" if len(text) > limit else "")
457
+
458
+
459
+ DOM_LAYOUT_SCRIPT = r"""
460
+ () => {
461
+ const root = document.documentElement || document.body;
462
+ const nodes = [];
463
+ const visualOnly = [];
464
+ const warnings = [];
465
+ const screenNodes = Array.from(document.querySelectorAll('.mb-screen'));
466
+ const screens = screenNodes.length ? screenNodes : [document.body];
467
+ const normalize = value => (value || '').replace(/\u00a0/g, ' ').replace(/[ \t]+/g, ' ').replace(/\s*\n\s*/g, '\n').trim();
468
+ const visible = (element, style, rect) => style.display !== 'none' && style.visibility !== 'hidden' && Number(style.opacity || 1) > 0 && rect.width > 0 && rect.height > 0;
469
+ const bounds = element => {
470
+ const rect = element.getBoundingClientRect();
471
+ const owner = element.closest?.('.mb-screen') || screens[0];
472
+ const screenRect = owner?.getBoundingClientRect?.() || {x: 0, y: 0};
473
+ return {
474
+ viewport: {x: Math.round(rect.x), y: Math.round(rect.y), width: Math.round(rect.width), height: Math.round(rect.height)},
475
+ document: {x: Math.round(rect.x + window.scrollX), y: Math.round(rect.y + window.scrollY), width: Math.round(rect.width), height: Math.round(rect.height)},
476
+ screen: {x: Math.round(rect.x - screenRect.x), y: Math.round(rect.y - screenRect.y), width: Math.round(rect.width), height: Math.round(rect.height)}
477
+ };
478
+ };
479
+ const screenId = element => {
480
+ const screen = element.closest?.('.mb-screen');
481
+ const index = screen ? screens.indexOf(screen) : 0;
482
+ return `screen-${String(Math.max(index, 0) + 1).padStart(3, '0')}`;
483
+ };
484
+ const allowedAttrs = new Set(['id', 'class', 'role', 'title', 'alt', 'type', 'name', 'placeholder', 'aria-label', 'aria-selected', 'aria-expanded', 'aria-controls', 'data-testid', 'href', 'src']);
485
+ const ignoredTags = new Set(['script', 'style', 'link', 'meta', 'noscript', 'template']);
486
+ const visit = (element, parentId = null, frameId = null) => {
487
+ if (!element || element.nodeType !== Node.ELEMENT_NODE) return;
488
+ const tag = element.tagName.toLowerCase();
489
+ if (ignoredTags.has(tag)) return;
490
+ const id = `dom-${nodes.length + 1}`;
491
+ const style = getComputedStyle(element);
492
+ const rect = element.getBoundingClientRect();
493
+ const directText = normalize(Array.from(element.childNodes).filter(node => node.nodeType === Node.TEXT_NODE).map(node => node.textContent).join(' '));
494
+ const semanticText = element.matches('.wRichText,button,[role=button],[role=tab],summary,table,th,td,label');
495
+ const node = {
496
+ id,
497
+ parent_id: parentId,
498
+ children: [],
499
+ frame_id: frameId,
500
+ screen_id: screenId(element),
501
+ tag,
502
+ role: element.getAttribute('role') || null,
503
+ attributes: {},
504
+ direct_text: directText,
505
+ text: directText || semanticText || element.children.length === 0 ? normalize(element.innerText || element.textContent || '') : null,
506
+ visibility: {visible: visible(element, style, rect), display: style.display, visibility: style.visibility, opacity: style.opacity},
507
+ bounds: bounds(element),
508
+ style: {position: style.position, z_index: style.zIndex, overflow: style.overflow, color: style.color, background_color: style.backgroundColor, font_size: style.fontSize, font_weight: style.fontWeight, transform: style.transform},
509
+ special: null
510
+ };
511
+ for (const attr of Array.from(element.attributes)) {
512
+ if (allowedAttrs.has(attr.name)) node.attributes[attr.name] = String(attr.value).slice(0, 2048);
513
+ }
514
+ nodes.push(node);
515
+ if (parentId) {
516
+ const parent = nodes.find(item => item.id === parentId);
517
+ if (parent) parent.children.push(id);
518
+ }
519
+ if (tag === 'iframe') {
520
+ node.special = {kind: 'iframe', src: element.getAttribute('src') || ''};
521
+ try {
522
+ if (element.contentDocument?.documentElement) visit(element.contentDocument.documentElement, id, id);
523
+ } catch (error) {
524
+ warnings.push(`跨域 iframe 无法读取:${element.getAttribute('src') || '[unknown]'}`);
525
+ }
526
+ }
527
+ if (tag === 'canvas') {
528
+ node.special = {kind: 'canvas', width: element.width, height: element.height};
529
+ visualOnly.push({node_id: id, kind: 'canvas', bounds: node.bounds});
530
+ }
531
+ if (tag === 'svg') {
532
+ node.special = {kind: 'svg'};
533
+ if (!normalize(element.textContent)) visualOnly.push({node_id: id, kind: 'svg', bounds: node.bounds});
534
+ }
535
+ if (visible(element, style, rect) && style.backgroundImage && style.backgroundImage !== 'none') {
536
+ visualOnly.push({node_id: id, kind: 'background_image', bounds: node.bounds});
537
+ }
538
+ if (element.shadowRoot) {
539
+ node.special = {...(node.special || {}), shadow_root: 'open'};
540
+ for (const child of Array.from(element.shadowRoot.children)) visit(child, id, frameId);
541
+ }
542
+ for (const child of Array.from(element.children)) visit(child, id, frameId);
543
+ };
544
+ visit(root);
545
+ return {nodes, screens: screens.map((screen, index) => ({id: `screen-${String(index + 1).padStart(3, '0')}`, title: normalize(screen.querySelector?.('.canvas-title')?.textContent) || `页面 ${index + 1}`, bounds: bounds(screen)})), visual_only: visualOnly, warnings, frame_count: nodes.filter(node => node.special?.kind === 'iframe').length};
546
+ }
547
+ """
@@ -0,0 +1,96 @@
1
+ """Explainable Chinese text classification for prototype requirements."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from typing import Any
7
+
8
+
9
+ CLASSIFICATION_PATTERNS: list[tuple[str, tuple[str, ...]]] = [
10
+ ("warning", ("注意", "谨慎", "禁止", "无关", "不可退款", "仅供娱乐")),
11
+ ("validation", ("若", "如果", "低于", "不足", "超过", "达到上限", "不可继续", "提示")),
12
+ ("persistence", ("不会重置", "不跟随", "保留", "整个活动期间有效", "退出活动清空")),
13
+ ("limit", ("上限", "最多", "至少", "1-3", "超过", "最高")),
14
+ ("timing", ("活动时间", "活动期间", "倒计时", "小时", "天", "有效期")),
15
+ ("probability", ("概率", "随机", "几率", "%")),
16
+ ("reward", ("奖励", "获得", "礼包", "礼物", "赠送", "奖池")),
17
+ ("interaction", ("点击", "选择", "切换", "弹出", "关闭", "打开", "输入", "按键")),
18
+ ]
19
+
20
+
21
+ def classify_blocks(blocks: list[dict[str, Any]]) -> tuple[list[dict[str, Any]], list[dict[str, Any]], list[str]]:
22
+ requirements: list[dict[str, Any]] = []
23
+ interactions: list[dict[str, Any]] = []
24
+ warnings: list[str] = []
25
+
26
+ for block in blocks:
27
+ if block.get("type") not in {"text", "heading", "label", "tab"}:
28
+ continue
29
+ text = str(block.get("text", "")).strip()
30
+ if not text:
31
+ continue
32
+ categories = [category for category, patterns in CLASSIFICATION_PATTERNS if any(pattern in text for pattern in patterns)]
33
+ if categories:
34
+ category = _select_primary_category(categories)
35
+ requirements.append(
36
+ {
37
+ "id": f"rule-{len(requirements) + 1:03d}",
38
+ "type": category,
39
+ "statement": text,
40
+ "conditions": _extract_conditions(text),
41
+ "feedback": _extract_feedback(text),
42
+ "source_block_ids": [block["id"]],
43
+ }
44
+ )
45
+ if "点击" in text or "选择" in text or "切换" in text or "弹出" in text:
46
+ interactions.append(
47
+ {
48
+ "id": f"interaction-{len(interactions) + 1:03d}",
49
+ "trigger": _extract_trigger(text),
50
+ "action": text,
51
+ "source_block_ids": [block["id"]],
52
+ }
53
+ )
54
+ if re.search(r"\b20\d{2}[./-]\d{1,2}[./-]\d{1,2}", text):
55
+ warnings.append("页面包含带具体日期的示例记录,请确认它们不是最终业务配置。")
56
+ if "谷歌文档" in text or "外部文档" in text:
57
+ warnings.append("页面引用了外部文档,外部文档内容未包含在本次提取结果中。")
58
+
59
+ if any(item["type"] == "probability" for item in requirements):
60
+ warnings.append("页面包含概率或随机规则,提取结果需要人工确认。")
61
+ return requirements, interactions, _unique(warnings)
62
+
63
+
64
+ def _select_primary_category(categories: list[str]) -> str:
65
+ # Keep validation and interaction semantics visible; generic rewards come later.
66
+ priority = ["warning", "validation", "persistence", "limit", "timing", "probability", "reward", "interaction"]
67
+ return min(categories, key=priority.index)
68
+
69
+
70
+ def _extract_conditions(text: str) -> list[str]:
71
+ conditions = re.findall(r"(?:若|如果|当|超过|低于|不足|达到)[^。;\n]*", text)
72
+ return [item.strip(" ,,::") for item in conditions[:5]]
73
+
74
+
75
+ def _extract_feedback(text: str) -> str | None:
76
+ match = re.search(r"(?:toast|提示|报错)\s*[::]\s*([^。;\n]+)", text, re.I)
77
+ return match.group(1).strip() if match else None
78
+
79
+
80
+ def _extract_trigger(text: str) -> str:
81
+ match = re.search(r"点击([^,。;\n]{0,24})", text)
82
+ if match:
83
+ return f"点击{match.group(1).strip()}".rstrip(",。")
84
+ match = re.search(r"(选择|切换|输入|弹出)[^,。;\n]{0,24}", text)
85
+ return match.group(0).strip() if match else "用户操作"
86
+
87
+
88
+ def _unique(values: list[str]) -> list[str]:
89
+ seen: set[str] = set()
90
+ result: list[str] = []
91
+ for value in values:
92
+ if value not in seen:
93
+ seen.add(value)
94
+ result.append(value)
95
+ return result
96
+