modao-prd-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. modao_prd_cli/__init__.py +1 -0
  2. modao_prd_cli/modao_prd/__init__.py +4 -0
  3. modao_prd_cli/modao_prd/__main__.py +7 -0
  4. modao_prd_cli/modao_prd/browser.py +578 -0
  5. modao_prd_cli/modao_prd/capture_evidence.py +547 -0
  6. modao_prd_cli/modao_prd/classifier.py +96 -0
  7. modao_prd_cli/modao_prd/cli.py +205 -0
  8. modao_prd_cli/modao_prd/errors.py +31 -0
  9. modao_prd_cli/modao_prd/evidence.py +207 -0
  10. modao_prd_cli/modao_prd/explorer.py +300 -0
  11. modao_prd_cli/modao_prd/extractor.py +465 -0
  12. modao_prd_cli/modao_prd/models.py +56 -0
  13. modao_prd_cli/modao_prd/normalizer.py +135 -0
  14. modao_prd_cli/modao_prd/schemas/coverage-1.0.json +15 -0
  15. modao_prd_cli/modao_prd/schemas/document-2.0.json +21 -0
  16. modao_prd_cli/modao_prd/schemas/document-2.1.json +31 -0
  17. modao_prd_cli/modao_prd/schemas/manifest-1.0.json +28 -0
  18. modao_prd_cli/modao_prd/tests/__init__.py +1 -0
  19. modao_prd_cli/modao_prd/tests/fixtures/modao_sample.html +20 -0
  20. modao_prd_cli/modao_prd/tests/test_browser.py +54 -0
  21. modao_prd_cli/modao_prd/tests/test_classifier.py +32 -0
  22. modao_prd_cli/modao_prd/tests/test_cli.py +88 -0
  23. modao_prd_cli/modao_prd/tests/test_extractor.py +57 -0
  24. modao_prd_cli/modao_prd/tests/test_full_e2e.py +23 -0
  25. modao_prd_cli/modao_prd/tests/test_writers.py +79 -0
  26. modao_prd_cli/modao_prd/writers.py +468 -0
  27. modao_prd_cli-0.1.0.dist-info/METADATA +108 -0
  28. modao_prd_cli-0.1.0.dist-info/RECORD +32 -0
  29. modao_prd_cli-0.1.0.dist-info/WHEEL +5 -0
  30. modao_prd_cli-0.1.0.dist-info/entry_points.txt +2 -0
  31. modao_prd_cli-0.1.0.dist-info/licenses/LICENSE +22 -0
  32. modao_prd_cli-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,300 @@
1
+ """Bounded, conservative exploration of public prototype states."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import re
7
+ import time
8
+ from typing import Any
9
+
10
+
11
+ HIGH_RISK_RE = re.compile(
12
+ r"抽奖|支付|购买|充值|领取|赠送|提交|确认|保存|发布|删除|兑换|消耗|下单|立即|开始|发起|登录|注册|下载|上传|授权|同意|拒绝",
13
+ re.I,
14
+ )
15
+ SAFE_HINT_RE = re.compile(
16
+ r"详情|规则|说明|帮助|更多|展开|收起|关闭|信息|问号|tooltip|info|tab",
17
+ re.I,
18
+ )
19
+
20
+
21
+ def explore_safe(
22
+ page: Any,
23
+ context: Any,
24
+ *,
25
+ requested_url: str,
26
+ timeout_ms: int,
27
+ max_states: int,
28
+ max_depth: int,
29
+ max_actions: int,
30
+ max_duration_seconds: int,
31
+ evidence_mode: str,
32
+ evidence_writer: Any,
33
+ max_item: int,
34
+ max_total: int,
35
+ raw: dict[str, Any],
36
+ ) -> None:
37
+ """Explore only low-risk disclosure controls, reloading each branch."""
38
+
39
+ started = time.monotonic()
40
+ actions = 0
41
+ skipped: list[dict[str, str]] = []
42
+ transitions: list[dict[str, Any]] = []
43
+ queue: list[tuple[str, int, list[dict[str, Any]], list[dict[str, Any]]]] = [("state-001", 0, _candidates(page), [])]
44
+ seen = {_fingerprint(page)}
45
+
46
+ while queue and len(raw.get("states", [])) < max_states and actions < max_actions:
47
+ if time.monotonic() - started >= max_duration_seconds:
48
+ raw.setdefault("warnings", []).append("安全探索达到整体时长上限,未继续访问剩余状态。")
49
+ raw.setdefault("coverage", {})["partial_reason"] = "max_duration"
50
+ break
51
+ source_state, depth, candidates, path = queue.pop(0)
52
+ if depth >= max_depth:
53
+ continue
54
+ for candidate in candidates:
55
+ if actions >= max_actions or len(raw.get("states", [])) >= max_states:
56
+ break
57
+ if time.monotonic() - started >= max_duration_seconds:
58
+ raw.setdefault("warnings", []).append("安全探索达到整体时长上限,未继续访问剩余状态。")
59
+ raw.setdefault("coverage", {})["partial_reason"] = "max_duration"
60
+ break
61
+ decision = _risk_decision(candidate)
62
+ if decision != "allow":
63
+ skipped.append({"label": candidate.get("label", ""), "reason": decision})
64
+ continue
65
+
66
+ actions += 1
67
+ try:
68
+ page.goto(requested_url, wait_until="domcontentloaded", timeout=timeout_ms)
69
+ page.wait_for_timeout(min(400, timeout_ms))
70
+ navigation_guard = _install_navigation_guard(context, requested_url)
71
+ try:
72
+ for replay in path:
73
+ _scroll_page(page)
74
+ locator = _locator_for_candidate(page, replay)
75
+ locator.scroll_into_view_if_needed(timeout=timeout_ms)
76
+ locator.click(timeout=timeout_ms)
77
+ page.wait_for_timeout(min(250, timeout_ms))
78
+ _scroll_page(page)
79
+ locator = _locator_for_candidate(page, candidate)
80
+ locator.scroll_into_view_if_needed(timeout=timeout_ms)
81
+ before = _fingerprint(page)
82
+ locator.click(timeout=timeout_ms)
83
+ finally:
84
+ _remove_navigation_guard(context, navigation_guard)
85
+ page.wait_for_timeout(min(450, timeout_ms))
86
+ _scroll_page(page)
87
+ after = _fingerprint(page)
88
+ if not _is_supported_share_url(page.url, requested_url):
89
+ skipped.append({"label": candidate.get("label", ""), "reason": "navigation_blocked"})
90
+ continue
91
+ if after == before:
92
+ continue
93
+ state_id = f"state-{len(raw.get('states', [])) + 1:03d}"
94
+ state = {
95
+ "id": state_id,
96
+ "url": page.url,
97
+ "depth": depth + 1,
98
+ "action_count": actions,
99
+ "trigger": candidate.get("label", ""),
100
+ "action": "安全探索控件点击",
101
+ "evidence_ids": [],
102
+ }
103
+ raw.setdefault("states", []).append(state)
104
+ transitions.append(
105
+ {
106
+ "id": f"transition-{len(transitions) + 1:03d}",
107
+ "source_state_id": source_state,
108
+ "target_state_id": state_id,
109
+ "trigger": candidate.get("label", ""),
110
+ "action": "click",
111
+ "result": "state_changed",
112
+ }
113
+ )
114
+ fingerprint = after
115
+ if fingerprint in seen:
116
+ continue
117
+ seen.add(fingerprint)
118
+ # State artifacts are captured by the browser callback attached
119
+ # in browser.py through this small local callback.
120
+ _capture_explored_state(
121
+ page,
122
+ context,
123
+ raw=raw,
124
+ state_id=state_id,
125
+ requested_url=requested_url,
126
+ evidence_mode=evidence_mode,
127
+ evidence_writer=evidence_writer,
128
+ max_item=max_item,
129
+ max_total=max_total,
130
+ )
131
+ state["evidence_ids"] = list(raw.get("states", [])[-1].get("evidence_ids", []))
132
+ if depth + 1 < max_depth:
133
+ queue.append((state_id, depth + 1, _candidates(page), path + [candidate]))
134
+ except Exception as exc:
135
+ raw.setdefault("warnings", []).append(
136
+ f"安全探索跳过“{candidate.get('label', '')}”:{_short_error(exc)}"
137
+ )
138
+
139
+ raw.setdefault("coverage", {})["actions_attempted"] = actions
140
+ raw["coverage"]["transitions"] = transitions
141
+ raw["coverage"]["skipped_actions"] = skipped
142
+ if actions >= max_actions:
143
+ raw["coverage"]["partial_reason"] = "max_actions"
144
+ elif len(raw.get("states", [])) >= max_states:
145
+ raw["coverage"]["partial_reason"] = "max_states"
146
+
147
+
148
+ def _capture_explored_state(
149
+ page: Any,
150
+ context: Any,
151
+ *,
152
+ raw: dict[str, Any],
153
+ state_id: str,
154
+ requested_url: str,
155
+ evidence_mode: str,
156
+ evidence_writer: Any,
157
+ max_item: int,
158
+ max_total: int,
159
+ ) -> None:
160
+ from .capture_evidence import capture_state_evidence
161
+
162
+ warnings = raw.setdefault("warnings", [])
163
+ network: list[dict[str, Any]] = []
164
+ evidence, dom, stats = capture_state_evidence(
165
+ page,
166
+ context,
167
+ writer=evidence_writer if evidence_mode != "none" else None,
168
+ state_id=state_id,
169
+ requested_url=requested_url,
170
+ evidence_mode=evidence_mode,
171
+ max_item=max_item,
172
+ max_total=max_total,
173
+ warnings=warnings,
174
+ network_records=network,
175
+ evidence_id_offset=len(raw.get("evidence", [])),
176
+ )
177
+ from .capture_evidence import merge_evidence
178
+
179
+ merge_evidence(raw.setdefault("evidence", []), evidence)
180
+ raw.setdefault("network", []).extend(network)
181
+ state_evidence_ids = [
182
+ item["id"]
183
+ for item in raw["evidence"]
184
+ if item.get("state_id") == state_id
185
+ ]
186
+ raw["states"][-1].update({"dom": dom, "evidence_stats": stats, "evidence_ids": state_evidence_ids})
187
+ try:
188
+ from .browser import _DOM_EXTRACTION_SCRIPT
189
+
190
+ extracted = page.evaluate(_DOM_EXTRACTION_SCRIPT)
191
+ records = raw.setdefault("records", [])
192
+ start_order = max((int(item.get("dom_order", -1)) for item in records), default=-1) + 1
193
+ for index, record in enumerate(extracted.get("records", [])):
194
+ records.append({**record, "state_id": state_id, "dom_order": start_order + index})
195
+ except Exception as exc:
196
+ warnings.append(f"状态 {state_id} 的结构化内容未提取:{_short_error(exc)}")
197
+ raw.setdefault("coverage", {})["state_count"] = len(raw["states"])
198
+
199
+
200
+ def _candidates(page: Any) -> list[dict[str, Any]]:
201
+ try:
202
+ return page.evaluate(
203
+ """() => Array.from(document.querySelectorAll('button,[role=button],[role=tab],summary,[aria-expanded],a'))
204
+ .map((element, index) => {
205
+ const style = getComputedStyle(element);
206
+ const rect = element.getBoundingClientRect();
207
+ const label = (element.innerText || element.getAttribute('aria-label') || element.getAttribute('title') || '').trim();
208
+ return {index, tag: element.tagName.toLowerCase(), role: element.getAttribute('role') || '',
209
+ label, expanded: element.getAttribute('aria-expanded'), visible: style.display !== 'none' &&
210
+ style.visibility !== 'hidden' && rect.width > 0 && rect.height > 0};
211
+ }).filter(item => item.visible)"""
212
+ )
213
+ except Exception:
214
+ return []
215
+
216
+
217
+ def _risk_decision(candidate: dict[str, Any]) -> str:
218
+ label = str(candidate.get("label") or "").strip()
219
+ role = str(candidate.get("role") or "")
220
+ tag = str(candidate.get("tag") or "")
221
+ if not label and role not in {"tab"} and candidate.get("expanded") is None:
222
+ return "ambiguous_control"
223
+ if HIGH_RISK_RE.search(label):
224
+ return "high_risk_semantics"
225
+ if tag in {"input", "select", "textarea"}:
226
+ return "form_control"
227
+ if tag == "a":
228
+ return "navigation_skipped"
229
+ if role == "tab" or candidate.get("expanded") is not None or (tag == "summary" and SAFE_HINT_RE.search(label)):
230
+ return "allow"
231
+ if tag == "button" and SAFE_HINT_RE.search(label):
232
+ return "allow"
233
+ return "ambiguous_control"
234
+
235
+
236
+ def _locator_for_candidate(page: Any, candidate: dict[str, Any]) -> Any:
237
+ # The index belongs to the exact selector list used in _candidates, so it
238
+ # remains deterministic after reloading the branch from the entry URL.
239
+ return page.locator("button,[role=button],[role=tab],summary,[aria-expanded],a").nth(int(candidate.get("index", 0)))
240
+
241
+
242
+ def _fingerprint(page: Any) -> str:
243
+ try:
244
+ value = page.evaluate(
245
+ """() => ({url: location.href, text: (document.body?.innerText || '').slice(0, 20000),
246
+ visible: Array.from(document.querySelectorAll('[aria-selected],[aria-expanded],.mb-screen')).map(x =>
247
+ [x.tagName, x.getAttribute('aria-selected'), x.getAttribute('aria-expanded'), x.className]).join('|')})"""
248
+ )
249
+ return hashlib.sha256(repr(value).encode("utf-8", "replace")).hexdigest()
250
+ except Exception:
251
+ return str(time.monotonic_ns())
252
+
253
+
254
+ def _scroll_page(page: Any) -> None:
255
+ try:
256
+ page.evaluate(
257
+ """() => {
258
+ const nodes = [document.scrollingElement, ...document.querySelectorAll('*')]
259
+ .filter(x => x && x.scrollHeight > x.clientHeight + 8 &&
260
+ ['auto', 'scroll'].includes(getComputedStyle(x).overflowY));
261
+ for (const node of nodes.slice(0, 80)) node.scrollTop = node.scrollHeight;
262
+ window.scrollTo(0, 0);
263
+ }"""
264
+ )
265
+ page.wait_for_timeout(180)
266
+ except Exception:
267
+ pass
268
+
269
+
270
+ def _short_error(exc: Exception, limit: int = 220) -> str:
271
+ text = " ".join(str(exc).split())
272
+ return text[:limit] + ("…" if len(text) > limit else "")
273
+
274
+
275
+ def _is_supported_share_url(url: str, requested_url: str) -> bool:
276
+ from .browser import validate_share_url
277
+
278
+ try:
279
+ requested = validate_share_url(requested_url)
280
+ actual = validate_share_url(url)
281
+ return requested.project_id == actual.project_id
282
+ except Exception:
283
+ return False
284
+
285
+
286
+ def _install_navigation_guard(context: Any, requested_url: str) -> Any:
287
+ from .capture_evidence import same_origin
288
+
289
+ def guard(route: Any) -> None:
290
+ if same_origin(route.request.url, requested_url):
291
+ route.continue_()
292
+ else:
293
+ route.abort()
294
+
295
+ context.route("**/*", guard)
296
+ return guard
297
+
298
+
299
+ def _remove_navigation_guard(context: Any, guard: Any) -> None:
300
+ context.unroute("**/*", guard)