modao-prd-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- modao_prd_cli/__init__.py +1 -0
- modao_prd_cli/modao_prd/__init__.py +4 -0
- modao_prd_cli/modao_prd/__main__.py +7 -0
- modao_prd_cli/modao_prd/browser.py +578 -0
- modao_prd_cli/modao_prd/capture_evidence.py +547 -0
- modao_prd_cli/modao_prd/classifier.py +96 -0
- modao_prd_cli/modao_prd/cli.py +205 -0
- modao_prd_cli/modao_prd/errors.py +31 -0
- modao_prd_cli/modao_prd/evidence.py +207 -0
- modao_prd_cli/modao_prd/explorer.py +300 -0
- modao_prd_cli/modao_prd/extractor.py +465 -0
- modao_prd_cli/modao_prd/models.py +56 -0
- modao_prd_cli/modao_prd/normalizer.py +135 -0
- modao_prd_cli/modao_prd/schemas/coverage-1.0.json +15 -0
- modao_prd_cli/modao_prd/schemas/document-2.0.json +21 -0
- modao_prd_cli/modao_prd/schemas/document-2.1.json +31 -0
- modao_prd_cli/modao_prd/schemas/manifest-1.0.json +28 -0
- modao_prd_cli/modao_prd/tests/__init__.py +1 -0
- modao_prd_cli/modao_prd/tests/fixtures/modao_sample.html +20 -0
- modao_prd_cli/modao_prd/tests/test_browser.py +54 -0
- modao_prd_cli/modao_prd/tests/test_classifier.py +32 -0
- modao_prd_cli/modao_prd/tests/test_cli.py +88 -0
- modao_prd_cli/modao_prd/tests/test_extractor.py +57 -0
- modao_prd_cli/modao_prd/tests/test_full_e2e.py +23 -0
- modao_prd_cli/modao_prd/tests/test_writers.py +79 -0
- modao_prd_cli/modao_prd/writers.py +468 -0
- modao_prd_cli-0.1.0.dist-info/METADATA +108 -0
- modao_prd_cli-0.1.0.dist-info/RECORD +32 -0
- modao_prd_cli-0.1.0.dist-info/WHEEL +5 -0
- modao_prd_cli-0.1.0.dist-info/entry_points.txt +2 -0
- modao_prd_cli-0.1.0.dist-info/licenses/LICENSE +22 -0
- modao_prd_cli-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,468 @@
|
|
|
1
|
+
"""Render the normalized document to Agent and human-friendly formats."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any, Iterable
|
|
9
|
+
|
|
10
|
+
from .evidence import EvidenceBundleWriter
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
FORMATS = {"json", "markdown", "ndjson", "all"}
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def write_outputs(
|
|
17
|
+
document: dict[str, Any],
|
|
18
|
+
output_root: str | Path,
|
|
19
|
+
output_format: str,
|
|
20
|
+
force: bool = False,
|
|
21
|
+
*,
|
|
22
|
+
evidence_writer: EvidenceBundleWriter | None = None,
|
|
23
|
+
) -> dict[str, Any]:
|
|
24
|
+
if output_format not in FORMATS:
|
|
25
|
+
raise ValueError(f"unsupported output format: {output_format}")
|
|
26
|
+
root = Path(output_root).expanduser().resolve()
|
|
27
|
+
root.mkdir(parents=True, exist_ok=True)
|
|
28
|
+
project_id = document.get("source", {}).get("project_id") or "unknown"
|
|
29
|
+
output_dir = root / project_id
|
|
30
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
31
|
+
selected = {"json", "markdown", "ndjson"} if output_format == "all" else {output_format}
|
|
32
|
+
paths = {name: output_dir / f"document.{name if name != 'markdown' else 'md'}" for name in selected}
|
|
33
|
+
existing = [path for path in paths.values() if path.exists()]
|
|
34
|
+
if existing and not force and evidence_writer is None:
|
|
35
|
+
raise FileExistsError(
|
|
36
|
+
"输出文件已存在;如需覆盖请显式使用 --force:" + ", ".join(str(path) for path in existing)
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
contents: dict[str, str] = {}
|
|
40
|
+
if "json" in selected:
|
|
41
|
+
# document.json is the machine/Agent entrypoint. Keep Markdown as the
|
|
42
|
+
# human-readable projection, while avoiding indentation whitespace in
|
|
43
|
+
# the JSON payload that Agents generally parse rather than display.
|
|
44
|
+
contents["json"] = json.dumps(document, ensure_ascii=False, separators=(",", ":"), default=str) + "\n"
|
|
45
|
+
if "markdown" in selected:
|
|
46
|
+
contents["markdown"] = render_markdown(document)
|
|
47
|
+
if "ndjson" in selected:
|
|
48
|
+
contents["ndjson"] = render_ndjson(document)
|
|
49
|
+
coverage = document.get("coverage") or {
|
|
50
|
+
"schema_version": "1.0",
|
|
51
|
+
"states": document.get("facts", {}).get("states", []),
|
|
52
|
+
"transitions": document.get("facts", {}).get("observed_interactions", []),
|
|
53
|
+
"summary": {"quality": document.get("stats", {}).get("quality", "unknown")},
|
|
54
|
+
}
|
|
55
|
+
coverage_path = output_dir / "coverage.json"
|
|
56
|
+
if evidence_writer is not None:
|
|
57
|
+
for name, content in contents.items():
|
|
58
|
+
relative = f"document.{name if name != 'markdown' else 'md'}"
|
|
59
|
+
media_type = "text/markdown" if name == "markdown" else ("application/x-ndjson" if name == "ndjson" else "application/json")
|
|
60
|
+
evidence_writer.add(relative, content, media_type=media_type, enforce_item_limit=False)
|
|
61
|
+
paths[name] = output_dir / relative
|
|
62
|
+
evidence_writer.add("coverage.json", json.dumps(coverage, ensure_ascii=False, indent=2, default=str) + "\n", media_type="application/json", enforce_item_limit=False)
|
|
63
|
+
if document.get("capture", {}).get("evidence") == "full":
|
|
64
|
+
network = document.get("network", [])
|
|
65
|
+
evidence_writer.add(
|
|
66
|
+
"evidence/network/index.ndjson",
|
|
67
|
+
"".join(json.dumps(item, ensure_ascii=False, default=str) + "\n" for item in network),
|
|
68
|
+
media_type="application/x-ndjson",
|
|
69
|
+
enforce_item_limit=False,
|
|
70
|
+
)
|
|
71
|
+
manifest_path = evidence_writer.finalize(
|
|
72
|
+
extra={
|
|
73
|
+
"project_id": project_id,
|
|
74
|
+
"document": "document.json" if "json" in selected else None,
|
|
75
|
+
"document_schema": document.get("schema_version"),
|
|
76
|
+
"document_profile": "agent_compact",
|
|
77
|
+
"coverage": "coverage.json",
|
|
78
|
+
"format": output_format,
|
|
79
|
+
"status": document.get("stats", {}).get("quality", "unknown"),
|
|
80
|
+
"capture": document.get("capture", {}),
|
|
81
|
+
}
|
|
82
|
+
)
|
|
83
|
+
else:
|
|
84
|
+
for name, content in contents.items():
|
|
85
|
+
paths[name].write_text(content, encoding="utf-8")
|
|
86
|
+
manifest_path = None
|
|
87
|
+
return {
|
|
88
|
+
"output_dir": str(output_dir),
|
|
89
|
+
"files": {name: str(path) for name, path in paths.items()},
|
|
90
|
+
"coverage": str(output_dir / "coverage.json") if evidence_writer is not None else None,
|
|
91
|
+
"manifest": str(manifest_path) if manifest_path else None,
|
|
92
|
+
"format": output_format,
|
|
93
|
+
"overwrote": bool(existing),
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def render_markdown(document: dict[str, Any]) -> str:
|
|
98
|
+
source = document.get("source", {})
|
|
99
|
+
facts = document.get("facts", {})
|
|
100
|
+
derived = document.get("derived", {})
|
|
101
|
+
model = document.get("document", {}) or {
|
|
102
|
+
"title": source.get("title"),
|
|
103
|
+
"screens": facts.get("screens", []),
|
|
104
|
+
"sections": [],
|
|
105
|
+
"blocks": facts.get("blocks", []),
|
|
106
|
+
}
|
|
107
|
+
model["screens"] = model.get("screens") or facts.get("screens", [])
|
|
108
|
+
model["blocks"] = model.get("blocks") or facts.get("blocks", [])
|
|
109
|
+
requirements = document.get("requirements") or derived.get("requirements", [])
|
|
110
|
+
interactions = document.get("interactions") or derived.get("declared_interactions", [])
|
|
111
|
+
tables = document.get("tables") or facts.get("tables", [])
|
|
112
|
+
assets = document.get("assets") or facts.get("assets", [])
|
|
113
|
+
stats = document.get("stats", {})
|
|
114
|
+
lines = [
|
|
115
|
+
f"# {source.get('title') or model.get('title') or '墨刀原型'}",
|
|
116
|
+
"",
|
|
117
|
+
"## 来源信息",
|
|
118
|
+
"",
|
|
119
|
+
f"- URL: {source.get('url', '')}",
|
|
120
|
+
f"- 访问方式: {source.get('access_mode', 'public_share')}",
|
|
121
|
+
f"- 抓取时间: {source.get('fetched_at', '')}",
|
|
122
|
+
f"- Schema: {document.get('schema_version', '')}",
|
|
123
|
+
"",
|
|
124
|
+
"## 抓取摘要",
|
|
125
|
+
"",
|
|
126
|
+
f"- 页面/画布:{stats.get('screen_count', 0)}",
|
|
127
|
+
f"- 页面分区:{stats.get('section_count', 0)}",
|
|
128
|
+
f"- 文本块:{stats.get('text_block_count', 0)}",
|
|
129
|
+
f"- 表格:{stats.get('table_count', 0)}",
|
|
130
|
+
f"- 图片:{stats.get('image_count', 0)}",
|
|
131
|
+
f"- 控件:{stats.get('control_count', 0)}",
|
|
132
|
+
f"- 业务规则:{stats.get('requirement_count', 0)}",
|
|
133
|
+
f"- 交互:{stats.get('interaction_count', 0)}",
|
|
134
|
+
f"- 解析质量:{stats.get('quality', 'unknown')}",
|
|
135
|
+
f"- 状态:{stats.get('state_count', len(facts.get('states', [])))}",
|
|
136
|
+
f"- 证据项:{stats.get('evidence_count', len(document.get('evidence', [])))}",
|
|
137
|
+
"",
|
|
138
|
+
]
|
|
139
|
+
warnings = _unique_strings(document.get("warnings", []))
|
|
140
|
+
if warnings:
|
|
141
|
+
lines.extend(["## 待确认项", ""])
|
|
142
|
+
lines.extend(f"- {warning}" for warning in warnings)
|
|
143
|
+
lines.append("")
|
|
144
|
+
lines.extend([
|
|
145
|
+
"## 页面/画布",
|
|
146
|
+
"",
|
|
147
|
+
])
|
|
148
|
+
blocks_by_id = {block.get("id"): block for block in model.get("blocks", [])}
|
|
149
|
+
requirements_by_block: dict[str, list[dict[str, Any]]] = {}
|
|
150
|
+
for requirement in requirements:
|
|
151
|
+
for block_id in requirement.get("source_block_ids", []):
|
|
152
|
+
requirements_by_block.setdefault(block_id, []).append(requirement)
|
|
153
|
+
interactions_by_block: dict[str, list[dict[str, Any]]] = {}
|
|
154
|
+
for interaction in interactions:
|
|
155
|
+
for block_id in interaction.get("source_block_ids", []):
|
|
156
|
+
interactions_by_block.setdefault(block_id, []).append(interaction)
|
|
157
|
+
tables_by_block = {
|
|
158
|
+
table.get("source_block_id"): table
|
|
159
|
+
for table in tables
|
|
160
|
+
if table.get("source_block_id")
|
|
161
|
+
}
|
|
162
|
+
assets_by_block = {
|
|
163
|
+
asset.get("block_id"): asset
|
|
164
|
+
for asset in assets
|
|
165
|
+
if asset.get("block_id")
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
sections = model.get("sections", []) or [
|
|
169
|
+
{"title": screen.get("title"), "screen_id": screen.get("id"), "block_ids": []}
|
|
170
|
+
for screen in model.get("screens", [])
|
|
171
|
+
]
|
|
172
|
+
rendered_rule_keys: set[tuple[str, str]] = set()
|
|
173
|
+
rendered_interaction_keys: set[tuple[str, str]] = set()
|
|
174
|
+
rendered_table_ids: set[str] = set()
|
|
175
|
+
rendered_asset_sources: set[str] = set()
|
|
176
|
+
rendered_block_ids: set[str] = set()
|
|
177
|
+
|
|
178
|
+
for section in sections:
|
|
179
|
+
section_title = str(section.get("title") or "未命名页面").strip()
|
|
180
|
+
lines.extend([f"### {section_title}", ""])
|
|
181
|
+
section_blocks = [blocks_by_id[block_id] for block_id in section.get("block_ids", []) if block_id in blocks_by_id]
|
|
182
|
+
section_block_ids = {block.get("id") for block in section_blocks}
|
|
183
|
+
rendered_block_ids.update(section_block_ids)
|
|
184
|
+
|
|
185
|
+
content_lines: list[str] = []
|
|
186
|
+
seen_content: set[tuple[str, str]] = set()
|
|
187
|
+
for block in section_blocks:
|
|
188
|
+
block_id = block.get("id")
|
|
189
|
+
block_type = block.get("type")
|
|
190
|
+
text = str(block.get("text") or "").strip()
|
|
191
|
+
if block_type in {"table", "image"} or not text or _is_noise_text(text):
|
|
192
|
+
continue
|
|
193
|
+
if block_type in {"heading", "tab"} and _is_section_heading(text, section_title):
|
|
194
|
+
continue
|
|
195
|
+
if block_type in {"text", "label"} and (block_id in requirements_by_block or block_id in interactions_by_block):
|
|
196
|
+
continue
|
|
197
|
+
key = (str(block_type), text)
|
|
198
|
+
if key in seen_content:
|
|
199
|
+
continue
|
|
200
|
+
seen_content.add(key)
|
|
201
|
+
if block_type in {"heading", "tab"}:
|
|
202
|
+
content_lines.extend([f"#### {text}", ""])
|
|
203
|
+
elif block_type == "button":
|
|
204
|
+
content_lines.append(f"- 按钮:{text}")
|
|
205
|
+
elif block_type == "label":
|
|
206
|
+
content_lines.extend([f"**{text}**", ""])
|
|
207
|
+
else:
|
|
208
|
+
content_lines.extend([text, ""])
|
|
209
|
+
|
|
210
|
+
if content_lines:
|
|
211
|
+
lines.extend(["#### 页面文案", ""])
|
|
212
|
+
lines.extend(content_lines)
|
|
213
|
+
|
|
214
|
+
section_requirements: list[dict[str, Any]] = []
|
|
215
|
+
section_examples: list[str] = []
|
|
216
|
+
seen_examples: set[str] = set()
|
|
217
|
+
for block_id in section_block_ids:
|
|
218
|
+
for requirement in requirements_by_block.get(block_id, []):
|
|
219
|
+
statement = str(requirement.get("statement") or "").strip()
|
|
220
|
+
if not statement:
|
|
221
|
+
continue
|
|
222
|
+
if _is_example_statement(statement):
|
|
223
|
+
if statement not in seen_examples:
|
|
224
|
+
seen_examples.add(statement)
|
|
225
|
+
section_examples.append(statement)
|
|
226
|
+
continue
|
|
227
|
+
if _is_rule_noise(statement):
|
|
228
|
+
continue
|
|
229
|
+
has_interaction = any(interactions_by_block.get(source_id) for source_id in requirement.get("source_block_ids", []))
|
|
230
|
+
if requirement.get("type") == "interaction" or (_is_interaction_like_rule(statement) and has_interaction):
|
|
231
|
+
continue
|
|
232
|
+
key = (str(requirement.get("type") or "unknown"), statement)
|
|
233
|
+
if key in rendered_rule_keys or not key[1]:
|
|
234
|
+
continue
|
|
235
|
+
rendered_rule_keys.add(key)
|
|
236
|
+
section_requirements.append(requirement)
|
|
237
|
+
if section_examples:
|
|
238
|
+
lines.extend(["#### 原型示例记录", ""])
|
|
239
|
+
lines.extend(f"- {statement}" for statement in section_examples)
|
|
240
|
+
lines.append("")
|
|
241
|
+
if section_requirements:
|
|
242
|
+
lines.extend(["#### 业务规则", ""])
|
|
243
|
+
for requirement in section_requirements:
|
|
244
|
+
_append_rule(lines, requirement)
|
|
245
|
+
|
|
246
|
+
section_interactions: list[dict[str, Any]] = []
|
|
247
|
+
for block_id in section_block_ids:
|
|
248
|
+
for interaction in interactions_by_block.get(block_id, []):
|
|
249
|
+
key = (str(interaction.get("trigger") or "").strip(), str(interaction.get("action") or "").strip())
|
|
250
|
+
if key in rendered_interaction_keys or not any(key):
|
|
251
|
+
continue
|
|
252
|
+
rendered_interaction_keys.add(key)
|
|
253
|
+
section_interactions.append(interaction)
|
|
254
|
+
if section_interactions:
|
|
255
|
+
lines.extend(["#### 交互说明", ""])
|
|
256
|
+
for interaction in section_interactions:
|
|
257
|
+
_append_interaction(lines, interaction)
|
|
258
|
+
|
|
259
|
+
section_tables = []
|
|
260
|
+
for block in section_blocks:
|
|
261
|
+
table = tables_by_block.get(block.get("id"))
|
|
262
|
+
if table and table.get("id") not in rendered_table_ids:
|
|
263
|
+
rendered_table_ids.add(table.get("id"))
|
|
264
|
+
section_tables.append(table)
|
|
265
|
+
if section_tables:
|
|
266
|
+
lines.extend(["#### 表格", ""])
|
|
267
|
+
for index, table in enumerate(section_tables, start=1):
|
|
268
|
+
table_title = _contextual_table_title(table, section_title, index)
|
|
269
|
+
lines.extend(_markdown_table(table, title=table_title))
|
|
270
|
+
|
|
271
|
+
section_assets = []
|
|
272
|
+
for block in section_blocks:
|
|
273
|
+
asset = assets_by_block.get(block.get("id"))
|
|
274
|
+
src = str((asset or {}).get("src") or "")
|
|
275
|
+
if asset and src and src not in rendered_asset_sources:
|
|
276
|
+
rendered_asset_sources.add(src)
|
|
277
|
+
section_assets.append(asset)
|
|
278
|
+
if section_assets:
|
|
279
|
+
lines.extend(["#### 图片资源", ""])
|
|
280
|
+
for asset in section_assets:
|
|
281
|
+
alt = asset.get("alt") or "图片"
|
|
282
|
+
lines.extend([f"", ""])
|
|
283
|
+
|
|
284
|
+
if not content_lines and not section_requirements and not section_interactions and not section_tables and not section_assets:
|
|
285
|
+
lines.extend(["暂无可提取内容。", ""])
|
|
286
|
+
|
|
287
|
+
unassigned_requirements = [
|
|
288
|
+
requirement
|
|
289
|
+
for requirement in requirements
|
|
290
|
+
if requirement.get("type") != "interaction"
|
|
291
|
+
and not any(block_id in rendered_block_ids for block_id in requirement.get("source_block_ids", []))
|
|
292
|
+
]
|
|
293
|
+
if unassigned_requirements:
|
|
294
|
+
lines.extend(["## 页面级规则", ""])
|
|
295
|
+
for requirement in unassigned_requirements:
|
|
296
|
+
_append_rule(lines, requirement)
|
|
297
|
+
|
|
298
|
+
lines.extend(["## 页面布局与状态", ""])
|
|
299
|
+
for state in facts.get("states", []):
|
|
300
|
+
state_id = state.get("id", "unknown")
|
|
301
|
+
evidence_ids = ", ".join(state.get("evidence_ids", [])) or "无"
|
|
302
|
+
lines.append(f"- `{state_id}`:深度 {state.get('depth', 0)},证据 `{evidence_ids}`")
|
|
303
|
+
if not facts.get("states"):
|
|
304
|
+
lines.append("- 未记录状态快照。")
|
|
305
|
+
lines.append("")
|
|
306
|
+
lines.extend(["## 统计", ""])
|
|
307
|
+
for key, label in (
|
|
308
|
+
("block_count", "内容块"),
|
|
309
|
+
("text_block_count", "文本块"),
|
|
310
|
+
("table_count", "表格"),
|
|
311
|
+
("image_count", "图片"),
|
|
312
|
+
("control_count", "控件"),
|
|
313
|
+
("requirement_count", "业务规则"),
|
|
314
|
+
("interaction_count", "交互"),
|
|
315
|
+
("warning_count", "警告"),
|
|
316
|
+
):
|
|
317
|
+
lines.append(f"- {label}:{stats.get(key, 0)}")
|
|
318
|
+
lines.append("")
|
|
319
|
+
lines.extend(["## 证据与覆盖范围", ""])
|
|
320
|
+
lines.append("- 原始事实应以 `document.json` 的 `facts` 和 `evidence` 为准。")
|
|
321
|
+
lines.append("- 本 Markdown 只提供可读摘要,不替代 HTML、DOM、ARIA、截图和网络证据。")
|
|
322
|
+
coverage = document.get("coverage") or {}
|
|
323
|
+
if coverage.get("partial_reason"):
|
|
324
|
+
lines.append(f"- 未覆盖原因:{coverage['partial_reason']}")
|
|
325
|
+
for item in coverage.get("skipped_actions", [])[:20]:
|
|
326
|
+
lines.append(f"- 跳过动作:{item.get('label', '')}({item.get('reason', '')})")
|
|
327
|
+
lines.append("")
|
|
328
|
+
return "\n".join(lines)
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def _append_rule(lines: list[str], requirement: dict[str, Any]) -> None:
|
|
332
|
+
label = {
|
|
333
|
+
"validation": "校验",
|
|
334
|
+
"persistence": "状态",
|
|
335
|
+
"limit": "限制",
|
|
336
|
+
"timing": "时间",
|
|
337
|
+
"probability": "概率",
|
|
338
|
+
"reward": "奖励",
|
|
339
|
+
"warning": "注意",
|
|
340
|
+
"interaction": "交互",
|
|
341
|
+
}.get(str(requirement.get("type")), str(requirement.get("type") or "规则"))
|
|
342
|
+
statement = str(requirement.get("statement") or "").strip()
|
|
343
|
+
statement_lines = statement.splitlines() or [""]
|
|
344
|
+
lines.append(f"- **{label}**:{statement_lines[0]}")
|
|
345
|
+
lines.extend(f" {line}" for line in statement_lines[1:])
|
|
346
|
+
feedback = str(requirement.get("feedback") or "").strip()
|
|
347
|
+
if feedback and feedback not in statement:
|
|
348
|
+
lines.append(f" - 反馈:{feedback}")
|
|
349
|
+
lines.append("")
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def _append_interaction(lines: list[str], interaction: dict[str, Any]) -> None:
|
|
353
|
+
trigger = _clean_trigger(str(interaction.get("trigger") or "用户操作").strip())
|
|
354
|
+
action = str(interaction.get("action") or "").strip()
|
|
355
|
+
if action.startswith(trigger):
|
|
356
|
+
action = action[len(trigger):].lstrip(" ,,::")
|
|
357
|
+
if action and action != trigger:
|
|
358
|
+
lines.append(f"- **{trigger}**:{action}")
|
|
359
|
+
else:
|
|
360
|
+
lines.append(f"- **{trigger}**")
|
|
361
|
+
lines.append("")
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def _clean_trigger(trigger: str) -> str:
|
|
365
|
+
trigger = trigger.rstrip(",,::;;。!?!?))").strip()
|
|
366
|
+
trigger = re.sub(r"(?:和|或|的|即)$", "", trigger).strip()
|
|
367
|
+
if not trigger:
|
|
368
|
+
return "用户操作"
|
|
369
|
+
return trigger
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
def _contextual_table_title(table: dict[str, Any], section_title: str, index: int) -> str:
|
|
373
|
+
title = str(table.get("title") or "").strip()
|
|
374
|
+
if not title or title in {"未命名表格", "表格"} or title == section_title:
|
|
375
|
+
return f"{section_title} - 表格 {index}"
|
|
376
|
+
return title
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def _is_noise_text(text: str) -> bool:
|
|
380
|
+
return bool(re.fullmatch(r"(?:x|×)?\d+[+]?", text, flags=re.IGNORECASE))
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def _is_section_heading(text: str, section_title: str) -> bool:
|
|
384
|
+
aliases = {
|
|
385
|
+
section_title,
|
|
386
|
+
re.sub(r"^TAB\s*\d+\s*[-—::]\s*", "", section_title, flags=re.IGNORECASE),
|
|
387
|
+
}
|
|
388
|
+
return text in aliases
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def _is_rule_noise(statement: str) -> bool:
|
|
392
|
+
if len(statement) > 14:
|
|
393
|
+
return False
|
|
394
|
+
if re.fullmatch(r"(?:礼物名称|奖励|奖励记录|星海礼物|任务奖励|BUFF礼物|剩余奖励份数)(?:([^)]*))?", statement):
|
|
395
|
+
return True
|
|
396
|
+
meaningful_cues = (
|
|
397
|
+
"用户", "活动", "点击", "选择", "切换", "消耗", "购买", "获得", "领取", "展示",
|
|
398
|
+
"记录", "播报", "抽奖", "配置", "集齐", "抽中", "兑换", "有效期", "上限", "提示",
|
|
399
|
+
)
|
|
400
|
+
return not any(cue in statement for cue in meaningful_cues)
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def _is_interaction_like_rule(statement: str) -> bool:
|
|
404
|
+
if not re.search(r"点击|选择|切换|弹出|关闭|打开|输入|按键", statement):
|
|
405
|
+
return False
|
|
406
|
+
condition_cues = (
|
|
407
|
+
"若", "如果", "超过", "不足", "上限", "提示", "toast", "不可", "不能", "低于",
|
|
408
|
+
"达到", "最多", "至少", "概率", "随机",
|
|
409
|
+
)
|
|
410
|
+
return not any(cue in statement for cue in condition_cues)
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def _is_example_statement(statement: str) -> bool:
|
|
414
|
+
return bool(re.search(r"\b20\d{2}[./-]\d{1,2}[./-]\d{1,2}\b", statement))
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def _unique_strings(values: list[Any]) -> list[str]:
|
|
418
|
+
result: list[str] = []
|
|
419
|
+
seen: set[str] = set()
|
|
420
|
+
for value in values:
|
|
421
|
+
text = str(value).strip()
|
|
422
|
+
if text and text not in seen:
|
|
423
|
+
seen.add(text)
|
|
424
|
+
result.append(text)
|
|
425
|
+
return result
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
def _markdown_table(table: dict[str, Any], *, title: str | None = None) -> list[str]:
|
|
429
|
+
columns = [str(column) for column in table.get("columns", [])]
|
|
430
|
+
if not columns:
|
|
431
|
+
return []
|
|
432
|
+
lines = [f"##### {title or table.get('title') or '表格'}", "", "| " + " | ".join(_escape_cell(column) for column in columns) + " |", "| " + " | ".join("---" for _ in columns) + " |"]
|
|
433
|
+
for row in table.get("rows", []):
|
|
434
|
+
lines.append("| " + " | ".join(_escape_cell(str(row.get(column, ""))) for column in columns) + " |")
|
|
435
|
+
lines.append("")
|
|
436
|
+
return lines
|
|
437
|
+
|
|
438
|
+
|
|
439
|
+
def render_ndjson(document: dict[str, Any]) -> str:
|
|
440
|
+
facts = document.get("facts", {})
|
|
441
|
+
derived = document.get("derived", {})
|
|
442
|
+
units: list[dict[str, Any]] = [
|
|
443
|
+
{
|
|
444
|
+
"type": "metadata",
|
|
445
|
+
"schema_version": document.get("schema_version"),
|
|
446
|
+
"title": document.get("source", {}).get("title"),
|
|
447
|
+
"url": document.get("source", {}).get("url"),
|
|
448
|
+
"project_id": document.get("source", {}).get("project_id"),
|
|
449
|
+
}
|
|
450
|
+
]
|
|
451
|
+
units.extend({"type": "screen", **screen} for screen in facts.get("screens", []) or document.get("document", {}).get("screens", []))
|
|
452
|
+
units.extend({"type": "state", **state} for state in facts.get("states", []))
|
|
453
|
+
units.extend({"type": "block", **block} for block in facts.get("blocks", []) or document.get("document", {}).get("blocks", []))
|
|
454
|
+
# One knowledge unit per derived rule. The previous output emitted the
|
|
455
|
+
# same rule once as ``rule`` and again as ``derived_rule``, doubling the
|
|
456
|
+
# amount of text an Agent had to index.
|
|
457
|
+
units.extend({"type": "derived_rule", **rule} for rule in derived.get("requirements", []) or document.get("requirements", []))
|
|
458
|
+
units.extend({"type": "interaction", **item} for item in derived.get("declared_interactions", []) or document.get("interactions", []))
|
|
459
|
+
units.extend({"type": "observed_interaction", **item} for item in facts.get("observed_interactions", []))
|
|
460
|
+
units.extend({"type": "table", **table} for table in facts.get("tables", []) or document.get("tables", []))
|
|
461
|
+
units.extend({"type": "evidence_ref", **item} for item in document.get("evidence", []))
|
|
462
|
+
units.append({"type": "coverage", **(document.get("coverage") or {})})
|
|
463
|
+
units.extend({"type": "warning", "message": warning} for warning in document.get("warnings", []))
|
|
464
|
+
return "\n".join(json.dumps(unit, ensure_ascii=False, default=str) for unit in units) + "\n"
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
def _escape_cell(value: str) -> str:
|
|
468
|
+
return value.replace("|", "\\|").replace("\n", "<br>")
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: modao-prd-cli
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Extract public Modao prototype shares into agent-friendly JSON, Markdown, and NDJSON
|
|
5
|
+
Author: modao-prd-cli contributors
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Keywords: modao,prototype,prd,cli,agent,playwright
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Environment :: Console
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Topic :: Software Development :: Documentation
|
|
16
|
+
Requires-Python: >=3.10
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
License-File: LICENSE
|
|
19
|
+
Requires-Dist: click<9.0,>=8.1
|
|
20
|
+
Requires-Dist: playwright<2.0,>=1.62
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Requires-Dist: build>=1.2; extra == "dev"
|
|
23
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
24
|
+
Requires-Dist: twine>=5; extra == "dev"
|
|
25
|
+
Dynamic: license-file
|
|
26
|
+
|
|
27
|
+
# modao-prd-cli
|
|
28
|
+
|
|
29
|
+
`modao-prd-cli` 将墨刀公开分享页采集为便于 Agent 读取的结构化文档和证据包。它使用本机 Chrome/Chromium 渲染页面,不登录墨刀、不保存 Cookie、不调用远程解析服务。
|
|
30
|
+
|
|
31
|
+
## 安装
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
python3 -m pip install modao-prd-cli
|
|
35
|
+
python3 -m playwright install chromium # doctor 未发现可用浏览器时执行
|
|
36
|
+
modao-prd-cli doctor --json
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
## 使用
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
modao-prd-cli inspect "https://modao.cc/proto/<project-id>/sharing?view_mode=inspect" --json
|
|
43
|
+
|
|
44
|
+
modao-prd-cli export "https://modao.cc/proto/<project-id>/sharing?view_mode=inspect" \
|
|
45
|
+
--format all
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
不传 `--output` 时,工具会从当前工作目录向上查找最近的 Git 根目录,并将结果写入该项目根目录下的 `.modao-prd/<project-id>/`。如果当前目录不属于 Git 项目,则使用当前工作目录作为项目根目录。首次默认导出会在项目根目录的 `.gitignore` 中幂等加入 `/.modao-prd/`。
|
|
49
|
+
|
|
50
|
+
如需指定其他位置,显式传入 `--output` 即可;此时不会修改项目的 `.gitignore`:
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
modao-prd-cli export "<墨刀公开分享链接>" \
|
|
54
|
+
--format all --output ./modao-export
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
导出默认使用精简证据模式:保留结构化事实、DOM 布局快照、整页截图、可读画布截图和有效原型图片,过滤运行时代码、埋点资源和常见 UI 图标。DOM 快照也会排除 `script/style` 等运行时节点,避免把框架代码混进需求证据。需要排查渲染问题或核对视觉/无障碍细节时,再显式开启完整证据(debug/full)模式:
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
modao-prd-cli export "<墨刀公开分享链接>" --format all \
|
|
61
|
+
--evidence full --headed --explore safe --max-states 50 --max-depth 3 \
|
|
62
|
+
--max-actions 200 --timeout 30 --max-duration 120 \
|
|
63
|
+
--max-item-mb 10 --max-total-mb 200
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
只支持 HTTPS 的墨刀公开分享路径:`https://modao.cc/proto/<project-id>/sharing`。
|
|
67
|
+
|
|
68
|
+
## 输出
|
|
69
|
+
|
|
70
|
+
```text
|
|
71
|
+
.modao-prd/<project-id>/
|
|
72
|
+
├── manifest.json
|
|
73
|
+
├── document.json
|
|
74
|
+
├── document.md
|
|
75
|
+
├── document.ndjson
|
|
76
|
+
├── coverage.json
|
|
77
|
+
└── evidence/
|
|
78
|
+
├── state-001.dom.json
|
|
79
|
+
├── state-001.screen-001.png # essential/full:每个画布一张完整大图
|
|
80
|
+
├── state-001.full.png # --evidence full
|
|
81
|
+
├── state-001.rendered.html # --evidence full
|
|
82
|
+
├── state-001.aria.yaml # --evidence full
|
|
83
|
+
├── assets/ # 过滤后的原型图片
|
|
84
|
+
└── network/ # --evidence full
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
读取顺序建议为 `manifest.json → document.json → coverage.json → evidence/`。
|
|
88
|
+
|
|
89
|
+
- `document.json` 使用 schema 2.1,是面向 Agent 的紧凑精简投影,区分 `facts`(页面事实)和 `derived`(确定性启发式推导);需要人读时使用 `document.md`,需要逐条检索时使用 `document.ndjson`。
|
|
90
|
+
- `document.md` 是人工阅读摘要,不替代原始证据。
|
|
91
|
+
- `document.ndjson` 是逐行知识单元,适合 RAG、向量化和增量处理。
|
|
92
|
+
- 默认精简模式输出 DOM/布局证据、可读画布截图和过滤后的原型图片;`--evidence full` 额外保存浏览器整页截图、渲染 HTML、ARIA、相关 JSON/HTML 网络正文和网络索引。精简模式不会为了“完整”保存无助于需求分析的浏览器整页截图、JS、CSS、埋点或通用图标。
|
|
93
|
+
- 每个画布默认只生成一张原始比例的完整大图,例如 `screen-001.png`,不再切分成多张图片;Agent 可按需放大图片,具体文本和结构优先从 `document.json`、`dom.json` 读取。
|
|
94
|
+
- `--headed` 只控制是否显示浏览器窗口,便于 debug,不会单独切换证据级别;完整证据必须显式指定 `--evidence full`。
|
|
95
|
+
- `coverage.json` 记录访问到的状态、动作、跳过项和部分完成原因。
|
|
96
|
+
|
|
97
|
+
`document.json` 不重复内嵌完整 DOM 父子关系,也不重复输出旧版的 `document.blocks`、顶层 `requirements`、`tables` 等别名;布局树保存在 `evidence/state-*.dom.json`,主文档通过状态和精简证据引用回查。所有页面事实尽量带 `state_id`、`screen_id` 和核心 `evidence_ids`。发生冲突时,以截图、HTML、DOM、ARIA 和原始表格为准;`derived` 只作为分析线索。
|
|
98
|
+
|
|
99
|
+
退出码为 `0`(在策略范围内完成)、`2`(得到可用但部分完成的证据包)、`1`(未得到可用文档)。
|
|
100
|
+
|
|
101
|
+
## 限制
|
|
102
|
+
|
|
103
|
+
- 不支持私有项目、登录态、评论和评审记录。
|
|
104
|
+
- 不点击抽奖、支付、购买、领取、提交、保存、发布、删除等高风险控件。
|
|
105
|
+
- 不访问页面中的外部链接;跨域 iframe、Canvas、视觉表格和图片文字只保存证据并标记未结构化区域。
|
|
106
|
+
- 不承诺理解业务语义;原型示例、日期、概率和外部文档引用需要人工确认。
|
|
107
|
+
|
|
108
|
+
Agent 使用说明见 [`skills/modao-prd-cli/SKILL.md`](skills/modao-prd-cli/SKILL.md)。
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
modao_prd_cli/__init__.py,sha256=8frud5wgkIUB8JCwDKuQJ7CqxlYyDh4SgppNCpDiMX0,49
|
|
2
|
+
modao_prd_cli/modao_prd/__init__.py,sha256=DQNC6xk7n2O-z8ljAGANkClf0aPRanXYp4M8zo1ySos,69
|
|
3
|
+
modao_prd_cli/modao_prd/__main__.py,sha256=JxLLND8sn_QRTC_gQwEzNef0C5uNVXCQX4viGAIY_oI,131
|
|
4
|
+
modao_prd_cli/modao_prd/browser.py,sha256=AzLtu54y_Wv3knqqFskfM0CbgbhlD9Vuw8PbyWuoGTw,22412
|
|
5
|
+
modao_prd_cli/modao_prd/capture_evidence.py,sha256=XHLvk4SrKo8mEw78g4rgs9MjTnf36YEGGoJbJsEerU0,23698
|
|
6
|
+
modao_prd_cli/modao_prd/classifier.py,sha256=iTUqptOStYfCFD8QMzZDzvnvR62PF2CPnbLxCya3ej8,4235
|
|
7
|
+
modao_prd_cli/modao_prd/cli.py,sha256=zjCcMPvA6y530-ux399tHvnQOGtGmRZ85kW5AfAP5RE,8417
|
|
8
|
+
modao_prd_cli/modao_prd/errors.py,sha256=c6N3m9KJ-BCv3D72PabKrXx9c0YU4kkCaB1Nf0eU2sQ,801
|
|
9
|
+
modao_prd_cli/modao_prd/evidence.py,sha256=WbhomxpCVrMlDSYD7MkWoDCLvIbiR7FmwI5vUI4B6jY,8536
|
|
10
|
+
modao_prd_cli/modao_prd/explorer.py,sha256=WYNPdnDYP3OYdWR0-QMvAycdZzAvKh0CUkmir5YQtFc,12117
|
|
11
|
+
modao_prd_cli/modao_prd/extractor.py,sha256=OWPeMkhJf-EPd2L6UKoXuS5l3LkdsFXSOxLIlh8ghiM,20129
|
|
12
|
+
modao_prd_cli/modao_prd/models.py,sha256=E7PDuaSZyPndv4r06a7bi1vDWsJUumOWSqZdB8knIpI,1296
|
|
13
|
+
modao_prd_cli/modao_prd/normalizer.py,sha256=BxHSkaHbupGoz2GLmZW8-48IwqLBDtGBNCyU7uUb-jo,5171
|
|
14
|
+
modao_prd_cli/modao_prd/writers.py,sha256=bOM8uaQ58aK55qtATTqcT62tJh4sh3iHha3ltIbVqCs,21393
|
|
15
|
+
modao_prd_cli/modao_prd/schemas/coverage-1.0.json,sha256=rYrkpuUReO9KxLVE2HTBDwIvbJsJ_cZV17SdYEfa9Jk,476
|
|
16
|
+
modao_prd_cli/modao_prd/schemas/document-2.0.json,sha256=Zp59aS5h9MUiDE1v85ncsQYqaL0USQabJKSDnQYfj88,908
|
|
17
|
+
modao_prd_cli/modao_prd/schemas/document-2.1.json,sha256=KnXl-uuZDLQFV5HuwCwg74fUgCVboXewhn7rxVlSWSM,1114
|
|
18
|
+
modao_prd_cli/modao_prd/schemas/manifest-1.0.json,sha256=B3aKAc59To8Y1pzkMk5fKjVJDfIM4e4T9B8pRnQcq8s,927
|
|
19
|
+
modao_prd_cli/modao_prd/tests/__init__.py,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
|
|
20
|
+
modao_prd_cli/modao_prd/tests/test_browser.py,sha256=wUyptN_JP_bj3-BJNfsMp5eGXbvdX3p9bQGmaRQT2gM,1795
|
|
21
|
+
modao_prd_cli/modao_prd/tests/test_classifier.py,sha256=9zdzZoVNlCrR-3qgiKN444jn7DX_N933zQEeRcmK6q8,1133
|
|
22
|
+
modao_prd_cli/modao_prd/tests/test_cli.py,sha256=0-e6UP-8RYKj1uaGKnSFmIZ471N13a1-psASieYU44Q,3216
|
|
23
|
+
modao_prd_cli/modao_prd/tests/test_extractor.py,sha256=FR0_d_Q9wlpN7E83Y528xAlafG8h8cUci8m2ZIkP-5A,2602
|
|
24
|
+
modao_prd_cli/modao_prd/tests/test_full_e2e.py,sha256=dXybSUd_h9SOjtOkmYXxLmiEPxDKpX9j7_-c1S_wdyg,887
|
|
25
|
+
modao_prd_cli/modao_prd/tests/test_writers.py,sha256=iBRucC220VIQy0YCELqQ1J7xCBVB-nyjz5x8nZGsxn8,3038
|
|
26
|
+
modao_prd_cli/modao_prd/tests/fixtures/modao_sample.html,sha256=bXR-HlNAacntbR7Pq8fLKc9AbweHgLshnKO__bMTRUw,951
|
|
27
|
+
modao_prd_cli-0.1.0.dist-info/licenses/LICENSE,sha256=lj2zmRPE10Lsf0i8ObqRf7KfXlau8tnE3lp_d3B8MMI,1084
|
|
28
|
+
modao_prd_cli-0.1.0.dist-info/METADATA,sha256=G3jz661WE2-PcUUPLLfq0XeexPD98A4dYz8l1dj6snI,5930
|
|
29
|
+
modao_prd_cli-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
30
|
+
modao_prd_cli-0.1.0.dist-info/entry_points.txt,sha256=52aJZY_q68cBiIvelyhOXwOqgkh-VZ2yji8VBZ_xxKs,67
|
|
31
|
+
modao_prd_cli-0.1.0.dist-info/top_level.txt,sha256=u4ESjIM5czfnE7XzXEOmJ95CDu6lN0s6gTYpPwOQ35w,14
|
|
32
|
+
modao_prd_cli-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 modao-prd-cli contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
22
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
modao_prd_cli
|