eduevidence 6.2.0 → 6.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/CHANGELOG.md +395 -0
  2. package/README.md +22 -13
  3. package/README.zh-CN.md +15 -8
  4. package/SKILL.md +10 -9
  5. package/benchmarks/evidence-library.json +277 -1
  6. package/docs/architecture.md +6 -3
  7. package/docs/j-ev-experimental.md +250 -0
  8. package/docs/reproducibility.md +138 -0
  9. package/domains/_neutral/copy/few_shots.json +21 -0
  10. package/domains/_neutral/copy/framing_lexicon.json +19 -0
  11. package/domains/_neutral/copy/module_labels.json +5 -0
  12. package/domains/_neutral/copy/module_labels_footer.json +102 -0
  13. package/domains/_neutral/copy/module_labels_modules.json +204 -0
  14. package/domains/_neutral/copy/module_labels_nav.json +126 -0
  15. package/domains/_neutral/copy/module_labels_summary.json +98 -0
  16. package/domains/_neutral/copy/module_labels_tables.json +164 -0
  17. package/domains/_neutral/copy/module_labels_v2.json +90 -0
  18. package/domains/_neutral/copy/risk_constructs.json +20 -0
  19. package/domains/_neutral/copy/section_titles.json +66 -0
  20. package/domains/_neutral/copy/terminology.json +11 -0
  21. package/domains/check_copy_packs.py +103 -0
  22. package/domains/education/copy/few_shots.json +22 -0
  23. package/domains/education/copy/framing_enums.json +167 -0
  24. package/domains/education/copy/framing_lexicon.json +166 -0
  25. package/domains/education/copy/module_labels.json +169 -0
  26. package/domains/education/copy/risk_constructs.json +48 -0
  27. package/domains/education/copy/section_titles.json +186 -0
  28. package/domains/education/copy/terminology.json +70 -0
  29. package/domains/education/manifest.json +1 -1
  30. package/domains/education/outcome_taxonomy.json +2 -2
  31. package/domains/manifest.json +1 -1
  32. package/domains/policy/copy/few_shots.json +22 -0
  33. package/domains/policy/copy/framing_enums.json +94 -0
  34. package/domains/policy/copy/framing_lexicon.json +174 -0
  35. package/domains/policy/copy/module_labels.json +168 -0
  36. package/domains/policy/copy/risk_constructs.json +33 -0
  37. package/domains/policy/copy/section_titles.json +186 -0
  38. package/domains/policy/copy/terminology.json +64 -0
  39. package/engine/capabilities.py +57 -5
  40. package/engine/decision_policy.py +88 -17
  41. package/engine/library_builtin.py +7 -4
  42. package/engine/tribunal.py +17 -23
  43. package/engine/versions.py +1 -1
  44. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +4 -4
  45. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +4 -4
  46. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +4 -4
  47. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
  48. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +4 -4
  49. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +4 -4
  50. package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
  51. package/examples/spaced-retrieval-practice/report.html +2522 -0
  52. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +4 -4
  53. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +4 -4
  54. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
  55. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +4 -4
  56. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +4 -4
  57. package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
  58. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +36 -36
  59. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +36 -36
  60. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +36 -36
  61. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +36 -36
  62. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +36 -36
  63. package/integrations/jev/__init__.py +115 -0
  64. package/integrations/jev/approval.py +212 -0
  65. package/integrations/jev/cli.py +84 -0
  66. package/integrations/jev/config.py +112 -0
  67. package/integrations/jev/gateway.py +128 -0
  68. package/integrations/jev/modes.py +38 -0
  69. package/integrations/jev/tools_classify.py +88 -0
  70. package/integrations/jev/tools_extract.py +111 -0
  71. package/integrations/jev/tools_rerank.py +71 -0
  72. package/integrations/jev/tools_screen.py +87 -0
  73. package/integrations/jev/tools_verify.py +95 -0
  74. package/integrations/jev_mcp.py +22 -0
  75. package/integrations/semantic_decide.py +286 -0
  76. package/integrations/semdecide_cli.py +55 -0
  77. package/package.json +9 -1
  78. package/pyproject.toml +1 -1
  79. package/references/report-copy-style.md +43 -3
  80. package/schemas/v2/decision-snapshot.schema.json +20 -9
  81. package/schemas/v2/intake.schema.json +191 -0
  82. package/scripts/build_evidence_library.py +15 -5
  83. package/scripts/dashboard_server.py +13 -2
  84. package/scripts/intake/__init__.py +31 -0
  85. package/scripts/intake/__main__.py +18 -0
  86. package/scripts/intake/background.py +78 -0
  87. package/scripts/intake/browser.py +79 -0
  88. package/scripts/intake/cli.py +57 -0
  89. package/scripts/intake/constants.py +57 -0
  90. package/scripts/intake/depth.py +53 -0
  91. package/scripts/intake/enhancements.py +106 -0
  92. package/scripts/intake/hooks.py +90 -0
  93. package/scripts/intake/prefs.py +76 -0
  94. package/scripts/intake/prompts.py +85 -0
  95. package/scripts/intake/session.py +152 -0
  96. package/scripts/lint_file_layers.py +126 -0
  97. package/scripts/orchestrator.py +68 -17
  98. package/scripts/pre_verdict_gate.py +21 -7
  99. package/scripts/skill_lint.py +11 -1
  100. package/scripts/skill_payload.py +3 -3
  101. package/scripts/test_adversarial_empirical.py +70 -6
  102. package/skill/agents/evidence-judge.md +49 -7
  103. package/skill/workflows/experimental-jev.md +170 -0
  104. package/skill/workflows/intake.md +120 -0
  105. package/visualization/eduevidence-report/scripts/build_infographics.py +32 -14
  106. package/visualization/eduevidence-report/scripts/build_report.py +75 -662
  107. package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
  108. package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
  109. package/visualization/eduevidence-report/scripts/zh_labels.py +61 -0
  110. package/scripts/build_esl_artifacts.py +0 -1921
  111. package/scripts/build_killer_demo.py +0 -295
  112. package/scripts/enrich_projects_human_and_lieflat.py +0 -315
  113. package/scripts/generate_new_projects.py +0 -686
  114. package/scripts/sync_killer_demo_report.py +0 -270
@@ -0,0 +1,120 @@
1
+ ---
2
+ name: intake
3
+ description: One-shot two-round user intake before any research workflow. User answers only mode and questions; the agent infers the rest and then runs unattended.
4
+ ---
5
+
6
+ # Intake — 一次性两轮引导
7
+
8
+ Use this **before** selecting `evidence-review` / `decision-and-pilot` / `evaluate-and-update`.
9
+ The user answers **only** mode and questions. Everything else is agent judgment.
10
+ After the wrap-up summary is confirmed once, run **unattended**.
11
+
12
+ Programmatic twin: `scripts/intake/` package (`session.py` + `prefs.py` + `hooks.py`) +
13
+ `schemas/v2/intake.schema.json`.
14
+ Non-question preferences persist to `~/.eduevidence/prefs.json`
15
+ (`enhancement`, `mcp`/`jev` mappings, `depth_preference`, `open_browser`,
16
+ `default_main_theme`). **Never record the research question in prefs.**
17
+ All five report themes stay rendered; `default_main_theme` only decides which one to open.
18
+
19
+ ---
20
+
21
+ ## 第一轮固定语句(逐字使用)
22
+
23
+ ```text
24
+ 1) 研究问题(必填)
25
+ 2) 执行增强(可多选,各附一句话简介):
26
+ [agent_mcp] Agent MCP — 多 CLI/多模型分工、独立子上下文、超时恢复(可选执行增强)
27
+ [jev] Jev — 轻量工具子集派发,适合把确定性小步交给受限工具面
28
+ [semdecide] SemDecide — 语义决策辅助,只结构化上游意图,不改下游确定性路由
29
+ [none] 均不启用 — 平台原生模式,零外部执行增强
30
+ 3) 深度:
31
+ [S] S 快检 — 单点问题,最小检索面,直接给边界结论
32
+ [M] M 标准 — 常规采用/比较问题,标准证据到决策流程
33
+ [L] L 深研 — 高影响、有争议或要试点,完整深研与决策扩展
34
+ [auto] 自动 — 不选则按指南判定
35
+ ```
36
+
37
+ ### 深度自动判定指南(仅当用户选 auto / 不选)
38
+
39
+ | 信号 | 判定 |
40
+ |---|---|
41
+ | 单点问题(查定义、查公式、单条事实) | **S** |
42
+ | 常规采用 / 是否有效 / 比较对照 | **M** |
43
+ | 高影响、有争议、要试点 / 全面部署 / 政策级 | **L** |
44
+ | 不确定 | **M** |
45
+
46
+ 用户显式选 S / M / L 时**不再**自动判定。CLI 同义词:`quick=S`,`standard=M`,`deep=L`。
47
+
48
+ ---
49
+
50
+ ## 第二轮(仅在需要时追问)
51
+
52
+ ### A. 增强项(仅当第一轮勾选了对应增强)
53
+
54
+ | 增强开启时 | 第二轮只问这一项 | 落盘位置 |
55
+ |---|---|---|
56
+ | **Agent MCP** | 角色 → CLI / 模型 **授权表**(只展示本机扫描到的真实 CLI/模型;确认后固化 `~/.eduevidence/agent_mcp_approval.json`,含 `role_mapping_hash`) | prefs.`mcp.role_mapping` + 批准文件 |
57
+ | **Jev** | 工具子集(如 `search` / `fetch` / `validate`;只派发确定性小步,禁止旁路 spawn) | prefs.`jev.tool_subset` |
58
+ | **SemDecide** | 用途说明(只结构化上游 ResearchIntent / Frame,供 `recommend_mode`;**不改**下游确定性路由) | prefs / intake.`semdecide.purpose` |
59
+
60
+ 未勾选的增强**不得**追问。勾选「均不启用」则整段跳过。
61
+
62
+ ### B. 背景智能深挖(始终执行)
63
+
64
+ 按题目推断领域 Frame 骨架后,针对该题追问 **3–6** 个影响边界的问题:
65
+
66
+ 1. **人群**(目标学习者 / 决策对象)
67
+ 2. **干预**(具体方法 / 工具及用法)
68
+ 3. **对照**(与什么比较;可用 business as usual)
69
+ 4. **主结果**(一个主要结果指标)
70
+ 5. **情境**(学制 / 班型 / 线上线下 / 支持条件)
71
+
72
+ 规则:
73
+
74
+ - **智能默认 + 一次确认**:能从题目直接推出的字段给出默认值,用户回车即采纳;推出不出的字段留空追问,**禁止编造**。
75
+ - 全部边界答完后,打印一次摘要,**一次确认**。
76
+ - 摘要确认后进入无人值守,不再逐步打断。
77
+
78
+ ### C. 收尾摘要(一次确认)
79
+
80
+ ```text
81
+ —— 收尾摘要(一次确认后无人值守)——
82
+ 问题:…
83
+ 增强:…
84
+ 深度:…(判定理由)
85
+ 边界:population / intervention / comparison / primary_outcome / context
86
+ 结束后打开报告:是|否 | 主主题:claude|academic|datalab|datalab-dark|presentation
87
+ 确认并开始无人值守执行?[Y/n]
88
+ ```
89
+
90
+ 用户确认后:写入 `intake.json`(过 `schemas/v2/intake.schema.json`),更新 `prefs.json`(不含研究问题),调用 `eduevidence run` / 工作流,**不再提问**。
91
+
92
+ ---
93
+
94
+ ## 无人值守与浏览器
95
+
96
+ - `python3 -m intake.cli --print-prompts`(或 `python3 scripts/intake/__main__.py --print-prompts`)
97
+ - `eduevidence run --yes` 或非 TTY:跳过全部提问,只用 prefs + CLI 标志,**不**弹浏览器。
98
+ - 结束时若 `prefs.open_browser` 为 true 且终端可交互:用 `webbrowser.open` 打开**主报告**。
99
+ - 五主题全部渲染并保留在 `reports-5themes/`;主报告 = `default_main_theme` 对应的那份,只决定**打开哪份**,不删除其余主题。
100
+ - `python3 scripts/dashboard_server.py --open`(或 `eduevidence dashboard --open`)启动后打开 Studio URL `/studio/`。
101
+
102
+ ## 与 Canonical Protocol 的关系
103
+
104
+ Intake 不是协议阶段,只是启动前的**一次性**用户契约收集。收集完成后的流程仍走:
105
+
106
+ ```text
107
+ Evidence Review → Decision & Pilot → Evaluate & Update
108
+ ```
109
+
110
+ Frame 阶段(`skill/task-briefs/frame.md`)继续负责完整 Research Frame 与 `NEEDS_USER_CONTEXT`;
111
+ 本文件的边界追问只是提前把用户才知道的边界收齐,避免中途反复打断。
112
+
113
+ ## 失败处理
114
+
115
+ | 失败 | 处理 |
116
+ |---|---|
117
+ | 研究问题为空 | 第一轮重问;不得带空问题进入工作流。 |
118
+ | 用户不确认收尾摘要 | 回到对应轮次修改,不得静默开跑。 |
119
+ | 增强不可用(如 Agent MCP 未安装) | 记录 `AGENT_MCP_UNAVAILABLE` 并降级 Platform Native;不阻断科学正确性。 |
120
+ | 边界字段用户拒绝提供 | 显式标注 `unknown + 如何获取`(FR-03),禁止用默认值冒充事实。 |
@@ -30,11 +30,9 @@ Usage:
30
30
  from __future__ import annotations
31
31
 
32
32
  import argparse
33
- import json
34
33
  import re
35
34
  import sys
36
35
  import xml.sax.saxutils as sax
37
- from pathlib import Path
38
36
  from typing import Any
39
37
 
40
38
  from adapter_contract import load_result, write_adapter_output
@@ -204,7 +202,27 @@ def _phase_short(name: Any, index: int, lang: str) -> str:
204
202
  return f"Phase {index + 1}"
205
203
 
206
204
 
207
- def intervention_svg(intervention: dict, lang: str = "zh") -> str:
205
+ def _ui_title(ui: dict | None, lang: str, key: str) -> str:
206
+ """Domain copy pack title when provided; else the built-in bilingual default."""
207
+ if ui:
208
+ title = ui.get(f"svg_{key}_title")
209
+ if title:
210
+ return str(title)
211
+ return TITLES[lang][key]
212
+
213
+
214
+ def _eval_nodes(lang: str, ui: dict | None = None) -> list[tuple[str, str]]:
215
+ """Eval node labels; retention/transfer keywords come from evaluation_measures."""
216
+ nodes = [list(n) for n in EVAL_NODES[lang]]
217
+ measures = list((ui or {}).get("evaluation_measures") or [])
218
+ if len(measures) > 2 and measures[2]:
219
+ nodes[2][1] = str(measures[2])
220
+ if len(measures) > 3 and measures[3]:
221
+ nodes[3][1] = str(measures[3])
222
+ return [(lab, kw) for lab, kw in nodes]
223
+
224
+
225
+ def intervention_svg(intervention: dict, lang: str = "zh", ui: dict | None = None) -> str:
208
226
  """干预时间线:只放阶段短名与活动数(HTML-02),长规则文本在 HTML 阶段块。"""
209
227
  phases = [intervention.get(p) for p in ("phase_1", "phase_2", "phase_3", "phase_4")]
210
228
  phases = [p for p in phases if isinstance(p, dict)]
@@ -220,12 +238,12 @@ def intervention_svg(intervention: dict, lang: str = "zh") -> str:
220
238
  boxes.append(_box(x, y, bw, 96, name, PALETTE["primary"], font_size=12, sub=sub))
221
239
  if i < len(phases) - 1:
222
240
  boxes.append(_arrow(x + bw, y + 48, x + bw + gap, y + 48))
223
- return _svg(TITLES[lang]["intervention"], "".join(boxes))
241
+ return _svg(_ui_title(ui, lang, "intervention"), "".join(boxes))
224
242
 
225
243
 
226
- def evaluation_svg(evaluation: dict, lang: str = "zh") -> str:
244
+ def evaluation_svg(evaluation: dict, lang: str = "zh", ui: dict | None = None) -> str:
227
245
  """评价设计流程:只放阶段关键词(HTML-02),评估长文本在 HTML 段落。"""
228
- nodes = EVAL_NODES[lang]
246
+ nodes = _eval_nodes(lang, ui)
229
247
  keys = ("baseline", "post_test", "retention_test", "transfer_test")
230
248
  bw, gap, y = 150, 12, 110
231
249
  boxes = []
@@ -236,22 +254,22 @@ def evaluation_svg(evaluation: dict, lang: str = "zh") -> str:
236
254
  f'fill="#fff" opacity="0.95">{_esc(keyword)}</text>')
237
255
  if i < len(nodes) - 1:
238
256
  boxes.append(_arrow(x + bw, y + 28, x + bw + gap, y + 28))
239
- return _svg(TITLES[lang]["evaluation"], "".join(boxes))
257
+ return _svg(_ui_title(ui, lang, "evaluation"), "".join(boxes))
240
258
 
241
259
 
242
- def render_infographics(result: dict, lang: str = "zh") -> dict[str, str]:
243
- """按语言渲染 4 张信息图(zh 数据 = result.zh.json,en 数据 = result.json)。"""
260
+ def render_infographics(result: dict, lang: str = "zh", ui: dict | None = None) -> dict[str, str]:
261
+ """按语言渲染 4 张信息图;ui 提供域文案标题/评价关键词(缺省用内置双语表)。"""
244
262
  return {
245
263
  "workflow": workflow_svg(lang),
246
264
  "tribunal": tribunal_svg(result.get("decision", {}), lang),
247
- "intervention": intervention_svg(result.get("intervention", {}), lang),
248
- "evaluation": evaluation_svg(result.get("evaluation", {}), lang),
265
+ "intervention": intervention_svg(result.get("intervention", {}), lang, ui=ui),
266
+ "evaluation": evaluation_svg(result.get("evaluation", {}), lang, ui=ui),
249
267
  }
250
268
 
251
269
 
252
- def build_all(result: dict, lang: str = "zh") -> dict[str, str]:
253
- """兼容别名:等价于 render_infographics(result, lang)。"""
254
- return render_infographics(result, lang)
270
+ def build_all(result: dict, lang: str = "zh", ui: dict | None = None) -> dict[str, str]:
271
+ """兼容别名:等价于 render_infographics(result, lang, ui)。"""
272
+ return render_infographics(result, lang, ui=ui)
255
273
 
256
274
 
257
275
  def main() -> int: