eduevidence 5.2.0 → 6.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (386) hide show
  1. package/CONTRIBUTING.md +105 -0
  2. package/README.md +142 -75
  3. package/README.zh-CN.md +73 -30
  4. package/SKILL.md +397 -131
  5. package/agents/openai.yaml +4 -0
  6. package/assets/readme/controlled-execution.svg +34 -0
  7. package/assets/readme/landing-tour.gif +0 -0
  8. package/assets/readme/logo.png +0 -0
  9. package/assets/readme/research-workflow.svg +56 -0
  10. package/assets/readme/studio-graph.png +0 -0
  11. package/assets/readme/studio-overview.png +0 -0
  12. package/assets/readme/studio-reports.png +0 -0
  13. package/assets/readme/studio-tour.gif +0 -0
  14. package/autoevolve/config.yaml +17 -0
  15. package/autoevolve/program.md +25 -0
  16. package/autoevolve/protected.manifest.yaml +34 -0
  17. package/benchmarks/adversarial/cases.jsonl +7 -0
  18. package/benchmarks/evidence-library.json +5268 -0
  19. package/benchmarks/partitions.json +8 -0
  20. package/bin/eduevidence.js +2 -1
  21. package/docs/architecture.md +496 -0
  22. package/docs/autoresearch-evolution-plan.md +2903 -0
  23. package/docs/autoresearch-implementation-status.md +101 -0
  24. package/docs/demo-storyboard.md +20 -0
  25. package/docs/demo-workplace-ai.md +92 -0
  26. package/docs/demo.md +32 -0
  27. package/docs/install-guide.md +150 -0
  28. package/docs/orchestration-role-model.md +1254 -0
  29. package/docs/release-closeout/README.md +17 -0
  30. package/docs/release-closeout/frontend-acceptance.md +23 -0
  31. package/docs/release-closeout/issues.md +19 -0
  32. package/docs/release-closeout/verification.md +28 -0
  33. package/docs/release-contract.md +108 -0
  34. package/docs/research-studio-guide.zh-CN.md +166 -0
  35. package/docs/sciverse-api.md +125 -0
  36. package/eduevidence_cli.py +29 -13
  37. package/engine/_resources.py +13 -0
  38. package/engine/autoevolve/__init__.py +3 -0
  39. package/engine/autoevolve/agent_view.py +167 -0
  40. package/engine/autoevolve/core.py +357 -0
  41. package/engine/autoevolve/events.py +11 -0
  42. package/engine/autoevolve/git_workspace.py +77 -0
  43. package/engine/autoevolve/projection.py +23 -0
  44. package/engine/autoevolve/runner.py +413 -0
  45. package/engine/autoevolve/trust.py +146 -0
  46. package/engine/autoresearch/__init__.py +6 -0
  47. package/engine/autoresearch/commit.py +132 -0
  48. package/engine/autoresearch/contracts.py +126 -0
  49. package/engine/autoresearch/controller.py +207 -0
  50. package/engine/autoresearch/events.py +12 -0
  51. package/engine/autoresearch/gap_priority.py +168 -0
  52. package/engine/autoresearch/projection.py +30 -0
  53. package/engine/autoresearch/research_memory.py +59 -0
  54. package/engine/autoresearch/saturation.py +91 -0
  55. package/engine/briefs.py +2 -1
  56. package/engine/capabilities.py +1 -0
  57. package/engine/contracts.py +3 -1
  58. package/engine/decision_policy.py +96 -0
  59. package/engine/evidence_graph.py +14 -10
  60. package/engine/evidencecore.py +7 -5
  61. package/engine/gaps.py +132 -73
  62. package/engine/ids.py +2 -0
  63. package/engine/judge_pack.py +65 -0
  64. package/engine/library.py +6 -2
  65. package/engine/library_builtin.py +3 -1
  66. package/engine/living.py +36 -5
  67. package/engine/meta_synthesis.py +3 -1
  68. package/engine/migration.py +88 -3
  69. package/engine/orchestration.py +460 -0
  70. package/engine/paths.py +2 -0
  71. package/engine/pilot.py +36 -33
  72. package/engine/project.py +2 -2
  73. package/engine/research_service.py +113 -0
  74. package/engine/studio_read_model.py +400 -0
  75. package/engine/taxonomy.py +211 -0
  76. package/engine/tribunal.py +44 -33
  77. package/engine/update.py +1 -0
  78. package/engine/versions.py +1 -1
  79. package/engine/worker_result.py +109 -0
  80. package/engine/workflows.py +70 -0
  81. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +2934 -0
  82. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
  83. package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
  84. package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
  85. package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
  86. package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
  87. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  88. package/examples/ai-coding-assistant-evidence/frame.json +48 -0
  89. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  90. package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
  91. package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
  92. package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
  93. package/examples/ai-coding-assistant-evidence/report_spec.json +230 -0
  94. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2934 -0
  95. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2934 -0
  96. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2934 -0
  97. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2934 -0
  98. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2934 -0
  99. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +2934 -0
  100. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +2934 -0
  101. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +2934 -0
  102. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +2934 -0
  103. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +2934 -0
  104. package/examples/ai-coding-assistant-evidence/result.json +1457 -0
  105. package/examples/ai-coding-assistant-evidence/result.zh.json +1457 -0
  106. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  107. package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
  108. package/examples/ai-coding-assistant-evidence/verdict.json +107 -0
  109. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  110. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  111. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  112. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  113. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  114. package/examples/spaced-retrieval-practice/frame.json +58 -0
  115. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  116. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  117. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  118. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  119. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  120. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  121. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  122. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  123. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  124. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  125. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  126. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  127. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  128. package/examples/spaced-retrieval-practice/result.json +942 -0
  129. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  130. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  131. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  132. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  133. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  134. package/examples/workplace-ai-assistant/claims.jsonl +4 -0
  135. package/examples/workplace-ai-assistant/evaluation.json +19 -0
  136. package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
  137. package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
  138. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  139. package/examples/workplace-ai-assistant/frame.json +41 -0
  140. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  141. package/examples/workplace-ai-assistant/intervention.json +27 -0
  142. package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
  143. package/examples/workplace-ai-assistant/methodology.json +60 -0
  144. package/examples/workplace-ai-assistant/report_spec.json +224 -0
  145. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2814 -0
  146. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2814 -0
  147. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2814 -0
  148. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2814 -0
  149. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2814 -0
  150. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  151. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  152. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  153. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  154. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  155. package/examples/workplace-ai-assistant/result.json +615 -0
  156. package/examples/workplace-ai-assistant/result.zh.json +615 -0
  157. package/examples/workplace-ai-assistant/search_log.json +19 -0
  158. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  159. package/examples/workplace-ai-assistant/sources.jsonl +3 -0
  160. package/examples/workplace-ai-assistant/validation_result.json +9 -0
  161. package/examples/workplace-ai-assistant/verdict.json +78 -0
  162. package/install.sh +7 -7
  163. package/integrations/agent_mcp.py +2 -2
  164. package/integrations/orchestration_dispatch.py +146 -0
  165. package/package.json +46 -3
  166. package/pyproject.toml +14 -22
  167. package/references/autoresearch.md +30 -0
  168. package/references/evaluation-policy.md +24 -0
  169. package/references/orchestration.md +22 -0
  170. package/references/report-copy-style.md +67 -0
  171. package/references/retrieval-compliance.md +75 -0
  172. package/references/retrieval-protocol.md +20 -0
  173. package/references/scientific-invariants.md +19 -0
  174. package/retrieval/audit.py +178 -0
  175. package/retrieval/fetch.py +96 -0
  176. package/retrieval/sciverse.py +398 -0
  177. package/retrieval/search.py +47 -7
  178. package/schemas/applicability.schema.json +94 -0
  179. package/schemas/chart-spec.schema.json +10 -3
  180. package/schemas/evidence.schema.json +316 -43
  181. package/schemas/fetch-result.schema.json +2 -1
  182. package/schemas/intervention.schema.json +106 -21
  183. package/schemas/report-result.schema.json +12 -4
  184. package/schemas/report-spec.schema.json +98 -100
  185. package/schemas/skeptic.schema.json +86 -0
  186. package/schemas/source.schema.json +21 -2
  187. package/schemas/v2/finding.schema.json +5 -1
  188. package/schemas/v2/methodology-audit.schema.json +5 -1
  189. package/schemas/v2/outcome.schema.json +28 -5
  190. package/schemas/v2/project.schema.json +2 -2
  191. package/schemas/v2/run.schema.json +1 -1
  192. package/schemas/v2/study.schema.json +5 -1
  193. package/schemas/vNext/autoevolve-session.schema.json +34 -0
  194. package/schemas/vNext/eval-snapshot.schema.json +77 -0
  195. package/schemas/vNext/execution-plan.schema.json +50 -0
  196. package/schemas/vNext/gap-priority.schema.json +54 -0
  197. package/schemas/vNext/negative-search-record.schema.json +68 -0
  198. package/schemas/vNext/research-iteration.schema.json +87 -0
  199. package/schemas/vNext/research-strategy.schema.json +62 -0
  200. package/schemas/vNext/skill-experiment.schema.json +90 -0
  201. package/schemas/vNext/task-spec.schema.json +156 -0
  202. package/schemas/vNext/worker-result.schema.json +60 -0
  203. package/schemas/verdict.schema.json +164 -28
  204. package/scripts/benchmark_judge.py +2 -2
  205. package/scripts/benchmark_v3.py +26 -43
  206. package/scripts/build_esl_artifacts.py +4 -4
  207. package/scripts/build_evidence_library.py +2 -2
  208. package/scripts/build_gh_pages.py +98 -0
  209. package/scripts/build_readme_diagrams.py +72 -0
  210. package/scripts/build_report_variants.py +101 -0
  211. package/scripts/build_result.py +74 -9
  212. package/scripts/check_autoresearch_invariants.py +95 -0
  213. package/scripts/check_package_parity.py +85 -0
  214. package/scripts/check_protocol_alignment.py +375 -0
  215. package/scripts/check_versioned_schemas.py +254 -0
  216. package/scripts/claim_audit.py +13 -8
  217. package/scripts/compute_confidence.py +10 -0
  218. package/scripts/daily_evolve.py +30 -0
  219. package/scripts/dashboard_server.py +130 -101
  220. package/scripts/did_regression.py +17 -32
  221. package/scripts/enrich_projects_human_and_lieflat.py +1 -1
  222. package/scripts/evidence_score.py +5 -2
  223. package/scripts/generate_metrics.py +4 -3
  224. package/scripts/generate_new_projects.py +5 -5
  225. package/scripts/orchestrator.py +286 -36
  226. package/scripts/pre_verdict_gate.py +224 -26
  227. package/scripts/quickstart.py +18 -2
  228. package/scripts/rebake_all_5themes.py +1 -2
  229. package/scripts/research_auto_cli.py +475 -0
  230. package/scripts/run_workspace.py +24 -8
  231. package/scripts/search_provenance.py +64 -0
  232. package/scripts/serve_web.py +9 -10
  233. package/scripts/skill_lint.py +1 -1
  234. package/scripts/skill_payload.py +81 -0
  235. package/scripts/test_adversarial_empirical.py +26 -19
  236. package/scripts/validate_schema.py +46 -2
  237. package/scripts/vnext_cli.py +133 -0
  238. package/setup.py +12 -0
  239. package/skill/agents/evaluation-designer.md +20 -4
  240. package/skill/agents/evidence-analyst.md +19 -3
  241. package/skill/agents/evidence-judge.md +50 -2
  242. package/skill/agents/evidence-retriever.md +20 -3
  243. package/skill/agents/intervention-designer.md +20 -4
  244. package/skill/agents/method-reviewer.md +18 -2
  245. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  246. package/skill/agents/skeptic.md +18 -2
  247. package/skill/roles/registry.yaml +45 -0
  248. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  249. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  250. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  251. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  252. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  253. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  254. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  255. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  256. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  257. package/skill/sub-skills/report-generation/SKILL.md +40 -6
  258. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  259. package/skill/sub-skills/study-design/SKILL.md +30 -9
  260. package/skill/task-briefs/adjudicate.md +32 -7
  261. package/skill/task-briefs/applicability.md +38 -0
  262. package/skill/task-briefs/audit.md +32 -7
  263. package/skill/task-briefs/challenge.md +34 -5
  264. package/skill/task-briefs/evaluate.md +30 -5
  265. package/skill/task-briefs/extract.md +31 -8
  266. package/skill/task-briefs/frame.md +39 -10
  267. package/skill/task-briefs/intervene.md +32 -6
  268. package/skill/task-briefs/present.md +32 -8
  269. package/skill/task-briefs/projection.md +37 -0
  270. package/skill/task-briefs/retrieve.md +36 -6
  271. package/skill/workflows/decision-and-pilot.md +85 -0
  272. package/skill/workflows/evaluate-and-update.md +93 -0
  273. package/skill/workflows/evidence-review.md +117 -0
  274. package/visualization/eduevidence-report/assets/base.css +2 -2
  275. package/visualization/eduevidence-report/assets/reader.css +752 -0
  276. package/visualization/eduevidence-report/assets/reader.js +132 -0
  277. package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
  278. package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
  279. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  280. package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
  281. package/visualization/eduevidence-report/scripts/build_report.py +561 -121
  282. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  283. package/visualization/eduevidence-report/scripts/lieflat_engine.py +371 -136
  284. package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
  285. package/visualization/eduevidence-report/themes/academic.css +1 -1
  286. package/visualization/eduevidence-report/themes/claude.css +1 -1
  287. package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
  288. package/visualization/eduevidence-report/themes/datalab.css +2 -2
  289. package/visualization/eduevidence-report/themes/presentation.css +2 -2
  290. package/web/README.md +18 -0
  291. package/web/architecture.html +14885 -0
  292. package/web/index.html +53 -0
  293. package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
  294. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  295. package/web/studio/assets/index-CQ6Keoyc.js +230 -0
  296. package/web/studio/config.json +1 -0
  297. package/web/studio/index.html +14 -0
  298. package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
  299. package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
  300. package/engine/__pycache__/bias.cpython-312.pyc +0 -0
  301. package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
  302. package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
  303. package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
  304. package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
  305. package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
  306. package/engine/__pycache__/events.cpython-312.pyc +0 -0
  307. package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
  308. package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
  309. package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
  310. package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
  311. package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
  312. package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
  313. package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
  314. package/engine/__pycache__/ids.cpython-312.pyc +0 -0
  315. package/engine/__pycache__/library.cpython-312.pyc +0 -0
  316. package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
  317. package/engine/__pycache__/living.cpython-312.pyc +0 -0
  318. package/engine/__pycache__/log.cpython-312.pyc +0 -0
  319. package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
  320. package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
  321. package/engine/__pycache__/migration.cpython-312.pyc +0 -0
  322. package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
  323. package/engine/__pycache__/paths.cpython-312.pyc +0 -0
  324. package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
  325. package/engine/__pycache__/planner.cpython-312.pyc +0 -0
  326. package/engine/__pycache__/project.cpython-312.pyc +0 -0
  327. package/engine/__pycache__/projections.cpython-312.pyc +0 -0
  328. package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
  329. package/engine/__pycache__/run.cpython-312.pyc +0 -0
  330. package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
  331. package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
  332. package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
  333. package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
  334. package/engine/__pycache__/update.cpython-312.pyc +0 -0
  335. package/engine/__pycache__/versions.cpython-312.pyc +0 -0
  336. package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
  337. package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
  338. package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
  339. package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
  340. package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
  341. package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
  342. package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
  343. package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
  344. package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
  345. package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
  346. package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
  347. package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
  348. package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
  349. package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
  350. package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
  351. package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
  352. package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
  353. package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
  354. package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
  355. package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
  356. package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
  357. package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
  358. package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
  359. package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
  360. package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
  361. package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
  362. package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
  363. package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
  364. package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
  365. package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
  366. package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
  367. package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
  368. package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
  369. package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
  370. package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
  371. package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
  372. package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
  373. package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
  374. package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
  375. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
  376. package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
  377. package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
  378. package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
  379. package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
  380. package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
  381. package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
  382. package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
  383. package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
  384. package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
  385. package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
  386. package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
@@ -0,0 +1,59 @@
1
+ from __future__ import annotations
2
+ import json
3
+ from pathlib import Path
4
+ from typing import Any
5
+ from .contracts import NegativeSearchRecord, ResearchIteration
6
+
7
+
8
+ class ResearchMemory:
9
+ def __init__(self, root: str | Path):
10
+ self.root = Path(root)
11
+ self.root.mkdir(parents=True, exist_ok=True)
12
+ self.iterations_path = self.root / "research-iterations.jsonl"
13
+ self.negative_path = self.root / "negative-searches.jsonl"
14
+
15
+ @staticmethod
16
+ def _append(path: Path, record: dict[str, Any]) -> None:
17
+ with path.open("a", encoding="utf-8") as f:
18
+ f.write(json.dumps(record, ensure_ascii=False, sort_keys=True) + "\n")
19
+
20
+ def append_iteration(self, iteration: ResearchIteration) -> None:
21
+ iteration.validate()
22
+ self._append(self.iterations_path, iteration.as_dict())
23
+
24
+ def append_negative_search(self, record: NegativeSearchRecord) -> None:
25
+ record.validate()
26
+ self._append(self.negative_path, record.__dict__)
27
+
28
+ def load_iterations(
29
+ self,
30
+ gap_id: str | None = None,
31
+ *,
32
+ gap_lineage_key: str | None = None,
33
+ ) -> list[dict[str, Any]]:
34
+ """Load iteration history, preferring stable lineage across revisions.
35
+
36
+ Legacy rows without `gap_lineage_key` remain queryable by `gap_id`.
37
+ When a lineage key is supplied, new keyed rows match by lineage and old
38
+ unkeyed rows may additionally match the supplied gap_id for migration.
39
+ """
40
+ if not self.iterations_path.exists():
41
+ return []
42
+ rows = [
43
+ json.loads(line)
44
+ for line in self.iterations_path.read_text(encoding="utf-8").splitlines()
45
+ if line.strip()
46
+ ]
47
+ if gap_lineage_key is not None:
48
+ return [
49
+ row for row in rows
50
+ if row.get("gap_lineage_key") == gap_lineage_key
51
+ or (
52
+ not row.get("gap_lineage_key")
53
+ and gap_id is not None
54
+ and row.get("gap_id") == gap_id
55
+ )
56
+ ]
57
+ if gap_id is not None:
58
+ return [row for row in rows if row.get("gap_id") == gap_id]
59
+ return rows
@@ -0,0 +1,91 @@
1
+ from __future__ import annotations
2
+ from dataclasses import dataclass
3
+ from typing import Any
4
+
5
+
6
+ @dataclass(frozen=True)
7
+ class SaturationResult:
8
+ saturated: bool
9
+ low_yield_streak: int
10
+ strategy_diversity_exhausted: bool
11
+ rationale: tuple[str, ...]
12
+
13
+
14
+ def _is_low_yield(row: dict[str, Any]) -> bool:
15
+ gain = row.get("evidence_gain") or {}
16
+ unique = int(gain.get("unique_eligible_evidence", 0) or 0)
17
+ direct = int(gain.get("direct_outcome_findings", 0) or 0)
18
+ delta = float(gain.get("decision_boundary_delta", 0) or 0)
19
+ duplicate_rate = float(gain.get("duplicate_rate", 0) or 0)
20
+ candidate_sources = row.get("candidate_sources") or []
21
+ no_candidates = len(candidate_sources) == 0
22
+ return (
23
+ unique == 0
24
+ and direct == 0
25
+ and abs(delta) < 1e-12
26
+ and (duplicate_rate >= 0.5 or no_candidates)
27
+ )
28
+
29
+
30
+ def detect_saturation(
31
+ iterations: list[dict[str, Any]],
32
+ *,
33
+ min_consecutive: int = 2,
34
+ available_strategy_types: set[str] | None = None,
35
+ ) -> SaturationResult:
36
+ """Detect bounded secondary-search saturation.
37
+
38
+ Strategy diversity is computed across the full history for the gap, while
39
+ the low-yield condition is intentionally a trailing streak. Empty searches
40
+ count as low-yield even when duplicate_rate is zero; otherwise a provider
41
+ returning no candidates could keep the loop alive forever.
42
+ """
43
+ attempted_all = {
44
+ str((row.get("strategy") or {}).get("experiment_type", ""))
45
+ for row in iterations
46
+ if str((row.get("strategy") or {}).get("experiment_type", ""))
47
+ }
48
+ streak = 0
49
+ for row in reversed(iterations):
50
+ if _is_low_yield(row):
51
+ streak += 1
52
+ else:
53
+ break
54
+
55
+ if available_strategy_types:
56
+ diversity_exhausted = available_strategy_types.issubset(attempted_all)
57
+ else:
58
+ diversity_exhausted = len(attempted_all) >= 2
59
+
60
+ rationale: list[str] = []
61
+ if streak >= min_consecutive:
62
+ rationale.append(
63
+ f"{streak} consecutive iterations produced no unique/direct evidence or decision-boundary change"
64
+ )
65
+ if diversity_exhausted:
66
+ rationale.append("strategy diversity exhausted for the configured search space")
67
+ return SaturationResult(
68
+ streak >= min_consecutive and diversity_exhausted,
69
+ streak,
70
+ diversity_exhausted,
71
+ tuple(rationale),
72
+ )
73
+
74
+
75
+ def transition_to_empirical(
76
+ *,
77
+ dvi_band: str,
78
+ decision_material: bool,
79
+ unresolved: bool,
80
+ saturation: SaturationResult,
81
+ ethics_feasible: bool,
82
+ ) -> tuple[bool, tuple[str, ...]]:
83
+ checks = [
84
+ (dvi_band.upper() == "HIGH", "gap DVI is HIGH"),
85
+ (decision_material, "gap is material to the decision"),
86
+ (unresolved, "gap remains unresolved"),
87
+ (saturation.saturated, "secondary search is saturated"),
88
+ (ethics_feasible, "empirical study is ethically/operationally feasible"),
89
+ ]
90
+ reasons = tuple(text for ok, text in checks if ok)
91
+ return all(ok for ok, _ in checks), reasons
package/engine/briefs.py CHANGED
@@ -14,6 +14,7 @@ from pathlib import Path
14
14
 
15
15
  from engine.contracts import load_schema, schema_path
16
16
  from engine.planner import PlanStep
17
+ from engine.project import ProjectWorkspace
17
18
 
18
19
 
19
20
  def _schema_section(schema: dict) -> str:
@@ -73,7 +74,7 @@ def build_task_brief(step: PlanStep, *, project: ProjectWorkspace,
73
74
  f"Validate the output with `engine.contracts.validate_record({schema_name!r}, record)` — "
74
75
  f"it must return [] (empty errors).",
75
76
  "",
76
- f"## Output path",
77
+ "## Output path",
77
78
  str(output_path),
78
79
  "",
79
80
  "## Inputs",
@@ -18,6 +18,7 @@ class CapabilitySpec:
18
18
  output_contracts: tuple[str, ...]
19
19
  deterministic_local: bool
20
20
  scientific_gate: str | None
21
+ workflow_ids: tuple[str, ...] = ()
21
22
 
22
23
 
23
24
  _REGISTRY: dict[str, CapabilitySpec] = {}
@@ -12,7 +12,9 @@ from typing import Callable
12
12
 
13
13
  from scripts.validate_schema import SchemaError, validate
14
14
 
15
- _REPO_SCHEMA_DIR = Path(__file__).resolve().parent.parent / "schemas" / "v2"
15
+ from engine._resources import resource_root
16
+
17
+ _REPO_SCHEMA_DIR = resource_root() / "schemas" / "v2"
16
18
 
17
19
 
18
20
  def _resolve_schema_dir() -> Path:
@@ -0,0 +1,96 @@
1
+ """Single source of truth for the ADOPT direct-evidence gate.
2
+
3
+ The four-state decision (ADOPT / PILOT / REJECT / INSUFFICIENT EVIDENCE) is
4
+ computed by two layers that must never drift apart:
5
+
6
+ * engine/tribunal.py - V2 Evidence Graph adjudication
7
+ * scripts/pre_verdict_gate.py - V1 run/example-pack gate enforcement
8
+
9
+ Before this module existed the rule lived only inside the tribunal, so the V1
10
+ gate could not enforce it and a hand-written verdict could carry any action it
11
+ liked. Both layers now resolve the 'which outcome categories count for this
12
+ domain' question here, and the V1 gate additionally re-derives direct-evidence
13
+ presence from the pack's own evidence records.
14
+
15
+ Stdlib only, consistent with the "Native Core" policy of engine/.
16
+ """
17
+ from __future__ import annotations
18
+
19
+ #: Category buckets that count as a domain's PRIMARY effect for the ADOPT gate.
20
+ #: The gate asks 'is there direct evidence on the outcome this decision is
21
+ #: actually about?' - for education that is a learning outcome (task
22
+ #: performance and process measures never qualify); for policy it is the
23
+ #: policy-effectiveness / cost class. Every entry must name a category the
24
+ #: domain registry declares, which check_protocol_alignment.py enforces.
25
+ PRIMARY_EFFECT_CATEGORIES: dict[str, tuple[str, ...]] = {
26
+ "education": ("learning",),
27
+ "policy": ("effectiveness", "cost"),
28
+ }
29
+
30
+ #: Directness (0-2) at which an evidence link may carry an ADOPT claim.
31
+ ADOPT_DIRECTNESS = 2
32
+
33
+ #: Confidence band required before ADOPT is possible at all.
34
+ ADOPT_REQUIRED_LABEL = "High"
35
+
36
+
37
+ def primary_effect_categories(domain: str) -> tuple[str, ...]:
38
+ """Categories that satisfy the ADOPT direct-evidence gate for a domain."""
39
+ from engine.taxonomy import categories as taxonomy_categories
40
+
41
+ declared = PRIMARY_EFFECT_CATEGORIES.get(domain)
42
+ if declared:
43
+ known = taxonomy_categories(domain)
44
+ missing = [c for c in declared if c not in known]
45
+ if missing:
46
+ raise ValueError(
47
+ 'domain ' + repr(domain) + ' ADOPT gate references undeclared '
48
+ 'categories ' + repr(missing) + '; declared: ' + repr(sorted(known)))
49
+ return declared
50
+ known = taxonomy_categories(domain)
51
+ if not known:
52
+ raise ValueError('domain ' + repr(domain) + ' declares no outcome categories')
53
+ return (next(iter(known)),)
54
+
55
+
56
+ def outcome_category(domain: str, value: str, primary: tuple[str, ...]) -> str | None:
57
+ """Resolve an outcome value to its category, or None when unresolvable.
58
+
59
+ The outcomes table stores CATEGORY buckets (learning / task_performance /
60
+ process / risk, plus each domain's own buckets), while V1 packs store raw
61
+ taxonomy tokens. Accept a category directly and, for a token, resolve it
62
+ through the registry. An unknown value returns None so callers fail
63
+ closed instead of silently treating it as decision-grade evidence.
64
+ """
65
+ from engine.taxonomy import TaxonomyError, category_of
66
+
67
+ if value in primary:
68
+ return value
69
+ try:
70
+ return category_of(domain, value)
71
+ except (TaxonomyError, ValueError):
72
+ return None
73
+
74
+
75
+ def decision_action(*, confidence_label: str, decisive_relations: dict[str, str],
76
+ has_direct_primary_evidence: bool) -> str:
77
+ """Gate-enforced decision action (uppercase four-state).
78
+
79
+ REJECT requires usable direct opposition evidence (an independent Study
80
+ folded to oppose_adoption). Low/Insufficient can never yield ADOPT.
81
+ ADOPT additionally requires direct evidence on the domain's PRIMARY
82
+ outcome category: High + decisive support WITHOUT such evidence downgrades
83
+ to PILOT - task performance and procedural efficiency are not
84
+ decision-grade effects. Moderate + decisive support -> PILOT; otherwise
85
+ INSUFFICIENT_EVIDENCE.
86
+ """
87
+ has_oppose = any(r == "oppose_adoption" for r in decisive_relations.values())
88
+ has_support = any(r == "support_adoption" for r in decisive_relations.values())
89
+ if has_oppose:
90
+ return "REJECT"
91
+ if (confidence_label == ADOPT_REQUIRED_LABEL and has_support
92
+ and has_direct_primary_evidence):
93
+ return "ADOPT"
94
+ if confidence_label in ("High", "Moderate") and has_support:
95
+ return "PILOT"
96
+ return "INSUFFICIENT_EVIDENCE"
@@ -54,7 +54,11 @@ class EvidenceNode:
54
54
  outcome_dimension: str = OutcomeDimension.GENERAL_MEASURE
55
55
  claim_id: Optional[str] = None
56
56
  outcome_id: Optional[str] = None
57
- effect_size: Dict[str, Any] = field(default_factory=lambda: {"metric": "Hedges g", "value": 0.0, "ci_lower": 0.0, "ci_upper": 0.0, "p_value": 0.05})
57
+ # A missing effect must stay missing: the old default fabricated a
58
+ # g = 0.0 with a p = 0.05, which would serialise as a real (and
59
+ # false) "no effect" result. Callers that need a number must supply
60
+ # one; the extractors already suppress charts when it is absent.
61
+ effect_size: Optional[Dict[str, Any]] = None
58
62
  sample_size: int = 0
59
63
  sample_description: str = ""
60
64
  study_design: str = "Quasi-Experimental" # RCT, Quasi-Experimental DID, Meta-Analysis, Observational
@@ -292,9 +296,9 @@ class EvidenceGraph:
292
296
  for ev in self.evidence.values():
293
297
  paper = self.papers.get(ev.paper_id)
294
298
  study_label = f"{paper.authors[0] if paper and paper.authors else ev.paper_id} ({paper.year if paper else ''})"
295
- effect_val = ev.effect_size.get("value", 0.0)
296
- ci_l = ev.effect_size.get("ci_lower")
297
- ci_u = ev.effect_size.get("ci_upper")
299
+ effect_val = (ev.effect_size or {}).get("value") or 0.0
300
+ ci_l = (ev.effect_size or {}).get("ci_lower")
301
+ ci_u = (ev.effect_size or {}).get("ci_upper")
298
302
  has_ci = ci_l is not None and ci_u is not None and float(ci_u) >= float(ci_l)
299
303
  points.append({
300
304
  "evidence_id": ev.evidence_id,
@@ -327,13 +331,13 @@ class EvidenceGraph:
327
331
  precision_counts = {"reported_ci": 0, "derived_from_sample_size": 0}
328
332
  excluded_no_precision = 0
329
333
  for n in nodes:
330
- eff = n.effect_size.get("value")
334
+ eff = (n.effect_size or {}).get("value")
331
335
  if eff is None or math.isnan(float(eff)) or math.isinf(float(eff)):
332
336
  continue
333
337
 
334
338
  # Statistical variance derivation (Borenstein et al. 2009)
335
- ci_l = n.effect_size.get("ci_lower")
336
- ci_u = n.effect_size.get("ci_upper")
339
+ ci_l = (n.effect_size or {}).get("ci_lower")
340
+ ci_u = (n.effect_size or {}).get("ci_upper")
337
341
  if ci_l is not None and ci_u is not None and float(ci_u) > float(ci_l):
338
342
  se = (float(ci_u) - float(ci_l)) / (2.0 * 1.95996)
339
343
  precision_counts["reported_ci"] += 1
@@ -422,7 +426,7 @@ class EvidenceGraph:
422
426
  "quote": p.summary,
423
427
  })
424
428
  for ev in self.evidence.values():
425
- effect_val = ev.effect_size.get("value", 0.0)
429
+ effect_val = (ev.effect_size or {}).get("value") or 0.0
426
430
  symbol_size = max(18, min(45, int(18 + abs(effect_val) * 20)))
427
431
  nodes.append({
428
432
  "id": ev.evidence_id,
@@ -433,8 +437,8 @@ class EvidenceGraph:
433
437
  "dimension": ev.outcome_dimension,
434
438
  "direction": ev.direction,
435
439
  "effect_size": effect_val,
436
- "ci_lower": ev.effect_size.get("ci_lower", "N/A"),
437
- "ci_upper": ev.effect_size.get("ci_upper", "N/A"),
440
+ "ci_lower": (ev.effect_size or {}).get("ci_lower", "N/A"),
441
+ "ci_upper": (ev.effect_size or {}).get("ci_upper", "N/A"),
438
442
  "sample_size": ev.sample_size,
439
443
  "wwc_rating": ev.wwc_rating,
440
444
  "quote": ev.key_quote,
@@ -12,8 +12,7 @@ v4 领域包机制:domains/ 注册表 + 领域契约加载 + frame 校验。
12
12
  education 域只是"指向现有契约"的注册:不新增任何逻辑路径、不引入新 schema
13
13
  或新校验器。领域选择(domain select)由主 agent 接 CLI 完成,引擎层不做选择。
14
14
 
15
- 路径解析:当前按仓库布局(domains/ 在仓库根目录)解析;wheel 安装场景的
16
- share/ 回退留给后续步骤(pyproject data-files 未包含 domains/)。
15
+ 路径解析:支持仓库、独立 Skill 与 wheel 的 share/eduevidence 资源布局。
17
16
  """
18
17
 
19
18
  from __future__ import annotations
@@ -22,7 +21,9 @@ import json
22
21
  from pathlib import Path
23
22
  from typing import Any
24
23
 
25
- REPO_ROOT = Path(__file__).resolve().parent.parent
24
+ from engine._resources import resource_root
25
+
26
+ REPO_ROOT = resource_root()
26
27
 
27
28
 
28
29
  def _resolve_domains_dir() -> Path:
@@ -114,7 +115,7 @@ def _validate_contracts(entry: dict) -> None:
114
115
 
115
116
  - frame_schema / outcome_taxonomy / methodology_checklist:文件存在且
116
117
  为可解析 JSON(指针引用另校验指针内容);
117
- - golds_dir / references_dir:目录存在(null 视为"无此契约",跳过)。
118
+ - references_dir:目录存在;golds_dir 属于独立 evaluator 资源,不是研究运行依赖。
118
119
  """
119
120
  domain_id = entry["id"]
120
121
 
@@ -142,7 +143,8 @@ def _validate_contracts(entry: dict) -> None:
142
143
  check_file("frame_schema")
143
144
  check_file("outcome_taxonomy")
144
145
  check_file("methodology_checklist")
145
- check_dir("golds_dir")
146
+ # Evaluation annotations (including holdout answers) are intentionally absent
147
+ # from shipped Skills. Benchmark consumers validate their own input corpus.
146
148
  check_dir("references_dir")
147
149
 
148
150