eduevidence 5.2.0 → 6.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +54 -42
- package/README.zh-CN.md +51 -28
- package/SKILL.md +390 -133
- package/agents/openai.yaml +4 -0
- package/assets/readme/controlled-execution.svg +34 -0
- package/assets/readme/logo.png +0 -0
- package/assets/readme/research-workflow.svg +56 -0
- package/assets/readme/studio-graph.png +0 -0
- package/assets/readme/studio-overview.png +0 -0
- package/assets/readme/studio-reports.png +0 -0
- package/autoevolve/config.yaml +17 -0
- package/autoevolve/program.md +25 -0
- package/autoevolve/protected.manifest.yaml +34 -0
- package/benchmarks/adversarial/cases.jsonl +7 -0
- package/benchmarks/evidence-library.json +5268 -0
- package/benchmarks/partitions.json +8 -0
- package/docs/architecture.md +220 -0
- package/docs/autoresearch-evolution-plan.md +2903 -0
- package/docs/autoresearch-implementation-status.md +101 -0
- package/docs/demo-storyboard.md +20 -0
- package/docs/demo-workplace-ai.md +92 -0
- package/docs/demo.md +32 -0
- package/docs/install-guide.md +150 -0
- package/docs/orchestration-role-model.md +1254 -0
- package/docs/release-closeout/README.md +17 -0
- package/docs/release-closeout/frontend-acceptance.md +23 -0
- package/docs/release-closeout/issues.md +19 -0
- package/docs/release-closeout/verification.md +28 -0
- package/docs/release-contract.md +108 -0
- package/docs/research-studio-guide.zh-CN.md +166 -0
- package/eduevidence_cli.py +17 -11
- package/engine/_resources.py +13 -0
- package/engine/autoevolve/__init__.py +3 -0
- package/engine/autoevolve/agent_view.py +167 -0
- package/engine/autoevolve/core.py +357 -0
- package/engine/autoevolve/events.py +11 -0
- package/engine/autoevolve/git_workspace.py +77 -0
- package/engine/autoevolve/projection.py +23 -0
- package/engine/autoevolve/runner.py +413 -0
- package/engine/autoevolve/trust.py +146 -0
- package/engine/autoresearch/__init__.py +6 -0
- package/engine/autoresearch/commit.py +132 -0
- package/engine/autoresearch/contracts.py +126 -0
- package/engine/autoresearch/controller.py +207 -0
- package/engine/autoresearch/events.py +12 -0
- package/engine/autoresearch/gap_priority.py +168 -0
- package/engine/autoresearch/projection.py +30 -0
- package/engine/autoresearch/research_memory.py +59 -0
- package/engine/autoresearch/saturation.py +91 -0
- package/engine/briefs.py +2 -1
- package/engine/capabilities.py +1 -0
- package/engine/contracts.py +3 -1
- package/engine/evidencecore.py +7 -5
- package/engine/gaps.py +90 -51
- package/engine/judge_pack.py +65 -0
- package/engine/library_builtin.py +3 -1
- package/engine/living.py +2 -1
- package/engine/meta_synthesis.py +3 -1
- package/engine/orchestration.py +460 -0
- package/engine/pilot.py +2 -1
- package/engine/project.py +2 -2
- package/engine/research_service.py +113 -0
- package/engine/studio_read_model.py +400 -0
- package/engine/tribunal.py +1 -2
- package/engine/update.py +1 -0
- package/engine/versions.py +1 -1
- package/engine/worker_result.py +109 -0
- package/engine/workflows.py +70 -0
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1720 -0
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
- package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
- package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
- package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/frame.json +48 -0
- package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
- package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
- package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +219 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1720 -0
- package/examples/ai-coding-assistant-evidence/result.json +1453 -0
- package/examples/ai-coding-assistant-evidence/result.zh.json +1453 -0
- package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +103 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -0
- package/examples/workplace-ai-assistant/evaluation.json +19 -0
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
- package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
- package/examples/workplace-ai-assistant/frame.json +41 -0
- package/examples/workplace-ai-assistant/intervention.json +27 -0
- package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
- package/examples/workplace-ai-assistant/methodology.json +60 -0
- package/examples/workplace-ai-assistant/report_spec.json +55 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2484 -0
- package/examples/workplace-ai-assistant/result.json +553 -0
- package/examples/workplace-ai-assistant/result.zh.json +553 -0
- package/examples/workplace-ai-assistant/search_log.json +19 -0
- package/examples/workplace-ai-assistant/sources.jsonl +3 -0
- package/examples/workplace-ai-assistant/validation_result.json +9 -0
- package/examples/workplace-ai-assistant/verdict.json +52 -0
- package/install.sh +7 -7
- package/integrations/orchestration_dispatch.py +146 -0
- package/package.json +37 -3
- package/pyproject.toml +11 -20
- package/references/autoresearch.md +30 -0
- package/references/evaluation-policy.md +24 -0
- package/references/orchestration.md +22 -0
- package/references/scientific-invariants.md +19 -0
- package/retrieval/audit.py +154 -0
- package/schemas/intervention.schema.json +106 -21
- package/schemas/report-result.schema.json +9 -1
- package/schemas/v2/project.schema.json +2 -2
- package/schemas/v2/run.schema.json +1 -1
- package/schemas/vNext/autoevolve-session.schema.json +1 -0
- package/schemas/vNext/eval-snapshot.schema.json +1 -0
- package/schemas/vNext/execution-plan.schema.json +1 -0
- package/schemas/vNext/gap-priority.schema.json +1 -0
- package/schemas/vNext/negative-search-record.schema.json +1 -0
- package/schemas/vNext/research-iteration.schema.json +1 -0
- package/schemas/vNext/research-strategy.schema.json +1 -0
- package/schemas/vNext/skill-experiment.schema.json +1 -0
- package/schemas/vNext/task-spec.schema.json +1 -0
- package/schemas/vNext/worker-result.schema.json +1 -0
- package/scripts/benchmark_judge.py +2 -2
- package/scripts/benchmark_v3.py +26 -43
- package/scripts/build_esl_artifacts.py +2 -2
- package/scripts/build_evidence_library.py +2 -2
- package/scripts/build_gh_pages.py +98 -0
- package/scripts/build_readme_diagrams.py +72 -0
- package/scripts/build_report_variants.py +85 -0
- package/scripts/check_autoresearch_invariants.py +95 -0
- package/scripts/daily_evolve.py +30 -0
- package/scripts/dashboard_server.py +130 -101
- package/scripts/did_regression.py +5 -30
- package/scripts/enrich_projects_human_and_lieflat.py +1 -1
- package/scripts/generate_metrics.py +4 -3
- package/scripts/generate_new_projects.py +1 -1
- package/scripts/orchestrator.py +172 -18
- package/scripts/rebake_all_5themes.py +1 -2
- package/scripts/research_auto_cli.py +475 -0
- package/scripts/run_workspace.py +17 -7
- package/scripts/search_provenance.py +64 -0
- package/scripts/serve_web.py +9 -10
- package/scripts/skill_lint.py +1 -1
- package/scripts/skill_payload.py +78 -0
- package/scripts/validate_schema.py +15 -1
- package/scripts/vnext_cli.py +133 -0
- package/setup.py +12 -0
- package/skill/roles/registry.yaml +45 -0
- package/skill/sub-skills/report-generation/SKILL.md +12 -6
- package/skill/task-briefs/applicability.md +3 -0
- package/skill/task-briefs/projection.md +3 -0
- package/skill/workflows/decision-and-pilot.md +10 -0
- package/skill/workflows/evaluate-and-update.md +10 -0
- package/skill/workflows/evidence-review.md +13 -0
- package/visualization/eduevidence-report/assets/base.css +2 -2
- package/visualization/eduevidence-report/assets/reader.css +752 -0
- package/visualization/eduevidence-report/assets/reader.js +132 -0
- package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
- package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
- package/visualization/eduevidence-report/scripts/build_report.py +58 -65
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +24 -100
- package/visualization/eduevidence-report/themes/academic.css +1 -1
- package/visualization/eduevidence-report/themes/claude.css +1 -1
- package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
- package/visualization/eduevidence-report/themes/datalab.css +2 -2
- package/visualization/eduevidence-report/themes/presentation.css +2 -2
- package/web/README.md +18 -0
- package/web/index.html +53 -0
- package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
- package/web/studio/assets/index-CzXocaGv.css +1 -0
- package/web/studio/assets/index-pa7jD7n4.js +230 -0
- package/web/studio/config.json +1 -0
- package/web/studio/index.html +14 -0
- package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
- package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/bias.cpython-312.pyc +0 -0
- package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
- package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
- package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
- package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
- package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
- package/engine/__pycache__/events.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
- package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
- package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
- package/engine/__pycache__/ids.cpython-312.pyc +0 -0
- package/engine/__pycache__/library.cpython-312.pyc +0 -0
- package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
- package/engine/__pycache__/living.cpython-312.pyc +0 -0
- package/engine/__pycache__/log.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/migration.cpython-312.pyc +0 -0
- package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
- package/engine/__pycache__/paths.cpython-312.pyc +0 -0
- package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
- package/engine/__pycache__/planner.cpython-312.pyc +0 -0
- package/engine/__pycache__/project.cpython-312.pyc +0 -0
- package/engine/__pycache__/projections.cpython-312.pyc +0 -0
- package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
- package/engine/__pycache__/run.cpython-312.pyc +0 -0
- package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
- package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
- package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
- package/engine/__pycache__/update.cpython-312.pyc +0 -0
- package/engine/__pycache__/versions.cpython-312.pyc +0 -0
- package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
- package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
- package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
- package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
- package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
- package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
- package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
- package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
- package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
- package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
- package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
- package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
- package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
- package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
- package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
- package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
package/install.sh
CHANGED
|
@@ -28,7 +28,7 @@ REPO_URL="https://github.com/37chengshan/eduevidence"
|
|
|
28
28
|
SKILL_NAME="eduevidence"
|
|
29
29
|
# Skill 本体(运行协议 + 确定性脚本 + 检索/集成层 + 展示层)。
|
|
30
30
|
# retrieval/ 与 integrations/ 会被 scripts/ 直接 import;visualization/ 负责最终 HTML 渲染。
|
|
31
|
-
SKILL_PAYLOAD=(SKILL.md engine domains skill references schemas scripts retrieval integrations visualization)
|
|
31
|
+
SKILL_PAYLOAD=(SKILL.md agents engine domains skill references schemas scripts retrieval integrations visualization web examples docs autoevolve eduevidence_cli.py)
|
|
32
32
|
# Agent MCP 声明文件:安装完成后写入 AGENT_MCP_INSTALLED=1,供
|
|
33
33
|
# integrations/agent_mcp.py 作为 env 后备来源读取(真实环境变量优先于该文件)。
|
|
34
34
|
AGENT_MCP_ENV_FILE="${AGENT_MCP_ENV_FILE:-$HOME/.eduevidence/env}"
|
|
@@ -149,7 +149,7 @@ host_skill_root() {
|
|
|
149
149
|
|
|
150
150
|
# ---------- 通用提示词(方式三:宿主不在支持列表时交给任意 AI) ----------
|
|
151
151
|
UNIVERSAL_PROMPT="请把 https://github.com/37chengshan/eduevidence 仓库中的 EduEvidence 安装为 skill:
|
|
152
|
-
1. 将仓库根目录的 SKILL.md、skill/、
|
|
152
|
+
1. 将仓库根目录的 完整分发目录(含 SKILL.md、engine/、domains/、skill/、schemas/、scripts/、web/studio/ 等)复制到你的 skill 目录
|
|
153
153
|
(如 ~/.claude/skills/eduevidence/、~/.omp/agent/skills/eduevidence/、~/.agents/skills/eduevidence/ 等),
|
|
154
154
|
或按你的 skill 装载机制导入;
|
|
155
155
|
2. 安装完成后确认能读取 SKILL.md,并能运行 scripts/ 下的确定性脚本;
|
|
@@ -268,13 +268,13 @@ local_setup() {
|
|
|
268
268
|
# 5. 自检:Schema 校验 + 报告渲染
|
|
269
269
|
echo "==> 自检:Schema 校验"
|
|
270
270
|
python scripts/validate_schema.py --schema schemas/verdict.schema.json \
|
|
271
|
-
--data examples/ai-coding-assistant/verdict.json
|
|
271
|
+
--data examples/ai-coding-assistant-evidence/verdict.json
|
|
272
272
|
python scripts/validate_schema.py --schema schemas/evidence.schema.json \
|
|
273
|
-
--data examples/ai-coding-assistant/evidence.jsonl
|
|
273
|
+
--data examples/ai-coding-assistant-evidence/evidence.jsonl
|
|
274
274
|
|
|
275
275
|
echo "==> 自检:渲染双语 HTML 报告"
|
|
276
276
|
python visualization/eduevidence-report/scripts/build_report.py \
|
|
277
|
-
--result examples/ai-coding-assistant/result.json \
|
|
277
|
+
--result examples/ai-coding-assistant-evidence/result.json \
|
|
278
278
|
--out /tmp/eduevidence-smoke.html
|
|
279
279
|
rm -f /tmp/eduevidence-smoke.html
|
|
280
280
|
|
|
@@ -291,7 +291,7 @@ local_setup() {
|
|
|
291
291
|
local_finish() {
|
|
292
292
|
echo ""
|
|
293
293
|
echo "安装完成。下一步:"
|
|
294
|
-
echo " 1. 查看示例报告: open examples/ai-coding-assistant/EduEvidence_Report.html"
|
|
294
|
+
echo " 1. 查看示例报告: open examples/ai-coding-assistant-evidence/EduEvidence_Report.html"
|
|
295
295
|
echo " 2. 渲染自己的 result.json(需同时准备 result.zh.json 中文平行数据):"
|
|
296
296
|
echo " python visualization/eduevidence-report/scripts/build_report.py \\"
|
|
297
297
|
echo " --result <你的 result.json> --out REPORT.html"
|
|
@@ -315,7 +315,7 @@ install_to_dir() {
|
|
|
315
315
|
rm -rf "$dest"
|
|
316
316
|
fi
|
|
317
317
|
mkdir -p "$dest"
|
|
318
|
-
|
|
318
|
+
python3 scripts/skill_payload.py "$PWD" "$dest"
|
|
319
319
|
echo " ✅ 已安装: $dest"
|
|
320
320
|
}
|
|
321
321
|
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""TaskSpec-aware dispatch and acceptance adapter for Agent MCP.
|
|
2
|
+
|
|
3
|
+
Agent MCP remains the only implementation of CLI/model approval and spawn
|
|
4
|
+
payload construction. EduEvidence adds two scientific contract gates around
|
|
5
|
+
it:
|
|
6
|
+
|
|
7
|
+
1. every delegated worker must carry a dispatch-ready TaskSpec;
|
|
8
|
+
2. every returned worker payload must be reconstructed and validated in the
|
|
9
|
+
main process before any artifact can enter Judge context.
|
|
10
|
+
|
|
11
|
+
Workers are staging-only. Raw host output is never a Judge input.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from typing import Any, Callable
|
|
16
|
+
|
|
17
|
+
from engine.orchestration import ExecutionMode, TaskSpec
|
|
18
|
+
from engine.worker_result import (
|
|
19
|
+
ArtifactValidator,
|
|
20
|
+
WorkerResult,
|
|
21
|
+
require_validated_artifacts_for_judge,
|
|
22
|
+
validate_worker_output,
|
|
23
|
+
)
|
|
24
|
+
from integrations.agent_mcp import safe_spawn
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
TASKSPEC_REQUIRED = "TASKSPEC_REQUIRED"
|
|
28
|
+
TASKSPEC_INVALID = "TASKSPEC_INVALID"
|
|
29
|
+
WORKER_OUTPUT_REJECTED = "WORKER_OUTPUT_REJECTED"
|
|
30
|
+
HostExecutor = Callable[[dict[str, Any]], dict[str, Any]]
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def dispatch_task(
|
|
34
|
+
task: TaskSpec,
|
|
35
|
+
prompt: str,
|
|
36
|
+
approval: dict[str, Any] | None,
|
|
37
|
+
*,
|
|
38
|
+
target_cli: str | None = None,
|
|
39
|
+
model: str | None = None,
|
|
40
|
+
allowed_clis: list[str] | None = None,
|
|
41
|
+
cwd: str = ".",
|
|
42
|
+
permission_mode: str = "plan",
|
|
43
|
+
context_mode: str = "compact",
|
|
44
|
+
summary_chars: int | None = None,
|
|
45
|
+
) -> dict[str, Any]:
|
|
46
|
+
"""Validate a dispatch-ready TaskSpec, then pass it through safe_spawn()."""
|
|
47
|
+
if not isinstance(task, TaskSpec):
|
|
48
|
+
return {
|
|
49
|
+
"status": TASKSPEC_REQUIRED,
|
|
50
|
+
"spawn_call": None,
|
|
51
|
+
"reason": "subagent dispatch requires a TaskSpec",
|
|
52
|
+
}
|
|
53
|
+
try:
|
|
54
|
+
task.validate_for_dispatch()
|
|
55
|
+
except ValueError as exc:
|
|
56
|
+
return {
|
|
57
|
+
"status": TASKSPEC_INVALID,
|
|
58
|
+
"spawn_call": None,
|
|
59
|
+
"reason": str(exc),
|
|
60
|
+
}
|
|
61
|
+
if task.execution_mode is not ExecutionMode.DELEGATED:
|
|
62
|
+
return {
|
|
63
|
+
"status": TASKSPEC_INVALID,
|
|
64
|
+
"spawn_call": None,
|
|
65
|
+
"reason": "only delegated TaskSpecs may be sent to Agent MCP",
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
worker_prompt = f"{task.to_prompt_contract()}\n\nWORKER INSTRUCTIONS:\n{prompt.strip()}"
|
|
69
|
+
result = safe_spawn(
|
|
70
|
+
task.role,
|
|
71
|
+
worker_prompt,
|
|
72
|
+
approval,
|
|
73
|
+
target_cli=target_cli,
|
|
74
|
+
model=model,
|
|
75
|
+
allowed_clis=allowed_clis,
|
|
76
|
+
cwd=cwd,
|
|
77
|
+
permission_mode=permission_mode,
|
|
78
|
+
context_mode=context_mode,
|
|
79
|
+
summary_chars=summary_chars,
|
|
80
|
+
timeout_seconds=task.timeout_seconds,
|
|
81
|
+
token_budget=task.token_budget,
|
|
82
|
+
)
|
|
83
|
+
if result.get("status") == "READY":
|
|
84
|
+
result["task_id"] = task.task_id
|
|
85
|
+
result["run_id"] = task.run_id
|
|
86
|
+
result["base_revision"] = task.base_revision
|
|
87
|
+
result["stage"] = task.stage
|
|
88
|
+
result["evidence_axis"] = task.evidence_axis
|
|
89
|
+
result["allowed_capabilities"] = list(task.allowed_capabilities)
|
|
90
|
+
result["expected_staging_outputs"] = list(task.expected_outputs)
|
|
91
|
+
result["output_contract"] = dict(task.output_contract)
|
|
92
|
+
return result
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def accept_worker_output(
|
|
96
|
+
task: TaskSpec,
|
|
97
|
+
raw_output: dict[str, Any],
|
|
98
|
+
*,
|
|
99
|
+
artifact_validator: ArtifactValidator | None = None,
|
|
100
|
+
) -> WorkerResult:
|
|
101
|
+
"""Main-process acceptance boundary for host/Agent-MCP worker output.
|
|
102
|
+
|
|
103
|
+
Worker self-attestation is ignored by `validate_worker_output`. Callers must
|
|
104
|
+
retain the originating TaskSpec and validate against that exact contract.
|
|
105
|
+
"""
|
|
106
|
+
return validate_worker_output(
|
|
107
|
+
task,
|
|
108
|
+
raw_output,
|
|
109
|
+
artifact_validator=artifact_validator,
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def execute_dispatched_task(
|
|
114
|
+
task: TaskSpec,
|
|
115
|
+
dispatch_result: dict[str, Any],
|
|
116
|
+
host_executor: HostExecutor,
|
|
117
|
+
*,
|
|
118
|
+
artifact_validator: ArtifactValidator | None = None,
|
|
119
|
+
) -> WorkerResult:
|
|
120
|
+
"""Execute one READY spawn call and immediately pass through acceptance.
|
|
121
|
+
|
|
122
|
+
This is the conformant host adapter. It intentionally does not expose a
|
|
123
|
+
helper that returns raw worker artifacts after execution.
|
|
124
|
+
"""
|
|
125
|
+
if dispatch_result.get("status") != "READY":
|
|
126
|
+
raise PermissionError("cannot execute a dispatch result that is not READY")
|
|
127
|
+
if dispatch_result.get("task_id") != task.task_id:
|
|
128
|
+
raise ValueError("dispatch result task_id does not match TaskSpec")
|
|
129
|
+
spawn_call = dispatch_result.get("spawn_call")
|
|
130
|
+
if not isinstance(spawn_call, dict):
|
|
131
|
+
raise ValueError("READY dispatch result must contain one spawn_call object")
|
|
132
|
+
raw = host_executor(dict(spawn_call))
|
|
133
|
+
if not isinstance(raw, dict):
|
|
134
|
+
raise ValueError("host executor must return one worker output object")
|
|
135
|
+
return accept_worker_output(
|
|
136
|
+
task,
|
|
137
|
+
raw,
|
|
138
|
+
artifact_validator=artifact_validator,
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def judge_artifacts(
|
|
143
|
+
results: list[WorkerResult] | tuple[WorkerResult, ...],
|
|
144
|
+
) -> list[dict[str, Any]]:
|
|
145
|
+
"""Return Judge inputs only when every worker result passed acceptance."""
|
|
146
|
+
return require_validated_artifacts_for_judge(results)
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "eduevidence",
|
|
3
|
-
"version": "
|
|
4
|
-
"description": "Evidence
|
|
3
|
+
"version": "6.0.0",
|
|
4
|
+
"description": "Evidence research and decision skill with education and organizational policy domains.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "EduEvidence Contributors",
|
|
7
7
|
"homepage": "https://github.com/37chengshan/eduevidence#readme",
|
|
@@ -42,9 +42,43 @@
|
|
|
42
42
|
"scripts/",
|
|
43
43
|
"retrieval/",
|
|
44
44
|
"integrations/",
|
|
45
|
-
"visualization/eduevidence-report/"
|
|
45
|
+
"visualization/eduevidence-report/",
|
|
46
|
+
"agents/",
|
|
47
|
+
"web/studio/",
|
|
48
|
+
"web/index.html",
|
|
49
|
+
"autoevolve/config.yaml",
|
|
50
|
+
"autoevolve/program.md",
|
|
51
|
+
"autoevolve/protected.manifest.yaml",
|
|
52
|
+
"setup.py",
|
|
53
|
+
"docs/architecture.md",
|
|
54
|
+
"docs/install-guide.md",
|
|
55
|
+
"docs/release-contract.md",
|
|
56
|
+
"docs/research-studio-guide.zh-CN.md",
|
|
57
|
+
"docs/autoresearch-evolution-plan.md",
|
|
58
|
+
"docs/orchestration-role-model.md",
|
|
59
|
+
"docs/autoresearch-implementation-status.md",
|
|
60
|
+
"benchmarks/evidence-library.json",
|
|
61
|
+
"benchmarks/partitions.json",
|
|
62
|
+
"benchmarks/adversarial/cases.jsonl",
|
|
63
|
+
"examples/ai-coding-assistant-evidence/*.json",
|
|
64
|
+
"examples/ai-coding-assistant-evidence/*.jsonl",
|
|
65
|
+
"examples/ai-coding-assistant-evidence/*.html",
|
|
66
|
+
"examples/ai-coding-assistant-evidence/reports-5themes/*.html",
|
|
67
|
+
"examples/workplace-ai-assistant/*.json",
|
|
68
|
+
"examples/workplace-ai-assistant/*.jsonl",
|
|
69
|
+
"examples/workplace-ai-assistant/*.html",
|
|
70
|
+
"examples/workplace-ai-assistant/reports-5themes/*.html",
|
|
71
|
+
"docs/demo-workplace-ai.md",
|
|
72
|
+
"docs/demo.md",
|
|
73
|
+
"docs/demo-storyboard.md",
|
|
74
|
+
"docs/release-closeout/*.md",
|
|
75
|
+
"assets/readme/"
|
|
46
76
|
],
|
|
47
77
|
"engines": {
|
|
48
78
|
"node": ">=18"
|
|
79
|
+
},
|
|
80
|
+
"publishConfig": {
|
|
81
|
+
"registry": "https://registry.npmjs.org/",
|
|
82
|
+
"access": "public"
|
|
49
83
|
}
|
|
50
84
|
}
|
package/pyproject.toml
CHANGED
|
@@ -4,8 +4,8 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "eduevidence"
|
|
7
|
-
version = "
|
|
8
|
-
description = "Evidence
|
|
7
|
+
version = "6.0.0"
|
|
8
|
+
description = "Evidence research and decision skill with education and organizational policy domains."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
11
11
|
license = { text = "MIT" }
|
|
@@ -24,32 +24,23 @@ dev = ["pytest>=7.0"]
|
|
|
24
24
|
eduevidence = "eduevidence_cli:main"
|
|
25
25
|
|
|
26
26
|
[tool.pytest.ini_options]
|
|
27
|
-
# scripts/ 只为收编历史遗留的红队压测 scripts/test_adversarial_empirical.py
|
|
28
|
-
# (E5:该文件是全仓唯一 test_*.py 命名的脚本,纳入默认收集后不再"死测试")。
|
|
29
27
|
testpaths = ["tests", "scripts"]
|
|
30
28
|
python_files = ["test_*.py"]
|
|
31
29
|
addopts = "-q"
|
|
32
30
|
filterwarnings = ["ignore::pytest.PytestReturnNotNoneWarning:scripts.*"]
|
|
33
31
|
|
|
34
|
-
# wheel 自包含 CLI:eduevidence_cli + engine + scripts(V1/V2 命令实现与
|
|
35
|
-
# validate_schema)+ schemas 数据文件(data-files 安装到 share/eduevidence/,
|
|
36
|
-
# engine.contracts / orchestrator 按回退路径解析)。Skill 本体(SKILL.md +
|
|
37
|
-
# skill/ + references/ + retrieval/ + integrations/ + visualization/ +
|
|
38
|
-
# assets/)由 install.sh 或源码包完整分发。
|
|
39
32
|
[tool.setuptools]
|
|
40
33
|
py-modules = ["eduevidence_cli"]
|
|
41
|
-
packages = ["engine", "scripts"]
|
|
42
34
|
|
|
43
|
-
[tool.setuptools.
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
"share/eduevidence/schemas/v3" = ["schemas/v3/*.json"]
|
|
47
|
-
"share/eduevidence/schemas/v4" = ["schemas/v4/*.json"]
|
|
48
|
-
"share/eduevidence/domains" = ["domains/manifest.json", "domains/education/manifest.json", "domains/education/outcome_taxonomy.json", "domains/policy/manifest.json", "domains/policy/frame.schema.json", "domains/policy/outcome_taxonomy.json", "domains/policy/methodology_checklist.json"]
|
|
49
|
-
"share/eduevidence/domains/policy/references" = ["domains/policy/references/*.md"]
|
|
50
|
-
"share/eduevidence/benchmarks" = ["benchmarks/evidence-library.json"]
|
|
35
|
+
[tool.setuptools.packages.find]
|
|
36
|
+
where = ["."]
|
|
37
|
+
include = ["engine*", "scripts*", "retrieval*", "integrations*"]
|
|
51
38
|
|
|
52
39
|
[tool.ruff.lint]
|
|
53
|
-
# E5 最小守门集:语法层错误 + 未定义/重定义名。风格类规则不设(存量代码
|
|
54
|
-
# 不做一次性大规模格式化,避免淹没 review)。
|
|
55
40
|
select = ["E9", "F63", "F7", "F82"]
|
|
41
|
+
|
|
42
|
+
[tool.ruff.lint.per-file-ignores]
|
|
43
|
+
# build_report.py enables postponed annotations; its legacy Optional annotation
|
|
44
|
+
# is not evaluated at runtime. Keep this F821 exception narrowly scoped rather
|
|
45
|
+
# than weakening undefined-name checks for the repository.
|
|
46
|
+
"visualization/eduevidence-report/scripts/build_report.py" = ["F821"]
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# Evidence Autoresearch
|
|
2
|
+
|
|
3
|
+
Use this reference for bounded autonomous evidence acquisition inside an existing EduEvidence Project.
|
|
4
|
+
|
|
5
|
+
## Loop
|
|
6
|
+
|
|
7
|
+
`GraphRevision + DecisionSnapshot + KnowledgeGaps → rank DVI → choose ONE gap → ONE falsifiable strategy → bounded execution → validate staging evidence → single-writer graph commit or no-gain log → re-adjudicate → next gap`.
|
|
8
|
+
|
|
9
|
+
## Hard rules
|
|
10
|
+
|
|
11
|
+
- A KnowledgeGap must already be grounded in the Evidence Graph.
|
|
12
|
+
- DVI is an ordinal HIGH/MEDIUM/LOW prioritization heuristic, never EVPI/EVSI or a probability.
|
|
13
|
+
- Run one primary research hypothesis per ResearchIteration.
|
|
14
|
+
- Validated evidence is append-only whether supportive, contradictory, neutral, or null.
|
|
15
|
+
- A no-gain iteration creates ResearchIteration memory but no GraphRevision.
|
|
16
|
+
- Negative search results may only state that no eligible evidence was found within the recorded search scope.
|
|
17
|
+
- Stop on resolved gap, loss of decision sensitivity, budget exhaustion, saturation, tool blockage, user stop, or need for new empirical evidence.
|
|
18
|
+
- Literature → Pilot requires HIGH DVI + decision materiality + unresolved gap + search saturation + ethical/operational feasibility.
|
|
19
|
+
- New decision outputs created by autonomous refresh remain candidate decisions until an appropriate human/review gate accepts them.
|
|
20
|
+
|
|
21
|
+
## CLI
|
|
22
|
+
|
|
23
|
+
```text
|
|
24
|
+
eduevidence research auto step --project <id>
|
|
25
|
+
eduevidence research auto step --project <id> --outcome-file <validated-staging.json>
|
|
26
|
+
eduevidence research auto status --project <id>
|
|
27
|
+
eduevidence research auto stop --project <id>
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
Without an execution artifact, `step` emits the selected GapPriority and ResearchStrategy and stops at `awaiting_execution`; it never fabricates search results.
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# Skill Autoresearch Evaluation Policy
|
|
2
|
+
|
|
3
|
+
Evaluate candidate EduEvidence implementations with **constraint-first Pareto evaluation**, never a single LLM score.
|
|
4
|
+
|
|
5
|
+
## Order
|
|
6
|
+
|
|
7
|
+
1. **L0 hard gates** — schema validity, provenance, graph integrity, study identity, no false precision, no unsupported ADOPT, synthetic/real separation, grounded gaps, protected integrity, privacy. Any failure = REJECT.
|
|
8
|
+
2. **L1 scientific correctness** — outcome separation, citation support, contradiction precision/recall, scope/decision calibration, methodology issue detection, gap correctness. Material regression = REJECT.
|
|
9
|
+
3. **L2 research quality** — direct evidence gain, unique eligible evidence yield, counter-evidence yield, applicability coverage, gap-resolution and saturation efficiency.
|
|
10
|
+
4. **L3 robustness** — S/M/L, domains, model families, repeated runs, provider variation, missing/adversarial conditions.
|
|
11
|
+
5. **L4 efficiency** — token, cost, latency, search/fetch calls, subagent count, parallel speedup.
|
|
12
|
+
6. **L5 simplicity** — LOC, loaded Skill context, branches, dependencies, maintenance surface. Equal evidence quality prefers the simpler candidate.
|
|
13
|
+
|
|
14
|
+
## Promotion
|
|
15
|
+
|
|
16
|
+
- Hard gate fail or scientific regression → `REJECT`.
|
|
17
|
+
- Candidate improvement within empirical noise → `RETEST`.
|
|
18
|
+
- Material quality improvement without core regression → `KEEP`.
|
|
19
|
+
- Real Pareto trade-off → `HUMAN_REVIEW`.
|
|
20
|
+
- Protected mutation → `INVALID` before scoring.
|
|
21
|
+
|
|
22
|
+
Use at least 3 repeated empirical runs, preferably 5, for stochastic model evaluation. DEV is visible to candidate experiments. HOLDOUT and ADVERSARIAL are evaluator-only promotion inputs. Temporal evaluation is timestamped and not permanent gold.
|
|
23
|
+
|
|
24
|
+
Skill Autoresearch operates only on fixture/benchmark research state and an `autoresearch/<run-tag>` branch/worktree. It never changes real user evidence state, merges main, releases, deploys, or launches a human-subject study automatically.
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
# Orchestration
|
|
2
|
+
|
|
3
|
+
Treat these concepts as distinct:
|
|
4
|
+
|
|
5
|
+
`Protocol Stage ≠ Scientific Role ≠ Capability ≠ Worker/Subagent ≠ Model/CLI`.
|
|
6
|
+
|
|
7
|
+
A Scientific Role is an accountability boundary. It does not imply a permanent agent. A Capability is reusable implementation. A Worker/Subagent is a temporary execution instance. Model/CLI selection is an execution adapter governed by existing user approval.
|
|
8
|
+
|
|
9
|
+
## Default topology
|
|
10
|
+
|
|
11
|
+
- **S**: lead-only, zero delegated workers.
|
|
12
|
+
- **M**: selectively delegate independent direct/counter evidence acquisition and, when useful, an independent Skeptic.
|
|
13
|
+
- **L**: split retrieval by evidence axis (direct causal; transfer/retention; null/negative/risk; applicability/freshness), then deterministic merge, optional independent Skeptic/Method Reviewer, single-writer commit, Judge, and high-impact independent final review.
|
|
14
|
+
- Hard parallel cap: 6. Never recursive swarm.
|
|
15
|
+
|
|
16
|
+
## TaskSpec and dispatch
|
|
17
|
+
|
|
18
|
+
Every delegated worker requires a validated TaskSpec. Workers are read-only against canonical project state and return staging artifacts only. Agent MCP delegation must pass through the TaskSpec dispatch adapter and the existing `safe_spawn()` approval gate. Do not route by provider name; providers are tools, while evidence axes are epistemic objectives.
|
|
19
|
+
|
|
20
|
+
## Single Writer
|
|
21
|
+
|
|
22
|
+
Parallelize independent evidence acquisition and analysis; serialize canonical state transitions. Only the lead/single-writer path may commit GraphRevision, DecisionSnapshot, persistent KnowledgeGap state, StudyDesign, PilotRun, or AnalysisRun. Judges consume validated artifacts, never unvalidated worker prose.
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# EduEvidence Scientific Invariants
|
|
2
|
+
|
|
3
|
+
These rules are protected scientific constraints. Autoresearch may optimize the research process, but must never optimize the evidence set toward a preferred conclusion.
|
|
4
|
+
|
|
5
|
+
1. **Optimize for decision integrity, not answer confidence.**
|
|
6
|
+
2. **Optimize the research process, never the conclusion.**
|
|
7
|
+
3. **Every iteration must improve the evidence state, improve the research system, or teach why an attempted path failed.**
|
|
8
|
+
4. Validated evidence is append-only regardless of whether it supports, contradicts, is null, or is neutral toward the current decision.
|
|
9
|
+
5. Search snippets are discovery metadata, never direct evidence.
|
|
10
|
+
6. Task performance is not learning; direct learning/retention/independent-transfer evidence is required before ADOPT for education interventions.
|
|
11
|
+
7. Missing uncertainty statistics must never be fabricated.
|
|
12
|
+
8. Causal estimators fail closed on invalid or unidentified designs.
|
|
13
|
+
9. StudyDesign requires an explicit evidence-grounded KnowledgeGap.
|
|
14
|
+
10. Negative search results describe the bounded search scope only; never infer that evidence does not exist globally.
|
|
15
|
+
11. Evidence Autoresearch may update project research state but may not modify Skill/repository code.
|
|
16
|
+
12. Skill Autoresearch may modify approved mutable repository surfaces but may not touch real user research state.
|
|
17
|
+
13. Holdout/gold/evaluator/schema/scientific-invariant surfaces are protected from automatic mutation.
|
|
18
|
+
14. Canonical project state has a single writer. Subagents return staging artifacts only.
|
|
19
|
+
15. No autonomous human-subject study launch, policy deployment, main-branch merge, or release.
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
"""Auditable, bounded search execution for evidence retrieval.
|
|
2
|
+
|
|
3
|
+
Provider implementations remain deliberately separate. This module records
|
|
4
|
+
the query intent, every provider attempt, screening decisions and saturation
|
|
5
|
+
stop reason so a search result is never mistaken for unobserved provenance.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import csv
|
|
10
|
+
import json
|
|
11
|
+
import time
|
|
12
|
+
from dataclasses import asdict, dataclass, field
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Any, Iterable
|
|
15
|
+
|
|
16
|
+
from retrieval.search import SearchHit
|
|
17
|
+
from retrieval.source import parse_doi_from_url, title_fingerprint
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class SearchQuery:
|
|
22
|
+
query_id: str
|
|
23
|
+
query: str
|
|
24
|
+
purpose: str # core | expansion | counter_evidence | citation_chain
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass(frozen=True)
|
|
28
|
+
class SearchPlan:
|
|
29
|
+
question: str
|
|
30
|
+
domain: str
|
|
31
|
+
concepts: tuple[str, ...]
|
|
32
|
+
synonyms: tuple[str, ...]
|
|
33
|
+
inclusion_criteria: tuple[str, ...]
|
|
34
|
+
exclusion_criteria: tuple[str, ...]
|
|
35
|
+
queries: tuple[SearchQuery, ...]
|
|
36
|
+
provider_budget: int = 10
|
|
37
|
+
policy_version: str = "2026.09"
|
|
38
|
+
|
|
39
|
+
@classmethod
|
|
40
|
+
def from_question(cls, question: str, *, domain: str = "education",
|
|
41
|
+
concepts: Iterable[str] = (), synonyms: Iterable[str] = ()) -> "SearchPlan":
|
|
42
|
+
terms = tuple(dict.fromkeys(x.strip() for x in (*concepts, *synonyms) if x.strip()))
|
|
43
|
+
core = " ".join(terms) or question
|
|
44
|
+
return cls(question, domain, tuple(concepts), tuple(synonyms), (), (), (
|
|
45
|
+
SearchQuery("Q1", core, "core"),
|
|
46
|
+
SearchQuery("Q2", f"{core} systematic review OR meta-analysis", "expansion"),
|
|
47
|
+
SearchQuery("Q3", f"{core} null negative harm bias limitation", "counter_evidence"),
|
|
48
|
+
))
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@dataclass
|
|
52
|
+
class SearchAttempt:
|
|
53
|
+
attempt_id: str
|
|
54
|
+
query_id: str
|
|
55
|
+
provider: str
|
|
56
|
+
started_at: float
|
|
57
|
+
ended_at: float
|
|
58
|
+
status: str
|
|
59
|
+
result_count: int
|
|
60
|
+
error: str = ""
|
|
61
|
+
retry_index: int = 0
|
|
62
|
+
|
|
63
|
+
def to_dict(self) -> dict[str, Any]:
|
|
64
|
+
value = asdict(self)
|
|
65
|
+
value["latency_ms"] = round((self.ended_at - self.started_at) * 1000, 2)
|
|
66
|
+
return value
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _hit_key(hit: SearchHit) -> tuple[str, str, str]:
|
|
70
|
+
doi = (hit.doi or parse_doi_from_url(hit.url) or "").lower()
|
|
71
|
+
return doi, hit.url.rstrip("/").lower(), title_fingerprint(hit.title)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def dedupe_hits(hits: Iterable[SearchHit]) -> list[SearchHit]:
|
|
75
|
+
"""Deduplicate DOI, canonical URL and normalized title, retaining priority."""
|
|
76
|
+
seen: set[tuple[str, str]] = set()
|
|
77
|
+
kept: list[SearchHit] = []
|
|
78
|
+
for hit in sorted(hits, key=lambda item: (item.score, item.citation_count or 0), reverse=True):
|
|
79
|
+
doi, url, title = _hit_key(hit)
|
|
80
|
+
keys = [("doi", doi), ("url", url), ("title", title)]
|
|
81
|
+
nonempty = [key for key in keys if key[1]]
|
|
82
|
+
if any(key in seen for key in nonempty):
|
|
83
|
+
continue
|
|
84
|
+
seen.update(nonempty)
|
|
85
|
+
kept.append(hit)
|
|
86
|
+
return kept
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
class AuditedSearchExecutor:
|
|
90
|
+
"""Runs an explicit plan with bounded retries and durable audit exports."""
|
|
91
|
+
|
|
92
|
+
def __init__(self, providers: Iterable[Any], *, max_retries: int = 1):
|
|
93
|
+
self.providers = list(providers)
|
|
94
|
+
self.max_retries = max_retries
|
|
95
|
+
|
|
96
|
+
def execute(self, plan: SearchPlan, output_dir: Path, *, limit: int = 10) -> list[SearchHit]:
|
|
97
|
+
output_dir = Path(output_dir)
|
|
98
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
99
|
+
attempts: list[SearchAttempt] = []
|
|
100
|
+
hits: list[SearchHit] = []
|
|
101
|
+
attempted: set[tuple[str, str]] = set()
|
|
102
|
+
counter_hits = 0
|
|
103
|
+
for query in plan.queries:
|
|
104
|
+
for provider in self.providers:
|
|
105
|
+
name = getattr(provider, "name", provider.__class__.__name__)
|
|
106
|
+
fingerprint = (name, query.query)
|
|
107
|
+
if fingerprint in attempted:
|
|
108
|
+
continue
|
|
109
|
+
attempted.add(fingerprint)
|
|
110
|
+
for retry in range(self.max_retries + 1):
|
|
111
|
+
started = time.time()
|
|
112
|
+
try:
|
|
113
|
+
result = provider.search(query.query, limit=limit)
|
|
114
|
+
ended = time.time()
|
|
115
|
+
attempts.append(SearchAttempt(
|
|
116
|
+
f"A-{len(attempts) + 1:04d}", query.query_id, name, started, ended,
|
|
117
|
+
"success", len(result), retry_index=retry,
|
|
118
|
+
))
|
|
119
|
+
hits.extend(result)
|
|
120
|
+
if query.purpose == "counter_evidence":
|
|
121
|
+
counter_hits += len(result)
|
|
122
|
+
break
|
|
123
|
+
except Exception as exc: # failures are recorded, never swallowed
|
|
124
|
+
ended = time.time()
|
|
125
|
+
attempts.append(SearchAttempt(
|
|
126
|
+
f"A-{len(attempts) + 1:04d}", query.query_id, name, started, ended,
|
|
127
|
+
"failed", 0, error=f"{type(exc).__name__}: {exc}", retry_index=retry,
|
|
128
|
+
))
|
|
129
|
+
unique = dedupe_hits(hits)
|
|
130
|
+
self._write_exports(output_dir, plan, attempts, unique, counter_hits)
|
|
131
|
+
return unique[:limit]
|
|
132
|
+
|
|
133
|
+
@staticmethod
|
|
134
|
+
def _write_exports(output_dir: Path, plan: SearchPlan, attempts: list[SearchAttempt],
|
|
135
|
+
hits: list[SearchHit], counter_hits: int) -> None:
|
|
136
|
+
(output_dir / "search-provenance.json").write_text(json.dumps({
|
|
137
|
+
"plan": {**asdict(plan), "queries": [asdict(q) for q in plan.queries]},
|
|
138
|
+
"search_completed": True,
|
|
139
|
+
"counter_evidence_queries_executed": sum(q.purpose == "counter_evidence" for q in plan.queries),
|
|
140
|
+
"counter_evidence_hit_count": counter_hits,
|
|
141
|
+
"stop_reason": "planned_queries_completed",
|
|
142
|
+
}, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
|
143
|
+
(output_dir / "search-attempts.jsonl").write_text(
|
|
144
|
+
"".join(json.dumps(item.to_dict(), ensure_ascii=False) + "\n" for item in attempts), encoding="utf-8")
|
|
145
|
+
with (output_dir / "source-screening.csv").open("w", newline="", encoding="utf-8") as fh:
|
|
146
|
+
writer = csv.DictWriter(fh, fieldnames=["title", "doi", "url", "provider", "year", "screening_status", "reason"])
|
|
147
|
+
writer.writeheader()
|
|
148
|
+
for hit in hits:
|
|
149
|
+
writer.writerow({"title": hit.title, "doi": hit.doi or parse_doi_from_url(hit.url) or "", "url": hit.url,
|
|
150
|
+
"provider": hit.provider, "year": hit.year or "", "screening_status": "candidate",
|
|
151
|
+
"reason": "discovery metadata only; fetch and validation required before evidence extraction"})
|
|
152
|
+
with (output_dir / "exclusion-log.csv").open("w", newline="", encoding="utf-8") as fh:
|
|
153
|
+
writer = csv.DictWriter(fh, fieldnames=["identifier", "reason"])
|
|
154
|
+
writer.writeheader()
|