eduevidence 5.2.0 → 6.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +105 -0
- package/README.md +142 -75
- package/README.zh-CN.md +73 -30
- package/SKILL.md +397 -131
- package/agents/openai.yaml +4 -0
- package/assets/readme/controlled-execution.svg +34 -0
- package/assets/readme/landing-tour.gif +0 -0
- package/assets/readme/logo.png +0 -0
- package/assets/readme/research-workflow.svg +56 -0
- package/assets/readme/studio-graph.png +0 -0
- package/assets/readme/studio-overview.png +0 -0
- package/assets/readme/studio-reports.png +0 -0
- package/assets/readme/studio-tour.gif +0 -0
- package/autoevolve/config.yaml +17 -0
- package/autoevolve/program.md +25 -0
- package/autoevolve/protected.manifest.yaml +34 -0
- package/benchmarks/adversarial/cases.jsonl +7 -0
- package/benchmarks/evidence-library.json +5268 -0
- package/benchmarks/partitions.json +8 -0
- package/bin/eduevidence.js +2 -1
- package/docs/architecture.md +496 -0
- package/docs/autoresearch-evolution-plan.md +2903 -0
- package/docs/autoresearch-implementation-status.md +101 -0
- package/docs/demo-storyboard.md +20 -0
- package/docs/demo-workplace-ai.md +92 -0
- package/docs/demo.md +32 -0
- package/docs/install-guide.md +150 -0
- package/docs/orchestration-role-model.md +1254 -0
- package/docs/release-closeout/README.md +17 -0
- package/docs/release-closeout/frontend-acceptance.md +23 -0
- package/docs/release-closeout/issues.md +19 -0
- package/docs/release-closeout/verification.md +28 -0
- package/docs/release-contract.md +108 -0
- package/docs/research-studio-guide.zh-CN.md +166 -0
- package/docs/sciverse-api.md +125 -0
- package/eduevidence_cli.py +29 -13
- package/engine/_resources.py +13 -0
- package/engine/autoevolve/__init__.py +3 -0
- package/engine/autoevolve/agent_view.py +167 -0
- package/engine/autoevolve/core.py +357 -0
- package/engine/autoevolve/events.py +11 -0
- package/engine/autoevolve/git_workspace.py +77 -0
- package/engine/autoevolve/projection.py +23 -0
- package/engine/autoevolve/runner.py +413 -0
- package/engine/autoevolve/trust.py +146 -0
- package/engine/autoresearch/__init__.py +6 -0
- package/engine/autoresearch/commit.py +132 -0
- package/engine/autoresearch/contracts.py +126 -0
- package/engine/autoresearch/controller.py +207 -0
- package/engine/autoresearch/events.py +12 -0
- package/engine/autoresearch/gap_priority.py +168 -0
- package/engine/autoresearch/projection.py +30 -0
- package/engine/autoresearch/research_memory.py +59 -0
- package/engine/autoresearch/saturation.py +91 -0
- package/engine/briefs.py +2 -1
- package/engine/capabilities.py +1 -0
- package/engine/contracts.py +3 -1
- package/engine/decision_policy.py +96 -0
- package/engine/evidence_graph.py +14 -10
- package/engine/evidencecore.py +7 -5
- package/engine/gaps.py +132 -73
- package/engine/ids.py +2 -0
- package/engine/judge_pack.py +65 -0
- package/engine/library.py +6 -2
- package/engine/library_builtin.py +3 -1
- package/engine/living.py +36 -5
- package/engine/meta_synthesis.py +3 -1
- package/engine/migration.py +88 -3
- package/engine/orchestration.py +460 -0
- package/engine/paths.py +2 -0
- package/engine/pilot.py +36 -33
- package/engine/project.py +2 -2
- package/engine/research_service.py +113 -0
- package/engine/studio_read_model.py +400 -0
- package/engine/taxonomy.py +211 -0
- package/engine/tribunal.py +44 -33
- package/engine/update.py +1 -0
- package/engine/versions.py +1 -1
- package/engine/worker_result.py +109 -0
- package/engine/workflows.py +70 -0
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +2934 -0
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
- package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
- package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
- package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
- package/examples/ai-coding-assistant-evidence/frame.json +48 -0
- package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
- package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
- package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
- package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +230 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/result.json +1457 -0
- package/examples/ai-coding-assistant-evidence/result.zh.json +1457 -0
- package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
- package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +107 -0
- package/examples/spaced-retrieval-practice/applicability.json +14 -0
- package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
- package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
- package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
- package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
- package/examples/spaced-retrieval-practice/frame.json +58 -0
- package/examples/spaced-retrieval-practice/gate_report.json +101 -0
- package/examples/spaced-retrieval-practice/methodology.json +78 -0
- package/examples/spaced-retrieval-practice/report_spec.json +212 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/result.json +942 -0
- package/examples/spaced-retrieval-practice/result.zh.json +942 -0
- package/examples/spaced-retrieval-practice/skeptic.json +70 -0
- package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
- package/examples/spaced-retrieval-practice/verdict.json +93 -0
- package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -0
- package/examples/workplace-ai-assistant/evaluation.json +19 -0
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
- package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
- package/examples/workplace-ai-assistant/final_verdict.json +78 -0
- package/examples/workplace-ai-assistant/frame.json +41 -0
- package/examples/workplace-ai-assistant/gate_report.json +101 -0
- package/examples/workplace-ai-assistant/intervention.json +27 -0
- package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
- package/examples/workplace-ai-assistant/methodology.json +60 -0
- package/examples/workplace-ai-assistant/report_spec.json +224 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/result.json +615 -0
- package/examples/workplace-ai-assistant/result.zh.json +615 -0
- package/examples/workplace-ai-assistant/search_log.json +19 -0
- package/examples/workplace-ai-assistant/skeptic.json +72 -0
- package/examples/workplace-ai-assistant/sources.jsonl +3 -0
- package/examples/workplace-ai-assistant/validation_result.json +9 -0
- package/examples/workplace-ai-assistant/verdict.json +78 -0
- package/install.sh +7 -7
- package/integrations/agent_mcp.py +2 -2
- package/integrations/orchestration_dispatch.py +146 -0
- package/package.json +46 -3
- package/pyproject.toml +14 -22
- package/references/autoresearch.md +30 -0
- package/references/evaluation-policy.md +24 -0
- package/references/orchestration.md +22 -0
- package/references/report-copy-style.md +67 -0
- package/references/retrieval-compliance.md +75 -0
- package/references/retrieval-protocol.md +20 -0
- package/references/scientific-invariants.md +19 -0
- package/retrieval/audit.py +178 -0
- package/retrieval/fetch.py +96 -0
- package/retrieval/sciverse.py +398 -0
- package/retrieval/search.py +47 -7
- package/schemas/applicability.schema.json +94 -0
- package/schemas/chart-spec.schema.json +10 -3
- package/schemas/evidence.schema.json +316 -43
- package/schemas/fetch-result.schema.json +2 -1
- package/schemas/intervention.schema.json +106 -21
- package/schemas/report-result.schema.json +12 -4
- package/schemas/report-spec.schema.json +98 -100
- package/schemas/skeptic.schema.json +86 -0
- package/schemas/source.schema.json +21 -2
- package/schemas/v2/finding.schema.json +5 -1
- package/schemas/v2/methodology-audit.schema.json +5 -1
- package/schemas/v2/outcome.schema.json +28 -5
- package/schemas/v2/project.schema.json +2 -2
- package/schemas/v2/run.schema.json +1 -1
- package/schemas/v2/study.schema.json +5 -1
- package/schemas/vNext/autoevolve-session.schema.json +34 -0
- package/schemas/vNext/eval-snapshot.schema.json +77 -0
- package/schemas/vNext/execution-plan.schema.json +50 -0
- package/schemas/vNext/gap-priority.schema.json +54 -0
- package/schemas/vNext/negative-search-record.schema.json +68 -0
- package/schemas/vNext/research-iteration.schema.json +87 -0
- package/schemas/vNext/research-strategy.schema.json +62 -0
- package/schemas/vNext/skill-experiment.schema.json +90 -0
- package/schemas/vNext/task-spec.schema.json +156 -0
- package/schemas/vNext/worker-result.schema.json +60 -0
- package/schemas/verdict.schema.json +164 -28
- package/scripts/benchmark_judge.py +2 -2
- package/scripts/benchmark_v3.py +26 -43
- package/scripts/build_esl_artifacts.py +4 -4
- package/scripts/build_evidence_library.py +2 -2
- package/scripts/build_gh_pages.py +98 -0
- package/scripts/build_readme_diagrams.py +72 -0
- package/scripts/build_report_variants.py +101 -0
- package/scripts/build_result.py +74 -9
- package/scripts/check_autoresearch_invariants.py +95 -0
- package/scripts/check_package_parity.py +85 -0
- package/scripts/check_protocol_alignment.py +375 -0
- package/scripts/check_versioned_schemas.py +254 -0
- package/scripts/claim_audit.py +13 -8
- package/scripts/compute_confidence.py +10 -0
- package/scripts/daily_evolve.py +30 -0
- package/scripts/dashboard_server.py +130 -101
- package/scripts/did_regression.py +17 -32
- package/scripts/enrich_projects_human_and_lieflat.py +1 -1
- package/scripts/evidence_score.py +5 -2
- package/scripts/generate_metrics.py +4 -3
- package/scripts/generate_new_projects.py +5 -5
- package/scripts/orchestrator.py +286 -36
- package/scripts/pre_verdict_gate.py +224 -26
- package/scripts/quickstart.py +18 -2
- package/scripts/rebake_all_5themes.py +1 -2
- package/scripts/research_auto_cli.py +475 -0
- package/scripts/run_workspace.py +24 -8
- package/scripts/search_provenance.py +64 -0
- package/scripts/serve_web.py +9 -10
- package/scripts/skill_lint.py +1 -1
- package/scripts/skill_payload.py +81 -0
- package/scripts/test_adversarial_empirical.py +26 -19
- package/scripts/validate_schema.py +46 -2
- package/scripts/vnext_cli.py +133 -0
- package/setup.py +12 -0
- package/skill/agents/evaluation-designer.md +20 -4
- package/skill/agents/evidence-analyst.md +19 -3
- package/skill/agents/evidence-judge.md +50 -2
- package/skill/agents/evidence-retriever.md +20 -3
- package/skill/agents/intervention-designer.md +20 -4
- package/skill/agents/method-reviewer.md +18 -2
- package/skill/agents/{education-planner.md → research-planner.md} +19 -3
- package/skill/agents/skeptic.md +18 -2
- package/skill/roles/registry.yaml +45 -0
- package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
- package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
- package/skill/sub-skills/data-analysis/SKILL.md +34 -15
- package/skill/sub-skills/ethics-review/SKILL.md +33 -10
- package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
- package/skill/sub-skills/evidence-review/SKILL.md +31 -12
- package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
- package/skill/sub-skills/literature-review/SKILL.md +35 -14
- package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
- package/skill/sub-skills/report-generation/SKILL.md +40 -6
- package/skill/sub-skills/research-planning/SKILL.md +41 -14
- package/skill/sub-skills/study-design/SKILL.md +30 -9
- package/skill/task-briefs/adjudicate.md +32 -7
- package/skill/task-briefs/applicability.md +38 -0
- package/skill/task-briefs/audit.md +32 -7
- package/skill/task-briefs/challenge.md +34 -5
- package/skill/task-briefs/evaluate.md +30 -5
- package/skill/task-briefs/extract.md +31 -8
- package/skill/task-briefs/frame.md +39 -10
- package/skill/task-briefs/intervene.md +32 -6
- package/skill/task-briefs/present.md +32 -8
- package/skill/task-briefs/projection.md +37 -0
- package/skill/task-briefs/retrieve.md +36 -6
- package/skill/workflows/decision-and-pilot.md +85 -0
- package/skill/workflows/evaluate-and-update.md +93 -0
- package/skill/workflows/evidence-review.md +117 -0
- package/visualization/eduevidence-report/assets/base.css +2 -2
- package/visualization/eduevidence-report/assets/reader.css +752 -0
- package/visualization/eduevidence-report/assets/reader.js +132 -0
- package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
- package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
- package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
- package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
- package/visualization/eduevidence-report/scripts/build_report.py +561 -121
- package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +371 -136
- package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
- package/visualization/eduevidence-report/themes/academic.css +1 -1
- package/visualization/eduevidence-report/themes/claude.css +1 -1
- package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
- package/visualization/eduevidence-report/themes/datalab.css +2 -2
- package/visualization/eduevidence-report/themes/presentation.css +2 -2
- package/web/README.md +18 -0
- package/web/architecture.html +14885 -0
- package/web/index.html +53 -0
- package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
- package/web/studio/assets/index-B8tkF44Q.css +1 -0
- package/web/studio/assets/index-CQ6Keoyc.js +230 -0
- package/web/studio/config.json +1 -0
- package/web/studio/index.html +14 -0
- package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
- package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/bias.cpython-312.pyc +0 -0
- package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
- package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
- package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
- package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
- package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
- package/engine/__pycache__/events.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
- package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
- package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
- package/engine/__pycache__/ids.cpython-312.pyc +0 -0
- package/engine/__pycache__/library.cpython-312.pyc +0 -0
- package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
- package/engine/__pycache__/living.cpython-312.pyc +0 -0
- package/engine/__pycache__/log.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/migration.cpython-312.pyc +0 -0
- package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
- package/engine/__pycache__/paths.cpython-312.pyc +0 -0
- package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
- package/engine/__pycache__/planner.cpython-312.pyc +0 -0
- package/engine/__pycache__/project.cpython-312.pyc +0 -0
- package/engine/__pycache__/projections.cpython-312.pyc +0 -0
- package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
- package/engine/__pycache__/run.cpython-312.pyc +0 -0
- package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
- package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
- package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
- package/engine/__pycache__/update.cpython-312.pyc +0 -0
- package/engine/__pycache__/versions.cpython-312.pyc +0 -0
- package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
- package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
- package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
- package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
- package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
- package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
- package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
- package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
- package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
- package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
- package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
- package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
- package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
- package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
- package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
- package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"""Atomic, append-only Graph commit for Evidence Autoresearch staging bundles.
|
|
2
|
+
|
|
3
|
+
One ResearchIteration may add several graph entity types, but it must create at
|
|
4
|
+
most one GraphRevision. Identical already-present entities are no-ops; the same
|
|
5
|
+
entity id with different scientific content is an append-only conflict rather
|
|
6
|
+
than an overwrite. The caller must supply the graph revision against which the
|
|
7
|
+
research request was created so stale results fail closed.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from engine.graph_store import GRAPH_TABLES, GraphMutation, GraphStore
|
|
14
|
+
|
|
15
|
+
_ID_KEYS = {
|
|
16
|
+
"sources": "source_id",
|
|
17
|
+
"studies": "study_id",
|
|
18
|
+
"findings": "finding_id",
|
|
19
|
+
"outcomes": "outcome_id",
|
|
20
|
+
"claims": "claim_id",
|
|
21
|
+
"evidence_links": "evidence_link_id",
|
|
22
|
+
"audits": "audit_id",
|
|
23
|
+
}
|
|
24
|
+
_VALID_SOURCE_STATUSES = {"valid", "accepted_partial"}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _rows(payload: dict[str, Any], table: str) -> list[dict]:
|
|
28
|
+
value = payload.get(table, [])
|
|
29
|
+
if value is None:
|
|
30
|
+
return []
|
|
31
|
+
if not isinstance(value, list) or any(not isinstance(row, dict) for row in value):
|
|
32
|
+
raise ValueError(f"staging bundle field {table!r} must be a list of objects")
|
|
33
|
+
return value
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def build_append_only_mutation(
|
|
37
|
+
store: GraphStore,
|
|
38
|
+
payload: dict[str, Any],
|
|
39
|
+
) -> tuple[GraphMutation, dict[str, list[str]]]:
|
|
40
|
+
"""Return only genuinely new entities plus their ids.
|
|
41
|
+
|
|
42
|
+
GraphStore performs schema and cross-entity validation on the final merged
|
|
43
|
+
snapshot. This function adds the Autoresearch-specific append-only and
|
|
44
|
+
validated-source gates before that atomic commit.
|
|
45
|
+
"""
|
|
46
|
+
upserts: dict[str, list[dict]] = {}
|
|
47
|
+
added: dict[str, list[str]] = {}
|
|
48
|
+
|
|
49
|
+
existing_tables = {table: store.read_table(table) for table in GRAPH_TABLES}
|
|
50
|
+
for table in GRAPH_TABLES:
|
|
51
|
+
id_key = _ID_KEYS[table]
|
|
52
|
+
existing = {row[id_key]: row for row in existing_tables[table]}
|
|
53
|
+
seen_incoming: dict[str, dict] = {}
|
|
54
|
+
fresh: list[dict] = []
|
|
55
|
+
fresh_ids: list[str] = []
|
|
56
|
+
for row in _rows(payload, table):
|
|
57
|
+
entity_id = row.get(id_key)
|
|
58
|
+
if not isinstance(entity_id, str) or not entity_id:
|
|
59
|
+
raise ValueError(f"{table} staging entity missing {id_key}")
|
|
60
|
+
prior_incoming = seen_incoming.get(entity_id)
|
|
61
|
+
if prior_incoming is not None:
|
|
62
|
+
if prior_incoming != row:
|
|
63
|
+
raise ValueError(
|
|
64
|
+
f"append-only conflict: duplicate incoming {table} id {entity_id} has different content"
|
|
65
|
+
)
|
|
66
|
+
continue
|
|
67
|
+
seen_incoming[entity_id] = row
|
|
68
|
+
prior = existing.get(entity_id)
|
|
69
|
+
if prior is not None:
|
|
70
|
+
if prior != row:
|
|
71
|
+
raise ValueError(
|
|
72
|
+
f"append-only conflict: {table} {entity_id} already exists with different content"
|
|
73
|
+
)
|
|
74
|
+
continue
|
|
75
|
+
if table == "sources" and row.get("validation_status") not in _VALID_SOURCE_STATUSES:
|
|
76
|
+
raise ValueError(
|
|
77
|
+
f"source {entity_id} has validation_status={row.get('validation_status')!r}; "
|
|
78
|
+
"only valid/accepted_partial sources may enter Evidence Autoresearch"
|
|
79
|
+
)
|
|
80
|
+
fresh.append(row)
|
|
81
|
+
fresh_ids.append(entity_id)
|
|
82
|
+
if fresh:
|
|
83
|
+
upserts[table] = fresh
|
|
84
|
+
added[table] = fresh_ids
|
|
85
|
+
|
|
86
|
+
# A newly added Study may only cite a validated existing/new Source. This
|
|
87
|
+
# preserves the Fetch -> Validate -> Extract gate in the atomic path.
|
|
88
|
+
source_status = {
|
|
89
|
+
row["source_id"]: row.get("validation_status")
|
|
90
|
+
for row in existing_tables["sources"]
|
|
91
|
+
}
|
|
92
|
+
for row in upserts.get("sources", []):
|
|
93
|
+
source_status[row["source_id"]] = row.get("validation_status")
|
|
94
|
+
for study in upserts.get("studies", []):
|
|
95
|
+
for source_id in study.get("source_ids", []):
|
|
96
|
+
status = source_status.get(source_id)
|
|
97
|
+
if status not in _VALID_SOURCE_STATUSES:
|
|
98
|
+
raise ValueError(
|
|
99
|
+
f"study {study['study_id']} references source {source_id} "
|
|
100
|
+
f"without validated provenance (status={status!r})"
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
return GraphMutation(upserts=upserts, retire_ids={}), added
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def commit_staging_bundle(
|
|
107
|
+
store: GraphStore,
|
|
108
|
+
*,
|
|
109
|
+
run_id: str,
|
|
110
|
+
expected_base_revision: int,
|
|
111
|
+
payload: dict[str, Any],
|
|
112
|
+
) -> int | None:
|
|
113
|
+
"""Atomically append one staging bundle, or return None for a true no-op."""
|
|
114
|
+
active = store.active_revision()
|
|
115
|
+
if active != expected_base_revision:
|
|
116
|
+
raise RuntimeError(
|
|
117
|
+
f"STALE_RESEARCH_STATE: request was based on graph revision "
|
|
118
|
+
f"{expected_base_revision}, active revision is {active}; re-plan before commit"
|
|
119
|
+
)
|
|
120
|
+
mutation, added = build_append_only_mutation(store, payload)
|
|
121
|
+
if not added:
|
|
122
|
+
return None
|
|
123
|
+
# Single Writer is an architectural invariant. Recheck immediately before
|
|
124
|
+
# commit so an intervening canonical transition is detected before write.
|
|
125
|
+
if store.active_revision() != expected_base_revision:
|
|
126
|
+
raise RuntimeError("STALE_RESEARCH_STATE: graph changed while validating staging bundle")
|
|
127
|
+
revision = store.commit(
|
|
128
|
+
run_id=run_id,
|
|
129
|
+
reason="autoresearch atomic validated evidence append",
|
|
130
|
+
mutation=mutation,
|
|
131
|
+
)
|
|
132
|
+
return revision.revision
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from dataclasses import dataclass, field, asdict
|
|
3
|
+
from datetime import datetime, timezone
|
|
4
|
+
from enum import Enum
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def utcnow() -> str:
|
|
9
|
+
return datetime.now(timezone.utc).isoformat()
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class ResearchExperimentType(str, Enum):
|
|
13
|
+
TARGETED_RETRIEVAL = "TARGETED_RETRIEVAL"
|
|
14
|
+
COUNTER_EVIDENCE_RETRIEVAL = "COUNTER_EVIDENCE_RETRIEVAL"
|
|
15
|
+
APPLICABILITY_RETRIEVAL = "APPLICABILITY_RETRIEVAL"
|
|
16
|
+
TEMPORAL_REFRESH = "TEMPORAL_REFRESH"
|
|
17
|
+
CITATION_CHAINING = "CITATION_CHAINING"
|
|
18
|
+
SCREENING_PRIORITY = "SCREENING_PRIORITY"
|
|
19
|
+
SOURCE_RECOVERY = "SOURCE_RECOVERY"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class IterationStatus(str, Enum):
|
|
23
|
+
COMPLETED_GAIN = "completed_gain"
|
|
24
|
+
COMPLETED_NO_GAIN = "completed_no_gain"
|
|
25
|
+
SEARCH_SATURATED = "search_saturated"
|
|
26
|
+
EMPIRICAL_NEEDED = "empirical_needed"
|
|
27
|
+
BUDGET_EXHAUSTED = "budget_exhausted"
|
|
28
|
+
TOOL_FAILURE = "tool_failure"
|
|
29
|
+
INVALID = "invalid"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass(frozen=True)
|
|
33
|
+
class ResearchBudget:
|
|
34
|
+
max_queries: int = 6
|
|
35
|
+
max_candidates: int = 30
|
|
36
|
+
max_fulltext_fetches: int = 12
|
|
37
|
+
|
|
38
|
+
def validate(self) -> None:
|
|
39
|
+
if min(self.max_queries, self.max_candidates, self.max_fulltext_fetches) < 0:
|
|
40
|
+
raise ValueError("research budget values must be non-negative")
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass(frozen=True)
|
|
44
|
+
class ResearchStrategy:
|
|
45
|
+
strategy_id: str
|
|
46
|
+
experiment_type: ResearchExperimentType
|
|
47
|
+
hypothesis: str
|
|
48
|
+
expected_gain: str
|
|
49
|
+
budget: ResearchBudget = field(default_factory=ResearchBudget)
|
|
50
|
+
|
|
51
|
+
def validate(self) -> None:
|
|
52
|
+
if not self.strategy_id.strip() or not self.hypothesis.strip() or not self.expected_gain.strip():
|
|
53
|
+
raise ValueError("strategy_id, hypothesis and expected_gain are required")
|
|
54
|
+
self.budget.validate()
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@dataclass(frozen=True)
|
|
58
|
+
class NegativeSearchRecord:
|
|
59
|
+
negative_search_id: str
|
|
60
|
+
research_iteration_id: str
|
|
61
|
+
gap_id: str
|
|
62
|
+
queries: tuple[str, ...]
|
|
63
|
+
providers: tuple[str, ...]
|
|
64
|
+
candidate_count: int
|
|
65
|
+
fetched_count: int
|
|
66
|
+
eligible_count: int
|
|
67
|
+
exclusion_reasons: dict[str, int] = field(default_factory=dict)
|
|
68
|
+
scope: dict[str, Any] = field(default_factory=dict)
|
|
69
|
+
searched_at: str = field(default_factory=utcnow)
|
|
70
|
+
conclusion: str = "no_eligible_evidence_found_within_search_scope"
|
|
71
|
+
|
|
72
|
+
def validate(self) -> None:
|
|
73
|
+
if self.eligible_count != 0:
|
|
74
|
+
raise ValueError("NegativeSearchRecord requires eligible_count == 0")
|
|
75
|
+
if min(self.candidate_count, self.fetched_count, self.eligible_count) < 0:
|
|
76
|
+
raise ValueError("counts must be non-negative")
|
|
77
|
+
if self.fetched_count > self.candidate_count:
|
|
78
|
+
raise ValueError("fetched_count cannot exceed candidate_count")
|
|
79
|
+
if self.conclusion != "no_eligible_evidence_found_within_search_scope":
|
|
80
|
+
raise ValueError("negative search conclusion must remain scope-bounded")
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@dataclass
|
|
84
|
+
class ResearchIteration:
|
|
85
|
+
iteration_id: str
|
|
86
|
+
project_id: str
|
|
87
|
+
base_graph_revision: int
|
|
88
|
+
gap_id: str
|
|
89
|
+
strategy: ResearchStrategy
|
|
90
|
+
gap_lineage_key: str | None = None
|
|
91
|
+
execution_plan_id: str | None = None
|
|
92
|
+
search_attempts: list[dict[str, Any]] = field(default_factory=list)
|
|
93
|
+
candidate_sources: list[str] = field(default_factory=list)
|
|
94
|
+
validated_evidence_ids: list[str] = field(default_factory=list)
|
|
95
|
+
negative_search_ids: list[str] = field(default_factory=list)
|
|
96
|
+
evidence_gain: dict[str, Any] = field(default_factory=dict)
|
|
97
|
+
new_graph_revision: int | None = None
|
|
98
|
+
decision_snapshot_id: str | None = None
|
|
99
|
+
status: IterationStatus | None = None
|
|
100
|
+
started_at: str = field(default_factory=utcnow)
|
|
101
|
+
completed_at: str | None = None
|
|
102
|
+
|
|
103
|
+
def validate(self) -> None:
|
|
104
|
+
self.strategy.validate()
|
|
105
|
+
if self.base_graph_revision < 0:
|
|
106
|
+
raise ValueError("base_graph_revision must be >= 0")
|
|
107
|
+
if self.gap_lineage_key is not None and not self.gap_lineage_key.startswith("KGK-"):
|
|
108
|
+
raise ValueError("gap_lineage_key must use the KGK- prefix")
|
|
109
|
+
if self.new_graph_revision is not None:
|
|
110
|
+
if not self.validated_evidence_ids:
|
|
111
|
+
raise ValueError("no-gain iteration must not create a GraphRevision")
|
|
112
|
+
if self.new_graph_revision <= self.base_graph_revision:
|
|
113
|
+
raise ValueError("new_graph_revision must advance base revision")
|
|
114
|
+
if self.status == IterationStatus.COMPLETED_NO_GAIN and self.new_graph_revision is not None:
|
|
115
|
+
raise ValueError("completed_no_gain cannot create GraphRevision")
|
|
116
|
+
|
|
117
|
+
def complete(self, status: IterationStatus) -> None:
|
|
118
|
+
self.status = status
|
|
119
|
+
self.completed_at = utcnow()
|
|
120
|
+
self.validate()
|
|
121
|
+
|
|
122
|
+
def as_dict(self) -> dict[str, Any]:
|
|
123
|
+
out = asdict(self)
|
|
124
|
+
out["strategy"]["experiment_type"] = self.strategy.experiment_type.value
|
|
125
|
+
out["status"] = self.status.value if self.status else None
|
|
126
|
+
return out
|
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from dataclasses import dataclass
|
|
3
|
+
from typing import Any, Callable
|
|
4
|
+
from .contracts import (
|
|
5
|
+
IterationStatus,
|
|
6
|
+
ResearchBudget,
|
|
7
|
+
ResearchExperimentType,
|
|
8
|
+
ResearchIteration,
|
|
9
|
+
ResearchStrategy,
|
|
10
|
+
)
|
|
11
|
+
from .gap_priority import GapPriority, rank_gaps
|
|
12
|
+
from .saturation import detect_saturation, transition_to_empirical
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class StepResult:
|
|
17
|
+
iteration: ResearchIteration
|
|
18
|
+
priority: GapPriority
|
|
19
|
+
next_action: str
|
|
20
|
+
rationale: tuple[str, ...]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def gap_lineage_key(gap: dict[str, Any]) -> str | None:
|
|
24
|
+
extensions = gap.get("extensions")
|
|
25
|
+
if not isinstance(extensions, dict):
|
|
26
|
+
return None
|
|
27
|
+
value = extensions.get("autoresearch_key")
|
|
28
|
+
if value is None:
|
|
29
|
+
return None
|
|
30
|
+
value = str(value).strip()
|
|
31
|
+
return value or None
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def history_matches_gap(row: dict[str, Any], gap: dict[str, Any]) -> bool:
|
|
35
|
+
"""Match research history across graph revisions without conflating gaps.
|
|
36
|
+
|
|
37
|
+
New iterations persist the stable KGK lineage key. Legacy rows that predate
|
|
38
|
+
the key remain readable and match only by revision-local gap_id.
|
|
39
|
+
"""
|
|
40
|
+
lineage = gap_lineage_key(gap)
|
|
41
|
+
row_lineage = row.get("gap_lineage_key")
|
|
42
|
+
if lineage and row_lineage:
|
|
43
|
+
return str(row_lineage) == lineage
|
|
44
|
+
return str(row.get("gap_id", "")) == str(gap.get("gap_id", ""))
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class EvidenceAutoresearchController:
|
|
48
|
+
"""Bounded gap-to-evidence loop with strategy memory."""
|
|
49
|
+
|
|
50
|
+
def __init__(self, *, max_iterations: int = 5):
|
|
51
|
+
if not 1 <= max_iterations <= 50:
|
|
52
|
+
raise ValueError("max_iterations must be 1..50")
|
|
53
|
+
self.max_iterations = max_iterations
|
|
54
|
+
|
|
55
|
+
def select_gap(self, gaps: list[dict[str, Any]], decision: dict[str, Any] | None = None) -> GapPriority:
|
|
56
|
+
ranked = rank_gaps(
|
|
57
|
+
[
|
|
58
|
+
g for g in gaps
|
|
59
|
+
if str(g.get("status", "open")).lower()
|
|
60
|
+
not in {"resolved", "low_decision_value", "search_saturated", "empirical_needed"}
|
|
61
|
+
],
|
|
62
|
+
decision=decision,
|
|
63
|
+
)
|
|
64
|
+
if not ranked:
|
|
65
|
+
raise ValueError("no unresolved KnowledgeGap available")
|
|
66
|
+
return ranked[0]
|
|
67
|
+
|
|
68
|
+
@staticmethod
|
|
69
|
+
def strategy_types_for(gap: dict[str, Any]) -> tuple[ResearchExperimentType, ...]:
|
|
70
|
+
gap_type = str(gap.get("gap_type", ""))
|
|
71
|
+
if gap_type == "unresolved_conflict":
|
|
72
|
+
return (
|
|
73
|
+
ResearchExperimentType.COUNTER_EVIDENCE_RETRIEVAL,
|
|
74
|
+
ResearchExperimentType.CITATION_CHAINING,
|
|
75
|
+
ResearchExperimentType.TARGETED_RETRIEVAL,
|
|
76
|
+
ResearchExperimentType.TEMPORAL_REFRESH,
|
|
77
|
+
)
|
|
78
|
+
if gap_type in {"population_gap", "context_gap"}:
|
|
79
|
+
return (
|
|
80
|
+
ResearchExperimentType.APPLICABILITY_RETRIEVAL,
|
|
81
|
+
ResearchExperimentType.TARGETED_RETRIEVAL,
|
|
82
|
+
ResearchExperimentType.CITATION_CHAINING,
|
|
83
|
+
ResearchExperimentType.TEMPORAL_REFRESH,
|
|
84
|
+
)
|
|
85
|
+
return (
|
|
86
|
+
ResearchExperimentType.TARGETED_RETRIEVAL,
|
|
87
|
+
ResearchExperimentType.CITATION_CHAINING,
|
|
88
|
+
ResearchExperimentType.TEMPORAL_REFRESH,
|
|
89
|
+
ResearchExperimentType.SOURCE_RECOVERY,
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
def build_strategy(
|
|
93
|
+
self,
|
|
94
|
+
priority: GapPriority,
|
|
95
|
+
gap: dict[str, Any],
|
|
96
|
+
history: list[dict[str, Any]] | None = None,
|
|
97
|
+
) -> ResearchStrategy:
|
|
98
|
+
history = history or []
|
|
99
|
+
attempted = {
|
|
100
|
+
str((row.get("strategy") or {}).get("experiment_type", ""))
|
|
101
|
+
for row in history
|
|
102
|
+
if history_matches_gap(row, gap)
|
|
103
|
+
}
|
|
104
|
+
choices = self.strategy_types_for(gap)
|
|
105
|
+
experiment_type = next(
|
|
106
|
+
(choice for choice in choices if choice.value not in attempted),
|
|
107
|
+
choices[-1],
|
|
108
|
+
)
|
|
109
|
+
gap_type = str(gap.get("gap_type", ""))
|
|
110
|
+
lineage = gap_lineage_key(gap)
|
|
111
|
+
hypothesis = (
|
|
112
|
+
f"{experiment_type.value} for {gap_type or 'gap'} will find "
|
|
113
|
+
f"decision-relevant evidence that materially improves directness "
|
|
114
|
+
f"or resolves {priority.gap_id}."
|
|
115
|
+
)
|
|
116
|
+
strategy_anchor = lineage or priority.gap_id
|
|
117
|
+
return ResearchStrategy(
|
|
118
|
+
f"STRAT-{strategy_anchor}-{experiment_type.value.lower()}",
|
|
119
|
+
experiment_type,
|
|
120
|
+
hypothesis,
|
|
121
|
+
"decision_relevant_evidence",
|
|
122
|
+
ResearchBudget(),
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
def step(
|
|
126
|
+
self,
|
|
127
|
+
*,
|
|
128
|
+
project_id: str,
|
|
129
|
+
base_graph_revision: int,
|
|
130
|
+
gaps: list[dict[str, Any]],
|
|
131
|
+
decision: dict[str, Any] | None,
|
|
132
|
+
history: list[dict[str, Any]],
|
|
133
|
+
executor: Callable[[ResearchStrategy, dict[str, Any]], dict[str, Any]],
|
|
134
|
+
graph_commit: Callable[[list[str]], int | None] | None = None,
|
|
135
|
+
decision_snapshot_id: str | None = None,
|
|
136
|
+
ethics_feasible: bool = False,
|
|
137
|
+
) -> StepResult:
|
|
138
|
+
priority = self.select_gap(gaps, decision)
|
|
139
|
+
gap = next(g for g in gaps if str(g.get("gap_id")) == priority.gap_id)
|
|
140
|
+
strategy = self.build_strategy(priority, gap, history)
|
|
141
|
+
iteration = ResearchIteration(
|
|
142
|
+
f"RIT-{len(history) + 1:04d}",
|
|
143
|
+
project_id,
|
|
144
|
+
base_graph_revision,
|
|
145
|
+
priority.gap_id,
|
|
146
|
+
strategy,
|
|
147
|
+
gap_lineage_key=gap_lineage_key(gap),
|
|
148
|
+
)
|
|
149
|
+
outcome = executor(strategy, gap)
|
|
150
|
+
if not isinstance(outcome, dict):
|
|
151
|
+
raise ValueError("research executor must return a staging JSON object")
|
|
152
|
+
if "resolved_gap_ids" in outcome:
|
|
153
|
+
raise ValueError(
|
|
154
|
+
"executor must not author KnowledgeGap RESOLVED state; append validated "
|
|
155
|
+
"evidence, re-derive gaps, and let the main adjudication engine resolve them"
|
|
156
|
+
)
|
|
157
|
+
valid = list(dict.fromkeys(outcome.get("validated_evidence_ids") or []))
|
|
158
|
+
iteration.validated_evidence_ids = valid
|
|
159
|
+
iteration.candidate_sources = list(outcome.get("candidate_sources") or [])
|
|
160
|
+
iteration.search_attempts = list(outcome.get("search_attempts") or [])
|
|
161
|
+
iteration.negative_search_ids = list(outcome.get("negative_search_ids") or [])
|
|
162
|
+
iteration.evidence_gain = dict(outcome.get("evidence_gain") or {})
|
|
163
|
+
|
|
164
|
+
if valid:
|
|
165
|
+
if graph_commit is None:
|
|
166
|
+
raise ValueError("validated evidence requires single-writer graph_commit callback")
|
|
167
|
+
committed_revision = graph_commit(valid)
|
|
168
|
+
if committed_revision is not None and committed_revision > base_graph_revision:
|
|
169
|
+
iteration.new_graph_revision = committed_revision
|
|
170
|
+
iteration.decision_snapshot_id = decision_snapshot_id
|
|
171
|
+
iteration.complete(IterationStatus.COMPLETED_GAIN)
|
|
172
|
+
return StepResult(
|
|
173
|
+
iteration,
|
|
174
|
+
priority,
|
|
175
|
+
"re_adjudicate",
|
|
176
|
+
("validated evidence appended; re-adjudication required before another research iteration",),
|
|
177
|
+
)
|
|
178
|
+
iteration.evidence_gain = {
|
|
179
|
+
**iteration.evidence_gain,
|
|
180
|
+
"duplicate_only": True,
|
|
181
|
+
"unique_eligible_evidence": 0,
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
iteration.complete(IterationStatus.COMPLETED_NO_GAIN)
|
|
185
|
+
combined = history + [iteration.as_dict()]
|
|
186
|
+
gap_history = [row for row in combined if history_matches_gap(row, gap)]
|
|
187
|
+
available = {item.value for item in self.strategy_types_for(gap)}
|
|
188
|
+
saturation = detect_saturation(gap_history, available_strategy_types=available)
|
|
189
|
+
empirical, reasons = transition_to_empirical(
|
|
190
|
+
dvi_band=priority.dvi_band.value,
|
|
191
|
+
decision_material=priority.decision_material,
|
|
192
|
+
unresolved=True,
|
|
193
|
+
saturation=saturation,
|
|
194
|
+
ethics_feasible=ethics_feasible,
|
|
195
|
+
)
|
|
196
|
+
if empirical:
|
|
197
|
+
iteration.status = IterationStatus.EMPIRICAL_NEEDED
|
|
198
|
+
return StepResult(iteration, priority, "empirical_evidence_needed", reasons)
|
|
199
|
+
if saturation.saturated:
|
|
200
|
+
iteration.status = IterationStatus.SEARCH_SATURATED
|
|
201
|
+
return StepResult(iteration, priority, "stop_search_saturated", saturation.rationale)
|
|
202
|
+
return StepResult(
|
|
203
|
+
iteration,
|
|
204
|
+
priority,
|
|
205
|
+
"next_iteration",
|
|
206
|
+
("no new validated evidence in this bounded iteration",),
|
|
207
|
+
)
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
EVIDENCE_AUTORESEARCH_EVENTS = (
|
|
2
|
+
"autoresearch.evidence.started",
|
|
3
|
+
"autoresearch.gap.ranked",
|
|
4
|
+
"autoresearch.hypothesis.created",
|
|
5
|
+
"autoresearch.iteration.started",
|
|
6
|
+
"autoresearch.search.completed",
|
|
7
|
+
"autoresearch.iteration.no_gain",
|
|
8
|
+
"autoresearch.iteration.evidence_gain",
|
|
9
|
+
"autoresearch.saturation.detected",
|
|
10
|
+
"autoresearch.empirical_needed",
|
|
11
|
+
"autoresearch.evidence.completed",
|
|
12
|
+
)
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from dataclasses import dataclass
|
|
3
|
+
from enum import Enum
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class Band(str, Enum):
|
|
8
|
+
HIGH = "HIGH"
|
|
9
|
+
MEDIUM = "MEDIUM"
|
|
10
|
+
LOW = "LOW"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
_LEVEL = {"low": 1, "medium": 2, "high": 3, "LOW": 1, "MEDIUM": 2, "HIGH": 3}
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass(frozen=True)
|
|
17
|
+
class GapPriority:
|
|
18
|
+
gap_id: str
|
|
19
|
+
dvi_band: Band
|
|
20
|
+
cost_band: Band
|
|
21
|
+
decision_material: bool
|
|
22
|
+
drivers: tuple[str, ...]
|
|
23
|
+
next_research_mode: str
|
|
24
|
+
score: int
|
|
25
|
+
|
|
26
|
+
def as_dict(self) -> dict[str, Any]:
|
|
27
|
+
return {
|
|
28
|
+
"gap_id": self.gap_id,
|
|
29
|
+
"dvi_band": self.dvi_band.value,
|
|
30
|
+
"cost_band": self.cost_band.value,
|
|
31
|
+
"decision_material": self.decision_material,
|
|
32
|
+
"drivers": list(self.drivers),
|
|
33
|
+
"next_research_mode": self.next_research_mode,
|
|
34
|
+
"score": self.score,
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _v(value: Any, default: int = 2) -> int:
|
|
39
|
+
if isinstance(value, int):
|
|
40
|
+
return max(1, min(3, value))
|
|
41
|
+
return _LEVEL.get(str(value), default)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _revision_number(value: Any, *, label: str) -> int:
|
|
45
|
+
if isinstance(value, bool):
|
|
46
|
+
raise ValueError(f"{label} must be an integer graph revision")
|
|
47
|
+
try:
|
|
48
|
+
revision = int(value)
|
|
49
|
+
except (TypeError, ValueError) as exc:
|
|
50
|
+
raise ValueError(f"{label} must be an integer graph revision") from exc
|
|
51
|
+
if revision < 0:
|
|
52
|
+
raise ValueError(f"{label} must be >= 0")
|
|
53
|
+
return revision
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _require_revision_bound_decision(
|
|
57
|
+
gap: dict[str, Any], decision: dict[str, Any] | None
|
|
58
|
+
) -> None:
|
|
59
|
+
"""Fail closed when a DecisionSnapshot and KnowledgeGap are from different revisions.
|
|
60
|
+
|
|
61
|
+
DVI may use the current DecisionSnapshot as a decision-sensitivity input. It
|
|
62
|
+
must never combine a stale decision with a newer gap (or vice versa), since
|
|
63
|
+
that could reorder the next research target using an obsolete boundary.
|
|
64
|
+
"""
|
|
65
|
+
if not decision:
|
|
66
|
+
return
|
|
67
|
+
if "graph_revision" not in decision:
|
|
68
|
+
raise ValueError(
|
|
69
|
+
"DecisionSnapshot used for DVI must include graph_revision"
|
|
70
|
+
)
|
|
71
|
+
if "derived_from_graph_revision" not in gap:
|
|
72
|
+
raise ValueError(
|
|
73
|
+
f"KnowledgeGap {gap.get('gap_id', '')!r} used for DVI must include "
|
|
74
|
+
"derived_from_graph_revision"
|
|
75
|
+
)
|
|
76
|
+
decision_revision = _revision_number(
|
|
77
|
+
decision.get("graph_revision"), label="DecisionSnapshot.graph_revision"
|
|
78
|
+
)
|
|
79
|
+
gap_revision = _revision_number(
|
|
80
|
+
gap.get("derived_from_graph_revision"),
|
|
81
|
+
label="KnowledgeGap.derived_from_graph_revision",
|
|
82
|
+
)
|
|
83
|
+
if decision_revision != gap_revision:
|
|
84
|
+
raise ValueError(
|
|
85
|
+
"stale DecisionSnapshot for DVI: "
|
|
86
|
+
f"decision revision {decision_revision} != gap revision {gap_revision}"
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _decision_material(gap: dict[str, Any]) -> bool:
|
|
91
|
+
"""Return a conservative, separately auditable materiality judgment.
|
|
92
|
+
|
|
93
|
+
Materiality is intentionally not derived from the DVI band. An explicit
|
|
94
|
+
boolean on the gap (or extensions.decision_material) wins. Otherwise only
|
|
95
|
+
gap types that can directly block/reverse an education decision are treated
|
|
96
|
+
as material by default.
|
|
97
|
+
"""
|
|
98
|
+
explicit = gap.get("decision_material")
|
|
99
|
+
if isinstance(explicit, bool):
|
|
100
|
+
return explicit
|
|
101
|
+
extensions = gap.get("extensions")
|
|
102
|
+
if isinstance(extensions, dict) and isinstance(extensions.get("decision_material"), bool):
|
|
103
|
+
return bool(extensions["decision_material"])
|
|
104
|
+
return str(gap.get("gap_type", "")) in {
|
|
105
|
+
"missing_transfer",
|
|
106
|
+
"missing_retention",
|
|
107
|
+
"unresolved_conflict",
|
|
108
|
+
"weak_causal_identification",
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def rank_gap(
|
|
113
|
+
gap: dict[str, Any],
|
|
114
|
+
*,
|
|
115
|
+
decision: dict[str, Any] | None = None,
|
|
116
|
+
expected_evidence_availability: str = "medium",
|
|
117
|
+
research_cost: str = "medium",
|
|
118
|
+
applicability_relevance: str = "medium",
|
|
119
|
+
risk_irreversibility: str = "medium",
|
|
120
|
+
) -> GapPriority:
|
|
121
|
+
"""Return an explainable ordinal research-priority band.
|
|
122
|
+
|
|
123
|
+
This is a conceptual DVI heuristic, not EVPI/EVSI and not a probability.
|
|
124
|
+
Decision materiality is a separate field and must not be inferred from the
|
|
125
|
+
resulting DVI band when deciding whether to bridge to empirical research.
|
|
126
|
+
"""
|
|
127
|
+
_require_revision_bound_decision(gap, decision)
|
|
128
|
+
decision = decision or {}
|
|
129
|
+
priority = _v(gap.get("priority", "medium"))
|
|
130
|
+
gap_type = str(gap.get("gap_type", ""))
|
|
131
|
+
material = _decision_material(gap)
|
|
132
|
+
decision_sensitive = 3 if material else priority
|
|
133
|
+
current_uncertainty = 3 if gap_type.startswith("missing_") or gap_type == "unresolved_conflict" else 2
|
|
134
|
+
directness_deficit = 3 if gap_type in {"missing_transfer", "missing_retention", "missing_outcome"} else 2
|
|
135
|
+
applicability = _v(applicability_relevance)
|
|
136
|
+
availability = _v(expected_evidence_availability)
|
|
137
|
+
cost = _v(research_cost)
|
|
138
|
+
risk = _v(risk_irreversibility)
|
|
139
|
+
score = decision_sensitive + current_uncertainty + directness_deficit + applicability + availability + risk - cost
|
|
140
|
+
if decision.get("recommended_action") in {"adopt", "ADOPT"} and gap_type in {"missing_transfer", "missing_retention"}:
|
|
141
|
+
score += 2
|
|
142
|
+
band = Band.HIGH if score >= 13 else Band.MEDIUM if score >= 9 else Band.LOW
|
|
143
|
+
cost_band = Band.HIGH if cost == 3 else Band.MEDIUM if cost == 2 else Band.LOW
|
|
144
|
+
drivers = (
|
|
145
|
+
f"gap_type={gap_type or 'unknown'}",
|
|
146
|
+
f"decision_material={material}",
|
|
147
|
+
f"decision_sensitivity={decision_sensitive}",
|
|
148
|
+
f"current_uncertainty={current_uncertainty}",
|
|
149
|
+
f"directness_deficit={directness_deficit}",
|
|
150
|
+
f"applicability_relevance={applicability}",
|
|
151
|
+
f"expected_evidence_availability={availability}",
|
|
152
|
+
f"risk_irreversibility={risk}",
|
|
153
|
+
f"research_cost={cost}",
|
|
154
|
+
)
|
|
155
|
+
mode = "secondary_evidence_search" if band is not Band.LOW else "defer"
|
|
156
|
+
return GapPriority(
|
|
157
|
+
str(gap.get("gap_id", "")),
|
|
158
|
+
band,
|
|
159
|
+
cost_band,
|
|
160
|
+
material,
|
|
161
|
+
drivers,
|
|
162
|
+
mode,
|
|
163
|
+
score,
|
|
164
|
+
)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def rank_gaps(gaps: list[dict[str, Any]], **kwargs: Any) -> list[GapPriority]:
|
|
168
|
+
return sorted((rank_gap(g, **kwargs) for g in gaps), key=lambda x: (-x.score, x.gap_id))
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from typing import Any
|
|
3
|
+
from .gap_priority import rank_gaps
|
|
4
|
+
from .saturation import detect_saturation
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def research_loop_projection(
|
|
8
|
+
*,
|
|
9
|
+
decision: dict[str, Any] | None,
|
|
10
|
+
gaps: list[dict[str, Any]],
|
|
11
|
+
iterations: list[dict[str, Any]],
|
|
12
|
+
revision: int,
|
|
13
|
+
) -> dict[str, Any]:
|
|
14
|
+
"""Projection-only state for Research Studio; never mutates canonical state."""
|
|
15
|
+
ranked = [item.as_dict() for item in rank_gaps(gaps, decision=decision)]
|
|
16
|
+
current = iterations[-1] if iterations else None
|
|
17
|
+
saturation = detect_saturation(iterations)
|
|
18
|
+
return {
|
|
19
|
+
"graph_revision": revision,
|
|
20
|
+
"decision": decision or {},
|
|
21
|
+
"gap_priorities": ranked,
|
|
22
|
+
"current_iteration": current,
|
|
23
|
+
"saturation": {
|
|
24
|
+
"saturated": saturation.saturated,
|
|
25
|
+
"low_yield_streak": saturation.low_yield_streak,
|
|
26
|
+
"strategy_diversity_exhausted": saturation.strategy_diversity_exhausted,
|
|
27
|
+
"rationale": list(saturation.rationale),
|
|
28
|
+
},
|
|
29
|
+
"iteration_count": len(iterations),
|
|
30
|
+
}
|