eduevidence 5.2.0 → 6.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +105 -0
- package/README.md +142 -75
- package/README.zh-CN.md +73 -30
- package/SKILL.md +397 -131
- package/agents/openai.yaml +4 -0
- package/assets/readme/controlled-execution.svg +34 -0
- package/assets/readme/landing-tour.gif +0 -0
- package/assets/readme/logo.png +0 -0
- package/assets/readme/research-workflow.svg +56 -0
- package/assets/readme/studio-graph.png +0 -0
- package/assets/readme/studio-overview.png +0 -0
- package/assets/readme/studio-reports.png +0 -0
- package/assets/readme/studio-tour.gif +0 -0
- package/autoevolve/config.yaml +17 -0
- package/autoevolve/program.md +25 -0
- package/autoevolve/protected.manifest.yaml +34 -0
- package/benchmarks/adversarial/cases.jsonl +7 -0
- package/benchmarks/evidence-library.json +5268 -0
- package/benchmarks/partitions.json +8 -0
- package/bin/eduevidence.js +2 -1
- package/docs/architecture.md +496 -0
- package/docs/autoresearch-evolution-plan.md +2903 -0
- package/docs/autoresearch-implementation-status.md +101 -0
- package/docs/demo-storyboard.md +20 -0
- package/docs/demo-workplace-ai.md +92 -0
- package/docs/demo.md +32 -0
- package/docs/install-guide.md +150 -0
- package/docs/orchestration-role-model.md +1254 -0
- package/docs/release-closeout/README.md +17 -0
- package/docs/release-closeout/frontend-acceptance.md +23 -0
- package/docs/release-closeout/issues.md +19 -0
- package/docs/release-closeout/verification.md +28 -0
- package/docs/release-contract.md +108 -0
- package/docs/research-studio-guide.zh-CN.md +166 -0
- package/docs/sciverse-api.md +125 -0
- package/eduevidence_cli.py +29 -13
- package/engine/_resources.py +13 -0
- package/engine/autoevolve/__init__.py +3 -0
- package/engine/autoevolve/agent_view.py +167 -0
- package/engine/autoevolve/core.py +357 -0
- package/engine/autoevolve/events.py +11 -0
- package/engine/autoevolve/git_workspace.py +77 -0
- package/engine/autoevolve/projection.py +23 -0
- package/engine/autoevolve/runner.py +413 -0
- package/engine/autoevolve/trust.py +146 -0
- package/engine/autoresearch/__init__.py +6 -0
- package/engine/autoresearch/commit.py +132 -0
- package/engine/autoresearch/contracts.py +126 -0
- package/engine/autoresearch/controller.py +207 -0
- package/engine/autoresearch/events.py +12 -0
- package/engine/autoresearch/gap_priority.py +168 -0
- package/engine/autoresearch/projection.py +30 -0
- package/engine/autoresearch/research_memory.py +59 -0
- package/engine/autoresearch/saturation.py +91 -0
- package/engine/briefs.py +2 -1
- package/engine/capabilities.py +1 -0
- package/engine/contracts.py +3 -1
- package/engine/decision_policy.py +96 -0
- package/engine/evidence_graph.py +14 -10
- package/engine/evidencecore.py +7 -5
- package/engine/gaps.py +132 -73
- package/engine/ids.py +2 -0
- package/engine/judge_pack.py +65 -0
- package/engine/library.py +6 -2
- package/engine/library_builtin.py +3 -1
- package/engine/living.py +36 -5
- package/engine/meta_synthesis.py +3 -1
- package/engine/migration.py +88 -3
- package/engine/orchestration.py +460 -0
- package/engine/paths.py +2 -0
- package/engine/pilot.py +36 -33
- package/engine/project.py +2 -2
- package/engine/research_service.py +113 -0
- package/engine/studio_read_model.py +400 -0
- package/engine/taxonomy.py +211 -0
- package/engine/tribunal.py +44 -33
- package/engine/update.py +1 -0
- package/engine/versions.py +1 -1
- package/engine/worker_result.py +109 -0
- package/engine/workflows.py +70 -0
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +2934 -0
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
- package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
- package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
- package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
- package/examples/ai-coding-assistant-evidence/frame.json +48 -0
- package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
- package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
- package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
- package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +230 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/result.json +1457 -0
- package/examples/ai-coding-assistant-evidence/result.zh.json +1457 -0
- package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
- package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +107 -0
- package/examples/spaced-retrieval-practice/applicability.json +14 -0
- package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
- package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
- package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
- package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
- package/examples/spaced-retrieval-practice/frame.json +58 -0
- package/examples/spaced-retrieval-practice/gate_report.json +101 -0
- package/examples/spaced-retrieval-practice/methodology.json +78 -0
- package/examples/spaced-retrieval-practice/report_spec.json +212 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/result.json +942 -0
- package/examples/spaced-retrieval-practice/result.zh.json +942 -0
- package/examples/spaced-retrieval-practice/skeptic.json +70 -0
- package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
- package/examples/spaced-retrieval-practice/verdict.json +93 -0
- package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -0
- package/examples/workplace-ai-assistant/evaluation.json +19 -0
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
- package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
- package/examples/workplace-ai-assistant/final_verdict.json +78 -0
- package/examples/workplace-ai-assistant/frame.json +41 -0
- package/examples/workplace-ai-assistant/gate_report.json +101 -0
- package/examples/workplace-ai-assistant/intervention.json +27 -0
- package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
- package/examples/workplace-ai-assistant/methodology.json +60 -0
- package/examples/workplace-ai-assistant/report_spec.json +224 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/result.json +615 -0
- package/examples/workplace-ai-assistant/result.zh.json +615 -0
- package/examples/workplace-ai-assistant/search_log.json +19 -0
- package/examples/workplace-ai-assistant/skeptic.json +72 -0
- package/examples/workplace-ai-assistant/sources.jsonl +3 -0
- package/examples/workplace-ai-assistant/validation_result.json +9 -0
- package/examples/workplace-ai-assistant/verdict.json +78 -0
- package/install.sh +7 -7
- package/integrations/agent_mcp.py +2 -2
- package/integrations/orchestration_dispatch.py +146 -0
- package/package.json +46 -3
- package/pyproject.toml +14 -22
- package/references/autoresearch.md +30 -0
- package/references/evaluation-policy.md +24 -0
- package/references/orchestration.md +22 -0
- package/references/report-copy-style.md +67 -0
- package/references/retrieval-compliance.md +75 -0
- package/references/retrieval-protocol.md +20 -0
- package/references/scientific-invariants.md +19 -0
- package/retrieval/audit.py +178 -0
- package/retrieval/fetch.py +96 -0
- package/retrieval/sciverse.py +398 -0
- package/retrieval/search.py +47 -7
- package/schemas/applicability.schema.json +94 -0
- package/schemas/chart-spec.schema.json +10 -3
- package/schemas/evidence.schema.json +316 -43
- package/schemas/fetch-result.schema.json +2 -1
- package/schemas/intervention.schema.json +106 -21
- package/schemas/report-result.schema.json +12 -4
- package/schemas/report-spec.schema.json +98 -100
- package/schemas/skeptic.schema.json +86 -0
- package/schemas/source.schema.json +21 -2
- package/schemas/v2/finding.schema.json +5 -1
- package/schemas/v2/methodology-audit.schema.json +5 -1
- package/schemas/v2/outcome.schema.json +28 -5
- package/schemas/v2/project.schema.json +2 -2
- package/schemas/v2/run.schema.json +1 -1
- package/schemas/v2/study.schema.json +5 -1
- package/schemas/vNext/autoevolve-session.schema.json +34 -0
- package/schemas/vNext/eval-snapshot.schema.json +77 -0
- package/schemas/vNext/execution-plan.schema.json +50 -0
- package/schemas/vNext/gap-priority.schema.json +54 -0
- package/schemas/vNext/negative-search-record.schema.json +68 -0
- package/schemas/vNext/research-iteration.schema.json +87 -0
- package/schemas/vNext/research-strategy.schema.json +62 -0
- package/schemas/vNext/skill-experiment.schema.json +90 -0
- package/schemas/vNext/task-spec.schema.json +156 -0
- package/schemas/vNext/worker-result.schema.json +60 -0
- package/schemas/verdict.schema.json +164 -28
- package/scripts/benchmark_judge.py +2 -2
- package/scripts/benchmark_v3.py +26 -43
- package/scripts/build_esl_artifacts.py +4 -4
- package/scripts/build_evidence_library.py +2 -2
- package/scripts/build_gh_pages.py +98 -0
- package/scripts/build_readme_diagrams.py +72 -0
- package/scripts/build_report_variants.py +101 -0
- package/scripts/build_result.py +74 -9
- package/scripts/check_autoresearch_invariants.py +95 -0
- package/scripts/check_package_parity.py +85 -0
- package/scripts/check_protocol_alignment.py +375 -0
- package/scripts/check_versioned_schemas.py +254 -0
- package/scripts/claim_audit.py +13 -8
- package/scripts/compute_confidence.py +10 -0
- package/scripts/daily_evolve.py +30 -0
- package/scripts/dashboard_server.py +130 -101
- package/scripts/did_regression.py +17 -32
- package/scripts/enrich_projects_human_and_lieflat.py +1 -1
- package/scripts/evidence_score.py +5 -2
- package/scripts/generate_metrics.py +4 -3
- package/scripts/generate_new_projects.py +5 -5
- package/scripts/orchestrator.py +286 -36
- package/scripts/pre_verdict_gate.py +224 -26
- package/scripts/quickstart.py +18 -2
- package/scripts/rebake_all_5themes.py +1 -2
- package/scripts/research_auto_cli.py +475 -0
- package/scripts/run_workspace.py +24 -8
- package/scripts/search_provenance.py +64 -0
- package/scripts/serve_web.py +9 -10
- package/scripts/skill_lint.py +1 -1
- package/scripts/skill_payload.py +81 -0
- package/scripts/test_adversarial_empirical.py +26 -19
- package/scripts/validate_schema.py +46 -2
- package/scripts/vnext_cli.py +133 -0
- package/setup.py +12 -0
- package/skill/agents/evaluation-designer.md +20 -4
- package/skill/agents/evidence-analyst.md +19 -3
- package/skill/agents/evidence-judge.md +50 -2
- package/skill/agents/evidence-retriever.md +20 -3
- package/skill/agents/intervention-designer.md +20 -4
- package/skill/agents/method-reviewer.md +18 -2
- package/skill/agents/{education-planner.md → research-planner.md} +19 -3
- package/skill/agents/skeptic.md +18 -2
- package/skill/roles/registry.yaml +45 -0
- package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
- package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
- package/skill/sub-skills/data-analysis/SKILL.md +34 -15
- package/skill/sub-skills/ethics-review/SKILL.md +33 -10
- package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
- package/skill/sub-skills/evidence-review/SKILL.md +31 -12
- package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
- package/skill/sub-skills/literature-review/SKILL.md +35 -14
- package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
- package/skill/sub-skills/report-generation/SKILL.md +40 -6
- package/skill/sub-skills/research-planning/SKILL.md +41 -14
- package/skill/sub-skills/study-design/SKILL.md +30 -9
- package/skill/task-briefs/adjudicate.md +32 -7
- package/skill/task-briefs/applicability.md +38 -0
- package/skill/task-briefs/audit.md +32 -7
- package/skill/task-briefs/challenge.md +34 -5
- package/skill/task-briefs/evaluate.md +30 -5
- package/skill/task-briefs/extract.md +31 -8
- package/skill/task-briefs/frame.md +39 -10
- package/skill/task-briefs/intervene.md +32 -6
- package/skill/task-briefs/present.md +32 -8
- package/skill/task-briefs/projection.md +37 -0
- package/skill/task-briefs/retrieve.md +36 -6
- package/skill/workflows/decision-and-pilot.md +85 -0
- package/skill/workflows/evaluate-and-update.md +93 -0
- package/skill/workflows/evidence-review.md +117 -0
- package/visualization/eduevidence-report/assets/base.css +2 -2
- package/visualization/eduevidence-report/assets/reader.css +752 -0
- package/visualization/eduevidence-report/assets/reader.js +132 -0
- package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
- package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
- package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
- package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
- package/visualization/eduevidence-report/scripts/build_report.py +561 -121
- package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +371 -136
- package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
- package/visualization/eduevidence-report/themes/academic.css +1 -1
- package/visualization/eduevidence-report/themes/claude.css +1 -1
- package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
- package/visualization/eduevidence-report/themes/datalab.css +2 -2
- package/visualization/eduevidence-report/themes/presentation.css +2 -2
- package/web/README.md +18 -0
- package/web/architecture.html +14885 -0
- package/web/index.html +53 -0
- package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
- package/web/studio/assets/index-B8tkF44Q.css +1 -0
- package/web/studio/assets/index-CQ6Keoyc.js +230 -0
- package/web/studio/config.json +1 -0
- package/web/studio/index.html +14 -0
- package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
- package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/bias.cpython-312.pyc +0 -0
- package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
- package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
- package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
- package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
- package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
- package/engine/__pycache__/events.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
- package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
- package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
- package/engine/__pycache__/ids.cpython-312.pyc +0 -0
- package/engine/__pycache__/library.cpython-312.pyc +0 -0
- package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
- package/engine/__pycache__/living.cpython-312.pyc +0 -0
- package/engine/__pycache__/log.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/migration.cpython-312.pyc +0 -0
- package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
- package/engine/__pycache__/paths.cpython-312.pyc +0 -0
- package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
- package/engine/__pycache__/planner.cpython-312.pyc +0 -0
- package/engine/__pycache__/project.cpython-312.pyc +0 -0
- package/engine/__pycache__/projections.cpython-312.pyc +0 -0
- package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
- package/engine/__pycache__/run.cpython-312.pyc +0 -0
- package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
- package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
- package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
- package/engine/__pycache__/update.cpython-312.pyc +0 -0
- package/engine/__pycache__/versions.cpython-312.pyc +0 -0
- package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
- package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
- package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
- package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
- package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
- package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
- package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
- package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
- package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
- package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
- package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
- package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
- package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
- package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
- package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
- package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
|
@@ -0,0 +1,460 @@
|
|
|
1
|
+
"""Canonical orchestration primitives for EduEvidence.
|
|
2
|
+
|
|
3
|
+
Protocol stage, scientific role, capability, worker/subagent, and model/CLI are
|
|
4
|
+
separate concepts. A RoleSpec describes scientific responsibility; a TaskSpec
|
|
5
|
+
is one bounded execution contract; an ExecutionPlan states sequencing and
|
|
6
|
+
parallel groups. Runtime model/CLI selection remains an adapter concern.
|
|
7
|
+
|
|
8
|
+
Scientific rule: workers produce staging artifacts only. Canonical project
|
|
9
|
+
state is committed by the single-writer lead path after validation.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
from dataclasses import asdict, dataclass, field
|
|
15
|
+
from enum import Enum
|
|
16
|
+
from typing import Any, Iterable
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class Complexity(str, Enum):
|
|
20
|
+
S = "S"
|
|
21
|
+
M = "M"
|
|
22
|
+
L = "L"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class ExecutionMode(str, Enum):
|
|
26
|
+
LOCAL = "local"
|
|
27
|
+
DELEGATED = "delegated"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
CANONICAL_STAGES = (
|
|
31
|
+
"frame",
|
|
32
|
+
"retrieve",
|
|
33
|
+
"extract",
|
|
34
|
+
"challenge",
|
|
35
|
+
"audit",
|
|
36
|
+
"adjudicate",
|
|
37
|
+
"applicability",
|
|
38
|
+
"intervene",
|
|
39
|
+
"evaluate",
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
CANONICAL_STATE_ARTIFACTS = frozenset({
|
|
43
|
+
"EvidenceGraph",
|
|
44
|
+
"GraphRevision",
|
|
45
|
+
"DecisionSnapshot",
|
|
46
|
+
"KnowledgeGap",
|
|
47
|
+
"StudyDesign",
|
|
48
|
+
"PilotRun",
|
|
49
|
+
"AnalysisRun",
|
|
50
|
+
})
|
|
51
|
+
|
|
52
|
+
DEFAULT_FORBIDDEN_WORKER_ACTIONS = (
|
|
53
|
+
"canonical_state_write",
|
|
54
|
+
"decision_promotion",
|
|
55
|
+
"recursive_worker_spawn",
|
|
56
|
+
"evaluator_mutation",
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass(frozen=True)
|
|
61
|
+
class RoleSpec:
|
|
62
|
+
"""A scientific responsibility, not an instruction to spawn an agent."""
|
|
63
|
+
|
|
64
|
+
name: str
|
|
65
|
+
responsibility: str
|
|
66
|
+
stages: tuple[str, ...]
|
|
67
|
+
capabilities: tuple[str, ...]
|
|
68
|
+
independence_required: bool = False
|
|
69
|
+
critical_path: bool = False
|
|
70
|
+
|
|
71
|
+
def __post_init__(self) -> None:
|
|
72
|
+
unknown = set(self.stages) - set(CANONICAL_STAGES)
|
|
73
|
+
if unknown:
|
|
74
|
+
raise ValueError(f"role {self.name!r} references unknown stages: {sorted(unknown)}")
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
ROLE_REGISTRY: dict[str, RoleSpec] = {
|
|
78
|
+
"research-planner": RoleSpec(
|
|
79
|
+
"research-planner",
|
|
80
|
+
"Own framing completeness, scope, comparison and outcome definition.",
|
|
81
|
+
("frame",),
|
|
82
|
+
("research-planning",),
|
|
83
|
+
critical_path=True,
|
|
84
|
+
),
|
|
85
|
+
"evidence-retriever": RoleSpec(
|
|
86
|
+
"evidence-retriever",
|
|
87
|
+
"Acquire candidate sources and record verifiable provenance without adjudicating them.",
|
|
88
|
+
("retrieve",),
|
|
89
|
+
("literature-review", "full-text-fetch", "source-validation"),
|
|
90
|
+
),
|
|
91
|
+
"evidence-analyst": RoleSpec(
|
|
92
|
+
"evidence-analyst",
|
|
93
|
+
"Extract structured study findings without deciding what the evidence body means.",
|
|
94
|
+
("extract",),
|
|
95
|
+
("evidence-extraction",),
|
|
96
|
+
),
|
|
97
|
+
"skeptic": RoleSpec(
|
|
98
|
+
"skeptic",
|
|
99
|
+
"Own independent counter-evidence and alternative-explanation coverage.",
|
|
100
|
+
("challenge",),
|
|
101
|
+
("contradiction-analysis", "counter-retrieval"),
|
|
102
|
+
independence_required=True,
|
|
103
|
+
critical_path=True,
|
|
104
|
+
),
|
|
105
|
+
"method-reviewer": RoleSpec(
|
|
106
|
+
"method-reviewer",
|
|
107
|
+
"Own study-level methodology appraisal and evidence-quality limitations.",
|
|
108
|
+
("audit",),
|
|
109
|
+
("methodology-audit",),
|
|
110
|
+
independence_required=True,
|
|
111
|
+
critical_path=True,
|
|
112
|
+
),
|
|
113
|
+
"evidence-judge": RoleSpec(
|
|
114
|
+
"evidence-judge",
|
|
115
|
+
"Own the bounded decision after evidence, challenge and audit gates pass.",
|
|
116
|
+
("adjudicate", "applicability"),
|
|
117
|
+
("evidence-review", "decision-adjudication", "applicability"),
|
|
118
|
+
critical_path=True,
|
|
119
|
+
),
|
|
120
|
+
"intervention-designer": RoleSpec(
|
|
121
|
+
"intervention-designer",
|
|
122
|
+
"Turn a grounded decision and KnowledgeGap into a bounded intervention or pilot.",
|
|
123
|
+
("intervene",),
|
|
124
|
+
("study-design", "intervention-design"),
|
|
125
|
+
),
|
|
126
|
+
"evaluation-designer": RoleSpec(
|
|
127
|
+
"evaluation-designer",
|
|
128
|
+
"Define estimable evaluation, retention/transfer measurement and update logic.",
|
|
129
|
+
("evaluate",),
|
|
130
|
+
("evaluation-design", "data-analysis"),
|
|
131
|
+
),
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
@dataclass(frozen=True)
|
|
136
|
+
class TaskSpec:
|
|
137
|
+
"""One bounded scientific task; delegation requires full runtime context."""
|
|
138
|
+
|
|
139
|
+
task_id: str
|
|
140
|
+
stage: str
|
|
141
|
+
role: str
|
|
142
|
+
objective: str
|
|
143
|
+
evidence_axis: str
|
|
144
|
+
run_id: str | None = None
|
|
145
|
+
base_revision: int | None = None
|
|
146
|
+
role_profile: str | None = None
|
|
147
|
+
reason_for_delegation: str | None = None
|
|
148
|
+
inputs: tuple[str, ...] = () # legacy alias kept during migration
|
|
149
|
+
input_artifacts: tuple[str, ...] = ()
|
|
150
|
+
allowed_capabilities: tuple[str, ...] = ()
|
|
151
|
+
forbidden_actions: tuple[str, ...] = DEFAULT_FORBIDDEN_WORKER_ACTIONS
|
|
152
|
+
scope: dict[str, Any] = field(default_factory=dict)
|
|
153
|
+
budget: dict[str, Any] = field(default_factory=dict)
|
|
154
|
+
expected_outputs: tuple[str, ...] = ()
|
|
155
|
+
output_contract: dict[str, Any] = field(default_factory=dict)
|
|
156
|
+
termination: dict[str, Any] = field(default_factory=dict)
|
|
157
|
+
execution_mode: ExecutionMode = ExecutionMode.LOCAL
|
|
158
|
+
independent: bool = False
|
|
159
|
+
read_only: bool = True
|
|
160
|
+
timeout_seconds: int = 1800
|
|
161
|
+
token_budget: int | None = None
|
|
162
|
+
metadata: dict[str, Any] = field(default_factory=dict)
|
|
163
|
+
|
|
164
|
+
def validate(self) -> None:
|
|
165
|
+
if not self.task_id.strip():
|
|
166
|
+
raise ValueError("task_id must be non-empty")
|
|
167
|
+
if self.stage not in CANONICAL_STAGES:
|
|
168
|
+
raise ValueError(f"unknown protocol stage: {self.stage!r}")
|
|
169
|
+
spec = ROLE_REGISTRY.get(self.role)
|
|
170
|
+
if spec is None:
|
|
171
|
+
raise ValueError(f"unknown scientific role: {self.role!r}")
|
|
172
|
+
if self.stage not in spec.stages:
|
|
173
|
+
raise ValueError(
|
|
174
|
+
f"role {self.role!r} does not own stage {self.stage!r}; owns {spec.stages}"
|
|
175
|
+
)
|
|
176
|
+
if self.role_profile not in (None, self.role):
|
|
177
|
+
raise ValueError("role_profile must identify the same scientific role")
|
|
178
|
+
if not self.objective.strip():
|
|
179
|
+
raise ValueError("objective must be non-empty")
|
|
180
|
+
if not self.evidence_axis.strip():
|
|
181
|
+
raise ValueError("evidence_axis must be non-empty")
|
|
182
|
+
if self.base_revision is not None and self.base_revision < 0:
|
|
183
|
+
raise ValueError("base_revision must be >= 0")
|
|
184
|
+
if self.timeout_seconds <= 0:
|
|
185
|
+
raise ValueError("timeout_seconds must be positive")
|
|
186
|
+
if self.token_budget is not None and self.token_budget <= 0:
|
|
187
|
+
raise ValueError("token_budget must be positive when provided")
|
|
188
|
+
unknown_caps = set(self.allowed_capabilities) - set(spec.capabilities)
|
|
189
|
+
if unknown_caps:
|
|
190
|
+
raise ValueError(
|
|
191
|
+
f"task grants capabilities outside role {self.role!r}: {sorted(unknown_caps)}"
|
|
192
|
+
)
|
|
193
|
+
forbidden = CANONICAL_STATE_ARTIFACTS.intersection(self.expected_outputs)
|
|
194
|
+
if self.execution_mode is ExecutionMode.DELEGATED:
|
|
195
|
+
if not self.read_only:
|
|
196
|
+
raise ValueError("delegated workers must be read-only against canonical state")
|
|
197
|
+
if forbidden:
|
|
198
|
+
raise ValueError(
|
|
199
|
+
"delegated workers may only return staging artifacts; canonical outputs forbidden: "
|
|
200
|
+
f"{sorted(forbidden)}"
|
|
201
|
+
)
|
|
202
|
+
if "canonical_state_write" not in self.forbidden_actions:
|
|
203
|
+
raise ValueError("delegated TaskSpec must explicitly forbid canonical_state_write")
|
|
204
|
+
if spec.independence_required and not self.independent:
|
|
205
|
+
raise ValueError(f"role {self.role!r} requires independent delegated execution")
|
|
206
|
+
|
|
207
|
+
def validate_for_dispatch(self) -> None:
|
|
208
|
+
self.validate()
|
|
209
|
+
if self.execution_mode is not ExecutionMode.DELEGATED:
|
|
210
|
+
raise ValueError("only delegated TaskSpecs may be dispatched")
|
|
211
|
+
if not self.run_id or not self.run_id.strip():
|
|
212
|
+
raise ValueError("delegated TaskSpec requires run_id")
|
|
213
|
+
if self.base_revision is None:
|
|
214
|
+
raise ValueError("delegated TaskSpec requires base_revision")
|
|
215
|
+
if not self.reason_for_delegation or not self.reason_for_delegation.strip():
|
|
216
|
+
raise ValueError("delegated TaskSpec requires reason_for_delegation")
|
|
217
|
+
if not self.allowed_capabilities:
|
|
218
|
+
raise ValueError("delegated TaskSpec requires explicit allowed_capabilities")
|
|
219
|
+
if not self.output_contract:
|
|
220
|
+
raise ValueError("delegated TaskSpec requires output_contract")
|
|
221
|
+
if not self.termination:
|
|
222
|
+
raise ValueError("delegated TaskSpec requires termination contract")
|
|
223
|
+
|
|
224
|
+
def to_dict(self) -> dict[str, Any]:
|
|
225
|
+
value = asdict(self)
|
|
226
|
+
value["execution_mode"] = self.execution_mode.value
|
|
227
|
+
return value
|
|
228
|
+
|
|
229
|
+
def to_prompt_contract(self) -> str:
|
|
230
|
+
"""Produce a stable, explicit worker envelope."""
|
|
231
|
+
self.validate_for_dispatch()
|
|
232
|
+
return (
|
|
233
|
+
f"TASK_ID: {self.task_id}\n"
|
|
234
|
+
f"RUN_ID: {self.run_id}\n"
|
|
235
|
+
f"BASE_GRAPH_REVISION: {self.base_revision}\n"
|
|
236
|
+
f"STAGE: {self.stage}\n"
|
|
237
|
+
f"SCIENTIFIC_ROLE: {self.role}\n"
|
|
238
|
+
f"EVIDENCE_AXIS: {self.evidence_axis}\n"
|
|
239
|
+
f"OBJECTIVE: {self.objective}\n"
|
|
240
|
+
f"REASON_FOR_DELEGATION: {self.reason_for_delegation}\n"
|
|
241
|
+
f"ALLOWED_CAPABILITIES: {', '.join(self.allowed_capabilities)}\n"
|
|
242
|
+
f"FORBIDDEN_ACTIONS: {', '.join(self.forbidden_actions)}\n"
|
|
243
|
+
f"INPUT_ARTIFACTS: {', '.join(self.input_artifacts or self.inputs) if (self.input_artifacts or self.inputs) else 'none'}\n"
|
|
244
|
+
f"SCOPE_JSON: {json.dumps(self.scope, ensure_ascii=False, sort_keys=True)}\n"
|
|
245
|
+
f"BUDGET_JSON: {json.dumps(self.budget, ensure_ascii=False, sort_keys=True)}\n"
|
|
246
|
+
f"OUTPUT_CONTRACT_JSON: {json.dumps(self.output_contract, ensure_ascii=False, sort_keys=True)}\n"
|
|
247
|
+
f"TERMINATION_JSON: {json.dumps(self.termination, ensure_ascii=False, sort_keys=True)}\n"
|
|
248
|
+
"CANONICAL_STATE_WRITE: FORBIDDEN\n"
|
|
249
|
+
"Return only the requested staging artifact(s)."
|
|
250
|
+
)
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
@dataclass(frozen=True)
|
|
254
|
+
class ExecutionPlan:
|
|
255
|
+
complexity: Complexity
|
|
256
|
+
tasks: tuple[TaskSpec, ...]
|
|
257
|
+
max_parallel_workers: int
|
|
258
|
+
parallel_groups: tuple[tuple[str, ...], ...] = ()
|
|
259
|
+
plan_id: str | None = None
|
|
260
|
+
|
|
261
|
+
@property
|
|
262
|
+
def delegated_tasks(self) -> tuple[TaskSpec, ...]:
|
|
263
|
+
return tuple(t for t in self.tasks if t.execution_mode is ExecutionMode.DELEGATED)
|
|
264
|
+
|
|
265
|
+
def validate(self) -> None:
|
|
266
|
+
if self.max_parallel_workers < 0:
|
|
267
|
+
raise ValueError("max_parallel_workers cannot be negative")
|
|
268
|
+
if self.max_parallel_workers > 6:
|
|
269
|
+
raise ValueError("bounded worker pool exceeds hard limit of 6")
|
|
270
|
+
ids: set[str] = set()
|
|
271
|
+
by_id: dict[str, TaskSpec] = {}
|
|
272
|
+
for task in self.tasks:
|
|
273
|
+
task.validate()
|
|
274
|
+
if task.task_id in ids:
|
|
275
|
+
raise ValueError(f"duplicate task id: {task.task_id}")
|
|
276
|
+
ids.add(task.task_id)
|
|
277
|
+
by_id[task.task_id] = task
|
|
278
|
+
if self.delegated_tasks and self.max_parallel_workers < 1:
|
|
279
|
+
raise ValueError("delegated plan requires max_parallel_workers >= 1")
|
|
280
|
+
|
|
281
|
+
grouped: list[str] = []
|
|
282
|
+
for group in self.parallel_groups:
|
|
283
|
+
if not group:
|
|
284
|
+
raise ValueError("parallel groups may not be empty")
|
|
285
|
+
if len(group) > self.max_parallel_workers:
|
|
286
|
+
raise ValueError("parallel group exceeds max_parallel_workers")
|
|
287
|
+
for task_id in group:
|
|
288
|
+
task = by_id.get(task_id)
|
|
289
|
+
if task is None:
|
|
290
|
+
raise ValueError(f"parallel group references unknown task {task_id}")
|
|
291
|
+
if task.execution_mode is not ExecutionMode.DELEGATED:
|
|
292
|
+
raise ValueError(f"parallel group may contain delegated tasks only: {task_id}")
|
|
293
|
+
grouped.append(task_id)
|
|
294
|
+
delegated_ids = [task.task_id for task in self.delegated_tasks]
|
|
295
|
+
if sorted(grouped) != sorted(delegated_ids):
|
|
296
|
+
raise ValueError("every delegated task must appear exactly once in parallel_groups")
|
|
297
|
+
|
|
298
|
+
def to_dict(self) -> dict[str, Any]:
|
|
299
|
+
return {
|
|
300
|
+
"plan_id": self.plan_id,
|
|
301
|
+
"complexity": self.complexity.value,
|
|
302
|
+
"tasks": [task.to_dict() for task in self.tasks],
|
|
303
|
+
"max_parallel_workers": self.max_parallel_workers,
|
|
304
|
+
"parallel_groups": [list(group) for group in self.parallel_groups],
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
class ExecutionPlanner:
|
|
309
|
+
"""Deterministic policy planner; it never chooses a concrete model/CLI."""
|
|
310
|
+
|
|
311
|
+
def plan(
|
|
312
|
+
self,
|
|
313
|
+
complexity: str | Complexity,
|
|
314
|
+
*,
|
|
315
|
+
run_id: str | None = None,
|
|
316
|
+
base_revision: int | None = None,
|
|
317
|
+
plan_id: str | None = None,
|
|
318
|
+
) -> ExecutionPlan:
|
|
319
|
+
level = complexity if isinstance(complexity, Complexity) else Complexity(complexity.upper())
|
|
320
|
+
if level is Complexity.S:
|
|
321
|
+
plan = ExecutionPlan(
|
|
322
|
+
level,
|
|
323
|
+
self._serial_tasks(run_id, base_revision),
|
|
324
|
+
max_parallel_workers=0,
|
|
325
|
+
parallel_groups=(),
|
|
326
|
+
plan_id=plan_id,
|
|
327
|
+
)
|
|
328
|
+
elif level is Complexity.M:
|
|
329
|
+
tasks = self._medium_tasks(run_id, base_revision)
|
|
330
|
+
plan = ExecutionPlan(
|
|
331
|
+
level,
|
|
332
|
+
tasks,
|
|
333
|
+
max_parallel_workers=2,
|
|
334
|
+
parallel_groups=(("retrieve-direct", "retrieve-counter"), ("challenge",)),
|
|
335
|
+
plan_id=plan_id,
|
|
336
|
+
)
|
|
337
|
+
else:
|
|
338
|
+
tasks = self._deep_tasks(run_id, base_revision)
|
|
339
|
+
plan = ExecutionPlan(
|
|
340
|
+
level,
|
|
341
|
+
tasks,
|
|
342
|
+
max_parallel_workers=4,
|
|
343
|
+
parallel_groups=(
|
|
344
|
+
(
|
|
345
|
+
"retrieve-direct",
|
|
346
|
+
"retrieve-transfer",
|
|
347
|
+
"retrieve-counter",
|
|
348
|
+
"retrieve-applicability",
|
|
349
|
+
),
|
|
350
|
+
("challenge", "audit"),
|
|
351
|
+
),
|
|
352
|
+
plan_id=plan_id,
|
|
353
|
+
)
|
|
354
|
+
plan.validate()
|
|
355
|
+
return plan
|
|
356
|
+
|
|
357
|
+
@staticmethod
|
|
358
|
+
def _base(
|
|
359
|
+
task_id: str,
|
|
360
|
+
stage: str,
|
|
361
|
+
role: str,
|
|
362
|
+
objective: str,
|
|
363
|
+
axis: str,
|
|
364
|
+
*,
|
|
365
|
+
run_id: str | None,
|
|
366
|
+
base_revision: int | None,
|
|
367
|
+
delegated: bool = False,
|
|
368
|
+
independent: bool = False,
|
|
369
|
+
outputs: Iterable[str] = (),
|
|
370
|
+
) -> TaskSpec:
|
|
371
|
+
role_spec = ROLE_REGISTRY[role]
|
|
372
|
+
timeout_seconds = 1800
|
|
373
|
+
token_budget = None
|
|
374
|
+
outputs = tuple(outputs)
|
|
375
|
+
return TaskSpec(
|
|
376
|
+
task_id=task_id,
|
|
377
|
+
stage=stage,
|
|
378
|
+
role=role,
|
|
379
|
+
role_profile=role,
|
|
380
|
+
objective=objective,
|
|
381
|
+
evidence_axis=axis,
|
|
382
|
+
run_id=run_id,
|
|
383
|
+
base_revision=base_revision,
|
|
384
|
+
reason_for_delegation=(
|
|
385
|
+
f"independent bounded {axis} work benefits from delegated context"
|
|
386
|
+
if delegated
|
|
387
|
+
else None
|
|
388
|
+
),
|
|
389
|
+
allowed_capabilities=role_spec.capabilities,
|
|
390
|
+
forbidden_actions=DEFAULT_FORBIDDEN_WORKER_ACTIONS,
|
|
391
|
+
scope={"evidence_axis": axis},
|
|
392
|
+
budget={"timeout_seconds": timeout_seconds, "token_budget": token_budget},
|
|
393
|
+
expected_outputs=outputs,
|
|
394
|
+
output_contract={
|
|
395
|
+
"artifact_types": list(outputs),
|
|
396
|
+
"canonical_state": False,
|
|
397
|
+
"validation_owner": "lead-orchestrator",
|
|
398
|
+
},
|
|
399
|
+
termination={
|
|
400
|
+
"max_seconds": timeout_seconds,
|
|
401
|
+
"conditions": ["output_contract_satisfied", "budget_exhausted", "tool_failure"],
|
|
402
|
+
},
|
|
403
|
+
execution_mode=ExecutionMode.DELEGATED if delegated else ExecutionMode.LOCAL,
|
|
404
|
+
independent=independent,
|
|
405
|
+
read_only=True,
|
|
406
|
+
timeout_seconds=timeout_seconds,
|
|
407
|
+
token_budget=token_budget,
|
|
408
|
+
)
|
|
409
|
+
|
|
410
|
+
def _serial_tasks(self, run_id, base_revision) -> tuple[TaskSpec, ...]:
|
|
411
|
+
return (
|
|
412
|
+
self._base("frame", "frame", "research-planner", "Structure the research question.", "frame", run_id=run_id, base_revision=base_revision),
|
|
413
|
+
self._base("retrieve", "retrieve", "evidence-retriever", "Acquire bounded evidence.", "direct+counter", run_id=run_id, base_revision=base_revision),
|
|
414
|
+
self._base("extract", "extract", "evidence-analyst", "Extract structured findings.", "all-eligible", run_id=run_id, base_revision=base_revision),
|
|
415
|
+
self._base("challenge", "challenge", "skeptic", "Challenge the provisional interpretation.", "counter-evidence", run_id=run_id, base_revision=base_revision),
|
|
416
|
+
self._base("audit", "audit", "method-reviewer", "Audit methodology and outcome validity.", "methodology", run_id=run_id, base_revision=base_revision),
|
|
417
|
+
self._base("judge", "adjudicate", "evidence-judge", "Emit an evidence-bounded decision.", "decision", run_id=run_id, base_revision=base_revision),
|
|
418
|
+
)
|
|
419
|
+
|
|
420
|
+
def _medium_tasks(self, run_id, base_revision) -> tuple[TaskSpec, ...]:
|
|
421
|
+
return (
|
|
422
|
+
self._base("frame", "frame", "research-planner", "Structure the research question.", "frame", run_id=run_id, base_revision=base_revision),
|
|
423
|
+
self._base("retrieve-direct", "retrieve", "evidence-retriever", "Retrieve direct decision-relevant evidence.", "direct-causal", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
|
|
424
|
+
self._base("retrieve-counter", "retrieve", "evidence-retriever", "Retrieve null, negative and contradictory evidence.", "counter-risk", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
|
|
425
|
+
self._base("extract", "extract", "evidence-analyst", "Merge validated sources and extract findings.", "all-eligible", run_id=run_id, base_revision=base_revision),
|
|
426
|
+
self._base("challenge", "challenge", "skeptic", "Independently test the provisional interpretation.", "counter-evidence", run_id=run_id, base_revision=base_revision, delegated=True, independent=True, outputs=("SkepticFindings",)),
|
|
427
|
+
self._base("audit", "audit", "method-reviewer", "Audit methodology and construct validity.", "methodology", run_id=run_id, base_revision=base_revision),
|
|
428
|
+
self._base("judge", "adjudicate", "evidence-judge", "Emit an evidence-bounded decision.", "decision", run_id=run_id, base_revision=base_revision),
|
|
429
|
+
)
|
|
430
|
+
|
|
431
|
+
def _deep_tasks(self, run_id, base_revision) -> tuple[TaskSpec, ...]:
|
|
432
|
+
return (
|
|
433
|
+
self._base("frame", "frame", "research-planner", "Structure the research question.", "frame", run_id=run_id, base_revision=base_revision),
|
|
434
|
+
self._base("retrieve-direct", "retrieve", "evidence-retriever", "Retrieve direct causal evidence.", "direct-causal", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
|
|
435
|
+
self._base("retrieve-transfer", "retrieve", "evidence-retriever", "Retrieve retention and independent-transfer evidence.", "transfer-retention", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
|
|
436
|
+
self._base("retrieve-counter", "retrieve", "evidence-retriever", "Retrieve null, negative, risk and contradiction evidence.", "counter-risk", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
|
|
437
|
+
self._base("retrieve-applicability", "retrieve", "evidence-retriever", "Retrieve subgroup, context and freshness evidence.", "applicability-freshness", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
|
|
438
|
+
self._base("extract", "extract", "evidence-analyst", "Deterministically merge and extract validated findings.", "all-eligible", run_id=run_id, base_revision=base_revision),
|
|
439
|
+
self._base("challenge", "challenge", "skeptic", "Independently challenge the merged interpretation.", "counter-evidence", run_id=run_id, base_revision=base_revision, delegated=True, independent=True, outputs=("SkepticFindings",)),
|
|
440
|
+
self._base("audit", "audit", "method-reviewer", "Independently audit methodology and outcome validity.", "methodology", run_id=run_id, base_revision=base_revision, delegated=True, independent=True, outputs=("MethodologyAudit",)),
|
|
441
|
+
self._base("judge", "adjudicate", "evidence-judge", "Emit an evidence-bounded decision after all gates.", "decision", run_id=run_id, base_revision=base_revision),
|
|
442
|
+
)
|
|
443
|
+
|
|
444
|
+
|
|
445
|
+
class CanonicalWriteGuard:
|
|
446
|
+
"""Fail-closed single-writer authority for canonical state mutations."""
|
|
447
|
+
|
|
448
|
+
def __init__(self, writer_id: str = "lead-orchestrator") -> None:
|
|
449
|
+
if not writer_id.strip():
|
|
450
|
+
raise ValueError("writer_id must be non-empty")
|
|
451
|
+
self.writer_id = writer_id
|
|
452
|
+
|
|
453
|
+
def require(self, actor_id: str, artifact_type: str) -> None:
|
|
454
|
+
if artifact_type not in CANONICAL_STATE_ARTIFACTS:
|
|
455
|
+
return
|
|
456
|
+
if actor_id != self.writer_id:
|
|
457
|
+
raise PermissionError(
|
|
458
|
+
f"canonical state {artifact_type} may only be written by {self.writer_id!r}; "
|
|
459
|
+
f"actor {actor_id!r} is staging-only"
|
|
460
|
+
)
|
package/engine/paths.py
CHANGED
package/engine/pilot.py
CHANGED
|
@@ -22,6 +22,7 @@ from engine.datasets import analysis_blocked_by_privacy, derive_csv_profile, ing
|
|
|
22
22
|
from engine.graph_store import GraphMutation, GraphStore
|
|
23
23
|
from engine.ids import new_local_id, new_run_id
|
|
24
24
|
from engine.project import ProjectWorkspace
|
|
25
|
+
from engine.taxonomy import category_of, tokens as taxonomy_tokens
|
|
25
26
|
from engine.synthesis import synthesize_project
|
|
26
27
|
from engine.tribunal import adjudicate, decision_diff, save_decision_snapshot
|
|
27
28
|
from engine.versions import (
|
|
@@ -30,15 +31,21 @@ from engine.versions import (
|
|
|
30
31
|
SOURCE_VALIDATION_POLICY_VERSION,
|
|
31
32
|
)
|
|
32
33
|
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
"
|
|
40
|
-
|
|
41
|
-
|
|
34
|
+
def outcome_taxonomy_tokens(domain: str = "education") -> set[str]:
|
|
35
|
+
"""Outcome tokens a pilot may measure, read from the domain registry.
|
|
36
|
+
|
|
37
|
+
This was a module-level set of the 20 education tokens, so a policy pilot
|
|
38
|
+
could not register at all. The domain registry (via engine/taxonomy.py) is
|
|
39
|
+
now the single authority.
|
|
40
|
+
"""
|
|
41
|
+
return set(taxonomy_tokens(domain))
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def project_domain(project: ProjectWorkspace) -> str:
|
|
45
|
+
"""The domain a project registers; defaults to education when absent."""
|
|
46
|
+
manifest = project.manifest()
|
|
47
|
+
return str(manifest.get("domain") or "education")
|
|
48
|
+
|
|
42
49
|
|
|
43
50
|
PILOT_STATUSES = ("registered", "data_imported", "analyzed", "adjudicated")
|
|
44
51
|
|
|
@@ -49,22 +56,6 @@ PII_COLUMN_HINTS = ("name", "student", "学号", "姓名", "email", "mail",
|
|
|
49
56
|
_DECISION_IMPLICATION = {"support": "support_adoption",
|
|
50
57
|
"contradict": "oppose_adoption", "neutral": "neutral"}
|
|
51
58
|
|
|
52
|
-
#: Outcome Taxonomy token -> graph outcome category enum (schemas/v2/outcome).
|
|
53
|
-
_OUTCOME_CATEGORY = {
|
|
54
|
-
"knowledge_gain": "learning", "concept_understanding": "learning",
|
|
55
|
-
"retention": "learning", "transfer": "learning",
|
|
56
|
-
"independent_problem_solving": "learning",
|
|
57
|
-
"completion_time": "task_performance", "accuracy": "task_performance",
|
|
58
|
-
"code_quality": "task_performance", "assignment_score": "task_performance",
|
|
59
|
-
"engagement": "process", "motivation": "process",
|
|
60
|
-
"cognitive_load": "process", "help_seeking": "process",
|
|
61
|
-
"metacognition": "process",
|
|
62
|
-
"ai_dependency": "risk", "over_reliance": "risk",
|
|
63
|
-
"reduced_effort": "risk", "reduced_transfer": "risk",
|
|
64
|
-
"academic_integrity_risk": "risk", "false_confidence": "risk",
|
|
65
|
-
}
|
|
66
|
-
|
|
67
|
-
|
|
68
59
|
def _now_iso() -> str:
|
|
69
60
|
return datetime.now(timezone.utc).isoformat()
|
|
70
61
|
|
|
@@ -83,7 +74,8 @@ def _load_pilot(project: ProjectWorkspace, pilot_id: str) -> dict:
|
|
|
83
74
|
def _save_pilot(project: ProjectWorkspace, pilot: dict) -> Path:
|
|
84
75
|
from scripts.validate_schema import SchemaError, validate # noqa: PLC0415
|
|
85
76
|
|
|
86
|
-
|
|
77
|
+
from engine._resources import resource_root
|
|
78
|
+
schema_path = (resource_root() / "schemas" / "v3"
|
|
87
79
|
/ "pilot-outcome.schema.json")
|
|
88
80
|
schema = json.loads(schema_path.read_text(encoding="utf-8"))
|
|
89
81
|
try:
|
|
@@ -110,10 +102,13 @@ def register_pilot(project: ProjectWorkspace, *,
|
|
|
110
102
|
raise ValueError(
|
|
111
103
|
f"decision snapshot {decision_snapshot_id} not found in this project; "
|
|
112
104
|
"a pilot must bind to a real adjudication")
|
|
113
|
-
|
|
105
|
+
domain = project_domain(project)
|
|
106
|
+
known_tokens = outcome_taxonomy_tokens(domain)
|
|
107
|
+
unknown = [o for o in outcome_columns if o not in known_tokens]
|
|
114
108
|
if unknown:
|
|
115
109
|
raise ValueError(
|
|
116
|
-
f"outcome_columns outside Outcome Taxonomy:
|
|
110
|
+
f"outcome_columns outside the {domain} Outcome Taxonomy: "
|
|
111
|
+
f"{sorted(unknown)}")
|
|
117
112
|
if not conditions or sample_size < 1:
|
|
118
113
|
raise ValueError("conditions must be non-empty and sample_size >= 1")
|
|
119
114
|
if anon_policy.get("no_pii_columns") is not True:
|
|
@@ -207,7 +202,13 @@ def link_analysis(project: ProjectWorkspace, pilot_id: str, *,
|
|
|
207
202
|
return pilot
|
|
208
203
|
|
|
209
204
|
|
|
210
|
-
def _ensure_outcome(store: GraphStore, outcome_id: str) -> dict:
|
|
205
|
+
def _ensure_outcome(store: GraphStore, outcome_id: str, domain: str) -> dict:
|
|
206
|
+
"""Create the pilot outcome row, classified by the DOMAIN registry.
|
|
207
|
+
|
|
208
|
+
``category_of`` raises for an unregistered token; the caller has already
|
|
209
|
+
validated the token against ``outcome_taxonomy_tokens(domain)``, so an
|
|
210
|
+
error here means the two disagreed and must not be silently absorbed.
|
|
211
|
+
"""
|
|
211
212
|
existing = store.get("outcomes", outcome_id)
|
|
212
213
|
if existing:
|
|
213
214
|
return existing
|
|
@@ -215,7 +216,7 @@ def _ensure_outcome(store: GraphStore, outcome_id: str) -> dict:
|
|
|
215
216
|
return {
|
|
216
217
|
"outcome_id": outcome_id,
|
|
217
218
|
"name": token,
|
|
218
|
-
"outcome_type":
|
|
219
|
+
"outcome_type": category_of(domain, token),
|
|
219
220
|
"extensions": {"pilot_outcome": True},
|
|
220
221
|
}
|
|
221
222
|
|
|
@@ -236,9 +237,11 @@ def redecide(project: ProjectWorkspace, pilot_id: str, *,
|
|
|
236
237
|
raise ValueError(f"invalid effect_direction {effect_direction!r}")
|
|
237
238
|
if relation_to_claim not in ("support", "contradict", "neutral"):
|
|
238
239
|
raise ValueError(f"invalid relation_to_claim {relation_to_claim!r}")
|
|
239
|
-
|
|
240
|
+
domain = project_domain(project)
|
|
241
|
+
if outcome_token not in outcome_taxonomy_tokens(domain):
|
|
240
242
|
raise ValueError(
|
|
241
|
-
f"outcome_token {outcome_token!r} outside
|
|
243
|
+
f"outcome_token {outcome_token!r} outside the {domain} "
|
|
244
|
+
"Outcome Taxonomy")
|
|
242
245
|
outcome_id = f"OUT-{outcome_token}"
|
|
243
246
|
pilot = _load_pilot(project, pilot_id)
|
|
244
247
|
if pilot["status"] not in ("analyzed", "data_imported"):
|
|
@@ -281,7 +284,7 @@ def redecide(project: ProjectWorkspace, pilot_id: str, *,
|
|
|
281
284
|
"identity_status": "resolved",
|
|
282
285
|
"extensions": {"pilot_id": pilot_id},
|
|
283
286
|
}
|
|
284
|
-
outcome = _ensure_outcome(store, outcome_id)
|
|
287
|
+
outcome = _ensure_outcome(store, outcome_id, domain)
|
|
285
288
|
estimate = None
|
|
286
289
|
if effect_estimate is not None:
|
|
287
290
|
estimate = {
|
package/engine/project.py
CHANGED
|
@@ -62,7 +62,7 @@ class ProjectWorkspace:
|
|
|
62
62
|
|
|
63
63
|
@classmethod
|
|
64
64
|
def create(cls, home: Path, *, question: str, title: str,
|
|
65
|
-
research_mode: str) -> "ProjectWorkspace":
|
|
65
|
+
research_mode: str, domain: str = "education") -> "ProjectWorkspace":
|
|
66
66
|
home = Path(home).expanduser().resolve()
|
|
67
67
|
project_id = new_project_id(question)
|
|
68
68
|
path = home / "projects" / project_id
|
|
@@ -72,7 +72,7 @@ class ProjectWorkspace:
|
|
|
72
72
|
manifest = {
|
|
73
73
|
"project_id": project_id,
|
|
74
74
|
"title": title,
|
|
75
|
-
"domain":
|
|
75
|
+
"domain": domain,
|
|
76
76
|
"question": question,
|
|
77
77
|
"research_mode": research_mode,
|
|
78
78
|
"decision_target": _default_decision_target(research_mode),
|