eduevidence 5.2.0 → 6.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +105 -0
- package/README.md +142 -75
- package/README.zh-CN.md +73 -30
- package/SKILL.md +397 -131
- package/agents/openai.yaml +4 -0
- package/assets/readme/controlled-execution.svg +34 -0
- package/assets/readme/landing-tour.gif +0 -0
- package/assets/readme/logo.png +0 -0
- package/assets/readme/research-workflow.svg +56 -0
- package/assets/readme/studio-graph.png +0 -0
- package/assets/readme/studio-overview.png +0 -0
- package/assets/readme/studio-reports.png +0 -0
- package/assets/readme/studio-tour.gif +0 -0
- package/autoevolve/config.yaml +17 -0
- package/autoevolve/program.md +25 -0
- package/autoevolve/protected.manifest.yaml +34 -0
- package/benchmarks/adversarial/cases.jsonl +7 -0
- package/benchmarks/evidence-library.json +5268 -0
- package/benchmarks/partitions.json +8 -0
- package/bin/eduevidence.js +2 -1
- package/docs/architecture.md +496 -0
- package/docs/autoresearch-evolution-plan.md +2903 -0
- package/docs/autoresearch-implementation-status.md +101 -0
- package/docs/demo-storyboard.md +20 -0
- package/docs/demo-workplace-ai.md +92 -0
- package/docs/demo.md +32 -0
- package/docs/install-guide.md +150 -0
- package/docs/orchestration-role-model.md +1254 -0
- package/docs/release-closeout/README.md +17 -0
- package/docs/release-closeout/frontend-acceptance.md +23 -0
- package/docs/release-closeout/issues.md +19 -0
- package/docs/release-closeout/verification.md +28 -0
- package/docs/release-contract.md +108 -0
- package/docs/research-studio-guide.zh-CN.md +166 -0
- package/docs/sciverse-api.md +125 -0
- package/eduevidence_cli.py +29 -13
- package/engine/_resources.py +13 -0
- package/engine/autoevolve/__init__.py +3 -0
- package/engine/autoevolve/agent_view.py +167 -0
- package/engine/autoevolve/core.py +357 -0
- package/engine/autoevolve/events.py +11 -0
- package/engine/autoevolve/git_workspace.py +77 -0
- package/engine/autoevolve/projection.py +23 -0
- package/engine/autoevolve/runner.py +413 -0
- package/engine/autoevolve/trust.py +146 -0
- package/engine/autoresearch/__init__.py +6 -0
- package/engine/autoresearch/commit.py +132 -0
- package/engine/autoresearch/contracts.py +126 -0
- package/engine/autoresearch/controller.py +207 -0
- package/engine/autoresearch/events.py +12 -0
- package/engine/autoresearch/gap_priority.py +168 -0
- package/engine/autoresearch/projection.py +30 -0
- package/engine/autoresearch/research_memory.py +59 -0
- package/engine/autoresearch/saturation.py +91 -0
- package/engine/briefs.py +2 -1
- package/engine/capabilities.py +1 -0
- package/engine/contracts.py +3 -1
- package/engine/decision_policy.py +96 -0
- package/engine/evidence_graph.py +14 -10
- package/engine/evidencecore.py +7 -5
- package/engine/gaps.py +132 -73
- package/engine/ids.py +2 -0
- package/engine/judge_pack.py +65 -0
- package/engine/library.py +6 -2
- package/engine/library_builtin.py +3 -1
- package/engine/living.py +36 -5
- package/engine/meta_synthesis.py +3 -1
- package/engine/migration.py +88 -3
- package/engine/orchestration.py +460 -0
- package/engine/paths.py +2 -0
- package/engine/pilot.py +36 -33
- package/engine/project.py +2 -2
- package/engine/research_service.py +113 -0
- package/engine/studio_read_model.py +400 -0
- package/engine/taxonomy.py +211 -0
- package/engine/tribunal.py +44 -33
- package/engine/update.py +1 -0
- package/engine/versions.py +1 -1
- package/engine/worker_result.py +109 -0
- package/engine/workflows.py +70 -0
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +2934 -0
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
- package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
- package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
- package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
- package/examples/ai-coding-assistant-evidence/frame.json +48 -0
- package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
- package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
- package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
- package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +230 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/result.json +1457 -0
- package/examples/ai-coding-assistant-evidence/result.zh.json +1457 -0
- package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
- package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +107 -0
- package/examples/spaced-retrieval-practice/applicability.json +14 -0
- package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
- package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
- package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
- package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
- package/examples/spaced-retrieval-practice/frame.json +58 -0
- package/examples/spaced-retrieval-practice/gate_report.json +101 -0
- package/examples/spaced-retrieval-practice/methodology.json +78 -0
- package/examples/spaced-retrieval-practice/report_spec.json +212 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/result.json +942 -0
- package/examples/spaced-retrieval-practice/result.zh.json +942 -0
- package/examples/spaced-retrieval-practice/skeptic.json +70 -0
- package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
- package/examples/spaced-retrieval-practice/verdict.json +93 -0
- package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -0
- package/examples/workplace-ai-assistant/evaluation.json +19 -0
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
- package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
- package/examples/workplace-ai-assistant/final_verdict.json +78 -0
- package/examples/workplace-ai-assistant/frame.json +41 -0
- package/examples/workplace-ai-assistant/gate_report.json +101 -0
- package/examples/workplace-ai-assistant/intervention.json +27 -0
- package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
- package/examples/workplace-ai-assistant/methodology.json +60 -0
- package/examples/workplace-ai-assistant/report_spec.json +224 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/result.json +615 -0
- package/examples/workplace-ai-assistant/result.zh.json +615 -0
- package/examples/workplace-ai-assistant/search_log.json +19 -0
- package/examples/workplace-ai-assistant/skeptic.json +72 -0
- package/examples/workplace-ai-assistant/sources.jsonl +3 -0
- package/examples/workplace-ai-assistant/validation_result.json +9 -0
- package/examples/workplace-ai-assistant/verdict.json +78 -0
- package/install.sh +7 -7
- package/integrations/agent_mcp.py +2 -2
- package/integrations/orchestration_dispatch.py +146 -0
- package/package.json +46 -3
- package/pyproject.toml +14 -22
- package/references/autoresearch.md +30 -0
- package/references/evaluation-policy.md +24 -0
- package/references/orchestration.md +22 -0
- package/references/report-copy-style.md +67 -0
- package/references/retrieval-compliance.md +75 -0
- package/references/retrieval-protocol.md +20 -0
- package/references/scientific-invariants.md +19 -0
- package/retrieval/audit.py +178 -0
- package/retrieval/fetch.py +96 -0
- package/retrieval/sciverse.py +398 -0
- package/retrieval/search.py +47 -7
- package/schemas/applicability.schema.json +94 -0
- package/schemas/chart-spec.schema.json +10 -3
- package/schemas/evidence.schema.json +316 -43
- package/schemas/fetch-result.schema.json +2 -1
- package/schemas/intervention.schema.json +106 -21
- package/schemas/report-result.schema.json +12 -4
- package/schemas/report-spec.schema.json +98 -100
- package/schemas/skeptic.schema.json +86 -0
- package/schemas/source.schema.json +21 -2
- package/schemas/v2/finding.schema.json +5 -1
- package/schemas/v2/methodology-audit.schema.json +5 -1
- package/schemas/v2/outcome.schema.json +28 -5
- package/schemas/v2/project.schema.json +2 -2
- package/schemas/v2/run.schema.json +1 -1
- package/schemas/v2/study.schema.json +5 -1
- package/schemas/vNext/autoevolve-session.schema.json +34 -0
- package/schemas/vNext/eval-snapshot.schema.json +77 -0
- package/schemas/vNext/execution-plan.schema.json +50 -0
- package/schemas/vNext/gap-priority.schema.json +54 -0
- package/schemas/vNext/negative-search-record.schema.json +68 -0
- package/schemas/vNext/research-iteration.schema.json +87 -0
- package/schemas/vNext/research-strategy.schema.json +62 -0
- package/schemas/vNext/skill-experiment.schema.json +90 -0
- package/schemas/vNext/task-spec.schema.json +156 -0
- package/schemas/vNext/worker-result.schema.json +60 -0
- package/schemas/verdict.schema.json +164 -28
- package/scripts/benchmark_judge.py +2 -2
- package/scripts/benchmark_v3.py +26 -43
- package/scripts/build_esl_artifacts.py +4 -4
- package/scripts/build_evidence_library.py +2 -2
- package/scripts/build_gh_pages.py +98 -0
- package/scripts/build_readme_diagrams.py +72 -0
- package/scripts/build_report_variants.py +101 -0
- package/scripts/build_result.py +74 -9
- package/scripts/check_autoresearch_invariants.py +95 -0
- package/scripts/check_package_parity.py +85 -0
- package/scripts/check_protocol_alignment.py +375 -0
- package/scripts/check_versioned_schemas.py +254 -0
- package/scripts/claim_audit.py +13 -8
- package/scripts/compute_confidence.py +10 -0
- package/scripts/daily_evolve.py +30 -0
- package/scripts/dashboard_server.py +130 -101
- package/scripts/did_regression.py +17 -32
- package/scripts/enrich_projects_human_and_lieflat.py +1 -1
- package/scripts/evidence_score.py +5 -2
- package/scripts/generate_metrics.py +4 -3
- package/scripts/generate_new_projects.py +5 -5
- package/scripts/orchestrator.py +286 -36
- package/scripts/pre_verdict_gate.py +224 -26
- package/scripts/quickstart.py +18 -2
- package/scripts/rebake_all_5themes.py +1 -2
- package/scripts/research_auto_cli.py +475 -0
- package/scripts/run_workspace.py +24 -8
- package/scripts/search_provenance.py +64 -0
- package/scripts/serve_web.py +9 -10
- package/scripts/skill_lint.py +1 -1
- package/scripts/skill_payload.py +81 -0
- package/scripts/test_adversarial_empirical.py +26 -19
- package/scripts/validate_schema.py +46 -2
- package/scripts/vnext_cli.py +133 -0
- package/setup.py +12 -0
- package/skill/agents/evaluation-designer.md +20 -4
- package/skill/agents/evidence-analyst.md +19 -3
- package/skill/agents/evidence-judge.md +50 -2
- package/skill/agents/evidence-retriever.md +20 -3
- package/skill/agents/intervention-designer.md +20 -4
- package/skill/agents/method-reviewer.md +18 -2
- package/skill/agents/{education-planner.md → research-planner.md} +19 -3
- package/skill/agents/skeptic.md +18 -2
- package/skill/roles/registry.yaml +45 -0
- package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
- package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
- package/skill/sub-skills/data-analysis/SKILL.md +34 -15
- package/skill/sub-skills/ethics-review/SKILL.md +33 -10
- package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
- package/skill/sub-skills/evidence-review/SKILL.md +31 -12
- package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
- package/skill/sub-skills/literature-review/SKILL.md +35 -14
- package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
- package/skill/sub-skills/report-generation/SKILL.md +40 -6
- package/skill/sub-skills/research-planning/SKILL.md +41 -14
- package/skill/sub-skills/study-design/SKILL.md +30 -9
- package/skill/task-briefs/adjudicate.md +32 -7
- package/skill/task-briefs/applicability.md +38 -0
- package/skill/task-briefs/audit.md +32 -7
- package/skill/task-briefs/challenge.md +34 -5
- package/skill/task-briefs/evaluate.md +30 -5
- package/skill/task-briefs/extract.md +31 -8
- package/skill/task-briefs/frame.md +39 -10
- package/skill/task-briefs/intervene.md +32 -6
- package/skill/task-briefs/present.md +32 -8
- package/skill/task-briefs/projection.md +37 -0
- package/skill/task-briefs/retrieve.md +36 -6
- package/skill/workflows/decision-and-pilot.md +85 -0
- package/skill/workflows/evaluate-and-update.md +93 -0
- package/skill/workflows/evidence-review.md +117 -0
- package/visualization/eduevidence-report/assets/base.css +2 -2
- package/visualization/eduevidence-report/assets/reader.css +752 -0
- package/visualization/eduevidence-report/assets/reader.js +132 -0
- package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
- package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
- package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
- package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
- package/visualization/eduevidence-report/scripts/build_report.py +561 -121
- package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +371 -136
- package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
- package/visualization/eduevidence-report/themes/academic.css +1 -1
- package/visualization/eduevidence-report/themes/claude.css +1 -1
- package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
- package/visualization/eduevidence-report/themes/datalab.css +2 -2
- package/visualization/eduevidence-report/themes/presentation.css +2 -2
- package/web/README.md +18 -0
- package/web/architecture.html +14885 -0
- package/web/index.html +53 -0
- package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
- package/web/studio/assets/index-B8tkF44Q.css +1 -0
- package/web/studio/assets/index-CQ6Keoyc.js +230 -0
- package/web/studio/config.json +1 -0
- package/web/studio/index.html +14 -0
- package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
- package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/bias.cpython-312.pyc +0 -0
- package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
- package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
- package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
- package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
- package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
- package/engine/__pycache__/events.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
- package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
- package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
- package/engine/__pycache__/ids.cpython-312.pyc +0 -0
- package/engine/__pycache__/library.cpython-312.pyc +0 -0
- package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
- package/engine/__pycache__/living.cpython-312.pyc +0 -0
- package/engine/__pycache__/log.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/migration.cpython-312.pyc +0 -0
- package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
- package/engine/__pycache__/paths.cpython-312.pyc +0 -0
- package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
- package/engine/__pycache__/planner.cpython-312.pyc +0 -0
- package/engine/__pycache__/project.cpython-312.pyc +0 -0
- package/engine/__pycache__/projections.cpython-312.pyc +0 -0
- package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
- package/engine/__pycache__/run.cpython-312.pyc +0 -0
- package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
- package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
- package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
- package/engine/__pycache__/update.cpython-312.pyc +0 -0
- package/engine/__pycache__/versions.cpython-312.pyc +0 -0
- package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
- package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
- package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
- package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
- package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
- package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
- package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
- package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
- package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
- package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
- package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
- package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
- package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
- package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
- package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
- package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
"""engine/taxonomy.py - Outcome taxonomy authority (domain registry backed).
|
|
2
|
+
|
|
3
|
+
Every domain registers its own outcome taxonomy in
|
|
4
|
+
``domains/<id>/outcome_taxonomy.json``. Each file declares the domain category
|
|
5
|
+
buckets (``categories``) and one entry per outcome token carrying an explicit
|
|
6
|
+
``category`` (``tokens[].category``).
|
|
7
|
+
|
|
8
|
+
This module is the ONLY reader of that contract. Before it existed, the token
|
|
9
|
+
set and the token-to-category mapping were hard-coded in ``engine/pilot.py``
|
|
10
|
+
with a ``.get(token, "learning")`` fallback, so a policy outcome was silently
|
|
11
|
+
classified as a learning outcome and policy runs could not complete. Callers
|
|
12
|
+
must no longer keep private copies of either table.
|
|
13
|
+
|
|
14
|
+
Fail-closed rule: an unknown token or an unregistered domain raises. Silent
|
|
15
|
+
classification is what produced the original defect; unknown input must be
|
|
16
|
+
visible as an error, never absorbed into a default bucket.
|
|
17
|
+
|
|
18
|
+
Stdlib only; results are cached per process.
|
|
19
|
+
"""
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import json
|
|
23
|
+
from typing import Any, Iterable
|
|
24
|
+
|
|
25
|
+
from engine.evidencecore import list_domains, load_domain
|
|
26
|
+
|
|
27
|
+
#: Bucket used when a caller must render an outcome whose category could not be
|
|
28
|
+
#: resolved. It is deliberately not a domain category: it marks the value as
|
|
29
|
+
#: unclassified instead of pretending it belongs to a real bucket.
|
|
30
|
+
UNCLASSIFIED = "unclassified"
|
|
31
|
+
|
|
32
|
+
_cache: dict[str, Any] = {}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class TaxonomyError(ValueError):
|
|
36
|
+
"""Raised when taxonomy data is missing, malformed, or unknown."""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _taxonomy_path(domain_id: str) -> str:
|
|
40
|
+
"""Registered relative path of a domain taxonomy file."""
|
|
41
|
+
entry = load_domain(domain_id)
|
|
42
|
+
raw = entry.get("outcome_taxonomy")
|
|
43
|
+
if not raw:
|
|
44
|
+
raise TaxonomyError(
|
|
45
|
+
"domain " + repr(domain_id) + " registers no outcome_taxonomy")
|
|
46
|
+
return str(raw).partition("#")[0]
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _load(domain_id: str) -> dict:
|
|
50
|
+
"""Load (and cache) one domain taxonomy, validating its shape."""
|
|
51
|
+
key = "taxonomy:" + domain_id
|
|
52
|
+
if key in _cache:
|
|
53
|
+
return _cache[key]
|
|
54
|
+
from engine._resources import resource_root
|
|
55
|
+
|
|
56
|
+
path = resource_root() / _taxonomy_path(domain_id)
|
|
57
|
+
if not path.is_file():
|
|
58
|
+
raise TaxonomyError(
|
|
59
|
+
"domain " + repr(domain_id) + ": taxonomy file missing: " + str(path))
|
|
60
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
61
|
+
categories = data.get("categories")
|
|
62
|
+
tokens = data.get("tokens")
|
|
63
|
+
if not isinstance(categories, dict) or not categories:
|
|
64
|
+
raise TaxonomyError(
|
|
65
|
+
"domain " + repr(domain_id) + ": taxonomy declares no categories")
|
|
66
|
+
if not isinstance(tokens, list) or not tokens:
|
|
67
|
+
raise TaxonomyError(
|
|
68
|
+
"domain " + repr(domain_id) + ": taxonomy declares no tokens")
|
|
69
|
+
seen: set[str] = set()
|
|
70
|
+
for item in tokens:
|
|
71
|
+
if not isinstance(item, dict) or not item.get("id"):
|
|
72
|
+
raise TaxonomyError(
|
|
73
|
+
"domain " + repr(domain_id) + ": token entry without an id")
|
|
74
|
+
token = str(item["id"])
|
|
75
|
+
if token in seen:
|
|
76
|
+
raise TaxonomyError(
|
|
77
|
+
"domain " + repr(domain_id) + ": duplicate token " + repr(token))
|
|
78
|
+
seen.add(token)
|
|
79
|
+
category = item.get("category")
|
|
80
|
+
if not category:
|
|
81
|
+
raise TaxonomyError(
|
|
82
|
+
"domain " + repr(domain_id) + ": token " + repr(token)
|
|
83
|
+
+ " declares no category (silent defaults are forbidden)")
|
|
84
|
+
if str(category) not in categories:
|
|
85
|
+
raise TaxonomyError(
|
|
86
|
+
"domain " + repr(domain_id) + ": token " + repr(token)
|
|
87
|
+
+ " uses category " + repr(category) + " absent from "
|
|
88
|
+
+ repr(sorted(categories)))
|
|
89
|
+
_cache[key] = data
|
|
90
|
+
return data
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _domain_ids(domain_id: str | None = None) -> tuple[str, ...]:
|
|
94
|
+
if domain_id:
|
|
95
|
+
return (domain_id,)
|
|
96
|
+
return tuple(d["id"] for d in list_domains())
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def tokens(domain_id: str) -> tuple[str, ...]:
|
|
100
|
+
"""Outcome tokens declared by one domain, in registry order."""
|
|
101
|
+
return tuple(str(item["id"]) for item in _load(domain_id)["tokens"])
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def categories(domain_id: str) -> dict[str, dict]:
|
|
105
|
+
"""Category buckets declared by one domain (id -> descriptor)."""
|
|
106
|
+
return dict(_load(domain_id)["categories"])
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def category_of(domain_id: str, token: str) -> str:
|
|
110
|
+
"""Category bucket for a token in a domain.
|
|
111
|
+
|
|
112
|
+
Raises TaxonomyError for an unknown token: a caller that cannot classify a
|
|
113
|
+
value must surface that, not guess. Use category_of_or_unclassified at
|
|
114
|
+
rendering boundaries where an unclassified value is acceptable.
|
|
115
|
+
"""
|
|
116
|
+
for item in _load(domain_id)["tokens"]:
|
|
117
|
+
if str(item["id"]) == token:
|
|
118
|
+
return str(item["category"])
|
|
119
|
+
raise TaxonomyError(
|
|
120
|
+
"domain " + repr(domain_id) + ": unknown outcome token " + repr(token)
|
|
121
|
+
+ " (" + str(len(tokens(domain_id))) + " tokens known)")
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def category_of_or_unclassified(domain_id: str, token: str) -> str:
|
|
125
|
+
"""Rendering-safe variant: unknown tokens map to UNCLASSIFIED."""
|
|
126
|
+
try:
|
|
127
|
+
return category_of(domain_id, token)
|
|
128
|
+
except TaxonomyError:
|
|
129
|
+
return UNCLASSIFIED
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def category_labels(domain_id: str, lang: str = "zh") -> dict[str, str]:
|
|
133
|
+
"""Display label per category bucket (falls back to the category id)."""
|
|
134
|
+
suffix = "_en" if lang == "en" else "_zh"
|
|
135
|
+
out: dict[str, str] = {}
|
|
136
|
+
for key, descriptor in categories(domain_id).items():
|
|
137
|
+
if isinstance(descriptor, dict):
|
|
138
|
+
label = descriptor.get("name" + suffix) or descriptor.get("name")
|
|
139
|
+
out[key] = str(label or key)
|
|
140
|
+
else:
|
|
141
|
+
out[key] = key
|
|
142
|
+
return out
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def all_tokens() -> dict[str, str]:
|
|
146
|
+
"""Every registered token -> owning domain, across registered domains.
|
|
147
|
+
|
|
148
|
+
A token registered by two domains is a contract conflict and raises: the
|
|
149
|
+
token would otherwise mean different things depending on lookup order.
|
|
150
|
+
"""
|
|
151
|
+
key = "all_tokens"
|
|
152
|
+
if key in _cache:
|
|
153
|
+
return _cache[key]
|
|
154
|
+
out: dict[str, str] = {}
|
|
155
|
+
for domain_id in _domain_ids():
|
|
156
|
+
for token in tokens(domain_id):
|
|
157
|
+
owner = out.get(token)
|
|
158
|
+
if owner is not None and owner != domain_id:
|
|
159
|
+
raise TaxonomyError(
|
|
160
|
+
"outcome token " + repr(token) + " is registered by both "
|
|
161
|
+
+ repr(owner) + " and " + repr(domain_id))
|
|
162
|
+
out[token] = domain_id
|
|
163
|
+
_cache[key] = out
|
|
164
|
+
return out
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def domain_of(token: str, default: str = "education") -> str:
|
|
168
|
+
"""Owning domain for a token; default when the token is unregistered."""
|
|
169
|
+
return all_tokens().get(token, default)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def all_tokens_ordered() -> tuple[str, ...]:
|
|
173
|
+
"""Every registered token in domain-registry then taxonomy order."""
|
|
174
|
+
ordered: list[str] = []
|
|
175
|
+
for domain_id in _domain_ids():
|
|
176
|
+
ordered.extend(tokens(domain_id))
|
|
177
|
+
return tuple(ordered)
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def all_categories() -> tuple[str, ...]:
|
|
181
|
+
"""Every category bucket across domains, de-duplicated, order preserved."""
|
|
182
|
+
ordered: list[str] = []
|
|
183
|
+
for domain_id in _domain_ids():
|
|
184
|
+
for name in categories(domain_id):
|
|
185
|
+
if name not in ordered:
|
|
186
|
+
ordered.append(name)
|
|
187
|
+
return tuple(ordered)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def categories_for_tokens(
|
|
191
|
+
token_list: Iterable[str], domain_id: str = "education"
|
|
192
|
+
) -> dict[str, list[str]]:
|
|
193
|
+
"""Group tokens by category; unregistered tokens land in UNCLASSIFIED."""
|
|
194
|
+
grouped: dict[str, list[str]] = {}
|
|
195
|
+
for token in token_list:
|
|
196
|
+
bucket = category_of_or_unclassified(domain_id, token)
|
|
197
|
+
grouped.setdefault(bucket, []).append(token)
|
|
198
|
+
return grouped
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def reset_cache() -> None:
|
|
202
|
+
"""Drop memoised taxonomies (tests and long-lived processes)."""
|
|
203
|
+
_cache.clear()
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
__all__ = [
|
|
207
|
+
"UNCLASSIFIED", "TaxonomyError",
|
|
208
|
+
"tokens", "categories", "category_of", "category_of_or_unclassified",
|
|
209
|
+
"category_labels", "all_tokens", "all_tokens_ordered", "all_categories",
|
|
210
|
+
"categories_for_tokens", "domain_of", "reset_cache",
|
|
211
|
+
]
|
package/engine/tribunal.py
CHANGED
|
@@ -38,8 +38,15 @@ from datetime import datetime, timezone
|
|
|
38
38
|
from pathlib import Path
|
|
39
39
|
|
|
40
40
|
from engine.contracts import validate_record
|
|
41
|
+
from engine.decision_policy import (
|
|
42
|
+
ADOPT_DIRECTNESS,
|
|
43
|
+
decision_action as _policy_decision_action,
|
|
44
|
+
outcome_category,
|
|
45
|
+
primary_effect_categories,
|
|
46
|
+
)
|
|
41
47
|
from engine.graph_store import GraphStore
|
|
42
48
|
from engine.ids import new_local_id
|
|
49
|
+
from engine.project import ProjectWorkspace
|
|
43
50
|
from engine.semantics import claim_relation, decision_implication
|
|
44
51
|
from engine.synthesis import ClaimSynthesis, synthesize_project
|
|
45
52
|
from engine.versions import (
|
|
@@ -229,16 +236,18 @@ def _confidence(store: GraphStore, syntheses: tuple[ClaimSynthesis, ...]) -> dic
|
|
|
229
236
|
"decisive_relations": dict(decisive)}
|
|
230
237
|
|
|
231
238
|
|
|
232
|
-
def
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
239
|
+
def _has_direct_primary_evidence(store: GraphStore,
|
|
240
|
+
decisive_relations: dict[str, str],
|
|
241
|
+
domain: str = "education") -> bool:
|
|
242
|
+
"""True when a decisive support_adoption Study measures a primary outcome.
|
|
236
243
|
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
244
|
+
Primary means the finding's outcome_type maps — through the domain
|
|
245
|
+
registry — into one of :func:`primary_effect_categories` (education:
|
|
246
|
+
learning; policy: effectiveness/cost), AND its evidence link carries
|
|
247
|
+
directness == 2. Task-performance, process and risk outcomes never
|
|
248
|
+
qualify, and a missing outcome record or missing directness fails closed.
|
|
241
249
|
"""
|
|
250
|
+
primary = primary_effect_categories(domain)
|
|
242
251
|
findings = {f["finding_id"]: f for f in store.read_table("findings")}
|
|
243
252
|
outcomes = {o["outcome_id"]: o for o in store.read_table("outcomes")}
|
|
244
253
|
links_by_finding: dict[str, list[dict]] = {}
|
|
@@ -253,39 +262,37 @@ def _has_direct_learning_evidence(store: GraphStore,
|
|
|
253
262
|
if fnd.get("study_id") not in support_studies:
|
|
254
263
|
continue
|
|
255
264
|
outcome = outcomes.get(fnd.get("outcome_id"))
|
|
256
|
-
if outcome is None
|
|
265
|
+
if outcome is None:
|
|
266
|
+
continue
|
|
267
|
+
value = str(outcome.get("outcome_type") or "")
|
|
268
|
+
category = outcome_category(domain, value, primary)
|
|
269
|
+
# Unknown value: fail closed rather than treating it as
|
|
270
|
+
# decision-grade evidence.
|
|
271
|
+
if category is None:
|
|
272
|
+
continue
|
|
273
|
+
if category not in primary:
|
|
257
274
|
continue
|
|
258
275
|
for link in links_by_finding.get(fid, []):
|
|
259
276
|
directness = link.get("directness")
|
|
260
|
-
if isinstance(directness, (int, float))
|
|
277
|
+
if (isinstance(directness, (int, float))
|
|
278
|
+
and int(directness) == ADOPT_DIRECTNESS):
|
|
261
279
|
return True
|
|
262
280
|
return False
|
|
263
281
|
|
|
264
282
|
|
|
265
283
|
def _decision_action(syn_statuses: dict[str, str], confidence: dict,
|
|
266
284
|
decisive_relations: dict[str, str],
|
|
267
|
-
|
|
268
|
-
"""Gate-enforced decision action.
|
|
269
|
-
|
|
270
|
-
REJECT requires usable direct opposition evidence (an independent Study
|
|
271
|
-
folded to oppose_adoption). Low/Insufficient can never yield ADOPT.
|
|
272
|
-
ADOPT additionally requires direct learning/transfer evidence: High +
|
|
273
|
-
decisive support WITHOUT a direct learning outcome downgrades to PILOT
|
|
274
|
-
(task performance / procedural efficiency is not learning). Moderate +
|
|
275
|
-
decisive support → PILOT; otherwise INSUFFICIENT_EVIDENCE.
|
|
276
|
-
"""
|
|
277
|
-
label = confidence["label"]
|
|
278
|
-
has_oppose = any(r == "oppose_adoption" for r in decisive_relations.values())
|
|
279
|
-
has_support = any(r == "support_adoption" for r in decisive_relations.values())
|
|
280
|
-
if has_oppose:
|
|
281
|
-
return "REJECT"
|
|
282
|
-
if label == "High" and has_support and has_direct_learning_evidence:
|
|
283
|
-
return "ADOPT"
|
|
284
|
-
if label in ("High", "Moderate") and has_support:
|
|
285
|
-
return "PILOT"
|
|
286
|
-
return "INSUFFICIENT_EVIDENCE"
|
|
287
|
-
|
|
285
|
+
has_direct_primary_evidence: bool = False) -> str:
|
|
286
|
+
"""Gate-enforced decision action; rule lives in engine.decision_policy.
|
|
288
287
|
|
|
288
|
+
The V1 Pre-Verdict Gate enforces the same rule, so the thresholds are
|
|
289
|
+
imported rather than restated here.
|
|
290
|
+
"""
|
|
291
|
+
return _policy_decision_action(
|
|
292
|
+
confidence_label=confidence["label"],
|
|
293
|
+
decisive_relations=decisive_relations,
|
|
294
|
+
has_direct_primary_evidence=has_direct_primary_evidence,
|
|
295
|
+
)
|
|
289
296
|
|
|
290
297
|
|
|
291
298
|
|
|
@@ -300,10 +307,11 @@ def adjudicate(store: GraphStore, *, project: ProjectWorkspace,
|
|
|
300
307
|
|
|
301
308
|
syn_statuses = {s.claim_id: s.status for s in syntheses}
|
|
302
309
|
decisive_relations = confidence.get("decisive_relations", {})
|
|
303
|
-
|
|
310
|
+
domain = str(project.manifest().get("domain") or "education")
|
|
311
|
+
direct_learning = _has_direct_primary_evidence(store, decisive_relations, domain)
|
|
304
312
|
|
|
305
313
|
decision = _decision_action(syn_statuses, confidence, decisive_relations,
|
|
306
|
-
|
|
314
|
+
has_direct_primary_evidence=direct_learning)
|
|
307
315
|
|
|
308
316
|
key_links: list[str] = []
|
|
309
317
|
for syn in syntheses:
|
|
@@ -352,6 +360,9 @@ def adjudicate(store: GraphStore, *, project: ProjectWorkspace,
|
|
|
352
360
|
"extensions": {"confidence_components": {
|
|
353
361
|
"decisive_studies": confidence.get("decisive_studies", 0),
|
|
354
362
|
"usable_studies": confidence.get("usable_studies", 0),
|
|
363
|
+
"has_direct_primary_evidence": direct_learning,
|
|
364
|
+
# Backward-compatible alias under the education-era name,
|
|
365
|
+
# so readers written against the older contract keep working.
|
|
355
366
|
"has_direct_learning_evidence": direct_learning,
|
|
356
367
|
}},
|
|
357
368
|
}
|
package/engine/update.py
CHANGED
package/engine/versions.py
CHANGED
|
@@ -4,7 +4,7 @@ Policy versions are frozen identifiers, not free-form strings: changing a
|
|
|
4
4
|
policy requires a new version, never silent mutation of an existing one.
|
|
5
5
|
"""
|
|
6
6
|
|
|
7
|
-
ENGINE_VERSION = "
|
|
7
|
+
ENGINE_VERSION = "6.2.0"
|
|
8
8
|
GRAPH_SCHEMA_VERSION = "2.0"
|
|
9
9
|
SOURCE_VALIDATION_POLICY_VERSION = "2026-08-12.v2"
|
|
10
10
|
METHODOLOGY_POLICY_VERSION = "2026-08-12.v2"
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""Validation boundary between delegated workers and canonical reasoning.
|
|
2
|
+
|
|
3
|
+
A worker may return candidate staging artifacts and prose, but it cannot mark
|
|
4
|
+
its own output trustworthy. The lead/main process reconstructs WorkerResult,
|
|
5
|
+
checks it against the originating TaskSpec and artifact validators, and only
|
|
6
|
+
then may validated artifacts enter a Judge context.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from typing import Any, Callable
|
|
12
|
+
|
|
13
|
+
from engine.orchestration import CANONICAL_STATE_ARTIFACTS, TaskSpec
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
ArtifactValidator = Callable[[dict[str, Any]], list[str]]
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True)
|
|
20
|
+
class WorkerResult:
|
|
21
|
+
task_id: str
|
|
22
|
+
status: str
|
|
23
|
+
staging_artifacts: tuple[dict[str, Any], ...]
|
|
24
|
+
validated: bool
|
|
25
|
+
validation_issues: tuple[str, ...] = ()
|
|
26
|
+
metrics: dict[str, Any] = field(default_factory=dict)
|
|
27
|
+
summary: str = ""
|
|
28
|
+
|
|
29
|
+
def to_dict(self) -> dict[str, Any]:
|
|
30
|
+
return {
|
|
31
|
+
"task_id": self.task_id,
|
|
32
|
+
"status": self.status,
|
|
33
|
+
"staging_artifacts": [dict(item) for item in self.staging_artifacts],
|
|
34
|
+
"validated": self.validated,
|
|
35
|
+
"validation_issues": list(self.validation_issues),
|
|
36
|
+
"metrics": dict(self.metrics),
|
|
37
|
+
"summary": self.summary,
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def validate_worker_output(
|
|
42
|
+
task: TaskSpec,
|
|
43
|
+
raw: dict[str, Any],
|
|
44
|
+
*,
|
|
45
|
+
artifact_validator: ArtifactValidator | None = None,
|
|
46
|
+
) -> WorkerResult:
|
|
47
|
+
"""Validate untrusted worker output in the lead process.
|
|
48
|
+
|
|
49
|
+
Any worker-supplied `validated` field is deliberately ignored.
|
|
50
|
+
"""
|
|
51
|
+
task.validate_for_dispatch()
|
|
52
|
+
issues: list[str] = []
|
|
53
|
+
if not isinstance(raw, dict):
|
|
54
|
+
raise ValueError("worker output must be an object")
|
|
55
|
+
if raw.get("task_id") != task.task_id:
|
|
56
|
+
issues.append("task_id mismatch")
|
|
57
|
+
status = str(raw.get("status", "failed"))
|
|
58
|
+
if status not in {"completed", "failed", "blocked"}:
|
|
59
|
+
issues.append(f"invalid worker status: {status}")
|
|
60
|
+
status = "failed"
|
|
61
|
+
artifacts = raw.get("staging_artifacts", [])
|
|
62
|
+
if not isinstance(artifacts, list) or any(not isinstance(item, dict) for item in artifacts):
|
|
63
|
+
issues.append("staging_artifacts must be a list of objects")
|
|
64
|
+
artifacts = []
|
|
65
|
+
|
|
66
|
+
expected = set(task.expected_outputs)
|
|
67
|
+
for index, artifact in enumerate(artifacts):
|
|
68
|
+
artifact_type = str(artifact.get("artifact_type", ""))
|
|
69
|
+
if not artifact_type:
|
|
70
|
+
issues.append(f"artifact[{index}] missing artifact_type")
|
|
71
|
+
continue
|
|
72
|
+
if artifact_type in CANONICAL_STATE_ARTIFACTS:
|
|
73
|
+
issues.append(f"artifact[{index}] attempts canonical output {artifact_type}")
|
|
74
|
+
if expected and artifact_type not in expected:
|
|
75
|
+
issues.append(
|
|
76
|
+
f"artifact[{index}] type {artifact_type} not allowed by TaskSpec output contract"
|
|
77
|
+
)
|
|
78
|
+
if artifact_validator is not None:
|
|
79
|
+
issues.extend(
|
|
80
|
+
f"artifact[{index}]: {problem}"
|
|
81
|
+
for problem in artifact_validator(artifact)
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
validated = status == "completed" and not issues
|
|
85
|
+
return WorkerResult(
|
|
86
|
+
task_id=task.task_id,
|
|
87
|
+
status=status,
|
|
88
|
+
staging_artifacts=tuple(dict(item) for item in artifacts),
|
|
89
|
+
validated=validated,
|
|
90
|
+
validation_issues=tuple(issues),
|
|
91
|
+
metrics=dict(raw.get("metrics") or {}),
|
|
92
|
+
summary=str(raw.get("summary", "")),
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def require_validated_artifacts_for_judge(
|
|
97
|
+
results: list[WorkerResult] | tuple[WorkerResult, ...],
|
|
98
|
+
) -> list[dict[str, Any]]:
|
|
99
|
+
"""Fail closed if a Judge context contains any unvalidated worker result."""
|
|
100
|
+
invalid = [result.task_id for result in results if not result.validated]
|
|
101
|
+
if invalid:
|
|
102
|
+
raise PermissionError(
|
|
103
|
+
"Judge may consume validated staging artifacts only; rejected task results: "
|
|
104
|
+
+ ",".join(invalid)
|
|
105
|
+
)
|
|
106
|
+
artifacts: list[dict[str, Any]] = []
|
|
107
|
+
for result in results:
|
|
108
|
+
artifacts.extend(dict(item) for item in result.staging_artifacts)
|
|
109
|
+
return artifacts
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""Canonical workflow registry for the Decision-Grade Evidence Engine.
|
|
2
|
+
|
|
3
|
+
The scientific protocol is deliberately independent from execution adapters and
|
|
4
|
+
projections. A renderer may fail or be replaced without changing what counts
|
|
5
|
+
as completed scientific work.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
SCIENTIFIC_STAGE_IDS = (
|
|
13
|
+
"frame", "retrieve", "extract", "challenge", "audit", "adjudicate",
|
|
14
|
+
"applicability", "intervene", "evaluate",
|
|
15
|
+
)
|
|
16
|
+
PROJECTION_STAGE_ID = "projection"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True)
|
|
20
|
+
class WorkflowSpec:
|
|
21
|
+
workflow_id: str
|
|
22
|
+
title: str
|
|
23
|
+
stage_ids: tuple[str, ...]
|
|
24
|
+
capability_ids: tuple[str, ...]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
_EVIDENCE_REVIEW = (
|
|
28
|
+
"research_framing", "literature_search", "counter_evidence_search",
|
|
29
|
+
"source_fetch", "source_validation", "study_extraction",
|
|
30
|
+
"finding_extraction", "methodology_appraisal", "claim_linking",
|
|
31
|
+
"evidence_synthesis", "tribunal", "applicability_analysis",
|
|
32
|
+
"knowledge_gap_detection",
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
_WORKFLOWS = {
|
|
36
|
+
"evidence_review": WorkflowSpec(
|
|
37
|
+
"evidence_review", "Evidence Review",
|
|
38
|
+
SCIENTIFIC_STAGE_IDS[:7], _EVIDENCE_REVIEW,
|
|
39
|
+
),
|
|
40
|
+
"decision_and_pilot": WorkflowSpec(
|
|
41
|
+
"decision_and_pilot", "Decision & Pilot",
|
|
42
|
+
SCIENTIFIC_STAGE_IDS[:8], _EVIDENCE_REVIEW + ("intervention_design",),
|
|
43
|
+
),
|
|
44
|
+
"evaluate_and_update": WorkflowSpec(
|
|
45
|
+
"evaluate_and_update", "Evaluate & Update",
|
|
46
|
+
("evaluate",), ("evaluation_design", "data_validation", "data_analysis"),
|
|
47
|
+
),
|
|
48
|
+
"full_research_cycle": WorkflowSpec(
|
|
49
|
+
"full_research_cycle", "Full Research Cycle",
|
|
50
|
+
SCIENTIFIC_STAGE_IDS,
|
|
51
|
+
_EVIDENCE_REVIEW + ("intervention_design", "evaluation_design", "data_validation", "data_analysis"),
|
|
52
|
+
),
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def workflow_registry() -> dict[str, WorkflowSpec]:
|
|
57
|
+
"""Return the immutable workflow catalogue keyed by user intent."""
|
|
58
|
+
return dict(_WORKFLOWS)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def workflow(workflow_id: str) -> WorkflowSpec:
|
|
62
|
+
try:
|
|
63
|
+
return _WORKFLOWS[workflow_id]
|
|
64
|
+
except KeyError as exc:
|
|
65
|
+
raise ValueError(f"unknown workflow {workflow_id!r}") from exc
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def execution_stages() -> tuple[str, ...]:
|
|
69
|
+
"""Canonical execution order, with Projection explicitly outside science."""
|
|
70
|
+
return SCIENTIFIC_STAGE_IDS + (PROJECTION_STAGE_ID,)
|