eduevidence 5.2.0 → 6.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +105 -0
- package/README.md +142 -75
- package/README.zh-CN.md +73 -30
- package/SKILL.md +397 -131
- package/agents/openai.yaml +4 -0
- package/assets/readme/controlled-execution.svg +34 -0
- package/assets/readme/landing-tour.gif +0 -0
- package/assets/readme/logo.png +0 -0
- package/assets/readme/research-workflow.svg +56 -0
- package/assets/readme/studio-graph.png +0 -0
- package/assets/readme/studio-overview.png +0 -0
- package/assets/readme/studio-reports.png +0 -0
- package/assets/readme/studio-tour.gif +0 -0
- package/autoevolve/config.yaml +17 -0
- package/autoevolve/program.md +25 -0
- package/autoevolve/protected.manifest.yaml +34 -0
- package/benchmarks/adversarial/cases.jsonl +7 -0
- package/benchmarks/evidence-library.json +5268 -0
- package/benchmarks/partitions.json +8 -0
- package/bin/eduevidence.js +2 -1
- package/docs/architecture.md +496 -0
- package/docs/autoresearch-evolution-plan.md +2903 -0
- package/docs/autoresearch-implementation-status.md +101 -0
- package/docs/demo-storyboard.md +20 -0
- package/docs/demo-workplace-ai.md +92 -0
- package/docs/demo.md +32 -0
- package/docs/install-guide.md +150 -0
- package/docs/orchestration-role-model.md +1254 -0
- package/docs/release-closeout/README.md +17 -0
- package/docs/release-closeout/frontend-acceptance.md +23 -0
- package/docs/release-closeout/issues.md +19 -0
- package/docs/release-closeout/verification.md +28 -0
- package/docs/release-contract.md +108 -0
- package/docs/research-studio-guide.zh-CN.md +166 -0
- package/docs/sciverse-api.md +125 -0
- package/eduevidence_cli.py +29 -13
- package/engine/_resources.py +13 -0
- package/engine/autoevolve/__init__.py +3 -0
- package/engine/autoevolve/agent_view.py +167 -0
- package/engine/autoevolve/core.py +357 -0
- package/engine/autoevolve/events.py +11 -0
- package/engine/autoevolve/git_workspace.py +77 -0
- package/engine/autoevolve/projection.py +23 -0
- package/engine/autoevolve/runner.py +413 -0
- package/engine/autoevolve/trust.py +146 -0
- package/engine/autoresearch/__init__.py +6 -0
- package/engine/autoresearch/commit.py +132 -0
- package/engine/autoresearch/contracts.py +126 -0
- package/engine/autoresearch/controller.py +207 -0
- package/engine/autoresearch/events.py +12 -0
- package/engine/autoresearch/gap_priority.py +168 -0
- package/engine/autoresearch/projection.py +30 -0
- package/engine/autoresearch/research_memory.py +59 -0
- package/engine/autoresearch/saturation.py +91 -0
- package/engine/briefs.py +2 -1
- package/engine/capabilities.py +1 -0
- package/engine/contracts.py +3 -1
- package/engine/decision_policy.py +96 -0
- package/engine/evidence_graph.py +14 -10
- package/engine/evidencecore.py +7 -5
- package/engine/gaps.py +132 -73
- package/engine/ids.py +2 -0
- package/engine/judge_pack.py +65 -0
- package/engine/library.py +6 -2
- package/engine/library_builtin.py +3 -1
- package/engine/living.py +36 -5
- package/engine/meta_synthesis.py +3 -1
- package/engine/migration.py +88 -3
- package/engine/orchestration.py +460 -0
- package/engine/paths.py +2 -0
- package/engine/pilot.py +36 -33
- package/engine/project.py +2 -2
- package/engine/research_service.py +113 -0
- package/engine/studio_read_model.py +400 -0
- package/engine/taxonomy.py +211 -0
- package/engine/tribunal.py +44 -33
- package/engine/update.py +1 -0
- package/engine/versions.py +1 -1
- package/engine/worker_result.py +109 -0
- package/engine/workflows.py +70 -0
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +2934 -0
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
- package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
- package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
- package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
- package/examples/ai-coding-assistant-evidence/frame.json +48 -0
- package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
- package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
- package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
- package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +230 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/result.json +1457 -0
- package/examples/ai-coding-assistant-evidence/result.zh.json +1457 -0
- package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
- package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +107 -0
- package/examples/spaced-retrieval-practice/applicability.json +14 -0
- package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
- package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
- package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
- package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
- package/examples/spaced-retrieval-practice/frame.json +58 -0
- package/examples/spaced-retrieval-practice/gate_report.json +101 -0
- package/examples/spaced-retrieval-practice/methodology.json +78 -0
- package/examples/spaced-retrieval-practice/report_spec.json +212 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/result.json +942 -0
- package/examples/spaced-retrieval-practice/result.zh.json +942 -0
- package/examples/spaced-retrieval-practice/skeptic.json +70 -0
- package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
- package/examples/spaced-retrieval-practice/verdict.json +93 -0
- package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -0
- package/examples/workplace-ai-assistant/evaluation.json +19 -0
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
- package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
- package/examples/workplace-ai-assistant/final_verdict.json +78 -0
- package/examples/workplace-ai-assistant/frame.json +41 -0
- package/examples/workplace-ai-assistant/gate_report.json +101 -0
- package/examples/workplace-ai-assistant/intervention.json +27 -0
- package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
- package/examples/workplace-ai-assistant/methodology.json +60 -0
- package/examples/workplace-ai-assistant/report_spec.json +224 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/result.json +615 -0
- package/examples/workplace-ai-assistant/result.zh.json +615 -0
- package/examples/workplace-ai-assistant/search_log.json +19 -0
- package/examples/workplace-ai-assistant/skeptic.json +72 -0
- package/examples/workplace-ai-assistant/sources.jsonl +3 -0
- package/examples/workplace-ai-assistant/validation_result.json +9 -0
- package/examples/workplace-ai-assistant/verdict.json +78 -0
- package/install.sh +7 -7
- package/integrations/agent_mcp.py +2 -2
- package/integrations/orchestration_dispatch.py +146 -0
- package/package.json +46 -3
- package/pyproject.toml +14 -22
- package/references/autoresearch.md +30 -0
- package/references/evaluation-policy.md +24 -0
- package/references/orchestration.md +22 -0
- package/references/report-copy-style.md +67 -0
- package/references/retrieval-compliance.md +75 -0
- package/references/retrieval-protocol.md +20 -0
- package/references/scientific-invariants.md +19 -0
- package/retrieval/audit.py +178 -0
- package/retrieval/fetch.py +96 -0
- package/retrieval/sciverse.py +398 -0
- package/retrieval/search.py +47 -7
- package/schemas/applicability.schema.json +94 -0
- package/schemas/chart-spec.schema.json +10 -3
- package/schemas/evidence.schema.json +316 -43
- package/schemas/fetch-result.schema.json +2 -1
- package/schemas/intervention.schema.json +106 -21
- package/schemas/report-result.schema.json +12 -4
- package/schemas/report-spec.schema.json +98 -100
- package/schemas/skeptic.schema.json +86 -0
- package/schemas/source.schema.json +21 -2
- package/schemas/v2/finding.schema.json +5 -1
- package/schemas/v2/methodology-audit.schema.json +5 -1
- package/schemas/v2/outcome.schema.json +28 -5
- package/schemas/v2/project.schema.json +2 -2
- package/schemas/v2/run.schema.json +1 -1
- package/schemas/v2/study.schema.json +5 -1
- package/schemas/vNext/autoevolve-session.schema.json +34 -0
- package/schemas/vNext/eval-snapshot.schema.json +77 -0
- package/schemas/vNext/execution-plan.schema.json +50 -0
- package/schemas/vNext/gap-priority.schema.json +54 -0
- package/schemas/vNext/negative-search-record.schema.json +68 -0
- package/schemas/vNext/research-iteration.schema.json +87 -0
- package/schemas/vNext/research-strategy.schema.json +62 -0
- package/schemas/vNext/skill-experiment.schema.json +90 -0
- package/schemas/vNext/task-spec.schema.json +156 -0
- package/schemas/vNext/worker-result.schema.json +60 -0
- package/schemas/verdict.schema.json +164 -28
- package/scripts/benchmark_judge.py +2 -2
- package/scripts/benchmark_v3.py +26 -43
- package/scripts/build_esl_artifacts.py +4 -4
- package/scripts/build_evidence_library.py +2 -2
- package/scripts/build_gh_pages.py +98 -0
- package/scripts/build_readme_diagrams.py +72 -0
- package/scripts/build_report_variants.py +101 -0
- package/scripts/build_result.py +74 -9
- package/scripts/check_autoresearch_invariants.py +95 -0
- package/scripts/check_package_parity.py +85 -0
- package/scripts/check_protocol_alignment.py +375 -0
- package/scripts/check_versioned_schemas.py +254 -0
- package/scripts/claim_audit.py +13 -8
- package/scripts/compute_confidence.py +10 -0
- package/scripts/daily_evolve.py +30 -0
- package/scripts/dashboard_server.py +130 -101
- package/scripts/did_regression.py +17 -32
- package/scripts/enrich_projects_human_and_lieflat.py +1 -1
- package/scripts/evidence_score.py +5 -2
- package/scripts/generate_metrics.py +4 -3
- package/scripts/generate_new_projects.py +5 -5
- package/scripts/orchestrator.py +286 -36
- package/scripts/pre_verdict_gate.py +224 -26
- package/scripts/quickstart.py +18 -2
- package/scripts/rebake_all_5themes.py +1 -2
- package/scripts/research_auto_cli.py +475 -0
- package/scripts/run_workspace.py +24 -8
- package/scripts/search_provenance.py +64 -0
- package/scripts/serve_web.py +9 -10
- package/scripts/skill_lint.py +1 -1
- package/scripts/skill_payload.py +81 -0
- package/scripts/test_adversarial_empirical.py +26 -19
- package/scripts/validate_schema.py +46 -2
- package/scripts/vnext_cli.py +133 -0
- package/setup.py +12 -0
- package/skill/agents/evaluation-designer.md +20 -4
- package/skill/agents/evidence-analyst.md +19 -3
- package/skill/agents/evidence-judge.md +50 -2
- package/skill/agents/evidence-retriever.md +20 -3
- package/skill/agents/intervention-designer.md +20 -4
- package/skill/agents/method-reviewer.md +18 -2
- package/skill/agents/{education-planner.md → research-planner.md} +19 -3
- package/skill/agents/skeptic.md +18 -2
- package/skill/roles/registry.yaml +45 -0
- package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
- package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
- package/skill/sub-skills/data-analysis/SKILL.md +34 -15
- package/skill/sub-skills/ethics-review/SKILL.md +33 -10
- package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
- package/skill/sub-skills/evidence-review/SKILL.md +31 -12
- package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
- package/skill/sub-skills/literature-review/SKILL.md +35 -14
- package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
- package/skill/sub-skills/report-generation/SKILL.md +40 -6
- package/skill/sub-skills/research-planning/SKILL.md +41 -14
- package/skill/sub-skills/study-design/SKILL.md +30 -9
- package/skill/task-briefs/adjudicate.md +32 -7
- package/skill/task-briefs/applicability.md +38 -0
- package/skill/task-briefs/audit.md +32 -7
- package/skill/task-briefs/challenge.md +34 -5
- package/skill/task-briefs/evaluate.md +30 -5
- package/skill/task-briefs/extract.md +31 -8
- package/skill/task-briefs/frame.md +39 -10
- package/skill/task-briefs/intervene.md +32 -6
- package/skill/task-briefs/present.md +32 -8
- package/skill/task-briefs/projection.md +37 -0
- package/skill/task-briefs/retrieve.md +36 -6
- package/skill/workflows/decision-and-pilot.md +85 -0
- package/skill/workflows/evaluate-and-update.md +93 -0
- package/skill/workflows/evidence-review.md +117 -0
- package/visualization/eduevidence-report/assets/base.css +2 -2
- package/visualization/eduevidence-report/assets/reader.css +752 -0
- package/visualization/eduevidence-report/assets/reader.js +132 -0
- package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
- package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
- package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
- package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
- package/visualization/eduevidence-report/scripts/build_report.py +561 -121
- package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +371 -136
- package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
- package/visualization/eduevidence-report/themes/academic.css +1 -1
- package/visualization/eduevidence-report/themes/claude.css +1 -1
- package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
- package/visualization/eduevidence-report/themes/datalab.css +2 -2
- package/visualization/eduevidence-report/themes/presentation.css +2 -2
- package/web/README.md +18 -0
- package/web/architecture.html +14885 -0
- package/web/index.html +53 -0
- package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
- package/web/studio/assets/index-B8tkF44Q.css +1 -0
- package/web/studio/assets/index-CQ6Keoyc.js +230 -0
- package/web/studio/config.json +1 -0
- package/web/studio/index.html +14 -0
- package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
- package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/bias.cpython-312.pyc +0 -0
- package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
- package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
- package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
- package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
- package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
- package/engine/__pycache__/events.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
- package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
- package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
- package/engine/__pycache__/ids.cpython-312.pyc +0 -0
- package/engine/__pycache__/library.cpython-312.pyc +0 -0
- package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
- package/engine/__pycache__/living.cpython-312.pyc +0 -0
- package/engine/__pycache__/log.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/migration.cpython-312.pyc +0 -0
- package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
- package/engine/__pycache__/paths.cpython-312.pyc +0 -0
- package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
- package/engine/__pycache__/planner.cpython-312.pyc +0 -0
- package/engine/__pycache__/project.cpython-312.pyc +0 -0
- package/engine/__pycache__/projections.cpython-312.pyc +0 -0
- package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
- package/engine/__pycache__/run.cpython-312.pyc +0 -0
- package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
- package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
- package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
- package/engine/__pycache__/update.cpython-312.pyc +0 -0
- package/engine/__pycache__/versions.cpython-312.pyc +0 -0
- package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
- package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
- package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
- package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
- package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
- package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
- package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
- package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
- package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
- package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
- package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
- package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
- package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
- package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
- package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
- package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
package/engine/gaps.py
CHANGED
|
@@ -4,10 +4,15 @@ A KnowledgeGap is not free-form "future work": it is derived from coverage —
|
|
|
4
4
|
the research frame's requested outcomes vs what the graph's Findings
|
|
5
5
|
actually measure. A task-performance Finding never covers a retention or
|
|
6
6
|
transfer gap.
|
|
7
|
+
|
|
8
|
+
`gap_id` identifies one revision-local gap artifact. `extensions.autoresearch_key`
|
|
9
|
+
is a stable semantic lineage key so bounded research memory survives graph
|
|
10
|
+
revisions when the same unresolved gap is re-derived.
|
|
7
11
|
"""
|
|
8
12
|
|
|
9
13
|
from __future__ import annotations
|
|
10
14
|
|
|
15
|
+
import hashlib
|
|
11
16
|
import json
|
|
12
17
|
from pathlib import Path
|
|
13
18
|
|
|
@@ -16,23 +21,57 @@ from engine.graph_store import GraphStore
|
|
|
16
21
|
from engine.ids import new_local_id
|
|
17
22
|
from engine.synthesis import ClaimSynthesis
|
|
18
23
|
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
24
|
+
#: Gap kind -> the outcome categories that count as covering it. Categories are
|
|
25
|
+
#: read from the domain registry, so a policy run classifies its outcomes with
|
|
26
|
+
#: the policy buckets (effectiveness / cost / equity / feasibility / risk)
|
|
27
|
+
#: instead of being measured against education vocabulary.
|
|
28
|
+
GAP_KIND_CATEGORIES: dict[str, tuple[str, ...]] = {
|
|
29
|
+
"learning": ("learning",),
|
|
30
|
+
"retention": ("learning",),
|
|
31
|
+
"transfer": ("learning",),
|
|
32
|
+
"task_performance": ("task_performance",),
|
|
33
|
+
"process": ("process",),
|
|
34
|
+
"risk": ("risk",),
|
|
35
|
+
"effectiveness": ("effectiveness",),
|
|
36
|
+
"cost": ("cost",),
|
|
37
|
+
"equity": ("equity",),
|
|
38
|
+
"feasibility": ("feasibility",),
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
#: Legacy aliases accepted when a frame names its requested outcomes with older
|
|
42
|
+
#: vocabulary; they resolve to a gap kind above.
|
|
43
|
+
GAP_KIND_ALIASES: dict[str, str] = {
|
|
44
|
+
"long_term": "retention", "learning_retention": "retention",
|
|
45
|
+
"transfer_learning": "transfer", "far_transfer": "transfer",
|
|
46
|
+
"assignment_score": "task_performance", "task_completion": "task_performance",
|
|
47
|
+
"policy_effectiveness": "effectiveness", "cost_effectiveness": "cost",
|
|
48
|
+
"implementation_risk": "risk",
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _autoresearch_key(
|
|
53
|
+
gap_type: str,
|
|
54
|
+
*,
|
|
55
|
+
related_claims: list[str],
|
|
56
|
+
related_outcomes: list[str],
|
|
57
|
+
semantic_token: str,
|
|
58
|
+
) -> str:
|
|
59
|
+
payload = {
|
|
60
|
+
"gap_type": gap_type,
|
|
61
|
+
"related_claims": sorted(related_claims),
|
|
62
|
+
"related_outcomes": sorted(related_outcomes),
|
|
63
|
+
"semantic_token": semantic_token.strip().lower(),
|
|
64
|
+
}
|
|
65
|
+
digest = hashlib.sha256(
|
|
66
|
+
json.dumps(payload, sort_keys=True, ensure_ascii=False).encode("utf-8")
|
|
67
|
+
).hexdigest()[:20]
|
|
68
|
+
return f"KGK-{digest}"
|
|
25
69
|
|
|
26
70
|
|
|
27
71
|
def derive_gaps(*, store: GraphStore,
|
|
28
72
|
syntheses: tuple[ClaimSynthesis, ...] | None = None,
|
|
29
73
|
frame: dict | None = None) -> list[dict]:
|
|
30
|
-
"""Derive structured gaps from graph coverage vs the research frame.
|
|
31
|
-
|
|
32
|
-
`frame` carries `requested_outcomes` (list of outcome names/types) and
|
|
33
|
-
optionally `target_population`. Findings' outcome types come from the
|
|
34
|
-
graph's outcomes table.
|
|
35
|
-
"""
|
|
74
|
+
"""Derive structured gaps from graph coverage vs the research frame."""
|
|
36
75
|
frame = frame or {}
|
|
37
76
|
requested = frame.get("requested_outcomes") or []
|
|
38
77
|
if not requested and frame.get("target_outcomes"):
|
|
@@ -47,59 +86,58 @@ def derive_gaps(*, store: GraphStore,
|
|
|
47
86
|
covered_types.add(o.get("outcome_type", ""))
|
|
48
87
|
|
|
49
88
|
claims = store.read_table("claims")
|
|
50
|
-
claim_ids = [c["claim_id"] for c in claims]
|
|
51
|
-
|
|
52
89
|
gaps: list[dict] = []
|
|
53
90
|
rev = store.active_revision()
|
|
54
91
|
|
|
55
92
|
def add(gap_type: str, priority: str, reasoning: str,
|
|
56
93
|
related_claims: list[str] | None = None,
|
|
57
|
-
related_outcomes: list[str] | None = None
|
|
94
|
+
related_outcomes: list[str] | None = None,
|
|
95
|
+
semantic_token: str = ""):
|
|
96
|
+
related_claims = related_claims or []
|
|
97
|
+
related_outcomes = related_outcomes or []
|
|
98
|
+
key = _autoresearch_key(
|
|
99
|
+
gap_type,
|
|
100
|
+
related_claims=related_claims,
|
|
101
|
+
related_outcomes=related_outcomes,
|
|
102
|
+
semantic_token=semantic_token or reasoning,
|
|
103
|
+
)
|
|
58
104
|
gaps.append({
|
|
59
105
|
"gap_id": new_local_id("GAP", {g["gap_id"] for g in gaps}),
|
|
60
106
|
"gap_type": gap_type,
|
|
61
|
-
"related_claim_ids": related_claims
|
|
62
|
-
"related_outcome_ids": related_outcomes
|
|
107
|
+
"related_claim_ids": related_claims,
|
|
108
|
+
"related_outcome_ids": related_outcomes,
|
|
63
109
|
"priority": priority,
|
|
64
110
|
"reasoning": reasoning,
|
|
65
111
|
"status": "open",
|
|
66
112
|
"derived_from_graph_revision": rev,
|
|
67
|
-
"extensions": {},
|
|
113
|
+
"extensions": {"autoresearch_key": key},
|
|
68
114
|
})
|
|
69
115
|
|
|
70
116
|
def _req_kind(req) -> tuple[str, str]:
|
|
71
|
-
"""Classify a requested outcome: retention | transfer |
|
|
72
|
-
task_performance | learning | other. Type-aware: names are matched
|
|
73
|
-
only within the outcome's declared type, never type-blind."""
|
|
74
117
|
if isinstance(req, dict):
|
|
75
118
|
req_name = str(req.get("name", "")).lower()
|
|
76
119
|
req_type = str(req.get("outcome_type", "")).lower()
|
|
77
120
|
else:
|
|
78
121
|
req_name, req_type = str(req).lower(), ""
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
if
|
|
82
|
-
|
|
83
|
-
if
|
|
84
|
-
return
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
return "other", req.get("name", "") if isinstance(req, dict) else str(req)
|
|
88
|
-
|
|
89
|
-
_RETENTION_COVER = _RETENTION_TYPES
|
|
90
|
-
_TRANSFER_COVER = _TRANSFER_TYPES
|
|
122
|
+
label = req.get("name", "") if isinstance(req, dict) else str(req)
|
|
123
|
+
kind = GAP_KIND_ALIASES.get(req_type) or GAP_KIND_ALIASES.get(req_name)
|
|
124
|
+
if kind is None:
|
|
125
|
+
kind = req_type if req_type in GAP_KIND_CATEGORIES else req_name
|
|
126
|
+
if kind in GAP_KIND_CATEGORIES:
|
|
127
|
+
return kind, label
|
|
128
|
+
return "other", label
|
|
129
|
+
|
|
91
130
|
def covered_for_kind(kind: str) -> bool:
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
if
|
|
99
|
-
return
|
|
100
|
-
return
|
|
101
|
-
|
|
102
|
-
# one pass per requested outcome; each gap emitted exactly once
|
|
131
|
+
"""True when the graph already carries an outcome for this gap kind.
|
|
132
|
+
|
|
133
|
+
``covered_types`` holds the category buckets stored on outcomes, which
|
|
134
|
+
is why this compares categories rather than V1 tokens.
|
|
135
|
+
"""
|
|
136
|
+
expected = GAP_KIND_CATEGORIES.get(kind)
|
|
137
|
+
if not expected:
|
|
138
|
+
return False
|
|
139
|
+
return bool(covered_types & set(expected))
|
|
140
|
+
|
|
103
141
|
seen: set[tuple[str, str]] = set()
|
|
104
142
|
for req in requested:
|
|
105
143
|
kind, label = _req_kind(req)
|
|
@@ -112,49 +150,70 @@ def derive_gaps(*, store: GraphStore,
|
|
|
112
150
|
if covered_for_kind(kind):
|
|
113
151
|
continue
|
|
114
152
|
if kind == "retention":
|
|
115
|
-
add(
|
|
153
|
+
add(
|
|
154
|
+
"missing_retention", "high",
|
|
116
155
|
f"frame requests retention outcome {label!r} but the graph has "
|
|
117
|
-
|
|
118
|
-
|
|
156
|
+
"no retention-type measurement; task-performance coverage does "
|
|
157
|
+
"not count (RULE 3)",
|
|
158
|
+
semantic_token=f"requested_outcome:{label}",
|
|
159
|
+
)
|
|
119
160
|
elif kind == "transfer":
|
|
120
|
-
add(
|
|
161
|
+
add(
|
|
162
|
+
"missing_transfer", "high",
|
|
121
163
|
f"frame requests transfer outcome {label!r} but the graph has "
|
|
122
|
-
|
|
123
|
-
|
|
164
|
+
"no transfer-type measurement; AI-assisted task performance "
|
|
165
|
+
"does not count (RULE 3)",
|
|
166
|
+
semantic_token=f"requested_outcome:{label}",
|
|
167
|
+
)
|
|
124
168
|
elif kind == "task_performance":
|
|
125
|
-
add(
|
|
126
|
-
|
|
127
|
-
f"covering finding"
|
|
169
|
+
add(
|
|
170
|
+
"missing_outcome", "medium",
|
|
171
|
+
f"frame requests task-performance outcome {label!r} with no covering finding",
|
|
172
|
+
semantic_token=f"requested_outcome:{label}",
|
|
173
|
+
)
|
|
128
174
|
elif kind == "learning":
|
|
129
|
-
add(
|
|
130
|
-
|
|
131
|
-
f"learning
|
|
175
|
+
add(
|
|
176
|
+
"missing_outcome", "medium",
|
|
177
|
+
f"frame requests learning outcome {label!r} with no covering learning finding; "
|
|
178
|
+
"task performance is not learning (RULE 3)",
|
|
179
|
+
semantic_token=f"requested_outcome:{label}",
|
|
180
|
+
)
|
|
132
181
|
else:
|
|
133
|
-
add(
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
182
|
+
add(
|
|
183
|
+
"missing_outcome", "medium",
|
|
184
|
+
f"frame requests outcome {label!r} with no covering finding",
|
|
185
|
+
semantic_token=f"requested_outcome:{label}",
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
claim_outcomes = {
|
|
189
|
+
c["claim_id"]: c.get("primary_outcome_ids", [])
|
|
190
|
+
for c in claims
|
|
191
|
+
}
|
|
137
192
|
|
|
138
|
-
# contradiction gaps
|
|
139
193
|
for syn in syntheses or ():
|
|
140
194
|
if syn.status == "contested":
|
|
141
|
-
add(
|
|
195
|
+
add(
|
|
196
|
+
"unresolved_conflict", "high",
|
|
142
197
|
f"claim {syn.claim_id} has independent contradictory studies "
|
|
143
|
-
f"({', '.join(syn.study_ids)})",
|
|
144
|
-
|
|
198
|
+
f"({', '.join(syn.study_ids)})",
|
|
199
|
+
[syn.claim_id],
|
|
200
|
+
claim_outcomes.get(syn.claim_id, []),
|
|
201
|
+
semantic_token=f"claim:{syn.claim_id}",
|
|
202
|
+
)
|
|
145
203
|
|
|
146
|
-
# methodology weakness / insufficient independence
|
|
147
204
|
if syntheses:
|
|
148
205
|
for syn in syntheses:
|
|
149
206
|
if syn.status == "insufficient" and len(syn.study_ids) < 2:
|
|
150
|
-
add(
|
|
151
|
-
|
|
152
|
-
f"
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
207
|
+
add(
|
|
208
|
+
"insufficient_sample_independence", "medium",
|
|
209
|
+
f"claim {syn.claim_id} rests on fewer than two independent studies",
|
|
210
|
+
[syn.claim_id],
|
|
211
|
+
claim_outcomes.get(syn.claim_id, []),
|
|
212
|
+
semantic_token=f"claim:{syn.claim_id}",
|
|
213
|
+
)
|
|
214
|
+
|
|
215
|
+
for gap in gaps:
|
|
216
|
+
errors = validate_record("knowledge-gap", gap)
|
|
158
217
|
if errors:
|
|
159
218
|
raise ValueError(f"invalid gap: {errors}")
|
|
160
219
|
return gaps
|
package/engine/ids.py
CHANGED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""Build a transparent, self-describing judge evidence pack from real artifacts."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import hashlib
|
|
5
|
+
import json
|
|
6
|
+
import shutil
|
|
7
|
+
from datetime import datetime, timezone
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
from engine.project import ProjectWorkspace
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _sha256(path: Path) -> str:
|
|
15
|
+
h = hashlib.sha256()
|
|
16
|
+
with path.open("rb") as fh:
|
|
17
|
+
for block in iter(lambda: fh.read(65536), b""):
|
|
18
|
+
h.update(block)
|
|
19
|
+
return h.hexdigest()
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def export_judge_pack(project: ProjectWorkspace, output_dir: Path) -> dict[str, Any]:
|
|
23
|
+
"""Copy available project evidence without inventing unavailable claims.
|
|
24
|
+
|
|
25
|
+
The manifest lists every required judge-pack category and explicitly marks
|
|
26
|
+
missing inputs. This makes the pack suitable for review while keeping its
|
|
27
|
+
limits auditable.
|
|
28
|
+
"""
|
|
29
|
+
output_dir = Path(output_dir).resolve()
|
|
30
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
31
|
+
candidates = {
|
|
32
|
+
"project_manifest": project.path / "project.json",
|
|
33
|
+
"graph": project.path / "graph",
|
|
34
|
+
"runs": project.path / "runs",
|
|
35
|
+
"decisions": project.path / "decisions",
|
|
36
|
+
"projections": project.path / "projections",
|
|
37
|
+
"reports": project.path / "reports",
|
|
38
|
+
"pilots": project.path / "pilots",
|
|
39
|
+
}
|
|
40
|
+
copied: list[dict[str, str]] = []
|
|
41
|
+
missing: list[str] = []
|
|
42
|
+
for name, source in candidates.items():
|
|
43
|
+
target = output_dir / name
|
|
44
|
+
if source.is_file():
|
|
45
|
+
shutil.copy2(source, target)
|
|
46
|
+
copied.append({"name": name, "path": target.name, "sha256": _sha256(target)})
|
|
47
|
+
elif source.is_dir() and any(source.rglob("*")):
|
|
48
|
+
shutil.copytree(source, target, dirs_exist_ok=True)
|
|
49
|
+
for item in sorted(path for path in target.rglob("*") if path.is_file()):
|
|
50
|
+
copied.append({"name": name, "path": str(item.relative_to(output_dir)), "sha256": _sha256(item)})
|
|
51
|
+
else:
|
|
52
|
+
missing.append(name)
|
|
53
|
+
manifest = {
|
|
54
|
+
"format": "eduevidence-judge-pack/2026.09",
|
|
55
|
+
"project_id": project.project_id,
|
|
56
|
+
"created_at": datetime.now(timezone.utc).isoformat(),
|
|
57
|
+
"copied_files": copied,
|
|
58
|
+
"missing_categories": missing,
|
|
59
|
+
"limitations": [
|
|
60
|
+
"Only immutable/project-scoped artifacts available at export time are included.",
|
|
61
|
+
"Benchmark, blinded-review and usability evidence must be supplied from completed study artifacts; they are never synthesized by this export.",
|
|
62
|
+
],
|
|
63
|
+
}
|
|
64
|
+
(output_dir / "manifest.json").write_text(json.dumps(manifest, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
|
65
|
+
return manifest
|
package/engine/library.py
CHANGED
|
@@ -8,6 +8,8 @@ changes an existing Project's conclusions — only an explicit import/sync
|
|
|
8
8
|
advances the Project graph.
|
|
9
9
|
"""
|
|
10
10
|
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
11
13
|
import hashlib
|
|
12
14
|
import json
|
|
13
15
|
import os
|
|
@@ -197,8 +199,10 @@ class ResearchLibrary:
|
|
|
197
199
|
outcomes.append({
|
|
198
200
|
"outcome_id": oid,
|
|
199
201
|
"name": f.get("measure", oid),
|
|
200
|
-
|
|
201
|
-
|
|
202
|
+
# No silent default: an unlabelled finding must not be filed as a
|
|
203
|
+
# learning outcome; that is how task-performance evidence used
|
|
204
|
+
# to reach the ADOPT gate on the V2 path.
|
|
205
|
+
"outcome_type": (f.get("extensions") or {}).get("outcome_type", ""),
|
|
202
206
|
"extensions": {
|
|
203
207
|
"auto_created_from_library_import": True,
|
|
204
208
|
"library_revision": lib_rev,
|
|
@@ -42,7 +42,9 @@ from functools import lru_cache
|
|
|
42
42
|
from pathlib import Path
|
|
43
43
|
from typing import Any
|
|
44
44
|
|
|
45
|
-
|
|
45
|
+
from engine._resources import resource_root
|
|
46
|
+
|
|
47
|
+
ROOT = resource_root()
|
|
46
48
|
def _resolve_library_path() -> Path:
|
|
47
49
|
"""Repository layout first; wheel-installed share/ layout as fallback."""
|
|
48
50
|
repo = ROOT / "benchmarks" / "evidence-library.json"
|
package/engine/living.py
CHANGED
|
@@ -43,7 +43,8 @@ from scripts.validate_schema import SchemaError, validate
|
|
|
43
43
|
def _resolve_v4_schema_dir() -> Path:
|
|
44
44
|
"""Repository layout first; wheel-installed share/ layout as fallback
|
|
45
45
|
(same pattern as engine/contracts._resolve_schema_dir)."""
|
|
46
|
-
|
|
46
|
+
from engine._resources import resource_root
|
|
47
|
+
repo = resource_root() / "schemas" / "v4"
|
|
47
48
|
if repo.is_dir():
|
|
48
49
|
return repo
|
|
49
50
|
import sys
|
|
@@ -226,6 +227,26 @@ def set_subscription_status(project: ProjectWorkspace, subscription_id: str,
|
|
|
226
227
|
return subscription
|
|
227
228
|
|
|
228
229
|
|
|
230
|
+
def _validated_outcome_type(value, domain: str) -> str:
|
|
231
|
+
"""Outcome token from a refresh payload, validated against the registry.
|
|
232
|
+
|
|
233
|
+
A missing or unknown token raises: the living path used to default to
|
|
234
|
+
a learning outcome, which silently promoted task-performance evidence.
|
|
235
|
+
"""
|
|
236
|
+
from engine.taxonomy import category_of, tokens as taxonomy_tokens
|
|
237
|
+
|
|
238
|
+
token = str(value or "").strip()
|
|
239
|
+
if not token:
|
|
240
|
+
raise ValueError(
|
|
241
|
+
"living evidence record must declare outcome_type; the engine "
|
|
242
|
+
"will not guess a category")
|
|
243
|
+
if token not in taxonomy_tokens(domain):
|
|
244
|
+
raise ValueError(
|
|
245
|
+
f"outcome_type {token!r} is not registered for domain {domain!r}")
|
|
246
|
+
category_of(domain, token) # fail closed on a malformed taxonomy
|
|
247
|
+
return token
|
|
248
|
+
|
|
249
|
+
|
|
229
250
|
def refresh(project: ProjectWorkspace, subscription_id: str, *,
|
|
230
251
|
new_evidence: list[dict] | None = None,
|
|
231
252
|
retriever: Callable[[dict], list[dict]] | None = None) -> dict:
|
|
@@ -298,8 +319,10 @@ def refresh(project: ProjectWorkspace, subscription_id: str, *,
|
|
|
298
319
|
|
|
299
320
|
# ---- normalize + validate fresh evidence ---------------------------
|
|
300
321
|
next_revision = store.active_revision() + 1
|
|
322
|
+
domain = str((project.manifest() or {}).get("domain") or "education")
|
|
301
323
|
mutation, new_hashes, finding_ids, summary_parts = _build_mutation(
|
|
302
|
-
project, store, subscription, fresh_packets, claims, next_revision
|
|
324
|
+
project, store, subscription, fresh_packets, claims, next_revision,
|
|
325
|
+
domain=domain)
|
|
303
326
|
|
|
304
327
|
revision = store.commit(
|
|
305
328
|
run_id=new_run_id(),
|
|
@@ -369,7 +392,7 @@ def _existing_drift_ids(project: ProjectWorkspace) -> set[str]:
|
|
|
369
392
|
|
|
370
393
|
|
|
371
394
|
def _build_mutation(project, store, subscription, fresh_packets, claims,
|
|
372
|
-
next_revision) -> tuple[GraphMutation, set[str], set[str], list[str]]:
|
|
395
|
+
next_revision, domain: str = "education") -> tuple[GraphMutation, set[str], set[str], list[str]]:
|
|
373
396
|
"""Normalize + validate each fresh record and assemble one GraphMutation."""
|
|
374
397
|
existing = {t: {row[_ID_KEY[t]] for row in store.read_table(t)}
|
|
375
398
|
for t in _GRAPH_TABLES}
|
|
@@ -417,6 +440,8 @@ def _build_mutation(project, store, subscription, fresh_packets, claims,
|
|
|
417
440
|
source_upserted = True
|
|
418
441
|
upserts["sources"].append(src)
|
|
419
442
|
|
|
443
|
+
# Outcome tokens are validated against the project domain registry
|
|
444
|
+
# rather than defaulted; see _validated_outcome_type().
|
|
420
445
|
# --- outcome (optional; reused when the id already exists) -------
|
|
421
446
|
outcome_upserted = False
|
|
422
447
|
outcome = None
|
|
@@ -438,7 +463,7 @@ def _build_mutation(project, store, subscription, fresh_packets, claims,
|
|
|
438
463
|
outcome = {
|
|
439
464
|
"outcome_id": out_id,
|
|
440
465
|
"name": outcome_pkt.get("name") or out_id[len("OUT-"):],
|
|
441
|
-
"outcome_type": outcome_pkt.get("outcome_type",
|
|
466
|
+
"outcome_type": _validated_outcome_type(outcome_pkt.get("outcome_type"), domain),
|
|
442
467
|
"extensions": outcome_pkt.get("extensions") or {},
|
|
443
468
|
}
|
|
444
469
|
existing["outcomes"].add(out_id)
|
|
@@ -502,7 +527,13 @@ def _build_mutation(project, store, subscription, fresh_packets, claims,
|
|
|
502
527
|
f"{label}: relation_to_claim must be one of "
|
|
503
528
|
f"{sorted(_RELATION_TO_IMPLICATION)}, got {relation!r}")
|
|
504
529
|
link.setdefault("decision_implication", _RELATION_TO_IMPLICATION[relation])
|
|
505
|
-
link.
|
|
530
|
+
# directness decides whether this link can carry an ADOPT claim.
|
|
531
|
+
# Defaulting it to 2 (direct) would let an unclassified refresh
|
|
532
|
+
# assert direct evidence it never established, so it is required.
|
|
533
|
+
if "directness" not in link:
|
|
534
|
+
raise ValueError(
|
|
535
|
+
f"{label}: directness is required for a living-evidence link "
|
|
536
|
+
"(2 = direct evidence for the claim; supply it explicitly)")
|
|
506
537
|
link.setdefault("applicability", {"scope_match": "direct"})
|
|
507
538
|
link.setdefault("reasoning_note",
|
|
508
539
|
f"living evidence refresh: {subscription['subscription_id']}")
|
package/engine/meta_synthesis.py
CHANGED
|
@@ -21,7 +21,9 @@ from engine.ids import new_local_id
|
|
|
21
21
|
from engine.library import ResearchLibrary
|
|
22
22
|
from scripts.validate_schema import SchemaError, validate
|
|
23
23
|
|
|
24
|
-
|
|
24
|
+
from engine._resources import resource_root
|
|
25
|
+
|
|
26
|
+
_SYNTHESIS_SCHEMA = (resource_root() / "schemas" / "v3"
|
|
25
27
|
/ "synthesis.schema.json")
|
|
26
28
|
|
|
27
29
|
|
package/engine/migration.py
CHANGED
|
@@ -30,6 +30,15 @@ from engine.versions import (
|
|
|
30
30
|
|
|
31
31
|
OUTCOME_TYPES = ("learning", "task_performance", "process", "risk")
|
|
32
32
|
|
|
33
|
+
#: V1 D5 Directness (0/1/2) -> V2 link directness + applicability scope_match.
|
|
34
|
+
#: Directness decides whether a link can carry an ADOPT claim, so a flat
|
|
35
|
+
#: hard-coded value silently capped every migrated pack at PILOT.
|
|
36
|
+
_D5_TO_DIRECTNESS = {
|
|
37
|
+
2: (2, "direct"),
|
|
38
|
+
1: (1, "partial"),
|
|
39
|
+
0: (1, "mismatch"),
|
|
40
|
+
}
|
|
41
|
+
|
|
33
42
|
_CLAIM_TO_IMPLICATION = {
|
|
34
43
|
"support": "support_adoption",
|
|
35
44
|
"contradict": "oppose_adoption",
|
|
@@ -75,6 +84,71 @@ def _map_decision_implication(ev: dict) -> str:
|
|
|
75
84
|
return _CLAIM_TO_IMPLICATION[_map_relation(ev.get("relation_to_claim") or ev.get("direction"))]
|
|
76
85
|
|
|
77
86
|
|
|
87
|
+
def _v1_effect_estimate(ev: dict) -> dict | None:
|
|
88
|
+
"""Effect magnitude from a V1 record, or None when it recorded none.
|
|
89
|
+
|
|
90
|
+
V1 kept numbers either on a top-level effect_size field or inside
|
|
91
|
+
extensions.raw_result; both are real sources, and absence stays None.
|
|
92
|
+
The V2 contract is narrow (metric + raw_text required, no extra keys), so
|
|
93
|
+
a V1 magnitude is normalized instead of copied verbatim: ci_lower/ci_upper
|
|
94
|
+
become ci_low/ci_high and keys the V2 contract does not declare are kept in
|
|
95
|
+
the raw_text so nothing recorded is silently lost.
|
|
96
|
+
"""
|
|
97
|
+
value = ev.get("effect_size")
|
|
98
|
+
if isinstance(value, dict) and value.get("value") is not None:
|
|
99
|
+
return _normalize_effect_estimate(value)
|
|
100
|
+
if isinstance(value, (int, float)):
|
|
101
|
+
return {"value": float(value), "source": "v1_effect_size"}
|
|
102
|
+
raw = (ev.get("extensions") or {}).get("raw_result")
|
|
103
|
+
if isinstance(raw, dict) and raw.get("value") is not None:
|
|
104
|
+
return _normalize_effect_estimate(raw)
|
|
105
|
+
return None
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _normalize_effect_estimate(raw: dict) -> dict:
|
|
109
|
+
"""Project a V1 magnitude onto the V2 effect_estimate contract."""
|
|
110
|
+
known = ("metric", "value", "unit", "ci_low", "ci_high", "p_value")
|
|
111
|
+
out: dict = {}
|
|
112
|
+
for key in known:
|
|
113
|
+
if key in raw and raw[key] is not None:
|
|
114
|
+
out[key] = raw[key]
|
|
115
|
+
for legacy, canonical in (("ci_lower", "ci_low"), ("ci_upper", "ci_high")):
|
|
116
|
+
if canonical not in out and raw.get(legacy) is not None:
|
|
117
|
+
out[canonical] = raw[legacy]
|
|
118
|
+
out.setdefault("metric", str(raw.get("source") or "v1_effect_size"))
|
|
119
|
+
extra = {k: v for k, v in raw.items()
|
|
120
|
+
if k not in known and k not in ("ci_lower", "ci_upper")
|
|
121
|
+
and v is not None}
|
|
122
|
+
text = raw.get("raw_text")
|
|
123
|
+
if not text:
|
|
124
|
+
text = json.dumps(extra, ensure_ascii=False, sort_keys=True) if extra else ""
|
|
125
|
+
out["raw_text"] = str(text)
|
|
126
|
+
return out
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _v1_directness(ev: dict) -> tuple[int, str, bool]:
|
|
130
|
+
"""Carry V1 D5 Directness across as (link directness, scope_match, recorded).
|
|
131
|
+
|
|
132
|
+
D5 is a 0/1/2 axis in the V1 evidence contract and the V2 ADOPT gate reads
|
|
133
|
+
the link's directness, so the mapping has to be explicit:
|
|
134
|
+
|
|
135
|
+
2 -> directness 2 / scope_match "direct"
|
|
136
|
+
1 -> directness 1 / scope_match "partial"
|
|
137
|
+
0 -> directness 1 / scope_match "mismatch" (0 can never gate an ADOPT)
|
|
138
|
+
|
|
139
|
+
A missing or non-integer D5 is not silently promoted: it migrates as
|
|
140
|
+
directness 1 / scope_match "partial" and the caller records a downgrade.
|
|
141
|
+
"""
|
|
142
|
+
dims = ev.get("quality_dimensions")
|
|
143
|
+
raw = dims.get("D5_directness") if isinstance(dims, dict) else None
|
|
144
|
+
if isinstance(raw, bool) or not isinstance(raw, int):
|
|
145
|
+
return 1, "partial", False
|
|
146
|
+
mapped = _D5_TO_DIRECTNESS.get(raw)
|
|
147
|
+
if mapped is None:
|
|
148
|
+
return 1, "partial", False
|
|
149
|
+
return mapped[0], mapped[1], True
|
|
150
|
+
|
|
151
|
+
|
|
78
152
|
def migrate_v1_pack(pack_dir: Path, *, home: Path,
|
|
79
153
|
title: str | None = None) -> MigrationResult:
|
|
80
154
|
"""Import a V1 pack directory into a new V2 Project graph.
|
|
@@ -273,7 +347,10 @@ def migrate_v1_pack(pack_dir: Path, *, home: Path,
|
|
|
273
347
|
"measure": ev.get("outcome_type") or "outcome",
|
|
274
348
|
"timepoint": None,
|
|
275
349
|
"effect_direction": _map_effect_direction(ev.get("effect_direction")),
|
|
276
|
-
|
|
350
|
+
# Carry the magnitude across the hop instead of dropping it:
|
|
351
|
+
# a migrated pack used to report 100% not_extractable, which
|
|
352
|
+
# reads as "no evidence" rather than "not migrated".
|
|
353
|
+
"effect_estimate": _v1_effect_estimate(ev),
|
|
277
354
|
"raw_result_text": ev.get("claim") or "unavailable",
|
|
278
355
|
"source_locator": ev.get("source_location") or "unavailable",
|
|
279
356
|
"extensions": {"v1_legacy": True},
|
|
@@ -281,6 +358,14 @@ def migrate_v1_pack(pack_dir: Path, *, home: Path,
|
|
|
281
358
|
claim_id = claim_by_evidence.get(ev["evidence_id"], f"CLM-{ev['evidence_id']}")
|
|
282
359
|
link_id = f"LNK-{ev['evidence_id']}"
|
|
283
360
|
|
|
361
|
+
directness, scope_match, d5_recorded = _v1_directness(ev)
|
|
362
|
+
if not d5_recorded:
|
|
363
|
+
report["downgrades"].append({
|
|
364
|
+
"evidence_id": ev["evidence_id"],
|
|
365
|
+
"from": "V1 quality_dimensions.D5_directness (absent or unreadable)",
|
|
366
|
+
"to": f"directness={directness}, scope_match={scope_match!r}",
|
|
367
|
+
})
|
|
368
|
+
|
|
284
369
|
links.append({
|
|
285
370
|
|
|
286
371
|
"evidence_link_id": link_id,
|
|
@@ -289,8 +374,8 @@ def migrate_v1_pack(pack_dir: Path, *, home: Path,
|
|
|
289
374
|
"relation_to_claim": _map_relation(
|
|
290
375
|
ev.get("relation_to_claim") or ev.get("direction")),
|
|
291
376
|
"decision_implication": _map_decision_implication(ev),
|
|
292
|
-
"directness":
|
|
293
|
-
"applicability": {"scope_match":
|
|
377
|
+
"directness": directness,
|
|
378
|
+
"applicability": {"scope_match": scope_match},
|
|
294
379
|
"reasoning_note": "migrated from V1 Evidence Object",
|
|
295
380
|
"created_in_revision": 1,
|
|
296
381
|
"extensions": {"v1_legacy": True},
|