eduevidence 5.2.0 → 6.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +105 -0
- package/README.md +142 -75
- package/README.zh-CN.md +73 -30
- package/SKILL.md +397 -131
- package/agents/openai.yaml +4 -0
- package/assets/readme/controlled-execution.svg +34 -0
- package/assets/readme/landing-tour.gif +0 -0
- package/assets/readme/logo.png +0 -0
- package/assets/readme/research-workflow.svg +56 -0
- package/assets/readme/studio-graph.png +0 -0
- package/assets/readme/studio-overview.png +0 -0
- package/assets/readme/studio-reports.png +0 -0
- package/assets/readme/studio-tour.gif +0 -0
- package/autoevolve/config.yaml +17 -0
- package/autoevolve/program.md +25 -0
- package/autoevolve/protected.manifest.yaml +34 -0
- package/benchmarks/adversarial/cases.jsonl +7 -0
- package/benchmarks/evidence-library.json +5268 -0
- package/benchmarks/partitions.json +8 -0
- package/bin/eduevidence.js +2 -1
- package/docs/architecture.md +496 -0
- package/docs/autoresearch-evolution-plan.md +2903 -0
- package/docs/autoresearch-implementation-status.md +101 -0
- package/docs/demo-storyboard.md +20 -0
- package/docs/demo-workplace-ai.md +92 -0
- package/docs/demo.md +32 -0
- package/docs/install-guide.md +150 -0
- package/docs/orchestration-role-model.md +1254 -0
- package/docs/release-closeout/README.md +17 -0
- package/docs/release-closeout/frontend-acceptance.md +23 -0
- package/docs/release-closeout/issues.md +19 -0
- package/docs/release-closeout/verification.md +28 -0
- package/docs/release-contract.md +108 -0
- package/docs/research-studio-guide.zh-CN.md +166 -0
- package/docs/sciverse-api.md +125 -0
- package/eduevidence_cli.py +29 -13
- package/engine/_resources.py +13 -0
- package/engine/autoevolve/__init__.py +3 -0
- package/engine/autoevolve/agent_view.py +167 -0
- package/engine/autoevolve/core.py +357 -0
- package/engine/autoevolve/events.py +11 -0
- package/engine/autoevolve/git_workspace.py +77 -0
- package/engine/autoevolve/projection.py +23 -0
- package/engine/autoevolve/runner.py +413 -0
- package/engine/autoevolve/trust.py +146 -0
- package/engine/autoresearch/__init__.py +6 -0
- package/engine/autoresearch/commit.py +132 -0
- package/engine/autoresearch/contracts.py +126 -0
- package/engine/autoresearch/controller.py +207 -0
- package/engine/autoresearch/events.py +12 -0
- package/engine/autoresearch/gap_priority.py +168 -0
- package/engine/autoresearch/projection.py +30 -0
- package/engine/autoresearch/research_memory.py +59 -0
- package/engine/autoresearch/saturation.py +91 -0
- package/engine/briefs.py +2 -1
- package/engine/capabilities.py +1 -0
- package/engine/contracts.py +3 -1
- package/engine/decision_policy.py +96 -0
- package/engine/evidence_graph.py +14 -10
- package/engine/evidencecore.py +7 -5
- package/engine/gaps.py +132 -73
- package/engine/ids.py +2 -0
- package/engine/judge_pack.py +65 -0
- package/engine/library.py +6 -2
- package/engine/library_builtin.py +3 -1
- package/engine/living.py +36 -5
- package/engine/meta_synthesis.py +3 -1
- package/engine/migration.py +88 -3
- package/engine/orchestration.py +460 -0
- package/engine/paths.py +2 -0
- package/engine/pilot.py +36 -33
- package/engine/project.py +2 -2
- package/engine/research_service.py +113 -0
- package/engine/studio_read_model.py +400 -0
- package/engine/taxonomy.py +211 -0
- package/engine/tribunal.py +44 -33
- package/engine/update.py +1 -0
- package/engine/versions.py +1 -1
- package/engine/worker_result.py +109 -0
- package/engine/workflows.py +70 -0
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +2934 -0
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
- package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
- package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
- package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
- package/examples/ai-coding-assistant-evidence/frame.json +48 -0
- package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
- package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
- package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
- package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +230 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/result.json +1457 -0
- package/examples/ai-coding-assistant-evidence/result.zh.json +1457 -0
- package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
- package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +107 -0
- package/examples/spaced-retrieval-practice/applicability.json +14 -0
- package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
- package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
- package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
- package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
- package/examples/spaced-retrieval-practice/frame.json +58 -0
- package/examples/spaced-retrieval-practice/gate_report.json +101 -0
- package/examples/spaced-retrieval-practice/methodology.json +78 -0
- package/examples/spaced-retrieval-practice/report_spec.json +212 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/result.json +942 -0
- package/examples/spaced-retrieval-practice/result.zh.json +942 -0
- package/examples/spaced-retrieval-practice/skeptic.json +70 -0
- package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
- package/examples/spaced-retrieval-practice/verdict.json +93 -0
- package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -0
- package/examples/workplace-ai-assistant/evaluation.json +19 -0
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
- package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
- package/examples/workplace-ai-assistant/final_verdict.json +78 -0
- package/examples/workplace-ai-assistant/frame.json +41 -0
- package/examples/workplace-ai-assistant/gate_report.json +101 -0
- package/examples/workplace-ai-assistant/intervention.json +27 -0
- package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
- package/examples/workplace-ai-assistant/methodology.json +60 -0
- package/examples/workplace-ai-assistant/report_spec.json +224 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/result.json +615 -0
- package/examples/workplace-ai-assistant/result.zh.json +615 -0
- package/examples/workplace-ai-assistant/search_log.json +19 -0
- package/examples/workplace-ai-assistant/skeptic.json +72 -0
- package/examples/workplace-ai-assistant/sources.jsonl +3 -0
- package/examples/workplace-ai-assistant/validation_result.json +9 -0
- package/examples/workplace-ai-assistant/verdict.json +78 -0
- package/install.sh +7 -7
- package/integrations/agent_mcp.py +2 -2
- package/integrations/orchestration_dispatch.py +146 -0
- package/package.json +46 -3
- package/pyproject.toml +14 -22
- package/references/autoresearch.md +30 -0
- package/references/evaluation-policy.md +24 -0
- package/references/orchestration.md +22 -0
- package/references/report-copy-style.md +67 -0
- package/references/retrieval-compliance.md +75 -0
- package/references/retrieval-protocol.md +20 -0
- package/references/scientific-invariants.md +19 -0
- package/retrieval/audit.py +178 -0
- package/retrieval/fetch.py +96 -0
- package/retrieval/sciverse.py +398 -0
- package/retrieval/search.py +47 -7
- package/schemas/applicability.schema.json +94 -0
- package/schemas/chart-spec.schema.json +10 -3
- package/schemas/evidence.schema.json +316 -43
- package/schemas/fetch-result.schema.json +2 -1
- package/schemas/intervention.schema.json +106 -21
- package/schemas/report-result.schema.json +12 -4
- package/schemas/report-spec.schema.json +98 -100
- package/schemas/skeptic.schema.json +86 -0
- package/schemas/source.schema.json +21 -2
- package/schemas/v2/finding.schema.json +5 -1
- package/schemas/v2/methodology-audit.schema.json +5 -1
- package/schemas/v2/outcome.schema.json +28 -5
- package/schemas/v2/project.schema.json +2 -2
- package/schemas/v2/run.schema.json +1 -1
- package/schemas/v2/study.schema.json +5 -1
- package/schemas/vNext/autoevolve-session.schema.json +34 -0
- package/schemas/vNext/eval-snapshot.schema.json +77 -0
- package/schemas/vNext/execution-plan.schema.json +50 -0
- package/schemas/vNext/gap-priority.schema.json +54 -0
- package/schemas/vNext/negative-search-record.schema.json +68 -0
- package/schemas/vNext/research-iteration.schema.json +87 -0
- package/schemas/vNext/research-strategy.schema.json +62 -0
- package/schemas/vNext/skill-experiment.schema.json +90 -0
- package/schemas/vNext/task-spec.schema.json +156 -0
- package/schemas/vNext/worker-result.schema.json +60 -0
- package/schemas/verdict.schema.json +164 -28
- package/scripts/benchmark_judge.py +2 -2
- package/scripts/benchmark_v3.py +26 -43
- package/scripts/build_esl_artifacts.py +4 -4
- package/scripts/build_evidence_library.py +2 -2
- package/scripts/build_gh_pages.py +98 -0
- package/scripts/build_readme_diagrams.py +72 -0
- package/scripts/build_report_variants.py +101 -0
- package/scripts/build_result.py +74 -9
- package/scripts/check_autoresearch_invariants.py +95 -0
- package/scripts/check_package_parity.py +85 -0
- package/scripts/check_protocol_alignment.py +375 -0
- package/scripts/check_versioned_schemas.py +254 -0
- package/scripts/claim_audit.py +13 -8
- package/scripts/compute_confidence.py +10 -0
- package/scripts/daily_evolve.py +30 -0
- package/scripts/dashboard_server.py +130 -101
- package/scripts/did_regression.py +17 -32
- package/scripts/enrich_projects_human_and_lieflat.py +1 -1
- package/scripts/evidence_score.py +5 -2
- package/scripts/generate_metrics.py +4 -3
- package/scripts/generate_new_projects.py +5 -5
- package/scripts/orchestrator.py +286 -36
- package/scripts/pre_verdict_gate.py +224 -26
- package/scripts/quickstart.py +18 -2
- package/scripts/rebake_all_5themes.py +1 -2
- package/scripts/research_auto_cli.py +475 -0
- package/scripts/run_workspace.py +24 -8
- package/scripts/search_provenance.py +64 -0
- package/scripts/serve_web.py +9 -10
- package/scripts/skill_lint.py +1 -1
- package/scripts/skill_payload.py +81 -0
- package/scripts/test_adversarial_empirical.py +26 -19
- package/scripts/validate_schema.py +46 -2
- package/scripts/vnext_cli.py +133 -0
- package/setup.py +12 -0
- package/skill/agents/evaluation-designer.md +20 -4
- package/skill/agents/evidence-analyst.md +19 -3
- package/skill/agents/evidence-judge.md +50 -2
- package/skill/agents/evidence-retriever.md +20 -3
- package/skill/agents/intervention-designer.md +20 -4
- package/skill/agents/method-reviewer.md +18 -2
- package/skill/agents/{education-planner.md → research-planner.md} +19 -3
- package/skill/agents/skeptic.md +18 -2
- package/skill/roles/registry.yaml +45 -0
- package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
- package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
- package/skill/sub-skills/data-analysis/SKILL.md +34 -15
- package/skill/sub-skills/ethics-review/SKILL.md +33 -10
- package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
- package/skill/sub-skills/evidence-review/SKILL.md +31 -12
- package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
- package/skill/sub-skills/literature-review/SKILL.md +35 -14
- package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
- package/skill/sub-skills/report-generation/SKILL.md +40 -6
- package/skill/sub-skills/research-planning/SKILL.md +41 -14
- package/skill/sub-skills/study-design/SKILL.md +30 -9
- package/skill/task-briefs/adjudicate.md +32 -7
- package/skill/task-briefs/applicability.md +38 -0
- package/skill/task-briefs/audit.md +32 -7
- package/skill/task-briefs/challenge.md +34 -5
- package/skill/task-briefs/evaluate.md +30 -5
- package/skill/task-briefs/extract.md +31 -8
- package/skill/task-briefs/frame.md +39 -10
- package/skill/task-briefs/intervene.md +32 -6
- package/skill/task-briefs/present.md +32 -8
- package/skill/task-briefs/projection.md +37 -0
- package/skill/task-briefs/retrieve.md +36 -6
- package/skill/workflows/decision-and-pilot.md +85 -0
- package/skill/workflows/evaluate-and-update.md +93 -0
- package/skill/workflows/evidence-review.md +117 -0
- package/visualization/eduevidence-report/assets/base.css +2 -2
- package/visualization/eduevidence-report/assets/reader.css +752 -0
- package/visualization/eduevidence-report/assets/reader.js +132 -0
- package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
- package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
- package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
- package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
- package/visualization/eduevidence-report/scripts/build_report.py +561 -121
- package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +371 -136
- package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
- package/visualization/eduevidence-report/themes/academic.css +1 -1
- package/visualization/eduevidence-report/themes/claude.css +1 -1
- package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
- package/visualization/eduevidence-report/themes/datalab.css +2 -2
- package/visualization/eduevidence-report/themes/presentation.css +2 -2
- package/web/README.md +18 -0
- package/web/architecture.html +14885 -0
- package/web/index.html +53 -0
- package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
- package/web/studio/assets/index-B8tkF44Q.css +1 -0
- package/web/studio/assets/index-CQ6Keoyc.js +230 -0
- package/web/studio/config.json +1 -0
- package/web/studio/index.html +14 -0
- package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
- package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/bias.cpython-312.pyc +0 -0
- package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
- package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
- package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
- package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
- package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
- package/engine/__pycache__/events.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
- package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
- package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
- package/engine/__pycache__/ids.cpython-312.pyc +0 -0
- package/engine/__pycache__/library.cpython-312.pyc +0 -0
- package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
- package/engine/__pycache__/living.cpython-312.pyc +0 -0
- package/engine/__pycache__/log.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/migration.cpython-312.pyc +0 -0
- package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
- package/engine/__pycache__/paths.cpython-312.pyc +0 -0
- package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
- package/engine/__pycache__/planner.cpython-312.pyc +0 -0
- package/engine/__pycache__/project.cpython-312.pyc +0 -0
- package/engine/__pycache__/projections.cpython-312.pyc +0 -0
- package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
- package/engine/__pycache__/run.cpython-312.pyc +0 -0
- package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
- package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
- package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
- package/engine/__pycache__/update.cpython-312.pyc +0 -0
- package/engine/__pycache__/versions.cpython-312.pyc +0 -0
- package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
- package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
- package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
- package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
- package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
- package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
- package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
- package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
- package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
- package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
- package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
- package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
- package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
- package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
- package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
- package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
import json
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Any
|
|
5
|
+
from .contracts import NegativeSearchRecord, ResearchIteration
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class ResearchMemory:
|
|
9
|
+
def __init__(self, root: str | Path):
|
|
10
|
+
self.root = Path(root)
|
|
11
|
+
self.root.mkdir(parents=True, exist_ok=True)
|
|
12
|
+
self.iterations_path = self.root / "research-iterations.jsonl"
|
|
13
|
+
self.negative_path = self.root / "negative-searches.jsonl"
|
|
14
|
+
|
|
15
|
+
@staticmethod
|
|
16
|
+
def _append(path: Path, record: dict[str, Any]) -> None:
|
|
17
|
+
with path.open("a", encoding="utf-8") as f:
|
|
18
|
+
f.write(json.dumps(record, ensure_ascii=False, sort_keys=True) + "\n")
|
|
19
|
+
|
|
20
|
+
def append_iteration(self, iteration: ResearchIteration) -> None:
|
|
21
|
+
iteration.validate()
|
|
22
|
+
self._append(self.iterations_path, iteration.as_dict())
|
|
23
|
+
|
|
24
|
+
def append_negative_search(self, record: NegativeSearchRecord) -> None:
|
|
25
|
+
record.validate()
|
|
26
|
+
self._append(self.negative_path, record.__dict__)
|
|
27
|
+
|
|
28
|
+
def load_iterations(
|
|
29
|
+
self,
|
|
30
|
+
gap_id: str | None = None,
|
|
31
|
+
*,
|
|
32
|
+
gap_lineage_key: str | None = None,
|
|
33
|
+
) -> list[dict[str, Any]]:
|
|
34
|
+
"""Load iteration history, preferring stable lineage across revisions.
|
|
35
|
+
|
|
36
|
+
Legacy rows without `gap_lineage_key` remain queryable by `gap_id`.
|
|
37
|
+
When a lineage key is supplied, new keyed rows match by lineage and old
|
|
38
|
+
unkeyed rows may additionally match the supplied gap_id for migration.
|
|
39
|
+
"""
|
|
40
|
+
if not self.iterations_path.exists():
|
|
41
|
+
return []
|
|
42
|
+
rows = [
|
|
43
|
+
json.loads(line)
|
|
44
|
+
for line in self.iterations_path.read_text(encoding="utf-8").splitlines()
|
|
45
|
+
if line.strip()
|
|
46
|
+
]
|
|
47
|
+
if gap_lineage_key is not None:
|
|
48
|
+
return [
|
|
49
|
+
row for row in rows
|
|
50
|
+
if row.get("gap_lineage_key") == gap_lineage_key
|
|
51
|
+
or (
|
|
52
|
+
not row.get("gap_lineage_key")
|
|
53
|
+
and gap_id is not None
|
|
54
|
+
and row.get("gap_id") == gap_id
|
|
55
|
+
)
|
|
56
|
+
]
|
|
57
|
+
if gap_id is not None:
|
|
58
|
+
return [row for row in rows if row.get("gap_id") == gap_id]
|
|
59
|
+
return rows
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from dataclasses import dataclass
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
@dataclass(frozen=True)
|
|
7
|
+
class SaturationResult:
|
|
8
|
+
saturated: bool
|
|
9
|
+
low_yield_streak: int
|
|
10
|
+
strategy_diversity_exhausted: bool
|
|
11
|
+
rationale: tuple[str, ...]
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _is_low_yield(row: dict[str, Any]) -> bool:
|
|
15
|
+
gain = row.get("evidence_gain") or {}
|
|
16
|
+
unique = int(gain.get("unique_eligible_evidence", 0) or 0)
|
|
17
|
+
direct = int(gain.get("direct_outcome_findings", 0) or 0)
|
|
18
|
+
delta = float(gain.get("decision_boundary_delta", 0) or 0)
|
|
19
|
+
duplicate_rate = float(gain.get("duplicate_rate", 0) or 0)
|
|
20
|
+
candidate_sources = row.get("candidate_sources") or []
|
|
21
|
+
no_candidates = len(candidate_sources) == 0
|
|
22
|
+
return (
|
|
23
|
+
unique == 0
|
|
24
|
+
and direct == 0
|
|
25
|
+
and abs(delta) < 1e-12
|
|
26
|
+
and (duplicate_rate >= 0.5 or no_candidates)
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def detect_saturation(
|
|
31
|
+
iterations: list[dict[str, Any]],
|
|
32
|
+
*,
|
|
33
|
+
min_consecutive: int = 2,
|
|
34
|
+
available_strategy_types: set[str] | None = None,
|
|
35
|
+
) -> SaturationResult:
|
|
36
|
+
"""Detect bounded secondary-search saturation.
|
|
37
|
+
|
|
38
|
+
Strategy diversity is computed across the full history for the gap, while
|
|
39
|
+
the low-yield condition is intentionally a trailing streak. Empty searches
|
|
40
|
+
count as low-yield even when duplicate_rate is zero; otherwise a provider
|
|
41
|
+
returning no candidates could keep the loop alive forever.
|
|
42
|
+
"""
|
|
43
|
+
attempted_all = {
|
|
44
|
+
str((row.get("strategy") or {}).get("experiment_type", ""))
|
|
45
|
+
for row in iterations
|
|
46
|
+
if str((row.get("strategy") or {}).get("experiment_type", ""))
|
|
47
|
+
}
|
|
48
|
+
streak = 0
|
|
49
|
+
for row in reversed(iterations):
|
|
50
|
+
if _is_low_yield(row):
|
|
51
|
+
streak += 1
|
|
52
|
+
else:
|
|
53
|
+
break
|
|
54
|
+
|
|
55
|
+
if available_strategy_types:
|
|
56
|
+
diversity_exhausted = available_strategy_types.issubset(attempted_all)
|
|
57
|
+
else:
|
|
58
|
+
diversity_exhausted = len(attempted_all) >= 2
|
|
59
|
+
|
|
60
|
+
rationale: list[str] = []
|
|
61
|
+
if streak >= min_consecutive:
|
|
62
|
+
rationale.append(
|
|
63
|
+
f"{streak} consecutive iterations produced no unique/direct evidence or decision-boundary change"
|
|
64
|
+
)
|
|
65
|
+
if diversity_exhausted:
|
|
66
|
+
rationale.append("strategy diversity exhausted for the configured search space")
|
|
67
|
+
return SaturationResult(
|
|
68
|
+
streak >= min_consecutive and diversity_exhausted,
|
|
69
|
+
streak,
|
|
70
|
+
diversity_exhausted,
|
|
71
|
+
tuple(rationale),
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def transition_to_empirical(
|
|
76
|
+
*,
|
|
77
|
+
dvi_band: str,
|
|
78
|
+
decision_material: bool,
|
|
79
|
+
unresolved: bool,
|
|
80
|
+
saturation: SaturationResult,
|
|
81
|
+
ethics_feasible: bool,
|
|
82
|
+
) -> tuple[bool, tuple[str, ...]]:
|
|
83
|
+
checks = [
|
|
84
|
+
(dvi_band.upper() == "HIGH", "gap DVI is HIGH"),
|
|
85
|
+
(decision_material, "gap is material to the decision"),
|
|
86
|
+
(unresolved, "gap remains unresolved"),
|
|
87
|
+
(saturation.saturated, "secondary search is saturated"),
|
|
88
|
+
(ethics_feasible, "empirical study is ethically/operationally feasible"),
|
|
89
|
+
]
|
|
90
|
+
reasons = tuple(text for ok, text in checks if ok)
|
|
91
|
+
return all(ok for ok, _ in checks), reasons
|
package/engine/briefs.py
CHANGED
|
@@ -14,6 +14,7 @@ from pathlib import Path
|
|
|
14
14
|
|
|
15
15
|
from engine.contracts import load_schema, schema_path
|
|
16
16
|
from engine.planner import PlanStep
|
|
17
|
+
from engine.project import ProjectWorkspace
|
|
17
18
|
|
|
18
19
|
|
|
19
20
|
def _schema_section(schema: dict) -> str:
|
|
@@ -73,7 +74,7 @@ def build_task_brief(step: PlanStep, *, project: ProjectWorkspace,
|
|
|
73
74
|
f"Validate the output with `engine.contracts.validate_record({schema_name!r}, record)` — "
|
|
74
75
|
f"it must return [] (empty errors).",
|
|
75
76
|
"",
|
|
76
|
-
|
|
77
|
+
"## Output path",
|
|
77
78
|
str(output_path),
|
|
78
79
|
"",
|
|
79
80
|
"## Inputs",
|
package/engine/capabilities.py
CHANGED
package/engine/contracts.py
CHANGED
|
@@ -12,7 +12,9 @@ from typing import Callable
|
|
|
12
12
|
|
|
13
13
|
from scripts.validate_schema import SchemaError, validate
|
|
14
14
|
|
|
15
|
-
|
|
15
|
+
from engine._resources import resource_root
|
|
16
|
+
|
|
17
|
+
_REPO_SCHEMA_DIR = resource_root() / "schemas" / "v2"
|
|
16
18
|
|
|
17
19
|
|
|
18
20
|
def _resolve_schema_dir() -> Path:
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""Single source of truth for the ADOPT direct-evidence gate.
|
|
2
|
+
|
|
3
|
+
The four-state decision (ADOPT / PILOT / REJECT / INSUFFICIENT EVIDENCE) is
|
|
4
|
+
computed by two layers that must never drift apart:
|
|
5
|
+
|
|
6
|
+
* engine/tribunal.py - V2 Evidence Graph adjudication
|
|
7
|
+
* scripts/pre_verdict_gate.py - V1 run/example-pack gate enforcement
|
|
8
|
+
|
|
9
|
+
Before this module existed the rule lived only inside the tribunal, so the V1
|
|
10
|
+
gate could not enforce it and a hand-written verdict could carry any action it
|
|
11
|
+
liked. Both layers now resolve the 'which outcome categories count for this
|
|
12
|
+
domain' question here, and the V1 gate additionally re-derives direct-evidence
|
|
13
|
+
presence from the pack's own evidence records.
|
|
14
|
+
|
|
15
|
+
Stdlib only, consistent with the "Native Core" policy of engine/.
|
|
16
|
+
"""
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
#: Category buckets that count as a domain's PRIMARY effect for the ADOPT gate.
|
|
20
|
+
#: The gate asks 'is there direct evidence on the outcome this decision is
|
|
21
|
+
#: actually about?' - for education that is a learning outcome (task
|
|
22
|
+
#: performance and process measures never qualify); for policy it is the
|
|
23
|
+
#: policy-effectiveness / cost class. Every entry must name a category the
|
|
24
|
+
#: domain registry declares, which check_protocol_alignment.py enforces.
|
|
25
|
+
PRIMARY_EFFECT_CATEGORIES: dict[str, tuple[str, ...]] = {
|
|
26
|
+
"education": ("learning",),
|
|
27
|
+
"policy": ("effectiveness", "cost"),
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
#: Directness (0-2) at which an evidence link may carry an ADOPT claim.
|
|
31
|
+
ADOPT_DIRECTNESS = 2
|
|
32
|
+
|
|
33
|
+
#: Confidence band required before ADOPT is possible at all.
|
|
34
|
+
ADOPT_REQUIRED_LABEL = "High"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def primary_effect_categories(domain: str) -> tuple[str, ...]:
|
|
38
|
+
"""Categories that satisfy the ADOPT direct-evidence gate for a domain."""
|
|
39
|
+
from engine.taxonomy import categories as taxonomy_categories
|
|
40
|
+
|
|
41
|
+
declared = PRIMARY_EFFECT_CATEGORIES.get(domain)
|
|
42
|
+
if declared:
|
|
43
|
+
known = taxonomy_categories(domain)
|
|
44
|
+
missing = [c for c in declared if c not in known]
|
|
45
|
+
if missing:
|
|
46
|
+
raise ValueError(
|
|
47
|
+
'domain ' + repr(domain) + ' ADOPT gate references undeclared '
|
|
48
|
+
'categories ' + repr(missing) + '; declared: ' + repr(sorted(known)))
|
|
49
|
+
return declared
|
|
50
|
+
known = taxonomy_categories(domain)
|
|
51
|
+
if not known:
|
|
52
|
+
raise ValueError('domain ' + repr(domain) + ' declares no outcome categories')
|
|
53
|
+
return (next(iter(known)),)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def outcome_category(domain: str, value: str, primary: tuple[str, ...]) -> str | None:
|
|
57
|
+
"""Resolve an outcome value to its category, or None when unresolvable.
|
|
58
|
+
|
|
59
|
+
The outcomes table stores CATEGORY buckets (learning / task_performance /
|
|
60
|
+
process / risk, plus each domain's own buckets), while V1 packs store raw
|
|
61
|
+
taxonomy tokens. Accept a category directly and, for a token, resolve it
|
|
62
|
+
through the registry. An unknown value returns None so callers fail
|
|
63
|
+
closed instead of silently treating it as decision-grade evidence.
|
|
64
|
+
"""
|
|
65
|
+
from engine.taxonomy import TaxonomyError, category_of
|
|
66
|
+
|
|
67
|
+
if value in primary:
|
|
68
|
+
return value
|
|
69
|
+
try:
|
|
70
|
+
return category_of(domain, value)
|
|
71
|
+
except (TaxonomyError, ValueError):
|
|
72
|
+
return None
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def decision_action(*, confidence_label: str, decisive_relations: dict[str, str],
|
|
76
|
+
has_direct_primary_evidence: bool) -> str:
|
|
77
|
+
"""Gate-enforced decision action (uppercase four-state).
|
|
78
|
+
|
|
79
|
+
REJECT requires usable direct opposition evidence (an independent Study
|
|
80
|
+
folded to oppose_adoption). Low/Insufficient can never yield ADOPT.
|
|
81
|
+
ADOPT additionally requires direct evidence on the domain's PRIMARY
|
|
82
|
+
outcome category: High + decisive support WITHOUT such evidence downgrades
|
|
83
|
+
to PILOT - task performance and procedural efficiency are not
|
|
84
|
+
decision-grade effects. Moderate + decisive support -> PILOT; otherwise
|
|
85
|
+
INSUFFICIENT_EVIDENCE.
|
|
86
|
+
"""
|
|
87
|
+
has_oppose = any(r == "oppose_adoption" for r in decisive_relations.values())
|
|
88
|
+
has_support = any(r == "support_adoption" for r in decisive_relations.values())
|
|
89
|
+
if has_oppose:
|
|
90
|
+
return "REJECT"
|
|
91
|
+
if (confidence_label == ADOPT_REQUIRED_LABEL and has_support
|
|
92
|
+
and has_direct_primary_evidence):
|
|
93
|
+
return "ADOPT"
|
|
94
|
+
if confidence_label in ("High", "Moderate") and has_support:
|
|
95
|
+
return "PILOT"
|
|
96
|
+
return "INSUFFICIENT_EVIDENCE"
|
package/engine/evidence_graph.py
CHANGED
|
@@ -54,7 +54,11 @@ class EvidenceNode:
|
|
|
54
54
|
outcome_dimension: str = OutcomeDimension.GENERAL_MEASURE
|
|
55
55
|
claim_id: Optional[str] = None
|
|
56
56
|
outcome_id: Optional[str] = None
|
|
57
|
-
|
|
57
|
+
# A missing effect must stay missing: the old default fabricated a
|
|
58
|
+
# g = 0.0 with a p = 0.05, which would serialise as a real (and
|
|
59
|
+
# false) "no effect" result. Callers that need a number must supply
|
|
60
|
+
# one; the extractors already suppress charts when it is absent.
|
|
61
|
+
effect_size: Optional[Dict[str, Any]] = None
|
|
58
62
|
sample_size: int = 0
|
|
59
63
|
sample_description: str = ""
|
|
60
64
|
study_design: str = "Quasi-Experimental" # RCT, Quasi-Experimental DID, Meta-Analysis, Observational
|
|
@@ -292,9 +296,9 @@ class EvidenceGraph:
|
|
|
292
296
|
for ev in self.evidence.values():
|
|
293
297
|
paper = self.papers.get(ev.paper_id)
|
|
294
298
|
study_label = f"{paper.authors[0] if paper and paper.authors else ev.paper_id} ({paper.year if paper else ''})"
|
|
295
|
-
effect_val = ev.effect_size.get("value"
|
|
296
|
-
ci_l = ev.effect_size.get("ci_lower")
|
|
297
|
-
ci_u = ev.effect_size.get("ci_upper")
|
|
299
|
+
effect_val = (ev.effect_size or {}).get("value") or 0.0
|
|
300
|
+
ci_l = (ev.effect_size or {}).get("ci_lower")
|
|
301
|
+
ci_u = (ev.effect_size or {}).get("ci_upper")
|
|
298
302
|
has_ci = ci_l is not None and ci_u is not None and float(ci_u) >= float(ci_l)
|
|
299
303
|
points.append({
|
|
300
304
|
"evidence_id": ev.evidence_id,
|
|
@@ -327,13 +331,13 @@ class EvidenceGraph:
|
|
|
327
331
|
precision_counts = {"reported_ci": 0, "derived_from_sample_size": 0}
|
|
328
332
|
excluded_no_precision = 0
|
|
329
333
|
for n in nodes:
|
|
330
|
-
eff = n.effect_size.get("value")
|
|
334
|
+
eff = (n.effect_size or {}).get("value")
|
|
331
335
|
if eff is None or math.isnan(float(eff)) or math.isinf(float(eff)):
|
|
332
336
|
continue
|
|
333
337
|
|
|
334
338
|
# Statistical variance derivation (Borenstein et al. 2009)
|
|
335
|
-
ci_l = n.effect_size.get("ci_lower")
|
|
336
|
-
ci_u = n.effect_size.get("ci_upper")
|
|
339
|
+
ci_l = (n.effect_size or {}).get("ci_lower")
|
|
340
|
+
ci_u = (n.effect_size or {}).get("ci_upper")
|
|
337
341
|
if ci_l is not None and ci_u is not None and float(ci_u) > float(ci_l):
|
|
338
342
|
se = (float(ci_u) - float(ci_l)) / (2.0 * 1.95996)
|
|
339
343
|
precision_counts["reported_ci"] += 1
|
|
@@ -422,7 +426,7 @@ class EvidenceGraph:
|
|
|
422
426
|
"quote": p.summary,
|
|
423
427
|
})
|
|
424
428
|
for ev in self.evidence.values():
|
|
425
|
-
effect_val = ev.effect_size.get("value"
|
|
429
|
+
effect_val = (ev.effect_size or {}).get("value") or 0.0
|
|
426
430
|
symbol_size = max(18, min(45, int(18 + abs(effect_val) * 20)))
|
|
427
431
|
nodes.append({
|
|
428
432
|
"id": ev.evidence_id,
|
|
@@ -433,8 +437,8 @@ class EvidenceGraph:
|
|
|
433
437
|
"dimension": ev.outcome_dimension,
|
|
434
438
|
"direction": ev.direction,
|
|
435
439
|
"effect_size": effect_val,
|
|
436
|
-
"ci_lower": ev.effect_size.get("ci_lower", "N/A"),
|
|
437
|
-
"ci_upper": ev.effect_size.get("ci_upper", "N/A"),
|
|
440
|
+
"ci_lower": (ev.effect_size or {}).get("ci_lower", "N/A"),
|
|
441
|
+
"ci_upper": (ev.effect_size or {}).get("ci_upper", "N/A"),
|
|
438
442
|
"sample_size": ev.sample_size,
|
|
439
443
|
"wwc_rating": ev.wwc_rating,
|
|
440
444
|
"quote": ev.key_quote,
|
package/engine/evidencecore.py
CHANGED
|
@@ -12,8 +12,7 @@ v4 领域包机制:domains/ 注册表 + 领域契约加载 + frame 校验。
|
|
|
12
12
|
education 域只是"指向现有契约"的注册:不新增任何逻辑路径、不引入新 schema
|
|
13
13
|
或新校验器。领域选择(domain select)由主 agent 接 CLI 完成,引擎层不做选择。
|
|
14
14
|
|
|
15
|
-
|
|
16
|
-
share/ 回退留给后续步骤(pyproject data-files 未包含 domains/)。
|
|
15
|
+
路径解析:支持仓库、独立 Skill 与 wheel 的 share/eduevidence 资源布局。
|
|
17
16
|
"""
|
|
18
17
|
|
|
19
18
|
from __future__ import annotations
|
|
@@ -22,7 +21,9 @@ import json
|
|
|
22
21
|
from pathlib import Path
|
|
23
22
|
from typing import Any
|
|
24
23
|
|
|
25
|
-
|
|
24
|
+
from engine._resources import resource_root
|
|
25
|
+
|
|
26
|
+
REPO_ROOT = resource_root()
|
|
26
27
|
|
|
27
28
|
|
|
28
29
|
def _resolve_domains_dir() -> Path:
|
|
@@ -114,7 +115,7 @@ def _validate_contracts(entry: dict) -> None:
|
|
|
114
115
|
|
|
115
116
|
- frame_schema / outcome_taxonomy / methodology_checklist:文件存在且
|
|
116
117
|
为可解析 JSON(指针引用另校验指针内容);
|
|
117
|
-
- golds_dir
|
|
118
|
+
- references_dir:目录存在;golds_dir 属于独立 evaluator 资源,不是研究运行依赖。
|
|
118
119
|
"""
|
|
119
120
|
domain_id = entry["id"]
|
|
120
121
|
|
|
@@ -142,7 +143,8 @@ def _validate_contracts(entry: dict) -> None:
|
|
|
142
143
|
check_file("frame_schema")
|
|
143
144
|
check_file("outcome_taxonomy")
|
|
144
145
|
check_file("methodology_checklist")
|
|
145
|
-
|
|
146
|
+
# Evaluation annotations (including holdout answers) are intentionally absent
|
|
147
|
+
# from shipped Skills. Benchmark consumers validate their own input corpus.
|
|
146
148
|
check_dir("references_dir")
|
|
147
149
|
|
|
148
150
|
|