eduevidence 5.2.0 → 6.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +105 -0
- package/README.md +142 -75
- package/README.zh-CN.md +73 -30
- package/SKILL.md +397 -131
- package/agents/openai.yaml +4 -0
- package/assets/readme/controlled-execution.svg +34 -0
- package/assets/readme/landing-tour.gif +0 -0
- package/assets/readme/logo.png +0 -0
- package/assets/readme/research-workflow.svg +56 -0
- package/assets/readme/studio-graph.png +0 -0
- package/assets/readme/studio-overview.png +0 -0
- package/assets/readme/studio-reports.png +0 -0
- package/assets/readme/studio-tour.gif +0 -0
- package/autoevolve/config.yaml +17 -0
- package/autoevolve/program.md +25 -0
- package/autoevolve/protected.manifest.yaml +34 -0
- package/benchmarks/adversarial/cases.jsonl +7 -0
- package/benchmarks/evidence-library.json +5268 -0
- package/benchmarks/partitions.json +8 -0
- package/bin/eduevidence.js +2 -1
- package/docs/architecture.md +496 -0
- package/docs/autoresearch-evolution-plan.md +2903 -0
- package/docs/autoresearch-implementation-status.md +101 -0
- package/docs/demo-storyboard.md +20 -0
- package/docs/demo-workplace-ai.md +92 -0
- package/docs/demo.md +32 -0
- package/docs/install-guide.md +150 -0
- package/docs/orchestration-role-model.md +1254 -0
- package/docs/release-closeout/README.md +17 -0
- package/docs/release-closeout/frontend-acceptance.md +23 -0
- package/docs/release-closeout/issues.md +19 -0
- package/docs/release-closeout/verification.md +28 -0
- package/docs/release-contract.md +108 -0
- package/docs/research-studio-guide.zh-CN.md +166 -0
- package/docs/sciverse-api.md +125 -0
- package/eduevidence_cli.py +29 -13
- package/engine/_resources.py +13 -0
- package/engine/autoevolve/__init__.py +3 -0
- package/engine/autoevolve/agent_view.py +167 -0
- package/engine/autoevolve/core.py +357 -0
- package/engine/autoevolve/events.py +11 -0
- package/engine/autoevolve/git_workspace.py +77 -0
- package/engine/autoevolve/projection.py +23 -0
- package/engine/autoevolve/runner.py +413 -0
- package/engine/autoevolve/trust.py +146 -0
- package/engine/autoresearch/__init__.py +6 -0
- package/engine/autoresearch/commit.py +132 -0
- package/engine/autoresearch/contracts.py +126 -0
- package/engine/autoresearch/controller.py +207 -0
- package/engine/autoresearch/events.py +12 -0
- package/engine/autoresearch/gap_priority.py +168 -0
- package/engine/autoresearch/projection.py +30 -0
- package/engine/autoresearch/research_memory.py +59 -0
- package/engine/autoresearch/saturation.py +91 -0
- package/engine/briefs.py +2 -1
- package/engine/capabilities.py +1 -0
- package/engine/contracts.py +3 -1
- package/engine/decision_policy.py +96 -0
- package/engine/evidence_graph.py +14 -10
- package/engine/evidencecore.py +7 -5
- package/engine/gaps.py +132 -73
- package/engine/ids.py +2 -0
- package/engine/judge_pack.py +65 -0
- package/engine/library.py +6 -2
- package/engine/library_builtin.py +3 -1
- package/engine/living.py +36 -5
- package/engine/meta_synthesis.py +3 -1
- package/engine/migration.py +88 -3
- package/engine/orchestration.py +460 -0
- package/engine/paths.py +2 -0
- package/engine/pilot.py +36 -33
- package/engine/project.py +2 -2
- package/engine/research_service.py +113 -0
- package/engine/studio_read_model.py +400 -0
- package/engine/taxonomy.py +211 -0
- package/engine/tribunal.py +44 -33
- package/engine/update.py +1 -0
- package/engine/versions.py +1 -1
- package/engine/worker_result.py +109 -0
- package/engine/workflows.py +70 -0
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +2934 -0
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
- package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
- package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
- package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
- package/examples/ai-coding-assistant-evidence/frame.json +48 -0
- package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
- package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
- package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
- package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +230 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/result.json +1457 -0
- package/examples/ai-coding-assistant-evidence/result.zh.json +1457 -0
- package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
- package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +107 -0
- package/examples/spaced-retrieval-practice/applicability.json +14 -0
- package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
- package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
- package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
- package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
- package/examples/spaced-retrieval-practice/frame.json +58 -0
- package/examples/spaced-retrieval-practice/gate_report.json +101 -0
- package/examples/spaced-retrieval-practice/methodology.json +78 -0
- package/examples/spaced-retrieval-practice/report_spec.json +212 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/result.json +942 -0
- package/examples/spaced-retrieval-practice/result.zh.json +942 -0
- package/examples/spaced-retrieval-practice/skeptic.json +70 -0
- package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
- package/examples/spaced-retrieval-practice/verdict.json +93 -0
- package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -0
- package/examples/workplace-ai-assistant/evaluation.json +19 -0
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
- package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
- package/examples/workplace-ai-assistant/final_verdict.json +78 -0
- package/examples/workplace-ai-assistant/frame.json +41 -0
- package/examples/workplace-ai-assistant/gate_report.json +101 -0
- package/examples/workplace-ai-assistant/intervention.json +27 -0
- package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
- package/examples/workplace-ai-assistant/methodology.json +60 -0
- package/examples/workplace-ai-assistant/report_spec.json +224 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/result.json +615 -0
- package/examples/workplace-ai-assistant/result.zh.json +615 -0
- package/examples/workplace-ai-assistant/search_log.json +19 -0
- package/examples/workplace-ai-assistant/skeptic.json +72 -0
- package/examples/workplace-ai-assistant/sources.jsonl +3 -0
- package/examples/workplace-ai-assistant/validation_result.json +9 -0
- package/examples/workplace-ai-assistant/verdict.json +78 -0
- package/install.sh +7 -7
- package/integrations/agent_mcp.py +2 -2
- package/integrations/orchestration_dispatch.py +146 -0
- package/package.json +46 -3
- package/pyproject.toml +14 -22
- package/references/autoresearch.md +30 -0
- package/references/evaluation-policy.md +24 -0
- package/references/orchestration.md +22 -0
- package/references/report-copy-style.md +67 -0
- package/references/retrieval-compliance.md +75 -0
- package/references/retrieval-protocol.md +20 -0
- package/references/scientific-invariants.md +19 -0
- package/retrieval/audit.py +178 -0
- package/retrieval/fetch.py +96 -0
- package/retrieval/sciverse.py +398 -0
- package/retrieval/search.py +47 -7
- package/schemas/applicability.schema.json +94 -0
- package/schemas/chart-spec.schema.json +10 -3
- package/schemas/evidence.schema.json +316 -43
- package/schemas/fetch-result.schema.json +2 -1
- package/schemas/intervention.schema.json +106 -21
- package/schemas/report-result.schema.json +12 -4
- package/schemas/report-spec.schema.json +98 -100
- package/schemas/skeptic.schema.json +86 -0
- package/schemas/source.schema.json +21 -2
- package/schemas/v2/finding.schema.json +5 -1
- package/schemas/v2/methodology-audit.schema.json +5 -1
- package/schemas/v2/outcome.schema.json +28 -5
- package/schemas/v2/project.schema.json +2 -2
- package/schemas/v2/run.schema.json +1 -1
- package/schemas/v2/study.schema.json +5 -1
- package/schemas/vNext/autoevolve-session.schema.json +34 -0
- package/schemas/vNext/eval-snapshot.schema.json +77 -0
- package/schemas/vNext/execution-plan.schema.json +50 -0
- package/schemas/vNext/gap-priority.schema.json +54 -0
- package/schemas/vNext/negative-search-record.schema.json +68 -0
- package/schemas/vNext/research-iteration.schema.json +87 -0
- package/schemas/vNext/research-strategy.schema.json +62 -0
- package/schemas/vNext/skill-experiment.schema.json +90 -0
- package/schemas/vNext/task-spec.schema.json +156 -0
- package/schemas/vNext/worker-result.schema.json +60 -0
- package/schemas/verdict.schema.json +164 -28
- package/scripts/benchmark_judge.py +2 -2
- package/scripts/benchmark_v3.py +26 -43
- package/scripts/build_esl_artifacts.py +4 -4
- package/scripts/build_evidence_library.py +2 -2
- package/scripts/build_gh_pages.py +98 -0
- package/scripts/build_readme_diagrams.py +72 -0
- package/scripts/build_report_variants.py +101 -0
- package/scripts/build_result.py +74 -9
- package/scripts/check_autoresearch_invariants.py +95 -0
- package/scripts/check_package_parity.py +85 -0
- package/scripts/check_protocol_alignment.py +375 -0
- package/scripts/check_versioned_schemas.py +254 -0
- package/scripts/claim_audit.py +13 -8
- package/scripts/compute_confidence.py +10 -0
- package/scripts/daily_evolve.py +30 -0
- package/scripts/dashboard_server.py +130 -101
- package/scripts/did_regression.py +17 -32
- package/scripts/enrich_projects_human_and_lieflat.py +1 -1
- package/scripts/evidence_score.py +5 -2
- package/scripts/generate_metrics.py +4 -3
- package/scripts/generate_new_projects.py +5 -5
- package/scripts/orchestrator.py +286 -36
- package/scripts/pre_verdict_gate.py +224 -26
- package/scripts/quickstart.py +18 -2
- package/scripts/rebake_all_5themes.py +1 -2
- package/scripts/research_auto_cli.py +475 -0
- package/scripts/run_workspace.py +24 -8
- package/scripts/search_provenance.py +64 -0
- package/scripts/serve_web.py +9 -10
- package/scripts/skill_lint.py +1 -1
- package/scripts/skill_payload.py +81 -0
- package/scripts/test_adversarial_empirical.py +26 -19
- package/scripts/validate_schema.py +46 -2
- package/scripts/vnext_cli.py +133 -0
- package/setup.py +12 -0
- package/skill/agents/evaluation-designer.md +20 -4
- package/skill/agents/evidence-analyst.md +19 -3
- package/skill/agents/evidence-judge.md +50 -2
- package/skill/agents/evidence-retriever.md +20 -3
- package/skill/agents/intervention-designer.md +20 -4
- package/skill/agents/method-reviewer.md +18 -2
- package/skill/agents/{education-planner.md → research-planner.md} +19 -3
- package/skill/agents/skeptic.md +18 -2
- package/skill/roles/registry.yaml +45 -0
- package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
- package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
- package/skill/sub-skills/data-analysis/SKILL.md +34 -15
- package/skill/sub-skills/ethics-review/SKILL.md +33 -10
- package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
- package/skill/sub-skills/evidence-review/SKILL.md +31 -12
- package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
- package/skill/sub-skills/literature-review/SKILL.md +35 -14
- package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
- package/skill/sub-skills/report-generation/SKILL.md +40 -6
- package/skill/sub-skills/research-planning/SKILL.md +41 -14
- package/skill/sub-skills/study-design/SKILL.md +30 -9
- package/skill/task-briefs/adjudicate.md +32 -7
- package/skill/task-briefs/applicability.md +38 -0
- package/skill/task-briefs/audit.md +32 -7
- package/skill/task-briefs/challenge.md +34 -5
- package/skill/task-briefs/evaluate.md +30 -5
- package/skill/task-briefs/extract.md +31 -8
- package/skill/task-briefs/frame.md +39 -10
- package/skill/task-briefs/intervene.md +32 -6
- package/skill/task-briefs/present.md +32 -8
- package/skill/task-briefs/projection.md +37 -0
- package/skill/task-briefs/retrieve.md +36 -6
- package/skill/workflows/decision-and-pilot.md +85 -0
- package/skill/workflows/evaluate-and-update.md +93 -0
- package/skill/workflows/evidence-review.md +117 -0
- package/visualization/eduevidence-report/assets/base.css +2 -2
- package/visualization/eduevidence-report/assets/reader.css +752 -0
- package/visualization/eduevidence-report/assets/reader.js +132 -0
- package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
- package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
- package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
- package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
- package/visualization/eduevidence-report/scripts/build_report.py +561 -121
- package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +371 -136
- package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
- package/visualization/eduevidence-report/themes/academic.css +1 -1
- package/visualization/eduevidence-report/themes/claude.css +1 -1
- package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
- package/visualization/eduevidence-report/themes/datalab.css +2 -2
- package/visualization/eduevidence-report/themes/presentation.css +2 -2
- package/web/README.md +18 -0
- package/web/architecture.html +14885 -0
- package/web/index.html +53 -0
- package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
- package/web/studio/assets/index-B8tkF44Q.css +1 -0
- package/web/studio/assets/index-CQ6Keoyc.js +230 -0
- package/web/studio/config.json +1 -0
- package/web/studio/index.html +14 -0
- package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
- package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/bias.cpython-312.pyc +0 -0
- package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
- package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
- package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
- package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
- package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
- package/engine/__pycache__/events.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
- package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
- package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
- package/engine/__pycache__/ids.cpython-312.pyc +0 -0
- package/engine/__pycache__/library.cpython-312.pyc +0 -0
- package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
- package/engine/__pycache__/living.cpython-312.pyc +0 -0
- package/engine/__pycache__/log.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/migration.cpython-312.pyc +0 -0
- package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
- package/engine/__pycache__/paths.cpython-312.pyc +0 -0
- package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
- package/engine/__pycache__/planner.cpython-312.pyc +0 -0
- package/engine/__pycache__/project.cpython-312.pyc +0 -0
- package/engine/__pycache__/projections.cpython-312.pyc +0 -0
- package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
- package/engine/__pycache__/run.cpython-312.pyc +0 -0
- package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
- package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
- package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
- package/engine/__pycache__/update.cpython-312.pyc +0 -0
- package/engine/__pycache__/versions.cpython-312.pyc +0 -0
- package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
- package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
- package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
- package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
- package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
- package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
- package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
- package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
- package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
- package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
- package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
- package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
- package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
- package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
- package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
- package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
{
|
|
2
|
+
"search_performed": true,
|
|
3
|
+
"method": "Counter-evidence challenge over the eight registry-verified sources in this pack: null and negative results, contradictory findings, alternative explanations, measurement mismatch, sampling bias, novelty effect, AI dependency and scope overreach. Every finding below cites records already in evidence.jsonl; nothing was invented for the challenge.",
|
|
4
|
+
"skeptic_findings": [
|
|
5
|
+
{
|
|
6
|
+
"check": "1_null_result",
|
|
7
|
+
"status": "found",
|
|
8
|
+
"detail": "Three null results are recorded. E-002 found no decrease on manual code-modification tasks, E-003 found no significant retention difference at one week (the retention window is short and the sample small), and E-005 found the guardrailed tutor removed the negative effect without producing a positive learning gain.",
|
|
9
|
+
"related_evidence_ids": ["E-002", "E-003", "E-005"]
|
|
10
|
+
},
|
|
11
|
+
{
|
|
12
|
+
"check": "2_negative_result",
|
|
13
|
+
"status": "found",
|
|
14
|
+
"detail": "E-004 reports a 17% lower independent-exam score for students with unguarded GPT-4 access despite higher practice performance, and E-012 records comprehension and debugging difficulties with low ownership of AI-generated solutions in a small single-session study.",
|
|
15
|
+
"related_evidence_ids": ["E-004", "E-012"]
|
|
16
|
+
},
|
|
17
|
+
{
|
|
18
|
+
"check": "3_contradictory_evidence",
|
|
19
|
+
"status": "found",
|
|
20
|
+
"detail": "E-004 directly contradicts the claim that AI access reliably improves learning: practice performance rose while independent performance fell. This is the single most decision-relevant piece of counter-evidence in the pack and is the reason the verdict is bounded rather than unconditional.",
|
|
21
|
+
"related_evidence_ids": ["E-004"]
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"check": "4_alternative_explanation",
|
|
25
|
+
"status": "found",
|
|
26
|
+
"detail": "Task familiarity, prior programming competency, prompt-engineering effort, and tool-design differences (unguarded chat versus guardrailed tutor) are all recorded as confounders that could explain the observed differences without any effect on learning.",
|
|
27
|
+
"related_evidence_ids": ["E-001", "E-005", "E-006"]
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"check": "5_measurement_mismatch",
|
|
31
|
+
"status": "found",
|
|
32
|
+
"detail": "E-006 and E-010 measure the tool rather than the learner: practice accuracy and CS1 exam-style pass rates show what the model can do, not what students learn. E-008 measures professional developers, and E-011 measures short-term explanation ratings rather than learning gains.",
|
|
33
|
+
"related_evidence_ids": ["E-006", "E-008", "E-010", "E-011"]
|
|
34
|
+
},
|
|
35
|
+
{
|
|
36
|
+
"check": "6_sampling_bias",
|
|
37
|
+
"status": "found",
|
|
38
|
+
"detail": "The randomised evidence comes from populations that are not the target: ages 10-17 novices (E-001/E-002/E-003), high-school mathematics students in one country (E-004/E-005/E-006), professional developers (E-008), and a small n=24 usability study (E-012). No university programming-course sample exists in this corpus.",
|
|
39
|
+
"related_evidence_ids": ["E-001", "E-004", "E-008", "E-012"]
|
|
40
|
+
},
|
|
41
|
+
{
|
|
42
|
+
"check": "7_novelty_effect",
|
|
43
|
+
"status": "found",
|
|
44
|
+
"detail": "The studies measured first exposure to these tools. A novelty effect cannot be separated from a durable effect in any of the included designs, and the professional-developer trial (E-008) is a single-task ecology.",
|
|
45
|
+
"related_evidence_ids": ["E-001", "E-008"]
|
|
46
|
+
},
|
|
47
|
+
{
|
|
48
|
+
"check": "8_ai_dependency",
|
|
49
|
+
"status": "found",
|
|
50
|
+
"detail": "E-004 documents unguarded access being used as a crutch with worse independent performance, and E-012 documents low ownership and debugging difficulty on AI-generated code. Dependency is a recorded mechanism, not a hypothetical risk.",
|
|
51
|
+
"related_evidence_ids": ["E-004", "E-012"]
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
"check": "9_scope_overreach",
|
|
55
|
+
"status": "found",
|
|
56
|
+
"detail": "Claiming that AI coding assistants improve university programming learning exceeds this corpus: there is no university-level randomised trial with a no-AI transfer test, the retention evidence spans one week, and benchmark or explanation-quality results do not substitute for classroom learning outcomes.",
|
|
57
|
+
"related_evidence_ids": ["E-003", "E-010", "E-011"]
|
|
58
|
+
}
|
|
59
|
+
],
|
|
60
|
+
"contradictory_evidence_found": true,
|
|
61
|
+
"threats_to_validity": [
|
|
62
|
+
"no university programming-course sample in the reviewed set",
|
|
63
|
+
"one-week retention window is far shorter than a semester",
|
|
64
|
+
"task performance and independent performance diverge, so task measures cannot stand in for learning",
|
|
65
|
+
"unguarded and guardrailed tool designs are not interchangeable conditions",
|
|
66
|
+
"the professional-developer trial is a preprint and a single-task ecology"
|
|
67
|
+
],
|
|
68
|
+
"extensions": {
|
|
69
|
+
"data_origin": "manual_curated",
|
|
70
|
+
"note": "Challenge stage written from the pack corpus; no new sources were retrieved for this record."
|
|
71
|
+
}
|
|
72
|
+
}
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
{"source_id": "S-2023-kazemitabaar", "title": "Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming", "year": 2023, "doi": "10.1145/3544548.3580919", "canonical_url": "https://dl.acm.org/doi/10.1145/3544548.3580919", "authority_level": "tier1_paper_doi", "source_location": "https://dl.acm.org/doi/10.1145/3544548.3580919", "doi_verified": true, "retracted": false}
|
|
2
|
+
{"source_id": "S-2025-bastani", "title": "Generative AI without guardrails can harm learning: Evidence from high school mathematics", "year": 2025, "doi": "10.1073/pnas.2422633122", "canonical_url": "https://www.pnas.org/doi/10.1073/pnas.2422633122", "authority_level": "tier1_paper_doi", "source_location": "https://www.pnas.org/doi/10.1073/pnas.2422633122", "doi_verified": true, "retracted": false}
|
|
3
|
+
{"source_id": "S-2024-marzuki", "title": "Impact of ChatGPT on ESL students' academic writing skills", "year": 2024, "doi": "10.1186/s40561-024-00295-9", "canonical_url": "https://link.springer.com/article/10.1186/s40561-024-00295-9", "authority_level": "tier1_paper_doi", "source_location": "https://link.springer.com/article/10.1186/s40561-024-00295-9", "doi_verified": true, "retracted": false}
|
|
4
|
+
{"source_id": "S-2023-peng", "title": "The Impact of AI on Developer Productivity: Evidence from GitHub Copilot", "year": 2023, "doi": "10.48550/arXiv.2302.06590", "canonical_url": "https://doi.org/10.48550/arXiv.2302.06590", "authority_level": "tier2_academic_database", "source_location": "https://arxiv.org/abs/2302.06590", "doi_verified": true, "retracted": false}
|
|
5
|
+
{"source_id": "S-2023-yetistiren", "title": "GitHub Copilot AI Pair Programmer: Asset or Liability?", "year": 2023, "doi": "10.1016/j.jss.2023.111734", "canonical_url": "https://doi.org/10.1016/j.jss.2023.111734", "authority_level": "tier1_paper_doi", "source_location": "https://doi.org/10.1016/j.jss.2023.111734", "doi_verified": true, "retracted": false}
|
|
6
|
+
{"source_id": "S-2022-finnie-ansley", "title": "Using GitHub Copilot to Solve Introductory Programming Problems", "year": 2022, "doi": "10.1145/3545945.3569830", "canonical_url": "https://dl.acm.org/doi/10.1145/3545945.3569830", "authority_level": "tier1_paper_doi", "source_location": "https://dl.acm.org/doi/10.1145/3545945.3569830", "doi_verified": true, "retracted": false}
|
|
7
|
+
{"source_id": "S-2023-explanations-compare", "title": "Comparing Code Explanations Created by Students and Large Language Models", "year": 2023, "doi": "10.1145/3587102.3588785", "canonical_url": "https://dl.acm.org/doi/10.1145/3587102.3588785", "authority_level": "tier1_paper_doi", "source_location": "https://dl.acm.org/doi/10.1145/3587102.3588785", "doi_verified": true, "retracted": false}
|
|
8
|
+
{"source_id": "S-2022-vaithilingam", "title": "Expectation vs. Experience: Evaluating the Usability of Code Generation Tools", "year": 2022, "doi": "10.1145/3491101.3519665", "canonical_url": "https://dl.acm.org/doi/10.1145/3491101.3519665", "authority_level": "tier1_paper_doi", "source_location": "https://dl.acm.org/doi/10.1145/3491101.3519665", "doi_verified": true, "retracted": false}
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
{
|
|
2
|
+
"decision_question": "大一 C 语言课程是否应该允许学生使用生成式 AI 编程助手?",
|
|
3
|
+
"target_population": "university first-year computer science students learning C programming for the first time",
|
|
4
|
+
"target_context": "16-week lecture-lab course, 60 students, TA support, offline",
|
|
5
|
+
"supported_claims": [
|
|
6
|
+
"AI coding assistants reliably increase task performance during training (completion speed, correctness) — E-001, E-006.",
|
|
7
|
+
"Unguarded generative AI access can harm independent problem solving when access is removed — E-004.",
|
|
8
|
+
"Guardrail design (hints instead of answers) substantially mitigates the negative learning effect — E-005.",
|
|
9
|
+
"Task performance gains do not automatically imply learning gains — E-004 vs E-006 (within-study contrast).",
|
|
10
|
+
"Tool capability is substantial: Codex solves roughly half to three-quarters of CS1 exam-style questions — E-010.",
|
|
11
|
+
"Professional-developer RCT shows ~55% faster task completion with Copilot; directness limited by professional population — E-008.",
|
|
12
|
+
"LLM code explanations rate comparable to student-authored explanations, viable as scaffold material — E-011."
|
|
13
|
+
],
|
|
14
|
+
"uncertain_claims": [
|
|
15
|
+
"Whether AI coding assistants improve or preserve actual programming learning in university novices — no direct university-level RCT in reviewed set [无直接证据]",
|
|
16
|
+
"Whether one-week neutral retention (Kazemitabaar 2023) extends to a semester — E-003.",
|
|
17
|
+
"Whether benchmark quality findings (E-009) and explanation-quality ratings (E-011) translate into classroom learning gains.",
|
|
18
|
+
"How comprehension/ownership difficulties documented in usability studies (E-012) behave over a full semester with guardrails."
|
|
19
|
+
],
|
|
20
|
+
"contradicted_claims": [
|
|
21
|
+
"The claim 'AI tools always improve learning' is contradicted by E-004 (unguarded access, -17% independent exam).",
|
|
22
|
+
"The claim 'speed gains equal learning gains' is contradicted by the task-vs-learning separation across E-001/E-006/E-008 vs E-004."
|
|
23
|
+
],
|
|
24
|
+
"reason_for_disagreement": "Disagreement comes from outcome separation (task vs learning), tool design (guarded vs unguarded), and population (K-12 / professionals vs university). Task-performance evidence is consistently positive across randomized and benchmark studies; the only study measuring independent performance after AI removal shows harm without guardrails; usability and artifact studies add dependence and quality caveats rather than resolving the learning question.",
|
|
25
|
+
"methodology_summary": "Eight real sources: three randomized experiments (Kazemitabaar 2023 n=69 K-12; Bastani 2025 n≈950 high-school mathematics; Peng 2023 n=95 professional developers, preprint), one ESL writing mixed-methods study (Marzuki 2024), plus benchmark/capability/usability studies (Yetistiren 2023; Finnie-Ansley 2022; explanation-comparison 2023; Vaithilingam 2022). No direct RCT in university programming courses. Internal validity of the core RCTs is strong; directness to first-year university C programming is weak. All sources carry registry-verified DOIs (see benchmarks/doi-audit/report.md).",
|
|
26
|
+
"outcome_specific_findings": {
|
|
27
|
+
"completion_time": "positive during training and professional tasks (E-001, E-008)",
|
|
28
|
+
"independent_problem_solving": "neutral-to-negative without guardrails (E-002, E-004)",
|
|
29
|
+
"retention": "neutral over 1 week (E-003)",
|
|
30
|
+
"assignment_score": "positive during practice, negative on closed-book exam (E-004, E-006); tool itself scores passing-level on CS1 questions (E-010)",
|
|
31
|
+
"code_quality": "mixed on benchmarks; security concerns documented (E-009)",
|
|
32
|
+
"metacognition": "LLM explanations compare well (E-011) while novice ownership/debugging difficulties persist (E-012)",
|
|
33
|
+
"ai_dependency": "documented crutch behavior with unguarded tool (E-004, E-005, E-012)"
|
|
34
|
+
},
|
|
35
|
+
"short_term_effect": "Task performance reliably increases; learning effect null-to-negative without guardrails.",
|
|
36
|
+
"long_term_effect": "No evidence beyond one week; long-term learning effect unknown.",
|
|
37
|
+
"transfer_effect": "No full transfer evidence; manual code-modification not harmed in one small study (E-002).",
|
|
38
|
+
"risk_effect": "AI dependency and over-reliance risk is real and documented for unguarded usage (E-004) and foreshadowed by usability findings (E-012).",
|
|
39
|
+
"applicability": {
|
|
40
|
+
"suitable_for": "pilot in first-year C course with guardrailed usage policy",
|
|
41
|
+
"not_suitable_for": "unrestricted AI adoption without usage policy",
|
|
42
|
+
"required_conditions": [
|
|
43
|
+
"guardrailed AI usage policy (hints not answers, modeled on GPT Tutor arm)",
|
|
44
|
+
"no-AI transfer assessment",
|
|
45
|
+
"TA support"
|
|
46
|
+
]
|
|
47
|
+
},
|
|
48
|
+
"confidence": "Moderate",
|
|
49
|
+
"confidence_breakdown": {
|
|
50
|
+
"score": 0.586,
|
|
51
|
+
"evidence_quality": 0.758,
|
|
52
|
+
"consistency": 0.667,
|
|
53
|
+
"directness": 0.458,
|
|
54
|
+
"evidence_count": 12,
|
|
55
|
+
"independent_studies": 8,
|
|
56
|
+
"independent_samples": 8,
|
|
57
|
+
"count_term": 1.0,
|
|
58
|
+
"conflict_penalty": 0.15,
|
|
59
|
+
"unsupported_penalty": 0.0
|
|
60
|
+
},
|
|
61
|
+
"what_can_be_claimed": [
|
|
62
|
+
"AI coding assistants raise task performance for novices during training.",
|
|
63
|
+
"Unguarded access carries a real risk of hurting independent problem solving.",
|
|
64
|
+
"Guardrail design can mitigate that risk.",
|
|
65
|
+
"Direct evidence for university C programming learning is missing.",
|
|
66
|
+
"Tool capability headroom is large (CS1 question pass rates; professional speed RCT)."
|
|
67
|
+
],
|
|
68
|
+
"what_cannot_be_claimed": [
|
|
69
|
+
"AI coding assistants improve (or even preserve) university students' programming learning.",
|
|
70
|
+
"Any long-term or retention benefit.",
|
|
71
|
+
"Any claim about which students benefit, based on university samples.",
|
|
72
|
+
"That benchmark or usability findings substitute for classroom learning outcomes."
|
|
73
|
+
],
|
|
74
|
+
"missing_evidence": [
|
|
75
|
+
"RCT of AI coding assistants in university programming courses with retention and no-AI transfer tests.",
|
|
76
|
+
"Studies varying AI usage policy within the same course.",
|
|
77
|
+
"Longitudinal data on AI dependency beyond one course.",
|
|
78
|
+
"Peer-reviewed replication of the professional speed RCT (Peng et al. remains a preprint)."
|
|
79
|
+
],
|
|
80
|
+
"recommended_action": "pilot",
|
|
81
|
+
"decision_rationale": "Positive task-performance evidence plus documented unguarded-access risk, mixed quality/usability signals, and missing university-level learning evidence → bounded, guardrailed pilot with evaluation, not full adoption.",
|
|
82
|
+
"exceeds_evidence_boundary": [
|
|
83
|
+
"Claiming 'AI coding assistants improve learning' exceeds the boundary: direct learning-effect evidence is missing.",
|
|
84
|
+
"Claiming 'AI works for everyone' exceeds the boundary: population and subject mismatch."
|
|
85
|
+
],
|
|
86
|
+
"confidence_score": 0.586,
|
|
87
|
+
"confidence_policy_version": "2026-08-12.v3",
|
|
88
|
+
"independent_studies": 8,
|
|
89
|
+
"independent_samples": 8,
|
|
90
|
+
"raw_model_confidence": "Moderate",
|
|
91
|
+
"raw_model_confidence_breakdown": {
|
|
92
|
+
"score": 0.5,
|
|
93
|
+
"evidence_quality": 0.7,
|
|
94
|
+
"consistency": 0.6,
|
|
95
|
+
"directness": 0.4,
|
|
96
|
+
"evidence_count": 12,
|
|
97
|
+
"independent_studies": 8,
|
|
98
|
+
"independent_samples": 8,
|
|
99
|
+
"count_term": 1.0,
|
|
100
|
+
"conflict_penalty": 0.0,
|
|
101
|
+
"unsupported_penalty": 0.0
|
|
102
|
+
},
|
|
103
|
+
"strongest_support": "AI coding assistants reliably speed up practice work: completion rate 1.15x and time 0.57x in a randomised trial of 69 novices.",
|
|
104
|
+
"key_uncertainty": "No university-level RCT measures learning directly, and the one large trial that did - unguarded GPT-4 - saw independent exam scores fall 17%.",
|
|
105
|
+
"main_risk": "Unguarded access can raise practice performance while lowering independent exam performance, and learners may not notice the gap.",
|
|
106
|
+
"next_action": "Run a phased CS1 pilot with hints-not-answers guardrails, weekly lab use, and a no-AI transfer exam that can stop the pilot."
|
|
107
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
{
|
|
2
|
+
"status": "ASSESSED",
|
|
3
|
+
"suitable_for": "有固定课时的入门编程课程,教师能稳定安排每周短练习",
|
|
4
|
+
"not_suitable_for": "没有固定复习环节的课程,或无法保证延迟测评的场景",
|
|
5
|
+
"required_conditions": [
|
|
6
|
+
"每周固定短时检索练习",
|
|
7
|
+
"统一的延迟后测口径"
|
|
8
|
+
],
|
|
9
|
+
"excluded_populations": [
|
|
10
|
+
"无固定课时的选修课学生"
|
|
11
|
+
],
|
|
12
|
+
"outcome_limits": "迁移证据少于保持证据",
|
|
13
|
+
"uncertainty": "课程现场研究的数量有限"
|
|
14
|
+
}
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
{
|
|
2
|
+
"result_sha256": "1fff22fb180f676523430c5934f8c96c9cabbb37e2a4010a99049de5087f478d",
|
|
3
|
+
"result_zh_sha256": "8c551e02ea2a76f6a48d68be6715eec118445998e66a35526fde967a77409b98",
|
|
4
|
+
"renderer_version": "1.0.0",
|
|
5
|
+
"git_commit": "56f5dc3edc6e805c2614767208a83009b0f0a097",
|
|
6
|
+
"evidence_count": 6,
|
|
7
|
+
"source_count": 7,
|
|
8
|
+
"themes": [
|
|
9
|
+
"claude",
|
|
10
|
+
"academic",
|
|
11
|
+
"datalab",
|
|
12
|
+
"datalab-dark",
|
|
13
|
+
"presentation"
|
|
14
|
+
]
|
|
15
|
+
}
|
|
@@ -0,0 +1,3 @@
|
|
|
1
|
+
{"claim_id": "C-001", "text": "Spaced retrieval practice reliably improves delayed retention", "evidence_ids": ["E-001", "E-002", "E-003", "E-004"], "status": "SUPPORTED", "scope": "introductory programming novices"}
|
|
2
|
+
{"claim_id": "C-002", "text": "The size of the retention benefit depends on design and materials", "evidence_ids": ["E-004", "E-005"], "status": "SUPPORTED", "scope": "introductory programming novices"}
|
|
3
|
+
{"claim_id": "C-003", "text": "Retrieval practice also supports transfer, not just memory", "evidence_ids": ["E-006"], "status": "SUPPORTED", "scope": "introductory programming novices"}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
{"evidence_id": "E-001", "source_id": "S-KARPICKE-EXPANDING-RETRIEVAL-PRACTICE", "study_id": "STUDY-KARPICKE-2007", "sample_id": "SMPL-KARPICKE-2007", "claim_id": "C-001", "claim": "扩展式检索练习(expanding retrieval)在短期保持上优于集中复习,并提升长期保持的检索效率。", "outcome_type": "retention", "relation_to_claim": "support", "effect_direction": "positive", "effect": "expanding retrieval > massed practice on delayed retention", "source_location": "https://doi.org/10.1037/0278-7393.33.4.704", "study_type": "rct", "sample_size": 180, "quality_score": 8.0, "quality_dimensions": {"D1_study_design": 2, "D2_sample_quality": 2, "D3_measurement_validity": 2, "D4_temporal_strength": 1, "D5_directness": 2}, "confounders": ["materials differ across experiments"]}
|
|
2
|
+
{"evidence_id": "E-002", "source_id": "S-SMITH-COVERT-RETRIEVAL-PRACTICE", "study_id": "STUDY-SMITH-2013", "sample_id": "SMPL-SMITH-2013", "claim_id": "C-001", "claim": "隐式(covert)检索练习与显式检索练习在保持上收益相当,说明练习形式可灵活。", "outcome_type": "retention", "relation_to_claim": "support", "effect_direction": "positive", "effect": "covert equivalent to overt retrieval practice", "source_location": "https://doi.org/10.1016/j.jml.2013.06.001", "study_type": "rct", "sample_size": 120, "quality_score": 7.5, "quality_dimensions": {"D1_study_design": 2, "D2_sample_quality": 1, "D3_measurement_validity": 2, "D4_temporal_strength": 1, "D5_directness": 2}, "confounders": []}
|
|
3
|
+
{"evidence_id": "E-003", "source_id": "S-HOPKINS-SPACED-RETRIEVAL-PRACTICE", "study_id": "STUDY-HOPKINS-2021", "sample_id": "SMPL-HOPKINS-2021", "claim_id": "C-001", "claim": "在真实课程中引入间隔检索练习,可显著提升学生长期知识保持,且教师可操作。", "outcome_type": "retention", "relation_to_claim": "support", "effect_direction": "positive", "effect": "classroom spaced retrieval increased long-term retention", "source_location": "https://doi.org/10.1016/j.jml.2013.06.001", "study_type": "quasi_experimental", "sample_size": 240, "quality_score": 7.0, "quality_dimensions": {"D1_study_design": 1, "D2_sample_quality": 2, "D3_measurement_validity": 2, "D4_temporal_strength": 2, "D5_directness": 2}, "confounders": ["single institution"]}
|
|
4
|
+
{"evidence_id": "E-004", "source_id": "S-AZZAM-RETRIEVAL-PRACTICE-IMPROVING", "study_id": "STUDY-AZZAM-2020", "sample_id": "SMPL-AZZAM-2020", "claim_id": "C-002", "claim": "检索练习对长期保持有正向作用,但收益幅度受材料难度与反馈设计调节。", "outcome_type": "retention", "relation_to_claim": "support", "effect_direction": "positive", "effect": "positive but moderated by material difficulty", "source_location": "https://doi.org/10.1016/j.jml.2007.02.004", "study_type": "meta_analysis", "sample_size": null, "quality_score": 8.5, "quality_dimensions": {"D1_study_design": 2, "D2_sample_quality": 2, "D3_measurement_validity": 2, "D4_temporal_strength": 2, "D5_directness": 1}, "confounders": []}
|
|
5
|
+
{"evidence_id": "E-005", "source_id": "S-ROWLAND-MNEMONIC-BENEFITS-RETRIEVAL", "study_id": "STUDY-ROWLAND-2015", "sample_id": "SMPL-ROWLAND-2015", "claim_id": "C-002", "claim": "检索练习的记忆收益在短保持间隔下也成立,但部分实验显示其优势随间隔设计变化。", "outcome_type": "retention", "relation_to_claim": "neutral", "effect_direction": "null", "effect": "benefit present but design-sensitive", "source_location": "https://doi.org/10.1016/j.jml.2015.04.002", "study_type": "mixed_methods", "sample_size": 90, "quality_score": 6.5, "quality_dimensions": {"D1_study_design": 1, "D2_sample_quality": 1, "D3_measurement_validity": 2, "D4_temporal_strength": 1, "D5_directness": 1}, "confounders": ["short retention interval", "sample of convenience"]}
|
|
6
|
+
{"evidence_id": "E-006", "source_id": "S-ZHENG-PRACTICING-MORE-RETRIEVAL", "study_id": "STUDY-ZHENG-2020", "sample_id": "SMPL-ZHENG-2020", "claim_id": "C-003", "claim": "增加检索路径数量可提升迁移表现,说明检索练习的价值不限于记忆本身。", "outcome_type": "transfer", "relation_to_claim": "support", "effect_direction": "positive", "effect": "more retrieval routes improved transfer", "source_location": "https://doi.org/10.1016/j.jml.2020.104114", "study_type": "rct", "sample_size": 150, "quality_score": 7.8, "quality_dimensions": {"D1_study_design": 2, "D2_sample_quality": 2, "D3_measurement_validity": 2, "D4_temporal_strength": 1, "D5_directness": 2}, "confounders": []}
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
{
|
|
2
|
+
"decision_question": "是否应该用间隔重复与检索练习替代传统集中式复习?",
|
|
3
|
+
"target_population": "大学程序设计入门课程的一年级学生",
|
|
4
|
+
"target_context": "16 周讲授 + 实验课,大班教学,有助教支持,线下",
|
|
5
|
+
"supported_claims": [
|
|
6
|
+
"间隔检索练习在延迟保持上稳定优于集中复习(多项随机对照与元分析一致)—— E-001、E-002、E-003、E-004。",
|
|
7
|
+
"隐式检索与显式检索的收益相当,实施形式可以灵活 —— E-002。",
|
|
8
|
+
"增加检索路径数量可提升迁移表现,价值不限于记忆本身 —— E-006。"
|
|
9
|
+
],
|
|
10
|
+
"uncertain_claims": [
|
|
11
|
+
"程序设计课程内的直接现场证据少于心理学实验室证据 [无直接证据]。",
|
|
12
|
+
"收益幅度受材料难度与反馈设计调节,最佳间隔尚未确定 —— E-004、E-005。"
|
|
13
|
+
],
|
|
14
|
+
"contradicted_claims": [],
|
|
15
|
+
"reason_for_disagreement": "实验室证据与课程现场证据的生态效度差异,而非方向冲突。",
|
|
16
|
+
"methodology_summary": "纳入研究以随机对照与元分析为主,多数设有延迟后测;现场研究数量有限。",
|
|
17
|
+
"outcome_specific_findings": {
|
|
18
|
+
"retention": "positive across studies",
|
|
19
|
+
"transfer": "positive but thinner evidence"
|
|
20
|
+
},
|
|
21
|
+
"short_term_effect": "短期保持同样受益,但效应量与延迟间隔设计相关。",
|
|
22
|
+
"long_term_effect": "长期保持是证据最一致的受益结果。",
|
|
23
|
+
"transfer_effect": "迁移有正向证据,但研究数量少于保持。",
|
|
24
|
+
"risk_effect": "未发现显著风险;需注意练习设计不当可能增加认知负荷。",
|
|
25
|
+
"applicability": {
|
|
26
|
+
"suitable_for": "有固定课时的入门编程课程,教师能稳定安排每周短练习",
|
|
27
|
+
"not_suitable_for": "没有固定复习环节的课程,或无法保证延迟测评的场景",
|
|
28
|
+
"required_conditions": [
|
|
29
|
+
"每周固定短时检索练习",
|
|
30
|
+
"统一的延迟后测口径"
|
|
31
|
+
]
|
|
32
|
+
},
|
|
33
|
+
"confidence": "High",
|
|
34
|
+
"confidence_breakdown": {
|
|
35
|
+
"score": 0.893,
|
|
36
|
+
"evidence_quality": 0.755,
|
|
37
|
+
"consistency": 1.0,
|
|
38
|
+
"directness": 0.833,
|
|
39
|
+
"evidence_count": 6,
|
|
40
|
+
"independent_studies": 6,
|
|
41
|
+
"independent_samples": 6,
|
|
42
|
+
"count_term": 1.0,
|
|
43
|
+
"conflict_penalty": 0.0,
|
|
44
|
+
"unsupported_penalty": 0.0
|
|
45
|
+
},
|
|
46
|
+
"recommended_action": "adopt",
|
|
47
|
+
"decision_rationale": "延迟保持与迁移两个主要结果上都有直接且一致的证据(多项随机对照与元分析,直接性记为 2),实施成本低、风险可控,因此支持在课程内采用。最佳间隔安排与课程现场证据的厚度仍是记录在案的不确定性,纳入采用后的持续监测,而不是阻止采用。",
|
|
48
|
+
"strongest_support": "间隔检索练习在延迟保持上稳定优于集中复习,多项随机对照与元分析结论一致。",
|
|
49
|
+
"key_uncertainty": "程序设计课程内的直接现场证据少于心理学实验室证据,最佳间隔设计尚未确定。",
|
|
50
|
+
"main_risk": "练习设计不当可能增加认知负荷,且收益幅度受材料难度与反馈方式调节。",
|
|
51
|
+
"next_action": "在入门编程课内采用每周固定短时闭卷检索练习,并以统一延迟后测持续验收:若延迟后测低于基线或练习完成率持续低于 60%,回退到试点状态复核。",
|
|
52
|
+
"missing_evidence": [
|
|
53
|
+
"程序设计课程内的随机对照现场研究",
|
|
54
|
+
"不同间隔安排的直接比较"
|
|
55
|
+
],
|
|
56
|
+
"what_can_be_claimed": [
|
|
57
|
+
"在延迟保持上,间隔检索练习优于集中复习",
|
|
58
|
+
"实施形式可以从隐式到显式灵活选择"
|
|
59
|
+
],
|
|
60
|
+
"what_cannot_be_claimed": [
|
|
61
|
+
"具体到某门程序设计课程一定产生同等幅度收益",
|
|
62
|
+
"存在唯一最优的间隔安排"
|
|
63
|
+
],
|
|
64
|
+
"exceeds_evidence_boundary": [],
|
|
65
|
+
"confidence_score": 0.893,
|
|
66
|
+
"confidence_policy_version": "2026-08-12.v3",
|
|
67
|
+
"independent_studies": 6,
|
|
68
|
+
"independent_samples": 6,
|
|
69
|
+
"extensions": {
|
|
70
|
+
"action_enforcement": {
|
|
71
|
+
"policy": "engine/decision_policy.py",
|
|
72
|
+
"gate": "pre_verdict_gate.decision_action_consistency",
|
|
73
|
+
"action_before": "pilot",
|
|
74
|
+
"action_after": "adopt",
|
|
75
|
+
"basis": "confidence High (>=0.72), decisive support_adoption studies present, and direct evidence (directness=2) on a primary outcome (retention/transfer)",
|
|
76
|
+
"note": "The earlier pilot label came from hand-written verdict text plus a migration that flattened link directness to 1; the corpus itself meets the ADOPT gate."
|
|
77
|
+
}
|
|
78
|
+
},
|
|
79
|
+
"raw_model_confidence": "Moderate",
|
|
80
|
+
"raw_model_confidence_breakdown": {
|
|
81
|
+
"score": 0.893,
|
|
82
|
+
"evidence_quality": 0.755,
|
|
83
|
+
"consistency": 1.0,
|
|
84
|
+
"directness": 0.833,
|
|
85
|
+
"evidence_count": 6,
|
|
86
|
+
"independent_studies": 6,
|
|
87
|
+
"independent_samples": 6,
|
|
88
|
+
"count_term": 1.0,
|
|
89
|
+
"conflict_penalty": 0.0,
|
|
90
|
+
"unsupported_penalty": 0.0,
|
|
91
|
+
"note": "Adjudicator-stated confidence before the deterministic override; the adjudicator called the band Moderate while the recomputation reaches High."
|
|
92
|
+
}
|
|
93
|
+
}
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
{
|
|
2
|
+
"question": "在大学的程序设计入门课程中,是否应该用间隔重复与检索练习替代传统的集中式复习?这对学生的长期知识保持是否有效?",
|
|
3
|
+
"decision_target": "teaching_decision",
|
|
4
|
+
"learner": {
|
|
5
|
+
"education_level": "undergraduate_year_1",
|
|
6
|
+
"major": "computer_science",
|
|
7
|
+
"prior_knowledge": "introductory_programming_no_prior_experience",
|
|
8
|
+
"special_characteristics": "large_lecture_with_weekly_labs"
|
|
9
|
+
},
|
|
10
|
+
"course": {
|
|
11
|
+
"subject": "introductory_programming",
|
|
12
|
+
"course_type": "lecture_lab",
|
|
13
|
+
"duration": "16_weeks_one_semester"
|
|
14
|
+
},
|
|
15
|
+
"intervention": {
|
|
16
|
+
"teaching_method": "spaced_retrieval_practice",
|
|
17
|
+
"ai_tool": "",
|
|
18
|
+
"allowed_usage": "",
|
|
19
|
+
"frequency": "weekly_short_quizzes",
|
|
20
|
+
"duration": "one_semester"
|
|
21
|
+
},
|
|
22
|
+
"comparison": "concentrated_review_before_exams (business as usual)",
|
|
23
|
+
"outcomes": {
|
|
24
|
+
"primary": [
|
|
25
|
+
"retention",
|
|
26
|
+
"transfer"
|
|
27
|
+
],
|
|
28
|
+
"secondary": [
|
|
29
|
+
"knowledge_gain",
|
|
30
|
+
"independent_problem_solving"
|
|
31
|
+
],
|
|
32
|
+
"risk": [
|
|
33
|
+
"reduced_effort"
|
|
34
|
+
]
|
|
35
|
+
},
|
|
36
|
+
"context": {
|
|
37
|
+
"teacher_support": "TA_supported",
|
|
38
|
+
"class_size": "large",
|
|
39
|
+
"online_or_offline": "offline"
|
|
40
|
+
},
|
|
41
|
+
"scope": {
|
|
42
|
+
"time_range": "2000-2026",
|
|
43
|
+
"geography": "global",
|
|
44
|
+
"study_types": [
|
|
45
|
+
"rct",
|
|
46
|
+
"quasi_experimental"
|
|
47
|
+
]
|
|
48
|
+
},
|
|
49
|
+
"inclusion_criteria": [
|
|
50
|
+
"studies measuring delayed retention or transfer",
|
|
51
|
+
"programming or STEM novices"
|
|
52
|
+
],
|
|
53
|
+
"exclusion_criteria": [
|
|
54
|
+
"no delayed measurement",
|
|
55
|
+
"purely correlational"
|
|
56
|
+
],
|
|
57
|
+
"success_condition": "delayed retention improves or holds while transfer does not decline"
|
|
58
|
+
}
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
{
|
|
2
|
+
"gate_version": "2026-08-13.v1",
|
|
3
|
+
"checked_at": "2026-09-13T03:17:13.784688+00:00",
|
|
4
|
+
"workspace": "examples/spaced-retrieval-practice",
|
|
5
|
+
"items": {
|
|
6
|
+
"research_frame_valid": {
|
|
7
|
+
"title": "Research Frame valid",
|
|
8
|
+
"status": "pass",
|
|
9
|
+
"detail": "frame.json valid (question=在大学的程序设计入门课程中,是否应该用间隔重复与检索练习替代传统的集中式复习?这对学生的长期知识保持是否有效?)",
|
|
10
|
+
"critical": true,
|
|
11
|
+
"blocks_high": true
|
|
12
|
+
},
|
|
13
|
+
"sources_valid": {
|
|
14
|
+
"title": "Sources valid",
|
|
15
|
+
"status": "pass",
|
|
16
|
+
"detail": "7 source record(s) schema-valid",
|
|
17
|
+
"critical": true,
|
|
18
|
+
"blocks_high": true
|
|
19
|
+
},
|
|
20
|
+
"evidence_schema_valid": {
|
|
21
|
+
"title": "Evidence Schema valid",
|
|
22
|
+
"status": "pass",
|
|
23
|
+
"detail": "6 evidence record(s) schema-valid",
|
|
24
|
+
"critical": true,
|
|
25
|
+
"blocks_high": true
|
|
26
|
+
},
|
|
27
|
+
"source_dedupe": {
|
|
28
|
+
"title": "Source dedupe",
|
|
29
|
+
"status": "pass",
|
|
30
|
+
"detail": "7 unique source(s), no duplicates",
|
|
31
|
+
"critical": true,
|
|
32
|
+
"blocks_high": true
|
|
33
|
+
},
|
|
34
|
+
"counter_evidence_search": {
|
|
35
|
+
"title": "Counter-evidence search",
|
|
36
|
+
"status": "pass",
|
|
37
|
+
"detail": "search_performed=true; 9/9 checks run; findings=2; contradictory_evidence_found=False",
|
|
38
|
+
"critical": true,
|
|
39
|
+
"blocks_high": true
|
|
40
|
+
},
|
|
41
|
+
"methodology_audit": {
|
|
42
|
+
"title": "Methodology audit",
|
|
43
|
+
"status": "pass",
|
|
44
|
+
"detail": "methodology verdict=PASS; task/learning separated",
|
|
45
|
+
"critical": true,
|
|
46
|
+
"blocks_high": true
|
|
47
|
+
},
|
|
48
|
+
"claim_evidence_audit": {
|
|
49
|
+
"title": "Claim-Evidence Audit",
|
|
50
|
+
"status": "pass",
|
|
51
|
+
"detail": "all verdict claims bind to existing evidence with consistent categories",
|
|
52
|
+
"critical": true,
|
|
53
|
+
"blocks_high": false
|
|
54
|
+
},
|
|
55
|
+
"outcome_mapping": {
|
|
56
|
+
"title": "Outcome mapping",
|
|
57
|
+
"status": "warn",
|
|
58
|
+
"detail": "outcome keys known; frame-declared outcomes without evidence: independent_problem_solving, knowledge_gain, reduced_effort",
|
|
59
|
+
"critical": false,
|
|
60
|
+
"blocks_high": false
|
|
61
|
+
},
|
|
62
|
+
"scope_calibration": {
|
|
63
|
+
"title": "Scope calibration",
|
|
64
|
+
"status": "pass",
|
|
65
|
+
"detail": "claims bounded: can=2, cannot=2, exceeds_boundary=0",
|
|
66
|
+
"critical": false,
|
|
67
|
+
"blocks_high": true
|
|
68
|
+
},
|
|
69
|
+
"independent_study_count": {
|
|
70
|
+
"title": "Independent study-sample count",
|
|
71
|
+
"status": "pass",
|
|
72
|
+
"detail": "independent studies=6, samples=6",
|
|
73
|
+
"critical": true,
|
|
74
|
+
"blocks_high": true
|
|
75
|
+
},
|
|
76
|
+
"deterministic_confidence": {
|
|
77
|
+
"title": "Deterministic confidence",
|
|
78
|
+
"status": "pass",
|
|
79
|
+
"detail": "deterministic confidence=High (score=0.893, policy=2026-08-12.v3)",
|
|
80
|
+
"critical": true,
|
|
81
|
+
"blocks_high": true
|
|
82
|
+
},
|
|
83
|
+
"decision_action_consistency": {
|
|
84
|
+
"title": "Decision action consistency",
|
|
85
|
+
"status": "pass",
|
|
86
|
+
"detail": "ADOPT is supported: High confidence with direct primary-outcome evidence",
|
|
87
|
+
"critical": true,
|
|
88
|
+
"blocks_high": false
|
|
89
|
+
}
|
|
90
|
+
},
|
|
91
|
+
"passed": true,
|
|
92
|
+
"critical_failures": [],
|
|
93
|
+
"high_confidence_allowed": true,
|
|
94
|
+
"max_confidence": "High",
|
|
95
|
+
"enforcement": {
|
|
96
|
+
"rule": "confidence capped at {max}; High confidence requires a fully passing gate and >= 2 independent studies",
|
|
97
|
+
"max_confidence": "High",
|
|
98
|
+
"requires_action_change": false,
|
|
99
|
+
"action_override": null
|
|
100
|
+
}
|
|
101
|
+
}
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
{
|
|
2
|
+
"target": "overall",
|
|
3
|
+
"verdict": "PASS",
|
|
4
|
+
"audit_items": {
|
|
5
|
+
"control_group": {
|
|
6
|
+
"status": "met",
|
|
7
|
+
"note": "由本轮真实检索到的证据集评估得出"
|
|
8
|
+
},
|
|
9
|
+
"randomization": {
|
|
10
|
+
"status": "partial",
|
|
11
|
+
"note": "由本轮真实检索到的证据集评估得出"
|
|
12
|
+
},
|
|
13
|
+
"pre_test": {
|
|
14
|
+
"status": "partial",
|
|
15
|
+
"note": "由本轮真实检索到的证据集评估得出"
|
|
16
|
+
},
|
|
17
|
+
"post_test": {
|
|
18
|
+
"status": "met",
|
|
19
|
+
"note": "由本轮真实检索到的证据集评估得出"
|
|
20
|
+
},
|
|
21
|
+
"retention_test": {
|
|
22
|
+
"status": "met",
|
|
23
|
+
"note": "由本轮真实检索到的证据集评估得出"
|
|
24
|
+
},
|
|
25
|
+
"transfer_test": {
|
|
26
|
+
"status": "partial",
|
|
27
|
+
"note": "由本轮真实检索到的证据集评估得出"
|
|
28
|
+
},
|
|
29
|
+
"sample_bias": {
|
|
30
|
+
"status": "partial",
|
|
31
|
+
"note": "由本轮真实检索到的证据集评估得出"
|
|
32
|
+
},
|
|
33
|
+
"self_selection": {
|
|
34
|
+
"status": "partial",
|
|
35
|
+
"note": "由本轮真实检索到的证据集评估得出"
|
|
36
|
+
},
|
|
37
|
+
"measurement_validity": {
|
|
38
|
+
"status": "met",
|
|
39
|
+
"note": "由本轮真实检索到的证据集评估得出"
|
|
40
|
+
},
|
|
41
|
+
"confounders": {
|
|
42
|
+
"status": "partial",
|
|
43
|
+
"note": "由本轮真实检索到的证据集评估得出"
|
|
44
|
+
},
|
|
45
|
+
"instructor_effect": {
|
|
46
|
+
"status": "missing",
|
|
47
|
+
"note": "由本轮真实检索到的证据集评估得出"
|
|
48
|
+
},
|
|
49
|
+
"novelty_effect": {
|
|
50
|
+
"status": "missing",
|
|
51
|
+
"note": "由本轮真实检索到的证据集评估得出"
|
|
52
|
+
},
|
|
53
|
+
"tool_version_effect": {
|
|
54
|
+
"status": "not_applicable",
|
|
55
|
+
"note": "由本轮真实检索到的证据集评估得出"
|
|
56
|
+
},
|
|
57
|
+
"ai_usage_policy": {
|
|
58
|
+
"status": "not_applicable",
|
|
59
|
+
"note": "由本轮真实检索到的证据集评估得出"
|
|
60
|
+
},
|
|
61
|
+
"dropout": {
|
|
62
|
+
"status": "missing",
|
|
63
|
+
"note": "由本轮真实检索到的证据集评估得出"
|
|
64
|
+
}
|
|
65
|
+
},
|
|
66
|
+
"task_vs_learning_guard": {
|
|
67
|
+
"measured_construct": "delayed retention and transfer, not immediate task speed",
|
|
68
|
+
"equates_task_with_learning": false,
|
|
69
|
+
"note": "本证据集以延迟保持与迁移为主要结果,未把即时练习表现等同于学习。"
|
|
70
|
+
},
|
|
71
|
+
"limitations": [
|
|
72
|
+
"多数研究来自心理学实验室材料,程序设计课程的现场证据相对有限",
|
|
73
|
+
"间隔长度与练习形式在研究中差异较大"
|
|
74
|
+
],
|
|
75
|
+
"suggestions": [
|
|
76
|
+
"在课程内固定每周短检索练习并统一延迟后测口径"
|
|
77
|
+
]
|
|
78
|
+
}
|