eduevidence 5.2.0 → 6.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +105 -0
- package/README.md +142 -75
- package/README.zh-CN.md +73 -30
- package/SKILL.md +397 -131
- package/agents/openai.yaml +4 -0
- package/assets/readme/controlled-execution.svg +34 -0
- package/assets/readme/landing-tour.gif +0 -0
- package/assets/readme/logo.png +0 -0
- package/assets/readme/research-workflow.svg +56 -0
- package/assets/readme/studio-graph.png +0 -0
- package/assets/readme/studio-overview.png +0 -0
- package/assets/readme/studio-reports.png +0 -0
- package/assets/readme/studio-tour.gif +0 -0
- package/autoevolve/config.yaml +17 -0
- package/autoevolve/program.md +25 -0
- package/autoevolve/protected.manifest.yaml +34 -0
- package/benchmarks/adversarial/cases.jsonl +7 -0
- package/benchmarks/evidence-library.json +5268 -0
- package/benchmarks/partitions.json +8 -0
- package/bin/eduevidence.js +2 -1
- package/docs/architecture.md +496 -0
- package/docs/autoresearch-evolution-plan.md +2903 -0
- package/docs/autoresearch-implementation-status.md +101 -0
- package/docs/demo-storyboard.md +20 -0
- package/docs/demo-workplace-ai.md +92 -0
- package/docs/demo.md +32 -0
- package/docs/install-guide.md +150 -0
- package/docs/orchestration-role-model.md +1254 -0
- package/docs/release-closeout/README.md +17 -0
- package/docs/release-closeout/frontend-acceptance.md +23 -0
- package/docs/release-closeout/issues.md +19 -0
- package/docs/release-closeout/verification.md +28 -0
- package/docs/release-contract.md +108 -0
- package/docs/research-studio-guide.zh-CN.md +166 -0
- package/docs/sciverse-api.md +125 -0
- package/eduevidence_cli.py +29 -13
- package/engine/_resources.py +13 -0
- package/engine/autoevolve/__init__.py +3 -0
- package/engine/autoevolve/agent_view.py +167 -0
- package/engine/autoevolve/core.py +357 -0
- package/engine/autoevolve/events.py +11 -0
- package/engine/autoevolve/git_workspace.py +77 -0
- package/engine/autoevolve/projection.py +23 -0
- package/engine/autoevolve/runner.py +413 -0
- package/engine/autoevolve/trust.py +146 -0
- package/engine/autoresearch/__init__.py +6 -0
- package/engine/autoresearch/commit.py +132 -0
- package/engine/autoresearch/contracts.py +126 -0
- package/engine/autoresearch/controller.py +207 -0
- package/engine/autoresearch/events.py +12 -0
- package/engine/autoresearch/gap_priority.py +168 -0
- package/engine/autoresearch/projection.py +30 -0
- package/engine/autoresearch/research_memory.py +59 -0
- package/engine/autoresearch/saturation.py +91 -0
- package/engine/briefs.py +2 -1
- package/engine/capabilities.py +1 -0
- package/engine/contracts.py +3 -1
- package/engine/decision_policy.py +96 -0
- package/engine/evidence_graph.py +14 -10
- package/engine/evidencecore.py +7 -5
- package/engine/gaps.py +132 -73
- package/engine/ids.py +2 -0
- package/engine/judge_pack.py +65 -0
- package/engine/library.py +6 -2
- package/engine/library_builtin.py +3 -1
- package/engine/living.py +36 -5
- package/engine/meta_synthesis.py +3 -1
- package/engine/migration.py +88 -3
- package/engine/orchestration.py +460 -0
- package/engine/paths.py +2 -0
- package/engine/pilot.py +36 -33
- package/engine/project.py +2 -2
- package/engine/research_service.py +113 -0
- package/engine/studio_read_model.py +400 -0
- package/engine/taxonomy.py +211 -0
- package/engine/tribunal.py +44 -33
- package/engine/update.py +1 -0
- package/engine/versions.py +1 -1
- package/engine/worker_result.py +109 -0
- package/engine/workflows.py +70 -0
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +2934 -0
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
- package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
- package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
- package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
- package/examples/ai-coding-assistant-evidence/frame.json +48 -0
- package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
- package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
- package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
- package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +230 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/result.json +1457 -0
- package/examples/ai-coding-assistant-evidence/result.zh.json +1457 -0
- package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
- package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +107 -0
- package/examples/spaced-retrieval-practice/applicability.json +14 -0
- package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
- package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
- package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
- package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
- package/examples/spaced-retrieval-practice/frame.json +58 -0
- package/examples/spaced-retrieval-practice/gate_report.json +101 -0
- package/examples/spaced-retrieval-practice/methodology.json +78 -0
- package/examples/spaced-retrieval-practice/report_spec.json +212 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/result.json +942 -0
- package/examples/spaced-retrieval-practice/result.zh.json +942 -0
- package/examples/spaced-retrieval-practice/skeptic.json +70 -0
- package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
- package/examples/spaced-retrieval-practice/verdict.json +93 -0
- package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -0
- package/examples/workplace-ai-assistant/evaluation.json +19 -0
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
- package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
- package/examples/workplace-ai-assistant/final_verdict.json +78 -0
- package/examples/workplace-ai-assistant/frame.json +41 -0
- package/examples/workplace-ai-assistant/gate_report.json +101 -0
- package/examples/workplace-ai-assistant/intervention.json +27 -0
- package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
- package/examples/workplace-ai-assistant/methodology.json +60 -0
- package/examples/workplace-ai-assistant/report_spec.json +224 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/result.json +615 -0
- package/examples/workplace-ai-assistant/result.zh.json +615 -0
- package/examples/workplace-ai-assistant/search_log.json +19 -0
- package/examples/workplace-ai-assistant/skeptic.json +72 -0
- package/examples/workplace-ai-assistant/sources.jsonl +3 -0
- package/examples/workplace-ai-assistant/validation_result.json +9 -0
- package/examples/workplace-ai-assistant/verdict.json +78 -0
- package/install.sh +7 -7
- package/integrations/agent_mcp.py +2 -2
- package/integrations/orchestration_dispatch.py +146 -0
- package/package.json +46 -3
- package/pyproject.toml +14 -22
- package/references/autoresearch.md +30 -0
- package/references/evaluation-policy.md +24 -0
- package/references/orchestration.md +22 -0
- package/references/report-copy-style.md +67 -0
- package/references/retrieval-compliance.md +75 -0
- package/references/retrieval-protocol.md +20 -0
- package/references/scientific-invariants.md +19 -0
- package/retrieval/audit.py +178 -0
- package/retrieval/fetch.py +96 -0
- package/retrieval/sciverse.py +398 -0
- package/retrieval/search.py +47 -7
- package/schemas/applicability.schema.json +94 -0
- package/schemas/chart-spec.schema.json +10 -3
- package/schemas/evidence.schema.json +316 -43
- package/schemas/fetch-result.schema.json +2 -1
- package/schemas/intervention.schema.json +106 -21
- package/schemas/report-result.schema.json +12 -4
- package/schemas/report-spec.schema.json +98 -100
- package/schemas/skeptic.schema.json +86 -0
- package/schemas/source.schema.json +21 -2
- package/schemas/v2/finding.schema.json +5 -1
- package/schemas/v2/methodology-audit.schema.json +5 -1
- package/schemas/v2/outcome.schema.json +28 -5
- package/schemas/v2/project.schema.json +2 -2
- package/schemas/v2/run.schema.json +1 -1
- package/schemas/v2/study.schema.json +5 -1
- package/schemas/vNext/autoevolve-session.schema.json +34 -0
- package/schemas/vNext/eval-snapshot.schema.json +77 -0
- package/schemas/vNext/execution-plan.schema.json +50 -0
- package/schemas/vNext/gap-priority.schema.json +54 -0
- package/schemas/vNext/negative-search-record.schema.json +68 -0
- package/schemas/vNext/research-iteration.schema.json +87 -0
- package/schemas/vNext/research-strategy.schema.json +62 -0
- package/schemas/vNext/skill-experiment.schema.json +90 -0
- package/schemas/vNext/task-spec.schema.json +156 -0
- package/schemas/vNext/worker-result.schema.json +60 -0
- package/schemas/verdict.schema.json +164 -28
- package/scripts/benchmark_judge.py +2 -2
- package/scripts/benchmark_v3.py +26 -43
- package/scripts/build_esl_artifacts.py +4 -4
- package/scripts/build_evidence_library.py +2 -2
- package/scripts/build_gh_pages.py +98 -0
- package/scripts/build_readme_diagrams.py +72 -0
- package/scripts/build_report_variants.py +101 -0
- package/scripts/build_result.py +74 -9
- package/scripts/check_autoresearch_invariants.py +95 -0
- package/scripts/check_package_parity.py +85 -0
- package/scripts/check_protocol_alignment.py +375 -0
- package/scripts/check_versioned_schemas.py +254 -0
- package/scripts/claim_audit.py +13 -8
- package/scripts/compute_confidence.py +10 -0
- package/scripts/daily_evolve.py +30 -0
- package/scripts/dashboard_server.py +130 -101
- package/scripts/did_regression.py +17 -32
- package/scripts/enrich_projects_human_and_lieflat.py +1 -1
- package/scripts/evidence_score.py +5 -2
- package/scripts/generate_metrics.py +4 -3
- package/scripts/generate_new_projects.py +5 -5
- package/scripts/orchestrator.py +286 -36
- package/scripts/pre_verdict_gate.py +224 -26
- package/scripts/quickstart.py +18 -2
- package/scripts/rebake_all_5themes.py +1 -2
- package/scripts/research_auto_cli.py +475 -0
- package/scripts/run_workspace.py +24 -8
- package/scripts/search_provenance.py +64 -0
- package/scripts/serve_web.py +9 -10
- package/scripts/skill_lint.py +1 -1
- package/scripts/skill_payload.py +81 -0
- package/scripts/test_adversarial_empirical.py +26 -19
- package/scripts/validate_schema.py +46 -2
- package/scripts/vnext_cli.py +133 -0
- package/setup.py +12 -0
- package/skill/agents/evaluation-designer.md +20 -4
- package/skill/agents/evidence-analyst.md +19 -3
- package/skill/agents/evidence-judge.md +50 -2
- package/skill/agents/evidence-retriever.md +20 -3
- package/skill/agents/intervention-designer.md +20 -4
- package/skill/agents/method-reviewer.md +18 -2
- package/skill/agents/{education-planner.md → research-planner.md} +19 -3
- package/skill/agents/skeptic.md +18 -2
- package/skill/roles/registry.yaml +45 -0
- package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
- package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
- package/skill/sub-skills/data-analysis/SKILL.md +34 -15
- package/skill/sub-skills/ethics-review/SKILL.md +33 -10
- package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
- package/skill/sub-skills/evidence-review/SKILL.md +31 -12
- package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
- package/skill/sub-skills/literature-review/SKILL.md +35 -14
- package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
- package/skill/sub-skills/report-generation/SKILL.md +40 -6
- package/skill/sub-skills/research-planning/SKILL.md +41 -14
- package/skill/sub-skills/study-design/SKILL.md +30 -9
- package/skill/task-briefs/adjudicate.md +32 -7
- package/skill/task-briefs/applicability.md +38 -0
- package/skill/task-briefs/audit.md +32 -7
- package/skill/task-briefs/challenge.md +34 -5
- package/skill/task-briefs/evaluate.md +30 -5
- package/skill/task-briefs/extract.md +31 -8
- package/skill/task-briefs/frame.md +39 -10
- package/skill/task-briefs/intervene.md +32 -6
- package/skill/task-briefs/present.md +32 -8
- package/skill/task-briefs/projection.md +37 -0
- package/skill/task-briefs/retrieve.md +36 -6
- package/skill/workflows/decision-and-pilot.md +85 -0
- package/skill/workflows/evaluate-and-update.md +93 -0
- package/skill/workflows/evidence-review.md +117 -0
- package/visualization/eduevidence-report/assets/base.css +2 -2
- package/visualization/eduevidence-report/assets/reader.css +752 -0
- package/visualization/eduevidence-report/assets/reader.js +132 -0
- package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
- package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
- package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
- package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
- package/visualization/eduevidence-report/scripts/build_report.py +561 -121
- package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +371 -136
- package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
- package/visualization/eduevidence-report/themes/academic.css +1 -1
- package/visualization/eduevidence-report/themes/claude.css +1 -1
- package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
- package/visualization/eduevidence-report/themes/datalab.css +2 -2
- package/visualization/eduevidence-report/themes/presentation.css +2 -2
- package/web/README.md +18 -0
- package/web/architecture.html +14885 -0
- package/web/index.html +53 -0
- package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
- package/web/studio/assets/index-B8tkF44Q.css +1 -0
- package/web/studio/assets/index-CQ6Keoyc.js +230 -0
- package/web/studio/config.json +1 -0
- package/web/studio/index.html +14 -0
- package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
- package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/bias.cpython-312.pyc +0 -0
- package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
- package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
- package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
- package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
- package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
- package/engine/__pycache__/events.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
- package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
- package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
- package/engine/__pycache__/ids.cpython-312.pyc +0 -0
- package/engine/__pycache__/library.cpython-312.pyc +0 -0
- package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
- package/engine/__pycache__/living.cpython-312.pyc +0 -0
- package/engine/__pycache__/log.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/migration.cpython-312.pyc +0 -0
- package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
- package/engine/__pycache__/paths.cpython-312.pyc +0 -0
- package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
- package/engine/__pycache__/planner.cpython-312.pyc +0 -0
- package/engine/__pycache__/project.cpython-312.pyc +0 -0
- package/engine/__pycache__/projections.cpython-312.pyc +0 -0
- package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
- package/engine/__pycache__/run.cpython-312.pyc +0 -0
- package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
- package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
- package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
- package/engine/__pycache__/update.cpython-312.pyc +0 -0
- package/engine/__pycache__/versions.cpython-312.pyc +0 -0
- package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
- package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
- package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
- package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
- package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
- package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
- package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
- package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
- package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
- package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
- package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
- package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
- package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
- package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
- package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
- package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
|
@@ -0,0 +1,413 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import shlex
|
|
6
|
+
import shutil
|
|
7
|
+
import subprocess
|
|
8
|
+
import time
|
|
9
|
+
from datetime import datetime, timezone
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from .agent_view import AgentMutationView
|
|
14
|
+
from .core import (
|
|
15
|
+
DailyProfile,
|
|
16
|
+
EvalSnapshot,
|
|
17
|
+
ExperimentLog,
|
|
18
|
+
PlateauTracker,
|
|
19
|
+
ProtectedManifest,
|
|
20
|
+
SkillExperiment,
|
|
21
|
+
promote,
|
|
22
|
+
)
|
|
23
|
+
from .git_workspace import GitWorkspace
|
|
24
|
+
from .trust import AgentIsolation, compute_eval_suite_hash
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _run_json(command: str, cwd: Path, env=None) -> dict:
|
|
28
|
+
argv = shlex.split(command)
|
|
29
|
+
if not argv:
|
|
30
|
+
raise ValueError("empty command")
|
|
31
|
+
completed = subprocess.run(
|
|
32
|
+
argv,
|
|
33
|
+
cwd=cwd,
|
|
34
|
+
check=True,
|
|
35
|
+
text=True,
|
|
36
|
+
capture_output=True,
|
|
37
|
+
env=env,
|
|
38
|
+
)
|
|
39
|
+
value = json.loads(completed.stdout)
|
|
40
|
+
if not isinstance(value, dict):
|
|
41
|
+
raise ValueError("command stdout must be one JSON object")
|
|
42
|
+
return value
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _trusted_eval_snapshot(
|
|
46
|
+
payload: dict[str, Any],
|
|
47
|
+
*,
|
|
48
|
+
suite_hash: str,
|
|
49
|
+
isolation_verified: bool,
|
|
50
|
+
) -> EvalSnapshot:
|
|
51
|
+
"""Build an EvalSnapshot while ignoring evaluator self-attestation.
|
|
52
|
+
|
|
53
|
+
Evaluation suite identity and holdout-isolation authority are runner-owned.
|
|
54
|
+
Any values with those names emitted by the evaluator are overwritten.
|
|
55
|
+
"""
|
|
56
|
+
trusted = dict(payload)
|
|
57
|
+
trusted["eval_suite_hash"] = suite_hash
|
|
58
|
+
trusted["holdout_isolation_verified"] = bool(isolation_verified)
|
|
59
|
+
return EvalSnapshot(**trusted)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _read_jsonl(path: Path, limit: int = 25) -> list[dict[str, Any]]:
|
|
63
|
+
if not path.is_file():
|
|
64
|
+
return []
|
|
65
|
+
rows = [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines() if line.strip()]
|
|
66
|
+
return rows[-limit:]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _session_root(repo: Path, tag: str) -> Path:
|
|
70
|
+
configured = os.environ.get("EDUEVIDENCE_AUTOEVOLVE_STATE_DIR")
|
|
71
|
+
base = (
|
|
72
|
+
Path(configured).expanduser().resolve()
|
|
73
|
+
if configured
|
|
74
|
+
else repo.parent / ".eduevidence-autoevolve-state"
|
|
75
|
+
)
|
|
76
|
+
root = base / tag
|
|
77
|
+
if root.exists():
|
|
78
|
+
raise FileExistsError(f"autoevolve session state already exists: {root}")
|
|
79
|
+
root.mkdir(parents=True)
|
|
80
|
+
return root
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _session_context(workspace: GitWorkspace, state_root: Path) -> dict[str, Any]:
|
|
84
|
+
repo_history = _read_jsonl(workspace.path / "autoevolve" / "experiments.jsonl")
|
|
85
|
+
current_history = _read_jsonl(state_root / "experiments.jsonl")
|
|
86
|
+
return {
|
|
87
|
+
"branch": workspace.branch,
|
|
88
|
+
"parent_revision": workspace.head(),
|
|
89
|
+
"prior_experiments": (repo_history + current_history)[-25:],
|
|
90
|
+
"holdout_policy": (
|
|
91
|
+
"This mutation view intentionally contains DEV benchmark material only. "
|
|
92
|
+
"Do not seek or infer hidden HOLDOUT/adversarial cases."
|
|
93
|
+
),
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _append_file(source: Path, target: Path, *, skip_first_line: bool = False) -> None:
|
|
98
|
+
if not source.is_file():
|
|
99
|
+
return
|
|
100
|
+
lines = source.read_text(encoding="utf-8").splitlines()
|
|
101
|
+
if skip_first_line and lines:
|
|
102
|
+
lines = lines[1:]
|
|
103
|
+
if not lines:
|
|
104
|
+
return
|
|
105
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
106
|
+
with target.open("a", encoding="utf-8") as handle:
|
|
107
|
+
for line in lines:
|
|
108
|
+
handle.write(line + "\n")
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _copy_public_session_files(state_root: Path, run_target: Path) -> None:
|
|
112
|
+
"""Persist only structured, non-candidate session records to Git."""
|
|
113
|
+
run_target.mkdir(parents=True, exist_ok=False)
|
|
114
|
+
for name in ("results.tsv", "experiments.jsonl", "daily-report.json"):
|
|
115
|
+
source = state_root / name
|
|
116
|
+
if source.is_file():
|
|
117
|
+
shutil.copy2(source, run_target / name)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _export_session(
|
|
121
|
+
workspace: GitWorkspace,
|
|
122
|
+
state_root: Path,
|
|
123
|
+
*,
|
|
124
|
+
tag: str,
|
|
125
|
+
best_experiment_id: str | None,
|
|
126
|
+
best_commit: str | None,
|
|
127
|
+
) -> str:
|
|
128
|
+
"""Persist audit memory only after candidate evaluation has finished.
|
|
129
|
+
|
|
130
|
+
Candidate patches/files remain local in the external state directory. They
|
|
131
|
+
are deliberately excluded from automatic branch persistence because an
|
|
132
|
+
external agent could have written sensitive/transient material into them.
|
|
133
|
+
"""
|
|
134
|
+
run_target = workspace.path / "autoevolve" / "runs" / tag
|
|
135
|
+
if run_target.exists():
|
|
136
|
+
raise FileExistsError(f"session run target exists: {run_target}")
|
|
137
|
+
run_target.parent.mkdir(parents=True, exist_ok=True)
|
|
138
|
+
_copy_public_session_files(state_root, run_target)
|
|
139
|
+
|
|
140
|
+
_append_file(
|
|
141
|
+
state_root / "results.tsv",
|
|
142
|
+
workspace.path / "autoevolve" / "results.tsv",
|
|
143
|
+
skip_first_line=True,
|
|
144
|
+
)
|
|
145
|
+
_append_file(
|
|
146
|
+
state_root / "experiments.jsonl",
|
|
147
|
+
workspace.path / "autoevolve" / "experiments.jsonl",
|
|
148
|
+
)
|
|
149
|
+
if best_experiment_id:
|
|
150
|
+
(workspace.path / "autoevolve" / "best.json").write_text(
|
|
151
|
+
json.dumps(
|
|
152
|
+
{
|
|
153
|
+
"best_experiment_id": best_experiment_id,
|
|
154
|
+
"candidate_commit": best_commit,
|
|
155
|
+
"session": tag,
|
|
156
|
+
},
|
|
157
|
+
indent=2,
|
|
158
|
+
)
|
|
159
|
+
+ "\n",
|
|
160
|
+
encoding="utf-8",
|
|
161
|
+
)
|
|
162
|
+
return workspace.commit(f"autoevolve: record session {tag}")
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
class DailyEvolutionRunner:
|
|
166
|
+
"""Run bounded branch-only autoresearch against an external approved agent.
|
|
167
|
+
|
|
168
|
+
Candidate mutation occurs in a sanitized view; evaluator execution occurs
|
|
169
|
+
in the full isolated worktree. Session logs live outside the candidate
|
|
170
|
+
worktree during experimentation so reject/revert can never erase history or
|
|
171
|
+
accidentally commit evaluator logs as part of a candidate code change.
|
|
172
|
+
"""
|
|
173
|
+
|
|
174
|
+
def __init__(self, repo: str | Path, *, profile: DailyProfile | None = None):
|
|
175
|
+
self.repo = Path(repo).resolve()
|
|
176
|
+
self.profile = profile or DailyProfile()
|
|
177
|
+
self.profile.validate()
|
|
178
|
+
|
|
179
|
+
def run(
|
|
180
|
+
self,
|
|
181
|
+
*,
|
|
182
|
+
agent_command: str,
|
|
183
|
+
eval_command: str,
|
|
184
|
+
run_tag: str | None = None,
|
|
185
|
+
push_branch: bool = False,
|
|
186
|
+
max_retests: int = 2,
|
|
187
|
+
) -> dict:
|
|
188
|
+
if max_retests < 0 or max_retests > 5:
|
|
189
|
+
raise ValueError("max_retests must be 0..5")
|
|
190
|
+
tag = run_tag or datetime.now(timezone.utc).strftime("%Y%m%d-%H%M%S")
|
|
191
|
+
workspace = GitWorkspace.create(self.repo, tag)
|
|
192
|
+
state_root = _session_root(self.repo, tag)
|
|
193
|
+
log = ExperimentLog(state_root)
|
|
194
|
+
manifest = ProtectedManifest.from_repo(workspace.path)
|
|
195
|
+
plateau = PlateauTracker()
|
|
196
|
+
isolation = AgentIsolation.from_environment()
|
|
197
|
+
trusted_suite_hash = compute_eval_suite_hash(workspace.path)
|
|
198
|
+
baseline = _trusted_eval_snapshot(
|
|
199
|
+
_run_json(eval_command, workspace.path),
|
|
200
|
+
suite_hash=trusted_suite_hash,
|
|
201
|
+
isolation_verified=isolation.verified,
|
|
202
|
+
)
|
|
203
|
+
statuses: list[str] = []
|
|
204
|
+
spent = float(baseline.cost)
|
|
205
|
+
best = None
|
|
206
|
+
best_commit = None
|
|
207
|
+
stop_reason = "completed"
|
|
208
|
+
started = time.monotonic()
|
|
209
|
+
|
|
210
|
+
for index in range(1, self.profile.max_experiments + 1):
|
|
211
|
+
if spent >= self.profile.max_cost_usd:
|
|
212
|
+
stop_reason = "cost_budget_exhausted"
|
|
213
|
+
break
|
|
214
|
+
if (time.monotonic() - started) / 60 >= self.profile.max_wall_minutes:
|
|
215
|
+
stop_reason = "wall_time_exhausted"
|
|
216
|
+
break
|
|
217
|
+
|
|
218
|
+
experiment_id = f"EXP-{index:04d}"
|
|
219
|
+
parent_revision = workspace.head()
|
|
220
|
+
protected_before = manifest.hash_tree(workspace.path)
|
|
221
|
+
view = AgentMutationView.create(
|
|
222
|
+
workspace.path,
|
|
223
|
+
session_context=_session_context(workspace, state_root),
|
|
224
|
+
)
|
|
225
|
+
candidate = None
|
|
226
|
+
attempted: list[str] = []
|
|
227
|
+
budget_hit = False
|
|
228
|
+
try:
|
|
229
|
+
env = os.environ.copy()
|
|
230
|
+
env.update(
|
|
231
|
+
{
|
|
232
|
+
"EDUEVIDENCE_EXPERIMENT_ID": experiment_id,
|
|
233
|
+
"EDUEVIDENCE_PROGRAM": str(view.path / "autoevolve" / "program.md"),
|
|
234
|
+
"EDUEVIDENCE_SESSION_CONTEXT": str(view.path / "autoevolve" / "session-context.json"),
|
|
235
|
+
"EDUEVIDENCE_HOLDOUT_ACCESS": "FORBIDDEN",
|
|
236
|
+
}
|
|
237
|
+
)
|
|
238
|
+
isolated_command, agent_env = isolation.wrap_command(
|
|
239
|
+
agent_command,
|
|
240
|
+
view.path,
|
|
241
|
+
env,
|
|
242
|
+
)
|
|
243
|
+
proposal = _run_json(isolated_command, view.path, agent_env)
|
|
244
|
+
hypothesis = str(proposal.get("hypothesis", "")).strip()
|
|
245
|
+
if not hypothesis:
|
|
246
|
+
raise ValueError("empty hypothesis")
|
|
247
|
+
proposal_cost = float(proposal.get("cost_usd", 0) or 0)
|
|
248
|
+
if proposal_cost < 0:
|
|
249
|
+
raise ValueError("agent cost_usd cannot be negative")
|
|
250
|
+
spent += proposal_cost
|
|
251
|
+
attempted = view.changed_files()
|
|
252
|
+
proposal_tiers = tuple(proposal.get("mutation_scope") or self.profile.mutation_tiers)
|
|
253
|
+
if not set(proposal_tiers).issubset(set(self.profile.mutation_tiers)):
|
|
254
|
+
status, reason = "INVALID", "agent requested a mutation tier not allowed by session profile"
|
|
255
|
+
else:
|
|
256
|
+
protected_ok, protected_bad = manifest.validate_changes(attempted)
|
|
257
|
+
scope_ok, scope_bad = manifest.validate_mutation_scope(
|
|
258
|
+
attempted,
|
|
259
|
+
mutation_tiers=self.profile.mutation_tiers,
|
|
260
|
+
allow_controlled=self.profile.allow_controlled,
|
|
261
|
+
)
|
|
262
|
+
if not attempted:
|
|
263
|
+
status, reason = "REJECT", "agent made no change"
|
|
264
|
+
elif not protected_ok:
|
|
265
|
+
status, reason = "INVALID", "protected mutation: " + ",".join(protected_bad)
|
|
266
|
+
elif not scope_ok:
|
|
267
|
+
status, reason = "INVALID", "mutation outside approved tier: " + ",".join(scope_bad)
|
|
268
|
+
elif spent >= self.profile.max_cost_usd:
|
|
269
|
+
status, reason = "REJECT", "session cost budget exhausted before candidate evaluation"
|
|
270
|
+
budget_hit = True
|
|
271
|
+
else:
|
|
272
|
+
view.sync_to(workspace.path, attempted)
|
|
273
|
+
protected_after_sync = manifest.hash_tree(workspace.path)
|
|
274
|
+
if protected_after_sync != protected_before:
|
|
275
|
+
status, reason = "INVALID", "protected tree hash changed"
|
|
276
|
+
elif compute_eval_suite_hash(workspace.path) != trusted_suite_hash:
|
|
277
|
+
status, reason = "INVALID", "trusted evaluation suite changed during experiment"
|
|
278
|
+
else:
|
|
279
|
+
eval_env = os.environ.copy()
|
|
280
|
+
eval_env["EDUEVIDENCE_EXPERIMENT_ID"] = experiment_id
|
|
281
|
+
eval_env["EDUEVIDENCE_EVAL_SUITE_HASH"] = trusted_suite_hash
|
|
282
|
+
for retest_index in range(max_retests + 1):
|
|
283
|
+
eval_env["EDUEVIDENCE_RETEST_INDEX"] = str(retest_index)
|
|
284
|
+
candidate = _trusted_eval_snapshot(
|
|
285
|
+
_run_json(eval_command, workspace.path, eval_env),
|
|
286
|
+
suite_hash=trusted_suite_hash,
|
|
287
|
+
isolation_verified=isolation.verified,
|
|
288
|
+
)
|
|
289
|
+
spent += float(candidate.cost)
|
|
290
|
+
status, reason = promote(baseline, candidate)
|
|
291
|
+
if status != "RETEST":
|
|
292
|
+
break
|
|
293
|
+
if spent >= self.profile.max_cost_usd:
|
|
294
|
+
budget_hit = True
|
|
295
|
+
reason += "; session cost budget exhausted"
|
|
296
|
+
break
|
|
297
|
+
if spent > self.profile.max_cost_usd and status == "KEEP":
|
|
298
|
+
status = "HUMAN_REVIEW"
|
|
299
|
+
reason = "candidate passed quality gates but session cost ceiling was exceeded"
|
|
300
|
+
budget_hit = True
|
|
301
|
+
|
|
302
|
+
protected_after = manifest.hash_tree(workspace.path)
|
|
303
|
+
experiment = SkillExperiment(
|
|
304
|
+
experiment_id=experiment_id,
|
|
305
|
+
session_id=tag,
|
|
306
|
+
parent_skill_revision=parent_revision,
|
|
307
|
+
hypothesis=hypothesis,
|
|
308
|
+
mutation_scope=proposal_tiers,
|
|
309
|
+
changed_files=attempted,
|
|
310
|
+
baseline_eval_id=baseline.eval_id,
|
|
311
|
+
candidate_eval_id=candidate.eval_id if candidate else None,
|
|
312
|
+
protected_hash_before=protected_before,
|
|
313
|
+
protected_hash_after=protected_after,
|
|
314
|
+
status=status,
|
|
315
|
+
promotion_reason=reason,
|
|
316
|
+
complexity_delta=(candidate.complexity - baseline.complexity) if candidate else 0.0,
|
|
317
|
+
)
|
|
318
|
+
if status == "KEEP":
|
|
319
|
+
experiment.candidate_commit = workspace.commit(
|
|
320
|
+
f"experiment: {experiment_id} {hypothesis[:72]}"
|
|
321
|
+
)
|
|
322
|
+
baseline = candidate
|
|
323
|
+
best = experiment_id
|
|
324
|
+
best_commit = experiment.candidate_commit
|
|
325
|
+
else:
|
|
326
|
+
if attempted:
|
|
327
|
+
artifact_dir = state_root / "candidates" / experiment_id
|
|
328
|
+
view.export_changes(artifact_dir / "files", attempted)
|
|
329
|
+
diff = workspace.diff()
|
|
330
|
+
if diff:
|
|
331
|
+
(artifact_dir / "candidate.diff").parent.mkdir(parents=True, exist_ok=True)
|
|
332
|
+
(artifact_dir / "candidate.diff").write_text(diff, encoding="utf-8")
|
|
333
|
+
workspace.restore()
|
|
334
|
+
log.append(experiment, candidate=candidate, description=reason)
|
|
335
|
+
statuses.append(status)
|
|
336
|
+
except Exception as exc:
|
|
337
|
+
if attempted:
|
|
338
|
+
try:
|
|
339
|
+
view.export_changes(state_root / "candidates" / experiment_id / "files", attempted)
|
|
340
|
+
except Exception:
|
|
341
|
+
pass
|
|
342
|
+
workspace.restore()
|
|
343
|
+
experiment = SkillExperiment(
|
|
344
|
+
experiment_id=experiment_id,
|
|
345
|
+
session_id=tag,
|
|
346
|
+
parent_skill_revision=parent_revision,
|
|
347
|
+
hypothesis="invalid-or-crashed",
|
|
348
|
+
mutation_scope=self.profile.mutation_tiers,
|
|
349
|
+
changed_files=attempted,
|
|
350
|
+
baseline_eval_id=baseline.eval_id,
|
|
351
|
+
protected_hash_before=protected_before,
|
|
352
|
+
protected_hash_after=manifest.hash_tree(workspace.path),
|
|
353
|
+
status="CRASH",
|
|
354
|
+
promotion_reason=str(exc),
|
|
355
|
+
)
|
|
356
|
+
log.append(experiment, description=str(exc))
|
|
357
|
+
statuses.append("CRASH")
|
|
358
|
+
finally:
|
|
359
|
+
view.cleanup()
|
|
360
|
+
|
|
361
|
+
if budget_hit:
|
|
362
|
+
stop_reason = "cost_budget_exhausted"
|
|
363
|
+
break
|
|
364
|
+
if plateau.plateau(statuses):
|
|
365
|
+
stop_reason = "plateau"
|
|
366
|
+
break
|
|
367
|
+
|
|
368
|
+
report = {
|
|
369
|
+
"run_tag": tag,
|
|
370
|
+
"branch": workspace.branch,
|
|
371
|
+
"experiments": len(statuses),
|
|
372
|
+
"statuses": statuses,
|
|
373
|
+
"best_experiment_id": best,
|
|
374
|
+
"best_candidate_commit": best_commit,
|
|
375
|
+
"cost": spent,
|
|
376
|
+
"wall_minutes": round((time.monotonic() - started) / 60, 3),
|
|
377
|
+
"plateau": plateau.plateau(statuses),
|
|
378
|
+
"stop_reason": stop_reason,
|
|
379
|
+
"promotion": "branch_only",
|
|
380
|
+
"branch_push_requested": push_branch,
|
|
381
|
+
"branch_pushed": False,
|
|
382
|
+
"mutation_view": (
|
|
383
|
+
"os_container_isolation" if isolation.verified else "dev_only_context_isolation"
|
|
384
|
+
),
|
|
385
|
+
"holdout_isolation_verified": isolation.verified,
|
|
386
|
+
"isolation_provider": isolation.mode,
|
|
387
|
+
"isolation_reason": isolation.reason,
|
|
388
|
+
"eval_suite_hash": trusted_suite_hash,
|
|
389
|
+
"security_note": (
|
|
390
|
+
"automatic KEEP requires runner-owned OS isolation; evaluator self-attestation is ignored"
|
|
391
|
+
),
|
|
392
|
+
"candidate_artifacts": "local session state only; never auto-pushed",
|
|
393
|
+
}
|
|
394
|
+
(state_root / "daily-report.json").write_text(
|
|
395
|
+
json.dumps(report, ensure_ascii=False, indent=2) + "\n",
|
|
396
|
+
encoding="utf-8",
|
|
397
|
+
)
|
|
398
|
+
metadata_commit = _export_session(
|
|
399
|
+
workspace,
|
|
400
|
+
state_root,
|
|
401
|
+
tag=tag,
|
|
402
|
+
best_experiment_id=best,
|
|
403
|
+
best_commit=best_commit,
|
|
404
|
+
)
|
|
405
|
+
report["session_metadata_commit"] = metadata_commit
|
|
406
|
+
if push_branch:
|
|
407
|
+
workspace.push()
|
|
408
|
+
report["branch_pushed"] = True
|
|
409
|
+
(state_root / "daily-report.json").write_text(
|
|
410
|
+
json.dumps(report, ensure_ascii=False, indent=2) + "\n",
|
|
411
|
+
encoding="utf-8",
|
|
412
|
+
)
|
|
413
|
+
return report
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import os
|
|
5
|
+
import shlex
|
|
6
|
+
import shutil
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Mapping
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
EVAL_SUITE_PATTERNS = (
|
|
13
|
+
"benchmarks/evaluator/**",
|
|
14
|
+
"benchmarks/holdout/**",
|
|
15
|
+
"benchmarks/adversarial/**",
|
|
16
|
+
"benchmarks/annotations/**",
|
|
17
|
+
"benchmarks/partitions.json",
|
|
18
|
+
"benchmarks/questions.jsonl",
|
|
19
|
+
"references/scientific-invariants.md",
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _matches(rel: str, pattern: str) -> bool:
|
|
24
|
+
import fnmatch
|
|
25
|
+
|
|
26
|
+
return fnmatch.fnmatch(rel, pattern) or (
|
|
27
|
+
pattern.endswith("/**") and rel.startswith(pattern[:-3])
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def compute_eval_suite_hash(root: str | Path) -> str:
|
|
32
|
+
"""Hash the trusted evaluation suite from repository files.
|
|
33
|
+
|
|
34
|
+
The evaluator does not supply this identity. The runner computes it from
|
|
35
|
+
protected evaluator/partition/gold/adversarial inputs before promotion.
|
|
36
|
+
"""
|
|
37
|
+
root = Path(root).resolve()
|
|
38
|
+
digest = hashlib.sha256()
|
|
39
|
+
files: list[Path] = []
|
|
40
|
+
for path in root.rglob("*"):
|
|
41
|
+
if not path.is_file() or path.is_symlink():
|
|
42
|
+
continue
|
|
43
|
+
rel = path.relative_to(root).as_posix()
|
|
44
|
+
if any(_matches(rel, pattern) for pattern in EVAL_SUITE_PATTERNS):
|
|
45
|
+
files.append(path)
|
|
46
|
+
if not files:
|
|
47
|
+
raise ValueError("trusted evaluation suite is empty")
|
|
48
|
+
for path in sorted(files):
|
|
49
|
+
rel = path.relative_to(root).as_posix()
|
|
50
|
+
digest.update(rel.encode("utf-8"))
|
|
51
|
+
digest.update(b"\0")
|
|
52
|
+
digest.update(path.read_bytes())
|
|
53
|
+
digest.update(b"\0")
|
|
54
|
+
return digest.hexdigest()
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@dataclass(frozen=True)
|
|
58
|
+
class AgentIsolation:
|
|
59
|
+
"""Runner-owned authority for mutation-agent OS isolation.
|
|
60
|
+
|
|
61
|
+
`verified` is true only when the runner itself can construct a supported
|
|
62
|
+
container boundary that mounts the sanitized mutation view and nothing from
|
|
63
|
+
the canonical repository. Evaluator JSON can never set this value.
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
mode: str
|
|
67
|
+
verified: bool
|
|
68
|
+
runtime: str | None = None
|
|
69
|
+
image: str | None = None
|
|
70
|
+
reason: str = ""
|
|
71
|
+
|
|
72
|
+
@classmethod
|
|
73
|
+
def from_environment(cls) -> "AgentIsolation":
|
|
74
|
+
mode = os.environ.get("EDUEVIDENCE_AUTOEVOLVE_ISOLATION", "none").strip().lower()
|
|
75
|
+
if mode in {"", "none", "off"}:
|
|
76
|
+
return cls("none", False, reason="no OS isolation provider configured")
|
|
77
|
+
if mode != "container":
|
|
78
|
+
return cls(mode, False, reason=f"unsupported isolation mode: {mode}")
|
|
79
|
+
|
|
80
|
+
requested = os.environ.get("EDUEVIDENCE_AUTOEVOLVE_CONTAINER_RUNTIME", "").strip()
|
|
81
|
+
runtimes = [requested] if requested else ["docker", "podman"]
|
|
82
|
+
runtime = next((item for item in runtimes if item and shutil.which(item)), None)
|
|
83
|
+
image = os.environ.get("EDUEVIDENCE_AUTOEVOLVE_ISOLATION_IMAGE", "").strip()
|
|
84
|
+
if not runtime:
|
|
85
|
+
return cls("container", False, reason="docker/podman runtime unavailable")
|
|
86
|
+
if not image:
|
|
87
|
+
return cls("container", False, runtime=runtime, reason="isolation image not configured")
|
|
88
|
+
return cls(
|
|
89
|
+
"container",
|
|
90
|
+
True,
|
|
91
|
+
runtime=runtime,
|
|
92
|
+
image=image,
|
|
93
|
+
reason="runner-owned container boundary",
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
def wrap_command(
|
|
97
|
+
self,
|
|
98
|
+
command: str,
|
|
99
|
+
view: str | Path,
|
|
100
|
+
env: Mapping[str, str],
|
|
101
|
+
) -> tuple[str, dict[str, str]]:
|
|
102
|
+
"""Return a command/environment suitable for `_run_json`.
|
|
103
|
+
|
|
104
|
+
Container mode exposes only the sanitized view at /workspace, disables
|
|
105
|
+
networking, uses a read-only container root, and gives the agent a
|
|
106
|
+
writable tmpfs. Only explicit EDUEVIDENCE_* control variables are
|
|
107
|
+
forwarded; host credentials are not inherited implicitly.
|
|
108
|
+
"""
|
|
109
|
+
if not self.verified:
|
|
110
|
+
return command, dict(env)
|
|
111
|
+
assert self.runtime and self.image
|
|
112
|
+
view = Path(view).resolve()
|
|
113
|
+
argv = [
|
|
114
|
+
self.runtime,
|
|
115
|
+
"run",
|
|
116
|
+
"--rm",
|
|
117
|
+
"--network",
|
|
118
|
+
"none",
|
|
119
|
+
"--read-only",
|
|
120
|
+
"--tmpfs",
|
|
121
|
+
"/tmp:rw,nosuid,nodev",
|
|
122
|
+
"--mount",
|
|
123
|
+
f"type=bind,src={view},dst=/workspace,rw",
|
|
124
|
+
"--workdir",
|
|
125
|
+
"/workspace",
|
|
126
|
+
]
|
|
127
|
+
view_prefix = str(view)
|
|
128
|
+
for key, value in env.items():
|
|
129
|
+
if not key.startswith("EDUEVIDENCE_"):
|
|
130
|
+
continue
|
|
131
|
+
text = str(value)
|
|
132
|
+
if text == view_prefix:
|
|
133
|
+
text = "/workspace"
|
|
134
|
+
elif text.startswith(view_prefix + os.sep):
|
|
135
|
+
rel = Path(text).relative_to(view).as_posix()
|
|
136
|
+
text = f"/workspace/{rel}"
|
|
137
|
+
argv.extend(["--env", f"{key}={text}"])
|
|
138
|
+
argv.extend([self.image, "sh", "-lc", command])
|
|
139
|
+
# The container runtime itself receives only the minimal host settings
|
|
140
|
+
# needed to launch. Model/API credentials are not implicitly forwarded.
|
|
141
|
+
host_env = {
|
|
142
|
+
key: value
|
|
143
|
+
for key, value in os.environ.items()
|
|
144
|
+
if key in {"PATH", "HOME", "TMPDIR"}
|
|
145
|
+
}
|
|
146
|
+
return shlex.join(argv), host_env
|