eduevidence 5.2.0 → 6.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +105 -0
- package/README.md +142 -75
- package/README.zh-CN.md +73 -30
- package/SKILL.md +397 -131
- package/agents/openai.yaml +4 -0
- package/assets/readme/controlled-execution.svg +34 -0
- package/assets/readme/landing-tour.gif +0 -0
- package/assets/readme/logo.png +0 -0
- package/assets/readme/research-workflow.svg +56 -0
- package/assets/readme/studio-graph.png +0 -0
- package/assets/readme/studio-overview.png +0 -0
- package/assets/readme/studio-reports.png +0 -0
- package/assets/readme/studio-tour.gif +0 -0
- package/autoevolve/config.yaml +17 -0
- package/autoevolve/program.md +25 -0
- package/autoevolve/protected.manifest.yaml +34 -0
- package/benchmarks/adversarial/cases.jsonl +7 -0
- package/benchmarks/evidence-library.json +5268 -0
- package/benchmarks/partitions.json +8 -0
- package/bin/eduevidence.js +2 -1
- package/docs/architecture.md +496 -0
- package/docs/autoresearch-evolution-plan.md +2903 -0
- package/docs/autoresearch-implementation-status.md +101 -0
- package/docs/demo-storyboard.md +20 -0
- package/docs/demo-workplace-ai.md +92 -0
- package/docs/demo.md +32 -0
- package/docs/install-guide.md +150 -0
- package/docs/orchestration-role-model.md +1254 -0
- package/docs/release-closeout/README.md +17 -0
- package/docs/release-closeout/frontend-acceptance.md +23 -0
- package/docs/release-closeout/issues.md +19 -0
- package/docs/release-closeout/verification.md +28 -0
- package/docs/release-contract.md +108 -0
- package/docs/research-studio-guide.zh-CN.md +166 -0
- package/docs/sciverse-api.md +125 -0
- package/eduevidence_cli.py +29 -13
- package/engine/_resources.py +13 -0
- package/engine/autoevolve/__init__.py +3 -0
- package/engine/autoevolve/agent_view.py +167 -0
- package/engine/autoevolve/core.py +357 -0
- package/engine/autoevolve/events.py +11 -0
- package/engine/autoevolve/git_workspace.py +77 -0
- package/engine/autoevolve/projection.py +23 -0
- package/engine/autoevolve/runner.py +413 -0
- package/engine/autoevolve/trust.py +146 -0
- package/engine/autoresearch/__init__.py +6 -0
- package/engine/autoresearch/commit.py +132 -0
- package/engine/autoresearch/contracts.py +126 -0
- package/engine/autoresearch/controller.py +207 -0
- package/engine/autoresearch/events.py +12 -0
- package/engine/autoresearch/gap_priority.py +168 -0
- package/engine/autoresearch/projection.py +30 -0
- package/engine/autoresearch/research_memory.py +59 -0
- package/engine/autoresearch/saturation.py +91 -0
- package/engine/briefs.py +2 -1
- package/engine/capabilities.py +1 -0
- package/engine/contracts.py +3 -1
- package/engine/decision_policy.py +96 -0
- package/engine/evidence_graph.py +14 -10
- package/engine/evidencecore.py +7 -5
- package/engine/gaps.py +132 -73
- package/engine/ids.py +2 -0
- package/engine/judge_pack.py +65 -0
- package/engine/library.py +6 -2
- package/engine/library_builtin.py +3 -1
- package/engine/living.py +36 -5
- package/engine/meta_synthesis.py +3 -1
- package/engine/migration.py +88 -3
- package/engine/orchestration.py +460 -0
- package/engine/paths.py +2 -0
- package/engine/pilot.py +36 -33
- package/engine/project.py +2 -2
- package/engine/research_service.py +113 -0
- package/engine/studio_read_model.py +400 -0
- package/engine/taxonomy.py +211 -0
- package/engine/tribunal.py +44 -33
- package/engine/update.py +1 -0
- package/engine/versions.py +1 -1
- package/engine/worker_result.py +109 -0
- package/engine/workflows.py +70 -0
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +2934 -0
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
- package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
- package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
- package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
- package/examples/ai-coding-assistant-evidence/frame.json +48 -0
- package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
- package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
- package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
- package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +230 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +2934 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +2934 -0
- package/examples/ai-coding-assistant-evidence/result.json +1457 -0
- package/examples/ai-coding-assistant-evidence/result.zh.json +1457 -0
- package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
- package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +107 -0
- package/examples/spaced-retrieval-practice/applicability.json +14 -0
- package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
- package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
- package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
- package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
- package/examples/spaced-retrieval-practice/frame.json +58 -0
- package/examples/spaced-retrieval-practice/gate_report.json +101 -0
- package/examples/spaced-retrieval-practice/methodology.json +78 -0
- package/examples/spaced-retrieval-practice/report_spec.json +212 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/result.json +942 -0
- package/examples/spaced-retrieval-practice/result.zh.json +942 -0
- package/examples/spaced-retrieval-practice/skeptic.json +70 -0
- package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
- package/examples/spaced-retrieval-practice/verdict.json +93 -0
- package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -0
- package/examples/workplace-ai-assistant/evaluation.json +19 -0
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
- package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
- package/examples/workplace-ai-assistant/final_verdict.json +78 -0
- package/examples/workplace-ai-assistant/frame.json +41 -0
- package/examples/workplace-ai-assistant/gate_report.json +101 -0
- package/examples/workplace-ai-assistant/intervention.json +27 -0
- package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
- package/examples/workplace-ai-assistant/methodology.json +60 -0
- package/examples/workplace-ai-assistant/report_spec.json +224 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/result.json +615 -0
- package/examples/workplace-ai-assistant/result.zh.json +615 -0
- package/examples/workplace-ai-assistant/search_log.json +19 -0
- package/examples/workplace-ai-assistant/skeptic.json +72 -0
- package/examples/workplace-ai-assistant/sources.jsonl +3 -0
- package/examples/workplace-ai-assistant/validation_result.json +9 -0
- package/examples/workplace-ai-assistant/verdict.json +78 -0
- package/install.sh +7 -7
- package/integrations/agent_mcp.py +2 -2
- package/integrations/orchestration_dispatch.py +146 -0
- package/package.json +46 -3
- package/pyproject.toml +14 -22
- package/references/autoresearch.md +30 -0
- package/references/evaluation-policy.md +24 -0
- package/references/orchestration.md +22 -0
- package/references/report-copy-style.md +67 -0
- package/references/retrieval-compliance.md +75 -0
- package/references/retrieval-protocol.md +20 -0
- package/references/scientific-invariants.md +19 -0
- package/retrieval/audit.py +178 -0
- package/retrieval/fetch.py +96 -0
- package/retrieval/sciverse.py +398 -0
- package/retrieval/search.py +47 -7
- package/schemas/applicability.schema.json +94 -0
- package/schemas/chart-spec.schema.json +10 -3
- package/schemas/evidence.schema.json +316 -43
- package/schemas/fetch-result.schema.json +2 -1
- package/schemas/intervention.schema.json +106 -21
- package/schemas/report-result.schema.json +12 -4
- package/schemas/report-spec.schema.json +98 -100
- package/schemas/skeptic.schema.json +86 -0
- package/schemas/source.schema.json +21 -2
- package/schemas/v2/finding.schema.json +5 -1
- package/schemas/v2/methodology-audit.schema.json +5 -1
- package/schemas/v2/outcome.schema.json +28 -5
- package/schemas/v2/project.schema.json +2 -2
- package/schemas/v2/run.schema.json +1 -1
- package/schemas/v2/study.schema.json +5 -1
- package/schemas/vNext/autoevolve-session.schema.json +34 -0
- package/schemas/vNext/eval-snapshot.schema.json +77 -0
- package/schemas/vNext/execution-plan.schema.json +50 -0
- package/schemas/vNext/gap-priority.schema.json +54 -0
- package/schemas/vNext/negative-search-record.schema.json +68 -0
- package/schemas/vNext/research-iteration.schema.json +87 -0
- package/schemas/vNext/research-strategy.schema.json +62 -0
- package/schemas/vNext/skill-experiment.schema.json +90 -0
- package/schemas/vNext/task-spec.schema.json +156 -0
- package/schemas/vNext/worker-result.schema.json +60 -0
- package/schemas/verdict.schema.json +164 -28
- package/scripts/benchmark_judge.py +2 -2
- package/scripts/benchmark_v3.py +26 -43
- package/scripts/build_esl_artifacts.py +4 -4
- package/scripts/build_evidence_library.py +2 -2
- package/scripts/build_gh_pages.py +98 -0
- package/scripts/build_readme_diagrams.py +72 -0
- package/scripts/build_report_variants.py +101 -0
- package/scripts/build_result.py +74 -9
- package/scripts/check_autoresearch_invariants.py +95 -0
- package/scripts/check_package_parity.py +85 -0
- package/scripts/check_protocol_alignment.py +375 -0
- package/scripts/check_versioned_schemas.py +254 -0
- package/scripts/claim_audit.py +13 -8
- package/scripts/compute_confidence.py +10 -0
- package/scripts/daily_evolve.py +30 -0
- package/scripts/dashboard_server.py +130 -101
- package/scripts/did_regression.py +17 -32
- package/scripts/enrich_projects_human_and_lieflat.py +1 -1
- package/scripts/evidence_score.py +5 -2
- package/scripts/generate_metrics.py +4 -3
- package/scripts/generate_new_projects.py +5 -5
- package/scripts/orchestrator.py +286 -36
- package/scripts/pre_verdict_gate.py +224 -26
- package/scripts/quickstart.py +18 -2
- package/scripts/rebake_all_5themes.py +1 -2
- package/scripts/research_auto_cli.py +475 -0
- package/scripts/run_workspace.py +24 -8
- package/scripts/search_provenance.py +64 -0
- package/scripts/serve_web.py +9 -10
- package/scripts/skill_lint.py +1 -1
- package/scripts/skill_payload.py +81 -0
- package/scripts/test_adversarial_empirical.py +26 -19
- package/scripts/validate_schema.py +46 -2
- package/scripts/vnext_cli.py +133 -0
- package/setup.py +12 -0
- package/skill/agents/evaluation-designer.md +20 -4
- package/skill/agents/evidence-analyst.md +19 -3
- package/skill/agents/evidence-judge.md +50 -2
- package/skill/agents/evidence-retriever.md +20 -3
- package/skill/agents/intervention-designer.md +20 -4
- package/skill/agents/method-reviewer.md +18 -2
- package/skill/agents/{education-planner.md → research-planner.md} +19 -3
- package/skill/agents/skeptic.md +18 -2
- package/skill/roles/registry.yaml +45 -0
- package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
- package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
- package/skill/sub-skills/data-analysis/SKILL.md +34 -15
- package/skill/sub-skills/ethics-review/SKILL.md +33 -10
- package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
- package/skill/sub-skills/evidence-review/SKILL.md +31 -12
- package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
- package/skill/sub-skills/literature-review/SKILL.md +35 -14
- package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
- package/skill/sub-skills/report-generation/SKILL.md +40 -6
- package/skill/sub-skills/research-planning/SKILL.md +41 -14
- package/skill/sub-skills/study-design/SKILL.md +30 -9
- package/skill/task-briefs/adjudicate.md +32 -7
- package/skill/task-briefs/applicability.md +38 -0
- package/skill/task-briefs/audit.md +32 -7
- package/skill/task-briefs/challenge.md +34 -5
- package/skill/task-briefs/evaluate.md +30 -5
- package/skill/task-briefs/extract.md +31 -8
- package/skill/task-briefs/frame.md +39 -10
- package/skill/task-briefs/intervene.md +32 -6
- package/skill/task-briefs/present.md +32 -8
- package/skill/task-briefs/projection.md +37 -0
- package/skill/task-briefs/retrieve.md +36 -6
- package/skill/workflows/decision-and-pilot.md +85 -0
- package/skill/workflows/evaluate-and-update.md +93 -0
- package/skill/workflows/evidence-review.md +117 -0
- package/visualization/eduevidence-report/assets/base.css +2 -2
- package/visualization/eduevidence-report/assets/reader.css +752 -0
- package/visualization/eduevidence-report/assets/reader.js +132 -0
- package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
- package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
- package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
- package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
- package/visualization/eduevidence-report/scripts/build_report.py +561 -121
- package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +371 -136
- package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
- package/visualization/eduevidence-report/themes/academic.css +1 -1
- package/visualization/eduevidence-report/themes/claude.css +1 -1
- package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
- package/visualization/eduevidence-report/themes/datalab.css +2 -2
- package/visualization/eduevidence-report/themes/presentation.css +2 -2
- package/web/README.md +18 -0
- package/web/architecture.html +14885 -0
- package/web/index.html +53 -0
- package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
- package/web/studio/assets/index-B8tkF44Q.css +1 -0
- package/web/studio/assets/index-CQ6Keoyc.js +230 -0
- package/web/studio/config.json +1 -0
- package/web/studio/index.html +14 -0
- package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
- package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/bias.cpython-312.pyc +0 -0
- package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
- package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
- package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
- package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
- package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
- package/engine/__pycache__/events.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
- package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
- package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
- package/engine/__pycache__/ids.cpython-312.pyc +0 -0
- package/engine/__pycache__/library.cpython-312.pyc +0 -0
- package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
- package/engine/__pycache__/living.cpython-312.pyc +0 -0
- package/engine/__pycache__/log.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/migration.cpython-312.pyc +0 -0
- package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
- package/engine/__pycache__/paths.cpython-312.pyc +0 -0
- package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
- package/engine/__pycache__/planner.cpython-312.pyc +0 -0
- package/engine/__pycache__/project.cpython-312.pyc +0 -0
- package/engine/__pycache__/projections.cpython-312.pyc +0 -0
- package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
- package/engine/__pycache__/run.cpython-312.pyc +0 -0
- package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
- package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
- package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
- package/engine/__pycache__/update.cpython-312.pyc +0 -0
- package/engine/__pycache__/versions.cpython-312.pyc +0 -0
- package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
- package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
- package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
- package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
- package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
- package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
- package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
- package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
- package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
- package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
- package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
- package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
- package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
- package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
- package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
- package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
"""Sanitized mutation view for Skill Autoresearch agents.
|
|
2
|
+
|
|
3
|
+
This prevents accidental holdout/gold exposure by default: the mutation agent
|
|
4
|
+
sees a tracked-file copy without .git and without evaluator/holdout/adversarial
|
|
5
|
+
benchmark material. `benchmarks/questions.jsonl` and gold annotations are
|
|
6
|
+
filtered to DEV ids only. It is context isolation, not an OS security sandbox;
|
|
7
|
+
a promotion evaluator must independently attest stronger holdout isolation.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import hashlib
|
|
12
|
+
import json
|
|
13
|
+
import shutil
|
|
14
|
+
import subprocess
|
|
15
|
+
import tempfile
|
|
16
|
+
from dataclasses import dataclass
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _sha(path: Path) -> str:
|
|
22
|
+
return hashlib.sha256(path.read_bytes()).hexdigest()
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _tracked_files(worktree: Path) -> list[str]:
|
|
26
|
+
output = subprocess.check_output(
|
|
27
|
+
["git", "ls-files", "-z"], cwd=worktree, text=False
|
|
28
|
+
)
|
|
29
|
+
return [item.decode("utf-8") for item in output.split(b"\0") if item]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _dev_ids(worktree: Path) -> set[str]:
|
|
33
|
+
path = worktree / "benchmarks" / "partitions.json"
|
|
34
|
+
if not path.is_file():
|
|
35
|
+
return set()
|
|
36
|
+
value = json.loads(path.read_text(encoding="utf-8"))
|
|
37
|
+
return {str(item) for item in value.get("dev", [])}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _copy_filtered_benchmark(source: Path, dest: Path, rel: str, dev_ids: set[str]) -> bool:
|
|
41
|
+
if rel == "benchmarks/partitions.json":
|
|
42
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
43
|
+
shutil.copy2(source, dest)
|
|
44
|
+
return True
|
|
45
|
+
if rel == "benchmarks/questions.jsonl":
|
|
46
|
+
rows = []
|
|
47
|
+
for line in source.read_text(encoding="utf-8").splitlines():
|
|
48
|
+
if not line.strip():
|
|
49
|
+
continue
|
|
50
|
+
row = json.loads(line)
|
|
51
|
+
if row.get("id") in dev_ids:
|
|
52
|
+
rows.append(json.dumps(row, ensure_ascii=False))
|
|
53
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
54
|
+
dest.write_text("\n".join(rows) + ("\n" if rows else ""), encoding="utf-8")
|
|
55
|
+
return True
|
|
56
|
+
if rel.startswith("benchmarks/annotations/gold-"):
|
|
57
|
+
qid = Path(rel).stem.removeprefix("gold-")
|
|
58
|
+
if qid in dev_ids:
|
|
59
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
60
|
+
shutil.copy2(source, dest)
|
|
61
|
+
return True
|
|
62
|
+
if rel.startswith("benchmarks/dev/"):
|
|
63
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
64
|
+
shutil.copy2(source, dest)
|
|
65
|
+
return True
|
|
66
|
+
return False
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@dataclass
|
|
70
|
+
class AgentMutationView:
|
|
71
|
+
path: Path
|
|
72
|
+
baseline_hashes: dict[str, str]
|
|
73
|
+
|
|
74
|
+
@classmethod
|
|
75
|
+
def create(
|
|
76
|
+
cls,
|
|
77
|
+
worktree: str | Path,
|
|
78
|
+
*,
|
|
79
|
+
session_context: dict[str, Any] | None = None,
|
|
80
|
+
) -> "AgentMutationView":
|
|
81
|
+
worktree = Path(worktree).resolve()
|
|
82
|
+
root = Path(tempfile.mkdtemp(prefix="eduevidence-autoevolve-agent-"))
|
|
83
|
+
dev_ids = _dev_ids(worktree)
|
|
84
|
+
for rel in _tracked_files(worktree):
|
|
85
|
+
source = worktree / rel
|
|
86
|
+
if not source.is_file() or source.is_symlink():
|
|
87
|
+
continue
|
|
88
|
+
dest = root / rel
|
|
89
|
+
if rel.startswith("benchmarks/"):
|
|
90
|
+
_copy_filtered_benchmark(source, dest, rel, dev_ids)
|
|
91
|
+
continue
|
|
92
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
93
|
+
shutil.copy2(source, dest)
|
|
94
|
+
if session_context is not None:
|
|
95
|
+
context_path = root / "autoevolve" / "session-context.json"
|
|
96
|
+
context_path.parent.mkdir(parents=True, exist_ok=True)
|
|
97
|
+
context_path.write_text(
|
|
98
|
+
json.dumps(session_context, ensure_ascii=False, indent=2) + "\n",
|
|
99
|
+
encoding="utf-8",
|
|
100
|
+
)
|
|
101
|
+
baseline = {
|
|
102
|
+
path.relative_to(root).as_posix(): _sha(path)
|
|
103
|
+
for path in root.rglob("*")
|
|
104
|
+
if path.is_file() and not path.is_symlink()
|
|
105
|
+
}
|
|
106
|
+
return cls(root, baseline)
|
|
107
|
+
|
|
108
|
+
def changed_files(self) -> list[str]:
|
|
109
|
+
current = {
|
|
110
|
+
path.relative_to(self.path).as_posix(): _sha(path)
|
|
111
|
+
for path in self.path.rglob("*")
|
|
112
|
+
if path.is_file() and not path.is_symlink()
|
|
113
|
+
}
|
|
114
|
+
changed = {
|
|
115
|
+
rel
|
|
116
|
+
for rel in set(self.baseline_hashes) | set(current)
|
|
117
|
+
if self.baseline_hashes.get(rel) != current.get(rel)
|
|
118
|
+
}
|
|
119
|
+
changed.update(
|
|
120
|
+
path.relative_to(self.path).as_posix()
|
|
121
|
+
for path in self.path.rglob("*")
|
|
122
|
+
if path.is_symlink()
|
|
123
|
+
)
|
|
124
|
+
return sorted(changed)
|
|
125
|
+
|
|
126
|
+
def sync_to(self, worktree: str | Path, changed: list[str]) -> None:
|
|
127
|
+
worktree = Path(worktree).resolve()
|
|
128
|
+
for rel in changed:
|
|
129
|
+
source = self.path / rel
|
|
130
|
+
target = (worktree / rel).resolve()
|
|
131
|
+
if worktree not in target.parents and target != worktree:
|
|
132
|
+
raise ValueError(f"unsafe mutation path: {rel}")
|
|
133
|
+
if source.is_symlink():
|
|
134
|
+
raise ValueError(f"symlink mutations are forbidden: {rel}")
|
|
135
|
+
if source.is_file():
|
|
136
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
137
|
+
shutil.copy2(source, target)
|
|
138
|
+
elif target.exists():
|
|
139
|
+
if target.is_dir():
|
|
140
|
+
shutil.rmtree(target)
|
|
141
|
+
else:
|
|
142
|
+
target.unlink()
|
|
143
|
+
|
|
144
|
+
def export_changes(self, destination: str | Path, changed: list[str]) -> None:
|
|
145
|
+
"""Preserve candidate files outside the worktree before a revert."""
|
|
146
|
+
destination = Path(destination)
|
|
147
|
+
destination.mkdir(parents=True, exist_ok=True)
|
|
148
|
+
manifest = []
|
|
149
|
+
for rel in changed:
|
|
150
|
+
source = self.path / rel
|
|
151
|
+
if source.is_symlink():
|
|
152
|
+
manifest.append({"path": rel, "status": "symlink_forbidden"})
|
|
153
|
+
continue
|
|
154
|
+
if source.is_file():
|
|
155
|
+
target = destination / rel
|
|
156
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
157
|
+
shutil.copy2(source, target)
|
|
158
|
+
manifest.append({"path": rel, "status": "present"})
|
|
159
|
+
else:
|
|
160
|
+
manifest.append({"path": rel, "status": "deleted"})
|
|
161
|
+
(destination / "manifest.json").write_text(
|
|
162
|
+
json.dumps(manifest, ensure_ascii=False, indent=2) + "\n",
|
|
163
|
+
encoding="utf-8",
|
|
164
|
+
)
|
|
165
|
+
|
|
166
|
+
def cleanup(self) -> None:
|
|
167
|
+
shutil.rmtree(self.path, ignore_errors=True)
|
|
@@ -0,0 +1,357 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import fnmatch
|
|
4
|
+
import hashlib
|
|
5
|
+
import json
|
|
6
|
+
from dataclasses import asdict, dataclass, field
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
PROTECTED_DEFAULTS = (
|
|
10
|
+
"autoevolve/protected.manifest.yaml",
|
|
11
|
+
"benchmarks/annotations/**",
|
|
12
|
+
"benchmarks/holdout/**",
|
|
13
|
+
"benchmarks/evaluator/**",
|
|
14
|
+
"schemas/**",
|
|
15
|
+
"references/scientific-invariants.md",
|
|
16
|
+
"scripts/pre_verdict_gate.py",
|
|
17
|
+
"scripts/compute_confidence.py",
|
|
18
|
+
"scripts/check_autoresearch_invariants.py",
|
|
19
|
+
"engine/graph_store.py",
|
|
20
|
+
"engine/study_design.py",
|
|
21
|
+
"engine/autoevolve/**",
|
|
22
|
+
".github/workflows/autoresearch-gates.yml",
|
|
23
|
+
)
|
|
24
|
+
SAFE_DEFAULTS = (
|
|
25
|
+
"skill/workflows/**",
|
|
26
|
+
"skill/agents/**",
|
|
27
|
+
"skill/sub-skills/**",
|
|
28
|
+
"retrieval/**",
|
|
29
|
+
"references/presentation/**",
|
|
30
|
+
)
|
|
31
|
+
CONTROLLED_DEFAULTS = (
|
|
32
|
+
"engine/semantics.py",
|
|
33
|
+
"engine/gaps.py",
|
|
34
|
+
"scripts/complexity_gate.py",
|
|
35
|
+
"engine/orchestration.py",
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass(frozen=True)
|
|
40
|
+
class EvalSnapshot:
|
|
41
|
+
eval_id: str
|
|
42
|
+
hard_gates_passed: bool
|
|
43
|
+
science_score: float
|
|
44
|
+
research_score: float
|
|
45
|
+
robustness: float
|
|
46
|
+
cost: float
|
|
47
|
+
latency: float
|
|
48
|
+
complexity: float
|
|
49
|
+
repeats: int = 1
|
|
50
|
+
noise_floor: float = 0.0
|
|
51
|
+
dev_passed: bool = False
|
|
52
|
+
holdout_passed: bool = False
|
|
53
|
+
adversarial_passed: bool = False
|
|
54
|
+
holdout_isolation_verified: bool = False
|
|
55
|
+
eval_suite_hash: str = ""
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass
|
|
59
|
+
class SkillExperiment:
|
|
60
|
+
experiment_id: str
|
|
61
|
+
session_id: str
|
|
62
|
+
parent_skill_revision: str
|
|
63
|
+
hypothesis: str
|
|
64
|
+
mutation_scope: tuple[str, ...]
|
|
65
|
+
changed_files: list[str] = field(default_factory=list)
|
|
66
|
+
candidate_commit: str | None = None
|
|
67
|
+
baseline_eval_id: str | None = None
|
|
68
|
+
candidate_eval_id: str | None = None
|
|
69
|
+
protected_hash_before: str | None = None
|
|
70
|
+
protected_hash_after: str | None = None
|
|
71
|
+
status: str = "created"
|
|
72
|
+
promotion_reason: str = ""
|
|
73
|
+
complexity_delta: float = 0.0
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _matches(path: str, patterns: tuple[str, ...]) -> bool:
|
|
77
|
+
normalized = path.replace("\\", "/")
|
|
78
|
+
return any(
|
|
79
|
+
fnmatch.fnmatch(normalized, pattern)
|
|
80
|
+
or (pattern.endswith("/**") and normalized.startswith(pattern[:-3]))
|
|
81
|
+
for pattern in patterns
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _parse_manifest(path: Path) -> tuple[tuple[str, ...], tuple[str, ...], tuple[str, ...]]:
|
|
86
|
+
"""Parse the intentionally tiny manifest subset without a YAML dependency."""
|
|
87
|
+
protected: list[str] = []
|
|
88
|
+
safe: list[str] = []
|
|
89
|
+
controlled: list[str] = []
|
|
90
|
+
section = ""
|
|
91
|
+
subsection = ""
|
|
92
|
+
for raw in path.read_text(encoding="utf-8").splitlines():
|
|
93
|
+
line = raw.split("#", 1)[0].rstrip()
|
|
94
|
+
if not line.strip():
|
|
95
|
+
continue
|
|
96
|
+
stripped = line.strip()
|
|
97
|
+
indent = len(line) - len(line.lstrip(" "))
|
|
98
|
+
if indent == 0 and stripped.endswith(":"):
|
|
99
|
+
section = stripped[:-1]
|
|
100
|
+
subsection = ""
|
|
101
|
+
continue
|
|
102
|
+
if section == "mutable" and indent == 2 and stripped.endswith(":"):
|
|
103
|
+
subsection = stripped[:-1]
|
|
104
|
+
continue
|
|
105
|
+
if stripped.startswith("- "):
|
|
106
|
+
value = stripped[2:].strip().strip('"\'')
|
|
107
|
+
if section == "protected":
|
|
108
|
+
protected.append(value)
|
|
109
|
+
elif section == "mutable" and subsection == "safe":
|
|
110
|
+
safe.append(value)
|
|
111
|
+
elif section == "mutable" and subsection == "controlled":
|
|
112
|
+
controlled.append(value)
|
|
113
|
+
if not protected or not safe:
|
|
114
|
+
raise ValueError(f"invalid autoresearch manifest: {path}")
|
|
115
|
+
return tuple(protected), tuple(safe), tuple(controlled)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
class ProtectedManifest:
|
|
119
|
+
def __init__(
|
|
120
|
+
self,
|
|
121
|
+
patterns=PROTECTED_DEFAULTS,
|
|
122
|
+
*,
|
|
123
|
+
safe_patterns=SAFE_DEFAULTS,
|
|
124
|
+
controlled_patterns=CONTROLLED_DEFAULTS,
|
|
125
|
+
):
|
|
126
|
+
self.patterns = tuple(patterns)
|
|
127
|
+
self.safe_patterns = tuple(safe_patterns)
|
|
128
|
+
self.controlled_patterns = tuple(controlled_patterns)
|
|
129
|
+
|
|
130
|
+
@classmethod
|
|
131
|
+
def from_repo(cls, root: str | Path) -> "ProtectedManifest":
|
|
132
|
+
path = Path(root) / "autoevolve" / "protected.manifest.yaml"
|
|
133
|
+
if not path.is_file():
|
|
134
|
+
return cls()
|
|
135
|
+
protected, safe, controlled = _parse_manifest(path)
|
|
136
|
+
merged_protected = tuple(dict.fromkeys((*protected, *PROTECTED_DEFAULTS)))
|
|
137
|
+
return cls(
|
|
138
|
+
merged_protected,
|
|
139
|
+
safe_patterns=safe,
|
|
140
|
+
controlled_patterns=controlled,
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
def is_protected(self, path: str) -> bool:
|
|
144
|
+
return _matches(path, self.patterns)
|
|
145
|
+
|
|
146
|
+
def classify(self, path: str) -> str:
|
|
147
|
+
if self.is_protected(path):
|
|
148
|
+
return "protected"
|
|
149
|
+
if _matches(path, self.safe_patterns):
|
|
150
|
+
return "safe"
|
|
151
|
+
if _matches(path, self.controlled_patterns):
|
|
152
|
+
return "controlled"
|
|
153
|
+
return "unknown"
|
|
154
|
+
|
|
155
|
+
def validate_changes(self, changed: list[str]) -> tuple[bool, list[str]]:
|
|
156
|
+
bad = [path for path in changed if self.is_protected(path)]
|
|
157
|
+
return (not bad, bad)
|
|
158
|
+
|
|
159
|
+
def validate_mutation_scope(
|
|
160
|
+
self,
|
|
161
|
+
changed: list[str],
|
|
162
|
+
*,
|
|
163
|
+
mutation_tiers: tuple[str, ...] | list[str],
|
|
164
|
+
allow_controlled: bool,
|
|
165
|
+
) -> tuple[bool, list[str]]:
|
|
166
|
+
tiers = set(mutation_tiers)
|
|
167
|
+
bad: list[str] = []
|
|
168
|
+
for path in changed:
|
|
169
|
+
kind = self.classify(path)
|
|
170
|
+
allowed = kind == "safe" and "safe" in tiers
|
|
171
|
+
allowed = allowed or (
|
|
172
|
+
kind == "controlled" and allow_controlled and "controlled" in tiers
|
|
173
|
+
)
|
|
174
|
+
if not allowed:
|
|
175
|
+
bad.append(path)
|
|
176
|
+
return (not bad, bad)
|
|
177
|
+
|
|
178
|
+
def hash_tree(self, root: str | Path) -> str:
|
|
179
|
+
root = Path(root)
|
|
180
|
+
digest = hashlib.sha256()
|
|
181
|
+
files = []
|
|
182
|
+
for file in root.rglob("*"):
|
|
183
|
+
if file.is_file() and self.is_protected(file.relative_to(root).as_posix()):
|
|
184
|
+
files.append(file)
|
|
185
|
+
for file in sorted(files):
|
|
186
|
+
rel = file.relative_to(root).as_posix()
|
|
187
|
+
digest.update(rel.encode())
|
|
188
|
+
digest.update(b"\0")
|
|
189
|
+
digest.update(file.read_bytes())
|
|
190
|
+
digest.update(b"\0")
|
|
191
|
+
return digest.hexdigest()
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _promotion_evidence_ready(baseline: EvalSnapshot, candidate: EvalSnapshot) -> tuple[bool, str]:
|
|
195
|
+
if not baseline.eval_suite_hash or baseline.eval_suite_hash != candidate.eval_suite_hash:
|
|
196
|
+
return False, "baseline/candidate eval suite hash missing or mismatched"
|
|
197
|
+
required = (
|
|
198
|
+
baseline.dev_passed,
|
|
199
|
+
baseline.holdout_passed,
|
|
200
|
+
baseline.adversarial_passed,
|
|
201
|
+
baseline.holdout_isolation_verified,
|
|
202
|
+
candidate.dev_passed,
|
|
203
|
+
candidate.holdout_passed,
|
|
204
|
+
candidate.adversarial_passed,
|
|
205
|
+
candidate.holdout_isolation_verified,
|
|
206
|
+
)
|
|
207
|
+
if not all(required):
|
|
208
|
+
return False, "DEV/HOLDOUT/adversarial gates and holdout isolation are required for automatic KEEP"
|
|
209
|
+
return True, ""
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def promote(
|
|
213
|
+
baseline: EvalSnapshot,
|
|
214
|
+
candidate: EvalSnapshot,
|
|
215
|
+
*,
|
|
216
|
+
simplicity_tolerance: float = 0.0,
|
|
217
|
+
efficiency_tolerance: float = 0.05,
|
|
218
|
+
minimum_repeats: int = 3,
|
|
219
|
+
) -> tuple[str, str]:
|
|
220
|
+
"""Constraint-first promotion; automatic KEEP is deliberately conservative."""
|
|
221
|
+
if not candidate.hard_gates_passed:
|
|
222
|
+
return "REJECT", "L0 hard gate failed"
|
|
223
|
+
if candidate.science_score < baseline.science_score:
|
|
224
|
+
return "REJECT", "scientific correctness regressed"
|
|
225
|
+
if min(baseline.repeats, candidate.repeats) < minimum_repeats:
|
|
226
|
+
return "RETEST", f"automatic promotion requires >= {minimum_repeats} repeated runs"
|
|
227
|
+
|
|
228
|
+
delta = candidate.research_score - baseline.research_score
|
|
229
|
+
noise = max(baseline.noise_floor, candidate.noise_floor)
|
|
230
|
+
if delta < -noise:
|
|
231
|
+
return "REJECT", "research-quality regression exceeds empirical noise floor"
|
|
232
|
+
|
|
233
|
+
regressions = []
|
|
234
|
+
if candidate.robustness < baseline.robustness:
|
|
235
|
+
regressions.append("robustness")
|
|
236
|
+
if baseline.cost > 0 and candidate.cost > baseline.cost * (1 + efficiency_tolerance):
|
|
237
|
+
regressions.append("cost")
|
|
238
|
+
if baseline.latency > 0 and candidate.latency > baseline.latency * (1 + efficiency_tolerance):
|
|
239
|
+
regressions.append("latency")
|
|
240
|
+
if candidate.complexity > baseline.complexity + simplicity_tolerance:
|
|
241
|
+
regressions.append("complexity")
|
|
242
|
+
|
|
243
|
+
if abs(delta) <= noise:
|
|
244
|
+
if regressions:
|
|
245
|
+
return "REJECT", "within noise floor with regression: " + ",".join(regressions)
|
|
246
|
+
if candidate.complexity < baseline.complexity - simplicity_tolerance:
|
|
247
|
+
ready, why = _promotion_evidence_ready(baseline, candidate)
|
|
248
|
+
if not ready:
|
|
249
|
+
return "HUMAN_REVIEW", why
|
|
250
|
+
return "KEEP", "equivalent research quality with simpler implementation"
|
|
251
|
+
return "RETEST", "candidate delta is within empirical noise floor"
|
|
252
|
+
|
|
253
|
+
if delta > noise and not regressions:
|
|
254
|
+
ready, why = _promotion_evidence_ready(baseline, candidate)
|
|
255
|
+
if not ready:
|
|
256
|
+
return "HUMAN_REVIEW", why
|
|
257
|
+
return "KEEP", "material research-quality improvement without Pareto regression"
|
|
258
|
+
return "HUMAN_REVIEW", "Pareto trade-off: " + ",".join(regressions or ["mixed metrics"])
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
class ExperimentLog:
|
|
262
|
+
HEADER = (
|
|
263
|
+
"experiment_id",
|
|
264
|
+
"parent_revision",
|
|
265
|
+
"candidate_commit",
|
|
266
|
+
"scope",
|
|
267
|
+
"hypothesis",
|
|
268
|
+
"eval_suite_hash",
|
|
269
|
+
"repeats",
|
|
270
|
+
"dev_passed",
|
|
271
|
+
"holdout_passed",
|
|
272
|
+
"adversarial_passed",
|
|
273
|
+
"holdout_isolation_verified",
|
|
274
|
+
"hard_gates",
|
|
275
|
+
"science_score",
|
|
276
|
+
"research_score",
|
|
277
|
+
"robustness",
|
|
278
|
+
"cost",
|
|
279
|
+
"latency",
|
|
280
|
+
"complexity_delta",
|
|
281
|
+
"status",
|
|
282
|
+
"description",
|
|
283
|
+
)
|
|
284
|
+
|
|
285
|
+
def __init__(self, root: str | Path):
|
|
286
|
+
self.root = Path(root)
|
|
287
|
+
self.root.mkdir(parents=True, exist_ok=True)
|
|
288
|
+
self.tsv = self.root / "results.tsv"
|
|
289
|
+
self.jsonl = self.root / "experiments.jsonl"
|
|
290
|
+
if not self.tsv.exists():
|
|
291
|
+
self.tsv.write_text("\t".join(self.HEADER) + "\n", encoding="utf-8")
|
|
292
|
+
|
|
293
|
+
def append(
|
|
294
|
+
self,
|
|
295
|
+
experiment: SkillExperiment,
|
|
296
|
+
*,
|
|
297
|
+
candidate: EvalSnapshot | None = None,
|
|
298
|
+
description: str = "",
|
|
299
|
+
) -> None:
|
|
300
|
+
row = [
|
|
301
|
+
experiment.experiment_id,
|
|
302
|
+
experiment.parent_skill_revision,
|
|
303
|
+
experiment.candidate_commit or "",
|
|
304
|
+
",".join(experiment.mutation_scope),
|
|
305
|
+
experiment.hypothesis,
|
|
306
|
+
str(candidate.eval_suite_hash if candidate else ""),
|
|
307
|
+
str(candidate.repeats if candidate else ""),
|
|
308
|
+
str(candidate.dev_passed if candidate else ""),
|
|
309
|
+
str(candidate.holdout_passed if candidate else ""),
|
|
310
|
+
str(candidate.adversarial_passed if candidate else ""),
|
|
311
|
+
str(candidate.holdout_isolation_verified if candidate else ""),
|
|
312
|
+
str(candidate.hard_gates_passed if candidate else ""),
|
|
313
|
+
str(candidate.science_score if candidate else ""),
|
|
314
|
+
str(candidate.research_score if candidate else ""),
|
|
315
|
+
str(candidate.robustness if candidate else ""),
|
|
316
|
+
str(candidate.cost if candidate else ""),
|
|
317
|
+
str(candidate.latency if candidate else ""),
|
|
318
|
+
str(experiment.complexity_delta),
|
|
319
|
+
experiment.status,
|
|
320
|
+
description,
|
|
321
|
+
]
|
|
322
|
+
with self.tsv.open("a", encoding="utf-8") as handle:
|
|
323
|
+
handle.write("\t".join(value.replace("\t", " ") for value in row) + "\n")
|
|
324
|
+
with self.jsonl.open("a", encoding="utf-8") as handle:
|
|
325
|
+
handle.write(json.dumps(asdict(experiment), ensure_ascii=False, sort_keys=True) + "\n")
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
class PlateauTracker:
|
|
329
|
+
def __init__(self, limit: int = 5):
|
|
330
|
+
self.limit = limit
|
|
331
|
+
|
|
332
|
+
def plateau(self, statuses: list[str]) -> bool:
|
|
333
|
+
valid = [status for status in statuses if status not in {"CRASH", "INVALID", "RETEST"}]
|
|
334
|
+
return len(valid) >= self.limit and all(status != "KEEP" for status in valid[-self.limit :])
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
@dataclass(frozen=True)
|
|
338
|
+
class DailyProfile:
|
|
339
|
+
max_experiments: int = 20
|
|
340
|
+
max_cost_usd: float = 5.0
|
|
341
|
+
max_wall_minutes: int = 180
|
|
342
|
+
mutation_tiers: tuple[str, ...] = ("safe",)
|
|
343
|
+
allow_controlled: bool = False
|
|
344
|
+
promotion: str = "branch_only"
|
|
345
|
+
|
|
346
|
+
def validate(self) -> None:
|
|
347
|
+
if not 1 <= self.max_experiments <= 50:
|
|
348
|
+
raise ValueError("daily max_experiments must be 1..50")
|
|
349
|
+
if self.max_cost_usd <= 0 or self.max_wall_minutes <= 0:
|
|
350
|
+
raise ValueError("daily budget must be positive")
|
|
351
|
+
if self.promotion != "branch_only":
|
|
352
|
+
raise ValueError("daily mode is branch_only")
|
|
353
|
+
unknown = set(self.mutation_tiers) - {"safe", "controlled"}
|
|
354
|
+
if unknown:
|
|
355
|
+
raise ValueError(f"unknown mutation tiers: {sorted(unknown)}")
|
|
356
|
+
if "controlled" in self.mutation_tiers and not self.allow_controlled:
|
|
357
|
+
raise ValueError("controlled mutation tier requires allow_controlled=true")
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
SKILL_AUTORESEARCH_EVENTS = (
|
|
2
|
+
"autoevolve.session.started",
|
|
3
|
+
"autoevolve.experiment.created",
|
|
4
|
+
"autoevolve.candidate.built",
|
|
5
|
+
"autoevolve.eval.completed",
|
|
6
|
+
"autoevolve.candidate.kept",
|
|
7
|
+
"autoevolve.candidate.rejected",
|
|
8
|
+
"autoevolve.candidate.retest",
|
|
9
|
+
"autoevolve.plateau",
|
|
10
|
+
"autoevolve.session.completed",
|
|
11
|
+
)
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
import subprocess
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def _run(repo: Path, *args: str, capture: bool = False) -> str:
|
|
10
|
+
completed = subprocess.run(
|
|
11
|
+
["git", *args], cwd=repo, check=True, text=True, capture_output=capture
|
|
12
|
+
)
|
|
13
|
+
return completed.stdout.rstrip("\n") if capture else ""
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def safe_tag(tag: str) -> str:
|
|
17
|
+
clean = re.sub(r"[^A-Za-z0-9._-]+", "-", tag).strip("-")
|
|
18
|
+
if not clean or clean in {".", ".."}:
|
|
19
|
+
raise ValueError("invalid run tag")
|
|
20
|
+
return clean[:80]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass(frozen=True)
|
|
24
|
+
class GitWorkspace:
|
|
25
|
+
repo: Path
|
|
26
|
+
path: Path
|
|
27
|
+
branch: str
|
|
28
|
+
|
|
29
|
+
@classmethod
|
|
30
|
+
def create(cls, repo: str | Path, tag: str):
|
|
31
|
+
repo = Path(repo).resolve()
|
|
32
|
+
tag = safe_tag(tag)
|
|
33
|
+
branch = f"autoresearch/{tag}"
|
|
34
|
+
current = _run(repo, "branch", "--show-current", capture=True)
|
|
35
|
+
if current.startswith("autoresearch/"):
|
|
36
|
+
raise ValueError("create the session from a non-autoresearch base branch")
|
|
37
|
+
path = repo / ".autoevolve-worktrees" / tag
|
|
38
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
39
|
+
if path.exists():
|
|
40
|
+
raise FileExistsError(path)
|
|
41
|
+
_run(repo, "worktree", "add", "-b", branch, str(path), "HEAD")
|
|
42
|
+
return cls(repo, path, branch)
|
|
43
|
+
|
|
44
|
+
def head(self) -> str:
|
|
45
|
+
return _run(self.path, "rev-parse", "HEAD", capture=True)
|
|
46
|
+
|
|
47
|
+
def diff(self) -> str:
|
|
48
|
+
return _run(self.path, "diff", "--binary", "HEAD", capture=True)
|
|
49
|
+
|
|
50
|
+
def changed_files(self) -> list[str]:
|
|
51
|
+
output = _run(self.path, "status", "--porcelain", capture=True)
|
|
52
|
+
return [line[3:] for line in output.splitlines() if line.strip()]
|
|
53
|
+
|
|
54
|
+
def restore(self) -> None:
|
|
55
|
+
output = _run(self.path, "status", "--porcelain", capture=True)
|
|
56
|
+
untracked = [line[3:] for line in output.splitlines() if line.startswith("?? ")]
|
|
57
|
+
_run(self.path, "restore", "--staged", "--worktree", ".")
|
|
58
|
+
for relative in untracked:
|
|
59
|
+
target = (self.path / relative).resolve()
|
|
60
|
+
if self.path not in target.parents and target != self.path:
|
|
61
|
+
raise ValueError("unsafe untracked path")
|
|
62
|
+
if target.is_file() or target.is_symlink():
|
|
63
|
+
target.unlink(missing_ok=True)
|
|
64
|
+
elif target.is_dir():
|
|
65
|
+
import shutil
|
|
66
|
+
shutil.rmtree(target)
|
|
67
|
+
|
|
68
|
+
def commit(self, message: str) -> str:
|
|
69
|
+
_run(self.path, "add", "-A")
|
|
70
|
+
_run(self.path, "commit", "-m", message)
|
|
71
|
+
return _run(self.path, "rev-parse", "HEAD", capture=True)
|
|
72
|
+
|
|
73
|
+
def push(self, remote: str = "origin") -> None:
|
|
74
|
+
"""Push only this experiment branch. Never force-push or merge."""
|
|
75
|
+
if not self.branch.startswith("autoresearch/"):
|
|
76
|
+
raise ValueError("only autoresearch branches may be pushed by autoevolve")
|
|
77
|
+
_run(self.path, "push", "-u", remote, self.branch)
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from typing import Any
|
|
3
|
+
from .core import PlateauTracker
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def skill_evolution_projection(
|
|
7
|
+
*,
|
|
8
|
+
baseline: dict[str, Any] | None,
|
|
9
|
+
best: dict[str, Any] | None,
|
|
10
|
+
experiments: list[dict[str, Any]],
|
|
11
|
+
protected_integrity: bool = True,
|
|
12
|
+
) -> dict[str, Any]:
|
|
13
|
+
"""Projection-only developer state; never mutates repository or user projects."""
|
|
14
|
+
statuses = [str(item.get("status", "")) for item in experiments]
|
|
15
|
+
return {
|
|
16
|
+
"baseline": baseline,
|
|
17
|
+
"best": best,
|
|
18
|
+
"experiments": experiments,
|
|
19
|
+
"experiment_count": len(experiments),
|
|
20
|
+
"plateau": PlateauTracker().plateau(statuses),
|
|
21
|
+
"protected_integrity": protected_integrity,
|
|
22
|
+
"promotion": "branch_only",
|
|
23
|
+
}
|