eduevidence 5.2.0 → 6.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +54 -42
- package/README.zh-CN.md +51 -28
- package/SKILL.md +390 -133
- package/agents/openai.yaml +4 -0
- package/assets/readme/controlled-execution.svg +34 -0
- package/assets/readme/logo.png +0 -0
- package/assets/readme/research-workflow.svg +56 -0
- package/assets/readme/studio-graph.png +0 -0
- package/assets/readme/studio-overview.png +0 -0
- package/assets/readme/studio-reports.png +0 -0
- package/autoevolve/config.yaml +17 -0
- package/autoevolve/program.md +25 -0
- package/autoevolve/protected.manifest.yaml +34 -0
- package/benchmarks/adversarial/cases.jsonl +7 -0
- package/benchmarks/evidence-library.json +5268 -0
- package/benchmarks/partitions.json +8 -0
- package/docs/architecture.md +220 -0
- package/docs/autoresearch-evolution-plan.md +2903 -0
- package/docs/autoresearch-implementation-status.md +101 -0
- package/docs/demo-storyboard.md +20 -0
- package/docs/demo-workplace-ai.md +92 -0
- package/docs/demo.md +32 -0
- package/docs/install-guide.md +150 -0
- package/docs/orchestration-role-model.md +1254 -0
- package/docs/release-closeout/README.md +17 -0
- package/docs/release-closeout/frontend-acceptance.md +23 -0
- package/docs/release-closeout/issues.md +19 -0
- package/docs/release-closeout/verification.md +28 -0
- package/docs/release-contract.md +108 -0
- package/docs/research-studio-guide.zh-CN.md +166 -0
- package/eduevidence_cli.py +17 -11
- package/engine/_resources.py +13 -0
- package/engine/autoevolve/__init__.py +3 -0
- package/engine/autoevolve/agent_view.py +167 -0
- package/engine/autoevolve/core.py +357 -0
- package/engine/autoevolve/events.py +11 -0
- package/engine/autoevolve/git_workspace.py +77 -0
- package/engine/autoevolve/projection.py +23 -0
- package/engine/autoevolve/runner.py +413 -0
- package/engine/autoevolve/trust.py +146 -0
- package/engine/autoresearch/__init__.py +6 -0
- package/engine/autoresearch/commit.py +132 -0
- package/engine/autoresearch/contracts.py +126 -0
- package/engine/autoresearch/controller.py +207 -0
- package/engine/autoresearch/events.py +12 -0
- package/engine/autoresearch/gap_priority.py +168 -0
- package/engine/autoresearch/projection.py +30 -0
- package/engine/autoresearch/research_memory.py +59 -0
- package/engine/autoresearch/saturation.py +91 -0
- package/engine/briefs.py +2 -1
- package/engine/capabilities.py +1 -0
- package/engine/contracts.py +3 -1
- package/engine/evidencecore.py +7 -5
- package/engine/gaps.py +90 -51
- package/engine/judge_pack.py +65 -0
- package/engine/library_builtin.py +3 -1
- package/engine/living.py +2 -1
- package/engine/meta_synthesis.py +3 -1
- package/engine/orchestration.py +460 -0
- package/engine/pilot.py +2 -1
- package/engine/project.py +2 -2
- package/engine/research_service.py +113 -0
- package/engine/studio_read_model.py +400 -0
- package/engine/tribunal.py +1 -2
- package/engine/update.py +1 -0
- package/engine/versions.py +1 -1
- package/engine/worker_result.py +109 -0
- package/engine/workflows.py +70 -0
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1720 -0
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
- package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
- package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
- package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/frame.json +48 -0
- package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
- package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
- package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +219 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1720 -0
- package/examples/ai-coding-assistant-evidence/result.json +1453 -0
- package/examples/ai-coding-assistant-evidence/result.zh.json +1453 -0
- package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +103 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -0
- package/examples/workplace-ai-assistant/evaluation.json +19 -0
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
- package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
- package/examples/workplace-ai-assistant/frame.json +41 -0
- package/examples/workplace-ai-assistant/intervention.json +27 -0
- package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
- package/examples/workplace-ai-assistant/methodology.json +60 -0
- package/examples/workplace-ai-assistant/report_spec.json +55 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2484 -0
- package/examples/workplace-ai-assistant/result.json +553 -0
- package/examples/workplace-ai-assistant/result.zh.json +553 -0
- package/examples/workplace-ai-assistant/search_log.json +19 -0
- package/examples/workplace-ai-assistant/sources.jsonl +3 -0
- package/examples/workplace-ai-assistant/validation_result.json +9 -0
- package/examples/workplace-ai-assistant/verdict.json +52 -0
- package/install.sh +7 -7
- package/integrations/orchestration_dispatch.py +146 -0
- package/package.json +37 -3
- package/pyproject.toml +11 -20
- package/references/autoresearch.md +30 -0
- package/references/evaluation-policy.md +24 -0
- package/references/orchestration.md +22 -0
- package/references/scientific-invariants.md +19 -0
- package/retrieval/audit.py +154 -0
- package/schemas/intervention.schema.json +106 -21
- package/schemas/report-result.schema.json +9 -1
- package/schemas/v2/project.schema.json +2 -2
- package/schemas/v2/run.schema.json +1 -1
- package/schemas/vNext/autoevolve-session.schema.json +1 -0
- package/schemas/vNext/eval-snapshot.schema.json +1 -0
- package/schemas/vNext/execution-plan.schema.json +1 -0
- package/schemas/vNext/gap-priority.schema.json +1 -0
- package/schemas/vNext/negative-search-record.schema.json +1 -0
- package/schemas/vNext/research-iteration.schema.json +1 -0
- package/schemas/vNext/research-strategy.schema.json +1 -0
- package/schemas/vNext/skill-experiment.schema.json +1 -0
- package/schemas/vNext/task-spec.schema.json +1 -0
- package/schemas/vNext/worker-result.schema.json +1 -0
- package/scripts/benchmark_judge.py +2 -2
- package/scripts/benchmark_v3.py +26 -43
- package/scripts/build_esl_artifacts.py +2 -2
- package/scripts/build_evidence_library.py +2 -2
- package/scripts/build_gh_pages.py +98 -0
- package/scripts/build_readme_diagrams.py +72 -0
- package/scripts/build_report_variants.py +85 -0
- package/scripts/check_autoresearch_invariants.py +95 -0
- package/scripts/daily_evolve.py +30 -0
- package/scripts/dashboard_server.py +130 -101
- package/scripts/did_regression.py +5 -30
- package/scripts/enrich_projects_human_and_lieflat.py +1 -1
- package/scripts/generate_metrics.py +4 -3
- package/scripts/generate_new_projects.py +1 -1
- package/scripts/orchestrator.py +172 -18
- package/scripts/rebake_all_5themes.py +1 -2
- package/scripts/research_auto_cli.py +475 -0
- package/scripts/run_workspace.py +17 -7
- package/scripts/search_provenance.py +64 -0
- package/scripts/serve_web.py +9 -10
- package/scripts/skill_lint.py +1 -1
- package/scripts/skill_payload.py +78 -0
- package/scripts/validate_schema.py +15 -1
- package/scripts/vnext_cli.py +133 -0
- package/setup.py +12 -0
- package/skill/roles/registry.yaml +45 -0
- package/skill/sub-skills/report-generation/SKILL.md +12 -6
- package/skill/task-briefs/applicability.md +3 -0
- package/skill/task-briefs/projection.md +3 -0
- package/skill/workflows/decision-and-pilot.md +10 -0
- package/skill/workflows/evaluate-and-update.md +10 -0
- package/skill/workflows/evidence-review.md +13 -0
- package/visualization/eduevidence-report/assets/base.css +2 -2
- package/visualization/eduevidence-report/assets/reader.css +752 -0
- package/visualization/eduevidence-report/assets/reader.js +132 -0
- package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
- package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
- package/visualization/eduevidence-report/scripts/build_report.py +58 -65
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +24 -100
- package/visualization/eduevidence-report/themes/academic.css +1 -1
- package/visualization/eduevidence-report/themes/claude.css +1 -1
- package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
- package/visualization/eduevidence-report/themes/datalab.css +2 -2
- package/visualization/eduevidence-report/themes/presentation.css +2 -2
- package/web/README.md +18 -0
- package/web/index.html +53 -0
- package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
- package/web/studio/assets/index-CzXocaGv.css +1 -0
- package/web/studio/assets/index-pa7jD7n4.js +230 -0
- package/web/studio/config.json +1 -0
- package/web/studio/index.html +14 -0
- package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
- package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/bias.cpython-312.pyc +0 -0
- package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
- package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
- package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
- package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
- package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
- package/engine/__pycache__/events.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
- package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
- package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
- package/engine/__pycache__/ids.cpython-312.pyc +0 -0
- package/engine/__pycache__/library.cpython-312.pyc +0 -0
- package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
- package/engine/__pycache__/living.cpython-312.pyc +0 -0
- package/engine/__pycache__/log.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/migration.cpython-312.pyc +0 -0
- package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
- package/engine/__pycache__/paths.cpython-312.pyc +0 -0
- package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
- package/engine/__pycache__/planner.cpython-312.pyc +0 -0
- package/engine/__pycache__/project.cpython-312.pyc +0 -0
- package/engine/__pycache__/projections.cpython-312.pyc +0 -0
- package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
- package/engine/__pycache__/run.cpython-312.pyc +0 -0
- package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
- package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
- package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
- package/engine/__pycache__/update.cpython-312.pyc +0 -0
- package/engine/__pycache__/versions.cpython-312.pyc +0 -0
- package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
- package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
- package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
- package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
- package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
- package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
- package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
- package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
- package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
- package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
- package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
- package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
- package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
- package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
- package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
- package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
|
@@ -5,42 +5,127 @@
|
|
|
5
5
|
"description": "A minimal verifiable pilot intervention. EduEvidence NEVER recommends full deployment directly — it always produces a small-scale, evaluable pilot with stop conditions. 扩展字段一律放在 extensions 内。",
|
|
6
6
|
"type": "object",
|
|
7
7
|
"additionalProperties": false,
|
|
8
|
-
"required": [
|
|
8
|
+
"required": [
|
|
9
|
+
"decision",
|
|
10
|
+
"pilot_duration",
|
|
11
|
+
"stop_conditions"
|
|
12
|
+
],
|
|
9
13
|
"properties": {
|
|
10
|
-
"decision": {
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
+
"decision": {
|
|
15
|
+
"type": "string",
|
|
16
|
+
"enum": [
|
|
17
|
+
"adopt",
|
|
18
|
+
"pilot",
|
|
19
|
+
"reject",
|
|
20
|
+
"insufficient_evidence"
|
|
21
|
+
]
|
|
22
|
+
},
|
|
23
|
+
"target_learners": {
|
|
24
|
+
"type": "string"
|
|
25
|
+
},
|
|
26
|
+
"learning_goals": {
|
|
27
|
+
"type": "array",
|
|
28
|
+
"items": {
|
|
29
|
+
"type": "string"
|
|
30
|
+
}
|
|
31
|
+
},
|
|
32
|
+
"pilot_duration": {
|
|
33
|
+
"type": "string",
|
|
34
|
+
"examples": [
|
|
35
|
+
"8_weeks"
|
|
36
|
+
]
|
|
37
|
+
},
|
|
14
38
|
"phase_1": {
|
|
15
39
|
"type": "object",
|
|
16
40
|
"additionalProperties": true,
|
|
17
41
|
"description": "Phase structure: name, activities, ai_usage_rule, outcome_check.",
|
|
18
42
|
"properties": {
|
|
19
|
-
"name": {
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
"
|
|
43
|
+
"name": {
|
|
44
|
+
"type": "string"
|
|
45
|
+
},
|
|
46
|
+
"activities": {
|
|
47
|
+
"type": "array",
|
|
48
|
+
"items": {
|
|
49
|
+
"type": "string"
|
|
50
|
+
}
|
|
51
|
+
},
|
|
52
|
+
"ai_usage_rule": {
|
|
53
|
+
"type": "string"
|
|
54
|
+
},
|
|
55
|
+
"outcome_check": {
|
|
56
|
+
"type": "string"
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
},
|
|
60
|
+
"phase_2": {
|
|
61
|
+
"type": "object",
|
|
62
|
+
"additionalProperties": true
|
|
63
|
+
},
|
|
64
|
+
"phase_3": {
|
|
65
|
+
"type": "object",
|
|
66
|
+
"additionalProperties": true
|
|
67
|
+
},
|
|
68
|
+
"phase_4": {
|
|
69
|
+
"type": "object",
|
|
70
|
+
"additionalProperties": true
|
|
71
|
+
},
|
|
72
|
+
"ai_usage_policy": {
|
|
73
|
+
"type": "string",
|
|
74
|
+
"description": "Explicit allowed/forbidden AI usage rules for students."
|
|
75
|
+
},
|
|
76
|
+
"teacher_role": {
|
|
77
|
+
"type": "string"
|
|
78
|
+
},
|
|
79
|
+
"student_role": {
|
|
80
|
+
"type": "string"
|
|
81
|
+
},
|
|
82
|
+
"reflection_requirement": {
|
|
83
|
+
"type": "string",
|
|
84
|
+
"description": "e.g. students must explain key logic generated by AI."
|
|
85
|
+
},
|
|
86
|
+
"assessment": {
|
|
87
|
+
"type": "string"
|
|
88
|
+
},
|
|
89
|
+
"risk_control": {
|
|
90
|
+
"type": "array",
|
|
91
|
+
"items": {
|
|
92
|
+
"type": "string"
|
|
93
|
+
}
|
|
94
|
+
},
|
|
95
|
+
"stop_conditions": {
|
|
96
|
+
"type": "array",
|
|
97
|
+
"items": {
|
|
98
|
+
"type": "string"
|
|
23
99
|
}
|
|
24
100
|
},
|
|
25
|
-
"phase_2": { "type": "object", "additionalProperties": true },
|
|
26
|
-
"phase_3": { "type": "object", "additionalProperties": true },
|
|
27
|
-
"phase_4": { "type": "object", "additionalProperties": true },
|
|
28
|
-
"ai_usage_policy": { "type": "string", "description": "Explicit allowed/forbidden AI usage rules for students." },
|
|
29
|
-
"teacher_role": { "type": "string" },
|
|
30
|
-
"student_role": { "type": "string" },
|
|
31
|
-
"reflection_requirement": { "type": "string", "description": "e.g. students must explain key logic generated by AI." },
|
|
32
|
-
"assessment": { "type": "string" },
|
|
33
|
-
"risk_control": { "type": "array", "items": { "type": "string" } },
|
|
34
|
-
"stop_conditions": { "type": "array", "items": { "type": "string" } },
|
|
35
101
|
"evidence_alignment": {
|
|
36
102
|
"type": "array",
|
|
37
|
-
"items": {
|
|
103
|
+
"items": {
|
|
104
|
+
"type": "string"
|
|
105
|
+
},
|
|
38
106
|
"description": "evidence_id(s) that this intervention traces back to."
|
|
39
107
|
},
|
|
40
108
|
"extensions": {
|
|
41
109
|
"type": "object",
|
|
42
110
|
"description": "结构化扩展字段的统一容器(P1-01)。未列入本 schema 的字段必须放在这里,禁止在顶层新增属性。",
|
|
43
111
|
"additionalProperties": true
|
|
112
|
+
},
|
|
113
|
+
"target_population": {
|
|
114
|
+
"type": "string",
|
|
115
|
+
"minLength": 1,
|
|
116
|
+
"description": "Target population for a non-teaching intervention."
|
|
117
|
+
}
|
|
118
|
+
},
|
|
119
|
+
"anyOf": [
|
|
120
|
+
{
|
|
121
|
+
"required": [
|
|
122
|
+
"target_learners"
|
|
123
|
+
]
|
|
124
|
+
},
|
|
125
|
+
{
|
|
126
|
+
"required": [
|
|
127
|
+
"target_population"
|
|
128
|
+
]
|
|
44
129
|
}
|
|
45
|
-
|
|
130
|
+
]
|
|
46
131
|
}
|
|
@@ -49,6 +49,14 @@
|
|
|
49
49
|
"manual_curated"
|
|
50
50
|
],
|
|
51
51
|
"description": "Provenance of the pack data (plan R4). real=pipeline-generated from real studies; manual_curated=hand-curated from real verified studies; synthetic=demonstration content, not real studies; hybrid=mixed."
|
|
52
|
+
},
|
|
53
|
+
"domain": {
|
|
54
|
+
"type": "string",
|
|
55
|
+
"enum": [
|
|
56
|
+
"education",
|
|
57
|
+
"policy"
|
|
58
|
+
],
|
|
59
|
+
"description": "Registered research domain; policy covers non-teaching organizational and public-policy interventions."
|
|
52
60
|
}
|
|
53
61
|
}
|
|
54
62
|
},
|
|
@@ -378,4 +386,4 @@
|
|
|
378
386
|
"description": "Optional pooled meta-analysis block (synthetic demo packs only unless engine-generated)."
|
|
379
387
|
}
|
|
380
388
|
}
|
|
381
|
-
}
|
|
389
|
+
}
|
|
@@ -13,9 +13,9 @@
|
|
|
13
13
|
"properties": {
|
|
14
14
|
"project_id": { "type": "string", "pattern": "^PRJ-" },
|
|
15
15
|
"title": { "type": "string", "minLength": 1 },
|
|
16
|
-
"domain": { "type": "string", "enum": ["education"] },
|
|
16
|
+
"domain": { "type": "string", "enum": ["education", "policy"] },
|
|
17
17
|
"question": { "type": "string", "minLength": 1 },
|
|
18
|
-
"research_mode": { "type": "string", "enum": ["evidence_review", "full_research_cycle"] },
|
|
18
|
+
"research_mode": { "type": "string", "enum": ["evidence_review", "decision_and_pilot", "evaluate_and_update", "full_research_cycle"] },
|
|
19
19
|
"decision_target": {
|
|
20
20
|
"type": "string",
|
|
21
21
|
"enum": ["evidence_review", "teaching_decision", "teaching_pilot", "evaluation_plan", "research_cycle"]
|
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
"project_id": { "type": "string", "pattern": "^PRJ-" },
|
|
16
16
|
"purpose": { "type": "string", "minLength": 1 },
|
|
17
17
|
"started_at": { "type": "string", "format": "date-time" },
|
|
18
|
-
"status": { "type": "string", "enum": ["running", "completed", "failed", "aborted"] },
|
|
18
|
+
"status": { "type": "string", "enum": ["queued", "running", "waiting_for_user", "waiting_for_user_data", "waiting_for_executor", "waiting_for_tool", "waiting_for_review", "blocked_scientific_gate", "blocked_contract_error", "completed", "failed_recoverable", "failed_terminal", "cancelled", "failed", "aborted"] },
|
|
19
19
|
"graph_revision_before": { "type": "integer", "minimum": 0 },
|
|
20
20
|
"graph_revision_after": { "type": ["integer", "null"], "minimum": 0 },
|
|
21
21
|
"capabilities": {
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"$schema":"http://json-schema.org/draft-07/schema#","type":"object","required":["session_id","status","max_experiments","promotion","experiments"],"properties":{"session_id":{"type":"string"},"status":{"enum":["initialized","running","plateau","budget_exhausted","completed","blocked"]},"max_experiments":{"type":"integer","minimum":1,"maximum":50},"max_cost_usd":{"type":"number","minimum":0},"max_wall_minutes":{"type":"integer","minimum":1},"promotion":{"const":"branch_only"},"experiments":{"type":"array","items":{"type":"string"}},"best_experiment_id":{"type":["string","null"]},"protected_integrity":{"type":"boolean"}},"additionalProperties":false}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"$schema":"http://json-schema.org/draft-07/schema#","type":"object","required":["eval_id","hard_gates_passed","science_score","research_score","robustness","cost","latency","complexity","repeats","noise_floor","dev_passed","holdout_passed","adversarial_passed","holdout_isolation_verified","eval_suite_hash"],"properties":{"eval_id":{"type":"string"},"hard_gates_passed":{"type":"boolean"},"science_score":{"type":"number"},"research_score":{"type":"number"},"robustness":{"type":"number"},"cost":{"type":"number","minimum":0},"latency":{"type":"number","minimum":0},"complexity":{"type":"number","minimum":0},"repeats":{"type":"integer","minimum":1},"noise_floor":{"type":"number","minimum":0},"dev_passed":{"type":"boolean"},"holdout_passed":{"type":"boolean"},"adversarial_passed":{"type":"boolean"},"holdout_isolation_verified":{"type":"boolean"},"eval_suite_hash":{"type":"string","minLength":1}},"additionalProperties":false}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"$schema":"http://json-schema.org/draft-07/schema#","type":"object","required":["complexity","tasks","max_parallel_workers","parallel_groups","plan_id"],"properties":{"complexity":{"enum":["S","M","L"]},"tasks":{"type":"array","items":{"$ref":"task-spec.schema.json"}},"max_parallel_workers":{"type":"integer","minimum":0,"maximum":6},"parallel_groups":{"type":"array","items":{"type":"array","minItems":1,"items":{"type":"string"}}},"plan_id":{"type":["string","null"]}},"additionalProperties":false}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"$schema":"http://json-schema.org/draft-07/schema#","type":"object","required":["gap_id","dvi_band","cost_band","decision_material","drivers","next_research_mode","score"],"properties":{"gap_id":{"type":"string"},"dvi_band":{"enum":["HIGH","MEDIUM","LOW"]},"cost_band":{"enum":["HIGH","MEDIUM","LOW"]},"decision_material":{"type":"boolean"},"drivers":{"type":"array","items":{"type":"string"}},"next_research_mode":{"enum":["secondary_evidence_search","defer","empirical_evidence_needed"]},"score":{"type":"integer"}},"additionalProperties":false}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"$schema":"http://json-schema.org/draft-07/schema#","type":"object","required":["negative_search_id","research_iteration_id","gap_id","queries","providers","candidate_count","fetched_count","eligible_count","conclusion"],"properties":{"negative_search_id":{"type":"string"},"research_iteration_id":{"type":"string"},"gap_id":{"type":"string"},"queries":{"type":"array","items":{"type":"string"}},"providers":{"type":"array","items":{"type":"string"}},"candidate_count":{"type":"integer","minimum":0},"fetched_count":{"type":"integer","minimum":0},"eligible_count":{"const":0},"exclusion_reasons":{"type":"object","additionalProperties":{"type":"integer","minimum":0}},"scope":{"type":"object"},"searched_at":{"type":"string"},"conclusion":{"const":"no_eligible_evidence_found_within_search_scope"}},"additionalProperties":false}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"$schema":"http://json-schema.org/draft-07/schema#","type":"object","required":["iteration_id","project_id","base_graph_revision","gap_id","strategy","status"],"properties":{"iteration_id":{"type":"string"},"project_id":{"type":"string"},"base_graph_revision":{"type":"integer","minimum":0},"gap_id":{"type":"string"},"gap_lineage_key":{"type":["string","null"],"pattern":"^KGK-"},"strategy":{"type":"object"},"validated_evidence_ids":{"type":"array","items":{"type":"string"}},"negative_search_ids":{"type":"array","items":{"type":"string"}},"evidence_gain":{"type":"object"},"new_graph_revision":{"type":["integer","null"]},"decision_snapshot_id":{"type":["string","null"]},"status":{"enum":["completed_gain","completed_no_gain","search_saturated","empirical_needed","budget_exhausted","tool_failure","invalid"]},"started_at":{"type":"string"},"completed_at":{"type":["string","null"]}},"additionalProperties":true}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"$schema":"http://json-schema.org/draft-07/schema#","type":"object","required":["strategy_id","experiment_type","hypothesis","expected_gain","budget"],"properties":{"strategy_id":{"type":"string","minLength":1},"experiment_type":{"enum":["TARGETED_RETRIEVAL","COUNTER_EVIDENCE_RETRIEVAL","APPLICABILITY_RETRIEVAL","TEMPORAL_REFRESH","CITATION_CHAINING","SCREENING_PRIORITY","SOURCE_RECOVERY"]},"hypothesis":{"type":"string","minLength":1},"expected_gain":{"type":"string","minLength":1},"budget":{"type":"object","required":["max_queries","max_candidates","max_fulltext_fetches"],"properties":{"max_queries":{"type":"integer","minimum":0},"max_candidates":{"type":"integer","minimum":0},"max_fulltext_fetches":{"type":"integer","minimum":0}},"additionalProperties":false}},"additionalProperties":false}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"$schema":"http://json-schema.org/draft-07/schema#","type":"object","required":["experiment_id","session_id","parent_skill_revision","hypothesis","mutation_scope","status"],"properties":{"experiment_id":{"type":"string"},"session_id":{"type":"string"},"parent_skill_revision":{"type":"string"},"hypothesis":{"type":"string","minLength":1},"mutation_scope":{"type":"array","items":{"type":"string"},"minItems":1},"changed_files":{"type":"array","items":{"type":"string"}},"candidate_commit":{"type":["string","null"]},"baseline_eval_id":{"type":["string","null"]},"candidate_eval_id":{"type":["string","null"]},"protected_hash_before":{"type":["string","null"]},"protected_hash_after":{"type":["string","null"]},"status":{"enum":["created","KEEP","REJECT","RETEST","HUMAN_REVIEW","CRASH","INVALID"]},"promotion_reason":{"type":"string"},"complexity_delta":{"type":"number"}},"additionalProperties":false}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"$schema":"http://json-schema.org/draft-07/schema#","type":"object","required":["task_id","run_id","base_revision","stage","role","role_profile","objective","reason_for_delegation","evidence_axis","inputs","input_artifacts","allowed_capabilities","forbidden_actions","scope","budget","expected_outputs","output_contract","termination","execution_mode","independent","read_only","timeout_seconds","token_budget","metadata"],"properties":{"task_id":{"type":"string","minLength":1},"run_id":{"type":["string","null"]},"base_revision":{"type":["integer","null"],"minimum":0},"stage":{"enum":["frame","retrieve","extract","challenge","audit","adjudicate","applicability","intervene","evaluate"]},"role":{"type":"string"},"role_profile":{"type":["string","null"]},"objective":{"type":"string","minLength":1},"reason_for_delegation":{"type":["string","null"]},"evidence_axis":{"type":"string","minLength":1},"inputs":{"type":"array","items":{"type":"string"}},"input_artifacts":{"type":"array","items":{"type":"string"}},"allowed_capabilities":{"type":"array","items":{"type":"string"}},"forbidden_actions":{"type":"array","items":{"type":"string"}},"scope":{"type":"object"},"budget":{"type":"object"},"expected_outputs":{"type":"array","items":{"type":"string"}},"output_contract":{"type":"object"},"termination":{"type":"object"},"execution_mode":{"enum":["local","delegated"]},"independent":{"type":"boolean"},"read_only":{"type":"boolean"},"timeout_seconds":{"type":"integer","minimum":1},"token_budget":{"type":["integer","null"],"minimum":1},"metadata":{"type":"object"}},"additionalProperties":false}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"$schema":"http://json-schema.org/draft-07/schema#","type":"object","required":["task_id","status","staging_artifacts","validated","validation_issues","metrics","summary"],"properties":{"task_id":{"type":"string","minLength":1},"status":{"enum":["completed","failed","blocked"]},"staging_artifacts":{"type":"array","items":{"type":"object","required":["artifact_type"],"properties":{"artifact_type":{"type":"string","minLength":1}},"additionalProperties":true}},"validated":{"type":"boolean"},"validation_issues":{"type":"array","items":{"type":"string"}},"metrics":{"type":"object"},"summary":{"type":"string"}},"additionalProperties":false}
|
|
@@ -45,7 +45,7 @@ from benchmark_evaluator import extract_json_block # noqa: E402
|
|
|
45
45
|
|
|
46
46
|
JUDGE_DIMS = ("citation_support", "outcome_correctness", "scope_calibration",
|
|
47
47
|
"contradiction_handling", "decision_calibration")
|
|
48
|
-
DEFAULT_JUDGE_MODEL = "
|
|
48
|
+
DEFAULT_JUDGE_MODEL = "" # 无默认:judge 模型必须显式指定或经 EDUEVIDENCE_LLM_MODEL 提供
|
|
49
49
|
DEFAULT_LIMIT = 60
|
|
50
50
|
HEURISTIC_METRICS = ("outcome_separation_accuracy", "decision_calibration",
|
|
51
51
|
"contradiction_recall", "contradiction_precision",
|
|
@@ -514,7 +514,7 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
514
514
|
p_run.add_argument("--questions", default="benchmarks/questions.jsonl")
|
|
515
515
|
p_run.add_argument("--annotations", default="benchmarks/annotations")
|
|
516
516
|
p_run.add_argument("--model", default=DEFAULT_JUDGE_MODEL)
|
|
517
|
-
p_run.add_argument("--thinking", default="
|
|
517
|
+
p_run.add_argument("--thinking", default="max", choices=["low", "high", "max"])
|
|
518
518
|
p_run.add_argument("--limit", type=int, default=DEFAULT_LIMIT,
|
|
519
519
|
help="max completed attempts to judge (default 60; <=0 = unlimited)")
|
|
520
520
|
p_run.set_defaults(func=_cmd_run)
|
package/scripts/benchmark_v3.py
CHANGED
|
@@ -31,6 +31,7 @@ from __future__ import annotations
|
|
|
31
31
|
import argparse
|
|
32
32
|
import json
|
|
33
33
|
import os
|
|
34
|
+
import subprocess
|
|
34
35
|
import sys
|
|
35
36
|
import urllib.error
|
|
36
37
|
import urllib.request
|
|
@@ -48,8 +49,6 @@ BASELINES = (
|
|
|
48
49
|
)
|
|
49
50
|
DEFAULT_BUDGET_TOKENS = 1_000_000
|
|
50
51
|
|
|
51
|
-
# ---------------------------------------------------------------- prompts
|
|
52
|
-
|
|
53
52
|
|
|
54
53
|
def _prompt_b0(q: dict) -> str:
|
|
55
54
|
return (
|
|
@@ -118,9 +117,6 @@ def build_prompt(baseline: str, q: dict) -> str:
|
|
|
118
117
|
return fn(q)
|
|
119
118
|
|
|
120
119
|
|
|
121
|
-
# ---------------------------------------------------------------- drivers
|
|
122
|
-
|
|
123
|
-
|
|
124
120
|
class ApiDriver:
|
|
125
121
|
"""OpenAI-compatible chat completions driver (no SDK dependency)."""
|
|
126
122
|
|
|
@@ -152,7 +148,7 @@ class ApiDriver:
|
|
|
152
148
|
headers={"Content-Type": "application/json",
|
|
153
149
|
"Authorization": f"Bearer {self.api_key}"},
|
|
154
150
|
method="POST")
|
|
155
|
-
with urllib.request.urlopen(req, timeout=self.timeout) as resp:
|
|
151
|
+
with urllib.request.urlopen(req, timeout=self.timeout) as resp:
|
|
156
152
|
payload = json.loads(resp.read().decode("utf-8"))
|
|
157
153
|
usage = payload.get("usage") or {}
|
|
158
154
|
text = (payload.get("choices") or [{}])[0].get("message", {}).get("content", "")
|
|
@@ -166,28 +162,21 @@ class ApiDriver:
|
|
|
166
162
|
|
|
167
163
|
|
|
168
164
|
class CliDriver:
|
|
169
|
-
"""omp CLI driver - host agent runtime (user-approved).
|
|
170
|
-
|
|
171
|
-
Calls `omp -p --no-session --model=<model> <prompt>` in a scratch dir;
|
|
172
|
-
captures stdout as the response. Token usage is estimated from text
|
|
173
|
-
length and recorded as such (manifest usage fields may stay null; the
|
|
174
|
-
run manifest environment records the exact invocation).
|
|
175
|
-
"""
|
|
165
|
+
"""omp CLI driver - host agent runtime (user-approved)."""
|
|
176
166
|
|
|
177
167
|
name = "cli"
|
|
178
168
|
|
|
179
|
-
def __init__(self, model: str | None = None, thinking: str = "
|
|
169
|
+
def __init__(self, model: str | None = None, thinking: str = "max",
|
|
180
170
|
timeout: int = 600):
|
|
181
|
-
self.model = model or os.environ.get("EDUEVIDENCE_LLM_MODEL", "
|
|
171
|
+
self.model = model or os.environ.get("EDUEVIDENCE_LLM_MODEL", "")
|
|
182
172
|
self.thinking = thinking
|
|
183
173
|
self.timeout = timeout
|
|
184
174
|
|
|
185
175
|
def available(self) -> bool:
|
|
186
176
|
import shutil
|
|
187
|
-
return shutil.which("omp") is not None
|
|
177
|
+
return bool(self.model) and shutil.which("omp") is not None
|
|
188
178
|
|
|
189
179
|
def call(self, prompt: str, *, no_tools: bool = False) -> tuple[str, dict]:
|
|
190
|
-
import subprocess
|
|
191
180
|
import tempfile
|
|
192
181
|
import time
|
|
193
182
|
|
|
@@ -200,7 +189,7 @@ class CliDriver:
|
|
|
200
189
|
t0 = time.monotonic()
|
|
201
190
|
with tempfile.TemporaryDirectory(prefix="eduevidence-bench-") as workdir:
|
|
202
191
|
proc = subprocess.run(cmd, capture_output=True, text=True,
|
|
203
|
-
|
|
192
|
+
timeout=self.timeout, cwd=workdir)
|
|
204
193
|
latency = time.monotonic() - t0
|
|
205
194
|
if proc.returncode != 0:
|
|
206
195
|
raise RuntimeError(
|
|
@@ -214,6 +203,7 @@ class CliDriver:
|
|
|
214
203
|
}
|
|
215
204
|
return text, usage
|
|
216
205
|
|
|
206
|
+
|
|
217
207
|
class SimDriver:
|
|
218
208
|
"""Deterministic simulation — harness validation ONLY. Never performance evidence."""
|
|
219
209
|
|
|
@@ -226,10 +216,8 @@ class SimDriver:
|
|
|
226
216
|
return True
|
|
227
217
|
|
|
228
218
|
def call(self, prompt: str, *, no_tools: bool = False) -> tuple[str, dict[str, Any]]:
|
|
229
|
-
from benchmark_v2 import simulate_question_result # noqa:
|
|
219
|
+
from benchmark_v2 import simulate_question_result # noqa: F401
|
|
230
220
|
|
|
231
|
-
# Deterministic pseudo-usage from prompt length; response is a stub
|
|
232
|
-
# that the evaluator must never use as model performance.
|
|
233
221
|
import random
|
|
234
222
|
rng = random.Random(len(prompt) * 7919 % 2**31)
|
|
235
223
|
usage = {
|
|
@@ -246,28 +234,26 @@ class SimDriver:
|
|
|
246
234
|
)
|
|
247
235
|
|
|
248
236
|
|
|
249
|
-
def make_driver(name: str) -> Any:
|
|
237
|
+
def make_driver(name: str, *, model: str | None = None, thinking: str = "max") -> Any:
|
|
250
238
|
if name == "api":
|
|
251
239
|
return ApiDriver()
|
|
252
240
|
if name == "cli":
|
|
253
|
-
return CliDriver()
|
|
241
|
+
return CliDriver(model=model, thinking=thinking)
|
|
254
242
|
if name == "sim":
|
|
255
243
|
return SimDriver()
|
|
256
244
|
raise ValueError(f"unknown driver: {name}")
|
|
257
245
|
|
|
258
246
|
|
|
259
|
-
# ---------------------------------------------------------------- run
|
|
260
|
-
|
|
261
|
-
|
|
262
247
|
def _now_iso() -> str:
|
|
263
248
|
return datetime.now(timezone.utc).isoformat()
|
|
264
249
|
|
|
265
250
|
|
|
266
251
|
def run_benchmark(*, questions: list[dict], baselines: list[str], repeats: int,
|
|
267
252
|
out_dir: Path, driver_name: str, budget_tokens: int | None,
|
|
268
|
-
temperature: float = 0.0, resume: bool = False
|
|
253
|
+
temperature: float = 0.0, resume: bool = False,
|
|
254
|
+
model: str | None = None, thinking: str = "max") -> dict[str, Any]:
|
|
269
255
|
out_dir.mkdir(parents=True, exist_ok=True)
|
|
270
|
-
driver = make_driver(driver_name)
|
|
256
|
+
driver = make_driver(driver_name, model=model, thinking=thinking)
|
|
271
257
|
if not driver.available():
|
|
272
258
|
raise RuntimeError(
|
|
273
259
|
f"driver '{driver_name}' unavailable (api needs EDUEVIDENCE_LLM_MODEL "
|
|
@@ -303,8 +289,6 @@ def run_benchmark(*, questions: list[dict], baselines: list[str], repeats: int,
|
|
|
303
289
|
total_tokens = 0
|
|
304
290
|
budget_stopped = False
|
|
305
291
|
import re as _re
|
|
306
|
-
# --resume: reuse previously completed attempts (their response artifacts
|
|
307
|
-
# live in out_dir); only unfinished attempts are re-run.
|
|
308
292
|
done_ids: set[str] = set()
|
|
309
293
|
resumed: dict[str, dict[str, Any]] = {}
|
|
310
294
|
if resume and out_dir.is_dir():
|
|
@@ -315,7 +299,7 @@ def run_benchmark(*, questions: list[dict], baselines: list[str], repeats: int,
|
|
|
315
299
|
except (OSError, _json.JSONDecodeError):
|
|
316
300
|
continue
|
|
317
301
|
aid = data.get("attempt_id")
|
|
318
|
-
if aid and
|
|
302
|
+
if aid and art.is_file():
|
|
319
303
|
done_ids.add(aid)
|
|
320
304
|
resumed[aid] = data
|
|
321
305
|
if done_ids:
|
|
@@ -333,9 +317,6 @@ def run_benchmark(*, questions: list[dict], baselines: list[str], repeats: int,
|
|
|
333
317
|
break
|
|
334
318
|
attempt_id = f"{question['id']}-{baseline}-a{attempt}"
|
|
335
319
|
if attempt_id in done_ids:
|
|
336
|
-
# Re-register resumed attempts in the manifest (status +
|
|
337
|
-
# usage read back from their artifact) so eval/report see
|
|
338
|
-
# the complete run.
|
|
339
320
|
art = out_dir / f"{attempt_id}.response.json"
|
|
340
321
|
data = resumed.get(attempt_id, {})
|
|
341
322
|
usage = data.get("usage") or {}
|
|
@@ -393,7 +374,7 @@ def run_benchmark(*, questions: list[dict], baselines: list[str], repeats: int,
|
|
|
393
374
|
}, ensure_ascii=False, indent=2), encoding="utf-8")
|
|
394
375
|
entry["artifacts"] = [artifact.name]
|
|
395
376
|
except (urllib.error.URLError, OSError, ValueError, KeyError,
|
|
396
|
-
subprocess.TimeoutExpired) as exc:
|
|
377
|
+
subprocess.TimeoutExpired) as exc:
|
|
397
378
|
entry.update({"status": "failed", "error": str(exc),
|
|
398
379
|
"finished_at": _now_iso()})
|
|
399
380
|
manifest["attempts"].append(entry)
|
|
@@ -405,8 +386,6 @@ def run_benchmark(*, questions: list[dict], baselines: list[str], repeats: int,
|
|
|
405
386
|
break
|
|
406
387
|
|
|
407
388
|
if budget_stopped:
|
|
408
|
-
# P2-2: record remaining attempts as budget_stopped so the report can
|
|
409
|
-
# distinguish "stopped by budget" from "never scheduled".
|
|
410
389
|
for question in questions:
|
|
411
390
|
if any(a["question_id"] == question["id"] for a in manifest["attempts"]):
|
|
412
391
|
continue
|
|
@@ -429,7 +408,7 @@ def run_benchmark(*, questions: list[dict], baselines: list[str], repeats: int,
|
|
|
429
408
|
tmp = out_dir / "manifest.json.tmp"
|
|
430
409
|
tmp.write_text(json.dumps(manifest, ensure_ascii=False, indent=2) + "\n",
|
|
431
410
|
encoding="utf-8")
|
|
432
|
-
tmp.replace(manifest_path)
|
|
411
|
+
tmp.replace(manifest_path)
|
|
433
412
|
_validate_manifest(manifest_path)
|
|
434
413
|
print(f"wrote {manifest_path} (attempts={len(manifest['attempts'])}, "
|
|
435
414
|
f"mode={manifest['run_mode']}, total_tokens~{total_tokens})")
|
|
@@ -444,13 +423,13 @@ def _questions_version() -> str:
|
|
|
444
423
|
proc = _sp.run(["git", "-C", str(repo), "rev-parse", "--short", "HEAD"],
|
|
445
424
|
capture_output=True, text=True, timeout=10)
|
|
446
425
|
return proc.stdout.strip() or "unknown"
|
|
447
|
-
except Exception:
|
|
426
|
+
except Exception:
|
|
448
427
|
return "unknown"
|
|
449
428
|
|
|
450
429
|
|
|
451
430
|
def _validate_manifest(path: Path) -> None:
|
|
452
431
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
453
|
-
from validate_schema import Validator, SchemaError
|
|
432
|
+
from validate_schema import Validator, SchemaError
|
|
454
433
|
|
|
455
434
|
import json as _json
|
|
456
435
|
schema = _json.loads(
|
|
@@ -479,12 +458,12 @@ def _cmd_run(args: argparse.Namespace) -> int:
|
|
|
479
458
|
baselines=baselines, repeats=args.repeats,
|
|
480
459
|
out_dir=Path(args.out), driver_name=args.driver,
|
|
481
460
|
budget_tokens=args.budget_tokens, temperature=args.temperature,
|
|
482
|
-
resume=args.resume)
|
|
461
|
+
resume=args.resume, model=args.model, thinking=args.thinking)
|
|
483
462
|
return 0
|
|
484
463
|
|
|
485
464
|
|
|
486
465
|
def _cmd_report(args: argparse.Namespace) -> int:
|
|
487
|
-
from benchmark_evaluator import report_from_run
|
|
466
|
+
from benchmark_evaluator import report_from_run
|
|
488
467
|
|
|
489
468
|
run_dir = Path(args.run)
|
|
490
469
|
manifest = json.loads((run_dir / "manifest.json").read_text(encoding="utf-8"))
|
|
@@ -508,6 +487,10 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
508
487
|
p_run.add_argument("--out", required=True)
|
|
509
488
|
p_run.add_argument("--budget-tokens", type=int, default=DEFAULT_BUDGET_TOKENS)
|
|
510
489
|
p_run.add_argument("--temperature", type=float, default=0.0)
|
|
490
|
+
p_run.add_argument("--model", default="",
|
|
491
|
+
help="OMP model for --driver cli (required; env EDUEVIDENCE_LLM_MODEL accepted; no unconfirmed default)")
|
|
492
|
+
p_run.add_argument("--thinking", default="max", choices=["low", "high", "max"],
|
|
493
|
+
help="reasoning effort for --driver cli")
|
|
511
494
|
p_run.add_argument("--resume", action="store_true",
|
|
512
495
|
help="skip attempts whose response artifacts already exist in --out")
|
|
513
496
|
p_run.set_defaults(func=_cmd_run)
|
|
@@ -536,7 +519,7 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
536
519
|
|
|
537
520
|
|
|
538
521
|
def _cmd_eval(args: argparse.Namespace) -> int:
|
|
539
|
-
from benchmark_evaluator import evaluate_run
|
|
522
|
+
from benchmark_evaluator import evaluate_run
|
|
540
523
|
|
|
541
524
|
run_dir = Path(args.run)
|
|
542
525
|
manifest = json.loads((run_dir / "manifest.json").read_text(encoding="utf-8"))
|
|
@@ -10,7 +10,7 @@ import sys
|
|
|
10
10
|
from pathlib import Path
|
|
11
11
|
|
|
12
12
|
# Add project root to sys.path
|
|
13
|
-
BASE_DIR = Path(
|
|
13
|
+
BASE_DIR = Path(__file__).resolve().parents[1]
|
|
14
14
|
sys.path.insert(0, str(BASE_DIR))
|
|
15
15
|
|
|
16
16
|
from engine.evidence_graph import (
|
|
@@ -18,7 +18,7 @@ from engine.evidence_graph import (
|
|
|
18
18
|
ClaimNode, RiskNode, GapNode, DecisionNode, GraphEdge
|
|
19
19
|
)
|
|
20
20
|
|
|
21
|
-
ESL_DIR = BASE_DIR / "examples" / "esl-academic-writing-ai"
|
|
21
|
+
ESL_DIR = BASE_DIR / "tests" / "fixtures" / "legacy-examples" / "esl-academic-writing-ai"
|
|
22
22
|
ESL_DIR.mkdir(parents=True, exist_ok=True)
|
|
23
23
|
THEMES_DIR = ESL_DIR / "reports-5themes"
|
|
24
24
|
THEMES_DIR.mkdir(parents=True, exist_ok=True)
|
|
@@ -59,8 +59,8 @@ ANNOTATIONS_DIR = ROOT / "benchmarks" / "annotations"
|
|
|
59
59
|
QUESTIONS_PATH = ROOT / "benchmarks" / "questions.jsonl"
|
|
60
60
|
EXAMPLE_EVIDENCE = {
|
|
61
61
|
"ai-coding-assistant": ROOT / "examples" / "ai-coding-assistant" / "evidence.jsonl",
|
|
62
|
-
"ai-tutor": ROOT / "examples" / "ai-tutor" / "evidence.jsonl",
|
|
63
|
-
"ai-writing-assistant": ROOT / "examples" / "ai-writing-assistant" / "evidence.jsonl",
|
|
62
|
+
"ai-tutor": ROOT / "tests" / "fixtures" / "legacy-examples" / "ai-tutor" / "evidence.jsonl",
|
|
63
|
+
"ai-writing-assistant": ROOT / "tests" / "fixtures" / "legacy-examples" / "ai-writing-assistant" / "evidence.jsonl",
|
|
64
64
|
}
|
|
65
65
|
|
|
66
66
|
_WS_RE = re.compile(r"\s+")
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Build public Pages: unchanged introduction + read-only example Studio.
|
|
3
|
+
|
|
4
|
+
Only repository examples are exported. The user's EDUEVIDENCE_HOME, local
|
|
5
|
+
projects, run events and Autoevolve session data never enter this artifact.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
import json
|
|
9
|
+
import re
|
|
10
|
+
import shutil
|
|
11
|
+
import sys
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
ROOT = Path(__file__).resolve().parent.parent
|
|
15
|
+
WEB_DIR = ROOT / 'web'
|
|
16
|
+
EXAMPLES_DIR = ROOT / 'examples'
|
|
17
|
+
OUT_DIR = ROOT / 'dist_gh_pages'
|
|
18
|
+
sys.path.insert(0, str(ROOT))
|
|
19
|
+
from engine.studio_read_model import StudioReader # noqa: E402
|
|
20
|
+
from scripts.build_report_variants import bake # noqa: E402
|
|
21
|
+
from scripts.dashboard_server import scan_local_projects, build_stats, build_viz_payload # noqa: E402
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def write_json(path: Path, payload):
|
|
25
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
26
|
+
path.write_text(json.dumps(payload, ensure_ascii=False, allow_nan=False, indent=2), encoding='utf-8')
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def main():
|
|
30
|
+
if not (WEB_DIR / 'studio' / 'index.html').is_file():
|
|
31
|
+
raise SystemExit('Build the frontend first: cd studio && npm ci && npm run build')
|
|
32
|
+
bake(EXAMPLES_DIR)
|
|
33
|
+
if OUT_DIR.exists():
|
|
34
|
+
shutil.rmtree(OUT_DIR)
|
|
35
|
+
OUT_DIR.mkdir(parents=True)
|
|
36
|
+
for item in WEB_DIR.iterdir():
|
|
37
|
+
if item.name == 'api':
|
|
38
|
+
continue
|
|
39
|
+
if item.is_dir():
|
|
40
|
+
shutil.copytree(item, OUT_DIR / item.name, dirs_exist_ok=True)
|
|
41
|
+
else:
|
|
42
|
+
shutil.copy2(item, OUT_DIR / item.name)
|
|
43
|
+
|
|
44
|
+
# Legacy landing endpoints kept without modifying source web/api artifacts.
|
|
45
|
+
projects = scan_local_projects()
|
|
46
|
+
for project in projects:
|
|
47
|
+
name = project['id']
|
|
48
|
+
project['html_report_path'] = f'reports/{name}/EduEvidence_Report.html' if project.get('html_report_path') else None
|
|
49
|
+
for variant in project.get('report_variants', []):
|
|
50
|
+
variant['path'] = f"reports/{name}/{Path(variant['path']).name}"
|
|
51
|
+
write_json(OUT_DIR / 'api' / 'projects.json', {'projects': projects, 'stats': build_stats(projects)})
|
|
52
|
+
for project in projects:
|
|
53
|
+
write_json(OUT_DIR / 'api' / 'projects' / project['id'] / 'viz.json', build_viz_payload(project['id']))
|
|
54
|
+
|
|
55
|
+
reader = StudioReader(EXAMPLES_DIR, ROOT / '.static-export-no-local-state', static=True)
|
|
56
|
+
catalog = reader.catalog()
|
|
57
|
+
write_json(OUT_DIR / 'api' / 'studio' / 'catalog.json', catalog)
|
|
58
|
+
write_json(OUT_DIR / 'api' / 'studio' / 'evolution.json', {'experiments': [], 'status': 'not_exported'})
|
|
59
|
+
for project in catalog['projects']:
|
|
60
|
+
key = project['id']
|
|
61
|
+
detail = reader.detail(key)
|
|
62
|
+
write_json(OUT_DIR / 'api' / 'studio' / 'projects' / f'{key}.json', detail)
|
|
63
|
+
name = key.removeprefix('example--')
|
|
64
|
+
source = EXAMPLES_DIR / name
|
|
65
|
+
target = OUT_DIR / 'reports' / name
|
|
66
|
+
target.mkdir(parents=True, exist_ok=True)
|
|
67
|
+
for path in (source / 'reports-5themes').glob('*.html'):
|
|
68
|
+
shutil.copy2(path, target / path.name)
|
|
69
|
+
current_default = source / 'reports-5themes' / 'EduEvidence_Report_claude.html'
|
|
70
|
+
if not current_default.exists():
|
|
71
|
+
current_default = source / 'EduEvidence_Report.html'
|
|
72
|
+
if current_default.is_file():
|
|
73
|
+
shutil.copy2(current_default, target / 'EduEvidence_Report.html')
|
|
74
|
+
write_json(OUT_DIR / 'studio' / 'config.json', {'mode': 'static', 'api_base': '../api/studio', 'readonly': True})
|
|
75
|
+
# Keep old public entry links functional without changing the landing design.
|
|
76
|
+
redirect = '<!doctype html><html lang="en"><meta charset="utf-8"><meta http-equiv="refresh" content="0;url=./studio/"><title>Research Studio</title><a href="./studio/">Open Research Studio</a><script>location.replace("./studio/"+location.hash)</script></html>'
|
|
77
|
+
(OUT_DIR / 'studio.html').write_text(redirect, encoding='utf-8')
|
|
78
|
+
landing = WEB_DIR / 'landing.html'
|
|
79
|
+
if landing.is_file():
|
|
80
|
+
page = landing.read_text(encoding='utf-8')
|
|
81
|
+
page = page.replace('href="/landing.html"', 'href="index.html"').replace('href="/index.html"', 'href="studio/"')
|
|
82
|
+
theme_alias = {'claude_research':'claude', 'academic_paper':'academic', 'datalab_light':'datalab', 'datalab_dark':'datalab-dark', 'presentation_judge':'presentation'}
|
|
83
|
+
def report_link(match):
|
|
84
|
+
parts = match.group(1).replace('&', '&').split('&')
|
|
85
|
+
project_id = parts[0]
|
|
86
|
+
theme = next((s.split('=', 1)[1] for s in parts[1:] if s.startswith('theme=')), 'default')
|
|
87
|
+
theme = theme_alias.get(theme, theme)
|
|
88
|
+
filename = 'EduEvidence_Report.html' if theme == 'default' else f'EduEvidence_Report_{theme}.html'
|
|
89
|
+
return f'href="reports/{project_id}/{filename}"'
|
|
90
|
+
page = re.sub(r'href="/report\?id=([^\"]+)"', report_link, page)
|
|
91
|
+
(OUT_DIR / 'index.html').write_text(page, encoding='utf-8')
|
|
92
|
+
(OUT_DIR / 'landing.html').write_text(page, encoding='utf-8')
|
|
93
|
+
(OUT_DIR / '.nojekyll').write_text('', encoding='utf-8')
|
|
94
|
+
print(f'Pages ready: {len(catalog["projects"])} public cases; local projects excluded')
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
if __name__ == '__main__':
|
|
98
|
+
main()
|