eduevidence 5.2.0 → 6.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +54 -42
- package/README.zh-CN.md +51 -28
- package/SKILL.md +390 -133
- package/agents/openai.yaml +4 -0
- package/assets/readme/controlled-execution.svg +34 -0
- package/assets/readme/logo.png +0 -0
- package/assets/readme/research-workflow.svg +56 -0
- package/assets/readme/studio-graph.png +0 -0
- package/assets/readme/studio-overview.png +0 -0
- package/assets/readme/studio-reports.png +0 -0
- package/autoevolve/config.yaml +17 -0
- package/autoevolve/program.md +25 -0
- package/autoevolve/protected.manifest.yaml +34 -0
- package/benchmarks/adversarial/cases.jsonl +7 -0
- package/benchmarks/evidence-library.json +5268 -0
- package/benchmarks/partitions.json +8 -0
- package/docs/architecture.md +220 -0
- package/docs/autoresearch-evolution-plan.md +2903 -0
- package/docs/autoresearch-implementation-status.md +101 -0
- package/docs/demo-storyboard.md +20 -0
- package/docs/demo-workplace-ai.md +92 -0
- package/docs/demo.md +32 -0
- package/docs/install-guide.md +150 -0
- package/docs/orchestration-role-model.md +1254 -0
- package/docs/release-closeout/README.md +17 -0
- package/docs/release-closeout/frontend-acceptance.md +23 -0
- package/docs/release-closeout/issues.md +19 -0
- package/docs/release-closeout/verification.md +28 -0
- package/docs/release-contract.md +108 -0
- package/docs/research-studio-guide.zh-CN.md +166 -0
- package/eduevidence_cli.py +17 -11
- package/engine/_resources.py +13 -0
- package/engine/autoevolve/__init__.py +3 -0
- package/engine/autoevolve/agent_view.py +167 -0
- package/engine/autoevolve/core.py +357 -0
- package/engine/autoevolve/events.py +11 -0
- package/engine/autoevolve/git_workspace.py +77 -0
- package/engine/autoevolve/projection.py +23 -0
- package/engine/autoevolve/runner.py +413 -0
- package/engine/autoevolve/trust.py +146 -0
- package/engine/autoresearch/__init__.py +6 -0
- package/engine/autoresearch/commit.py +132 -0
- package/engine/autoresearch/contracts.py +126 -0
- package/engine/autoresearch/controller.py +207 -0
- package/engine/autoresearch/events.py +12 -0
- package/engine/autoresearch/gap_priority.py +168 -0
- package/engine/autoresearch/projection.py +30 -0
- package/engine/autoresearch/research_memory.py +59 -0
- package/engine/autoresearch/saturation.py +91 -0
- package/engine/briefs.py +2 -1
- package/engine/capabilities.py +1 -0
- package/engine/contracts.py +3 -1
- package/engine/evidencecore.py +7 -5
- package/engine/gaps.py +90 -51
- package/engine/judge_pack.py +65 -0
- package/engine/library_builtin.py +3 -1
- package/engine/living.py +2 -1
- package/engine/meta_synthesis.py +3 -1
- package/engine/orchestration.py +460 -0
- package/engine/pilot.py +2 -1
- package/engine/project.py +2 -2
- package/engine/research_service.py +113 -0
- package/engine/studio_read_model.py +400 -0
- package/engine/tribunal.py +1 -2
- package/engine/update.py +1 -0
- package/engine/versions.py +1 -1
- package/engine/worker_result.py +109 -0
- package/engine/workflows.py +70 -0
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1720 -0
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
- package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
- package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
- package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/frame.json +48 -0
- package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
- package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
- package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +219 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1720 -0
- package/examples/ai-coding-assistant-evidence/result.json +1453 -0
- package/examples/ai-coding-assistant-evidence/result.zh.json +1453 -0
- package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +103 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -0
- package/examples/workplace-ai-assistant/evaluation.json +19 -0
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
- package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
- package/examples/workplace-ai-assistant/frame.json +41 -0
- package/examples/workplace-ai-assistant/intervention.json +27 -0
- package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
- package/examples/workplace-ai-assistant/methodology.json +60 -0
- package/examples/workplace-ai-assistant/report_spec.json +55 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2484 -0
- package/examples/workplace-ai-assistant/result.json +553 -0
- package/examples/workplace-ai-assistant/result.zh.json +553 -0
- package/examples/workplace-ai-assistant/search_log.json +19 -0
- package/examples/workplace-ai-assistant/sources.jsonl +3 -0
- package/examples/workplace-ai-assistant/validation_result.json +9 -0
- package/examples/workplace-ai-assistant/verdict.json +52 -0
- package/install.sh +7 -7
- package/integrations/orchestration_dispatch.py +146 -0
- package/package.json +37 -3
- package/pyproject.toml +11 -20
- package/references/autoresearch.md +30 -0
- package/references/evaluation-policy.md +24 -0
- package/references/orchestration.md +22 -0
- package/references/scientific-invariants.md +19 -0
- package/retrieval/audit.py +154 -0
- package/schemas/intervention.schema.json +106 -21
- package/schemas/report-result.schema.json +9 -1
- package/schemas/v2/project.schema.json +2 -2
- package/schemas/v2/run.schema.json +1 -1
- package/schemas/vNext/autoevolve-session.schema.json +1 -0
- package/schemas/vNext/eval-snapshot.schema.json +1 -0
- package/schemas/vNext/execution-plan.schema.json +1 -0
- package/schemas/vNext/gap-priority.schema.json +1 -0
- package/schemas/vNext/negative-search-record.schema.json +1 -0
- package/schemas/vNext/research-iteration.schema.json +1 -0
- package/schemas/vNext/research-strategy.schema.json +1 -0
- package/schemas/vNext/skill-experiment.schema.json +1 -0
- package/schemas/vNext/task-spec.schema.json +1 -0
- package/schemas/vNext/worker-result.schema.json +1 -0
- package/scripts/benchmark_judge.py +2 -2
- package/scripts/benchmark_v3.py +26 -43
- package/scripts/build_esl_artifacts.py +2 -2
- package/scripts/build_evidence_library.py +2 -2
- package/scripts/build_gh_pages.py +98 -0
- package/scripts/build_readme_diagrams.py +72 -0
- package/scripts/build_report_variants.py +85 -0
- package/scripts/check_autoresearch_invariants.py +95 -0
- package/scripts/daily_evolve.py +30 -0
- package/scripts/dashboard_server.py +130 -101
- package/scripts/did_regression.py +5 -30
- package/scripts/enrich_projects_human_and_lieflat.py +1 -1
- package/scripts/generate_metrics.py +4 -3
- package/scripts/generate_new_projects.py +1 -1
- package/scripts/orchestrator.py +172 -18
- package/scripts/rebake_all_5themes.py +1 -2
- package/scripts/research_auto_cli.py +475 -0
- package/scripts/run_workspace.py +17 -7
- package/scripts/search_provenance.py +64 -0
- package/scripts/serve_web.py +9 -10
- package/scripts/skill_lint.py +1 -1
- package/scripts/skill_payload.py +78 -0
- package/scripts/validate_schema.py +15 -1
- package/scripts/vnext_cli.py +133 -0
- package/setup.py +12 -0
- package/skill/roles/registry.yaml +45 -0
- package/skill/sub-skills/report-generation/SKILL.md +12 -6
- package/skill/task-briefs/applicability.md +3 -0
- package/skill/task-briefs/projection.md +3 -0
- package/skill/workflows/decision-and-pilot.md +10 -0
- package/skill/workflows/evaluate-and-update.md +10 -0
- package/skill/workflows/evidence-review.md +13 -0
- package/visualization/eduevidence-report/assets/base.css +2 -2
- package/visualization/eduevidence-report/assets/reader.css +752 -0
- package/visualization/eduevidence-report/assets/reader.js +132 -0
- package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
- package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
- package/visualization/eduevidence-report/scripts/build_report.py +58 -65
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +24 -100
- package/visualization/eduevidence-report/themes/academic.css +1 -1
- package/visualization/eduevidence-report/themes/claude.css +1 -1
- package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
- package/visualization/eduevidence-report/themes/datalab.css +2 -2
- package/visualization/eduevidence-report/themes/presentation.css +2 -2
- package/web/README.md +18 -0
- package/web/index.html +53 -0
- package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
- package/web/studio/assets/index-CzXocaGv.css +1 -0
- package/web/studio/assets/index-pa7jD7n4.js +230 -0
- package/web/studio/config.json +1 -0
- package/web/studio/index.html +14 -0
- package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
- package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/bias.cpython-312.pyc +0 -0
- package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
- package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
- package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
- package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
- package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
- package/engine/__pycache__/events.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
- package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
- package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
- package/engine/__pycache__/ids.cpython-312.pyc +0 -0
- package/engine/__pycache__/library.cpython-312.pyc +0 -0
- package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
- package/engine/__pycache__/living.cpython-312.pyc +0 -0
- package/engine/__pycache__/log.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/migration.cpython-312.pyc +0 -0
- package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
- package/engine/__pycache__/paths.cpython-312.pyc +0 -0
- package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
- package/engine/__pycache__/planner.cpython-312.pyc +0 -0
- package/engine/__pycache__/project.cpython-312.pyc +0 -0
- package/engine/__pycache__/projections.cpython-312.pyc +0 -0
- package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
- package/engine/__pycache__/run.cpython-312.pyc +0 -0
- package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
- package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
- package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
- package/engine/__pycache__/update.cpython-312.pyc +0 -0
- package/engine/__pycache__/versions.cpython-312.pyc +0 -0
- package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
- package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
- package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
- package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
- package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
- package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
- package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
- package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
- package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
- package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
- package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
- package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
- package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
- package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
- package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
- package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
|
@@ -10,10 +10,11 @@ Metrics:
|
|
|
10
10
|
- engine_version from engine/versions.py (the version authority)
|
|
11
11
|
- test_functions grep 'def test_' across tests/
|
|
12
12
|
- test_files number of collected test modules in tests/
|
|
13
|
-
- schema_count schemas/*.json
|
|
13
|
+
- schema_count schemas/*.json recursively
|
|
14
14
|
- reference_doc_count references/*.md
|
|
15
15
|
- gold_annotation_count benchmarks/annotations/gold-Q*.json
|
|
16
|
-
- example_packs examples/*/ directories shipping result.json
|
|
16
|
+
- example_packs real examples/*/ directories shipping result.json
|
|
17
|
+
(compatibility symlink aliases are excluded)
|
|
17
18
|
|
|
18
19
|
Usage:
|
|
19
20
|
python3 scripts/generate_metrics.py # regenerate docs/metrics.json
|
|
@@ -56,7 +57,7 @@ def collect() -> dict:
|
|
|
56
57
|
|
|
57
58
|
example_packs = sorted(
|
|
58
59
|
p.name for p in (REPO_ROOT / "examples").iterdir()
|
|
59
|
-
if p.is_dir() and (p / "result.json").exists()
|
|
60
|
+
if not p.is_symlink() and p.is_dir() and (p / "result.json").exists()
|
|
60
61
|
) if (REPO_ROOT / "examples").is_dir() else []
|
|
61
62
|
|
|
62
63
|
return {
|
package/scripts/orchestrator.py
CHANGED
|
@@ -10,14 +10,14 @@ stages); the two deterministic stages it executes locally are:
|
|
|
10
10
|
adjudicate — Pre-Verdict Gate (scripts/pre_verdict_gate.py) + deterministic
|
|
11
11
|
confidence (scripts/compute_confidence.py) producing
|
|
12
12
|
final_verdict.json from raw_verdict.json + evidence.jsonl
|
|
13
|
-
|
|
13
|
+
projection — assemble result.json and renderable projections from artifacts
|
|
14
14
|
(decision = final_verdict.json; claims carry claim_id per
|
|
15
15
|
report-result.schema.json)
|
|
16
16
|
|
|
17
17
|
Stage machine (execution_plan.json / state.json):
|
|
18
18
|
|
|
19
19
|
frame -> retrieve -> extract -> challenge -> audit -> adjudicate
|
|
20
|
-
-> intervene -> evaluate ->
|
|
20
|
+
-> applicability -> intervene -> evaluate -> projection
|
|
21
21
|
|
|
22
22
|
Each stage writes exactly one primary artifact and is schema-gated against
|
|
23
23
|
schemas/*. When the artifact is missing the orchestrator either seeds it from
|
|
@@ -51,6 +51,9 @@ for _p in (str(ROOT), str(ROOT / "scripts")):
|
|
|
51
51
|
if _p not in sys.path:
|
|
52
52
|
sys.path.insert(0, _p)
|
|
53
53
|
|
|
54
|
+
from engine._resources import resource_root # noqa: E402
|
|
55
|
+
ROOT = resource_root()
|
|
56
|
+
|
|
54
57
|
from run_workspace import (RESOURCE_POLICY_VERSION, STAGES, RunWorkspace, # noqa: E402
|
|
55
58
|
load_json, load_jsonl, next_run_id, save_jsonl)
|
|
56
59
|
from pre_verdict_gate import apply_enforcement, evaluate_workspace # noqa: E402
|
|
@@ -70,9 +73,10 @@ STAGE_SPEC: dict[str, dict[str, Any]] = {
|
|
|
70
73
|
"challenge": {"artifact": "skeptic.json", "schema": None, "jsonl": False, "local": False},
|
|
71
74
|
"audit": {"artifact": "methodology.json", "schema": "methodology.schema.json", "jsonl": False, "local": False},
|
|
72
75
|
"adjudicate": {"artifact": "final_verdict.json", "schema": "verdict.schema.json", "jsonl": False, "local": True},
|
|
76
|
+
"applicability": {"artifact": "applicability.json", "schema": None, "jsonl": False, "local": False},
|
|
73
77
|
"intervene": {"artifact": "intervention.json", "schema": "intervention.schema.json", "jsonl": False, "local": False},
|
|
74
78
|
"evaluate": {"artifact": "evaluation.json", "schema": "evaluation.schema.json", "jsonl": False, "local": False},
|
|
75
|
-
"
|
|
79
|
+
"projection": {"artifact": "result.json", "schema": "report-result.schema.json", "jsonl": False, "local": True},
|
|
76
80
|
}
|
|
77
81
|
|
|
78
82
|
#: Phase 33 — canonical failure -> handling-action mapping. Extends the
|
|
@@ -143,11 +147,14 @@ _STAGE_BRIEFS: dict[str, str] = {
|
|
|
143
147
|
"adjudicate": ("Judge the evidence: write raw_verdict.json (model verdict). The orchestrator "
|
|
144
148
|
"then runs the Pre-Verdict Gate and deterministic confidence to produce "
|
|
145
149
|
"final_verdict.json."),
|
|
150
|
+
"applicability": ("Assess whether supported effects apply to the target population, setting, "
|
|
151
|
+
"implementation constraints and outcomes; write applicability.json. Do not "
|
|
152
|
+
"upgrade a decision merely because evidence is present."),
|
|
146
153
|
"intervene": ("Design the minimal verifiable teaching intervention (phased pilot, "
|
|
147
154
|
"stop conditions, evidence alignment); write intervention.json."),
|
|
148
155
|
"evaluate": ("Design the evaluation plan (baseline/post/retention/transfer, task vs learning "
|
|
149
156
|
"separation); write evaluation.json."),
|
|
150
|
-
"
|
|
157
|
+
"projection": ("Translate result.json into result.zh.json and render report_spec.json / "
|
|
151
158
|
"report.html via the visualization layer."),
|
|
152
159
|
}
|
|
153
160
|
|
|
@@ -179,6 +186,7 @@ def init_run(
|
|
|
179
186
|
run_id: str | None = None,
|
|
180
187
|
approve_agent_mcp: bool = False,
|
|
181
188
|
scp_available: bool | None = None,
|
|
189
|
+
approval_record: dict | None = None,
|
|
182
190
|
) -> RunWorkspace:
|
|
183
191
|
"""Create the run workspace + manifest + planning artifacts (Phase 11-13)."""
|
|
184
192
|
depth = DEPTH_ALIASES.get(depth, depth)
|
|
@@ -275,6 +283,16 @@ def init_run(
|
|
|
275
283
|
"reason": ("user-approved via --approve-agent-mcp"
|
|
276
284
|
if approve_agent_mcp else "not yet approved; runs in platform-native mode"),
|
|
277
285
|
}
|
|
286
|
+
if approval_record:
|
|
287
|
+
agent_mcp_approval["approval_global_path"] = str(_global_approval_path())
|
|
288
|
+
agent_mcp_approval["role_mapping_hash"] = approval_record.get("role_mapping_hash")
|
|
289
|
+
agent_mcp_approval["roles"] = approval_record.get("roles", {})
|
|
290
|
+
agent_mcp_approval["approved_at"] = _utc_now()
|
|
291
|
+
agent_mcp_approval["reason"] = "user-confirmed role mapping (global approval, hash-verified)"
|
|
292
|
+
elif approve_agent_mcp:
|
|
293
|
+
agent_mcp_approval["reason"] = (
|
|
294
|
+
"user-approved via --approve-agent-mcp (boolean only; a role mapping "
|
|
295
|
+
"in the global approval is required before any spawn)")
|
|
278
296
|
|
|
279
297
|
for name, data in (("capability_plan", capability_plan),
|
|
280
298
|
("resource_plan", resource_plan),
|
|
@@ -306,11 +324,11 @@ def schema_gate(ws: RunWorkspace, stage: str) -> dict[str, Any]:
|
|
|
306
324
|
spec = STAGE_SPEC[stage]
|
|
307
325
|
artifact = spec["artifact"]
|
|
308
326
|
schema_name = spec["schema"]
|
|
309
|
-
if schema_name is None: #
|
|
327
|
+
if schema_name is None: # lightweight parseability contract
|
|
310
328
|
data = load_json(ws.path / artifact)
|
|
311
329
|
ok = bool(data) and isinstance(data, dict)
|
|
312
330
|
return {"passed": ok, "stage": stage, "artifact": artifact,
|
|
313
|
-
"schema": None, "issues": [] if ok else ["
|
|
331
|
+
"schema": None, "issues": [] if ok else [f"{artifact} missing or unparseable"]}
|
|
314
332
|
|
|
315
333
|
from validate_schema import SchemaError, Validator
|
|
316
334
|
|
|
@@ -520,14 +538,14 @@ def _assemble_result(ws: RunWorkspace, manifest: dict[str, Any]) -> dict[str, An
|
|
|
520
538
|
}
|
|
521
539
|
|
|
522
540
|
|
|
523
|
-
def
|
|
524
|
-
|
|
525
|
-
"""
|
|
541
|
+
def _run_projection(ws: RunWorkspace, manifest: dict[str, Any], question: str,
|
|
542
|
+
demo_pack: Path | None = None) -> dict[str, Any]:
|
|
543
|
+
"""Build projections after science; this is not a scientific protocol stage."""
|
|
526
544
|
required = ("final_verdict.json", "intervention.json", "evaluation.json")
|
|
527
545
|
missing = [name for name in required if not (ws.path / name).is_file()
|
|
528
546
|
or not load_json(ws.path / name)]
|
|
529
547
|
if missing:
|
|
530
|
-
ws.write_brief("
|
|
548
|
+
ws.write_brief("projection", question, _STAGE_BRIEFS["projection"])
|
|
531
549
|
return {"status": "pending",
|
|
532
550
|
"detail": f"missing prerequisite artifacts: {', '.join(missing)}"}
|
|
533
551
|
|
|
@@ -550,7 +568,7 @@ def _run_present(ws: RunWorkspace, manifest: dict[str, Any], question: str,
|
|
|
550
568
|
result = _assemble_result(ws, manifest)
|
|
551
569
|
(ws.path / "result.json").write_text(
|
|
552
570
|
json.dumps(result, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
|
553
|
-
gate = schema_gate(ws, "
|
|
571
|
+
gate = schema_gate(ws, "projection")
|
|
554
572
|
if not gate["passed"]:
|
|
555
573
|
return {"status": "failed", "detail": f"result.json schema gate: {gate['issues']}"}
|
|
556
574
|
missing_render = [n for n in ("result.zh.json", "report_spec.json", "report.html")
|
|
@@ -576,6 +594,10 @@ def run_stage(ws: RunWorkspace, stage: str, *, demo_pack: Path | None = None) ->
|
|
|
576
594
|
Deterministic stages execute locally; external stages are either seeded
|
|
577
595
|
from ``demo_pack`` (demo/test mode) or handed off via a task brief.
|
|
578
596
|
"""
|
|
597
|
+
# Compatibility for callers of the retired name. State/manifests only
|
|
598
|
+
# record ``projection`` from this point forward.
|
|
599
|
+
if stage == "present":
|
|
600
|
+
stage = "projection"
|
|
579
601
|
ws.trace("stage_started", stage=stage)
|
|
580
602
|
log.info("stage=%s run=%s start", stage, ws.run_id)
|
|
581
603
|
spec = STAGE_SPEC[stage]
|
|
@@ -598,8 +620,8 @@ def run_stage(ws: RunWorkspace, stage: str, *, demo_pack: Path | None = None) ->
|
|
|
598
620
|
# local deterministic stages
|
|
599
621
|
if stage == "adjudicate":
|
|
600
622
|
result = _run_adjudicate(ws, question, demo_pack=demo_pack)
|
|
601
|
-
elif stage == "
|
|
602
|
-
result =
|
|
623
|
+
elif stage == "projection":
|
|
624
|
+
result = _run_projection(ws, ws.load_manifest(), question, demo_pack=demo_pack)
|
|
603
625
|
else:
|
|
604
626
|
# demo/test seeding
|
|
605
627
|
if demo_pack is not None:
|
|
@@ -666,6 +688,21 @@ def _seed_from_demo(ws: RunWorkspace, stage: str, demo_pack: Path) -> dict[str,
|
|
|
666
688
|
if (pack / "methodology.json").is_file():
|
|
667
689
|
(ws.path / "methodology.json").write_bytes((pack / "methodology.json").read_bytes())
|
|
668
690
|
return {"seeded": True, "detail": "methodology.json seeded from demo pack"}
|
|
691
|
+
elif stage == "applicability":
|
|
692
|
+
if (pack / "applicability.json").is_file():
|
|
693
|
+
(ws.path / "applicability.json").write_bytes((pack / "applicability.json").read_bytes())
|
|
694
|
+
else:
|
|
695
|
+
verdict = load_json(ws.path / "final_verdict.json") or load_json(pack / "verdict.json")
|
|
696
|
+
value = verdict.get("applicability") if isinstance(verdict, dict) else None
|
|
697
|
+
# A demo can only carry the decision's existing applicability
|
|
698
|
+
# boundary; absence remains explicit rather than inferred.
|
|
699
|
+
payload = value if isinstance(value, dict) and value else {
|
|
700
|
+
"status": "NOT_CAPTURED",
|
|
701
|
+
"reason": "demo pack does not provide an applicability assessment",
|
|
702
|
+
}
|
|
703
|
+
(ws.path / "applicability.json").write_text(
|
|
704
|
+
json.dumps(payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
|
705
|
+
return {"seeded": True, "detail": "applicability.json seeded from decision boundary (demo)"}
|
|
669
706
|
elif stage == "intervene":
|
|
670
707
|
if (pack / "intervention.json").is_file():
|
|
671
708
|
(ws.path / "intervention.json").write_bytes((pack / "intervention.json").read_bytes())
|
|
@@ -732,6 +769,9 @@ def interactive_agent_mcp_setup(approved: bool) -> bool:
|
|
|
732
769
|
|
|
733
770
|
保留 GitHub 版全部行为(--approve-agent-mcp 旗标、agent_mcp_approval.json、
|
|
734
771
|
safe_spawn 门);此处只补 run 启动时的交互提示层。非交互终端直接返回原值。
|
|
772
|
+
新增:Agent MCP 可用时,生成角色→CLI→模型推荐表,询问用户是否采用并
|
|
773
|
+
固化到全局 ~/.eduevidence/agent_mcp_approval.json(含 hash 防篡改);
|
|
774
|
+
确认后返回 True,调用方可把该记录带入本次 run 的批准工件。
|
|
735
775
|
"""
|
|
736
776
|
if approved:
|
|
737
777
|
return True
|
|
@@ -779,13 +819,77 @@ def interactive_agent_mcp_setup(approved: bool) -> bool:
|
|
|
779
819
|
return False
|
|
780
820
|
|
|
781
821
|
answer = input("是否启用 Agent MCP 增强模式(推荐)?[Y/n] ").strip().lower()
|
|
782
|
-
|
|
822
|
+
if answer in ("n", "no"):
|
|
823
|
+
return False
|
|
824
|
+
|
|
825
|
+
# Agent MCP 可用:生成推荐表并询问是否固化到全局(简短的 8 角色表)。
|
|
826
|
+
approved_now, _ = _confirm_global_approval()
|
|
827
|
+
return approved_now
|
|
828
|
+
|
|
829
|
+
|
|
830
|
+
def _global_approval_path() -> Path:
|
|
831
|
+
"""全局用户批准文件:EDUEVIDENCE_HOME 优先,缺省 ~/.eduevidence。"""
|
|
832
|
+
home = Path(os.environ.get("EDUEVIDENCE_HOME", "~/.eduevidence")).expanduser()
|
|
833
|
+
return home / "agent_mcp_approval.json"
|
|
834
|
+
|
|
835
|
+
|
|
836
|
+
def _confirm_global_approval() -> tuple[bool, dict | None]:
|
|
837
|
+
"""构建角色推荐表 → 展示 → 询问 → 固化;返回 (approved, approval_record)。
|
|
838
|
+
|
|
839
|
+
- available CLIs 只扫描本机真实存在的(omp/codex/claude/grok/opencode),
|
|
840
|
+
不猜模型;无任何可用 CLI 时回退为布尔批准。
|
|
841
|
+
- 任何展示内容都只来自已验证模型清单;推荐行缺少 cli/model 的角色
|
|
842
|
+
不进映射(safe_spawn 会对该角色保持关闭)。
|
|
843
|
+
- 用户确认后写 ~/.eduevidence/agent_mcp_approval.json(含 role_mapping_hash),
|
|
844
|
+
后续 run 加载并校验:映射变更即失效,需重新确认。
|
|
845
|
+
"""
|
|
846
|
+
import shutil as _shutil
|
|
847
|
+
from integrations.agent_mcp import (build_recommendation_table,
|
|
848
|
+
scan_available_models, write_approval)
|
|
849
|
+
|
|
850
|
+
allowed_clis = [c for c in ("omp", "codex", "claude", "grok", "opencode")
|
|
851
|
+
if _shutil.which(c)]
|
|
852
|
+
if not allowed_clis:
|
|
853
|
+
return True, None
|
|
854
|
+
inventory = scan_available_models(allowed_clis, timeout=20)
|
|
855
|
+
table = build_recommendation_table(allowed_clis, inventory)
|
|
856
|
+
rows = [r for r in table["recommendations"] if r.get("cli") and r.get("model")]
|
|
857
|
+
if not rows:
|
|
858
|
+
print("[startup] 未能从已安装 CLI 解析到已验证模型;按布尔批准启用。")
|
|
859
|
+
return True, None
|
|
860
|
+
|
|
861
|
+
print("[startup] 角色 → CLI / 模型 推荐表(仅基于本机扫描到的可用模型,无固定推荐):")
|
|
862
|
+
for r in rows:
|
|
863
|
+
print(f" {r['role']:<20} → {r['cli']} / {r['model']}")
|
|
864
|
+
summary = table.get("summary", {})
|
|
865
|
+
print(f" cross_model_review(反证异族复核): {summary.get('cross_model_review', 'unknown')}"
|
|
866
|
+
f" | 角色数: {summary.get('role_count', len(rows))}")
|
|
867
|
+
answer = input("采用推荐表并固化到全局批准文件?[Y/n] ").strip().lower()
|
|
868
|
+
if answer in ("n", "no"):
|
|
869
|
+
print("[startup] 未固化;本次以平台原生模式运行(可用 --approve-agent-mcp 跳过询问)。")
|
|
870
|
+
return False, None
|
|
871
|
+
roles = {r["role"]: {"cli": r["cli"], "model": r["model"]} for r in rows}
|
|
872
|
+
path = _global_approval_path()
|
|
873
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
874
|
+
record = write_approval(path, roles, sorted(allowed_clis))
|
|
875
|
+
print(f"[startup] 已固化 → {path}")
|
|
876
|
+
return True, record
|
|
877
|
+
|
|
878
|
+
|
|
879
|
+
def load_global_approval() -> dict | None:
|
|
880
|
+
"""加载全局批准文件(missing/corrupt -> None;有效期交由
|
|
881
|
+
integrations.agent_mcp.is_approval_current 判定)。"""
|
|
882
|
+
from integrations.agent_mcp import load_approval
|
|
883
|
+
return load_approval(_global_approval_path())
|
|
783
884
|
|
|
784
885
|
|
|
785
886
|
def _cmd_run(args: argparse.Namespace) -> int:
|
|
786
887
|
approve = args.approve_agent_mcp or interactive_agent_mcp_setup(args.approve_agent_mcp)
|
|
888
|
+
approval_record = None
|
|
889
|
+
if approve:
|
|
890
|
+
approval_record = load_global_approval()
|
|
787
891
|
ws = init_run(Path(args.runs_dir), args.question, depth=args.depth, run_id=args.run_id,
|
|
788
|
-
approve_agent_mcp=approve)
|
|
892
|
+
approve_agent_mcp=approve, approval_record=approval_record)
|
|
789
893
|
print(f"workspace created: {ws.path}")
|
|
790
894
|
print(f"manifest: {json.dumps(ws.load_manifest(), ensure_ascii=False, indent=2)}")
|
|
791
895
|
if args.dry_run:
|
|
@@ -984,9 +1088,15 @@ def _cmd_synthesize(args) -> int:
|
|
|
984
1088
|
def _cmd_benchmark(args) -> int:
|
|
985
1089
|
import benchmark_v3 as bv3
|
|
986
1090
|
if args.action == "run":
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
|
|
1091
|
+
argv = ["run", "--baselines", args.baselines, "--questions", args.questions,
|
|
1092
|
+
"--repeats", str(args.repeats), "--out", args.out,
|
|
1093
|
+
"--budget-tokens", str(args.budget), "--model", args.model,
|
|
1094
|
+
"--thinking", args.thinking]
|
|
1095
|
+
if getattr(args, "ids", None):
|
|
1096
|
+
argv.extend(["--ids", args.ids])
|
|
1097
|
+
if args.driver:
|
|
1098
|
+
argv.extend(["--driver", args.driver])
|
|
1099
|
+
return bv3.main(argv)
|
|
990
1100
|
if args.action == "eval":
|
|
991
1101
|
return bv3.main(["eval", "--run", args.run, "--annotations", args.annotations])
|
|
992
1102
|
if args.action == "report":
|
|
@@ -1172,6 +1282,17 @@ def _cmd_report(args) -> int:
|
|
|
1172
1282
|
return 0
|
|
1173
1283
|
|
|
1174
1284
|
|
|
1285
|
+
def _cmd_export(args) -> int:
|
|
1286
|
+
from engine.judge_pack import export_judge_pack
|
|
1287
|
+
from engine.project import ProjectWorkspace
|
|
1288
|
+
project = ProjectWorkspace.open(_home(args), args.project)
|
|
1289
|
+
output = Path(args.out) if args.out else project.path / "exports" / "judge-pack"
|
|
1290
|
+
manifest = export_judge_pack(project, output)
|
|
1291
|
+
print(json.dumps({"output": str(output), "files": len(manifest["copied_files"]),
|
|
1292
|
+
"missing_categories": manifest["missing_categories"]}, ensure_ascii=False))
|
|
1293
|
+
return 0
|
|
1294
|
+
|
|
1295
|
+
|
|
1175
1296
|
def _cmd_migrate(args) -> int:
|
|
1176
1297
|
from engine.migration import migrate_v1_pack
|
|
1177
1298
|
result = migrate_v1_pack(args.pack, home=_home(args), title=args.title)
|
|
@@ -1194,6 +1315,17 @@ def _cmd_search(args) -> int:
|
|
|
1194
1315
|
return 0
|
|
1195
1316
|
|
|
1196
1317
|
|
|
1318
|
+
def _cmd_search_plan(args) -> int:
|
|
1319
|
+
from search_provenance import main as search_plan_main
|
|
1320
|
+
argv = [args.query, "--out", str(args.out), "--domain", args.domain,
|
|
1321
|
+
"--limit", str(args.limit), "--channel", args.channel, "--policy", args.policy]
|
|
1322
|
+
for concept in args.concept:
|
|
1323
|
+
argv.extend(["--concept", concept])
|
|
1324
|
+
for synonym in args.synonym:
|
|
1325
|
+
argv.extend(["--synonym", synonym])
|
|
1326
|
+
return search_plan_main(argv)
|
|
1327
|
+
|
|
1328
|
+
|
|
1197
1329
|
def _cmd_did(args) -> int:
|
|
1198
1330
|
from did_regression import run_did_analysis
|
|
1199
1331
|
res = run_did_analysis(str(args.csv))
|
|
@@ -1329,6 +1461,13 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
1329
1461
|
p_report.add_argument("--home", default=None)
|
|
1330
1462
|
p_report.set_defaults(func=_cmd_report)
|
|
1331
1463
|
|
|
1464
|
+
p_export = sub.add_parser("export", help="export a project evidence pack")
|
|
1465
|
+
p_export.add_argument("kind", choices=["judge-pack"])
|
|
1466
|
+
p_export.add_argument("project", help="project id")
|
|
1467
|
+
p_export.add_argument("--home", default=None)
|
|
1468
|
+
p_export.add_argument("--out", default=None)
|
|
1469
|
+
p_export.set_defaults(func=_cmd_export)
|
|
1470
|
+
|
|
1332
1471
|
|
|
1333
1472
|
p_pilot = sub.add_parser("pilot", help="V3 Decision-to-Outcome Loop")
|
|
1334
1473
|
p_pilot.add_argument("action", choices=["register", "import", "analyze-link", "redecide"])
|
|
@@ -1363,11 +1502,15 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
1363
1502
|
p_bench.add_argument("action", choices=["run", "eval", "report"])
|
|
1364
1503
|
p_bench.add_argument("--baselines", default="B2_standard_agent,B3_eduevidence_single")
|
|
1365
1504
|
p_bench.add_argument("--questions", default="benchmarks/questions.jsonl")
|
|
1505
|
+
p_bench.add_argument("--ids", default=None, help="comma-separated question ids to run")
|
|
1366
1506
|
p_bench.add_argument("--repeats", type=int, default=3)
|
|
1367
1507
|
p_bench.add_argument("--driver", default=None, choices=["api", "cli", "sim"],
|
|
1368
1508
|
help="api | cli (omp) | sim (harness validation only); default: auto (api > cli > sim)")
|
|
1369
1509
|
p_bench.add_argument("--out", default="benchmarks/empirical/run-001")
|
|
1370
1510
|
p_bench.add_argument("--budget", type=int, default=1000000)
|
|
1511
|
+
p_bench.add_argument("--model", default="",
|
|
1512
|
+
help="model for --driver cli (required; no unconfirmed default)")
|
|
1513
|
+
p_bench.add_argument("--thinking", default="max", choices=["low", "high", "max"])
|
|
1371
1514
|
p_bench.add_argument("--run", default=None, help="run dir (eval/report)")
|
|
1372
1515
|
p_bench.add_argument("--annotations", default="benchmarks/annotations")
|
|
1373
1516
|
p_bench.add_argument("--report", default="benchmarks/empirical/v3-report.md")
|
|
@@ -1414,6 +1557,17 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
1414
1557
|
p_srch.add_argument("--academic", action="store_true", help="academic only")
|
|
1415
1558
|
p_srch.set_defaults(func=_cmd_search)
|
|
1416
1559
|
|
|
1560
|
+
p_sp = sub.add_parser("search-plan", help="audited, bounded search with provenance export")
|
|
1561
|
+
p_sp.add_argument("query", help="research question")
|
|
1562
|
+
p_sp.add_argument("--out", required=True, type=Path)
|
|
1563
|
+
p_sp.add_argument("--domain", default="education", choices=["education", "policy"])
|
|
1564
|
+
p_sp.add_argument("--concept", action="append", default=[])
|
|
1565
|
+
p_sp.add_argument("--synonym", action="append", default=[])
|
|
1566
|
+
p_sp.add_argument("--limit", type=int, default=10)
|
|
1567
|
+
p_sp.add_argument("--channel", default="all", choices=["all", "academic", "web"])
|
|
1568
|
+
p_sp.add_argument("--policy", default="2026.09")
|
|
1569
|
+
p_sp.set_defaults(func=_cmd_search_plan)
|
|
1570
|
+
|
|
1417
1571
|
p_did = sub.add_parser("did", help="run DID regression on classroom CSV")
|
|
1418
1572
|
p_did.add_argument("csv", type=Path, help="CSV file path")
|
|
1419
1573
|
p_did.set_defaults(func=_cmd_did)
|
|
@@ -10,8 +10,7 @@ from pathlib import Path
|
|
|
10
10
|
ROOT = Path(__file__).resolve().parent.parent
|
|
11
11
|
|
|
12
12
|
PROJECTS = [
|
|
13
|
-
"examples/
|
|
14
|
-
"examples/esl-academic-writing-ai",
|
|
13
|
+
"examples/workplace-ai-assistant",
|
|
15
14
|
"examples/ai-coding-assistant-evidence"
|
|
16
15
|
]
|
|
17
16
|
|