eduevidence 5.2.0 → 6.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +54 -42
- package/README.zh-CN.md +51 -28
- package/SKILL.md +390 -133
- package/agents/openai.yaml +4 -0
- package/assets/readme/controlled-execution.svg +34 -0
- package/assets/readme/logo.png +0 -0
- package/assets/readme/research-workflow.svg +56 -0
- package/assets/readme/studio-graph.png +0 -0
- package/assets/readme/studio-overview.png +0 -0
- package/assets/readme/studio-reports.png +0 -0
- package/autoevolve/config.yaml +17 -0
- package/autoevolve/program.md +25 -0
- package/autoevolve/protected.manifest.yaml +34 -0
- package/benchmarks/adversarial/cases.jsonl +7 -0
- package/benchmarks/evidence-library.json +5268 -0
- package/benchmarks/partitions.json +8 -0
- package/docs/architecture.md +220 -0
- package/docs/autoresearch-evolution-plan.md +2903 -0
- package/docs/autoresearch-implementation-status.md +101 -0
- package/docs/demo-storyboard.md +20 -0
- package/docs/demo-workplace-ai.md +92 -0
- package/docs/demo.md +32 -0
- package/docs/install-guide.md +150 -0
- package/docs/orchestration-role-model.md +1254 -0
- package/docs/release-closeout/README.md +17 -0
- package/docs/release-closeout/frontend-acceptance.md +23 -0
- package/docs/release-closeout/issues.md +19 -0
- package/docs/release-closeout/verification.md +28 -0
- package/docs/release-contract.md +108 -0
- package/docs/research-studio-guide.zh-CN.md +166 -0
- package/eduevidence_cli.py +17 -11
- package/engine/_resources.py +13 -0
- package/engine/autoevolve/__init__.py +3 -0
- package/engine/autoevolve/agent_view.py +167 -0
- package/engine/autoevolve/core.py +357 -0
- package/engine/autoevolve/events.py +11 -0
- package/engine/autoevolve/git_workspace.py +77 -0
- package/engine/autoevolve/projection.py +23 -0
- package/engine/autoevolve/runner.py +413 -0
- package/engine/autoevolve/trust.py +146 -0
- package/engine/autoresearch/__init__.py +6 -0
- package/engine/autoresearch/commit.py +132 -0
- package/engine/autoresearch/contracts.py +126 -0
- package/engine/autoresearch/controller.py +207 -0
- package/engine/autoresearch/events.py +12 -0
- package/engine/autoresearch/gap_priority.py +168 -0
- package/engine/autoresearch/projection.py +30 -0
- package/engine/autoresearch/research_memory.py +59 -0
- package/engine/autoresearch/saturation.py +91 -0
- package/engine/briefs.py +2 -1
- package/engine/capabilities.py +1 -0
- package/engine/contracts.py +3 -1
- package/engine/evidencecore.py +7 -5
- package/engine/gaps.py +90 -51
- package/engine/judge_pack.py +65 -0
- package/engine/library_builtin.py +3 -1
- package/engine/living.py +2 -1
- package/engine/meta_synthesis.py +3 -1
- package/engine/orchestration.py +460 -0
- package/engine/pilot.py +2 -1
- package/engine/project.py +2 -2
- package/engine/research_service.py +113 -0
- package/engine/studio_read_model.py +400 -0
- package/engine/tribunal.py +1 -2
- package/engine/update.py +1 -0
- package/engine/versions.py +1 -1
- package/engine/worker_result.py +109 -0
- package/engine/workflows.py +70 -0
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1720 -0
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +15 -0
- package/examples/ai-coding-assistant-evidence/citation_check.json +79 -0
- package/examples/ai-coding-assistant-evidence/claims.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/evaluation.json +35 -0
- package/examples/ai-coding-assistant-evidence/evidence.jsonl +12 -0
- package/examples/ai-coding-assistant-evidence/frame.json +48 -0
- package/examples/ai-coding-assistant-evidence/intervention.json +51 -0
- package/examples/ai-coding-assistant-evidence/methodology.json +36 -0
- package/examples/ai-coding-assistant-evidence/raw_verdict.json +86 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +219 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +2614 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1720 -0
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1720 -0
- package/examples/ai-coding-assistant-evidence/result.json +1453 -0
- package/examples/ai-coding-assistant-evidence/result.zh.json +1453 -0
- package/examples/ai-coding-assistant-evidence/sources.jsonl +8 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +103 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -0
- package/examples/workplace-ai-assistant/evaluation.json +19 -0
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -0
- package/examples/workplace-ai-assistant/evidence_graph.json +444 -0
- package/examples/workplace-ai-assistant/frame.json +41 -0
- package/examples/workplace-ai-assistant/intervention.json +27 -0
- package/examples/workplace-ai-assistant/legacy-link-check.json +16 -0
- package/examples/workplace-ai-assistant/methodology.json +60 -0
- package/examples/workplace-ai-assistant/report_spec.json +55 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +2484 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +2484 -0
- package/examples/workplace-ai-assistant/result.json +553 -0
- package/examples/workplace-ai-assistant/result.zh.json +553 -0
- package/examples/workplace-ai-assistant/search_log.json +19 -0
- package/examples/workplace-ai-assistant/sources.jsonl +3 -0
- package/examples/workplace-ai-assistant/validation_result.json +9 -0
- package/examples/workplace-ai-assistant/verdict.json +52 -0
- package/install.sh +7 -7
- package/integrations/orchestration_dispatch.py +146 -0
- package/package.json +37 -3
- package/pyproject.toml +11 -20
- package/references/autoresearch.md +30 -0
- package/references/evaluation-policy.md +24 -0
- package/references/orchestration.md +22 -0
- package/references/scientific-invariants.md +19 -0
- package/retrieval/audit.py +154 -0
- package/schemas/intervention.schema.json +106 -21
- package/schemas/report-result.schema.json +9 -1
- package/schemas/v2/project.schema.json +2 -2
- package/schemas/v2/run.schema.json +1 -1
- package/schemas/vNext/autoevolve-session.schema.json +1 -0
- package/schemas/vNext/eval-snapshot.schema.json +1 -0
- package/schemas/vNext/execution-plan.schema.json +1 -0
- package/schemas/vNext/gap-priority.schema.json +1 -0
- package/schemas/vNext/negative-search-record.schema.json +1 -0
- package/schemas/vNext/research-iteration.schema.json +1 -0
- package/schemas/vNext/research-strategy.schema.json +1 -0
- package/schemas/vNext/skill-experiment.schema.json +1 -0
- package/schemas/vNext/task-spec.schema.json +1 -0
- package/schemas/vNext/worker-result.schema.json +1 -0
- package/scripts/benchmark_judge.py +2 -2
- package/scripts/benchmark_v3.py +26 -43
- package/scripts/build_esl_artifacts.py +2 -2
- package/scripts/build_evidence_library.py +2 -2
- package/scripts/build_gh_pages.py +98 -0
- package/scripts/build_readme_diagrams.py +72 -0
- package/scripts/build_report_variants.py +85 -0
- package/scripts/check_autoresearch_invariants.py +95 -0
- package/scripts/daily_evolve.py +30 -0
- package/scripts/dashboard_server.py +130 -101
- package/scripts/did_regression.py +5 -30
- package/scripts/enrich_projects_human_and_lieflat.py +1 -1
- package/scripts/generate_metrics.py +4 -3
- package/scripts/generate_new_projects.py +1 -1
- package/scripts/orchestrator.py +172 -18
- package/scripts/rebake_all_5themes.py +1 -2
- package/scripts/research_auto_cli.py +475 -0
- package/scripts/run_workspace.py +17 -7
- package/scripts/search_provenance.py +64 -0
- package/scripts/serve_web.py +9 -10
- package/scripts/skill_lint.py +1 -1
- package/scripts/skill_payload.py +78 -0
- package/scripts/validate_schema.py +15 -1
- package/scripts/vnext_cli.py +133 -0
- package/setup.py +12 -0
- package/skill/roles/registry.yaml +45 -0
- package/skill/sub-skills/report-generation/SKILL.md +12 -6
- package/skill/task-briefs/applicability.md +3 -0
- package/skill/task-briefs/projection.md +3 -0
- package/skill/workflows/decision-and-pilot.md +10 -0
- package/skill/workflows/evaluate-and-update.md +10 -0
- package/skill/workflows/evidence-review.md +13 -0
- package/visualization/eduevidence-report/assets/base.css +2 -2
- package/visualization/eduevidence-report/assets/reader.css +752 -0
- package/visualization/eduevidence-report/assets/reader.js +132 -0
- package/visualization/eduevidence-report/references/chart-selection-catalog.md +109 -0
- package/visualization/eduevidence-report/references/lieflat-composition.md +3 -1
- package/visualization/eduevidence-report/scripts/build_report.py +58 -65
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +24 -100
- package/visualization/eduevidence-report/themes/academic.css +1 -1
- package/visualization/eduevidence-report/themes/claude.css +1 -1
- package/visualization/eduevidence-report/themes/datalab-dark.css +2 -2
- package/visualization/eduevidence-report/themes/datalab.css +2 -2
- package/visualization/eduevidence-report/themes/presentation.css +2 -2
- package/web/README.md +18 -0
- package/web/index.html +53 -0
- package/web/studio/THIRD_PARTY_LICENSES.txt +146 -0
- package/web/studio/assets/index-CzXocaGv.css +1 -0
- package/web/studio/assets/index-pa7jD7n4.js +230 -0
- package/web/studio/config.json +1 -0
- package/web/studio/index.html +14 -0
- package/engine/__pycache__/__init__.cpython-312.pyc +0 -0
- package/engine/__pycache__/analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/bias.cpython-312.pyc +0 -0
- package/engine/__pycache__/briefs.cpython-312.pyc +0 -0
- package/engine/__pycache__/capabilities.cpython-312.pyc +0 -0
- package/engine/__pycache__/citation_check.cpython-312.pyc +0 -0
- package/engine/__pycache__/contracts.cpython-312.pyc +0 -0
- package/engine/__pycache__/datasets.cpython-312.pyc +0 -0
- package/engine/__pycache__/events.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_graph.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidence_review.cpython-312.pyc +0 -0
- package/engine/__pycache__/evidencecore.cpython-312.pyc +0 -0
- package/engine/__pycache__/gap_lens.cpython-312.pyc +0 -0
- package/engine/__pycache__/gaps.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_store.cpython-312.pyc +0 -0
- package/engine/__pycache__/graph_validate.cpython-312.pyc +0 -0
- package/engine/__pycache__/ids.cpython-312.pyc +0 -0
- package/engine/__pycache__/library.cpython-312.pyc +0 -0
- package/engine/__pycache__/library_builtin.cpython-312.pyc +0 -0
- package/engine/__pycache__/living.cpython-312.pyc +0 -0
- package/engine/__pycache__/log.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_analysis.cpython-312.pyc +0 -0
- package/engine/__pycache__/meta_synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/migration.cpython-312.pyc +0 -0
- package/engine/__pycache__/mode_router.cpython-312.pyc +0 -0
- package/engine/__pycache__/paths.cpython-312.pyc +0 -0
- package/engine/__pycache__/pilot.cpython-312.pyc +0 -0
- package/engine/__pycache__/planner.cpython-312.pyc +0 -0
- package/engine/__pycache__/project.cpython-312.pyc +0 -0
- package/engine/__pycache__/projections.cpython-312.pyc +0 -0
- package/engine/__pycache__/robustness.cpython-312.pyc +0 -0
- package/engine/__pycache__/run.cpython-312.pyc +0 -0
- package/engine/__pycache__/semantics.cpython-312.pyc +0 -0
- package/engine/__pycache__/study_design.cpython-312.pyc +0 -0
- package/engine/__pycache__/synthesis.cpython-312.pyc +0 -0
- package/engine/__pycache__/tribunal.cpython-312.pyc +0 -0
- package/engine/__pycache__/update.cpython-312.pyc +0 -0
- package/engine/__pycache__/versions.cpython-312.pyc +0 -0
- package/integrations/__pycache__/__init__.cpython-312.pyc +0 -0
- package/integrations/__pycache__/agent_mcp.cpython-312.pyc +0 -0
- package/integrations/__pycache__/smart_web_fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/__init__.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/corpus_store.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/dedupe.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/failures.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/fetch.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/search.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/source.cpython-312.pyc +0 -0
- package/retrieval/__pycache__/validate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/__init__.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_evaluator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_judge.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_routing.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v2.cpython-312.pyc +0 -0
- package/scripts/__pycache__/benchmark_v3.cpython-312.pyc +0 -0
- package/scripts/__pycache__/build_result.cpython-312.pyc +0 -0
- package/scripts/__pycache__/claim_audit.cpython-312.pyc +0 -0
- package/scripts/__pycache__/complexity_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/compute_confidence.cpython-312.pyc +0 -0
- package/scripts/__pycache__/dashboard_server.cpython-312.pyc +0 -0
- package/scripts/__pycache__/did_regression.cpython-312.pyc +0 -0
- package/scripts/__pycache__/effect_calculator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_matrix.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_score.cpython-312.pyc +0 -0
- package/scripts/__pycache__/evidence_semantics.cpython-312.pyc +0 -0
- package/scripts/__pycache__/fetch_benchmark.cpython-312.pyc +0 -0
- package/scripts/__pycache__/lint_report_layout.cpython-312.pyc +0 -0
- package/scripts/__pycache__/orchestrator.cpython-312.pyc +0 -0
- package/scripts/__pycache__/pre_verdict_gate.cpython-312.pyc +0 -0
- package/scripts/__pycache__/recompute_demo_quality.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/render_report_html.cpython-312.pyc +0 -0
- package/scripts/__pycache__/run_workspace.cpython-312.pyc +0 -0
- package/scripts/__pycache__/skill_lint.cpython-312.pyc +0 -0
- package/scripts/__pycache__/startup_probe.cpython-312.pyc +0 -0
- package/scripts/__pycache__/sync_killer_demo_report.cpython-312.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.0.2.pyc +0 -0
- package/scripts/__pycache__/test_adversarial_empirical.cpython-312-pytest-9.1.1.pyc +0 -0
- package/scripts/__pycache__/validate_schema.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/adapter_contract.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_artifact_manifest.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_charts.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_figures.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_infographics.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/build_report.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/charts_data.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/lieflat_engine.cpython-312.pyc +0 -0
- package/visualization/eduevidence-report/scripts/__pycache__/zh_labels.cpython-312.pyc +0 -0
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
import json
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Any
|
|
5
|
+
from .contracts import NegativeSearchRecord, ResearchIteration
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class ResearchMemory:
|
|
9
|
+
def __init__(self, root: str | Path):
|
|
10
|
+
self.root = Path(root)
|
|
11
|
+
self.root.mkdir(parents=True, exist_ok=True)
|
|
12
|
+
self.iterations_path = self.root / "research-iterations.jsonl"
|
|
13
|
+
self.negative_path = self.root / "negative-searches.jsonl"
|
|
14
|
+
|
|
15
|
+
@staticmethod
|
|
16
|
+
def _append(path: Path, record: dict[str, Any]) -> None:
|
|
17
|
+
with path.open("a", encoding="utf-8") as f:
|
|
18
|
+
f.write(json.dumps(record, ensure_ascii=False, sort_keys=True) + "\n")
|
|
19
|
+
|
|
20
|
+
def append_iteration(self, iteration: ResearchIteration) -> None:
|
|
21
|
+
iteration.validate()
|
|
22
|
+
self._append(self.iterations_path, iteration.as_dict())
|
|
23
|
+
|
|
24
|
+
def append_negative_search(self, record: NegativeSearchRecord) -> None:
|
|
25
|
+
record.validate()
|
|
26
|
+
self._append(self.negative_path, record.__dict__)
|
|
27
|
+
|
|
28
|
+
def load_iterations(
|
|
29
|
+
self,
|
|
30
|
+
gap_id: str | None = None,
|
|
31
|
+
*,
|
|
32
|
+
gap_lineage_key: str | None = None,
|
|
33
|
+
) -> list[dict[str, Any]]:
|
|
34
|
+
"""Load iteration history, preferring stable lineage across revisions.
|
|
35
|
+
|
|
36
|
+
Legacy rows without `gap_lineage_key` remain queryable by `gap_id`.
|
|
37
|
+
When a lineage key is supplied, new keyed rows match by lineage and old
|
|
38
|
+
unkeyed rows may additionally match the supplied gap_id for migration.
|
|
39
|
+
"""
|
|
40
|
+
if not self.iterations_path.exists():
|
|
41
|
+
return []
|
|
42
|
+
rows = [
|
|
43
|
+
json.loads(line)
|
|
44
|
+
for line in self.iterations_path.read_text(encoding="utf-8").splitlines()
|
|
45
|
+
if line.strip()
|
|
46
|
+
]
|
|
47
|
+
if gap_lineage_key is not None:
|
|
48
|
+
return [
|
|
49
|
+
row for row in rows
|
|
50
|
+
if row.get("gap_lineage_key") == gap_lineage_key
|
|
51
|
+
or (
|
|
52
|
+
not row.get("gap_lineage_key")
|
|
53
|
+
and gap_id is not None
|
|
54
|
+
and row.get("gap_id") == gap_id
|
|
55
|
+
)
|
|
56
|
+
]
|
|
57
|
+
if gap_id is not None:
|
|
58
|
+
return [row for row in rows if row.get("gap_id") == gap_id]
|
|
59
|
+
return rows
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from dataclasses import dataclass
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
@dataclass(frozen=True)
|
|
7
|
+
class SaturationResult:
|
|
8
|
+
saturated: bool
|
|
9
|
+
low_yield_streak: int
|
|
10
|
+
strategy_diversity_exhausted: bool
|
|
11
|
+
rationale: tuple[str, ...]
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _is_low_yield(row: dict[str, Any]) -> bool:
|
|
15
|
+
gain = row.get("evidence_gain") or {}
|
|
16
|
+
unique = int(gain.get("unique_eligible_evidence", 0) or 0)
|
|
17
|
+
direct = int(gain.get("direct_outcome_findings", 0) or 0)
|
|
18
|
+
delta = float(gain.get("decision_boundary_delta", 0) or 0)
|
|
19
|
+
duplicate_rate = float(gain.get("duplicate_rate", 0) or 0)
|
|
20
|
+
candidate_sources = row.get("candidate_sources") or []
|
|
21
|
+
no_candidates = len(candidate_sources) == 0
|
|
22
|
+
return (
|
|
23
|
+
unique == 0
|
|
24
|
+
and direct == 0
|
|
25
|
+
and abs(delta) < 1e-12
|
|
26
|
+
and (duplicate_rate >= 0.5 or no_candidates)
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def detect_saturation(
|
|
31
|
+
iterations: list[dict[str, Any]],
|
|
32
|
+
*,
|
|
33
|
+
min_consecutive: int = 2,
|
|
34
|
+
available_strategy_types: set[str] | None = None,
|
|
35
|
+
) -> SaturationResult:
|
|
36
|
+
"""Detect bounded secondary-search saturation.
|
|
37
|
+
|
|
38
|
+
Strategy diversity is computed across the full history for the gap, while
|
|
39
|
+
the low-yield condition is intentionally a trailing streak. Empty searches
|
|
40
|
+
count as low-yield even when duplicate_rate is zero; otherwise a provider
|
|
41
|
+
returning no candidates could keep the loop alive forever.
|
|
42
|
+
"""
|
|
43
|
+
attempted_all = {
|
|
44
|
+
str((row.get("strategy") or {}).get("experiment_type", ""))
|
|
45
|
+
for row in iterations
|
|
46
|
+
if str((row.get("strategy") or {}).get("experiment_type", ""))
|
|
47
|
+
}
|
|
48
|
+
streak = 0
|
|
49
|
+
for row in reversed(iterations):
|
|
50
|
+
if _is_low_yield(row):
|
|
51
|
+
streak += 1
|
|
52
|
+
else:
|
|
53
|
+
break
|
|
54
|
+
|
|
55
|
+
if available_strategy_types:
|
|
56
|
+
diversity_exhausted = available_strategy_types.issubset(attempted_all)
|
|
57
|
+
else:
|
|
58
|
+
diversity_exhausted = len(attempted_all) >= 2
|
|
59
|
+
|
|
60
|
+
rationale: list[str] = []
|
|
61
|
+
if streak >= min_consecutive:
|
|
62
|
+
rationale.append(
|
|
63
|
+
f"{streak} consecutive iterations produced no unique/direct evidence or decision-boundary change"
|
|
64
|
+
)
|
|
65
|
+
if diversity_exhausted:
|
|
66
|
+
rationale.append("strategy diversity exhausted for the configured search space")
|
|
67
|
+
return SaturationResult(
|
|
68
|
+
streak >= min_consecutive and diversity_exhausted,
|
|
69
|
+
streak,
|
|
70
|
+
diversity_exhausted,
|
|
71
|
+
tuple(rationale),
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def transition_to_empirical(
|
|
76
|
+
*,
|
|
77
|
+
dvi_band: str,
|
|
78
|
+
decision_material: bool,
|
|
79
|
+
unresolved: bool,
|
|
80
|
+
saturation: SaturationResult,
|
|
81
|
+
ethics_feasible: bool,
|
|
82
|
+
) -> tuple[bool, tuple[str, ...]]:
|
|
83
|
+
checks = [
|
|
84
|
+
(dvi_band.upper() == "HIGH", "gap DVI is HIGH"),
|
|
85
|
+
(decision_material, "gap is material to the decision"),
|
|
86
|
+
(unresolved, "gap remains unresolved"),
|
|
87
|
+
(saturation.saturated, "secondary search is saturated"),
|
|
88
|
+
(ethics_feasible, "empirical study is ethically/operationally feasible"),
|
|
89
|
+
]
|
|
90
|
+
reasons = tuple(text for ok, text in checks if ok)
|
|
91
|
+
return all(ok for ok, _ in checks), reasons
|
package/engine/briefs.py
CHANGED
|
@@ -14,6 +14,7 @@ from pathlib import Path
|
|
|
14
14
|
|
|
15
15
|
from engine.contracts import load_schema, schema_path
|
|
16
16
|
from engine.planner import PlanStep
|
|
17
|
+
from engine.project import ProjectWorkspace
|
|
17
18
|
|
|
18
19
|
|
|
19
20
|
def _schema_section(schema: dict) -> str:
|
|
@@ -73,7 +74,7 @@ def build_task_brief(step: PlanStep, *, project: ProjectWorkspace,
|
|
|
73
74
|
f"Validate the output with `engine.contracts.validate_record({schema_name!r}, record)` — "
|
|
74
75
|
f"it must return [] (empty errors).",
|
|
75
76
|
"",
|
|
76
|
-
|
|
77
|
+
"## Output path",
|
|
77
78
|
str(output_path),
|
|
78
79
|
"",
|
|
79
80
|
"## Inputs",
|
package/engine/capabilities.py
CHANGED
package/engine/contracts.py
CHANGED
|
@@ -12,7 +12,9 @@ from typing import Callable
|
|
|
12
12
|
|
|
13
13
|
from scripts.validate_schema import SchemaError, validate
|
|
14
14
|
|
|
15
|
-
|
|
15
|
+
from engine._resources import resource_root
|
|
16
|
+
|
|
17
|
+
_REPO_SCHEMA_DIR = resource_root() / "schemas" / "v2"
|
|
16
18
|
|
|
17
19
|
|
|
18
20
|
def _resolve_schema_dir() -> Path:
|
package/engine/evidencecore.py
CHANGED
|
@@ -12,8 +12,7 @@ v4 领域包机制:domains/ 注册表 + 领域契约加载 + frame 校验。
|
|
|
12
12
|
education 域只是"指向现有契约"的注册:不新增任何逻辑路径、不引入新 schema
|
|
13
13
|
或新校验器。领域选择(domain select)由主 agent 接 CLI 完成,引擎层不做选择。
|
|
14
14
|
|
|
15
|
-
|
|
16
|
-
share/ 回退留给后续步骤(pyproject data-files 未包含 domains/)。
|
|
15
|
+
路径解析:支持仓库、独立 Skill 与 wheel 的 share/eduevidence 资源布局。
|
|
17
16
|
"""
|
|
18
17
|
|
|
19
18
|
from __future__ import annotations
|
|
@@ -22,7 +21,9 @@ import json
|
|
|
22
21
|
from pathlib import Path
|
|
23
22
|
from typing import Any
|
|
24
23
|
|
|
25
|
-
|
|
24
|
+
from engine._resources import resource_root
|
|
25
|
+
|
|
26
|
+
REPO_ROOT = resource_root()
|
|
26
27
|
|
|
27
28
|
|
|
28
29
|
def _resolve_domains_dir() -> Path:
|
|
@@ -114,7 +115,7 @@ def _validate_contracts(entry: dict) -> None:
|
|
|
114
115
|
|
|
115
116
|
- frame_schema / outcome_taxonomy / methodology_checklist:文件存在且
|
|
116
117
|
为可解析 JSON(指针引用另校验指针内容);
|
|
117
|
-
- golds_dir
|
|
118
|
+
- references_dir:目录存在;golds_dir 属于独立 evaluator 资源,不是研究运行依赖。
|
|
118
119
|
"""
|
|
119
120
|
domain_id = entry["id"]
|
|
120
121
|
|
|
@@ -142,7 +143,8 @@ def _validate_contracts(entry: dict) -> None:
|
|
|
142
143
|
check_file("frame_schema")
|
|
143
144
|
check_file("outcome_taxonomy")
|
|
144
145
|
check_file("methodology_checklist")
|
|
145
|
-
|
|
146
|
+
# Evaluation annotations (including holdout answers) are intentionally absent
|
|
147
|
+
# from shipped Skills. Benchmark consumers validate their own input corpus.
|
|
146
148
|
check_dir("references_dir")
|
|
147
149
|
|
|
148
150
|
|
package/engine/gaps.py
CHANGED
|
@@ -4,10 +4,15 @@ A KnowledgeGap is not free-form "future work": it is derived from coverage —
|
|
|
4
4
|
the research frame's requested outcomes vs what the graph's Findings
|
|
5
5
|
actually measure. A task-performance Finding never covers a retention or
|
|
6
6
|
transfer gap.
|
|
7
|
+
|
|
8
|
+
`gap_id` identifies one revision-local gap artifact. `extensions.autoresearch_key`
|
|
9
|
+
is a stable semantic lineage key so bounded research memory survives graph
|
|
10
|
+
revisions when the same unresolved gap is re-derived.
|
|
7
11
|
"""
|
|
8
12
|
|
|
9
13
|
from __future__ import annotations
|
|
10
14
|
|
|
15
|
+
import hashlib
|
|
11
16
|
import json
|
|
12
17
|
from pathlib import Path
|
|
13
18
|
|
|
@@ -16,23 +21,35 @@ from engine.graph_store import GraphStore
|
|
|
16
21
|
from engine.ids import new_local_id
|
|
17
22
|
from engine.synthesis import ClaimSynthesis
|
|
18
23
|
|
|
19
|
-
# outcome_type names for timepoint-like gaps (frame.requested_outcomes entries
|
|
20
|
-
# may carry outcome_type or be plain strings; we match on outcome_type)
|
|
21
24
|
_RETENTION_TYPES = {"retention", "long_term", "learning_retention"}
|
|
22
25
|
_TRANSFER_TYPES = {"transfer", "transfer_learning", "far_transfer"}
|
|
23
26
|
_TASK_PERFORMANCE = {"task_performance", "assignment_score", "task_completion"}
|
|
24
27
|
_LEARNING = {"learning"}
|
|
25
28
|
|
|
26
29
|
|
|
30
|
+
def _autoresearch_key(
|
|
31
|
+
gap_type: str,
|
|
32
|
+
*,
|
|
33
|
+
related_claims: list[str],
|
|
34
|
+
related_outcomes: list[str],
|
|
35
|
+
semantic_token: str,
|
|
36
|
+
) -> str:
|
|
37
|
+
payload = {
|
|
38
|
+
"gap_type": gap_type,
|
|
39
|
+
"related_claims": sorted(related_claims),
|
|
40
|
+
"related_outcomes": sorted(related_outcomes),
|
|
41
|
+
"semantic_token": semantic_token.strip().lower(),
|
|
42
|
+
}
|
|
43
|
+
digest = hashlib.sha256(
|
|
44
|
+
json.dumps(payload, sort_keys=True, ensure_ascii=False).encode("utf-8")
|
|
45
|
+
).hexdigest()[:20]
|
|
46
|
+
return f"KGK-{digest}"
|
|
47
|
+
|
|
48
|
+
|
|
27
49
|
def derive_gaps(*, store: GraphStore,
|
|
28
50
|
syntheses: tuple[ClaimSynthesis, ...] | None = None,
|
|
29
51
|
frame: dict | None = None) -> list[dict]:
|
|
30
|
-
"""Derive structured gaps from graph coverage vs the research frame.
|
|
31
|
-
|
|
32
|
-
`frame` carries `requested_outcomes` (list of outcome names/types) and
|
|
33
|
-
optionally `target_population`. Findings' outcome types come from the
|
|
34
|
-
graph's outcomes table.
|
|
35
|
-
"""
|
|
52
|
+
"""Derive structured gaps from graph coverage vs the research frame."""
|
|
36
53
|
frame = frame or {}
|
|
37
54
|
requested = frame.get("requested_outcomes") or []
|
|
38
55
|
if not requested and frame.get("target_outcomes"):
|
|
@@ -47,30 +64,34 @@ def derive_gaps(*, store: GraphStore,
|
|
|
47
64
|
covered_types.add(o.get("outcome_type", ""))
|
|
48
65
|
|
|
49
66
|
claims = store.read_table("claims")
|
|
50
|
-
claim_ids = [c["claim_id"] for c in claims]
|
|
51
|
-
|
|
52
67
|
gaps: list[dict] = []
|
|
53
68
|
rev = store.active_revision()
|
|
54
69
|
|
|
55
70
|
def add(gap_type: str, priority: str, reasoning: str,
|
|
56
71
|
related_claims: list[str] | None = None,
|
|
57
|
-
related_outcomes: list[str] | None = None
|
|
72
|
+
related_outcomes: list[str] | None = None,
|
|
73
|
+
semantic_token: str = ""):
|
|
74
|
+
related_claims = related_claims or []
|
|
75
|
+
related_outcomes = related_outcomes or []
|
|
76
|
+
key = _autoresearch_key(
|
|
77
|
+
gap_type,
|
|
78
|
+
related_claims=related_claims,
|
|
79
|
+
related_outcomes=related_outcomes,
|
|
80
|
+
semantic_token=semantic_token or reasoning,
|
|
81
|
+
)
|
|
58
82
|
gaps.append({
|
|
59
83
|
"gap_id": new_local_id("GAP", {g["gap_id"] for g in gaps}),
|
|
60
84
|
"gap_type": gap_type,
|
|
61
|
-
"related_claim_ids": related_claims
|
|
62
|
-
"related_outcome_ids": related_outcomes
|
|
85
|
+
"related_claim_ids": related_claims,
|
|
86
|
+
"related_outcome_ids": related_outcomes,
|
|
63
87
|
"priority": priority,
|
|
64
88
|
"reasoning": reasoning,
|
|
65
89
|
"status": "open",
|
|
66
90
|
"derived_from_graph_revision": rev,
|
|
67
|
-
"extensions": {},
|
|
91
|
+
"extensions": {"autoresearch_key": key},
|
|
68
92
|
})
|
|
69
93
|
|
|
70
94
|
def _req_kind(req) -> tuple[str, str]:
|
|
71
|
-
"""Classify a requested outcome: retention | transfer |
|
|
72
|
-
task_performance | learning | other. Type-aware: names are matched
|
|
73
|
-
only within the outcome's declared type, never type-blind."""
|
|
74
95
|
if isinstance(req, dict):
|
|
75
96
|
req_name = str(req.get("name", "")).lower()
|
|
76
97
|
req_type = str(req.get("outcome_type", "")).lower()
|
|
@@ -86,20 +107,17 @@ def derive_gaps(*, store: GraphStore,
|
|
|
86
107
|
return "learning", req.get("name", "") if isinstance(req, dict) else str(req)
|
|
87
108
|
return "other", req.get("name", "") if isinstance(req, dict) else str(req)
|
|
88
109
|
|
|
89
|
-
_RETENTION_COVER = _RETENTION_TYPES
|
|
90
|
-
_TRANSFER_COVER = _TRANSFER_TYPES
|
|
91
110
|
def covered_for_kind(kind: str) -> bool:
|
|
92
111
|
if kind == "retention":
|
|
93
|
-
return bool(covered_types &
|
|
112
|
+
return bool(covered_types & _RETENTION_TYPES)
|
|
94
113
|
if kind == "transfer":
|
|
95
|
-
return bool(covered_types &
|
|
114
|
+
return bool(covered_types & _TRANSFER_TYPES)
|
|
96
115
|
if kind == "task_performance":
|
|
97
116
|
return bool(covered_types & _TASK_PERFORMANCE)
|
|
98
117
|
if kind == "learning":
|
|
99
118
|
return bool(covered_types & _LEARNING)
|
|
100
119
|
return False
|
|
101
120
|
|
|
102
|
-
# one pass per requested outcome; each gap emitted exactly once
|
|
103
121
|
seen: set[tuple[str, str]] = set()
|
|
104
122
|
for req in requested:
|
|
105
123
|
kind, label = _req_kind(req)
|
|
@@ -112,49 +130,70 @@ def derive_gaps(*, store: GraphStore,
|
|
|
112
130
|
if covered_for_kind(kind):
|
|
113
131
|
continue
|
|
114
132
|
if kind == "retention":
|
|
115
|
-
add(
|
|
133
|
+
add(
|
|
134
|
+
"missing_retention", "high",
|
|
116
135
|
f"frame requests retention outcome {label!r} but the graph has "
|
|
117
|
-
|
|
118
|
-
|
|
136
|
+
"no retention-type measurement; task-performance coverage does "
|
|
137
|
+
"not count (RULE 3)",
|
|
138
|
+
semantic_token=f"requested_outcome:{label}",
|
|
139
|
+
)
|
|
119
140
|
elif kind == "transfer":
|
|
120
|
-
add(
|
|
141
|
+
add(
|
|
142
|
+
"missing_transfer", "high",
|
|
121
143
|
f"frame requests transfer outcome {label!r} but the graph has "
|
|
122
|
-
|
|
123
|
-
|
|
144
|
+
"no transfer-type measurement; AI-assisted task performance "
|
|
145
|
+
"does not count (RULE 3)",
|
|
146
|
+
semantic_token=f"requested_outcome:{label}",
|
|
147
|
+
)
|
|
124
148
|
elif kind == "task_performance":
|
|
125
|
-
add(
|
|
126
|
-
|
|
127
|
-
f"covering finding"
|
|
149
|
+
add(
|
|
150
|
+
"missing_outcome", "medium",
|
|
151
|
+
f"frame requests task-performance outcome {label!r} with no covering finding",
|
|
152
|
+
semantic_token=f"requested_outcome:{label}",
|
|
153
|
+
)
|
|
128
154
|
elif kind == "learning":
|
|
129
|
-
add(
|
|
130
|
-
|
|
131
|
-
f"learning
|
|
155
|
+
add(
|
|
156
|
+
"missing_outcome", "medium",
|
|
157
|
+
f"frame requests learning outcome {label!r} with no covering learning finding; "
|
|
158
|
+
"task performance is not learning (RULE 3)",
|
|
159
|
+
semantic_token=f"requested_outcome:{label}",
|
|
160
|
+
)
|
|
132
161
|
else:
|
|
133
|
-
add(
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
162
|
+
add(
|
|
163
|
+
"missing_outcome", "medium",
|
|
164
|
+
f"frame requests outcome {label!r} with no covering finding",
|
|
165
|
+
semantic_token=f"requested_outcome:{label}",
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
claim_outcomes = {
|
|
169
|
+
c["claim_id"]: c.get("primary_outcome_ids", [])
|
|
170
|
+
for c in claims
|
|
171
|
+
}
|
|
137
172
|
|
|
138
|
-
# contradiction gaps
|
|
139
173
|
for syn in syntheses or ():
|
|
140
174
|
if syn.status == "contested":
|
|
141
|
-
add(
|
|
175
|
+
add(
|
|
176
|
+
"unresolved_conflict", "high",
|
|
142
177
|
f"claim {syn.claim_id} has independent contradictory studies "
|
|
143
|
-
f"({', '.join(syn.study_ids)})",
|
|
144
|
-
|
|
178
|
+
f"({', '.join(syn.study_ids)})",
|
|
179
|
+
[syn.claim_id],
|
|
180
|
+
claim_outcomes.get(syn.claim_id, []),
|
|
181
|
+
semantic_token=f"claim:{syn.claim_id}",
|
|
182
|
+
)
|
|
145
183
|
|
|
146
|
-
# methodology weakness / insufficient independence
|
|
147
184
|
if syntheses:
|
|
148
185
|
for syn in syntheses:
|
|
149
186
|
if syn.status == "insufficient" and len(syn.study_ids) < 2:
|
|
150
|
-
add(
|
|
151
|
-
|
|
152
|
-
f"
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
187
|
+
add(
|
|
188
|
+
"insufficient_sample_independence", "medium",
|
|
189
|
+
f"claim {syn.claim_id} rests on fewer than two independent studies",
|
|
190
|
+
[syn.claim_id],
|
|
191
|
+
claim_outcomes.get(syn.claim_id, []),
|
|
192
|
+
semantic_token=f"claim:{syn.claim_id}",
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
for gap in gaps:
|
|
196
|
+
errors = validate_record("knowledge-gap", gap)
|
|
158
197
|
if errors:
|
|
159
198
|
raise ValueError(f"invalid gap: {errors}")
|
|
160
199
|
return gaps
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""Build a transparent, self-describing judge evidence pack from real artifacts."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import hashlib
|
|
5
|
+
import json
|
|
6
|
+
import shutil
|
|
7
|
+
from datetime import datetime, timezone
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
from engine.project import ProjectWorkspace
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _sha256(path: Path) -> str:
|
|
15
|
+
h = hashlib.sha256()
|
|
16
|
+
with path.open("rb") as fh:
|
|
17
|
+
for block in iter(lambda: fh.read(65536), b""):
|
|
18
|
+
h.update(block)
|
|
19
|
+
return h.hexdigest()
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def export_judge_pack(project: ProjectWorkspace, output_dir: Path) -> dict[str, Any]:
|
|
23
|
+
"""Copy available project evidence without inventing unavailable claims.
|
|
24
|
+
|
|
25
|
+
The manifest lists every required judge-pack category and explicitly marks
|
|
26
|
+
missing inputs. This makes the pack suitable for review while keeping its
|
|
27
|
+
limits auditable.
|
|
28
|
+
"""
|
|
29
|
+
output_dir = Path(output_dir).resolve()
|
|
30
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
31
|
+
candidates = {
|
|
32
|
+
"project_manifest": project.path / "project.json",
|
|
33
|
+
"graph": project.path / "graph",
|
|
34
|
+
"runs": project.path / "runs",
|
|
35
|
+
"decisions": project.path / "decisions",
|
|
36
|
+
"projections": project.path / "projections",
|
|
37
|
+
"reports": project.path / "reports",
|
|
38
|
+
"pilots": project.path / "pilots",
|
|
39
|
+
}
|
|
40
|
+
copied: list[dict[str, str]] = []
|
|
41
|
+
missing: list[str] = []
|
|
42
|
+
for name, source in candidates.items():
|
|
43
|
+
target = output_dir / name
|
|
44
|
+
if source.is_file():
|
|
45
|
+
shutil.copy2(source, target)
|
|
46
|
+
copied.append({"name": name, "path": target.name, "sha256": _sha256(target)})
|
|
47
|
+
elif source.is_dir() and any(source.rglob("*")):
|
|
48
|
+
shutil.copytree(source, target, dirs_exist_ok=True)
|
|
49
|
+
for item in sorted(path for path in target.rglob("*") if path.is_file()):
|
|
50
|
+
copied.append({"name": name, "path": str(item.relative_to(output_dir)), "sha256": _sha256(item)})
|
|
51
|
+
else:
|
|
52
|
+
missing.append(name)
|
|
53
|
+
manifest = {
|
|
54
|
+
"format": "eduevidence-judge-pack/2026.09",
|
|
55
|
+
"project_id": project.project_id,
|
|
56
|
+
"created_at": datetime.now(timezone.utc).isoformat(),
|
|
57
|
+
"copied_files": copied,
|
|
58
|
+
"missing_categories": missing,
|
|
59
|
+
"limitations": [
|
|
60
|
+
"Only immutable/project-scoped artifacts available at export time are included.",
|
|
61
|
+
"Benchmark, blinded-review and usability evidence must be supplied from completed study artifacts; they are never synthesized by this export.",
|
|
62
|
+
],
|
|
63
|
+
}
|
|
64
|
+
(output_dir / "manifest.json").write_text(json.dumps(manifest, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
|
65
|
+
return manifest
|
|
@@ -42,7 +42,9 @@ from functools import lru_cache
|
|
|
42
42
|
from pathlib import Path
|
|
43
43
|
from typing import Any
|
|
44
44
|
|
|
45
|
-
|
|
45
|
+
from engine._resources import resource_root
|
|
46
|
+
|
|
47
|
+
ROOT = resource_root()
|
|
46
48
|
def _resolve_library_path() -> Path:
|
|
47
49
|
"""Repository layout first; wheel-installed share/ layout as fallback."""
|
|
48
50
|
repo = ROOT / "benchmarks" / "evidence-library.json"
|
package/engine/living.py
CHANGED
|
@@ -43,7 +43,8 @@ from scripts.validate_schema import SchemaError, validate
|
|
|
43
43
|
def _resolve_v4_schema_dir() -> Path:
|
|
44
44
|
"""Repository layout first; wheel-installed share/ layout as fallback
|
|
45
45
|
(same pattern as engine/contracts._resolve_schema_dir)."""
|
|
46
|
-
|
|
46
|
+
from engine._resources import resource_root
|
|
47
|
+
repo = resource_root() / "schemas" / "v4"
|
|
47
48
|
if repo.is_dir():
|
|
48
49
|
return repo
|
|
49
50
|
import sys
|
package/engine/meta_synthesis.py
CHANGED
|
@@ -21,7 +21,9 @@ from engine.ids import new_local_id
|
|
|
21
21
|
from engine.library import ResearchLibrary
|
|
22
22
|
from scripts.validate_schema import SchemaError, validate
|
|
23
23
|
|
|
24
|
-
|
|
24
|
+
from engine._resources import resource_root
|
|
25
|
+
|
|
26
|
+
_SYNTHESIS_SCHEMA = (resource_root() / "schemas" / "v3"
|
|
25
27
|
/ "synthesis.schema.json")
|
|
26
28
|
|
|
27
29
|
|