eduevidence 6.0.0 → 6.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +395 -0
- package/CONTRIBUTING.md +105 -0
- package/README.md +113 -49
- package/README.zh-CN.md +39 -12
- package/SKILL.md +15 -5
- package/assets/readme/landing-tour.gif +0 -0
- package/assets/readme/studio-tour.gif +0 -0
- package/benchmarks/evidence-library.json +277 -1
- package/bin/eduevidence.js +2 -1
- package/docs/architecture.md +325 -46
- package/docs/demo-workplace-ai.md +1 -1
- package/docs/install-guide.md +1 -1
- package/docs/j-ev-experimental.md +250 -0
- package/docs/orchestration-role-model.md +1 -1
- package/docs/release-closeout/README.md +1 -1
- package/docs/reproducibility.md +138 -0
- package/docs/sciverse-api.md +125 -0
- package/domains/_neutral/copy/few_shots.json +21 -0
- package/domains/_neutral/copy/framing_lexicon.json +19 -0
- package/domains/_neutral/copy/module_labels.json +5 -0
- package/domains/_neutral/copy/module_labels_footer.json +102 -0
- package/domains/_neutral/copy/module_labels_modules.json +204 -0
- package/domains/_neutral/copy/module_labels_nav.json +126 -0
- package/domains/_neutral/copy/module_labels_summary.json +98 -0
- package/domains/_neutral/copy/module_labels_tables.json +164 -0
- package/domains/_neutral/copy/module_labels_v2.json +90 -0
- package/domains/_neutral/copy/risk_constructs.json +20 -0
- package/domains/_neutral/copy/section_titles.json +66 -0
- package/domains/_neutral/copy/terminology.json +11 -0
- package/domains/check_copy_packs.py +103 -0
- package/domains/education/copy/few_shots.json +22 -0
- package/domains/education/copy/framing_enums.json +167 -0
- package/domains/education/copy/framing_lexicon.json +166 -0
- package/domains/education/copy/module_labels.json +169 -0
- package/domains/education/copy/risk_constructs.json +48 -0
- package/domains/education/copy/section_titles.json +186 -0
- package/domains/education/copy/terminology.json +70 -0
- package/domains/education/manifest.json +1 -1
- package/domains/education/outcome_taxonomy.json +2 -2
- package/domains/manifest.json +1 -1
- package/domains/policy/copy/few_shots.json +22 -0
- package/domains/policy/copy/framing_enums.json +94 -0
- package/domains/policy/copy/framing_lexicon.json +174 -0
- package/domains/policy/copy/module_labels.json +168 -0
- package/domains/policy/copy/risk_constructs.json +33 -0
- package/domains/policy/copy/section_titles.json +186 -0
- package/domains/policy/copy/terminology.json +64 -0
- package/eduevidence_cli.py +10 -0
- package/engine/capabilities.py +57 -5
- package/engine/decision_policy.py +167 -0
- package/engine/evidence_graph.py +14 -10
- package/engine/gaps.py +42 -22
- package/engine/ids.py +2 -0
- package/engine/library.py +6 -2
- package/engine/library_builtin.py +7 -4
- package/engine/living.py +34 -4
- package/engine/migration.py +88 -3
- package/engine/orchestration.py +5 -5
- package/engine/paths.py +2 -0
- package/engine/pilot.py +34 -32
- package/engine/taxonomy.py +211 -0
- package/engine/tribunal.py +49 -43
- package/engine/versions.py +1 -1
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1361 -147
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
- package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
- package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
- package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +448 -128
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +448 -128
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +448 -128
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +448 -128
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +448 -128
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
- package/examples/ai-coding-assistant-evidence/result.json +13 -9
- package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
- package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
- package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
- package/examples/spaced-retrieval-practice/applicability.json +14 -0
- package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
- package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
- package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
- package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
- package/examples/spaced-retrieval-practice/frame.json +58 -0
- package/examples/spaced-retrieval-practice/gate_report.json +101 -0
- package/examples/spaced-retrieval-practice/methodology.json +78 -0
- package/examples/spaced-retrieval-practice/report.html +2522 -0
- package/examples/spaced-retrieval-practice/report_spec.json +212 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/result.json +942 -0
- package/examples/spaced-retrieval-practice/result.zh.json +942 -0
- package/examples/spaced-retrieval-practice/skeptic.json +70 -0
- package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
- package/examples/spaced-retrieval-practice/verdict.json +93 -0
- package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
- package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -4
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
- package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
- package/examples/workplace-ai-assistant/final_verdict.json +78 -0
- package/examples/workplace-ai-assistant/gate_report.json +101 -0
- package/examples/workplace-ai-assistant/report_spec.json +209 -40
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +449 -119
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +449 -119
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +449 -119
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +449 -119
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +449 -119
- package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/result.json +82 -20
- package/examples/workplace-ai-assistant/result.zh.json +82 -20
- package/examples/workplace-ai-assistant/skeptic.json +72 -0
- package/examples/workplace-ai-assistant/verdict.json +36 -10
- package/integrations/agent_mcp.py +2 -2
- package/integrations/jev/__init__.py +115 -0
- package/integrations/jev/approval.py +212 -0
- package/integrations/jev/cli.py +84 -0
- package/integrations/jev/config.py +112 -0
- package/integrations/jev/gateway.py +128 -0
- package/integrations/jev/modes.py +38 -0
- package/integrations/jev/tools_classify.py +88 -0
- package/integrations/jev/tools_extract.py +111 -0
- package/integrations/jev/tools_rerank.py +71 -0
- package/integrations/jev/tools_screen.py +87 -0
- package/integrations/jev/tools_verify.py +95 -0
- package/integrations/jev_mcp.py +22 -0
- package/integrations/semantic_decide.py +286 -0
- package/integrations/semdecide_cli.py +55 -0
- package/package.json +19 -2
- package/pyproject.toml +4 -3
- package/references/report-copy-style.md +107 -0
- package/references/retrieval-compliance.md +75 -0
- package/references/retrieval-protocol.md +20 -0
- package/retrieval/audit.py +27 -3
- package/retrieval/fetch.py +96 -0
- package/retrieval/sciverse.py +398 -0
- package/retrieval/search.py +47 -7
- package/schemas/applicability.schema.json +94 -0
- package/schemas/chart-spec.schema.json +10 -3
- package/schemas/evidence.schema.json +316 -43
- package/schemas/fetch-result.schema.json +2 -1
- package/schemas/report-result.schema.json +3 -3
- package/schemas/report-spec.schema.json +98 -100
- package/schemas/skeptic.schema.json +86 -0
- package/schemas/source.schema.json +21 -2
- package/schemas/v2/decision-snapshot.schema.json +20 -9
- package/schemas/v2/finding.schema.json +5 -1
- package/schemas/v2/intake.schema.json +191 -0
- package/schemas/v2/methodology-audit.schema.json +5 -1
- package/schemas/v2/outcome.schema.json +28 -5
- package/schemas/v2/study.schema.json +5 -1
- package/schemas/vNext/autoevolve-session.schema.json +34 -1
- package/schemas/vNext/eval-snapshot.schema.json +77 -1
- package/schemas/vNext/execution-plan.schema.json +50 -1
- package/schemas/vNext/gap-priority.schema.json +54 -1
- package/schemas/vNext/negative-search-record.schema.json +68 -1
- package/schemas/vNext/research-iteration.schema.json +87 -1
- package/schemas/vNext/research-strategy.schema.json +62 -1
- package/schemas/vNext/skill-experiment.schema.json +90 -1
- package/schemas/vNext/task-spec.schema.json +156 -1
- package/schemas/vNext/worker-result.schema.json +60 -1
- package/schemas/verdict.schema.json +164 -28
- package/scripts/build_evidence_library.py +15 -5
- package/scripts/build_report_variants.py +18 -2
- package/scripts/build_result.py +74 -9
- package/scripts/check_package_parity.py +85 -0
- package/scripts/check_protocol_alignment.py +375 -0
- package/scripts/check_versioned_schemas.py +254 -0
- package/scripts/claim_audit.py +13 -8
- package/scripts/compute_confidence.py +10 -0
- package/scripts/dashboard_server.py +13 -2
- package/scripts/did_regression.py +12 -2
- package/scripts/evidence_score.py +5 -2
- package/scripts/intake/__init__.py +31 -0
- package/scripts/intake/__main__.py +18 -0
- package/scripts/intake/background.py +78 -0
- package/scripts/intake/browser.py +79 -0
- package/scripts/intake/cli.py +57 -0
- package/scripts/intake/constants.py +57 -0
- package/scripts/intake/depth.py +53 -0
- package/scripts/intake/enhancements.py +106 -0
- package/scripts/intake/hooks.py +90 -0
- package/scripts/intake/prefs.py +76 -0
- package/scripts/intake/prompts.py +85 -0
- package/scripts/intake/session.py +152 -0
- package/scripts/lint_file_layers.py +126 -0
- package/scripts/orchestrator.py +187 -40
- package/scripts/pre_verdict_gate.py +241 -29
- package/scripts/quickstart.py +18 -2
- package/scripts/run_workspace.py +7 -1
- package/scripts/skill_lint.py +11 -1
- package/scripts/skill_payload.py +6 -3
- package/scripts/test_adversarial_empirical.py +96 -25
- package/scripts/validate_schema.py +31 -1
- package/skill/agents/evaluation-designer.md +20 -4
- package/skill/agents/evidence-analyst.md +19 -3
- package/skill/agents/evidence-judge.md +98 -8
- package/skill/agents/evidence-retriever.md +20 -3
- package/skill/agents/intervention-designer.md +20 -4
- package/skill/agents/method-reviewer.md +18 -2
- package/skill/agents/{education-planner.md → research-planner.md} +19 -3
- package/skill/agents/skeptic.md +18 -2
- package/skill/roles/registry.yaml +11 -11
- package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
- package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
- package/skill/sub-skills/data-analysis/SKILL.md +34 -15
- package/skill/sub-skills/ethics-review/SKILL.md +33 -10
- package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
- package/skill/sub-skills/evidence-review/SKILL.md +31 -12
- package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
- package/skill/sub-skills/literature-review/SKILL.md +35 -14
- package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
- package/skill/sub-skills/report-generation/SKILL.md +28 -0
- package/skill/sub-skills/research-planning/SKILL.md +41 -14
- package/skill/sub-skills/study-design/SKILL.md +30 -9
- package/skill/task-briefs/adjudicate.md +32 -7
- package/skill/task-briefs/applicability.md +37 -2
- package/skill/task-briefs/audit.md +32 -7
- package/skill/task-briefs/challenge.md +34 -5
- package/skill/task-briefs/evaluate.md +30 -5
- package/skill/task-briefs/extract.md +31 -8
- package/skill/task-briefs/frame.md +39 -10
- package/skill/task-briefs/intervene.md +32 -6
- package/skill/task-briefs/present.md +32 -8
- package/skill/task-briefs/projection.md +36 -2
- package/skill/task-briefs/retrieve.md +36 -6
- package/skill/workflows/decision-and-pilot.md +76 -1
- package/skill/workflows/evaluate-and-update.md +83 -0
- package/skill/workflows/evidence-review.md +104 -0
- package/skill/workflows/experimental-jev.md +170 -0
- package/skill/workflows/intake.md +120 -0
- package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
- package/visualization/eduevidence-report/scripts/build_infographics.py +37 -15
- package/visualization/eduevidence-report/scripts/build_report.py +435 -575
- package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
- package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
- package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
- package/visualization/eduevidence-report/scripts/zh_labels.py +141 -1
- package/web/architecture.html +14885 -0
- package/web/studio/assets/index-B8tkF44Q.css +1 -0
- package/web/studio/index.html +2 -2
- package/scripts/build_esl_artifacts.py +0 -1921
- package/scripts/build_killer_demo.py +0 -295
- package/scripts/enrich_projects_human_and_lieflat.py +0 -315
- package/scripts/generate_new_projects.py +0 -686
- package/scripts/sync_killer_demo_report.py +0 -270
- package/web/studio/assets/index-CzXocaGv.css +0 -1
- /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""check_package_parity.py - prove the shipped package matches the source tree.
|
|
3
|
+
|
|
4
|
+
The upload package is a projection of this repository. If a source file changed
|
|
5
|
+
and the package still carries the old bytes, reviewers receive a different
|
|
6
|
+
product from the one under source control. CI previously compared only
|
|
7
|
+
SKILL.md, so a renamed role file went unnoticed for a whole change set.
|
|
8
|
+
|
|
9
|
+
Compares every file the shared payload allowlist ships: content must match and
|
|
10
|
+
nothing may be missing. Stdlib only; exit 1 on any drift.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import hashlib
|
|
15
|
+
import sys
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
ROOT = Path(__file__).resolve().parent.parent
|
|
19
|
+
sys.path.insert(0, str(ROOT))
|
|
20
|
+
sys.path.insert(0, str(ROOT / "scripts"))
|
|
21
|
+
|
|
22
|
+
PACKAGE = ROOT / "dist" / "eduevidence-submission"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def digest(path: Path) -> str:
|
|
26
|
+
return hashlib.sha256(path.read_bytes()).hexdigest()
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def main() -> int:
|
|
30
|
+
if not PACKAGE.is_dir():
|
|
31
|
+
print(f"ERROR: no package at {PACKAGE}; run bash packaging/make_upload.sh")
|
|
32
|
+
return 1
|
|
33
|
+
|
|
34
|
+
from skill_payload import payload_files
|
|
35
|
+
|
|
36
|
+
expected = set(payload_files(ROOT))
|
|
37
|
+
# The manifest and the packaging notes are written by the build, not copied.
|
|
38
|
+
generated = {"submission-manifest.json", "UPLOAD-README.md", "START-HERE.md",
|
|
39
|
+
"scp-manifest.json", "upload-layout.md", "README.zh-CN.md"}
|
|
40
|
+
expected |= {name for name in generated if (PACKAGE / name).is_file()}
|
|
41
|
+
|
|
42
|
+
missing = sorted(rel for rel in expected if not (PACKAGE / rel).is_file())
|
|
43
|
+
|
|
44
|
+
# Reverse direction: a file the package carries but the source does not is
|
|
45
|
+
# stale output from an earlier build. A one-way check cannot see that,
|
|
46
|
+
# which is how pre-rename report copies once survived inside a package.
|
|
47
|
+
unexpected = []
|
|
48
|
+
for path in sorted(PACKAGE.rglob("*")):
|
|
49
|
+
if not path.is_file():
|
|
50
|
+
continue
|
|
51
|
+
rel = path.relative_to(PACKAGE).as_posix()
|
|
52
|
+
if rel in expected or rel == "submission-manifest.json":
|
|
53
|
+
continue
|
|
54
|
+
if any(part in {"__pycache__", ".git"} for part in path.parts):
|
|
55
|
+
continue
|
|
56
|
+
if path.suffix in {".pyc", ".pyo"} or path.name == ".DS_Store":
|
|
57
|
+
continue
|
|
58
|
+
if not (ROOT / rel).exists():
|
|
59
|
+
unexpected.append(rel)
|
|
60
|
+
differing = []
|
|
61
|
+
for rel in sorted(expected):
|
|
62
|
+
source = ROOT / rel
|
|
63
|
+
shipped = PACKAGE / rel
|
|
64
|
+
if not source.is_file() or not shipped.is_file():
|
|
65
|
+
continue
|
|
66
|
+
if digest(source) != digest(shipped):
|
|
67
|
+
differing.append(rel)
|
|
68
|
+
|
|
69
|
+
if missing or differing or unexpected:
|
|
70
|
+
print("ERROR: package does not match the source tree", file=sys.stderr)
|
|
71
|
+
for rel in missing[:20]:
|
|
72
|
+
print(f" missing from package: {rel}", file=sys.stderr)
|
|
73
|
+
for rel in differing[:20]:
|
|
74
|
+
print(f" differs from source: {rel}", file=sys.stderr)
|
|
75
|
+
for rel in unexpected[:20]:
|
|
76
|
+
print(f" stale in package: {rel}", file=sys.stderr)
|
|
77
|
+
print(" fix: bash packaging/make_upload.sh", file=sys.stderr)
|
|
78
|
+
return 1
|
|
79
|
+
|
|
80
|
+
print(f"package parity OK ({len(expected)} files byte-identical to source)")
|
|
81
|
+
return 0
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
if __name__ == "__main__":
|
|
85
|
+
sys.exit(main())
|
|
@@ -0,0 +1,375 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""check_protocol_alignment.py — Protocol five-way alignment gate.
|
|
3
|
+
|
|
4
|
+
Every scientific contract in this repository is declared in more than one
|
|
5
|
+
place: the workflow registry (engine/workflows.py), the capability registry
|
|
6
|
+
(engine/capabilities.py), the role registry (skill/roles/registry.yaml), the
|
|
7
|
+
role prompts (skill/agents/*.md), the stage briefs (skill/task-briefs/*.md),
|
|
8
|
+
the sub-skill recipes (skill/sub-skills/*/SKILL.md), the routing requirements
|
|
9
|
+
(integrations/agent_mcp.py) and the packaging manifest (packaging/*).
|
|
10
|
+
|
|
11
|
+
Drift between them is silent: a role can lose its brief, a capability can
|
|
12
|
+
exist with no recipe, a version can be bumped in one file only. This gate
|
|
13
|
+
makes that drift fail loudly. Stdlib only; a non-zero exit blocks CI.
|
|
14
|
+
|
|
15
|
+
Usage:
|
|
16
|
+
python3 scripts/check_protocol_alignment.py
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import json
|
|
21
|
+
import re
|
|
22
|
+
import sys
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
|
|
25
|
+
ROOT = Path(__file__).resolve().parent.parent
|
|
26
|
+
sys.path.insert(0, str(ROOT))
|
|
27
|
+
|
|
28
|
+
from engine.capabilities import capability_registry # noqa: E402
|
|
29
|
+
from engine.versions import ENGINE_VERSION # noqa: E402
|
|
30
|
+
from engine.workflows import execution_stages, workflow_registry # noqa: E402
|
|
31
|
+
|
|
32
|
+
FRONTMATTER_RE = re.compile(r"^---\s*\n(.*?)\n---\s*\n", re.DOTALL)
|
|
33
|
+
|
|
34
|
+
#: Projection-layer capabilities sit outside the scientific stage model
|
|
35
|
+
#: (engine/workflows.py: Projection is not a scientific stage), so they are
|
|
36
|
+
#: owned by the projection brief rather than by a scientific role.
|
|
37
|
+
PROJECTION_CAPABILITIES = {"report_projection", "report_rendering"}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def list_domains_from_registry() -> list[dict]:
|
|
41
|
+
"""Registered domains (evidencecore is the registry owner)."""
|
|
42
|
+
from engine.evidencecore import list_domains
|
|
43
|
+
|
|
44
|
+
return list_domains()
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _frontmatter(path: Path) -> dict[str, str]:
|
|
48
|
+
"""Parse the flat key: value frontmatter used by skills and role prompts."""
|
|
49
|
+
match = FRONTMATTER_RE.match(path.read_text(encoding="utf-8"))
|
|
50
|
+
if not match:
|
|
51
|
+
return {}
|
|
52
|
+
fields: dict[str, str] = {}
|
|
53
|
+
for line in match.group(1).splitlines():
|
|
54
|
+
if line.strip().startswith("#") or ":" not in line:
|
|
55
|
+
continue
|
|
56
|
+
key, _, value = line.partition(":")
|
|
57
|
+
fields[key.strip()] = value.split("#", 1)[0].strip()
|
|
58
|
+
return fields
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _registry_roles() -> dict[str, dict]:
|
|
62
|
+
"""The role registry is a small fixed-shape YAML file; parse it narrowly."""
|
|
63
|
+
text = (ROOT / "skill" / "roles" / "registry.yaml").read_text(encoding="utf-8")
|
|
64
|
+
roles: dict[str, dict] = {}
|
|
65
|
+
current: str | None = None
|
|
66
|
+
for raw in text.splitlines():
|
|
67
|
+
if not raw.strip() or raw.strip().startswith("#"):
|
|
68
|
+
continue
|
|
69
|
+
if raw.startswith("roles:"):
|
|
70
|
+
continue
|
|
71
|
+
if raw.startswith("execution:"):
|
|
72
|
+
break
|
|
73
|
+
if re.match(r"^ [A-Za-z0-9_-]+:\s*$", raw):
|
|
74
|
+
current = raw.strip().rstrip(":")
|
|
75
|
+
roles[current] = {}
|
|
76
|
+
continue
|
|
77
|
+
if current and raw.strip().startswith("stages:"):
|
|
78
|
+
stages = raw.split(":", 1)[1].strip().strip("[]")
|
|
79
|
+
roles[current]["stages"] = [s.strip() for s in stages.split(",") if s.strip()]
|
|
80
|
+
elif current and raw.strip().startswith("capabilities:"):
|
|
81
|
+
caps = raw.split(":", 1)[1].strip().strip("[]")
|
|
82
|
+
roles[current]["capabilities"] = [c.strip() for c in caps.split(",") if c.strip()]
|
|
83
|
+
elif current and ":" in raw.strip():
|
|
84
|
+
key, _, value = raw.strip().partition(":")
|
|
85
|
+
roles[current][key.strip()] = value.strip()
|
|
86
|
+
return roles
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _merge_role(roles: dict[str, dict], name: str, stage: str, capability: str,
|
|
90
|
+
critical: bool | None = None, independence: bool | None = None) -> None:
|
|
91
|
+
entry = roles.setdefault(name, {"stages": [], "capabilities": []})
|
|
92
|
+
if stage not in entry["stages"]:
|
|
93
|
+
entry["stages"].append(stage)
|
|
94
|
+
for cap in capability.split("+"):
|
|
95
|
+
cap = cap.strip()
|
|
96
|
+
if cap and cap not in entry["capabilities"]:
|
|
97
|
+
entry["capabilities"].append(cap)
|
|
98
|
+
if critical is not None:
|
|
99
|
+
entry["critical_path"] = "true" if critical else "false"
|
|
100
|
+
if independence:
|
|
101
|
+
entry["independence_required"] = "true"
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def check() -> list[str]:
|
|
105
|
+
errors: list[str] = []
|
|
106
|
+
|
|
107
|
+
stages = list(execution_stages())
|
|
108
|
+
scientific_stages = [s for s in stages if s != "projection"]
|
|
109
|
+
|
|
110
|
+
# ---------------------------------------------------------------- briefs
|
|
111
|
+
brief_dir = ROOT / "skill" / "task-briefs"
|
|
112
|
+
briefs = {p.stem for p in brief_dir.glob("*.md")}
|
|
113
|
+
for stage in stages:
|
|
114
|
+
if stage not in briefs:
|
|
115
|
+
errors.append(f"stage {stage!r} has no task brief in skill/task-briefs/")
|
|
116
|
+
for extra in sorted(briefs - set(stages) - {"present"}):
|
|
117
|
+
errors.append(f"task brief {extra!r} does not map to a canonical stage")
|
|
118
|
+
|
|
119
|
+
# ---------------------------------------------------------------- roles
|
|
120
|
+
registry = _registry_roles()
|
|
121
|
+
if not registry:
|
|
122
|
+
errors.append("skill/roles/registry.yaml declares no roles")
|
|
123
|
+
agent_files = {p.stem: p for p in (ROOT / "skill" / "agents").glob("*.md")}
|
|
124
|
+
if set(registry) != set(agent_files):
|
|
125
|
+
errors.append("role registry and skill/agents/*.md disagree: "
|
|
126
|
+
f"registry-only={sorted(set(registry) - set(agent_files))} "
|
|
127
|
+
f"agent-only={sorted(set(agent_files) - set(registry))}")
|
|
128
|
+
|
|
129
|
+
stage_owner: dict[str, str] = {}
|
|
130
|
+
for role, entry in registry.items():
|
|
131
|
+
for stage in entry.get("stages", []):
|
|
132
|
+
if stage in stage_owner:
|
|
133
|
+
errors.append(f"stage {stage!r} is owned by both {stage_owner[stage]!r} and {role!r}")
|
|
134
|
+
stage_owner[stage] = role
|
|
135
|
+
for stage in scientific_stages:
|
|
136
|
+
if stage not in stage_owner:
|
|
137
|
+
errors.append(f"stage {stage!r} has no owning role in the registry")
|
|
138
|
+
|
|
139
|
+
# Registry capabilities must be engine capability IDs: a free-text label
|
|
140
|
+
# here silently detaches the role from the capability it claims to run.
|
|
141
|
+
for role, entry in registry.items():
|
|
142
|
+
for cap in entry.get("capabilities", []):
|
|
143
|
+
if cap not in capability_registry():
|
|
144
|
+
errors.append(f"registry role {role!r} declares capability {cap!r}, "
|
|
145
|
+
"which is not in engine/capabilities.py")
|
|
146
|
+
|
|
147
|
+
# Independence is graded: the skeptic must come from a different model
|
|
148
|
+
# family, the method reviewer must be separated from the content judgement.
|
|
149
|
+
independence = {role: entry.get("independence_required")
|
|
150
|
+
for role, entry in registry.items() if entry.get("independence_required")}
|
|
151
|
+
if independence != {"skeptic": "different-model-family",
|
|
152
|
+
"method-reviewer": "role-separation"}:
|
|
153
|
+
expected = {"skeptic": "different-model-family", "method-reviewer": "role-separation"}
|
|
154
|
+
errors.append("independence_required must be "
|
|
155
|
+
f"{expected}, found {independence}")
|
|
156
|
+
|
|
157
|
+
# ------------------------------------------------- role prompt frontmatter
|
|
158
|
+
for role, path in agent_files.items():
|
|
159
|
+
fields = _frontmatter(path)
|
|
160
|
+
if fields.get("name") != role:
|
|
161
|
+
errors.append(f"{path.name}: frontmatter name {fields.get('name')!r} != filename {role!r}")
|
|
162
|
+
if fields.get("role_id") != role:
|
|
163
|
+
errors.append(f"{path.name}: missing role_id: {role}")
|
|
164
|
+
if not fields.get("capabilities"):
|
|
165
|
+
errors.append(f"{path.name}: missing capabilities declaration")
|
|
166
|
+
if not fields.get("output_contracts"):
|
|
167
|
+
errors.append(f"{path.name}: missing output_contracts declaration")
|
|
168
|
+
for banned in ("default_cli", "default_model"):
|
|
169
|
+
if banned in fields:
|
|
170
|
+
errors.append(f"{path.name}: {banned} must not be bound in the role prompt "
|
|
171
|
+
"(model/CLI choice is a user-confirmed routing decision)")
|
|
172
|
+
for token in ("claude-", "gpt-", "deepseek-", "glm-", "kimi-"):
|
|
173
|
+
if token in fields.get("recommended_reasoning", ""):
|
|
174
|
+
errors.append(f"{path.name}: recommended_reasoning must not contain a model name")
|
|
175
|
+
if role in registry:
|
|
176
|
+
declared = set(registry[role].get("capabilities", []))
|
|
177
|
+
prompt_caps = {c.strip() for c in fields.get("capabilities", "").split(",") if c.strip()}
|
|
178
|
+
unknown = {c for c in prompt_caps if c not in capability_registry()}
|
|
179
|
+
unmapped = {c for c in unknown if not c.startswith("(")}
|
|
180
|
+
if unmapped:
|
|
181
|
+
errors.append(f"{path.name}: capabilities not in the capability registry: {sorted(unmapped)}")
|
|
182
|
+
missing = {c for c in declared if c in capability_registry()} - prompt_caps
|
|
183
|
+
if missing:
|
|
184
|
+
errors.append(f"{path.name}: registry capabilities absent from the prompt: "
|
|
185
|
+
f"{sorted(missing)}")
|
|
186
|
+
if registry.get(role, {}).get("critical_path") == "true":
|
|
187
|
+
if fields.get("critical_path") != "true":
|
|
188
|
+
errors.append(f"{path.name}: registry marks this role critical_path but the "
|
|
189
|
+
"prompt does not declare critical_path: true")
|
|
190
|
+
|
|
191
|
+
# ----------------------------------------------- routing-side requirements
|
|
192
|
+
from integrations.agent_mcp import ROLE_REQUIREMENTS # noqa: E402
|
|
193
|
+
if set(ROLE_REQUIREMENTS) != set(registry):
|
|
194
|
+
errors.append("integrations.agent_mcp.ROLE_REQUIREMENTS and the role registry disagree: "
|
|
195
|
+
f"routing-only={sorted(set(ROLE_REQUIREMENTS) - set(registry))} "
|
|
196
|
+
f"registry-only={sorted(set(registry) - set(ROLE_REQUIREMENTS))}")
|
|
197
|
+
for role, reqs in ROLE_REQUIREMENTS.items():
|
|
198
|
+
for banned in ("default_cli", "default_model", "model", "cli"):
|
|
199
|
+
if banned in reqs:
|
|
200
|
+
errors.append(f"ROLE_REQUIREMENTS[{role!r}] must not bind {banned!r}")
|
|
201
|
+
wants_family = registry.get(role, {}).get("independence_required") == "different-model-family"
|
|
202
|
+
if wants_family and reqs.get("independence") != "different-model-family":
|
|
203
|
+
errors.append(f"ROLE_REQUIREMENTS[{role!r}] must require a different model family")
|
|
204
|
+
|
|
205
|
+
# ------------------------------------------------------------ capabilities
|
|
206
|
+
capabilities = capability_registry()
|
|
207
|
+
sub_skills = sorted(p for p in (ROOT / "skill" / "sub-skills").iterdir()
|
|
208
|
+
if p.is_dir() and not p.name.startswith("."))
|
|
209
|
+
if len(sub_skills) < 5:
|
|
210
|
+
errors.append(f"expected at least 5 sub-skills, found {len(sub_skills)}")
|
|
211
|
+
mapped: set[str] = set()
|
|
212
|
+
for skill_dir in sub_skills:
|
|
213
|
+
path = skill_dir / "SKILL.md"
|
|
214
|
+
if not path.is_file():
|
|
215
|
+
errors.append(f"sub-skill {skill_dir.name} has no SKILL.md")
|
|
216
|
+
continue
|
|
217
|
+
fields = _frontmatter(path)
|
|
218
|
+
if fields.get("name") != skill_dir.name:
|
|
219
|
+
errors.append(f"{skill_dir.name}/SKILL.md: name {fields.get('name')!r} != directory")
|
|
220
|
+
declared = fields.get("capability", "")
|
|
221
|
+
if not declared:
|
|
222
|
+
errors.append(f"{skill_dir.name}/SKILL.md: missing capability declaration")
|
|
223
|
+
continue
|
|
224
|
+
for cap in declared.split("+"):
|
|
225
|
+
cap = cap.strip().split()[0] if cap.strip() else ""
|
|
226
|
+
if cap and not cap.startswith("("):
|
|
227
|
+
mapped.add(cap)
|
|
228
|
+
if cap not in capabilities:
|
|
229
|
+
errors.append(f"{skill_dir.name}/SKILL.md: capability {cap!r} is not in "
|
|
230
|
+
"engine/capabilities.py")
|
|
231
|
+
# Every registered capability must be owned by at least one role, and every
|
|
232
|
+
# recipe capability must be one the engine actually registers.
|
|
233
|
+
owned: set[str] = set()
|
|
234
|
+
for entry in registry.values():
|
|
235
|
+
owned.update(entry.get("capabilities", []))
|
|
236
|
+
for cap in capabilities:
|
|
237
|
+
if cap not in owned and cap not in PROJECTION_CAPABILITIES:
|
|
238
|
+
errors.append(f"capability {cap!r} is registered in engine/capabilities.py "
|
|
239
|
+
"but no role owns it")
|
|
240
|
+
for cap in sorted(mapped):
|
|
241
|
+
if cap in capabilities and cap not in owned and cap not in PROJECTION_CAPABILITIES:
|
|
242
|
+
errors.append(f"sub-skill capability {cap!r} is owned by no role")
|
|
243
|
+
|
|
244
|
+
# ---------------------------------------------------------------- workflows
|
|
245
|
+
workflow_dir = ROOT / "skill" / "workflows"
|
|
246
|
+
workflow_files = {p.stem for p in workflow_dir.glob("*.md")}
|
|
247
|
+
registry_workflows = workflow_registry()
|
|
248
|
+
public = {w for w in registry_workflows if w != "full_research_cycle"}
|
|
249
|
+
normalised = {name.replace("_", "-") for name in public}
|
|
250
|
+
if not normalised <= workflow_files:
|
|
251
|
+
errors.append("workflows missing a runbook: "
|
|
252
|
+
f"{sorted(normalised - workflow_files)}")
|
|
253
|
+
skill_text = (ROOT / "SKILL.md").read_text(encoding="utf-8")
|
|
254
|
+
for name in public:
|
|
255
|
+
reference = f"skill/workflows/{name.replace('_', '-')}.md"
|
|
256
|
+
if reference not in skill_text:
|
|
257
|
+
errors.append(f"SKILL.md does not route to {reference}")
|
|
258
|
+
|
|
259
|
+
skill_md = root_skill_text = skill_text # alias for readability
|
|
260
|
+
for stage in scientific_stages:
|
|
261
|
+
if f"| {stage.capitalize()} " not in skill_md and stage not in skill_md:
|
|
262
|
+
errors.append(f"SKILL.md does not mention stage {stage!r}")
|
|
263
|
+
|
|
264
|
+
# ------------------------------------------------------------- taxonomy
|
|
265
|
+
# The registry is the authority for outcome tokens and their categories;
|
|
266
|
+
# the JSON Schemas carry static enums because JSON Schema cannot read a
|
|
267
|
+
# file at validation time. This dimension is what keeps the static enums
|
|
268
|
+
# honest: drift between a schema enum and the registry fails the gate.
|
|
269
|
+
from engine.taxonomy import (
|
|
270
|
+
all_tokens_ordered,
|
|
271
|
+
categories as taxonomy_categories,
|
|
272
|
+
)
|
|
273
|
+
|
|
274
|
+
registered_tokens = set(all_tokens_ordered())
|
|
275
|
+
evidence_path = ROOT / "schemas" / "evidence.schema.json"
|
|
276
|
+
if evidence_path.is_file():
|
|
277
|
+
evidence_schema = json.loads(evidence_path.read_text(encoding="utf-8"))
|
|
278
|
+
enum = (evidence_schema.get("properties", {})
|
|
279
|
+
.get("outcome_type", {}).get("enum"))
|
|
280
|
+
if not isinstance(enum, list) or not enum:
|
|
281
|
+
errors.append("schemas/evidence.schema.json declares no outcome_type enum")
|
|
282
|
+
else:
|
|
283
|
+
missing = sorted(registered_tokens - set(enum))
|
|
284
|
+
extra = sorted(set(enum) - registered_tokens)
|
|
285
|
+
if missing:
|
|
286
|
+
errors.append(
|
|
287
|
+
"outcome_type enum is missing registered token(s): " + repr(missing))
|
|
288
|
+
if extra:
|
|
289
|
+
errors.append(
|
|
290
|
+
"outcome_type enum declares unregistered token(s): " + repr(extra))
|
|
291
|
+
|
|
292
|
+
# Every domain's ADOPT-gate categories must be categories it declares.
|
|
293
|
+
from engine.tribunal import primary_effect_categories
|
|
294
|
+
|
|
295
|
+
for domain_id in (d["id"] for d in list_domains_from_registry()):
|
|
296
|
+
declared = set(taxonomy_categories(domain_id))
|
|
297
|
+
try:
|
|
298
|
+
primary = primary_effect_categories(domain_id)
|
|
299
|
+
except ValueError as exc:
|
|
300
|
+
errors.append(f"domain {domain_id!r}: ADOPT gate misconfigured: {exc}")
|
|
301
|
+
continue
|
|
302
|
+
for category in primary:
|
|
303
|
+
if category not in declared:
|
|
304
|
+
errors.append(
|
|
305
|
+
f"domain {domain_id!r}: ADOPT gate names undeclared category "
|
|
306
|
+
+ repr(category))
|
|
307
|
+
|
|
308
|
+
# Every V2 outcome bucket must be a category some domain declares.
|
|
309
|
+
v2_outcome = ROOT / "schemas" / "v2" / "outcome.schema.json"
|
|
310
|
+
if v2_outcome.is_file():
|
|
311
|
+
v2_schema = json.loads(v2_outcome.read_text(encoding="utf-8"))
|
|
312
|
+
buckets = (v2_schema.get("properties", {})
|
|
313
|
+
.get("outcome_type", {}).get("enum"))
|
|
314
|
+
if isinstance(buckets, list) and buckets:
|
|
315
|
+
every = set()
|
|
316
|
+
for domain_id in (d["id"] for d in list_domains_from_registry()):
|
|
317
|
+
every.update(taxonomy_categories(domain_id))
|
|
318
|
+
orphan = sorted(set(buckets) - every)
|
|
319
|
+
# Reverse direction too: a category a domain declares but the V2
|
|
320
|
+
# contract omits would reject that domain's outcomes at the graph
|
|
321
|
+
# layer, which is exactly how policy was blocked.
|
|
322
|
+
missing_bucket = sorted(every - set(buckets))
|
|
323
|
+
if missing_bucket:
|
|
324
|
+
errors.append(
|
|
325
|
+
"schemas/v2/outcome.schema.json is missing category bucket(s) "
|
|
326
|
+
"that domains declare: " + repr(missing_bucket))
|
|
327
|
+
if orphan:
|
|
328
|
+
errors.append(
|
|
329
|
+
"schemas/v2/outcome.schema.json declares bucket(s) no domain "
|
|
330
|
+
"registers: " + repr(orphan))
|
|
331
|
+
|
|
332
|
+
# ---------------------------------------------------------------- versions
|
|
333
|
+
for relative in ("packaging/scp-manifest.json",):
|
|
334
|
+
path = ROOT / relative
|
|
335
|
+
if not path.is_file():
|
|
336
|
+
continue
|
|
337
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
338
|
+
declared = (data.get("skill") or {}).get("version")
|
|
339
|
+
if declared != ENGINE_VERSION:
|
|
340
|
+
errors.append(f"{relative}: skill.version {declared!r} != ENGINE_VERSION {ENGINE_VERSION!r}")
|
|
341
|
+
start_here = ROOT / "packaging" / "START-HERE.md"
|
|
342
|
+
if start_here.is_file():
|
|
343
|
+
text = start_here.read_text(encoding="utf-8")
|
|
344
|
+
major_minor = ".".join(ENGINE_VERSION.split(".")[:2])
|
|
345
|
+
if f"EduEvidence {major_minor}" not in text:
|
|
346
|
+
errors.append(f"packaging/START-HERE.md does not state EduEvidence {major_minor}")
|
|
347
|
+
|
|
348
|
+
# ------------------------------------------------------- docs must not drift
|
|
349
|
+
for doc in ("docs/architecture.md", "README.zh-CN.md", "docs/install-guide.md"):
|
|
350
|
+
path = ROOT / doc
|
|
351
|
+
if not path.is_file():
|
|
352
|
+
continue
|
|
353
|
+
text = path.read_text(encoding="utf-8")
|
|
354
|
+
for stale in ("752 个测试", "752 tests"):
|
|
355
|
+
if stale in text:
|
|
356
|
+
errors.append(f"{doc}: stale test count {stale!r}; use docs/metrics.json")
|
|
357
|
+
|
|
358
|
+
return errors
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def main() -> int:
|
|
362
|
+
print("[*] Checking protocol alignment across workflows, roles, capabilities and packaging...")
|
|
363
|
+
errors = check()
|
|
364
|
+
if errors:
|
|
365
|
+
print(f"[-] Protocol alignment FAILED with {len(errors)} error(s):", file=sys.stderr)
|
|
366
|
+
for error in errors:
|
|
367
|
+
print(f" • {error}", file=sys.stderr)
|
|
368
|
+
return 1
|
|
369
|
+
print("[+] Protocol alignment PASSED: stages, briefs, roles, prompts, capabilities, "
|
|
370
|
+
"sub-skills, workflows and packaging agree.")
|
|
371
|
+
return 0
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
if __name__ == "__main__":
|
|
375
|
+
sys.exit(main())
|