eduevidence 6.0.0 → 6.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (267) hide show
  1. package/CHANGELOG.md +395 -0
  2. package/CONTRIBUTING.md +105 -0
  3. package/README.md +113 -49
  4. package/README.zh-CN.md +39 -12
  5. package/SKILL.md +15 -5
  6. package/assets/readme/landing-tour.gif +0 -0
  7. package/assets/readme/studio-tour.gif +0 -0
  8. package/benchmarks/evidence-library.json +277 -1
  9. package/bin/eduevidence.js +2 -1
  10. package/docs/architecture.md +325 -46
  11. package/docs/demo-workplace-ai.md +1 -1
  12. package/docs/install-guide.md +1 -1
  13. package/docs/j-ev-experimental.md +250 -0
  14. package/docs/orchestration-role-model.md +1 -1
  15. package/docs/release-closeout/README.md +1 -1
  16. package/docs/reproducibility.md +138 -0
  17. package/docs/sciverse-api.md +125 -0
  18. package/domains/_neutral/copy/few_shots.json +21 -0
  19. package/domains/_neutral/copy/framing_lexicon.json +19 -0
  20. package/domains/_neutral/copy/module_labels.json +5 -0
  21. package/domains/_neutral/copy/module_labels_footer.json +102 -0
  22. package/domains/_neutral/copy/module_labels_modules.json +204 -0
  23. package/domains/_neutral/copy/module_labels_nav.json +126 -0
  24. package/domains/_neutral/copy/module_labels_summary.json +98 -0
  25. package/domains/_neutral/copy/module_labels_tables.json +164 -0
  26. package/domains/_neutral/copy/module_labels_v2.json +90 -0
  27. package/domains/_neutral/copy/risk_constructs.json +20 -0
  28. package/domains/_neutral/copy/section_titles.json +66 -0
  29. package/domains/_neutral/copy/terminology.json +11 -0
  30. package/domains/check_copy_packs.py +103 -0
  31. package/domains/education/copy/few_shots.json +22 -0
  32. package/domains/education/copy/framing_enums.json +167 -0
  33. package/domains/education/copy/framing_lexicon.json +166 -0
  34. package/domains/education/copy/module_labels.json +169 -0
  35. package/domains/education/copy/risk_constructs.json +48 -0
  36. package/domains/education/copy/section_titles.json +186 -0
  37. package/domains/education/copy/terminology.json +70 -0
  38. package/domains/education/manifest.json +1 -1
  39. package/domains/education/outcome_taxonomy.json +2 -2
  40. package/domains/manifest.json +1 -1
  41. package/domains/policy/copy/few_shots.json +22 -0
  42. package/domains/policy/copy/framing_enums.json +94 -0
  43. package/domains/policy/copy/framing_lexicon.json +174 -0
  44. package/domains/policy/copy/module_labels.json +168 -0
  45. package/domains/policy/copy/risk_constructs.json +33 -0
  46. package/domains/policy/copy/section_titles.json +186 -0
  47. package/domains/policy/copy/terminology.json +64 -0
  48. package/eduevidence_cli.py +10 -0
  49. package/engine/capabilities.py +57 -5
  50. package/engine/decision_policy.py +167 -0
  51. package/engine/evidence_graph.py +14 -10
  52. package/engine/gaps.py +42 -22
  53. package/engine/ids.py +2 -0
  54. package/engine/library.py +6 -2
  55. package/engine/library_builtin.py +7 -4
  56. package/engine/living.py +34 -4
  57. package/engine/migration.py +88 -3
  58. package/engine/orchestration.py +5 -5
  59. package/engine/paths.py +2 -0
  60. package/engine/pilot.py +34 -32
  61. package/engine/taxonomy.py +211 -0
  62. package/engine/tribunal.py +49 -43
  63. package/engine/versions.py +1 -1
  64. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1361 -147
  65. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
  66. package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
  67. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  68. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  69. package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
  70. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +448 -128
  71. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +448 -128
  72. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +448 -128
  73. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +448 -128
  74. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +448 -128
  75. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
  76. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
  77. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
  78. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
  79. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
  80. package/examples/ai-coding-assistant-evidence/result.json +13 -9
  81. package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
  82. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  83. package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
  84. package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
  85. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  86. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  87. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  88. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  89. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  90. package/examples/spaced-retrieval-practice/frame.json +58 -0
  91. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  92. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  93. package/examples/spaced-retrieval-practice/report.html +2522 -0
  94. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  95. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  96. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  97. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  98. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  99. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  100. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  101. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  102. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  103. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  104. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  105. package/examples/spaced-retrieval-practice/result.json +942 -0
  106. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  107. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  108. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  109. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  110. package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
  111. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  112. package/examples/workplace-ai-assistant/claims.jsonl +4 -4
  113. package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
  114. package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
  115. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  116. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  117. package/examples/workplace-ai-assistant/report_spec.json +209 -40
  118. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +449 -119
  119. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +449 -119
  120. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +449 -119
  121. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +449 -119
  122. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +449 -119
  123. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  124. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  125. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  126. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  127. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  128. package/examples/workplace-ai-assistant/result.json +82 -20
  129. package/examples/workplace-ai-assistant/result.zh.json +82 -20
  130. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  131. package/examples/workplace-ai-assistant/verdict.json +36 -10
  132. package/integrations/agent_mcp.py +2 -2
  133. package/integrations/jev/__init__.py +115 -0
  134. package/integrations/jev/approval.py +212 -0
  135. package/integrations/jev/cli.py +84 -0
  136. package/integrations/jev/config.py +112 -0
  137. package/integrations/jev/gateway.py +128 -0
  138. package/integrations/jev/modes.py +38 -0
  139. package/integrations/jev/tools_classify.py +88 -0
  140. package/integrations/jev/tools_extract.py +111 -0
  141. package/integrations/jev/tools_rerank.py +71 -0
  142. package/integrations/jev/tools_screen.py +87 -0
  143. package/integrations/jev/tools_verify.py +95 -0
  144. package/integrations/jev_mcp.py +22 -0
  145. package/integrations/semantic_decide.py +286 -0
  146. package/integrations/semdecide_cli.py +55 -0
  147. package/package.json +19 -2
  148. package/pyproject.toml +4 -3
  149. package/references/report-copy-style.md +107 -0
  150. package/references/retrieval-compliance.md +75 -0
  151. package/references/retrieval-protocol.md +20 -0
  152. package/retrieval/audit.py +27 -3
  153. package/retrieval/fetch.py +96 -0
  154. package/retrieval/sciverse.py +398 -0
  155. package/retrieval/search.py +47 -7
  156. package/schemas/applicability.schema.json +94 -0
  157. package/schemas/chart-spec.schema.json +10 -3
  158. package/schemas/evidence.schema.json +316 -43
  159. package/schemas/fetch-result.schema.json +2 -1
  160. package/schemas/report-result.schema.json +3 -3
  161. package/schemas/report-spec.schema.json +98 -100
  162. package/schemas/skeptic.schema.json +86 -0
  163. package/schemas/source.schema.json +21 -2
  164. package/schemas/v2/decision-snapshot.schema.json +20 -9
  165. package/schemas/v2/finding.schema.json +5 -1
  166. package/schemas/v2/intake.schema.json +191 -0
  167. package/schemas/v2/methodology-audit.schema.json +5 -1
  168. package/schemas/v2/outcome.schema.json +28 -5
  169. package/schemas/v2/study.schema.json +5 -1
  170. package/schemas/vNext/autoevolve-session.schema.json +34 -1
  171. package/schemas/vNext/eval-snapshot.schema.json +77 -1
  172. package/schemas/vNext/execution-plan.schema.json +50 -1
  173. package/schemas/vNext/gap-priority.schema.json +54 -1
  174. package/schemas/vNext/negative-search-record.schema.json +68 -1
  175. package/schemas/vNext/research-iteration.schema.json +87 -1
  176. package/schemas/vNext/research-strategy.schema.json +62 -1
  177. package/schemas/vNext/skill-experiment.schema.json +90 -1
  178. package/schemas/vNext/task-spec.schema.json +156 -1
  179. package/schemas/vNext/worker-result.schema.json +60 -1
  180. package/schemas/verdict.schema.json +164 -28
  181. package/scripts/build_evidence_library.py +15 -5
  182. package/scripts/build_report_variants.py +18 -2
  183. package/scripts/build_result.py +74 -9
  184. package/scripts/check_package_parity.py +85 -0
  185. package/scripts/check_protocol_alignment.py +375 -0
  186. package/scripts/check_versioned_schemas.py +254 -0
  187. package/scripts/claim_audit.py +13 -8
  188. package/scripts/compute_confidence.py +10 -0
  189. package/scripts/dashboard_server.py +13 -2
  190. package/scripts/did_regression.py +12 -2
  191. package/scripts/evidence_score.py +5 -2
  192. package/scripts/intake/__init__.py +31 -0
  193. package/scripts/intake/__main__.py +18 -0
  194. package/scripts/intake/background.py +78 -0
  195. package/scripts/intake/browser.py +79 -0
  196. package/scripts/intake/cli.py +57 -0
  197. package/scripts/intake/constants.py +57 -0
  198. package/scripts/intake/depth.py +53 -0
  199. package/scripts/intake/enhancements.py +106 -0
  200. package/scripts/intake/hooks.py +90 -0
  201. package/scripts/intake/prefs.py +76 -0
  202. package/scripts/intake/prompts.py +85 -0
  203. package/scripts/intake/session.py +152 -0
  204. package/scripts/lint_file_layers.py +126 -0
  205. package/scripts/orchestrator.py +187 -40
  206. package/scripts/pre_verdict_gate.py +241 -29
  207. package/scripts/quickstart.py +18 -2
  208. package/scripts/run_workspace.py +7 -1
  209. package/scripts/skill_lint.py +11 -1
  210. package/scripts/skill_payload.py +6 -3
  211. package/scripts/test_adversarial_empirical.py +96 -25
  212. package/scripts/validate_schema.py +31 -1
  213. package/skill/agents/evaluation-designer.md +20 -4
  214. package/skill/agents/evidence-analyst.md +19 -3
  215. package/skill/agents/evidence-judge.md +98 -8
  216. package/skill/agents/evidence-retriever.md +20 -3
  217. package/skill/agents/intervention-designer.md +20 -4
  218. package/skill/agents/method-reviewer.md +18 -2
  219. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  220. package/skill/agents/skeptic.md +18 -2
  221. package/skill/roles/registry.yaml +11 -11
  222. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  223. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  224. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  225. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  226. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  227. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  228. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  229. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  230. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  231. package/skill/sub-skills/report-generation/SKILL.md +28 -0
  232. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  233. package/skill/sub-skills/study-design/SKILL.md +30 -9
  234. package/skill/task-briefs/adjudicate.md +32 -7
  235. package/skill/task-briefs/applicability.md +37 -2
  236. package/skill/task-briefs/audit.md +32 -7
  237. package/skill/task-briefs/challenge.md +34 -5
  238. package/skill/task-briefs/evaluate.md +30 -5
  239. package/skill/task-briefs/extract.md +31 -8
  240. package/skill/task-briefs/frame.md +39 -10
  241. package/skill/task-briefs/intervene.md +32 -6
  242. package/skill/task-briefs/present.md +32 -8
  243. package/skill/task-briefs/projection.md +36 -2
  244. package/skill/task-briefs/retrieve.md +36 -6
  245. package/skill/workflows/decision-and-pilot.md +76 -1
  246. package/skill/workflows/evaluate-and-update.md +83 -0
  247. package/skill/workflows/evidence-review.md +104 -0
  248. package/skill/workflows/experimental-jev.md +170 -0
  249. package/skill/workflows/intake.md +120 -0
  250. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  251. package/visualization/eduevidence-report/scripts/build_infographics.py +37 -15
  252. package/visualization/eduevidence-report/scripts/build_report.py +435 -575
  253. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  254. package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
  255. package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
  256. package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
  257. package/visualization/eduevidence-report/scripts/zh_labels.py +141 -1
  258. package/web/architecture.html +14885 -0
  259. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  260. package/web/studio/index.html +2 -2
  261. package/scripts/build_esl_artifacts.py +0 -1921
  262. package/scripts/build_killer_demo.py +0 -295
  263. package/scripts/enrich_projects_human_and_lieflat.py +0 -315
  264. package/scripts/generate_new_projects.py +0 -686
  265. package/scripts/sync_killer_demo_report.py +0 -270
  266. package/web/studio/assets/index-CzXocaGv.css +0 -1
  267. /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
@@ -0,0 +1,85 @@
1
+ #!/usr/bin/env python3
2
+ """check_package_parity.py - prove the shipped package matches the source tree.
3
+
4
+ The upload package is a projection of this repository. If a source file changed
5
+ and the package still carries the old bytes, reviewers receive a different
6
+ product from the one under source control. CI previously compared only
7
+ SKILL.md, so a renamed role file went unnoticed for a whole change set.
8
+
9
+ Compares every file the shared payload allowlist ships: content must match and
10
+ nothing may be missing. Stdlib only; exit 1 on any drift.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import hashlib
15
+ import sys
16
+ from pathlib import Path
17
+
18
+ ROOT = Path(__file__).resolve().parent.parent
19
+ sys.path.insert(0, str(ROOT))
20
+ sys.path.insert(0, str(ROOT / "scripts"))
21
+
22
+ PACKAGE = ROOT / "dist" / "eduevidence-submission"
23
+
24
+
25
+ def digest(path: Path) -> str:
26
+ return hashlib.sha256(path.read_bytes()).hexdigest()
27
+
28
+
29
+ def main() -> int:
30
+ if not PACKAGE.is_dir():
31
+ print(f"ERROR: no package at {PACKAGE}; run bash packaging/make_upload.sh")
32
+ return 1
33
+
34
+ from skill_payload import payload_files
35
+
36
+ expected = set(payload_files(ROOT))
37
+ # The manifest and the packaging notes are written by the build, not copied.
38
+ generated = {"submission-manifest.json", "UPLOAD-README.md", "START-HERE.md",
39
+ "scp-manifest.json", "upload-layout.md", "README.zh-CN.md"}
40
+ expected |= {name for name in generated if (PACKAGE / name).is_file()}
41
+
42
+ missing = sorted(rel for rel in expected if not (PACKAGE / rel).is_file())
43
+
44
+ # Reverse direction: a file the package carries but the source does not is
45
+ # stale output from an earlier build. A one-way check cannot see that,
46
+ # which is how pre-rename report copies once survived inside a package.
47
+ unexpected = []
48
+ for path in sorted(PACKAGE.rglob("*")):
49
+ if not path.is_file():
50
+ continue
51
+ rel = path.relative_to(PACKAGE).as_posix()
52
+ if rel in expected or rel == "submission-manifest.json":
53
+ continue
54
+ if any(part in {"__pycache__", ".git"} for part in path.parts):
55
+ continue
56
+ if path.suffix in {".pyc", ".pyo"} or path.name == ".DS_Store":
57
+ continue
58
+ if not (ROOT / rel).exists():
59
+ unexpected.append(rel)
60
+ differing = []
61
+ for rel in sorted(expected):
62
+ source = ROOT / rel
63
+ shipped = PACKAGE / rel
64
+ if not source.is_file() or not shipped.is_file():
65
+ continue
66
+ if digest(source) != digest(shipped):
67
+ differing.append(rel)
68
+
69
+ if missing or differing or unexpected:
70
+ print("ERROR: package does not match the source tree", file=sys.stderr)
71
+ for rel in missing[:20]:
72
+ print(f" missing from package: {rel}", file=sys.stderr)
73
+ for rel in differing[:20]:
74
+ print(f" differs from source: {rel}", file=sys.stderr)
75
+ for rel in unexpected[:20]:
76
+ print(f" stale in package: {rel}", file=sys.stderr)
77
+ print(" fix: bash packaging/make_upload.sh", file=sys.stderr)
78
+ return 1
79
+
80
+ print(f"package parity OK ({len(expected)} files byte-identical to source)")
81
+ return 0
82
+
83
+
84
+ if __name__ == "__main__":
85
+ sys.exit(main())
@@ -0,0 +1,375 @@
1
+ #!/usr/bin/env python3
2
+ """check_protocol_alignment.py — Protocol five-way alignment gate.
3
+
4
+ Every scientific contract in this repository is declared in more than one
5
+ place: the workflow registry (engine/workflows.py), the capability registry
6
+ (engine/capabilities.py), the role registry (skill/roles/registry.yaml), the
7
+ role prompts (skill/agents/*.md), the stage briefs (skill/task-briefs/*.md),
8
+ the sub-skill recipes (skill/sub-skills/*/SKILL.md), the routing requirements
9
+ (integrations/agent_mcp.py) and the packaging manifest (packaging/*).
10
+
11
+ Drift between them is silent: a role can lose its brief, a capability can
12
+ exist with no recipe, a version can be bumped in one file only. This gate
13
+ makes that drift fail loudly. Stdlib only; a non-zero exit blocks CI.
14
+
15
+ Usage:
16
+ python3 scripts/check_protocol_alignment.py
17
+ """
18
+ from __future__ import annotations
19
+
20
+ import json
21
+ import re
22
+ import sys
23
+ from pathlib import Path
24
+
25
+ ROOT = Path(__file__).resolve().parent.parent
26
+ sys.path.insert(0, str(ROOT))
27
+
28
+ from engine.capabilities import capability_registry # noqa: E402
29
+ from engine.versions import ENGINE_VERSION # noqa: E402
30
+ from engine.workflows import execution_stages, workflow_registry # noqa: E402
31
+
32
+ FRONTMATTER_RE = re.compile(r"^---\s*\n(.*?)\n---\s*\n", re.DOTALL)
33
+
34
+ #: Projection-layer capabilities sit outside the scientific stage model
35
+ #: (engine/workflows.py: Projection is not a scientific stage), so they are
36
+ #: owned by the projection brief rather than by a scientific role.
37
+ PROJECTION_CAPABILITIES = {"report_projection", "report_rendering"}
38
+
39
+
40
+ def list_domains_from_registry() -> list[dict]:
41
+ """Registered domains (evidencecore is the registry owner)."""
42
+ from engine.evidencecore import list_domains
43
+
44
+ return list_domains()
45
+
46
+
47
+ def _frontmatter(path: Path) -> dict[str, str]:
48
+ """Parse the flat key: value frontmatter used by skills and role prompts."""
49
+ match = FRONTMATTER_RE.match(path.read_text(encoding="utf-8"))
50
+ if not match:
51
+ return {}
52
+ fields: dict[str, str] = {}
53
+ for line in match.group(1).splitlines():
54
+ if line.strip().startswith("#") or ":" not in line:
55
+ continue
56
+ key, _, value = line.partition(":")
57
+ fields[key.strip()] = value.split("#", 1)[0].strip()
58
+ return fields
59
+
60
+
61
+ def _registry_roles() -> dict[str, dict]:
62
+ """The role registry is a small fixed-shape YAML file; parse it narrowly."""
63
+ text = (ROOT / "skill" / "roles" / "registry.yaml").read_text(encoding="utf-8")
64
+ roles: dict[str, dict] = {}
65
+ current: str | None = None
66
+ for raw in text.splitlines():
67
+ if not raw.strip() or raw.strip().startswith("#"):
68
+ continue
69
+ if raw.startswith("roles:"):
70
+ continue
71
+ if raw.startswith("execution:"):
72
+ break
73
+ if re.match(r"^ [A-Za-z0-9_-]+:\s*$", raw):
74
+ current = raw.strip().rstrip(":")
75
+ roles[current] = {}
76
+ continue
77
+ if current and raw.strip().startswith("stages:"):
78
+ stages = raw.split(":", 1)[1].strip().strip("[]")
79
+ roles[current]["stages"] = [s.strip() for s in stages.split(",") if s.strip()]
80
+ elif current and raw.strip().startswith("capabilities:"):
81
+ caps = raw.split(":", 1)[1].strip().strip("[]")
82
+ roles[current]["capabilities"] = [c.strip() for c in caps.split(",") if c.strip()]
83
+ elif current and ":" in raw.strip():
84
+ key, _, value = raw.strip().partition(":")
85
+ roles[current][key.strip()] = value.strip()
86
+ return roles
87
+
88
+
89
+ def _merge_role(roles: dict[str, dict], name: str, stage: str, capability: str,
90
+ critical: bool | None = None, independence: bool | None = None) -> None:
91
+ entry = roles.setdefault(name, {"stages": [], "capabilities": []})
92
+ if stage not in entry["stages"]:
93
+ entry["stages"].append(stage)
94
+ for cap in capability.split("+"):
95
+ cap = cap.strip()
96
+ if cap and cap not in entry["capabilities"]:
97
+ entry["capabilities"].append(cap)
98
+ if critical is not None:
99
+ entry["critical_path"] = "true" if critical else "false"
100
+ if independence:
101
+ entry["independence_required"] = "true"
102
+
103
+
104
+ def check() -> list[str]:
105
+ errors: list[str] = []
106
+
107
+ stages = list(execution_stages())
108
+ scientific_stages = [s for s in stages if s != "projection"]
109
+
110
+ # ---------------------------------------------------------------- briefs
111
+ brief_dir = ROOT / "skill" / "task-briefs"
112
+ briefs = {p.stem for p in brief_dir.glob("*.md")}
113
+ for stage in stages:
114
+ if stage not in briefs:
115
+ errors.append(f"stage {stage!r} has no task brief in skill/task-briefs/")
116
+ for extra in sorted(briefs - set(stages) - {"present"}):
117
+ errors.append(f"task brief {extra!r} does not map to a canonical stage")
118
+
119
+ # ---------------------------------------------------------------- roles
120
+ registry = _registry_roles()
121
+ if not registry:
122
+ errors.append("skill/roles/registry.yaml declares no roles")
123
+ agent_files = {p.stem: p for p in (ROOT / "skill" / "agents").glob("*.md")}
124
+ if set(registry) != set(agent_files):
125
+ errors.append("role registry and skill/agents/*.md disagree: "
126
+ f"registry-only={sorted(set(registry) - set(agent_files))} "
127
+ f"agent-only={sorted(set(agent_files) - set(registry))}")
128
+
129
+ stage_owner: dict[str, str] = {}
130
+ for role, entry in registry.items():
131
+ for stage in entry.get("stages", []):
132
+ if stage in stage_owner:
133
+ errors.append(f"stage {stage!r} is owned by both {stage_owner[stage]!r} and {role!r}")
134
+ stage_owner[stage] = role
135
+ for stage in scientific_stages:
136
+ if stage not in stage_owner:
137
+ errors.append(f"stage {stage!r} has no owning role in the registry")
138
+
139
+ # Registry capabilities must be engine capability IDs: a free-text label
140
+ # here silently detaches the role from the capability it claims to run.
141
+ for role, entry in registry.items():
142
+ for cap in entry.get("capabilities", []):
143
+ if cap not in capability_registry():
144
+ errors.append(f"registry role {role!r} declares capability {cap!r}, "
145
+ "which is not in engine/capabilities.py")
146
+
147
+ # Independence is graded: the skeptic must come from a different model
148
+ # family, the method reviewer must be separated from the content judgement.
149
+ independence = {role: entry.get("independence_required")
150
+ for role, entry in registry.items() if entry.get("independence_required")}
151
+ if independence != {"skeptic": "different-model-family",
152
+ "method-reviewer": "role-separation"}:
153
+ expected = {"skeptic": "different-model-family", "method-reviewer": "role-separation"}
154
+ errors.append("independence_required must be "
155
+ f"{expected}, found {independence}")
156
+
157
+ # ------------------------------------------------- role prompt frontmatter
158
+ for role, path in agent_files.items():
159
+ fields = _frontmatter(path)
160
+ if fields.get("name") != role:
161
+ errors.append(f"{path.name}: frontmatter name {fields.get('name')!r} != filename {role!r}")
162
+ if fields.get("role_id") != role:
163
+ errors.append(f"{path.name}: missing role_id: {role}")
164
+ if not fields.get("capabilities"):
165
+ errors.append(f"{path.name}: missing capabilities declaration")
166
+ if not fields.get("output_contracts"):
167
+ errors.append(f"{path.name}: missing output_contracts declaration")
168
+ for banned in ("default_cli", "default_model"):
169
+ if banned in fields:
170
+ errors.append(f"{path.name}: {banned} must not be bound in the role prompt "
171
+ "(model/CLI choice is a user-confirmed routing decision)")
172
+ for token in ("claude-", "gpt-", "deepseek-", "glm-", "kimi-"):
173
+ if token in fields.get("recommended_reasoning", ""):
174
+ errors.append(f"{path.name}: recommended_reasoning must not contain a model name")
175
+ if role in registry:
176
+ declared = set(registry[role].get("capabilities", []))
177
+ prompt_caps = {c.strip() for c in fields.get("capabilities", "").split(",") if c.strip()}
178
+ unknown = {c for c in prompt_caps if c not in capability_registry()}
179
+ unmapped = {c for c in unknown if not c.startswith("(")}
180
+ if unmapped:
181
+ errors.append(f"{path.name}: capabilities not in the capability registry: {sorted(unmapped)}")
182
+ missing = {c for c in declared if c in capability_registry()} - prompt_caps
183
+ if missing:
184
+ errors.append(f"{path.name}: registry capabilities absent from the prompt: "
185
+ f"{sorted(missing)}")
186
+ if registry.get(role, {}).get("critical_path") == "true":
187
+ if fields.get("critical_path") != "true":
188
+ errors.append(f"{path.name}: registry marks this role critical_path but the "
189
+ "prompt does not declare critical_path: true")
190
+
191
+ # ----------------------------------------------- routing-side requirements
192
+ from integrations.agent_mcp import ROLE_REQUIREMENTS # noqa: E402
193
+ if set(ROLE_REQUIREMENTS) != set(registry):
194
+ errors.append("integrations.agent_mcp.ROLE_REQUIREMENTS and the role registry disagree: "
195
+ f"routing-only={sorted(set(ROLE_REQUIREMENTS) - set(registry))} "
196
+ f"registry-only={sorted(set(registry) - set(ROLE_REQUIREMENTS))}")
197
+ for role, reqs in ROLE_REQUIREMENTS.items():
198
+ for banned in ("default_cli", "default_model", "model", "cli"):
199
+ if banned in reqs:
200
+ errors.append(f"ROLE_REQUIREMENTS[{role!r}] must not bind {banned!r}")
201
+ wants_family = registry.get(role, {}).get("independence_required") == "different-model-family"
202
+ if wants_family and reqs.get("independence") != "different-model-family":
203
+ errors.append(f"ROLE_REQUIREMENTS[{role!r}] must require a different model family")
204
+
205
+ # ------------------------------------------------------------ capabilities
206
+ capabilities = capability_registry()
207
+ sub_skills = sorted(p for p in (ROOT / "skill" / "sub-skills").iterdir()
208
+ if p.is_dir() and not p.name.startswith("."))
209
+ if len(sub_skills) < 5:
210
+ errors.append(f"expected at least 5 sub-skills, found {len(sub_skills)}")
211
+ mapped: set[str] = set()
212
+ for skill_dir in sub_skills:
213
+ path = skill_dir / "SKILL.md"
214
+ if not path.is_file():
215
+ errors.append(f"sub-skill {skill_dir.name} has no SKILL.md")
216
+ continue
217
+ fields = _frontmatter(path)
218
+ if fields.get("name") != skill_dir.name:
219
+ errors.append(f"{skill_dir.name}/SKILL.md: name {fields.get('name')!r} != directory")
220
+ declared = fields.get("capability", "")
221
+ if not declared:
222
+ errors.append(f"{skill_dir.name}/SKILL.md: missing capability declaration")
223
+ continue
224
+ for cap in declared.split("+"):
225
+ cap = cap.strip().split()[0] if cap.strip() else ""
226
+ if cap and not cap.startswith("("):
227
+ mapped.add(cap)
228
+ if cap not in capabilities:
229
+ errors.append(f"{skill_dir.name}/SKILL.md: capability {cap!r} is not in "
230
+ "engine/capabilities.py")
231
+ # Every registered capability must be owned by at least one role, and every
232
+ # recipe capability must be one the engine actually registers.
233
+ owned: set[str] = set()
234
+ for entry in registry.values():
235
+ owned.update(entry.get("capabilities", []))
236
+ for cap in capabilities:
237
+ if cap not in owned and cap not in PROJECTION_CAPABILITIES:
238
+ errors.append(f"capability {cap!r} is registered in engine/capabilities.py "
239
+ "but no role owns it")
240
+ for cap in sorted(mapped):
241
+ if cap in capabilities and cap not in owned and cap not in PROJECTION_CAPABILITIES:
242
+ errors.append(f"sub-skill capability {cap!r} is owned by no role")
243
+
244
+ # ---------------------------------------------------------------- workflows
245
+ workflow_dir = ROOT / "skill" / "workflows"
246
+ workflow_files = {p.stem for p in workflow_dir.glob("*.md")}
247
+ registry_workflows = workflow_registry()
248
+ public = {w for w in registry_workflows if w != "full_research_cycle"}
249
+ normalised = {name.replace("_", "-") for name in public}
250
+ if not normalised <= workflow_files:
251
+ errors.append("workflows missing a runbook: "
252
+ f"{sorted(normalised - workflow_files)}")
253
+ skill_text = (ROOT / "SKILL.md").read_text(encoding="utf-8")
254
+ for name in public:
255
+ reference = f"skill/workflows/{name.replace('_', '-')}.md"
256
+ if reference not in skill_text:
257
+ errors.append(f"SKILL.md does not route to {reference}")
258
+
259
+ skill_md = root_skill_text = skill_text # alias for readability
260
+ for stage in scientific_stages:
261
+ if f"| {stage.capitalize()} " not in skill_md and stage not in skill_md:
262
+ errors.append(f"SKILL.md does not mention stage {stage!r}")
263
+
264
+ # ------------------------------------------------------------- taxonomy
265
+ # The registry is the authority for outcome tokens and their categories;
266
+ # the JSON Schemas carry static enums because JSON Schema cannot read a
267
+ # file at validation time. This dimension is what keeps the static enums
268
+ # honest: drift between a schema enum and the registry fails the gate.
269
+ from engine.taxonomy import (
270
+ all_tokens_ordered,
271
+ categories as taxonomy_categories,
272
+ )
273
+
274
+ registered_tokens = set(all_tokens_ordered())
275
+ evidence_path = ROOT / "schemas" / "evidence.schema.json"
276
+ if evidence_path.is_file():
277
+ evidence_schema = json.loads(evidence_path.read_text(encoding="utf-8"))
278
+ enum = (evidence_schema.get("properties", {})
279
+ .get("outcome_type", {}).get("enum"))
280
+ if not isinstance(enum, list) or not enum:
281
+ errors.append("schemas/evidence.schema.json declares no outcome_type enum")
282
+ else:
283
+ missing = sorted(registered_tokens - set(enum))
284
+ extra = sorted(set(enum) - registered_tokens)
285
+ if missing:
286
+ errors.append(
287
+ "outcome_type enum is missing registered token(s): " + repr(missing))
288
+ if extra:
289
+ errors.append(
290
+ "outcome_type enum declares unregistered token(s): " + repr(extra))
291
+
292
+ # Every domain's ADOPT-gate categories must be categories it declares.
293
+ from engine.tribunal import primary_effect_categories
294
+
295
+ for domain_id in (d["id"] for d in list_domains_from_registry()):
296
+ declared = set(taxonomy_categories(domain_id))
297
+ try:
298
+ primary = primary_effect_categories(domain_id)
299
+ except ValueError as exc:
300
+ errors.append(f"domain {domain_id!r}: ADOPT gate misconfigured: {exc}")
301
+ continue
302
+ for category in primary:
303
+ if category not in declared:
304
+ errors.append(
305
+ f"domain {domain_id!r}: ADOPT gate names undeclared category "
306
+ + repr(category))
307
+
308
+ # Every V2 outcome bucket must be a category some domain declares.
309
+ v2_outcome = ROOT / "schemas" / "v2" / "outcome.schema.json"
310
+ if v2_outcome.is_file():
311
+ v2_schema = json.loads(v2_outcome.read_text(encoding="utf-8"))
312
+ buckets = (v2_schema.get("properties", {})
313
+ .get("outcome_type", {}).get("enum"))
314
+ if isinstance(buckets, list) and buckets:
315
+ every = set()
316
+ for domain_id in (d["id"] for d in list_domains_from_registry()):
317
+ every.update(taxonomy_categories(domain_id))
318
+ orphan = sorted(set(buckets) - every)
319
+ # Reverse direction too: a category a domain declares but the V2
320
+ # contract omits would reject that domain's outcomes at the graph
321
+ # layer, which is exactly how policy was blocked.
322
+ missing_bucket = sorted(every - set(buckets))
323
+ if missing_bucket:
324
+ errors.append(
325
+ "schemas/v2/outcome.schema.json is missing category bucket(s) "
326
+ "that domains declare: " + repr(missing_bucket))
327
+ if orphan:
328
+ errors.append(
329
+ "schemas/v2/outcome.schema.json declares bucket(s) no domain "
330
+ "registers: " + repr(orphan))
331
+
332
+ # ---------------------------------------------------------------- versions
333
+ for relative in ("packaging/scp-manifest.json",):
334
+ path = ROOT / relative
335
+ if not path.is_file():
336
+ continue
337
+ data = json.loads(path.read_text(encoding="utf-8"))
338
+ declared = (data.get("skill") or {}).get("version")
339
+ if declared != ENGINE_VERSION:
340
+ errors.append(f"{relative}: skill.version {declared!r} != ENGINE_VERSION {ENGINE_VERSION!r}")
341
+ start_here = ROOT / "packaging" / "START-HERE.md"
342
+ if start_here.is_file():
343
+ text = start_here.read_text(encoding="utf-8")
344
+ major_minor = ".".join(ENGINE_VERSION.split(".")[:2])
345
+ if f"EduEvidence {major_minor}" not in text:
346
+ errors.append(f"packaging/START-HERE.md does not state EduEvidence {major_minor}")
347
+
348
+ # ------------------------------------------------------- docs must not drift
349
+ for doc in ("docs/architecture.md", "README.zh-CN.md", "docs/install-guide.md"):
350
+ path = ROOT / doc
351
+ if not path.is_file():
352
+ continue
353
+ text = path.read_text(encoding="utf-8")
354
+ for stale in ("752 个测试", "752 tests"):
355
+ if stale in text:
356
+ errors.append(f"{doc}: stale test count {stale!r}; use docs/metrics.json")
357
+
358
+ return errors
359
+
360
+
361
+ def main() -> int:
362
+ print("[*] Checking protocol alignment across workflows, roles, capabilities and packaging...")
363
+ errors = check()
364
+ if errors:
365
+ print(f"[-] Protocol alignment FAILED with {len(errors)} error(s):", file=sys.stderr)
366
+ for error in errors:
367
+ print(f" • {error}", file=sys.stderr)
368
+ return 1
369
+ print("[+] Protocol alignment PASSED: stages, briefs, roles, prompts, capabilities, "
370
+ "sub-skills, workflows and packaging agree.")
371
+ return 0
372
+
373
+
374
+ if __name__ == "__main__":
375
+ sys.exit(main())