eduevidence 6.0.0 → 6.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (189) hide show
  1. package/CONTRIBUTING.md +105 -0
  2. package/README.md +93 -38
  3. package/README.zh-CN.md +26 -6
  4. package/SKILL.md +11 -2
  5. package/assets/readme/landing-tour.gif +0 -0
  6. package/assets/readme/studio-tour.gif +0 -0
  7. package/bin/eduevidence.js +2 -1
  8. package/docs/architecture.md +319 -43
  9. package/docs/demo-workplace-ai.md +1 -1
  10. package/docs/install-guide.md +1 -1
  11. package/docs/orchestration-role-model.md +1 -1
  12. package/docs/release-closeout/README.md +1 -1
  13. package/docs/sciverse-api.md +125 -0
  14. package/eduevidence_cli.py +10 -0
  15. package/engine/decision_policy.py +96 -0
  16. package/engine/evidence_graph.py +14 -10
  17. package/engine/gaps.py +42 -22
  18. package/engine/ids.py +2 -0
  19. package/engine/library.py +6 -2
  20. package/engine/living.py +34 -4
  21. package/engine/migration.py +88 -3
  22. package/engine/orchestration.py +5 -5
  23. package/engine/paths.py +2 -0
  24. package/engine/pilot.py +34 -32
  25. package/engine/taxonomy.py +211 -0
  26. package/engine/tribunal.py +43 -31
  27. package/engine/versions.py +1 -1
  28. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1360 -146
  29. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
  30. package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
  31. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  32. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  33. package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
  34. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +447 -127
  35. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +447 -127
  36. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +447 -127
  37. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +447 -127
  38. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +447 -127
  39. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
  40. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
  41. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
  42. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
  43. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
  44. package/examples/ai-coding-assistant-evidence/result.json +13 -9
  45. package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
  46. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  47. package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
  48. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  49. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  50. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  51. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  52. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  53. package/examples/spaced-retrieval-practice/frame.json +58 -0
  54. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  55. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  56. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  57. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  58. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  59. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  60. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  61. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  62. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  63. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  64. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  65. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  66. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  67. package/examples/spaced-retrieval-practice/result.json +942 -0
  68. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  69. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  70. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  71. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  72. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  73. package/examples/workplace-ai-assistant/claims.jsonl +4 -4
  74. package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
  75. package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
  76. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  77. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  78. package/examples/workplace-ai-assistant/report_spec.json +209 -40
  79. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +435 -105
  80. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +435 -105
  81. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +435 -105
  82. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +435 -105
  83. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +435 -105
  84. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  85. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  86. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  87. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  88. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  89. package/examples/workplace-ai-assistant/result.json +82 -20
  90. package/examples/workplace-ai-assistant/result.zh.json +82 -20
  91. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  92. package/examples/workplace-ai-assistant/verdict.json +36 -10
  93. package/integrations/agent_mcp.py +2 -2
  94. package/package.json +12 -3
  95. package/pyproject.toml +4 -3
  96. package/references/report-copy-style.md +67 -0
  97. package/references/retrieval-compliance.md +75 -0
  98. package/references/retrieval-protocol.md +20 -0
  99. package/retrieval/audit.py +27 -3
  100. package/retrieval/fetch.py +96 -0
  101. package/retrieval/sciverse.py +398 -0
  102. package/retrieval/search.py +47 -7
  103. package/schemas/applicability.schema.json +94 -0
  104. package/schemas/chart-spec.schema.json +10 -3
  105. package/schemas/evidence.schema.json +316 -43
  106. package/schemas/fetch-result.schema.json +2 -1
  107. package/schemas/report-result.schema.json +3 -3
  108. package/schemas/report-spec.schema.json +98 -100
  109. package/schemas/skeptic.schema.json +86 -0
  110. package/schemas/source.schema.json +21 -2
  111. package/schemas/v2/finding.schema.json +5 -1
  112. package/schemas/v2/methodology-audit.schema.json +5 -1
  113. package/schemas/v2/outcome.schema.json +28 -5
  114. package/schemas/v2/study.schema.json +5 -1
  115. package/schemas/vNext/autoevolve-session.schema.json +34 -1
  116. package/schemas/vNext/eval-snapshot.schema.json +77 -1
  117. package/schemas/vNext/execution-plan.schema.json +50 -1
  118. package/schemas/vNext/gap-priority.schema.json +54 -1
  119. package/schemas/vNext/negative-search-record.schema.json +68 -1
  120. package/schemas/vNext/research-iteration.schema.json +87 -1
  121. package/schemas/vNext/research-strategy.schema.json +62 -1
  122. package/schemas/vNext/skill-experiment.schema.json +90 -1
  123. package/schemas/vNext/task-spec.schema.json +156 -1
  124. package/schemas/vNext/worker-result.schema.json +60 -1
  125. package/schemas/verdict.schema.json +164 -28
  126. package/scripts/build_esl_artifacts.py +2 -2
  127. package/scripts/build_report_variants.py +18 -2
  128. package/scripts/build_result.py +74 -9
  129. package/scripts/check_package_parity.py +85 -0
  130. package/scripts/check_protocol_alignment.py +375 -0
  131. package/scripts/check_versioned_schemas.py +254 -0
  132. package/scripts/claim_audit.py +13 -8
  133. package/scripts/compute_confidence.py +10 -0
  134. package/scripts/did_regression.py +12 -2
  135. package/scripts/evidence_score.py +5 -2
  136. package/scripts/generate_new_projects.py +4 -4
  137. package/scripts/orchestrator.py +120 -24
  138. package/scripts/pre_verdict_gate.py +224 -26
  139. package/scripts/quickstart.py +18 -2
  140. package/scripts/run_workspace.py +7 -1
  141. package/scripts/skill_payload.py +4 -1
  142. package/scripts/test_adversarial_empirical.py +26 -19
  143. package/scripts/validate_schema.py +31 -1
  144. package/skill/agents/evaluation-designer.md +20 -4
  145. package/skill/agents/evidence-analyst.md +19 -3
  146. package/skill/agents/evidence-judge.md +50 -2
  147. package/skill/agents/evidence-retriever.md +20 -3
  148. package/skill/agents/intervention-designer.md +20 -4
  149. package/skill/agents/method-reviewer.md +18 -2
  150. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  151. package/skill/agents/skeptic.md +18 -2
  152. package/skill/roles/registry.yaml +11 -11
  153. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  154. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  155. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  156. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  157. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  158. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  159. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  160. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  161. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  162. package/skill/sub-skills/report-generation/SKILL.md +28 -0
  163. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  164. package/skill/sub-skills/study-design/SKILL.md +30 -9
  165. package/skill/task-briefs/adjudicate.md +32 -7
  166. package/skill/task-briefs/applicability.md +37 -2
  167. package/skill/task-briefs/audit.md +32 -7
  168. package/skill/task-briefs/challenge.md +34 -5
  169. package/skill/task-briefs/evaluate.md +30 -5
  170. package/skill/task-briefs/extract.md +31 -8
  171. package/skill/task-briefs/frame.md +39 -10
  172. package/skill/task-briefs/intervene.md +32 -6
  173. package/skill/task-briefs/present.md +32 -8
  174. package/skill/task-briefs/projection.md +36 -2
  175. package/skill/task-briefs/retrieve.md +36 -6
  176. package/skill/workflows/decision-and-pilot.md +76 -1
  177. package/skill/workflows/evaluate-and-update.md +83 -0
  178. package/skill/workflows/evidence-review.md +104 -0
  179. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  180. package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
  181. package/visualization/eduevidence-report/scripts/build_report.py +512 -65
  182. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  183. package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
  184. package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
  185. package/web/architecture.html +14885 -0
  186. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  187. package/web/studio/index.html +2 -2
  188. package/web/studio/assets/index-CzXocaGv.css +0 -1
  189. /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
@@ -30,14 +30,14 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
30
30
  from evidence_semantics import effect_direction
31
31
  from engine.versions import ENGINE_VERSION
32
32
 
33
- OUTCOME_ORDER = [
34
- "knowledge_gain", "concept_understanding", "retention", "transfer",
35
- "independent_problem_solving", "completion_time", "accuracy",
36
- "code_quality", "assignment_score", "engagement", "motivation",
37
- "cognitive_load", "help_seeking", "metacognition", "ai_dependency",
38
- "over_reliance", "reduced_effort", "reduced_transfer",
39
- "academic_integrity_risk", "false_confidence",
40
- ]
33
+ def _outcome_order() -> list[str]:
34
+ """Registered outcome tokens in registry order (was a hard-coded list)."""
35
+ from engine.taxonomy import all_tokens_ordered
36
+
37
+ return list(all_tokens_ordered())
38
+
39
+
40
+ OUTCOME_ORDER = _outcome_order()
41
41
 
42
42
 
43
43
  def _load_json(path: Path) -> dict[str, Any] | None:
@@ -221,12 +221,74 @@ def build_claims(evidence: list[dict[str, Any]]) -> list[dict[str, Any]]:
221
221
  return list(claims.values())
222
222
 
223
223
 
224
+ def _applicability(pack_dir: Path, verdict: dict) -> dict:
225
+ """Stage-7 applicability assessment, falling back to the verdict boundary.
226
+
227
+ applicability.json is the dedicated deliverable of the Applicability stage.
228
+ It used to be written and then ignored, because the renderer only looked at
229
+ the verdict; this is where it re-enters the result.
230
+ """
231
+ import json as _json
232
+
233
+ path = pack_dir / "applicability.json"
234
+ if path.is_file():
235
+ try:
236
+ data = _json.loads(path.read_text(encoding="utf-8"))
237
+ except (OSError, _json.JSONDecodeError):
238
+ data = None
239
+ if isinstance(data, dict) and data and data.get("status") != "NOT_CAPTURED":
240
+ return data
241
+ value = verdict.get("applicability") if isinstance(verdict, dict) else None
242
+ return value if isinstance(value, dict) else {}
243
+
244
+ def _derive_study_audits(methodology: list[dict], evidence: list[dict]) -> list[dict]:
245
+ """Per-study audit rows derived from the audits and evidence present.
246
+
247
+ Each row names the study and reports the audit verdict that covers it,
248
+ so the per-study axis the schema advertises actually exists downstream.
249
+ Rows are only emitted for studies the audits or evidence actually name.
250
+ """
251
+ by_study: dict[str, dict] = {}
252
+ for audit in methodology:
253
+ if not isinstance(audit, dict):
254
+ continue
255
+ target = audit.get("target") or "overall"
256
+ if target == "overall":
257
+ # The aggregate audit is not a study row; label it as the
258
+ # body-of-evidence review so it cannot be mistaken for one.
259
+ target = "body_of_evidence"
260
+ entry = by_study.setdefault(target, {
261
+ "study_id": target,
262
+ "verdict": audit.get("verdict"),
263
+ "audit_items": audit.get("audit_items") or {},
264
+ "limitations": list(audit.get("limitations") or []),
265
+ "task_vs_learning_guard": audit.get("task_vs_learning_guard"),
266
+ })
267
+ entry.setdefault("evidence_ids", [])
268
+ known = {e.get("study_id") for e in evidence if e.get("study_id")}
269
+ for study_id in sorted(known):
270
+ by_study.setdefault(study_id, {
271
+ "study_id": study_id,
272
+ "verdict": None,
273
+ "audit_items": {},
274
+ "limitations": [],
275
+ "task_vs_learning_guard": None,
276
+ "evidence_ids": [e.get("evidence_id") for e in evidence
277
+ if e.get("study_id") == study_id],
278
+ })
279
+ return list(by_study.values())
280
+
281
+
224
282
  def build_result(pack_dir: Path, *, mode: str = "platform_native") -> dict[str, Any]:
225
283
  frame = _load_json(pack_dir / "frame.json") or {}
226
284
  evidence = _load_jsonl(pack_dir / "evidence.jsonl")
227
285
  # methodology.json is a single MethodologyAudit object (or a JSONL list)
228
286
  methodology_single = _load_json(pack_dir / "methodology.json")
229
287
  methodology = [methodology_single] if methodology_single else _load_jsonl(pack_dir / "methodology.jsonl")
288
+ # Per-study audits: the contract advertised them but nothing produced
289
+ # them, so a multi-study review silently shipped a single audit object.
290
+ # Derive the per-study axis from the audits actually present.
291
+ study_audits = _derive_study_audits(methodology, evidence)
230
292
  verdict = _load_json(pack_dir / "verdict.json") or {}
231
293
  intervention = _load_json(pack_dir / "intervention.json") or {}
232
294
  evaluation = _load_json(pack_dir / "evaluation.json") or {}
@@ -268,9 +330,12 @@ def build_result(pack_dir: Path, *, mode: str = "platform_native") -> dict[str,
268
330
  "sources": sources,
269
331
  "evidence": evidence,
270
332
  "methodology_reviews": methodology,
333
+ "study_audits": study_audits,
271
334
  "conflicts": [{"reason_for_disagreement": verdict.get("reason_for_disagreement", "")}]
272
335
  if verdict.get("reason_for_disagreement") else [],
273
- "applicability": verdict.get("applicability", {}),
336
+ # Prefer the dedicated stage-7 assessment; fall back to the verdict-embedded
337
+ # boundary when the run carries no separate applicability.json.
338
+ "applicability": _applicability(pack_dir, verdict),
274
339
  "intervention": intervention,
275
340
  "evaluation": evaluation,
276
341
  "benchmark": {},
@@ -0,0 +1,85 @@
1
+ #!/usr/bin/env python3
2
+ """check_package_parity.py - prove the shipped package matches the source tree.
3
+
4
+ The upload package is a projection of this repository. If a source file changed
5
+ and the package still carries the old bytes, reviewers receive a different
6
+ product from the one under source control. CI previously compared only
7
+ SKILL.md, so a renamed role file went unnoticed for a whole change set.
8
+
9
+ Compares every file the shared payload allowlist ships: content must match and
10
+ nothing may be missing. Stdlib only; exit 1 on any drift.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import hashlib
15
+ import sys
16
+ from pathlib import Path
17
+
18
+ ROOT = Path(__file__).resolve().parent.parent
19
+ sys.path.insert(0, str(ROOT))
20
+ sys.path.insert(0, str(ROOT / "scripts"))
21
+
22
+ PACKAGE = ROOT / "dist" / "eduevidence-submission"
23
+
24
+
25
+ def digest(path: Path) -> str:
26
+ return hashlib.sha256(path.read_bytes()).hexdigest()
27
+
28
+
29
+ def main() -> int:
30
+ if not PACKAGE.is_dir():
31
+ print(f"ERROR: no package at {PACKAGE}; run bash packaging/make_upload.sh")
32
+ return 1
33
+
34
+ from skill_payload import payload_files
35
+
36
+ expected = set(payload_files(ROOT))
37
+ # The manifest and the packaging notes are written by the build, not copied.
38
+ generated = {"submission-manifest.json", "UPLOAD-README.md", "START-HERE.md",
39
+ "scp-manifest.json", "upload-layout.md", "README.zh-CN.md"}
40
+ expected |= {name for name in generated if (PACKAGE / name).is_file()}
41
+
42
+ missing = sorted(rel for rel in expected if not (PACKAGE / rel).is_file())
43
+
44
+ # Reverse direction: a file the package carries but the source does not is
45
+ # stale output from an earlier build. A one-way check cannot see that,
46
+ # which is how pre-rename report copies once survived inside a package.
47
+ unexpected = []
48
+ for path in sorted(PACKAGE.rglob("*")):
49
+ if not path.is_file():
50
+ continue
51
+ rel = path.relative_to(PACKAGE).as_posix()
52
+ if rel in expected or rel == "submission-manifest.json":
53
+ continue
54
+ if any(part in {"__pycache__", ".git"} for part in path.parts):
55
+ continue
56
+ if path.suffix in {".pyc", ".pyo"} or path.name == ".DS_Store":
57
+ continue
58
+ if not (ROOT / rel).exists():
59
+ unexpected.append(rel)
60
+ differing = []
61
+ for rel in sorted(expected):
62
+ source = ROOT / rel
63
+ shipped = PACKAGE / rel
64
+ if not source.is_file() or not shipped.is_file():
65
+ continue
66
+ if digest(source) != digest(shipped):
67
+ differing.append(rel)
68
+
69
+ if missing or differing or unexpected:
70
+ print("ERROR: package does not match the source tree", file=sys.stderr)
71
+ for rel in missing[:20]:
72
+ print(f" missing from package: {rel}", file=sys.stderr)
73
+ for rel in differing[:20]:
74
+ print(f" differs from source: {rel}", file=sys.stderr)
75
+ for rel in unexpected[:20]:
76
+ print(f" stale in package: {rel}", file=sys.stderr)
77
+ print(" fix: bash packaging/make_upload.sh", file=sys.stderr)
78
+ return 1
79
+
80
+ print(f"package parity OK ({len(expected)} files byte-identical to source)")
81
+ return 0
82
+
83
+
84
+ if __name__ == "__main__":
85
+ sys.exit(main())
@@ -0,0 +1,375 @@
1
+ #!/usr/bin/env python3
2
+ """check_protocol_alignment.py — Protocol five-way alignment gate.
3
+
4
+ Every scientific contract in this repository is declared in more than one
5
+ place: the workflow registry (engine/workflows.py), the capability registry
6
+ (engine/capabilities.py), the role registry (skill/roles/registry.yaml), the
7
+ role prompts (skill/agents/*.md), the stage briefs (skill/task-briefs/*.md),
8
+ the sub-skill recipes (skill/sub-skills/*/SKILL.md), the routing requirements
9
+ (integrations/agent_mcp.py) and the packaging manifest (packaging/*).
10
+
11
+ Drift between them is silent: a role can lose its brief, a capability can
12
+ exist with no recipe, a version can be bumped in one file only. This gate
13
+ makes that drift fail loudly. Stdlib only; a non-zero exit blocks CI.
14
+
15
+ Usage:
16
+ python3 scripts/check_protocol_alignment.py
17
+ """
18
+ from __future__ import annotations
19
+
20
+ import json
21
+ import re
22
+ import sys
23
+ from pathlib import Path
24
+
25
+ ROOT = Path(__file__).resolve().parent.parent
26
+ sys.path.insert(0, str(ROOT))
27
+
28
+ from engine.capabilities import capability_registry # noqa: E402
29
+ from engine.versions import ENGINE_VERSION # noqa: E402
30
+ from engine.workflows import execution_stages, workflow_registry # noqa: E402
31
+
32
+ FRONTMATTER_RE = re.compile(r"^---\s*\n(.*?)\n---\s*\n", re.DOTALL)
33
+
34
+ #: Projection-layer capabilities sit outside the scientific stage model
35
+ #: (engine/workflows.py: Projection is not a scientific stage), so they are
36
+ #: owned by the projection brief rather than by a scientific role.
37
+ PROJECTION_CAPABILITIES = {"report_projection", "report_rendering"}
38
+
39
+
40
+ def list_domains_from_registry() -> list[dict]:
41
+ """Registered domains (evidencecore is the registry owner)."""
42
+ from engine.evidencecore import list_domains
43
+
44
+ return list_domains()
45
+
46
+
47
+ def _frontmatter(path: Path) -> dict[str, str]:
48
+ """Parse the flat key: value frontmatter used by skills and role prompts."""
49
+ match = FRONTMATTER_RE.match(path.read_text(encoding="utf-8"))
50
+ if not match:
51
+ return {}
52
+ fields: dict[str, str] = {}
53
+ for line in match.group(1).splitlines():
54
+ if line.strip().startswith("#") or ":" not in line:
55
+ continue
56
+ key, _, value = line.partition(":")
57
+ fields[key.strip()] = value.split("#", 1)[0].strip()
58
+ return fields
59
+
60
+
61
+ def _registry_roles() -> dict[str, dict]:
62
+ """The role registry is a small fixed-shape YAML file; parse it narrowly."""
63
+ text = (ROOT / "skill" / "roles" / "registry.yaml").read_text(encoding="utf-8")
64
+ roles: dict[str, dict] = {}
65
+ current: str | None = None
66
+ for raw in text.splitlines():
67
+ if not raw.strip() or raw.strip().startswith("#"):
68
+ continue
69
+ if raw.startswith("roles:"):
70
+ continue
71
+ if raw.startswith("execution:"):
72
+ break
73
+ if re.match(r"^ [A-Za-z0-9_-]+:\s*$", raw):
74
+ current = raw.strip().rstrip(":")
75
+ roles[current] = {}
76
+ continue
77
+ if current and raw.strip().startswith("stages:"):
78
+ stages = raw.split(":", 1)[1].strip().strip("[]")
79
+ roles[current]["stages"] = [s.strip() for s in stages.split(",") if s.strip()]
80
+ elif current and raw.strip().startswith("capabilities:"):
81
+ caps = raw.split(":", 1)[1].strip().strip("[]")
82
+ roles[current]["capabilities"] = [c.strip() for c in caps.split(",") if c.strip()]
83
+ elif current and ":" in raw.strip():
84
+ key, _, value = raw.strip().partition(":")
85
+ roles[current][key.strip()] = value.strip()
86
+ return roles
87
+
88
+
89
+ def _merge_role(roles: dict[str, dict], name: str, stage: str, capability: str,
90
+ critical: bool | None = None, independence: bool | None = None) -> None:
91
+ entry = roles.setdefault(name, {"stages": [], "capabilities": []})
92
+ if stage not in entry["stages"]:
93
+ entry["stages"].append(stage)
94
+ for cap in capability.split("+"):
95
+ cap = cap.strip()
96
+ if cap and cap not in entry["capabilities"]:
97
+ entry["capabilities"].append(cap)
98
+ if critical is not None:
99
+ entry["critical_path"] = "true" if critical else "false"
100
+ if independence:
101
+ entry["independence_required"] = "true"
102
+
103
+
104
+ def check() -> list[str]:
105
+ errors: list[str] = []
106
+
107
+ stages = list(execution_stages())
108
+ scientific_stages = [s for s in stages if s != "projection"]
109
+
110
+ # ---------------------------------------------------------------- briefs
111
+ brief_dir = ROOT / "skill" / "task-briefs"
112
+ briefs = {p.stem for p in brief_dir.glob("*.md")}
113
+ for stage in stages:
114
+ if stage not in briefs:
115
+ errors.append(f"stage {stage!r} has no task brief in skill/task-briefs/")
116
+ for extra in sorted(briefs - set(stages) - {"present"}):
117
+ errors.append(f"task brief {extra!r} does not map to a canonical stage")
118
+
119
+ # ---------------------------------------------------------------- roles
120
+ registry = _registry_roles()
121
+ if not registry:
122
+ errors.append("skill/roles/registry.yaml declares no roles")
123
+ agent_files = {p.stem: p for p in (ROOT / "skill" / "agents").glob("*.md")}
124
+ if set(registry) != set(agent_files):
125
+ errors.append("role registry and skill/agents/*.md disagree: "
126
+ f"registry-only={sorted(set(registry) - set(agent_files))} "
127
+ f"agent-only={sorted(set(agent_files) - set(registry))}")
128
+
129
+ stage_owner: dict[str, str] = {}
130
+ for role, entry in registry.items():
131
+ for stage in entry.get("stages", []):
132
+ if stage in stage_owner:
133
+ errors.append(f"stage {stage!r} is owned by both {stage_owner[stage]!r} and {role!r}")
134
+ stage_owner[stage] = role
135
+ for stage in scientific_stages:
136
+ if stage not in stage_owner:
137
+ errors.append(f"stage {stage!r} has no owning role in the registry")
138
+
139
+ # Registry capabilities must be engine capability IDs: a free-text label
140
+ # here silently detaches the role from the capability it claims to run.
141
+ for role, entry in registry.items():
142
+ for cap in entry.get("capabilities", []):
143
+ if cap not in capability_registry():
144
+ errors.append(f"registry role {role!r} declares capability {cap!r}, "
145
+ "which is not in engine/capabilities.py")
146
+
147
+ # Independence is graded: the skeptic must come from a different model
148
+ # family, the method reviewer must be separated from the content judgement.
149
+ independence = {role: entry.get("independence_required")
150
+ for role, entry in registry.items() if entry.get("independence_required")}
151
+ if independence != {"skeptic": "different-model-family",
152
+ "method-reviewer": "role-separation"}:
153
+ expected = {"skeptic": "different-model-family", "method-reviewer": "role-separation"}
154
+ errors.append("independence_required must be "
155
+ f"{expected}, found {independence}")
156
+
157
+ # ------------------------------------------------- role prompt frontmatter
158
+ for role, path in agent_files.items():
159
+ fields = _frontmatter(path)
160
+ if fields.get("name") != role:
161
+ errors.append(f"{path.name}: frontmatter name {fields.get('name')!r} != filename {role!r}")
162
+ if fields.get("role_id") != role:
163
+ errors.append(f"{path.name}: missing role_id: {role}")
164
+ if not fields.get("capabilities"):
165
+ errors.append(f"{path.name}: missing capabilities declaration")
166
+ if not fields.get("output_contracts"):
167
+ errors.append(f"{path.name}: missing output_contracts declaration")
168
+ for banned in ("default_cli", "default_model"):
169
+ if banned in fields:
170
+ errors.append(f"{path.name}: {banned} must not be bound in the role prompt "
171
+ "(model/CLI choice is a user-confirmed routing decision)")
172
+ for token in ("claude-", "gpt-", "deepseek-", "glm-", "kimi-"):
173
+ if token in fields.get("recommended_reasoning", ""):
174
+ errors.append(f"{path.name}: recommended_reasoning must not contain a model name")
175
+ if role in registry:
176
+ declared = set(registry[role].get("capabilities", []))
177
+ prompt_caps = {c.strip() for c in fields.get("capabilities", "").split(",") if c.strip()}
178
+ unknown = {c for c in prompt_caps if c not in capability_registry()}
179
+ unmapped = {c for c in unknown if not c.startswith("(")}
180
+ if unmapped:
181
+ errors.append(f"{path.name}: capabilities not in the capability registry: {sorted(unmapped)}")
182
+ missing = {c for c in declared if c in capability_registry()} - prompt_caps
183
+ if missing:
184
+ errors.append(f"{path.name}: registry capabilities absent from the prompt: "
185
+ f"{sorted(missing)}")
186
+ if registry.get(role, {}).get("critical_path") == "true":
187
+ if fields.get("critical_path") != "true":
188
+ errors.append(f"{path.name}: registry marks this role critical_path but the "
189
+ "prompt does not declare critical_path: true")
190
+
191
+ # ----------------------------------------------- routing-side requirements
192
+ from integrations.agent_mcp import ROLE_REQUIREMENTS # noqa: E402
193
+ if set(ROLE_REQUIREMENTS) != set(registry):
194
+ errors.append("integrations.agent_mcp.ROLE_REQUIREMENTS and the role registry disagree: "
195
+ f"routing-only={sorted(set(ROLE_REQUIREMENTS) - set(registry))} "
196
+ f"registry-only={sorted(set(registry) - set(ROLE_REQUIREMENTS))}")
197
+ for role, reqs in ROLE_REQUIREMENTS.items():
198
+ for banned in ("default_cli", "default_model", "model", "cli"):
199
+ if banned in reqs:
200
+ errors.append(f"ROLE_REQUIREMENTS[{role!r}] must not bind {banned!r}")
201
+ wants_family = registry.get(role, {}).get("independence_required") == "different-model-family"
202
+ if wants_family and reqs.get("independence") != "different-model-family":
203
+ errors.append(f"ROLE_REQUIREMENTS[{role!r}] must require a different model family")
204
+
205
+ # ------------------------------------------------------------ capabilities
206
+ capabilities = capability_registry()
207
+ sub_skills = sorted(p for p in (ROOT / "skill" / "sub-skills").iterdir()
208
+ if p.is_dir() and not p.name.startswith("."))
209
+ if len(sub_skills) < 5:
210
+ errors.append(f"expected at least 5 sub-skills, found {len(sub_skills)}")
211
+ mapped: set[str] = set()
212
+ for skill_dir in sub_skills:
213
+ path = skill_dir / "SKILL.md"
214
+ if not path.is_file():
215
+ errors.append(f"sub-skill {skill_dir.name} has no SKILL.md")
216
+ continue
217
+ fields = _frontmatter(path)
218
+ if fields.get("name") != skill_dir.name:
219
+ errors.append(f"{skill_dir.name}/SKILL.md: name {fields.get('name')!r} != directory")
220
+ declared = fields.get("capability", "")
221
+ if not declared:
222
+ errors.append(f"{skill_dir.name}/SKILL.md: missing capability declaration")
223
+ continue
224
+ for cap in declared.split("+"):
225
+ cap = cap.strip().split()[0] if cap.strip() else ""
226
+ if cap and not cap.startswith("("):
227
+ mapped.add(cap)
228
+ if cap not in capabilities:
229
+ errors.append(f"{skill_dir.name}/SKILL.md: capability {cap!r} is not in "
230
+ "engine/capabilities.py")
231
+ # Every registered capability must be owned by at least one role, and every
232
+ # recipe capability must be one the engine actually registers.
233
+ owned: set[str] = set()
234
+ for entry in registry.values():
235
+ owned.update(entry.get("capabilities", []))
236
+ for cap in capabilities:
237
+ if cap not in owned and cap not in PROJECTION_CAPABILITIES:
238
+ errors.append(f"capability {cap!r} is registered in engine/capabilities.py "
239
+ "but no role owns it")
240
+ for cap in sorted(mapped):
241
+ if cap in capabilities and cap not in owned and cap not in PROJECTION_CAPABILITIES:
242
+ errors.append(f"sub-skill capability {cap!r} is owned by no role")
243
+
244
+ # ---------------------------------------------------------------- workflows
245
+ workflow_dir = ROOT / "skill" / "workflows"
246
+ workflow_files = {p.stem for p in workflow_dir.glob("*.md")}
247
+ registry_workflows = workflow_registry()
248
+ public = {w for w in registry_workflows if w != "full_research_cycle"}
249
+ normalised = {name.replace("_", "-") for name in public}
250
+ if not normalised <= workflow_files:
251
+ errors.append("workflows missing a runbook: "
252
+ f"{sorted(normalised - workflow_files)}")
253
+ skill_text = (ROOT / "SKILL.md").read_text(encoding="utf-8")
254
+ for name in public:
255
+ reference = f"skill/workflows/{name.replace('_', '-')}.md"
256
+ if reference not in skill_text:
257
+ errors.append(f"SKILL.md does not route to {reference}")
258
+
259
+ skill_md = root_skill_text = skill_text # alias for readability
260
+ for stage in scientific_stages:
261
+ if f"| {stage.capitalize()} " not in skill_md and stage not in skill_md:
262
+ errors.append(f"SKILL.md does not mention stage {stage!r}")
263
+
264
+ # ------------------------------------------------------------- taxonomy
265
+ # The registry is the authority for outcome tokens and their categories;
266
+ # the JSON Schemas carry static enums because JSON Schema cannot read a
267
+ # file at validation time. This dimension is what keeps the static enums
268
+ # honest: drift between a schema enum and the registry fails the gate.
269
+ from engine.taxonomy import (
270
+ all_tokens_ordered,
271
+ categories as taxonomy_categories,
272
+ )
273
+
274
+ registered_tokens = set(all_tokens_ordered())
275
+ evidence_path = ROOT / "schemas" / "evidence.schema.json"
276
+ if evidence_path.is_file():
277
+ evidence_schema = json.loads(evidence_path.read_text(encoding="utf-8"))
278
+ enum = (evidence_schema.get("properties", {})
279
+ .get("outcome_type", {}).get("enum"))
280
+ if not isinstance(enum, list) or not enum:
281
+ errors.append("schemas/evidence.schema.json declares no outcome_type enum")
282
+ else:
283
+ missing = sorted(registered_tokens - set(enum))
284
+ extra = sorted(set(enum) - registered_tokens)
285
+ if missing:
286
+ errors.append(
287
+ "outcome_type enum is missing registered token(s): " + repr(missing))
288
+ if extra:
289
+ errors.append(
290
+ "outcome_type enum declares unregistered token(s): " + repr(extra))
291
+
292
+ # Every domain's ADOPT-gate categories must be categories it declares.
293
+ from engine.tribunal import primary_effect_categories
294
+
295
+ for domain_id in (d["id"] for d in list_domains_from_registry()):
296
+ declared = set(taxonomy_categories(domain_id))
297
+ try:
298
+ primary = primary_effect_categories(domain_id)
299
+ except ValueError as exc:
300
+ errors.append(f"domain {domain_id!r}: ADOPT gate misconfigured: {exc}")
301
+ continue
302
+ for category in primary:
303
+ if category not in declared:
304
+ errors.append(
305
+ f"domain {domain_id!r}: ADOPT gate names undeclared category "
306
+ + repr(category))
307
+
308
+ # Every V2 outcome bucket must be a category some domain declares.
309
+ v2_outcome = ROOT / "schemas" / "v2" / "outcome.schema.json"
310
+ if v2_outcome.is_file():
311
+ v2_schema = json.loads(v2_outcome.read_text(encoding="utf-8"))
312
+ buckets = (v2_schema.get("properties", {})
313
+ .get("outcome_type", {}).get("enum"))
314
+ if isinstance(buckets, list) and buckets:
315
+ every = set()
316
+ for domain_id in (d["id"] for d in list_domains_from_registry()):
317
+ every.update(taxonomy_categories(domain_id))
318
+ orphan = sorted(set(buckets) - every)
319
+ # Reverse direction too: a category a domain declares but the V2
320
+ # contract omits would reject that domain's outcomes at the graph
321
+ # layer, which is exactly how policy was blocked.
322
+ missing_bucket = sorted(every - set(buckets))
323
+ if missing_bucket:
324
+ errors.append(
325
+ "schemas/v2/outcome.schema.json is missing category bucket(s) "
326
+ "that domains declare: " + repr(missing_bucket))
327
+ if orphan:
328
+ errors.append(
329
+ "schemas/v2/outcome.schema.json declares bucket(s) no domain "
330
+ "registers: " + repr(orphan))
331
+
332
+ # ---------------------------------------------------------------- versions
333
+ for relative in ("packaging/scp-manifest.json",):
334
+ path = ROOT / relative
335
+ if not path.is_file():
336
+ continue
337
+ data = json.loads(path.read_text(encoding="utf-8"))
338
+ declared = (data.get("skill") or {}).get("version")
339
+ if declared != ENGINE_VERSION:
340
+ errors.append(f"{relative}: skill.version {declared!r} != ENGINE_VERSION {ENGINE_VERSION!r}")
341
+ start_here = ROOT / "packaging" / "START-HERE.md"
342
+ if start_here.is_file():
343
+ text = start_here.read_text(encoding="utf-8")
344
+ major_minor = ".".join(ENGINE_VERSION.split(".")[:2])
345
+ if f"EduEvidence {major_minor}" not in text:
346
+ errors.append(f"packaging/START-HERE.md does not state EduEvidence {major_minor}")
347
+
348
+ # ------------------------------------------------------- docs must not drift
349
+ for doc in ("docs/architecture.md", "README.zh-CN.md", "docs/install-guide.md"):
350
+ path = ROOT / doc
351
+ if not path.is_file():
352
+ continue
353
+ text = path.read_text(encoding="utf-8")
354
+ for stale in ("752 个测试", "752 tests"):
355
+ if stale in text:
356
+ errors.append(f"{doc}: stale test count {stale!r}; use docs/metrics.json")
357
+
358
+ return errors
359
+
360
+
361
+ def main() -> int:
362
+ print("[*] Checking protocol alignment across workflows, roles, capabilities and packaging...")
363
+ errors = check()
364
+ if errors:
365
+ print(f"[-] Protocol alignment FAILED with {len(errors)} error(s):", file=sys.stderr)
366
+ for error in errors:
367
+ print(f" • {error}", file=sys.stderr)
368
+ return 1
369
+ print("[+] Protocol alignment PASSED: stages, briefs, roles, prompts, capabilities, "
370
+ "sub-skills, workflows and packaging agree.")
371
+ return 0
372
+
373
+
374
+ if __name__ == "__main__":
375
+ sys.exit(main())