eduevidence 6.0.0 → 6.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (189) hide show
  1. package/CONTRIBUTING.md +105 -0
  2. package/README.md +93 -38
  3. package/README.zh-CN.md +26 -6
  4. package/SKILL.md +11 -2
  5. package/assets/readme/landing-tour.gif +0 -0
  6. package/assets/readme/studio-tour.gif +0 -0
  7. package/bin/eduevidence.js +2 -1
  8. package/docs/architecture.md +319 -43
  9. package/docs/demo-workplace-ai.md +1 -1
  10. package/docs/install-guide.md +1 -1
  11. package/docs/orchestration-role-model.md +1 -1
  12. package/docs/release-closeout/README.md +1 -1
  13. package/docs/sciverse-api.md +125 -0
  14. package/eduevidence_cli.py +10 -0
  15. package/engine/decision_policy.py +96 -0
  16. package/engine/evidence_graph.py +14 -10
  17. package/engine/gaps.py +42 -22
  18. package/engine/ids.py +2 -0
  19. package/engine/library.py +6 -2
  20. package/engine/living.py +34 -4
  21. package/engine/migration.py +88 -3
  22. package/engine/orchestration.py +5 -5
  23. package/engine/paths.py +2 -0
  24. package/engine/pilot.py +34 -32
  25. package/engine/taxonomy.py +211 -0
  26. package/engine/tribunal.py +43 -31
  27. package/engine/versions.py +1 -1
  28. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1360 -146
  29. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
  30. package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
  31. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  32. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  33. package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
  34. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +447 -127
  35. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +447 -127
  36. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +447 -127
  37. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +447 -127
  38. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +447 -127
  39. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
  40. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
  41. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
  42. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
  43. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
  44. package/examples/ai-coding-assistant-evidence/result.json +13 -9
  45. package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
  46. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  47. package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
  48. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  49. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  50. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  51. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  52. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  53. package/examples/spaced-retrieval-practice/frame.json +58 -0
  54. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  55. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  56. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  57. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  58. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  59. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  60. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  61. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  62. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  63. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  64. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  65. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  66. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  67. package/examples/spaced-retrieval-practice/result.json +942 -0
  68. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  69. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  70. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  71. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  72. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  73. package/examples/workplace-ai-assistant/claims.jsonl +4 -4
  74. package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
  75. package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
  76. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  77. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  78. package/examples/workplace-ai-assistant/report_spec.json +209 -40
  79. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +435 -105
  80. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +435 -105
  81. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +435 -105
  82. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +435 -105
  83. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +435 -105
  84. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  85. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  86. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  87. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  88. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  89. package/examples/workplace-ai-assistant/result.json +82 -20
  90. package/examples/workplace-ai-assistant/result.zh.json +82 -20
  91. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  92. package/examples/workplace-ai-assistant/verdict.json +36 -10
  93. package/integrations/agent_mcp.py +2 -2
  94. package/package.json +12 -3
  95. package/pyproject.toml +4 -3
  96. package/references/report-copy-style.md +67 -0
  97. package/references/retrieval-compliance.md +75 -0
  98. package/references/retrieval-protocol.md +20 -0
  99. package/retrieval/audit.py +27 -3
  100. package/retrieval/fetch.py +96 -0
  101. package/retrieval/sciverse.py +398 -0
  102. package/retrieval/search.py +47 -7
  103. package/schemas/applicability.schema.json +94 -0
  104. package/schemas/chart-spec.schema.json +10 -3
  105. package/schemas/evidence.schema.json +316 -43
  106. package/schemas/fetch-result.schema.json +2 -1
  107. package/schemas/report-result.schema.json +3 -3
  108. package/schemas/report-spec.schema.json +98 -100
  109. package/schemas/skeptic.schema.json +86 -0
  110. package/schemas/source.schema.json +21 -2
  111. package/schemas/v2/finding.schema.json +5 -1
  112. package/schemas/v2/methodology-audit.schema.json +5 -1
  113. package/schemas/v2/outcome.schema.json +28 -5
  114. package/schemas/v2/study.schema.json +5 -1
  115. package/schemas/vNext/autoevolve-session.schema.json +34 -1
  116. package/schemas/vNext/eval-snapshot.schema.json +77 -1
  117. package/schemas/vNext/execution-plan.schema.json +50 -1
  118. package/schemas/vNext/gap-priority.schema.json +54 -1
  119. package/schemas/vNext/negative-search-record.schema.json +68 -1
  120. package/schemas/vNext/research-iteration.schema.json +87 -1
  121. package/schemas/vNext/research-strategy.schema.json +62 -1
  122. package/schemas/vNext/skill-experiment.schema.json +90 -1
  123. package/schemas/vNext/task-spec.schema.json +156 -1
  124. package/schemas/vNext/worker-result.schema.json +60 -1
  125. package/schemas/verdict.schema.json +164 -28
  126. package/scripts/build_esl_artifacts.py +2 -2
  127. package/scripts/build_report_variants.py +18 -2
  128. package/scripts/build_result.py +74 -9
  129. package/scripts/check_package_parity.py +85 -0
  130. package/scripts/check_protocol_alignment.py +375 -0
  131. package/scripts/check_versioned_schemas.py +254 -0
  132. package/scripts/claim_audit.py +13 -8
  133. package/scripts/compute_confidence.py +10 -0
  134. package/scripts/did_regression.py +12 -2
  135. package/scripts/evidence_score.py +5 -2
  136. package/scripts/generate_new_projects.py +4 -4
  137. package/scripts/orchestrator.py +120 -24
  138. package/scripts/pre_verdict_gate.py +224 -26
  139. package/scripts/quickstart.py +18 -2
  140. package/scripts/run_workspace.py +7 -1
  141. package/scripts/skill_payload.py +4 -1
  142. package/scripts/test_adversarial_empirical.py +26 -19
  143. package/scripts/validate_schema.py +31 -1
  144. package/skill/agents/evaluation-designer.md +20 -4
  145. package/skill/agents/evidence-analyst.md +19 -3
  146. package/skill/agents/evidence-judge.md +50 -2
  147. package/skill/agents/evidence-retriever.md +20 -3
  148. package/skill/agents/intervention-designer.md +20 -4
  149. package/skill/agents/method-reviewer.md +18 -2
  150. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  151. package/skill/agents/skeptic.md +18 -2
  152. package/skill/roles/registry.yaml +11 -11
  153. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  154. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  155. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  156. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  157. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  158. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  159. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  160. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  161. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  162. package/skill/sub-skills/report-generation/SKILL.md +28 -0
  163. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  164. package/skill/sub-skills/study-design/SKILL.md +30 -9
  165. package/skill/task-briefs/adjudicate.md +32 -7
  166. package/skill/task-briefs/applicability.md +37 -2
  167. package/skill/task-briefs/audit.md +32 -7
  168. package/skill/task-briefs/challenge.md +34 -5
  169. package/skill/task-briefs/evaluate.md +30 -5
  170. package/skill/task-briefs/extract.md +31 -8
  171. package/skill/task-briefs/frame.md +39 -10
  172. package/skill/task-briefs/intervene.md +32 -6
  173. package/skill/task-briefs/present.md +32 -8
  174. package/skill/task-briefs/projection.md +36 -2
  175. package/skill/task-briefs/retrieve.md +36 -6
  176. package/skill/workflows/decision-and-pilot.md +76 -1
  177. package/skill/workflows/evaluate-and-update.md +83 -0
  178. package/skill/workflows/evidence-review.md +104 -0
  179. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  180. package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
  181. package/visualization/eduevidence-report/scripts/build_report.py +512 -65
  182. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  183. package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
  184. package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
  185. package/web/architecture.html +14885 -0
  186. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  187. package/web/studio/index.html +2 -2
  188. package/web/studio/assets/index-CzXocaGv.css +0 -1
  189. /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
@@ -75,8 +75,8 @@ class RoleSpec:
75
75
 
76
76
 
77
77
  ROLE_REGISTRY: dict[str, RoleSpec] = {
78
- "education-planner": RoleSpec(
79
- "education-planner",
78
+ "research-planner": RoleSpec(
79
+ "research-planner",
80
80
  "Own framing completeness, scope, comparison and outcome definition.",
81
81
  ("frame",),
82
82
  ("research-planning",),
@@ -409,7 +409,7 @@ class ExecutionPlanner:
409
409
 
410
410
  def _serial_tasks(self, run_id, base_revision) -> tuple[TaskSpec, ...]:
411
411
  return (
412
- self._base("frame", "frame", "education-planner", "Structure the research question.", "frame", run_id=run_id, base_revision=base_revision),
412
+ self._base("frame", "frame", "research-planner", "Structure the research question.", "frame", run_id=run_id, base_revision=base_revision),
413
413
  self._base("retrieve", "retrieve", "evidence-retriever", "Acquire bounded evidence.", "direct+counter", run_id=run_id, base_revision=base_revision),
414
414
  self._base("extract", "extract", "evidence-analyst", "Extract structured findings.", "all-eligible", run_id=run_id, base_revision=base_revision),
415
415
  self._base("challenge", "challenge", "skeptic", "Challenge the provisional interpretation.", "counter-evidence", run_id=run_id, base_revision=base_revision),
@@ -419,7 +419,7 @@ class ExecutionPlanner:
419
419
 
420
420
  def _medium_tasks(self, run_id, base_revision) -> tuple[TaskSpec, ...]:
421
421
  return (
422
- self._base("frame", "frame", "education-planner", "Structure the research question.", "frame", run_id=run_id, base_revision=base_revision),
422
+ self._base("frame", "frame", "research-planner", "Structure the research question.", "frame", run_id=run_id, base_revision=base_revision),
423
423
  self._base("retrieve-direct", "retrieve", "evidence-retriever", "Retrieve direct decision-relevant evidence.", "direct-causal", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
424
424
  self._base("retrieve-counter", "retrieve", "evidence-retriever", "Retrieve null, negative and contradictory evidence.", "counter-risk", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
425
425
  self._base("extract", "extract", "evidence-analyst", "Merge validated sources and extract findings.", "all-eligible", run_id=run_id, base_revision=base_revision),
@@ -430,7 +430,7 @@ class ExecutionPlanner:
430
430
 
431
431
  def _deep_tasks(self, run_id, base_revision) -> tuple[TaskSpec, ...]:
432
432
  return (
433
- self._base("frame", "frame", "education-planner", "Structure the research question.", "frame", run_id=run_id, base_revision=base_revision),
433
+ self._base("frame", "frame", "research-planner", "Structure the research question.", "frame", run_id=run_id, base_revision=base_revision),
434
434
  self._base("retrieve-direct", "retrieve", "evidence-retriever", "Retrieve direct causal evidence.", "direct-causal", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
435
435
  self._base("retrieve-transfer", "retrieve", "evidence-retriever", "Retrieve retention and independent-transfer evidence.", "transfer-retention", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
436
436
  self._base("retrieve-counter", "retrieve", "evidence-retriever", "Retrieve null, negative, risk and contradiction evidence.", "counter-risk", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
package/engine/paths.py CHANGED
@@ -4,6 +4,8 @@ EDUEVIDENCE_HOME (default `~/.eduevidence`) is the root that owns the Shared
4
4
  Research Library and all Projects. An explicit path always wins.
5
5
  """
6
6
 
7
+ from __future__ import annotations
8
+
7
9
  from pathlib import Path
8
10
  import os
9
11
 
package/engine/pilot.py CHANGED
@@ -22,6 +22,7 @@ from engine.datasets import analysis_blocked_by_privacy, derive_csv_profile, ing
22
22
  from engine.graph_store import GraphMutation, GraphStore
23
23
  from engine.ids import new_local_id, new_run_id
24
24
  from engine.project import ProjectWorkspace
25
+ from engine.taxonomy import category_of, tokens as taxonomy_tokens
25
26
  from engine.synthesis import synthesize_project
26
27
  from engine.tribunal import adjudicate, decision_diff, save_decision_snapshot
27
28
  from engine.versions import (
@@ -30,15 +31,21 @@ from engine.versions import (
30
31
  SOURCE_VALIDATION_POLICY_VERSION,
31
32
  )
32
33
 
33
- #: Outcome Taxonomy tokens that pilots may measure (mirrors outcome-taxonomy.md).
34
- OUTCOME_TAXONOMY = {
35
- "knowledge_gain", "concept_understanding", "retention", "transfer",
36
- "independent_problem_solving", "completion_time", "accuracy",
37
- "code_quality", "assignment_score", "engagement", "motivation",
38
- "cognitive_load", "help_seeking", "metacognition", "ai_dependency",
39
- "over_reliance", "reduced_effort", "reduced_transfer",
40
- "academic_integrity_risk", "false_confidence",
41
- }
34
+ def outcome_taxonomy_tokens(domain: str = "education") -> set[str]:
35
+ """Outcome tokens a pilot may measure, read from the domain registry.
36
+
37
+ This was a module-level set of the 20 education tokens, so a policy pilot
38
+ could not register at all. The domain registry (via engine/taxonomy.py) is
39
+ now the single authority.
40
+ """
41
+ return set(taxonomy_tokens(domain))
42
+
43
+
44
+ def project_domain(project: ProjectWorkspace) -> str:
45
+ """The domain a project registers; defaults to education when absent."""
46
+ manifest = project.manifest()
47
+ return str(manifest.get("domain") or "education")
48
+
42
49
 
43
50
  PILOT_STATUSES = ("registered", "data_imported", "analyzed", "adjudicated")
44
51
 
@@ -49,22 +56,6 @@ PII_COLUMN_HINTS = ("name", "student", "学号", "姓名", "email", "mail",
49
56
  _DECISION_IMPLICATION = {"support": "support_adoption",
50
57
  "contradict": "oppose_adoption", "neutral": "neutral"}
51
58
 
52
- #: Outcome Taxonomy token -> graph outcome category enum (schemas/v2/outcome).
53
- _OUTCOME_CATEGORY = {
54
- "knowledge_gain": "learning", "concept_understanding": "learning",
55
- "retention": "learning", "transfer": "learning",
56
- "independent_problem_solving": "learning",
57
- "completion_time": "task_performance", "accuracy": "task_performance",
58
- "code_quality": "task_performance", "assignment_score": "task_performance",
59
- "engagement": "process", "motivation": "process",
60
- "cognitive_load": "process", "help_seeking": "process",
61
- "metacognition": "process",
62
- "ai_dependency": "risk", "over_reliance": "risk",
63
- "reduced_effort": "risk", "reduced_transfer": "risk",
64
- "academic_integrity_risk": "risk", "false_confidence": "risk",
65
- }
66
-
67
-
68
59
  def _now_iso() -> str:
69
60
  return datetime.now(timezone.utc).isoformat()
70
61
 
@@ -111,10 +102,13 @@ def register_pilot(project: ProjectWorkspace, *,
111
102
  raise ValueError(
112
103
  f"decision snapshot {decision_snapshot_id} not found in this project; "
113
104
  "a pilot must bind to a real adjudication")
114
- unknown = [o for o in outcome_columns if o not in OUTCOME_TAXONOMY]
105
+ domain = project_domain(project)
106
+ known_tokens = outcome_taxonomy_tokens(domain)
107
+ unknown = [o for o in outcome_columns if o not in known_tokens]
115
108
  if unknown:
116
109
  raise ValueError(
117
- f"outcome_columns outside Outcome Taxonomy: {sorted(unknown)}")
110
+ f"outcome_columns outside the {domain} Outcome Taxonomy: "
111
+ f"{sorted(unknown)}")
118
112
  if not conditions or sample_size < 1:
119
113
  raise ValueError("conditions must be non-empty and sample_size >= 1")
120
114
  if anon_policy.get("no_pii_columns") is not True:
@@ -208,7 +202,13 @@ def link_analysis(project: ProjectWorkspace, pilot_id: str, *,
208
202
  return pilot
209
203
 
210
204
 
211
- def _ensure_outcome(store: GraphStore, outcome_id: str) -> dict:
205
+ def _ensure_outcome(store: GraphStore, outcome_id: str, domain: str) -> dict:
206
+ """Create the pilot outcome row, classified by the DOMAIN registry.
207
+
208
+ ``category_of`` raises for an unregistered token; the caller has already
209
+ validated the token against ``outcome_taxonomy_tokens(domain)``, so an
210
+ error here means the two disagreed and must not be silently absorbed.
211
+ """
212
212
  existing = store.get("outcomes", outcome_id)
213
213
  if existing:
214
214
  return existing
@@ -216,7 +216,7 @@ def _ensure_outcome(store: GraphStore, outcome_id: str) -> dict:
216
216
  return {
217
217
  "outcome_id": outcome_id,
218
218
  "name": token,
219
- "outcome_type": _OUTCOME_CATEGORY.get(token, "learning"),
219
+ "outcome_type": category_of(domain, token),
220
220
  "extensions": {"pilot_outcome": True},
221
221
  }
222
222
 
@@ -237,9 +237,11 @@ def redecide(project: ProjectWorkspace, pilot_id: str, *,
237
237
  raise ValueError(f"invalid effect_direction {effect_direction!r}")
238
238
  if relation_to_claim not in ("support", "contradict", "neutral"):
239
239
  raise ValueError(f"invalid relation_to_claim {relation_to_claim!r}")
240
- if outcome_token not in OUTCOME_TAXONOMY:
240
+ domain = project_domain(project)
241
+ if outcome_token not in outcome_taxonomy_tokens(domain):
241
242
  raise ValueError(
242
- f"outcome_token {outcome_token!r} outside Outcome Taxonomy")
243
+ f"outcome_token {outcome_token!r} outside the {domain} "
244
+ "Outcome Taxonomy")
243
245
  outcome_id = f"OUT-{outcome_token}"
244
246
  pilot = _load_pilot(project, pilot_id)
245
247
  if pilot["status"] not in ("analyzed", "data_imported"):
@@ -282,7 +284,7 @@ def redecide(project: ProjectWorkspace, pilot_id: str, *,
282
284
  "identity_status": "resolved",
283
285
  "extensions": {"pilot_id": pilot_id},
284
286
  }
285
- outcome = _ensure_outcome(store, outcome_id)
287
+ outcome = _ensure_outcome(store, outcome_id, domain)
286
288
  estimate = None
287
289
  if effect_estimate is not None:
288
290
  estimate = {
@@ -0,0 +1,211 @@
1
+ """engine/taxonomy.py - Outcome taxonomy authority (domain registry backed).
2
+
3
+ Every domain registers its own outcome taxonomy in
4
+ ``domains/<id>/outcome_taxonomy.json``. Each file declares the domain category
5
+ buckets (``categories``) and one entry per outcome token carrying an explicit
6
+ ``category`` (``tokens[].category``).
7
+
8
+ This module is the ONLY reader of that contract. Before it existed, the token
9
+ set and the token-to-category mapping were hard-coded in ``engine/pilot.py``
10
+ with a ``.get(token, "learning")`` fallback, so a policy outcome was silently
11
+ classified as a learning outcome and policy runs could not complete. Callers
12
+ must no longer keep private copies of either table.
13
+
14
+ Fail-closed rule: an unknown token or an unregistered domain raises. Silent
15
+ classification is what produced the original defect; unknown input must be
16
+ visible as an error, never absorbed into a default bucket.
17
+
18
+ Stdlib only; results are cached per process.
19
+ """
20
+ from __future__ import annotations
21
+
22
+ import json
23
+ from typing import Any, Iterable
24
+
25
+ from engine.evidencecore import list_domains, load_domain
26
+
27
+ #: Bucket used when a caller must render an outcome whose category could not be
28
+ #: resolved. It is deliberately not a domain category: it marks the value as
29
+ #: unclassified instead of pretending it belongs to a real bucket.
30
+ UNCLASSIFIED = "unclassified"
31
+
32
+ _cache: dict[str, Any] = {}
33
+
34
+
35
+ class TaxonomyError(ValueError):
36
+ """Raised when taxonomy data is missing, malformed, or unknown."""
37
+
38
+
39
+ def _taxonomy_path(domain_id: str) -> str:
40
+ """Registered relative path of a domain taxonomy file."""
41
+ entry = load_domain(domain_id)
42
+ raw = entry.get("outcome_taxonomy")
43
+ if not raw:
44
+ raise TaxonomyError(
45
+ "domain " + repr(domain_id) + " registers no outcome_taxonomy")
46
+ return str(raw).partition("#")[0]
47
+
48
+
49
+ def _load(domain_id: str) -> dict:
50
+ """Load (and cache) one domain taxonomy, validating its shape."""
51
+ key = "taxonomy:" + domain_id
52
+ if key in _cache:
53
+ return _cache[key]
54
+ from engine._resources import resource_root
55
+
56
+ path = resource_root() / _taxonomy_path(domain_id)
57
+ if not path.is_file():
58
+ raise TaxonomyError(
59
+ "domain " + repr(domain_id) + ": taxonomy file missing: " + str(path))
60
+ data = json.loads(path.read_text(encoding="utf-8"))
61
+ categories = data.get("categories")
62
+ tokens = data.get("tokens")
63
+ if not isinstance(categories, dict) or not categories:
64
+ raise TaxonomyError(
65
+ "domain " + repr(domain_id) + ": taxonomy declares no categories")
66
+ if not isinstance(tokens, list) or not tokens:
67
+ raise TaxonomyError(
68
+ "domain " + repr(domain_id) + ": taxonomy declares no tokens")
69
+ seen: set[str] = set()
70
+ for item in tokens:
71
+ if not isinstance(item, dict) or not item.get("id"):
72
+ raise TaxonomyError(
73
+ "domain " + repr(domain_id) + ": token entry without an id")
74
+ token = str(item["id"])
75
+ if token in seen:
76
+ raise TaxonomyError(
77
+ "domain " + repr(domain_id) + ": duplicate token " + repr(token))
78
+ seen.add(token)
79
+ category = item.get("category")
80
+ if not category:
81
+ raise TaxonomyError(
82
+ "domain " + repr(domain_id) + ": token " + repr(token)
83
+ + " declares no category (silent defaults are forbidden)")
84
+ if str(category) not in categories:
85
+ raise TaxonomyError(
86
+ "domain " + repr(domain_id) + ": token " + repr(token)
87
+ + " uses category " + repr(category) + " absent from "
88
+ + repr(sorted(categories)))
89
+ _cache[key] = data
90
+ return data
91
+
92
+
93
+ def _domain_ids(domain_id: str | None = None) -> tuple[str, ...]:
94
+ if domain_id:
95
+ return (domain_id,)
96
+ return tuple(d["id"] for d in list_domains())
97
+
98
+
99
+ def tokens(domain_id: str) -> tuple[str, ...]:
100
+ """Outcome tokens declared by one domain, in registry order."""
101
+ return tuple(str(item["id"]) for item in _load(domain_id)["tokens"])
102
+
103
+
104
+ def categories(domain_id: str) -> dict[str, dict]:
105
+ """Category buckets declared by one domain (id -> descriptor)."""
106
+ return dict(_load(domain_id)["categories"])
107
+
108
+
109
+ def category_of(domain_id: str, token: str) -> str:
110
+ """Category bucket for a token in a domain.
111
+
112
+ Raises TaxonomyError for an unknown token: a caller that cannot classify a
113
+ value must surface that, not guess. Use category_of_or_unclassified at
114
+ rendering boundaries where an unclassified value is acceptable.
115
+ """
116
+ for item in _load(domain_id)["tokens"]:
117
+ if str(item["id"]) == token:
118
+ return str(item["category"])
119
+ raise TaxonomyError(
120
+ "domain " + repr(domain_id) + ": unknown outcome token " + repr(token)
121
+ + " (" + str(len(tokens(domain_id))) + " tokens known)")
122
+
123
+
124
+ def category_of_or_unclassified(domain_id: str, token: str) -> str:
125
+ """Rendering-safe variant: unknown tokens map to UNCLASSIFIED."""
126
+ try:
127
+ return category_of(domain_id, token)
128
+ except TaxonomyError:
129
+ return UNCLASSIFIED
130
+
131
+
132
+ def category_labels(domain_id: str, lang: str = "zh") -> dict[str, str]:
133
+ """Display label per category bucket (falls back to the category id)."""
134
+ suffix = "_en" if lang == "en" else "_zh"
135
+ out: dict[str, str] = {}
136
+ for key, descriptor in categories(domain_id).items():
137
+ if isinstance(descriptor, dict):
138
+ label = descriptor.get("name" + suffix) or descriptor.get("name")
139
+ out[key] = str(label or key)
140
+ else:
141
+ out[key] = key
142
+ return out
143
+
144
+
145
+ def all_tokens() -> dict[str, str]:
146
+ """Every registered token -> owning domain, across registered domains.
147
+
148
+ A token registered by two domains is a contract conflict and raises: the
149
+ token would otherwise mean different things depending on lookup order.
150
+ """
151
+ key = "all_tokens"
152
+ if key in _cache:
153
+ return _cache[key]
154
+ out: dict[str, str] = {}
155
+ for domain_id in _domain_ids():
156
+ for token in tokens(domain_id):
157
+ owner = out.get(token)
158
+ if owner is not None and owner != domain_id:
159
+ raise TaxonomyError(
160
+ "outcome token " + repr(token) + " is registered by both "
161
+ + repr(owner) + " and " + repr(domain_id))
162
+ out[token] = domain_id
163
+ _cache[key] = out
164
+ return out
165
+
166
+
167
+ def domain_of(token: str, default: str = "education") -> str:
168
+ """Owning domain for a token; default when the token is unregistered."""
169
+ return all_tokens().get(token, default)
170
+
171
+
172
+ def all_tokens_ordered() -> tuple[str, ...]:
173
+ """Every registered token in domain-registry then taxonomy order."""
174
+ ordered: list[str] = []
175
+ for domain_id in _domain_ids():
176
+ ordered.extend(tokens(domain_id))
177
+ return tuple(ordered)
178
+
179
+
180
+ def all_categories() -> tuple[str, ...]:
181
+ """Every category bucket across domains, de-duplicated, order preserved."""
182
+ ordered: list[str] = []
183
+ for domain_id in _domain_ids():
184
+ for name in categories(domain_id):
185
+ if name not in ordered:
186
+ ordered.append(name)
187
+ return tuple(ordered)
188
+
189
+
190
+ def categories_for_tokens(
191
+ token_list: Iterable[str], domain_id: str = "education"
192
+ ) -> dict[str, list[str]]:
193
+ """Group tokens by category; unregistered tokens land in UNCLASSIFIED."""
194
+ grouped: dict[str, list[str]] = {}
195
+ for token in token_list:
196
+ bucket = category_of_or_unclassified(domain_id, token)
197
+ grouped.setdefault(bucket, []).append(token)
198
+ return grouped
199
+
200
+
201
+ def reset_cache() -> None:
202
+ """Drop memoised taxonomies (tests and long-lived processes)."""
203
+ _cache.clear()
204
+
205
+
206
+ __all__ = [
207
+ "UNCLASSIFIED", "TaxonomyError",
208
+ "tokens", "categories", "category_of", "category_of_or_unclassified",
209
+ "category_labels", "all_tokens", "all_tokens_ordered", "all_categories",
210
+ "categories_for_tokens", "domain_of", "reset_cache",
211
+ ]
@@ -38,6 +38,12 @@ from datetime import datetime, timezone
38
38
  from pathlib import Path
39
39
 
40
40
  from engine.contracts import validate_record
41
+ from engine.decision_policy import (
42
+ ADOPT_DIRECTNESS,
43
+ decision_action as _policy_decision_action,
44
+ outcome_category,
45
+ primary_effect_categories,
46
+ )
41
47
  from engine.graph_store import GraphStore
42
48
  from engine.ids import new_local_id
43
49
  from engine.project import ProjectWorkspace
@@ -230,16 +236,18 @@ def _confidence(store: GraphStore, syntheses: tuple[ClaimSynthesis, ...]) -> dic
230
236
  "decisive_relations": dict(decisive)}
231
237
 
232
238
 
233
- def _has_direct_learning_evidence(store: GraphStore,
234
- decisive_relations: dict[str, str]) -> bool:
235
- """True when at least one decisive support_adoption Study measures a
236
- learning outcome directly.
239
+ def _has_direct_primary_evidence(store: GraphStore,
240
+ decisive_relations: dict[str, str],
241
+ domain: str = "education") -> bool:
242
+ """True when a decisive support_adoption Study measures a primary outcome.
237
243
 
238
- Learning evidence means the finding's outcome is declared
239
- outcome_type == "learning" AND its evidence link carries directness == 2.
240
- Task performance / process / risk outcomes never qualify, and a missing
241
- outcome record or missing directness fails closed (False).
244
+ Primary means the finding's outcome_type maps — through the domain
245
+ registry — into one of :func:`primary_effect_categories` (education:
246
+ learning; policy: effectiveness/cost), AND its evidence link carries
247
+ directness == 2. Task-performance, process and risk outcomes never
248
+ qualify, and a missing outcome record or missing directness fails closed.
242
249
  """
250
+ primary = primary_effect_categories(domain)
243
251
  findings = {f["finding_id"]: f for f in store.read_table("findings")}
244
252
  outcomes = {o["outcome_id"]: o for o in store.read_table("outcomes")}
245
253
  links_by_finding: dict[str, list[dict]] = {}
@@ -254,37 +262,37 @@ def _has_direct_learning_evidence(store: GraphStore,
254
262
  if fnd.get("study_id") not in support_studies:
255
263
  continue
256
264
  outcome = outcomes.get(fnd.get("outcome_id"))
257
- if outcome is None or outcome.get("outcome_type") != "learning":
265
+ if outcome is None:
266
+ continue
267
+ value = str(outcome.get("outcome_type") or "")
268
+ category = outcome_category(domain, value, primary)
269
+ # Unknown value: fail closed rather than treating it as
270
+ # decision-grade evidence.
271
+ if category is None:
272
+ continue
273
+ if category not in primary:
258
274
  continue
259
275
  for link in links_by_finding.get(fid, []):
260
276
  directness = link.get("directness")
261
- if isinstance(directness, (int, float)) and int(directness) == 2:
277
+ if (isinstance(directness, (int, float))
278
+ and int(directness) == ADOPT_DIRECTNESS):
262
279
  return True
263
280
  return False
264
281
 
265
282
 
266
283
  def _decision_action(syn_statuses: dict[str, str], confidence: dict,
267
284
  decisive_relations: dict[str, str],
268
- has_direct_learning_evidence: bool = False) -> str:
269
- """Gate-enforced decision action.
270
-
271
- REJECT requires usable direct opposition evidence (an independent Study
272
- folded to oppose_adoption). Low/Insufficient can never yield ADOPT.
273
- ADOPT additionally requires direct learning/transfer evidence: High +
274
- decisive support WITHOUT a direct learning outcome downgrades to PILOT
275
- (task performance / procedural efficiency is not learning). Moderate +
276
- decisive support → PILOT; otherwise INSUFFICIENT_EVIDENCE.
285
+ has_direct_primary_evidence: bool = False) -> str:
286
+ """Gate-enforced decision action; rule lives in engine.decision_policy.
287
+
288
+ The V1 Pre-Verdict Gate enforces the same rule, so the thresholds are
289
+ imported rather than restated here.
277
290
  """
278
- label = confidence["label"]
279
- has_oppose = any(r == "oppose_adoption" for r in decisive_relations.values())
280
- has_support = any(r == "support_adoption" for r in decisive_relations.values())
281
- if has_oppose:
282
- return "REJECT"
283
- if label == "High" and has_support and has_direct_learning_evidence:
284
- return "ADOPT"
285
- if label in ("High", "Moderate") and has_support:
286
- return "PILOT"
287
- return "INSUFFICIENT_EVIDENCE"
291
+ return _policy_decision_action(
292
+ confidence_label=confidence["label"],
293
+ decisive_relations=decisive_relations,
294
+ has_direct_primary_evidence=has_direct_primary_evidence,
295
+ )
288
296
 
289
297
 
290
298
 
@@ -299,10 +307,11 @@ def adjudicate(store: GraphStore, *, project: ProjectWorkspace,
299
307
 
300
308
  syn_statuses = {s.claim_id: s.status for s in syntheses}
301
309
  decisive_relations = confidence.get("decisive_relations", {})
302
- direct_learning = _has_direct_learning_evidence(store, decisive_relations)
310
+ domain = str(project.manifest().get("domain") or "education")
311
+ direct_learning = _has_direct_primary_evidence(store, decisive_relations, domain)
303
312
 
304
313
  decision = _decision_action(syn_statuses, confidence, decisive_relations,
305
- has_direct_learning_evidence=direct_learning)
314
+ has_direct_primary_evidence=direct_learning)
306
315
 
307
316
  key_links: list[str] = []
308
317
  for syn in syntheses:
@@ -351,6 +360,9 @@ def adjudicate(store: GraphStore, *, project: ProjectWorkspace,
351
360
  "extensions": {"confidence_components": {
352
361
  "decisive_studies": confidence.get("decisive_studies", 0),
353
362
  "usable_studies": confidence.get("usable_studies", 0),
363
+ "has_direct_primary_evidence": direct_learning,
364
+ # Backward-compatible alias under the education-era name,
365
+ # so readers written against the older contract keep working.
354
366
  "has_direct_learning_evidence": direct_learning,
355
367
  }},
356
368
  }
@@ -4,7 +4,7 @@ Policy versions are frozen identifiers, not free-form strings: changing a
4
4
  policy requires a new version, never silent mutation of an existing one.
5
5
  """
6
6
 
7
- ENGINE_VERSION = "6.0.0"
7
+ ENGINE_VERSION = "6.2.0"
8
8
  GRAPH_SCHEMA_VERSION = "2.0"
9
9
  SOURCE_VALIDATION_POLICY_VERSION = "2026-08-12.v2"
10
10
  METHODOLOGY_POLICY_VERSION = "2026-08-12.v2"