eduevidence 6.0.0 → 6.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +105 -0
- package/README.md +93 -38
- package/README.zh-CN.md +26 -6
- package/SKILL.md +11 -2
- package/assets/readme/landing-tour.gif +0 -0
- package/assets/readme/studio-tour.gif +0 -0
- package/bin/eduevidence.js +2 -1
- package/docs/architecture.md +319 -43
- package/docs/demo-workplace-ai.md +1 -1
- package/docs/install-guide.md +1 -1
- package/docs/orchestration-role-model.md +1 -1
- package/docs/release-closeout/README.md +1 -1
- package/docs/sciverse-api.md +125 -0
- package/eduevidence_cli.py +10 -0
- package/engine/decision_policy.py +96 -0
- package/engine/evidence_graph.py +14 -10
- package/engine/gaps.py +42 -22
- package/engine/ids.py +2 -0
- package/engine/library.py +6 -2
- package/engine/living.py +34 -4
- package/engine/migration.py +88 -3
- package/engine/orchestration.py +5 -5
- package/engine/paths.py +2 -0
- package/engine/pilot.py +34 -32
- package/engine/taxonomy.py +211 -0
- package/engine/tribunal.py +43 -31
- package/engine/versions.py +1 -1
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1360 -146
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
- package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
- package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
- package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +447 -127
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +447 -127
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +447 -127
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +447 -127
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +447 -127
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
- package/examples/ai-coding-assistant-evidence/result.json +13 -9
- package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
- package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
- package/examples/spaced-retrieval-practice/applicability.json +14 -0
- package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
- package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
- package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
- package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
- package/examples/spaced-retrieval-practice/frame.json +58 -0
- package/examples/spaced-retrieval-practice/gate_report.json +101 -0
- package/examples/spaced-retrieval-practice/methodology.json +78 -0
- package/examples/spaced-retrieval-practice/report_spec.json +212 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/result.json +942 -0
- package/examples/spaced-retrieval-practice/result.zh.json +942 -0
- package/examples/spaced-retrieval-practice/skeptic.json +70 -0
- package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
- package/examples/spaced-retrieval-practice/verdict.json +93 -0
- package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -4
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
- package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
- package/examples/workplace-ai-assistant/final_verdict.json +78 -0
- package/examples/workplace-ai-assistant/gate_report.json +101 -0
- package/examples/workplace-ai-assistant/report_spec.json +209 -40
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +435 -105
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +435 -105
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +435 -105
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +435 -105
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +435 -105
- package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/result.json +82 -20
- package/examples/workplace-ai-assistant/result.zh.json +82 -20
- package/examples/workplace-ai-assistant/skeptic.json +72 -0
- package/examples/workplace-ai-assistant/verdict.json +36 -10
- package/integrations/agent_mcp.py +2 -2
- package/package.json +12 -3
- package/pyproject.toml +4 -3
- package/references/report-copy-style.md +67 -0
- package/references/retrieval-compliance.md +75 -0
- package/references/retrieval-protocol.md +20 -0
- package/retrieval/audit.py +27 -3
- package/retrieval/fetch.py +96 -0
- package/retrieval/sciverse.py +398 -0
- package/retrieval/search.py +47 -7
- package/schemas/applicability.schema.json +94 -0
- package/schemas/chart-spec.schema.json +10 -3
- package/schemas/evidence.schema.json +316 -43
- package/schemas/fetch-result.schema.json +2 -1
- package/schemas/report-result.schema.json +3 -3
- package/schemas/report-spec.schema.json +98 -100
- package/schemas/skeptic.schema.json +86 -0
- package/schemas/source.schema.json +21 -2
- package/schemas/v2/finding.schema.json +5 -1
- package/schemas/v2/methodology-audit.schema.json +5 -1
- package/schemas/v2/outcome.schema.json +28 -5
- package/schemas/v2/study.schema.json +5 -1
- package/schemas/vNext/autoevolve-session.schema.json +34 -1
- package/schemas/vNext/eval-snapshot.schema.json +77 -1
- package/schemas/vNext/execution-plan.schema.json +50 -1
- package/schemas/vNext/gap-priority.schema.json +54 -1
- package/schemas/vNext/negative-search-record.schema.json +68 -1
- package/schemas/vNext/research-iteration.schema.json +87 -1
- package/schemas/vNext/research-strategy.schema.json +62 -1
- package/schemas/vNext/skill-experiment.schema.json +90 -1
- package/schemas/vNext/task-spec.schema.json +156 -1
- package/schemas/vNext/worker-result.schema.json +60 -1
- package/schemas/verdict.schema.json +164 -28
- package/scripts/build_esl_artifacts.py +2 -2
- package/scripts/build_report_variants.py +18 -2
- package/scripts/build_result.py +74 -9
- package/scripts/check_package_parity.py +85 -0
- package/scripts/check_protocol_alignment.py +375 -0
- package/scripts/check_versioned_schemas.py +254 -0
- package/scripts/claim_audit.py +13 -8
- package/scripts/compute_confidence.py +10 -0
- package/scripts/did_regression.py +12 -2
- package/scripts/evidence_score.py +5 -2
- package/scripts/generate_new_projects.py +4 -4
- package/scripts/orchestrator.py +120 -24
- package/scripts/pre_verdict_gate.py +224 -26
- package/scripts/quickstart.py +18 -2
- package/scripts/run_workspace.py +7 -1
- package/scripts/skill_payload.py +4 -1
- package/scripts/test_adversarial_empirical.py +26 -19
- package/scripts/validate_schema.py +31 -1
- package/skill/agents/evaluation-designer.md +20 -4
- package/skill/agents/evidence-analyst.md +19 -3
- package/skill/agents/evidence-judge.md +50 -2
- package/skill/agents/evidence-retriever.md +20 -3
- package/skill/agents/intervention-designer.md +20 -4
- package/skill/agents/method-reviewer.md +18 -2
- package/skill/agents/{education-planner.md → research-planner.md} +19 -3
- package/skill/agents/skeptic.md +18 -2
- package/skill/roles/registry.yaml +11 -11
- package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
- package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
- package/skill/sub-skills/data-analysis/SKILL.md +34 -15
- package/skill/sub-skills/ethics-review/SKILL.md +33 -10
- package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
- package/skill/sub-skills/evidence-review/SKILL.md +31 -12
- package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
- package/skill/sub-skills/literature-review/SKILL.md +35 -14
- package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
- package/skill/sub-skills/report-generation/SKILL.md +28 -0
- package/skill/sub-skills/research-planning/SKILL.md +41 -14
- package/skill/sub-skills/study-design/SKILL.md +30 -9
- package/skill/task-briefs/adjudicate.md +32 -7
- package/skill/task-briefs/applicability.md +37 -2
- package/skill/task-briefs/audit.md +32 -7
- package/skill/task-briefs/challenge.md +34 -5
- package/skill/task-briefs/evaluate.md +30 -5
- package/skill/task-briefs/extract.md +31 -8
- package/skill/task-briefs/frame.md +39 -10
- package/skill/task-briefs/intervene.md +32 -6
- package/skill/task-briefs/present.md +32 -8
- package/skill/task-briefs/projection.md +36 -2
- package/skill/task-briefs/retrieve.md +36 -6
- package/skill/workflows/decision-and-pilot.md +76 -1
- package/skill/workflows/evaluate-and-update.md +83 -0
- package/skill/workflows/evidence-review.md +104 -0
- package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
- package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
- package/visualization/eduevidence-report/scripts/build_report.py +512 -65
- package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
- package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
- package/web/architecture.html +14885 -0
- package/web/studio/assets/index-B8tkF44Q.css +1 -0
- package/web/studio/index.html +2 -2
- package/web/studio/assets/index-CzXocaGv.css +0 -1
- /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
package/engine/orchestration.py
CHANGED
|
@@ -75,8 +75,8 @@ class RoleSpec:
|
|
|
75
75
|
|
|
76
76
|
|
|
77
77
|
ROLE_REGISTRY: dict[str, RoleSpec] = {
|
|
78
|
-
"
|
|
79
|
-
"
|
|
78
|
+
"research-planner": RoleSpec(
|
|
79
|
+
"research-planner",
|
|
80
80
|
"Own framing completeness, scope, comparison and outcome definition.",
|
|
81
81
|
("frame",),
|
|
82
82
|
("research-planning",),
|
|
@@ -409,7 +409,7 @@ class ExecutionPlanner:
|
|
|
409
409
|
|
|
410
410
|
def _serial_tasks(self, run_id, base_revision) -> tuple[TaskSpec, ...]:
|
|
411
411
|
return (
|
|
412
|
-
self._base("frame", "frame", "
|
|
412
|
+
self._base("frame", "frame", "research-planner", "Structure the research question.", "frame", run_id=run_id, base_revision=base_revision),
|
|
413
413
|
self._base("retrieve", "retrieve", "evidence-retriever", "Acquire bounded evidence.", "direct+counter", run_id=run_id, base_revision=base_revision),
|
|
414
414
|
self._base("extract", "extract", "evidence-analyst", "Extract structured findings.", "all-eligible", run_id=run_id, base_revision=base_revision),
|
|
415
415
|
self._base("challenge", "challenge", "skeptic", "Challenge the provisional interpretation.", "counter-evidence", run_id=run_id, base_revision=base_revision),
|
|
@@ -419,7 +419,7 @@ class ExecutionPlanner:
|
|
|
419
419
|
|
|
420
420
|
def _medium_tasks(self, run_id, base_revision) -> tuple[TaskSpec, ...]:
|
|
421
421
|
return (
|
|
422
|
-
self._base("frame", "frame", "
|
|
422
|
+
self._base("frame", "frame", "research-planner", "Structure the research question.", "frame", run_id=run_id, base_revision=base_revision),
|
|
423
423
|
self._base("retrieve-direct", "retrieve", "evidence-retriever", "Retrieve direct decision-relevant evidence.", "direct-causal", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
|
|
424
424
|
self._base("retrieve-counter", "retrieve", "evidence-retriever", "Retrieve null, negative and contradictory evidence.", "counter-risk", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
|
|
425
425
|
self._base("extract", "extract", "evidence-analyst", "Merge validated sources and extract findings.", "all-eligible", run_id=run_id, base_revision=base_revision),
|
|
@@ -430,7 +430,7 @@ class ExecutionPlanner:
|
|
|
430
430
|
|
|
431
431
|
def _deep_tasks(self, run_id, base_revision) -> tuple[TaskSpec, ...]:
|
|
432
432
|
return (
|
|
433
|
-
self._base("frame", "frame", "
|
|
433
|
+
self._base("frame", "frame", "research-planner", "Structure the research question.", "frame", run_id=run_id, base_revision=base_revision),
|
|
434
434
|
self._base("retrieve-direct", "retrieve", "evidence-retriever", "Retrieve direct causal evidence.", "direct-causal", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
|
|
435
435
|
self._base("retrieve-transfer", "retrieve", "evidence-retriever", "Retrieve retention and independent-transfer evidence.", "transfer-retention", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
|
|
436
436
|
self._base("retrieve-counter", "retrieve", "evidence-retriever", "Retrieve null, negative, risk and contradiction evidence.", "counter-risk", run_id=run_id, base_revision=base_revision, delegated=True, outputs=("SourceCandidates",)),
|
package/engine/paths.py
CHANGED
package/engine/pilot.py
CHANGED
|
@@ -22,6 +22,7 @@ from engine.datasets import analysis_blocked_by_privacy, derive_csv_profile, ing
|
|
|
22
22
|
from engine.graph_store import GraphMutation, GraphStore
|
|
23
23
|
from engine.ids import new_local_id, new_run_id
|
|
24
24
|
from engine.project import ProjectWorkspace
|
|
25
|
+
from engine.taxonomy import category_of, tokens as taxonomy_tokens
|
|
25
26
|
from engine.synthesis import synthesize_project
|
|
26
27
|
from engine.tribunal import adjudicate, decision_diff, save_decision_snapshot
|
|
27
28
|
from engine.versions import (
|
|
@@ -30,15 +31,21 @@ from engine.versions import (
|
|
|
30
31
|
SOURCE_VALIDATION_POLICY_VERSION,
|
|
31
32
|
)
|
|
32
33
|
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
"
|
|
40
|
-
|
|
41
|
-
|
|
34
|
+
def outcome_taxonomy_tokens(domain: str = "education") -> set[str]:
|
|
35
|
+
"""Outcome tokens a pilot may measure, read from the domain registry.
|
|
36
|
+
|
|
37
|
+
This was a module-level set of the 20 education tokens, so a policy pilot
|
|
38
|
+
could not register at all. The domain registry (via engine/taxonomy.py) is
|
|
39
|
+
now the single authority.
|
|
40
|
+
"""
|
|
41
|
+
return set(taxonomy_tokens(domain))
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def project_domain(project: ProjectWorkspace) -> str:
|
|
45
|
+
"""The domain a project registers; defaults to education when absent."""
|
|
46
|
+
manifest = project.manifest()
|
|
47
|
+
return str(manifest.get("domain") or "education")
|
|
48
|
+
|
|
42
49
|
|
|
43
50
|
PILOT_STATUSES = ("registered", "data_imported", "analyzed", "adjudicated")
|
|
44
51
|
|
|
@@ -49,22 +56,6 @@ PII_COLUMN_HINTS = ("name", "student", "学号", "姓名", "email", "mail",
|
|
|
49
56
|
_DECISION_IMPLICATION = {"support": "support_adoption",
|
|
50
57
|
"contradict": "oppose_adoption", "neutral": "neutral"}
|
|
51
58
|
|
|
52
|
-
#: Outcome Taxonomy token -> graph outcome category enum (schemas/v2/outcome).
|
|
53
|
-
_OUTCOME_CATEGORY = {
|
|
54
|
-
"knowledge_gain": "learning", "concept_understanding": "learning",
|
|
55
|
-
"retention": "learning", "transfer": "learning",
|
|
56
|
-
"independent_problem_solving": "learning",
|
|
57
|
-
"completion_time": "task_performance", "accuracy": "task_performance",
|
|
58
|
-
"code_quality": "task_performance", "assignment_score": "task_performance",
|
|
59
|
-
"engagement": "process", "motivation": "process",
|
|
60
|
-
"cognitive_load": "process", "help_seeking": "process",
|
|
61
|
-
"metacognition": "process",
|
|
62
|
-
"ai_dependency": "risk", "over_reliance": "risk",
|
|
63
|
-
"reduced_effort": "risk", "reduced_transfer": "risk",
|
|
64
|
-
"academic_integrity_risk": "risk", "false_confidence": "risk",
|
|
65
|
-
}
|
|
66
|
-
|
|
67
|
-
|
|
68
59
|
def _now_iso() -> str:
|
|
69
60
|
return datetime.now(timezone.utc).isoformat()
|
|
70
61
|
|
|
@@ -111,10 +102,13 @@ def register_pilot(project: ProjectWorkspace, *,
|
|
|
111
102
|
raise ValueError(
|
|
112
103
|
f"decision snapshot {decision_snapshot_id} not found in this project; "
|
|
113
104
|
"a pilot must bind to a real adjudication")
|
|
114
|
-
|
|
105
|
+
domain = project_domain(project)
|
|
106
|
+
known_tokens = outcome_taxonomy_tokens(domain)
|
|
107
|
+
unknown = [o for o in outcome_columns if o not in known_tokens]
|
|
115
108
|
if unknown:
|
|
116
109
|
raise ValueError(
|
|
117
|
-
f"outcome_columns outside Outcome Taxonomy:
|
|
110
|
+
f"outcome_columns outside the {domain} Outcome Taxonomy: "
|
|
111
|
+
f"{sorted(unknown)}")
|
|
118
112
|
if not conditions or sample_size < 1:
|
|
119
113
|
raise ValueError("conditions must be non-empty and sample_size >= 1")
|
|
120
114
|
if anon_policy.get("no_pii_columns") is not True:
|
|
@@ -208,7 +202,13 @@ def link_analysis(project: ProjectWorkspace, pilot_id: str, *,
|
|
|
208
202
|
return pilot
|
|
209
203
|
|
|
210
204
|
|
|
211
|
-
def _ensure_outcome(store: GraphStore, outcome_id: str) -> dict:
|
|
205
|
+
def _ensure_outcome(store: GraphStore, outcome_id: str, domain: str) -> dict:
|
|
206
|
+
"""Create the pilot outcome row, classified by the DOMAIN registry.
|
|
207
|
+
|
|
208
|
+
``category_of`` raises for an unregistered token; the caller has already
|
|
209
|
+
validated the token against ``outcome_taxonomy_tokens(domain)``, so an
|
|
210
|
+
error here means the two disagreed and must not be silently absorbed.
|
|
211
|
+
"""
|
|
212
212
|
existing = store.get("outcomes", outcome_id)
|
|
213
213
|
if existing:
|
|
214
214
|
return existing
|
|
@@ -216,7 +216,7 @@ def _ensure_outcome(store: GraphStore, outcome_id: str) -> dict:
|
|
|
216
216
|
return {
|
|
217
217
|
"outcome_id": outcome_id,
|
|
218
218
|
"name": token,
|
|
219
|
-
"outcome_type":
|
|
219
|
+
"outcome_type": category_of(domain, token),
|
|
220
220
|
"extensions": {"pilot_outcome": True},
|
|
221
221
|
}
|
|
222
222
|
|
|
@@ -237,9 +237,11 @@ def redecide(project: ProjectWorkspace, pilot_id: str, *,
|
|
|
237
237
|
raise ValueError(f"invalid effect_direction {effect_direction!r}")
|
|
238
238
|
if relation_to_claim not in ("support", "contradict", "neutral"):
|
|
239
239
|
raise ValueError(f"invalid relation_to_claim {relation_to_claim!r}")
|
|
240
|
-
|
|
240
|
+
domain = project_domain(project)
|
|
241
|
+
if outcome_token not in outcome_taxonomy_tokens(domain):
|
|
241
242
|
raise ValueError(
|
|
242
|
-
f"outcome_token {outcome_token!r} outside
|
|
243
|
+
f"outcome_token {outcome_token!r} outside the {domain} "
|
|
244
|
+
"Outcome Taxonomy")
|
|
243
245
|
outcome_id = f"OUT-{outcome_token}"
|
|
244
246
|
pilot = _load_pilot(project, pilot_id)
|
|
245
247
|
if pilot["status"] not in ("analyzed", "data_imported"):
|
|
@@ -282,7 +284,7 @@ def redecide(project: ProjectWorkspace, pilot_id: str, *,
|
|
|
282
284
|
"identity_status": "resolved",
|
|
283
285
|
"extensions": {"pilot_id": pilot_id},
|
|
284
286
|
}
|
|
285
|
-
outcome = _ensure_outcome(store, outcome_id)
|
|
287
|
+
outcome = _ensure_outcome(store, outcome_id, domain)
|
|
286
288
|
estimate = None
|
|
287
289
|
if effect_estimate is not None:
|
|
288
290
|
estimate = {
|
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
"""engine/taxonomy.py - Outcome taxonomy authority (domain registry backed).
|
|
2
|
+
|
|
3
|
+
Every domain registers its own outcome taxonomy in
|
|
4
|
+
``domains/<id>/outcome_taxonomy.json``. Each file declares the domain category
|
|
5
|
+
buckets (``categories``) and one entry per outcome token carrying an explicit
|
|
6
|
+
``category`` (``tokens[].category``).
|
|
7
|
+
|
|
8
|
+
This module is the ONLY reader of that contract. Before it existed, the token
|
|
9
|
+
set and the token-to-category mapping were hard-coded in ``engine/pilot.py``
|
|
10
|
+
with a ``.get(token, "learning")`` fallback, so a policy outcome was silently
|
|
11
|
+
classified as a learning outcome and policy runs could not complete. Callers
|
|
12
|
+
must no longer keep private copies of either table.
|
|
13
|
+
|
|
14
|
+
Fail-closed rule: an unknown token or an unregistered domain raises. Silent
|
|
15
|
+
classification is what produced the original defect; unknown input must be
|
|
16
|
+
visible as an error, never absorbed into a default bucket.
|
|
17
|
+
|
|
18
|
+
Stdlib only; results are cached per process.
|
|
19
|
+
"""
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import json
|
|
23
|
+
from typing import Any, Iterable
|
|
24
|
+
|
|
25
|
+
from engine.evidencecore import list_domains, load_domain
|
|
26
|
+
|
|
27
|
+
#: Bucket used when a caller must render an outcome whose category could not be
|
|
28
|
+
#: resolved. It is deliberately not a domain category: it marks the value as
|
|
29
|
+
#: unclassified instead of pretending it belongs to a real bucket.
|
|
30
|
+
UNCLASSIFIED = "unclassified"
|
|
31
|
+
|
|
32
|
+
_cache: dict[str, Any] = {}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class TaxonomyError(ValueError):
|
|
36
|
+
"""Raised when taxonomy data is missing, malformed, or unknown."""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _taxonomy_path(domain_id: str) -> str:
|
|
40
|
+
"""Registered relative path of a domain taxonomy file."""
|
|
41
|
+
entry = load_domain(domain_id)
|
|
42
|
+
raw = entry.get("outcome_taxonomy")
|
|
43
|
+
if not raw:
|
|
44
|
+
raise TaxonomyError(
|
|
45
|
+
"domain " + repr(domain_id) + " registers no outcome_taxonomy")
|
|
46
|
+
return str(raw).partition("#")[0]
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _load(domain_id: str) -> dict:
|
|
50
|
+
"""Load (and cache) one domain taxonomy, validating its shape."""
|
|
51
|
+
key = "taxonomy:" + domain_id
|
|
52
|
+
if key in _cache:
|
|
53
|
+
return _cache[key]
|
|
54
|
+
from engine._resources import resource_root
|
|
55
|
+
|
|
56
|
+
path = resource_root() / _taxonomy_path(domain_id)
|
|
57
|
+
if not path.is_file():
|
|
58
|
+
raise TaxonomyError(
|
|
59
|
+
"domain " + repr(domain_id) + ": taxonomy file missing: " + str(path))
|
|
60
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
61
|
+
categories = data.get("categories")
|
|
62
|
+
tokens = data.get("tokens")
|
|
63
|
+
if not isinstance(categories, dict) or not categories:
|
|
64
|
+
raise TaxonomyError(
|
|
65
|
+
"domain " + repr(domain_id) + ": taxonomy declares no categories")
|
|
66
|
+
if not isinstance(tokens, list) or not tokens:
|
|
67
|
+
raise TaxonomyError(
|
|
68
|
+
"domain " + repr(domain_id) + ": taxonomy declares no tokens")
|
|
69
|
+
seen: set[str] = set()
|
|
70
|
+
for item in tokens:
|
|
71
|
+
if not isinstance(item, dict) or not item.get("id"):
|
|
72
|
+
raise TaxonomyError(
|
|
73
|
+
"domain " + repr(domain_id) + ": token entry without an id")
|
|
74
|
+
token = str(item["id"])
|
|
75
|
+
if token in seen:
|
|
76
|
+
raise TaxonomyError(
|
|
77
|
+
"domain " + repr(domain_id) + ": duplicate token " + repr(token))
|
|
78
|
+
seen.add(token)
|
|
79
|
+
category = item.get("category")
|
|
80
|
+
if not category:
|
|
81
|
+
raise TaxonomyError(
|
|
82
|
+
"domain " + repr(domain_id) + ": token " + repr(token)
|
|
83
|
+
+ " declares no category (silent defaults are forbidden)")
|
|
84
|
+
if str(category) not in categories:
|
|
85
|
+
raise TaxonomyError(
|
|
86
|
+
"domain " + repr(domain_id) + ": token " + repr(token)
|
|
87
|
+
+ " uses category " + repr(category) + " absent from "
|
|
88
|
+
+ repr(sorted(categories)))
|
|
89
|
+
_cache[key] = data
|
|
90
|
+
return data
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _domain_ids(domain_id: str | None = None) -> tuple[str, ...]:
|
|
94
|
+
if domain_id:
|
|
95
|
+
return (domain_id,)
|
|
96
|
+
return tuple(d["id"] for d in list_domains())
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def tokens(domain_id: str) -> tuple[str, ...]:
|
|
100
|
+
"""Outcome tokens declared by one domain, in registry order."""
|
|
101
|
+
return tuple(str(item["id"]) for item in _load(domain_id)["tokens"])
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def categories(domain_id: str) -> dict[str, dict]:
|
|
105
|
+
"""Category buckets declared by one domain (id -> descriptor)."""
|
|
106
|
+
return dict(_load(domain_id)["categories"])
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def category_of(domain_id: str, token: str) -> str:
|
|
110
|
+
"""Category bucket for a token in a domain.
|
|
111
|
+
|
|
112
|
+
Raises TaxonomyError for an unknown token: a caller that cannot classify a
|
|
113
|
+
value must surface that, not guess. Use category_of_or_unclassified at
|
|
114
|
+
rendering boundaries where an unclassified value is acceptable.
|
|
115
|
+
"""
|
|
116
|
+
for item in _load(domain_id)["tokens"]:
|
|
117
|
+
if str(item["id"]) == token:
|
|
118
|
+
return str(item["category"])
|
|
119
|
+
raise TaxonomyError(
|
|
120
|
+
"domain " + repr(domain_id) + ": unknown outcome token " + repr(token)
|
|
121
|
+
+ " (" + str(len(tokens(domain_id))) + " tokens known)")
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def category_of_or_unclassified(domain_id: str, token: str) -> str:
|
|
125
|
+
"""Rendering-safe variant: unknown tokens map to UNCLASSIFIED."""
|
|
126
|
+
try:
|
|
127
|
+
return category_of(domain_id, token)
|
|
128
|
+
except TaxonomyError:
|
|
129
|
+
return UNCLASSIFIED
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def category_labels(domain_id: str, lang: str = "zh") -> dict[str, str]:
|
|
133
|
+
"""Display label per category bucket (falls back to the category id)."""
|
|
134
|
+
suffix = "_en" if lang == "en" else "_zh"
|
|
135
|
+
out: dict[str, str] = {}
|
|
136
|
+
for key, descriptor in categories(domain_id).items():
|
|
137
|
+
if isinstance(descriptor, dict):
|
|
138
|
+
label = descriptor.get("name" + suffix) or descriptor.get("name")
|
|
139
|
+
out[key] = str(label or key)
|
|
140
|
+
else:
|
|
141
|
+
out[key] = key
|
|
142
|
+
return out
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def all_tokens() -> dict[str, str]:
|
|
146
|
+
"""Every registered token -> owning domain, across registered domains.
|
|
147
|
+
|
|
148
|
+
A token registered by two domains is a contract conflict and raises: the
|
|
149
|
+
token would otherwise mean different things depending on lookup order.
|
|
150
|
+
"""
|
|
151
|
+
key = "all_tokens"
|
|
152
|
+
if key in _cache:
|
|
153
|
+
return _cache[key]
|
|
154
|
+
out: dict[str, str] = {}
|
|
155
|
+
for domain_id in _domain_ids():
|
|
156
|
+
for token in tokens(domain_id):
|
|
157
|
+
owner = out.get(token)
|
|
158
|
+
if owner is not None and owner != domain_id:
|
|
159
|
+
raise TaxonomyError(
|
|
160
|
+
"outcome token " + repr(token) + " is registered by both "
|
|
161
|
+
+ repr(owner) + " and " + repr(domain_id))
|
|
162
|
+
out[token] = domain_id
|
|
163
|
+
_cache[key] = out
|
|
164
|
+
return out
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def domain_of(token: str, default: str = "education") -> str:
|
|
168
|
+
"""Owning domain for a token; default when the token is unregistered."""
|
|
169
|
+
return all_tokens().get(token, default)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def all_tokens_ordered() -> tuple[str, ...]:
|
|
173
|
+
"""Every registered token in domain-registry then taxonomy order."""
|
|
174
|
+
ordered: list[str] = []
|
|
175
|
+
for domain_id in _domain_ids():
|
|
176
|
+
ordered.extend(tokens(domain_id))
|
|
177
|
+
return tuple(ordered)
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def all_categories() -> tuple[str, ...]:
|
|
181
|
+
"""Every category bucket across domains, de-duplicated, order preserved."""
|
|
182
|
+
ordered: list[str] = []
|
|
183
|
+
for domain_id in _domain_ids():
|
|
184
|
+
for name in categories(domain_id):
|
|
185
|
+
if name not in ordered:
|
|
186
|
+
ordered.append(name)
|
|
187
|
+
return tuple(ordered)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def categories_for_tokens(
|
|
191
|
+
token_list: Iterable[str], domain_id: str = "education"
|
|
192
|
+
) -> dict[str, list[str]]:
|
|
193
|
+
"""Group tokens by category; unregistered tokens land in UNCLASSIFIED."""
|
|
194
|
+
grouped: dict[str, list[str]] = {}
|
|
195
|
+
for token in token_list:
|
|
196
|
+
bucket = category_of_or_unclassified(domain_id, token)
|
|
197
|
+
grouped.setdefault(bucket, []).append(token)
|
|
198
|
+
return grouped
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def reset_cache() -> None:
|
|
202
|
+
"""Drop memoised taxonomies (tests and long-lived processes)."""
|
|
203
|
+
_cache.clear()
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
__all__ = [
|
|
207
|
+
"UNCLASSIFIED", "TaxonomyError",
|
|
208
|
+
"tokens", "categories", "category_of", "category_of_or_unclassified",
|
|
209
|
+
"category_labels", "all_tokens", "all_tokens_ordered", "all_categories",
|
|
210
|
+
"categories_for_tokens", "domain_of", "reset_cache",
|
|
211
|
+
]
|
package/engine/tribunal.py
CHANGED
|
@@ -38,6 +38,12 @@ from datetime import datetime, timezone
|
|
|
38
38
|
from pathlib import Path
|
|
39
39
|
|
|
40
40
|
from engine.contracts import validate_record
|
|
41
|
+
from engine.decision_policy import (
|
|
42
|
+
ADOPT_DIRECTNESS,
|
|
43
|
+
decision_action as _policy_decision_action,
|
|
44
|
+
outcome_category,
|
|
45
|
+
primary_effect_categories,
|
|
46
|
+
)
|
|
41
47
|
from engine.graph_store import GraphStore
|
|
42
48
|
from engine.ids import new_local_id
|
|
43
49
|
from engine.project import ProjectWorkspace
|
|
@@ -230,16 +236,18 @@ def _confidence(store: GraphStore, syntheses: tuple[ClaimSynthesis, ...]) -> dic
|
|
|
230
236
|
"decisive_relations": dict(decisive)}
|
|
231
237
|
|
|
232
238
|
|
|
233
|
-
def
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
239
|
+
def _has_direct_primary_evidence(store: GraphStore,
|
|
240
|
+
decisive_relations: dict[str, str],
|
|
241
|
+
domain: str = "education") -> bool:
|
|
242
|
+
"""True when a decisive support_adoption Study measures a primary outcome.
|
|
237
243
|
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
244
|
+
Primary means the finding's outcome_type maps — through the domain
|
|
245
|
+
registry — into one of :func:`primary_effect_categories` (education:
|
|
246
|
+
learning; policy: effectiveness/cost), AND its evidence link carries
|
|
247
|
+
directness == 2. Task-performance, process and risk outcomes never
|
|
248
|
+
qualify, and a missing outcome record or missing directness fails closed.
|
|
242
249
|
"""
|
|
250
|
+
primary = primary_effect_categories(domain)
|
|
243
251
|
findings = {f["finding_id"]: f for f in store.read_table("findings")}
|
|
244
252
|
outcomes = {o["outcome_id"]: o for o in store.read_table("outcomes")}
|
|
245
253
|
links_by_finding: dict[str, list[dict]] = {}
|
|
@@ -254,37 +262,37 @@ def _has_direct_learning_evidence(store: GraphStore,
|
|
|
254
262
|
if fnd.get("study_id") not in support_studies:
|
|
255
263
|
continue
|
|
256
264
|
outcome = outcomes.get(fnd.get("outcome_id"))
|
|
257
|
-
if outcome is None
|
|
265
|
+
if outcome is None:
|
|
266
|
+
continue
|
|
267
|
+
value = str(outcome.get("outcome_type") or "")
|
|
268
|
+
category = outcome_category(domain, value, primary)
|
|
269
|
+
# Unknown value: fail closed rather than treating it as
|
|
270
|
+
# decision-grade evidence.
|
|
271
|
+
if category is None:
|
|
272
|
+
continue
|
|
273
|
+
if category not in primary:
|
|
258
274
|
continue
|
|
259
275
|
for link in links_by_finding.get(fid, []):
|
|
260
276
|
directness = link.get("directness")
|
|
261
|
-
if isinstance(directness, (int, float))
|
|
277
|
+
if (isinstance(directness, (int, float))
|
|
278
|
+
and int(directness) == ADOPT_DIRECTNESS):
|
|
262
279
|
return True
|
|
263
280
|
return False
|
|
264
281
|
|
|
265
282
|
|
|
266
283
|
def _decision_action(syn_statuses: dict[str, str], confidence: dict,
|
|
267
284
|
decisive_relations: dict[str, str],
|
|
268
|
-
|
|
269
|
-
"""Gate-enforced decision action.
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
ADOPT additionally requires direct learning/transfer evidence: High +
|
|
274
|
-
decisive support WITHOUT a direct learning outcome downgrades to PILOT
|
|
275
|
-
(task performance / procedural efficiency is not learning). Moderate +
|
|
276
|
-
decisive support → PILOT; otherwise INSUFFICIENT_EVIDENCE.
|
|
285
|
+
has_direct_primary_evidence: bool = False) -> str:
|
|
286
|
+
"""Gate-enforced decision action; rule lives in engine.decision_policy.
|
|
287
|
+
|
|
288
|
+
The V1 Pre-Verdict Gate enforces the same rule, so the thresholds are
|
|
289
|
+
imported rather than restated here.
|
|
277
290
|
"""
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
if label == "High" and has_support and has_direct_learning_evidence:
|
|
284
|
-
return "ADOPT"
|
|
285
|
-
if label in ("High", "Moderate") and has_support:
|
|
286
|
-
return "PILOT"
|
|
287
|
-
return "INSUFFICIENT_EVIDENCE"
|
|
291
|
+
return _policy_decision_action(
|
|
292
|
+
confidence_label=confidence["label"],
|
|
293
|
+
decisive_relations=decisive_relations,
|
|
294
|
+
has_direct_primary_evidence=has_direct_primary_evidence,
|
|
295
|
+
)
|
|
288
296
|
|
|
289
297
|
|
|
290
298
|
|
|
@@ -299,10 +307,11 @@ def adjudicate(store: GraphStore, *, project: ProjectWorkspace,
|
|
|
299
307
|
|
|
300
308
|
syn_statuses = {s.claim_id: s.status for s in syntheses}
|
|
301
309
|
decisive_relations = confidence.get("decisive_relations", {})
|
|
302
|
-
|
|
310
|
+
domain = str(project.manifest().get("domain") or "education")
|
|
311
|
+
direct_learning = _has_direct_primary_evidence(store, decisive_relations, domain)
|
|
303
312
|
|
|
304
313
|
decision = _decision_action(syn_statuses, confidence, decisive_relations,
|
|
305
|
-
|
|
314
|
+
has_direct_primary_evidence=direct_learning)
|
|
306
315
|
|
|
307
316
|
key_links: list[str] = []
|
|
308
317
|
for syn in syntheses:
|
|
@@ -351,6 +360,9 @@ def adjudicate(store: GraphStore, *, project: ProjectWorkspace,
|
|
|
351
360
|
"extensions": {"confidence_components": {
|
|
352
361
|
"decisive_studies": confidence.get("decisive_studies", 0),
|
|
353
362
|
"usable_studies": confidence.get("usable_studies", 0),
|
|
363
|
+
"has_direct_primary_evidence": direct_learning,
|
|
364
|
+
# Backward-compatible alias under the education-era name,
|
|
365
|
+
# so readers written against the older contract keep working.
|
|
354
366
|
"has_direct_learning_evidence": direct_learning,
|
|
355
367
|
}},
|
|
356
368
|
}
|
package/engine/versions.py
CHANGED
|
@@ -4,7 +4,7 @@ Policy versions are frozen identifiers, not free-form strings: changing a
|
|
|
4
4
|
policy requires a new version, never silent mutation of an existing one.
|
|
5
5
|
"""
|
|
6
6
|
|
|
7
|
-
ENGINE_VERSION = "6.
|
|
7
|
+
ENGINE_VERSION = "6.2.0"
|
|
8
8
|
GRAPH_SCHEMA_VERSION = "2.0"
|
|
9
9
|
SOURCE_VALIDATION_POLICY_VERSION = "2026-08-12.v2"
|
|
10
10
|
METHODOLOGY_POLICY_VERSION = "2026-08-12.v2"
|