eduevidence 6.0.0 → 6.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +395 -0
- package/CONTRIBUTING.md +105 -0
- package/README.md +113 -49
- package/README.zh-CN.md +39 -12
- package/SKILL.md +15 -5
- package/assets/readme/landing-tour.gif +0 -0
- package/assets/readme/studio-tour.gif +0 -0
- package/benchmarks/evidence-library.json +277 -1
- package/bin/eduevidence.js +2 -1
- package/docs/architecture.md +325 -46
- package/docs/demo-workplace-ai.md +1 -1
- package/docs/install-guide.md +1 -1
- package/docs/j-ev-experimental.md +250 -0
- package/docs/orchestration-role-model.md +1 -1
- package/docs/release-closeout/README.md +1 -1
- package/docs/reproducibility.md +138 -0
- package/docs/sciverse-api.md +125 -0
- package/domains/_neutral/copy/few_shots.json +21 -0
- package/domains/_neutral/copy/framing_lexicon.json +19 -0
- package/domains/_neutral/copy/module_labels.json +5 -0
- package/domains/_neutral/copy/module_labels_footer.json +102 -0
- package/domains/_neutral/copy/module_labels_modules.json +204 -0
- package/domains/_neutral/copy/module_labels_nav.json +126 -0
- package/domains/_neutral/copy/module_labels_summary.json +98 -0
- package/domains/_neutral/copy/module_labels_tables.json +164 -0
- package/domains/_neutral/copy/module_labels_v2.json +90 -0
- package/domains/_neutral/copy/risk_constructs.json +20 -0
- package/domains/_neutral/copy/section_titles.json +66 -0
- package/domains/_neutral/copy/terminology.json +11 -0
- package/domains/check_copy_packs.py +103 -0
- package/domains/education/copy/few_shots.json +22 -0
- package/domains/education/copy/framing_enums.json +167 -0
- package/domains/education/copy/framing_lexicon.json +166 -0
- package/domains/education/copy/module_labels.json +169 -0
- package/domains/education/copy/risk_constructs.json +48 -0
- package/domains/education/copy/section_titles.json +186 -0
- package/domains/education/copy/terminology.json +70 -0
- package/domains/education/manifest.json +1 -1
- package/domains/education/outcome_taxonomy.json +2 -2
- package/domains/manifest.json +1 -1
- package/domains/policy/copy/few_shots.json +22 -0
- package/domains/policy/copy/framing_enums.json +94 -0
- package/domains/policy/copy/framing_lexicon.json +174 -0
- package/domains/policy/copy/module_labels.json +168 -0
- package/domains/policy/copy/risk_constructs.json +33 -0
- package/domains/policy/copy/section_titles.json +186 -0
- package/domains/policy/copy/terminology.json +64 -0
- package/eduevidence_cli.py +10 -0
- package/engine/capabilities.py +57 -5
- package/engine/decision_policy.py +167 -0
- package/engine/evidence_graph.py +14 -10
- package/engine/gaps.py +42 -22
- package/engine/ids.py +2 -0
- package/engine/library.py +6 -2
- package/engine/library_builtin.py +7 -4
- package/engine/living.py +34 -4
- package/engine/migration.py +88 -3
- package/engine/orchestration.py +5 -5
- package/engine/paths.py +2 -0
- package/engine/pilot.py +34 -32
- package/engine/taxonomy.py +211 -0
- package/engine/tribunal.py +49 -43
- package/engine/versions.py +1 -1
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1361 -147
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
- package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
- package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
- package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +448 -128
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +448 -128
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +448 -128
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +448 -128
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +448 -128
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
- package/examples/ai-coding-assistant-evidence/result.json +13 -9
- package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
- package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
- package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
- package/examples/spaced-retrieval-practice/applicability.json +14 -0
- package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
- package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
- package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
- package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
- package/examples/spaced-retrieval-practice/frame.json +58 -0
- package/examples/spaced-retrieval-practice/gate_report.json +101 -0
- package/examples/spaced-retrieval-practice/methodology.json +78 -0
- package/examples/spaced-retrieval-practice/report.html +2522 -0
- package/examples/spaced-retrieval-practice/report_spec.json +212 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/result.json +942 -0
- package/examples/spaced-retrieval-practice/result.zh.json +942 -0
- package/examples/spaced-retrieval-practice/skeptic.json +70 -0
- package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
- package/examples/spaced-retrieval-practice/verdict.json +93 -0
- package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
- package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -4
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
- package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
- package/examples/workplace-ai-assistant/final_verdict.json +78 -0
- package/examples/workplace-ai-assistant/gate_report.json +101 -0
- package/examples/workplace-ai-assistant/report_spec.json +209 -40
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +449 -119
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +449 -119
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +449 -119
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +449 -119
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +449 -119
- package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/result.json +82 -20
- package/examples/workplace-ai-assistant/result.zh.json +82 -20
- package/examples/workplace-ai-assistant/skeptic.json +72 -0
- package/examples/workplace-ai-assistant/verdict.json +36 -10
- package/integrations/agent_mcp.py +2 -2
- package/integrations/jev/__init__.py +115 -0
- package/integrations/jev/approval.py +212 -0
- package/integrations/jev/cli.py +84 -0
- package/integrations/jev/config.py +112 -0
- package/integrations/jev/gateway.py +128 -0
- package/integrations/jev/modes.py +38 -0
- package/integrations/jev/tools_classify.py +88 -0
- package/integrations/jev/tools_extract.py +111 -0
- package/integrations/jev/tools_rerank.py +71 -0
- package/integrations/jev/tools_screen.py +87 -0
- package/integrations/jev/tools_verify.py +95 -0
- package/integrations/jev_mcp.py +22 -0
- package/integrations/semantic_decide.py +286 -0
- package/integrations/semdecide_cli.py +55 -0
- package/package.json +19 -2
- package/pyproject.toml +4 -3
- package/references/report-copy-style.md +107 -0
- package/references/retrieval-compliance.md +75 -0
- package/references/retrieval-protocol.md +20 -0
- package/retrieval/audit.py +27 -3
- package/retrieval/fetch.py +96 -0
- package/retrieval/sciverse.py +398 -0
- package/retrieval/search.py +47 -7
- package/schemas/applicability.schema.json +94 -0
- package/schemas/chart-spec.schema.json +10 -3
- package/schemas/evidence.schema.json +316 -43
- package/schemas/fetch-result.schema.json +2 -1
- package/schemas/report-result.schema.json +3 -3
- package/schemas/report-spec.schema.json +98 -100
- package/schemas/skeptic.schema.json +86 -0
- package/schemas/source.schema.json +21 -2
- package/schemas/v2/decision-snapshot.schema.json +20 -9
- package/schemas/v2/finding.schema.json +5 -1
- package/schemas/v2/intake.schema.json +191 -0
- package/schemas/v2/methodology-audit.schema.json +5 -1
- package/schemas/v2/outcome.schema.json +28 -5
- package/schemas/v2/study.schema.json +5 -1
- package/schemas/vNext/autoevolve-session.schema.json +34 -1
- package/schemas/vNext/eval-snapshot.schema.json +77 -1
- package/schemas/vNext/execution-plan.schema.json +50 -1
- package/schemas/vNext/gap-priority.schema.json +54 -1
- package/schemas/vNext/negative-search-record.schema.json +68 -1
- package/schemas/vNext/research-iteration.schema.json +87 -1
- package/schemas/vNext/research-strategy.schema.json +62 -1
- package/schemas/vNext/skill-experiment.schema.json +90 -1
- package/schemas/vNext/task-spec.schema.json +156 -1
- package/schemas/vNext/worker-result.schema.json +60 -1
- package/schemas/verdict.schema.json +164 -28
- package/scripts/build_evidence_library.py +15 -5
- package/scripts/build_report_variants.py +18 -2
- package/scripts/build_result.py +74 -9
- package/scripts/check_package_parity.py +85 -0
- package/scripts/check_protocol_alignment.py +375 -0
- package/scripts/check_versioned_schemas.py +254 -0
- package/scripts/claim_audit.py +13 -8
- package/scripts/compute_confidence.py +10 -0
- package/scripts/dashboard_server.py +13 -2
- package/scripts/did_regression.py +12 -2
- package/scripts/evidence_score.py +5 -2
- package/scripts/intake/__init__.py +31 -0
- package/scripts/intake/__main__.py +18 -0
- package/scripts/intake/background.py +78 -0
- package/scripts/intake/browser.py +79 -0
- package/scripts/intake/cli.py +57 -0
- package/scripts/intake/constants.py +57 -0
- package/scripts/intake/depth.py +53 -0
- package/scripts/intake/enhancements.py +106 -0
- package/scripts/intake/hooks.py +90 -0
- package/scripts/intake/prefs.py +76 -0
- package/scripts/intake/prompts.py +85 -0
- package/scripts/intake/session.py +152 -0
- package/scripts/lint_file_layers.py +126 -0
- package/scripts/orchestrator.py +187 -40
- package/scripts/pre_verdict_gate.py +241 -29
- package/scripts/quickstart.py +18 -2
- package/scripts/run_workspace.py +7 -1
- package/scripts/skill_lint.py +11 -1
- package/scripts/skill_payload.py +6 -3
- package/scripts/test_adversarial_empirical.py +96 -25
- package/scripts/validate_schema.py +31 -1
- package/skill/agents/evaluation-designer.md +20 -4
- package/skill/agents/evidence-analyst.md +19 -3
- package/skill/agents/evidence-judge.md +98 -8
- package/skill/agents/evidence-retriever.md +20 -3
- package/skill/agents/intervention-designer.md +20 -4
- package/skill/agents/method-reviewer.md +18 -2
- package/skill/agents/{education-planner.md → research-planner.md} +19 -3
- package/skill/agents/skeptic.md +18 -2
- package/skill/roles/registry.yaml +11 -11
- package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
- package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
- package/skill/sub-skills/data-analysis/SKILL.md +34 -15
- package/skill/sub-skills/ethics-review/SKILL.md +33 -10
- package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
- package/skill/sub-skills/evidence-review/SKILL.md +31 -12
- package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
- package/skill/sub-skills/literature-review/SKILL.md +35 -14
- package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
- package/skill/sub-skills/report-generation/SKILL.md +28 -0
- package/skill/sub-skills/research-planning/SKILL.md +41 -14
- package/skill/sub-skills/study-design/SKILL.md +30 -9
- package/skill/task-briefs/adjudicate.md +32 -7
- package/skill/task-briefs/applicability.md +37 -2
- package/skill/task-briefs/audit.md +32 -7
- package/skill/task-briefs/challenge.md +34 -5
- package/skill/task-briefs/evaluate.md +30 -5
- package/skill/task-briefs/extract.md +31 -8
- package/skill/task-briefs/frame.md +39 -10
- package/skill/task-briefs/intervene.md +32 -6
- package/skill/task-briefs/present.md +32 -8
- package/skill/task-briefs/projection.md +36 -2
- package/skill/task-briefs/retrieve.md +36 -6
- package/skill/workflows/decision-and-pilot.md +76 -1
- package/skill/workflows/evaluate-and-update.md +83 -0
- package/skill/workflows/evidence-review.md +104 -0
- package/skill/workflows/experimental-jev.md +170 -0
- package/skill/workflows/intake.md +120 -0
- package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
- package/visualization/eduevidence-report/scripts/build_infographics.py +37 -15
- package/visualization/eduevidence-report/scripts/build_report.py +435 -575
- package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
- package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
- package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
- package/visualization/eduevidence-report/scripts/zh_labels.py +141 -1
- package/web/architecture.html +14885 -0
- package/web/studio/assets/index-B8tkF44Q.css +1 -0
- package/web/studio/index.html +2 -2
- package/scripts/build_esl_artifacts.py +0 -1921
- package/scripts/build_killer_demo.py +0 -295
- package/scripts/enrich_projects_human_and_lieflat.py +0 -315
- package/scripts/generate_new_projects.py +0 -686
- package/scripts/sync_killer_demo_report.py +0 -270
- package/web/studio/assets/index-CzXocaGv.css +0 -1
- /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
|
@@ -1 +1,156 @@
|
|
|
1
|
-
{
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"title": "TaskSpec",
|
|
4
|
+
"description": "A dispatchable unit of work for one role (engine/orchestration.py TaskSpec). Fixes the stage, role, evidence axis, run and base revision, the granted capabilities, forbidden actions, budget, expected outputs and the output contract. Delegated specs are staging-only: canonical-state artifacts are refused by construction.",
|
|
5
|
+
"type": "object",
|
|
6
|
+
"required": [
|
|
7
|
+
"task_id",
|
|
8
|
+
"run_id",
|
|
9
|
+
"base_revision",
|
|
10
|
+
"stage",
|
|
11
|
+
"role",
|
|
12
|
+
"role_profile",
|
|
13
|
+
"objective",
|
|
14
|
+
"reason_for_delegation",
|
|
15
|
+
"evidence_axis",
|
|
16
|
+
"inputs",
|
|
17
|
+
"input_artifacts",
|
|
18
|
+
"allowed_capabilities",
|
|
19
|
+
"forbidden_actions",
|
|
20
|
+
"scope",
|
|
21
|
+
"budget",
|
|
22
|
+
"expected_outputs",
|
|
23
|
+
"output_contract",
|
|
24
|
+
"termination",
|
|
25
|
+
"execution_mode",
|
|
26
|
+
"independent",
|
|
27
|
+
"read_only",
|
|
28
|
+
"timeout_seconds",
|
|
29
|
+
"token_budget",
|
|
30
|
+
"metadata"
|
|
31
|
+
],
|
|
32
|
+
"properties": {
|
|
33
|
+
"task_id": {
|
|
34
|
+
"type": "string",
|
|
35
|
+
"minLength": 1
|
|
36
|
+
},
|
|
37
|
+
"run_id": {
|
|
38
|
+
"type": [
|
|
39
|
+
"string",
|
|
40
|
+
"null"
|
|
41
|
+
]
|
|
42
|
+
},
|
|
43
|
+
"base_revision": {
|
|
44
|
+
"type": [
|
|
45
|
+
"integer",
|
|
46
|
+
"null"
|
|
47
|
+
],
|
|
48
|
+
"minimum": 0
|
|
49
|
+
},
|
|
50
|
+
"stage": {
|
|
51
|
+
"enum": [
|
|
52
|
+
"frame",
|
|
53
|
+
"retrieve",
|
|
54
|
+
"extract",
|
|
55
|
+
"challenge",
|
|
56
|
+
"audit",
|
|
57
|
+
"adjudicate",
|
|
58
|
+
"applicability",
|
|
59
|
+
"intervene",
|
|
60
|
+
"evaluate"
|
|
61
|
+
]
|
|
62
|
+
},
|
|
63
|
+
"role": {
|
|
64
|
+
"type": "string"
|
|
65
|
+
},
|
|
66
|
+
"role_profile": {
|
|
67
|
+
"type": [
|
|
68
|
+
"string",
|
|
69
|
+
"null"
|
|
70
|
+
]
|
|
71
|
+
},
|
|
72
|
+
"objective": {
|
|
73
|
+
"type": "string",
|
|
74
|
+
"minLength": 1
|
|
75
|
+
},
|
|
76
|
+
"reason_for_delegation": {
|
|
77
|
+
"type": [
|
|
78
|
+
"string",
|
|
79
|
+
"null"
|
|
80
|
+
]
|
|
81
|
+
},
|
|
82
|
+
"evidence_axis": {
|
|
83
|
+
"type": "string",
|
|
84
|
+
"minLength": 1
|
|
85
|
+
},
|
|
86
|
+
"inputs": {
|
|
87
|
+
"type": "array",
|
|
88
|
+
"items": {
|
|
89
|
+
"type": "string"
|
|
90
|
+
}
|
|
91
|
+
},
|
|
92
|
+
"input_artifacts": {
|
|
93
|
+
"type": "array",
|
|
94
|
+
"items": {
|
|
95
|
+
"type": "string"
|
|
96
|
+
}
|
|
97
|
+
},
|
|
98
|
+
"allowed_capabilities": {
|
|
99
|
+
"type": "array",
|
|
100
|
+
"items": {
|
|
101
|
+
"type": "string"
|
|
102
|
+
}
|
|
103
|
+
},
|
|
104
|
+
"forbidden_actions": {
|
|
105
|
+
"type": "array",
|
|
106
|
+
"items": {
|
|
107
|
+
"type": "string"
|
|
108
|
+
}
|
|
109
|
+
},
|
|
110
|
+
"scope": {
|
|
111
|
+
"type": "object"
|
|
112
|
+
},
|
|
113
|
+
"budget": {
|
|
114
|
+
"type": "object"
|
|
115
|
+
},
|
|
116
|
+
"expected_outputs": {
|
|
117
|
+
"type": "array",
|
|
118
|
+
"items": {
|
|
119
|
+
"type": "string"
|
|
120
|
+
}
|
|
121
|
+
},
|
|
122
|
+
"output_contract": {
|
|
123
|
+
"type": "object"
|
|
124
|
+
},
|
|
125
|
+
"termination": {
|
|
126
|
+
"type": "object"
|
|
127
|
+
},
|
|
128
|
+
"execution_mode": {
|
|
129
|
+
"enum": [
|
|
130
|
+
"local",
|
|
131
|
+
"delegated"
|
|
132
|
+
]
|
|
133
|
+
},
|
|
134
|
+
"independent": {
|
|
135
|
+
"type": "boolean"
|
|
136
|
+
},
|
|
137
|
+
"read_only": {
|
|
138
|
+
"type": "boolean"
|
|
139
|
+
},
|
|
140
|
+
"timeout_seconds": {
|
|
141
|
+
"type": "integer",
|
|
142
|
+
"minimum": 1
|
|
143
|
+
},
|
|
144
|
+
"token_budget": {
|
|
145
|
+
"type": [
|
|
146
|
+
"integer",
|
|
147
|
+
"null"
|
|
148
|
+
],
|
|
149
|
+
"minimum": 1
|
|
150
|
+
},
|
|
151
|
+
"metadata": {
|
|
152
|
+
"type": "object"
|
|
153
|
+
}
|
|
154
|
+
},
|
|
155
|
+
"additionalProperties": false
|
|
156
|
+
}
|
|
@@ -1 +1,60 @@
|
|
|
1
|
-
{
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"title": "WorkerResult",
|
|
4
|
+
"description": "The lead process's verdict on one worker's submission (engine/worker_result.py). validate_worker_output() reconstructs this from untrusted worker output: any worker-supplied validated flag is ignored, canonical-state artifacts are rejected, and only results that pass here may enter a Judge context.",
|
|
5
|
+
"type": "object",
|
|
6
|
+
"required": [
|
|
7
|
+
"task_id",
|
|
8
|
+
"status",
|
|
9
|
+
"staging_artifacts",
|
|
10
|
+
"validated",
|
|
11
|
+
"validation_issues",
|
|
12
|
+
"metrics",
|
|
13
|
+
"summary"
|
|
14
|
+
],
|
|
15
|
+
"properties": {
|
|
16
|
+
"task_id": {
|
|
17
|
+
"type": "string",
|
|
18
|
+
"minLength": 1
|
|
19
|
+
},
|
|
20
|
+
"status": {
|
|
21
|
+
"enum": [
|
|
22
|
+
"completed",
|
|
23
|
+
"failed",
|
|
24
|
+
"blocked"
|
|
25
|
+
]
|
|
26
|
+
},
|
|
27
|
+
"staging_artifacts": {
|
|
28
|
+
"type": "array",
|
|
29
|
+
"items": {
|
|
30
|
+
"type": "object",
|
|
31
|
+
"required": [
|
|
32
|
+
"artifact_type"
|
|
33
|
+
],
|
|
34
|
+
"properties": {
|
|
35
|
+
"artifact_type": {
|
|
36
|
+
"type": "string",
|
|
37
|
+
"minLength": 1
|
|
38
|
+
}
|
|
39
|
+
},
|
|
40
|
+
"additionalProperties": true
|
|
41
|
+
}
|
|
42
|
+
},
|
|
43
|
+
"validated": {
|
|
44
|
+
"type": "boolean"
|
|
45
|
+
},
|
|
46
|
+
"validation_issues": {
|
|
47
|
+
"type": "array",
|
|
48
|
+
"items": {
|
|
49
|
+
"type": "string"
|
|
50
|
+
}
|
|
51
|
+
},
|
|
52
|
+
"metrics": {
|
|
53
|
+
"type": "object"
|
|
54
|
+
},
|
|
55
|
+
"summary": {
|
|
56
|
+
"type": "string"
|
|
57
|
+
}
|
|
58
|
+
},
|
|
59
|
+
"additionalProperties": false
|
|
60
|
+
}
|
|
@@ -5,52 +5,188 @@
|
|
|
5
5
|
"description": "Output of the Evidence Tribunal: what the evidence supports, cannot support, and the recommended action. Decision is one of ADOPT / PILOT / REJECT / INSUFFICIENT EVIDENCE (expressed here as adopt|pilot|reject|insufficient_evidence). confidence 由 scripts/compute_confidence.py 确定性计算并覆盖模型值;confidence_score 是规则化指数(0-1),不是概率。扩展字段一律放在 extensions 内。",
|
|
6
6
|
"type": "object",
|
|
7
7
|
"additionalProperties": false,
|
|
8
|
-
"required": [
|
|
8
|
+
"required": [
|
|
9
|
+
"decision_question",
|
|
10
|
+
"recommended_action",
|
|
11
|
+
"confidence"
|
|
12
|
+
],
|
|
9
13
|
"properties": {
|
|
10
|
-
"decision_question": {
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
"
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
"
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
"
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
14
|
+
"decision_question": {
|
|
15
|
+
"type": "string"
|
|
16
|
+
},
|
|
17
|
+
"target_population": {
|
|
18
|
+
"type": "string"
|
|
19
|
+
},
|
|
20
|
+
"target_context": {
|
|
21
|
+
"type": "string"
|
|
22
|
+
},
|
|
23
|
+
"supported_claims": {
|
|
24
|
+
"type": "array",
|
|
25
|
+
"items": {
|
|
26
|
+
"type": "string"
|
|
27
|
+
}
|
|
28
|
+
},
|
|
29
|
+
"uncertain_claims": {
|
|
30
|
+
"type": "array",
|
|
31
|
+
"items": {
|
|
32
|
+
"type": "string"
|
|
33
|
+
}
|
|
34
|
+
},
|
|
35
|
+
"contradicted_claims": {
|
|
36
|
+
"type": "array",
|
|
37
|
+
"items": {
|
|
38
|
+
"type": "string"
|
|
39
|
+
}
|
|
40
|
+
},
|
|
41
|
+
"reason_for_disagreement": {
|
|
42
|
+
"type": "string"
|
|
43
|
+
},
|
|
44
|
+
"methodology_summary": {
|
|
45
|
+
"type": "string"
|
|
46
|
+
},
|
|
47
|
+
"outcome_specific_findings": {
|
|
48
|
+
"type": "object",
|
|
49
|
+
"additionalProperties": true
|
|
50
|
+
},
|
|
51
|
+
"short_term_effect": {
|
|
52
|
+
"type": [
|
|
53
|
+
"string",
|
|
54
|
+
"null"
|
|
55
|
+
]
|
|
56
|
+
},
|
|
57
|
+
"long_term_effect": {
|
|
58
|
+
"type": [
|
|
59
|
+
"string",
|
|
60
|
+
"null"
|
|
61
|
+
]
|
|
62
|
+
},
|
|
63
|
+
"transfer_effect": {
|
|
64
|
+
"type": [
|
|
65
|
+
"string",
|
|
66
|
+
"null"
|
|
67
|
+
]
|
|
68
|
+
},
|
|
69
|
+
"risk_effect": {
|
|
70
|
+
"type": [
|
|
71
|
+
"string",
|
|
72
|
+
"null"
|
|
73
|
+
]
|
|
74
|
+
},
|
|
75
|
+
"applicability": {
|
|
76
|
+
"type": "object",
|
|
77
|
+
"additionalProperties": true
|
|
78
|
+
},
|
|
79
|
+
"confidence": {
|
|
80
|
+
"type": "string",
|
|
81
|
+
"enum": [
|
|
82
|
+
"High",
|
|
83
|
+
"Moderate",
|
|
84
|
+
"Low",
|
|
85
|
+
"Insufficient"
|
|
86
|
+
]
|
|
87
|
+
},
|
|
25
88
|
"confidence_score": {
|
|
26
|
-
"type": [
|
|
89
|
+
"type": [
|
|
90
|
+
"number",
|
|
91
|
+
"null"
|
|
92
|
+
],
|
|
27
93
|
"minimum": 0,
|
|
28
94
|
"maximum": 1,
|
|
29
95
|
"description": "规则化置信度指数(0-1),由 compute_confidence.py 覆盖模型值。不是概率,禁止宣传为百分比。"
|
|
30
96
|
},
|
|
31
|
-
"confidence_policy_version": {
|
|
32
|
-
|
|
33
|
-
|
|
97
|
+
"confidence_policy_version": {
|
|
98
|
+
"type": "string",
|
|
99
|
+
"description": "确定性置信度策略版本号(如 2026-08-12.v1)。"
|
|
100
|
+
},
|
|
101
|
+
"independent_studies": {
|
|
102
|
+
"type": [
|
|
103
|
+
"integer",
|
|
104
|
+
"null"
|
|
105
|
+
],
|
|
106
|
+
"minimum": 0,
|
|
107
|
+
"description": "独立研究数(按 study_id/source_id 去重)。"
|
|
108
|
+
},
|
|
109
|
+
"independent_samples": {
|
|
110
|
+
"type": [
|
|
111
|
+
"integer",
|
|
112
|
+
"null"
|
|
113
|
+
],
|
|
114
|
+
"minimum": 0,
|
|
115
|
+
"description": "独立样本数(按 sample_id 去重)。"
|
|
116
|
+
},
|
|
34
117
|
"confidence_breakdown": {
|
|
35
118
|
"type": "object",
|
|
36
119
|
"additionalProperties": true,
|
|
37
120
|
"description": "Rule-based components: evidence_quality, consistency, directness, evidence_count, independent_studies, independent_samples, conflict_penalty, unsupported_penalty."
|
|
38
121
|
},
|
|
39
122
|
"raw_model_confidence": {
|
|
40
|
-
"type": [
|
|
123
|
+
"type": [
|
|
124
|
+
"string",
|
|
125
|
+
"null"
|
|
126
|
+
],
|
|
41
127
|
"description": "模型原始 confidence 输出(被确定性值覆盖前的值,仅供审计比对)。"
|
|
42
128
|
},
|
|
43
|
-
"raw_model_confidence_breakdown": {
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
"
|
|
48
|
-
|
|
49
|
-
|
|
129
|
+
"raw_model_confidence_breakdown": {
|
|
130
|
+
"type": "object",
|
|
131
|
+
"additionalProperties": true
|
|
132
|
+
},
|
|
133
|
+
"what_can_be_claimed": {
|
|
134
|
+
"type": "array",
|
|
135
|
+
"items": {
|
|
136
|
+
"type": "string"
|
|
137
|
+
}
|
|
138
|
+
},
|
|
139
|
+
"what_cannot_be_claimed": {
|
|
140
|
+
"type": "array",
|
|
141
|
+
"items": {
|
|
142
|
+
"type": "string"
|
|
143
|
+
}
|
|
144
|
+
},
|
|
145
|
+
"missing_evidence": {
|
|
146
|
+
"type": "array",
|
|
147
|
+
"items": {
|
|
148
|
+
"type": "string"
|
|
149
|
+
}
|
|
150
|
+
},
|
|
151
|
+
"recommended_action": {
|
|
152
|
+
"type": "string",
|
|
153
|
+
"enum": [
|
|
154
|
+
"adopt",
|
|
155
|
+
"pilot",
|
|
156
|
+
"reject",
|
|
157
|
+
"insufficient_evidence"
|
|
158
|
+
]
|
|
159
|
+
},
|
|
160
|
+
"decision_rationale": {
|
|
161
|
+
"type": "string"
|
|
162
|
+
},
|
|
163
|
+
"exceeds_evidence_boundary": {
|
|
164
|
+
"type": "array",
|
|
165
|
+
"items": {
|
|
166
|
+
"type": "string"
|
|
167
|
+
},
|
|
168
|
+
"description": "Conclusions that currently go beyond the evidence boundary."
|
|
169
|
+
},
|
|
50
170
|
"extensions": {
|
|
51
171
|
"type": "object",
|
|
52
172
|
"description": "结构化扩展字段的统一容器(P1-01)。未列入本 schema 的字段必须放在这里,禁止在顶层新增属性。",
|
|
53
173
|
"additionalProperties": true
|
|
174
|
+
},
|
|
175
|
+
"strongest_support": {
|
|
176
|
+
"type": "string",
|
|
177
|
+
"description": "The single strongest conclusion the evidence supports, as a complete reader-facing sentence (<=60 chars zh / ~15 words en). Written by the adjudicator, not assembled by the renderer."
|
|
178
|
+
},
|
|
179
|
+
"key_uncertainty": {
|
|
180
|
+
"type": "string",
|
|
181
|
+
"description": "The decision-relevant uncertainty or counter-evidence, as a complete reader-facing sentence (<=70 chars zh / ~18 words en)."
|
|
182
|
+
},
|
|
183
|
+
"main_risk": {
|
|
184
|
+
"type": "string",
|
|
185
|
+
"description": "The principal risk of acting, as a complete reader-facing sentence (<=60 chars zh / ~15 words en)."
|
|
186
|
+
},
|
|
187
|
+
"next_action": {
|
|
188
|
+
"type": "string",
|
|
189
|
+
"description": "The recommended next step, as a complete reader-facing sentence (<=80 chars zh / ~20 words en)."
|
|
54
190
|
}
|
|
55
191
|
}
|
|
56
192
|
}
|
|
@@ -17,9 +17,15 @@ schemas/v4/evidence-library.schema.json using the repo's zero-dependency
|
|
|
17
17
|
validator (scripts/validate_schema.py).
|
|
18
18
|
|
|
19
19
|
Direction semantics (adoption-relevant, conservative):
|
|
20
|
-
support -> evidence favors adopting the intervention
|
|
21
|
-
|
|
20
|
+
support -> evidence favors adopting the intervention
|
|
21
|
+
(offline screening cap => pilot; never adopt)
|
|
22
|
+
contradict -> evidence opposes adopting the intervention
|
|
23
|
+
(oppose-only => reject; mixed with support is an unresolved
|
|
24
|
+
conflict and must fall to INSUFFICIENT_EVIDENCE via
|
|
25
|
+
engine.decision_policy.decision_outcome)
|
|
22
26
|
neutral -> inconclusive
|
|
27
|
+
The library never emits adopt: full ADOPT is decided only by
|
|
28
|
+
engine.decision_policy.decision_outcome (High + support + direct primary).
|
|
23
29
|
For gold units the coarse rule is: if a question's expected decision range is
|
|
24
30
|
purely reject-oriented ("reject" present and "pilot" absent), its
|
|
25
31
|
key_claims/key_supporting_sources are harmful evidence => contradict, and its
|
|
@@ -266,10 +272,14 @@ def build(generated_at: str | None = None) -> tuple[dict[str, Any], int]:
|
|
|
266
272
|
"key_claims/key_supporting_sources/known_contradictions/correct_outcome_types)"
|
|
267
273
|
"+ 3 个示例工作流 evidence.jsonl(ai-coding-assistant / ai-tutor / ai-writing-assistant)"
|
|
268
274
|
"抽取生成;按 (source_id, outcome_token, claim_text) 去重合并。"
|
|
269
|
-
"direction 语义为采纳方向:support
|
|
270
|
-
"contradict
|
|
275
|
+
"direction 语义为采纳方向:support=支持采纳(离线初筛上限 => pilot,永不输出 adopt),"
|
|
276
|
+
"contradict=反对采纳(oppose-only 时 => reject;与 support 并存属未决冲突,应交 "
|
|
277
|
+
"engine.decision_policy.decision_outcome 判为 INSUFFICIENT_EVIDENCE),neutral=中性;"
|
|
278
|
+
"金标准条目按 expected_decision_range "
|
|
271
279
|
"粗粒度映射方向(纯 reject 问题反向映射),conflict 与混合方向问题的单条断言方向可能不精确。"
|
|
272
|
-
"仅用于离线初步裁决(preliminary,保守),从不直接给出 adopt
|
|
280
|
+
"仅用于离线初步裁决(preliminary,保守),从不直接给出 adopt;"
|
|
281
|
+
"完整 ADOPT 只能由 engine.decision_policy.decision_outcome 给出"
|
|
282
|
+
"(High + support + 主要结果 directness 2)。"
|
|
273
283
|
),
|
|
274
284
|
}
|
|
275
285
|
_validate(library)
|
|
@@ -40,7 +40,9 @@ def bake(examples: Path, *, force: bool = False) -> list[dict]:
|
|
|
40
40
|
if not force and manifest.is_file():
|
|
41
41
|
try:
|
|
42
42
|
prior = json.loads(manifest.read_text(encoding='utf-8'))
|
|
43
|
-
|
|
43
|
+
main = prior.get('main_report') or {}
|
|
44
|
+
main_ok = bool(main.get('file')) and (directory / main['file']).is_file() and hashlib.sha256((directory / main['file']).read_bytes()).hexdigest() == main.get('sha256')
|
|
45
|
+
valid = prior.get('cache_key') == cache_key and main_ok and all(
|
|
44
46
|
(out_dir / record['file']).is_file() and hashlib.sha256((out_dir / record['file']).read_bytes()).hexdigest() == record['sha256']
|
|
45
47
|
for record in prior.get('reports', [])) and len(prior.get('reports', [])) == len(THEMES)
|
|
46
48
|
if valid:
|
|
@@ -64,8 +66,22 @@ def bake(examples: Path, *, force: bool = False) -> list[dict]:
|
|
|
64
66
|
raise RuntimeError(f'{directory.name}/{theme}: renderer rejected input\n{completed.stdout}\n{completed.stderr}')
|
|
65
67
|
os.replace(temporary, target)
|
|
66
68
|
records.append({'theme': theme, 'file': target.name, 'sha256': hashlib.sha256(target.read_bytes()).hexdigest()})
|
|
69
|
+
# Also refresh the pack-root report. This used to write only the themed
|
|
70
|
+
# variants, so examples/*/EduEvidence_Report.html kept whatever bytes it
|
|
71
|
+
# was first rendered with - the packaged example shipped a report the
|
|
72
|
+
# current renderer would not produce.
|
|
73
|
+
from_default = next((r for r in records if r['theme'] == 'claude'), records[0])
|
|
74
|
+
main_target = directory / 'EduEvidence_Report.html'
|
|
75
|
+
main_temp = directory / '.EduEvidence_Report.pending.html'
|
|
76
|
+
main_temp.write_bytes((out_dir / from_default['file']).read_bytes())
|
|
77
|
+
os.replace(main_temp, main_target)
|
|
78
|
+
main_sha = hashlib.sha256(main_target.read_bytes()).hexdigest()
|
|
79
|
+
|
|
67
80
|
value = {'schema_version': 1, 'project': directory.name, 'cache_key': cache_key,
|
|
68
|
-
'result_sha256': result_hash, 'renderer_sha256': engine_hash,
|
|
81
|
+
'result_sha256': result_hash, 'renderer_sha256': engine_hash,
|
|
82
|
+
'main_report': {'file': main_target.name, 'sha256': main_sha,
|
|
83
|
+
'theme': from_default['theme']},
|
|
84
|
+
'reports': records}
|
|
69
85
|
manifest.write_text(json.dumps(value, indent=2) + '\n', encoding='utf-8')
|
|
70
86
|
reports.append(value)
|
|
71
87
|
print(f'{directory.name}: {len(records)} verified report variants')
|
package/scripts/build_result.py
CHANGED
|
@@ -30,14 +30,14 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
|
|
30
30
|
from evidence_semantics import effect_direction
|
|
31
31
|
from engine.versions import ENGINE_VERSION
|
|
32
32
|
|
|
33
|
-
|
|
34
|
-
"
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
33
|
+
def _outcome_order() -> list[str]:
|
|
34
|
+
"""Registered outcome tokens in registry order (was a hard-coded list)."""
|
|
35
|
+
from engine.taxonomy import all_tokens_ordered
|
|
36
|
+
|
|
37
|
+
return list(all_tokens_ordered())
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
OUTCOME_ORDER = _outcome_order()
|
|
41
41
|
|
|
42
42
|
|
|
43
43
|
def _load_json(path: Path) -> dict[str, Any] | None:
|
|
@@ -221,12 +221,74 @@ def build_claims(evidence: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
|
|
221
221
|
return list(claims.values())
|
|
222
222
|
|
|
223
223
|
|
|
224
|
+
def _applicability(pack_dir: Path, verdict: dict) -> dict:
|
|
225
|
+
"""Stage-7 applicability assessment, falling back to the verdict boundary.
|
|
226
|
+
|
|
227
|
+
applicability.json is the dedicated deliverable of the Applicability stage.
|
|
228
|
+
It used to be written and then ignored, because the renderer only looked at
|
|
229
|
+
the verdict; this is where it re-enters the result.
|
|
230
|
+
"""
|
|
231
|
+
import json as _json
|
|
232
|
+
|
|
233
|
+
path = pack_dir / "applicability.json"
|
|
234
|
+
if path.is_file():
|
|
235
|
+
try:
|
|
236
|
+
data = _json.loads(path.read_text(encoding="utf-8"))
|
|
237
|
+
except (OSError, _json.JSONDecodeError):
|
|
238
|
+
data = None
|
|
239
|
+
if isinstance(data, dict) and data and data.get("status") != "NOT_CAPTURED":
|
|
240
|
+
return data
|
|
241
|
+
value = verdict.get("applicability") if isinstance(verdict, dict) else None
|
|
242
|
+
return value if isinstance(value, dict) else {}
|
|
243
|
+
|
|
244
|
+
def _derive_study_audits(methodology: list[dict], evidence: list[dict]) -> list[dict]:
|
|
245
|
+
"""Per-study audit rows derived from the audits and evidence present.
|
|
246
|
+
|
|
247
|
+
Each row names the study and reports the audit verdict that covers it,
|
|
248
|
+
so the per-study axis the schema advertises actually exists downstream.
|
|
249
|
+
Rows are only emitted for studies the audits or evidence actually name.
|
|
250
|
+
"""
|
|
251
|
+
by_study: dict[str, dict] = {}
|
|
252
|
+
for audit in methodology:
|
|
253
|
+
if not isinstance(audit, dict):
|
|
254
|
+
continue
|
|
255
|
+
target = audit.get("target") or "overall"
|
|
256
|
+
if target == "overall":
|
|
257
|
+
# The aggregate audit is not a study row; label it as the
|
|
258
|
+
# body-of-evidence review so it cannot be mistaken for one.
|
|
259
|
+
target = "body_of_evidence"
|
|
260
|
+
entry = by_study.setdefault(target, {
|
|
261
|
+
"study_id": target,
|
|
262
|
+
"verdict": audit.get("verdict"),
|
|
263
|
+
"audit_items": audit.get("audit_items") or {},
|
|
264
|
+
"limitations": list(audit.get("limitations") or []),
|
|
265
|
+
"task_vs_learning_guard": audit.get("task_vs_learning_guard"),
|
|
266
|
+
})
|
|
267
|
+
entry.setdefault("evidence_ids", [])
|
|
268
|
+
known = {e.get("study_id") for e in evidence if e.get("study_id")}
|
|
269
|
+
for study_id in sorted(known):
|
|
270
|
+
by_study.setdefault(study_id, {
|
|
271
|
+
"study_id": study_id,
|
|
272
|
+
"verdict": None,
|
|
273
|
+
"audit_items": {},
|
|
274
|
+
"limitations": [],
|
|
275
|
+
"task_vs_learning_guard": None,
|
|
276
|
+
"evidence_ids": [e.get("evidence_id") for e in evidence
|
|
277
|
+
if e.get("study_id") == study_id],
|
|
278
|
+
})
|
|
279
|
+
return list(by_study.values())
|
|
280
|
+
|
|
281
|
+
|
|
224
282
|
def build_result(pack_dir: Path, *, mode: str = "platform_native") -> dict[str, Any]:
|
|
225
283
|
frame = _load_json(pack_dir / "frame.json") or {}
|
|
226
284
|
evidence = _load_jsonl(pack_dir / "evidence.jsonl")
|
|
227
285
|
# methodology.json is a single MethodologyAudit object (or a JSONL list)
|
|
228
286
|
methodology_single = _load_json(pack_dir / "methodology.json")
|
|
229
287
|
methodology = [methodology_single] if methodology_single else _load_jsonl(pack_dir / "methodology.jsonl")
|
|
288
|
+
# Per-study audits: the contract advertised them but nothing produced
|
|
289
|
+
# them, so a multi-study review silently shipped a single audit object.
|
|
290
|
+
# Derive the per-study axis from the audits actually present.
|
|
291
|
+
study_audits = _derive_study_audits(methodology, evidence)
|
|
230
292
|
verdict = _load_json(pack_dir / "verdict.json") or {}
|
|
231
293
|
intervention = _load_json(pack_dir / "intervention.json") or {}
|
|
232
294
|
evaluation = _load_json(pack_dir / "evaluation.json") or {}
|
|
@@ -268,9 +330,12 @@ def build_result(pack_dir: Path, *, mode: str = "platform_native") -> dict[str,
|
|
|
268
330
|
"sources": sources,
|
|
269
331
|
"evidence": evidence,
|
|
270
332
|
"methodology_reviews": methodology,
|
|
333
|
+
"study_audits": study_audits,
|
|
271
334
|
"conflicts": [{"reason_for_disagreement": verdict.get("reason_for_disagreement", "")}]
|
|
272
335
|
if verdict.get("reason_for_disagreement") else [],
|
|
273
|
-
|
|
336
|
+
# Prefer the dedicated stage-7 assessment; fall back to the verdict-embedded
|
|
337
|
+
# boundary when the run carries no separate applicability.json.
|
|
338
|
+
"applicability": _applicability(pack_dir, verdict),
|
|
274
339
|
"intervention": intervention,
|
|
275
340
|
"evaluation": evaluation,
|
|
276
341
|
"benchmark": {},
|