eduevidence 6.2.0 → 6.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +395 -0
- package/README.md +22 -13
- package/README.zh-CN.md +15 -8
- package/SKILL.md +10 -9
- package/benchmarks/evidence-library.json +277 -1
- package/docs/architecture.md +6 -3
- package/docs/j-ev-experimental.md +250 -0
- package/docs/reproducibility.md +138 -0
- package/domains/_neutral/copy/few_shots.json +21 -0
- package/domains/_neutral/copy/framing_lexicon.json +19 -0
- package/domains/_neutral/copy/module_labels.json +5 -0
- package/domains/_neutral/copy/module_labels_footer.json +102 -0
- package/domains/_neutral/copy/module_labels_modules.json +204 -0
- package/domains/_neutral/copy/module_labels_nav.json +126 -0
- package/domains/_neutral/copy/module_labels_summary.json +98 -0
- package/domains/_neutral/copy/module_labels_tables.json +164 -0
- package/domains/_neutral/copy/module_labels_v2.json +90 -0
- package/domains/_neutral/copy/risk_constructs.json +20 -0
- package/domains/_neutral/copy/section_titles.json +66 -0
- package/domains/_neutral/copy/terminology.json +11 -0
- package/domains/check_copy_packs.py +103 -0
- package/domains/education/copy/few_shots.json +22 -0
- package/domains/education/copy/framing_enums.json +167 -0
- package/domains/education/copy/framing_lexicon.json +166 -0
- package/domains/education/copy/module_labels.json +169 -0
- package/domains/education/copy/risk_constructs.json +48 -0
- package/domains/education/copy/section_titles.json +186 -0
- package/domains/education/copy/terminology.json +70 -0
- package/domains/education/manifest.json +1 -1
- package/domains/education/outcome_taxonomy.json +2 -2
- package/domains/manifest.json +1 -1
- package/domains/policy/copy/few_shots.json +22 -0
- package/domains/policy/copy/framing_enums.json +94 -0
- package/domains/policy/copy/framing_lexicon.json +174 -0
- package/domains/policy/copy/module_labels.json +168 -0
- package/domains/policy/copy/risk_constructs.json +33 -0
- package/domains/policy/copy/section_titles.json +186 -0
- package/domains/policy/copy/terminology.json +64 -0
- package/engine/capabilities.py +57 -5
- package/engine/decision_policy.py +88 -17
- package/engine/library_builtin.py +7 -4
- package/engine/tribunal.py +17 -23
- package/engine/versions.py +1 -1
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +4 -4
- package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
- package/examples/spaced-retrieval-practice/report.html +2522 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +4 -4
- package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +36 -36
- package/integrations/jev/__init__.py +115 -0
- package/integrations/jev/approval.py +212 -0
- package/integrations/jev/cli.py +84 -0
- package/integrations/jev/config.py +112 -0
- package/integrations/jev/gateway.py +128 -0
- package/integrations/jev/modes.py +38 -0
- package/integrations/jev/tools_classify.py +88 -0
- package/integrations/jev/tools_extract.py +111 -0
- package/integrations/jev/tools_rerank.py +71 -0
- package/integrations/jev/tools_screen.py +87 -0
- package/integrations/jev/tools_verify.py +95 -0
- package/integrations/jev_mcp.py +22 -0
- package/integrations/semantic_decide.py +286 -0
- package/integrations/semdecide_cli.py +55 -0
- package/package.json +9 -1
- package/pyproject.toml +1 -1
- package/references/report-copy-style.md +43 -3
- package/schemas/v2/decision-snapshot.schema.json +20 -9
- package/schemas/v2/intake.schema.json +191 -0
- package/scripts/build_evidence_library.py +15 -5
- package/scripts/dashboard_server.py +13 -2
- package/scripts/intake/__init__.py +31 -0
- package/scripts/intake/__main__.py +18 -0
- package/scripts/intake/background.py +78 -0
- package/scripts/intake/browser.py +79 -0
- package/scripts/intake/cli.py +57 -0
- package/scripts/intake/constants.py +57 -0
- package/scripts/intake/depth.py +53 -0
- package/scripts/intake/enhancements.py +106 -0
- package/scripts/intake/hooks.py +90 -0
- package/scripts/intake/prefs.py +76 -0
- package/scripts/intake/prompts.py +85 -0
- package/scripts/intake/session.py +152 -0
- package/scripts/lint_file_layers.py +126 -0
- package/scripts/orchestrator.py +68 -17
- package/scripts/pre_verdict_gate.py +21 -7
- package/scripts/skill_lint.py +11 -1
- package/scripts/skill_payload.py +3 -3
- package/scripts/test_adversarial_empirical.py +70 -6
- package/skill/agents/evidence-judge.md +49 -7
- package/skill/workflows/experimental-jev.md +170 -0
- package/skill/workflows/intake.md +120 -0
- package/visualization/eduevidence-report/scripts/build_infographics.py +32 -14
- package/visualization/eduevidence-report/scripts/build_report.py +75 -662
- package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
- package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
- package/visualization/eduevidence-report/scripts/zh_labels.py +61 -0
- package/scripts/build_esl_artifacts.py +0 -1921
- package/scripts/build_killer_demo.py +0 -295
- package/scripts/enrich_projects_human_and_lieflat.py +0 -315
- package/scripts/generate_new_projects.py +0 -686
- package/scripts/sync_killer_demo_report.py +0 -270
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
{
|
|
2
|
+
"domain": "policy",
|
|
3
|
+
"purpose": "Domain-sensitive module/UI labels (policy voice). target_learners displays as 目标人群; intervention is 干预方案.",
|
|
4
|
+
"labels": {
|
|
5
|
+
"outcome_separation_title": {
|
|
6
|
+
"zh": "结果分离 · 过程产出 ≠ 政策效果",
|
|
7
|
+
"en": "Outcome Separation · Process output ≠ policy effect"
|
|
8
|
+
},
|
|
9
|
+
"outcome_sep_note": {
|
|
10
|
+
"zh": "将不同结果类型分开裁决,避免把试点期产出或效率提升直接等同于真实政策效果。",
|
|
11
|
+
"en": "Outcomes are adjudicated separately so pilot throughput or efficiency gains are not silently treated as real policy effects."
|
|
12
|
+
},
|
|
13
|
+
"outcome_group_task": {
|
|
14
|
+
"zh": "过程 / 近端产出",
|
|
15
|
+
"en": "Process / proximal output"
|
|
16
|
+
},
|
|
17
|
+
"outcome_group_learning": {
|
|
18
|
+
"zh": "效果 / 目标结果",
|
|
19
|
+
"en": "Effect / target outcome"
|
|
20
|
+
},
|
|
21
|
+
"outcome_group_risk": {
|
|
22
|
+
"zh": "风险 / 实施风险",
|
|
23
|
+
"en": "Risk / implementation risk"
|
|
24
|
+
},
|
|
25
|
+
"outcome_group_other": {
|
|
26
|
+
"zh": "其他结果",
|
|
27
|
+
"en": "Other outcomes"
|
|
28
|
+
},
|
|
29
|
+
"method_guard": {
|
|
30
|
+
"zh": "产出 vs 效果护栏",
|
|
31
|
+
"en": "Output vs effect guard"
|
|
32
|
+
},
|
|
33
|
+
"applicability": {
|
|
34
|
+
"zh": [
|
|
35
|
+
"适用于谁",
|
|
36
|
+
"适用辖区",
|
|
37
|
+
"适用结果",
|
|
38
|
+
"适用条件",
|
|
39
|
+
"目标人群",
|
|
40
|
+
"目标情境"
|
|
41
|
+
],
|
|
42
|
+
"en": [
|
|
43
|
+
"Suitable for",
|
|
44
|
+
"Jurisdiction",
|
|
45
|
+
"Outcomes",
|
|
46
|
+
"Conditions",
|
|
47
|
+
"Target population",
|
|
48
|
+
"Target context"
|
|
49
|
+
]
|
|
50
|
+
},
|
|
51
|
+
"intervention_learners": {
|
|
52
|
+
"zh": "目标人群",
|
|
53
|
+
"en": "Target population"
|
|
54
|
+
},
|
|
55
|
+
"intervention_population": {
|
|
56
|
+
"zh": "目标人群",
|
|
57
|
+
"en": "Target population"
|
|
58
|
+
},
|
|
59
|
+
"intervention_duration": {
|
|
60
|
+
"zh": "试点时长",
|
|
61
|
+
"en": "Pilot duration"
|
|
62
|
+
},
|
|
63
|
+
"intervention_policy": {
|
|
64
|
+
"zh": "干预使用规则",
|
|
65
|
+
"en": "Intervention usage policy"
|
|
66
|
+
},
|
|
67
|
+
"intervention_rule": {
|
|
68
|
+
"zh": "干预规则",
|
|
69
|
+
"en": "Intervention rule"
|
|
70
|
+
},
|
|
71
|
+
"intervention_activities": {
|
|
72
|
+
"zh": "活动",
|
|
73
|
+
"en": "Activities"
|
|
74
|
+
},
|
|
75
|
+
"intervention_check": {
|
|
76
|
+
"zh": "结果检查",
|
|
77
|
+
"en": "Outcome check"
|
|
78
|
+
},
|
|
79
|
+
"intervention_stop": {
|
|
80
|
+
"zh": "停止条件",
|
|
81
|
+
"en": "Stop conditions"
|
|
82
|
+
},
|
|
83
|
+
"intervention_timeline": {
|
|
84
|
+
"zh": "干预方案时间线信息图",
|
|
85
|
+
"en": "Intervention plan timeline infographic"
|
|
86
|
+
},
|
|
87
|
+
"evaluation_measures": {
|
|
88
|
+
"zh": [
|
|
89
|
+
"基线",
|
|
90
|
+
"试点结束",
|
|
91
|
+
"持续监测",
|
|
92
|
+
"扩展评估"
|
|
93
|
+
],
|
|
94
|
+
"en": [
|
|
95
|
+
"Baseline",
|
|
96
|
+
"Pilot end",
|
|
97
|
+
"Durability",
|
|
98
|
+
"Scale-out"
|
|
99
|
+
]
|
|
100
|
+
},
|
|
101
|
+
"evaluation_metrics": {
|
|
102
|
+
"zh": [
|
|
103
|
+
"过程指标",
|
|
104
|
+
"效果指标",
|
|
105
|
+
"风险指标"
|
|
106
|
+
],
|
|
107
|
+
"en": [
|
|
108
|
+
"Process metrics",
|
|
109
|
+
"Effect metrics",
|
|
110
|
+
"Risk metrics"
|
|
111
|
+
]
|
|
112
|
+
},
|
|
113
|
+
"evaluation_question": {
|
|
114
|
+
"zh": "研究问题",
|
|
115
|
+
"en": "Research question"
|
|
116
|
+
},
|
|
117
|
+
"evaluation_threshold": {
|
|
118
|
+
"zh": "成功阈值",
|
|
119
|
+
"en": "Success threshold"
|
|
120
|
+
},
|
|
121
|
+
"evaluation_plan": {
|
|
122
|
+
"zh": "分析计划",
|
|
123
|
+
"en": "Analysis plan"
|
|
124
|
+
},
|
|
125
|
+
"evaluation_figure": {
|
|
126
|
+
"zh": "评价设计信息图",
|
|
127
|
+
"en": "Evaluation design infographic"
|
|
128
|
+
},
|
|
129
|
+
"svg_intervention_title": {
|
|
130
|
+
"zh": "干预方案时间线",
|
|
131
|
+
"en": "Intervention plan timeline"
|
|
132
|
+
},
|
|
133
|
+
"svg_intervention_desc": {
|
|
134
|
+
"zh": "各试点阶段的短名称与活动数量;完整干预使用规则见阶段说明块。",
|
|
135
|
+
"en": "Short phase names and activity counts; full intervention usage rules are in the phase blocks."
|
|
136
|
+
},
|
|
137
|
+
"svg_evaluation_title": {
|
|
138
|
+
"zh": "评价设计流程",
|
|
139
|
+
"en": "Evaluation design flow"
|
|
140
|
+
},
|
|
141
|
+
"svg_evaluation_desc": {
|
|
142
|
+
"zh": "基线、试点结束、持续监测与扩展评估的评价流程;完整指标与分析计划见评估章节。",
|
|
143
|
+
"en": "Evaluation flow across baseline, pilot end, durability and scale-out; full metrics and analysis plan are in the evaluation section."
|
|
144
|
+
},
|
|
145
|
+
"svg_workflow_desc": {
|
|
146
|
+
"zh": "从问题框架、检索、抓取验证、证据抽取、反方质疑、方法审计、裁决到适用性与干预评价的完整流程。",
|
|
147
|
+
"en": "Research flow from framing, retrieval, fetch/verify, extraction, challenge, method audit and adjudication to applicability and intervention evaluation."
|
|
148
|
+
},
|
|
149
|
+
"decision_kpi": {
|
|
150
|
+
"zh": [
|
|
151
|
+
"决策",
|
|
152
|
+
"置信度",
|
|
153
|
+
"证据最充分的结果",
|
|
154
|
+
"最不确定的结果",
|
|
155
|
+
"主要风险",
|
|
156
|
+
"来源数量"
|
|
157
|
+
],
|
|
158
|
+
"en": [
|
|
159
|
+
"Decision",
|
|
160
|
+
"Confidence",
|
|
161
|
+
"Best-supported outcome",
|
|
162
|
+
"Most uncertain outcome",
|
|
163
|
+
"Main risk",
|
|
164
|
+
"Sources"
|
|
165
|
+
]
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
}
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
{
|
|
2
|
+
"domain": "policy",
|
|
3
|
+
"purpose": "Outcome grouping + separation guard constructs (process output is not policy effect). No teaching vocabulary.",
|
|
4
|
+
"outcome_groups": {
|
|
5
|
+
"task": [
|
|
6
|
+
"cost_effectiveness",
|
|
7
|
+
"feasibility"
|
|
8
|
+
],
|
|
9
|
+
"learning": [
|
|
10
|
+
"policy_effectiveness",
|
|
11
|
+
"equity"
|
|
12
|
+
],
|
|
13
|
+
"risk": [
|
|
14
|
+
"implementation_risk"
|
|
15
|
+
]
|
|
16
|
+
},
|
|
17
|
+
"separation_title": {
|
|
18
|
+
"zh": "结果分离 · 过程产出 ≠ 政策效果",
|
|
19
|
+
"en": "Outcome Separation · Process output ≠ policy effect"
|
|
20
|
+
},
|
|
21
|
+
"separation_note": {
|
|
22
|
+
"zh": "将不同结果类型分开裁决,避免把试点期产出或效率提升直接等同于真实政策效果。",
|
|
23
|
+
"en": "Outcomes are adjudicated separately so pilot throughput or efficiency gains are not silently treated as real policy effects."
|
|
24
|
+
},
|
|
25
|
+
"method_guard": {
|
|
26
|
+
"zh": "产出 vs 效果护栏",
|
|
27
|
+
"en": "Output vs effect guard"
|
|
28
|
+
},
|
|
29
|
+
"guard_principle": {
|
|
30
|
+
"zh": "过程产出提升 ≠ 政策效果实现;ADOPT 需要独立效果、持续性与公平证据。",
|
|
31
|
+
"en": "Process output is not policy effect; ADOPT requires independent effect, durability and equity evidence."
|
|
32
|
+
}
|
|
33
|
+
}
|
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
{
|
|
2
|
+
"domain": "policy",
|
|
3
|
+
"purpose": "Section titles/leads + default full-report chapter plan (policy voice; no teaching vocabulary).",
|
|
4
|
+
"section_titles": {
|
|
5
|
+
"zh": {
|
|
6
|
+
"01": "01 执行决策",
|
|
7
|
+
"02": "02 结果证据概览",
|
|
8
|
+
"03": "03 证据矩阵",
|
|
9
|
+
"04": "04 证据裁决",
|
|
10
|
+
"05": "05 方法学审计",
|
|
11
|
+
"06": "06 冲突分析",
|
|
12
|
+
"07": "07 主张-证据追溯",
|
|
13
|
+
"08": "08 适用性",
|
|
14
|
+
"09": "09 干预方案",
|
|
15
|
+
"10": "10 评价方案",
|
|
16
|
+
"11": "11 基准测试",
|
|
17
|
+
"12": "12 来源与溯源"
|
|
18
|
+
},
|
|
19
|
+
"en": {
|
|
20
|
+
"01": "01 Executive Decision",
|
|
21
|
+
"02": "02 Outcome Evidence Overview",
|
|
22
|
+
"03": "03 Evidence Matrix",
|
|
23
|
+
"04": "04 Evidence Tribunal",
|
|
24
|
+
"05": "05 Methodology Audit",
|
|
25
|
+
"06": "06 Conflict Analysis",
|
|
26
|
+
"07": "07 Claim-Evidence Trace",
|
|
27
|
+
"08": "08 Applicability",
|
|
28
|
+
"09": "09 Intervention Plan",
|
|
29
|
+
"10": "10 Evaluation Plan",
|
|
30
|
+
"11": "11 Benchmark",
|
|
31
|
+
"12": "12 Sources & Provenance"
|
|
32
|
+
}
|
|
33
|
+
},
|
|
34
|
+
"section_leads": {
|
|
35
|
+
"zh": {
|
|
36
|
+
"01": "本节先给结论:最终怎么裁决、置信度多高、靠哪几条证据。",
|
|
37
|
+
"02": "一图看清:哪些政策效果有支持证据、哪些被反驳。",
|
|
38
|
+
"03": "每条证据来自哪项研究、测了什么、方向与质量如何;可筛选、可搜索。",
|
|
39
|
+
"04": "证据允许主张什么、不允许主张什么;缺失的关键证据是什么。",
|
|
40
|
+
"05": "研究质量可靠吗?哪些方法学问题让结论打折。",
|
|
41
|
+
"06": "不同研究为何结论不同;分歧出在哪一环。",
|
|
42
|
+
"07": "从结论到证据到原始来源,每一步都可追查。",
|
|
43
|
+
"08": "结论适用于谁、什么人群与效果、需要什么条件。",
|
|
44
|
+
"09": "试点怎么分阶段放开干预规则;什么情况必须叫停。",
|
|
45
|
+
"10": "如何验证效果:指标、对照、成功阈值。",
|
|
46
|
+
"11": "EduEvidence 自身基准表现:引用精度与成本。",
|
|
47
|
+
"12": "每篇文献是谁、出自哪里、如何获取。"
|
|
48
|
+
},
|
|
49
|
+
"en": {
|
|
50
|
+
"01": "The verdict first: what we decide, at what confidence, on which evidence.",
|
|
51
|
+
"02": "At a glance: which policy outcomes have supporting evidence, which are contradicted.",
|
|
52
|
+
"03": "Where each piece of evidence comes from, what it measures, its direction and quality — filterable and searchable.",
|
|
53
|
+
"04": "What the evidence lets us claim, what it does not, and what is still missing.",
|
|
54
|
+
"05": "How reliable are these studies, and which methodological concerns discount the conclusions.",
|
|
55
|
+
"06": "Why studies disagree — and where exactly they diverge.",
|
|
56
|
+
"07": "Every step from conclusion to evidence to source stays traceable.",
|
|
57
|
+
"08": "Who the conclusion applies to, for which population and outcomes, under what conditions.",
|
|
58
|
+
"09": "How intervention rules phase in during a pilot, and when we must stop.",
|
|
59
|
+
"10": "How we verify real effects: metrics, comparison, success threshold.",
|
|
60
|
+
"11": "How EduEvidence itself performs: citation precision and cost.",
|
|
61
|
+
"12": "Who wrote each cited study, where it came from, how it was fetched."
|
|
62
|
+
}
|
|
63
|
+
},
|
|
64
|
+
"default_chapters": [
|
|
65
|
+
{
|
|
66
|
+
"key": "decision",
|
|
67
|
+
"title_zh": "结论、裁决与研究边界",
|
|
68
|
+
"title_en": "Decision, Adjudication & Research Boundary",
|
|
69
|
+
"lead_zh": "先明确最终裁决与研究边界,再解释为什么。",
|
|
70
|
+
"lead_en": "State the final adjudication and research boundary before explaining why.",
|
|
71
|
+
"modules": [
|
|
72
|
+
"decision",
|
|
73
|
+
"scope"
|
|
74
|
+
]
|
|
75
|
+
},
|
|
76
|
+
{
|
|
77
|
+
"key": "evidence",
|
|
78
|
+
"title_zh": "关键证据与结果分离",
|
|
79
|
+
"title_en": "Key Evidence & Outcome Separation",
|
|
80
|
+
"lead_zh": "把过程产出、真实效果、持续性与风险放在同一证据地图中,但不混为一谈。",
|
|
81
|
+
"lead_en": "Place process outputs, real effects, durability and risk on one evidence map without conflating them.",
|
|
82
|
+
"modules": [
|
|
83
|
+
"retrieval",
|
|
84
|
+
"outcomes",
|
|
85
|
+
"evidence"
|
|
86
|
+
]
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
"key": "quality",
|
|
90
|
+
"title_zh": "证据可信度、反证与方法审计",
|
|
91
|
+
"title_en": "Evidence Quality, Counterevidence & Method Audit",
|
|
92
|
+
"lead_zh": "检查证据为什么可信、哪里冲突,以及哪些结论必须降级。",
|
|
93
|
+
"lead_en": "Examine why evidence is credible, where it conflicts, and which conclusions require downgrading.",
|
|
94
|
+
"modules": [
|
|
95
|
+
"quality",
|
|
96
|
+
"conflicts",
|
|
97
|
+
"trace"
|
|
98
|
+
]
|
|
99
|
+
},
|
|
100
|
+
{
|
|
101
|
+
"key": "action",
|
|
102
|
+
"title_zh": "适用范围与政策行动",
|
|
103
|
+
"title_en": "Applicability & Policy Action",
|
|
104
|
+
"lead_zh": "把可外推范围、护栏和政策动作连接到具体证据。",
|
|
105
|
+
"lead_en": "Connect applicability, guardrails and policy actions to specific evidence.",
|
|
106
|
+
"modules": [
|
|
107
|
+
"applicability",
|
|
108
|
+
"intervention"
|
|
109
|
+
]
|
|
110
|
+
},
|
|
111
|
+
{
|
|
112
|
+
"key": "evaluation",
|
|
113
|
+
"title_zh": "试点设计、评估与停止条件",
|
|
114
|
+
"title_en": "Pilot, Evaluation & Stop Conditions",
|
|
115
|
+
"lead_zh": "用独立效果测量验证试点,并预先写清停止条件。",
|
|
116
|
+
"lead_en": "Validate the pilot with independent effect measures and pre-specified stop conditions.",
|
|
117
|
+
"modules": [
|
|
118
|
+
"evaluation"
|
|
119
|
+
]
|
|
120
|
+
},
|
|
121
|
+
{
|
|
122
|
+
"key": "sources",
|
|
123
|
+
"title_zh": "来源、溯源与附录",
|
|
124
|
+
"title_en": "Sources, Traceability & Appendix",
|
|
125
|
+
"lead_zh": "保留原始来源、URL、证据 ID 和获取信息,确保可回查。",
|
|
126
|
+
"lead_en": "Preserve original sources, URLs, evidence IDs and retrieval metadata for auditability.",
|
|
127
|
+
"modules": [
|
|
128
|
+
"sources"
|
|
129
|
+
]
|
|
130
|
+
}
|
|
131
|
+
],
|
|
132
|
+
"brief_blocks": {
|
|
133
|
+
"zh": {
|
|
134
|
+
"decision": [
|
|
135
|
+
"先看结论",
|
|
136
|
+
"该不该做、置信度多高、最关键的证据边界在哪。"
|
|
137
|
+
],
|
|
138
|
+
"lieflat": [
|
|
139
|
+
"Lieflat 实证手作画廊",
|
|
140
|
+
"AI 按数据形状从 Lieflat 目录选型编排;每张图的数字都可溯源到 result.json。"
|
|
141
|
+
],
|
|
142
|
+
"outcomes": [
|
|
143
|
+
"过程产出 ≠ 政策效果",
|
|
144
|
+
"只展示真正有解释力的结果分离;正向、负向与零效应按 effect_direction 编码。"
|
|
145
|
+
],
|
|
146
|
+
"tribunal": [
|
|
147
|
+
"证据裁决",
|
|
148
|
+
"支持、不确定、被反驳与缺失证据分开放置,不把长段落平铺在同一层。"
|
|
149
|
+
],
|
|
150
|
+
"action": [
|
|
151
|
+
"从证据到行动",
|
|
152
|
+
"适用性、护栏、停止条件与评价连成一条可执行路径。"
|
|
153
|
+
],
|
|
154
|
+
"sources": [
|
|
155
|
+
"关键来源",
|
|
156
|
+
"摘要页只列最关键的来源;完整溯源在完整报告中展开。"
|
|
157
|
+
]
|
|
158
|
+
},
|
|
159
|
+
"en": {
|
|
160
|
+
"decision": [
|
|
161
|
+
"Decision first",
|
|
162
|
+
"What to do, how confident we are, and the most important evidence boundary."
|
|
163
|
+
],
|
|
164
|
+
"lieflat": [
|
|
165
|
+
"Lieflat Editorial Gallery",
|
|
166
|
+
"Charts selected and composed by AI from the Lieflat catalog; every number traces back to result.json."
|
|
167
|
+
],
|
|
168
|
+
"outcomes": [
|
|
169
|
+
"Process output ≠ policy effect",
|
|
170
|
+
"Only informative outcome separation; positive, negative and null effects use effect_direction."
|
|
171
|
+
],
|
|
172
|
+
"tribunal": [
|
|
173
|
+
"Evidence tribunal",
|
|
174
|
+
"Supported, uncertain, contradicted and missing evidence stay separated instead of flattened into long prose."
|
|
175
|
+
],
|
|
176
|
+
"action": [
|
|
177
|
+
"Evidence to action",
|
|
178
|
+
"Applicability, guardrails, stop conditions and evaluation form one executable path."
|
|
179
|
+
],
|
|
180
|
+
"sources": [
|
|
181
|
+
"Key sources",
|
|
182
|
+
"Only the key sources in the brief; full traceability expands in the full report."
|
|
183
|
+
]
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
}
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
{
|
|
2
|
+
"domain": "policy",
|
|
3
|
+
"purpose": "Domain terminology (policy voice). target_learners → 目标人群; intervention → 干预方案; no teaching vocabulary.",
|
|
4
|
+
"methodology_labels": {
|
|
5
|
+
"zh": {
|
|
6
|
+
"evidence_level": "证据等级",
|
|
7
|
+
"causal_identification": "因果识别",
|
|
8
|
+
"external_validity": "外部效度",
|
|
9
|
+
"cost_evidence": "成本证据",
|
|
10
|
+
"stakeholder_representation": "利益相关者代表",
|
|
11
|
+
"implementation_evidence": "实施证据",
|
|
12
|
+
"equity_analysis": "公平性分析",
|
|
13
|
+
"effect_size_reported": "效应量报告",
|
|
14
|
+
"uncertainty_quantified": "不确定性量化",
|
|
15
|
+
"comparator_clarity": "对照与反事实清晰",
|
|
16
|
+
"publication_bias_risk": "发表偏倚风险",
|
|
17
|
+
"generalizability_claims": "可推广性声明"
|
|
18
|
+
},
|
|
19
|
+
"en": {
|
|
20
|
+
"evidence_level": "Evidence level",
|
|
21
|
+
"causal_identification": "Causal identification",
|
|
22
|
+
"external_validity": "External validity",
|
|
23
|
+
"cost_evidence": "Cost evidence",
|
|
24
|
+
"stakeholder_representation": "Stakeholder representation",
|
|
25
|
+
"implementation_evidence": "Implementation evidence",
|
|
26
|
+
"equity_analysis": "Equity analysis",
|
|
27
|
+
"effect_size_reported": "Effect size reported",
|
|
28
|
+
"uncertainty_quantified": "Uncertainty quantified",
|
|
29
|
+
"comparator_clarity": "Comparator clarity",
|
|
30
|
+
"publication_bias_risk": "Publication-bias risk",
|
|
31
|
+
"generalizability_claims": "Generalizability claims"
|
|
32
|
+
}
|
|
33
|
+
},
|
|
34
|
+
"field_labels": {
|
|
35
|
+
"zh": {
|
|
36
|
+
"target_learners": "目标人群",
|
|
37
|
+
"target_population": "目标人群",
|
|
38
|
+
"ai_usage_policy": "干预使用规则",
|
|
39
|
+
"ai_usage_rule": "干预规则",
|
|
40
|
+
"task_vs_learning_guard": "产出 vs 效果护栏",
|
|
41
|
+
"learning_metrics": "效果指标",
|
|
42
|
+
"process_metrics": "过程指标",
|
|
43
|
+
"risk_metrics": "风险指标",
|
|
44
|
+
"task_performance": "过程产出",
|
|
45
|
+
"learning_gain": "目标效果",
|
|
46
|
+
"retention": "持续性",
|
|
47
|
+
"transfer": "扩散效果"
|
|
48
|
+
},
|
|
49
|
+
"en": {
|
|
50
|
+
"target_learners": "Target population",
|
|
51
|
+
"target_population": "Target population",
|
|
52
|
+
"ai_usage_policy": "Intervention usage policy",
|
|
53
|
+
"ai_usage_rule": "Intervention rule",
|
|
54
|
+
"task_vs_learning_guard": "Output vs effect guard",
|
|
55
|
+
"learning_metrics": "Effect metrics",
|
|
56
|
+
"process_metrics": "Process metrics",
|
|
57
|
+
"risk_metrics": "Risk metrics",
|
|
58
|
+
"task_performance": "Process output",
|
|
59
|
+
"learning_gain": "Target effect",
|
|
60
|
+
"retention": "Durability",
|
|
61
|
+
"transfer": "Spillover"
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
}
|
package/engine/capabilities.py
CHANGED
|
@@ -19,20 +19,32 @@ class CapabilitySpec:
|
|
|
19
19
|
deterministic_local: bool
|
|
20
20
|
scientific_gate: str | None
|
|
21
21
|
workflow_ids: tuple[str, ...] = ()
|
|
22
|
+
experimental: bool = False
|
|
22
23
|
|
|
23
24
|
|
|
24
25
|
_REGISTRY: dict[str, CapabilitySpec] = {}
|
|
25
26
|
|
|
27
|
+
#: Jev / SemDecide Tier-0 acceleration IDs (opt-in via --experimental).
|
|
28
|
+
#: Not role-owned scientific work: omitted from capability_registry() by
|
|
29
|
+
#: default so the protocol role-ownership gate stays unchanged. Pass
|
|
30
|
+
#: include_experimental=True when wiring skill/workflows/experimental-jev.md.
|
|
31
|
+
EXPERIMENTAL_CAPABILITY_IDS: frozenset[str] = frozenset({
|
|
32
|
+
"content_screen", "semantic_rerank", "field_extract",
|
|
33
|
+
"classify_check", "claim_verify",
|
|
34
|
+
})
|
|
35
|
+
|
|
26
36
|
|
|
27
37
|
def _register(capability_id: str, *, input_contracts: tuple[str, ...],
|
|
28
38
|
output_contracts: tuple[str, ...], deterministic_local: bool,
|
|
29
|
-
scientific_gate: str | None = None
|
|
39
|
+
scientific_gate: str | None = None,
|
|
40
|
+
experimental: bool = False) -> None:
|
|
30
41
|
_REGISTRY[capability_id] = CapabilitySpec(
|
|
31
42
|
capability_id=capability_id,
|
|
32
43
|
input_contracts=input_contracts,
|
|
33
44
|
output_contracts=output_contracts,
|
|
34
45
|
deterministic_local=deterministic_local,
|
|
35
46
|
scientific_gate=scientific_gate,
|
|
47
|
+
experimental=experimental,
|
|
36
48
|
)
|
|
37
49
|
|
|
38
50
|
|
|
@@ -90,10 +102,50 @@ _register("intervention_design", input_contracts=("decision-snapshot",),
|
|
|
90
102
|
_register("evaluation_design", input_contracts=("decision-snapshot",),
|
|
91
103
|
output_contracts=("evaluation",), deterministic_local=False)
|
|
92
104
|
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
105
|
+
# Jev experimental Tier-0 (opt-in --experimental; mode 1/2/3).
|
|
106
|
+
# Acceleration only — never replaces Skeptic, pre_verdict_gate, methodology
|
|
107
|
+
# audit, or tribunal. Uncertain / invalid answers fail closed to the large model.
|
|
108
|
+
_register("content_screen", input_contracts=("fetch-result",),
|
|
109
|
+
output_contracts=("screen-verdict",), deterministic_local=False,
|
|
110
|
+
scientific_gate="experimental Tier-0; advisory screen before context; "
|
|
111
|
+
"RULE 2 snippet != evidence",
|
|
112
|
+
experimental=True)
|
|
113
|
+
_register("semantic_rerank", input_contracts=("source",),
|
|
114
|
+
output_contracts=("ranked-candidates",), deterministic_local=False,
|
|
115
|
+
scientific_gate="experimental Tier-0; ranking is discovery order only",
|
|
116
|
+
experimental=True)
|
|
117
|
+
_register("field_extract", input_contracts=("source",),
|
|
118
|
+
output_contracts=("field-values",), deterministic_local=False,
|
|
119
|
+
scientific_gate="experimental Tier-0; verbatim substring values only, "
|
|
120
|
+
"never model-written fields",
|
|
121
|
+
experimental=True)
|
|
122
|
+
_register("classify_check", input_contracts=("source",),
|
|
123
|
+
output_contracts=("classification",), deterministic_local=False,
|
|
124
|
+
scientific_gate="experimental Tier-0; labels never replace "
|
|
125
|
+
"methodology_appraisal or outcome separation",
|
|
126
|
+
experimental=True)
|
|
127
|
+
_register("claim_verify", input_contracts=("claim", "source"),
|
|
128
|
+
output_contracts=("claim-verdict",), deterministic_local=False,
|
|
129
|
+
scientific_gate="experimental Tier-0; does NOT replace Skeptic nine "
|
|
130
|
+
"checks or Pre-Verdict Gate",
|
|
131
|
+
experimental=True)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def capability_registry(*, include_experimental: bool = False) -> dict[str, CapabilitySpec]:
|
|
135
|
+
"""Return the capability registry.
|
|
136
|
+
|
|
137
|
+
Experimental Jev Tier-0 IDs are omitted by default (they are acceleration
|
|
138
|
+
helpers, not role-owned scientific stages). Opt in with
|
|
139
|
+
``include_experimental=True`` when resolving --experimental modes 1/2/3.
|
|
140
|
+
"""
|
|
141
|
+
if include_experimental:
|
|
142
|
+
return dict(_REGISTRY)
|
|
143
|
+
return {cid: spec for cid, spec in _REGISTRY.items() if not spec.experimental}
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def experimental_capability_registry() -> dict[str, CapabilitySpec]:
|
|
147
|
+
"""Return only the Jev/SemDecide Tier-0 experimental capabilities."""
|
|
148
|
+
return {cid: spec for cid, spec in _REGISTRY.items() if spec.experimental}
|
|
97
149
|
|
|
98
150
|
|
|
99
151
|
def capability(capability_id: str) -> CapabilitySpec | None:
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
"""Single source of truth for the
|
|
1
|
+
"""Single source of truth for the four-state decision gate.
|
|
2
2
|
|
|
3
3
|
The four-state decision (ADOPT / PILOT / REJECT / INSUFFICIENT EVIDENCE) is
|
|
4
4
|
computed by two layers that must never drift apart:
|
|
@@ -12,6 +12,15 @@ liked. Both layers now resolve the 'which outcome categories count for this
|
|
|
12
12
|
domain' question here, and the V1 gate additionally re-derives direct-evidence
|
|
13
13
|
presence from the pack's own evidence records.
|
|
14
14
|
|
|
15
|
+
Decision matrix (gate-enforced):
|
|
16
|
+
|
|
17
|
+
conflict/mixed/support+oppose -> INSUFFICIENT_EVIDENCE (unresolved_conflict)
|
|
18
|
+
oppose_adoption (oppose-only) -> REJECT
|
|
19
|
+
High + support + direct_primary -> ADOPT
|
|
20
|
+
High + support + missing direct -> PILOT (downgrade_reason=missing_direct_primary)
|
|
21
|
+
Moderate+ support (+/- direct) -> PILOT (confidence_band | missing_direct_primary)
|
|
22
|
+
Low / 无支持 / 间接无 Moderate -> INSUFFICIENT_EVIDENCE
|
|
23
|
+
|
|
15
24
|
Stdlib only, consistent with the "Native Core" policy of engine/.
|
|
16
25
|
"""
|
|
17
26
|
from __future__ import annotations
|
|
@@ -33,6 +42,18 @@ ADOPT_DIRECTNESS = 2
|
|
|
33
42
|
#: Confidence band required before ADOPT is possible at all.
|
|
34
43
|
ADOPT_REQUIRED_LABEL = "High"
|
|
35
44
|
|
|
45
|
+
#: Confidence band that can still reach PILOT given a verifiable primary path.
|
|
46
|
+
PILOT_REQUIRED_LABEL = "Moderate"
|
|
47
|
+
|
|
48
|
+
#: Auditable reasons a decision was stepped down from a higher action.
|
|
49
|
+
DOWNGRADE_MISSING_DIRECT_PRIMARY = "missing_direct_primary"
|
|
50
|
+
DOWNGRADE_CONFIDENCE_BAND = "confidence_band"
|
|
51
|
+
DOWNGRADE_GATE_CRITICAL = "gate_critical"
|
|
52
|
+
DOWNGRADE_UNRESOLVED_CONFLICT = "unresolved_conflict"
|
|
53
|
+
|
|
54
|
+
#: Relation values that mark unresolved conflict rather than a clean vote.
|
|
55
|
+
CONFLICT_RELATIONS = frozenset({"conflict", "mixed"})
|
|
56
|
+
|
|
36
57
|
|
|
37
58
|
def primary_effect_categories(domain: str) -> tuple[str, ...]:
|
|
38
59
|
"""Categories that satisfy the ADOPT direct-evidence gate for a domain."""
|
|
@@ -72,25 +93,75 @@ def outcome_category(domain: str, value: str, primary: tuple[str, ...]) -> str |
|
|
|
72
93
|
return None
|
|
73
94
|
|
|
74
95
|
|
|
75
|
-
def
|
|
76
|
-
|
|
77
|
-
"""Gate-enforced decision action
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
96
|
+
def decision_outcome(*, confidence_label: str, decisive_relations: dict[str, str],
|
|
97
|
+
has_direct_primary_evidence: bool) -> dict[str, str | None]:
|
|
98
|
+
"""Gate-enforced decision action plus auditable downgrade_reason.
|
|
99
|
+
|
|
100
|
+
Returns ``{"action": <uppercase four-state>, "downgrade_reason": <str|None>}``.
|
|
101
|
+
|
|
102
|
+
REJECT requires usable direct opposition evidence and no decisive support
|
|
103
|
+
(oppose-only). Mixed support+oppose or any conflict/conditional/mixed
|
|
104
|
+
relation is an unresolved conflict → INSUFFICIENT_EVIDENCE (never ADOPT,
|
|
105
|
+
never a bare REJECT). ADOPT requires High confidence, decisive support,
|
|
106
|
+
AND direct evidence on the domain's PRIMARY outcome category. PILOT is
|
|
107
|
+
the bounded-trial exit: High + support missing that direct primary
|
|
108
|
+
evidence (``missing_direct_primary``), or Moderate + support on a
|
|
109
|
+
verifiable primary-outcome path (``confidence_band``).
|
|
86
110
|
"""
|
|
87
111
|
has_oppose = any(r == "oppose_adoption" for r in decisive_relations.values())
|
|
88
112
|
has_support = any(r == "support_adoption" for r in decisive_relations.values())
|
|
113
|
+
has_conflict = any(r in CONFLICT_RELATIONS for r in decisive_relations.values())
|
|
114
|
+
|
|
115
|
+
# Unresolved conflict (mixed votes or explicit conflict tags) fails closed
|
|
116
|
+
# before any ADOPT/PILOT/REJECT claim.
|
|
117
|
+
if has_conflict or (has_oppose and has_support):
|
|
118
|
+
return {"action": "INSUFFICIENT_EVIDENCE",
|
|
119
|
+
"downgrade_reason": DOWNGRADE_UNRESOLVED_CONFLICT}
|
|
120
|
+
|
|
89
121
|
if has_oppose:
|
|
90
|
-
return "REJECT"
|
|
122
|
+
return {"action": "REJECT", "downgrade_reason": None}
|
|
123
|
+
|
|
91
124
|
if (confidence_label == ADOPT_REQUIRED_LABEL and has_support
|
|
92
125
|
and has_direct_primary_evidence):
|
|
93
|
-
return "ADOPT"
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
126
|
+
return {"action": "ADOPT", "downgrade_reason": None}
|
|
127
|
+
|
|
128
|
+
if confidence_label == ADOPT_REQUIRED_LABEL and has_support:
|
|
129
|
+
return {"action": "PILOT", "downgrade_reason": DOWNGRADE_MISSING_DIRECT_PRIMARY}
|
|
130
|
+
|
|
131
|
+
# Bounded-trial exit: Moderate + support reaches PILOT even without a
|
|
132
|
+
# verified primary path (Gate C allows PILOT / INSUFFICIENT / REJECT).
|
|
133
|
+
if confidence_label == PILOT_REQUIRED_LABEL and has_support:
|
|
134
|
+
reason = (None if has_direct_primary_evidence
|
|
135
|
+
else DOWNGRADE_MISSING_DIRECT_PRIMARY)
|
|
136
|
+
if has_direct_primary_evidence:
|
|
137
|
+
reason = DOWNGRADE_CONFIDENCE_BAND
|
|
138
|
+
return {"action": "PILOT", "downgrade_reason": reason}
|
|
139
|
+
|
|
140
|
+
# Low / 间接 (Moderate without a verifiable primary path) -> INSUFFICIENT.
|
|
141
|
+
if confidence_label in ("Low", "Insufficient"):
|
|
142
|
+
return {"action": "INSUFFICIENT_EVIDENCE",
|
|
143
|
+
"downgrade_reason": DOWNGRADE_CONFIDENCE_BAND}
|
|
144
|
+
if confidence_label == PILOT_REQUIRED_LABEL and has_support:
|
|
145
|
+
return {"action": "INSUFFICIENT_EVIDENCE",
|
|
146
|
+
"downgrade_reason": DOWNGRADE_MISSING_DIRECT_PRIMARY}
|
|
147
|
+
return {"action": "INSUFFICIENT_EVIDENCE", "downgrade_reason": None}
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def decision_action(*, confidence_label: str, decisive_relations: dict[str, str],
|
|
151
|
+
has_direct_primary_evidence: bool) -> str:
|
|
152
|
+
"""Gate-enforced decision action (uppercase four-state)."""
|
|
153
|
+
return decision_outcome(
|
|
154
|
+
confidence_label=confidence_label,
|
|
155
|
+
decisive_relations=decisive_relations,
|
|
156
|
+
has_direct_primary_evidence=has_direct_primary_evidence,
|
|
157
|
+
)["action"]
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def downgrade_reason(*, confidence_label: str, decisive_relations: dict[str, str],
|
|
161
|
+
has_direct_primary_evidence: bool) -> str | None:
|
|
162
|
+
"""Auditable reason the action was stepped down (or None when not downgraded)."""
|
|
163
|
+
return decision_outcome(
|
|
164
|
+
confidence_label=confidence_label,
|
|
165
|
+
decisive_relations=decisive_relations,
|
|
166
|
+
has_direct_primary_evidence=has_direct_primary_evidence,
|
|
167
|
+
)["downgrade_reason"]
|