eduevidence 6.0.0 → 6.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +105 -0
- package/README.md +93 -38
- package/README.zh-CN.md +26 -6
- package/SKILL.md +11 -2
- package/assets/readme/landing-tour.gif +0 -0
- package/assets/readme/studio-tour.gif +0 -0
- package/bin/eduevidence.js +2 -1
- package/docs/architecture.md +319 -43
- package/docs/demo-workplace-ai.md +1 -1
- package/docs/install-guide.md +1 -1
- package/docs/orchestration-role-model.md +1 -1
- package/docs/release-closeout/README.md +1 -1
- package/docs/sciverse-api.md +125 -0
- package/eduevidence_cli.py +10 -0
- package/engine/decision_policy.py +96 -0
- package/engine/evidence_graph.py +14 -10
- package/engine/gaps.py +42 -22
- package/engine/ids.py +2 -0
- package/engine/library.py +6 -2
- package/engine/living.py +34 -4
- package/engine/migration.py +88 -3
- package/engine/orchestration.py +5 -5
- package/engine/paths.py +2 -0
- package/engine/pilot.py +34 -32
- package/engine/taxonomy.py +211 -0
- package/engine/tribunal.py +43 -31
- package/engine/versions.py +1 -1
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1360 -146
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
- package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
- package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
- package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +447 -127
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +447 -127
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +447 -127
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +447 -127
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +447 -127
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
- package/examples/ai-coding-assistant-evidence/result.json +13 -9
- package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
- package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
- package/examples/spaced-retrieval-practice/applicability.json +14 -0
- package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
- package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
- package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
- package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
- package/examples/spaced-retrieval-practice/frame.json +58 -0
- package/examples/spaced-retrieval-practice/gate_report.json +101 -0
- package/examples/spaced-retrieval-practice/methodology.json +78 -0
- package/examples/spaced-retrieval-practice/report_spec.json +212 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/result.json +942 -0
- package/examples/spaced-retrieval-practice/result.zh.json +942 -0
- package/examples/spaced-retrieval-practice/skeptic.json +70 -0
- package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
- package/examples/spaced-retrieval-practice/verdict.json +93 -0
- package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -4
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
- package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
- package/examples/workplace-ai-assistant/final_verdict.json +78 -0
- package/examples/workplace-ai-assistant/gate_report.json +101 -0
- package/examples/workplace-ai-assistant/report_spec.json +209 -40
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +435 -105
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +435 -105
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +435 -105
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +435 -105
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +435 -105
- package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/result.json +82 -20
- package/examples/workplace-ai-assistant/result.zh.json +82 -20
- package/examples/workplace-ai-assistant/skeptic.json +72 -0
- package/examples/workplace-ai-assistant/verdict.json +36 -10
- package/integrations/agent_mcp.py +2 -2
- package/package.json +12 -3
- package/pyproject.toml +4 -3
- package/references/report-copy-style.md +67 -0
- package/references/retrieval-compliance.md +75 -0
- package/references/retrieval-protocol.md +20 -0
- package/retrieval/audit.py +27 -3
- package/retrieval/fetch.py +96 -0
- package/retrieval/sciverse.py +398 -0
- package/retrieval/search.py +47 -7
- package/schemas/applicability.schema.json +94 -0
- package/schemas/chart-spec.schema.json +10 -3
- package/schemas/evidence.schema.json +316 -43
- package/schemas/fetch-result.schema.json +2 -1
- package/schemas/report-result.schema.json +3 -3
- package/schemas/report-spec.schema.json +98 -100
- package/schemas/skeptic.schema.json +86 -0
- package/schemas/source.schema.json +21 -2
- package/schemas/v2/finding.schema.json +5 -1
- package/schemas/v2/methodology-audit.schema.json +5 -1
- package/schemas/v2/outcome.schema.json +28 -5
- package/schemas/v2/study.schema.json +5 -1
- package/schemas/vNext/autoevolve-session.schema.json +34 -1
- package/schemas/vNext/eval-snapshot.schema.json +77 -1
- package/schemas/vNext/execution-plan.schema.json +50 -1
- package/schemas/vNext/gap-priority.schema.json +54 -1
- package/schemas/vNext/negative-search-record.schema.json +68 -1
- package/schemas/vNext/research-iteration.schema.json +87 -1
- package/schemas/vNext/research-strategy.schema.json +62 -1
- package/schemas/vNext/skill-experiment.schema.json +90 -1
- package/schemas/vNext/task-spec.schema.json +156 -1
- package/schemas/vNext/worker-result.schema.json +60 -1
- package/schemas/verdict.schema.json +164 -28
- package/scripts/build_esl_artifacts.py +2 -2
- package/scripts/build_report_variants.py +18 -2
- package/scripts/build_result.py +74 -9
- package/scripts/check_package_parity.py +85 -0
- package/scripts/check_protocol_alignment.py +375 -0
- package/scripts/check_versioned_schemas.py +254 -0
- package/scripts/claim_audit.py +13 -8
- package/scripts/compute_confidence.py +10 -0
- package/scripts/did_regression.py +12 -2
- package/scripts/evidence_score.py +5 -2
- package/scripts/generate_new_projects.py +4 -4
- package/scripts/orchestrator.py +120 -24
- package/scripts/pre_verdict_gate.py +224 -26
- package/scripts/quickstart.py +18 -2
- package/scripts/run_workspace.py +7 -1
- package/scripts/skill_payload.py +4 -1
- package/scripts/test_adversarial_empirical.py +26 -19
- package/scripts/validate_schema.py +31 -1
- package/skill/agents/evaluation-designer.md +20 -4
- package/skill/agents/evidence-analyst.md +19 -3
- package/skill/agents/evidence-judge.md +50 -2
- package/skill/agents/evidence-retriever.md +20 -3
- package/skill/agents/intervention-designer.md +20 -4
- package/skill/agents/method-reviewer.md +18 -2
- package/skill/agents/{education-planner.md → research-planner.md} +19 -3
- package/skill/agents/skeptic.md +18 -2
- package/skill/roles/registry.yaml +11 -11
- package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
- package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
- package/skill/sub-skills/data-analysis/SKILL.md +34 -15
- package/skill/sub-skills/ethics-review/SKILL.md +33 -10
- package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
- package/skill/sub-skills/evidence-review/SKILL.md +31 -12
- package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
- package/skill/sub-skills/literature-review/SKILL.md +35 -14
- package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
- package/skill/sub-skills/report-generation/SKILL.md +28 -0
- package/skill/sub-skills/research-planning/SKILL.md +41 -14
- package/skill/sub-skills/study-design/SKILL.md +30 -9
- package/skill/task-briefs/adjudicate.md +32 -7
- package/skill/task-briefs/applicability.md +37 -2
- package/skill/task-briefs/audit.md +32 -7
- package/skill/task-briefs/challenge.md +34 -5
- package/skill/task-briefs/evaluate.md +30 -5
- package/skill/task-briefs/extract.md +31 -8
- package/skill/task-briefs/frame.md +39 -10
- package/skill/task-briefs/intervene.md +32 -6
- package/skill/task-briefs/present.md +32 -8
- package/skill/task-briefs/projection.md +36 -2
- package/skill/task-briefs/retrieve.md +36 -6
- package/skill/workflows/decision-and-pilot.md +76 -1
- package/skill/workflows/evaluate-and-update.md +83 -0
- package/skill/workflows/evidence-review.md +104 -0
- package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
- package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
- package/visualization/eduevidence-report/scripts/build_report.py +512 -65
- package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
- package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
- package/web/architecture.html +14885 -0
- package/web/studio/assets/index-B8tkF44Q.css +1 -0
- package/web/studio/index.html +2 -2
- package/web/studio/assets/index-CzXocaGv.css +0 -1
- /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
# Sciverse 检索通道(API 契约存档)
|
|
2
|
+
|
|
3
|
+
本条记录 EduEvidence 接入 Sciverse 开放平台所用的最小契约,依据官方 `openapi.yaml` **v0.14.2**(`opendatalab/Sciverse-Agent-Tools`)整理。目的是让检索链在**没有网络**时也能被复核:字段名、错误码、限制与语义都在这里,代码变更需同步更新本文件。
|
|
4
|
+
|
|
5
|
+
## 1. 基本信息
|
|
6
|
+
|
|
7
|
+
| 项 | 值 |
|
|
8
|
+
|---|---|
|
|
9
|
+
| Base URL | `https://api.sciverse.space` |
|
|
10
|
+
| 鉴权 | `Authorization: Bearer <SCIVERSE_API_TOKEN>`(HTTP Bearer) |
|
|
11
|
+
| Token 来源 | 控制台 Tokens 页;同账号可通用于 Sciverse / DianShi / SeqStudio 已开通能力 |
|
|
12
|
+
| 实现 | `retrieval/sciverse.py`(stdlib-only) |
|
|
13
|
+
| 通道类型 | key-based 学术通道,在 `MultiSearchRouter.academic_key_providers` 中优先于零配置学术通道 |
|
|
14
|
+
| 无 token 行为 | 通道静默失活(`SCIVERSE_UNAVAILABLE`),零配置通道继续工作 |
|
|
15
|
+
|
|
16
|
+
## 2. 使用的四个端点
|
|
17
|
+
|
|
18
|
+
### 2.1 `POST /meta-search` — 结构化元数据检索
|
|
19
|
+
|
|
20
|
+
用于 Source 级命中:标题、作者、摘要、期刊、年份、DOI。
|
|
21
|
+
|
|
22
|
+
- 请求(所用子集):`collection`(papers/authors/sources)、`query`(BM25)、`filters_advanced[]`(`{field, operator, value}`)、`page`、`page_size`(≤50)。
|
|
23
|
+
- 响应:`results[]`(注意是 `results`,不是 `hits`)、`total_count`、`page`、`page_size`、`total_pages`、`next_cursor`。
|
|
24
|
+
- 关键字段:`unique_id`(元数据全局唯一 ID,**始终存在**)、`doc_id`(全文内容哈希 sha256,**仅当存在全文**)、`is_content_accessible`、`doi`、`author[].name`、`publication_published_year`、`publication_venue_name_unified`。
|
|
25
|
+
- 过滤操作符:`FILTER_OP_EQ/NE/GT/GTE/LT/LTE/IN/NIN/CONTAINS/MATCH/MATCH_PHRASE`;`doi` 用 `EQ`(服务端去 `doi.org` 前缀并转小写后精确匹配)。
|
|
26
|
+
- 排序与加权:`sort_by_year`(`auto` 在带 query 时保留相关性排序)、`freshness_boost` / `impact_boost` / `language_affinity`(`NONE|MILD|STRONG`,仅在 query 非空时生效;加权生效时**不支持深翻页**)。
|
|
27
|
+
|
|
28
|
+
### 2.2 `POST /agentic-search` — 自然语言语义检索(RAG)
|
|
29
|
+
|
|
30
|
+
用于 chunk 级**定位子**:
|
|
31
|
+
|
|
32
|
+
- 请求:`query`(1–200 字最佳)、`top_k`(1–100)、`mode`(`fast` ~200ms / `balanced` ~600ms / `quality` ~2–4s)、可选 `filters`、`source_types`(`web` / `pdf`)。
|
|
33
|
+
- 响应:`hits[]`,每条含 `chunk_id`、`doc_id`、`score`、`title`、`offset`(**Unicode 码点**,可直接作为 `/content` 的 offset)、`page_no`、`source_type`、`chunk`、`abstract`。
|
|
34
|
+
- **限制**:`balanced` 模式服务端约截断至 50 条;同一篇论文最多返回约 3 个 chunk,因此高 `top_k` 需要足够多的不同论文。
|
|
35
|
+
- **软过滤语义**:`filters` 在召回阶段与语义检索同时下推,但 chunk 侧元数据缺失的文档不会被排除(按年份过滤时,缺年份的 chunk 仍可能返回)。结论表述必须写"近似范围";需要严格范围时改用 `meta-search` 的结构化过滤并核对返回记录。
|
|
36
|
+
- 唯一的硬过滤字段是 `doc_id`(命中绝不越出给定集合;去重后上限默认 1000)。
|
|
37
|
+
|
|
38
|
+
### 2.3 `GET /content` — 按码点区间读原文
|
|
39
|
+
|
|
40
|
+
把 chunk 定位子扩展为可抽取的正文:
|
|
41
|
+
|
|
42
|
+
- 参数:`doc_id`(必填)、`offset`(默认 0)、`limit`(默认 4096,服务端上限 524288,超出静默钳制)。
|
|
43
|
+
- **必须显式传 `offset`**:省略时服务端返回整篇全文并忽略 `limit`。
|
|
44
|
+
- 响应:`text`、`bytes_returned`(UTF-8 字节数,仅供参考)、`next_offset`(下一段起始码点,翻页用它而非字节数)、`more`。
|
|
45
|
+
- 单位:`offset` / `limit` 均以 **Unicode 码点**计,与 Python `len` 一致,不是字节。
|
|
46
|
+
|
|
47
|
+
### 2.4 `POST /meta-paper-relations` — 引用 / 被引 / 相关工作
|
|
48
|
+
|
|
49
|
+
用于引文链审计(`SearchQuery.purpose = citation_chain`):
|
|
50
|
+
|
|
51
|
+
- 请求:`unique_id`(**不是 doc_id**)、`relation`(`CITATIONS` 被引 / `REFERENCES` 参考文献 / `RELATED_WORKS`)、`page`、`page_size`(≤200)。
|
|
52
|
+
- 响应:`items[]`(`id` / `id_type` / `title`)、`total_count`、`page`、`page_size`、`total_pages`。
|
|
53
|
+
- 方向语义:`CITATIONS` 是"谁引用了我",`REFERENCES` 是"我引用了谁",两者相反。
|
|
54
|
+
- **上限**:关系数超 10000 返回 `429`;`page × page_size` 超 10000 返回 `400`。两种情况改用 `meta-search` 的 `references_unique_id` 反查(支持深翻页与任意排序)。
|
|
55
|
+
- `total_count` 只统计库内命中,与论文自身 `citation_count` 可能有约 ±1% 差异。
|
|
56
|
+
|
|
57
|
+
## 3. 错误处理与状态映射
|
|
58
|
+
|
|
59
|
+
`retrieval/sciverse.py` 把 HTTP 与网络异常统一映射为**定型状态**,绝不把异常抛进检索管道:
|
|
60
|
+
|
|
61
|
+
| HTTP / 情形 | 状态 | 管道行为 |
|
|
62
|
+
|---|---|---|
|
|
63
|
+
| 200 | `ok` | 正常解析 |
|
|
64
|
+
| 401 / 403 | `SCIVERSE_UNAUTHORIZED` | 记录尝试,切换其他通道 |
|
|
65
|
+
| 400 / 404 / 429 | `SCIVERSE_BAD_REQUEST` | 同上(429 视为配额/上限耗尽) |
|
|
66
|
+
| 502 / 503 / 5xx | `SCIVERSE_UPSTREAM_ERROR` | 同上 |
|
|
67
|
+
| 超时 / DNS / TLS | `SCIVERSE_NETWORK_ERROR` | 同上 |
|
|
68
|
+
| 未配置 token | `SCIVERSE_UNAVAILABLE` | 通道失活,不产生请求 |
|
|
69
|
+
|
|
70
|
+
错误信息只保留服务端 `ApiError.message` 或状态描述,**不携带 Authorization 头或 token 片段**。
|
|
71
|
+
|
|
72
|
+
## 4. 在证据链中的位置(RULE 2 的机器化)
|
|
73
|
+
|
|
74
|
+
```text
|
|
75
|
+
/meta-search → Source 级命中(DOI → https://doi.org/<doi>;无 DOI 则标记 needs_manual_location)
|
|
76
|
+
/agentic-search → chunk 定位子(doc_id + offset)→ 写入 chunks.jsonl,标注 discovery_only_requires_content_fetch
|
|
77
|
+
/content → 按定位读原文 → 经 retrieval/validate.py 校验门 → 才可进入 Extract
|
|
78
|
+
/meta-paper-relations → 引文链记录(写入审计导出,供筛选与饱和判断)
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
要点:**chunk 不是证据**。它是发现线索;只有 `/content` 读到的正文通过校验门(长度、错误页、登录页、验证码、标题匹配等)之后,才允许被抽取为 Evidence Object。这条纪律由 `retrieval/fetch.py::fetch_sciverse_content()` 实现,产出与 `FetchResult` 同形,下游零改动。
|
|
82
|
+
|
|
83
|
+
引用目标永远是论文本身(DOI / `unique_id`),**不是 Sciverse** —— 它是读取路径,不是来源。
|
|
84
|
+
|
|
85
|
+
## 5. 配置
|
|
86
|
+
|
|
87
|
+
```bash
|
|
88
|
+
export SCIVERSE_API_TOKEN=sv-... # 必需;未设置时通道失活
|
|
89
|
+
# 可选:指向测试环境
|
|
90
|
+
export SCIVERSE_BASE_URL=https://api.sciverse.space
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
宿主侧如已安装官方 Skill / MCP,可作为**补充**而非替代(本仓库的实现不依赖它):
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
npx skills add https://sciverse.space
|
|
97
|
+
# 或 MCP server
|
|
98
|
+
npm install -g sciverse-mcp-server
|
|
99
|
+
export SCIVERSE_API_TOKEN=sv-...
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
## 6. 验证
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
# 契约与降级行为(离线,mock HTTP)
|
|
106
|
+
python -m pytest tests/test_sciverse_channel.py -q
|
|
107
|
+
|
|
108
|
+
# 真实连通冒烟(需 token)
|
|
109
|
+
python - <<'PY'
|
|
110
|
+
from retrieval import sciverse as s
|
|
111
|
+
print('available:', s.available())
|
|
112
|
+
print('meta:', s.meta_search('generative AI coding assistants learning outcomes', limit=3).status)
|
|
113
|
+
r = s.agentic_search('unguarded GPT-4 access harms independent exam performance', top_k=3)
|
|
114
|
+
recs = s.chunk_records(r)
|
|
115
|
+
print('chunks:', len(recs))
|
|
116
|
+
if recs:
|
|
117
|
+
c = s.read_content(recs[0]['doc_id'], offset=recs[0]['offset'], limit=600)
|
|
118
|
+
print('content:', c.status, len(c.data.get('text') or ''))
|
|
119
|
+
PY
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
## 7. 合规
|
|
123
|
+
|
|
124
|
+
配额、限速、缓存与署名规则统一见 `references/retrieval-compliance.md`(§3 Sciverse 附加约定)。规范漂移时以官方 `openapi.yaml` 为准,并同步更新本文件与 `references/retrieval-protocol.md`。
|
|
125
|
+
|
package/eduevidence_cli.py
CHANGED
|
@@ -8,6 +8,16 @@ only two intercepted command domains:
|
|
|
8
8
|
eduevidence evolve ...
|
|
9
9
|
"""
|
|
10
10
|
import sys
|
|
11
|
+
|
|
12
|
+
MIN_PYTHON = (3, 10)
|
|
13
|
+
|
|
14
|
+
if sys.version_info < MIN_PYTHON:
|
|
15
|
+
import platform
|
|
16
|
+
sys.stderr.write(
|
|
17
|
+
f"EduEvidence requires Python {MIN_PYTHON[0]}.{MIN_PYTHON[1]} or newer; "
|
|
18
|
+
f"this interpreter is {platform.python_version()}.\n"
|
|
19
|
+
"Install a supported Python and re-run, or set PYTHON to one.\n")
|
|
20
|
+
raise SystemExit(2)
|
|
11
21
|
from pathlib import Path
|
|
12
22
|
|
|
13
23
|
ROOT = Path(__file__).resolve().parent
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
"""Single source of truth for the ADOPT direct-evidence gate.
|
|
2
|
+
|
|
3
|
+
The four-state decision (ADOPT / PILOT / REJECT / INSUFFICIENT EVIDENCE) is
|
|
4
|
+
computed by two layers that must never drift apart:
|
|
5
|
+
|
|
6
|
+
* engine/tribunal.py - V2 Evidence Graph adjudication
|
|
7
|
+
* scripts/pre_verdict_gate.py - V1 run/example-pack gate enforcement
|
|
8
|
+
|
|
9
|
+
Before this module existed the rule lived only inside the tribunal, so the V1
|
|
10
|
+
gate could not enforce it and a hand-written verdict could carry any action it
|
|
11
|
+
liked. Both layers now resolve the 'which outcome categories count for this
|
|
12
|
+
domain' question here, and the V1 gate additionally re-derives direct-evidence
|
|
13
|
+
presence from the pack's own evidence records.
|
|
14
|
+
|
|
15
|
+
Stdlib only, consistent with the "Native Core" policy of engine/.
|
|
16
|
+
"""
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
#: Category buckets that count as a domain's PRIMARY effect for the ADOPT gate.
|
|
20
|
+
#: The gate asks 'is there direct evidence on the outcome this decision is
|
|
21
|
+
#: actually about?' - for education that is a learning outcome (task
|
|
22
|
+
#: performance and process measures never qualify); for policy it is the
|
|
23
|
+
#: policy-effectiveness / cost class. Every entry must name a category the
|
|
24
|
+
#: domain registry declares, which check_protocol_alignment.py enforces.
|
|
25
|
+
PRIMARY_EFFECT_CATEGORIES: dict[str, tuple[str, ...]] = {
|
|
26
|
+
"education": ("learning",),
|
|
27
|
+
"policy": ("effectiveness", "cost"),
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
#: Directness (0-2) at which an evidence link may carry an ADOPT claim.
|
|
31
|
+
ADOPT_DIRECTNESS = 2
|
|
32
|
+
|
|
33
|
+
#: Confidence band required before ADOPT is possible at all.
|
|
34
|
+
ADOPT_REQUIRED_LABEL = "High"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def primary_effect_categories(domain: str) -> tuple[str, ...]:
|
|
38
|
+
"""Categories that satisfy the ADOPT direct-evidence gate for a domain."""
|
|
39
|
+
from engine.taxonomy import categories as taxonomy_categories
|
|
40
|
+
|
|
41
|
+
declared = PRIMARY_EFFECT_CATEGORIES.get(domain)
|
|
42
|
+
if declared:
|
|
43
|
+
known = taxonomy_categories(domain)
|
|
44
|
+
missing = [c for c in declared if c not in known]
|
|
45
|
+
if missing:
|
|
46
|
+
raise ValueError(
|
|
47
|
+
'domain ' + repr(domain) + ' ADOPT gate references undeclared '
|
|
48
|
+
'categories ' + repr(missing) + '; declared: ' + repr(sorted(known)))
|
|
49
|
+
return declared
|
|
50
|
+
known = taxonomy_categories(domain)
|
|
51
|
+
if not known:
|
|
52
|
+
raise ValueError('domain ' + repr(domain) + ' declares no outcome categories')
|
|
53
|
+
return (next(iter(known)),)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def outcome_category(domain: str, value: str, primary: tuple[str, ...]) -> str | None:
|
|
57
|
+
"""Resolve an outcome value to its category, or None when unresolvable.
|
|
58
|
+
|
|
59
|
+
The outcomes table stores CATEGORY buckets (learning / task_performance /
|
|
60
|
+
process / risk, plus each domain's own buckets), while V1 packs store raw
|
|
61
|
+
taxonomy tokens. Accept a category directly and, for a token, resolve it
|
|
62
|
+
through the registry. An unknown value returns None so callers fail
|
|
63
|
+
closed instead of silently treating it as decision-grade evidence.
|
|
64
|
+
"""
|
|
65
|
+
from engine.taxonomy import TaxonomyError, category_of
|
|
66
|
+
|
|
67
|
+
if value in primary:
|
|
68
|
+
return value
|
|
69
|
+
try:
|
|
70
|
+
return category_of(domain, value)
|
|
71
|
+
except (TaxonomyError, ValueError):
|
|
72
|
+
return None
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def decision_action(*, confidence_label: str, decisive_relations: dict[str, str],
|
|
76
|
+
has_direct_primary_evidence: bool) -> str:
|
|
77
|
+
"""Gate-enforced decision action (uppercase four-state).
|
|
78
|
+
|
|
79
|
+
REJECT requires usable direct opposition evidence (an independent Study
|
|
80
|
+
folded to oppose_adoption). Low/Insufficient can never yield ADOPT.
|
|
81
|
+
ADOPT additionally requires direct evidence on the domain's PRIMARY
|
|
82
|
+
outcome category: High + decisive support WITHOUT such evidence downgrades
|
|
83
|
+
to PILOT - task performance and procedural efficiency are not
|
|
84
|
+
decision-grade effects. Moderate + decisive support -> PILOT; otherwise
|
|
85
|
+
INSUFFICIENT_EVIDENCE.
|
|
86
|
+
"""
|
|
87
|
+
has_oppose = any(r == "oppose_adoption" for r in decisive_relations.values())
|
|
88
|
+
has_support = any(r == "support_adoption" for r in decisive_relations.values())
|
|
89
|
+
if has_oppose:
|
|
90
|
+
return "REJECT"
|
|
91
|
+
if (confidence_label == ADOPT_REQUIRED_LABEL and has_support
|
|
92
|
+
and has_direct_primary_evidence):
|
|
93
|
+
return "ADOPT"
|
|
94
|
+
if confidence_label in ("High", "Moderate") and has_support:
|
|
95
|
+
return "PILOT"
|
|
96
|
+
return "INSUFFICIENT_EVIDENCE"
|
package/engine/evidence_graph.py
CHANGED
|
@@ -54,7 +54,11 @@ class EvidenceNode:
|
|
|
54
54
|
outcome_dimension: str = OutcomeDimension.GENERAL_MEASURE
|
|
55
55
|
claim_id: Optional[str] = None
|
|
56
56
|
outcome_id: Optional[str] = None
|
|
57
|
-
|
|
57
|
+
# A missing effect must stay missing: the old default fabricated a
|
|
58
|
+
# g = 0.0 with a p = 0.05, which would serialise as a real (and
|
|
59
|
+
# false) "no effect" result. Callers that need a number must supply
|
|
60
|
+
# one; the extractors already suppress charts when it is absent.
|
|
61
|
+
effect_size: Optional[Dict[str, Any]] = None
|
|
58
62
|
sample_size: int = 0
|
|
59
63
|
sample_description: str = ""
|
|
60
64
|
study_design: str = "Quasi-Experimental" # RCT, Quasi-Experimental DID, Meta-Analysis, Observational
|
|
@@ -292,9 +296,9 @@ class EvidenceGraph:
|
|
|
292
296
|
for ev in self.evidence.values():
|
|
293
297
|
paper = self.papers.get(ev.paper_id)
|
|
294
298
|
study_label = f"{paper.authors[0] if paper and paper.authors else ev.paper_id} ({paper.year if paper else ''})"
|
|
295
|
-
effect_val = ev.effect_size.get("value"
|
|
296
|
-
ci_l = ev.effect_size.get("ci_lower")
|
|
297
|
-
ci_u = ev.effect_size.get("ci_upper")
|
|
299
|
+
effect_val = (ev.effect_size or {}).get("value") or 0.0
|
|
300
|
+
ci_l = (ev.effect_size or {}).get("ci_lower")
|
|
301
|
+
ci_u = (ev.effect_size or {}).get("ci_upper")
|
|
298
302
|
has_ci = ci_l is not None and ci_u is not None and float(ci_u) >= float(ci_l)
|
|
299
303
|
points.append({
|
|
300
304
|
"evidence_id": ev.evidence_id,
|
|
@@ -327,13 +331,13 @@ class EvidenceGraph:
|
|
|
327
331
|
precision_counts = {"reported_ci": 0, "derived_from_sample_size": 0}
|
|
328
332
|
excluded_no_precision = 0
|
|
329
333
|
for n in nodes:
|
|
330
|
-
eff = n.effect_size.get("value")
|
|
334
|
+
eff = (n.effect_size or {}).get("value")
|
|
331
335
|
if eff is None or math.isnan(float(eff)) or math.isinf(float(eff)):
|
|
332
336
|
continue
|
|
333
337
|
|
|
334
338
|
# Statistical variance derivation (Borenstein et al. 2009)
|
|
335
|
-
ci_l = n.effect_size.get("ci_lower")
|
|
336
|
-
ci_u = n.effect_size.get("ci_upper")
|
|
339
|
+
ci_l = (n.effect_size or {}).get("ci_lower")
|
|
340
|
+
ci_u = (n.effect_size or {}).get("ci_upper")
|
|
337
341
|
if ci_l is not None and ci_u is not None and float(ci_u) > float(ci_l):
|
|
338
342
|
se = (float(ci_u) - float(ci_l)) / (2.0 * 1.95996)
|
|
339
343
|
precision_counts["reported_ci"] += 1
|
|
@@ -422,7 +426,7 @@ class EvidenceGraph:
|
|
|
422
426
|
"quote": p.summary,
|
|
423
427
|
})
|
|
424
428
|
for ev in self.evidence.values():
|
|
425
|
-
effect_val = ev.effect_size.get("value"
|
|
429
|
+
effect_val = (ev.effect_size or {}).get("value") or 0.0
|
|
426
430
|
symbol_size = max(18, min(45, int(18 + abs(effect_val) * 20)))
|
|
427
431
|
nodes.append({
|
|
428
432
|
"id": ev.evidence_id,
|
|
@@ -433,8 +437,8 @@ class EvidenceGraph:
|
|
|
433
437
|
"dimension": ev.outcome_dimension,
|
|
434
438
|
"direction": ev.direction,
|
|
435
439
|
"effect_size": effect_val,
|
|
436
|
-
"ci_lower": ev.effect_size.get("ci_lower", "N/A"),
|
|
437
|
-
"ci_upper": ev.effect_size.get("ci_upper", "N/A"),
|
|
440
|
+
"ci_lower": (ev.effect_size or {}).get("ci_lower", "N/A"),
|
|
441
|
+
"ci_upper": (ev.effect_size or {}).get("ci_upper", "N/A"),
|
|
438
442
|
"sample_size": ev.sample_size,
|
|
439
443
|
"wwc_rating": ev.wwc_rating,
|
|
440
444
|
"quote": ev.key_quote,
|
package/engine/gaps.py
CHANGED
|
@@ -21,10 +21,32 @@ from engine.graph_store import GraphStore
|
|
|
21
21
|
from engine.ids import new_local_id
|
|
22
22
|
from engine.synthesis import ClaimSynthesis
|
|
23
23
|
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
24
|
+
#: Gap kind -> the outcome categories that count as covering it. Categories are
|
|
25
|
+
#: read from the domain registry, so a policy run classifies its outcomes with
|
|
26
|
+
#: the policy buckets (effectiveness / cost / equity / feasibility / risk)
|
|
27
|
+
#: instead of being measured against education vocabulary.
|
|
28
|
+
GAP_KIND_CATEGORIES: dict[str, tuple[str, ...]] = {
|
|
29
|
+
"learning": ("learning",),
|
|
30
|
+
"retention": ("learning",),
|
|
31
|
+
"transfer": ("learning",),
|
|
32
|
+
"task_performance": ("task_performance",),
|
|
33
|
+
"process": ("process",),
|
|
34
|
+
"risk": ("risk",),
|
|
35
|
+
"effectiveness": ("effectiveness",),
|
|
36
|
+
"cost": ("cost",),
|
|
37
|
+
"equity": ("equity",),
|
|
38
|
+
"feasibility": ("feasibility",),
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
#: Legacy aliases accepted when a frame names its requested outcomes with older
|
|
42
|
+
#: vocabulary; they resolve to a gap kind above.
|
|
43
|
+
GAP_KIND_ALIASES: dict[str, str] = {
|
|
44
|
+
"long_term": "retention", "learning_retention": "retention",
|
|
45
|
+
"transfer_learning": "transfer", "far_transfer": "transfer",
|
|
46
|
+
"assignment_score": "task_performance", "task_completion": "task_performance",
|
|
47
|
+
"policy_effectiveness": "effectiveness", "cost_effectiveness": "cost",
|
|
48
|
+
"implementation_risk": "risk",
|
|
49
|
+
}
|
|
28
50
|
|
|
29
51
|
|
|
30
52
|
def _autoresearch_key(
|
|
@@ -97,26 +119,24 @@ def derive_gaps(*, store: GraphStore,
|
|
|
97
119
|
req_type = str(req.get("outcome_type", "")).lower()
|
|
98
120
|
else:
|
|
99
121
|
req_name, req_type = str(req).lower(), ""
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
if
|
|
103
|
-
|
|
104
|
-
if
|
|
105
|
-
return
|
|
106
|
-
|
|
107
|
-
return "learning", req.get("name", "") if isinstance(req, dict) else str(req)
|
|
108
|
-
return "other", req.get("name", "") if isinstance(req, dict) else str(req)
|
|
122
|
+
label = req.get("name", "") if isinstance(req, dict) else str(req)
|
|
123
|
+
kind = GAP_KIND_ALIASES.get(req_type) or GAP_KIND_ALIASES.get(req_name)
|
|
124
|
+
if kind is None:
|
|
125
|
+
kind = req_type if req_type in GAP_KIND_CATEGORIES else req_name
|
|
126
|
+
if kind in GAP_KIND_CATEGORIES:
|
|
127
|
+
return kind, label
|
|
128
|
+
return "other", label
|
|
109
129
|
|
|
110
130
|
def covered_for_kind(kind: str) -> bool:
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
if
|
|
118
|
-
return
|
|
119
|
-
return
|
|
131
|
+
"""True when the graph already carries an outcome for this gap kind.
|
|
132
|
+
|
|
133
|
+
``covered_types`` holds the category buckets stored on outcomes, which
|
|
134
|
+
is why this compares categories rather than V1 tokens.
|
|
135
|
+
"""
|
|
136
|
+
expected = GAP_KIND_CATEGORIES.get(kind)
|
|
137
|
+
if not expected:
|
|
138
|
+
return False
|
|
139
|
+
return bool(covered_types & set(expected))
|
|
120
140
|
|
|
121
141
|
seen: set[tuple[str, str]] = set()
|
|
122
142
|
for req in requested:
|
package/engine/ids.py
CHANGED
package/engine/library.py
CHANGED
|
@@ -8,6 +8,8 @@ changes an existing Project's conclusions — only an explicit import/sync
|
|
|
8
8
|
advances the Project graph.
|
|
9
9
|
"""
|
|
10
10
|
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
11
13
|
import hashlib
|
|
12
14
|
import json
|
|
13
15
|
import os
|
|
@@ -197,8 +199,10 @@ class ResearchLibrary:
|
|
|
197
199
|
outcomes.append({
|
|
198
200
|
"outcome_id": oid,
|
|
199
201
|
"name": f.get("measure", oid),
|
|
200
|
-
|
|
201
|
-
|
|
202
|
+
# No silent default: an unlabelled finding must not be filed as a
|
|
203
|
+
# learning outcome; that is how task-performance evidence used
|
|
204
|
+
# to reach the ADOPT gate on the V2 path.
|
|
205
|
+
"outcome_type": (f.get("extensions") or {}).get("outcome_type", ""),
|
|
202
206
|
"extensions": {
|
|
203
207
|
"auto_created_from_library_import": True,
|
|
204
208
|
"library_revision": lib_rev,
|
package/engine/living.py
CHANGED
|
@@ -227,6 +227,26 @@ def set_subscription_status(project: ProjectWorkspace, subscription_id: str,
|
|
|
227
227
|
return subscription
|
|
228
228
|
|
|
229
229
|
|
|
230
|
+
def _validated_outcome_type(value, domain: str) -> str:
|
|
231
|
+
"""Outcome token from a refresh payload, validated against the registry.
|
|
232
|
+
|
|
233
|
+
A missing or unknown token raises: the living path used to default to
|
|
234
|
+
a learning outcome, which silently promoted task-performance evidence.
|
|
235
|
+
"""
|
|
236
|
+
from engine.taxonomy import category_of, tokens as taxonomy_tokens
|
|
237
|
+
|
|
238
|
+
token = str(value or "").strip()
|
|
239
|
+
if not token:
|
|
240
|
+
raise ValueError(
|
|
241
|
+
"living evidence record must declare outcome_type; the engine "
|
|
242
|
+
"will not guess a category")
|
|
243
|
+
if token not in taxonomy_tokens(domain):
|
|
244
|
+
raise ValueError(
|
|
245
|
+
f"outcome_type {token!r} is not registered for domain {domain!r}")
|
|
246
|
+
category_of(domain, token) # fail closed on a malformed taxonomy
|
|
247
|
+
return token
|
|
248
|
+
|
|
249
|
+
|
|
230
250
|
def refresh(project: ProjectWorkspace, subscription_id: str, *,
|
|
231
251
|
new_evidence: list[dict] | None = None,
|
|
232
252
|
retriever: Callable[[dict], list[dict]] | None = None) -> dict:
|
|
@@ -299,8 +319,10 @@ def refresh(project: ProjectWorkspace, subscription_id: str, *,
|
|
|
299
319
|
|
|
300
320
|
# ---- normalize + validate fresh evidence ---------------------------
|
|
301
321
|
next_revision = store.active_revision() + 1
|
|
322
|
+
domain = str((project.manifest() or {}).get("domain") or "education")
|
|
302
323
|
mutation, new_hashes, finding_ids, summary_parts = _build_mutation(
|
|
303
|
-
project, store, subscription, fresh_packets, claims, next_revision
|
|
324
|
+
project, store, subscription, fresh_packets, claims, next_revision,
|
|
325
|
+
domain=domain)
|
|
304
326
|
|
|
305
327
|
revision = store.commit(
|
|
306
328
|
run_id=new_run_id(),
|
|
@@ -370,7 +392,7 @@ def _existing_drift_ids(project: ProjectWorkspace) -> set[str]:
|
|
|
370
392
|
|
|
371
393
|
|
|
372
394
|
def _build_mutation(project, store, subscription, fresh_packets, claims,
|
|
373
|
-
next_revision) -> tuple[GraphMutation, set[str], set[str], list[str]]:
|
|
395
|
+
next_revision, domain: str = "education") -> tuple[GraphMutation, set[str], set[str], list[str]]:
|
|
374
396
|
"""Normalize + validate each fresh record and assemble one GraphMutation."""
|
|
375
397
|
existing = {t: {row[_ID_KEY[t]] for row in store.read_table(t)}
|
|
376
398
|
for t in _GRAPH_TABLES}
|
|
@@ -418,6 +440,8 @@ def _build_mutation(project, store, subscription, fresh_packets, claims,
|
|
|
418
440
|
source_upserted = True
|
|
419
441
|
upserts["sources"].append(src)
|
|
420
442
|
|
|
443
|
+
# Outcome tokens are validated against the project domain registry
|
|
444
|
+
# rather than defaulted; see _validated_outcome_type().
|
|
421
445
|
# --- outcome (optional; reused when the id already exists) -------
|
|
422
446
|
outcome_upserted = False
|
|
423
447
|
outcome = None
|
|
@@ -439,7 +463,7 @@ def _build_mutation(project, store, subscription, fresh_packets, claims,
|
|
|
439
463
|
outcome = {
|
|
440
464
|
"outcome_id": out_id,
|
|
441
465
|
"name": outcome_pkt.get("name") or out_id[len("OUT-"):],
|
|
442
|
-
"outcome_type": outcome_pkt.get("outcome_type",
|
|
466
|
+
"outcome_type": _validated_outcome_type(outcome_pkt.get("outcome_type"), domain),
|
|
443
467
|
"extensions": outcome_pkt.get("extensions") or {},
|
|
444
468
|
}
|
|
445
469
|
existing["outcomes"].add(out_id)
|
|
@@ -503,7 +527,13 @@ def _build_mutation(project, store, subscription, fresh_packets, claims,
|
|
|
503
527
|
f"{label}: relation_to_claim must be one of "
|
|
504
528
|
f"{sorted(_RELATION_TO_IMPLICATION)}, got {relation!r}")
|
|
505
529
|
link.setdefault("decision_implication", _RELATION_TO_IMPLICATION[relation])
|
|
506
|
-
link.
|
|
530
|
+
# directness decides whether this link can carry an ADOPT claim.
|
|
531
|
+
# Defaulting it to 2 (direct) would let an unclassified refresh
|
|
532
|
+
# assert direct evidence it never established, so it is required.
|
|
533
|
+
if "directness" not in link:
|
|
534
|
+
raise ValueError(
|
|
535
|
+
f"{label}: directness is required for a living-evidence link "
|
|
536
|
+
"(2 = direct evidence for the claim; supply it explicitly)")
|
|
507
537
|
link.setdefault("applicability", {"scope_match": "direct"})
|
|
508
538
|
link.setdefault("reasoning_note",
|
|
509
539
|
f"living evidence refresh: {subscription['subscription_id']}")
|
package/engine/migration.py
CHANGED
|
@@ -30,6 +30,15 @@ from engine.versions import (
|
|
|
30
30
|
|
|
31
31
|
OUTCOME_TYPES = ("learning", "task_performance", "process", "risk")
|
|
32
32
|
|
|
33
|
+
#: V1 D5 Directness (0/1/2) -> V2 link directness + applicability scope_match.
|
|
34
|
+
#: Directness decides whether a link can carry an ADOPT claim, so a flat
|
|
35
|
+
#: hard-coded value silently capped every migrated pack at PILOT.
|
|
36
|
+
_D5_TO_DIRECTNESS = {
|
|
37
|
+
2: (2, "direct"),
|
|
38
|
+
1: (1, "partial"),
|
|
39
|
+
0: (1, "mismatch"),
|
|
40
|
+
}
|
|
41
|
+
|
|
33
42
|
_CLAIM_TO_IMPLICATION = {
|
|
34
43
|
"support": "support_adoption",
|
|
35
44
|
"contradict": "oppose_adoption",
|
|
@@ -75,6 +84,71 @@ def _map_decision_implication(ev: dict) -> str:
|
|
|
75
84
|
return _CLAIM_TO_IMPLICATION[_map_relation(ev.get("relation_to_claim") or ev.get("direction"))]
|
|
76
85
|
|
|
77
86
|
|
|
87
|
+
def _v1_effect_estimate(ev: dict) -> dict | None:
|
|
88
|
+
"""Effect magnitude from a V1 record, or None when it recorded none.
|
|
89
|
+
|
|
90
|
+
V1 kept numbers either on a top-level effect_size field or inside
|
|
91
|
+
extensions.raw_result; both are real sources, and absence stays None.
|
|
92
|
+
The V2 contract is narrow (metric + raw_text required, no extra keys), so
|
|
93
|
+
a V1 magnitude is normalized instead of copied verbatim: ci_lower/ci_upper
|
|
94
|
+
become ci_low/ci_high and keys the V2 contract does not declare are kept in
|
|
95
|
+
the raw_text so nothing recorded is silently lost.
|
|
96
|
+
"""
|
|
97
|
+
value = ev.get("effect_size")
|
|
98
|
+
if isinstance(value, dict) and value.get("value") is not None:
|
|
99
|
+
return _normalize_effect_estimate(value)
|
|
100
|
+
if isinstance(value, (int, float)):
|
|
101
|
+
return {"value": float(value), "source": "v1_effect_size"}
|
|
102
|
+
raw = (ev.get("extensions") or {}).get("raw_result")
|
|
103
|
+
if isinstance(raw, dict) and raw.get("value") is not None:
|
|
104
|
+
return _normalize_effect_estimate(raw)
|
|
105
|
+
return None
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _normalize_effect_estimate(raw: dict) -> dict:
|
|
109
|
+
"""Project a V1 magnitude onto the V2 effect_estimate contract."""
|
|
110
|
+
known = ("metric", "value", "unit", "ci_low", "ci_high", "p_value")
|
|
111
|
+
out: dict = {}
|
|
112
|
+
for key in known:
|
|
113
|
+
if key in raw and raw[key] is not None:
|
|
114
|
+
out[key] = raw[key]
|
|
115
|
+
for legacy, canonical in (("ci_lower", "ci_low"), ("ci_upper", "ci_high")):
|
|
116
|
+
if canonical not in out and raw.get(legacy) is not None:
|
|
117
|
+
out[canonical] = raw[legacy]
|
|
118
|
+
out.setdefault("metric", str(raw.get("source") or "v1_effect_size"))
|
|
119
|
+
extra = {k: v for k, v in raw.items()
|
|
120
|
+
if k not in known and k not in ("ci_lower", "ci_upper")
|
|
121
|
+
and v is not None}
|
|
122
|
+
text = raw.get("raw_text")
|
|
123
|
+
if not text:
|
|
124
|
+
text = json.dumps(extra, ensure_ascii=False, sort_keys=True) if extra else ""
|
|
125
|
+
out["raw_text"] = str(text)
|
|
126
|
+
return out
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _v1_directness(ev: dict) -> tuple[int, str, bool]:
|
|
130
|
+
"""Carry V1 D5 Directness across as (link directness, scope_match, recorded).
|
|
131
|
+
|
|
132
|
+
D5 is a 0/1/2 axis in the V1 evidence contract and the V2 ADOPT gate reads
|
|
133
|
+
the link's directness, so the mapping has to be explicit:
|
|
134
|
+
|
|
135
|
+
2 -> directness 2 / scope_match "direct"
|
|
136
|
+
1 -> directness 1 / scope_match "partial"
|
|
137
|
+
0 -> directness 1 / scope_match "mismatch" (0 can never gate an ADOPT)
|
|
138
|
+
|
|
139
|
+
A missing or non-integer D5 is not silently promoted: it migrates as
|
|
140
|
+
directness 1 / scope_match "partial" and the caller records a downgrade.
|
|
141
|
+
"""
|
|
142
|
+
dims = ev.get("quality_dimensions")
|
|
143
|
+
raw = dims.get("D5_directness") if isinstance(dims, dict) else None
|
|
144
|
+
if isinstance(raw, bool) or not isinstance(raw, int):
|
|
145
|
+
return 1, "partial", False
|
|
146
|
+
mapped = _D5_TO_DIRECTNESS.get(raw)
|
|
147
|
+
if mapped is None:
|
|
148
|
+
return 1, "partial", False
|
|
149
|
+
return mapped[0], mapped[1], True
|
|
150
|
+
|
|
151
|
+
|
|
78
152
|
def migrate_v1_pack(pack_dir: Path, *, home: Path,
|
|
79
153
|
title: str | None = None) -> MigrationResult:
|
|
80
154
|
"""Import a V1 pack directory into a new V2 Project graph.
|
|
@@ -273,7 +347,10 @@ def migrate_v1_pack(pack_dir: Path, *, home: Path,
|
|
|
273
347
|
"measure": ev.get("outcome_type") or "outcome",
|
|
274
348
|
"timepoint": None,
|
|
275
349
|
"effect_direction": _map_effect_direction(ev.get("effect_direction")),
|
|
276
|
-
|
|
350
|
+
# Carry the magnitude across the hop instead of dropping it:
|
|
351
|
+
# a migrated pack used to report 100% not_extractable, which
|
|
352
|
+
# reads as "no evidence" rather than "not migrated".
|
|
353
|
+
"effect_estimate": _v1_effect_estimate(ev),
|
|
277
354
|
"raw_result_text": ev.get("claim") or "unavailable",
|
|
278
355
|
"source_locator": ev.get("source_location") or "unavailable",
|
|
279
356
|
"extensions": {"v1_legacy": True},
|
|
@@ -281,6 +358,14 @@ def migrate_v1_pack(pack_dir: Path, *, home: Path,
|
|
|
281
358
|
claim_id = claim_by_evidence.get(ev["evidence_id"], f"CLM-{ev['evidence_id']}")
|
|
282
359
|
link_id = f"LNK-{ev['evidence_id']}"
|
|
283
360
|
|
|
361
|
+
directness, scope_match, d5_recorded = _v1_directness(ev)
|
|
362
|
+
if not d5_recorded:
|
|
363
|
+
report["downgrades"].append({
|
|
364
|
+
"evidence_id": ev["evidence_id"],
|
|
365
|
+
"from": "V1 quality_dimensions.D5_directness (absent or unreadable)",
|
|
366
|
+
"to": f"directness={directness}, scope_match={scope_match!r}",
|
|
367
|
+
})
|
|
368
|
+
|
|
284
369
|
links.append({
|
|
285
370
|
|
|
286
371
|
"evidence_link_id": link_id,
|
|
@@ -289,8 +374,8 @@ def migrate_v1_pack(pack_dir: Path, *, home: Path,
|
|
|
289
374
|
"relation_to_claim": _map_relation(
|
|
290
375
|
ev.get("relation_to_claim") or ev.get("direction")),
|
|
291
376
|
"decision_implication": _map_decision_implication(ev),
|
|
292
|
-
"directness":
|
|
293
|
-
"applicability": {"scope_match":
|
|
377
|
+
"directness": directness,
|
|
378
|
+
"applicability": {"scope_match": scope_match},
|
|
294
379
|
"reasoning_note": "migrated from V1 Evidence Object",
|
|
295
380
|
"created_in_revision": 1,
|
|
296
381
|
"extensions": {"v1_legacy": True},
|