eduevidence 6.0.0 → 6.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (267) hide show
  1. package/CHANGELOG.md +395 -0
  2. package/CONTRIBUTING.md +105 -0
  3. package/README.md +113 -49
  4. package/README.zh-CN.md +39 -12
  5. package/SKILL.md +15 -5
  6. package/assets/readme/landing-tour.gif +0 -0
  7. package/assets/readme/studio-tour.gif +0 -0
  8. package/benchmarks/evidence-library.json +277 -1
  9. package/bin/eduevidence.js +2 -1
  10. package/docs/architecture.md +325 -46
  11. package/docs/demo-workplace-ai.md +1 -1
  12. package/docs/install-guide.md +1 -1
  13. package/docs/j-ev-experimental.md +250 -0
  14. package/docs/orchestration-role-model.md +1 -1
  15. package/docs/release-closeout/README.md +1 -1
  16. package/docs/reproducibility.md +138 -0
  17. package/docs/sciverse-api.md +125 -0
  18. package/domains/_neutral/copy/few_shots.json +21 -0
  19. package/domains/_neutral/copy/framing_lexicon.json +19 -0
  20. package/domains/_neutral/copy/module_labels.json +5 -0
  21. package/domains/_neutral/copy/module_labels_footer.json +102 -0
  22. package/domains/_neutral/copy/module_labels_modules.json +204 -0
  23. package/domains/_neutral/copy/module_labels_nav.json +126 -0
  24. package/domains/_neutral/copy/module_labels_summary.json +98 -0
  25. package/domains/_neutral/copy/module_labels_tables.json +164 -0
  26. package/domains/_neutral/copy/module_labels_v2.json +90 -0
  27. package/domains/_neutral/copy/risk_constructs.json +20 -0
  28. package/domains/_neutral/copy/section_titles.json +66 -0
  29. package/domains/_neutral/copy/terminology.json +11 -0
  30. package/domains/check_copy_packs.py +103 -0
  31. package/domains/education/copy/few_shots.json +22 -0
  32. package/domains/education/copy/framing_enums.json +167 -0
  33. package/domains/education/copy/framing_lexicon.json +166 -0
  34. package/domains/education/copy/module_labels.json +169 -0
  35. package/domains/education/copy/risk_constructs.json +48 -0
  36. package/domains/education/copy/section_titles.json +186 -0
  37. package/domains/education/copy/terminology.json +70 -0
  38. package/domains/education/manifest.json +1 -1
  39. package/domains/education/outcome_taxonomy.json +2 -2
  40. package/domains/manifest.json +1 -1
  41. package/domains/policy/copy/few_shots.json +22 -0
  42. package/domains/policy/copy/framing_enums.json +94 -0
  43. package/domains/policy/copy/framing_lexicon.json +174 -0
  44. package/domains/policy/copy/module_labels.json +168 -0
  45. package/domains/policy/copy/risk_constructs.json +33 -0
  46. package/domains/policy/copy/section_titles.json +186 -0
  47. package/domains/policy/copy/terminology.json +64 -0
  48. package/eduevidence_cli.py +10 -0
  49. package/engine/capabilities.py +57 -5
  50. package/engine/decision_policy.py +167 -0
  51. package/engine/evidence_graph.py +14 -10
  52. package/engine/gaps.py +42 -22
  53. package/engine/ids.py +2 -0
  54. package/engine/library.py +6 -2
  55. package/engine/library_builtin.py +7 -4
  56. package/engine/living.py +34 -4
  57. package/engine/migration.py +88 -3
  58. package/engine/orchestration.py +5 -5
  59. package/engine/paths.py +2 -0
  60. package/engine/pilot.py +34 -32
  61. package/engine/taxonomy.py +211 -0
  62. package/engine/tribunal.py +49 -43
  63. package/engine/versions.py +1 -1
  64. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1361 -147
  65. package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
  66. package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
  67. package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
  68. package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
  69. package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
  70. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +448 -128
  71. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +448 -128
  72. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +448 -128
  73. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +448 -128
  74. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +448 -128
  75. package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
  76. package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
  77. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
  78. package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
  79. package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
  80. package/examples/ai-coding-assistant-evidence/result.json +13 -9
  81. package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
  82. package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
  83. package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
  84. package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
  85. package/examples/spaced-retrieval-practice/applicability.json +14 -0
  86. package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
  87. package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
  88. package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
  89. package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
  90. package/examples/spaced-retrieval-practice/frame.json +58 -0
  91. package/examples/spaced-retrieval-practice/gate_report.json +101 -0
  92. package/examples/spaced-retrieval-practice/methodology.json +78 -0
  93. package/examples/spaced-retrieval-practice/report.html +2522 -0
  94. package/examples/spaced-retrieval-practice/report_spec.json +212 -0
  95. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
  96. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
  97. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
  98. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
  99. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
  100. package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
  101. package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
  102. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
  103. package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
  104. package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
  105. package/examples/spaced-retrieval-practice/result.json +942 -0
  106. package/examples/spaced-retrieval-practice/result.zh.json +942 -0
  107. package/examples/spaced-retrieval-practice/skeptic.json +70 -0
  108. package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
  109. package/examples/spaced-retrieval-practice/verdict.json +93 -0
  110. package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
  111. package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
  112. package/examples/workplace-ai-assistant/claims.jsonl +4 -4
  113. package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
  114. package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
  115. package/examples/workplace-ai-assistant/final_verdict.json +78 -0
  116. package/examples/workplace-ai-assistant/gate_report.json +101 -0
  117. package/examples/workplace-ai-assistant/report_spec.json +209 -40
  118. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +449 -119
  119. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +449 -119
  120. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +449 -119
  121. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +449 -119
  122. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +449 -119
  123. package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
  124. package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
  125. package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
  126. package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
  127. package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
  128. package/examples/workplace-ai-assistant/result.json +82 -20
  129. package/examples/workplace-ai-assistant/result.zh.json +82 -20
  130. package/examples/workplace-ai-assistant/skeptic.json +72 -0
  131. package/examples/workplace-ai-assistant/verdict.json +36 -10
  132. package/integrations/agent_mcp.py +2 -2
  133. package/integrations/jev/__init__.py +115 -0
  134. package/integrations/jev/approval.py +212 -0
  135. package/integrations/jev/cli.py +84 -0
  136. package/integrations/jev/config.py +112 -0
  137. package/integrations/jev/gateway.py +128 -0
  138. package/integrations/jev/modes.py +38 -0
  139. package/integrations/jev/tools_classify.py +88 -0
  140. package/integrations/jev/tools_extract.py +111 -0
  141. package/integrations/jev/tools_rerank.py +71 -0
  142. package/integrations/jev/tools_screen.py +87 -0
  143. package/integrations/jev/tools_verify.py +95 -0
  144. package/integrations/jev_mcp.py +22 -0
  145. package/integrations/semantic_decide.py +286 -0
  146. package/integrations/semdecide_cli.py +55 -0
  147. package/package.json +19 -2
  148. package/pyproject.toml +4 -3
  149. package/references/report-copy-style.md +107 -0
  150. package/references/retrieval-compliance.md +75 -0
  151. package/references/retrieval-protocol.md +20 -0
  152. package/retrieval/audit.py +27 -3
  153. package/retrieval/fetch.py +96 -0
  154. package/retrieval/sciverse.py +398 -0
  155. package/retrieval/search.py +47 -7
  156. package/schemas/applicability.schema.json +94 -0
  157. package/schemas/chart-spec.schema.json +10 -3
  158. package/schemas/evidence.schema.json +316 -43
  159. package/schemas/fetch-result.schema.json +2 -1
  160. package/schemas/report-result.schema.json +3 -3
  161. package/schemas/report-spec.schema.json +98 -100
  162. package/schemas/skeptic.schema.json +86 -0
  163. package/schemas/source.schema.json +21 -2
  164. package/schemas/v2/decision-snapshot.schema.json +20 -9
  165. package/schemas/v2/finding.schema.json +5 -1
  166. package/schemas/v2/intake.schema.json +191 -0
  167. package/schemas/v2/methodology-audit.schema.json +5 -1
  168. package/schemas/v2/outcome.schema.json +28 -5
  169. package/schemas/v2/study.schema.json +5 -1
  170. package/schemas/vNext/autoevolve-session.schema.json +34 -1
  171. package/schemas/vNext/eval-snapshot.schema.json +77 -1
  172. package/schemas/vNext/execution-plan.schema.json +50 -1
  173. package/schemas/vNext/gap-priority.schema.json +54 -1
  174. package/schemas/vNext/negative-search-record.schema.json +68 -1
  175. package/schemas/vNext/research-iteration.schema.json +87 -1
  176. package/schemas/vNext/research-strategy.schema.json +62 -1
  177. package/schemas/vNext/skill-experiment.schema.json +90 -1
  178. package/schemas/vNext/task-spec.schema.json +156 -1
  179. package/schemas/vNext/worker-result.schema.json +60 -1
  180. package/schemas/verdict.schema.json +164 -28
  181. package/scripts/build_evidence_library.py +15 -5
  182. package/scripts/build_report_variants.py +18 -2
  183. package/scripts/build_result.py +74 -9
  184. package/scripts/check_package_parity.py +85 -0
  185. package/scripts/check_protocol_alignment.py +375 -0
  186. package/scripts/check_versioned_schemas.py +254 -0
  187. package/scripts/claim_audit.py +13 -8
  188. package/scripts/compute_confidence.py +10 -0
  189. package/scripts/dashboard_server.py +13 -2
  190. package/scripts/did_regression.py +12 -2
  191. package/scripts/evidence_score.py +5 -2
  192. package/scripts/intake/__init__.py +31 -0
  193. package/scripts/intake/__main__.py +18 -0
  194. package/scripts/intake/background.py +78 -0
  195. package/scripts/intake/browser.py +79 -0
  196. package/scripts/intake/cli.py +57 -0
  197. package/scripts/intake/constants.py +57 -0
  198. package/scripts/intake/depth.py +53 -0
  199. package/scripts/intake/enhancements.py +106 -0
  200. package/scripts/intake/hooks.py +90 -0
  201. package/scripts/intake/prefs.py +76 -0
  202. package/scripts/intake/prompts.py +85 -0
  203. package/scripts/intake/session.py +152 -0
  204. package/scripts/lint_file_layers.py +126 -0
  205. package/scripts/orchestrator.py +187 -40
  206. package/scripts/pre_verdict_gate.py +241 -29
  207. package/scripts/quickstart.py +18 -2
  208. package/scripts/run_workspace.py +7 -1
  209. package/scripts/skill_lint.py +11 -1
  210. package/scripts/skill_payload.py +6 -3
  211. package/scripts/test_adversarial_empirical.py +96 -25
  212. package/scripts/validate_schema.py +31 -1
  213. package/skill/agents/evaluation-designer.md +20 -4
  214. package/skill/agents/evidence-analyst.md +19 -3
  215. package/skill/agents/evidence-judge.md +98 -8
  216. package/skill/agents/evidence-retriever.md +20 -3
  217. package/skill/agents/intervention-designer.md +20 -4
  218. package/skill/agents/method-reviewer.md +18 -2
  219. package/skill/agents/{education-planner.md → research-planner.md} +19 -3
  220. package/skill/agents/skeptic.md +18 -2
  221. package/skill/roles/registry.yaml +11 -11
  222. package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
  223. package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
  224. package/skill/sub-skills/data-analysis/SKILL.md +34 -15
  225. package/skill/sub-skills/ethics-review/SKILL.md +33 -10
  226. package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
  227. package/skill/sub-skills/evidence-review/SKILL.md +31 -12
  228. package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
  229. package/skill/sub-skills/literature-review/SKILL.md +35 -14
  230. package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
  231. package/skill/sub-skills/report-generation/SKILL.md +28 -0
  232. package/skill/sub-skills/research-planning/SKILL.md +41 -14
  233. package/skill/sub-skills/study-design/SKILL.md +30 -9
  234. package/skill/task-briefs/adjudicate.md +32 -7
  235. package/skill/task-briefs/applicability.md +37 -2
  236. package/skill/task-briefs/audit.md +32 -7
  237. package/skill/task-briefs/challenge.md +34 -5
  238. package/skill/task-briefs/evaluate.md +30 -5
  239. package/skill/task-briefs/extract.md +31 -8
  240. package/skill/task-briefs/frame.md +39 -10
  241. package/skill/task-briefs/intervene.md +32 -6
  242. package/skill/task-briefs/present.md +32 -8
  243. package/skill/task-briefs/projection.md +36 -2
  244. package/skill/task-briefs/retrieve.md +36 -6
  245. package/skill/workflows/decision-and-pilot.md +76 -1
  246. package/skill/workflows/evaluate-and-update.md +83 -0
  247. package/skill/workflows/evidence-review.md +104 -0
  248. package/skill/workflows/experimental-jev.md +170 -0
  249. package/skill/workflows/intake.md +120 -0
  250. package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
  251. package/visualization/eduevidence-report/scripts/build_infographics.py +37 -15
  252. package/visualization/eduevidence-report/scripts/build_report.py +435 -575
  253. package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
  254. package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
  255. package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
  256. package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
  257. package/visualization/eduevidence-report/scripts/zh_labels.py +141 -1
  258. package/web/architecture.html +14885 -0
  259. package/web/studio/assets/index-B8tkF44Q.css +1 -0
  260. package/web/studio/index.html +2 -2
  261. package/scripts/build_esl_artifacts.py +0 -1921
  262. package/scripts/build_killer_demo.py +0 -295
  263. package/scripts/enrich_projects_human_and_lieflat.py +0 -315
  264. package/scripts/generate_new_projects.py +0 -686
  265. package/scripts/sync_killer_demo_report.py +0 -270
  266. package/web/studio/assets/index-CzXocaGv.css +0 -1
  267. /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
@@ -142,12 +142,36 @@ class AuditedSearchExecutor:
142
142
  }, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
143
143
  (output_dir / "search-attempts.jsonl").write_text(
144
144
  "".join(json.dumps(item.to_dict(), ensure_ascii=False) + "\n" for item in attempts), encoding="utf-8")
145
+ # Sciverse chunk locators are a working set, never evidence: /content
146
+ # must expand them and the fetch/validate gate must pass first (RULE 2).
147
+ chunk_records = [
148
+ {"chunk_id": hit.chunk_id, "doc_id": hit.doc_id, "offset": hit.offset,
149
+ "title": hit.title, "doi": hit.doi, "provider": hit.provider,
150
+ "score": hit.score, "locator_state": "discovery_only_requires_content_fetch"}
151
+ for hit in hits
152
+ if getattr(hit, "doc_id", None)
153
+ ]
154
+ if chunk_records:
155
+ (output_dir / "chunks.jsonl").write_text(
156
+ "".join(json.dumps(record, ensure_ascii=False) + "\n" for record in chunk_records),
157
+ encoding="utf-8")
145
158
  with (output_dir / "source-screening.csv").open("w", newline="", encoding="utf-8") as fh:
146
- writer = csv.DictWriter(fh, fieldnames=["title", "doi", "url", "provider", "year", "screening_status", "reason"])
159
+ writer = csv.DictWriter(fh, fieldnames=["title", "doi", "url", "provider", "year",
160
+ "doc_id", "chunk_id", "offset", "screening_status", "reason"])
147
161
  writer.writeheader()
148
162
  for hit in hits:
149
- writer.writerow({"title": hit.title, "doi": hit.doi or parse_doi_from_url(hit.url) or "", "url": hit.url,
150
- "provider": hit.provider, "year": hit.year or "", "screening_status": "candidate",
163
+ doi = hit.doi or parse_doi_from_url(hit.url) or ""
164
+ url = hit.url
165
+ # Never fabricate a location for a DOI-less record; the explicit
166
+ # marker routes it to manual screening instead.
167
+ if not url and not doi:
168
+ url = "needs_manual_location"
169
+ writer.writerow({"title": hit.title, "doi": doi, "url": url,
170
+ "provider": hit.provider, "year": hit.year or "",
171
+ "doc_id": getattr(hit, "doc_id", None) or "",
172
+ "chunk_id": getattr(hit, "chunk_id", None) or "",
173
+ "offset": getattr(hit, "offset", None) if getattr(hit, "offset", None) is not None else "",
174
+ "screening_status": "candidate",
151
175
  "reason": "discovery metadata only; fetch and validation required before evidence extraction"})
152
176
  with (output_dir / "exclusion-log.csv").open("w", newline="", encoding="utf-8") as fh:
153
177
  writer = csv.DictWriter(fh, fieldnames=["identifier", "reason"])
@@ -210,6 +210,98 @@ def _fetch_raw_html(url: str, timeout: int) -> tuple[int, str, str]:
210
210
  return _http_get(url, timeout=timeout)
211
211
 
212
212
 
213
+ def fetch_sciverse_content(
214
+ doc_id: str,
215
+ *,
216
+ offset: int = 0,
217
+ limit: int = 4096,
218
+ canonical_url: str | None = None,
219
+ expect_title: str | None = None,
220
+ ) -> dict[str, Any]:
221
+ """Read a Sciverse full-text slice through the same FetchResult contract.
222
+
223
+ This is the machine-enforced half of RULE 2: a ``/agentic-search`` chunk is
224
+ a locator (doc_id + code-point offset), and only the text returned here —
225
+ after passing the Fetch Validation Gate — may enter Evidence Extraction.
226
+
227
+ ``offset``/``limit`` count Unicode code points, exactly like Python ``len``.
228
+ The locator is recorded inside ``extensions`` because the FetchResult
229
+ contract is closed (``additionalProperties: false``). ``canonical_url``
230
+ carries the source's own citation pointer when one exists; the locator URL
231
+ is used otherwise and is a provenance pointer, never a citation target.
232
+ """
233
+ from retrieval.sciverse import (
234
+ STATUS_OK,
235
+ STATUS_UNAVAILABLE,
236
+ read_content,
237
+ )
238
+
239
+ fetched_at = datetime.now(timezone.utc).isoformat()
240
+ resolved_url = canonical_url or f"https://sciverse.space/doc/{doc_id}"
241
+ response = read_content(doc_id, offset=offset, limit=limit)
242
+
243
+ if response.status == STATUS_UNAVAILABLE:
244
+ return FetchResult(
245
+ original_url=resolved_url,
246
+ fetch_provider="sciverse_content",
247
+ fetch_status="FETCH_FAILED",
248
+ fetched_at=fetched_at,
249
+ validation={"passed": False,
250
+ "checks": {"http_success": False, "body_length_ok": False},
251
+ "issues": ["SCIVERSE_UNAVAILABLE: no API token configured"]},
252
+ extensions={"sciverse": {"doc_id": doc_id, "offset": offset,
253
+ "status": response.status, "error": response.error}},
254
+ ).to_dict()
255
+
256
+ if response.status != STATUS_OK:
257
+ return FetchResult(
258
+ original_url=resolved_url,
259
+ fetch_provider="sciverse_content",
260
+ fetch_status="FETCH_FAILED",
261
+ fetched_at=fetched_at,
262
+ validation={"passed": False,
263
+ "checks": {"http_success": False, "body_length_ok": False},
264
+ "issues": [f"{response.status}: {response.error}".strip(": ")]},
265
+ extensions={"sciverse": {"doc_id": doc_id, "offset": offset,
266
+ "status": response.status,
267
+ "http_status": response.http_status,
268
+ "error": response.error}},
269
+ ).to_dict()
270
+
271
+ content = str(response.data.get("text") or "")
272
+ next_offset = response.data.get("next_offset")
273
+ more = bool(response.data.get("more"))
274
+ candidate = FetchResult(
275
+ original_url=resolved_url,
276
+ resolved_url=resolved_url,
277
+ fetch_method="smart_web_fetch",
278
+ fetch_provider="sciverse_content",
279
+ fetch_status="FETCH_VALID" if content.strip() else "FETCH_FAILED",
280
+ fetched_at=fetched_at,
281
+ raw_size=len(content.encode("utf-8")),
282
+ content=content,
283
+ extensions={"sciverse": {
284
+ "doc_id": doc_id,
285
+ "offset": offset,
286
+ "limit": limit,
287
+ "next_offset": next_offset,
288
+ "more": more,
289
+ "location_unit": "unicode_code_point",
290
+ }},
291
+ )
292
+ candidate.clean_size = candidate.raw_size
293
+ candidate.content_length = candidate.raw_size
294
+ candidate.content_hash = _hash(content) if content else ""
295
+ # A Sciverse locator URL is not an http(s) target: the scheme/URL-match and
296
+ # private-target checks do not apply; length and error-page checks do.
297
+ candidate.validation = validate_fetch_result(
298
+ candidate.to_dict(), expect_title=expect_title,
299
+ )
300
+ if not candidate.validation.get("passed") and candidate.fetch_status == "FETCH_VALID":
301
+ candidate.fetch_status = "FETCH_PARTIAL"
302
+ return candidate.to_dict()
303
+
304
+
213
305
  _PROVIDER_FETCHERS: dict[str, Callable[[str, int], tuple[int, str, str]]] = {
214
306
  "builtin": _fetch_builtin,
215
307
  "jina_reader": _fetch_jina_reader,
@@ -299,6 +391,9 @@ class FetchResult:
299
391
  fallback_chain: list[str] = field(default_factory=list)
300
392
  content: str = ""
301
393
  validation: dict[str, Any] = field(default_factory=dict)
394
+ #: Structured extension container (schema: fetch-result.extensions).
395
+ #: Sciverse locators live here because the top level is closed.
396
+ extensions: dict[str, Any] = field(default_factory=dict)
302
397
 
303
398
  def to_dict(self) -> dict[str, Any]:
304
399
  return {
@@ -317,6 +412,7 @@ class FetchResult:
317
412
  "fallback_chain": self.fallback_chain,
318
413
  "content": self.content if self.fetch_status != "FETCH_FAILED" else "",
319
414
  "validation": self.validation,
415
+ "extensions": self.extensions,
320
416
  }
321
417
 
322
418
 
@@ -0,0 +1,398 @@
1
+ #!/usr/bin/env python3
2
+ """sciverse.py — Sciverse Open Platform retrieval channel (key-based).
3
+
4
+ Sciverse exposes citation-grade academic retrieval to agents through four
5
+ endpoints used here (canonical spec: openapi.yaml v0.14.2 of
6
+ opendatalab/Sciverse-Agent-Tools, archived in docs/sciverse-api.md):
7
+
8
+ POST /meta-search structured metadata search -> Source-level hits
9
+ POST /agentic-search semantic chunk retrieval -> chunk locators
10
+ GET /content read original text slice -> evidence content
11
+ POST /meta-paper-relations citations / references -> citation chain
12
+
13
+ Scientific discipline encoded here (RULE 2): an /agentic-search chunk is a
14
+ discovery locator, never evidence. It must be expanded through ``/content``
15
+ and pass the fetch/validate gate before any extraction stage may use it.
16
+
17
+ The module is stdlib-only and never raises for network or API content issues:
18
+ every failure is reported as a typed status so the pipeline degrades cleanly
19
+ to the zero-config channels.
20
+ """
21
+ from __future__ import annotations
22
+
23
+ import json
24
+ import os
25
+ import urllib.error
26
+ import urllib.parse
27
+ import urllib.request
28
+ from dataclasses import dataclass, field
29
+ from typing import Any, List, Optional
30
+
31
+ from engine.log import get_log
32
+
33
+ log = get_log("sciverse")
34
+
35
+ BASE_URL = os.environ.get("SCIVERSE_BASE_URL", "https://api.sciverse.space").rstrip("/")
36
+ TOKEN_ENV = "SCIVERSE_API_TOKEN"
37
+ DEFAULT_TIMEOUT = 20
38
+ USER_AGENT = "EduEvidence-Research-Agent/6.1 (+https://github.com/37chengshan/eduevidence)"
39
+
40
+ #: Reading window used when expanding a chunk locator into evidence content.
41
+ DEFAULT_CONTENT_LIMIT = 4096
42
+ MAX_CONTENT_LIMIT = 16384 # LLM-facing ceiling recommended by the spec
43
+
44
+ #: Typed failure statuses (never exceptions) surfaced to callers and audits.
45
+ STATUS_OK = "ok"
46
+ STATUS_UNAVAILABLE = "SCIVERSE_UNAVAILABLE" # no token configured
47
+ STATUS_UNAUTHORIZED = "SCIVERSE_UNAUTHORIZED" # 401/403
48
+ STATUS_BAD_REQUEST = "SCIVERSE_BAD_REQUEST" # 400/404/429
49
+ STATUS_UPSTREAM = "SCIVERSE_UPSTREAM_ERROR" # 502/503/5xx
50
+ STATUS_NETWORK = "SCIVERSE_NETWORK_ERROR" # timeout / DNS / TLS
51
+
52
+
53
+ def api_token() -> str:
54
+ """Bearer token from the environment; empty string means unavailable."""
55
+ return os.environ.get(TOKEN_ENV, "").strip()
56
+
57
+
58
+ def available() -> bool:
59
+ return bool(api_token())
60
+
61
+
62
+ @dataclass
63
+ class SciverseResponse:
64
+ """One endpoint call outcome. ``data`` is empty unless ``ok`` is True."""
65
+
66
+ status: str
67
+ data: dict[str, Any] = field(default_factory=dict)
68
+ error: str = ""
69
+ http_status: Optional[int] = None
70
+
71
+ @property
72
+ def ok(self) -> bool:
73
+ return self.status == STATUS_OK
74
+
75
+
76
+ def _classify(http_status: int) -> str:
77
+ if http_status in (401, 403):
78
+ return STATUS_UNAUTHORIZED
79
+ if http_status in (400, 404, 429):
80
+ return STATUS_BAD_REQUEST
81
+ if http_status >= 500:
82
+ return STATUS_UPSTREAM
83
+ return STATUS_UPSTREAM
84
+
85
+
86
+ def _error_message(body: str) -> str:
87
+ """Extract ApiError.message without leaking tokens or full payloads."""
88
+ try:
89
+ payload = json.loads(body)
90
+ except (ValueError, TypeError):
91
+ return body.strip()[:200]
92
+ if isinstance(payload, dict):
93
+ for key in ("message", "error", "detail"):
94
+ value = payload.get(key)
95
+ if isinstance(value, str) and value.strip():
96
+ return value.strip()[:200]
97
+ return ""
98
+
99
+
100
+ def _request(
101
+ method: str,
102
+ path: str,
103
+ *,
104
+ params: Optional[dict[str, Any]] = None,
105
+ payload: Optional[dict[str, Any]] = None,
106
+ timeout: int = DEFAULT_TIMEOUT,
107
+ ) -> SciverseResponse:
108
+ """Single typed HTTP call. Never raises; never logs the token."""
109
+ token = api_token()
110
+ if not token:
111
+ return SciverseResponse(STATUS_UNAVAILABLE, error=f"{TOKEN_ENV} is not set")
112
+
113
+ url = BASE_URL + path
114
+ if params:
115
+ clean = {k: v for k, v in params.items() if v is not None}
116
+ if clean:
117
+ url = f"{url}?{urllib.parse.urlencode(clean)}"
118
+
119
+ headers = {
120
+ "Authorization": f"Bearer {token}",
121
+ "Accept": "application/json",
122
+ "User-Agent": USER_AGENT,
123
+ }
124
+ body_bytes: Optional[bytes] = None
125
+ if payload is not None:
126
+ body_bytes = json.dumps(payload).encode("utf-8")
127
+ headers["Content-Type"] = "application/json"
128
+
129
+ req = urllib.request.Request(url, data=body_bytes, headers=headers, method=method)
130
+ try:
131
+ with urllib.request.urlopen(req, timeout=timeout) as resp:
132
+ raw = resp.read().decode("utf-8", errors="replace")
133
+ data = json.loads(raw) if raw.strip() else {}
134
+ return SciverseResponse(STATUS_OK, data=data if isinstance(data, dict) else {},
135
+ http_status=resp.status)
136
+ except urllib.error.HTTPError as exc:
137
+ detail = ""
138
+ try:
139
+ detail = _error_message(exc.read().decode("utf-8", errors="replace"))
140
+ except Exception: # pragma: no cover - body already consumed/closed
141
+ detail = ""
142
+ status = _classify(exc.code)
143
+ log.debug("sciverse http error path=%s status=%s", path, exc.code)
144
+ return SciverseResponse(status, error=detail or f"HTTP {exc.code}", http_status=exc.code)
145
+ except Exception as exc: # timeouts, DNS, TLS, decode
146
+ log.debug("sciverse network error path=%s err=%s", path, type(exc).__name__)
147
+ return SciverseResponse(STATUS_NETWORK, error=type(exc).__name__)
148
+
149
+
150
+ # ---------------------------------------------------------------------------
151
+ # Endpoint wrappers (core four)
152
+ # ---------------------------------------------------------------------------
153
+
154
+ def meta_search(
155
+ query: str,
156
+ *,
157
+ limit: int = 10,
158
+ year_from: Optional[int] = None,
159
+ year_to: Optional[int] = None,
160
+ filters: Optional[list[dict[str, Any]]] = None,
161
+ collection: str = "papers",
162
+ ) -> SciverseResponse:
163
+ """Structured metadata search (``/meta-search``). Source-level hits.
164
+
165
+ Returns paper metadata including ``unique_id`` (always) and ``doc_id``
166
+ (only when full text exists) — both are required for downstream
167
+ ``/content`` and ``/meta-paper-relations`` calls.
168
+ """
169
+ payload: dict[str, Any] = {
170
+ "collection": collection,
171
+ "query": query,
172
+ "page": 1,
173
+ "page_size": max(1, min(limit, 50)),
174
+ }
175
+ advanced: list[dict[str, Any]] = list(filters or [])
176
+ if year_from is not None:
177
+ advanced.append({"field": "publication_published_year",
178
+ "operator": "FILTER_OP_GTE", "value": year_from})
179
+ if year_to is not None:
180
+ advanced.append({"field": "publication_published_year",
181
+ "operator": "FILTER_OP_LTE", "value": year_to})
182
+ if advanced:
183
+ payload["filters_advanced"] = advanced
184
+ return _request("POST", "/meta-search", payload=payload)
185
+
186
+
187
+ def agentic_search(
188
+ query: str,
189
+ *,
190
+ top_k: int = 10,
191
+ mode: str = "balanced",
192
+ filters: Optional[dict[str, Any]] = None,
193
+ ) -> SciverseResponse:
194
+ """Natural-language chunk retrieval (``/agentic-search``).
195
+
196
+ Each hit is a locator (chunk_id/doc_id/offset/score) — a discovery aid.
197
+ ``offset`` is a Unicode code-point offset usable directly by
198
+ :func:`read_content`.
199
+ """
200
+ payload: dict[str, Any] = {
201
+ "query": query,
202
+ "top_k": max(1, min(top_k, 100)),
203
+ "mode": mode if mode in ("fast", "balanced", "quality") else "balanced",
204
+ }
205
+ if filters:
206
+ payload["filters"] = filters
207
+ return _request("POST", "/agentic-search", payload=payload)
208
+
209
+
210
+ def read_content(
211
+ doc_id: str,
212
+ *,
213
+ offset: int = 0,
214
+ limit: int = DEFAULT_CONTENT_LIMIT,
215
+ ) -> SciverseResponse:
216
+ """Read original text by code-point range (``/content``).
217
+
218
+ ``offset`` is always sent explicitly: omitting it makes the server return
219
+ the whole document and ignore ``limit``.
220
+ """
221
+ return _request("GET", "/content", params={
222
+ "doc_id": doc_id,
223
+ "offset": max(0, int(offset)),
224
+ "limit": max(1, min(int(limit), MAX_CONTENT_LIMIT)),
225
+ })
226
+
227
+
228
+ def paper_relations(
229
+ unique_id: str,
230
+ *,
231
+ relation: str = "REFERENCES",
232
+ page: int = 1,
233
+ page_size: int = 25,
234
+ ) -> SciverseResponse:
235
+ """Paginate citations / references / related works (``/meta-paper-relations``).
236
+
237
+ ``unique_id`` (not ``doc_id``) identifies the target paper. CITATIONS is
238
+ incoming (who cites me); REFERENCES is outgoing (who I cite).
239
+ """
240
+ relation_name = relation.upper()
241
+ if relation_name not in ("CITATIONS", "REFERENCES", "RELATED_WORKS"):
242
+ relation_name = "REFERENCES"
243
+ return _request("POST", "/meta-paper-relations", payload={
244
+ "unique_id": unique_id,
245
+ "relation": relation_name,
246
+ "page": max(1, int(page)),
247
+ "page_size": max(1, min(int(page_size), 200)),
248
+ })
249
+
250
+
251
+ # ---------------------------------------------------------------------------
252
+ # Chunk / relation record helpers (run-workspace working sets)
253
+ # ---------------------------------------------------------------------------
254
+
255
+ def chunk_records(response: SciverseResponse, *, query_id: str = "") -> list[dict[str, Any]]:
256
+ """Normalize ``/agentic-search`` hits into chunk-locator records.
257
+
258
+ These records are a working set, never evidence: they carry the locator
259
+ (doc_id/offset) that :func:`read_content` needs.
260
+ """
261
+ if not response.ok:
262
+ return []
263
+ records: list[dict[str, Any]] = []
264
+ for hit in response.data.get("hits") or []:
265
+ if not isinstance(hit, dict):
266
+ continue
267
+ doc_id = str(hit.get("doc_id") or "")
268
+ if not doc_id:
269
+ continue
270
+ records.append({
271
+ "chunk_id": str(hit.get("chunk_id") or ""),
272
+ "doc_id": doc_id,
273
+ "offset": int(hit.get("offset") or 0),
274
+ "score": float(hit.get("score") or 0.0),
275
+ "chunk": str(hit.get("chunk") or ""),
276
+ "title": str(hit.get("title") or ""),
277
+ "page_no": hit.get("page_no"),
278
+ "source_type": str(hit.get("source_type") or ""),
279
+ "query_id": query_id,
280
+ })
281
+ return records
282
+
283
+
284
+ def relation_records(response: SciverseResponse, *, unique_id: str = "",
285
+ relation: str = "") -> list[dict[str, Any]]:
286
+ """Normalize a ``/meta-paper-relations`` page into citation-chain records."""
287
+ if not response.ok:
288
+ return []
289
+ records: list[dict[str, Any]] = []
290
+ for item in response.data.get("items") or []:
291
+ if not isinstance(item, dict):
292
+ continue
293
+ records.append({
294
+ "source_unique_id": unique_id,
295
+ "relation": relation,
296
+ "id": str(item.get("id") or ""),
297
+ "id_type": str(item.get("id_type") or ""),
298
+ "title": str(item.get("title") or ""),
299
+ })
300
+ return records
301
+
302
+
303
+ # ---------------------------------------------------------------------------
304
+ # Router provider (priority: key-based academic channel)
305
+ # ---------------------------------------------------------------------------
306
+
307
+ class SciverseProvider:
308
+ """MultiSearchRouter-compatible provider over ``/meta-search`` + ``/agentic-search``.
309
+
310
+ ``meta_search`` is the Source-level channel; ``agentic`` (opt-in via
311
+ ``include_chunks``) additionally records chunk locators on the hit so the
312
+ run workspace can expand them through ``/content``.
313
+ """
314
+
315
+ name = "sciverse"
316
+
317
+ def __init__(self, *, token: Optional[str] = None, include_chunks: bool = True):
318
+ if token is not None:
319
+ os.environ[TOKEN_ENV] = token
320
+ self.include_chunks = include_chunks
321
+
322
+ def is_available(self) -> bool:
323
+ return available()
324
+
325
+ def search(self, query: str, limit: int = 10) -> List[Any]:
326
+ from retrieval.search import SearchHit
327
+
328
+ if not self.is_available():
329
+ return []
330
+ response = meta_search(query, limit=limit)
331
+ if not response.ok:
332
+ log.debug("sciverse meta-search unavailable status=%s", response.status)
333
+ return []
334
+
335
+ chunk_index: dict[str, dict[str, Any]] = {}
336
+ if self.include_chunks:
337
+ semantic = agentic_search(query, top_k=max(1, min(limit, 50)))
338
+ for record in chunk_records(semantic):
339
+ chunk_index.setdefault(record["doc_id"], record)
340
+
341
+ hits: List[Any] = []
342
+ for item in response.data.get("results") or []:
343
+ if not isinstance(item, dict):
344
+ continue
345
+ title = str(item.get("title") or "Untitled")
346
+ doi = _normalize_doi(item.get("doi"))
347
+ doc_id = str(item.get("doc_id") or "")
348
+ unique_id = str(item.get("unique_id") or "")
349
+ locator = chunk_index.get(doc_id, {}) if doc_id else {}
350
+
351
+ authors = [
352
+ str(a.get("name")) for a in (item.get("author") or [])
353
+ if isinstance(a, dict) and a.get("name")
354
+ ]
355
+ hit = SearchHit(
356
+ title=title,
357
+ url=_source_url(doi),
358
+ snippet=str(item.get("abstract") or title),
359
+ provider=self.name,
360
+ doi=doi,
361
+ year=item.get("publication_published_year"),
362
+ citation_count=item.get("citation_count"),
363
+ authors=authors[:5],
364
+ is_academic=True,
365
+ score=1.35,
366
+ doc_id=doc_id or None,
367
+ chunk_id=locator.get("chunk_id"),
368
+ offset=locator.get("offset"),
369
+ unique_id=unique_id or None,
370
+ )
371
+ hits.append(hit)
372
+ return hits
373
+
374
+
375
+ def _normalize_doi(value: Any) -> Optional[str]:
376
+ if not isinstance(value, str) or not value.strip():
377
+ return None
378
+ doi = value.strip()
379
+ for prefix in ("https://doi.org/", "http://doi.org/", "doi:"):
380
+ if doi.lower().startswith(prefix):
381
+ doi = doi[len(prefix):]
382
+ return doi or None
383
+
384
+
385
+ def _source_url(doi: Optional[str]) -> str:
386
+ """Canonical citation pointer. Never fabricate a URL for a DOI-less record:
387
+ the empty string marks it as ``needs_manual_location`` downstream."""
388
+ return f"https://doi.org/{doi}" if doi else ""
389
+
390
+
391
+ __all__ = [
392
+ "BASE_URL", "TOKEN_ENV", "DEFAULT_CONTENT_LIMIT", "MAX_CONTENT_LIMIT",
393
+ "STATUS_OK", "STATUS_UNAVAILABLE", "STATUS_UNAUTHORIZED", "STATUS_BAD_REQUEST",
394
+ "STATUS_UPSTREAM", "STATUS_NETWORK",
395
+ "SciverseResponse", "SciverseProvider",
396
+ "api_token", "available", "meta_search", "agentic_search", "read_content",
397
+ "paper_relations", "chunk_records", "relation_records",
398
+ ]
@@ -11,12 +11,10 @@ Supports both:
11
11
  - DuckDuckGo (Zero-auth general web search fallback)
12
12
 
13
13
  2. User-Configured Search Channels (Key-Based):
14
+ - Sciverse (SCIVERSE_API_TOKEN) — citation-grade academic retrieval with
15
+ doc_id/chunk offset provenance (/meta-search + /agentic-search)
14
16
  - Tavily (TAVILY_API_KEY)
15
17
  - Brave Search (BRAVE_API_KEY)
16
- - SerpAPI (SERPAPI_API_KEY)
17
- - Serper (SERPER_API_KEY)
18
- - Exa (EXA_API_KEY)
19
- - Bocha (BOCHA_API_KEY)
20
18
 
21
19
  Pure stdlib HTTP client with robust error handling, SSL verification,
22
20
  timeout safeguards, and intelligent multi-source deduplication.
@@ -52,6 +50,16 @@ class SearchHit:
52
50
  authors: List[str] = field(default_factory=list)
53
51
  is_academic: bool = False
54
52
  score: float = 1.0
53
+ #: Sciverse locators (None for every other provider). ``doc_id`` addresses
54
+ #: the full-text artifact for /content; ``chunk_id``/``offset`` locate the
55
+ #: retrieved passage (Unicode code points) inside it. They are discovery
56
+ #: locators, never evidence (RULE 2).
57
+ doc_id: Optional[str] = None
58
+ chunk_id: Optional[str] = None
59
+ offset: Optional[int] = None
60
+ #: Sciverse metadata identifier (always present); required by
61
+ #: /meta-paper-relations to page the citation chain.
62
+ unique_id: Optional[str] = None
55
63
 
56
64
  def to_dict(self) -> dict:
57
65
  return asdict(self)
@@ -389,6 +397,15 @@ class MultiSearchRouter:
389
397
  """Orchestrates zero-config and configured search channels with deduplication."""
390
398
 
391
399
  def __init__(self):
400
+ self.academic_key_providers = []
401
+ try: # optional channel: activating it must never break the router
402
+ from retrieval.sciverse import SciverseProvider
403
+
404
+ provider = SciverseProvider()
405
+ if provider.is_available():
406
+ self.academic_key_providers.append(provider)
407
+ except Exception: # pragma: no cover - import/环境异常时静默降级
408
+ log.debug("sciverse provider unavailable at router init", exc_info=True)
392
409
  self.zero_config_academic = [
393
410
  OpenAlexProvider(),
394
411
  SemanticScholarProvider(),
@@ -406,6 +423,9 @@ class MultiSearchRouter:
406
423
 
407
424
  def get_provider_status(self) -> List[dict]:
408
425
  status = []
426
+ for p in self.academic_key_providers:
427
+ status.append({"provider": p.name, "type": "academic_key",
428
+ "status": "active", "requires_key": True})
409
429
  for p in self.zero_config_academic:
410
430
  status.append({"provider": p.name, "type": "academic_zero_config", "status": "active", "requires_key": False})
411
431
  for p in self.zero_config_web:
@@ -423,6 +443,7 @@ class MultiSearchRouter:
423
443
  def search(self, query: str, limit: int = 15, academic_only: bool = False) -> List[SearchHit]:
424
444
  all_hits: List[SearchHit] = []
425
445
  seen_urls = set()
446
+ seen_locators: set[str] = set()
426
447
 
427
448
  # 1. Try configured high-priority commercial providers if active
428
449
  if not academic_only:
@@ -437,7 +458,26 @@ class MultiSearchRouter:
437
458
  except Exception:
438
459
  pass
439
460
 
440
- # 2. Run Zero-Config Academic Providers
461
+ # 2. Run key-based academic providers (Sciverse: citation-grade
462
+ # retrieval with doc_id/offset provenance) before zero-config ones.
463
+ for kp in self.academic_key_providers:
464
+ try:
465
+ hits = kp.search(query, limit=limit)
466
+ for h in hits:
467
+ locator = getattr(h, "doc_id", None)
468
+ if locator and locator in seen_locators:
469
+ continue
470
+ if h.url and h.url in seen_urls:
471
+ continue
472
+ if h.url:
473
+ seen_urls.add(h.url)
474
+ if locator:
475
+ seen_locators.add(locator)
476
+ all_hits.append(h)
477
+ except Exception:
478
+ pass
479
+
480
+ # 3. Run Zero-Config Academic Providers
441
481
  for ap in self.zero_config_academic:
442
482
  try:
443
483
  hits = ap.search(query, limit=limit)
@@ -448,7 +488,7 @@ class MultiSearchRouter:
448
488
  except Exception:
449
489
  pass
450
490
 
451
- # 3. Run Zero-Config Web/Dynamic Providers if not academic_only
491
+ # 4. Run Zero-Config Web/Dynamic Providers if not academic_only
452
492
  if not academic_only:
453
493
  for wp in self.zero_config_web:
454
494
  try:
@@ -460,7 +500,7 @@ class MultiSearchRouter:
460
500
  except Exception:
461
501
  pass
462
502
 
463
- # 4. Fallback to verified offline domain corpus if external search returned 0 hits
503
+ # 5. Fallback to verified offline domain corpus if external search returned 0 hits
464
504
  if not all_hits:
465
505
  try:
466
506
  from retrieval.corpus_store import DomainCorpusStore