eduevidence 6.2.0 → 6.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/CHANGELOG.md +395 -0
  2. package/README.md +22 -13
  3. package/README.zh-CN.md +15 -8
  4. package/SKILL.md +10 -9
  5. package/benchmarks/evidence-library.json +277 -1
  6. package/docs/architecture.md +6 -3
  7. package/docs/j-ev-experimental.md +250 -0
  8. package/docs/reproducibility.md +138 -0
  9. package/domains/_neutral/copy/few_shots.json +21 -0
  10. package/domains/_neutral/copy/framing_lexicon.json +19 -0
  11. package/domains/_neutral/copy/module_labels.json +5 -0
  12. package/domains/_neutral/copy/module_labels_footer.json +102 -0
  13. package/domains/_neutral/copy/module_labels_modules.json +204 -0
  14. package/domains/_neutral/copy/module_labels_nav.json +126 -0
  15. package/domains/_neutral/copy/module_labels_summary.json +98 -0
  16. package/domains/_neutral/copy/module_labels_tables.json +164 -0
  17. package/domains/_neutral/copy/module_labels_v2.json +90 -0
  18. package/domains/_neutral/copy/risk_constructs.json +20 -0
  19. package/domains/_neutral/copy/section_titles.json +66 -0
  20. package/domains/_neutral/copy/terminology.json +11 -0
  21. package/domains/check_copy_packs.py +103 -0
  22. package/domains/education/copy/few_shots.json +22 -0
  23. package/domains/education/copy/framing_enums.json +167 -0
  24. package/domains/education/copy/framing_lexicon.json +166 -0
  25. package/domains/education/copy/module_labels.json +169 -0
  26. package/domains/education/copy/risk_constructs.json +48 -0
  27. package/domains/education/copy/section_titles.json +186 -0
  28. package/domains/education/copy/terminology.json +70 -0
  29. package/domains/education/manifest.json +1 -1
  30. package/domains/education/outcome_taxonomy.json +2 -2
  31. package/domains/manifest.json +1 -1
  32. package/domains/policy/copy/few_shots.json +22 -0
  33. package/domains/policy/copy/framing_enums.json +94 -0
  34. package/domains/policy/copy/framing_lexicon.json +174 -0
  35. package/domains/policy/copy/module_labels.json +168 -0
  36. package/domains/policy/copy/risk_constructs.json +33 -0
  37. package/domains/policy/copy/section_titles.json +186 -0
  38. package/domains/policy/copy/terminology.json +64 -0
  39. package/engine/capabilities.py +57 -5
  40. package/engine/decision_policy.py +88 -17
  41. package/engine/library_builtin.py +7 -4
  42. package/engine/tribunal.py +17 -23
  43. package/engine/versions.py +1 -1
  44. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +4 -4
  45. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +4 -4
  46. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +4 -4
  47. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
  48. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +4 -4
  49. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +4 -4
  50. package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
  51. package/examples/spaced-retrieval-practice/report.html +2522 -0
  52. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +4 -4
  53. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +4 -4
  54. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
  55. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +4 -4
  56. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +4 -4
  57. package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
  58. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +36 -36
  59. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +36 -36
  60. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +36 -36
  61. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +36 -36
  62. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +36 -36
  63. package/integrations/jev/__init__.py +115 -0
  64. package/integrations/jev/approval.py +212 -0
  65. package/integrations/jev/cli.py +84 -0
  66. package/integrations/jev/config.py +112 -0
  67. package/integrations/jev/gateway.py +128 -0
  68. package/integrations/jev/modes.py +38 -0
  69. package/integrations/jev/tools_classify.py +88 -0
  70. package/integrations/jev/tools_extract.py +111 -0
  71. package/integrations/jev/tools_rerank.py +71 -0
  72. package/integrations/jev/tools_screen.py +87 -0
  73. package/integrations/jev/tools_verify.py +95 -0
  74. package/integrations/jev_mcp.py +22 -0
  75. package/integrations/semantic_decide.py +286 -0
  76. package/integrations/semdecide_cli.py +55 -0
  77. package/package.json +9 -1
  78. package/pyproject.toml +1 -1
  79. package/references/report-copy-style.md +43 -3
  80. package/schemas/v2/decision-snapshot.schema.json +20 -9
  81. package/schemas/v2/intake.schema.json +191 -0
  82. package/scripts/build_evidence_library.py +15 -5
  83. package/scripts/dashboard_server.py +13 -2
  84. package/scripts/intake/__init__.py +31 -0
  85. package/scripts/intake/__main__.py +18 -0
  86. package/scripts/intake/background.py +78 -0
  87. package/scripts/intake/browser.py +79 -0
  88. package/scripts/intake/cli.py +57 -0
  89. package/scripts/intake/constants.py +57 -0
  90. package/scripts/intake/depth.py +53 -0
  91. package/scripts/intake/enhancements.py +106 -0
  92. package/scripts/intake/hooks.py +90 -0
  93. package/scripts/intake/prefs.py +76 -0
  94. package/scripts/intake/prompts.py +85 -0
  95. package/scripts/intake/session.py +152 -0
  96. package/scripts/lint_file_layers.py +126 -0
  97. package/scripts/orchestrator.py +68 -17
  98. package/scripts/pre_verdict_gate.py +21 -7
  99. package/scripts/skill_lint.py +11 -1
  100. package/scripts/skill_payload.py +3 -3
  101. package/scripts/test_adversarial_empirical.py +70 -6
  102. package/skill/agents/evidence-judge.md +49 -7
  103. package/skill/workflows/experimental-jev.md +170 -0
  104. package/skill/workflows/intake.md +120 -0
  105. package/visualization/eduevidence-report/scripts/build_infographics.py +32 -14
  106. package/visualization/eduevidence-report/scripts/build_report.py +75 -662
  107. package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
  108. package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
  109. package/visualization/eduevidence-report/scripts/zh_labels.py +61 -0
  110. package/scripts/build_esl_artifacts.py +0 -1921
  111. package/scripts/build_killer_demo.py +0 -295
  112. package/scripts/enrich_projects_human_and_lieflat.py +0 -315
  113. package/scripts/generate_new_projects.py +0 -686
  114. package/scripts/sync_killer_demo_report.py +0 -270
@@ -24,7 +24,8 @@ Conservative verdict rules (offline preliminary gate):
24
24
  any matched contradict entry -> reject
25
25
  else any matched support entry -> pilot
26
26
  else -> insufficient_evidence
27
- adopt is NEVER returned by the preliminary gate.
27
+ The preliminary gate is a screening bound only; a full ADOPT still requires
28
+ the deterministic decision_action gate (High + support + direct primary).
28
29
 
29
30
  Output:
30
31
  {"verdict": ..., "coverage": {"matched_entries": [...],
@@ -228,8 +229,9 @@ def preliminary_verdict(question: str, *, top_k: int = 10) -> dict[str, Any]:
228
229
  claim_text + effect_summary + title; the top_k entries are considered and an
229
230
  entry counts as matched when overlap >= MATCH_THRESHOLD and it shares at
230
231
  least MIN_SHARED_BIGRAMS tokens. Verdict: contradict => reject, else
231
- support => pilot, else insufficient_evidence. Never adopt. Never crashes on
232
- empty/blank questions.
232
+ support => pilot, else insufficient_evidence. This is a screening bound
233
+ only; full ADOPT is decided by engine.decision_policy.decision_action.
234
+ Never crashes on empty/blank questions.
233
235
  """
234
236
  try:
235
237
  top_k = int(top_k)
@@ -298,6 +300,7 @@ def _build_note(
298
300
  f"离线初步裁决在 top_k={top_k} 内匹配到 {len(matched)} 条内置证据:"
299
301
  f"support={counts['support']}、contradict={counts['contradict']}、"
300
302
  f"neutral={counts['neutral']};匹配结局词:{outcome_str}。"
301
- "本裁决为初步(preliminary=true)且保守,从不直接给出 adopt,"
303
+ "本裁决为初步(preliminary=true)筛查上界;完整四态裁决仍须经 "
304
+ "decision_action 闸门(High + support + 主结果直接证据才可 adopt)。"
302
305
  "建议结合完整证据库与在线检索复核。"
303
306
  )
@@ -40,14 +40,14 @@ from pathlib import Path
40
40
  from engine.contracts import validate_record
41
41
  from engine.decision_policy import (
42
42
  ADOPT_DIRECTNESS,
43
- decision_action as _policy_decision_action,
43
+ decision_outcome,
44
44
  outcome_category,
45
45
  primary_effect_categories,
46
46
  )
47
47
  from engine.graph_store import GraphStore
48
48
  from engine.ids import new_local_id
49
49
  from engine.project import ProjectWorkspace
50
- from engine.semantics import claim_relation, decision_implication
50
+ from engine.semantics import decision_implication
51
51
  from engine.synthesis import ClaimSynthesis, synthesize_project
52
52
  from engine.versions import (
53
53
  CONFIDENCE_POLICY_VERSION,
@@ -199,6 +199,7 @@ def _confidence(store: GraphStore, syntheses: tuple[ClaimSynthesis, ...]) -> dic
199
199
  if relation in ("support_adoption", "oppose_adoption"):
200
200
  decisive[sid] = relation
201
201
  elif relation == "conditional":
202
+ decisive[sid] = "conditional"
202
203
  critical_uncertainty_units += 1
203
204
  quality_sum += (
204
205
  audit.get("design_quality", 0) + audit.get("sample_quality", 0)
@@ -209,7 +210,8 @@ def _confidence(store: GraphStore, syntheses: tuple[ClaimSynthesis, ...]) -> dic
209
210
  directness_sum += sum(dirs) / len(dirs) / 2.0
210
211
  directness_count += 1
211
212
 
212
- n_decisive = len(decisive)
213
+ n_decisive = sum(1 for r in decisive.values()
214
+ if r in ("support_adoption", "oppose_adoption"))
213
215
  n_usable = len(usable_studies)
214
216
  if n_decisive == 0:
215
217
  return {"score": 0.0, "label": "Insufficient",
@@ -280,22 +282,6 @@ def _has_direct_primary_evidence(store: GraphStore,
280
282
  return False
281
283
 
282
284
 
283
- def _decision_action(syn_statuses: dict[str, str], confidence: dict,
284
- decisive_relations: dict[str, str],
285
- has_direct_primary_evidence: bool = False) -> str:
286
- """Gate-enforced decision action; rule lives in engine.decision_policy.
287
-
288
- The V1 Pre-Verdict Gate enforces the same rule, so the thresholds are
289
- imported rather than restated here.
290
- """
291
- return _policy_decision_action(
292
- confidence_label=confidence["label"],
293
- decisive_relations=decisive_relations,
294
- has_direct_primary_evidence=has_direct_primary_evidence,
295
- )
296
-
297
-
298
-
299
285
  def adjudicate(store: GraphStore, *, project: ProjectWorkspace,
300
286
  claim_syntheses: tuple[ClaimSynthesis, ...] | None = None,
301
287
  applicability: dict | None = None,
@@ -305,13 +291,19 @@ def adjudicate(store: GraphStore, *, project: ProjectWorkspace,
305
291
  confidence = _confidence(store, syntheses)
306
292
  applicability = applicability or {"boundary": "evidence scope", "notes": ""}
307
293
 
308
- syn_statuses = {s.claim_id: s.status for s in syntheses}
309
294
  decisive_relations = confidence.get("decisive_relations", {})
310
295
  domain = str(project.manifest().get("domain") or "education")
311
296
  direct_learning = _has_direct_primary_evidence(store, decisive_relations, domain)
312
297
 
313
- decision = _decision_action(syn_statuses, confidence, decisive_relations,
314
- has_direct_primary_evidence=direct_learning)
298
+ # Single authority: engine.decision_policy owns the four-state matrix, and
299
+ # engine/tribunal.py keeps no private copy of the rule that could drift.
300
+ outcome = decision_outcome(
301
+ confidence_label=str(confidence.get("label") or ""),
302
+ decisive_relations=decisive_relations,
303
+ has_direct_primary_evidence=direct_learning,
304
+ )
305
+ decision = str(outcome["action"])
306
+ downgrade_reason = outcome.get("downgrade_reason")
315
307
 
316
308
  key_links: list[str] = []
317
309
  for syn in syntheses:
@@ -322,7 +314,8 @@ def adjudicate(store: GraphStore, *, project: ProjectWorkspace,
322
314
  for syn in syntheses:
323
315
  risks.extend(syn.unresolved_conflicts)
324
316
  if not risks and decision == "ADOPT":
325
- risks.append("long-term retention/transfer may still be untested")
317
+ risks.append(
318
+ "primary-outcome durability beyond the observed window may still be untested")
326
319
 
327
320
  missing: list[str] = []
328
321
  for syn in syntheses:
@@ -343,6 +336,7 @@ def adjudicate(store: GraphStore, *, project: ProjectWorkspace,
343
336
  snapshot = {
344
337
  "decision_snapshot_id": new_local_id("DEC", set()),
345
338
  "decision": decision,
339
+ "downgrade_reason": downgrade_reason,
346
340
  "confidence_label": confidence["label"],
347
341
  "confidence_score_internal": confidence["score"],
348
342
  "claim_assessments": claim_assessments,
@@ -4,7 +4,7 @@ Policy versions are frozen identifiers, not free-form strings: changing a
4
4
  policy requires a new version, never silent mutation of an existing one.
5
5
  """
6
6
 
7
- ENGINE_VERSION = "6.2.0"
7
+ ENGINE_VERSION = "6.3.0"
8
8
  GRAPH_SCHEMA_VERSION = "2.0"
9
9
  SOURCE_VALIDATION_POLICY_VERSION = "2026-08-12.v2"
10
10
  METHODOLOGY_POLICY_VERSION = "2026-08-12.v2"