eduevidence 6.2.0 → 6.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/CHANGELOG.md +395 -0
  2. package/README.md +22 -13
  3. package/README.zh-CN.md +15 -8
  4. package/SKILL.md +10 -9
  5. package/benchmarks/evidence-library.json +277 -1
  6. package/docs/architecture.md +6 -3
  7. package/docs/j-ev-experimental.md +250 -0
  8. package/docs/reproducibility.md +138 -0
  9. package/domains/_neutral/copy/few_shots.json +21 -0
  10. package/domains/_neutral/copy/framing_lexicon.json +19 -0
  11. package/domains/_neutral/copy/module_labels.json +5 -0
  12. package/domains/_neutral/copy/module_labels_footer.json +102 -0
  13. package/domains/_neutral/copy/module_labels_modules.json +204 -0
  14. package/domains/_neutral/copy/module_labels_nav.json +126 -0
  15. package/domains/_neutral/copy/module_labels_summary.json +98 -0
  16. package/domains/_neutral/copy/module_labels_tables.json +164 -0
  17. package/domains/_neutral/copy/module_labels_v2.json +90 -0
  18. package/domains/_neutral/copy/risk_constructs.json +20 -0
  19. package/domains/_neutral/copy/section_titles.json +66 -0
  20. package/domains/_neutral/copy/terminology.json +11 -0
  21. package/domains/check_copy_packs.py +103 -0
  22. package/domains/education/copy/few_shots.json +22 -0
  23. package/domains/education/copy/framing_enums.json +167 -0
  24. package/domains/education/copy/framing_lexicon.json +166 -0
  25. package/domains/education/copy/module_labels.json +169 -0
  26. package/domains/education/copy/risk_constructs.json +48 -0
  27. package/domains/education/copy/section_titles.json +186 -0
  28. package/domains/education/copy/terminology.json +70 -0
  29. package/domains/education/manifest.json +1 -1
  30. package/domains/education/outcome_taxonomy.json +2 -2
  31. package/domains/manifest.json +1 -1
  32. package/domains/policy/copy/few_shots.json +22 -0
  33. package/domains/policy/copy/framing_enums.json +94 -0
  34. package/domains/policy/copy/framing_lexicon.json +174 -0
  35. package/domains/policy/copy/module_labels.json +168 -0
  36. package/domains/policy/copy/risk_constructs.json +33 -0
  37. package/domains/policy/copy/section_titles.json +186 -0
  38. package/domains/policy/copy/terminology.json +64 -0
  39. package/engine/capabilities.py +57 -5
  40. package/engine/decision_policy.py +88 -17
  41. package/engine/library_builtin.py +7 -4
  42. package/engine/tribunal.py +17 -23
  43. package/engine/versions.py +1 -1
  44. package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +4 -4
  45. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +4 -4
  46. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +4 -4
  47. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
  48. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +4 -4
  49. package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +4 -4
  50. package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
  51. package/examples/spaced-retrieval-practice/report.html +2522 -0
  52. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +4 -4
  53. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +4 -4
  54. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
  55. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +4 -4
  56. package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +4 -4
  57. package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
  58. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +36 -36
  59. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +36 -36
  60. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +36 -36
  61. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +36 -36
  62. package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +36 -36
  63. package/integrations/jev/__init__.py +115 -0
  64. package/integrations/jev/approval.py +212 -0
  65. package/integrations/jev/cli.py +84 -0
  66. package/integrations/jev/config.py +112 -0
  67. package/integrations/jev/gateway.py +128 -0
  68. package/integrations/jev/modes.py +38 -0
  69. package/integrations/jev/tools_classify.py +88 -0
  70. package/integrations/jev/tools_extract.py +111 -0
  71. package/integrations/jev/tools_rerank.py +71 -0
  72. package/integrations/jev/tools_screen.py +87 -0
  73. package/integrations/jev/tools_verify.py +95 -0
  74. package/integrations/jev_mcp.py +22 -0
  75. package/integrations/semantic_decide.py +286 -0
  76. package/integrations/semdecide_cli.py +55 -0
  77. package/package.json +9 -1
  78. package/pyproject.toml +1 -1
  79. package/references/report-copy-style.md +43 -3
  80. package/schemas/v2/decision-snapshot.schema.json +20 -9
  81. package/schemas/v2/intake.schema.json +191 -0
  82. package/scripts/build_evidence_library.py +15 -5
  83. package/scripts/dashboard_server.py +13 -2
  84. package/scripts/intake/__init__.py +31 -0
  85. package/scripts/intake/__main__.py +18 -0
  86. package/scripts/intake/background.py +78 -0
  87. package/scripts/intake/browser.py +79 -0
  88. package/scripts/intake/cli.py +57 -0
  89. package/scripts/intake/constants.py +57 -0
  90. package/scripts/intake/depth.py +53 -0
  91. package/scripts/intake/enhancements.py +106 -0
  92. package/scripts/intake/hooks.py +90 -0
  93. package/scripts/intake/prefs.py +76 -0
  94. package/scripts/intake/prompts.py +85 -0
  95. package/scripts/intake/session.py +152 -0
  96. package/scripts/lint_file_layers.py +126 -0
  97. package/scripts/orchestrator.py +68 -17
  98. package/scripts/pre_verdict_gate.py +21 -7
  99. package/scripts/skill_lint.py +11 -1
  100. package/scripts/skill_payload.py +3 -3
  101. package/scripts/test_adversarial_empirical.py +70 -6
  102. package/skill/agents/evidence-judge.md +49 -7
  103. package/skill/workflows/experimental-jev.md +170 -0
  104. package/skill/workflows/intake.md +120 -0
  105. package/visualization/eduevidence-report/scripts/build_infographics.py +32 -14
  106. package/visualization/eduevidence-report/scripts/build_report.py +75 -662
  107. package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
  108. package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
  109. package/visualization/eduevidence-report/scripts/zh_labels.py +61 -0
  110. package/scripts/build_esl_artifacts.py +0 -1921
  111. package/scripts/build_killer_demo.py +0 -295
  112. package/scripts/enrich_projects_human_and_lieflat.py +0 -315
  113. package/scripts/generate_new_projects.py +0 -686
  114. package/scripts/sync_killer_demo_report.py +0 -270
@@ -60,6 +60,13 @@ from pre_verdict_gate import apply_enforcement, evaluate_workspace # noqa: E402
60
60
  from engine.versions import ENGINE_VERSION # noqa: E402
61
61
  from engine.log import enable_console_logging, get_log # noqa: E402
62
62
 
63
+ # One-shot two-round intake + prefs (scripts/intake/). Soft-import so the
64
+ # orchestrator still starts if intake is stripped from a minimal package.
65
+ try:
66
+ from intake import hooks as _intake_hooks # noqa: E402
67
+ except Exception: # pragma: no cover - optional adapter
68
+ _intake_hooks = None
69
+
63
70
  log = get_log("orchestrator")
64
71
 
65
72
  DEPTH_ALIASES = {"quick": "S", "standard": "M", "deep": "L"}
@@ -546,6 +553,10 @@ def _run_adjudicate(ws: RunWorkspace, question: str,
546
553
  json.dumps(final, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
547
554
 
548
555
  post = evaluate_workspace(ws.path, require_final=True)
556
+ if not post.get("passed") or post.get("critical_failures"):
557
+ final = apply_enforcement(final, post)
558
+ (ws.path / "final_verdict.json").write_text(
559
+ json.dumps(final, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
549
560
  gate_report = {"run_id": ws.run_id, "stage": "adjudicate",
550
561
  "pre": {"passed": pre["passed"], "max_confidence": pre["max_confidence"],
551
562
  "critical_failures": pre["critical_failures"]},
@@ -571,8 +582,7 @@ def _run_adjudicate(ws: RunWorkspace, question: str,
571
582
 
572
583
  def _assemble_result(ws: RunWorkspace, manifest: dict[str, Any]) -> dict[str, Any]:
573
584
  """Assemble result.json from workspace artifacts (decision=final_verdict.json)."""
574
- from build_result import (NOT_CAPTURED_USAGE, OUTCOME_ORDER,
575
- aggregate_outcomes, build_claims,
585
+ from build_result import (NOT_CAPTURED_USAGE, aggregate_outcomes, build_claims,
576
586
  build_outcome_mapping, derive_provenance)
577
587
 
578
588
  frame = load_json(ws.path / "frame.json")
@@ -976,13 +986,45 @@ def load_global_approval() -> dict | None:
976
986
 
977
987
 
978
988
  def _cmd_run(args: argparse.Namespace) -> int:
979
- approve = args.approve_agent_mcp or interactive_agent_mcp_setup(args.approve_agent_mcp)
980
- approval_record = None
981
- if approve:
982
- approval_record = load_global_approval()
983
- ws = init_run(Path(args.runs_dir), args.question, depth=args.depth, run_id=args.run_id,
989
+ """Create a run workspace. Delegates intake/open to scripts/intake/hooks.py."""
990
+ assume_yes = bool(getattr(args, "yes", False))
991
+ domain = getattr(args, "domain", "education")
992
+ enhancements = list(getattr(args, "enhancement", None) or []) or None
993
+ intake_record: dict[str, Any] | None = None
994
+
995
+ if _intake_hooks is not None:
996
+ ctx = _intake_hooks.prepare_run(
997
+ question=args.question,
998
+ depth=getattr(args, "depth", None),
999
+ enhancements=enhancements,
1000
+ domain=domain,
1001
+ assume_yes=assume_yes,
1002
+ )
1003
+ if ctx.get("error"):
1004
+ print(f"ERROR: intake failed: {ctx['error']}", file=sys.stderr)
1005
+ return 2
1006
+ question, depth_arg = ctx["question"], ctx["depth"]
1007
+ enhancements, intake_record = ctx["enhancements"], ctx["intake_record"]
1008
+ wants_mcp = ctx["wants_mcp"]
1009
+ else:
1010
+ question = args.question
1011
+ depth_arg = DEPTH_ALIASES.get(getattr(args, "depth", None) or "M",
1012
+ getattr(args, "depth", None) or "M")
1013
+ if depth_arg not in DEPTHS:
1014
+ depth_arg = "M"
1015
+ wants_mcp = bool(enhancements and "agent_mcp" in enhancements
1016
+ and "none" not in enhancements)
1017
+
1018
+ approve = bool(args.approve_agent_mcp)
1019
+ if wants_mcp and not assume_yes and sys.stdin.isatty():
1020
+ approve = approve or interactive_agent_mcp_setup(approve)
1021
+ approval_record = load_global_approval() if approve else None
1022
+
1023
+ ws = init_run(Path(args.runs_dir), question, depth=depth_arg, run_id=args.run_id,
984
1024
  approve_agent_mcp=approve, approval_record=approval_record,
985
- domain=getattr(args, "domain", "education"))
1025
+ domain=domain)
1026
+ if _intake_hooks is not None:
1027
+ _intake_hooks.write_intake_artifact(ws.path, intake_record)
986
1028
  print(f"workspace created: {ws.path}")
987
1029
  print(f"manifest: {json.dumps(ws.load_manifest(), ensure_ascii=False, indent=2)}")
988
1030
  if args.dry_run:
@@ -990,9 +1032,9 @@ def _cmd_run(args: argparse.Namespace) -> int:
990
1032
  return 0
991
1033
  summary = advance(ws, demo_pack=args.demo_pack)
992
1034
  print(json.dumps(summary, ensure_ascii=False, indent=2))
993
- if summary["failures"]:
994
- return 1
995
- return 0
1035
+ if _intake_hooks is not None:
1036
+ _intake_hooks.after_run(ws.path, intake_record=intake_record, summary=summary)
1037
+ return 1 if summary["failures"] else 0
996
1038
 
997
1039
 
998
1040
  def _print_status(ws) -> None:
@@ -1087,7 +1129,7 @@ def _cmd_domain(args) -> int:
1087
1129
 
1088
1130
  def _cmd_living(args) -> int:
1089
1131
  from engine.project import ProjectWorkspace
1090
- from engine.living import create_subscription, refresh, set_subscription_status
1132
+ from engine.living import create_subscription, refresh
1091
1133
  home = _home(args)
1092
1134
  ws = ProjectWorkspace.open(home, args.project)
1093
1135
  if args.action == "subscribe":
@@ -1107,7 +1149,6 @@ def _cmd_living(args) -> int:
1107
1149
  print(drift.get("summary", ""))
1108
1150
  return 0
1109
1151
  if args.action == "status":
1110
- from pathlib import Path as _P
1111
1152
  import json as _json
1112
1153
  p = ws.path / "living" / "subscriptions" / f"{args.subscription}.json"
1113
1154
  if not p.is_file():
@@ -1319,7 +1360,7 @@ def _cmd_study(args) -> int:
1319
1360
  print(e, file=sys.stderr)
1320
1361
  return 1
1321
1362
  from engine.study_design import save_study_design
1322
- path = save_study_design(ws, design)
1363
+ save_study_design(ws, design) # persists study-designs/<id>.json
1323
1364
  print(design["design_id"])
1324
1365
  return 0
1325
1366
 
@@ -1397,7 +1438,8 @@ def _cmd_migrate(args) -> int:
1397
1438
 
1398
1439
  def _cmd_dashboard(args) -> int:
1399
1440
  from dashboard_server import run_dashboard_server
1400
- run_dashboard_server(host=args.host, port=args.port)
1441
+ run_dashboard_server(host=args.host, port=args.port,
1442
+ open_browser=bool(getattr(args, "open", False)))
1401
1443
  return 0
1402
1444
 
1403
1445
 
@@ -1455,13 +1497,20 @@ def main(argv: list[str] | None = None) -> int:
1455
1497
  p_run.add_argument("--domain", default="education", metavar="ID",
1456
1498
  help="registered research domain (default: education; "
1457
1499
  "see `domain list`)")
1458
- p_run.add_argument("--depth", default="M", choices=["quick", "standard", "deep", "S", "M", "L"],
1459
- help="complexity depth (default: standard/M)")
1500
+ p_run.add_argument("--depth", default=None,
1501
+ choices=["quick", "standard", "deep", "S", "M", "L", "auto"],
1502
+ help="complexity depth (S/M/L/auto or quick/standard/deep; "
1503
+ "default: intake/prefs auto guideline)")
1460
1504
  p_run.add_argument("--run-id", default=None, help="explicit run id (default: timestamp)")
1461
1505
  p_run.add_argument("--demo-pack", default=None, type=Path,
1462
1506
  help="seed external stages from an example pack (demo/test mode)")
1463
1507
  p_run.add_argument("--approve-agent-mcp", action="store_true",
1464
1508
  help="record agent-mcp approval in the manifest")
1509
+ p_run.add_argument("--yes", action="store_true",
1510
+ help="skip the two-round intake prompts; use prefs + flags unattended")
1511
+ p_run.add_argument("--enhancement", action="append", default=None,
1512
+ choices=["agent_mcp", "jev", "semdecide", "none"],
1513
+ help="execution enhancement (repeatable; used by intake)")
1465
1514
  p_run.add_argument("--dry-run", action="store_true", help="initialize only, do not advance")
1466
1515
  p_run.add_argument("--runs-dir", default=os.environ.get("EDUEVIDENCE_RUNS_DIR", str(ROOT / "runs")),
1467
1516
  help="directory holding run workspaces (default: <repo>/runs)")
@@ -1645,6 +1694,8 @@ def main(argv: list[str] | None = None) -> int:
1645
1694
  p_dash = sub.add_parser("dashboard", help="start local research & token dashboard")
1646
1695
  p_dash.add_argument("--host", default="127.0.0.1")
1647
1696
  p_dash.add_argument("--port", type=int, default=8765)
1697
+ p_dash.add_argument("--open", action="store_true",
1698
+ help="open the Studio URL in the system browser once the server is up")
1648
1699
  p_dash.set_defaults(func=_cmd_dashboard)
1649
1700
 
1650
1701
  p_srch = sub.add_parser("search", help="multi-channel hybrid search")
@@ -1,7 +1,7 @@
1
1
  #!/usr/bin/env python3
2
2
  """pre_verdict_gate.py — Pre-Verdict Gate (Phase 15).
3
3
 
4
- An 11-item checklist that must pass (or be explicitly degraded) before a
4
+ A 12-item checklist that must pass (or be explicitly degraded) before a
5
5
  verdict may carry a high confidence label. The gate is deterministic and
6
6
  reads ONLY the run workspace — no model call, no network.
7
7
 
@@ -417,19 +417,25 @@ def check_decision_action(ws: Path) -> dict[str, str]:
417
417
  label = action or "unset"
418
418
  return _item_res("pass", "action=" + label + " is within the conservative bound")
419
419
 
420
- from engine.decision_policy import ADOPT_REQUIRED_LABEL, decision_action
420
+ from engine.decision_policy import (
421
+ ADOPT_REQUIRED_LABEL,
422
+ decision_outcome,
423
+ )
421
424
 
422
425
  evidence = _load_ws_jsonl(ws, "evidence.jsonl")
423
426
  relations = [decision_relation(ev) for ev in evidence]
424
427
  decisive = {str(index): rel for index, rel in enumerate(relations)
425
- if rel in ("support_adoption", "oppose_adoption")}
428
+ if rel in ("support_adoption", "oppose_adoption",
429
+ "conditional", "conflict", "mixed")}
426
430
  label = str(verdict.get("confidence") or "")
427
431
  summary = _primary_evidence_summary(ws)
428
- expected = decision_action(
432
+ outcome = decision_outcome(
429
433
  confidence_label=label,
430
434
  decisive_relations=decisive,
431
435
  has_direct_primary_evidence=bool(summary.get("direct")),
432
436
  )
437
+ expected = outcome["action"]
438
+ downgrade_reason = outcome.get("downgrade_reason")
433
439
  if expected == "ADOPT":
434
440
  detail = "ADOPT is supported: " + label + " confidence with direct primary-outcome evidence"
435
441
  return _item_res("pass", detail)
@@ -440,8 +446,11 @@ def check_decision_action(ws: Path) -> dict[str, str]:
440
446
  reasons.append("no decisive supporting evidence")
441
447
  if not summary.get("direct"):
442
448
  reasons.append("no primary-outcome evidence at directness 2")
449
+ if downgrade_reason:
450
+ reasons.append("downgrade_reason=" + downgrade_reason)
443
451
  detail = ("recommended_action=adopt is not supported (" + "; ".join(reasons)
444
- + "); the evidence bounds this decision to pilot")
452
+ + "); the evidence bounds this decision to "
453
+ + str(expected).lower())
445
454
  return _item_res("fail", detail)
446
455
 
447
456
 
@@ -580,7 +589,7 @@ GATE_ITEMS: list[dict[str, Any]] = [
580
589
 
581
590
 
582
591
  def evaluate_workspace(workspace: Path, *, require_final: bool = True) -> dict[str, Any]:
583
- """Run the 11-item gate over a run workspace. Pure, deterministic, read-only."""
592
+ """Run the 12-item gate over a run workspace. Pure, deterministic, read-only."""
584
593
  ws = Path(workspace)
585
594
  items: dict[str, dict[str, Any]] = {}
586
595
  for spec in GATE_ITEMS:
@@ -641,11 +650,15 @@ def apply_enforcement(verdict: dict[str, Any], gate: dict[str, Any]) -> dict[str
641
650
  current = out.get("confidence", "Insufficient")
642
651
  if CONFIDENCE_RANK.get(current, 0) > CONFIDENCE_RANK.get(cap, 0):
643
652
  out["confidence"] = cap
653
+ downgrade_reason = None
644
654
  if not gate.get("passed", False) and out.get("recommended_action") == "adopt":
645
655
  out["recommended_action"] = "pilot"
656
+ downgrade_reason = "gate_critical"
646
657
  action_item = (gate.get("items") or {}).get("decision_action_consistency") or {}
647
658
  if action_item.get("status") == "fail" and out.get("recommended_action") == "adopt":
648
659
  out["recommended_action"] = "pilot"
660
+ if downgrade_reason is None:
661
+ downgrade_reason = "missing_direct_primary"
649
662
  extensions = out.setdefault("extensions", {})
650
663
  if not isinstance(extensions, dict):
651
664
  extensions = {}
@@ -659,6 +672,7 @@ def apply_enforcement(verdict: dict[str, Any], gate: dict[str, Any]) -> dict[str
659
672
  "max_confidence": cap,
660
673
  "confidence_before": current,
661
674
  "action_before": verdict.get("recommended_action"),
675
+ "downgrade_reason": downgrade_reason,
662
676
  })
663
677
  return out
664
678
 
@@ -667,7 +681,7 @@ def apply_enforcement(verdict: dict[str, Any], gate: dict[str, Any]) -> dict[str
667
681
 
668
682
 
669
683
  def main(argv: list[str] | None = None) -> int:
670
- parser = argparse.ArgumentParser(description="EduEvidence Pre-Verdict Gate (11-item checklist)")
684
+ parser = argparse.ArgumentParser(description="EduEvidence Pre-Verdict Gate (12-item checklist)")
671
685
  parser.add_argument("--workspace", required=True, help="run workspace directory (runs/<run_id>)")
672
686
  parser.add_argument("--require-final", action="store_true",
673
687
  help="item 11 fails when final_verdict.json is missing (default: warn)")
@@ -13,7 +13,6 @@ Usage:
13
13
  """
14
14
  from __future__ import annotations
15
15
 
16
- import os
17
16
  import re
18
17
  import sys
19
18
  from pathlib import Path
@@ -130,6 +129,17 @@ def lint_skill() -> list[str]:
130
129
  if re.search(r"theme-btn|theme-switcher|data-theme-target", ltext):
131
130
  errors.append("scripts/render_report_html.py still contains runtime theme switcher")
132
131
 
132
+ # 10. File-layer hard gate (also runnable alone: python3 scripts/lint_file_layers.py)
133
+ try:
134
+ sys.path.insert(0, str(ROOT / "scripts"))
135
+ from lint_file_layers import check_file_layers # noqa: PLC0415
136
+ layer_errors, layer_warnings = check_file_layers()
137
+ errors.extend(layer_errors)
138
+ for w in layer_warnings:
139
+ print(f" ! {w}")
140
+ except Exception as exc: # pragma: no cover - lint must still run
141
+ errors.append(f"file-layer gate failed to load: {exc}")
142
+
133
143
  return errors
134
144
 
135
145
 
@@ -7,6 +7,7 @@ import sys
7
7
 
8
8
  TREES = (
9
9
  "agents", "engine", "domains", "skill", "references", "schemas", "scripts",
10
+ "integrations", "visualization/eduevidence-report",
10
11
  "retrieval", "integrations", "visualization/eduevidence-report", "web/studio",
11
12
  "assets/readme",
12
13
  )
@@ -22,7 +23,7 @@ DOCS = (
22
23
  "reproducibility.md", "release-contract.md", "autoresearch-evolution-plan.md",
23
24
  "orchestration-role-model.md", "autoresearch-implementation-status.md",
24
25
  "research-studio-guide.zh-CN.md", "demo-workplace-ai.md",
25
- "sciverse-api.md",
26
+ "sciverse-api.md", "j-ev-experimental.md",
26
27
  "release-closeout/README.md", "release-closeout/issues.md",
27
28
  "release-closeout/frontend-acceptance.md", "release-closeout/verification.md",
28
29
  )
@@ -37,8 +38,7 @@ EXAMPLE_FILES = (
37
38
  )
38
39
  RETIRED_DEMO_SCRIPTS = {
39
40
  "scripts/build_esl_artifacts.py", "scripts/generate_new_projects.py",
40
- "scripts/enrich_projects_human_and_lieflat.py", "scripts/build_killer_demo.py",
41
- "scripts/sync_killer_demo_report.py",
41
+ "scripts/enrich_projects_human_and_lieflat.py",
42
42
  }
43
43
 
44
44
 
@@ -44,7 +44,6 @@ from engine.evidence_graph import (
44
44
  )
45
45
  from engine.gap_lens import gap_lens
46
46
  from engine.semantics import OutcomeClassifier, OutcomeDimension
47
- from engine.tribunal import _confidence, _decision_action, _study_implication
48
47
  from retrieval.corpus_store import DomainCorpusStore, corpus_store
49
48
  from retrieval.search import search_evidence, search_router
50
49
  from scripts.did_regression import run_did_analysis
@@ -444,15 +443,21 @@ def test_offline_corpus_and_network_isolation():
444
443
  def test_dashboard_server_adversarial():
445
444
  print_section("TEST 6: Dashboard Server Concurrency, SSE Disconnection & Source Leakage")
446
445
 
447
- import socketserver
446
+ import http.server
448
447
  from scripts.dashboard_server import StudioHandler
449
448
 
450
- class _ReuseServer(socketserver.TCPServer):
451
- # allow_reuse_address 必须在 bind() 之前生效(构造时读取类属性)
449
+ class _AdversarialServer(http.server.ThreadingHTTPServer):
450
+ # Exercise the class the shipped server actually runs
451
+ # (scripts/dashboard_server.py run_dashboard_server -> ThreadingHTTPServer).
452
+ # A single-threaded socketserver.TCPServer answers a 30-client burst from a
453
+ # 5-slot accept queue, and the kernel resets the overflow ([Errno 54]
454
+ # on macOS). That harness choice - not the product - caused the resets.
452
455
  allow_reuse_address = True
456
+ request_queue_size = 128 # the 30 simultaneous connects below must fit
457
+ daemon_threads = True
453
458
 
454
459
  test_port = 0 # 临时端口:避免与常驻服务/上次残留监听冲突
455
- server = _ReuseServer(("127.0.0.1", test_port), StudioHandler)
460
+ server = _AdversarialServer(("127.0.0.1", test_port), StudioHandler)
456
461
  test_port = server.server_address[1]
457
462
 
458
463
  t = threading.Thread(target=server.serve_forever, daemon=True)
@@ -517,9 +522,68 @@ def test_dashboard_server_adversarial():
517
522
  futures = [executor.submit(fetch_data, i) for i in range(30)]
518
523
  statuses = [f.result() for f in futures]
519
524
  elapsed = time.time() - t0
525
+ assert statuses == [200] * 30, (
526
+ "30 concurrent /api/projects requests did not all succeed: "
527
+ f"statuses={sorted(set(statuses))}")
520
528
  print(f" - 30 concurrent requests finished in {elapsed:.3f}s (all status: 200).")
521
- print(f" - Note: TCPServer is single-threaded; requests are strictly serialized.")
529
+ print(f" - Note: the shipped server is ThreadingHTTPServer; requests are served concurrently.")
522
530
  results["concurrency_30_time"] = elapsed
531
+ results["concurrency_30_statuses"] = statuses
532
+
533
+ # 4. Host / CSRF guards (dashboard_server.py do_GET) must fail closed on
534
+ # the projection API, without blocking legitimate same-origin readers.
535
+ def raw_get(request: str) -> tuple[int, str]:
536
+ """Send one raw HTTP request; return (status_code, body snippet)."""
537
+ sock = socket.create_connection(("127.0.0.1", test_port), timeout=5)
538
+ sock.settimeout(3)
539
+ try:
540
+ sock.sendall(request.encode("latin-1"))
541
+ buf = b""
542
+ while len(buf) < 16384:
543
+ try:
544
+ chunk = sock.recv(4096)
545
+ except socket.timeout:
546
+ break
547
+ if not chunk:
548
+ break
549
+ buf += chunk
550
+ finally:
551
+ sock.close()
552
+ head, _, body = buf.partition(b"\r\n\r\n")
553
+ first_line = head.split(b"\r\n", 1)[0].decode("latin-1", errors="ignore")
554
+ fields = first_line.split(" ")
555
+ status = int(fields[1]) if len(fields) > 1 and fields[1].isdigit() else 0
556
+ return status, body.decode("utf-8", errors="ignore")[:200]
557
+
558
+ untrusted_status, untrusted_body = raw_get(
559
+ "GET /api/projects HTTP/1.1\r\nHost: evil.example.com\r\n\r\n")
560
+ assert untrusted_status == 403, (
561
+ "an untrusted Host header reached the projection API: "
562
+ f"status={untrusted_status}")
563
+ assert "untrusted host" in untrusted_body, untrusted_body
564
+
565
+ csrf_status, csrf_body = raw_get(
566
+ f"GET /api/projects HTTP/1.1\r\nHost: 127.0.0.1:{test_port}\r\n"
567
+ "Sec-Fetch-Site: cross-site\r\n\r\n")
568
+ assert csrf_status == 403, (
569
+ f"a cross-site request reached the projection API: status={csrf_status}")
570
+ assert "cross-origin" in csrf_body, csrf_body
571
+
572
+ same_origin_status, _ = raw_get(
573
+ f"GET /api/projects HTTP/1.1\r\nHost: 127.0.0.1:{test_port}\r\n"
574
+ "Sec-Fetch-Site: same-origin\r\n\r\n")
575
+ assert same_origin_status == 200, (
576
+ f"a legitimate same-origin reader was blocked: status={same_origin_status}")
577
+
578
+ print("\n[*] Host / CSRF guard probes:")
579
+ print(f" - untrusted Host -> {untrusted_status} ({untrusted_body.strip()})")
580
+ print(f" - Sec-Fetch-Site: cross-site -> {csrf_status} ({csrf_body.strip()})")
581
+ print(f" - Sec-Fetch-Site: same-origin -> {same_origin_status}")
582
+ results["host_csrf_guard"] = {
583
+ "untrusted_host": untrusted_status,
584
+ "cross_site": csrf_status,
585
+ "same_origin": same_origin_status,
586
+ }
523
587
 
524
588
  finally:
525
589
  server.shutdown()
@@ -24,12 +24,54 @@ critical_path: true
24
24
 
25
25
  ## 四态决策规则(硬标准)
26
26
 
27
- | 决策 | 要求 |
28
- |---|---|
29
- | ADOPT | 多项关键 Outcome 有较强直接证据 + 风险可控 + 场景匹配 |
30
- | PILOT | 有积极证据,但长期效果/迁移/风险仍不明确 |
31
- | REJECT | 关键结果稳定负效应,或风险明显大于收益 |
32
- | INSUFFICIENT EVIDENCE | 来源不足 / 直接性差 / 设计弱 / 冲突无法解释 |
27
+ | 决策 | 要求 | downgrade_reason |
28
+ |---|---|---|
29
+ | ADOPT | High + support + 主结果 directness=2 直接证据 + 风险可控 + 场景匹配 | (无) |
30
+ | PILOT | High + support 但缺主结果直接证据;或 Moderate + support + 可验证主结果路径 | `missing_direct_primary` / `confidence_band` |
31
+ | REJECT | 关键结果稳定负效应(oppose),或风险明显大于收益 | (无) |
32
+ | INSUFFICIENT EVIDENCE | Low / 间接(Moderate 无可验证主结果路径)/ 冲突无法解释 / 来源不足 | `confidence_band` / `missing_direct_primary` |
33
+
34
+ 闸门强制降级(`GATE_CRITICAL_FAILURE`)时记录 `downgrade_reason=gate_critical`。
35
+
36
+ ### 四态均衡 Few-shot
37
+
38
+ **ADOPT**(High + support + 主结果直接证据)
39
+ ```json
40
+ {
41
+ "confidence": "High",
42
+ "recommended_action": "adopt",
43
+ "decision_rationale": "延迟保持与迁移两项主要结果上均有直接且一致的支持证据,风险可控,场景高度匹配,因此可以推广。"
44
+ }
45
+ ```
46
+
47
+ **PILOT**(Moderate + support + 可验证主结果路径;或 High 缺主结果直接证据)
48
+ ```json
49
+ {
50
+ "confidence": "Moderate",
51
+ "recommended_action": "pilot",
52
+ "decision_rationale": "任务表现与部分学习指标方向积极,主要结果上存在可验证的测量路径,但长期保持与迁移仍不确定,因此建议有边界的试点。"
53
+ }
54
+ ```
55
+
56
+ **REJECT**(oppose / 稳定负效应)
57
+ ```json
58
+ {
59
+ "confidence": "Moderate",
60
+ "recommended_action": "reject",
61
+ "decision_rationale": "多项独立研究在关键结果上呈现稳定负效应,且风险大于收益,因此不建议采纳。"
62
+ }
63
+ ```
64
+
65
+ **INSUFFICIENT EVIDENCE**(Low / 间接 / 冲突)
66
+ ```json
67
+ {
68
+ "confidence": "Low",
69
+ "recommended_action": "insufficient_evidence",
70
+ "decision_rationale": "现有来源多为间接证据且结论互相冲突,主要结果上缺少可验证路径,因此证据不足以支持采纳或试点。"
71
+ }
72
+ ```
73
+
74
+ 禁止把四态收成「一律 PILOT」:ADOPT 与 INSUFFICIENT、REJECT 都必须按上表真实可达。
33
75
 
34
76
  ## 输入
35
77
 
@@ -154,6 +196,6 @@ critical_path: true
154
196
  | 失败 | 处理 |
155
197
  |---|---|
156
198
  | `PRE_VERDICT_FAILED` | 修复前置产物后重跑闸门,不得跳过。 |
157
- | `GATE_CRITICAL_FAILURE` | 封顶置信度,强制降级为 PILOT 或 INSUFFICIENT EVIDENCE。 |
199
+ | `GATE_CRITICAL_FAILURE` | 封顶置信度,强制降级为 PILOT 或 INSUFFICIENT EVIDENCE,并记录 `downgrade_reason=gate_critical`。 |
158
200
  | `CONFLICT_UNRESOLVED` | 保持不确定,不强行裁决。 |
159
201
  | 证据只支持任务表现 | 不得产出学习效果类结论。 |
@@ -0,0 +1,170 @@
1
+ ---
2
+ name: experimental-jev
3
+ description: Optional Jev / SemDecide Tier-0 acceleration overlay. Modes 0 standard / 1 jev-mcp / 2 semdecide / 3 hybrid. Fail-closed. Never replaces Pre-Verdict Gate or Skeptic.
4
+ ---
5
+
6
+ # Experimental Jev Overlay
7
+
8
+ Use only when the run explicitly enables the overlay: `eduevidence run --enhancement jev`
9
+ (and/or `--enhancement semdecide`, which is recorded in the run's `intake.json`), or
10
+ `EDU_EXPERIMENTAL_JEV=<mode>` for the detect CLIs (`--experimental <0|1|2|3>` is a flag of
11
+ `integrations/jev_mcp.py` and `integrations/semantic_decide.py`, not of the host CLI).
12
+ This is an **acceleration overlay** on the scientific protocol — not a fourth workflow,
13
+ not a substitute for scientific gates, and not a source of evidence.
14
+
15
+ Default without an explicit selection: **mode 0 standard** (existing large-model pipeline only).
16
+
17
+ ## Modes
18
+
19
+ | Mode | Name | Backing | When |
20
+ |---:|---|---|---|
21
+ | 0 | standard | host large model only | default; no experimental flag |
22
+ | 1 | jev-mcp | `integrations/jev/` → FreeJev or Vercel AI Gateway `typesafe-ai/jev` (urllib) or documented MCP (`https://freejev.org/mcp`, `npx -y @jkudish/jev-mcp`) | Tier-0 typed judgments, fast & cheap |
23
+ | 2 | semdecide | `integrations/semantic_decide.py` → `semdecide is / filter / choose` | Unix-pipeline predicates, filtering, routing |
24
+ | 3 | hybrid | 1 + 2 together | screen/rerank/extract/classify/verify via Jev; local predicates via SemDecide |
25
+
26
+ Enable:
27
+
28
+ ```bash
29
+ # resolve mode + detect backends (no network judgment yet)
30
+ python3 integrations/jev_mcp.py --experimental 3
31
+ python3 integrations/semantic_decide.py --experimental 3
32
+
33
+ # user confirmation gate (writes ~/.eduevidence/jev_mcp_approval.json)
34
+ python3 integrations/jev_mcp.py --experimental 3 --approve
35
+
36
+ # host CLI (documented entry; the enhancement selection turns the overlay on for the run)
37
+ eduevidence run --question "..." --run-id <id> --enhancement jev --enhancement semdecide
38
+ ```
39
+
40
+ Secrets live only in `~/.eduevidence/env` — never in the repository, never on argv.
41
+
42
+ | Provider | Env | Decide endpoint | MCP (documented only) |
43
+ |---|---|---|---|
44
+ | `freejev` | `FREEJEV_API_KEY`, `JEV_PROVIDER=freejev` | `https://freejev.org/api/v1/decide` | `https://freejev.org/mcp` |
45
+ | `vercel` (default) | `AI_GATEWAY_API_KEY` or `TYPESAFE_API_KEY`, `JEV_PROVIDER=vercel` | Vercel AI Gateway System One | `npx -y @jkudish/jev-mcp` |
46
+
47
+ **FreeJev `request_id` is an idempotency key — never retry the same `request_id`.**
48
+ On timeout / 5xx / unknown outcome, fail-closed escalate this call to the large
49
+ model. A deliberate re-ask must use a **new** `request_id` and its own provenance row.
50
+
51
+ ## Non-negotiable boundaries
52
+
53
+ 1. **Fail-closed.** Any Jev `JEV_INVALID_RESPONSE` / `JEV_PROVIDER_ERROR`, and any
54
+ SemDecide exit **2 / 3 / 4** (plus `SEMDECIDE_UNAVAILABLE` / `SEMDECIDE_TIMEOUT`),
55
+ escalates to the large model (or a human). Never invent, never silently
56
+ accept, never force a binary on an uncertain probability.
57
+ 2. **FreeJev `request_id` 不重试.** Same `request_id` is never resent; escalate instead.
58
+ 2. **Does not replace `scripts/pre_verdict_gate.py`.** Stage Adjudicate still runs the
59
+ Pre-Verdict Gate before any verdict is final.
60
+ 3. **Does not replace Skeptic.** Stage Challenge still executes all nine fixed checks
61
+ (`skill/sub-skills/contradiction-analysis`). `claim_verify` may pre-triage claims only.
62
+ 4. **Does not replace methodology audit / tribunal / outcome separation.**
63
+ 5. **Snippet ≠ evidence (RULE 2).** `content_screen` / `semantic_rerank` operate on
64
+ discovery material only. Only fetched, validated content enters Extract.
65
+ 6. **Verbatim extract only.** `field_extract` returns regex-matched substrings chosen by
66
+ Jev — never model-authored field values. `review` / `not_found` are not extraction.
67
+ 7. **Approval gate.** Network Tier-0 calls require
68
+ `~/.eduevidence/jev_mcp_approval.json` (tools hash + provider + model). Missing or
69
+ mismatched approval → `JEV_APPROVAL_REQUIRED`, escalate.
70
+
71
+ ## Nine-stage mapping (`engine/workflows.py::SCIENTIFIC_STAGE_IDS`)
72
+
73
+ | # | Stage | Mode 1 / 3 Jev Tier-0 | Mode 2 / 3 SemDecide | Still mandatory (never skipped) |
74
+ |---:|---|---|---|---|
75
+ | 1 | Frame | — | `choose` optional route for domain/frame vocabulary (uncertain → planner) | Frame schema; no intervention advice before framing |
76
+ | 2 | Retrieve | `content_screen` (pre-context injection/substance/relevance); `semantic_rerank` (discovery order); `classify_check` (inclusion pre-triage) | `filter` on candidate JSONL; `is` for cheap exclusion predicates | Search plan + counter-evidence queries; Fetch/Validate gate; provenance export |
77
+ | 3 | Extract | `field_extract` (verbatim n / effect / CI candidates); `classify_check` (outcome type pre-label) | `is` for task-vs-learning guard pre-check | Evidence Objects schema; outcome separation; finding ≠ study merge ban |
78
+ | 4 | Challenge | `claim_verify` **assist only** (claim↔evidence relation triage) | `is` for "is there a contradiction signal?" assist | **All nine Skeptic checks**; `NO CONTRADICTORY EVIDENCE FOUND` rule |
79
+ | 5 | Audit | `claim_verify` assist on claim support | `is` for checklist pre-filter | Methodology audit + `task_vs_learning_guard` |
80
+ | 6 | Adjudicate | — (Tier-0 must not score final verdict) | — | Four-state verdict; confidence breakdown; **`scripts/pre_verdict_gate.py`** |
81
+ | 7 | Applicability | `classify_check` optional population/context tags | `choose` optional applicability bounds | Population / conditions / exclusions / uncertainty stated |
82
+ | 8 | Intervene | — | — | Intervention design gates (`intervention_design`) |
83
+ | 9 | Evaluate | — | — | Evaluation design + data validation (`evaluation_design`, `data_validation`) |
84
+
85
+ Projection (report render) stays outside the nine stages and outside this overlay.
86
+
87
+ ## Capability ↔ tool
88
+
89
+ Registered in `engine/capabilities.py` (experimental set; resolve with
90
+ `capability_registry(include_experimental=True)` or `experimental_capability_registry()` —
91
+ omitted from the default scientific registry because Tier-0 is not role-owned work):
92
+
93
+ | capability_id | Tool | Mode |
94
+ |---|---|---|
95
+ | `content_screen` | `jev_mcp.screen` | 1, 3 |
96
+ | `semantic_rerank` | `jev_mcp.rerank` | 1, 3 |
97
+ | `field_extract` | `jev_mcp.extract` | 1, 3 |
98
+ | `classify_check` | `jev_mcp.classify` | 1, 3 |
99
+ | `claim_verify` | `jev_mcp.verify` | 1, 3 |
100
+
101
+ SemDecide wrappers (`integrations/semantic_decide.py`) support stages without new
102
+ capability IDs: `is` / `filter` / `choose`.
103
+
104
+ ## Fail-closed matrix
105
+
106
+ Aligned with `integrations/semantic_decide.py` (`EXIT_MEANINGS`, `ESCALATE_EXIT_CODES`,
107
+ `requires_llm_escalation`, `escalate_plan`) and `integrations/jev/gateway.py` status codes.
108
+
109
+ | Signal | Meaning | Action |
110
+ |---|---|---|
111
+ | Jev `JEV_INVALID_RESPONSE` | typed answer / `answers` envelope missing or malformed | escalate to large model; mark `review` |
112
+ | Jev `JEV_PROVIDER_ERROR` | transport / HTTP 401, 403, 429, 529, 5xx | escalate this call only; record attempt |
113
+ | FreeJev `request_id` failure/timeout | idempotent decide attempt already keyed | **do not retry same `request_id`**; escalate / new id only |
114
+ | `JEV_APPROVAL_REQUIRED` | no/wrong approval file | do not call network; escalate / ask user to approve |
115
+ | `JEV_UNAVAILABLE` | no key and no MCP path | mode 0 standard pipeline |
116
+ | SemDecide exit `0` | `true/selected/match` | semantic result usable |
117
+ | SemDecide exit `1` | `false/no match` | usable; `filter` exit 1 is **not** an escalation by itself |
118
+ | SemDecide exit `2` | `invalid input` | repair input and re-run, or hand the whole decision to the large model |
119
+ | SemDecide exit `3` | `uncertain` | escalate to large model / human; never force true/false |
120
+ | SemDecide exit `4` | `provider failure` | degrade to large model for this call only; record the attempt |
121
+ | `SEMDECIDE_UNAVAILABLE` | binary missing / spawn failed | escalate (`requires_llm_escalation`) |
122
+ | `SEMDECIDE_TIMEOUT` | CLI timed out | escalate (`escalate=ESCALATE_TO_LLM`) |
123
+ | Screen `block` / `review` | injection or low trust | do not enter context; discard or human review |
124
+ | Extract `review` / `not_found` | provisional or absent value | not an extracted field; large model re-extract |
125
+ | Classify `review` | low confidence / margin | large model re-label |
126
+ | Verify `review` / `unknown` | low confidence / invalid | large model claim check; never auto-adopt |
127
+
128
+ ## Degradation path
129
+
130
+ ```
131
+ mode 3 hybrid
132
+ ├─ provider key + approval → Tier-0 screen/rerank/extract/classify/verify
133
+ │ (FreeJev: FREEJEV_API_KEY + JEV_PROVIDER=freejev; request_id never retried)
134
+ ├─ semdecide on PATH → is/filter/choose
135
+ ├─ exit 2/3/4 / SEMDECIDE_TIMEOUT
136
+ │ / JEV_INVALID_RESPONSE / JEV_PROVIDER_ERROR
137
+ │ → large model for THAT call (fail-closed)
138
+ ├─ Jev unavailable → SemDecide only (mode 2 slice)
139
+ ├─ semdecide unavailable → Jev only (mode 1 slice)
140
+ └─ both unavailable / overlay not selected → mode 0 standard (Platform Native)
141
+ ```
142
+
143
+ Scientific gates do not degrade: Pre-Verdict Gate, Skeptic nine checks, methodology
144
+ audit, and the four-state tribunal always run on the host path.
145
+
146
+ ## Acceptance checklist
147
+
148
+ - [ ] The overlay selection is recorded on the run (`intake.json` enhancements); absent
149
+ selection means mode 0.
150
+ - [ ] Approval file present and `tools_hash` matches before any gateway call.
151
+ - [ ] Every Tier-0 result stored with `status` / `confidence` / `escalate` provenance.
152
+ - [ ] All SemDecide exit 2/3/4, `SEMDECIDE_UNAVAILABLE`, and `SEMDECIDE_TIMEOUT` rows
153
+ show an escalation record (no silent coercion). `filter` exit 1 alone is not escalated.
154
+ - [ ] FreeJev calls never reuse a `request_id` for retry; each attempt is one shot.
155
+ - [ ] Stage 4 Challenge still has the nine Skeptic checks + explicit contradiction statement.
156
+ - [ ] Stage 6 Adjudicate still ran `scripts/pre_verdict_gate.py` before finalising.
157
+ - [ ] A/B metrics logged per `docs/j-ev-experimental.md` (speedup口径 + quality gates).
158
+ - [ ] No secrets in artifacts (`AI_GATEWAY_API_KEY` / `TYPESAFE_API_KEY` / `FREEJEV_API_KEY`);
159
+ keys only from `~/.eduevidence/env` / process env.
160
+
161
+ ## Minimal example
162
+
163
+ ```bash
164
+ export EDU_EXPERIMENTAL_JEV=3 # detect CLIs read this; the host run uses --enhancement
165
+ python3 integrations/jev_mcp.py --experimental 3 --approve
166
+ eduevidence run --question "Should first-year CS students use generative AI coding assistants?" \
167
+ --run-id ai-cs1-jev --enhancement jev --enhancement semdecide
168
+ # Scientific path unchanged: Frame → Retrieve → Extract → Challenge → Audit → Adjudicate → Applicability
169
+ # Tier-0 only pre-screens, re-ranks, extracts verbatim fields, pre-labels, pre-verifies claims.
170
+ ```