eduevidence 6.0.0 → 6.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +395 -0
- package/CONTRIBUTING.md +105 -0
- package/README.md +113 -49
- package/README.zh-CN.md +39 -12
- package/SKILL.md +15 -5
- package/assets/readme/landing-tour.gif +0 -0
- package/assets/readme/studio-tour.gif +0 -0
- package/benchmarks/evidence-library.json +277 -1
- package/bin/eduevidence.js +2 -1
- package/docs/architecture.md +325 -46
- package/docs/demo-workplace-ai.md +1 -1
- package/docs/install-guide.md +1 -1
- package/docs/j-ev-experimental.md +250 -0
- package/docs/orchestration-role-model.md +1 -1
- package/docs/release-closeout/README.md +1 -1
- package/docs/reproducibility.md +138 -0
- package/docs/sciverse-api.md +125 -0
- package/domains/_neutral/copy/few_shots.json +21 -0
- package/domains/_neutral/copy/framing_lexicon.json +19 -0
- package/domains/_neutral/copy/module_labels.json +5 -0
- package/domains/_neutral/copy/module_labels_footer.json +102 -0
- package/domains/_neutral/copy/module_labels_modules.json +204 -0
- package/domains/_neutral/copy/module_labels_nav.json +126 -0
- package/domains/_neutral/copy/module_labels_summary.json +98 -0
- package/domains/_neutral/copy/module_labels_tables.json +164 -0
- package/domains/_neutral/copy/module_labels_v2.json +90 -0
- package/domains/_neutral/copy/risk_constructs.json +20 -0
- package/domains/_neutral/copy/section_titles.json +66 -0
- package/domains/_neutral/copy/terminology.json +11 -0
- package/domains/check_copy_packs.py +103 -0
- package/domains/education/copy/few_shots.json +22 -0
- package/domains/education/copy/framing_enums.json +167 -0
- package/domains/education/copy/framing_lexicon.json +166 -0
- package/domains/education/copy/module_labels.json +169 -0
- package/domains/education/copy/risk_constructs.json +48 -0
- package/domains/education/copy/section_titles.json +186 -0
- package/domains/education/copy/terminology.json +70 -0
- package/domains/education/manifest.json +1 -1
- package/domains/education/outcome_taxonomy.json +2 -2
- package/domains/manifest.json +1 -1
- package/domains/policy/copy/few_shots.json +22 -0
- package/domains/policy/copy/framing_enums.json +94 -0
- package/domains/policy/copy/framing_lexicon.json +174 -0
- package/domains/policy/copy/module_labels.json +168 -0
- package/domains/policy/copy/risk_constructs.json +33 -0
- package/domains/policy/copy/section_titles.json +186 -0
- package/domains/policy/copy/terminology.json +64 -0
- package/eduevidence_cli.py +10 -0
- package/engine/capabilities.py +57 -5
- package/engine/decision_policy.py +167 -0
- package/engine/evidence_graph.py +14 -10
- package/engine/gaps.py +42 -22
- package/engine/ids.py +2 -0
- package/engine/library.py +6 -2
- package/engine/library_builtin.py +7 -4
- package/engine/living.py +34 -4
- package/engine/migration.py +88 -3
- package/engine/orchestration.py +5 -5
- package/engine/paths.py +2 -0
- package/engine/pilot.py +34 -32
- package/engine/taxonomy.py +211 -0
- package/engine/tribunal.py +49 -43
- package/engine/versions.py +1 -1
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1361 -147
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
- package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
- package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
- package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +448 -128
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +448 -128
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +448 -128
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +448 -128
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +448 -128
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
- package/examples/ai-coding-assistant-evidence/result.json +13 -9
- package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
- package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
- package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
- package/examples/spaced-retrieval-practice/applicability.json +14 -0
- package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
- package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
- package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
- package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
- package/examples/spaced-retrieval-practice/frame.json +58 -0
- package/examples/spaced-retrieval-practice/gate_report.json +101 -0
- package/examples/spaced-retrieval-practice/methodology.json +78 -0
- package/examples/spaced-retrieval-practice/report.html +2522 -0
- package/examples/spaced-retrieval-practice/report_spec.json +212 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/result.json +942 -0
- package/examples/spaced-retrieval-practice/result.zh.json +942 -0
- package/examples/spaced-retrieval-practice/skeptic.json +70 -0
- package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
- package/examples/spaced-retrieval-practice/verdict.json +93 -0
- package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
- package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -4
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
- package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
- package/examples/workplace-ai-assistant/final_verdict.json +78 -0
- package/examples/workplace-ai-assistant/gate_report.json +101 -0
- package/examples/workplace-ai-assistant/report_spec.json +209 -40
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +449 -119
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +449 -119
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +449 -119
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +449 -119
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +449 -119
- package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/result.json +82 -20
- package/examples/workplace-ai-assistant/result.zh.json +82 -20
- package/examples/workplace-ai-assistant/skeptic.json +72 -0
- package/examples/workplace-ai-assistant/verdict.json +36 -10
- package/integrations/agent_mcp.py +2 -2
- package/integrations/jev/__init__.py +115 -0
- package/integrations/jev/approval.py +212 -0
- package/integrations/jev/cli.py +84 -0
- package/integrations/jev/config.py +112 -0
- package/integrations/jev/gateway.py +128 -0
- package/integrations/jev/modes.py +38 -0
- package/integrations/jev/tools_classify.py +88 -0
- package/integrations/jev/tools_extract.py +111 -0
- package/integrations/jev/tools_rerank.py +71 -0
- package/integrations/jev/tools_screen.py +87 -0
- package/integrations/jev/tools_verify.py +95 -0
- package/integrations/jev_mcp.py +22 -0
- package/integrations/semantic_decide.py +286 -0
- package/integrations/semdecide_cli.py +55 -0
- package/package.json +19 -2
- package/pyproject.toml +4 -3
- package/references/report-copy-style.md +107 -0
- package/references/retrieval-compliance.md +75 -0
- package/references/retrieval-protocol.md +20 -0
- package/retrieval/audit.py +27 -3
- package/retrieval/fetch.py +96 -0
- package/retrieval/sciverse.py +398 -0
- package/retrieval/search.py +47 -7
- package/schemas/applicability.schema.json +94 -0
- package/schemas/chart-spec.schema.json +10 -3
- package/schemas/evidence.schema.json +316 -43
- package/schemas/fetch-result.schema.json +2 -1
- package/schemas/report-result.schema.json +3 -3
- package/schemas/report-spec.schema.json +98 -100
- package/schemas/skeptic.schema.json +86 -0
- package/schemas/source.schema.json +21 -2
- package/schemas/v2/decision-snapshot.schema.json +20 -9
- package/schemas/v2/finding.schema.json +5 -1
- package/schemas/v2/intake.schema.json +191 -0
- package/schemas/v2/methodology-audit.schema.json +5 -1
- package/schemas/v2/outcome.schema.json +28 -5
- package/schemas/v2/study.schema.json +5 -1
- package/schemas/vNext/autoevolve-session.schema.json +34 -1
- package/schemas/vNext/eval-snapshot.schema.json +77 -1
- package/schemas/vNext/execution-plan.schema.json +50 -1
- package/schemas/vNext/gap-priority.schema.json +54 -1
- package/schemas/vNext/negative-search-record.schema.json +68 -1
- package/schemas/vNext/research-iteration.schema.json +87 -1
- package/schemas/vNext/research-strategy.schema.json +62 -1
- package/schemas/vNext/skill-experiment.schema.json +90 -1
- package/schemas/vNext/task-spec.schema.json +156 -1
- package/schemas/vNext/worker-result.schema.json +60 -1
- package/schemas/verdict.schema.json +164 -28
- package/scripts/build_evidence_library.py +15 -5
- package/scripts/build_report_variants.py +18 -2
- package/scripts/build_result.py +74 -9
- package/scripts/check_package_parity.py +85 -0
- package/scripts/check_protocol_alignment.py +375 -0
- package/scripts/check_versioned_schemas.py +254 -0
- package/scripts/claim_audit.py +13 -8
- package/scripts/compute_confidence.py +10 -0
- package/scripts/dashboard_server.py +13 -2
- package/scripts/did_regression.py +12 -2
- package/scripts/evidence_score.py +5 -2
- package/scripts/intake/__init__.py +31 -0
- package/scripts/intake/__main__.py +18 -0
- package/scripts/intake/background.py +78 -0
- package/scripts/intake/browser.py +79 -0
- package/scripts/intake/cli.py +57 -0
- package/scripts/intake/constants.py +57 -0
- package/scripts/intake/depth.py +53 -0
- package/scripts/intake/enhancements.py +106 -0
- package/scripts/intake/hooks.py +90 -0
- package/scripts/intake/prefs.py +76 -0
- package/scripts/intake/prompts.py +85 -0
- package/scripts/intake/session.py +152 -0
- package/scripts/lint_file_layers.py +126 -0
- package/scripts/orchestrator.py +187 -40
- package/scripts/pre_verdict_gate.py +241 -29
- package/scripts/quickstart.py +18 -2
- package/scripts/run_workspace.py +7 -1
- package/scripts/skill_lint.py +11 -1
- package/scripts/skill_payload.py +6 -3
- package/scripts/test_adversarial_empirical.py +96 -25
- package/scripts/validate_schema.py +31 -1
- package/skill/agents/evaluation-designer.md +20 -4
- package/skill/agents/evidence-analyst.md +19 -3
- package/skill/agents/evidence-judge.md +98 -8
- package/skill/agents/evidence-retriever.md +20 -3
- package/skill/agents/intervention-designer.md +20 -4
- package/skill/agents/method-reviewer.md +18 -2
- package/skill/agents/{education-planner.md → research-planner.md} +19 -3
- package/skill/agents/skeptic.md +18 -2
- package/skill/roles/registry.yaml +11 -11
- package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
- package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
- package/skill/sub-skills/data-analysis/SKILL.md +34 -15
- package/skill/sub-skills/ethics-review/SKILL.md +33 -10
- package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
- package/skill/sub-skills/evidence-review/SKILL.md +31 -12
- package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
- package/skill/sub-skills/literature-review/SKILL.md +35 -14
- package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
- package/skill/sub-skills/report-generation/SKILL.md +28 -0
- package/skill/sub-skills/research-planning/SKILL.md +41 -14
- package/skill/sub-skills/study-design/SKILL.md +30 -9
- package/skill/task-briefs/adjudicate.md +32 -7
- package/skill/task-briefs/applicability.md +37 -2
- package/skill/task-briefs/audit.md +32 -7
- package/skill/task-briefs/challenge.md +34 -5
- package/skill/task-briefs/evaluate.md +30 -5
- package/skill/task-briefs/extract.md +31 -8
- package/skill/task-briefs/frame.md +39 -10
- package/skill/task-briefs/intervene.md +32 -6
- package/skill/task-briefs/present.md +32 -8
- package/skill/task-briefs/projection.md +36 -2
- package/skill/task-briefs/retrieve.md +36 -6
- package/skill/workflows/decision-and-pilot.md +76 -1
- package/skill/workflows/evaluate-and-update.md +83 -0
- package/skill/workflows/evidence-review.md +104 -0
- package/skill/workflows/experimental-jev.md +170 -0
- package/skill/workflows/intake.md +120 -0
- package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
- package/visualization/eduevidence-report/scripts/build_infographics.py +37 -15
- package/visualization/eduevidence-report/scripts/build_report.py +435 -575
- package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
- package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
- package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
- package/visualization/eduevidence-report/scripts/zh_labels.py +141 -1
- package/web/architecture.html +14885 -0
- package/web/studio/assets/index-B8tkF44Q.css +1 -0
- package/web/studio/index.html +2 -2
- package/scripts/build_esl_artifacts.py +0 -1921
- package/scripts/build_killer_demo.py +0 -295
- package/scripts/enrich_projects_human_and_lieflat.py +0 -315
- package/scripts/generate_new_projects.py +0 -686
- package/scripts/sync_killer_demo_report.py +0 -270
- package/web/studio/assets/index-CzXocaGv.css +0 -1
- /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
|
@@ -7,4 +7,79 @@ description: Convert a bounded evidence decision into a reversible, measurable p
|
|
|
7
7
|
|
|
8
8
|
Use only after an Evidence Review has produced an auditable decision snapshot. Add `Intervene` with a minimal pilot, explicit stop conditions, owner, population, and outcome measures. A pilot is not an adoption claim.
|
|
9
9
|
|
|
10
|
-
When this workflow is reached from Evidence Autoresearch, require the bridge in `references/autoresearch.md`: the KnowledgeGap is HIGH-DVI and decision-material, remains unresolved, bounded secondary search is saturated, and the empirical study is ethically/operationally feasible. Then still pass the existing grounded StudyDesign gate;
|
|
10
|
+
When this workflow is reached from Evidence Autoresearch, require the bridge in `references/autoresearch.md`: the KnowledgeGap is HIGH-DVI and decision-material, remains unresolved, bounded secondary search is saturated, and the empirical study is ethically/operationally feasible. Then still pass the existing grounded StudyDesign gate; "few papers found" alone never authorizes a pilot.
|
|
11
|
+
|
|
12
|
+
## When this workflow applies
|
|
13
|
+
|
|
14
|
+
- The review reached `PILOT` (or `ADOPT` with conditions) and the decision must become something a team can actually run.
|
|
15
|
+
- The user asks for an intervention plan, a rollout, a phased adoption, a stop rule, or "what would we do next week".
|
|
16
|
+
- A bounded empirical gap must be closed before the institution can move further.
|
|
17
|
+
|
|
18
|
+
Not for: producing the review itself (`evidence-review.md`), analysing data that already exists (`evaluate-and-update.md`), or recommending institution-wide deployment — that is an adoption claim this workflow is designed to prevent.
|
|
19
|
+
|
|
20
|
+
## Prerequisites
|
|
21
|
+
|
|
22
|
+
1. A completed, schema-valid `final_verdict.json` that passed the Pre-Verdict Gate. A pilot design without a review is a guess with extra steps.
|
|
23
|
+
2. An applicability statement — the pilot population must sit inside the supported population.
|
|
24
|
+
3. If the pilot closes an empirical gap: an explicit evidence-grounded KnowledgeGap ID. No KnowledgeGap, no study design.
|
|
25
|
+
|
|
26
|
+
## Runbook
|
|
27
|
+
|
|
28
|
+
| # | Stage | Role | Input | Artifact | Gate |
|
|
29
|
+
|---|---|---|---|---|---|
|
|
30
|
+
| 1–7 | (inherit the review) | — | prior run | `final_verdict.json` + `applicability.json` | the review's own gates must be green |
|
|
31
|
+
| 8 | Intervene | intervention-designer | verdict + frame | `intervention.json` | `schemas/intervention.schema.json`; minimal reversible pilot, not a deployment plan |
|
|
32
|
+
| 8b | (grounding check) | — | `intervention.json` | KnowledgeGap reference | `scripts/complexity_gate.py` + grounded StudyDesign gate |
|
|
33
|
+
| 9 | Evaluate | evaluation-designer | `intervention.json` + verdict | `evaluation.json` | `schemas/evaluation.schema.json`; task vs learning separated, retention and transfer present |
|
|
34
|
+
|
|
35
|
+
### Stage 8 — Intervene
|
|
36
|
+
|
|
37
|
+
Design the minimum viable pilot: phased rollout with explicit usage rules and guardrails, an owner, the participating population, and the evidence it aligns to. Every phase needs a **stop condition** (including a harm/risk stop) and a decision point — weeks, not quarters. Resist scope creep: a pilot that cannot be stopped cheaply is not a pilot.
|
|
38
|
+
|
|
39
|
+
### Stage 8b — Grounding check
|
|
40
|
+
|
|
41
|
+
A design may proceed only when it answers a real, unresolved gap that the review identified and the evidence cannot close. If the gap is already answered by existing evidence, the correct output is an applicability-bounded decision, not a new trial.
|
|
42
|
+
|
|
43
|
+
### Stage 9 — Evaluate
|
|
44
|
+
|
|
45
|
+
Define the measurement plan: baseline, immediate post-test, retention, and transfer, plus process and risk indicators. Success thresholds must be pre-registered; failure and stop thresholds are equally mandatory. Make the analysis plan explicit enough that `scripts/did_regression.py` can execute it on the returned data without further decisions.
|
|
46
|
+
|
|
47
|
+
## Failure handling and fallbacks
|
|
48
|
+
|
|
49
|
+
| Failure | Handling |
|
|
50
|
+
|---|---|
|
|
51
|
+
| `NEEDS_USER_CONTEXT` | Ask for population, duration, staffing, and constraints; never invent an operating context. |
|
|
52
|
+
| `INSUFFICIENT_EVIDENCE` | Stop here and return to the review — a pilot cannot substitute for missing evidence about safety or harm. |
|
|
53
|
+
| `SCOPE_MISMATCH` | Shrink the pilot population until it sits inside the supported population. |
|
|
54
|
+
| No grounded KnowledgeGap | Do not design the study; report what evidence would be needed instead. |
|
|
55
|
+
| Ethics review unresolved | Block the pilot; see `skill/sub-skills/ethics-review/SKILL.md`. |
|
|
56
|
+
|
|
57
|
+
## Human hand-off points
|
|
58
|
+
|
|
59
|
+
- Before the pilot design is accepted: the course owner confirms feasibility, staffing, and time budget.
|
|
60
|
+
- Before enrolment: ethics/IRB review where human participants or personal telemetry are involved.
|
|
61
|
+
- Before every phase transition: the stop-condition check is a human decision, not an automatic continuation.
|
|
62
|
+
|
|
63
|
+
## Resuming and updating
|
|
64
|
+
|
|
65
|
+
Pilot outcome data returns through `reference/` and is re-injected by the `evaluate-and-update` workflow, which commits a new Evidence Graph revision and a new decision snapshot. Never edit a pilot's success threshold after seeing the data.
|
|
66
|
+
|
|
67
|
+
## Minimal example
|
|
68
|
+
|
|
69
|
+
```bash
|
|
70
|
+
eduevidence project create --question "Unguarded AI coding assistance in CS1" --domain education # policy / other registered domains work identically
|
|
71
|
+
eduevidence project show --project PRJ-... # confirm the decision snapshot
|
|
72
|
+
eduevidence study design --project PRJ-... --gap GAP-RETENTION-001 # grounded design
|
|
73
|
+
# write intervention.json (phases + stop conditions) and evaluation.json (thresholds)
|
|
74
|
+
eduevidence pilot register --project PRJ-... --decision DEC-... --title "Phased CS1 pilot"
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
## Acceptance checklist
|
|
78
|
+
|
|
79
|
+
- [ ] The pilot traces to a schema-valid verdict and an applicability boundary.
|
|
80
|
+
- [ ] Every phase has an owner, a duration, a stop condition, and a decision point.
|
|
81
|
+
- [ ] Guardrails and usage rules are explicit (what participants may and may not do).
|
|
82
|
+
- [ ] Evaluation separates task performance from learning and includes retention + transfer.
|
|
83
|
+
- [ ] Success and failure thresholds were set before enrolment.
|
|
84
|
+
- [ ] The KnowledgeGap reference exists, or the workflow exited without a study design.
|
|
85
|
+
- [ ] Nothing in the output claims adoption, deployment, or proven benefit.
|
|
@@ -8,3 +8,86 @@ description: Re-inject validated outcome data into an evidence graph and re-adju
|
|
|
8
8
|
Use when pilot or field data exists. Validate provenance and missingness before analysis, fail closed when inference is not estimable, commit a graph revision, then produce a new decision snapshot and a decision diff.
|
|
9
9
|
|
|
10
10
|
For autonomous/living refreshes, preserve prior revisions and treat the newly re-adjudicated decision as a candidate update until the applicable review/human gate accepts it. New evidence does not need to flip the action; unchanged action with changed certainty, applicability, or boundary is a valid revision.
|
|
11
|
+
|
|
12
|
+
## When this workflow applies
|
|
13
|
+
|
|
14
|
+
- Pilot, field, or survey data has been collected and must now update the decision.
|
|
15
|
+
- New literature or a retraction changes what the existing decision can rest on.
|
|
16
|
+
- A scheduled Living Evidence refresh fires and must feed a decision diff.
|
|
17
|
+
|
|
18
|
+
Not for: running the original review (`evidence-review.md`) or designing the pilot that produced the data (`decision-and-pilot.md`).
|
|
19
|
+
|
|
20
|
+
## Prerequisites
|
|
21
|
+
|
|
22
|
+
1. The dataset plus its collection provenance: who collected it, when, from which population, under which consent.
|
|
23
|
+
2. The prior decision snapshot and the pilot's pre-registered thresholds — they must not be revised now.
|
|
24
|
+
3. Confirmation that the analysis plan was fixed before the data arrived.
|
|
25
|
+
|
|
26
|
+
## Runbook
|
|
27
|
+
|
|
28
|
+
| # | Stage | Role | Input | Artifact | Gate |
|
|
29
|
+
|---|---|---|---|---|---|
|
|
30
|
+
| 1 | Validate data | (deterministic) | raw dataset | `dataset-manifest` | provenance, hash, and missingness checked before any analysis |
|
|
31
|
+
| 2 | Analyse | evaluation-designer / `data-analysis` | manifest + analysis plan | `analysis-run` | fail closed when not estimable; never fabricate p-values |
|
|
32
|
+
| 3 | Merge | (deterministic) | analysis + prior graph | new Evidence Graph revision | append-only; prior revisions preserved |
|
|
33
|
+
| 4 | Re-adjudicate | evidence-judge | revised graph | new decision snapshot | verdict schema + Pre-Verdict Gate re-run |
|
|
34
|
+
| 5 | Diff | (deterministic) | old vs new snapshot | `decision-diff` | every change states which evidence caused it |
|
|
35
|
+
|
|
36
|
+
### Step 1 — Validate before analysing
|
|
37
|
+
|
|
38
|
+
Profile the dataset for missingness, attrition, and provenance. `provenance/hash/missingness before analysis` is a hard gate: an undocumented dataset is not evidence, and a treatment/control label that cannot be verified is not a comparison.
|
|
39
|
+
|
|
40
|
+
### Step 2 — Analyse under the pre-registered plan
|
|
41
|
+
|
|
42
|
+
Run the fixed plan — for the standard classroom case, Difference-in-Differences via `scripts/did_regression.py`, with Hedges' g via `scripts/effect_calculator.py`. If the design is not estimable (no baseline, no control, attrition beyond tolerance), the correct output is `ANALYSIS_NOT_ESTIMABLE`, not a weaker statistic presented as if it were the plan.
|
|
43
|
+
|
|
44
|
+
### Step 3 — Commit a revision, never an overwrite
|
|
45
|
+
|
|
46
|
+
A local finding enters the graph as a new node (`EVD-LOCAL-*`) and a new graph revision. Existing sources, findings, and prior decisions stay untouched — the Single Writer rule applies to the commit.
|
|
47
|
+
|
|
48
|
+
### Step 4 — Re-adjudicate against the whole body of evidence
|
|
49
|
+
|
|
50
|
+
Local data is one study among many. Re-run the four-state decision over the complete evidence set; a single favourable classroom result never outvotes a body of contrary evidence, and a null local result does not erase positive evidence elsewhere.
|
|
51
|
+
|
|
52
|
+
### Step 5 — Publish a decision diff
|
|
53
|
+
|
|
54
|
+
The deliverable is the diff: what changed in action, confidence, applicability, or boundary, and which evidence moved each one. Unchanged action with changed uncertainty is a real result and should be reported as such.
|
|
55
|
+
|
|
56
|
+
## Failure handling and fallbacks
|
|
57
|
+
|
|
58
|
+
| Failure | Handling |
|
|
59
|
+
|---|---|
|
|
60
|
+
| `ANALYSIS_NOT_ESTIMABLE` / `ANALYSIS_CAPABILITY_UNAVAILABLE` | Fail closed: report the design limitation; never substitute a weaker estimator silently. |
|
|
61
|
+
| Missingness above tolerance | Report attrition explicitly; downgrade certainty rather than dropping participants silently. |
|
|
62
|
+
| Thresholds changed post-hoc | Stop; the change is a protocol deviation and must be recorded, not absorbed. |
|
|
63
|
+
| New evidence contradicts on method only | Keep the finding, strip its support role (methodology gate). |
|
|
64
|
+
| Retraction of a cited source | Run `scripts/retraction_watch.py`, remove the source's support, re-adjudicate. |
|
|
65
|
+
|
|
66
|
+
## Human hand-off points
|
|
67
|
+
|
|
68
|
+
- Before analysis: data owner confirms consent and de-identification.
|
|
69
|
+
- Before the new decision is effective: the review/human gate accepts or rejects the candidate update (`living` refresh).
|
|
70
|
+
- When the diff changes a stop condition: the course owner decides whether the pilot continues.
|
|
71
|
+
|
|
72
|
+
## Resuming and updating
|
|
73
|
+
|
|
74
|
+
Autonomous refreshes create candidate updates only. Living Evidence subscriptions keep prior revisions and surface each refresh as a diff for acceptance; nothing overwrites a published decision. Repeated no-change refreshes are recorded, not re-announced.
|
|
75
|
+
|
|
76
|
+
## Minimal example
|
|
77
|
+
|
|
78
|
+
```bash
|
|
79
|
+
eduevidence data ingest --project PRJ-... --file pilot-results.csv
|
|
80
|
+
eduevidence analyze --project PRJ-... --outcome retention --design did
|
|
81
|
+
eduevidence adjudicate --project PRJ-... # new decision snapshot + diff
|
|
82
|
+
eduevidence living status --project PRJ-... # scheduled refresh state
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
## Acceptance checklist
|
|
86
|
+
|
|
87
|
+
- [ ] Dataset provenance, hashing, and missingness recorded before analysis.
|
|
88
|
+
- [ ] Analysis followed the pre-registered plan; non-estimable designs failed closed.
|
|
89
|
+
- [ ] A new graph revision was committed; no prior revision was modified.
|
|
90
|
+
- [ ] Re-adjudication used the full evidence set, not the local data alone.
|
|
91
|
+
- [ ] The decision diff states each change and the evidence that caused it.
|
|
92
|
+
- [ ] Any retracted source was removed from the support structure.
|
|
93
|
+
|
|
@@ -11,3 +11,107 @@ Run `Frame → Retrieve → Extract → Challenge → Audit → Adjudicate → A
|
|
|
11
11
|
The required outputs are a search plan and attempt log, validated sources, claim-level evidence links, a methodology audit, a decision boundary, and applicability limits. Search snippets are discovery metadata, never evidence.
|
|
12
12
|
|
|
13
13
|
When the user asks to continue autonomously, identify the next most decision-relevant evidence, or keep iterating until the evidence state reaches a bounded stopping condition, load `references/autoresearch.md`. Keep the public workflow unchanged: Evidence Autoresearch is a meta-layer over this review, not a fourth user-facing workflow. Preserve append-only evidence and the Single Writer rule.
|
|
14
|
+
|
|
15
|
+
## When this workflow applies
|
|
16
|
+
|
|
17
|
+
- A decision question can be answered from **existing** research — across any field: teaching methods, curriculums, AI tools, policies, programmes, clinical or organisational practice, cultural or media interventions.
|
|
18
|
+
- The user needs to know what the evidence supports, what it cannot support, for whom, and under which conditions.
|
|
19
|
+
- The user explicitly asks for a review, synthesis, appraisal, or an evidence-bounded recommendation.
|
|
20
|
+
|
|
21
|
+
Not for: designing a new study or pilot (use `decision-and-pilot.md`), re-injecting field data into an existing decision (use `evaluate-and-update.md`), or a quick literature list with no decision target.
|
|
22
|
+
|
|
23
|
+
The **domain** changes the frame vocabulary, not the protocol. `education` frames populations as learner/course; `policy` frames them as decision object/population/stakeholders; other domains register their own frame contract under `domains/`. Every stage below is domain-independent.
|
|
24
|
+
|
|
25
|
+
## Prerequisites
|
|
26
|
+
|
|
27
|
+
1. A research question plus the decision it feeds — if the decision target is missing, run the `frame` stage before anything else.
|
|
28
|
+
2. A run workspace (`eduevidence run --question "..."`) so every artifact has a durable home; do not carry state in chat.
|
|
29
|
+
3. A Complexity Gate level (S/M/L) from `scripts/complexity_gate.py`; it fixes the retrieval breadth, not the protocol.
|
|
30
|
+
|
|
31
|
+
## Runbook
|
|
32
|
+
|
|
33
|
+
| # | Stage | Role | Input | Artifact | Gate |
|
|
34
|
+
|---|---|---|---|---|---|
|
|
35
|
+
| 1 | Frame | research-planner | user question | `frame.json` | the selected domain's frame schema (`domains/<id>/manifest.json` → `frame_schema`); no intervention advice before framing completes |
|
|
36
|
+
| 2 | Retrieve | evidence-retriever | `frame.json` | `sources.jsonl` + `fetch/` | `schemas/source.schema.json`; Fetch/Validate gate inside Retrieve (RULE 2) |
|
|
37
|
+
| 3 | Extract | evidence-analyst | `sources.jsonl` + `fetch/` | `evidence.jsonl` | `schemas/evidence.schema.json`; outcome separation (task ≠ learning) |
|
|
38
|
+
| 4 | Challenge | skeptic | `evidence.jsonl` | `skeptic.json` | all 9 fixed checks executed; no fabricated counter-evidence |
|
|
39
|
+
| 5 | Audit | method-reviewer | `evidence.jsonl` + fetched text | `methodology.json` | `schemas/methodology.schema.json`; `task_vs_learning_guard` present |
|
|
40
|
+
| 6 | Adjudicate | evidence-judge | all of the above | `raw_verdict.json` → `final_verdict.json` | `schemas/verdict.schema.json`; Pre-Verdict Gate before finalising |
|
|
41
|
+
| 7 | Applicability | evidence-judge | `evidence.jsonl` + verdict | `applicability.json` | supported population, conditions, exclusions and uncertainty stated |
|
|
42
|
+
|
|
43
|
+
Execute the stages in order. A stage may be delegated to a sub-agent, a script, or the host model itself — the artifact and its gate are what count, never which adapter produced it.
|
|
44
|
+
|
|
45
|
+
### Stage 1 — Frame
|
|
46
|
+
|
|
47
|
+
Produce the PICO-style frame with decision target, scope, and inclusion/exclusion criteria, using the vocabulary of the selected domain (education: learner/course; policy: decision object/population/stakeholders). Ask the user only for inputs that cannot be inferred (target population, intervention variant, comparison, primary outcome); never invent defaults for them. Emit the suggested complexity level alongside the frame.
|
|
48
|
+
|
|
49
|
+
### Stage 2 — Retrieve
|
|
50
|
+
|
|
51
|
+
Write an auditable search plan (`SearchPlan`) with core, expansion, and **independent counter-evidence** queries, then execute it through `scripts/search_provenance.py` so attempts, screening decisions and exclusions are exported. `Fetch`/`Validate` are mandatory gates inside this stage: a snippet or abstract is a discovery aid; only fetched, validated content can be extracted from. Zero-config channels (OpenAlex / Semantic Scholar / CrossRef / AIHot / AgentSearch) work without keys; Sciverse (`SCIVERSE_API_TOKEN`) adds citation-grade retrieval with `doc_id`/`offset` provenance, and its chunk hits must be expanded through `/content` before use.
|
|
52
|
+
|
|
53
|
+
### Stage 3 — Extract
|
|
54
|
+
|
|
55
|
+
Extract claim-level Evidence Objects with effect sizes, CIs, sample sizes, outcome type and source location. Keep `relation_to_claim` on the link, never on the study. Do not merge task-performance and learning outcomes into one record.
|
|
56
|
+
|
|
57
|
+
### Stage 4 — Challenge
|
|
58
|
+
|
|
59
|
+
Run the nine fixed checks: null results, negative results, contradictory findings, alternative explanations, measurement mismatch, sampling bias, novelty effect, AI dependency, scope overreach. If nothing is found, state `NO CONTRADICTORY EVIDENCE FOUND` — absence of counter-evidence is a finding, not a gap to fill with invented sources.
|
|
60
|
+
|
|
61
|
+
### Stage 5 — Audit
|
|
62
|
+
|
|
63
|
+
Appraise each study against the methodology checklist and WWC 5.0 / GRADE-informed criteria. The auditor judges whether the evidence stands, never what it says. Record the `task_vs_learning_guard` verdict explicitly.
|
|
64
|
+
|
|
65
|
+
### Stage 6 — Adjudicate
|
|
66
|
+
|
|
67
|
+
Integrate frame, evidence matrix, skeptic findings and methodology audit into one of four states — `ADOPT` / `PILOT` / `REJECT` / `INSUFFICIENT EVIDENCE` — with `what_can_be_claimed` / `what_cannot_be_claimed` and a confidence breakdown. Run `scripts/pre_verdict_gate.py` **before** the verdict is treated as final; a critical failure caps confidence and forbids a high-confidence adoption claim.
|
|
68
|
+
|
|
69
|
+
### Stage 7 — Applicability
|
|
70
|
+
|
|
71
|
+
State who the evidence applies to, in which contexts, for which outcomes, under which conditions, and where it stops. A positive average effect never transfers automatically.
|
|
72
|
+
|
|
73
|
+
## Failure handling and fallbacks
|
|
74
|
+
|
|
75
|
+
| Failure | Handling |
|
|
76
|
+
|---|---|
|
|
77
|
+
| `SEARCH_NO_RESULT` | Broaden terms, switch discovery provider, or record a negative-search record — never lower the evidence bar silently. |
|
|
78
|
+
| `FETCH_FAILED` | Run the provider degradation chain; if it is exhausted, discard the source and return to Retrieve. |
|
|
79
|
+
| `FETCH_PARTIAL` | Require rule or human confirmation before extraction; never reconstruct missing text from memory. |
|
|
80
|
+
| `SCHEMA_INVALID` | Block the stage, repair the artifact, re-run the gate; do not advance. |
|
|
81
|
+
| `METHODOLOGY_TOO_WEAK` | Keep the study in the matrix but strip its support role. |
|
|
82
|
+
| `CONFLICT_UNRESOLVED` | Stay uncertain; do not force adjudication. |
|
|
83
|
+
| `INSUFFICIENT_EVIDENCE` | Emit `INSUFFICIENT EVIDENCE` with the specific evidence that would change the decision. |
|
|
84
|
+
|
|
85
|
+
## Human hand-off points
|
|
86
|
+
|
|
87
|
+
- After Frame, when population / intervention / comparison / outcome inputs are missing (`NEEDS_USER_CONTEXT`).
|
|
88
|
+
- After Challenge, when counter-evidence is absent, thin, or directly contradicts the working conclusion.
|
|
89
|
+
- After Adjudicate, before the decision is used for a real change in practice, curriculum, policy or operations.
|
|
90
|
+
|
|
91
|
+
## Resuming and updating
|
|
92
|
+
|
|
93
|
+
Resume with `eduevidence resume --run-id <id>`: the workspace continues from the first stage whose artifact is missing or schema-invalid. Evidence is append-only — a correction adds a new revision and a new decision snapshot; never overwrite an earlier one. To refresh the same question later, run the `evaluate-and-update` workflow so the change lands as a decision diff rather than a silently rewritten report.
|
|
94
|
+
|
|
95
|
+
## Minimal example
|
|
96
|
+
|
|
97
|
+
```bash
|
|
98
|
+
eduevidence run --question "Should first-year CS students use generative AI coding assistants?" --run-id ai-cs1
|
|
99
|
+
python scripts/search_provenance.py "first-year CS generative AI coding assistant learning outcomes" \
|
|
100
|
+
--out runs/ai-cs1/provenance --concept "AI coding assistant"
|
|
101
|
+
eduevidence status --run-id ai-cs1
|
|
102
|
+
eduevidence gate --run-id ai-cs1
|
|
103
|
+
eduevidence report --run-id ai-cs1 --theme claude
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
The shipped reference result for this exact question is `examples/ai-coding-assistant-evidence/` — read its `frame.json` → `sources.jsonl` → `evidence.jsonl` → `skeptic.json` → `methodology.json` → `verdict.json` chain to see what a completed review looks like.
|
|
107
|
+
|
|
108
|
+
## Acceptance checklist
|
|
109
|
+
|
|
110
|
+
- [ ] Frame passes its schema and states scope plus inclusion/exclusion criteria.
|
|
111
|
+
- [ ] Search plan shows core, expansion and independent counter-evidence queries with attempts and exclusions exported.
|
|
112
|
+
- [ ] Every extracted finding traces to fetched, validated content — no snippet-only evidence.
|
|
113
|
+
- [ ] All nine skeptic checks ran; the contradiction statement is explicit either way.
|
|
114
|
+
- [ ] Methodology audit separates task performance from learning and records the guard verdict.
|
|
115
|
+
- [ ] Verdict is one of the four states, carries its confidence breakdown, and passed the Pre-Verdict Gate.
|
|
116
|
+
- [ ] Applicability states population, conditions, outcome limits and uncertainty.
|
|
117
|
+
- [ ] Reports and exports are projections of the stored artifacts, not a new source of truth.
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: experimental-jev
|
|
3
|
+
description: Optional Jev / SemDecide Tier-0 acceleration overlay. Modes 0 standard / 1 jev-mcp / 2 semdecide / 3 hybrid. Fail-closed. Never replaces Pre-Verdict Gate or Skeptic.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Experimental Jev Overlay
|
|
7
|
+
|
|
8
|
+
Use only when the run explicitly enables the overlay: `eduevidence run --enhancement jev`
|
|
9
|
+
(and/or `--enhancement semdecide`, which is recorded in the run's `intake.json`), or
|
|
10
|
+
`EDU_EXPERIMENTAL_JEV=<mode>` for the detect CLIs (`--experimental <0|1|2|3>` is a flag of
|
|
11
|
+
`integrations/jev_mcp.py` and `integrations/semantic_decide.py`, not of the host CLI).
|
|
12
|
+
This is an **acceleration overlay** on the scientific protocol — not a fourth workflow,
|
|
13
|
+
not a substitute for scientific gates, and not a source of evidence.
|
|
14
|
+
|
|
15
|
+
Default without an explicit selection: **mode 0 standard** (existing large-model pipeline only).
|
|
16
|
+
|
|
17
|
+
## Modes
|
|
18
|
+
|
|
19
|
+
| Mode | Name | Backing | When |
|
|
20
|
+
|---:|---|---|---|
|
|
21
|
+
| 0 | standard | host large model only | default; no experimental flag |
|
|
22
|
+
| 1 | jev-mcp | `integrations/jev/` → FreeJev or Vercel AI Gateway `typesafe-ai/jev` (urllib) or documented MCP (`https://freejev.org/mcp`, `npx -y @jkudish/jev-mcp`) | Tier-0 typed judgments, fast & cheap |
|
|
23
|
+
| 2 | semdecide | `integrations/semantic_decide.py` → `semdecide is / filter / choose` | Unix-pipeline predicates, filtering, routing |
|
|
24
|
+
| 3 | hybrid | 1 + 2 together | screen/rerank/extract/classify/verify via Jev; local predicates via SemDecide |
|
|
25
|
+
|
|
26
|
+
Enable:
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
# resolve mode + detect backends (no network judgment yet)
|
|
30
|
+
python3 integrations/jev_mcp.py --experimental 3
|
|
31
|
+
python3 integrations/semantic_decide.py --experimental 3
|
|
32
|
+
|
|
33
|
+
# user confirmation gate (writes ~/.eduevidence/jev_mcp_approval.json)
|
|
34
|
+
python3 integrations/jev_mcp.py --experimental 3 --approve
|
|
35
|
+
|
|
36
|
+
# host CLI (documented entry; the enhancement selection turns the overlay on for the run)
|
|
37
|
+
eduevidence run --question "..." --run-id <id> --enhancement jev --enhancement semdecide
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
Secrets live only in `~/.eduevidence/env` — never in the repository, never on argv.
|
|
41
|
+
|
|
42
|
+
| Provider | Env | Decide endpoint | MCP (documented only) |
|
|
43
|
+
|---|---|---|---|
|
|
44
|
+
| `freejev` | `FREEJEV_API_KEY`, `JEV_PROVIDER=freejev` | `https://freejev.org/api/v1/decide` | `https://freejev.org/mcp` |
|
|
45
|
+
| `vercel` (default) | `AI_GATEWAY_API_KEY` or `TYPESAFE_API_KEY`, `JEV_PROVIDER=vercel` | Vercel AI Gateway System One | `npx -y @jkudish/jev-mcp` |
|
|
46
|
+
|
|
47
|
+
**FreeJev `request_id` is an idempotency key — never retry the same `request_id`.**
|
|
48
|
+
On timeout / 5xx / unknown outcome, fail-closed escalate this call to the large
|
|
49
|
+
model. A deliberate re-ask must use a **new** `request_id` and its own provenance row.
|
|
50
|
+
|
|
51
|
+
## Non-negotiable boundaries
|
|
52
|
+
|
|
53
|
+
1. **Fail-closed.** Any Jev `JEV_INVALID_RESPONSE` / `JEV_PROVIDER_ERROR`, and any
|
|
54
|
+
SemDecide exit **2 / 3 / 4** (plus `SEMDECIDE_UNAVAILABLE` / `SEMDECIDE_TIMEOUT`),
|
|
55
|
+
escalates to the large model (or a human). Never invent, never silently
|
|
56
|
+
accept, never force a binary on an uncertain probability.
|
|
57
|
+
2. **FreeJev `request_id` 不重试.** Same `request_id` is never resent; escalate instead.
|
|
58
|
+
2. **Does not replace `scripts/pre_verdict_gate.py`.** Stage Adjudicate still runs the
|
|
59
|
+
Pre-Verdict Gate before any verdict is final.
|
|
60
|
+
3. **Does not replace Skeptic.** Stage Challenge still executes all nine fixed checks
|
|
61
|
+
(`skill/sub-skills/contradiction-analysis`). `claim_verify` may pre-triage claims only.
|
|
62
|
+
4. **Does not replace methodology audit / tribunal / outcome separation.**
|
|
63
|
+
5. **Snippet ≠ evidence (RULE 2).** `content_screen` / `semantic_rerank` operate on
|
|
64
|
+
discovery material only. Only fetched, validated content enters Extract.
|
|
65
|
+
6. **Verbatim extract only.** `field_extract` returns regex-matched substrings chosen by
|
|
66
|
+
Jev — never model-authored field values. `review` / `not_found` are not extraction.
|
|
67
|
+
7. **Approval gate.** Network Tier-0 calls require
|
|
68
|
+
`~/.eduevidence/jev_mcp_approval.json` (tools hash + provider + model). Missing or
|
|
69
|
+
mismatched approval → `JEV_APPROVAL_REQUIRED`, escalate.
|
|
70
|
+
|
|
71
|
+
## Nine-stage mapping (`engine/workflows.py::SCIENTIFIC_STAGE_IDS`)
|
|
72
|
+
|
|
73
|
+
| # | Stage | Mode 1 / 3 Jev Tier-0 | Mode 2 / 3 SemDecide | Still mandatory (never skipped) |
|
|
74
|
+
|---:|---|---|---|---|
|
|
75
|
+
| 1 | Frame | — | `choose` optional route for domain/frame vocabulary (uncertain → planner) | Frame schema; no intervention advice before framing |
|
|
76
|
+
| 2 | Retrieve | `content_screen` (pre-context injection/substance/relevance); `semantic_rerank` (discovery order); `classify_check` (inclusion pre-triage) | `filter` on candidate JSONL; `is` for cheap exclusion predicates | Search plan + counter-evidence queries; Fetch/Validate gate; provenance export |
|
|
77
|
+
| 3 | Extract | `field_extract` (verbatim n / effect / CI candidates); `classify_check` (outcome type pre-label) | `is` for task-vs-learning guard pre-check | Evidence Objects schema; outcome separation; finding ≠ study merge ban |
|
|
78
|
+
| 4 | Challenge | `claim_verify` **assist only** (claim↔evidence relation triage) | `is` for "is there a contradiction signal?" assist | **All nine Skeptic checks**; `NO CONTRADICTORY EVIDENCE FOUND` rule |
|
|
79
|
+
| 5 | Audit | `claim_verify` assist on claim support | `is` for checklist pre-filter | Methodology audit + `task_vs_learning_guard` |
|
|
80
|
+
| 6 | Adjudicate | — (Tier-0 must not score final verdict) | — | Four-state verdict; confidence breakdown; **`scripts/pre_verdict_gate.py`** |
|
|
81
|
+
| 7 | Applicability | `classify_check` optional population/context tags | `choose` optional applicability bounds | Population / conditions / exclusions / uncertainty stated |
|
|
82
|
+
| 8 | Intervene | — | — | Intervention design gates (`intervention_design`) |
|
|
83
|
+
| 9 | Evaluate | — | — | Evaluation design + data validation (`evaluation_design`, `data_validation`) |
|
|
84
|
+
|
|
85
|
+
Projection (report render) stays outside the nine stages and outside this overlay.
|
|
86
|
+
|
|
87
|
+
## Capability ↔ tool
|
|
88
|
+
|
|
89
|
+
Registered in `engine/capabilities.py` (experimental set; resolve with
|
|
90
|
+
`capability_registry(include_experimental=True)` or `experimental_capability_registry()` —
|
|
91
|
+
omitted from the default scientific registry because Tier-0 is not role-owned work):
|
|
92
|
+
|
|
93
|
+
| capability_id | Tool | Mode |
|
|
94
|
+
|---|---|---|
|
|
95
|
+
| `content_screen` | `jev_mcp.screen` | 1, 3 |
|
|
96
|
+
| `semantic_rerank` | `jev_mcp.rerank` | 1, 3 |
|
|
97
|
+
| `field_extract` | `jev_mcp.extract` | 1, 3 |
|
|
98
|
+
| `classify_check` | `jev_mcp.classify` | 1, 3 |
|
|
99
|
+
| `claim_verify` | `jev_mcp.verify` | 1, 3 |
|
|
100
|
+
|
|
101
|
+
SemDecide wrappers (`integrations/semantic_decide.py`) support stages without new
|
|
102
|
+
capability IDs: `is` / `filter` / `choose`.
|
|
103
|
+
|
|
104
|
+
## Fail-closed matrix
|
|
105
|
+
|
|
106
|
+
Aligned with `integrations/semantic_decide.py` (`EXIT_MEANINGS`, `ESCALATE_EXIT_CODES`,
|
|
107
|
+
`requires_llm_escalation`, `escalate_plan`) and `integrations/jev/gateway.py` status codes.
|
|
108
|
+
|
|
109
|
+
| Signal | Meaning | Action |
|
|
110
|
+
|---|---|---|
|
|
111
|
+
| Jev `JEV_INVALID_RESPONSE` | typed answer / `answers` envelope missing or malformed | escalate to large model; mark `review` |
|
|
112
|
+
| Jev `JEV_PROVIDER_ERROR` | transport / HTTP 401, 403, 429, 529, 5xx | escalate this call only; record attempt |
|
|
113
|
+
| FreeJev `request_id` failure/timeout | idempotent decide attempt already keyed | **do not retry same `request_id`**; escalate / new id only |
|
|
114
|
+
| `JEV_APPROVAL_REQUIRED` | no/wrong approval file | do not call network; escalate / ask user to approve |
|
|
115
|
+
| `JEV_UNAVAILABLE` | no key and no MCP path | mode 0 standard pipeline |
|
|
116
|
+
| SemDecide exit `0` | `true/selected/match` | semantic result usable |
|
|
117
|
+
| SemDecide exit `1` | `false/no match` | usable; `filter` exit 1 is **not** an escalation by itself |
|
|
118
|
+
| SemDecide exit `2` | `invalid input` | repair input and re-run, or hand the whole decision to the large model |
|
|
119
|
+
| SemDecide exit `3` | `uncertain` | escalate to large model / human; never force true/false |
|
|
120
|
+
| SemDecide exit `4` | `provider failure` | degrade to large model for this call only; record the attempt |
|
|
121
|
+
| `SEMDECIDE_UNAVAILABLE` | binary missing / spawn failed | escalate (`requires_llm_escalation`) |
|
|
122
|
+
| `SEMDECIDE_TIMEOUT` | CLI timed out | escalate (`escalate=ESCALATE_TO_LLM`) |
|
|
123
|
+
| Screen `block` / `review` | injection or low trust | do not enter context; discard or human review |
|
|
124
|
+
| Extract `review` / `not_found` | provisional or absent value | not an extracted field; large model re-extract |
|
|
125
|
+
| Classify `review` | low confidence / margin | large model re-label |
|
|
126
|
+
| Verify `review` / `unknown` | low confidence / invalid | large model claim check; never auto-adopt |
|
|
127
|
+
|
|
128
|
+
## Degradation path
|
|
129
|
+
|
|
130
|
+
```
|
|
131
|
+
mode 3 hybrid
|
|
132
|
+
├─ provider key + approval → Tier-0 screen/rerank/extract/classify/verify
|
|
133
|
+
│ (FreeJev: FREEJEV_API_KEY + JEV_PROVIDER=freejev; request_id never retried)
|
|
134
|
+
├─ semdecide on PATH → is/filter/choose
|
|
135
|
+
├─ exit 2/3/4 / SEMDECIDE_TIMEOUT
|
|
136
|
+
│ / JEV_INVALID_RESPONSE / JEV_PROVIDER_ERROR
|
|
137
|
+
│ → large model for THAT call (fail-closed)
|
|
138
|
+
├─ Jev unavailable → SemDecide only (mode 2 slice)
|
|
139
|
+
├─ semdecide unavailable → Jev only (mode 1 slice)
|
|
140
|
+
└─ both unavailable / overlay not selected → mode 0 standard (Platform Native)
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
Scientific gates do not degrade: Pre-Verdict Gate, Skeptic nine checks, methodology
|
|
144
|
+
audit, and the four-state tribunal always run on the host path.
|
|
145
|
+
|
|
146
|
+
## Acceptance checklist
|
|
147
|
+
|
|
148
|
+
- [ ] The overlay selection is recorded on the run (`intake.json` enhancements); absent
|
|
149
|
+
selection means mode 0.
|
|
150
|
+
- [ ] Approval file present and `tools_hash` matches before any gateway call.
|
|
151
|
+
- [ ] Every Tier-0 result stored with `status` / `confidence` / `escalate` provenance.
|
|
152
|
+
- [ ] All SemDecide exit 2/3/4, `SEMDECIDE_UNAVAILABLE`, and `SEMDECIDE_TIMEOUT` rows
|
|
153
|
+
show an escalation record (no silent coercion). `filter` exit 1 alone is not escalated.
|
|
154
|
+
- [ ] FreeJev calls never reuse a `request_id` for retry; each attempt is one shot.
|
|
155
|
+
- [ ] Stage 4 Challenge still has the nine Skeptic checks + explicit contradiction statement.
|
|
156
|
+
- [ ] Stage 6 Adjudicate still ran `scripts/pre_verdict_gate.py` before finalising.
|
|
157
|
+
- [ ] A/B metrics logged per `docs/j-ev-experimental.md` (speedup口径 + quality gates).
|
|
158
|
+
- [ ] No secrets in artifacts (`AI_GATEWAY_API_KEY` / `TYPESAFE_API_KEY` / `FREEJEV_API_KEY`);
|
|
159
|
+
keys only from `~/.eduevidence/env` / process env.
|
|
160
|
+
|
|
161
|
+
## Minimal example
|
|
162
|
+
|
|
163
|
+
```bash
|
|
164
|
+
export EDU_EXPERIMENTAL_JEV=3 # detect CLIs read this; the host run uses --enhancement
|
|
165
|
+
python3 integrations/jev_mcp.py --experimental 3 --approve
|
|
166
|
+
eduevidence run --question "Should first-year CS students use generative AI coding assistants?" \
|
|
167
|
+
--run-id ai-cs1-jev --enhancement jev --enhancement semdecide
|
|
168
|
+
# Scientific path unchanged: Frame → Retrieve → Extract → Challenge → Audit → Adjudicate → Applicability
|
|
169
|
+
# Tier-0 only pre-screens, re-ranks, extracts verbatim fields, pre-labels, pre-verifies claims.
|
|
170
|
+
```
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: intake
|
|
3
|
+
description: One-shot two-round user intake before any research workflow. User answers only mode and questions; the agent infers the rest and then runs unattended.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Intake — 一次性两轮引导
|
|
7
|
+
|
|
8
|
+
Use this **before** selecting `evidence-review` / `decision-and-pilot` / `evaluate-and-update`.
|
|
9
|
+
The user answers **only** mode and questions. Everything else is agent judgment.
|
|
10
|
+
After the wrap-up summary is confirmed once, run **unattended**.
|
|
11
|
+
|
|
12
|
+
Programmatic twin: `scripts/intake/` package (`session.py` + `prefs.py` + `hooks.py`) +
|
|
13
|
+
`schemas/v2/intake.schema.json`.
|
|
14
|
+
Non-question preferences persist to `~/.eduevidence/prefs.json`
|
|
15
|
+
(`enhancement`, `mcp`/`jev` mappings, `depth_preference`, `open_browser`,
|
|
16
|
+
`default_main_theme`). **Never record the research question in prefs.**
|
|
17
|
+
All five report themes stay rendered; `default_main_theme` only decides which one to open.
|
|
18
|
+
|
|
19
|
+
---
|
|
20
|
+
|
|
21
|
+
## 第一轮固定语句(逐字使用)
|
|
22
|
+
|
|
23
|
+
```text
|
|
24
|
+
1) 研究问题(必填)
|
|
25
|
+
2) 执行增强(可多选,各附一句话简介):
|
|
26
|
+
[agent_mcp] Agent MCP — 多 CLI/多模型分工、独立子上下文、超时恢复(可选执行增强)
|
|
27
|
+
[jev] Jev — 轻量工具子集派发,适合把确定性小步交给受限工具面
|
|
28
|
+
[semdecide] SemDecide — 语义决策辅助,只结构化上游意图,不改下游确定性路由
|
|
29
|
+
[none] 均不启用 — 平台原生模式,零外部执行增强
|
|
30
|
+
3) 深度:
|
|
31
|
+
[S] S 快检 — 单点问题,最小检索面,直接给边界结论
|
|
32
|
+
[M] M 标准 — 常规采用/比较问题,标准证据到决策流程
|
|
33
|
+
[L] L 深研 — 高影响、有争议或要试点,完整深研与决策扩展
|
|
34
|
+
[auto] 自动 — 不选则按指南判定
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
### 深度自动判定指南(仅当用户选 auto / 不选)
|
|
38
|
+
|
|
39
|
+
| 信号 | 判定 |
|
|
40
|
+
|---|---|
|
|
41
|
+
| 单点问题(查定义、查公式、单条事实) | **S** |
|
|
42
|
+
| 常规采用 / 是否有效 / 比较对照 | **M** |
|
|
43
|
+
| 高影响、有争议、要试点 / 全面部署 / 政策级 | **L** |
|
|
44
|
+
| 不确定 | **M** |
|
|
45
|
+
|
|
46
|
+
用户显式选 S / M / L 时**不再**自动判定。CLI 同义词:`quick=S`,`standard=M`,`deep=L`。
|
|
47
|
+
|
|
48
|
+
---
|
|
49
|
+
|
|
50
|
+
## 第二轮(仅在需要时追问)
|
|
51
|
+
|
|
52
|
+
### A. 增强项(仅当第一轮勾选了对应增强)
|
|
53
|
+
|
|
54
|
+
| 增强开启时 | 第二轮只问这一项 | 落盘位置 |
|
|
55
|
+
|---|---|---|
|
|
56
|
+
| **Agent MCP** | 角色 → CLI / 模型 **授权表**(只展示本机扫描到的真实 CLI/模型;确认后固化 `~/.eduevidence/agent_mcp_approval.json`,含 `role_mapping_hash`) | prefs.`mcp.role_mapping` + 批准文件 |
|
|
57
|
+
| **Jev** | 工具子集(如 `search` / `fetch` / `validate`;只派发确定性小步,禁止旁路 spawn) | prefs.`jev.tool_subset` |
|
|
58
|
+
| **SemDecide** | 用途说明(只结构化上游 ResearchIntent / Frame,供 `recommend_mode`;**不改**下游确定性路由) | prefs / intake.`semdecide.purpose` |
|
|
59
|
+
|
|
60
|
+
未勾选的增强**不得**追问。勾选「均不启用」则整段跳过。
|
|
61
|
+
|
|
62
|
+
### B. 背景智能深挖(始终执行)
|
|
63
|
+
|
|
64
|
+
按题目推断领域 Frame 骨架后,针对该题追问 **3–6** 个影响边界的问题:
|
|
65
|
+
|
|
66
|
+
1. **人群**(目标学习者 / 决策对象)
|
|
67
|
+
2. **干预**(具体方法 / 工具及用法)
|
|
68
|
+
3. **对照**(与什么比较;可用 business as usual)
|
|
69
|
+
4. **主结果**(一个主要结果指标)
|
|
70
|
+
5. **情境**(学制 / 班型 / 线上线下 / 支持条件)
|
|
71
|
+
|
|
72
|
+
规则:
|
|
73
|
+
|
|
74
|
+
- **智能默认 + 一次确认**:能从题目直接推出的字段给出默认值,用户回车即采纳;推出不出的字段留空追问,**禁止编造**。
|
|
75
|
+
- 全部边界答完后,打印一次摘要,**一次确认**。
|
|
76
|
+
- 摘要确认后进入无人值守,不再逐步打断。
|
|
77
|
+
|
|
78
|
+
### C. 收尾摘要(一次确认)
|
|
79
|
+
|
|
80
|
+
```text
|
|
81
|
+
—— 收尾摘要(一次确认后无人值守)——
|
|
82
|
+
问题:…
|
|
83
|
+
增强:…
|
|
84
|
+
深度:…(判定理由)
|
|
85
|
+
边界:population / intervention / comparison / primary_outcome / context
|
|
86
|
+
结束后打开报告:是|否 | 主主题:claude|academic|datalab|datalab-dark|presentation
|
|
87
|
+
确认并开始无人值守执行?[Y/n]
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
用户确认后:写入 `intake.json`(过 `schemas/v2/intake.schema.json`),更新 `prefs.json`(不含研究问题),调用 `eduevidence run` / 工作流,**不再提问**。
|
|
91
|
+
|
|
92
|
+
---
|
|
93
|
+
|
|
94
|
+
## 无人值守与浏览器
|
|
95
|
+
|
|
96
|
+
- `python3 -m intake.cli --print-prompts`(或 `python3 scripts/intake/__main__.py --print-prompts`)
|
|
97
|
+
- `eduevidence run --yes` 或非 TTY:跳过全部提问,只用 prefs + CLI 标志,**不**弹浏览器。
|
|
98
|
+
- 结束时若 `prefs.open_browser` 为 true 且终端可交互:用 `webbrowser.open` 打开**主报告**。
|
|
99
|
+
- 五主题全部渲染并保留在 `reports-5themes/`;主报告 = `default_main_theme` 对应的那份,只决定**打开哪份**,不删除其余主题。
|
|
100
|
+
- `python3 scripts/dashboard_server.py --open`(或 `eduevidence dashboard --open`)启动后打开 Studio URL `/studio/`。
|
|
101
|
+
|
|
102
|
+
## 与 Canonical Protocol 的关系
|
|
103
|
+
|
|
104
|
+
Intake 不是协议阶段,只是启动前的**一次性**用户契约收集。收集完成后的流程仍走:
|
|
105
|
+
|
|
106
|
+
```text
|
|
107
|
+
Evidence Review → Decision & Pilot → Evaluate & Update
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Frame 阶段(`skill/task-briefs/frame.md`)继续负责完整 Research Frame 与 `NEEDS_USER_CONTEXT`;
|
|
111
|
+
本文件的边界追问只是提前把用户才知道的边界收齐,避免中途反复打断。
|
|
112
|
+
|
|
113
|
+
## 失败处理
|
|
114
|
+
|
|
115
|
+
| 失败 | 处理 |
|
|
116
|
+
|---|---|
|
|
117
|
+
| 研究问题为空 | 第一轮重问;不得带空问题进入工作流。 |
|
|
118
|
+
| 用户不确认收尾摘要 | 回到对应轮次修改,不得静默开跑。 |
|
|
119
|
+
| 增强不可用(如 Agent MCP 未安装) | 记录 `AGENT_MCP_UNAVAILABLE` 并降级 Platform Native;不阻断科学正确性。 |
|
|
120
|
+
| 边界字段用户拒绝提供 | 显式标注 `unknown + 如何获取`(FR-03),禁止用默认值冒充事实。 |
|
|
@@ -32,7 +32,7 @@ from pathlib import Path
|
|
|
32
32
|
from typing import Any
|
|
33
33
|
|
|
34
34
|
from adapter_contract import load_result, write_adapter_output
|
|
35
|
-
from zh_labels import
|
|
35
|
+
from zh_labels import label
|
|
36
36
|
from build_charts import effect_outcomes
|
|
37
37
|
|
|
38
38
|
OKABE_ITO = ["#E69F00", "#56B4E9", "#009E73", "#F0E442", "#0072B2", "#D55E00", "#CC79A7", "#000000"]
|
|
@@ -82,7 +82,26 @@ def _linear_ticks(maxv: int) -> list[int]:
|
|
|
82
82
|
return ticks
|
|
83
83
|
|
|
84
84
|
|
|
85
|
+
def _caption_width(caption: str, font_size: int = 11) -> int:
|
|
86
|
+
"""Conservative width of an italic caption line at the SVG font stack.
|
|
87
|
+
|
|
88
|
+
The caption is a single unwrapped line, so a long title used to run past
|
|
89
|
+
the fixed 720px canvas and get clipped. CJK glyphs are a full em wide;
|
|
90
|
+
latin is bounded by the same cap-height factor used elsewhere in this
|
|
91
|
+
module.
|
|
92
|
+
"""
|
|
93
|
+
cjk = sum(1 for ch in caption if "\u4e00" <= ch <= "\u9fff")
|
|
94
|
+
latin = len(caption) - cjk
|
|
95
|
+
# 0.75em is the measured upper bound for the latin part of the stack
|
|
96
|
+
# (Helvetica/Arial italic); ~2 full-width latin words at the start of the
|
|
97
|
+
# caption also shift the whole line right by the x=20 origin.
|
|
98
|
+
return int(cjk * font_size + latin * font_size * 0.78) + 60
|
|
99
|
+
|
|
100
|
+
|
|
85
101
|
def _figure_svg(title: str, caption: str, body: str, w: int = 720, h: int = 300) -> str:
|
|
102
|
+
# Grow (never shrink) the canvas so the unwrapped caption fits instead of
|
|
103
|
+
# being cut off at the right edge; the body keeps its designed layout.
|
|
104
|
+
w = max(w, _caption_width(caption))
|
|
86
105
|
return (f'<svg viewBox="0 0 {w} {h}" xmlns="http://www.w3.org/2000/svg" role="img" '
|
|
87
106
|
f'aria-label="{_esc(caption)}">'
|
|
88
107
|
f'<rect width="{w}" height="{h}" fill="#FFFFFF"/>'
|
|
@@ -337,8 +356,11 @@ def render_figures(figure_data: dict, theme: str = "okabe_ito", lang: str = "zh"
|
|
|
337
356
|
# 绝不把 relation_to_claim 的 support/contradict 当作 outcome 好坏;计数轴整数刻度。)
|
|
338
357
|
outcomes = figure_data.get("outcomes", [])
|
|
339
358
|
if outcomes:
|
|
340
|
-
|
|
341
|
-
|
|
359
|
+
# Both locales render a curated label. The English axis used to print
|
|
360
|
+
# the storage token ("knowledge_gain") because only zh went through
|
|
361
|
+
# OUTCOME_ZH; the English report then showed raw enum values as its
|
|
362
|
+
# headline comparison chart's axis.
|
|
363
|
+
names = [label(lang, "outcome", o["outcome_type"]) for o in outcomes]
|
|
342
364
|
series = [{"name": s["name"],
|
|
343
365
|
"data": [o.get(s["data"], 0) for o in outcomes]}
|
|
344
366
|
for s in DIR_SERIES[lang]]
|