eduevidence 6.2.0 → 6.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +395 -0
- package/README.md +22 -13
- package/README.zh-CN.md +15 -8
- package/SKILL.md +10 -9
- package/benchmarks/evidence-library.json +277 -1
- package/docs/architecture.md +6 -3
- package/docs/j-ev-experimental.md +250 -0
- package/docs/reproducibility.md +138 -0
- package/domains/_neutral/copy/few_shots.json +21 -0
- package/domains/_neutral/copy/framing_lexicon.json +19 -0
- package/domains/_neutral/copy/module_labels.json +5 -0
- package/domains/_neutral/copy/module_labels_footer.json +102 -0
- package/domains/_neutral/copy/module_labels_modules.json +204 -0
- package/domains/_neutral/copy/module_labels_nav.json +126 -0
- package/domains/_neutral/copy/module_labels_summary.json +98 -0
- package/domains/_neutral/copy/module_labels_tables.json +164 -0
- package/domains/_neutral/copy/module_labels_v2.json +90 -0
- package/domains/_neutral/copy/risk_constructs.json +20 -0
- package/domains/_neutral/copy/section_titles.json +66 -0
- package/domains/_neutral/copy/terminology.json +11 -0
- package/domains/check_copy_packs.py +103 -0
- package/domains/education/copy/few_shots.json +22 -0
- package/domains/education/copy/framing_enums.json +167 -0
- package/domains/education/copy/framing_lexicon.json +166 -0
- package/domains/education/copy/module_labels.json +169 -0
- package/domains/education/copy/risk_constructs.json +48 -0
- package/domains/education/copy/section_titles.json +186 -0
- package/domains/education/copy/terminology.json +70 -0
- package/domains/education/manifest.json +1 -1
- package/domains/education/outcome_taxonomy.json +2 -2
- package/domains/manifest.json +1 -1
- package/domains/policy/copy/few_shots.json +22 -0
- package/domains/policy/copy/framing_enums.json +94 -0
- package/domains/policy/copy/framing_lexicon.json +174 -0
- package/domains/policy/copy/module_labels.json +168 -0
- package/domains/policy/copy/risk_constructs.json +33 -0
- package/domains/policy/copy/section_titles.json +186 -0
- package/domains/policy/copy/terminology.json +64 -0
- package/engine/capabilities.py +57 -5
- package/engine/decision_policy.py +88 -17
- package/engine/library_builtin.py +7 -4
- package/engine/tribunal.py +17 -23
- package/engine/versions.py +1 -1
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +4 -4
- package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
- package/examples/spaced-retrieval-practice/report.html +2522 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +4 -4
- package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +36 -36
- package/integrations/jev/__init__.py +115 -0
- package/integrations/jev/approval.py +212 -0
- package/integrations/jev/cli.py +84 -0
- package/integrations/jev/config.py +112 -0
- package/integrations/jev/gateway.py +128 -0
- package/integrations/jev/modes.py +38 -0
- package/integrations/jev/tools_classify.py +88 -0
- package/integrations/jev/tools_extract.py +111 -0
- package/integrations/jev/tools_rerank.py +71 -0
- package/integrations/jev/tools_screen.py +87 -0
- package/integrations/jev/tools_verify.py +95 -0
- package/integrations/jev_mcp.py +22 -0
- package/integrations/semantic_decide.py +286 -0
- package/integrations/semdecide_cli.py +55 -0
- package/package.json +9 -1
- package/pyproject.toml +1 -1
- package/references/report-copy-style.md +43 -3
- package/schemas/v2/decision-snapshot.schema.json +20 -9
- package/schemas/v2/intake.schema.json +191 -0
- package/scripts/build_evidence_library.py +15 -5
- package/scripts/dashboard_server.py +13 -2
- package/scripts/intake/__init__.py +31 -0
- package/scripts/intake/__main__.py +18 -0
- package/scripts/intake/background.py +78 -0
- package/scripts/intake/browser.py +79 -0
- package/scripts/intake/cli.py +57 -0
- package/scripts/intake/constants.py +57 -0
- package/scripts/intake/depth.py +53 -0
- package/scripts/intake/enhancements.py +106 -0
- package/scripts/intake/hooks.py +90 -0
- package/scripts/intake/prefs.py +76 -0
- package/scripts/intake/prompts.py +85 -0
- package/scripts/intake/session.py +152 -0
- package/scripts/lint_file_layers.py +126 -0
- package/scripts/orchestrator.py +68 -17
- package/scripts/pre_verdict_gate.py +21 -7
- package/scripts/skill_lint.py +11 -1
- package/scripts/skill_payload.py +3 -3
- package/scripts/test_adversarial_empirical.py +70 -6
- package/skill/agents/evidence-judge.md +49 -7
- package/skill/workflows/experimental-jev.md +170 -0
- package/skill/workflows/intake.md +120 -0
- package/visualization/eduevidence-report/scripts/build_infographics.py +32 -14
- package/visualization/eduevidence-report/scripts/build_report.py +75 -662
- package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
- package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
- package/visualization/eduevidence-report/scripts/zh_labels.py +61 -0
- package/scripts/build_esl_artifacts.py +0 -1921
- package/scripts/build_killer_demo.py +0 -295
- package/scripts/enrich_projects_human_and_lieflat.py +0 -315
- package/scripts/generate_new_projects.py +0 -686
- package/scripts/sync_killer_demo_report.py +0 -270
package/SKILL.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
name: eduevidence
|
|
3
3
|
description: "Decision-grade evidence synthesis for education and applied social science intervention decisions. Use when a user needs to determine whether, when, for whom, or how to adopt, pilot, evaluate, or revise a teaching method, curriculum change, AI tool, program, or policy intervention. Run an auditable evidence-to-decision workflow spanning systematic retrieval, counter-evidence challenge, methodological quality and evidence-certainty appraisal, provenance-traceable evidence graphs, applicability boundaries, evidence-grounded gap detection, preregistration-ready study or pilot design, empirical evidence re-injection, and decision revision."
|
|
4
4
|
---
|
|
5
|
-
# EduEvidence 6.
|
|
5
|
+
# EduEvidence 6.3 — Decision-Grade Evidence Engine
|
|
6
6
|
> **AI4SS Track | Art–Science Integration · General Intelligence**
|
|
7
7
|
> **From empirical questions to decision-grade evidence and evidence-to-action loops.**
|
|
8
8
|
|
|
@@ -28,6 +28,7 @@ Treat roles, models, Agent MCP, retrievers, scripts, and HTML reports as **execu
|
|
|
28
28
|
---
|
|
29
29
|
|
|
30
30
|
## 2. Route the Request to a Workflow
|
|
31
|
+
Before selecting a primary workflow, run the one-shot two-round intake in `skill/workflows/intake.md` (research question → execution enhancements → depth; then enhancement mappings + Frame boundary confirmation). Persist non-question preferences via `scripts/intake/` to `~/.eduevidence/prefs.json`. After the wrap-up summary is confirmed once, continue unattended.
|
|
31
32
|
Load exactly one primary workflow for the current task:
|
|
32
33
|
- `skill/workflows/evidence-review.md` — use for evidence synthesis and decision appraisal from existing research.
|
|
33
34
|
- `skill/workflows/decision-and-pilot.md` — use when the evidence must be converted into an actionable pilot or intervention plan.
|
|
@@ -109,7 +110,7 @@ Applicability → Intervene → Evaluate
|
|
|
109
110
|
| 3 | **Extract** | `evidence.schema.json` | Extract findings, effect sizes, confidence intervals, sample sizes, outcomes, population characteristics, and study design information when available. |
|
|
110
111
|
| 4 | **Challenge** | `evidence.schema.json` | Search explicitly for null findings, negative findings, contradictory evidence, alternative explanations, and confounders. |
|
|
111
112
|
| 5 | **Audit** | `methodology.schema.json` | Appraise study quality and evidence certainty. Apply WWC 5.0 criteria where relevant to education-study design; use GRADE-informed certainty assessment at the body-of-evidence level where appropriate. |
|
|
112
|
-
| 6 | **Adjudicate** | `verdict.schema.json` | Integrate evidence and emit a bounded decision: `ADOPT`, `PILOT`, `
|
|
113
|
+
| 6 | **Adjudicate** | `verdict.schema.json` | Integrate evidence and emit a bounded decision: `ADOPT`, `PILOT`, `REJECT`, or `INSUFFICIENT_EVIDENCE`. Pass the Pre-Verdict Gate before finalizing. |
|
|
113
114
|
| 7 | **Applicability** | `references/applicability-policy.md` | State who the evidence applies to, in which contexts, for which outcomes, and under what conditions. |
|
|
114
115
|
| 8 | **Intervene** | `intervention.schema.json` | Design the minimum viable intervention or pilot and define explicit success, failure, and stop conditions. |
|
|
115
116
|
| 9 | **Evaluate** | `evaluation.schema.json` | Define baseline, post-intervention, retention/maintenance, transfer, and decision-update logic. |
|
|
@@ -138,7 +139,7 @@ Include, where relevant:
|
|
|
138
139
|
### Gate C — No Direct Learning Evidence, No ADOPT
|
|
139
140
|
For education interventions, do not issue `ADOPT` based only on task speed, task completion, productivity, usability, preference, or subjective experience.
|
|
140
141
|
Require direct evidence on learning, retention, independent transfer, or another explicitly decision-relevant outcome before `ADOPT` can be considered.
|
|
141
|
-
If direct evidence is missing, bound the decision to `PILOT`, `
|
|
142
|
+
If direct evidence is missing, bound the decision to `PILOT`, `INSUFFICIENT_EVIDENCE`, or `REJECT` as justified by the evidence.
|
|
142
143
|
### Gate D — No False Precision
|
|
143
144
|
Never invent missing uncertainty statistics.
|
|
144
145
|
- If a confidence interval is not reported, do not fabricate one.
|
|
@@ -303,13 +304,13 @@ The example demonstrates:
|
|
|
303
304
|
- Empirical evidence re-injection followed by decision revision.
|
|
304
305
|
Do not generalize the flagship verdict to unrelated populations, courses, tools, or policy contexts.
|
|
305
306
|
|
|
306
|
-
The four-state
|
|
307
|
+
The four-state matrix (`engine.decision_policy.decision_outcome`) is the gate-enforced landing; each public case also records the adjudicator's own `recommended_action`. Recomputed from the pack's own `evidence.jsonl` + confidence label:
|
|
307
308
|
|
|
308
|
-
- `examples/ai-coding-assistant-evidence/` - `
|
|
309
|
-
- `examples/spaced-retrieval-practice/` - `ADOPT` / High / 0.893. Retention and transfer, the primary outcomes, carry direct evidence at directness 2.
|
|
310
|
-
- `examples/workplace-ai-assistant/` - `
|
|
309
|
+
- `examples/ai-coding-assistant-evidence/` - matrix `INSUFFICIENT_EVIDENCE` (`unresolved_conflict`: the decisive relations contain both `support_adoption` and `oppose_adoption`, so the conflict check fires first) / Moderate / 0.586. The pack records `PILOT`; primary evidence stops at task performance (no directness-2 learning outcome), so the decision is bounded — but the matrix does not let a conflicted relation set claim `PILOT`.
|
|
310
|
+
- `examples/spaced-retrieval-practice/` - matrix `ADOPT` / High / 0.893. Retention and transfer, the primary outcomes, carry direct evidence at directness 2. The pack records `ADOPT`; this is the only public case where stated and matrix landings agree.
|
|
311
|
+
- `examples/workplace-ai-assistant/` - matrix `INSUFFICIENT_EVIDENCE` with `downgrade_reason=None` / Moderate / 0.578, using the policy domain contract: all four decisive relations are `conditional`, and `conditional` is not a conflict relation (`CONFLICT_RELATIONS` = `{conflict, mixed}`), so no downgrade reason is recorded and, with no decisive `support_adoption`, Moderate confidence cannot reach `PILOT`. The pack records `PILOT`.
|
|
311
312
|
|
|
312
|
-
A verdict never awards itself an action: the Pre-Verdict Gate re-derives primary-outcome directness from the evidence corpus and caps an unsupported `ADOPT` to `PILOT`. Never present a case as ADOPT without that derivation passing.
|
|
313
|
+
A verdict never awards itself an action: the Pre-Verdict Gate re-derives primary-outcome directness from the evidence corpus and caps an unsupported `ADOPT` to `PILOT`. Never present a case as ADOPT without that derivation passing. Where the matrix landing is `INSUFFICIENT_EVIDENCE`, do not describe the case as having landed at `PILOT` — report both the recorded action and the matrix bound, and do not invent evidence to force a stronger landing.
|
|
313
314
|
|
|
314
315
|
---
|
|
315
316
|
|
|
@@ -451,7 +452,7 @@ This appendix keeps the deterministic repository gates (skill lint / version / m
|
|
|
451
452
|
- **EduEvidence Research Engine** — the runnable engine delivered as a Skill package (`engine/`, `retrieval/`, `scripts/`, `schemas/`).
|
|
452
453
|
- **Shared Research Library** — verified external facts (`Source` / `Study` / `Finding` / `Audit`) may be reused across project snapshots; interpretive objects (`Claim` / `EvidenceLink` / `Applicability` / `Decision`) stay project-local.
|
|
453
454
|
- **No new study design without evidence grounding** — every study or pilot must cite an explicit, evidence-supported `KnowledgeGap` identifier.
|
|
454
|
-
- **Schema 版本口径**:schemas/ 顶层
|
|
455
|
+
- **Schema 版本口径**:schemas/ 顶层 15 个 = V1 契约(evidence.schema.json 当前修订 1.1、education-frame / verdict / report-spec 等);schemas/v2/ 18 个 = V2 契约(evidence-link / research-intent / intake / study / graph-revision / project 等);另有 v3 3 个、v4 4 个、vNext 10 个,递归合计 50 个(计数以 `scripts/generate_metrics.py` 为准)。文档与代理配置一律以此口径命名。
|
|
455
456
|
- Five baked report themes (presentation systems, not science):
|
|
456
457
|
|
|
457
458
|
├─ Claude Research [Light]
|
|
@@ -4941,6 +4941,282 @@
|
|
|
4941
4941
|
"assessment_edtech"
|
|
4942
4942
|
]
|
|
4943
4943
|
},
|
|
4944
|
+
{
|
|
4945
|
+
"entry_id": "lib-ai-coding-assistant-E-001",
|
|
4946
|
+
"source_id": "S-2023-kazemitabaar",
|
|
4947
|
+
"title": "Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming",
|
|
4948
|
+
"year": 2023,
|
|
4949
|
+
"outcome_token": "completion_time",
|
|
4950
|
+
"outcome_tokens": [
|
|
4951
|
+
"completion_time"
|
|
4952
|
+
],
|
|
4953
|
+
"direction": "support",
|
|
4954
|
+
"study_type": "rct",
|
|
4955
|
+
"claim_text": "AI coding assistants significantly increase task completion speed and completion rate during training.",
|
|
4956
|
+
"effect_summary": "1.15x completion rate, 0.57x time, 1.8x correctness",
|
|
4957
|
+
"confidence_markers": [
|
|
4958
|
+
"evidence_level:strong",
|
|
4959
|
+
"quality_score:9.0",
|
|
4960
|
+
"confidence:0.7",
|
|
4961
|
+
"decision_relation:support_adoption"
|
|
4962
|
+
],
|
|
4963
|
+
"domains": [
|
|
4964
|
+
"ai-coding-assistant"
|
|
4965
|
+
]
|
|
4966
|
+
},
|
|
4967
|
+
{
|
|
4968
|
+
"entry_id": "lib-ai-coding-assistant-E-002",
|
|
4969
|
+
"source_id": "S-2023-kazemitabaar",
|
|
4970
|
+
"title": "Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming",
|
|
4971
|
+
"year": 2023,
|
|
4972
|
+
"outcome_token": "independent_problem_solving",
|
|
4973
|
+
"outcome_tokens": [
|
|
4974
|
+
"independent_problem_solving"
|
|
4975
|
+
],
|
|
4976
|
+
"direction": "neutral",
|
|
4977
|
+
"study_type": "rct",
|
|
4978
|
+
"claim_text": "Access to AI code generation did not decrease performance on manual code-modification tasks.",
|
|
4979
|
+
"effect_summary": "no significant difference between groups",
|
|
4980
|
+
"confidence_markers": [
|
|
4981
|
+
"evidence_level:moderate",
|
|
4982
|
+
"quality_score:7.0",
|
|
4983
|
+
"confidence:0.5",
|
|
4984
|
+
"decision_relation:neutral"
|
|
4985
|
+
],
|
|
4986
|
+
"domains": [
|
|
4987
|
+
"ai-coding-assistant"
|
|
4988
|
+
]
|
|
4989
|
+
},
|
|
4990
|
+
{
|
|
4991
|
+
"entry_id": "lib-ai-coding-assistant-E-003",
|
|
4992
|
+
"source_id": "S-2023-kazemitabaar",
|
|
4993
|
+
"title": "Studying the effect of AI Code Generators on Supporting Novice Learners in Introductory Programming",
|
|
4994
|
+
"year": 2023,
|
|
4995
|
+
"outcome_token": "retention",
|
|
4996
|
+
"outcome_tokens": [
|
|
4997
|
+
"retention"
|
|
4998
|
+
],
|
|
4999
|
+
"direction": "neutral",
|
|
5000
|
+
"study_type": "rct",
|
|
5001
|
+
"claim_text": "One week after training, retention differences between Codex and baseline groups did not reach statistical significance.",
|
|
5002
|
+
"effect_summary": "slightly better for Codex group but not significant",
|
|
5003
|
+
"confidence_markers": [
|
|
5004
|
+
"evidence_level:strong",
|
|
5005
|
+
"quality_score:9.0",
|
|
5006
|
+
"confidence:0.5",
|
|
5007
|
+
"decision_relation:neutral"
|
|
5008
|
+
],
|
|
5009
|
+
"domains": [
|
|
5010
|
+
"ai-coding-assistant"
|
|
5011
|
+
]
|
|
5012
|
+
},
|
|
5013
|
+
{
|
|
5014
|
+
"entry_id": "lib-ai-coding-assistant-E-004",
|
|
5015
|
+
"source_id": "S-2025-bastani",
|
|
5016
|
+
"title": "Generative AI without guardrails can harm learning: Evidence from high school mathematics",
|
|
5017
|
+
"year": 2025,
|
|
5018
|
+
"outcome_token": "independent_problem_solving",
|
|
5019
|
+
"outcome_tokens": [
|
|
5020
|
+
"independent_problem_solving"
|
|
5021
|
+
],
|
|
5022
|
+
"direction": "contradict",
|
|
5023
|
+
"study_type": "rct",
|
|
5024
|
+
"claim_text": "Students with unguarded GPT-4 access performed 17% worse on the independent exam than the control group, despite higher practice performance.",
|
|
5025
|
+
"effect_summary": "negative_17_percent_on_independent_exam",
|
|
5026
|
+
"confidence_markers": [
|
|
5027
|
+
"evidence_level:strong",
|
|
5028
|
+
"quality_score:8.0",
|
|
5029
|
+
"confidence:0.75",
|
|
5030
|
+
"decision_relation:oppose_adoption"
|
|
5031
|
+
],
|
|
5032
|
+
"domains": [
|
|
5033
|
+
"ai-coding-assistant"
|
|
5034
|
+
]
|
|
5035
|
+
},
|
|
5036
|
+
{
|
|
5037
|
+
"entry_id": "lib-ai-coding-assistant-E-005",
|
|
5038
|
+
"source_id": "S-2025-bastani",
|
|
5039
|
+
"title": "Generative AI without guardrails can harm learning: Evidence from high school mathematics",
|
|
5040
|
+
"year": 2025,
|
|
5041
|
+
"outcome_token": "independent_problem_solving",
|
|
5042
|
+
"outcome_tokens": [
|
|
5043
|
+
"independent_problem_solving"
|
|
5044
|
+
],
|
|
5045
|
+
"direction": "support",
|
|
5046
|
+
"study_type": "rct",
|
|
5047
|
+
"claim_text": "Guardrail design of the AI tutor (hints instead of answers, teacher-informed prompts) largely eliminated the negative learning effect.",
|
|
5048
|
+
"effect_summary": "negative effect essentially eradicated, no positive effect observed",
|
|
5049
|
+
"confidence_markers": [
|
|
5050
|
+
"evidence_level:strong",
|
|
5051
|
+
"quality_score:8.0",
|
|
5052
|
+
"confidence:0.75",
|
|
5053
|
+
"decision_relation:conditional"
|
|
5054
|
+
],
|
|
5055
|
+
"domains": [
|
|
5056
|
+
"ai-coding-assistant"
|
|
5057
|
+
]
|
|
5058
|
+
},
|
|
5059
|
+
{
|
|
5060
|
+
"entry_id": "lib-ai-coding-assistant-E-006",
|
|
5061
|
+
"source_id": "S-2025-bastani",
|
|
5062
|
+
"title": "Generative AI without guardrails can harm learning: Evidence from high school mathematics",
|
|
5063
|
+
"year": 2025,
|
|
5064
|
+
"outcome_token": "assignment_score",
|
|
5065
|
+
"outcome_tokens": [
|
|
5066
|
+
"assignment_score"
|
|
5067
|
+
],
|
|
5068
|
+
"direction": "support",
|
|
5069
|
+
"study_type": "rct",
|
|
5070
|
+
"claim_text": "Access to GPT-4 during practice improves task performance (48% for GPT Base, 127% for GPT Tutor) — but this task performance does not transfer to independent exam performance.",
|
|
5071
|
+
"effect_summary": "48-127 percent improvement on practice problems",
|
|
5072
|
+
"confidence_markers": [
|
|
5073
|
+
"evidence_level:strong",
|
|
5074
|
+
"quality_score:8.0",
|
|
5075
|
+
"confidence:0.75",
|
|
5076
|
+
"decision_relation:conditional"
|
|
5077
|
+
],
|
|
5078
|
+
"domains": [
|
|
5079
|
+
"ai-coding-assistant"
|
|
5080
|
+
]
|
|
5081
|
+
},
|
|
5082
|
+
{
|
|
5083
|
+
"entry_id": "lib-ai-coding-assistant-E-007",
|
|
5084
|
+
"source_id": "S-2024-marzuki",
|
|
5085
|
+
"title": "Impact of ChatGPT on ESL students' academic writing skills",
|
|
5086
|
+
"year": 2024,
|
|
5087
|
+
"outcome_token": "knowledge_gain",
|
|
5088
|
+
"outcome_tokens": [
|
|
5089
|
+
"knowledge_gain"
|
|
5090
|
+
],
|
|
5091
|
+
"direction": "support",
|
|
5092
|
+
"study_type": "mixed_methods",
|
|
5093
|
+
"claim_text": "ChatGPT as a formative feedback tool produced a significant positive impact on students' academic writing skills with positive student perceptions.",
|
|
5094
|
+
"effect_summary": "significant positive impact on writing skills",
|
|
5095
|
+
"confidence_markers": [
|
|
5096
|
+
"evidence_level:moderate",
|
|
5097
|
+
"quality_score:6.0",
|
|
5098
|
+
"confidence:0.55",
|
|
5099
|
+
"decision_relation:support_adoption"
|
|
5100
|
+
],
|
|
5101
|
+
"domains": [
|
|
5102
|
+
"ai-coding-assistant"
|
|
5103
|
+
]
|
|
5104
|
+
},
|
|
5105
|
+
{
|
|
5106
|
+
"entry_id": "lib-ai-coding-assistant-E-008",
|
|
5107
|
+
"source_id": "S-2023-peng",
|
|
5108
|
+
"title": "The Impact of AI on Developer Productivity: Evidence from GitHub Copilot",
|
|
5109
|
+
"year": 2023,
|
|
5110
|
+
"outcome_token": "completion_time",
|
|
5111
|
+
"outcome_tokens": [
|
|
5112
|
+
"completion_time"
|
|
5113
|
+
],
|
|
5114
|
+
"direction": "support",
|
|
5115
|
+
"study_type": "rct",
|
|
5116
|
+
"claim_text": "Professional developers with Copilot access completed a standardized coding task about 55% faster than the control group (RCT, n=95).",
|
|
5117
|
+
"effect_summary": "~55.8% faster task completion in Copilot group",
|
|
5118
|
+
"confidence_markers": [
|
|
5119
|
+
"evidence_level:moderate",
|
|
5120
|
+
"quality_score:8.0",
|
|
5121
|
+
"confidence:0.6",
|
|
5122
|
+
"decision_relation:conditional"
|
|
5123
|
+
],
|
|
5124
|
+
"domains": [
|
|
5125
|
+
"ai-coding-assistant"
|
|
5126
|
+
]
|
|
5127
|
+
},
|
|
5128
|
+
{
|
|
5129
|
+
"entry_id": "lib-ai-coding-assistant-E-009",
|
|
5130
|
+
"source_id": "S-2023-yetistiren",
|
|
5131
|
+
"title": "GitHub Copilot AI Pair Programmer: Asset or Liability?",
|
|
5132
|
+
"year": 2023,
|
|
5133
|
+
"outcome_token": "code_quality",
|
|
5134
|
+
"outcome_tokens": [
|
|
5135
|
+
"code_quality"
|
|
5136
|
+
],
|
|
5137
|
+
"direction": "support",
|
|
5138
|
+
"study_type": "observational",
|
|
5139
|
+
"claim_text": "Systematic benchmark evaluation reports mixed quality results for Copilot-generated code relative to human code: correctness competitive on parts of the benchmark while security-relevant defects are documented.",
|
|
5140
|
+
"effect_summary": "mixed quality profile; no single-direction summary",
|
|
5141
|
+
"confidence_markers": [
|
|
5142
|
+
"evidence_level:moderate",
|
|
5143
|
+
"quality_score:7.0",
|
|
5144
|
+
"confidence:0.55",
|
|
5145
|
+
"decision_relation:conditional"
|
|
5146
|
+
],
|
|
5147
|
+
"domains": [
|
|
5148
|
+
"ai-coding-assistant"
|
|
5149
|
+
]
|
|
5150
|
+
},
|
|
5151
|
+
{
|
|
5152
|
+
"entry_id": "lib-ai-coding-assistant-E-010",
|
|
5153
|
+
"source_id": "S-2022-finnie-ansley",
|
|
5154
|
+
"title": "Using GitHub Copilot to Solve Introductory Programming Problems",
|
|
5155
|
+
"year": 2022,
|
|
5156
|
+
"outcome_token": "assignment_score",
|
|
5157
|
+
"outcome_tokens": [
|
|
5158
|
+
"assignment_score"
|
|
5159
|
+
],
|
|
5160
|
+
"direction": "support",
|
|
5161
|
+
"study_type": "observational",
|
|
5162
|
+
"claim_text": "Codex produced passing-level solutions for roughly half to three-quarters of CS1 exam-style questions depending on the dataset, indicating substantial task-capability headroom available to novices.",
|
|
5163
|
+
"effect_summary": "passing solutions on ~50-75% of questions across datasets",
|
|
5164
|
+
"confidence_markers": [
|
|
5165
|
+
"evidence_level:moderate",
|
|
5166
|
+
"quality_score:7.0",
|
|
5167
|
+
"confidence:0.55",
|
|
5168
|
+
"decision_relation:conditional"
|
|
5169
|
+
],
|
|
5170
|
+
"domains": [
|
|
5171
|
+
"ai-coding-assistant"
|
|
5172
|
+
]
|
|
5173
|
+
},
|
|
5174
|
+
{
|
|
5175
|
+
"entry_id": "lib-ai-coding-assistant-E-011",
|
|
5176
|
+
"source_id": "S-2023-explanations-compare",
|
|
5177
|
+
"title": "Comparing Code Explanations Created by Students and Large Language Models",
|
|
5178
|
+
"year": 2023,
|
|
5179
|
+
"outcome_token": "metacognition",
|
|
5180
|
+
"outcome_tokens": [
|
|
5181
|
+
"metacognition"
|
|
5182
|
+
],
|
|
5183
|
+
"direction": "support",
|
|
5184
|
+
"study_type": "observational",
|
|
5185
|
+
"claim_text": "Controlled comparisons find LLM-generated code explanations comparable to (in places better than) student-authored explanations, suggesting viability as explanatory scaffold material rather than as a replacement for student explanation practice.",
|
|
5186
|
+
"effect_summary": "comparable-or-better rated quality vs student explanations",
|
|
5187
|
+
"confidence_markers": [
|
|
5188
|
+
"evidence_level:moderate",
|
|
5189
|
+
"quality_score:7.0",
|
|
5190
|
+
"confidence:0.55",
|
|
5191
|
+
"decision_relation:conditional"
|
|
5192
|
+
],
|
|
5193
|
+
"domains": [
|
|
5194
|
+
"ai-coding-assistant"
|
|
5195
|
+
]
|
|
5196
|
+
},
|
|
5197
|
+
{
|
|
5198
|
+
"entry_id": "lib-ai-coding-assistant-E-012",
|
|
5199
|
+
"source_id": "S-2022-vaithilingam",
|
|
5200
|
+
"title": "Expectation vs. Experience: Evaluating the Usability of Code Generation Tools",
|
|
5201
|
+
"year": 2022,
|
|
5202
|
+
"outcome_token": "over_reliance",
|
|
5203
|
+
"outcome_tokens": [
|
|
5204
|
+
"over_reliance"
|
|
5205
|
+
],
|
|
5206
|
+
"direction": "support",
|
|
5207
|
+
"study_type": "qualitative",
|
|
5208
|
+
"claim_text": "Despite faster first-task completion, participants struggled to understand and debug AI-generated solutions and reported low ownership of the final program - documenting metacognitive and dependence risks that pure speed metrics miss.",
|
|
5209
|
+
"effect_summary": "documented comprehension/ownership difficulties despite speed gain",
|
|
5210
|
+
"confidence_markers": [
|
|
5211
|
+
"evidence_level:moderate",
|
|
5212
|
+
"quality_score:7.0",
|
|
5213
|
+
"confidence:0.55",
|
|
5214
|
+
"decision_relation:conditional"
|
|
5215
|
+
],
|
|
5216
|
+
"domains": [
|
|
5217
|
+
"ai-coding-assistant"
|
|
5218
|
+
]
|
|
5219
|
+
},
|
|
4944
5220
|
{
|
|
4945
5221
|
"entry_id": "lib-ai-tutor-E-001",
|
|
4946
5222
|
"source_id": "S-2025-bastani",
|
|
@@ -5264,5 +5540,5 @@
|
|
|
5264
5540
|
]
|
|
5265
5541
|
}
|
|
5266
5542
|
],
|
|
5267
|
-
"coverage_note": "内置证据库:由 30 份金标准标注(benchmarks/annotations/gold-Q01..Q30 的 key_claims/key_supporting_sources/known_contradictions/correct_outcome_types)+ 3 个示例工作流 evidence.jsonl(ai-coding-assistant / ai-tutor / ai-writing-assistant)抽取生成;按 (source_id, outcome_token, claim_text) 去重合并。direction 语义为采纳方向:support
|
|
5543
|
+
"coverage_note": "内置证据库:由 30 份金标准标注(benchmarks/annotations/gold-Q01..Q30 的 key_claims/key_supporting_sources/known_contradictions/correct_outcome_types)+ 3 个示例工作流 evidence.jsonl(ai-coding-assistant / ai-tutor / ai-writing-assistant)抽取生成;按 (source_id, outcome_token, claim_text) 去重合并。direction 语义为采纳方向:support=支持采纳(离线初筛上限 => pilot,永不输出 adopt),contradict=反对采纳(oppose-only 时 => reject;与 support 并存属未决冲突,应交 engine.decision_policy.decision_outcome 判为 INSUFFICIENT_EVIDENCE),neutral=中性;金标准条目按 expected_decision_range 粗粒度映射方向(纯 reject 问题反向映射),conflict 与混合方向问题的单条断言方向可能不精确。仅用于离线初步裁决(preliminary,保守),从不直接给出 adopt;完整 ADOPT 只能由 engine.decision_policy.decision_outcome 给出(High + support + 主要结果 directness 2)。"
|
|
5268
5544
|
}
|
package/docs/architecture.md
CHANGED
|
@@ -183,7 +183,7 @@ motion/motion.css + motion/motion.js(data-lieflat reveal:滚入播放、
|
|
|
183
183
|
点击重播 + timer 清理、prefers-reduced-motion 降级、打印全开)
|
|
184
184
|
```
|
|
185
185
|
|
|
186
|
-
学术图(outcome-comparison / benchmark / forest)保持主题无关的出版级渲染;Lieflat 部分只渲染 `resolve_visual_layout` 校验通过的条目。注册表、契约与推荐组合见 `visualization/eduevidence-report/references/lieflat-composition.md`,schema 见 `visualization/eduevidence-report/schemas/visual-layout.schema.json
|
|
186
|
+
学术图(outcome-comparison / benchmark / forest)保持主题无关的出版级渲染;Lieflat 部分只渲染 `resolve_visual_layout` 校验通过的条目。注册表、契约与推荐组合见 `visualization/eduevidence-report/references/lieflat-composition.md`,schema 见 `visualization/eduevidence-report/schemas/visual-layout.schema.json`(完整路径;短路径 `references/lieflat-composition.md` 与 `/lieflat/lupi`、`docs/lieflat*` 均为已废弃死路径,禁止引用)。
|
|
187
187
|
|
|
188
188
|
该协议是**可剥离的**:即使没有 Agent 框架,只要按此协议组织检索、抽取、审计与裁决,也能得到可复现的决策链。
|
|
189
189
|
|
|
@@ -276,8 +276,11 @@ edu/
|
|
|
276
276
|
├── integrations/ # 集成层(Agent MCP 增强 + Smart Web Fetch)
|
|
277
277
|
├── visualization/ # 呈现层(eduevidence-report:build_report / build_charts /
|
|
278
278
|
│ # build_infographics / build_figures / charts_data 提取器 /
|
|
279
|
-
│ # lieflat_engine 注册表渲染器 / motion / themes / schemas
|
|
280
|
-
│ # lieflat-
|
|
279
|
+
│ # lieflat_engine 注册表渲染器 / motion / themes / schemas。
|
|
280
|
+
│ # 注册表文档:references/lieflat-composition.md;
|
|
281
|
+
│ # schema:schemas/visual-layout.schema.json。
|
|
282
|
+
│ # 注:历史路径 /lieflat/lupi 与 docs/lieflat* 已废弃,
|
|
283
|
+
│ # 不在 skill_payload / npm files 内;勿再引用。)
|
|
281
284
|
├── benchmarks/ # 基准评测(questions / annotations / baselines /
|
|
282
285
|
│ # evaluator / results / v2)
|
|
283
286
|
├── examples/ # 端到端示例
|
|
@@ -0,0 +1,250 @@
|
|
|
1
|
+
# Jev 生态实验模式(Experimental Jev / SemDecide Tier-0)
|
|
2
|
+
|
|
3
|
+
> **实验性加速层,不是科学门。** Tier-0(screen / rerank / extract / classify / verify)
|
|
4
|
+
> 只做窄域类型化判断,**不替代** `scripts/pre_verdict_gate.py`、Skeptic 九项检查、
|
|
5
|
+
> 方法学审计与 Tribunal。不确定 / 非法结果 **fail-closed** 升级大模型。
|
|
6
|
+
>
|
|
7
|
+
> 入口开关:`--experimental <0|1|2|3>` 或 `EDU_EXPERIMENTAL_JEV=<mode>`。
|
|
8
|
+
> 密钥仅存 `~/.eduevidence/env`(`AI_GATEWAY_API_KEY` + `JEV_PROVIDER=vercel`,可选
|
|
9
|
+
> `TYPESAFE_API_KEY`;或 FreeJev:`FREEJEV_API_KEY` + `JEV_PROVIDER=freejev`),
|
|
10
|
+
> **禁止写入仓库**。
|
|
11
|
+
|
|
12
|
+
## 1. 定位
|
|
13
|
+
|
|
14
|
+
| 组件 | 实现 | 上游 |
|
|
15
|
+
|---|---|---|
|
|
16
|
+
| `integrations/jev_mcp.py` | detect → approval → urllib 调 Vercel AI Gateway `typesafe-ai/jev` | TypeSafe System One / `@jkudish/jev-mcp` |
|
|
17
|
+
| `integrations/semantic_decide.py` | subprocess `semdecide is / filter / choose` | [sharziki/semdecide](https://github.com/sharziki/semdecide) |
|
|
18
|
+
| `engine/capabilities.py` | 注册 5 个实验能力(`experimental_capability_registry()`;默认科学注册表不含它们,避免协议角色门误伤) | 本仓库能力注册表 |
|
|
19
|
+
| `skill/workflows/experimental-jev.md` | 模式 / 九阶段映射 / fail-closed | Skill 工作流覆盖层 |
|
|
20
|
+
|
|
21
|
+
原则(与 agent-mcp 一致):**检测 → 推荐 → 用户确认 → 调用 → 降级**。不迁移、不复制
|
|
22
|
+
上游实现;上游不可用时整体退回 Platform Native / mode 0,科学协议不变。
|
|
23
|
+
|
|
24
|
+
## 2. 启用方式
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
# 0) 密钥(本地,勿提交)
|
|
28
|
+
# ~/.eduevidence/env
|
|
29
|
+
# # Vercel AI Gateway 路径
|
|
30
|
+
# export AI_GATEWAY_API_KEY=...
|
|
31
|
+
# export JEV_PROVIDER=vercel
|
|
32
|
+
# # 或 FreeJev 路径
|
|
33
|
+
# export FREEJEV_API_KEY=...
|
|
34
|
+
# export JEV_PROVIDER=freejev
|
|
35
|
+
|
|
36
|
+
# 1) 检测
|
|
37
|
+
python3 integrations/jev_mcp.py --experimental 3
|
|
38
|
+
python3 integrations/semantic_decide.py --experimental 3
|
|
39
|
+
|
|
40
|
+
# 2) 用户确认门 → ~/.eduevidence/jev_mcp_approval.json
|
|
41
|
+
python3 integrations/jev_mcp.py --experimental 3 --approve
|
|
42
|
+
|
|
43
|
+
# 3) 运行时打开 overlay(宿主 CLI 用 --enhancement 选择;写入 run 的 intake.json)
|
|
44
|
+
eduevidence run --question "..." --run-id <id> --enhancement jev --enhancement semdecide
|
|
45
|
+
# 检测类 CLI 用模式号:--experimental <0|1|2|3>,或 export EDU_EXPERIMENTAL_JEV=3
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
| Mode | 名称 | 后端 |
|
|
49
|
+
|---:|---|---|
|
|
50
|
+
| 0 | standard | 仅大模型(默认,无 `--experimental`) |
|
|
51
|
+
| 1 | jev-mcp | FreeJev / Vercel Gateway urllib **或** 文档化 MCP |
|
|
52
|
+
| 2 | semdecide | `semdecide is / filter / choose` |
|
|
53
|
+
| 3 | hybrid | 1 + 2 |
|
|
54
|
+
|
|
55
|
+
### Provider(Mode 1 / 3)
|
|
56
|
+
|
|
57
|
+
| `JEV_PROVIDER` | 密钥 | Decide 端点 | 备注 |
|
|
58
|
+
|---|---|---|---|
|
|
59
|
+
| `freejev` | `FREEJEV_API_KEY` | `https://freejev.org/api/v1/decide` | 仅 `FREEJEV_API_KEY` 时可自动选中 |
|
|
60
|
+
| `vercel`(默认) | `AI_GATEWAY_API_KEY`(或 `TYPESAFE_API_KEY`) | `https://ai-gateway.vercel.sh/typesafe/v1/systemone` | `model=typesafe-ai/jev` |
|
|
61
|
+
|
|
62
|
+
MCP 备选(不自动 spawn,仅文档化):
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
# FreeJev MCP
|
|
66
|
+
# https://freejev.org/mcp + FREEJEV_API_KEY
|
|
67
|
+
# npx 包路径
|
|
68
|
+
claude mcp add jev -- npx -y @jkudish/jev-mcp
|
|
69
|
+
# 通用 JSON:npx -y @jkudish/jev-mcp + TYPESAFE_API_KEY / AI_GATEWAY_API_KEY
|
|
70
|
+
python3 integrations/jev_mcp.py --mcp-guide
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
**FreeJev `request_id` 不重试。** Decide 调用的 `request_id` 是幂等键:同一
|
|
74
|
+
`request_id` **禁止重试**(超时 / 5xx / 未知结果一律不重发)。本调用
|
|
75
|
+
fail-closed 升级大模型;若业务上必须再问一次,使用**新的** `request_id` 发起
|
|
76
|
+
新调用,并单独记录 provenance。
|
|
77
|
+
|
|
78
|
+
## 3. Tier-0 API 列表
|
|
79
|
+
|
|
80
|
+
传输:`POST` 到 provider decide 端点(Vercel System One 或 FreeJev),
|
|
81
|
+
`Authorization: Bearer $AI_GATEWAY_API_KEY` / `$FREEJEV_API_KEY`,
|
|
82
|
+
`model=typesafe-ai/jev`。
|
|
83
|
+
请求体 TypeSafe System One:`{ model, state, questions }` → `{ model, answers, usage }`。
|
|
84
|
+
FreeJev 另带 `request_id`(见上:**不重试**)。
|
|
85
|
+
|
|
86
|
+
| # | Python API | capability_id | MCP 工具 | SemDecide 对应 | 问题形态 | 默认阈值 |
|
|
87
|
+
|---|---|---|---|---|---|---|
|
|
88
|
+
| 1 | `jev_mcp.screen(text, purpose)` | `content_screen` | `jev_screen` | — | 3× Noul:injection / substance / relevance | `block_at=0.75`, `review_at=0.25` |
|
|
89
|
+
| 2 | `jev_mcp.rerank(query, candidates)` | `semantic_rerank` | `jev_rerank` | — | 1× Noul / candidate | `auto_accept=0.70` |
|
|
90
|
+
| 3 | `jev_mcp.extract(document, fields)` | `field_extract` | `jev_extract` | — | 1× Choice / field(regex 候选 + `none_of_them`) | `auto_accept=0.80`, `margin=0.40` |
|
|
91
|
+
| 4 | `jev_mcp.classify(items, classes)` | `classify_check` | `jev_classify` | `choose` 可路由 | 1× Choice / item | `auto_accept=0.85`, `margin=0.50` |
|
|
92
|
+
| 5 | `jev_mcp.verify(claims, evidence)` | `claim_verify` | `jev_verify` | `is` 可预筛 | 1× Choice / claim:supports / contradicts / says_nothing | `auto_accept=0.80` |
|
|
93
|
+
|
|
94
|
+
统一入口:`jev_mcp.safe_call(capability_id, ...)`(先过 approval 门)。
|
|
95
|
+
|
|
96
|
+
SemDecide 封装(exit 2/3/4 → 升级大模型):
|
|
97
|
+
|
|
98
|
+
| API | CLI | 用途 |
|
|
99
|
+
|---|---|---|
|
|
100
|
+
| `semdecide.is_predicate(text, predicate)` | `semdecide is` | 语义谓词 |
|
|
101
|
+
| `semdecide.filter_records(records, predicate)` | `semdecide filter` | JSONL 语义过滤(保序) |
|
|
102
|
+
| `semdecide.choose_route(text, question, options)` | `semdecide choose` | 命名选项路由 |
|
|
103
|
+
|
|
104
|
+
### 状态与 fail-closed(与 `integrations/semantic_decide.py` 对齐)
|
|
105
|
+
|
|
106
|
+
Jev / Tier-0:
|
|
107
|
+
|
|
108
|
+
| 状态 | 含义 | 管道行为 |
|
|
109
|
+
|---|---|---|
|
|
110
|
+
| `ok` + `action/auto` | 可自动采用(仍不是科学门) | 记录 provenance,继续阶段 |
|
|
111
|
+
| `review` / `not_found` / `JEV_INVALID_RESPONSE` | 低置信或非法答案(envelope 缺 `answers` 等) | **升级大模型**;extract 的 `review` 值不得当作已抽取 |
|
|
112
|
+
| `JEV_APPROVAL_REQUIRED` | 无/错 approval | 不发网络请求 |
|
|
113
|
+
| `JEV_UNAVAILABLE` | 无密钥 / 无可用传输 | 本调用升级大模型;hybrid 可切 mode 2 切片 |
|
|
114
|
+
| `JEV_PROVIDER_ERROR` | 上游失败(HTTP 401/403/429/529/5xx) | 本调用升级大模型;FreeJev **`request_id` 不重试** |
|
|
115
|
+
|
|
116
|
+
SemDecide(`EXIT_MEANINGS` / `ESCALATE_EXIT_CODES={2,3,4}`):
|
|
117
|
+
|
|
118
|
+
| exit / status | `EXIT_MEANINGS` | 管道行为 |
|
|
119
|
+
|---|---|---|
|
|
120
|
+
| `0` | `true/selected/match` | 语义结果可用 |
|
|
121
|
+
| `1` | `false/no match` | 语义结果可用;`filter` 的 exit 1 **不**单独升级 |
|
|
122
|
+
| **`2`** | `invalid input` | 修输入重跑,或整段决策**升级大模型**(`escalate_plan`) |
|
|
123
|
+
| **`3`** | `uncertain` | **升级大模型 / 人工**;禁止强行 true/false |
|
|
124
|
+
| **`4`** | `provider failure` | **升级大模型**(仅本调用);记录 attempt |
|
|
125
|
+
| `SEMDECIDE_UNAVAILABLE` | binary 缺失 / spawn 失败 | **升级大模型**(`requires_llm_escalation`) |
|
|
126
|
+
| `SEMDECIDE_TIMEOUT` | 超时 | **升级大模型**(`result.escalate=ESCALATE_TO_LLM`) |
|
|
127
|
+
|
|
128
|
+
`screen` 的 recommendation(pass / review / block / skip)仅为咨询;拦截执行权在调用方。
|
|
129
|
+
|
|
130
|
+
## 4. 阈值
|
|
131
|
+
|
|
132
|
+
| 参数 | 默认 | 调节建议 |
|
|
133
|
+
|---|---:|---|
|
|
134
|
+
| `screen_block_at` | 0.75 | 注入概率 ≥ 则 block(咨询) |
|
|
135
|
+
| `screen_review_at` | 0.25 | 注入概率 ≥ 则 review |
|
|
136
|
+
| `verify_auto_accept` | 0.80 | 关系裁决 confidence ≥ 才 `auto` |
|
|
137
|
+
| `classify_auto_accept` | 0.85 | top 概率门槛 |
|
|
138
|
+
| `classify_minimum_margin` | 0.50 | 冠亚军差门槛(二者同时满足才 auto) |
|
|
139
|
+
| `extract_auto_accept` | 0.80 | 字段挑选 confidence |
|
|
140
|
+
| `extract_minimum_margin` | 0.40 | 候选差门槛;失败 → `review` |
|
|
141
|
+
| `rerank_auto_accept` | 0.70 | 相关 Noul 门槛 |
|
|
142
|
+
| SemDecide `--threshold` | 0.70 | `is` / `filter` |
|
|
143
|
+
| SemDecide `--min-confidence` | 0.70 | `choose`;低于则 exit 3 |
|
|
144
|
+
| SemDecide `--uncertainty-margin` | 0.0 | 边缘区不强行二值化 |
|
|
145
|
+
|
|
146
|
+
阈值是起点(TypeSafe cookbook 基线),**先在自有数据上校准再强制**。
|
|
147
|
+
|
|
148
|
+
## 5. 九阶段映射(加速点)
|
|
149
|
+
|
|
150
|
+
`frame → retrieve → extract → challenge → audit → adjudicate → applicability → intervene → evaluate`
|
|
151
|
+
|
|
152
|
+
| 阶段 | Jev Tier-0 | SemDecide | 不变的科学门 |
|
|
153
|
+
|---|---|---|---|
|
|
154
|
+
| Frame | — | `choose` 可选 | frame schema |
|
|
155
|
+
| Retrieve | screen / rerank / classify | `filter`, `is` | 检索计划 + 反方检索 + Fetch/Validate |
|
|
156
|
+
| Extract | extract / classify | `is` 预检 | Evidence Objects、task ≠ learning |
|
|
157
|
+
| Challenge | **verify 仅辅助** | `is` 辅助 | **Skeptic 九项检查** |
|
|
158
|
+
| Audit | verify 辅助 | `is` 预筛 | 方法学审计 + guard |
|
|
159
|
+
| Adjudicate | —(禁止 Tier-0 定裁决) | — | **Pre-Verdict Gate** + 四态裁决 |
|
|
160
|
+
| Applicability | classify 可选 | `choose` 可选 | 人群/条件/排除/不确定性 |
|
|
161
|
+
| Intervene | — | — | intervention_design |
|
|
162
|
+
| Evaluate | — | — | evaluation_design / data_validation |
|
|
163
|
+
|
|
164
|
+
完整矩阵见 `skill/workflows/experimental-jev.md`。
|
|
165
|
+
|
|
166
|
+
## 6. 加速比口径(speedup)
|
|
167
|
+
|
|
168
|
+
只比较**同一问题、同一 Complexity Gate、同一九阶段协议**下 mode 0 基线 vs 实验模式。
|
|
169
|
+
|
|
170
|
+
| 指标 | 定义 | 统计范围 |
|
|
171
|
+
|---|---|---|
|
|
172
|
+
| `T_stage` | 单阶段 wall-clock(s),从进入阶段到产物过 schema/gate | 仅计 **gate 通过** 的阶段 |
|
|
173
|
+
| `T_run` | Frame→Applicability(或声明的阶段子集)累计 | 同上 |
|
|
174
|
+
| `speedup_stage` | `T_stage(mode0) / T_stage(exp)` | per stage |
|
|
175
|
+
| `speedup_run` | `T_run(mode0) / T_run(exp)` | per run |
|
|
176
|
+
| `cost_tokens` | 主模型 + Tier-0 `usage.input_tokens/output_tokens` | 双方同口径合计 |
|
|
177
|
+
| `escalation_rate` | fail-closed 升级次数 / Tier-0 调用次数 | 实验侧 |
|
|
178
|
+
| `gate_pass_rate` | 过科学门的阶段数 / 执行阶段数 | 双方 |
|
|
179
|
+
|
|
180
|
+
规则:
|
|
181
|
+
|
|
182
|
+
1. **质量门优先**:`gate_pass_rate(exp) < gate_pass_rate(mode0)` 时 speedup **无效**(质量退化不算加速)。
|
|
183
|
+
2. Tier-0 调用耗时与 token 计入实验侧成本,不得只计大模型。
|
|
184
|
+
3. 升级到大模型的调用耗时计入实验侧(含往返),避免“看起来快”。
|
|
185
|
+
4. 未过 gate / 中断 / 人工改写的 run **不进入** speedup 均值,单独列表。
|
|
186
|
+
5. 报告格式:每 run 一行 `{run_id, mode, T_run, speedup_run, cost_tokens, escalation_rate, gate_pass_rate}`。
|
|
187
|
+
|
|
188
|
+
示例(示意,非承诺值):
|
|
189
|
+
|
|
190
|
+
| run | mode | T_run | speedup_run | escalations | gate_pass |
|
|
191
|
+
|---|---:|---:|---:|---:|---:|
|
|
192
|
+
| ai-cs1-b | 0 | 3120s | 1.00× | — | 7/7 |
|
|
193
|
+
| ai-cs1-j | 3 | 1810s | 1.72× | 4/38 | 7/7 |
|
|
194
|
+
|
|
195
|
+
## 7. A/B 验收
|
|
196
|
+
|
|
197
|
+
| 项 | 通过条件 |
|
|
198
|
+
|---|---|
|
|
199
|
+
| 科学完整性 | 挑战阶段九项 Skeptic 全跑;Adjudicate 前 `pre_verdict_gate` 必跑且结论不被 Tier-0 覆盖 |
|
|
200
|
+
| 质量非劣 | `gate_pass_rate` 不低于基线;Pre-Verdict critical 失败数不增加 |
|
|
201
|
+
| 校准 | `verify`/`classify` 的 auto 集抽样人工一致率 ≥ 约定阈值(建议 ≥ 0.9) |
|
|
202
|
+
| Fail-closed | 抽检 100% 的 SemDecide exit 2/3/4、`SEMDECIDE_TIMEOUT`/`SEMDECIDE_UNAVAILABLE` 与 `JEV_INVALID_RESPONSE`/`JEV_PROVIDER_ERROR` 都有升级记录 |
|
|
203
|
+
| 密钥卫生 | 仓库与 run 产物无 `AI_GATEWAY_API_KEY` / `TYPESAFE_API_KEY` / `FREEJEV_API_KEY` 明文 |
|
|
204
|
+
| FreeJev 幂等 | 抽检无同一 `request_id` 重试;失败一律升级大模型 |
|
|
205
|
+
| 可复现 | mode、approval `tools_hash`、阈值、上游 model id 写入 run 元数据 |
|
|
206
|
+
| 加速有效 | `speedup_run ≥ 1.2`(目标,可调)且质量非劣;否则保持 mode 0 |
|
|
207
|
+
|
|
208
|
+
## 8. 降级路径
|
|
209
|
+
|
|
210
|
+
```
|
|
211
|
+
--experimental 3 (hybrid)
|
|
212
|
+
├─ provider key + jev_mcp_approval.json → Tier-0 screen/rerank/extract/classify/verify
|
|
213
|
+
│ (FreeJev: FREEJEV_API_KEY + JEV_PROVIDER=freejev;request_id 不重试)
|
|
214
|
+
├─ semdecide on PATH → is/filter/choose
|
|
215
|
+
├─ exit 2/3/4 / SEMDECIDE_TIMEOUT
|
|
216
|
+
│ / JEV_INVALID_RESPONSE / JEV_PROVIDER_ERROR
|
|
217
|
+
│ → 单次升级大模型(fail-closed)
|
|
218
|
+
├─ Jev 不可用 → 仅 SemDecide(mode 2 切片)
|
|
219
|
+
├─ semdecide 不可用 → 仅 Jev(mode 1 切片)
|
|
220
|
+
└─ 双不可用 / 未开 --experimental → mode 0 standard
|
|
221
|
+
|
|
222
|
+
科学门永不降级:Pre-Verdict Gate / Skeptic 九项 / 方法学审计 / Tribunal 始终在宿主路径执行。
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
## 9. API 速查(代码)
|
|
226
|
+
|
|
227
|
+
```python
|
|
228
|
+
from integrations.jev_mcp import (
|
|
229
|
+
detect_jev_mcp, write_approval, load_approval,
|
|
230
|
+
screen, rerank, extract, classify, verify, safe_call,
|
|
231
|
+
resolve_experimental_mode, npx_mcp_guide,
|
|
232
|
+
)
|
|
233
|
+
from integrations.semantic_decide import (
|
|
234
|
+
detect_semdecide, is_predicate, filter_records, choose_route,
|
|
235
|
+
requires_llm_escalation, escalate_plan,
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
mode = resolve_experimental_mode(3) # 0..3
|
|
239
|
+
approval = load_approval() # ~/.eduevidence/jev_mcp_approval.json
|
|
240
|
+
verdict = safe_call("claim_verify", claims, evidence, approval=approval)
|
|
241
|
+
if verdict.get("escalate"):
|
|
242
|
+
... # fail-closed → large model
|
|
243
|
+
```
|
|
244
|
+
|
|
245
|
+
## 10. 变更记录
|
|
246
|
+
|
|
247
|
+
| 日期 | 内容 |
|
|
248
|
+
|---|---|
|
|
249
|
+
| 2026-09-22 | 首版:Tier-0 五能力、approval 门、mode 0–3、阈值、加速比口径、A/B 验收、降级路径 |
|
|
250
|
+
| 2026-09-22 | 收口:FreeJev(`FREEJEV_API_KEY` / `JEV_PROVIDER=freejev` / MCP `https://freejev.org/mcp` / `request_id` 不重试);fail-closed 表与 `semantic_decide.py` 对齐 |
|