eduevidence 6.2.0 → 6.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +395 -0
- package/README.md +22 -13
- package/README.zh-CN.md +15 -8
- package/SKILL.md +10 -9
- package/benchmarks/evidence-library.json +277 -1
- package/docs/architecture.md +6 -3
- package/docs/j-ev-experimental.md +250 -0
- package/docs/reproducibility.md +138 -0
- package/domains/_neutral/copy/few_shots.json +21 -0
- package/domains/_neutral/copy/framing_lexicon.json +19 -0
- package/domains/_neutral/copy/module_labels.json +5 -0
- package/domains/_neutral/copy/module_labels_footer.json +102 -0
- package/domains/_neutral/copy/module_labels_modules.json +204 -0
- package/domains/_neutral/copy/module_labels_nav.json +126 -0
- package/domains/_neutral/copy/module_labels_summary.json +98 -0
- package/domains/_neutral/copy/module_labels_tables.json +164 -0
- package/domains/_neutral/copy/module_labels_v2.json +90 -0
- package/domains/_neutral/copy/risk_constructs.json +20 -0
- package/domains/_neutral/copy/section_titles.json +66 -0
- package/domains/_neutral/copy/terminology.json +11 -0
- package/domains/check_copy_packs.py +103 -0
- package/domains/education/copy/few_shots.json +22 -0
- package/domains/education/copy/framing_enums.json +167 -0
- package/domains/education/copy/framing_lexicon.json +166 -0
- package/domains/education/copy/module_labels.json +169 -0
- package/domains/education/copy/risk_constructs.json +48 -0
- package/domains/education/copy/section_titles.json +186 -0
- package/domains/education/copy/terminology.json +70 -0
- package/domains/education/manifest.json +1 -1
- package/domains/education/outcome_taxonomy.json +2 -2
- package/domains/manifest.json +1 -1
- package/domains/policy/copy/few_shots.json +22 -0
- package/domains/policy/copy/framing_enums.json +94 -0
- package/domains/policy/copy/framing_lexicon.json +174 -0
- package/domains/policy/copy/module_labels.json +168 -0
- package/domains/policy/copy/risk_constructs.json +33 -0
- package/domains/policy/copy/section_titles.json +186 -0
- package/domains/policy/copy/terminology.json +64 -0
- package/engine/capabilities.py +57 -5
- package/engine/decision_policy.py +88 -17
- package/engine/library_builtin.py +7 -4
- package/engine/tribunal.py +17 -23
- package/engine/versions.py +1 -1
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +4 -4
- package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
- package/examples/spaced-retrieval-practice/report.html +2522 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +4 -4
- package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +36 -36
- package/integrations/jev/__init__.py +115 -0
- package/integrations/jev/approval.py +212 -0
- package/integrations/jev/cli.py +84 -0
- package/integrations/jev/config.py +112 -0
- package/integrations/jev/gateway.py +128 -0
- package/integrations/jev/modes.py +38 -0
- package/integrations/jev/tools_classify.py +88 -0
- package/integrations/jev/tools_extract.py +111 -0
- package/integrations/jev/tools_rerank.py +71 -0
- package/integrations/jev/tools_screen.py +87 -0
- package/integrations/jev/tools_verify.py +95 -0
- package/integrations/jev_mcp.py +22 -0
- package/integrations/semantic_decide.py +286 -0
- package/integrations/semdecide_cli.py +55 -0
- package/package.json +9 -1
- package/pyproject.toml +1 -1
- package/references/report-copy-style.md +43 -3
- package/schemas/v2/decision-snapshot.schema.json +20 -9
- package/schemas/v2/intake.schema.json +191 -0
- package/scripts/build_evidence_library.py +15 -5
- package/scripts/dashboard_server.py +13 -2
- package/scripts/intake/__init__.py +31 -0
- package/scripts/intake/__main__.py +18 -0
- package/scripts/intake/background.py +78 -0
- package/scripts/intake/browser.py +79 -0
- package/scripts/intake/cli.py +57 -0
- package/scripts/intake/constants.py +57 -0
- package/scripts/intake/depth.py +53 -0
- package/scripts/intake/enhancements.py +106 -0
- package/scripts/intake/hooks.py +90 -0
- package/scripts/intake/prefs.py +76 -0
- package/scripts/intake/prompts.py +85 -0
- package/scripts/intake/session.py +152 -0
- package/scripts/lint_file_layers.py +126 -0
- package/scripts/orchestrator.py +68 -17
- package/scripts/pre_verdict_gate.py +21 -7
- package/scripts/skill_lint.py +11 -1
- package/scripts/skill_payload.py +3 -3
- package/scripts/test_adversarial_empirical.py +70 -6
- package/skill/agents/evidence-judge.md +49 -7
- package/skill/workflows/experimental-jev.md +170 -0
- package/skill/workflows/intake.md +120 -0
- package/visualization/eduevidence-report/scripts/build_infographics.py +32 -14
- package/visualization/eduevidence-report/scripts/build_report.py +75 -662
- package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
- package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
- package/visualization/eduevidence-report/scripts/zh_labels.py +61 -0
- package/scripts/build_esl_artifacts.py +0 -1921
- package/scripts/build_killer_demo.py +0 -295
- package/scripts/enrich_projects_human_and_lieflat.py +0 -315
- package/scripts/generate_new_projects.py +0 -686
- package/scripts/sync_killer_demo_report.py +0 -270
|
@@ -24,7 +24,8 @@ Conservative verdict rules (offline preliminary gate):
|
|
|
24
24
|
any matched contradict entry -> reject
|
|
25
25
|
else any matched support entry -> pilot
|
|
26
26
|
else -> insufficient_evidence
|
|
27
|
-
|
|
27
|
+
The preliminary gate is a screening bound only; a full ADOPT still requires
|
|
28
|
+
the deterministic decision_action gate (High + support + direct primary).
|
|
28
29
|
|
|
29
30
|
Output:
|
|
30
31
|
{"verdict": ..., "coverage": {"matched_entries": [...],
|
|
@@ -228,8 +229,9 @@ def preliminary_verdict(question: str, *, top_k: int = 10) -> dict[str, Any]:
|
|
|
228
229
|
claim_text + effect_summary + title; the top_k entries are considered and an
|
|
229
230
|
entry counts as matched when overlap >= MATCH_THRESHOLD and it shares at
|
|
230
231
|
least MIN_SHARED_BIGRAMS tokens. Verdict: contradict => reject, else
|
|
231
|
-
support => pilot, else insufficient_evidence.
|
|
232
|
-
|
|
232
|
+
support => pilot, else insufficient_evidence. This is a screening bound
|
|
233
|
+
only; full ADOPT is decided by engine.decision_policy.decision_action.
|
|
234
|
+
Never crashes on empty/blank questions.
|
|
233
235
|
"""
|
|
234
236
|
try:
|
|
235
237
|
top_k = int(top_k)
|
|
@@ -298,6 +300,7 @@ def _build_note(
|
|
|
298
300
|
f"离线初步裁决在 top_k={top_k} 内匹配到 {len(matched)} 条内置证据:"
|
|
299
301
|
f"support={counts['support']}、contradict={counts['contradict']}、"
|
|
300
302
|
f"neutral={counts['neutral']};匹配结局词:{outcome_str}。"
|
|
301
|
-
"本裁决为初步(preliminary=true
|
|
303
|
+
"本裁决为初步(preliminary=true)筛查上界;完整四态裁决仍须经 "
|
|
304
|
+
"decision_action 闸门(High + support + 主结果直接证据才可 adopt)。"
|
|
302
305
|
"建议结合完整证据库与在线检索复核。"
|
|
303
306
|
)
|
package/engine/tribunal.py
CHANGED
|
@@ -40,14 +40,14 @@ from pathlib import Path
|
|
|
40
40
|
from engine.contracts import validate_record
|
|
41
41
|
from engine.decision_policy import (
|
|
42
42
|
ADOPT_DIRECTNESS,
|
|
43
|
-
|
|
43
|
+
decision_outcome,
|
|
44
44
|
outcome_category,
|
|
45
45
|
primary_effect_categories,
|
|
46
46
|
)
|
|
47
47
|
from engine.graph_store import GraphStore
|
|
48
48
|
from engine.ids import new_local_id
|
|
49
49
|
from engine.project import ProjectWorkspace
|
|
50
|
-
from engine.semantics import
|
|
50
|
+
from engine.semantics import decision_implication
|
|
51
51
|
from engine.synthesis import ClaimSynthesis, synthesize_project
|
|
52
52
|
from engine.versions import (
|
|
53
53
|
CONFIDENCE_POLICY_VERSION,
|
|
@@ -199,6 +199,7 @@ def _confidence(store: GraphStore, syntheses: tuple[ClaimSynthesis, ...]) -> dic
|
|
|
199
199
|
if relation in ("support_adoption", "oppose_adoption"):
|
|
200
200
|
decisive[sid] = relation
|
|
201
201
|
elif relation == "conditional":
|
|
202
|
+
decisive[sid] = "conditional"
|
|
202
203
|
critical_uncertainty_units += 1
|
|
203
204
|
quality_sum += (
|
|
204
205
|
audit.get("design_quality", 0) + audit.get("sample_quality", 0)
|
|
@@ -209,7 +210,8 @@ def _confidence(store: GraphStore, syntheses: tuple[ClaimSynthesis, ...]) -> dic
|
|
|
209
210
|
directness_sum += sum(dirs) / len(dirs) / 2.0
|
|
210
211
|
directness_count += 1
|
|
211
212
|
|
|
212
|
-
n_decisive =
|
|
213
|
+
n_decisive = sum(1 for r in decisive.values()
|
|
214
|
+
if r in ("support_adoption", "oppose_adoption"))
|
|
213
215
|
n_usable = len(usable_studies)
|
|
214
216
|
if n_decisive == 0:
|
|
215
217
|
return {"score": 0.0, "label": "Insufficient",
|
|
@@ -280,22 +282,6 @@ def _has_direct_primary_evidence(store: GraphStore,
|
|
|
280
282
|
return False
|
|
281
283
|
|
|
282
284
|
|
|
283
|
-
def _decision_action(syn_statuses: dict[str, str], confidence: dict,
|
|
284
|
-
decisive_relations: dict[str, str],
|
|
285
|
-
has_direct_primary_evidence: bool = False) -> str:
|
|
286
|
-
"""Gate-enforced decision action; rule lives in engine.decision_policy.
|
|
287
|
-
|
|
288
|
-
The V1 Pre-Verdict Gate enforces the same rule, so the thresholds are
|
|
289
|
-
imported rather than restated here.
|
|
290
|
-
"""
|
|
291
|
-
return _policy_decision_action(
|
|
292
|
-
confidence_label=confidence["label"],
|
|
293
|
-
decisive_relations=decisive_relations,
|
|
294
|
-
has_direct_primary_evidence=has_direct_primary_evidence,
|
|
295
|
-
)
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
285
|
def adjudicate(store: GraphStore, *, project: ProjectWorkspace,
|
|
300
286
|
claim_syntheses: tuple[ClaimSynthesis, ...] | None = None,
|
|
301
287
|
applicability: dict | None = None,
|
|
@@ -305,13 +291,19 @@ def adjudicate(store: GraphStore, *, project: ProjectWorkspace,
|
|
|
305
291
|
confidence = _confidence(store, syntheses)
|
|
306
292
|
applicability = applicability or {"boundary": "evidence scope", "notes": ""}
|
|
307
293
|
|
|
308
|
-
syn_statuses = {s.claim_id: s.status for s in syntheses}
|
|
309
294
|
decisive_relations = confidence.get("decisive_relations", {})
|
|
310
295
|
domain = str(project.manifest().get("domain") or "education")
|
|
311
296
|
direct_learning = _has_direct_primary_evidence(store, decisive_relations, domain)
|
|
312
297
|
|
|
313
|
-
|
|
314
|
-
|
|
298
|
+
# Single authority: engine.decision_policy owns the four-state matrix, and
|
|
299
|
+
# engine/tribunal.py keeps no private copy of the rule that could drift.
|
|
300
|
+
outcome = decision_outcome(
|
|
301
|
+
confidence_label=str(confidence.get("label") or ""),
|
|
302
|
+
decisive_relations=decisive_relations,
|
|
303
|
+
has_direct_primary_evidence=direct_learning,
|
|
304
|
+
)
|
|
305
|
+
decision = str(outcome["action"])
|
|
306
|
+
downgrade_reason = outcome.get("downgrade_reason")
|
|
315
307
|
|
|
316
308
|
key_links: list[str] = []
|
|
317
309
|
for syn in syntheses:
|
|
@@ -322,7 +314,8 @@ def adjudicate(store: GraphStore, *, project: ProjectWorkspace,
|
|
|
322
314
|
for syn in syntheses:
|
|
323
315
|
risks.extend(syn.unresolved_conflicts)
|
|
324
316
|
if not risks and decision == "ADOPT":
|
|
325
|
-
risks.append(
|
|
317
|
+
risks.append(
|
|
318
|
+
"primary-outcome durability beyond the observed window may still be untested")
|
|
326
319
|
|
|
327
320
|
missing: list[str] = []
|
|
328
321
|
for syn in syntheses:
|
|
@@ -343,6 +336,7 @@ def adjudicate(store: GraphStore, *, project: ProjectWorkspace,
|
|
|
343
336
|
snapshot = {
|
|
344
337
|
"decision_snapshot_id": new_local_id("DEC", set()),
|
|
345
338
|
"decision": decision,
|
|
339
|
+
"downgrade_reason": downgrade_reason,
|
|
346
340
|
"confidence_label": confidence["label"],
|
|
347
341
|
"confidence_score_internal": confidence["score"],
|
|
348
342
|
"claim_assessments": claim_assessments,
|
package/engine/versions.py
CHANGED
|
@@ -4,7 +4,7 @@ Policy versions are frozen identifiers, not free-form strings: changing a
|
|
|
4
4
|
policy requires a new version, never silent mutation of an existing one.
|
|
5
5
|
"""
|
|
6
6
|
|
|
7
|
-
ENGINE_VERSION = "6.
|
|
7
|
+
ENGINE_VERSION = "6.3.0"
|
|
8
8
|
GRAPH_SCHEMA_VERSION = "2.0"
|
|
9
9
|
SOURCE_VALIDATION_POLICY_VERSION = "2026-08-12.v2"
|
|
10
10
|
METHODOLOGY_POLICY_VERSION = "2026-08-12.v2"
|