eduevidence 6.2.0 → 6.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +395 -0
- package/README.md +22 -13
- package/README.zh-CN.md +15 -8
- package/SKILL.md +10 -9
- package/benchmarks/evidence-library.json +277 -1
- package/docs/architecture.md +6 -3
- package/docs/j-ev-experimental.md +250 -0
- package/docs/reproducibility.md +138 -0
- package/domains/_neutral/copy/few_shots.json +21 -0
- package/domains/_neutral/copy/framing_lexicon.json +19 -0
- package/domains/_neutral/copy/module_labels.json +5 -0
- package/domains/_neutral/copy/module_labels_footer.json +102 -0
- package/domains/_neutral/copy/module_labels_modules.json +204 -0
- package/domains/_neutral/copy/module_labels_nav.json +126 -0
- package/domains/_neutral/copy/module_labels_summary.json +98 -0
- package/domains/_neutral/copy/module_labels_tables.json +164 -0
- package/domains/_neutral/copy/module_labels_v2.json +90 -0
- package/domains/_neutral/copy/risk_constructs.json +20 -0
- package/domains/_neutral/copy/section_titles.json +66 -0
- package/domains/_neutral/copy/terminology.json +11 -0
- package/domains/check_copy_packs.py +103 -0
- package/domains/education/copy/few_shots.json +22 -0
- package/domains/education/copy/framing_enums.json +167 -0
- package/domains/education/copy/framing_lexicon.json +166 -0
- package/domains/education/copy/module_labels.json +169 -0
- package/domains/education/copy/risk_constructs.json +48 -0
- package/domains/education/copy/section_titles.json +186 -0
- package/domains/education/copy/terminology.json +70 -0
- package/domains/education/manifest.json +1 -1
- package/domains/education/outcome_taxonomy.json +2 -2
- package/domains/manifest.json +1 -1
- package/domains/policy/copy/few_shots.json +22 -0
- package/domains/policy/copy/framing_enums.json +94 -0
- package/domains/policy/copy/framing_lexicon.json +174 -0
- package/domains/policy/copy/module_labels.json +168 -0
- package/domains/policy/copy/risk_constructs.json +33 -0
- package/domains/policy/copy/section_titles.json +186 -0
- package/domains/policy/copy/terminology.json +64 -0
- package/engine/capabilities.py +57 -5
- package/engine/decision_policy.py +88 -17
- package/engine/library_builtin.py +7 -4
- package/engine/tribunal.py +17 -23
- package/engine/versions.py +1 -1
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +4 -4
- package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
- package/examples/spaced-retrieval-practice/report.html +2522 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +4 -4
- package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +36 -36
- package/integrations/jev/__init__.py +115 -0
- package/integrations/jev/approval.py +212 -0
- package/integrations/jev/cli.py +84 -0
- package/integrations/jev/config.py +112 -0
- package/integrations/jev/gateway.py +128 -0
- package/integrations/jev/modes.py +38 -0
- package/integrations/jev/tools_classify.py +88 -0
- package/integrations/jev/tools_extract.py +111 -0
- package/integrations/jev/tools_rerank.py +71 -0
- package/integrations/jev/tools_screen.py +87 -0
- package/integrations/jev/tools_verify.py +95 -0
- package/integrations/jev_mcp.py +22 -0
- package/integrations/semantic_decide.py +286 -0
- package/integrations/semdecide_cli.py +55 -0
- package/package.json +9 -1
- package/pyproject.toml +1 -1
- package/references/report-copy-style.md +43 -3
- package/schemas/v2/decision-snapshot.schema.json +20 -9
- package/schemas/v2/intake.schema.json +191 -0
- package/scripts/build_evidence_library.py +15 -5
- package/scripts/dashboard_server.py +13 -2
- package/scripts/intake/__init__.py +31 -0
- package/scripts/intake/__main__.py +18 -0
- package/scripts/intake/background.py +78 -0
- package/scripts/intake/browser.py +79 -0
- package/scripts/intake/cli.py +57 -0
- package/scripts/intake/constants.py +57 -0
- package/scripts/intake/depth.py +53 -0
- package/scripts/intake/enhancements.py +106 -0
- package/scripts/intake/hooks.py +90 -0
- package/scripts/intake/prefs.py +76 -0
- package/scripts/intake/prompts.py +85 -0
- package/scripts/intake/session.py +152 -0
- package/scripts/lint_file_layers.py +126 -0
- package/scripts/orchestrator.py +68 -17
- package/scripts/pre_verdict_gate.py +21 -7
- package/scripts/skill_lint.py +11 -1
- package/scripts/skill_payload.py +3 -3
- package/scripts/test_adversarial_empirical.py +70 -6
- package/skill/agents/evidence-judge.md +49 -7
- package/skill/workflows/experimental-jev.md +170 -0
- package/skill/workflows/intake.md +120 -0
- package/visualization/eduevidence-report/scripts/build_infographics.py +32 -14
- package/visualization/eduevidence-report/scripts/build_report.py +75 -662
- package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
- package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
- package/visualization/eduevidence-report/scripts/zh_labels.py +61 -0
- package/scripts/build_esl_artifacts.py +0 -1921
- package/scripts/build_killer_demo.py +0 -295
- package/scripts/enrich_projects_human_and_lieflat.py +0 -315
- package/scripts/generate_new_projects.py +0 -686
- package/scripts/sync_killer_demo_report.py +0 -270
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""classify_check — batch label items against a closed class catalog."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from .config import ESCALATE_TO_LLM, JEV_INVALID_RESPONSE, THRESHOLDS
|
|
7
|
+
from .gateway import _prob, system_one
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def classify(items: list[dict[str, Any]], classes: list[dict[str, Any]],
|
|
11
|
+
*, purpose: str = "classify", approval: dict[str, Any] | None = None,
|
|
12
|
+
auto_accept: float | None = None,
|
|
13
|
+
minimum_margin: float | None = None,
|
|
14
|
+
timeout: float = 20.0) -> dict[str, Any]:
|
|
15
|
+
"""classify_check — batch label items against a closed class catalog."""
|
|
16
|
+
auto_accept = (THRESHOLDS["classify_auto_accept"]
|
|
17
|
+
if auto_accept is None else auto_accept)
|
|
18
|
+
minimum_margin = (THRESHOLDS["classify_minimum_margin"]
|
|
19
|
+
if minimum_margin is None else minimum_margin)
|
|
20
|
+
class_map = {str(c.get("id")): str(c.get("description") or c.get("id"))
|
|
21
|
+
for c in classes[:250] if c.get("id")}
|
|
22
|
+
if not class_map:
|
|
23
|
+
return {"status": JEV_INVALID_RESPONSE, "tool": "classify", "results": [],
|
|
24
|
+
"error": "empty class catalog", "escalate": ESCALATE_TO_LLM}
|
|
25
|
+
state = {
|
|
26
|
+
"purpose": purpose[:500],
|
|
27
|
+
"classes": class_map,
|
|
28
|
+
"items": [
|
|
29
|
+
{"id": str(it.get("id", i)), "text": str(it.get("text", ""))[:2_000]}
|
|
30
|
+
for i, it in enumerate(items[:64])
|
|
31
|
+
],
|
|
32
|
+
}
|
|
33
|
+
questions = {
|
|
34
|
+
f"i{j}": {
|
|
35
|
+
"type": "choice",
|
|
36
|
+
"instructions": {
|
|
37
|
+
"purpose": purpose[:500],
|
|
38
|
+
"item": state["items"][j]["text"],
|
|
39
|
+
"question": "Which class best describes this item?",
|
|
40
|
+
},
|
|
41
|
+
"criteria": {cid: None for cid in class_map},
|
|
42
|
+
}
|
|
43
|
+
for j in range(len(state["items"]))
|
|
44
|
+
}
|
|
45
|
+
result = system_one(state, questions, approval=approval, tool="classify",
|
|
46
|
+
timeout=timeout)
|
|
47
|
+
if result.get("status") != "ok":
|
|
48
|
+
result["results"] = []
|
|
49
|
+
return result
|
|
50
|
+
|
|
51
|
+
outputs: list[dict[str, Any]] = []
|
|
52
|
+
for j, item in enumerate(state["items"]):
|
|
53
|
+
ans = result["answers"].get(f"i{j}") or {}
|
|
54
|
+
label = ans.get("choice")
|
|
55
|
+
probs = {str(k): _prob(v) or 0.0 for k, v in (ans.get("probabilities") or {}).items()}
|
|
56
|
+
conf = _prob(ans.get("confidence"))
|
|
57
|
+
values = sorted(probs.values(), reverse=True)
|
|
58
|
+
margin = (values[0] - values[1]) if len(values) >= 2 else 1.0
|
|
59
|
+
if label not in class_map:
|
|
60
|
+
outputs.append({"id": item["id"], "classification": None,
|
|
61
|
+
"status": JEV_INVALID_RESPONSE, "confidence": conf})
|
|
62
|
+
continue
|
|
63
|
+
auto = (conf is not None and conf >= auto_accept and margin >= minimum_margin)
|
|
64
|
+
outputs.append({
|
|
65
|
+
"id": item["id"],
|
|
66
|
+
"classification": label,
|
|
67
|
+
"margin": margin,
|
|
68
|
+
"confidence": conf,
|
|
69
|
+
"decision": "auto" if auto else "review",
|
|
70
|
+
"probabilities": probs,
|
|
71
|
+
})
|
|
72
|
+
by_class: dict[str, int] = {}
|
|
73
|
+
for row in outputs:
|
|
74
|
+
if row.get("classification"):
|
|
75
|
+
by_class[row["classification"]] = by_class.get(row["classification"], 0) + 1
|
|
76
|
+
result["results"] = outputs
|
|
77
|
+
result["summary"] = {
|
|
78
|
+
"items": len(outputs),
|
|
79
|
+
"auto": sum(1 for r in outputs if r.get("decision") == "auto"),
|
|
80
|
+
"review": sum(1 for r in outputs if r.get("decision") != "auto"),
|
|
81
|
+
"by_class": by_class,
|
|
82
|
+
}
|
|
83
|
+
result["auto_accept"] = auto_accept
|
|
84
|
+
result["minimum_margin"] = minimum_margin
|
|
85
|
+
if any(r.get("status") == JEV_INVALID_RESPONSE for r in outputs):
|
|
86
|
+
result["status"] = JEV_INVALID_RESPONSE
|
|
87
|
+
result["escalate"] = ESCALATE_TO_LLM
|
|
88
|
+
return result
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""field_extract — pick verbatim regex matches; never model-written values."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import re as _re
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from .config import ESCALATE_TO_LLM, JEV_INVALID_RESPONSE, THRESHOLDS
|
|
8
|
+
from .gateway import _prob, system_one
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def extract(document: str, fields: list[dict[str, Any]], *,
|
|
12
|
+
approval: dict[str, Any] | None = None,
|
|
13
|
+
auto_accept: float | None = None,
|
|
14
|
+
minimum_margin: float | None = None,
|
|
15
|
+
timeout: float = 20.0) -> dict[str, Any]:
|
|
16
|
+
"""field_extract — pick verbatim regex matches; never model-written values."""
|
|
17
|
+
auto_accept = THRESHOLDS["extract_auto_accept"] if auto_accept is None else auto_accept
|
|
18
|
+
minimum_margin = (THRESHOLDS["extract_minimum_margin"]
|
|
19
|
+
if minimum_margin is None else minimum_margin)
|
|
20
|
+
doc = document[:50_000]
|
|
21
|
+
prepared: list[dict[str, Any]] = []
|
|
22
|
+
questions: dict[str, Any] = {}
|
|
23
|
+
for i, field in enumerate(fields[:32]):
|
|
24
|
+
fid = str(field.get("id") or f"field{i}")
|
|
25
|
+
pattern = field.get("pattern") or ""
|
|
26
|
+
try:
|
|
27
|
+
matches = [m.group(0) for m in _re.finditer(pattern, doc)]
|
|
28
|
+
except _re.error as exc:
|
|
29
|
+
prepared.append({"id": fid, "value": None, "status": "invalid_pattern",
|
|
30
|
+
"reason": str(exc)})
|
|
31
|
+
continue
|
|
32
|
+
matches = [m for m in matches if 0 < len(m) <= 2_000][:20]
|
|
33
|
+
if not matches:
|
|
34
|
+
prepared.append({"id": fid, "value": None, "status": "not_found",
|
|
35
|
+
"reason": "no_regex_matches", "candidates_considered": 0})
|
|
36
|
+
continue
|
|
37
|
+
uniq: list[str] = []
|
|
38
|
+
for m in matches:
|
|
39
|
+
if m not in uniq:
|
|
40
|
+
uniq.append(m)
|
|
41
|
+
criteria = {f"m{j}": (m[:200] if m else m) for j, m in enumerate(uniq)}
|
|
42
|
+
criteria["none_of_them"] = "None of the candidate matches is this field's value"
|
|
43
|
+
questions[f"f{i}"] = {
|
|
44
|
+
"type": "choice",
|
|
45
|
+
"instructions": {
|
|
46
|
+
"description": str(field.get("description") or fid)[:500],
|
|
47
|
+
"question": "Which candidate is the field's real value in the document?",
|
|
48
|
+
},
|
|
49
|
+
"criteria": {k: None for k in criteria},
|
|
50
|
+
}
|
|
51
|
+
prepared.append({
|
|
52
|
+
"id": fid, "question_key": f"f{i}", "status": "pending",
|
|
53
|
+
"candidates": uniq, "value": None,
|
|
54
|
+
})
|
|
55
|
+
|
|
56
|
+
if not questions:
|
|
57
|
+
return {"status": "ok", "tool": "extract", "results": prepared,
|
|
58
|
+
"usage": {"input_tokens": 0, "output_tokens": 0},
|
|
59
|
+
"note": "all fields zero-match or invalid; no API call made",
|
|
60
|
+
"escalate": None}
|
|
61
|
+
|
|
62
|
+
result = system_one({"document": doc}, questions, approval=approval,
|
|
63
|
+
tool="extract", timeout=timeout)
|
|
64
|
+
if result.get("status") != "ok":
|
|
65
|
+
result["results"] = prepared
|
|
66
|
+
return result
|
|
67
|
+
|
|
68
|
+
outputs: list[dict[str, Any]] = []
|
|
69
|
+
for row in prepared:
|
|
70
|
+
if row.get("status") in ("not_found", "invalid_pattern"):
|
|
71
|
+
outputs.append(row)
|
|
72
|
+
continue
|
|
73
|
+
ans = result["answers"].get(row["question_key"]) or {}
|
|
74
|
+
choice = ans.get("choice")
|
|
75
|
+
probs = ans.get("probabilities") or {}
|
|
76
|
+
conf = _prob(ans.get("confidence"))
|
|
77
|
+
candidates = row["candidates"]
|
|
78
|
+
if choice == "none_of_them":
|
|
79
|
+
outputs.append({"id": row["id"], "value": None, "status": "not_found",
|
|
80
|
+
"reason": "none_of_them", "confidence": conf,
|
|
81
|
+
"candidates_considered": len(candidates)})
|
|
82
|
+
continue
|
|
83
|
+
try:
|
|
84
|
+
idx = int(str(choice)[1:]) if str(choice).startswith("m") else -1
|
|
85
|
+
except ValueError:
|
|
86
|
+
idx = -1
|
|
87
|
+
if idx < 0 or idx >= len(candidates):
|
|
88
|
+
outputs.append({"id": row["id"], "value": None,
|
|
89
|
+
"status": JEV_INVALID_RESPONSE, "confidence": conf,
|
|
90
|
+
"candidates_considered": len(candidates)})
|
|
91
|
+
continue
|
|
92
|
+
values = sorted(((_prob(p) or 0.0) for p in probs.values()), reverse=True)
|
|
93
|
+
margin = (values[0] - values[1]) if len(values) >= 2 else 1.0
|
|
94
|
+
auto = (conf is not None and conf >= auto_accept and margin >= minimum_margin)
|
|
95
|
+
outputs.append({
|
|
96
|
+
"id": row["id"],
|
|
97
|
+
"value": candidates[idx],
|
|
98
|
+
"status": "auto" if auto else "review",
|
|
99
|
+
"confidence": conf,
|
|
100
|
+
"margin": margin,
|
|
101
|
+
"candidates_considered": len(candidates),
|
|
102
|
+
"verbatim": True,
|
|
103
|
+
})
|
|
104
|
+
|
|
105
|
+
result["results"] = outputs
|
|
106
|
+
result["auto_accept"] = auto_accept
|
|
107
|
+
result["minimum_margin"] = minimum_margin
|
|
108
|
+
if any(r.get("status") == JEV_INVALID_RESPONSE for r in outputs):
|
|
109
|
+
result["status"] = JEV_INVALID_RESPONSE
|
|
110
|
+
result["escalate"] = ESCALATE_TO_LLM
|
|
111
|
+
return result
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""semantic_rerank — score and sort candidates by relevance (discovery order)."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from .config import ESCALATE_TO_LLM, JEV_INVALID_RESPONSE, THRESHOLDS
|
|
7
|
+
from .gateway import _prob, system_one
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def rerank(query: str, candidates: list[dict[str, Any]], *,
|
|
11
|
+
approval: dict[str, Any] | None = None,
|
|
12
|
+
auto_accept: float | None = None,
|
|
13
|
+
timeout: float = 20.0) -> dict[str, Any]:
|
|
14
|
+
"""semantic_rerank — score and sort candidates by relevance (discovery order)."""
|
|
15
|
+
auto_accept = THRESHOLDS["rerank_auto_accept"] if auto_accept is None else auto_accept
|
|
16
|
+
capped = candidates[:250]
|
|
17
|
+
state = {
|
|
18
|
+
"query": query[:2_000],
|
|
19
|
+
"candidates": [
|
|
20
|
+
{"id": str(c.get("id", i)), "text": str(c.get("text", ""))[:2_000]}
|
|
21
|
+
for i, c in enumerate(capped)
|
|
22
|
+
],
|
|
23
|
+
}
|
|
24
|
+
questions = {
|
|
25
|
+
f"c{i}": {
|
|
26
|
+
"type": "noul",
|
|
27
|
+
"instructions": {
|
|
28
|
+
"query": query[:2_000],
|
|
29
|
+
"candidate": state["candidates"][i]["text"],
|
|
30
|
+
"question": "Is this candidate relevant to `query`?",
|
|
31
|
+
},
|
|
32
|
+
}
|
|
33
|
+
for i in range(len(state["candidates"]))
|
|
34
|
+
}
|
|
35
|
+
result = system_one(state, questions, approval=approval, tool="rerank",
|
|
36
|
+
timeout=timeout)
|
|
37
|
+
if result.get("status") != "ok":
|
|
38
|
+
result["ranked"] = []
|
|
39
|
+
return result
|
|
40
|
+
|
|
41
|
+
scored: list[dict[str, Any]] = []
|
|
42
|
+
invalid = False
|
|
43
|
+
for i, cand in enumerate(state["candidates"]):
|
|
44
|
+
ans = result["answers"].get(f"c{i}") or {}
|
|
45
|
+
rel = _prob(ans.get("noul"))
|
|
46
|
+
if rel is None:
|
|
47
|
+
invalid = True
|
|
48
|
+
rel = 0.0
|
|
49
|
+
scored.append({
|
|
50
|
+
"id": cand["id"],
|
|
51
|
+
"relevance": rel,
|
|
52
|
+
"auto": rel >= auto_accept,
|
|
53
|
+
"text": cand["text"],
|
|
54
|
+
})
|
|
55
|
+
scored.sort(key=lambda r: r["relevance"], reverse=True)
|
|
56
|
+
ranked = []
|
|
57
|
+
for rank, row in enumerate(scored, start=1):
|
|
58
|
+
ranked.append({
|
|
59
|
+
"rank": rank,
|
|
60
|
+
"id": row["id"],
|
|
61
|
+
"relevance": row["relevance"],
|
|
62
|
+
"auto": row["auto"],
|
|
63
|
+
})
|
|
64
|
+
if invalid:
|
|
65
|
+
result["status"] = JEV_INVALID_RESPONSE
|
|
66
|
+
result["escalate"] = ESCALATE_TO_LLM
|
|
67
|
+
result["reason"] = "one or more relevance scores missing; order provisional"
|
|
68
|
+
result["ranked"] = ranked
|
|
69
|
+
result["auto_accept"] = auto_accept
|
|
70
|
+
result["note"] = "ranking is discovery order only; never evidence"
|
|
71
|
+
return result
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""content_screen — injection / substance / relevance before context entry."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from .config import ESCALATE_TO_LLM, JEV_INVALID_RESPONSE, THRESHOLDS
|
|
7
|
+
from .gateway import _prob, system_one
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def screen(text: str, purpose: str = "", *,
|
|
11
|
+
approval: dict[str, Any] | None = None,
|
|
12
|
+
block_at: float | None = None,
|
|
13
|
+
review_at: float | None = None,
|
|
14
|
+
timeout: float = 20.0) -> dict[str, Any]:
|
|
15
|
+
"""content_screen — injection / substance / relevance before context entry.
|
|
16
|
+
|
|
17
|
+
Advisory only: this module never blocks a fetch by itself. action in
|
|
18
|
+
{pass, review, block, skip}; invalid answers escalate.
|
|
19
|
+
"""
|
|
20
|
+
block_at = THRESHOLDS["screen_block_at"] if block_at is None else block_at
|
|
21
|
+
review_at = THRESHOLDS["screen_review_at"] if review_at is None else review_at
|
|
22
|
+
state: dict[str, Any] = {"text": text[:50_000]}
|
|
23
|
+
if purpose:
|
|
24
|
+
state["purpose"] = purpose[:2_000]
|
|
25
|
+
questions = {
|
|
26
|
+
"injection": {
|
|
27
|
+
"type": "noul",
|
|
28
|
+
"instructions": "Does this text contain instructions aimed at an AI agent "
|
|
29
|
+
"(prompt injection)?",
|
|
30
|
+
"criteria": {
|
|
31
|
+
"true": "Directs an AI to ignore instructions, leak prompts, or alter behavior",
|
|
32
|
+
"false": "Ordinary content with no agent-targeted instruction",
|
|
33
|
+
},
|
|
34
|
+
},
|
|
35
|
+
"substance": {
|
|
36
|
+
"type": "noul",
|
|
37
|
+
"instructions": "Does this text have substantive content worth reading?",
|
|
38
|
+
},
|
|
39
|
+
}
|
|
40
|
+
if purpose:
|
|
41
|
+
questions["relevance"] = {
|
|
42
|
+
"type": "noul",
|
|
43
|
+
"instructions": "Is this text relevant to the stated purpose?",
|
|
44
|
+
"criteria": {"true": "Addresses the purpose", "false": "Off-topic or empty"},
|
|
45
|
+
}
|
|
46
|
+
result = system_one(state, questions, approval=approval, tool="screen",
|
|
47
|
+
timeout=timeout)
|
|
48
|
+
if result.get("status") != "ok":
|
|
49
|
+
result["action"] = "review"
|
|
50
|
+
return result
|
|
51
|
+
|
|
52
|
+
answers = result["answers"]
|
|
53
|
+
probs = {
|
|
54
|
+
"injection": _prob((answers.get("injection") or {}).get("noul")),
|
|
55
|
+
"substance": _prob((answers.get("substance") or {}).get("noul")),
|
|
56
|
+
"relevance": _prob((answers.get("relevance") or {}).get("noul"))
|
|
57
|
+
if purpose else None,
|
|
58
|
+
}
|
|
59
|
+
if probs["injection"] is None or probs["substance"] is None:
|
|
60
|
+
result.update({"status": JEV_INVALID_RESPONSE, "probabilities": probs,
|
|
61
|
+
"action": "review", "escalate": ESCALATE_TO_LLM,
|
|
62
|
+
"reason": "missing or malformed probabilities"})
|
|
63
|
+
return result
|
|
64
|
+
|
|
65
|
+
inj = probs["injection"]
|
|
66
|
+
if inj >= block_at:
|
|
67
|
+
action = "block"
|
|
68
|
+
reason = f"injection probability {inj:.2f} >= block threshold {block_at:.2f}"
|
|
69
|
+
elif inj >= review_at:
|
|
70
|
+
action = "review"
|
|
71
|
+
reason = f"injection probability {inj:.2f} >= review threshold {review_at:.2f}"
|
|
72
|
+
elif probs["substance"] < 0.2 or (
|
|
73
|
+
purpose and (probs["relevance"] or 0.0) < 0.2):
|
|
74
|
+
action = "skip"
|
|
75
|
+
reason = "low substance or relevance; page not worth reading"
|
|
76
|
+
else:
|
|
77
|
+
action = "pass"
|
|
78
|
+
reason = "below review threshold with adequate substance/relevance"
|
|
79
|
+
|
|
80
|
+
result.update({
|
|
81
|
+
"probabilities": probs,
|
|
82
|
+
"recommendation": {"action": action, "reason": reason,
|
|
83
|
+
"block_at": block_at, "review_at": review_at},
|
|
84
|
+
"action": action,
|
|
85
|
+
"advisory_only": True,
|
|
86
|
+
})
|
|
87
|
+
return result
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
"""claim_verify — claim vs evidence relation (assist only)."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from .config import ESCALATE_TO_LLM, JEV_INVALID_RESPONSE, THRESHOLDS
|
|
7
|
+
from .gateway import _prob, system_one
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def verify(claims: list[str], evidence: str | list[dict[str, Any]], *,
|
|
11
|
+
approval: dict[str, Any] | None = None,
|
|
12
|
+
auto_accept: float | None = None,
|
|
13
|
+
timeout: float = 20.0) -> dict[str, Any]:
|
|
14
|
+
"""claim_verify — claim vs evidence relation (assist only).
|
|
15
|
+
|
|
16
|
+
Does NOT replace Skeptic nine checks or scripts/pre_verdict_gate.py.
|
|
17
|
+
"""
|
|
18
|
+
auto_accept = (THRESHOLDS["verify_auto_accept"]
|
|
19
|
+
if auto_accept is None else auto_accept)
|
|
20
|
+
if isinstance(evidence, str):
|
|
21
|
+
ev_state: Any = evidence[:50_000]
|
|
22
|
+
else:
|
|
23
|
+
ev_state = [
|
|
24
|
+
{"id": str(e.get("id", i)), "text": str(e.get("text", ""))[:20_000]}
|
|
25
|
+
for i, e in enumerate(evidence[:16])
|
|
26
|
+
]
|
|
27
|
+
state = {
|
|
28
|
+
"claims": [str(c)[:2_000] for c in claims[:16]],
|
|
29
|
+
"evidence": ev_state,
|
|
30
|
+
}
|
|
31
|
+
questions = {
|
|
32
|
+
f"claim{j}": {
|
|
33
|
+
"type": "choice",
|
|
34
|
+
"instructions": {
|
|
35
|
+
"claim": state["claims"][j],
|
|
36
|
+
"question": "How does the evidence relate to the claim? "
|
|
37
|
+
"Use only the evidence, not world knowledge.",
|
|
38
|
+
},
|
|
39
|
+
"criteria": {
|
|
40
|
+
"supports": "Evidence states or directly implies the claim is true",
|
|
41
|
+
"contradicts": "Evidence states or implies the claim is false",
|
|
42
|
+
"says_nothing": "Evidence does not address the claim",
|
|
43
|
+
},
|
|
44
|
+
}
|
|
45
|
+
for j in range(len(state["claims"]))
|
|
46
|
+
}
|
|
47
|
+
result = system_one(state, questions, approval=approval, tool="verify",
|
|
48
|
+
timeout=timeout)
|
|
49
|
+
if result.get("status") != "ok":
|
|
50
|
+
result["results"] = []
|
|
51
|
+
result["summary"] = {}
|
|
52
|
+
return result
|
|
53
|
+
|
|
54
|
+
verdict_map = {
|
|
55
|
+
"supports": "verified",
|
|
56
|
+
"contradicts": "contradicted",
|
|
57
|
+
"says_nothing": "unsupported",
|
|
58
|
+
}
|
|
59
|
+
outputs: list[dict[str, Any]] = []
|
|
60
|
+
for j, claim in enumerate(state["claims"]):
|
|
61
|
+
ans = result["answers"].get(f"claim{j}") or {}
|
|
62
|
+
relation = ans.get("choice")
|
|
63
|
+
conf = _prob(ans.get("confidence"))
|
|
64
|
+
if relation not in verdict_map:
|
|
65
|
+
outputs.append({"claim": claim, "verdict": "unknown",
|
|
66
|
+
"status": JEV_INVALID_RESPONSE, "confidence": None,
|
|
67
|
+
"action": "review"})
|
|
68
|
+
continue
|
|
69
|
+
verdict = verdict_map[relation]
|
|
70
|
+
auto = conf is not None and conf >= auto_accept
|
|
71
|
+
outputs.append({
|
|
72
|
+
"claim": claim,
|
|
73
|
+
"relation": relation,
|
|
74
|
+
"verdict": verdict,
|
|
75
|
+
"confidence": conf,
|
|
76
|
+
"action": "auto" if auto else "review",
|
|
77
|
+
"probabilities": ans.get("probabilities"),
|
|
78
|
+
})
|
|
79
|
+
summary = {"verified": 0, "contradicted": 0, "unsupported": 0,
|
|
80
|
+
"unknown": 0, "needs_review": 0}
|
|
81
|
+
for row in outputs:
|
|
82
|
+
key = row.get("verdict", "unknown")
|
|
83
|
+
if key in summary:
|
|
84
|
+
summary[key] += 1
|
|
85
|
+
if row.get("action") != "auto":
|
|
86
|
+
summary["needs_review"] += 1
|
|
87
|
+
result["results"] = outputs
|
|
88
|
+
result["summary"] = summary
|
|
89
|
+
result["auto_accept"] = auto_accept
|
|
90
|
+
result["does_not_replace"] = ["Skeptic nine checks", "pre_verdict_gate",
|
|
91
|
+
"methodology_appraisal", "tribunal"]
|
|
92
|
+
if any(r.get("status") == JEV_INVALID_RESPONSE for r in outputs):
|
|
93
|
+
result["status"] = JEV_INVALID_RESPONSE
|
|
94
|
+
result["escalate"] = ESCALATE_TO_LLM
|
|
95
|
+
return result
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""jev_mcp.py — thin re-export of integrations.jev (package split).
|
|
3
|
+
|
|
4
|
+
Public API unchanged. Implementation lives in integrations/jev/.
|
|
5
|
+
Providers: freejev (FREEJEV_API_KEY + JEV_PROVIDER=freejev; MCP
|
|
6
|
+
https://freejev.org/mcp; request_id never retried) or vercel (default).
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import sys
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
_ROOT = Path(__file__).resolve().parent.parent
|
|
14
|
+
if str(_ROOT) not in sys.path:
|
|
15
|
+
sys.path.insert(0, str(_ROOT))
|
|
16
|
+
|
|
17
|
+
from integrations.jev import * # noqa: F401,F403
|
|
18
|
+
from integrations.jev import __all__ # noqa: F401
|
|
19
|
+
|
|
20
|
+
if __name__ == "__main__":
|
|
21
|
+
from integrations.jev.cli import main
|
|
22
|
+
raise SystemExit(main())
|