eduevidence 6.2.0 → 6.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +395 -0
- package/README.md +22 -13
- package/README.zh-CN.md +15 -8
- package/SKILL.md +10 -9
- package/benchmarks/evidence-library.json +277 -1
- package/docs/architecture.md +6 -3
- package/docs/j-ev-experimental.md +250 -0
- package/docs/reproducibility.md +138 -0
- package/domains/_neutral/copy/few_shots.json +21 -0
- package/domains/_neutral/copy/framing_lexicon.json +19 -0
- package/domains/_neutral/copy/module_labels.json +5 -0
- package/domains/_neutral/copy/module_labels_footer.json +102 -0
- package/domains/_neutral/copy/module_labels_modules.json +204 -0
- package/domains/_neutral/copy/module_labels_nav.json +126 -0
- package/domains/_neutral/copy/module_labels_summary.json +98 -0
- package/domains/_neutral/copy/module_labels_tables.json +164 -0
- package/domains/_neutral/copy/module_labels_v2.json +90 -0
- package/domains/_neutral/copy/risk_constructs.json +20 -0
- package/domains/_neutral/copy/section_titles.json +66 -0
- package/domains/_neutral/copy/terminology.json +11 -0
- package/domains/check_copy_packs.py +103 -0
- package/domains/education/copy/few_shots.json +22 -0
- package/domains/education/copy/framing_enums.json +167 -0
- package/domains/education/copy/framing_lexicon.json +166 -0
- package/domains/education/copy/module_labels.json +169 -0
- package/domains/education/copy/risk_constructs.json +48 -0
- package/domains/education/copy/section_titles.json +186 -0
- package/domains/education/copy/terminology.json +70 -0
- package/domains/education/manifest.json +1 -1
- package/domains/education/outcome_taxonomy.json +2 -2
- package/domains/manifest.json +1 -1
- package/domains/policy/copy/few_shots.json +22 -0
- package/domains/policy/copy/framing_enums.json +94 -0
- package/domains/policy/copy/framing_lexicon.json +174 -0
- package/domains/policy/copy/module_labels.json +168 -0
- package/domains/policy/copy/risk_constructs.json +33 -0
- package/domains/policy/copy/section_titles.json +186 -0
- package/domains/policy/copy/terminology.json +64 -0
- package/engine/capabilities.py +57 -5
- package/engine/decision_policy.py +88 -17
- package/engine/library_builtin.py +7 -4
- package/engine/tribunal.py +17 -23
- package/engine/versions.py +1 -1
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +4 -4
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +4 -4
- package/examples/spaced-retrieval-practice/EduEvidence_Report.html +2728 -0
- package/examples/spaced-retrieval-practice/report.html +2522 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +4 -4
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +4 -4
- package/examples/workplace-ai-assistant/EduEvidence_Report.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +36 -36
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +36 -36
- package/integrations/jev/__init__.py +115 -0
- package/integrations/jev/approval.py +212 -0
- package/integrations/jev/cli.py +84 -0
- package/integrations/jev/config.py +112 -0
- package/integrations/jev/gateway.py +128 -0
- package/integrations/jev/modes.py +38 -0
- package/integrations/jev/tools_classify.py +88 -0
- package/integrations/jev/tools_extract.py +111 -0
- package/integrations/jev/tools_rerank.py +71 -0
- package/integrations/jev/tools_screen.py +87 -0
- package/integrations/jev/tools_verify.py +95 -0
- package/integrations/jev_mcp.py +22 -0
- package/integrations/semantic_decide.py +286 -0
- package/integrations/semdecide_cli.py +55 -0
- package/package.json +9 -1
- package/pyproject.toml +1 -1
- package/references/report-copy-style.md +43 -3
- package/schemas/v2/decision-snapshot.schema.json +20 -9
- package/schemas/v2/intake.schema.json +191 -0
- package/scripts/build_evidence_library.py +15 -5
- package/scripts/dashboard_server.py +13 -2
- package/scripts/intake/__init__.py +31 -0
- package/scripts/intake/__main__.py +18 -0
- package/scripts/intake/background.py +78 -0
- package/scripts/intake/browser.py +79 -0
- package/scripts/intake/cli.py +57 -0
- package/scripts/intake/constants.py +57 -0
- package/scripts/intake/depth.py +53 -0
- package/scripts/intake/enhancements.py +106 -0
- package/scripts/intake/hooks.py +90 -0
- package/scripts/intake/prefs.py +76 -0
- package/scripts/intake/prompts.py +85 -0
- package/scripts/intake/session.py +152 -0
- package/scripts/lint_file_layers.py +126 -0
- package/scripts/orchestrator.py +68 -17
- package/scripts/pre_verdict_gate.py +21 -7
- package/scripts/skill_lint.py +11 -1
- package/scripts/skill_payload.py +3 -3
- package/scripts/test_adversarial_empirical.py +70 -6
- package/skill/agents/evidence-judge.md +49 -7
- package/skill/workflows/experimental-jev.md +170 -0
- package/skill/workflows/intake.md +120 -0
- package/visualization/eduevidence-report/scripts/build_infographics.py +32 -14
- package/visualization/eduevidence-report/scripts/build_report.py +75 -662
- package/visualization/eduevidence-report/scripts/report_copy_pack.py +296 -0
- package/visualization/eduevidence-report/scripts/report_copy_policy_guard.py +47 -0
- package/visualization/eduevidence-report/scripts/zh_labels.py +61 -0
- package/scripts/build_esl_artifacts.py +0 -1921
- package/scripts/build_killer_demo.py +0 -295
- package/scripts/enrich_projects_human_and_lieflat.py +0 -315
- package/scripts/generate_new_projects.py +0 -686
- package/scripts/sync_killer_demo_report.py +0 -270
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
"""integrations.jev — Jev (TypeSafe System One) experimental Tier-0 package.
|
|
2
|
+
|
|
3
|
+
Experimental only (`--experimental` / mode 1 or 3). Never a scientific gate
|
|
4
|
+
replacement. Public API is re-exported here and via integrations/jev_mcp.py.
|
|
5
|
+
|
|
6
|
+
Providers: freejev (FREEJEV_API_KEY + JEV_PROVIDER=freejev, MCP
|
|
7
|
+
https://freejev.org/mcp, request_id never retried) or vercel (default).
|
|
8
|
+
Fail-closed: invalid / provider errors escalate to the large model.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from .approval import (
|
|
15
|
+
build_approval_record,
|
|
16
|
+
detect_jev_mcp,
|
|
17
|
+
is_approval_current,
|
|
18
|
+
load_approval,
|
|
19
|
+
require_jev_mcp,
|
|
20
|
+
write_approval,
|
|
21
|
+
)
|
|
22
|
+
from .cli import npx_mcp_guide
|
|
23
|
+
from .config import (
|
|
24
|
+
DEFAULT_MODEL,
|
|
25
|
+
ESCALATE_TO_LLM,
|
|
26
|
+
JEV_APPROVAL_PATH,
|
|
27
|
+
JEV_APPROVAL_REQUIRED,
|
|
28
|
+
JEV_ENV_FILE,
|
|
29
|
+
JEV_INVALID_RESPONSE,
|
|
30
|
+
JEV_PROVIDER_ERROR,
|
|
31
|
+
JEV_UNAVAILABLE,
|
|
32
|
+
MCP_JEV_TOOLS,
|
|
33
|
+
MODE_HYBRID,
|
|
34
|
+
MODE_JEV_MCP,
|
|
35
|
+
MODE_SEMDECIDE,
|
|
36
|
+
MODE_STANDARD,
|
|
37
|
+
NPM_JEV_MCP,
|
|
38
|
+
THRESHOLDS,
|
|
39
|
+
TIER0_TOOLS,
|
|
40
|
+
VERCEL_TYPE_SAFE_BASE,
|
|
41
|
+
api_key,
|
|
42
|
+
model_id,
|
|
43
|
+
provider,
|
|
44
|
+
system_one_url,
|
|
45
|
+
)
|
|
46
|
+
from .gateway import system_one
|
|
47
|
+
from .modes import resolve_experimental_mode
|
|
48
|
+
from .tools_classify import classify
|
|
49
|
+
from .tools_extract import extract
|
|
50
|
+
from .tools_rerank import rerank
|
|
51
|
+
from .tools_screen import screen
|
|
52
|
+
from .tools_verify import verify
|
|
53
|
+
|
|
54
|
+
# Map capability_id -> wrapper (engine/capabilities.py experimental set).
|
|
55
|
+
CAPABILITY_TO_TOOL = {
|
|
56
|
+
"content_screen": screen,
|
|
57
|
+
"semantic_rerank": rerank,
|
|
58
|
+
"field_extract": extract,
|
|
59
|
+
"classify_check": classify,
|
|
60
|
+
"claim_verify": verify,
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def safe_call(capability_id: str, *args: Any,
|
|
65
|
+
approval: dict[str, Any] | None = None,
|
|
66
|
+
**kwargs: Any) -> dict[str, Any]:
|
|
67
|
+
"""Capability-routed Tier-0 call with the approval gate applied."""
|
|
68
|
+
fn = CAPABILITY_TO_TOOL.get(capability_id)
|
|
69
|
+
if fn is None:
|
|
70
|
+
return {"status": JEV_INVALID_RESPONSE,
|
|
71
|
+
"error": f"unknown capability {capability_id!r}",
|
|
72
|
+
"escalate": ESCALATE_TO_LLM}
|
|
73
|
+
kwargs.setdefault("approval", approval)
|
|
74
|
+
return fn(*args, **kwargs)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
__all__ = [
|
|
78
|
+
"CAPABILITY_TO_TOOL",
|
|
79
|
+
"DEFAULT_MODEL",
|
|
80
|
+
"ESCALATE_TO_LLM",
|
|
81
|
+
"JEV_APPROVAL_PATH",
|
|
82
|
+
"JEV_APPROVAL_REQUIRED",
|
|
83
|
+
"JEV_ENV_FILE",
|
|
84
|
+
"JEV_INVALID_RESPONSE",
|
|
85
|
+
"JEV_PROVIDER_ERROR",
|
|
86
|
+
"JEV_UNAVAILABLE",
|
|
87
|
+
"MCP_JEV_TOOLS",
|
|
88
|
+
"MODE_HYBRID",
|
|
89
|
+
"MODE_JEV_MCP",
|
|
90
|
+
"MODE_SEMDECIDE",
|
|
91
|
+
"MODE_STANDARD",
|
|
92
|
+
"NPM_JEV_MCP",
|
|
93
|
+
"THRESHOLDS",
|
|
94
|
+
"TIER0_TOOLS",
|
|
95
|
+
"VERCEL_TYPE_SAFE_BASE",
|
|
96
|
+
"api_key",
|
|
97
|
+
"build_approval_record",
|
|
98
|
+
"classify",
|
|
99
|
+
"detect_jev_mcp",
|
|
100
|
+
"extract",
|
|
101
|
+
"is_approval_current",
|
|
102
|
+
"load_approval",
|
|
103
|
+
"model_id",
|
|
104
|
+
"npx_mcp_guide",
|
|
105
|
+
"provider",
|
|
106
|
+
"require_jev_mcp",
|
|
107
|
+
"rerank",
|
|
108
|
+
"resolve_experimental_mode",
|
|
109
|
+
"safe_call",
|
|
110
|
+
"screen",
|
|
111
|
+
"system_one",
|
|
112
|
+
"system_one_url",
|
|
113
|
+
"verify",
|
|
114
|
+
"write_approval",
|
|
115
|
+
]
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
"""Availability detect + user approval gate (~/.eduevidence/jev_mcp_approval.json)."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import hashlib
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
from datetime import datetime, timezone
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
from .config import (
|
|
12
|
+
JEV_APPROVAL_PATH,
|
|
13
|
+
JEV_APPROVAL_REQUIRED,
|
|
14
|
+
JEV_UNAVAILABLE,
|
|
15
|
+
MCP_JEV_TOOLS,
|
|
16
|
+
MODE_JEV_MCP,
|
|
17
|
+
NPM_JEV_MCP,
|
|
18
|
+
TIER0_TOOLS,
|
|
19
|
+
THRESHOLDS,
|
|
20
|
+
api_key,
|
|
21
|
+
model_id,
|
|
22
|
+
provider,
|
|
23
|
+
system_one_url,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def detect_jev_mcp() -> dict[str, Any]:
|
|
28
|
+
"""Probe Jev gateway credentials and (optional) local MCP binary.
|
|
29
|
+
|
|
30
|
+
Providers: freejev (FREEJEV_API_KEY + JEV_PROVIDER=freejev; MCP
|
|
31
|
+
https://freejev.org/mcp; request_id never retried) or vercel (default).
|
|
32
|
+
|
|
33
|
+
States:
|
|
34
|
+
- available: provider key present and provider is supported
|
|
35
|
+
- mcp_documented: key missing but npx/node present (documented path only)
|
|
36
|
+
- unavailable: neither usable transport
|
|
37
|
+
"""
|
|
38
|
+
key = api_key()
|
|
39
|
+
prov = provider()
|
|
40
|
+
reasons: list[str] = []
|
|
41
|
+
has_npx = any(
|
|
42
|
+
os.path.exists(os.path.join(d, "npx"))
|
|
43
|
+
for d in os.environ.get("PATH", "").split(os.pathsep) if d
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
if not key:
|
|
47
|
+
reasons.append("AI_GATEWAY_API_KEY not set in env or ~/.eduevidence/env")
|
|
48
|
+
if prov not in ("vercel", "typesafe", "compatible", "openrouter", "cloudflare",
|
|
49
|
+
"freejev"):
|
|
50
|
+
reasons.append(f"unsupported JEV_PROVIDER={prov!r}")
|
|
51
|
+
|
|
52
|
+
if key and prov in ("vercel", "typesafe", "compatible", "openrouter", "cloudflare",
|
|
53
|
+
"freejev"):
|
|
54
|
+
state = "available"
|
|
55
|
+
available = True
|
|
56
|
+
mode = "jev_gateway"
|
|
57
|
+
hint = ""
|
|
58
|
+
elif has_npx:
|
|
59
|
+
state = "mcp_documented"
|
|
60
|
+
available = False
|
|
61
|
+
mode = "mcp_documented"
|
|
62
|
+
hint = (f"Gateway key missing — register MCP server via `{NPM_JEV_MCP}` "
|
|
63
|
+
"in the host agent, or set AI_GATEWAY_API_KEY + JEV_PROVIDER=vercel "
|
|
64
|
+
"in ~/.eduevidence/env")
|
|
65
|
+
else:
|
|
66
|
+
state = "unavailable"
|
|
67
|
+
available = False
|
|
68
|
+
mode = "none"
|
|
69
|
+
hint = "Install Node 20+ for MCP path, or configure Vercel AI Gateway key"
|
|
70
|
+
|
|
71
|
+
return {
|
|
72
|
+
"available": available,
|
|
73
|
+
"state": state,
|
|
74
|
+
"mode": mode,
|
|
75
|
+
"provider": prov if key else None,
|
|
76
|
+
"model": model_id() if key else None,
|
|
77
|
+
"endpoint": system_one_url() if key else None,
|
|
78
|
+
"reason": "; ".join(reasons) or "gateway credentials present",
|
|
79
|
+
"reasons": reasons,
|
|
80
|
+
"hint": hint,
|
|
81
|
+
"tier0_tools": list(TIER0_TOOLS),
|
|
82
|
+
"mcp_alternate": {
|
|
83
|
+
"command": NPM_JEV_MCP,
|
|
84
|
+
"tools": dict(MCP_JEV_TOOLS),
|
|
85
|
+
"env": ["TYPESAFE_API_KEY or AI_GATEWAY_API_KEY", "JEV_PROVIDER"],
|
|
86
|
+
},
|
|
87
|
+
"thresholds": dict(THRESHOLDS),
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def require_jev_mcp() -> dict[str, Any]:
|
|
92
|
+
report = detect_jev_mcp()
|
|
93
|
+
if not report["available"]:
|
|
94
|
+
raise RuntimeError(JEV_UNAVAILABLE)
|
|
95
|
+
return report
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _tools_hash(tools: tuple[str, ...] | list[str]) -> str:
|
|
99
|
+
payload = json.dumps(sorted(tools), ensure_ascii=False, separators=(",", ":"))
|
|
100
|
+
return hashlib.sha256(payload.encode("utf-8")).hexdigest()
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def build_approval_record(
|
|
104
|
+
tools: list[str] | tuple[str, ...] = TIER0_TOOLS,
|
|
105
|
+
*,
|
|
106
|
+
provider_name: str | None = None,
|
|
107
|
+
model: str | None = None,
|
|
108
|
+
mode: int = MODE_JEV_MCP,
|
|
109
|
+
) -> dict[str, Any]:
|
|
110
|
+
"""Build the approval record the user confirms before any network call."""
|
|
111
|
+
selected = tuple(sorted({str(t) for t in tools if str(t) in TIER0_TOOLS}))
|
|
112
|
+
return {
|
|
113
|
+
"approved": True,
|
|
114
|
+
"approved_at": datetime.now(timezone.utc).isoformat(),
|
|
115
|
+
"mode": int(mode),
|
|
116
|
+
"provider": provider_name or provider(),
|
|
117
|
+
"model": model or model_id(),
|
|
118
|
+
"tools": list(selected),
|
|
119
|
+
"tools_hash": _tools_hash(selected),
|
|
120
|
+
"endpoint": system_one_url(),
|
|
121
|
+
"schema_version": 1,
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def write_approval(path: str | Path | None = None, **kwargs: Any) -> dict[str, Any]:
|
|
126
|
+
record = build_approval_record(**kwargs)
|
|
127
|
+
out = Path(os.path.expanduser(path or JEV_APPROVAL_PATH))
|
|
128
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
129
|
+
out.write_text(json.dumps(record, ensure_ascii=False, indent=2) + "\n",
|
|
130
|
+
encoding="utf-8")
|
|
131
|
+
return record
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def load_approval(path: str | Path | None = None) -> dict[str, Any] | None:
|
|
135
|
+
try:
|
|
136
|
+
return json.loads(
|
|
137
|
+
Path(os.path.expanduser(path or JEV_APPROVAL_PATH)).read_text(encoding="utf-8"))
|
|
138
|
+
except (OSError, json.JSONDecodeError):
|
|
139
|
+
return None
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def is_approval_current(
|
|
143
|
+
approval: dict[str, Any] | None,
|
|
144
|
+
tools: list[str] | tuple[str, ...] = TIER0_TOOLS,
|
|
145
|
+
*,
|
|
146
|
+
provider_name: str | None = None,
|
|
147
|
+
model: str | None = None,
|
|
148
|
+
mode: int | None = None,
|
|
149
|
+
path: str | Path | None = None,
|
|
150
|
+
) -> tuple[bool, list[str]]:
|
|
151
|
+
"""True when the approval still matches the current proposal.
|
|
152
|
+
|
|
153
|
+
``path`` names the approval file to re-read from disk; ``None`` means the
|
|
154
|
+
configured one (``~/.eduevidence/jev_mcp_approval.json`` or
|
|
155
|
+
``JEV_APPROVAL_PATH``). An explicitly empty/missing path is never treated
|
|
156
|
+
as "current": the disk check below still fails closed, so an in-memory
|
|
157
|
+
record cannot pass because the file could not be read.
|
|
158
|
+
"""
|
|
159
|
+
changes: list[str] = []
|
|
160
|
+
if not approval or not approval.get("approved"):
|
|
161
|
+
changes.append("approval missing or not approved")
|
|
162
|
+
return False, changes
|
|
163
|
+
stored = tuple(sorted(approval.get("tools") or []))
|
|
164
|
+
if approval.get("tools_hash") != _tools_hash(stored):
|
|
165
|
+
changes.append("approval file tampered (tools_hash mismatch)")
|
|
166
|
+
wanted = tuple(sorted({str(t) for t in tools if str(t) in TIER0_TOOLS}))
|
|
167
|
+
if wanted != stored:
|
|
168
|
+
changes.append("tool set changed")
|
|
169
|
+
if provider_name is not None and approval.get("provider") != provider_name:
|
|
170
|
+
changes.append("provider changed")
|
|
171
|
+
elif provider_name is None and approval.get("provider") != provider():
|
|
172
|
+
changes.append("provider changed")
|
|
173
|
+
if model is not None and approval.get("model") != model:
|
|
174
|
+
changes.append("model changed")
|
|
175
|
+
elif model is None and approval.get("model") != model_id():
|
|
176
|
+
changes.append("model changed")
|
|
177
|
+
if mode is not None and int(approval.get("mode", -1)) != int(mode):
|
|
178
|
+
changes.append("experimental mode changed")
|
|
179
|
+
disk = load_approval(path)
|
|
180
|
+
if not disk:
|
|
181
|
+
changes.append("approval file missing on disk")
|
|
182
|
+
elif disk.get("tools_hash") != approval.get("tools_hash"):
|
|
183
|
+
changes.append("approval file tampered (disk tools_hash mismatch)")
|
|
184
|
+
return (not changes), changes
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _check_approval(approval: dict[str, Any] | None, tool: str) -> str | None:
|
|
188
|
+
"""None when allowed, else JEV_APPROVAL_REQUIRED. Fail closed.
|
|
189
|
+
|
|
190
|
+
Accepts only records that still match the on-disk approval file.
|
|
191
|
+
In-memory forged dicts are rejected. Provider/endpoint must match live
|
|
192
|
+
config exactly — no vercel bypass.
|
|
193
|
+
"""
|
|
194
|
+
if not approval or not approval.get("approved"):
|
|
195
|
+
return JEV_APPROVAL_REQUIRED
|
|
196
|
+
stored = tuple(sorted(approval.get("tools") or []))
|
|
197
|
+
if approval.get("tools_hash") != _tools_hash(stored):
|
|
198
|
+
return JEV_APPROVAL_REQUIRED
|
|
199
|
+
if tool not in TIER0_TOOLS or tool not in stored:
|
|
200
|
+
return JEV_APPROVAL_REQUIRED
|
|
201
|
+
disk = load_approval()
|
|
202
|
+
if not disk or not disk.get("approved"):
|
|
203
|
+
return JEV_APPROVAL_REQUIRED
|
|
204
|
+
if disk.get("tools_hash") != approval.get("tools_hash"):
|
|
205
|
+
return JEV_APPROVAL_REQUIRED
|
|
206
|
+
if tuple(sorted(disk.get("tools") or [])) != stored:
|
|
207
|
+
return JEV_APPROVAL_REQUIRED
|
|
208
|
+
if approval.get("provider") != provider():
|
|
209
|
+
return JEV_APPROVAL_REQUIRED
|
|
210
|
+
if approval.get("endpoint") and approval.get("endpoint") != system_one_url():
|
|
211
|
+
return JEV_APPROVAL_REQUIRED
|
|
212
|
+
return None
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""CLI: --experimental / --approve / --mcp-guide (npx @jkudish/jev-mcp path)."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import json
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from .approval import detect_jev_mcp, write_approval
|
|
9
|
+
from .config import (
|
|
10
|
+
MCP_JEV_TOOLS,
|
|
11
|
+
MODE_HYBRID,
|
|
12
|
+
MODE_JEV_MCP,
|
|
13
|
+
MODE_STANDARD,
|
|
14
|
+
NPM_JEV_MCP,
|
|
15
|
+
TIER0_TOOLS,
|
|
16
|
+
)
|
|
17
|
+
from .modes import resolve_experimental_mode
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def npx_mcp_guide() -> dict[str, Any]:
|
|
21
|
+
"""Documented alternate: host-registered @jkudish/jev-mcp MCP server."""
|
|
22
|
+
return {
|
|
23
|
+
"command": NPM_JEV_MCP,
|
|
24
|
+
"requires": ["Node.js >= 20", "TYPESAFE_API_KEY or AI_GATEWAY_API_KEY"],
|
|
25
|
+
"env": {
|
|
26
|
+
"AI_GATEWAY_API_KEY": "set in ~/.eduevidence/env — never commit",
|
|
27
|
+
"JEV_PROVIDER": "vercel | freejev",
|
|
28
|
+
"TYPESAFE_API_KEY": "optional direct TypeSafe key",
|
|
29
|
+
"FREEJEV_API_KEY": "FreeJev key when JEV_PROVIDER=freejev",
|
|
30
|
+
},
|
|
31
|
+
"freejev": {
|
|
32
|
+
"mcp": "https://freejev.org/mcp",
|
|
33
|
+
"decide": "https://freejev.org/api/v1/decide",
|
|
34
|
+
"request_id": "idempotency key — never retry the same request_id",
|
|
35
|
+
},
|
|
36
|
+
"client_snippets": {
|
|
37
|
+
"claude_code": "claude mcp add jev -- npx -y @jkudish/jev-mcp",
|
|
38
|
+
"generic_json": {
|
|
39
|
+
"mcpServers": {
|
|
40
|
+
"jev": {
|
|
41
|
+
"command": "npx",
|
|
42
|
+
"args": ["-y", "@jkudish/jev-mcp"],
|
|
43
|
+
"env": {"TYPESAFE_API_KEY": "<from ~/.eduevidence/env>"},
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
},
|
|
47
|
+
},
|
|
48
|
+
"tools": dict(MCP_JEV_TOOLS),
|
|
49
|
+
"note": "MCP path is host-side; this module only documents it and does "
|
|
50
|
+
"not spawn or migrate jev-mcp implementation.",
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def main(argv: list[str] | None = None) -> int:
|
|
55
|
+
parser = argparse.ArgumentParser(
|
|
56
|
+
description="Jev experimental detect / approve / Tier-0 probe")
|
|
57
|
+
parser.add_argument("--experimental", nargs="?", const="1", default=None,
|
|
58
|
+
metavar="MODE",
|
|
59
|
+
help="enable experimental mode 0|1|2|3 (default 1) and "
|
|
60
|
+
"print the resolved plan")
|
|
61
|
+
parser.add_argument("--approve", action="store_true",
|
|
62
|
+
help="write ~/.eduevidence/jev_mcp_approval.json for "
|
|
63
|
+
"the current provider/model/tools")
|
|
64
|
+
parser.add_argument("--mcp-guide", action="store_true",
|
|
65
|
+
help="print the documented npx @jkudish/jev-mcp path")
|
|
66
|
+
args = parser.parse_args(argv)
|
|
67
|
+
|
|
68
|
+
out: dict[str, Any] = {"detect": detect_jev_mcp()}
|
|
69
|
+
mode = resolve_experimental_mode(args.experimental)
|
|
70
|
+
if args.experimental is not None:
|
|
71
|
+
out["experimental"] = {
|
|
72
|
+
"mode": mode,
|
|
73
|
+
"mode_name": {0: "standard", 1: "jev-mcp", 2: "semdecide",
|
|
74
|
+
3: "hybrid"}.get(mode, str(mode)),
|
|
75
|
+
"enabled": mode != MODE_STANDARD,
|
|
76
|
+
"tier0_tools": list(TIER0_TOOLS) if mode in (MODE_JEV_MCP, MODE_HYBRID) else [],
|
|
77
|
+
"fail_closed": "uncertain / invalid / provider errors escalate to large model",
|
|
78
|
+
}
|
|
79
|
+
if args.approve:
|
|
80
|
+
out["approval"] = write_approval()
|
|
81
|
+
if args.mcp_guide:
|
|
82
|
+
out["mcp_alternate"] = npx_mcp_guide()
|
|
83
|
+
print(json.dumps(out, ensure_ascii=False, indent=2))
|
|
84
|
+
return 0
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""Jev env / paths / constants (real env > ~/.eduevidence/env > default).
|
|
2
|
+
|
|
3
|
+
Providers (Mode 1 / 3):
|
|
4
|
+
- freejev: FREEJEV_API_KEY + JEV_PROVIDER=freejev
|
|
5
|
+
Decide: https://freejev.org/api/v1/decide
|
|
6
|
+
MCP (documented only): https://freejev.org/mcp
|
|
7
|
+
request_id is an idempotency key — NEVER retry the same request_id;
|
|
8
|
+
fail-closed escalate this call (new ask => new request_id).
|
|
9
|
+
- vercel (default): AI_GATEWAY_API_KEY or TYPESAFE_API_KEY + JEV_PROVIDER=vercel
|
|
10
|
+
Decide: https://ai-gateway.vercel.sh/typesafe/v1/systemone
|
|
11
|
+
MCP (documented only): npx -y @jkudish/jev-mcp
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import os
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
JEV_ENV_FILE = os.environ.get("JEV_ENV_FILE", "~/.eduevidence/env")
|
|
19
|
+
JEV_APPROVAL_PATH = os.environ.get(
|
|
20
|
+
"JEV_APPROVAL_FILE", "~/.eduevidence/jev_mcp_approval.json")
|
|
21
|
+
|
|
22
|
+
JEV_UNAVAILABLE = "JEV_UNAVAILABLE"
|
|
23
|
+
JEV_APPROVAL_REQUIRED = "JEV_APPROVAL_REQUIRED"
|
|
24
|
+
JEV_INVALID_RESPONSE = "JEV_INVALID_RESPONSE"
|
|
25
|
+
JEV_PROVIDER_ERROR = "JEV_PROVIDER_ERROR"
|
|
26
|
+
ESCALATE_TO_LLM = "ESCALATE_TO_LLM"
|
|
27
|
+
|
|
28
|
+
MODE_STANDARD = 0
|
|
29
|
+
MODE_JEV_MCP = 1
|
|
30
|
+
MODE_SEMDECIDE = 2
|
|
31
|
+
MODE_HYBRID = 3
|
|
32
|
+
|
|
33
|
+
TIER0_TOOLS = ("screen", "rerank", "extract", "classify", "verify")
|
|
34
|
+
|
|
35
|
+
VERCEL_TYPE_SAFE_BASE = "https://ai-gateway.vercel.sh/typesafe"
|
|
36
|
+
FREEJEV_DECIDE_URL = "https://freejev.org/api/v1/decide"
|
|
37
|
+
FREEJEV_MCP_URL = "https://freejev.org/mcp"
|
|
38
|
+
DEFAULT_MODEL = "typesafe-ai/jev"
|
|
39
|
+
|
|
40
|
+
NPM_JEV_MCP = "npx -y @jkudish/jev-mcp"
|
|
41
|
+
MCP_JEV_TOOLS = {
|
|
42
|
+
"screen": "jev_screen",
|
|
43
|
+
"rerank": "jev_rerank",
|
|
44
|
+
"extract": "jev_extract",
|
|
45
|
+
"classify": "jev_classify",
|
|
46
|
+
"verify": "jev_verify",
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
THRESHOLDS = {
|
|
50
|
+
"screen_block_at": 0.75,
|
|
51
|
+
"screen_review_at": 0.25,
|
|
52
|
+
"verify_auto_accept": 0.80,
|
|
53
|
+
"classify_auto_accept": 0.85,
|
|
54
|
+
"classify_minimum_margin": 0.50,
|
|
55
|
+
"extract_auto_accept": 0.80,
|
|
56
|
+
"extract_minimum_margin": 0.40,
|
|
57
|
+
"rerank_auto_accept": 0.70,
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _env_file_values(path: str | Path | None = None) -> dict[str, str]:
|
|
62
|
+
"""Parse KEY=VALUE lines from ~/.eduevidence/env (best-effort)."""
|
|
63
|
+
try:
|
|
64
|
+
text = Path(os.path.expanduser(path or JEV_ENV_FILE)).read_text(encoding="utf-8")
|
|
65
|
+
except OSError:
|
|
66
|
+
return {}
|
|
67
|
+
values: dict[str, str] = {}
|
|
68
|
+
for line in text.splitlines():
|
|
69
|
+
line = line.strip()
|
|
70
|
+
if not line or line.startswith("#") or "=" not in line:
|
|
71
|
+
continue
|
|
72
|
+
if line.startswith("export "):
|
|
73
|
+
line = line[len("export "):]
|
|
74
|
+
key, _, value = line.partition("=")
|
|
75
|
+
values[key.strip()] = value.strip().strip("\"'")
|
|
76
|
+
return values
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
_ENV_FILE_VALUES = _env_file_values()
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _effective_env(key: str, default: str = "") -> str:
|
|
83
|
+
return os.environ.get(key) or _ENV_FILE_VALUES.get(key) or default
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def provider() -> str:
|
|
87
|
+
explicit = (_effective_env("JEV_PROVIDER") or "").lower()
|
|
88
|
+
if explicit:
|
|
89
|
+
return explicit
|
|
90
|
+
if _effective_env("FREEJEV_API_KEY") and not (
|
|
91
|
+
_effective_env("AI_GATEWAY_API_KEY")
|
|
92
|
+
or _effective_env("TYPESAFE_API_KEY")):
|
|
93
|
+
return "freejev"
|
|
94
|
+
return "vercel"
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def api_key() -> str:
|
|
98
|
+
"""Gateway / provider key. Never log or commit this value."""
|
|
99
|
+
return (_effective_env("AI_GATEWAY_API_KEY")
|
|
100
|
+
or _effective_env("TYPESAFE_API_KEY")
|
|
101
|
+
or _effective_env("FREEJEV_API_KEY"))
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def model_id() -> str:
|
|
105
|
+
return _effective_env("JEV_MCP_MODEL", DEFAULT_MODEL) or DEFAULT_MODEL
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def system_one_url() -> str:
|
|
109
|
+
if provider() == "freejev":
|
|
110
|
+
return FREEJEV_DECIDE_URL
|
|
111
|
+
base = _effective_env("JEV_TYPE_SAFE_BASE", VERCEL_TYPE_SAFE_BASE).rstrip("/")
|
|
112
|
+
return f"{base}/v1/systemone"
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""System One transport (stdlib urllib only) + provider error mapping."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
import urllib.error
|
|
6
|
+
import urllib.request
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from .approval import _check_approval, detect_jev_mcp, load_approval
|
|
10
|
+
from .config import (
|
|
11
|
+
ESCALATE_TO_LLM,
|
|
12
|
+
JEV_APPROVAL_REQUIRED,
|
|
13
|
+
JEV_INVALID_RESPONSE,
|
|
14
|
+
JEV_PROVIDER_ERROR,
|
|
15
|
+
JEV_UNAVAILABLE,
|
|
16
|
+
api_key,
|
|
17
|
+
model_id,
|
|
18
|
+
system_one_url,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _http_post_json(url: str, payload: dict[str, Any], *,
|
|
23
|
+
timeout: float = 20.0) -> tuple[int, dict[str, Any] | None, str]:
|
|
24
|
+
"""POST JSON with Bearer key. Returns (status, body|None, error)."""
|
|
25
|
+
key = api_key()
|
|
26
|
+
if not key:
|
|
27
|
+
return 0, None, "AI_GATEWAY_API_KEY missing"
|
|
28
|
+
data = json.dumps(payload, ensure_ascii=False).encode("utf-8")
|
|
29
|
+
req = urllib.request.Request(
|
|
30
|
+
url, data=data, method="POST",
|
|
31
|
+
headers={
|
|
32
|
+
"Authorization": f"Bearer {key}",
|
|
33
|
+
"Content-Type": "application/json",
|
|
34
|
+
"Accept": "application/json",
|
|
35
|
+
},
|
|
36
|
+
)
|
|
37
|
+
try:
|
|
38
|
+
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
39
|
+
raw = resp.read().decode("utf-8", "replace")
|
|
40
|
+
try:
|
|
41
|
+
return resp.status, json.loads(raw), ""
|
|
42
|
+
except json.JSONDecodeError:
|
|
43
|
+
return resp.status, None, "response is not JSON"
|
|
44
|
+
except urllib.error.HTTPError as exc:
|
|
45
|
+
# Never echo Authorization / key fragments.
|
|
46
|
+
detail = ""
|
|
47
|
+
try:
|
|
48
|
+
detail = exc.read().decode("utf-8", "replace")[:400]
|
|
49
|
+
except OSError:
|
|
50
|
+
pass
|
|
51
|
+
return exc.code, None, f"HTTP {exc.code}: {detail}"
|
|
52
|
+
except (urllib.error.URLError, TimeoutError, OSError) as exc:
|
|
53
|
+
return 0, None, f"network error: {type(exc).__name__}"
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _map_provider_error(status: int) -> str:
|
|
57
|
+
"""Map HTTP status to a fail-closed JEV_* status code."""
|
|
58
|
+
return (JEV_PROVIDER_ERROR if status in (401, 403, 429, 529) or status >= 500
|
|
59
|
+
else JEV_INVALID_RESPONSE)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def system_one(state: Any, questions: dict[str, Any], *,
|
|
63
|
+
approval: dict[str, Any] | None,
|
|
64
|
+
tool: str | None = None,
|
|
65
|
+
timeout: float = 20.0) -> dict[str, Any]:
|
|
66
|
+
"""UNIFIED network entry. Business code MUST NOT POST directly.
|
|
67
|
+
|
|
68
|
+
Gate: approval exists -> tool approved -> credentials present -> POST.
|
|
69
|
+
Fail-closed: every transport / envelope problem returns a structured
|
|
70
|
+
error that callers must escalate to the large model (never invent answers).
|
|
71
|
+
|
|
72
|
+
FreeJev: any request_id is a one-shot idempotency key — never retry the
|
|
73
|
+
same request_id after timeout / 5xx / unknown outcome. Escalate instead;
|
|
74
|
+
a deliberate re-ask must mint a new request_id.
|
|
75
|
+
"""
|
|
76
|
+
approval = approval if approval is not None else load_approval()
|
|
77
|
+
if tool is not None:
|
|
78
|
+
failure = _check_approval(approval, tool)
|
|
79
|
+
if failure is not None:
|
|
80
|
+
return {"status": failure, "tool": tool, "answers": None,
|
|
81
|
+
"escalate": ESCALATE_TO_LLM}
|
|
82
|
+
elif not approval or not approval.get("approved"):
|
|
83
|
+
return {"status": JEV_APPROVAL_REQUIRED, "answers": None,
|
|
84
|
+
"escalate": ESCALATE_TO_LLM}
|
|
85
|
+
|
|
86
|
+
report = detect_jev_mcp()
|
|
87
|
+
if not report["available"]:
|
|
88
|
+
return {"status": JEV_UNAVAILABLE, "tool": tool, "answers": None,
|
|
89
|
+
"escalate": ESCALATE_TO_LLM, "hint": report.get("hint", "")}
|
|
90
|
+
|
|
91
|
+
payload = {
|
|
92
|
+
"model": model_id(),
|
|
93
|
+
"state": state,
|
|
94
|
+
"questions": questions,
|
|
95
|
+
}
|
|
96
|
+
status, body, err = _http_post_json(system_one_url(), payload, timeout=timeout)
|
|
97
|
+
if body is None:
|
|
98
|
+
code = _map_provider_error(status)
|
|
99
|
+
return {"status": code, "tool": tool, "http_status": status,
|
|
100
|
+
"error": err, "answers": None, "escalate": ESCALATE_TO_LLM}
|
|
101
|
+
|
|
102
|
+
answers = body.get("answers")
|
|
103
|
+
usage = body.get("usage") or {}
|
|
104
|
+
if not isinstance(answers, dict):
|
|
105
|
+
return {"status": JEV_INVALID_RESPONSE, "tool": tool,
|
|
106
|
+
"error": "missing or non-object answers envelope",
|
|
107
|
+
"answers": None, "escalate": ESCALATE_TO_LLM}
|
|
108
|
+
return {
|
|
109
|
+
"status": "ok",
|
|
110
|
+
"tool": tool,
|
|
111
|
+
"model": body.get("model") or model_id(),
|
|
112
|
+
"answers": answers,
|
|
113
|
+
"usage": {
|
|
114
|
+
"input_tokens": usage.get("input_tokens"),
|
|
115
|
+
"output_tokens": usage.get("output_tokens"),
|
|
116
|
+
},
|
|
117
|
+
"escalate": None,
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _prob(value: Any) -> float | None:
|
|
122
|
+
try:
|
|
123
|
+
p = float(value)
|
|
124
|
+
except (TypeError, ValueError):
|
|
125
|
+
return None
|
|
126
|
+
if p != p or p < 0.0 or p > 1.0: # NaN or out of range
|
|
127
|
+
return None
|
|
128
|
+
return p
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Experimental mode resolution (0|1|2|3)."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from .config import (
|
|
5
|
+
MODE_HYBRID,
|
|
6
|
+
MODE_JEV_MCP,
|
|
7
|
+
MODE_SEMDECIDE,
|
|
8
|
+
MODE_STANDARD,
|
|
9
|
+
_effective_env,
|
|
10
|
+
)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def resolve_experimental_mode(flag: str | int | bool | None = None) -> int:
|
|
14
|
+
"""Map --experimental / env to mode 0..3.
|
|
15
|
+
|
|
16
|
+
Accepts 0|1|2|3, True (=1), None (env EDU_EXPERIMENTAL_JEV, else 0).
|
|
17
|
+
"""
|
|
18
|
+
if flag is None:
|
|
19
|
+
flag = _effective_env("EDU_EXPERIMENTAL_JEV", "") or _effective_env(
|
|
20
|
+
"EXPERIMENTAL", "")
|
|
21
|
+
if flag is True:
|
|
22
|
+
return MODE_JEV_MCP
|
|
23
|
+
if flag is False or flag is None or flag == "":
|
|
24
|
+
return MODE_STANDARD
|
|
25
|
+
text = str(flag).strip().lower()
|
|
26
|
+
if text in ("0", "standard", "off", "false", "none"):
|
|
27
|
+
return MODE_STANDARD
|
|
28
|
+
if text in ("1", "jev", "jev-mcp", "jev_mcp", "true", "on"):
|
|
29
|
+
return MODE_JEV_MCP
|
|
30
|
+
if text in ("2", "semdecide", "sem"):
|
|
31
|
+
return MODE_SEMDECIDE
|
|
32
|
+
if text in ("3", "hybrid"):
|
|
33
|
+
return MODE_HYBRID
|
|
34
|
+
try:
|
|
35
|
+
value = int(text)
|
|
36
|
+
except ValueError:
|
|
37
|
+
return MODE_STANDARD
|
|
38
|
+
return value if value in (0, 1, 2, 3) else MODE_STANDARD
|