sys1bench 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sys1bench/__init__.py +3 -0
- sys1bench/adapters/__init__.py +19 -0
- sys1bench/adapters/base.py +159 -0
- sys1bench/adapters/embed_knn.py +61 -0
- sys1bench/adapters/encoder_finetuned.py +114 -0
- sys1bench/adapters/generic_http.py +161 -0
- sys1bench/adapters/hybrid_router.py +49 -0
- sys1bench/adapters/jev_openrouter.py +172 -0
- sys1bench/adapters/jev_typesafe.py +165 -0
- sys1bench/adapters/laya_local.py +192 -0
- sys1bench/adapters/llm_constrained.py +104 -0
- sys1bench/adapters/majority_prior.py +53 -0
- sys1bench/adapters/mock.py +77 -0
- sys1bench/adapters/nli_zeroshot.py +50 -0
- sys1bench/adapters/regex_keyword.py +70 -0
- sys1bench/analysis/__init__.py +4 -0
- sys1bench/analysis/arms.py +46 -0
- sys1bench/analysis/decision_value.py +43 -0
- sys1bench/analysis/decomposition.py +47 -0
- sys1bench/analysis/meta_eval.py +67 -0
- sys1bench/analysis/stats.py +76 -0
- sys1bench/cli.py +434 -0
- sys1bench/data/__init__.py +37 -0
- sys1bench/data/canary/canary.jsonl +200 -0
- sys1bench/data/framings/guardrail_intent.yaml +21 -0
- sys1bench/data/framings/log_triage.yaml +20 -0
- sys1bench/data/framings/phishing_email.yaml +25 -0
- sys1bench/data/framings/policy_compliance.yaml +18 -0
- sys1bench/data/framings/support_tickets.yaml +41 -0
- sys1bench/framing/__init__.py +10 -0
- sys1bench/framing/expand.py +214 -0
- sys1bench/framing/perturb.py +67 -0
- sys1bench/generators/__init__.py +10 -0
- sys1bench/generators/base.py +105 -0
- sys1bench/generators/guardrail_intent.py +95 -0
- sys1bench/generators/log_triage.py +100 -0
- sys1bench/generators/multilingual_tickets.py +84 -0
- sys1bench/generators/phishing_email.py +127 -0
- sys1bench/generators/policy_compliance.py +90 -0
- sys1bench/generators/rag_relevance.py +122 -0
- sys1bench/generators/support_tickets.py +212 -0
- sys1bench/metrics/__init__.py +9 -0
- sys1bench/metrics/calibration.py +164 -0
- sys1bench/metrics/consistency.py +71 -0
- sys1bench/metrics/decision_value.py +43 -0
- sys1bench/metrics/efficiency.py +35 -0
- sys1bench/metrics/ordinal.py +80 -0
- sys1bench/metrics/robustness.py +46 -0
- sys1bench/metrics/selective.py +67 -0
- sys1bench/report/__init__.py +1 -0
- sys1bench/report/dashboard.py +177 -0
- sys1bench/report/latex.py +40 -0
- sys1bench/report/plots.py +143 -0
- sys1bench/report/results_doc.py +237 -0
- sys1bench/report/scorecard.py +170 -0
- sys1bench/runners/__init__.py +2 -0
- sys1bench/runners/benchmark_runner.py +149 -0
- sys1bench/runners/cache.py +53 -0
- sys1bench/runners/canary.py +43 -0
- sys1bench/runners/hybrid_sweep.py +37 -0
- sys1bench/runners/noul_consistency.py +121 -0
- sys1bench/runners/ordinal_probes.py +90 -0
- sys1bench/runners/robustness.py +112 -0
- sys1bench/runners/suite.py +183 -0
- sys1bench/runners/sweeps.py +190 -0
- sys1bench/schemas.py +274 -0
- sys1bench-0.3.0.dist-info/METADATA +195 -0
- sys1bench-0.3.0.dist-info/RECORD +72 -0
- sys1bench-0.3.0.dist-info/WHEEL +5 -0
- sys1bench-0.3.0.dist-info/entry_points.txt +2 -0
- sys1bench-0.3.0.dist-info/licenses/LICENSE +202 -0
- sys1bench-0.3.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"""TypeSafe Jev via the OpenRouter alpha Decisions API.
|
|
2
|
+
|
|
3
|
+
Endpoint: POST https://openrouter.ai/api/alpha/decisions
|
|
4
|
+
Body: {"model": "typesafe/jev-1.13", "state": <str|obj|list>, "questions": {key: {type, instructions, criteria}}}
|
|
5
|
+
Types: choice (<=255 options), score (2..10 levels), boolean (our "noul").
|
|
6
|
+
Limits: 32k state tokens, 64k per request, 1200 rpm. Input billed at $0.042/M tokens, output free.
|
|
7
|
+
|
|
8
|
+
Response parsing is defensive: the alpha API's field names may move. Everything received is stored
|
|
9
|
+
verbatim in `raw`, so re-parsing later never needs another API call. The model id string returned by
|
|
10
|
+
the provider is recorded because `jev-latest` can move silently; pin `jev-1.13`.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import os
|
|
16
|
+
import random
|
|
17
|
+
import time
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
import httpx
|
|
21
|
+
|
|
22
|
+
from ..schemas import (
|
|
23
|
+
Answer,
|
|
24
|
+
DecisionRequest,
|
|
25
|
+
DecisionResponse,
|
|
26
|
+
LatencyRecord,
|
|
27
|
+
ModelCapabilities,
|
|
28
|
+
ProviderRecord,
|
|
29
|
+
Question,
|
|
30
|
+
)
|
|
31
|
+
from .base import BaseAdapter, finalize_answer, register
|
|
32
|
+
|
|
33
|
+
DEFAULT_URL = "https://openrouter.ai/api/alpha/decisions"
|
|
34
|
+
PRICE_PER_M_INPUT = 0.042
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _to_vendor_question(q: Question) -> dict[str, Any]:
|
|
38
|
+
if q.type == "noul":
|
|
39
|
+
return {"type": "boolean", "instructions": q.instructions}
|
|
40
|
+
if q.type == "choice":
|
|
41
|
+
opts = [{"key": c.key, "description": c.description} for c in q.criteria] # type: ignore[union-attr]
|
|
42
|
+
if q.allow_abstain:
|
|
43
|
+
opts.append({"key": q.abstain_key, "description": "None of the listed options applies."})
|
|
44
|
+
return {"type": "choice", "instructions": q.instructions, "criteria": opts}
|
|
45
|
+
levels = [{"level": c.level, "description": c.description} for c in q.criteria] # type: ignore[union-attr]
|
|
46
|
+
return {"type": "score", "instructions": q.instructions, "criteria": levels}
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _first(d: dict, *names: str, default=None):
|
|
50
|
+
for n in names:
|
|
51
|
+
if n in d and d[n] is not None:
|
|
52
|
+
return d[n]
|
|
53
|
+
return default
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def parse_answer(q: Question, payload: dict[str, Any]) -> Answer:
|
|
57
|
+
"""Map a vendor answer object to the contract. Handles the shapes seen in public harnesses:
|
|
58
|
+
choice: {"choice": key, "probabilities": {key: p} | [p...], "confidence": c}
|
|
59
|
+
score: {"score"|"expected": x, "distribution"|"probabilities": {...}|[...], "confidence": c}
|
|
60
|
+
boolean:{"probability"|"p_true"|"value": p, "confidence": c}
|
|
61
|
+
"""
|
|
62
|
+
keys = q.option_keys
|
|
63
|
+
conf = _first(payload, "confidence")
|
|
64
|
+
if q.type == "noul":
|
|
65
|
+
p = _first(payload, "probability", "p_true", "prob", "value")
|
|
66
|
+
if isinstance(p, bool):
|
|
67
|
+
p = 1.0 if p else 0.0
|
|
68
|
+
if p is None:
|
|
69
|
+
return Answer.failed(q, "no probability field")
|
|
70
|
+
return finalize_answer(q, [float(p), 1.0 - float(p)], confidence=conf)
|
|
71
|
+
dist = _first(payload, "probabilities", "distribution", "probs")
|
|
72
|
+
if dist is None:
|
|
73
|
+
return Answer.failed(q, "no distribution field")
|
|
74
|
+
if isinstance(dist, dict):
|
|
75
|
+
probs = [float(dist.get(k, dist.get(int(k) if k.isdigit() else k, 0.0))) for k in keys]
|
|
76
|
+
extra = set(map(str, dist)) - set(keys)
|
|
77
|
+
abst = q.allow_abstain and q.abstain_key in extra
|
|
78
|
+
elif isinstance(dist, list):
|
|
79
|
+
if all(isinstance(x, dict) for x in dist):
|
|
80
|
+
m = {str(_first(x, "key", "level", "option")): float(_first(x, "probability", "p", "prob")) for x in dist}
|
|
81
|
+
probs = [m.get(k, 0.0) for k in keys]
|
|
82
|
+
abst = q.allow_abstain and q.abstain_key in m
|
|
83
|
+
else:
|
|
84
|
+
probs = [float(x) for x in dist[: len(keys)]]
|
|
85
|
+
abst = False
|
|
86
|
+
else:
|
|
87
|
+
return Answer.failed(q, f"unparseable distribution {type(dist).__name__}")
|
|
88
|
+
if abst:
|
|
89
|
+
# abstain mass is recorded, then folded out so the vector stays over the manifest's options
|
|
90
|
+
total = sum(probs)
|
|
91
|
+
if total > 0:
|
|
92
|
+
probs = [p / total for p in probs]
|
|
93
|
+
ans = finalize_answer(q, probs, confidence=conf, abstained=bool(abst and max(probs) < 0.5))
|
|
94
|
+
return ans
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
@register("jev_openrouter")
|
|
98
|
+
class JevOpenRouterAdapter(BaseAdapter):
|
|
99
|
+
def __init__(self, model_id: str = "typesafe/jev-1.13", url: str = DEFAULT_URL, api_key: str | None = None,
|
|
100
|
+
timeout_s: float = 30.0, max_retries: int = 5, route: str = "openrouter", **tunables) -> None:
|
|
101
|
+
super().__init__(model_id, **tunables)
|
|
102
|
+
if model_id.endswith("latest"):
|
|
103
|
+
raise ValueError("pin an explicit Jev version (e.g. typesafe/jev-1.13); 'latest' moves silently")
|
|
104
|
+
self.url, self.timeout_s, self.max_retries, self.route = url, timeout_s, max_retries, route
|
|
105
|
+
self.api_key = api_key or os.environ.get("OPENROUTER_API_KEY")
|
|
106
|
+
self._client = httpx.Client(timeout=timeout_s)
|
|
107
|
+
|
|
108
|
+
@property
|
|
109
|
+
def capabilities(self) -> ModelCapabilities:
|
|
110
|
+
return ModelCapabilities(name=self.model_id, deployment="hosted", max_options=255, max_score_levels=10,
|
|
111
|
+
max_state_tokens=32_000, emits_confidence=True, supports_native_abstain=False,
|
|
112
|
+
supports_batching=True, languages=None)
|
|
113
|
+
|
|
114
|
+
def _post(self, body: dict[str, Any]) -> tuple[dict[str, Any], float, dict[str, str]]:
|
|
115
|
+
if not self.api_key:
|
|
116
|
+
raise RuntimeError("OPENROUTER_API_KEY not set")
|
|
117
|
+
headers = {"Authorization": f"Bearer {self.api_key}", "Content-Type": "application/json"}
|
|
118
|
+
delay = 0.5
|
|
119
|
+
last_exc: Exception | None = None
|
|
120
|
+
for attempt in range(self.max_retries):
|
|
121
|
+
t0 = time.perf_counter()
|
|
122
|
+
try:
|
|
123
|
+
r = self._client.post(self.url, json=body, headers=headers)
|
|
124
|
+
ms = (time.perf_counter() - t0) * 1000
|
|
125
|
+
if r.status_code in (429, 500, 502, 503, 504):
|
|
126
|
+
raise httpx.HTTPStatusError(f"{r.status_code}", request=r.request, response=r)
|
|
127
|
+
r.raise_for_status()
|
|
128
|
+
return r.json(), ms, dict(r.headers)
|
|
129
|
+
except (httpx.TransportError, httpx.HTTPStatusError) as e:
|
|
130
|
+
last_exc = e
|
|
131
|
+
time.sleep(delay + random.uniform(0, delay))
|
|
132
|
+
delay = min(delay * 2, 16.0)
|
|
133
|
+
raise RuntimeError(f"jev request failed after {self.max_retries} attempts: {last_exc}")
|
|
134
|
+
|
|
135
|
+
def parse_raw_answer(self, q: Question, payload: dict) -> Answer:
|
|
136
|
+
return parse_answer(q, payload)
|
|
137
|
+
|
|
138
|
+
def decide(self, request: DecisionRequest) -> DecisionResponse:
|
|
139
|
+
body = {"model": self.model_id, "state": request.state,
|
|
140
|
+
"questions": {k: _to_vendor_question(q) for k, q in request.questions.items()}}
|
|
141
|
+
try:
|
|
142
|
+
data, ms, headers = self._post(body)
|
|
143
|
+
except Exception as e:
|
|
144
|
+
return DecisionResponse(
|
|
145
|
+
answers={k: Answer.failed(q, "transport") for k, q in request.questions.items()},
|
|
146
|
+
latency=LatencyRecord(client_ms=float("nan"), route=self.route, timestamp=time.time()),
|
|
147
|
+
provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id, route=self.route),
|
|
148
|
+
transport_error=str(e),
|
|
149
|
+
)
|
|
150
|
+
raw_answers = _first(data, "answers", "decisions", "results", default={})
|
|
151
|
+
answers = {}
|
|
152
|
+
for k, q in request.questions.items():
|
|
153
|
+
payload = raw_answers.get(k) if isinstance(raw_answers, dict) else None
|
|
154
|
+
answers[k] = parse_answer(q, payload) if isinstance(payload, dict) else Answer.failed(q, "missing answer")
|
|
155
|
+
usage = data.get("usage", {}) or {}
|
|
156
|
+
in_tok = _first(usage, "prompt_tokens", "input_tokens")
|
|
157
|
+
cost = (in_tok / 1e6) * PRICE_PER_M_INPUT if in_tok else _first(usage, "cost")
|
|
158
|
+
server_ms = None
|
|
159
|
+
for h in ("x-processing-ms", "openrouter-processing-ms", "x-response-time"):
|
|
160
|
+
if h in headers:
|
|
161
|
+
try:
|
|
162
|
+
server_ms = float(str(headers[h]).rstrip("ms"))
|
|
163
|
+
except ValueError:
|
|
164
|
+
pass
|
|
165
|
+
return DecisionResponse(
|
|
166
|
+
answers=answers,
|
|
167
|
+
latency=LatencyRecord(client_ms=ms, server_ms=server_ms, route=self.route, timestamp=time.time()),
|
|
168
|
+
provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id,
|
|
169
|
+
model_id_returned=_first(data, "model"), generation_id=_first(data, "id", "generation_id"),
|
|
170
|
+
route=self.route, billed_input_tokens=in_tok, billed_output_tokens=0, cost_usd=cost),
|
|
171
|
+
raw=data,
|
|
172
|
+
)
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
"""TypeSafe Jev via the first-party API (preferred over the OpenRouter alpha route).
|
|
2
|
+
|
|
3
|
+
Endpoint: POST https://api.typesafe.ai/v1/systemone Authorization: Bearer <TypeSafe_API_KEY>
|
|
4
|
+
Body: {"state": <str|obj|list>, "model": "jev-1.13.0", "questions": {id: Question}}
|
|
5
|
+
Question: choice -> {"type": "choice", "instructions": str|obj, "criteria": {option: description|null}} (<=255 options)
|
|
6
|
+
score -> {"type": "score", "instructions": str|obj, "criteria": [level_description, ...]} (2..10 levels)
|
|
7
|
+
noul -> {"type": "noul", "instructions": str|obj, "criteria": {"true": ..., "false": ...}?}
|
|
8
|
+
Answer: choice -> {"type","choice","probabilities":{option: p},"confidence"}
|
|
9
|
+
score -> {"type","score" (expected level index),"legend":{"0": desc,...},"probabilities":{"0": p,...},"confidence"}
|
|
10
|
+
noul -> {"type","noul": P(yes)}
|
|
11
|
+
Response also carries "model" (versioned id that answered) and "usage": {input_tokens, output_tokens}.
|
|
12
|
+
Limits (models page, 2026-09-22): 64k tokens per request, 32k for state plus longest question, 1200 rpm, 250k tok/s.
|
|
13
|
+
Price: $0.042 per million input tokens; output free.
|
|
14
|
+
|
|
15
|
+
Score probabilities are keyed by *level index*, so they are mapped back onto the manifest's level values by
|
|
16
|
+
position. Option order in `criteria` is preserved as JSON object order, which is what the permutation suite varies.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import os
|
|
22
|
+
import random
|
|
23
|
+
import time
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from typing import Any
|
|
26
|
+
|
|
27
|
+
import httpx
|
|
28
|
+
|
|
29
|
+
from ..schemas import Answer, DecisionRequest, DecisionResponse, LatencyRecord, ModelCapabilities, ProviderRecord, Question
|
|
30
|
+
from .base import BaseAdapter, finalize_answer, register
|
|
31
|
+
|
|
32
|
+
DEFAULT_URL = "https://api.typesafe.ai/v1/systemone"
|
|
33
|
+
PRICE_PER_M_INPUT = 0.042
|
|
34
|
+
ENV_KEYS = ("TypeSafe_API_KEY", "TYPESAFE_API_KEY")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _load_dotenv_key() -> str | None:
|
|
38
|
+
for k in ENV_KEYS:
|
|
39
|
+
if os.environ.get(k):
|
|
40
|
+
return os.environ[k]
|
|
41
|
+
for candidate in (Path.cwd() / ".env", Path(__file__).resolve().parents[3] / ".env"):
|
|
42
|
+
if candidate.exists():
|
|
43
|
+
for line in candidate.read_text().splitlines():
|
|
44
|
+
line = line.strip()
|
|
45
|
+
if "=" in line and not line.startswith("#"):
|
|
46
|
+
name, _, val = line.partition("=")
|
|
47
|
+
if name.strip() in ENV_KEYS and val.strip():
|
|
48
|
+
return val.strip().strip('"').strip("'")
|
|
49
|
+
return None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def to_vendor_question(q: Question) -> dict[str, Any]:
|
|
53
|
+
if q.type == "noul":
|
|
54
|
+
return {"type": "noul", "instructions": q.instructions}
|
|
55
|
+
if q.type == "choice":
|
|
56
|
+
crit: dict[str, Any] = {c.key: (c.description or None) for c in q.criteria} # type: ignore[union-attr]
|
|
57
|
+
if q.allow_abstain:
|
|
58
|
+
crit[q.abstain_key] = "None of the listed options applies."
|
|
59
|
+
return {"type": "choice", "instructions": q.instructions, "criteria": crit}
|
|
60
|
+
levels = [(c.description or f"level {c.level}") for c in q.criteria] # type: ignore[union-attr]
|
|
61
|
+
return {"type": "score", "instructions": q.instructions, "criteria": levels}
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def parse_answer(q: Question, payload: dict[str, Any]) -> Answer:
|
|
65
|
+
keys = q.option_keys
|
|
66
|
+
if q.type == "noul":
|
|
67
|
+
p = payload.get("noul")
|
|
68
|
+
if p is None:
|
|
69
|
+
return Answer.failed(q, "no noul field")
|
|
70
|
+
return finalize_answer(q, [float(p), 1.0 - float(p)])
|
|
71
|
+
dist = payload.get("probabilities")
|
|
72
|
+
if not isinstance(dist, dict):
|
|
73
|
+
return Answer.failed(q, "no probabilities map")
|
|
74
|
+
conf = payload.get("confidence")
|
|
75
|
+
if q.type == "score":
|
|
76
|
+
# keyed by level index as string, in the order we sent the levels
|
|
77
|
+
probs = [float(dist.get(str(i), 0.0)) for i in range(len(keys))]
|
|
78
|
+
return finalize_answer(q, probs, confidence=conf)
|
|
79
|
+
probs = [float(dist.get(k, 0.0)) for k in keys]
|
|
80
|
+
abstained = False
|
|
81
|
+
if q.allow_abstain and q.abstain_key in dist:
|
|
82
|
+
p_abst = float(dist[q.abstain_key])
|
|
83
|
+
abstained = p_abst >= max(probs)
|
|
84
|
+
s = sum(probs)
|
|
85
|
+
probs = [p / s for p in probs] if s > 0 else [1.0 / len(keys)] * len(keys)
|
|
86
|
+
return finalize_answer(q, probs, confidence=conf, abstained=abstained)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
@register("jev_typesafe")
|
|
90
|
+
class JevTypeSafeAdapter(BaseAdapter):
|
|
91
|
+
def __init__(self, model_id: str = "jev-1.13.0", url: str = DEFAULT_URL, api_key: str | None = None,
|
|
92
|
+
timeout_s: float = 30.0, max_retries: int = 6, allow_alias: bool = False, **tunables) -> None:
|
|
93
|
+
super().__init__(model_id, **tunables)
|
|
94
|
+
if (model_id.endswith("latest") or model_id.endswith("preview")) and not allow_alias:
|
|
95
|
+
raise ValueError("pin a versioned Jev id (e.g. jev-1.13.0); aliases move silently. Pass allow_alias=True to override.")
|
|
96
|
+
self.url, self.timeout_s, self.max_retries = url, timeout_s, max_retries
|
|
97
|
+
self.api_key = api_key or _load_dotenv_key()
|
|
98
|
+
self._client = httpx.Client(timeout=timeout_s, http2=False)
|
|
99
|
+
|
|
100
|
+
@property
|
|
101
|
+
def capabilities(self) -> ModelCapabilities:
|
|
102
|
+
return ModelCapabilities(name=self.model_id, deployment="hosted", max_options=255, max_score_levels=10,
|
|
103
|
+
max_state_tokens=32_000, emits_confidence=True, supports_native_abstain=False,
|
|
104
|
+
supports_batching=True, languages={"en"})
|
|
105
|
+
|
|
106
|
+
def _post(self, body: dict[str, Any]) -> tuple[dict[str, Any], float, dict[str, str]]:
|
|
107
|
+
if not self.api_key:
|
|
108
|
+
raise RuntimeError("TypeSafe_API_KEY not set (env or .env)")
|
|
109
|
+
headers = {"Authorization": f"Bearer {self.api_key}", "Content-Type": "application/json"}
|
|
110
|
+
delay, last = 0.5, None
|
|
111
|
+
for _ in range(self.max_retries):
|
|
112
|
+
t0 = time.perf_counter()
|
|
113
|
+
try:
|
|
114
|
+
r = self._client.post(self.url, json=body, headers=headers)
|
|
115
|
+
ms = (time.perf_counter() - t0) * 1000
|
|
116
|
+
if r.status_code in (429, 500, 502, 503, 504, 529):
|
|
117
|
+
ra = r.headers.get("retry-after")
|
|
118
|
+
if ra:
|
|
119
|
+
try:
|
|
120
|
+
delay = max(delay, float(ra))
|
|
121
|
+
except ValueError:
|
|
122
|
+
pass
|
|
123
|
+
raise httpx.HTTPStatusError(f"{r.status_code}: {r.text[:200]}", request=r.request, response=r)
|
|
124
|
+
if 400 <= r.status_code < 500 and r.status_code != 429:
|
|
125
|
+
# request rejected (400 over the 32k state limit, 413, 422 malformed): never retry, classify as vendor rejection
|
|
126
|
+
raise ValueError(f"{r.status_code} rejected: {r.text[:300]}")
|
|
127
|
+
r.raise_for_status()
|
|
128
|
+
return r.json(), ms, dict(r.headers)
|
|
129
|
+
except (httpx.TransportError, httpx.HTTPStatusError) as e:
|
|
130
|
+
last = e
|
|
131
|
+
time.sleep(delay + random.uniform(0, delay))
|
|
132
|
+
delay = min(delay * 2, 20.0)
|
|
133
|
+
raise RuntimeError(f"typesafe request failed after {self.max_retries} attempts: {last}")
|
|
134
|
+
|
|
135
|
+
def parse_raw_answer(self, q: Question, payload: dict) -> Answer:
|
|
136
|
+
return parse_answer(q, payload)
|
|
137
|
+
|
|
138
|
+
def decide(self, request: DecisionRequest) -> DecisionResponse:
|
|
139
|
+
body = {"state": request.state, "model": self.model_id,
|
|
140
|
+
"questions": {k: to_vendor_question(q) for k, q in request.questions.items()}}
|
|
141
|
+
try:
|
|
142
|
+
data, ms, headers = self._post(body)
|
|
143
|
+
except Exception as e:
|
|
144
|
+
kind = "rejected_by_vendor" if isinstance(e, ValueError) else "transport"
|
|
145
|
+
return DecisionResponse(
|
|
146
|
+
answers={k: Answer.failed(q, kind) for k, q in request.questions.items()},
|
|
147
|
+
latency=LatencyRecord(client_ms=float("nan"), route="typesafe", timestamp=time.time()),
|
|
148
|
+
provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id, route="typesafe"),
|
|
149
|
+
transport_error=str(e)[:500],
|
|
150
|
+
)
|
|
151
|
+
raw_answers = data.get("answers", {}) or {}
|
|
152
|
+
answers = {k: (parse_answer(q, raw_answers[k]) if isinstance(raw_answers.get(k), dict) else Answer.failed(q, "missing answer"))
|
|
153
|
+
for k, q in request.questions.items()}
|
|
154
|
+
usage = data.get("usage", {}) or {}
|
|
155
|
+
in_tok = usage.get("input_tokens")
|
|
156
|
+
return DecisionResponse(
|
|
157
|
+
answers=answers,
|
|
158
|
+
latency=LatencyRecord(client_ms=ms, route="typesafe", timestamp=time.time()),
|
|
159
|
+
provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id, model_id_returned=data.get("model"),
|
|
160
|
+
version_hash=str(data.get("model") or self.model_id),
|
|
161
|
+
generation_id=headers.get("x-request-id") or headers.get("request-id"), route="typesafe",
|
|
162
|
+
billed_input_tokens=in_tok, billed_output_tokens=usage.get("output_tokens"),
|
|
163
|
+
cost_usd=(in_tok / 1e6 * PRICE_PER_M_INPUT) if in_tok is not None else None),
|
|
164
|
+
raw=data,
|
|
165
|
+
)
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
"""Convai Laya, run locally (Apache 2.0) through the `laya` package's Router (verified against laya 0.3.4).
|
|
2
|
+
|
|
3
|
+
from laya import Router
|
|
4
|
+
router = Router(preload=True, device="cuda")
|
|
5
|
+
out = router.predict(state, questions, model=None | "english" | "multilingual" | "typed-decisions")
|
|
6
|
+
|
|
7
|
+
Question shape matches TypeSafe's: choice criteria is {option: description}, score criteria is an ordered list of
|
|
8
|
+
level descriptions, noul has no criteria. Answers:
|
|
9
|
+
choice -> {"type","choice","probabilities":{opt: p},"confidence","action":{"act_probability"}}
|
|
10
|
+
score -> {"type","score","legend","probabilities":{"0": p,...},"confidence","action":{...}}
|
|
11
|
+
noul -> {"type","noul": P(yes),"confidence","action":{...}}
|
|
12
|
+
`action.act_probability` is Laya's act/escalate head; we treat act_probability < 0.5 as a native abstention.
|
|
13
|
+
Result also carries "routing": {"model","repo","reason",...} and "usage".
|
|
14
|
+
|
|
15
|
+
Tunables: `head_max_len` and `max_len` are set on the checkpoint's `agent.cfg` before inference (Suite C ablates them;
|
|
16
|
+
the 77-label collapse at the default budget is a configuration effect). Checkpoint defaults: english 512/192,
|
|
17
|
+
multilingual 1024/256, typed-decisions 1024/256. The package prints a warning and falls back to CPU when CUDA is
|
|
18
|
+
unavailable or out of memory; we record the device actually used in `hardware` so latency tables stay honest.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import platform
|
|
24
|
+
import time
|
|
25
|
+
from typing import Any
|
|
26
|
+
|
|
27
|
+
from ..schemas import Answer, DecisionRequest, DecisionResponse, LatencyRecord, ModelCapabilities, ProviderRecord, Question
|
|
28
|
+
from .base import BaseAdapter, FatalAdapterError, finalize_answer, register
|
|
29
|
+
|
|
30
|
+
CHECKPOINTS = {"english": ("convaiinnovations/laya", 512, 192), "multilingual": ("convaiinnovations/laya/multilingual", 1024, 256),
|
|
31
|
+
"typed-decisions": ("convaiinnovations/laya/typed-decisions", 1024, 256), "router": ("convaiinnovations/laya", 1024, 256)}
|
|
32
|
+
ALIASES = {"laya": "english", "laya-multilingual": "multilingual", "laya-typed-decisions": "typed-decisions"}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def to_vendor_question(q: Question) -> dict[str, Any]:
|
|
36
|
+
if q.type == "noul":
|
|
37
|
+
return {"type": "noul", "instructions": q.instructions}
|
|
38
|
+
if q.type == "choice":
|
|
39
|
+
crit = {c.key: (c.description or c.key) for c in q.criteria} # type: ignore[union-attr]
|
|
40
|
+
if q.allow_abstain:
|
|
41
|
+
crit[q.abstain_key] = "None of the listed options applies."
|
|
42
|
+
return {"type": "choice", "instructions": q.instructions, "criteria": crit}
|
|
43
|
+
return {"type": "score", "instructions": q.instructions,
|
|
44
|
+
"criteria": [(c.description or f"level {c.level}") for c in q.criteria]} # type: ignore[union-attr]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def parse_answer(q: Question, payload: dict[str, Any], abstain_threshold: float = 0.5) -> Answer:
|
|
48
|
+
keys = q.option_keys
|
|
49
|
+
conf = payload.get("confidence")
|
|
50
|
+
act = (payload.get("action") or {}).get("act_probability")
|
|
51
|
+
abstained = act is not None and float(act) < abstain_threshold
|
|
52
|
+
if q.type == "noul":
|
|
53
|
+
p = payload.get("noul")
|
|
54
|
+
if p is None:
|
|
55
|
+
return Answer.failed(q, "no noul field")
|
|
56
|
+
a = finalize_answer(q, [float(p), 1.0 - float(p)], confidence=None, abstained=abstained)
|
|
57
|
+
else:
|
|
58
|
+
dist = payload.get("probabilities")
|
|
59
|
+
if not isinstance(dist, dict):
|
|
60
|
+
return Answer.failed(q, "no probabilities map")
|
|
61
|
+
if q.type == "score":
|
|
62
|
+
probs = [float(dist.get(str(i), 0.0)) for i in range(len(keys))]
|
|
63
|
+
else:
|
|
64
|
+
probs = [float(dist.get(k, 0.0)) for k in keys]
|
|
65
|
+
if q.allow_abstain and q.abstain_key in dist:
|
|
66
|
+
abstained = abstained or float(dist[q.abstain_key]) >= max(probs)
|
|
67
|
+
s = sum(probs)
|
|
68
|
+
probs = [x / s for x in probs] if s > 0 else [1.0 / len(keys)] * len(keys)
|
|
69
|
+
a = finalize_answer(q, probs, confidence=conf, abstained=abstained)
|
|
70
|
+
if a.ok and act is not None:
|
|
71
|
+
a = a.model_copy(update={"raw_prob_sum": a.raw_prob_sum})
|
|
72
|
+
return a
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@register("laya_local")
|
|
76
|
+
class LayaLocalAdapter(BaseAdapter):
|
|
77
|
+
def __init__(self, model_id: str = "convaiinnovations/laya", checkpoint: str = "english", device: str = "cuda",
|
|
78
|
+
head_max_len: int | None = None, max_len: int | None = None, batch_size: int = 1,
|
|
79
|
+
hardware: str | None = None, abstain_threshold: float = 0.5, strict_device: bool = True, **tunables) -> None:
|
|
80
|
+
super().__init__(model_id, **tunables)
|
|
81
|
+
checkpoint = ALIASES.get(checkpoint, checkpoint)
|
|
82
|
+
if checkpoint not in CHECKPOINTS:
|
|
83
|
+
raise ValueError(f"unknown Laya checkpoint {checkpoint!r}; known: {sorted(CHECKPOINTS)} (+ aliases {sorted(ALIASES)})")
|
|
84
|
+
self.checkpoint, self.device, self.batch_size, self.abstain_threshold = checkpoint, device, batch_size, abstain_threshold
|
|
85
|
+
self.strict_device = strict_device
|
|
86
|
+
_, dflt_max, dflt_head = CHECKPOINTS[checkpoint]
|
|
87
|
+
self.head_max_len = head_max_len or dflt_head
|
|
88
|
+
self.max_len = max_len or dflt_max
|
|
89
|
+
self.hardware = hardware
|
|
90
|
+
self.tunables.update({"head_max_len": self.head_max_len, "max_len": self.max_len, "checkpoint": checkpoint})
|
|
91
|
+
self._router = None
|
|
92
|
+
|
|
93
|
+
@property
|
|
94
|
+
def capabilities(self) -> ModelCapabilities:
|
|
95
|
+
return ModelCapabilities(name=f"laya:{self.checkpoint}", deployment="local", max_options=255,
|
|
96
|
+
max_state_tokens=self.max_len - self.head_max_len, emits_confidence=True,
|
|
97
|
+
supports_native_abstain=True, supports_batching=True,
|
|
98
|
+
languages={"en"} if self.checkpoint == "english" else None,
|
|
99
|
+
tunables={"head_max_len": self.head_max_len, "max_len": self.max_len})
|
|
100
|
+
|
|
101
|
+
def _load(self):
|
|
102
|
+
if self._router is not None:
|
|
103
|
+
return self._router
|
|
104
|
+
try:
|
|
105
|
+
from laya import Router # type: ignore
|
|
106
|
+
except ImportError as e: # pragma: no cover
|
|
107
|
+
raise RuntimeError("install Laya: pip install laya (github.com/NandhaKishorM/laya)") from e
|
|
108
|
+
# Lazy router: load only the checkpoint we evaluate (preloading all three needs ~6 GB and, on a shared GPU,
|
|
109
|
+
# silently lands on CPU). "router" mode keeps the full preload because it dispatches per language.
|
|
110
|
+
if self.checkpoint == "router":
|
|
111
|
+
router = Router(preload=True, device=self.device)
|
|
112
|
+
agents = list((getattr(router, "_agents", {}) or {}).values())
|
|
113
|
+
else:
|
|
114
|
+
router = Router(preload=False, device=self.device)
|
|
115
|
+
agents = [router.load(self.checkpoint)]
|
|
116
|
+
for ag in agents:
|
|
117
|
+
if hasattr(ag, "cfg"):
|
|
118
|
+
ag.cfg["head_max_len"] = self.head_max_len
|
|
119
|
+
ag.cfg["max_len"] = self.max_len
|
|
120
|
+
dev = str(getattr(ag, "device", ""))
|
|
121
|
+
if self.device.startswith("cuda") and "cuda" not in dev and hasattr(ag, "model"):
|
|
122
|
+
try: # the package falls back to CPU when CUDA context creation fails at import; retry the move
|
|
123
|
+
import torch # type: ignore
|
|
124
|
+
|
|
125
|
+
ag.model.to("cuda")
|
|
126
|
+
ag.device = torch.device("cuda") if hasattr(torch, "device") else "cuda"
|
|
127
|
+
except Exception:
|
|
128
|
+
pass
|
|
129
|
+
dev = str(getattr(ag, "device", ""))
|
|
130
|
+
if self.device.startswith("cuda") and "cuda" not in dev and self.strict_device:
|
|
131
|
+
del router, agents # release the half-loaded model before aborting
|
|
132
|
+
raise FatalAdapterError(f"Laya checkpoint {self.checkpoint!r} loaded on {dev or 'cpu'} although device={self.device!r} was requested; "
|
|
133
|
+
"refusing to record CPU latency as GPU. Free GPU memory or pass strict_device=false.")
|
|
134
|
+
ag = agents[0] if agents else None
|
|
135
|
+
dev = str(getattr(ag, "device", self.device)) if ag is not None else self.device
|
|
136
|
+
if self.hardware is None:
|
|
137
|
+
gpu = None
|
|
138
|
+
try:
|
|
139
|
+
import torch # type: ignore
|
|
140
|
+
|
|
141
|
+
if "cuda" in dev and torch.cuda.is_available():
|
|
142
|
+
gpu = torch.cuda.get_device_name(0)
|
|
143
|
+
except Exception: # pragma: no cover
|
|
144
|
+
pass
|
|
145
|
+
self.hardware = f"{gpu} ({dev})" if gpu else f"CPU {platform.machine()} ({dev})"
|
|
146
|
+
self._router = router # only after every check passed
|
|
147
|
+
return self._router
|
|
148
|
+
|
|
149
|
+
def parse_raw_answer(self, q: Question, payload: dict) -> Answer:
|
|
150
|
+
return parse_answer(q, payload, self.abstain_threshold)
|
|
151
|
+
|
|
152
|
+
def decide(self, request: DecisionRequest) -> DecisionResponse:
|
|
153
|
+
router = self._load()
|
|
154
|
+
vq = {k: to_vendor_question(q) for k, q in request.questions.items()}
|
|
155
|
+
model = None if self.checkpoint == "router" else self.checkpoint
|
|
156
|
+
t0 = time.perf_counter()
|
|
157
|
+
try:
|
|
158
|
+
out = router.predict(request.state, vq, model=model)
|
|
159
|
+
except Exception as e:
|
|
160
|
+
# laya raises ValueError("... options exceed head_max_len=N") when the option strings do not fit the budget:
|
|
161
|
+
# that is the model refusing the request, not a harness bug, so it is classified like a vendor 4xx.
|
|
162
|
+
kind = "rejected_by_model" if isinstance(e, ValueError) else "adapter_exception"
|
|
163
|
+
return DecisionResponse(
|
|
164
|
+
answers={k: Answer.failed(q, kind) for k, q in request.questions.items()},
|
|
165
|
+
latency=LatencyRecord(client_ms=float("nan"), timestamp=time.time()),
|
|
166
|
+
provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id, hardware=self.hardware),
|
|
167
|
+
transport_error=repr(e)[:500])
|
|
168
|
+
ms = (time.perf_counter() - t0) * 1000
|
|
169
|
+
raw_answers = out.get("answers", {}) if isinstance(out, dict) else {}
|
|
170
|
+
# laya truncates silently to max_len - head_max_len state tokens and reports nothing; estimate it (~4 chars/token)
|
|
171
|
+
from ..generators.base import estimate_tokens
|
|
172
|
+
|
|
173
|
+
est_tokens = estimate_tokens(request.state_text())
|
|
174
|
+
truncated = bool(out.get("truncated", False)) if isinstance(out, dict) else False
|
|
175
|
+
truncated = truncated or est_tokens > (self.max_len - self.head_max_len)
|
|
176
|
+
answers = {}
|
|
177
|
+
for k, q in request.questions.items():
|
|
178
|
+
a = parse_answer(q, raw_answers[k], self.abstain_threshold) if isinstance(raw_answers.get(k), dict) else Answer.failed(q, "missing answer")
|
|
179
|
+
if a.ok and truncated:
|
|
180
|
+
a = a.model_copy(update={"truncated": True})
|
|
181
|
+
answers[k] = a
|
|
182
|
+
routing = out.get("routing", {}) if isinstance(out, dict) else {}
|
|
183
|
+
usage = out.get("usage", {}) if isinstance(out, dict) else {}
|
|
184
|
+
return DecisionResponse(
|
|
185
|
+
answers=answers,
|
|
186
|
+
latency=LatencyRecord(client_ms=ms, compute_ms=ms, batch_size=self.batch_size, timestamp=time.time()),
|
|
187
|
+
provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id,
|
|
188
|
+
model_id_returned=f"laya:{routing.get('model', self.checkpoint)}",
|
|
189
|
+
version_hash=f"{routing.get('repo', '')}|head{self.head_max_len}|max{self.max_len}",
|
|
190
|
+
hardware=self.hardware, billed_input_tokens=usage.get("input_tokens"), cost_usd=0.0),
|
|
191
|
+
raw=out if isinstance(out, dict) else {"out": str(out)},
|
|
192
|
+
)
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""Constrained-decoding LLM baseline with first-token logprobs, so LLMs get calibration metrics.
|
|
2
|
+
|
|
3
|
+
Backend A (default): any OpenAI-compatible server (vLLM, OpenRouter) with `logprobs`. The prompt lists
|
|
4
|
+
options as single-letter keys; probabilities are the renormalised logprobs of the first generated token
|
|
5
|
+
over the option letters, which is the standard "logit-based" calibration protocol for MCQ.
|
|
6
|
+
Backend B: vLLM + Outlines regex-constrained generation (import guarded); same probability extraction.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import math
|
|
12
|
+
import os
|
|
13
|
+
import string
|
|
14
|
+
import time
|
|
15
|
+
|
|
16
|
+
import httpx
|
|
17
|
+
|
|
18
|
+
from ..schemas import (
|
|
19
|
+
Answer,
|
|
20
|
+
DecisionRequest,
|
|
21
|
+
DecisionResponse,
|
|
22
|
+
LatencyRecord,
|
|
23
|
+
ModelCapabilities,
|
|
24
|
+
ProviderRecord,
|
|
25
|
+
Question,
|
|
26
|
+
)
|
|
27
|
+
from .base import BaseAdapter, finalize_answer, register
|
|
28
|
+
|
|
29
|
+
LETTERS = string.ascii_uppercase + string.ascii_lowercase + string.digits
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def build_prompt(state: str, q: Question) -> tuple[str, list[str]]:
|
|
33
|
+
keys = q.option_keys
|
|
34
|
+
letters = list(LETTERS[: len(keys)])
|
|
35
|
+
if q.type == "noul":
|
|
36
|
+
lines = ["A. yes", "B. no"]
|
|
37
|
+
elif q.type == "choice":
|
|
38
|
+
lines = [f"{L}. {c.key}: {c.description}" for L, c in zip(letters, q.criteria)] # type: ignore[union-attr]
|
|
39
|
+
else:
|
|
40
|
+
lines = [f"{L}. level {c.level}: {c.description}" for L, c in zip(letters, q.criteria)] # type: ignore[union-attr]
|
|
41
|
+
prompt = (f"State:\n{state}\n\nQuestion: {q.instructions}\nOptions:\n" + "\n".join(lines) +
|
|
42
|
+
"\n\nAnswer with the single letter of the best option.\nAnswer:")
|
|
43
|
+
return prompt, letters
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@register("llm_constrained")
|
|
47
|
+
class LLMConstrainedAdapter(BaseAdapter):
|
|
48
|
+
def __init__(self, model_id: str = "Qwen/Qwen2.5-3B-Instruct", base_url: str = "http://localhost:8000/v1",
|
|
49
|
+
api_key_env: str = "OPENAI_API_KEY", top_logprobs: int = 20, price_per_m_input: float | None = None,
|
|
50
|
+
price_per_m_output: float | None = None, timeout_s: float = 60.0, **tunables) -> None:
|
|
51
|
+
super().__init__(model_id, **tunables)
|
|
52
|
+
self.base_url, self.top_logprobs = base_url.rstrip("/"), top_logprobs
|
|
53
|
+
self.key = os.environ.get(api_key_env, "EMPTY")
|
|
54
|
+
self.pi, self.po = price_per_m_input, price_per_m_output
|
|
55
|
+
self._client = httpx.Client(timeout=timeout_s)
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def capabilities(self) -> ModelCapabilities:
|
|
59
|
+
return ModelCapabilities(name=f"llm:{self.model_id}", deployment="hosted" if "localhost" not in self.base_url else "local",
|
|
60
|
+
max_options=len(LETTERS), emits_confidence=False, supports_batching=False, deterministic=False)
|
|
61
|
+
|
|
62
|
+
def _one(self, state: str, q: Question) -> tuple[Answer, dict]:
|
|
63
|
+
prompt, letters = build_prompt(state, q)
|
|
64
|
+
body = {"model": self.model_id, "prompt": prompt, "max_tokens": 1, "temperature": 0.0,
|
|
65
|
+
"logprobs": self.top_logprobs}
|
|
66
|
+
r = self._client.post(f"{self.base_url}/completions", json=body,
|
|
67
|
+
headers={"Authorization": f"Bearer {self.key}"})
|
|
68
|
+
r.raise_for_status()
|
|
69
|
+
data = r.json()
|
|
70
|
+
lp = data["choices"][0].get("logprobs", {})
|
|
71
|
+
top = (lp.get("top_logprobs") or [{}])[0]
|
|
72
|
+
scores = []
|
|
73
|
+
for L in letters:
|
|
74
|
+
cands = [v for k, v in top.items() if k.strip() == L]
|
|
75
|
+
scores.append(max(cands) if cands else -30.0)
|
|
76
|
+
m = max(scores)
|
|
77
|
+
w = [math.exp(s - m) for s in scores]
|
|
78
|
+
z = sum(w)
|
|
79
|
+
probs = [x / z for x in w]
|
|
80
|
+
return finalize_answer(q, probs), data
|
|
81
|
+
|
|
82
|
+
def decide(self, request: DecisionRequest) -> DecisionResponse:
|
|
83
|
+
t0 = time.perf_counter()
|
|
84
|
+
answers, raws, in_tok, out_tok = {}, {}, 0, 0
|
|
85
|
+
err = None
|
|
86
|
+
for k, q in request.questions.items():
|
|
87
|
+
try:
|
|
88
|
+
a, data = self._one(request.state_text(), q)
|
|
89
|
+
answers[k] = a
|
|
90
|
+
raws[k] = data
|
|
91
|
+
u = data.get("usage", {})
|
|
92
|
+
in_tok += u.get("prompt_tokens", 0)
|
|
93
|
+
out_tok += u.get("completion_tokens", 0)
|
|
94
|
+
except Exception as e: # transport or parse
|
|
95
|
+
answers[k] = Answer.failed(q, "transport")
|
|
96
|
+
err = str(e)
|
|
97
|
+
ms = (time.perf_counter() - t0) * 1000
|
|
98
|
+
cost = None
|
|
99
|
+
if self.pi is not None:
|
|
100
|
+
cost = in_tok / 1e6 * self.pi + out_tok / 1e6 * (self.po or 0.0)
|
|
101
|
+
return DecisionResponse(answers=answers, latency=LatencyRecord(client_ms=ms, timestamp=time.time()),
|
|
102
|
+
provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id,
|
|
103
|
+
billed_input_tokens=in_tok, billed_output_tokens=out_tok, cost_usd=cost),
|
|
104
|
+
raw=raws, transport_error=err)
|