sys1bench 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. sys1bench/__init__.py +3 -0
  2. sys1bench/adapters/__init__.py +19 -0
  3. sys1bench/adapters/base.py +159 -0
  4. sys1bench/adapters/embed_knn.py +61 -0
  5. sys1bench/adapters/encoder_finetuned.py +114 -0
  6. sys1bench/adapters/generic_http.py +161 -0
  7. sys1bench/adapters/hybrid_router.py +49 -0
  8. sys1bench/adapters/jev_openrouter.py +172 -0
  9. sys1bench/adapters/jev_typesafe.py +165 -0
  10. sys1bench/adapters/laya_local.py +192 -0
  11. sys1bench/adapters/llm_constrained.py +104 -0
  12. sys1bench/adapters/majority_prior.py +53 -0
  13. sys1bench/adapters/mock.py +77 -0
  14. sys1bench/adapters/nli_zeroshot.py +50 -0
  15. sys1bench/adapters/regex_keyword.py +70 -0
  16. sys1bench/analysis/__init__.py +4 -0
  17. sys1bench/analysis/arms.py +46 -0
  18. sys1bench/analysis/decision_value.py +43 -0
  19. sys1bench/analysis/decomposition.py +47 -0
  20. sys1bench/analysis/meta_eval.py +67 -0
  21. sys1bench/analysis/stats.py +76 -0
  22. sys1bench/cli.py +434 -0
  23. sys1bench/data/__init__.py +37 -0
  24. sys1bench/data/canary/canary.jsonl +200 -0
  25. sys1bench/data/framings/guardrail_intent.yaml +21 -0
  26. sys1bench/data/framings/log_triage.yaml +20 -0
  27. sys1bench/data/framings/phishing_email.yaml +25 -0
  28. sys1bench/data/framings/policy_compliance.yaml +18 -0
  29. sys1bench/data/framings/support_tickets.yaml +41 -0
  30. sys1bench/framing/__init__.py +10 -0
  31. sys1bench/framing/expand.py +214 -0
  32. sys1bench/framing/perturb.py +67 -0
  33. sys1bench/generators/__init__.py +10 -0
  34. sys1bench/generators/base.py +105 -0
  35. sys1bench/generators/guardrail_intent.py +95 -0
  36. sys1bench/generators/log_triage.py +100 -0
  37. sys1bench/generators/multilingual_tickets.py +84 -0
  38. sys1bench/generators/phishing_email.py +127 -0
  39. sys1bench/generators/policy_compliance.py +90 -0
  40. sys1bench/generators/rag_relevance.py +122 -0
  41. sys1bench/generators/support_tickets.py +212 -0
  42. sys1bench/metrics/__init__.py +9 -0
  43. sys1bench/metrics/calibration.py +164 -0
  44. sys1bench/metrics/consistency.py +71 -0
  45. sys1bench/metrics/decision_value.py +43 -0
  46. sys1bench/metrics/efficiency.py +35 -0
  47. sys1bench/metrics/ordinal.py +80 -0
  48. sys1bench/metrics/robustness.py +46 -0
  49. sys1bench/metrics/selective.py +67 -0
  50. sys1bench/report/__init__.py +1 -0
  51. sys1bench/report/dashboard.py +177 -0
  52. sys1bench/report/latex.py +40 -0
  53. sys1bench/report/plots.py +143 -0
  54. sys1bench/report/results_doc.py +237 -0
  55. sys1bench/report/scorecard.py +170 -0
  56. sys1bench/runners/__init__.py +2 -0
  57. sys1bench/runners/benchmark_runner.py +149 -0
  58. sys1bench/runners/cache.py +53 -0
  59. sys1bench/runners/canary.py +43 -0
  60. sys1bench/runners/hybrid_sweep.py +37 -0
  61. sys1bench/runners/noul_consistency.py +121 -0
  62. sys1bench/runners/ordinal_probes.py +90 -0
  63. sys1bench/runners/robustness.py +112 -0
  64. sys1bench/runners/suite.py +183 -0
  65. sys1bench/runners/sweeps.py +190 -0
  66. sys1bench/schemas.py +274 -0
  67. sys1bench-0.3.0.dist-info/METADATA +195 -0
  68. sys1bench-0.3.0.dist-info/RECORD +72 -0
  69. sys1bench-0.3.0.dist-info/WHEEL +5 -0
  70. sys1bench-0.3.0.dist-info/entry_points.txt +2 -0
  71. sys1bench-0.3.0.dist-info/licenses/LICENSE +202 -0
  72. sys1bench-0.3.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,172 @@
1
+ """TypeSafe Jev via the OpenRouter alpha Decisions API.
2
+
3
+ Endpoint: POST https://openrouter.ai/api/alpha/decisions
4
+ Body: {"model": "typesafe/jev-1.13", "state": <str|obj|list>, "questions": {key: {type, instructions, criteria}}}
5
+ Types: choice (<=255 options), score (2..10 levels), boolean (our "noul").
6
+ Limits: 32k state tokens, 64k per request, 1200 rpm. Input billed at $0.042/M tokens, output free.
7
+
8
+ Response parsing is defensive: the alpha API's field names may move. Everything received is stored
9
+ verbatim in `raw`, so re-parsing later never needs another API call. The model id string returned by
10
+ the provider is recorded because `jev-latest` can move silently; pin `jev-1.13`.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import os
16
+ import random
17
+ import time
18
+ from typing import Any
19
+
20
+ import httpx
21
+
22
+ from ..schemas import (
23
+ Answer,
24
+ DecisionRequest,
25
+ DecisionResponse,
26
+ LatencyRecord,
27
+ ModelCapabilities,
28
+ ProviderRecord,
29
+ Question,
30
+ )
31
+ from .base import BaseAdapter, finalize_answer, register
32
+
33
+ DEFAULT_URL = "https://openrouter.ai/api/alpha/decisions"
34
+ PRICE_PER_M_INPUT = 0.042
35
+
36
+
37
+ def _to_vendor_question(q: Question) -> dict[str, Any]:
38
+ if q.type == "noul":
39
+ return {"type": "boolean", "instructions": q.instructions}
40
+ if q.type == "choice":
41
+ opts = [{"key": c.key, "description": c.description} for c in q.criteria] # type: ignore[union-attr]
42
+ if q.allow_abstain:
43
+ opts.append({"key": q.abstain_key, "description": "None of the listed options applies."})
44
+ return {"type": "choice", "instructions": q.instructions, "criteria": opts}
45
+ levels = [{"level": c.level, "description": c.description} for c in q.criteria] # type: ignore[union-attr]
46
+ return {"type": "score", "instructions": q.instructions, "criteria": levels}
47
+
48
+
49
+ def _first(d: dict, *names: str, default=None):
50
+ for n in names:
51
+ if n in d and d[n] is not None:
52
+ return d[n]
53
+ return default
54
+
55
+
56
+ def parse_answer(q: Question, payload: dict[str, Any]) -> Answer:
57
+ """Map a vendor answer object to the contract. Handles the shapes seen in public harnesses:
58
+ choice: {"choice": key, "probabilities": {key: p} | [p...], "confidence": c}
59
+ score: {"score"|"expected": x, "distribution"|"probabilities": {...}|[...], "confidence": c}
60
+ boolean:{"probability"|"p_true"|"value": p, "confidence": c}
61
+ """
62
+ keys = q.option_keys
63
+ conf = _first(payload, "confidence")
64
+ if q.type == "noul":
65
+ p = _first(payload, "probability", "p_true", "prob", "value")
66
+ if isinstance(p, bool):
67
+ p = 1.0 if p else 0.0
68
+ if p is None:
69
+ return Answer.failed(q, "no probability field")
70
+ return finalize_answer(q, [float(p), 1.0 - float(p)], confidence=conf)
71
+ dist = _first(payload, "probabilities", "distribution", "probs")
72
+ if dist is None:
73
+ return Answer.failed(q, "no distribution field")
74
+ if isinstance(dist, dict):
75
+ probs = [float(dist.get(k, dist.get(int(k) if k.isdigit() else k, 0.0))) for k in keys]
76
+ extra = set(map(str, dist)) - set(keys)
77
+ abst = q.allow_abstain and q.abstain_key in extra
78
+ elif isinstance(dist, list):
79
+ if all(isinstance(x, dict) for x in dist):
80
+ m = {str(_first(x, "key", "level", "option")): float(_first(x, "probability", "p", "prob")) for x in dist}
81
+ probs = [m.get(k, 0.0) for k in keys]
82
+ abst = q.allow_abstain and q.abstain_key in m
83
+ else:
84
+ probs = [float(x) for x in dist[: len(keys)]]
85
+ abst = False
86
+ else:
87
+ return Answer.failed(q, f"unparseable distribution {type(dist).__name__}")
88
+ if abst:
89
+ # abstain mass is recorded, then folded out so the vector stays over the manifest's options
90
+ total = sum(probs)
91
+ if total > 0:
92
+ probs = [p / total for p in probs]
93
+ ans = finalize_answer(q, probs, confidence=conf, abstained=bool(abst and max(probs) < 0.5))
94
+ return ans
95
+
96
+
97
+ @register("jev_openrouter")
98
+ class JevOpenRouterAdapter(BaseAdapter):
99
+ def __init__(self, model_id: str = "typesafe/jev-1.13", url: str = DEFAULT_URL, api_key: str | None = None,
100
+ timeout_s: float = 30.0, max_retries: int = 5, route: str = "openrouter", **tunables) -> None:
101
+ super().__init__(model_id, **tunables)
102
+ if model_id.endswith("latest"):
103
+ raise ValueError("pin an explicit Jev version (e.g. typesafe/jev-1.13); 'latest' moves silently")
104
+ self.url, self.timeout_s, self.max_retries, self.route = url, timeout_s, max_retries, route
105
+ self.api_key = api_key or os.environ.get("OPENROUTER_API_KEY")
106
+ self._client = httpx.Client(timeout=timeout_s)
107
+
108
+ @property
109
+ def capabilities(self) -> ModelCapabilities:
110
+ return ModelCapabilities(name=self.model_id, deployment="hosted", max_options=255, max_score_levels=10,
111
+ max_state_tokens=32_000, emits_confidence=True, supports_native_abstain=False,
112
+ supports_batching=True, languages=None)
113
+
114
+ def _post(self, body: dict[str, Any]) -> tuple[dict[str, Any], float, dict[str, str]]:
115
+ if not self.api_key:
116
+ raise RuntimeError("OPENROUTER_API_KEY not set")
117
+ headers = {"Authorization": f"Bearer {self.api_key}", "Content-Type": "application/json"}
118
+ delay = 0.5
119
+ last_exc: Exception | None = None
120
+ for attempt in range(self.max_retries):
121
+ t0 = time.perf_counter()
122
+ try:
123
+ r = self._client.post(self.url, json=body, headers=headers)
124
+ ms = (time.perf_counter() - t0) * 1000
125
+ if r.status_code in (429, 500, 502, 503, 504):
126
+ raise httpx.HTTPStatusError(f"{r.status_code}", request=r.request, response=r)
127
+ r.raise_for_status()
128
+ return r.json(), ms, dict(r.headers)
129
+ except (httpx.TransportError, httpx.HTTPStatusError) as e:
130
+ last_exc = e
131
+ time.sleep(delay + random.uniform(0, delay))
132
+ delay = min(delay * 2, 16.0)
133
+ raise RuntimeError(f"jev request failed after {self.max_retries} attempts: {last_exc}")
134
+
135
+ def parse_raw_answer(self, q: Question, payload: dict) -> Answer:
136
+ return parse_answer(q, payload)
137
+
138
+ def decide(self, request: DecisionRequest) -> DecisionResponse:
139
+ body = {"model": self.model_id, "state": request.state,
140
+ "questions": {k: _to_vendor_question(q) for k, q in request.questions.items()}}
141
+ try:
142
+ data, ms, headers = self._post(body)
143
+ except Exception as e:
144
+ return DecisionResponse(
145
+ answers={k: Answer.failed(q, "transport") for k, q in request.questions.items()},
146
+ latency=LatencyRecord(client_ms=float("nan"), route=self.route, timestamp=time.time()),
147
+ provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id, route=self.route),
148
+ transport_error=str(e),
149
+ )
150
+ raw_answers = _first(data, "answers", "decisions", "results", default={})
151
+ answers = {}
152
+ for k, q in request.questions.items():
153
+ payload = raw_answers.get(k) if isinstance(raw_answers, dict) else None
154
+ answers[k] = parse_answer(q, payload) if isinstance(payload, dict) else Answer.failed(q, "missing answer")
155
+ usage = data.get("usage", {}) or {}
156
+ in_tok = _first(usage, "prompt_tokens", "input_tokens")
157
+ cost = (in_tok / 1e6) * PRICE_PER_M_INPUT if in_tok else _first(usage, "cost")
158
+ server_ms = None
159
+ for h in ("x-processing-ms", "openrouter-processing-ms", "x-response-time"):
160
+ if h in headers:
161
+ try:
162
+ server_ms = float(str(headers[h]).rstrip("ms"))
163
+ except ValueError:
164
+ pass
165
+ return DecisionResponse(
166
+ answers=answers,
167
+ latency=LatencyRecord(client_ms=ms, server_ms=server_ms, route=self.route, timestamp=time.time()),
168
+ provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id,
169
+ model_id_returned=_first(data, "model"), generation_id=_first(data, "id", "generation_id"),
170
+ route=self.route, billed_input_tokens=in_tok, billed_output_tokens=0, cost_usd=cost),
171
+ raw=data,
172
+ )
@@ -0,0 +1,165 @@
1
+ """TypeSafe Jev via the first-party API (preferred over the OpenRouter alpha route).
2
+
3
+ Endpoint: POST https://api.typesafe.ai/v1/systemone Authorization: Bearer <TypeSafe_API_KEY>
4
+ Body: {"state": <str|obj|list>, "model": "jev-1.13.0", "questions": {id: Question}}
5
+ Question: choice -> {"type": "choice", "instructions": str|obj, "criteria": {option: description|null}} (<=255 options)
6
+ score -> {"type": "score", "instructions": str|obj, "criteria": [level_description, ...]} (2..10 levels)
7
+ noul -> {"type": "noul", "instructions": str|obj, "criteria": {"true": ..., "false": ...}?}
8
+ Answer: choice -> {"type","choice","probabilities":{option: p},"confidence"}
9
+ score -> {"type","score" (expected level index),"legend":{"0": desc,...},"probabilities":{"0": p,...},"confidence"}
10
+ noul -> {"type","noul": P(yes)}
11
+ Response also carries "model" (versioned id that answered) and "usage": {input_tokens, output_tokens}.
12
+ Limits (models page, 2026-09-22): 64k tokens per request, 32k for state plus longest question, 1200 rpm, 250k tok/s.
13
+ Price: $0.042 per million input tokens; output free.
14
+
15
+ Score probabilities are keyed by *level index*, so they are mapped back onto the manifest's level values by
16
+ position. Option order in `criteria` is preserved as JSON object order, which is what the permutation suite varies.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import os
22
+ import random
23
+ import time
24
+ from pathlib import Path
25
+ from typing import Any
26
+
27
+ import httpx
28
+
29
+ from ..schemas import Answer, DecisionRequest, DecisionResponse, LatencyRecord, ModelCapabilities, ProviderRecord, Question
30
+ from .base import BaseAdapter, finalize_answer, register
31
+
32
+ DEFAULT_URL = "https://api.typesafe.ai/v1/systemone"
33
+ PRICE_PER_M_INPUT = 0.042
34
+ ENV_KEYS = ("TypeSafe_API_KEY", "TYPESAFE_API_KEY")
35
+
36
+
37
+ def _load_dotenv_key() -> str | None:
38
+ for k in ENV_KEYS:
39
+ if os.environ.get(k):
40
+ return os.environ[k]
41
+ for candidate in (Path.cwd() / ".env", Path(__file__).resolve().parents[3] / ".env"):
42
+ if candidate.exists():
43
+ for line in candidate.read_text().splitlines():
44
+ line = line.strip()
45
+ if "=" in line and not line.startswith("#"):
46
+ name, _, val = line.partition("=")
47
+ if name.strip() in ENV_KEYS and val.strip():
48
+ return val.strip().strip('"').strip("'")
49
+ return None
50
+
51
+
52
+ def to_vendor_question(q: Question) -> dict[str, Any]:
53
+ if q.type == "noul":
54
+ return {"type": "noul", "instructions": q.instructions}
55
+ if q.type == "choice":
56
+ crit: dict[str, Any] = {c.key: (c.description or None) for c in q.criteria} # type: ignore[union-attr]
57
+ if q.allow_abstain:
58
+ crit[q.abstain_key] = "None of the listed options applies."
59
+ return {"type": "choice", "instructions": q.instructions, "criteria": crit}
60
+ levels = [(c.description or f"level {c.level}") for c in q.criteria] # type: ignore[union-attr]
61
+ return {"type": "score", "instructions": q.instructions, "criteria": levels}
62
+
63
+
64
+ def parse_answer(q: Question, payload: dict[str, Any]) -> Answer:
65
+ keys = q.option_keys
66
+ if q.type == "noul":
67
+ p = payload.get("noul")
68
+ if p is None:
69
+ return Answer.failed(q, "no noul field")
70
+ return finalize_answer(q, [float(p), 1.0 - float(p)])
71
+ dist = payload.get("probabilities")
72
+ if not isinstance(dist, dict):
73
+ return Answer.failed(q, "no probabilities map")
74
+ conf = payload.get("confidence")
75
+ if q.type == "score":
76
+ # keyed by level index as string, in the order we sent the levels
77
+ probs = [float(dist.get(str(i), 0.0)) for i in range(len(keys))]
78
+ return finalize_answer(q, probs, confidence=conf)
79
+ probs = [float(dist.get(k, 0.0)) for k in keys]
80
+ abstained = False
81
+ if q.allow_abstain and q.abstain_key in dist:
82
+ p_abst = float(dist[q.abstain_key])
83
+ abstained = p_abst >= max(probs)
84
+ s = sum(probs)
85
+ probs = [p / s for p in probs] if s > 0 else [1.0 / len(keys)] * len(keys)
86
+ return finalize_answer(q, probs, confidence=conf, abstained=abstained)
87
+
88
+
89
+ @register("jev_typesafe")
90
+ class JevTypeSafeAdapter(BaseAdapter):
91
+ def __init__(self, model_id: str = "jev-1.13.0", url: str = DEFAULT_URL, api_key: str | None = None,
92
+ timeout_s: float = 30.0, max_retries: int = 6, allow_alias: bool = False, **tunables) -> None:
93
+ super().__init__(model_id, **tunables)
94
+ if (model_id.endswith("latest") or model_id.endswith("preview")) and not allow_alias:
95
+ raise ValueError("pin a versioned Jev id (e.g. jev-1.13.0); aliases move silently. Pass allow_alias=True to override.")
96
+ self.url, self.timeout_s, self.max_retries = url, timeout_s, max_retries
97
+ self.api_key = api_key or _load_dotenv_key()
98
+ self._client = httpx.Client(timeout=timeout_s, http2=False)
99
+
100
+ @property
101
+ def capabilities(self) -> ModelCapabilities:
102
+ return ModelCapabilities(name=self.model_id, deployment="hosted", max_options=255, max_score_levels=10,
103
+ max_state_tokens=32_000, emits_confidence=True, supports_native_abstain=False,
104
+ supports_batching=True, languages={"en"})
105
+
106
+ def _post(self, body: dict[str, Any]) -> tuple[dict[str, Any], float, dict[str, str]]:
107
+ if not self.api_key:
108
+ raise RuntimeError("TypeSafe_API_KEY not set (env or .env)")
109
+ headers = {"Authorization": f"Bearer {self.api_key}", "Content-Type": "application/json"}
110
+ delay, last = 0.5, None
111
+ for _ in range(self.max_retries):
112
+ t0 = time.perf_counter()
113
+ try:
114
+ r = self._client.post(self.url, json=body, headers=headers)
115
+ ms = (time.perf_counter() - t0) * 1000
116
+ if r.status_code in (429, 500, 502, 503, 504, 529):
117
+ ra = r.headers.get("retry-after")
118
+ if ra:
119
+ try:
120
+ delay = max(delay, float(ra))
121
+ except ValueError:
122
+ pass
123
+ raise httpx.HTTPStatusError(f"{r.status_code}: {r.text[:200]}", request=r.request, response=r)
124
+ if 400 <= r.status_code < 500 and r.status_code != 429:
125
+ # request rejected (400 over the 32k state limit, 413, 422 malformed): never retry, classify as vendor rejection
126
+ raise ValueError(f"{r.status_code} rejected: {r.text[:300]}")
127
+ r.raise_for_status()
128
+ return r.json(), ms, dict(r.headers)
129
+ except (httpx.TransportError, httpx.HTTPStatusError) as e:
130
+ last = e
131
+ time.sleep(delay + random.uniform(0, delay))
132
+ delay = min(delay * 2, 20.0)
133
+ raise RuntimeError(f"typesafe request failed after {self.max_retries} attempts: {last}")
134
+
135
+ def parse_raw_answer(self, q: Question, payload: dict) -> Answer:
136
+ return parse_answer(q, payload)
137
+
138
+ def decide(self, request: DecisionRequest) -> DecisionResponse:
139
+ body = {"state": request.state, "model": self.model_id,
140
+ "questions": {k: to_vendor_question(q) for k, q in request.questions.items()}}
141
+ try:
142
+ data, ms, headers = self._post(body)
143
+ except Exception as e:
144
+ kind = "rejected_by_vendor" if isinstance(e, ValueError) else "transport"
145
+ return DecisionResponse(
146
+ answers={k: Answer.failed(q, kind) for k, q in request.questions.items()},
147
+ latency=LatencyRecord(client_ms=float("nan"), route="typesafe", timestamp=time.time()),
148
+ provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id, route="typesafe"),
149
+ transport_error=str(e)[:500],
150
+ )
151
+ raw_answers = data.get("answers", {}) or {}
152
+ answers = {k: (parse_answer(q, raw_answers[k]) if isinstance(raw_answers.get(k), dict) else Answer.failed(q, "missing answer"))
153
+ for k, q in request.questions.items()}
154
+ usage = data.get("usage", {}) or {}
155
+ in_tok = usage.get("input_tokens")
156
+ return DecisionResponse(
157
+ answers=answers,
158
+ latency=LatencyRecord(client_ms=ms, route="typesafe", timestamp=time.time()),
159
+ provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id, model_id_returned=data.get("model"),
160
+ version_hash=str(data.get("model") or self.model_id),
161
+ generation_id=headers.get("x-request-id") or headers.get("request-id"), route="typesafe",
162
+ billed_input_tokens=in_tok, billed_output_tokens=usage.get("output_tokens"),
163
+ cost_usd=(in_tok / 1e6 * PRICE_PER_M_INPUT) if in_tok is not None else None),
164
+ raw=data,
165
+ )
@@ -0,0 +1,192 @@
1
+ """Convai Laya, run locally (Apache 2.0) through the `laya` package's Router (verified against laya 0.3.4).
2
+
3
+ from laya import Router
4
+ router = Router(preload=True, device="cuda")
5
+ out = router.predict(state, questions, model=None | "english" | "multilingual" | "typed-decisions")
6
+
7
+ Question shape matches TypeSafe's: choice criteria is {option: description}, score criteria is an ordered list of
8
+ level descriptions, noul has no criteria. Answers:
9
+ choice -> {"type","choice","probabilities":{opt: p},"confidence","action":{"act_probability"}}
10
+ score -> {"type","score","legend","probabilities":{"0": p,...},"confidence","action":{...}}
11
+ noul -> {"type","noul": P(yes),"confidence","action":{...}}
12
+ `action.act_probability` is Laya's act/escalate head; we treat act_probability < 0.5 as a native abstention.
13
+ Result also carries "routing": {"model","repo","reason",...} and "usage".
14
+
15
+ Tunables: `head_max_len` and `max_len` are set on the checkpoint's `agent.cfg` before inference (Suite C ablates them;
16
+ the 77-label collapse at the default budget is a configuration effect). Checkpoint defaults: english 512/192,
17
+ multilingual 1024/256, typed-decisions 1024/256. The package prints a warning and falls back to CPU when CUDA is
18
+ unavailable or out of memory; we record the device actually used in `hardware` so latency tables stay honest.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import platform
24
+ import time
25
+ from typing import Any
26
+
27
+ from ..schemas import Answer, DecisionRequest, DecisionResponse, LatencyRecord, ModelCapabilities, ProviderRecord, Question
28
+ from .base import BaseAdapter, FatalAdapterError, finalize_answer, register
29
+
30
+ CHECKPOINTS = {"english": ("convaiinnovations/laya", 512, 192), "multilingual": ("convaiinnovations/laya/multilingual", 1024, 256),
31
+ "typed-decisions": ("convaiinnovations/laya/typed-decisions", 1024, 256), "router": ("convaiinnovations/laya", 1024, 256)}
32
+ ALIASES = {"laya": "english", "laya-multilingual": "multilingual", "laya-typed-decisions": "typed-decisions"}
33
+
34
+
35
+ def to_vendor_question(q: Question) -> dict[str, Any]:
36
+ if q.type == "noul":
37
+ return {"type": "noul", "instructions": q.instructions}
38
+ if q.type == "choice":
39
+ crit = {c.key: (c.description or c.key) for c in q.criteria} # type: ignore[union-attr]
40
+ if q.allow_abstain:
41
+ crit[q.abstain_key] = "None of the listed options applies."
42
+ return {"type": "choice", "instructions": q.instructions, "criteria": crit}
43
+ return {"type": "score", "instructions": q.instructions,
44
+ "criteria": [(c.description or f"level {c.level}") for c in q.criteria]} # type: ignore[union-attr]
45
+
46
+
47
+ def parse_answer(q: Question, payload: dict[str, Any], abstain_threshold: float = 0.5) -> Answer:
48
+ keys = q.option_keys
49
+ conf = payload.get("confidence")
50
+ act = (payload.get("action") or {}).get("act_probability")
51
+ abstained = act is not None and float(act) < abstain_threshold
52
+ if q.type == "noul":
53
+ p = payload.get("noul")
54
+ if p is None:
55
+ return Answer.failed(q, "no noul field")
56
+ a = finalize_answer(q, [float(p), 1.0 - float(p)], confidence=None, abstained=abstained)
57
+ else:
58
+ dist = payload.get("probabilities")
59
+ if not isinstance(dist, dict):
60
+ return Answer.failed(q, "no probabilities map")
61
+ if q.type == "score":
62
+ probs = [float(dist.get(str(i), 0.0)) for i in range(len(keys))]
63
+ else:
64
+ probs = [float(dist.get(k, 0.0)) for k in keys]
65
+ if q.allow_abstain and q.abstain_key in dist:
66
+ abstained = abstained or float(dist[q.abstain_key]) >= max(probs)
67
+ s = sum(probs)
68
+ probs = [x / s for x in probs] if s > 0 else [1.0 / len(keys)] * len(keys)
69
+ a = finalize_answer(q, probs, confidence=conf, abstained=abstained)
70
+ if a.ok and act is not None:
71
+ a = a.model_copy(update={"raw_prob_sum": a.raw_prob_sum})
72
+ return a
73
+
74
+
75
+ @register("laya_local")
76
+ class LayaLocalAdapter(BaseAdapter):
77
+ def __init__(self, model_id: str = "convaiinnovations/laya", checkpoint: str = "english", device: str = "cuda",
78
+ head_max_len: int | None = None, max_len: int | None = None, batch_size: int = 1,
79
+ hardware: str | None = None, abstain_threshold: float = 0.5, strict_device: bool = True, **tunables) -> None:
80
+ super().__init__(model_id, **tunables)
81
+ checkpoint = ALIASES.get(checkpoint, checkpoint)
82
+ if checkpoint not in CHECKPOINTS:
83
+ raise ValueError(f"unknown Laya checkpoint {checkpoint!r}; known: {sorted(CHECKPOINTS)} (+ aliases {sorted(ALIASES)})")
84
+ self.checkpoint, self.device, self.batch_size, self.abstain_threshold = checkpoint, device, batch_size, abstain_threshold
85
+ self.strict_device = strict_device
86
+ _, dflt_max, dflt_head = CHECKPOINTS[checkpoint]
87
+ self.head_max_len = head_max_len or dflt_head
88
+ self.max_len = max_len or dflt_max
89
+ self.hardware = hardware
90
+ self.tunables.update({"head_max_len": self.head_max_len, "max_len": self.max_len, "checkpoint": checkpoint})
91
+ self._router = None
92
+
93
+ @property
94
+ def capabilities(self) -> ModelCapabilities:
95
+ return ModelCapabilities(name=f"laya:{self.checkpoint}", deployment="local", max_options=255,
96
+ max_state_tokens=self.max_len - self.head_max_len, emits_confidence=True,
97
+ supports_native_abstain=True, supports_batching=True,
98
+ languages={"en"} if self.checkpoint == "english" else None,
99
+ tunables={"head_max_len": self.head_max_len, "max_len": self.max_len})
100
+
101
+ def _load(self):
102
+ if self._router is not None:
103
+ return self._router
104
+ try:
105
+ from laya import Router # type: ignore
106
+ except ImportError as e: # pragma: no cover
107
+ raise RuntimeError("install Laya: pip install laya (github.com/NandhaKishorM/laya)") from e
108
+ # Lazy router: load only the checkpoint we evaluate (preloading all three needs ~6 GB and, on a shared GPU,
109
+ # silently lands on CPU). "router" mode keeps the full preload because it dispatches per language.
110
+ if self.checkpoint == "router":
111
+ router = Router(preload=True, device=self.device)
112
+ agents = list((getattr(router, "_agents", {}) or {}).values())
113
+ else:
114
+ router = Router(preload=False, device=self.device)
115
+ agents = [router.load(self.checkpoint)]
116
+ for ag in agents:
117
+ if hasattr(ag, "cfg"):
118
+ ag.cfg["head_max_len"] = self.head_max_len
119
+ ag.cfg["max_len"] = self.max_len
120
+ dev = str(getattr(ag, "device", ""))
121
+ if self.device.startswith("cuda") and "cuda" not in dev and hasattr(ag, "model"):
122
+ try: # the package falls back to CPU when CUDA context creation fails at import; retry the move
123
+ import torch # type: ignore
124
+
125
+ ag.model.to("cuda")
126
+ ag.device = torch.device("cuda") if hasattr(torch, "device") else "cuda"
127
+ except Exception:
128
+ pass
129
+ dev = str(getattr(ag, "device", ""))
130
+ if self.device.startswith("cuda") and "cuda" not in dev and self.strict_device:
131
+ del router, agents # release the half-loaded model before aborting
132
+ raise FatalAdapterError(f"Laya checkpoint {self.checkpoint!r} loaded on {dev or 'cpu'} although device={self.device!r} was requested; "
133
+ "refusing to record CPU latency as GPU. Free GPU memory or pass strict_device=false.")
134
+ ag = agents[0] if agents else None
135
+ dev = str(getattr(ag, "device", self.device)) if ag is not None else self.device
136
+ if self.hardware is None:
137
+ gpu = None
138
+ try:
139
+ import torch # type: ignore
140
+
141
+ if "cuda" in dev and torch.cuda.is_available():
142
+ gpu = torch.cuda.get_device_name(0)
143
+ except Exception: # pragma: no cover
144
+ pass
145
+ self.hardware = f"{gpu} ({dev})" if gpu else f"CPU {platform.machine()} ({dev})"
146
+ self._router = router # only after every check passed
147
+ return self._router
148
+
149
+ def parse_raw_answer(self, q: Question, payload: dict) -> Answer:
150
+ return parse_answer(q, payload, self.abstain_threshold)
151
+
152
+ def decide(self, request: DecisionRequest) -> DecisionResponse:
153
+ router = self._load()
154
+ vq = {k: to_vendor_question(q) for k, q in request.questions.items()}
155
+ model = None if self.checkpoint == "router" else self.checkpoint
156
+ t0 = time.perf_counter()
157
+ try:
158
+ out = router.predict(request.state, vq, model=model)
159
+ except Exception as e:
160
+ # laya raises ValueError("... options exceed head_max_len=N") when the option strings do not fit the budget:
161
+ # that is the model refusing the request, not a harness bug, so it is classified like a vendor 4xx.
162
+ kind = "rejected_by_model" if isinstance(e, ValueError) else "adapter_exception"
163
+ return DecisionResponse(
164
+ answers={k: Answer.failed(q, kind) for k, q in request.questions.items()},
165
+ latency=LatencyRecord(client_ms=float("nan"), timestamp=time.time()),
166
+ provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id, hardware=self.hardware),
167
+ transport_error=repr(e)[:500])
168
+ ms = (time.perf_counter() - t0) * 1000
169
+ raw_answers = out.get("answers", {}) if isinstance(out, dict) else {}
170
+ # laya truncates silently to max_len - head_max_len state tokens and reports nothing; estimate it (~4 chars/token)
171
+ from ..generators.base import estimate_tokens
172
+
173
+ est_tokens = estimate_tokens(request.state_text())
174
+ truncated = bool(out.get("truncated", False)) if isinstance(out, dict) else False
175
+ truncated = truncated or est_tokens > (self.max_len - self.head_max_len)
176
+ answers = {}
177
+ for k, q in request.questions.items():
178
+ a = parse_answer(q, raw_answers[k], self.abstain_threshold) if isinstance(raw_answers.get(k), dict) else Answer.failed(q, "missing answer")
179
+ if a.ok and truncated:
180
+ a = a.model_copy(update={"truncated": True})
181
+ answers[k] = a
182
+ routing = out.get("routing", {}) if isinstance(out, dict) else {}
183
+ usage = out.get("usage", {}) if isinstance(out, dict) else {}
184
+ return DecisionResponse(
185
+ answers=answers,
186
+ latency=LatencyRecord(client_ms=ms, compute_ms=ms, batch_size=self.batch_size, timestamp=time.time()),
187
+ provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id,
188
+ model_id_returned=f"laya:{routing.get('model', self.checkpoint)}",
189
+ version_hash=f"{routing.get('repo', '')}|head{self.head_max_len}|max{self.max_len}",
190
+ hardware=self.hardware, billed_input_tokens=usage.get("input_tokens"), cost_usd=0.0),
191
+ raw=out if isinstance(out, dict) else {"out": str(out)},
192
+ )
@@ -0,0 +1,104 @@
1
+ """Constrained-decoding LLM baseline with first-token logprobs, so LLMs get calibration metrics.
2
+
3
+ Backend A (default): any OpenAI-compatible server (vLLM, OpenRouter) with `logprobs`. The prompt lists
4
+ options as single-letter keys; probabilities are the renormalised logprobs of the first generated token
5
+ over the option letters, which is the standard "logit-based" calibration protocol for MCQ.
6
+ Backend B: vLLM + Outlines regex-constrained generation (import guarded); same probability extraction.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import math
12
+ import os
13
+ import string
14
+ import time
15
+
16
+ import httpx
17
+
18
+ from ..schemas import (
19
+ Answer,
20
+ DecisionRequest,
21
+ DecisionResponse,
22
+ LatencyRecord,
23
+ ModelCapabilities,
24
+ ProviderRecord,
25
+ Question,
26
+ )
27
+ from .base import BaseAdapter, finalize_answer, register
28
+
29
+ LETTERS = string.ascii_uppercase + string.ascii_lowercase + string.digits
30
+
31
+
32
+ def build_prompt(state: str, q: Question) -> tuple[str, list[str]]:
33
+ keys = q.option_keys
34
+ letters = list(LETTERS[: len(keys)])
35
+ if q.type == "noul":
36
+ lines = ["A. yes", "B. no"]
37
+ elif q.type == "choice":
38
+ lines = [f"{L}. {c.key}: {c.description}" for L, c in zip(letters, q.criteria)] # type: ignore[union-attr]
39
+ else:
40
+ lines = [f"{L}. level {c.level}: {c.description}" for L, c in zip(letters, q.criteria)] # type: ignore[union-attr]
41
+ prompt = (f"State:\n{state}\n\nQuestion: {q.instructions}\nOptions:\n" + "\n".join(lines) +
42
+ "\n\nAnswer with the single letter of the best option.\nAnswer:")
43
+ return prompt, letters
44
+
45
+
46
+ @register("llm_constrained")
47
+ class LLMConstrainedAdapter(BaseAdapter):
48
+ def __init__(self, model_id: str = "Qwen/Qwen2.5-3B-Instruct", base_url: str = "http://localhost:8000/v1",
49
+ api_key_env: str = "OPENAI_API_KEY", top_logprobs: int = 20, price_per_m_input: float | None = None,
50
+ price_per_m_output: float | None = None, timeout_s: float = 60.0, **tunables) -> None:
51
+ super().__init__(model_id, **tunables)
52
+ self.base_url, self.top_logprobs = base_url.rstrip("/"), top_logprobs
53
+ self.key = os.environ.get(api_key_env, "EMPTY")
54
+ self.pi, self.po = price_per_m_input, price_per_m_output
55
+ self._client = httpx.Client(timeout=timeout_s)
56
+
57
+ @property
58
+ def capabilities(self) -> ModelCapabilities:
59
+ return ModelCapabilities(name=f"llm:{self.model_id}", deployment="hosted" if "localhost" not in self.base_url else "local",
60
+ max_options=len(LETTERS), emits_confidence=False, supports_batching=False, deterministic=False)
61
+
62
+ def _one(self, state: str, q: Question) -> tuple[Answer, dict]:
63
+ prompt, letters = build_prompt(state, q)
64
+ body = {"model": self.model_id, "prompt": prompt, "max_tokens": 1, "temperature": 0.0,
65
+ "logprobs": self.top_logprobs}
66
+ r = self._client.post(f"{self.base_url}/completions", json=body,
67
+ headers={"Authorization": f"Bearer {self.key}"})
68
+ r.raise_for_status()
69
+ data = r.json()
70
+ lp = data["choices"][0].get("logprobs", {})
71
+ top = (lp.get("top_logprobs") or [{}])[0]
72
+ scores = []
73
+ for L in letters:
74
+ cands = [v for k, v in top.items() if k.strip() == L]
75
+ scores.append(max(cands) if cands else -30.0)
76
+ m = max(scores)
77
+ w = [math.exp(s - m) for s in scores]
78
+ z = sum(w)
79
+ probs = [x / z for x in w]
80
+ return finalize_answer(q, probs), data
81
+
82
+ def decide(self, request: DecisionRequest) -> DecisionResponse:
83
+ t0 = time.perf_counter()
84
+ answers, raws, in_tok, out_tok = {}, {}, 0, 0
85
+ err = None
86
+ for k, q in request.questions.items():
87
+ try:
88
+ a, data = self._one(request.state_text(), q)
89
+ answers[k] = a
90
+ raws[k] = data
91
+ u = data.get("usage", {})
92
+ in_tok += u.get("prompt_tokens", 0)
93
+ out_tok += u.get("completion_tokens", 0)
94
+ except Exception as e: # transport or parse
95
+ answers[k] = Answer.failed(q, "transport")
96
+ err = str(e)
97
+ ms = (time.perf_counter() - t0) * 1000
98
+ cost = None
99
+ if self.pi is not None:
100
+ cost = in_tok / 1e6 * self.pi + out_tok / 1e6 * (self.po or 0.0)
101
+ return DecisionResponse(answers=answers, latency=LatencyRecord(client_ms=ms, timestamp=time.time()),
102
+ provider=ProviderRecord(adapter_id=self.adapter_id, model_id_requested=self.model_id,
103
+ billed_input_tokens=in_tok, billed_output_tokens=out_tok, cost_usd=cost),
104
+ raw=raws, transport_error=err)