jevguard-core 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- jevguard/__init__.py +63 -0
- jevguard/cache.py +331 -0
- jevguard/calibrator.py +262 -0
- jevguard/cli.py +106 -0
- jevguard/client.py +440 -0
- jevguard/exceptions.py +59 -0
- jevguard/memory.py +188 -0
- jevguard/models.py +192 -0
- jevguard/optimizer.py +241 -0
- jevguard_core-1.0.0.dist-info/METADATA +293 -0
- jevguard_core-1.0.0.dist-info/RECORD +15 -0
- jevguard_core-1.0.0.dist-info/WHEEL +5 -0
- jevguard_core-1.0.0.dist-info/entry_points.txt +2 -0
- jevguard_core-1.0.0.dist-info/licenses/LICENSE +21 -0
- jevguard_core-1.0.0.dist-info/top_level.txt +1 -0
jevguard/__init__.py
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""
|
|
2
|
+
JevGuard - High-Performance Deterministic Runtime for TypeSafe AI / Jev.
|
|
3
|
+
Provides closed-world optimization, certainty calibration, state pruning,
|
|
4
|
+
zero-token deterministic caching, and episodic session memory.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from .models import (
|
|
8
|
+
Question,
|
|
9
|
+
Noul,
|
|
10
|
+
Score,
|
|
11
|
+
Choice,
|
|
12
|
+
NoulAnswer,
|
|
13
|
+
ScoreAnswer,
|
|
14
|
+
ChoiceAnswer,
|
|
15
|
+
EvaluationResponse,
|
|
16
|
+
EvaluationResult
|
|
17
|
+
)
|
|
18
|
+
from .client import JevGuardClient
|
|
19
|
+
from .optimizer import StatePruner, QuestionOptimizer, ESCAPE_OPTION_KEY
|
|
20
|
+
from .calibrator import ResponseCalibrator, CertaintyCalibrator
|
|
21
|
+
from .cache import DeterministicCache, SemanticCache
|
|
22
|
+
from .memory import EpisodicMemory
|
|
23
|
+
from .exceptions import (
|
|
24
|
+
JevGuardError,
|
|
25
|
+
JevGuardConfigError,
|
|
26
|
+
JevGuardNetworkError,
|
|
27
|
+
JevGuardTimeoutError,
|
|
28
|
+
JevGuardHTTPError,
|
|
29
|
+
JevGuardAuthenticationError,
|
|
30
|
+
JevGuardRateLimitError,
|
|
31
|
+
JevGuardServerError
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
__version__ = "1.0.0"
|
|
35
|
+
|
|
36
|
+
__all__ = [
|
|
37
|
+
"JevGuardClient",
|
|
38
|
+
"Question",
|
|
39
|
+
"Noul",
|
|
40
|
+
"Score",
|
|
41
|
+
"Choice",
|
|
42
|
+
"NoulAnswer",
|
|
43
|
+
"ScoreAnswer",
|
|
44
|
+
"ChoiceAnswer",
|
|
45
|
+
"EvaluationResponse",
|
|
46
|
+
"EvaluationResult",
|
|
47
|
+
"StatePruner",
|
|
48
|
+
"QuestionOptimizer",
|
|
49
|
+
"ResponseCalibrator",
|
|
50
|
+
"CertaintyCalibrator",
|
|
51
|
+
"DeterministicCache",
|
|
52
|
+
"SemanticCache",
|
|
53
|
+
"EpisodicMemory",
|
|
54
|
+
"ESCAPE_OPTION_KEY",
|
|
55
|
+
"JevGuardError",
|
|
56
|
+
"JevGuardConfigError",
|
|
57
|
+
"JevGuardNetworkError",
|
|
58
|
+
"JevGuardTimeoutError",
|
|
59
|
+
"JevGuardHTTPError",
|
|
60
|
+
"JevGuardAuthenticationError",
|
|
61
|
+
"JevGuardRateLimitError",
|
|
62
|
+
"JevGuardServerError"
|
|
63
|
+
]
|
jevguard/cache.py
ADDED
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
"""
|
|
2
|
+
jevguard.cache - Zero-Token Deterministic Hashing Cache for TypeSafe AI / Jev.
|
|
3
|
+
Stores decisions indexed by SHA-256 fingerprints of canonical JSON with volatile key masking.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import contextlib
|
|
7
|
+
import hashlib
|
|
8
|
+
import json
|
|
9
|
+
import logging
|
|
10
|
+
import sqlite3
|
|
11
|
+
import threading
|
|
12
|
+
import time
|
|
13
|
+
from typing import Any, Dict, Generator, Iterable, Optional, Set
|
|
14
|
+
|
|
15
|
+
import os
|
|
16
|
+
|
|
17
|
+
logger = logging.getLogger("jevguard.cache")
|
|
18
|
+
|
|
19
|
+
DEFAULT_VOLATILE_KEYS: Set[str] = {
|
|
20
|
+
"timestamp",
|
|
21
|
+
"trace_id",
|
|
22
|
+
"span_id",
|
|
23
|
+
"request_id",
|
|
24
|
+
"correlation_id",
|
|
25
|
+
"nonce",
|
|
26
|
+
"traceid",
|
|
27
|
+
"requestid",
|
|
28
|
+
"x_trace_id",
|
|
29
|
+
"x_request_id",
|
|
30
|
+
"x_correlation_id",
|
|
31
|
+
"xtraceid",
|
|
32
|
+
"xrequestid"
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class DeterministicCache:
|
|
37
|
+
"""Provides instant 0-token response retrieval for repeated queries."""
|
|
38
|
+
|
|
39
|
+
def __init__(
|
|
40
|
+
self,
|
|
41
|
+
db_path: str = "jevguard_cache.db",
|
|
42
|
+
max_memory_items: int = 500,
|
|
43
|
+
default_ignore_keys: Optional[Iterable[str]] = None,
|
|
44
|
+
ttl_seconds: Optional[float] = None
|
|
45
|
+
):
|
|
46
|
+
self.db_path = db_path
|
|
47
|
+
self.max_memory_items = max_memory_items
|
|
48
|
+
raw_ttl = os.environ.get("JEVGUARD_CACHE_TTL")
|
|
49
|
+
if raw_ttl is not None:
|
|
50
|
+
try:
|
|
51
|
+
self.ttl_seconds = float(raw_ttl)
|
|
52
|
+
except (ValueError, TypeError):
|
|
53
|
+
self.ttl_seconds = 3600.0
|
|
54
|
+
elif ttl_seconds is not None:
|
|
55
|
+
self.ttl_seconds = float(ttl_seconds)
|
|
56
|
+
else:
|
|
57
|
+
self.ttl_seconds = 3600.0
|
|
58
|
+
self.default_ignore_keys = (
|
|
59
|
+
set(DEFAULT_VOLATILE_KEYS).union({str(k).strip().lower().replace("-", "_") for k in default_ignore_keys})
|
|
60
|
+
if default_ignore_keys is not None
|
|
61
|
+
else set(DEFAULT_VOLATILE_KEYS)
|
|
62
|
+
)
|
|
63
|
+
self._memory_lru: Dict[str, Dict[str, Any]] = {}
|
|
64
|
+
self._lock = threading.RLock()
|
|
65
|
+
self._local = threading.local()
|
|
66
|
+
self._shared_conn: Optional[sqlite3.Connection] = None
|
|
67
|
+
|
|
68
|
+
if self.db_path == ":memory:":
|
|
69
|
+
self._shared_conn = sqlite3.connect(":memory:", check_same_thread=False, timeout=60.0)
|
|
70
|
+
self._shared_conn.row_factory = sqlite3.Row
|
|
71
|
+
try:
|
|
72
|
+
self._shared_conn.execute("PRAGMA busy_timeout = 60000")
|
|
73
|
+
except Exception:
|
|
74
|
+
pass
|
|
75
|
+
|
|
76
|
+
self.stats = {
|
|
77
|
+
"hits": 0,
|
|
78
|
+
"misses": 0,
|
|
79
|
+
"tokens_saved": 0
|
|
80
|
+
}
|
|
81
|
+
self._init_db()
|
|
82
|
+
|
|
83
|
+
@contextlib.contextmanager
|
|
84
|
+
def _get_connection(self) -> Generator[sqlite3.Connection, None, None]:
|
|
85
|
+
if self._shared_conn is not None:
|
|
86
|
+
with self._lock:
|
|
87
|
+
yield self._shared_conn
|
|
88
|
+
else:
|
|
89
|
+
conn = getattr(self._local, "conn", None)
|
|
90
|
+
if conn is None:
|
|
91
|
+
conn = sqlite3.connect(self.db_path, timeout=60.0)
|
|
92
|
+
conn.row_factory = sqlite3.Row
|
|
93
|
+
try:
|
|
94
|
+
conn.execute("PRAGMA busy_timeout = 60000")
|
|
95
|
+
conn.execute("PRAGMA journal_mode=WAL")
|
|
96
|
+
conn.execute("PRAGMA synchronous=NORMAL")
|
|
97
|
+
except Exception as e:
|
|
98
|
+
logger.debug("PRAGMA setup notice: %s", e)
|
|
99
|
+
self._local.conn = conn
|
|
100
|
+
with self._lock:
|
|
101
|
+
yield conn
|
|
102
|
+
|
|
103
|
+
def _init_db(self) -> None:
|
|
104
|
+
with self._get_connection() as conn:
|
|
105
|
+
conn.execute("""
|
|
106
|
+
CREATE TABLE IF NOT EXISTS evaluation_cache (
|
|
107
|
+
fingerprint TEXT PRIMARY KEY,
|
|
108
|
+
model TEXT NOT NULL,
|
|
109
|
+
response_json TEXT NOT NULL,
|
|
110
|
+
created_at REAL NOT NULL,
|
|
111
|
+
hit_count INTEGER DEFAULT 0,
|
|
112
|
+
tokens_estimate INTEGER DEFAULT 0
|
|
113
|
+
)
|
|
114
|
+
""")
|
|
115
|
+
conn.commit()
|
|
116
|
+
|
|
117
|
+
@classmethod
|
|
118
|
+
def _strip_volatile_keys(cls, data: Any, ignore_keys: Set[str], seen: Optional[Set[int]] = None) -> Any:
|
|
119
|
+
if seen is None:
|
|
120
|
+
seen = set()
|
|
121
|
+
|
|
122
|
+
if isinstance(data, (dict, list, tuple, set, frozenset)):
|
|
123
|
+
obj_id = id(data)
|
|
124
|
+
if obj_id in seen:
|
|
125
|
+
return "<cyclic_ref>"
|
|
126
|
+
seen.add(obj_id)
|
|
127
|
+
else:
|
|
128
|
+
obj_id = None
|
|
129
|
+
|
|
130
|
+
try:
|
|
131
|
+
if isinstance(data, dict):
|
|
132
|
+
cleaned = {}
|
|
133
|
+
for k, v in data.items():
|
|
134
|
+
norm_k = str(k).strip().lower().replace("-", "_")
|
|
135
|
+
if norm_k in ignore_keys:
|
|
136
|
+
continue
|
|
137
|
+
cleaned[str(k)] = cls._strip_volatile_keys(v, ignore_keys, seen=seen)
|
|
138
|
+
return cleaned
|
|
139
|
+
elif isinstance(data, list):
|
|
140
|
+
return [cls._strip_volatile_keys(item, ignore_keys, seen=seen) for item in data]
|
|
141
|
+
elif isinstance(data, tuple):
|
|
142
|
+
return tuple(cls._strip_volatile_keys(item, ignore_keys, seen=seen) for item in data)
|
|
143
|
+
elif isinstance(data, (set, frozenset)):
|
|
144
|
+
items = [cls._strip_volatile_keys(item, ignore_keys, seen=seen) for item in data]
|
|
145
|
+
try:
|
|
146
|
+
return sorted(items)
|
|
147
|
+
except TypeError:
|
|
148
|
+
return sorted(items, key=lambda x: str(x))
|
|
149
|
+
return data
|
|
150
|
+
finally:
|
|
151
|
+
if obj_id is not None:
|
|
152
|
+
seen.remove(obj_id)
|
|
153
|
+
|
|
154
|
+
@classmethod
|
|
155
|
+
def compute_fingerprint(
|
|
156
|
+
cls,
|
|
157
|
+
model: str = "jev-latest",
|
|
158
|
+
state: Any = None,
|
|
159
|
+
wire_questions: Optional[Dict[str, Any]] = None,
|
|
160
|
+
ignore_keys: Optional[Iterable[str]] = None
|
|
161
|
+
) -> str:
|
|
162
|
+
# Detect positional invocation without model: compute_fingerprint(state, questions, ignore_keys)
|
|
163
|
+
if isinstance(model, (dict, list)):
|
|
164
|
+
if wire_questions is not None and not isinstance(wire_questions, dict):
|
|
165
|
+
ignore_keys = wire_questions
|
|
166
|
+
wire_questions = state if isinstance(state, dict) else {}
|
|
167
|
+
state = model
|
|
168
|
+
model = "jev-latest"
|
|
169
|
+
|
|
170
|
+
target_model = model or "jev-latest"
|
|
171
|
+
target_state = state if state is not None else {}
|
|
172
|
+
target_questions = wire_questions if wire_questions is not None else {}
|
|
173
|
+
|
|
174
|
+
keys_to_ignore = (
|
|
175
|
+
set(DEFAULT_VOLATILE_KEYS).union({str(k).strip().lower().replace("-", "_") for k in ignore_keys})
|
|
176
|
+
if ignore_keys is not None
|
|
177
|
+
else set(DEFAULT_VOLATILE_KEYS)
|
|
178
|
+
)
|
|
179
|
+
filtered_state = cls._strip_volatile_keys(target_state, keys_to_ignore) if keys_to_ignore else target_state
|
|
180
|
+
|
|
181
|
+
canonical_struct = {
|
|
182
|
+
"model": target_model.strip().lower(),
|
|
183
|
+
"state": filtered_state,
|
|
184
|
+
"questions": target_questions
|
|
185
|
+
}
|
|
186
|
+
canonical_bytes = json.dumps(
|
|
187
|
+
canonical_struct,
|
|
188
|
+
sort_keys=True,
|
|
189
|
+
separators=(",", ":"),
|
|
190
|
+
ensure_ascii=True
|
|
191
|
+
).encode("utf-8")
|
|
192
|
+
return hashlib.sha256(canonical_bytes).hexdigest()
|
|
193
|
+
|
|
194
|
+
def set(
|
|
195
|
+
self,
|
|
196
|
+
fingerprint: str,
|
|
197
|
+
data: Dict[str, Any],
|
|
198
|
+
model: str = "jev-latest",
|
|
199
|
+
input_tokens_estimate: int = 0
|
|
200
|
+
) -> None:
|
|
201
|
+
"""Alias for put() following standard key-value cache conventions."""
|
|
202
|
+
self.put(fingerprint, model, data, input_tokens_estimate)
|
|
203
|
+
|
|
204
|
+
def get(self, fingerprint: str) -> Optional[Dict[str, Any]]:
|
|
205
|
+
with self._lock:
|
|
206
|
+
if fingerprint in self._memory_lru:
|
|
207
|
+
item = self._memory_lru[fingerprint]
|
|
208
|
+
entry_time = item.get("created_at", 0.0)
|
|
209
|
+
if self.ttl_seconds > 0 and (time.time() - entry_time) > self.ttl_seconds:
|
|
210
|
+
del self._memory_lru[fingerprint]
|
|
211
|
+
else:
|
|
212
|
+
self.stats["hits"] += 1
|
|
213
|
+
tokens = item.get("tokens_estimate", 0)
|
|
214
|
+
self.stats["tokens_saved"] += tokens
|
|
215
|
+
self._promote_lru(fingerprint, item["data"], tokens, created_at=entry_time)
|
|
216
|
+
return item["data"]
|
|
217
|
+
|
|
218
|
+
try:
|
|
219
|
+
with self._get_connection() as conn:
|
|
220
|
+
cur = conn.execute(
|
|
221
|
+
"SELECT response_json, tokens_estimate, created_at FROM evaluation_cache WHERE fingerprint = ?",
|
|
222
|
+
(fingerprint,)
|
|
223
|
+
)
|
|
224
|
+
row = cur.fetchone()
|
|
225
|
+
if row:
|
|
226
|
+
entry_created_at = row["created_at"]
|
|
227
|
+
if self.ttl_seconds > 0 and (time.time() - entry_created_at) > self.ttl_seconds:
|
|
228
|
+
conn.execute("DELETE FROM evaluation_cache WHERE fingerprint = ?", (fingerprint,))
|
|
229
|
+
conn.commit()
|
|
230
|
+
else:
|
|
231
|
+
with self._lock:
|
|
232
|
+
self.stats["hits"] += 1
|
|
233
|
+
data = json.loads(row["response_json"])
|
|
234
|
+
tokens = row["tokens_estimate"]
|
|
235
|
+
self.stats["tokens_saved"] += tokens
|
|
236
|
+
self._promote_lru(fingerprint, data, tokens, created_at=entry_created_at)
|
|
237
|
+
return data
|
|
238
|
+
except Exception as err:
|
|
239
|
+
logger.warning("Cache lookup error for %s: %s", fingerprint, err)
|
|
240
|
+
|
|
241
|
+
with self._lock:
|
|
242
|
+
self.stats["misses"] += 1
|
|
243
|
+
return None
|
|
244
|
+
|
|
245
|
+
def put(self, fingerprint: str, model: str, response_data: Dict[str, Any], input_tokens_estimate: int) -> None:
|
|
246
|
+
now = time.time()
|
|
247
|
+
try:
|
|
248
|
+
raw_json = json.dumps(response_data, separators=(",", ":"))
|
|
249
|
+
with self._get_connection() as conn:
|
|
250
|
+
conn.execute("""
|
|
251
|
+
INSERT OR REPLACE INTO evaluation_cache (
|
|
252
|
+
fingerprint, model, response_json, created_at, hit_count, tokens_estimate
|
|
253
|
+
) VALUES (?, ?, ?, ?, COALESCE((SELECT hit_count FROM evaluation_cache WHERE fingerprint = ?), 0), ?)
|
|
254
|
+
""", (fingerprint, model, raw_json, now, fingerprint, input_tokens_estimate))
|
|
255
|
+
if self.ttl_seconds > 0:
|
|
256
|
+
conn.execute("DELETE FROM evaluation_cache WHERE created_at < ?", (now - self.ttl_seconds,))
|
|
257
|
+
conn.commit()
|
|
258
|
+
|
|
259
|
+
with self._lock:
|
|
260
|
+
if self.ttl_seconds > 0:
|
|
261
|
+
expired_keys = [k for k, v in self._memory_lru.items() if (now - v.get("created_at", 0.0)) > self.ttl_seconds]
|
|
262
|
+
for k in expired_keys:
|
|
263
|
+
del self._memory_lru[k]
|
|
264
|
+
self._promote_lru(fingerprint, response_data, input_tokens_estimate, created_at=now)
|
|
265
|
+
except Exception as err:
|
|
266
|
+
logger.warning("Cache store error for %s: %s", fingerprint, err)
|
|
267
|
+
|
|
268
|
+
def _promote_lru(self, fingerprint: str, data: Dict[str, Any], tokens_estimate: int, created_at: Optional[float] = None) -> None:
|
|
269
|
+
if fingerprint in self._memory_lru:
|
|
270
|
+
existing = self._memory_lru.pop(fingerprint)
|
|
271
|
+
hit_count = existing.get("hit_count", 0) + 1
|
|
272
|
+
entry_time = existing.get("created_at", created_at or time.time())
|
|
273
|
+
else:
|
|
274
|
+
hit_count = 0
|
|
275
|
+
entry_time = created_at or time.time()
|
|
276
|
+
|
|
277
|
+
if len(self._memory_lru) >= self.max_memory_items:
|
|
278
|
+
oldest_key = next(iter(self._memory_lru))
|
|
279
|
+
del self._memory_lru[oldest_key]
|
|
280
|
+
self._memory_lru[fingerprint] = {
|
|
281
|
+
"data": data,
|
|
282
|
+
"tokens_estimate": tokens_estimate,
|
|
283
|
+
"hit_count": hit_count,
|
|
284
|
+
"created_at": entry_time
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
def clear(self) -> None:
|
|
288
|
+
with self._lock:
|
|
289
|
+
self._memory_lru.clear()
|
|
290
|
+
self.stats["hits"] = 0
|
|
291
|
+
self.stats["misses"] = 0
|
|
292
|
+
self.stats["tokens_saved"] = 0
|
|
293
|
+
try:
|
|
294
|
+
with self._get_connection() as conn:
|
|
295
|
+
conn.execute("DELETE FROM evaluation_cache")
|
|
296
|
+
conn.commit()
|
|
297
|
+
except Exception as err:
|
|
298
|
+
logger.warning("Cache clear error: %s", err)
|
|
299
|
+
|
|
300
|
+
def get_stats(self) -> Dict[str, Any]:
|
|
301
|
+
with self._lock:
|
|
302
|
+
total = self.stats["hits"] + self.stats["misses"]
|
|
303
|
+
rate = round((self.stats["hits"] / total) * 100, 2) if total > 0 else 0.0
|
|
304
|
+
return {
|
|
305
|
+
"hits": self.stats["hits"],
|
|
306
|
+
"misses": self.stats["misses"],
|
|
307
|
+
"hit_rate_pct": rate,
|
|
308
|
+
"tokens_saved": self.stats["tokens_saved"]
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
def close(self) -> None:
|
|
312
|
+
with self._lock:
|
|
313
|
+
if self._shared_conn is not None:
|
|
314
|
+
try:
|
|
315
|
+
self._shared_conn.close()
|
|
316
|
+
except Exception:
|
|
317
|
+
pass
|
|
318
|
+
self._shared_conn = None
|
|
319
|
+
conn = getattr(self._local, "conn", None)
|
|
320
|
+
if conn is not None:
|
|
321
|
+
try:
|
|
322
|
+
conn.close()
|
|
323
|
+
except Exception:
|
|
324
|
+
pass
|
|
325
|
+
self._local.conn = None
|
|
326
|
+
|
|
327
|
+
def __del__(self) -> None:
|
|
328
|
+
self.close()
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
SemanticCache = DeterministicCache
|
jevguard/calibrator.py
ADDED
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
"""
|
|
2
|
+
jevguard.calibrator - Certainty and Dispersion Analyzer for TypeSafe AI / Jev.
|
|
3
|
+
Identifies low confidence (< 0.40) and flat probability distributions (gap < 0.15),
|
|
4
|
+
marking results as AMBIGUOUS_STATE to prevent false certainty.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import math
|
|
8
|
+
from typing import Any, Dict, List, Optional, Tuple
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class ResponseCalibrator:
|
|
12
|
+
"""Evaluates probability spread on Jev decisions."""
|
|
13
|
+
|
|
14
|
+
MIN_TOP_PROBABILITY: float = 0.40
|
|
15
|
+
MIN_DISPERSION_GAP: float = 0.15
|
|
16
|
+
DEFAULT_NOUL_UNCERTAINTY_MARGIN: float = 0.12
|
|
17
|
+
|
|
18
|
+
def __init__(
|
|
19
|
+
self,
|
|
20
|
+
min_top_prob: float = MIN_TOP_PROBABILITY,
|
|
21
|
+
min_dispersion_gap: float = MIN_DISPERSION_GAP,
|
|
22
|
+
noul_uncertainty_margin: float = DEFAULT_NOUL_UNCERTAINTY_MARGIN
|
|
23
|
+
):
|
|
24
|
+
self.min_top_prob = min_top_prob
|
|
25
|
+
self.min_dispersion_gap = min_dispersion_gap
|
|
26
|
+
self.noul_margin = noul_uncertainty_margin
|
|
27
|
+
|
|
28
|
+
def calibrate(self, raw_answers: Dict[str, Any]) -> Tuple[Dict[str, Any], Dict[str, Any]]:
|
|
29
|
+
calibrated: Dict[str, Any] = {}
|
|
30
|
+
ambiguous_questions: List[str] = []
|
|
31
|
+
|
|
32
|
+
for name, ans in raw_answers.items():
|
|
33
|
+
if not isinstance(ans, dict):
|
|
34
|
+
calibrated[name] = {
|
|
35
|
+
"is_ambiguous": True,
|
|
36
|
+
"status": "AMBIGUOUS_STATE",
|
|
37
|
+
"calibration": {"reasons": ["invalid_answer_structure"]},
|
|
38
|
+
"raw": ans
|
|
39
|
+
}
|
|
40
|
+
ambiguous_questions.append(name)
|
|
41
|
+
continue
|
|
42
|
+
|
|
43
|
+
item = dict(ans)
|
|
44
|
+
q_type = str(item.get("type", "")).strip().lower()
|
|
45
|
+
|
|
46
|
+
if q_type == "choice":
|
|
47
|
+
self._calibrate_choice(item)
|
|
48
|
+
elif q_type == "score":
|
|
49
|
+
self._calibrate_score(item)
|
|
50
|
+
elif q_type == "noul":
|
|
51
|
+
self._calibrate_noul(item)
|
|
52
|
+
else:
|
|
53
|
+
item["is_ambiguous"] = True
|
|
54
|
+
item["status"] = "AMBIGUOUS_STATE"
|
|
55
|
+
item["calibration"] = {
|
|
56
|
+
"reasons": ["unknown_question_type"],
|
|
57
|
+
"question_type": q_type
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
if item.get("is_ambiguous", False):
|
|
61
|
+
ambiguous_questions.append(name)
|
|
62
|
+
|
|
63
|
+
calibrated[name] = item
|
|
64
|
+
|
|
65
|
+
summary = {
|
|
66
|
+
"total_evaluated": len(calibrated),
|
|
67
|
+
"ambiguous_count": len(ambiguous_questions),
|
|
68
|
+
"ambiguous_questions": ambiguous_questions,
|
|
69
|
+
"has_ambiguity": len(ambiguous_questions) > 0,
|
|
70
|
+
"verdict": "AMBIGUOUS_STATE" if ambiguous_questions else "CONFIDENT"
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
return calibrated, summary
|
|
74
|
+
|
|
75
|
+
def _calibrate_choice(self, item: Dict[str, Any]) -> None:
|
|
76
|
+
probs = item.get("probabilities", {})
|
|
77
|
+
parsed_pairs: List[Tuple[str, float]] = []
|
|
78
|
+
has_invalid = False
|
|
79
|
+
if isinstance(probs, dict):
|
|
80
|
+
for k, v in probs.items():
|
|
81
|
+
try:
|
|
82
|
+
val = float(v)
|
|
83
|
+
if math.isnan(val) or math.isinf(val):
|
|
84
|
+
has_invalid = True
|
|
85
|
+
else:
|
|
86
|
+
parsed_pairs.append((str(k), val))
|
|
87
|
+
except (ValueError, TypeError):
|
|
88
|
+
has_invalid = True
|
|
89
|
+
|
|
90
|
+
if not parsed_pairs:
|
|
91
|
+
try:
|
|
92
|
+
raw_conf = item.get("confidence", 0.0)
|
|
93
|
+
conf = float(raw_conf)
|
|
94
|
+
if math.isnan(conf) or math.isinf(conf):
|
|
95
|
+
conf = 0.0
|
|
96
|
+
has_invalid = True
|
|
97
|
+
except (ValueError, TypeError):
|
|
98
|
+
conf = 0.0
|
|
99
|
+
has_invalid = True
|
|
100
|
+
|
|
101
|
+
reasons = []
|
|
102
|
+
if has_invalid:
|
|
103
|
+
reasons.append("invalid_probability")
|
|
104
|
+
if conf < self.min_top_prob:
|
|
105
|
+
reasons.append("low_confidence")
|
|
106
|
+
|
|
107
|
+
is_amb = len(reasons) > 0
|
|
108
|
+
item["is_ambiguous"] = is_amb
|
|
109
|
+
item["status"] = "AMBIGUOUS_STATE" if is_amb else "CONFIDENT"
|
|
110
|
+
item["calibration"] = {
|
|
111
|
+
"top_choice": item.get("choice"),
|
|
112
|
+
"top_probability": round(conf, 4),
|
|
113
|
+
"runner_up_choice": None,
|
|
114
|
+
"runner_up_probability": 0.0,
|
|
115
|
+
"dispersion_gap": round(conf, 4),
|
|
116
|
+
"reasons": reasons
|
|
117
|
+
}
|
|
118
|
+
return
|
|
119
|
+
|
|
120
|
+
sorted_pairs = sorted(parsed_pairs, key=lambda x: x[1], reverse=True)
|
|
121
|
+
top_k, top_p = sorted_pairs[0]
|
|
122
|
+
runner_k, runner_p = sorted_pairs[1] if len(sorted_pairs) > 1 else (None, 0.0)
|
|
123
|
+
gap = top_p - runner_p
|
|
124
|
+
|
|
125
|
+
reasons = []
|
|
126
|
+
if has_invalid:
|
|
127
|
+
reasons.append("invalid_probability")
|
|
128
|
+
if top_p < self.min_top_prob:
|
|
129
|
+
reasons.append("low_confidence")
|
|
130
|
+
if len(sorted_pairs) > 1 and gap < self.min_dispersion_gap:
|
|
131
|
+
reasons.append("flat_distribution")
|
|
132
|
+
|
|
133
|
+
is_amb = len(reasons) > 0
|
|
134
|
+
item["is_ambiguous"] = is_amb
|
|
135
|
+
item["status"] = "AMBIGUOUS_STATE" if is_amb else "CONFIDENT"
|
|
136
|
+
item["calibration"] = {
|
|
137
|
+
"top_choice": top_k,
|
|
138
|
+
"top_probability": round(top_p, 4),
|
|
139
|
+
"runner_up_choice": runner_k,
|
|
140
|
+
"runner_up_probability": round(runner_p, 4),
|
|
141
|
+
"dispersion_gap": round(gap, 4),
|
|
142
|
+
"reasons": reasons
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
def _calibrate_score(self, item: Dict[str, Any]) -> None:
|
|
146
|
+
reasons: List[str] = []
|
|
147
|
+
raw_conf = item.get("confidence")
|
|
148
|
+
if raw_conf is None:
|
|
149
|
+
conf = 0.0
|
|
150
|
+
reasons.append("low_confidence")
|
|
151
|
+
else:
|
|
152
|
+
try:
|
|
153
|
+
conf = float(raw_conf)
|
|
154
|
+
if math.isnan(conf) or math.isinf(conf):
|
|
155
|
+
conf = 0.0
|
|
156
|
+
reasons.append("invalid_probability")
|
|
157
|
+
except (ValueError, TypeError):
|
|
158
|
+
conf = 0.0
|
|
159
|
+
reasons.append("invalid_probability")
|
|
160
|
+
|
|
161
|
+
probs = item.get("probabilities", {})
|
|
162
|
+
|
|
163
|
+
if conf < self.min_top_prob and "low_confidence" not in reasons:
|
|
164
|
+
reasons.append("low_confidence")
|
|
165
|
+
|
|
166
|
+
dispersion_gap = conf
|
|
167
|
+
if isinstance(probs, dict) and len(probs) >= 2:
|
|
168
|
+
parsed_probs: List[float] = []
|
|
169
|
+
for v in probs.values():
|
|
170
|
+
try:
|
|
171
|
+
val = float(v)
|
|
172
|
+
if math.isnan(val) or math.isinf(val):
|
|
173
|
+
if "invalid_probability" not in reasons:
|
|
174
|
+
reasons.append("invalid_probability")
|
|
175
|
+
else:
|
|
176
|
+
parsed_probs.append(val)
|
|
177
|
+
except (ValueError, TypeError):
|
|
178
|
+
if "invalid_probability" not in reasons:
|
|
179
|
+
reasons.append("invalid_probability")
|
|
180
|
+
if len(parsed_probs) >= 2:
|
|
181
|
+
sorted_probs = sorted(parsed_probs, reverse=True)
|
|
182
|
+
top_p = sorted_probs[0]
|
|
183
|
+
runner_p = sorted_probs[1]
|
|
184
|
+
dispersion_gap = top_p - runner_p
|
|
185
|
+
if top_p < self.min_top_prob and "low_confidence" not in reasons:
|
|
186
|
+
reasons.append("low_confidence")
|
|
187
|
+
if dispersion_gap < self.min_dispersion_gap and "flat_distribution" not in reasons:
|
|
188
|
+
reasons.append("flat_distribution")
|
|
189
|
+
|
|
190
|
+
is_amb = len(reasons) > 0
|
|
191
|
+
item["is_ambiguous"] = is_amb
|
|
192
|
+
item["status"] = "AMBIGUOUS_STATE" if is_amb else "CONFIDENT"
|
|
193
|
+
item["calibration"] = {
|
|
194
|
+
"confidence": round(conf, 4),
|
|
195
|
+
"dispersion_gap": round(dispersion_gap, 4),
|
|
196
|
+
"reasons": reasons
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
def _calibrate_noul(self, item: Dict[str, Any]) -> None:
|
|
200
|
+
val = item.get("noul")
|
|
201
|
+
if val is None:
|
|
202
|
+
item["is_ambiguous"] = True
|
|
203
|
+
item["status"] = "AMBIGUOUS_STATE"
|
|
204
|
+
item["calibration"] = {
|
|
205
|
+
"probability": 0.0,
|
|
206
|
+
"boundary_distance": 0.0,
|
|
207
|
+
"reasons": ["missing_noul_value"]
|
|
208
|
+
}
|
|
209
|
+
return
|
|
210
|
+
try:
|
|
211
|
+
prob = float(val)
|
|
212
|
+
except (ValueError, TypeError):
|
|
213
|
+
item["is_ambiguous"] = True
|
|
214
|
+
item["status"] = "AMBIGUOUS_STATE"
|
|
215
|
+
item["calibration"] = {
|
|
216
|
+
"probability": 0.0,
|
|
217
|
+
"boundary_distance": 0.0,
|
|
218
|
+
"reasons": ["invalid_probability"]
|
|
219
|
+
}
|
|
220
|
+
return
|
|
221
|
+
|
|
222
|
+
if math.isnan(prob) or math.isinf(prob):
|
|
223
|
+
item["is_ambiguous"] = True
|
|
224
|
+
item["status"] = "AMBIGUOUS_STATE"
|
|
225
|
+
item["calibration"] = {
|
|
226
|
+
"probability": 0.0,
|
|
227
|
+
"boundary_distance": 0.0,
|
|
228
|
+
"reasons": ["invalid_probability"]
|
|
229
|
+
}
|
|
230
|
+
return
|
|
231
|
+
|
|
232
|
+
dist = abs(prob - 0.50)
|
|
233
|
+
is_amb = dist < self.noul_margin
|
|
234
|
+
|
|
235
|
+
item["is_ambiguous"] = is_amb
|
|
236
|
+
item["status"] = "AMBIGUOUS_STATE" if is_amb else "CONFIDENT"
|
|
237
|
+
item["calibration"] = {
|
|
238
|
+
"probability": round(prob, 4),
|
|
239
|
+
"boundary_distance": round(dist, 4),
|
|
240
|
+
"reasons": ["boundary_uncertainty"] if is_amb else []
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
def calibrate_choice(self, item: Dict[str, Any]) -> Dict[str, Any]:
|
|
244
|
+
"""Convenience method to calibrate a single choice answer dictionary in-place."""
|
|
245
|
+
c = dict(item)
|
|
246
|
+
self._calibrate_choice(c)
|
|
247
|
+
return c
|
|
248
|
+
|
|
249
|
+
def calibrate_score(self, item: Dict[str, Any]) -> Dict[str, Any]:
|
|
250
|
+
"""Convenience method to calibrate a single score answer dictionary in-place."""
|
|
251
|
+
s = dict(item)
|
|
252
|
+
self._calibrate_score(s)
|
|
253
|
+
return s
|
|
254
|
+
|
|
255
|
+
def calibrate_noul(self, item: Dict[str, Any]) -> Dict[str, Any]:
|
|
256
|
+
"""Convenience method to calibrate a single noul answer dictionary in-place."""
|
|
257
|
+
n = dict(item)
|
|
258
|
+
self._calibrate_noul(n)
|
|
259
|
+
return n
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
CertaintyCalibrator = ResponseCalibrator
|