agent-learning 0.4.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning/__init__.py +160 -0
- agent_learning/_version.py +3 -0
- agent_learning/capture.py +271 -0
- agent_learning/classifiers/__init__.py +37 -0
- agent_learning/classifiers/base.py +185 -0
- agent_learning/classifiers/router.py +236 -0
- agent_learning/classifiers/scorers/__init__.py +34 -0
- agent_learning/classifiers/scorers/_base.py +162 -0
- agent_learning/classifiers/scorers/adherence.py +26 -0
- agent_learning/classifiers/scorers/completion.py +26 -0
- agent_learning/classifiers/scorers/intent.py +25 -0
- agent_learning/cli.py +386 -0
- agent_learning/config.py +524 -0
- agent_learning/learners/__init__.py +6 -0
- agent_learning/learners/base.py +38 -0
- agent_learning/learners/reinforce.py +153 -0
- agent_learning/metrics/__init__.py +22 -0
- agent_learning/metrics/base.py +233 -0
- agent_learning/metrics/intent_resolution.py +50 -0
- agent_learning/metrics/registry.py +42 -0
- agent_learning/metrics/task_adherence.py +42 -0
- agent_learning/metrics/task_completion.py +54 -0
- agent_learning/policy/__init__.py +7 -0
- agent_learning/policy/base.py +59 -0
- agent_learning/policy/contextual_softmax.py +243 -0
- agent_learning/policy/softmax_bandit.py +157 -0
- agent_learning/py.typed +1 -0
- agent_learning/rewards/__init__.py +6 -0
- agent_learning/rewards/shaping.py +121 -0
- agent_learning/rewards/writer.py +130 -0
- agent_learning/scorers/__init__.py +187 -0
- agent_learning/scorers/base.py +49 -0
- agent_learning/scorers/llm/__init__.py +19 -0
- agent_learning/scorers/llm/_base.py +126 -0
- agent_learning/scorers/llm/adherence.py +21 -0
- agent_learning/scorers/llm/completion.py +21 -0
- agent_learning/scorers/llm/intent.py +21 -0
- agent_learning/scorers/nlp/__init__.py +16 -0
- agent_learning/scorers/nlp/_base.py +99 -0
- agent_learning/scorers/nlp/adherence.py +17 -0
- agent_learning/scorers/nlp/completion.py +17 -0
- agent_learning/scorers/nlp/intent.py +17 -0
- agent_learning/scorers/nlp_text/__init__.py +26 -0
- agent_learning/scorers/nlp_text/_base.py +234 -0
- agent_learning/scorers/nlp_text/adherence.py +94 -0
- agent_learning/scorers/nlp_text/completion.py +91 -0
- agent_learning/scorers/nlp_text/intent.py +59 -0
- agent_learning/scorers/slm/__init__.py +25 -0
- agent_learning/scorers/slm/_base.py +292 -0
- agent_learning/scorers/slm/adherence.py +98 -0
- agent_learning/scorers/slm/completion.py +111 -0
- agent_learning/scorers/slm/intent.py +80 -0
- agent_learning/scorers/stdlib/__init__.py +38 -0
- agent_learning/scorers/stdlib/_text.py +87 -0
- agent_learning/scorers/stdlib/adherence.py +157 -0
- agent_learning/scorers/stdlib/completion.py +117 -0
- agent_learning/scorers/stdlib/intent.py +182 -0
- agent_learning/storage/__init__.py +14 -0
- agent_learning/storage/base.py +156 -0
- agent_learning/storage/cosmos.py +506 -0
- agent_learning/storage/local.py +353 -0
- agent_learning/storage/memory.py +209 -0
- agent_learning/training/__init__.py +5 -0
- agent_learning/training/runner.py +172 -0
- agent_learning/types.py +507 -0
- agent_learning-0.4.1.dist-info/METADATA +82 -0
- agent_learning-0.4.1.dist-info/RECORD +71 -0
- agent_learning-0.4.1.dist-info/WHEEL +5 -0
- agent_learning-0.4.1.dist-info/entry_points.txt +2 -0
- agent_learning-0.4.1.dist-info/licenses/LICENSE +21 -0
- agent_learning-0.4.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
"""Multi-class router/classifier over a context vector.
|
|
2
|
+
|
|
3
|
+
A :class:`RouterClassifier` maps a fixed-dimensional context vector
|
|
4
|
+
(usually called ``phi``) to one of a known set of class ids, plus a
|
|
5
|
+
calibrated confidence in ``[0, 1]``. It is the deterministic, in-SDK
|
|
6
|
+
replacement for any LLM-based or heuristic routing layer in front of
|
|
7
|
+
a policy.
|
|
8
|
+
|
|
9
|
+
Two inference modes are supported:
|
|
10
|
+
|
|
11
|
+
- ``mode="logreg"`` (set by :meth:`fit`) trains a multinomial
|
|
12
|
+
logistic regression on whatever class ids appear in the training
|
|
13
|
+
rows. This mode cannot route to a class id absent from the
|
|
14
|
+
training set, so it is brittle on classes that have never been
|
|
15
|
+
seen during training.
|
|
16
|
+
- ``mode="prototype"`` (set by :meth:`fit_from_catalog`) stores a
|
|
17
|
+
``phi`` prototype per class id. At inference time it picks the
|
|
18
|
+
class whose prototype is closest (cosine similarity) to the query
|
|
19
|
+
``phi``. This mode generalises to every class in the catalog,
|
|
20
|
+
including ones whose training examples are zero, because the
|
|
21
|
+
catalog itself provides the per-class representation.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import math
|
|
27
|
+
from dataclasses import dataclass, field
|
|
28
|
+
from typing import Dict, List
|
|
29
|
+
|
|
30
|
+
from .base import (
|
|
31
|
+
ClassifierResult,
|
|
32
|
+
_dot,
|
|
33
|
+
_softmax,
|
|
34
|
+
fit_multinomial_logreg,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class RouterClassifier:
|
|
40
|
+
"""Multi-class router over a context vector.
|
|
41
|
+
|
|
42
|
+
Training rows for :meth:`fit` are dicts with:
|
|
43
|
+
|
|
44
|
+
- ``"phi"``: ``list[float]`` of length :attr:`feature_dim`.
|
|
45
|
+
- ``"class_id"``: ground-truth class id (``str``).
|
|
46
|
+
|
|
47
|
+
Catalog rows for :meth:`fit_from_catalog` are dicts with the same
|
|
48
|
+
two keys; the catalog enumerates every class id the router is
|
|
49
|
+
allowed to predict.
|
|
50
|
+
|
|
51
|
+
Inference takes a ``phi`` vector and returns the predicted
|
|
52
|
+
``class_id`` with its softmax (or cosine-derived) probability. If
|
|
53
|
+
the top probability is below :attr:`refusal_threshold` the
|
|
54
|
+
classifier emits ``label="refused"`` with the complement
|
|
55
|
+
probability as confidence.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
feature_dim: int = 0
|
|
59
|
+
refusal_threshold: float = 0.40
|
|
60
|
+
epochs: int = 20
|
|
61
|
+
learning_rate: float = 0.10
|
|
62
|
+
weight_decay: float = 1e-4
|
|
63
|
+
batch_size: int = 256
|
|
64
|
+
seed: int = 42
|
|
65
|
+
|
|
66
|
+
# Filled in by fit() or fit_from_catalog().
|
|
67
|
+
classes: List[str] = field(default_factory=list)
|
|
68
|
+
weights: List[List[float]] = field(default_factory=list)
|
|
69
|
+
prototypes: List[List[float]] = field(default_factory=list)
|
|
70
|
+
mode: str = "logreg" # "logreg" or "prototype"
|
|
71
|
+
version: int = 0
|
|
72
|
+
|
|
73
|
+
def fit(self, training_rows: List[dict]) -> "RouterClassifier":
|
|
74
|
+
if not training_rows:
|
|
75
|
+
return self
|
|
76
|
+
self._infer_feature_dim(training_rows)
|
|
77
|
+
unique = sorted({row["class_id"] for row in training_rows})
|
|
78
|
+
cls_to_idx = {cid: i for i, cid in enumerate(unique)}
|
|
79
|
+
prepared: List[dict] = []
|
|
80
|
+
for row in training_rows:
|
|
81
|
+
phi = row["phi"]
|
|
82
|
+
self._check_phi(phi)
|
|
83
|
+
features = list(phi) + [1.0] # append bias
|
|
84
|
+
prepared.append({
|
|
85
|
+
"features": features,
|
|
86
|
+
"label": cls_to_idx[row["class_id"]],
|
|
87
|
+
})
|
|
88
|
+
weights = fit_multinomial_logreg(
|
|
89
|
+
prepared,
|
|
90
|
+
feature_dim=self.feature_dim,
|
|
91
|
+
num_classes=len(unique),
|
|
92
|
+
epochs=self.epochs,
|
|
93
|
+
learning_rate=self.learning_rate,
|
|
94
|
+
weight_decay=self.weight_decay,
|
|
95
|
+
batch_size=self.batch_size,
|
|
96
|
+
seed=self.seed,
|
|
97
|
+
)
|
|
98
|
+
self.classes = unique
|
|
99
|
+
self.weights = weights
|
|
100
|
+
self.mode = "logreg"
|
|
101
|
+
self.version += 1
|
|
102
|
+
return self
|
|
103
|
+
|
|
104
|
+
def fit_from_catalog(self, catalog_rows: List[dict]) -> "RouterClassifier":
|
|
105
|
+
"""Build a prototype-based router from catalog rows.
|
|
106
|
+
|
|
107
|
+
Each catalog row must carry ``"class_id"`` and ``"phi"``.
|
|
108
|
+
At inference time the router picks the class whose prototype
|
|
109
|
+
is closest (cosine similarity) to the query ``phi``.
|
|
110
|
+
"""
|
|
111
|
+
if not catalog_rows:
|
|
112
|
+
return self
|
|
113
|
+
self._infer_feature_dim(catalog_rows)
|
|
114
|
+
classes: List[str] = []
|
|
115
|
+
prototypes: List[List[float]] = []
|
|
116
|
+
for row in catalog_rows:
|
|
117
|
+
cid = str(row["class_id"])
|
|
118
|
+
phi = list(row["phi"])
|
|
119
|
+
self._check_phi(phi)
|
|
120
|
+
classes.append(cid)
|
|
121
|
+
prototypes.append(phi)
|
|
122
|
+
self.classes = classes
|
|
123
|
+
self.prototypes = prototypes
|
|
124
|
+
self.weights = []
|
|
125
|
+
self.mode = "prototype"
|
|
126
|
+
self.version += 1
|
|
127
|
+
return self
|
|
128
|
+
|
|
129
|
+
def predict(self, features: Dict[str, object]) -> ClassifierResult:
|
|
130
|
+
phi = features.get("phi")
|
|
131
|
+
if phi is None:
|
|
132
|
+
raise ValueError("RouterClassifier.predict requires features['phi']")
|
|
133
|
+
return self._predict_from_phi(list(phi)) # type: ignore[arg-type]
|
|
134
|
+
|
|
135
|
+
def predict_from_phi(self, phi: List[float]) -> ClassifierResult:
|
|
136
|
+
"""Convenience: predict directly from a context vector."""
|
|
137
|
+
return self._predict_from_phi(phi)
|
|
138
|
+
|
|
139
|
+
def _predict_from_phi(self, phi: List[float]) -> ClassifierResult:
|
|
140
|
+
if not self.classes:
|
|
141
|
+
return ClassifierResult(label="refused", confidence=0.0)
|
|
142
|
+
if self.mode == "prototype":
|
|
143
|
+
return self._predict_prototype(phi)
|
|
144
|
+
if not self.weights:
|
|
145
|
+
return ClassifierResult(label="refused", confidence=0.0)
|
|
146
|
+
x = list(phi) + [1.0]
|
|
147
|
+
logits = [_dot(self.weights[k], x) for k in range(len(self.classes))]
|
|
148
|
+
probs = _softmax(logits)
|
|
149
|
+
# argmax
|
|
150
|
+
best_idx = 0
|
|
151
|
+
best_p = probs[0]
|
|
152
|
+
for i in range(1, len(probs)):
|
|
153
|
+
if probs[i] > best_p:
|
|
154
|
+
best_p = probs[i]
|
|
155
|
+
best_idx = i
|
|
156
|
+
if best_p < self.refusal_threshold:
|
|
157
|
+
return ClassifierResult(
|
|
158
|
+
label="refused",
|
|
159
|
+
confidence=1.0 - best_p,
|
|
160
|
+
features={"top_candidate": float(best_idx), "top_probability": best_p},
|
|
161
|
+
)
|
|
162
|
+
return ClassifierResult(
|
|
163
|
+
label=self.classes[best_idx],
|
|
164
|
+
confidence=best_p,
|
|
165
|
+
features={"top_probability": best_p},
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
def _predict_prototype(self, phi: List[float]) -> ClassifierResult:
|
|
169
|
+
"""Cosine-similarity nearest-prototype routing."""
|
|
170
|
+
x_norm = math.sqrt(sum(v * v for v in phi)) or 1.0
|
|
171
|
+
sims: List[float] = []
|
|
172
|
+
for proto in self.prototypes:
|
|
173
|
+
p_norm = math.sqrt(sum(v * v for v in proto)) or 1.0
|
|
174
|
+
sims.append(_dot(phi, proto) / (x_norm * p_norm))
|
|
175
|
+
probs = _softmax(sims)
|
|
176
|
+
best_idx = 0
|
|
177
|
+
best_p = probs[0]
|
|
178
|
+
for i in range(1, len(probs)):
|
|
179
|
+
if probs[i] > best_p:
|
|
180
|
+
best_p = probs[i]
|
|
181
|
+
best_idx = i
|
|
182
|
+
if best_p < self.refusal_threshold:
|
|
183
|
+
return ClassifierResult(
|
|
184
|
+
label="refused",
|
|
185
|
+
confidence=1.0 - best_p,
|
|
186
|
+
features={"top_candidate": float(best_idx), "top_probability": best_p},
|
|
187
|
+
)
|
|
188
|
+
return ClassifierResult(
|
|
189
|
+
label=self.classes[best_idx],
|
|
190
|
+
confidence=best_p,
|
|
191
|
+
features={"top_probability": best_p, "top_similarity": sims[best_idx]},
|
|
192
|
+
)
|
|
193
|
+
|
|
194
|
+
def _infer_feature_dim(self, rows: List[dict]) -> None:
|
|
195
|
+
if self.feature_dim > 0:
|
|
196
|
+
return
|
|
197
|
+
for row in rows:
|
|
198
|
+
phi = row.get("phi")
|
|
199
|
+
if phi is not None:
|
|
200
|
+
self.feature_dim = len(phi)
|
|
201
|
+
return
|
|
202
|
+
|
|
203
|
+
def _check_phi(self, phi) -> None:
|
|
204
|
+
if self.feature_dim and len(phi) != self.feature_dim:
|
|
205
|
+
raise ValueError(
|
|
206
|
+
f"phi length {len(phi)} != feature_dim {self.feature_dim}"
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
def to_snapshot(self) -> Dict[str, object]:
|
|
210
|
+
"""Serialise to a JSON-friendly snapshot for persistence."""
|
|
211
|
+
return {
|
|
212
|
+
"type": "router_snapshot",
|
|
213
|
+
"version": self.version,
|
|
214
|
+
"mode": self.mode,
|
|
215
|
+
"feature_dim": self.feature_dim,
|
|
216
|
+
"refusal_threshold": self.refusal_threshold,
|
|
217
|
+
"classes": list(self.classes),
|
|
218
|
+
"weights": [list(w) for w in self.weights],
|
|
219
|
+
"prototypes": [list(p) for p in self.prototypes],
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
@classmethod
|
|
223
|
+
def from_snapshot(cls, doc: Dict[str, object]) -> "RouterClassifier":
|
|
224
|
+
inst = cls(
|
|
225
|
+
feature_dim=int(doc.get("feature_dim", 0)),
|
|
226
|
+
refusal_threshold=float(doc.get("refusal_threshold", 0.40)),
|
|
227
|
+
)
|
|
228
|
+
inst.classes = list(doc.get("classes", [])) # type: ignore[arg-type]
|
|
229
|
+
inst.weights = [list(w) for w in doc.get("weights", [])] # type: ignore[arg-type]
|
|
230
|
+
inst.prototypes = [list(p) for p in doc.get("prototypes", [])] # type: ignore[arg-type]
|
|
231
|
+
inst.mode = str(doc.get("mode", "logreg"))
|
|
232
|
+
inst.version = int(doc.get("version", 0))
|
|
233
|
+
return inst
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
__all__ = ["RouterClassifier"]
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""Binary ``{pass, fail}`` scorers for RL reward shaping.
|
|
2
|
+
|
|
3
|
+
Each scorer is a binary classifier over a ``(context, action)`` pair:
|
|
4
|
+
|
|
5
|
+
- :class:`agent_learning.classifiers.scorers.intent.IntentScorer` —
|
|
6
|
+
did the action address the requester's intent?
|
|
7
|
+
- :class:`agent_learning.classifiers.scorers.adherence.AdherenceScorer`
|
|
8
|
+
— did the action respect the contract / constraints of the task?
|
|
9
|
+
- :class:`agent_learning.classifiers.scorers.completion.CompletionScorer`
|
|
10
|
+
— did the action surface every required output?
|
|
11
|
+
|
|
12
|
+
All three scorers share the same surface:
|
|
13
|
+
|
|
14
|
+
- ``fit(training_rows)`` learns a binary logistic regression. Each
|
|
15
|
+
row carries ``"phi"`` (the context vector), ``"action_id"`` (the
|
|
16
|
+
chosen action), and ``"label"`` (``0`` or ``1``).
|
|
17
|
+
- ``predict(features)`` returns a :class:`ClassifierResult`.
|
|
18
|
+
- ``score(phi=..., action_id=...)`` is the convenience wrapper
|
|
19
|
+
matching the call shape any LLM-based scorer would expose.
|
|
20
|
+
|
|
21
|
+
The three scorers are intentionally identical in structure — they
|
|
22
|
+
differ only in their training labels and in the meaning callers
|
|
23
|
+
assign to their outputs.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from .adherence import AdherenceScorer
|
|
27
|
+
from .completion import CompletionScorer
|
|
28
|
+
from .intent import IntentScorer
|
|
29
|
+
|
|
30
|
+
__all__ = [
|
|
31
|
+
"AdherenceScorer",
|
|
32
|
+
"CompletionScorer",
|
|
33
|
+
"IntentScorer",
|
|
34
|
+
]
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
"""Shared base for binary scorers.
|
|
2
|
+
|
|
3
|
+
All scorers in this package are binary classifiers over the same
|
|
4
|
+
feature shape:
|
|
5
|
+
|
|
6
|
+
[phi (feature_dim)] ++ [action one-hot (len(actions))] ++ [bias (1)]
|
|
7
|
+
|
|
8
|
+
Per-scorer subclasses override :attr:`label_name` for the snapshot
|
|
9
|
+
payload. They are otherwise identical — the difference between an
|
|
10
|
+
intent, adherence, or completion scorer lives entirely in the binary
|
|
11
|
+
labels used to train it.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from typing import Dict, List
|
|
18
|
+
|
|
19
|
+
from ..base import (
|
|
20
|
+
ClassifierResult,
|
|
21
|
+
_dot,
|
|
22
|
+
_sigmoid,
|
|
23
|
+
fit_binary_logreg,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class BinaryScorer:
|
|
29
|
+
"""Common base for the binary scorers in this package.
|
|
30
|
+
|
|
31
|
+
Both :attr:`feature_dim` and :attr:`actions` are inferred from
|
|
32
|
+
the first training row when :meth:`fit` is called, so callers
|
|
33
|
+
only need to set them explicitly if they want to build a scorer
|
|
34
|
+
without ever calling ``fit`` (e.g. when restoring from a
|
|
35
|
+
snapshot).
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
label_name: str = "score"
|
|
39
|
+
feature_dim: int = 0
|
|
40
|
+
actions: List[str] = field(default_factory=list)
|
|
41
|
+
epochs: int = 20
|
|
42
|
+
learning_rate: float = 0.10
|
|
43
|
+
weight_decay: float = 1e-4
|
|
44
|
+
batch_size: int = 256
|
|
45
|
+
seed: int = 42
|
|
46
|
+
|
|
47
|
+
# Filled by fit()
|
|
48
|
+
weights: List[float] = field(default_factory=list)
|
|
49
|
+
version: int = 0
|
|
50
|
+
|
|
51
|
+
@property
|
|
52
|
+
def vector_dim(self) -> int:
|
|
53
|
+
"""Length of the feature vector *excluding* the bias entry."""
|
|
54
|
+
return self.feature_dim + len(self.actions)
|
|
55
|
+
|
|
56
|
+
def _build_features(self, phi: List[float], action_id: str) -> List[float]:
|
|
57
|
+
if len(phi) != self.feature_dim:
|
|
58
|
+
raise ValueError(
|
|
59
|
+
f"phi has length {len(phi)}, expected {self.feature_dim}"
|
|
60
|
+
)
|
|
61
|
+
action_onehot = [0.0] * len(self.actions)
|
|
62
|
+
if action_id in self.actions:
|
|
63
|
+
action_onehot[self.actions.index(action_id)] = 1.0
|
|
64
|
+
return list(phi) + action_onehot + [1.0] # bias
|
|
65
|
+
|
|
66
|
+
def fit(self, training_rows: List[dict]) -> "BinaryScorer":
|
|
67
|
+
if not training_rows:
|
|
68
|
+
return self
|
|
69
|
+
self._infer_dims(training_rows)
|
|
70
|
+
prepared: List[dict] = []
|
|
71
|
+
for row in training_rows:
|
|
72
|
+
feats = self._build_features(row["phi"], row["action_id"])
|
|
73
|
+
prepared.append({"features": feats, "label": int(row["label"])})
|
|
74
|
+
weights = fit_binary_logreg(
|
|
75
|
+
prepared,
|
|
76
|
+
feature_dim=self.vector_dim,
|
|
77
|
+
epochs=self.epochs,
|
|
78
|
+
learning_rate=self.learning_rate,
|
|
79
|
+
weight_decay=self.weight_decay,
|
|
80
|
+
batch_size=self.batch_size,
|
|
81
|
+
seed=self.seed,
|
|
82
|
+
)
|
|
83
|
+
self.weights = weights
|
|
84
|
+
self.version += 1
|
|
85
|
+
return self
|
|
86
|
+
|
|
87
|
+
def _infer_dims(self, rows: List[dict]) -> None:
|
|
88
|
+
if self.feature_dim == 0:
|
|
89
|
+
for row in rows:
|
|
90
|
+
phi = row.get("phi")
|
|
91
|
+
if phi is not None:
|
|
92
|
+
self.feature_dim = len(phi)
|
|
93
|
+
break
|
|
94
|
+
if not self.actions:
|
|
95
|
+
seen: List[str] = []
|
|
96
|
+
for row in rows:
|
|
97
|
+
aid = row.get("action_id")
|
|
98
|
+
if aid is not None and aid not in seen:
|
|
99
|
+
seen.append(aid)
|
|
100
|
+
self.actions = sorted(seen)
|
|
101
|
+
|
|
102
|
+
def predict(self, features: Dict[str, object]) -> ClassifierResult:
|
|
103
|
+
phi = features.get("phi")
|
|
104
|
+
action_id = features.get("action_id")
|
|
105
|
+
if phi is None or action_id is None:
|
|
106
|
+
raise ValueError(
|
|
107
|
+
"scorer predict requires features['phi'] and features['action_id']"
|
|
108
|
+
)
|
|
109
|
+
return self._predict(list(phi), str(action_id)) # type: ignore[arg-type]
|
|
110
|
+
|
|
111
|
+
def score(self, *, phi: List[float], action_id: str) -> ClassifierResult:
|
|
112
|
+
"""Convenience wrapper matching a typical LLM scorer call shape."""
|
|
113
|
+
return self._predict(list(phi), action_id)
|
|
114
|
+
|
|
115
|
+
def _predict(self, phi: List[float], action_id: str) -> ClassifierResult:
|
|
116
|
+
if not self.weights:
|
|
117
|
+
return ClassifierResult(label="fail", confidence=0.0)
|
|
118
|
+
x = self._build_features(phi, action_id)
|
|
119
|
+
p = _sigmoid(_dot(self.weights, x))
|
|
120
|
+
if p >= 0.5:
|
|
121
|
+
return ClassifierResult(
|
|
122
|
+
label="pass",
|
|
123
|
+
confidence=p,
|
|
124
|
+
features={"probability": p, "action_id": _hash_str(action_id)},
|
|
125
|
+
)
|
|
126
|
+
return ClassifierResult(
|
|
127
|
+
label="fail",
|
|
128
|
+
confidence=1.0 - p,
|
|
129
|
+
features={"probability": p, "action_id": _hash_str(action_id)},
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
def to_snapshot(self) -> Dict[str, object]:
|
|
133
|
+
return {
|
|
134
|
+
"type": "scorer_snapshot",
|
|
135
|
+
"label_name": self.label_name,
|
|
136
|
+
"version": self.version,
|
|
137
|
+
"feature_dim": self.feature_dim,
|
|
138
|
+
"actions": list(self.actions),
|
|
139
|
+
"weights": list(self.weights),
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
@classmethod
|
|
143
|
+
def from_snapshot(cls, doc: Dict[str, object]) -> "BinaryScorer":
|
|
144
|
+
inst = cls(
|
|
145
|
+
label_name=str(doc.get("label_name", "score")),
|
|
146
|
+
feature_dim=int(doc.get("feature_dim", 0)),
|
|
147
|
+
actions=list(doc.get("actions", [])), # type: ignore[arg-type]
|
|
148
|
+
)
|
|
149
|
+
inst.weights = list(doc.get("weights", [])) # type: ignore[arg-type]
|
|
150
|
+
inst.version = int(doc.get("version", 0))
|
|
151
|
+
return inst
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _hash_str(s: str) -> float:
|
|
155
|
+
"""Stable deterministic hash for trace/explainability output."""
|
|
156
|
+
h = 0
|
|
157
|
+
for ch in s:
|
|
158
|
+
h = (h * 31 + ord(ch)) & 0xFFFF
|
|
159
|
+
return float(h)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
__all__ = ["BinaryScorer"]
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Task-adherence scorer.
|
|
2
|
+
|
|
3
|
+
Binary ``{pass, fail}`` classifier that predicts whether the chosen
|
|
4
|
+
action respected the task's contract or constraints. Domain-specific
|
|
5
|
+
contracts are entirely the caller's concern — this class only sees
|
|
6
|
+
the context vector, the action id, and the training labels.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from ._base import BinaryScorer
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class AdherenceScorer(BinaryScorer):
|
|
15
|
+
"""Predict whether an action respects the task's contract.
|
|
16
|
+
|
|
17
|
+
Training rows expect ``label = 1`` when the action is scored as
|
|
18
|
+
adhere to the contract and ``label = 0`` otherwise.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
def __init__(self, **kwargs):
|
|
22
|
+
kwargs.setdefault("label_name", "adherence")
|
|
23
|
+
super().__init__(**kwargs)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
__all__ = ["AdherenceScorer"]
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Task-completion scorer.
|
|
2
|
+
|
|
3
|
+
Binary ``{pass, fail}`` classifier that predicts whether the chosen
|
|
4
|
+
action produced a complete result. "Complete" is whatever the
|
|
5
|
+
training labels encode; the classifier itself sees only the context
|
|
6
|
+
vector, the action id, and the binary label.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from ._base import BinaryScorer
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class CompletionScorer(BinaryScorer):
|
|
15
|
+
"""Predict whether an action produced a complete result.
|
|
16
|
+
|
|
17
|
+
Training rows expect ``label = 1`` when the action's output is
|
|
18
|
+
scored as complete and ``label = 0`` otherwise.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
def __init__(self, **kwargs):
|
|
22
|
+
kwargs.setdefault("label_name", "completion")
|
|
23
|
+
super().__init__(**kwargs)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
__all__ = ["CompletionScorer"]
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""Intent-resolution scorer.
|
|
2
|
+
|
|
3
|
+
Binary ``{pass, fail}`` classifier that predicts whether the chosen
|
|
4
|
+
action addresses the requester's intent given a context vector. A
|
|
5
|
+
deterministic, non-LLM drop-in for any LLM-based intent evaluator.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from ._base import BinaryScorer
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class IntentScorer(BinaryScorer):
|
|
14
|
+
"""Predict whether the chosen action addresses the requester's intent.
|
|
15
|
+
|
|
16
|
+
Training rows expect ``label = 1`` when the action is scored as
|
|
17
|
+
address the intent and ``label = 0`` otherwise.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
def __init__(self, **kwargs):
|
|
21
|
+
kwargs.setdefault("label_name", "intent")
|
|
22
|
+
super().__init__(**kwargs)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
__all__ = ["IntentScorer"]
|