agent-learning 0.4.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning/__init__.py +160 -0
- agent_learning/_version.py +3 -0
- agent_learning/capture.py +271 -0
- agent_learning/classifiers/__init__.py +37 -0
- agent_learning/classifiers/base.py +185 -0
- agent_learning/classifiers/router.py +236 -0
- agent_learning/classifiers/scorers/__init__.py +34 -0
- agent_learning/classifiers/scorers/_base.py +162 -0
- agent_learning/classifiers/scorers/adherence.py +26 -0
- agent_learning/classifiers/scorers/completion.py +26 -0
- agent_learning/classifiers/scorers/intent.py +25 -0
- agent_learning/cli.py +386 -0
- agent_learning/config.py +524 -0
- agent_learning/learners/__init__.py +6 -0
- agent_learning/learners/base.py +38 -0
- agent_learning/learners/reinforce.py +153 -0
- agent_learning/metrics/__init__.py +22 -0
- agent_learning/metrics/base.py +233 -0
- agent_learning/metrics/intent_resolution.py +50 -0
- agent_learning/metrics/registry.py +42 -0
- agent_learning/metrics/task_adherence.py +42 -0
- agent_learning/metrics/task_completion.py +54 -0
- agent_learning/policy/__init__.py +7 -0
- agent_learning/policy/base.py +59 -0
- agent_learning/policy/contextual_softmax.py +243 -0
- agent_learning/policy/softmax_bandit.py +157 -0
- agent_learning/py.typed +1 -0
- agent_learning/rewards/__init__.py +6 -0
- agent_learning/rewards/shaping.py +121 -0
- agent_learning/rewards/writer.py +130 -0
- agent_learning/scorers/__init__.py +187 -0
- agent_learning/scorers/base.py +49 -0
- agent_learning/scorers/llm/__init__.py +19 -0
- agent_learning/scorers/llm/_base.py +126 -0
- agent_learning/scorers/llm/adherence.py +21 -0
- agent_learning/scorers/llm/completion.py +21 -0
- agent_learning/scorers/llm/intent.py +21 -0
- agent_learning/scorers/nlp/__init__.py +16 -0
- agent_learning/scorers/nlp/_base.py +99 -0
- agent_learning/scorers/nlp/adherence.py +17 -0
- agent_learning/scorers/nlp/completion.py +17 -0
- agent_learning/scorers/nlp/intent.py +17 -0
- agent_learning/scorers/nlp_text/__init__.py +26 -0
- agent_learning/scorers/nlp_text/_base.py +234 -0
- agent_learning/scorers/nlp_text/adherence.py +94 -0
- agent_learning/scorers/nlp_text/completion.py +91 -0
- agent_learning/scorers/nlp_text/intent.py +59 -0
- agent_learning/scorers/slm/__init__.py +25 -0
- agent_learning/scorers/slm/_base.py +292 -0
- agent_learning/scorers/slm/adherence.py +98 -0
- agent_learning/scorers/slm/completion.py +111 -0
- agent_learning/scorers/slm/intent.py +80 -0
- agent_learning/scorers/stdlib/__init__.py +38 -0
- agent_learning/scorers/stdlib/_text.py +87 -0
- agent_learning/scorers/stdlib/adherence.py +157 -0
- agent_learning/scorers/stdlib/completion.py +117 -0
- agent_learning/scorers/stdlib/intent.py +182 -0
- agent_learning/storage/__init__.py +14 -0
- agent_learning/storage/base.py +156 -0
- agent_learning/storage/cosmos.py +506 -0
- agent_learning/storage/local.py +353 -0
- agent_learning/storage/memory.py +209 -0
- agent_learning/training/__init__.py +5 -0
- agent_learning/training/runner.py +172 -0
- agent_learning/types.py +507 -0
- agent_learning-0.4.1.dist-info/METADATA +82 -0
- agent_learning-0.4.1.dist-info/RECORD +71 -0
- agent_learning-0.4.1.dist-info/WHEEL +5 -0
- agent_learning-0.4.1.dist-info/entry_points.txt +2 -0
- agent_learning-0.4.1.dist-info/licenses/LICENSE +21 -0
- agent_learning-0.4.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
"""Public scorer interface for the agent-learning SDK.
|
|
2
|
+
|
|
3
|
+
Two selectors on :class:`agent_learning.config.ScoreRuntimeConfig`
|
|
4
|
+
decide which backend ``build_scorers`` returns:
|
|
5
|
+
|
|
6
|
+
- ``cfg.tier`` (preferred, new). One of:
|
|
7
|
+
- ``"stdlib"``: Tier 1, pure-stdlib text scorers. Zero external
|
|
8
|
+
dependencies.
|
|
9
|
+
- ``"nlp"``: Tier 2, in-SDK feature-based scorers over
|
|
10
|
+
``(phi, action_id)``. Currently routed to the existing
|
|
11
|
+
:mod:`.nlp` package.
|
|
12
|
+
- ``"slm"``: Tier 3, Microsoft Phi-4-mini-instruct via the
|
|
13
|
+
``[slm]`` extra. Stub raises until the extra is wired.
|
|
14
|
+
- ``"llm"``: Tier 4, ``azure-ai-evaluation`` evaluators behind
|
|
15
|
+
Azure OpenAI. Requires the ``[llm]`` extra.
|
|
16
|
+
- ``cfg.mode`` (legacy). Used only when ``cfg.tier`` is None. Values
|
|
17
|
+
``"nlp"`` and ``"llm"`` map to the same backends Tier 2 and Tier 4
|
|
18
|
+
resolve to, preserving the v0.1 API.
|
|
19
|
+
|
|
20
|
+
Callers never branch on the tier themselves; the factory hides the
|
|
21
|
+
choice so the reward shaper and learner stay backend-agnostic.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from typing import Tuple
|
|
27
|
+
|
|
28
|
+
from ..config import ScoreRuntimeConfig
|
|
29
|
+
from .base import Scorer, ScoreResult
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _build_stdlib(cfg: ScoreRuntimeConfig) -> Tuple[Scorer, Scorer, Scorer]:
|
|
33
|
+
from .stdlib import (
|
|
34
|
+
StdlibAdherenceScorer,
|
|
35
|
+
StdlibCompletionScorer,
|
|
36
|
+
StdlibIntentScorer,
|
|
37
|
+
)
|
|
38
|
+
return (
|
|
39
|
+
StdlibIntentScorer.load_or_default(
|
|
40
|
+
cfg.stdlib.snapshot_dir,
|
|
41
|
+
feature_dim=cfg.stdlib.feature_dim,
|
|
42
|
+
pass_threshold=cfg.stdlib.pass_threshold,
|
|
43
|
+
),
|
|
44
|
+
StdlibAdherenceScorer.load_or_default(
|
|
45
|
+
cfg.stdlib.snapshot_dir,
|
|
46
|
+
pass_threshold=cfg.stdlib.pass_threshold,
|
|
47
|
+
),
|
|
48
|
+
StdlibCompletionScorer.load_or_default(
|
|
49
|
+
cfg.stdlib.snapshot_dir,
|
|
50
|
+
pass_threshold=cfg.stdlib.pass_threshold,
|
|
51
|
+
),
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _build_nlp(cfg: ScoreRuntimeConfig) -> Tuple[Scorer, Scorer, Scorer]:
|
|
56
|
+
"""Tier 2: TF-IDF + scikit-learn scorers over query/response text.
|
|
57
|
+
|
|
58
|
+
Requires the ``[nlp]`` extra. The scorers are imported lazily so
|
|
59
|
+
callers that never select ``tier="nlp"`` don't need scikit-learn
|
|
60
|
+
installed.
|
|
61
|
+
"""
|
|
62
|
+
from .nlp_text import (
|
|
63
|
+
NlpTextAdherenceScorer,
|
|
64
|
+
NlpTextCompletionScorer,
|
|
65
|
+
NlpTextIntentScorer,
|
|
66
|
+
)
|
|
67
|
+
nlp_text_cfg = cfg.nlp_text
|
|
68
|
+
return (
|
|
69
|
+
NlpTextIntentScorer.load_or_default(
|
|
70
|
+
nlp_text_cfg.snapshot_dir,
|
|
71
|
+
pass_threshold=nlp_text_cfg.pass_threshold,
|
|
72
|
+
max_features=nlp_text_cfg.max_features,
|
|
73
|
+
ngram_min=nlp_text_cfg.ngram_min,
|
|
74
|
+
ngram_max=nlp_text_cfg.ngram_max,
|
|
75
|
+
),
|
|
76
|
+
NlpTextAdherenceScorer.load_or_default(
|
|
77
|
+
nlp_text_cfg.snapshot_dir,
|
|
78
|
+
pass_threshold=nlp_text_cfg.pass_threshold,
|
|
79
|
+
max_features=nlp_text_cfg.max_features,
|
|
80
|
+
ngram_min=nlp_text_cfg.ngram_min,
|
|
81
|
+
ngram_max=nlp_text_cfg.ngram_max,
|
|
82
|
+
),
|
|
83
|
+
NlpTextCompletionScorer.load_or_default(
|
|
84
|
+
nlp_text_cfg.snapshot_dir,
|
|
85
|
+
pass_threshold=nlp_text_cfg.pass_threshold,
|
|
86
|
+
max_features=nlp_text_cfg.max_features,
|
|
87
|
+
ngram_min=nlp_text_cfg.ngram_min,
|
|
88
|
+
ngram_max=nlp_text_cfg.ngram_max,
|
|
89
|
+
),
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _build_nlp_legacy(cfg: ScoreRuntimeConfig) -> Tuple[Scorer, Scorer, Scorer]:
|
|
94
|
+
"""Back-compat path for callers using ``mode="nlp"``.
|
|
95
|
+
|
|
96
|
+
Routes to the original :class:`agent_learning.classifiers.scorers.BinaryScorer`
|
|
97
|
+
stack over ``(phi, action_id)``. Preserves the v0.1 API for callers
|
|
98
|
+
that haven't migrated to the tier-based selector yet.
|
|
99
|
+
"""
|
|
100
|
+
from .nlp import (
|
|
101
|
+
NlpAdherenceScorer,
|
|
102
|
+
NlpCompletionScorer,
|
|
103
|
+
NlpIntentScorer,
|
|
104
|
+
)
|
|
105
|
+
return (
|
|
106
|
+
NlpIntentScorer.load_or_default(cfg.nlp),
|
|
107
|
+
NlpAdherenceScorer.load_or_default(cfg.nlp),
|
|
108
|
+
NlpCompletionScorer.load_or_default(cfg.nlp),
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _build_slm(cfg: ScoreRuntimeConfig) -> Tuple[Scorer, Scorer, Scorer]:
|
|
113
|
+
"""Tier 3: Phi-4-mini-instruct INT4 ONNX scorers.
|
|
114
|
+
|
|
115
|
+
Requires the ``[slm]`` extra and a local copy of the
|
|
116
|
+
Phi-4-mini-instruct INT4 ONNX bundle. Scorers are imported lazily so
|
|
117
|
+
callers that never select ``tier="slm"`` don't need
|
|
118
|
+
``onnxruntime-genai`` installed.
|
|
119
|
+
"""
|
|
120
|
+
from .slm import (
|
|
121
|
+
SlmAdherenceScorer,
|
|
122
|
+
SlmCompletionScorer,
|
|
123
|
+
SlmIntentScorer,
|
|
124
|
+
)
|
|
125
|
+
slm_cfg = cfg.slm
|
|
126
|
+
return (
|
|
127
|
+
SlmIntentScorer.load_or_default(
|
|
128
|
+
slm_cfg.model_dir,
|
|
129
|
+
pass_threshold=slm_cfg.pass_threshold,
|
|
130
|
+
max_new_tokens=slm_cfg.max_new_tokens,
|
|
131
|
+
temperature=slm_cfg.temperature,
|
|
132
|
+
),
|
|
133
|
+
SlmAdherenceScorer.load_or_default(
|
|
134
|
+
slm_cfg.model_dir,
|
|
135
|
+
pass_threshold=slm_cfg.pass_threshold,
|
|
136
|
+
max_new_tokens=slm_cfg.max_new_tokens,
|
|
137
|
+
temperature=slm_cfg.temperature,
|
|
138
|
+
),
|
|
139
|
+
SlmCompletionScorer.load_or_default(
|
|
140
|
+
slm_cfg.model_dir,
|
|
141
|
+
pass_threshold=slm_cfg.pass_threshold,
|
|
142
|
+
max_new_tokens=slm_cfg.max_new_tokens,
|
|
143
|
+
temperature=slm_cfg.temperature,
|
|
144
|
+
),
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _build_llm(cfg: ScoreRuntimeConfig) -> Tuple[Scorer, Scorer, Scorer]:
|
|
149
|
+
from .llm import (
|
|
150
|
+
LlmAdherenceScorer,
|
|
151
|
+
LlmCompletionScorer,
|
|
152
|
+
LlmIntentScorer,
|
|
153
|
+
)
|
|
154
|
+
return (
|
|
155
|
+
LlmIntentScorer(cfg.llm),
|
|
156
|
+
LlmAdherenceScorer(cfg.llm),
|
|
157
|
+
LlmCompletionScorer(cfg.llm),
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def build_scorers(cfg: ScoreRuntimeConfig) -> Tuple[Scorer, Scorer, Scorer]:
|
|
162
|
+
"""Return the ``(intent, adherence, completion)`` scorer trio.
|
|
163
|
+
|
|
164
|
+
Routing order: ``cfg.tier`` if set, else ``cfg.mode``.
|
|
165
|
+
"""
|
|
166
|
+
tier = cfg.tier
|
|
167
|
+
if tier is None:
|
|
168
|
+
# Legacy mode fallback. mode="nlp" routes to the BinaryScorer
|
|
169
|
+
# stack, NOT the new TF-IDF scorers (those are reachable via
|
|
170
|
+
# tier="nlp").
|
|
171
|
+
if cfg.mode == "nlp":
|
|
172
|
+
return _build_nlp_legacy(cfg)
|
|
173
|
+
if cfg.mode == "llm":
|
|
174
|
+
return _build_llm(cfg)
|
|
175
|
+
raise ValueError(f"unknown score_mode: {cfg.mode!r}")
|
|
176
|
+
if tier == "stdlib":
|
|
177
|
+
return _build_stdlib(cfg)
|
|
178
|
+
if tier == "nlp":
|
|
179
|
+
return _build_nlp(cfg)
|
|
180
|
+
if tier == "slm":
|
|
181
|
+
return _build_slm(cfg)
|
|
182
|
+
if tier == "llm":
|
|
183
|
+
return _build_llm(cfg)
|
|
184
|
+
raise ValueError(f"unknown score tier: {tier!r}")
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
__all__ = ["Scorer", "ScoreResult", "build_scorers"]
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""Shared types for the scoring backends.
|
|
2
|
+
|
|
3
|
+
Each backend produces the same :class:`ScoreResult` shape so the reward
|
|
4
|
+
shaper and learner stay backend-agnostic.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from typing import Dict, Protocol, runtime_checkable
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass(frozen=True)
|
|
14
|
+
class ScoreResult:
|
|
15
|
+
"""The contract every scoring backend honors.
|
|
16
|
+
|
|
17
|
+
Attributes:
|
|
18
|
+
label: ``"pass"`` or ``"fail"``.
|
|
19
|
+
confidence: probability of the predicted label in ``[0, 1]``.
|
|
20
|
+
normalized: a continuous score in ``[0, 1]`` suitable for the
|
|
21
|
+
reward shaper. Equals ``confidence`` when ``label == "pass"``
|
|
22
|
+
and ``1.0 - confidence`` when ``label == "fail"``.
|
|
23
|
+
features: optional debugging payload. Never consumed by the
|
|
24
|
+
reward shaper or the learner.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
label: str
|
|
28
|
+
confidence: float
|
|
29
|
+
normalized: float
|
|
30
|
+
features: Dict[str, object] = field(default_factory=dict)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@runtime_checkable
|
|
34
|
+
class Scorer(Protocol):
|
|
35
|
+
"""Structural type for both NLP and LLM scorers."""
|
|
36
|
+
|
|
37
|
+
name: str
|
|
38
|
+
|
|
39
|
+
def score(self, **kwargs: object) -> ScoreResult:
|
|
40
|
+
"""Score one episode.
|
|
41
|
+
|
|
42
|
+
NLP backends typically require ``phi`` and ``action_id``.
|
|
43
|
+
LLM backends typically require ``request`` and ``response``.
|
|
44
|
+
Unused keyword arguments are ignored by each backend so the
|
|
45
|
+
same call site works regardless of mode.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
__all__ = ["Scorer", "ScoreResult"]
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""LLM-backed scorers (Tier 4, requires ``azure-ai-evaluation``).
|
|
2
|
+
|
|
3
|
+
These wrappers stay shallow on purpose: each one defers to the
|
|
4
|
+
matching evaluator from the ``azure-ai-evaluation`` package and
|
|
5
|
+
projects the evaluator's verdict onto the SDK-neutral
|
|
6
|
+
:class:`agent_learning.scorers.base.ScoreResult` shape.
|
|
7
|
+
|
|
8
|
+
The evaluator import is deferred until first call so installing the
|
|
9
|
+
SDK without ``azure-ai-evaluation`` stays inexpensive when callers
|
|
10
|
+
only use the NLP backend.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from .adherence import LlmAdherenceScorer
|
|
16
|
+
from .completion import LlmCompletionScorer
|
|
17
|
+
from .intent import LlmIntentScorer
|
|
18
|
+
|
|
19
|
+
__all__ = ["LlmAdherenceScorer", "LlmCompletionScorer", "LlmIntentScorer"]
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""Shared wrapper that adapts an ``azure-ai-evaluation`` evaluator to
|
|
2
|
+
the SDK :class:`ScoreResult` contract.
|
|
3
|
+
|
|
4
|
+
The evaluator import is lazy: the ``azure-ai-evaluation`` package is
|
|
5
|
+
only imported the first time :meth:`_evaluator` is invoked, so callers
|
|
6
|
+
who never use the LLM backend pay no import cost.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import Any, Optional
|
|
13
|
+
|
|
14
|
+
from ...config import ScoreConfig
|
|
15
|
+
from ..base import ScoreResult
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass
|
|
19
|
+
class _LlmScorerWrapper:
|
|
20
|
+
"""Common base for the three concrete LLM scorers."""
|
|
21
|
+
|
|
22
|
+
cfg: ScoreConfig
|
|
23
|
+
name: str = "llm-scorer"
|
|
24
|
+
label_name: str = "llm-scorer"
|
|
25
|
+
evaluator_attr: str = ""
|
|
26
|
+
_cached_evaluator: Optional[Any] = field(default=None, repr=False)
|
|
27
|
+
|
|
28
|
+
def _evaluator(self) -> Any:
|
|
29
|
+
if self._cached_evaluator is not None:
|
|
30
|
+
return self._cached_evaluator
|
|
31
|
+
try:
|
|
32
|
+
module = __import__("azure.ai.evaluation", fromlist=[self.evaluator_attr])
|
|
33
|
+
except ImportError as exc:
|
|
34
|
+
raise ImportError(
|
|
35
|
+
"LLM scorers require the optional 'azure-ai-evaluation' package. "
|
|
36
|
+
"Install it with: pip install azure-ai-evaluation"
|
|
37
|
+
) from exc
|
|
38
|
+
evaluator_cls = getattr(module, self.evaluator_attr)
|
|
39
|
+
self._cached_evaluator = evaluator_cls(model_config=self.cfg.to_model_config())
|
|
40
|
+
return self._cached_evaluator
|
|
41
|
+
|
|
42
|
+
def score(
|
|
43
|
+
self,
|
|
44
|
+
*,
|
|
45
|
+
query: Optional[str] = None,
|
|
46
|
+
response: Optional[str] = None,
|
|
47
|
+
request: Optional[str] = None,
|
|
48
|
+
**kwargs: object,
|
|
49
|
+
) -> ScoreResult:
|
|
50
|
+
"""Run the underlying evaluator and project to a ScoreResult.
|
|
51
|
+
|
|
52
|
+
Either ``query`` or ``request`` may be used to supply the
|
|
53
|
+
prompt. Extra keyword arguments are forwarded to the evaluator
|
|
54
|
+
so callers can pass evaluator-specific fields like ``context``
|
|
55
|
+
or ``tool_calls`` without the SDK needing to learn them.
|
|
56
|
+
"""
|
|
57
|
+
if query is None:
|
|
58
|
+
query = request
|
|
59
|
+
if query is None or response is None:
|
|
60
|
+
raise ValueError(
|
|
61
|
+
f"{self.name} LLM scorer requires query (or request) and response"
|
|
62
|
+
)
|
|
63
|
+
payload = {"query": query, "response": response, **kwargs}
|
|
64
|
+
# Drop NLP-specific kwargs that would confuse the evaluator.
|
|
65
|
+
payload.pop("phi", None)
|
|
66
|
+
payload.pop("action_id", None)
|
|
67
|
+
evaluator = self._evaluator()
|
|
68
|
+
result = evaluator(**payload)
|
|
69
|
+
return _project_to_score(result, threshold=self.cfg.threshold, name=self.name)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _project_to_score(
|
|
73
|
+
result: Any, *, threshold: float, name: str
|
|
74
|
+
) -> ScoreResult:
|
|
75
|
+
"""Map an evaluator return value onto a :class:`ScoreResult`.
|
|
76
|
+
|
|
77
|
+
``azure-ai-evaluation`` evaluators return a mapping with a numeric
|
|
78
|
+
score keyed by either ``"score"`` or ``"<evaluator>_score"``. The
|
|
79
|
+
helper accepts either shape so the SDK works across evaluator
|
|
80
|
+
versions.
|
|
81
|
+
"""
|
|
82
|
+
if not isinstance(result, dict):
|
|
83
|
+
raise TypeError(
|
|
84
|
+
f"{name} evaluator returned {type(result).__name__}, expected mapping"
|
|
85
|
+
)
|
|
86
|
+
raw: Optional[float] = None
|
|
87
|
+
for key in ("score", f"{name}_score", "result"):
|
|
88
|
+
if key in result and isinstance(result[key], (int, float)):
|
|
89
|
+
raw = float(result[key])
|
|
90
|
+
break
|
|
91
|
+
if raw is None:
|
|
92
|
+
# Fall back to the first numeric value in the result mapping.
|
|
93
|
+
for value in result.values():
|
|
94
|
+
if isinstance(value, (int, float)):
|
|
95
|
+
raw = float(value)
|
|
96
|
+
break
|
|
97
|
+
if raw is None:
|
|
98
|
+
raise ValueError(
|
|
99
|
+
f"{name} evaluator returned no numeric score; got keys {list(result)!r}"
|
|
100
|
+
)
|
|
101
|
+
normalized = _normalize(raw)
|
|
102
|
+
label = "pass" if normalized >= threshold else "fail"
|
|
103
|
+
confidence = normalized if label == "pass" else 1.0 - normalized
|
|
104
|
+
return ScoreResult(
|
|
105
|
+
label=label,
|
|
106
|
+
confidence=confidence,
|
|
107
|
+
normalized=normalized,
|
|
108
|
+
features={"raw": raw, "evaluator_result": result},
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _normalize(raw: float) -> float:
|
|
113
|
+
"""Project a raw evaluator score into ``[0, 1]``.
|
|
114
|
+
|
|
115
|
+
Common ``azure-ai-evaluation`` evaluators emit scores on a 1-5
|
|
116
|
+
Likert scale. Scores already in ``[0, 1]`` are passed through
|
|
117
|
+
unchanged.
|
|
118
|
+
"""
|
|
119
|
+
if 0.0 <= raw <= 1.0:
|
|
120
|
+
return raw
|
|
121
|
+
if 1.0 <= raw <= 5.0:
|
|
122
|
+
return (raw - 1.0) / 4.0
|
|
123
|
+
return max(0.0, min(1.0, raw))
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
__all__ = ["_LlmScorerWrapper"]
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""LLM-backed task-adherence scorer."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from ...config import ScoreConfig
|
|
6
|
+
from ._base import _LlmScorerWrapper
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class LlmAdherenceScorer(_LlmScorerWrapper):
|
|
10
|
+
"""Defers to ``azure.ai.evaluation.TaskAdherenceEvaluator``."""
|
|
11
|
+
|
|
12
|
+
def __init__(self, cfg: ScoreConfig):
|
|
13
|
+
super().__init__(
|
|
14
|
+
cfg=cfg,
|
|
15
|
+
name="adherence",
|
|
16
|
+
label_name="adherence",
|
|
17
|
+
evaluator_attr="TaskAdherenceEvaluator",
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
__all__ = ["LlmAdherenceScorer"]
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""LLM-backed task-completion scorer."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from ...config import ScoreConfig
|
|
6
|
+
from ._base import _LlmScorerWrapper
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class LlmCompletionScorer(_LlmScorerWrapper):
|
|
10
|
+
"""Defers to ``azure.ai.evaluation.TaskCompletionEvaluator``."""
|
|
11
|
+
|
|
12
|
+
def __init__(self, cfg: ScoreConfig):
|
|
13
|
+
super().__init__(
|
|
14
|
+
cfg=cfg,
|
|
15
|
+
name="completion",
|
|
16
|
+
label_name="completion",
|
|
17
|
+
evaluator_attr="TaskCompletionEvaluator",
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
__all__ = ["LlmCompletionScorer"]
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""LLM-backed intent-resolution scorer."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from ...config import ScoreConfig
|
|
6
|
+
from ._base import _LlmScorerWrapper
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class LlmIntentScorer(_LlmScorerWrapper):
|
|
10
|
+
"""Defers to ``azure.ai.evaluation.IntentResolutionEvaluator``."""
|
|
11
|
+
|
|
12
|
+
def __init__(self, cfg: ScoreConfig):
|
|
13
|
+
super().__init__(
|
|
14
|
+
cfg=cfg,
|
|
15
|
+
name="intent",
|
|
16
|
+
label_name="intent",
|
|
17
|
+
evaluator_attr="IntentResolutionEvaluator",
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
__all__ = ["LlmIntentScorer"]
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""Native, in-process scorers (Tier 0, pure standard library).
|
|
2
|
+
|
|
3
|
+
These wrap the existing :class:`agent_learning.classifiers.scorers.BinaryScorer`
|
|
4
|
+
implementations so the SDK can ship a working scorer stack with no
|
|
5
|
+
external service dependencies. The wrappers translate the underlying
|
|
6
|
+
:class:`agent_learning.classifiers.base.ClassifierResult` into the
|
|
7
|
+
backend-neutral :class:`agent_learning.scorers.base.ScoreResult`.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from .adherence import NlpAdherenceScorer
|
|
13
|
+
from .completion import NlpCompletionScorer
|
|
14
|
+
from .intent import NlpIntentScorer
|
|
15
|
+
|
|
16
|
+
__all__ = ["NlpAdherenceScorer", "NlpCompletionScorer", "NlpIntentScorer"]
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""Shared wrapper that adapts a :class:`BinaryScorer` to the SDK
|
|
2
|
+
:class:`ScoreResult` contract.
|
|
3
|
+
|
|
4
|
+
The wrapper reads a snapshot from ``cfg.snapshot_dir/{name}.json`` at
|
|
5
|
+
load time. When no snapshot is present the scorer stays in its unfitted
|
|
6
|
+
"always-fail with zero confidence" state, which keeps the reward
|
|
7
|
+
shaper safe to call but encourages operators to train the scorers.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
import os
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from typing import List, Optional
|
|
16
|
+
|
|
17
|
+
from ...classifiers.scorers._base import BinaryScorer
|
|
18
|
+
from ...config import NlpScoreConfig
|
|
19
|
+
from ..base import ScoreResult
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class _NlpScorerWrapper:
|
|
24
|
+
"""Common base for the three concrete NLP scorers."""
|
|
25
|
+
|
|
26
|
+
name: str
|
|
27
|
+
label_name: str
|
|
28
|
+
_impl: BinaryScorer
|
|
29
|
+
_pass_threshold: float = 0.5
|
|
30
|
+
|
|
31
|
+
@classmethod
|
|
32
|
+
def _build(cls, name: str, cfg: NlpScoreConfig) -> "_NlpScorerWrapper":
|
|
33
|
+
impl = BinaryScorer(label_name=name)
|
|
34
|
+
snapshot_path = os.path.join(cfg.snapshot_dir, f"{name}.json")
|
|
35
|
+
if os.path.isfile(snapshot_path):
|
|
36
|
+
with open(snapshot_path, "r", encoding="utf-8") as fh:
|
|
37
|
+
impl = BinaryScorer.from_snapshot(json.load(fh))
|
|
38
|
+
impl.label_name = name
|
|
39
|
+
return cls(
|
|
40
|
+
name=name,
|
|
41
|
+
label_name=name,
|
|
42
|
+
_impl=impl,
|
|
43
|
+
_pass_threshold=float(cfg.pass_threshold),
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
def fit(self, training_rows: List[dict]) -> "_NlpScorerWrapper":
|
|
47
|
+
"""Fit the underlying logistic-regression head.
|
|
48
|
+
|
|
49
|
+
Training rows must look like ``{"phi": [...], "action_id": str, "label": 0|1}``.
|
|
50
|
+
"""
|
|
51
|
+
self._impl.fit(training_rows)
|
|
52
|
+
return self
|
|
53
|
+
|
|
54
|
+
def save(self, snapshot_dir: str) -> str:
|
|
55
|
+
"""Persist the fitted weights to ``{snapshot_dir}/{name}.json``."""
|
|
56
|
+
os.makedirs(snapshot_dir, exist_ok=True)
|
|
57
|
+
path = os.path.join(snapshot_dir, f"{self.name}.json")
|
|
58
|
+
with open(path, "w", encoding="utf-8") as fh:
|
|
59
|
+
json.dump(self._impl.to_snapshot(), fh)
|
|
60
|
+
return path
|
|
61
|
+
|
|
62
|
+
def score(
|
|
63
|
+
self,
|
|
64
|
+
*,
|
|
65
|
+
phi: Optional[List[float]] = None,
|
|
66
|
+
action_id: Optional[str] = None,
|
|
67
|
+
**_: object,
|
|
68
|
+
) -> ScoreResult:
|
|
69
|
+
"""Score one episode given a context vector and chosen action.
|
|
70
|
+
|
|
71
|
+
Extra keyword arguments are ignored so the same call site works
|
|
72
|
+
when the LLM backend is swapped in.
|
|
73
|
+
"""
|
|
74
|
+
if phi is None or action_id is None:
|
|
75
|
+
raise ValueError(
|
|
76
|
+
f"{self.name} NLP scorer requires phi and action_id"
|
|
77
|
+
)
|
|
78
|
+
result = self._impl.score(phi=list(phi), action_id=str(action_id))
|
|
79
|
+
probability = float(result.features.get("probability", result.confidence))
|
|
80
|
+
label = "pass" if probability >= self._pass_threshold else "fail"
|
|
81
|
+
if label == "pass":
|
|
82
|
+
confidence = probability
|
|
83
|
+
normalized = probability
|
|
84
|
+
else:
|
|
85
|
+
confidence = 1.0 - probability
|
|
86
|
+
normalized = probability # keep the raw "pass probability" for shaping
|
|
87
|
+
features = {
|
|
88
|
+
"probability": probability,
|
|
89
|
+
"fitted": bool(self._impl.weights),
|
|
90
|
+
}
|
|
91
|
+
return ScoreResult(
|
|
92
|
+
label=label,
|
|
93
|
+
confidence=confidence,
|
|
94
|
+
normalized=normalized,
|
|
95
|
+
features=features,
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
__all__ = ["_NlpScorerWrapper"]
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""NLP task-adherence scorer."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from ...config import NlpScoreConfig
|
|
6
|
+
from ._base import _NlpScorerWrapper
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class NlpAdherenceScorer(_NlpScorerWrapper):
|
|
10
|
+
"""Predict whether the response adheres to the requested task contract."""
|
|
11
|
+
|
|
12
|
+
@classmethod
|
|
13
|
+
def load_or_default(cls, cfg: NlpScoreConfig) -> "NlpAdherenceScorer":
|
|
14
|
+
return cls._build("adherence", cfg) # type: ignore[return-value]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
__all__ = ["NlpAdherenceScorer"]
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""NLP task-completion scorer."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from ...config import NlpScoreConfig
|
|
6
|
+
from ._base import _NlpScorerWrapper
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class NlpCompletionScorer(_NlpScorerWrapper):
|
|
10
|
+
"""Predict whether the response completes the requested task."""
|
|
11
|
+
|
|
12
|
+
@classmethod
|
|
13
|
+
def load_or_default(cls, cfg: NlpScoreConfig) -> "NlpCompletionScorer":
|
|
14
|
+
return cls._build("completion", cfg) # type: ignore[return-value]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
__all__ = ["NlpCompletionScorer"]
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""NLP intent-resolution scorer."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from ...config import NlpScoreConfig
|
|
6
|
+
from ._base import _NlpScorerWrapper
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class NlpIntentScorer(_NlpScorerWrapper):
|
|
10
|
+
"""Predict whether the chosen action addresses the requester's intent."""
|
|
11
|
+
|
|
12
|
+
@classmethod
|
|
13
|
+
def load_or_default(cls, cfg: NlpScoreConfig) -> "NlpIntentScorer":
|
|
14
|
+
return cls._build("intent", cfg) # type: ignore[return-value]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
__all__ = ["NlpIntentScorer"]
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Tier 2 NLP scorers (TF-IDF + scikit-learn) over raw query/response text.
|
|
2
|
+
|
|
3
|
+
These scorers operate on the same ``(query, response)`` pair the Tier 1
|
|
4
|
+
stdlib scorers accept and the same Tier 4 LLM scorers accept, but use a
|
|
5
|
+
TF-IDF vectorizer + logistic-regression head from scikit-learn for the
|
|
6
|
+
intent and adherence/completion classifiers. The adherence scorer also
|
|
7
|
+
applies the same deterministic rule engine the stdlib backend uses;
|
|
8
|
+
the rule-engine score is combined with the learned probability.
|
|
9
|
+
|
|
10
|
+
Requires the ``[nlp]`` extra (``pip install
|
|
11
|
+
agent-learning[nlp]``). All scikit-learn imports are lazy so
|
|
12
|
+
the package can still be imported without the extra; calling
|
|
13
|
+
:meth:`fit` or :meth:`score` raises a helpful ``ImportError`` then.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from .adherence import NlpTextAdherenceScorer
|
|
19
|
+
from .completion import NlpTextCompletionScorer
|
|
20
|
+
from .intent import NlpTextIntentScorer
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
"NlpTextAdherenceScorer",
|
|
24
|
+
"NlpTextCompletionScorer",
|
|
25
|
+
"NlpTextIntentScorer",
|
|
26
|
+
]
|