agent-learning 0.4.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. agent_learning/__init__.py +160 -0
  2. agent_learning/_version.py +3 -0
  3. agent_learning/capture.py +271 -0
  4. agent_learning/classifiers/__init__.py +37 -0
  5. agent_learning/classifiers/base.py +185 -0
  6. agent_learning/classifiers/router.py +236 -0
  7. agent_learning/classifiers/scorers/__init__.py +34 -0
  8. agent_learning/classifiers/scorers/_base.py +162 -0
  9. agent_learning/classifiers/scorers/adherence.py +26 -0
  10. agent_learning/classifiers/scorers/completion.py +26 -0
  11. agent_learning/classifiers/scorers/intent.py +25 -0
  12. agent_learning/cli.py +386 -0
  13. agent_learning/config.py +524 -0
  14. agent_learning/learners/__init__.py +6 -0
  15. agent_learning/learners/base.py +38 -0
  16. agent_learning/learners/reinforce.py +153 -0
  17. agent_learning/metrics/__init__.py +22 -0
  18. agent_learning/metrics/base.py +233 -0
  19. agent_learning/metrics/intent_resolution.py +50 -0
  20. agent_learning/metrics/registry.py +42 -0
  21. agent_learning/metrics/task_adherence.py +42 -0
  22. agent_learning/metrics/task_completion.py +54 -0
  23. agent_learning/policy/__init__.py +7 -0
  24. agent_learning/policy/base.py +59 -0
  25. agent_learning/policy/contextual_softmax.py +243 -0
  26. agent_learning/policy/softmax_bandit.py +157 -0
  27. agent_learning/py.typed +1 -0
  28. agent_learning/rewards/__init__.py +6 -0
  29. agent_learning/rewards/shaping.py +121 -0
  30. agent_learning/rewards/writer.py +130 -0
  31. agent_learning/scorers/__init__.py +187 -0
  32. agent_learning/scorers/base.py +49 -0
  33. agent_learning/scorers/llm/__init__.py +19 -0
  34. agent_learning/scorers/llm/_base.py +126 -0
  35. agent_learning/scorers/llm/adherence.py +21 -0
  36. agent_learning/scorers/llm/completion.py +21 -0
  37. agent_learning/scorers/llm/intent.py +21 -0
  38. agent_learning/scorers/nlp/__init__.py +16 -0
  39. agent_learning/scorers/nlp/_base.py +99 -0
  40. agent_learning/scorers/nlp/adherence.py +17 -0
  41. agent_learning/scorers/nlp/completion.py +17 -0
  42. agent_learning/scorers/nlp/intent.py +17 -0
  43. agent_learning/scorers/nlp_text/__init__.py +26 -0
  44. agent_learning/scorers/nlp_text/_base.py +234 -0
  45. agent_learning/scorers/nlp_text/adherence.py +94 -0
  46. agent_learning/scorers/nlp_text/completion.py +91 -0
  47. agent_learning/scorers/nlp_text/intent.py +59 -0
  48. agent_learning/scorers/slm/__init__.py +25 -0
  49. agent_learning/scorers/slm/_base.py +292 -0
  50. agent_learning/scorers/slm/adherence.py +98 -0
  51. agent_learning/scorers/slm/completion.py +111 -0
  52. agent_learning/scorers/slm/intent.py +80 -0
  53. agent_learning/scorers/stdlib/__init__.py +38 -0
  54. agent_learning/scorers/stdlib/_text.py +87 -0
  55. agent_learning/scorers/stdlib/adherence.py +157 -0
  56. agent_learning/scorers/stdlib/completion.py +117 -0
  57. agent_learning/scorers/stdlib/intent.py +182 -0
  58. agent_learning/storage/__init__.py +14 -0
  59. agent_learning/storage/base.py +156 -0
  60. agent_learning/storage/cosmos.py +506 -0
  61. agent_learning/storage/local.py +353 -0
  62. agent_learning/storage/memory.py +209 -0
  63. agent_learning/training/__init__.py +5 -0
  64. agent_learning/training/runner.py +172 -0
  65. agent_learning/types.py +507 -0
  66. agent_learning-0.4.1.dist-info/METADATA +82 -0
  67. agent_learning-0.4.1.dist-info/RECORD +71 -0
  68. agent_learning-0.4.1.dist-info/WHEEL +5 -0
  69. agent_learning-0.4.1.dist-info/entry_points.txt +2 -0
  70. agent_learning-0.4.1.dist-info/licenses/LICENSE +21 -0
  71. agent_learning-0.4.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,187 @@
1
+ """Public scorer interface for the agent-learning SDK.
2
+
3
+ Two selectors on :class:`agent_learning.config.ScoreRuntimeConfig`
4
+ decide which backend ``build_scorers`` returns:
5
+
6
+ - ``cfg.tier`` (preferred, new). One of:
7
+ - ``"stdlib"``: Tier 1, pure-stdlib text scorers. Zero external
8
+ dependencies.
9
+ - ``"nlp"``: Tier 2, in-SDK feature-based scorers over
10
+ ``(phi, action_id)``. Currently routed to the existing
11
+ :mod:`.nlp` package.
12
+ - ``"slm"``: Tier 3, Microsoft Phi-4-mini-instruct via the
13
+ ``[slm]`` extra. Stub raises until the extra is wired.
14
+ - ``"llm"``: Tier 4, ``azure-ai-evaluation`` evaluators behind
15
+ Azure OpenAI. Requires the ``[llm]`` extra.
16
+ - ``cfg.mode`` (legacy). Used only when ``cfg.tier`` is None. Values
17
+ ``"nlp"`` and ``"llm"`` map to the same backends Tier 2 and Tier 4
18
+ resolve to, preserving the v0.1 API.
19
+
20
+ Callers never branch on the tier themselves; the factory hides the
21
+ choice so the reward shaper and learner stay backend-agnostic.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ from typing import Tuple
27
+
28
+ from ..config import ScoreRuntimeConfig
29
+ from .base import Scorer, ScoreResult
30
+
31
+
32
+ def _build_stdlib(cfg: ScoreRuntimeConfig) -> Tuple[Scorer, Scorer, Scorer]:
33
+ from .stdlib import (
34
+ StdlibAdherenceScorer,
35
+ StdlibCompletionScorer,
36
+ StdlibIntentScorer,
37
+ )
38
+ return (
39
+ StdlibIntentScorer.load_or_default(
40
+ cfg.stdlib.snapshot_dir,
41
+ feature_dim=cfg.stdlib.feature_dim,
42
+ pass_threshold=cfg.stdlib.pass_threshold,
43
+ ),
44
+ StdlibAdherenceScorer.load_or_default(
45
+ cfg.stdlib.snapshot_dir,
46
+ pass_threshold=cfg.stdlib.pass_threshold,
47
+ ),
48
+ StdlibCompletionScorer.load_or_default(
49
+ cfg.stdlib.snapshot_dir,
50
+ pass_threshold=cfg.stdlib.pass_threshold,
51
+ ),
52
+ )
53
+
54
+
55
+ def _build_nlp(cfg: ScoreRuntimeConfig) -> Tuple[Scorer, Scorer, Scorer]:
56
+ """Tier 2: TF-IDF + scikit-learn scorers over query/response text.
57
+
58
+ Requires the ``[nlp]`` extra. The scorers are imported lazily so
59
+ callers that never select ``tier="nlp"`` don't need scikit-learn
60
+ installed.
61
+ """
62
+ from .nlp_text import (
63
+ NlpTextAdherenceScorer,
64
+ NlpTextCompletionScorer,
65
+ NlpTextIntentScorer,
66
+ )
67
+ nlp_text_cfg = cfg.nlp_text
68
+ return (
69
+ NlpTextIntentScorer.load_or_default(
70
+ nlp_text_cfg.snapshot_dir,
71
+ pass_threshold=nlp_text_cfg.pass_threshold,
72
+ max_features=nlp_text_cfg.max_features,
73
+ ngram_min=nlp_text_cfg.ngram_min,
74
+ ngram_max=nlp_text_cfg.ngram_max,
75
+ ),
76
+ NlpTextAdherenceScorer.load_or_default(
77
+ nlp_text_cfg.snapshot_dir,
78
+ pass_threshold=nlp_text_cfg.pass_threshold,
79
+ max_features=nlp_text_cfg.max_features,
80
+ ngram_min=nlp_text_cfg.ngram_min,
81
+ ngram_max=nlp_text_cfg.ngram_max,
82
+ ),
83
+ NlpTextCompletionScorer.load_or_default(
84
+ nlp_text_cfg.snapshot_dir,
85
+ pass_threshold=nlp_text_cfg.pass_threshold,
86
+ max_features=nlp_text_cfg.max_features,
87
+ ngram_min=nlp_text_cfg.ngram_min,
88
+ ngram_max=nlp_text_cfg.ngram_max,
89
+ ),
90
+ )
91
+
92
+
93
+ def _build_nlp_legacy(cfg: ScoreRuntimeConfig) -> Tuple[Scorer, Scorer, Scorer]:
94
+ """Back-compat path for callers using ``mode="nlp"``.
95
+
96
+ Routes to the original :class:`agent_learning.classifiers.scorers.BinaryScorer`
97
+ stack over ``(phi, action_id)``. Preserves the v0.1 API for callers
98
+ that haven't migrated to the tier-based selector yet.
99
+ """
100
+ from .nlp import (
101
+ NlpAdherenceScorer,
102
+ NlpCompletionScorer,
103
+ NlpIntentScorer,
104
+ )
105
+ return (
106
+ NlpIntentScorer.load_or_default(cfg.nlp),
107
+ NlpAdherenceScorer.load_or_default(cfg.nlp),
108
+ NlpCompletionScorer.load_or_default(cfg.nlp),
109
+ )
110
+
111
+
112
+ def _build_slm(cfg: ScoreRuntimeConfig) -> Tuple[Scorer, Scorer, Scorer]:
113
+ """Tier 3: Phi-4-mini-instruct INT4 ONNX scorers.
114
+
115
+ Requires the ``[slm]`` extra and a local copy of the
116
+ Phi-4-mini-instruct INT4 ONNX bundle. Scorers are imported lazily so
117
+ callers that never select ``tier="slm"`` don't need
118
+ ``onnxruntime-genai`` installed.
119
+ """
120
+ from .slm import (
121
+ SlmAdherenceScorer,
122
+ SlmCompletionScorer,
123
+ SlmIntentScorer,
124
+ )
125
+ slm_cfg = cfg.slm
126
+ return (
127
+ SlmIntentScorer.load_or_default(
128
+ slm_cfg.model_dir,
129
+ pass_threshold=slm_cfg.pass_threshold,
130
+ max_new_tokens=slm_cfg.max_new_tokens,
131
+ temperature=slm_cfg.temperature,
132
+ ),
133
+ SlmAdherenceScorer.load_or_default(
134
+ slm_cfg.model_dir,
135
+ pass_threshold=slm_cfg.pass_threshold,
136
+ max_new_tokens=slm_cfg.max_new_tokens,
137
+ temperature=slm_cfg.temperature,
138
+ ),
139
+ SlmCompletionScorer.load_or_default(
140
+ slm_cfg.model_dir,
141
+ pass_threshold=slm_cfg.pass_threshold,
142
+ max_new_tokens=slm_cfg.max_new_tokens,
143
+ temperature=slm_cfg.temperature,
144
+ ),
145
+ )
146
+
147
+
148
+ def _build_llm(cfg: ScoreRuntimeConfig) -> Tuple[Scorer, Scorer, Scorer]:
149
+ from .llm import (
150
+ LlmAdherenceScorer,
151
+ LlmCompletionScorer,
152
+ LlmIntentScorer,
153
+ )
154
+ return (
155
+ LlmIntentScorer(cfg.llm),
156
+ LlmAdherenceScorer(cfg.llm),
157
+ LlmCompletionScorer(cfg.llm),
158
+ )
159
+
160
+
161
+ def build_scorers(cfg: ScoreRuntimeConfig) -> Tuple[Scorer, Scorer, Scorer]:
162
+ """Return the ``(intent, adherence, completion)`` scorer trio.
163
+
164
+ Routing order: ``cfg.tier`` if set, else ``cfg.mode``.
165
+ """
166
+ tier = cfg.tier
167
+ if tier is None:
168
+ # Legacy mode fallback. mode="nlp" routes to the BinaryScorer
169
+ # stack, NOT the new TF-IDF scorers (those are reachable via
170
+ # tier="nlp").
171
+ if cfg.mode == "nlp":
172
+ return _build_nlp_legacy(cfg)
173
+ if cfg.mode == "llm":
174
+ return _build_llm(cfg)
175
+ raise ValueError(f"unknown score_mode: {cfg.mode!r}")
176
+ if tier == "stdlib":
177
+ return _build_stdlib(cfg)
178
+ if tier == "nlp":
179
+ return _build_nlp(cfg)
180
+ if tier == "slm":
181
+ return _build_slm(cfg)
182
+ if tier == "llm":
183
+ return _build_llm(cfg)
184
+ raise ValueError(f"unknown score tier: {tier!r}")
185
+
186
+
187
+ __all__ = ["Scorer", "ScoreResult", "build_scorers"]
@@ -0,0 +1,49 @@
1
+ """Shared types for the scoring backends.
2
+
3
+ Each backend produces the same :class:`ScoreResult` shape so the reward
4
+ shaper and learner stay backend-agnostic.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from dataclasses import dataclass, field
10
+ from typing import Dict, Protocol, runtime_checkable
11
+
12
+
13
+ @dataclass(frozen=True)
14
+ class ScoreResult:
15
+ """The contract every scoring backend honors.
16
+
17
+ Attributes:
18
+ label: ``"pass"`` or ``"fail"``.
19
+ confidence: probability of the predicted label in ``[0, 1]``.
20
+ normalized: a continuous score in ``[0, 1]`` suitable for the
21
+ reward shaper. Equals ``confidence`` when ``label == "pass"``
22
+ and ``1.0 - confidence`` when ``label == "fail"``.
23
+ features: optional debugging payload. Never consumed by the
24
+ reward shaper or the learner.
25
+ """
26
+
27
+ label: str
28
+ confidence: float
29
+ normalized: float
30
+ features: Dict[str, object] = field(default_factory=dict)
31
+
32
+
33
+ @runtime_checkable
34
+ class Scorer(Protocol):
35
+ """Structural type for both NLP and LLM scorers."""
36
+
37
+ name: str
38
+
39
+ def score(self, **kwargs: object) -> ScoreResult:
40
+ """Score one episode.
41
+
42
+ NLP backends typically require ``phi`` and ``action_id``.
43
+ LLM backends typically require ``request`` and ``response``.
44
+ Unused keyword arguments are ignored by each backend so the
45
+ same call site works regardless of mode.
46
+ """
47
+
48
+
49
+ __all__ = ["Scorer", "ScoreResult"]
@@ -0,0 +1,19 @@
1
+ """LLM-backed scorers (Tier 4, requires ``azure-ai-evaluation``).
2
+
3
+ These wrappers stay shallow on purpose: each one defers to the
4
+ matching evaluator from the ``azure-ai-evaluation`` package and
5
+ projects the evaluator's verdict onto the SDK-neutral
6
+ :class:`agent_learning.scorers.base.ScoreResult` shape.
7
+
8
+ The evaluator import is deferred until first call so installing the
9
+ SDK without ``azure-ai-evaluation`` stays inexpensive when callers
10
+ only use the NLP backend.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from .adherence import LlmAdherenceScorer
16
+ from .completion import LlmCompletionScorer
17
+ from .intent import LlmIntentScorer
18
+
19
+ __all__ = ["LlmAdherenceScorer", "LlmCompletionScorer", "LlmIntentScorer"]
@@ -0,0 +1,126 @@
1
+ """Shared wrapper that adapts an ``azure-ai-evaluation`` evaluator to
2
+ the SDK :class:`ScoreResult` contract.
3
+
4
+ The evaluator import is lazy: the ``azure-ai-evaluation`` package is
5
+ only imported the first time :meth:`_evaluator` is invoked, so callers
6
+ who never use the LLM backend pay no import cost.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from dataclasses import dataclass, field
12
+ from typing import Any, Optional
13
+
14
+ from ...config import ScoreConfig
15
+ from ..base import ScoreResult
16
+
17
+
18
+ @dataclass
19
+ class _LlmScorerWrapper:
20
+ """Common base for the three concrete LLM scorers."""
21
+
22
+ cfg: ScoreConfig
23
+ name: str = "llm-scorer"
24
+ label_name: str = "llm-scorer"
25
+ evaluator_attr: str = ""
26
+ _cached_evaluator: Optional[Any] = field(default=None, repr=False)
27
+
28
+ def _evaluator(self) -> Any:
29
+ if self._cached_evaluator is not None:
30
+ return self._cached_evaluator
31
+ try:
32
+ module = __import__("azure.ai.evaluation", fromlist=[self.evaluator_attr])
33
+ except ImportError as exc:
34
+ raise ImportError(
35
+ "LLM scorers require the optional 'azure-ai-evaluation' package. "
36
+ "Install it with: pip install azure-ai-evaluation"
37
+ ) from exc
38
+ evaluator_cls = getattr(module, self.evaluator_attr)
39
+ self._cached_evaluator = evaluator_cls(model_config=self.cfg.to_model_config())
40
+ return self._cached_evaluator
41
+
42
+ def score(
43
+ self,
44
+ *,
45
+ query: Optional[str] = None,
46
+ response: Optional[str] = None,
47
+ request: Optional[str] = None,
48
+ **kwargs: object,
49
+ ) -> ScoreResult:
50
+ """Run the underlying evaluator and project to a ScoreResult.
51
+
52
+ Either ``query`` or ``request`` may be used to supply the
53
+ prompt. Extra keyword arguments are forwarded to the evaluator
54
+ so callers can pass evaluator-specific fields like ``context``
55
+ or ``tool_calls`` without the SDK needing to learn them.
56
+ """
57
+ if query is None:
58
+ query = request
59
+ if query is None or response is None:
60
+ raise ValueError(
61
+ f"{self.name} LLM scorer requires query (or request) and response"
62
+ )
63
+ payload = {"query": query, "response": response, **kwargs}
64
+ # Drop NLP-specific kwargs that would confuse the evaluator.
65
+ payload.pop("phi", None)
66
+ payload.pop("action_id", None)
67
+ evaluator = self._evaluator()
68
+ result = evaluator(**payload)
69
+ return _project_to_score(result, threshold=self.cfg.threshold, name=self.name)
70
+
71
+
72
+ def _project_to_score(
73
+ result: Any, *, threshold: float, name: str
74
+ ) -> ScoreResult:
75
+ """Map an evaluator return value onto a :class:`ScoreResult`.
76
+
77
+ ``azure-ai-evaluation`` evaluators return a mapping with a numeric
78
+ score keyed by either ``"score"`` or ``"<evaluator>_score"``. The
79
+ helper accepts either shape so the SDK works across evaluator
80
+ versions.
81
+ """
82
+ if not isinstance(result, dict):
83
+ raise TypeError(
84
+ f"{name} evaluator returned {type(result).__name__}, expected mapping"
85
+ )
86
+ raw: Optional[float] = None
87
+ for key in ("score", f"{name}_score", "result"):
88
+ if key in result and isinstance(result[key], (int, float)):
89
+ raw = float(result[key])
90
+ break
91
+ if raw is None:
92
+ # Fall back to the first numeric value in the result mapping.
93
+ for value in result.values():
94
+ if isinstance(value, (int, float)):
95
+ raw = float(value)
96
+ break
97
+ if raw is None:
98
+ raise ValueError(
99
+ f"{name} evaluator returned no numeric score; got keys {list(result)!r}"
100
+ )
101
+ normalized = _normalize(raw)
102
+ label = "pass" if normalized >= threshold else "fail"
103
+ confidence = normalized if label == "pass" else 1.0 - normalized
104
+ return ScoreResult(
105
+ label=label,
106
+ confidence=confidence,
107
+ normalized=normalized,
108
+ features={"raw": raw, "evaluator_result": result},
109
+ )
110
+
111
+
112
+ def _normalize(raw: float) -> float:
113
+ """Project a raw evaluator score into ``[0, 1]``.
114
+
115
+ Common ``azure-ai-evaluation`` evaluators emit scores on a 1-5
116
+ Likert scale. Scores already in ``[0, 1]`` are passed through
117
+ unchanged.
118
+ """
119
+ if 0.0 <= raw <= 1.0:
120
+ return raw
121
+ if 1.0 <= raw <= 5.0:
122
+ return (raw - 1.0) / 4.0
123
+ return max(0.0, min(1.0, raw))
124
+
125
+
126
+ __all__ = ["_LlmScorerWrapper"]
@@ -0,0 +1,21 @@
1
+ """LLM-backed task-adherence scorer."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from ...config import ScoreConfig
6
+ from ._base import _LlmScorerWrapper
7
+
8
+
9
+ class LlmAdherenceScorer(_LlmScorerWrapper):
10
+ """Defers to ``azure.ai.evaluation.TaskAdherenceEvaluator``."""
11
+
12
+ def __init__(self, cfg: ScoreConfig):
13
+ super().__init__(
14
+ cfg=cfg,
15
+ name="adherence",
16
+ label_name="adherence",
17
+ evaluator_attr="TaskAdherenceEvaluator",
18
+ )
19
+
20
+
21
+ __all__ = ["LlmAdherenceScorer"]
@@ -0,0 +1,21 @@
1
+ """LLM-backed task-completion scorer."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from ...config import ScoreConfig
6
+ from ._base import _LlmScorerWrapper
7
+
8
+
9
+ class LlmCompletionScorer(_LlmScorerWrapper):
10
+ """Defers to ``azure.ai.evaluation.TaskCompletionEvaluator``."""
11
+
12
+ def __init__(self, cfg: ScoreConfig):
13
+ super().__init__(
14
+ cfg=cfg,
15
+ name="completion",
16
+ label_name="completion",
17
+ evaluator_attr="TaskCompletionEvaluator",
18
+ )
19
+
20
+
21
+ __all__ = ["LlmCompletionScorer"]
@@ -0,0 +1,21 @@
1
+ """LLM-backed intent-resolution scorer."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from ...config import ScoreConfig
6
+ from ._base import _LlmScorerWrapper
7
+
8
+
9
+ class LlmIntentScorer(_LlmScorerWrapper):
10
+ """Defers to ``azure.ai.evaluation.IntentResolutionEvaluator``."""
11
+
12
+ def __init__(self, cfg: ScoreConfig):
13
+ super().__init__(
14
+ cfg=cfg,
15
+ name="intent",
16
+ label_name="intent",
17
+ evaluator_attr="IntentResolutionEvaluator",
18
+ )
19
+
20
+
21
+ __all__ = ["LlmIntentScorer"]
@@ -0,0 +1,16 @@
1
+ """Native, in-process scorers (Tier 0, pure standard library).
2
+
3
+ These wrap the existing :class:`agent_learning.classifiers.scorers.BinaryScorer`
4
+ implementations so the SDK can ship a working scorer stack with no
5
+ external service dependencies. The wrappers translate the underlying
6
+ :class:`agent_learning.classifiers.base.ClassifierResult` into the
7
+ backend-neutral :class:`agent_learning.scorers.base.ScoreResult`.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from .adherence import NlpAdherenceScorer
13
+ from .completion import NlpCompletionScorer
14
+ from .intent import NlpIntentScorer
15
+
16
+ __all__ = ["NlpAdherenceScorer", "NlpCompletionScorer", "NlpIntentScorer"]
@@ -0,0 +1,99 @@
1
+ """Shared wrapper that adapts a :class:`BinaryScorer` to the SDK
2
+ :class:`ScoreResult` contract.
3
+
4
+ The wrapper reads a snapshot from ``cfg.snapshot_dir/{name}.json`` at
5
+ load time. When no snapshot is present the scorer stays in its unfitted
6
+ "always-fail with zero confidence" state, which keeps the reward
7
+ shaper safe to call but encourages operators to train the scorers.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ import os
14
+ from dataclasses import dataclass
15
+ from typing import List, Optional
16
+
17
+ from ...classifiers.scorers._base import BinaryScorer
18
+ from ...config import NlpScoreConfig
19
+ from ..base import ScoreResult
20
+
21
+
22
+ @dataclass
23
+ class _NlpScorerWrapper:
24
+ """Common base for the three concrete NLP scorers."""
25
+
26
+ name: str
27
+ label_name: str
28
+ _impl: BinaryScorer
29
+ _pass_threshold: float = 0.5
30
+
31
+ @classmethod
32
+ def _build(cls, name: str, cfg: NlpScoreConfig) -> "_NlpScorerWrapper":
33
+ impl = BinaryScorer(label_name=name)
34
+ snapshot_path = os.path.join(cfg.snapshot_dir, f"{name}.json")
35
+ if os.path.isfile(snapshot_path):
36
+ with open(snapshot_path, "r", encoding="utf-8") as fh:
37
+ impl = BinaryScorer.from_snapshot(json.load(fh))
38
+ impl.label_name = name
39
+ return cls(
40
+ name=name,
41
+ label_name=name,
42
+ _impl=impl,
43
+ _pass_threshold=float(cfg.pass_threshold),
44
+ )
45
+
46
+ def fit(self, training_rows: List[dict]) -> "_NlpScorerWrapper":
47
+ """Fit the underlying logistic-regression head.
48
+
49
+ Training rows must look like ``{"phi": [...], "action_id": str, "label": 0|1}``.
50
+ """
51
+ self._impl.fit(training_rows)
52
+ return self
53
+
54
+ def save(self, snapshot_dir: str) -> str:
55
+ """Persist the fitted weights to ``{snapshot_dir}/{name}.json``."""
56
+ os.makedirs(snapshot_dir, exist_ok=True)
57
+ path = os.path.join(snapshot_dir, f"{self.name}.json")
58
+ with open(path, "w", encoding="utf-8") as fh:
59
+ json.dump(self._impl.to_snapshot(), fh)
60
+ return path
61
+
62
+ def score(
63
+ self,
64
+ *,
65
+ phi: Optional[List[float]] = None,
66
+ action_id: Optional[str] = None,
67
+ **_: object,
68
+ ) -> ScoreResult:
69
+ """Score one episode given a context vector and chosen action.
70
+
71
+ Extra keyword arguments are ignored so the same call site works
72
+ when the LLM backend is swapped in.
73
+ """
74
+ if phi is None or action_id is None:
75
+ raise ValueError(
76
+ f"{self.name} NLP scorer requires phi and action_id"
77
+ )
78
+ result = self._impl.score(phi=list(phi), action_id=str(action_id))
79
+ probability = float(result.features.get("probability", result.confidence))
80
+ label = "pass" if probability >= self._pass_threshold else "fail"
81
+ if label == "pass":
82
+ confidence = probability
83
+ normalized = probability
84
+ else:
85
+ confidence = 1.0 - probability
86
+ normalized = probability # keep the raw "pass probability" for shaping
87
+ features = {
88
+ "probability": probability,
89
+ "fitted": bool(self._impl.weights),
90
+ }
91
+ return ScoreResult(
92
+ label=label,
93
+ confidence=confidence,
94
+ normalized=normalized,
95
+ features=features,
96
+ )
97
+
98
+
99
+ __all__ = ["_NlpScorerWrapper"]
@@ -0,0 +1,17 @@
1
+ """NLP task-adherence scorer."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from ...config import NlpScoreConfig
6
+ from ._base import _NlpScorerWrapper
7
+
8
+
9
+ class NlpAdherenceScorer(_NlpScorerWrapper):
10
+ """Predict whether the response adheres to the requested task contract."""
11
+
12
+ @classmethod
13
+ def load_or_default(cls, cfg: NlpScoreConfig) -> "NlpAdherenceScorer":
14
+ return cls._build("adherence", cfg) # type: ignore[return-value]
15
+
16
+
17
+ __all__ = ["NlpAdherenceScorer"]
@@ -0,0 +1,17 @@
1
+ """NLP task-completion scorer."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from ...config import NlpScoreConfig
6
+ from ._base import _NlpScorerWrapper
7
+
8
+
9
+ class NlpCompletionScorer(_NlpScorerWrapper):
10
+ """Predict whether the response completes the requested task."""
11
+
12
+ @classmethod
13
+ def load_or_default(cls, cfg: NlpScoreConfig) -> "NlpCompletionScorer":
14
+ return cls._build("completion", cfg) # type: ignore[return-value]
15
+
16
+
17
+ __all__ = ["NlpCompletionScorer"]
@@ -0,0 +1,17 @@
1
+ """NLP intent-resolution scorer."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from ...config import NlpScoreConfig
6
+ from ._base import _NlpScorerWrapper
7
+
8
+
9
+ class NlpIntentScorer(_NlpScorerWrapper):
10
+ """Predict whether the chosen action addresses the requester's intent."""
11
+
12
+ @classmethod
13
+ def load_or_default(cls, cfg: NlpScoreConfig) -> "NlpIntentScorer":
14
+ return cls._build("intent", cfg) # type: ignore[return-value]
15
+
16
+
17
+ __all__ = ["NlpIntentScorer"]
@@ -0,0 +1,26 @@
1
+ """Tier 2 NLP scorers (TF-IDF + scikit-learn) over raw query/response text.
2
+
3
+ These scorers operate on the same ``(query, response)`` pair the Tier 1
4
+ stdlib scorers accept and the same Tier 4 LLM scorers accept, but use a
5
+ TF-IDF vectorizer + logistic-regression head from scikit-learn for the
6
+ intent and adherence/completion classifiers. The adherence scorer also
7
+ applies the same deterministic rule engine the stdlib backend uses;
8
+ the rule-engine score is combined with the learned probability.
9
+
10
+ Requires the ``[nlp]`` extra (``pip install
11
+ agent-learning[nlp]``). All scikit-learn imports are lazy so
12
+ the package can still be imported without the extra; calling
13
+ :meth:`fit` or :meth:`score` raises a helpful ``ImportError`` then.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ from .adherence import NlpTextAdherenceScorer
19
+ from .completion import NlpTextCompletionScorer
20
+ from .intent import NlpTextIntentScorer
21
+
22
+ __all__ = [
23
+ "NlpTextAdherenceScorer",
24
+ "NlpTextCompletionScorer",
25
+ "NlpTextIntentScorer",
26
+ ]