agent-learning 0.4.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning/__init__.py +160 -0
- agent_learning/_version.py +3 -0
- agent_learning/capture.py +271 -0
- agent_learning/classifiers/__init__.py +37 -0
- agent_learning/classifiers/base.py +185 -0
- agent_learning/classifiers/router.py +236 -0
- agent_learning/classifiers/scorers/__init__.py +34 -0
- agent_learning/classifiers/scorers/_base.py +162 -0
- agent_learning/classifiers/scorers/adherence.py +26 -0
- agent_learning/classifiers/scorers/completion.py +26 -0
- agent_learning/classifiers/scorers/intent.py +25 -0
- agent_learning/cli.py +386 -0
- agent_learning/config.py +524 -0
- agent_learning/learners/__init__.py +6 -0
- agent_learning/learners/base.py +38 -0
- agent_learning/learners/reinforce.py +153 -0
- agent_learning/metrics/__init__.py +22 -0
- agent_learning/metrics/base.py +233 -0
- agent_learning/metrics/intent_resolution.py +50 -0
- agent_learning/metrics/registry.py +42 -0
- agent_learning/metrics/task_adherence.py +42 -0
- agent_learning/metrics/task_completion.py +54 -0
- agent_learning/policy/__init__.py +7 -0
- agent_learning/policy/base.py +59 -0
- agent_learning/policy/contextual_softmax.py +243 -0
- agent_learning/policy/softmax_bandit.py +157 -0
- agent_learning/py.typed +1 -0
- agent_learning/rewards/__init__.py +6 -0
- agent_learning/rewards/shaping.py +121 -0
- agent_learning/rewards/writer.py +130 -0
- agent_learning/scorers/__init__.py +187 -0
- agent_learning/scorers/base.py +49 -0
- agent_learning/scorers/llm/__init__.py +19 -0
- agent_learning/scorers/llm/_base.py +126 -0
- agent_learning/scorers/llm/adherence.py +21 -0
- agent_learning/scorers/llm/completion.py +21 -0
- agent_learning/scorers/llm/intent.py +21 -0
- agent_learning/scorers/nlp/__init__.py +16 -0
- agent_learning/scorers/nlp/_base.py +99 -0
- agent_learning/scorers/nlp/adherence.py +17 -0
- agent_learning/scorers/nlp/completion.py +17 -0
- agent_learning/scorers/nlp/intent.py +17 -0
- agent_learning/scorers/nlp_text/__init__.py +26 -0
- agent_learning/scorers/nlp_text/_base.py +234 -0
- agent_learning/scorers/nlp_text/adherence.py +94 -0
- agent_learning/scorers/nlp_text/completion.py +91 -0
- agent_learning/scorers/nlp_text/intent.py +59 -0
- agent_learning/scorers/slm/__init__.py +25 -0
- agent_learning/scorers/slm/_base.py +292 -0
- agent_learning/scorers/slm/adherence.py +98 -0
- agent_learning/scorers/slm/completion.py +111 -0
- agent_learning/scorers/slm/intent.py +80 -0
- agent_learning/scorers/stdlib/__init__.py +38 -0
- agent_learning/scorers/stdlib/_text.py +87 -0
- agent_learning/scorers/stdlib/adherence.py +157 -0
- agent_learning/scorers/stdlib/completion.py +117 -0
- agent_learning/scorers/stdlib/intent.py +182 -0
- agent_learning/storage/__init__.py +14 -0
- agent_learning/storage/base.py +156 -0
- agent_learning/storage/cosmos.py +506 -0
- agent_learning/storage/local.py +353 -0
- agent_learning/storage/memory.py +209 -0
- agent_learning/training/__init__.py +5 -0
- agent_learning/training/runner.py +172 -0
- agent_learning/types.py +507 -0
- agent_learning-0.4.1.dist-info/METADATA +82 -0
- agent_learning-0.4.1.dist-info/RECORD +71 -0
- agent_learning-0.4.1.dist-info/WHEEL +5 -0
- agent_learning-0.4.1.dist-info/entry_points.txt +2 -0
- agent_learning-0.4.1.dist-info/licenses/LICENSE +21 -0
- agent_learning-0.4.1.dist-info/top_level.txt +1 -0
agent_learning/config.py
ADDED
|
@@ -0,0 +1,524 @@
|
|
|
1
|
+
"""Environment-driven configuration for the agent-learning SDK.
|
|
2
|
+
|
|
3
|
+
All settings can be overridden via environment variables; programmatic
|
|
4
|
+
overrides are also accepted via dataclass field assignment.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import os
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from typing import Any, Literal, Optional
|
|
12
|
+
|
|
13
|
+
ScoreMode = Literal["nlp", "llm"]
|
|
14
|
+
ScoreTier = Literal["stdlib", "nlp", "slm", "llm"]
|
|
15
|
+
CredentialMode = Literal[
|
|
16
|
+
"default",
|
|
17
|
+
"managed-identity",
|
|
18
|
+
"workload-identity",
|
|
19
|
+
"environment",
|
|
20
|
+
"azure-cli",
|
|
21
|
+
"none",
|
|
22
|
+
]
|
|
23
|
+
_VALID_CREDENTIAL_MODES: frozenset[str] = frozenset(
|
|
24
|
+
{"default", "managed-identity", "workload-identity", "environment", "azure-cli", "none"}
|
|
25
|
+
)
|
|
26
|
+
_VALID_SCORE_TIERS: frozenset[str] = frozenset(
|
|
27
|
+
{"stdlib", "nlp", "slm", "llm"}
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _env_bool(name: str, default: bool) -> bool:
|
|
32
|
+
"""Parse a boolean environment variable using common conventions."""
|
|
33
|
+
raw = os.getenv(name)
|
|
34
|
+
if raw is None:
|
|
35
|
+
return default
|
|
36
|
+
return raw.strip().lower() in {"1", "true", "yes", "on"}
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _env_credential_mode() -> Optional[CredentialMode]:
|
|
40
|
+
"""Read AGENT_LEARNING_SCORE_CREDENTIAL_MODE and validate."""
|
|
41
|
+
raw = os.getenv("AGENT_LEARNING_SCORE_CREDENTIAL_MODE")
|
|
42
|
+
if raw is None or raw == "":
|
|
43
|
+
return None
|
|
44
|
+
value = raw.strip().lower()
|
|
45
|
+
if value in _VALID_CREDENTIAL_MODES:
|
|
46
|
+
return value # type: ignore[return-value]
|
|
47
|
+
return None
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _env_float(name: str, default: float) -> float:
|
|
51
|
+
raw = os.getenv(name)
|
|
52
|
+
if raw is None or raw == "":
|
|
53
|
+
return default
|
|
54
|
+
try:
|
|
55
|
+
return float(raw)
|
|
56
|
+
except ValueError:
|
|
57
|
+
return default
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _env_int(name: str, default: int) -> int:
|
|
61
|
+
raw = os.getenv(name)
|
|
62
|
+
if raw is None or raw == "":
|
|
63
|
+
return default
|
|
64
|
+
try:
|
|
65
|
+
return int(raw)
|
|
66
|
+
except ValueError:
|
|
67
|
+
return default
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
# ---------------------------------------------------------------------------
|
|
71
|
+
# Cosmos DB
|
|
72
|
+
# ---------------------------------------------------------------------------
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass
|
|
76
|
+
class CosmosConfig:
|
|
77
|
+
"""Cosmos DB connection + container configuration."""
|
|
78
|
+
|
|
79
|
+
endpoint: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_COSMOS_ENDPOINT", ""))
|
|
80
|
+
database_name: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_COSMOS_DATABASE", "dq_rl"))
|
|
81
|
+
auth_mode: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_COSMOS_AUTH_MODE", "aad"))
|
|
82
|
+
account_key: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_COSMOS_KEY", ""))
|
|
83
|
+
partition_key_field: str = field(
|
|
84
|
+
default_factory=lambda: os.getenv("AGENT_LEARNING_PARTITION_KEY_FIELD", "agent_id")
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
container_episodes: str = field(
|
|
88
|
+
default_factory=lambda: os.getenv("AGENT_LEARNING_CONTAINER_EPISODES", "learning_episodes")
|
|
89
|
+
)
|
|
90
|
+
container_rewards: str = field(
|
|
91
|
+
default_factory=lambda: os.getenv("AGENT_LEARNING_CONTAINER_REWARDS", "learning_rewards")
|
|
92
|
+
)
|
|
93
|
+
container_metrics: str = field(
|
|
94
|
+
default_factory=lambda: os.getenv("AGENT_LEARNING_CONTAINER_METRICS", "learning_metrics")
|
|
95
|
+
)
|
|
96
|
+
container_policies: str = field(
|
|
97
|
+
default_factory=lambda: os.getenv("AGENT_LEARNING_CONTAINER_POLICIES", "learning_policies")
|
|
98
|
+
)
|
|
99
|
+
container_runs: str = field(
|
|
100
|
+
default_factory=lambda: os.getenv("AGENT_LEARNING_CONTAINER_RUNS", "learning_runs")
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
@property
|
|
104
|
+
def enabled(self) -> bool:
|
|
105
|
+
return bool(self.endpoint)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
# ---------------------------------------------------------------------------
|
|
109
|
+
# Score / evaluator model
|
|
110
|
+
# ---------------------------------------------------------------------------
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
@dataclass
|
|
114
|
+
class ScoreConfig:
|
|
115
|
+
"""Configuration for the LLM scorer used by metric evaluators.
|
|
116
|
+
|
|
117
|
+
Authentication paths, in priority order:
|
|
118
|
+
|
|
119
|
+
1. ``credential`` set explicitly to a ``TokenCredential`` instance.
|
|
120
|
+
Wins over every other field. The SDK passes a bearer-token
|
|
121
|
+
provider built from it to the underlying evaluator.
|
|
122
|
+
2. ``credential_mode`` set to one of ``"default"``,
|
|
123
|
+
``"managed-identity"``, ``"workload-identity"``,
|
|
124
|
+
``"environment"``, ``"azure-cli"`` (or ``"none"`` to opt out
|
|
125
|
+
even when an api_key is also present). The SDK lazy-imports
|
|
126
|
+
``azure-identity`` and builds the matching credential.
|
|
127
|
+
``user_assigned_client_id`` is forwarded when provided.
|
|
128
|
+
3. ``api_key`` (legacy path). Used only when no credential is
|
|
129
|
+
resolved.
|
|
130
|
+
|
|
131
|
+
Environment variables ``AGENT_LEARNING_SCORE_CREDENTIAL_MODE`` and
|
|
132
|
+
``AGENT_LEARNING_SCORE_USER_ASSIGNED_CLIENT_ID`` mirror the two
|
|
133
|
+
string fields so the same SDK build switches between developer
|
|
134
|
+
laptop, CI, and production AKS without code changes.
|
|
135
|
+
"""
|
|
136
|
+
|
|
137
|
+
azure_endpoint: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_SCORE_ENDPOINT", ""))
|
|
138
|
+
azure_deployment: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_SCORE_DEPLOYMENT", ""))
|
|
139
|
+
api_key: Optional[str] = field(default_factory=lambda: os.getenv("AGENT_LEARNING_SCORE_API_KEY") or None)
|
|
140
|
+
api_version: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_SCORE_API_VERSION", "2024-10-21"))
|
|
141
|
+
# Pass-through threshold for IntentResolutionEvaluator (defaults to its own default)
|
|
142
|
+
threshold: int = field(default_factory=lambda: _env_int("AGENT_LEARNING_INTENT_THRESHOLD", 3))
|
|
143
|
+
# Managed-identity / TokenCredential authentication (optional).
|
|
144
|
+
credential: Optional[Any] = field(default=None, repr=False)
|
|
145
|
+
credential_mode: Optional[CredentialMode] = field(default_factory=_env_credential_mode)
|
|
146
|
+
user_assigned_client_id: Optional[str] = field(
|
|
147
|
+
default_factory=lambda: os.getenv("AGENT_LEARNING_SCORE_USER_ASSIGNED_CLIENT_ID") or None
|
|
148
|
+
)
|
|
149
|
+
# Azure AD token scope for the resolved credential. Defaults to the
|
|
150
|
+
# Cognitive Services scope used by Azure OpenAI and Azure AI Foundry.
|
|
151
|
+
credential_scope: str = field(
|
|
152
|
+
default_factory=lambda: os.getenv(
|
|
153
|
+
"AGENT_LEARNING_SCORE_CREDENTIAL_SCOPE",
|
|
154
|
+
"https://cognitiveservices.azure.com/.default",
|
|
155
|
+
)
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
@property
|
|
159
|
+
def enabled(self) -> bool:
|
|
160
|
+
return bool(self.azure_endpoint and self.azure_deployment)
|
|
161
|
+
|
|
162
|
+
def resolve_credential(self) -> Optional[Any]:
|
|
163
|
+
"""Resolve the active TokenCredential, or ``None``.
|
|
164
|
+
|
|
165
|
+
Returns the explicit ``credential`` field when set. Otherwise
|
|
166
|
+
builds a credential from ``credential_mode`` via a lazy import
|
|
167
|
+
of ``azure-identity``. Returns ``None`` when no credential is
|
|
168
|
+
configured (caller falls back to ``api_key``).
|
|
169
|
+
"""
|
|
170
|
+
if self.credential is not None:
|
|
171
|
+
return self.credential
|
|
172
|
+
mode = self.credential_mode
|
|
173
|
+
if mode is None or mode == "none":
|
|
174
|
+
return None
|
|
175
|
+
try:
|
|
176
|
+
import azure.identity as az_id # type: ignore[import-not-found]
|
|
177
|
+
except ImportError as exc:
|
|
178
|
+
raise ImportError(
|
|
179
|
+
"Credential-based auth for the LLM scorer requires the optional "
|
|
180
|
+
"'azure-identity' package. Install it with: pip install azure-identity"
|
|
181
|
+
) from exc
|
|
182
|
+
if mode == "default":
|
|
183
|
+
kwargs: dict = {}
|
|
184
|
+
if self.user_assigned_client_id:
|
|
185
|
+
kwargs["managed_identity_client_id"] = self.user_assigned_client_id
|
|
186
|
+
return az_id.DefaultAzureCredential(**kwargs)
|
|
187
|
+
if mode == "managed-identity":
|
|
188
|
+
if self.user_assigned_client_id:
|
|
189
|
+
return az_id.ManagedIdentityCredential(client_id=self.user_assigned_client_id)
|
|
190
|
+
return az_id.ManagedIdentityCredential()
|
|
191
|
+
if mode == "workload-identity":
|
|
192
|
+
if self.user_assigned_client_id:
|
|
193
|
+
return az_id.WorkloadIdentityCredential(client_id=self.user_assigned_client_id)
|
|
194
|
+
return az_id.WorkloadIdentityCredential()
|
|
195
|
+
if mode == "environment":
|
|
196
|
+
return az_id.EnvironmentCredential()
|
|
197
|
+
if mode == "azure-cli":
|
|
198
|
+
return az_id.AzureCliCredential()
|
|
199
|
+
raise ValueError(f"Unknown credential_mode: {mode!r}")
|
|
200
|
+
|
|
201
|
+
def to_model_config(self) -> dict:
|
|
202
|
+
"""Build the dict expected by azure-ai-evaluation evaluators.
|
|
203
|
+
|
|
204
|
+
When a TokenCredential is resolved, the dict carries an
|
|
205
|
+
``azure_ad_token_provider`` callable (built via
|
|
206
|
+
``azure.identity.get_bearer_token_provider``) and the
|
|
207
|
+
``api_key`` field is omitted. Otherwise the legacy ``api_key``
|
|
208
|
+
path is used.
|
|
209
|
+
"""
|
|
210
|
+
cfg: dict = {
|
|
211
|
+
"azure_endpoint": self.azure_endpoint,
|
|
212
|
+
"azure_deployment": self.azure_deployment,
|
|
213
|
+
"api_version": self.api_version,
|
|
214
|
+
}
|
|
215
|
+
credential = self.resolve_credential()
|
|
216
|
+
if credential is not None:
|
|
217
|
+
try:
|
|
218
|
+
from azure.identity import ( # type: ignore[import-not-found]
|
|
219
|
+
get_bearer_token_provider,
|
|
220
|
+
)
|
|
221
|
+
except ImportError as exc:
|
|
222
|
+
raise ImportError(
|
|
223
|
+
"Credential-based auth for the LLM scorer requires the optional "
|
|
224
|
+
"'azure-identity' package. Install it with: pip install azure-identity"
|
|
225
|
+
) from exc
|
|
226
|
+
cfg["azure_ad_token_provider"] = get_bearer_token_provider(
|
|
227
|
+
credential, self.credential_scope
|
|
228
|
+
)
|
|
229
|
+
elif self.api_key:
|
|
230
|
+
cfg["api_key"] = self.api_key
|
|
231
|
+
return cfg
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
@dataclass
|
|
235
|
+
class NlpScoreConfig:
|
|
236
|
+
"""Configuration for the in-SDK NLP scoring stack (Tier 0, pure stdlib).
|
|
237
|
+
|
|
238
|
+
The NLP scorers read their fitted weights from ``snapshot_dir`` at
|
|
239
|
+
load time and fall back to an unfitted (always-pass) policy when no
|
|
240
|
+
snapshot is present. Optional dependencies (scikit-learn,
|
|
241
|
+
sentence-transformers, etc.) are detected at import time. Tier 0
|
|
242
|
+
requires nothing beyond the Python standard library.
|
|
243
|
+
"""
|
|
244
|
+
|
|
245
|
+
snapshot_dir: str = field(
|
|
246
|
+
default_factory=lambda: os.getenv(
|
|
247
|
+
"AGENT_LEARNING_NLP_SCORE_DIR", "./data/agent-learning/nlp-scores"
|
|
248
|
+
)
|
|
249
|
+
)
|
|
250
|
+
pass_threshold: float = field(
|
|
251
|
+
default_factory=lambda: _env_float("AGENT_LEARNING_NLP_PASS_THRESHOLD", 0.5)
|
|
252
|
+
)
|
|
253
|
+
enable_semantic: bool = field(
|
|
254
|
+
default_factory=lambda: _env_bool("AGENT_LEARNING_NLP_SEMANTIC", False)
|
|
255
|
+
)
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def _env_score_mode(default: ScoreMode = "llm") -> ScoreMode:
|
|
259
|
+
raw = os.getenv("AGENT_LEARNING_SCORE_MODE")
|
|
260
|
+
if raw is None or raw == "":
|
|
261
|
+
return default
|
|
262
|
+
value = raw.strip().lower()
|
|
263
|
+
if value in ("nlp", "llm"):
|
|
264
|
+
return value # type: ignore[return-value]
|
|
265
|
+
return default
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def _env_score_tier() -> Optional[ScoreTier]:
|
|
269
|
+
"""Read AGENT_LEARNING_SCORE_TIER and validate.
|
|
270
|
+
|
|
271
|
+
Returns one of ``"stdlib"``, ``"nlp"``, ``"slm"``, ``"llm"`` or
|
|
272
|
+
``None`` when the env var is unset or invalid. The factory in
|
|
273
|
+
scoring factory falls back to ``mode`` when ``tier``
|
|
274
|
+
is None.
|
|
275
|
+
"""
|
|
276
|
+
raw = os.getenv("AGENT_LEARNING_SCORE_TIER")
|
|
277
|
+
if raw is None or raw == "":
|
|
278
|
+
return None
|
|
279
|
+
value = raw.strip().lower()
|
|
280
|
+
if value in _VALID_SCORE_TIERS:
|
|
281
|
+
return value # type: ignore[return-value]
|
|
282
|
+
return None
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
@dataclass
|
|
286
|
+
class StdlibScoreConfig:
|
|
287
|
+
"""Configuration for the Tier 1 stdlib scorers.
|
|
288
|
+
|
|
289
|
+
All Tier 1 scorers run on the Python standard library alone. The
|
|
290
|
+
intent scorer optionally loads fitted bag-of-words weights from
|
|
291
|
+
``snapshot_dir``; the adherence and completion scorers are pure
|
|
292
|
+
rule engines with no persisted state.
|
|
293
|
+
"""
|
|
294
|
+
|
|
295
|
+
snapshot_dir: str = field(
|
|
296
|
+
default_factory=lambda: os.getenv(
|
|
297
|
+
"AGENT_LEARNING_STDLIB_SCORE_DIR",
|
|
298
|
+
"./data/agent-learning/stdlib-scores",
|
|
299
|
+
)
|
|
300
|
+
)
|
|
301
|
+
pass_threshold: float = field(
|
|
302
|
+
default_factory=lambda: _env_float(
|
|
303
|
+
"AGENT_LEARNING_STDLIB_PASS_THRESHOLD", 0.5
|
|
304
|
+
)
|
|
305
|
+
)
|
|
306
|
+
feature_dim: int = field(
|
|
307
|
+
default_factory=lambda: _env_int(
|
|
308
|
+
"AGENT_LEARNING_STDLIB_FEATURE_DIM", 1024
|
|
309
|
+
)
|
|
310
|
+
)
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
@dataclass
|
|
314
|
+
class NlpTextScoreConfig:
|
|
315
|
+
"""Configuration for the Tier 2 NLP text scorers.
|
|
316
|
+
|
|
317
|
+
Tier 2 wraps a TF-IDF vectorizer + scikit-learn logistic
|
|
318
|
+
regression around the response text. The fitted vectorizer and
|
|
319
|
+
classifier are persisted to ``{snapshot_dir}/{name}.nlp_text.joblib``
|
|
320
|
+
with a sibling JSON header at ``{name}.nlp_text.json``. When no
|
|
321
|
+
snapshot is present the scorers fall back to a rule-engine signal
|
|
322
|
+
only (adherence, completion) or the pass threshold (intent).
|
|
323
|
+
|
|
324
|
+
Requires the ``[nlp]`` extra (``pip install
|
|
325
|
+
agent-learning[nlp]``).
|
|
326
|
+
"""
|
|
327
|
+
|
|
328
|
+
snapshot_dir: str = field(
|
|
329
|
+
default_factory=lambda: os.getenv(
|
|
330
|
+
"AGENT_LEARNING_NLP_TEXT_SCORE_DIR",
|
|
331
|
+
"./data/agent-learning/nlp-text-scores",
|
|
332
|
+
)
|
|
333
|
+
)
|
|
334
|
+
pass_threshold: float = field(
|
|
335
|
+
default_factory=lambda: _env_float(
|
|
336
|
+
"AGENT_LEARNING_NLP_TEXT_PASS_THRESHOLD", 0.5
|
|
337
|
+
)
|
|
338
|
+
)
|
|
339
|
+
max_features: int = field(
|
|
340
|
+
default_factory=lambda: _env_int(
|
|
341
|
+
"AGENT_LEARNING_NLP_TEXT_MAX_FEATURES", 20000
|
|
342
|
+
)
|
|
343
|
+
)
|
|
344
|
+
ngram_min: int = field(
|
|
345
|
+
default_factory=lambda: _env_int("AGENT_LEARNING_NLP_TEXT_NGRAM_MIN", 1)
|
|
346
|
+
)
|
|
347
|
+
ngram_max: int = field(
|
|
348
|
+
default_factory=lambda: _env_int("AGENT_LEARNING_NLP_TEXT_NGRAM_MAX", 2)
|
|
349
|
+
)
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
@dataclass
|
|
353
|
+
class SlmScoreConfig:
|
|
354
|
+
"""Configuration for the Tier 3 small-language-model scorers.
|
|
355
|
+
|
|
356
|
+
Tier 3 wraps a locally-hosted instance of Microsoft
|
|
357
|
+
Phi-4-mini-instruct (3.8 B parameters, 4-bit ONNX) via
|
|
358
|
+
``onnxruntime-genai``. The model is loaded from ``model_dir`` on
|
|
359
|
+
first use and reused across the three scorers in the same process.
|
|
360
|
+
|
|
361
|
+
Requires the ``[slm]`` extra (``pip install
|
|
362
|
+
agent-learning[slm]``). The default ``model_dir``
|
|
363
|
+
expects a Phi-4-mini-instruct INT4 ONNX bundle laid out under
|
|
364
|
+
``./models/phi-4-mini-instruct-int4-onnx``; set
|
|
365
|
+
``AGENT_LEARNING_SLM_MODEL_DIR`` to point at any local path.
|
|
366
|
+
"""
|
|
367
|
+
|
|
368
|
+
model_dir: str = field(
|
|
369
|
+
default_factory=lambda: os.getenv(
|
|
370
|
+
"AGENT_LEARNING_SLM_MODEL_DIR",
|
|
371
|
+
"./models/phi-4-mini-instruct-int4-onnx",
|
|
372
|
+
)
|
|
373
|
+
)
|
|
374
|
+
pass_threshold: float = field(
|
|
375
|
+
default_factory=lambda: _env_float(
|
|
376
|
+
"AGENT_LEARNING_SLM_PASS_THRESHOLD", 0.5
|
|
377
|
+
)
|
|
378
|
+
)
|
|
379
|
+
max_new_tokens: int = field(
|
|
380
|
+
default_factory=lambda: _env_int(
|
|
381
|
+
"AGENT_LEARNING_SLM_MAX_NEW_TOKENS", 64
|
|
382
|
+
)
|
|
383
|
+
)
|
|
384
|
+
temperature: float = field(
|
|
385
|
+
default_factory=lambda: _env_float(
|
|
386
|
+
"AGENT_LEARNING_SLM_TEMPERATURE", 0.0
|
|
387
|
+
)
|
|
388
|
+
)
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
@dataclass
|
|
392
|
+
class ScoreRuntimeConfig:
|
|
393
|
+
"""Top-level switch across the four scoring tiers.
|
|
394
|
+
|
|
395
|
+
Two selectors exist, in priority order:
|
|
396
|
+
|
|
397
|
+
1. ``tier`` (preferred). One of ``"stdlib"`` (Tier 1, zero deps),
|
|
398
|
+
``"nlp"`` (Tier 2, scikit-learn + rapidfuzz, ``[nlp]`` extra),
|
|
399
|
+
``"slm"`` (Tier 3, Phi-4-mini-instruct ONNX, ``[slm]`` extra),
|
|
400
|
+
or ``"llm"`` (Tier 4, azure-ai-evaluation, ``[llm]`` extra).
|
|
401
|
+
2. ``mode`` (legacy). One of ``"nlp"`` (the existing phi+action_id
|
|
402
|
+
binary scoring stack) or ``"llm"`` (Azure AI Evaluation).
|
|
403
|
+
Used when ``tier`` is ``None``.
|
|
404
|
+
|
|
405
|
+
The default ``mode`` stays at ``"llm"`` for backwards compatibility
|
|
406
|
+
with callers already wired through azure-ai-evaluation. Set
|
|
407
|
+
``AGENT_LEARNING_SCORE_TIER=stdlib`` to opt in to the new Tier 1
|
|
408
|
+
text-based scorers.
|
|
409
|
+
"""
|
|
410
|
+
|
|
411
|
+
mode: ScoreMode = field(default_factory=_env_score_mode)
|
|
412
|
+
tier: Optional[ScoreTier] = field(default_factory=_env_score_tier)
|
|
413
|
+
stdlib: StdlibScoreConfig = field(default_factory=StdlibScoreConfig)
|
|
414
|
+
nlp: NlpScoreConfig = field(default_factory=NlpScoreConfig)
|
|
415
|
+
nlp_text: NlpTextScoreConfig = field(default_factory=NlpTextScoreConfig)
|
|
416
|
+
slm: SlmScoreConfig = field(default_factory=SlmScoreConfig)
|
|
417
|
+
llm: ScoreConfig = field(default_factory=ScoreConfig)
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
# ---------------------------------------------------------------------------
|
|
421
|
+
# Capture / shaping
|
|
422
|
+
# ---------------------------------------------------------------------------
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
@dataclass
|
|
426
|
+
class CaptureConfig:
|
|
427
|
+
"""Configuration for episode capture."""
|
|
428
|
+
|
|
429
|
+
enabled: bool = field(default_factory=lambda: _env_bool("AGENT_LEARNING_ENABLE_CAPTURE", False))
|
|
430
|
+
agent_id: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_AGENT_ID", "default"))
|
|
431
|
+
agent_name: Optional[str] = field(
|
|
432
|
+
default_factory=lambda: os.getenv("AGENT_LEARNING_AGENT_NAME") or None
|
|
433
|
+
)
|
|
434
|
+
task_id: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_TASK_ID", "default"))
|
|
435
|
+
task_name: Optional[str] = field(
|
|
436
|
+
default_factory=lambda: os.getenv("AGENT_LEARNING_TASK_NAME") or None
|
|
437
|
+
)
|
|
438
|
+
local_fallback_dir: str = field(
|
|
439
|
+
default_factory=lambda: os.getenv("AGENT_LEARNING_DATA_DIR", "./data/agent-learning")
|
|
440
|
+
)
|
|
441
|
+
max_output_length: int = field(default_factory=lambda: _env_int("AGENT_LEARNING_MAX_OUTPUT_LEN", 10000))
|
|
442
|
+
redact_secrets: bool = field(default_factory=lambda: _env_bool("AGENT_LEARNING_REDACT_SECRETS", True))
|
|
443
|
+
|
|
444
|
+
|
|
445
|
+
@dataclass
|
|
446
|
+
class ShapingConfig:
|
|
447
|
+
"""Weights and penalties used to combine metric scores into a scalar reward.
|
|
448
|
+
|
|
449
|
+
Each weight is multiplied against the metric's normalized score (in
|
|
450
|
+
[0, 1]) before summing. Weights need not sum to 1 — the shaper will
|
|
451
|
+
return the raw weighted sum, clamped to [-1, 1].
|
|
452
|
+
|
|
453
|
+
The defaults bias the credit-assignment toward task completion (the
|
|
454
|
+
metric the action template directly controls) and away from intent
|
|
455
|
+
resolution (typically computed by an upstream classifier the policy
|
|
456
|
+
cannot influence).
|
|
457
|
+
"""
|
|
458
|
+
|
|
459
|
+
intent_resolution_weight: float = field(
|
|
460
|
+
default_factory=lambda: _env_float("AGENT_LEARNING_W_INTENT", 0.10)
|
|
461
|
+
)
|
|
462
|
+
task_adherence_weight: float = field(
|
|
463
|
+
default_factory=lambda: _env_float("AGENT_LEARNING_W_ADHERENCE", 0.20)
|
|
464
|
+
)
|
|
465
|
+
task_completion_weight: float = field(
|
|
466
|
+
default_factory=lambda: _env_float("AGENT_LEARNING_W_COMPLETION", 0.50)
|
|
467
|
+
)
|
|
468
|
+
latency_penalty_threshold_ms: int = field(
|
|
469
|
+
default_factory=lambda: _env_int("AGENT_LEARNING_LATENCY_THRESHOLD_MS", 15000)
|
|
470
|
+
)
|
|
471
|
+
latency_penalty_value: float = field(
|
|
472
|
+
default_factory=lambda: _env_float("AGENT_LEARNING_LATENCY_PENALTY", -0.1)
|
|
473
|
+
)
|
|
474
|
+
# Routing terms. Computed by the calling system and passed to
|
|
475
|
+
# ``RewardShaper.shape(..., routing_correct=..., hallucinated_class=...)``.
|
|
476
|
+
# The values below are the magnitudes used when the relevant condition
|
|
477
|
+
# fires.
|
|
478
|
+
route_correct_reward: float = field(
|
|
479
|
+
default_factory=lambda: _env_float("AGENT_LEARNING_R_ROUTE_CORRECT", 0.20)
|
|
480
|
+
)
|
|
481
|
+
route_wrong_penalty: float = field(
|
|
482
|
+
default_factory=lambda: _env_float("AGENT_LEARNING_P_ROUTE_WRONG", -0.30)
|
|
483
|
+
)
|
|
484
|
+
hallucinated_class_penalty: float = field(
|
|
485
|
+
default_factory=lambda: _env_float("AGENT_LEARNING_P_HALLU", -0.25)
|
|
486
|
+
)
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
# ---------------------------------------------------------------------------
|
|
490
|
+
# Learner
|
|
491
|
+
# ---------------------------------------------------------------------------
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
@dataclass
|
|
495
|
+
class LearnerConfig:
|
|
496
|
+
"""Hyperparameters for the REINFORCE-with-baseline learner."""
|
|
497
|
+
|
|
498
|
+
learning_rate: float = field(default_factory=lambda: _env_float("AGENT_LEARNING_LR", 0.05))
|
|
499
|
+
baseline_decay: float = field(default_factory=lambda: _env_float("AGENT_LEARNING_BASELINE_DECAY", 0.9))
|
|
500
|
+
entropy_bonus: float = field(default_factory=lambda: _env_float("AGENT_LEARNING_ENTROPY_BONUS", 0.01))
|
|
501
|
+
max_logit_abs: float = field(default_factory=lambda: _env_float("AGENT_LEARNING_MAX_LOGIT", 10.0))
|
|
502
|
+
importance_clip: float = field(
|
|
503
|
+
default_factory=lambda: _env_float("AGENT_LEARNING_IMPORTANCE_CLIP", 5.0)
|
|
504
|
+
)
|
|
505
|
+
min_train_episodes: int = field(
|
|
506
|
+
default_factory=lambda: _env_int("AGENT_LEARNING_MIN_TRAIN_EPISODES", 5)
|
|
507
|
+
)
|
|
508
|
+
|
|
509
|
+
|
|
510
|
+
__all__ = [
|
|
511
|
+
"CaptureConfig",
|
|
512
|
+
"CosmosConfig",
|
|
513
|
+
"CredentialMode",
|
|
514
|
+
"ScoreConfig",
|
|
515
|
+
"ScoreMode",
|
|
516
|
+
"ScoreRuntimeConfig",
|
|
517
|
+
"ScoreTier",
|
|
518
|
+
"LearnerConfig",
|
|
519
|
+
"NlpScoreConfig",
|
|
520
|
+
"NlpTextScoreConfig",
|
|
521
|
+
"ShapingConfig",
|
|
522
|
+
"SlmScoreConfig",
|
|
523
|
+
"StdlibScoreConfig",
|
|
524
|
+
]
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Learner interface."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from abc import ABC, abstractmethod
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
from typing import Any, Dict, Iterable, List
|
|
8
|
+
|
|
9
|
+
from ..policy.base import Policy
|
|
10
|
+
from ..types import Episode, Reward
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass
|
|
14
|
+
class LearnerResult:
|
|
15
|
+
"""Summary returned by :meth:`Learner.update`."""
|
|
16
|
+
|
|
17
|
+
episodes_used: int
|
|
18
|
+
mean_reward: float
|
|
19
|
+
baseline_before: float
|
|
20
|
+
baseline_after: float
|
|
21
|
+
logit_deltas: Dict[str, float] = field(default_factory=dict)
|
|
22
|
+
extra: Dict[str, Any] = field(default_factory=dict)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class Learner(ABC):
|
|
26
|
+
"""Abstract base class for an agent-learning algorithm."""
|
|
27
|
+
|
|
28
|
+
@abstractmethod
|
|
29
|
+
def update(
|
|
30
|
+
self,
|
|
31
|
+
policy: Policy,
|
|
32
|
+
episodes: Iterable[Episode],
|
|
33
|
+
rewards: Iterable[Reward],
|
|
34
|
+
) -> LearnerResult:
|
|
35
|
+
"""Apply one update to ``policy`` using the supplied data."""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
__all__ = ["Learner", "LearnerResult"]
|