agent-learning 0.4.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. agent_learning/__init__.py +160 -0
  2. agent_learning/_version.py +3 -0
  3. agent_learning/capture.py +271 -0
  4. agent_learning/classifiers/__init__.py +37 -0
  5. agent_learning/classifiers/base.py +185 -0
  6. agent_learning/classifiers/router.py +236 -0
  7. agent_learning/classifiers/scorers/__init__.py +34 -0
  8. agent_learning/classifiers/scorers/_base.py +162 -0
  9. agent_learning/classifiers/scorers/adherence.py +26 -0
  10. agent_learning/classifiers/scorers/completion.py +26 -0
  11. agent_learning/classifiers/scorers/intent.py +25 -0
  12. agent_learning/cli.py +386 -0
  13. agent_learning/config.py +524 -0
  14. agent_learning/learners/__init__.py +6 -0
  15. agent_learning/learners/base.py +38 -0
  16. agent_learning/learners/reinforce.py +153 -0
  17. agent_learning/metrics/__init__.py +22 -0
  18. agent_learning/metrics/base.py +233 -0
  19. agent_learning/metrics/intent_resolution.py +50 -0
  20. agent_learning/metrics/registry.py +42 -0
  21. agent_learning/metrics/task_adherence.py +42 -0
  22. agent_learning/metrics/task_completion.py +54 -0
  23. agent_learning/policy/__init__.py +7 -0
  24. agent_learning/policy/base.py +59 -0
  25. agent_learning/policy/contextual_softmax.py +243 -0
  26. agent_learning/policy/softmax_bandit.py +157 -0
  27. agent_learning/py.typed +1 -0
  28. agent_learning/rewards/__init__.py +6 -0
  29. agent_learning/rewards/shaping.py +121 -0
  30. agent_learning/rewards/writer.py +130 -0
  31. agent_learning/scorers/__init__.py +187 -0
  32. agent_learning/scorers/base.py +49 -0
  33. agent_learning/scorers/llm/__init__.py +19 -0
  34. agent_learning/scorers/llm/_base.py +126 -0
  35. agent_learning/scorers/llm/adherence.py +21 -0
  36. agent_learning/scorers/llm/completion.py +21 -0
  37. agent_learning/scorers/llm/intent.py +21 -0
  38. agent_learning/scorers/nlp/__init__.py +16 -0
  39. agent_learning/scorers/nlp/_base.py +99 -0
  40. agent_learning/scorers/nlp/adherence.py +17 -0
  41. agent_learning/scorers/nlp/completion.py +17 -0
  42. agent_learning/scorers/nlp/intent.py +17 -0
  43. agent_learning/scorers/nlp_text/__init__.py +26 -0
  44. agent_learning/scorers/nlp_text/_base.py +234 -0
  45. agent_learning/scorers/nlp_text/adherence.py +94 -0
  46. agent_learning/scorers/nlp_text/completion.py +91 -0
  47. agent_learning/scorers/nlp_text/intent.py +59 -0
  48. agent_learning/scorers/slm/__init__.py +25 -0
  49. agent_learning/scorers/slm/_base.py +292 -0
  50. agent_learning/scorers/slm/adherence.py +98 -0
  51. agent_learning/scorers/slm/completion.py +111 -0
  52. agent_learning/scorers/slm/intent.py +80 -0
  53. agent_learning/scorers/stdlib/__init__.py +38 -0
  54. agent_learning/scorers/stdlib/_text.py +87 -0
  55. agent_learning/scorers/stdlib/adherence.py +157 -0
  56. agent_learning/scorers/stdlib/completion.py +117 -0
  57. agent_learning/scorers/stdlib/intent.py +182 -0
  58. agent_learning/storage/__init__.py +14 -0
  59. agent_learning/storage/base.py +156 -0
  60. agent_learning/storage/cosmos.py +506 -0
  61. agent_learning/storage/local.py +353 -0
  62. agent_learning/storage/memory.py +209 -0
  63. agent_learning/training/__init__.py +5 -0
  64. agent_learning/training/runner.py +172 -0
  65. agent_learning/types.py +507 -0
  66. agent_learning-0.4.1.dist-info/METADATA +82 -0
  67. agent_learning-0.4.1.dist-info/RECORD +71 -0
  68. agent_learning-0.4.1.dist-info/WHEEL +5 -0
  69. agent_learning-0.4.1.dist-info/entry_points.txt +2 -0
  70. agent_learning-0.4.1.dist-info/licenses/LICENSE +21 -0
  71. agent_learning-0.4.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,524 @@
1
+ """Environment-driven configuration for the agent-learning SDK.
2
+
3
+ All settings can be overridden via environment variables; programmatic
4
+ overrides are also accepted via dataclass field assignment.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import os
10
+ from dataclasses import dataclass, field
11
+ from typing import Any, Literal, Optional
12
+
13
+ ScoreMode = Literal["nlp", "llm"]
14
+ ScoreTier = Literal["stdlib", "nlp", "slm", "llm"]
15
+ CredentialMode = Literal[
16
+ "default",
17
+ "managed-identity",
18
+ "workload-identity",
19
+ "environment",
20
+ "azure-cli",
21
+ "none",
22
+ ]
23
+ _VALID_CREDENTIAL_MODES: frozenset[str] = frozenset(
24
+ {"default", "managed-identity", "workload-identity", "environment", "azure-cli", "none"}
25
+ )
26
+ _VALID_SCORE_TIERS: frozenset[str] = frozenset(
27
+ {"stdlib", "nlp", "slm", "llm"}
28
+ )
29
+
30
+
31
+ def _env_bool(name: str, default: bool) -> bool:
32
+ """Parse a boolean environment variable using common conventions."""
33
+ raw = os.getenv(name)
34
+ if raw is None:
35
+ return default
36
+ return raw.strip().lower() in {"1", "true", "yes", "on"}
37
+
38
+
39
+ def _env_credential_mode() -> Optional[CredentialMode]:
40
+ """Read AGENT_LEARNING_SCORE_CREDENTIAL_MODE and validate."""
41
+ raw = os.getenv("AGENT_LEARNING_SCORE_CREDENTIAL_MODE")
42
+ if raw is None or raw == "":
43
+ return None
44
+ value = raw.strip().lower()
45
+ if value in _VALID_CREDENTIAL_MODES:
46
+ return value # type: ignore[return-value]
47
+ return None
48
+
49
+
50
+ def _env_float(name: str, default: float) -> float:
51
+ raw = os.getenv(name)
52
+ if raw is None or raw == "":
53
+ return default
54
+ try:
55
+ return float(raw)
56
+ except ValueError:
57
+ return default
58
+
59
+
60
+ def _env_int(name: str, default: int) -> int:
61
+ raw = os.getenv(name)
62
+ if raw is None or raw == "":
63
+ return default
64
+ try:
65
+ return int(raw)
66
+ except ValueError:
67
+ return default
68
+
69
+
70
+ # ---------------------------------------------------------------------------
71
+ # Cosmos DB
72
+ # ---------------------------------------------------------------------------
73
+
74
+
75
+ @dataclass
76
+ class CosmosConfig:
77
+ """Cosmos DB connection + container configuration."""
78
+
79
+ endpoint: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_COSMOS_ENDPOINT", ""))
80
+ database_name: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_COSMOS_DATABASE", "dq_rl"))
81
+ auth_mode: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_COSMOS_AUTH_MODE", "aad"))
82
+ account_key: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_COSMOS_KEY", ""))
83
+ partition_key_field: str = field(
84
+ default_factory=lambda: os.getenv("AGENT_LEARNING_PARTITION_KEY_FIELD", "agent_id")
85
+ )
86
+
87
+ container_episodes: str = field(
88
+ default_factory=lambda: os.getenv("AGENT_LEARNING_CONTAINER_EPISODES", "learning_episodes")
89
+ )
90
+ container_rewards: str = field(
91
+ default_factory=lambda: os.getenv("AGENT_LEARNING_CONTAINER_REWARDS", "learning_rewards")
92
+ )
93
+ container_metrics: str = field(
94
+ default_factory=lambda: os.getenv("AGENT_LEARNING_CONTAINER_METRICS", "learning_metrics")
95
+ )
96
+ container_policies: str = field(
97
+ default_factory=lambda: os.getenv("AGENT_LEARNING_CONTAINER_POLICIES", "learning_policies")
98
+ )
99
+ container_runs: str = field(
100
+ default_factory=lambda: os.getenv("AGENT_LEARNING_CONTAINER_RUNS", "learning_runs")
101
+ )
102
+
103
+ @property
104
+ def enabled(self) -> bool:
105
+ return bool(self.endpoint)
106
+
107
+
108
+ # ---------------------------------------------------------------------------
109
+ # Score / evaluator model
110
+ # ---------------------------------------------------------------------------
111
+
112
+
113
+ @dataclass
114
+ class ScoreConfig:
115
+ """Configuration for the LLM scorer used by metric evaluators.
116
+
117
+ Authentication paths, in priority order:
118
+
119
+ 1. ``credential`` set explicitly to a ``TokenCredential`` instance.
120
+ Wins over every other field. The SDK passes a bearer-token
121
+ provider built from it to the underlying evaluator.
122
+ 2. ``credential_mode`` set to one of ``"default"``,
123
+ ``"managed-identity"``, ``"workload-identity"``,
124
+ ``"environment"``, ``"azure-cli"`` (or ``"none"`` to opt out
125
+ even when an api_key is also present). The SDK lazy-imports
126
+ ``azure-identity`` and builds the matching credential.
127
+ ``user_assigned_client_id`` is forwarded when provided.
128
+ 3. ``api_key`` (legacy path). Used only when no credential is
129
+ resolved.
130
+
131
+ Environment variables ``AGENT_LEARNING_SCORE_CREDENTIAL_MODE`` and
132
+ ``AGENT_LEARNING_SCORE_USER_ASSIGNED_CLIENT_ID`` mirror the two
133
+ string fields so the same SDK build switches between developer
134
+ laptop, CI, and production AKS without code changes.
135
+ """
136
+
137
+ azure_endpoint: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_SCORE_ENDPOINT", ""))
138
+ azure_deployment: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_SCORE_DEPLOYMENT", ""))
139
+ api_key: Optional[str] = field(default_factory=lambda: os.getenv("AGENT_LEARNING_SCORE_API_KEY") or None)
140
+ api_version: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_SCORE_API_VERSION", "2024-10-21"))
141
+ # Pass-through threshold for IntentResolutionEvaluator (defaults to its own default)
142
+ threshold: int = field(default_factory=lambda: _env_int("AGENT_LEARNING_INTENT_THRESHOLD", 3))
143
+ # Managed-identity / TokenCredential authentication (optional).
144
+ credential: Optional[Any] = field(default=None, repr=False)
145
+ credential_mode: Optional[CredentialMode] = field(default_factory=_env_credential_mode)
146
+ user_assigned_client_id: Optional[str] = field(
147
+ default_factory=lambda: os.getenv("AGENT_LEARNING_SCORE_USER_ASSIGNED_CLIENT_ID") or None
148
+ )
149
+ # Azure AD token scope for the resolved credential. Defaults to the
150
+ # Cognitive Services scope used by Azure OpenAI and Azure AI Foundry.
151
+ credential_scope: str = field(
152
+ default_factory=lambda: os.getenv(
153
+ "AGENT_LEARNING_SCORE_CREDENTIAL_SCOPE",
154
+ "https://cognitiveservices.azure.com/.default",
155
+ )
156
+ )
157
+
158
+ @property
159
+ def enabled(self) -> bool:
160
+ return bool(self.azure_endpoint and self.azure_deployment)
161
+
162
+ def resolve_credential(self) -> Optional[Any]:
163
+ """Resolve the active TokenCredential, or ``None``.
164
+
165
+ Returns the explicit ``credential`` field when set. Otherwise
166
+ builds a credential from ``credential_mode`` via a lazy import
167
+ of ``azure-identity``. Returns ``None`` when no credential is
168
+ configured (caller falls back to ``api_key``).
169
+ """
170
+ if self.credential is not None:
171
+ return self.credential
172
+ mode = self.credential_mode
173
+ if mode is None or mode == "none":
174
+ return None
175
+ try:
176
+ import azure.identity as az_id # type: ignore[import-not-found]
177
+ except ImportError as exc:
178
+ raise ImportError(
179
+ "Credential-based auth for the LLM scorer requires the optional "
180
+ "'azure-identity' package. Install it with: pip install azure-identity"
181
+ ) from exc
182
+ if mode == "default":
183
+ kwargs: dict = {}
184
+ if self.user_assigned_client_id:
185
+ kwargs["managed_identity_client_id"] = self.user_assigned_client_id
186
+ return az_id.DefaultAzureCredential(**kwargs)
187
+ if mode == "managed-identity":
188
+ if self.user_assigned_client_id:
189
+ return az_id.ManagedIdentityCredential(client_id=self.user_assigned_client_id)
190
+ return az_id.ManagedIdentityCredential()
191
+ if mode == "workload-identity":
192
+ if self.user_assigned_client_id:
193
+ return az_id.WorkloadIdentityCredential(client_id=self.user_assigned_client_id)
194
+ return az_id.WorkloadIdentityCredential()
195
+ if mode == "environment":
196
+ return az_id.EnvironmentCredential()
197
+ if mode == "azure-cli":
198
+ return az_id.AzureCliCredential()
199
+ raise ValueError(f"Unknown credential_mode: {mode!r}")
200
+
201
+ def to_model_config(self) -> dict:
202
+ """Build the dict expected by azure-ai-evaluation evaluators.
203
+
204
+ When a TokenCredential is resolved, the dict carries an
205
+ ``azure_ad_token_provider`` callable (built via
206
+ ``azure.identity.get_bearer_token_provider``) and the
207
+ ``api_key`` field is omitted. Otherwise the legacy ``api_key``
208
+ path is used.
209
+ """
210
+ cfg: dict = {
211
+ "azure_endpoint": self.azure_endpoint,
212
+ "azure_deployment": self.azure_deployment,
213
+ "api_version": self.api_version,
214
+ }
215
+ credential = self.resolve_credential()
216
+ if credential is not None:
217
+ try:
218
+ from azure.identity import ( # type: ignore[import-not-found]
219
+ get_bearer_token_provider,
220
+ )
221
+ except ImportError as exc:
222
+ raise ImportError(
223
+ "Credential-based auth for the LLM scorer requires the optional "
224
+ "'azure-identity' package. Install it with: pip install azure-identity"
225
+ ) from exc
226
+ cfg["azure_ad_token_provider"] = get_bearer_token_provider(
227
+ credential, self.credential_scope
228
+ )
229
+ elif self.api_key:
230
+ cfg["api_key"] = self.api_key
231
+ return cfg
232
+
233
+
234
+ @dataclass
235
+ class NlpScoreConfig:
236
+ """Configuration for the in-SDK NLP scoring stack (Tier 0, pure stdlib).
237
+
238
+ The NLP scorers read their fitted weights from ``snapshot_dir`` at
239
+ load time and fall back to an unfitted (always-pass) policy when no
240
+ snapshot is present. Optional dependencies (scikit-learn,
241
+ sentence-transformers, etc.) are detected at import time. Tier 0
242
+ requires nothing beyond the Python standard library.
243
+ """
244
+
245
+ snapshot_dir: str = field(
246
+ default_factory=lambda: os.getenv(
247
+ "AGENT_LEARNING_NLP_SCORE_DIR", "./data/agent-learning/nlp-scores"
248
+ )
249
+ )
250
+ pass_threshold: float = field(
251
+ default_factory=lambda: _env_float("AGENT_LEARNING_NLP_PASS_THRESHOLD", 0.5)
252
+ )
253
+ enable_semantic: bool = field(
254
+ default_factory=lambda: _env_bool("AGENT_LEARNING_NLP_SEMANTIC", False)
255
+ )
256
+
257
+
258
+ def _env_score_mode(default: ScoreMode = "llm") -> ScoreMode:
259
+ raw = os.getenv("AGENT_LEARNING_SCORE_MODE")
260
+ if raw is None or raw == "":
261
+ return default
262
+ value = raw.strip().lower()
263
+ if value in ("nlp", "llm"):
264
+ return value # type: ignore[return-value]
265
+ return default
266
+
267
+
268
+ def _env_score_tier() -> Optional[ScoreTier]:
269
+ """Read AGENT_LEARNING_SCORE_TIER and validate.
270
+
271
+ Returns one of ``"stdlib"``, ``"nlp"``, ``"slm"``, ``"llm"`` or
272
+ ``None`` when the env var is unset or invalid. The factory in
273
+ scoring factory falls back to ``mode`` when ``tier``
274
+ is None.
275
+ """
276
+ raw = os.getenv("AGENT_LEARNING_SCORE_TIER")
277
+ if raw is None or raw == "":
278
+ return None
279
+ value = raw.strip().lower()
280
+ if value in _VALID_SCORE_TIERS:
281
+ return value # type: ignore[return-value]
282
+ return None
283
+
284
+
285
+ @dataclass
286
+ class StdlibScoreConfig:
287
+ """Configuration for the Tier 1 stdlib scorers.
288
+
289
+ All Tier 1 scorers run on the Python standard library alone. The
290
+ intent scorer optionally loads fitted bag-of-words weights from
291
+ ``snapshot_dir``; the adherence and completion scorers are pure
292
+ rule engines with no persisted state.
293
+ """
294
+
295
+ snapshot_dir: str = field(
296
+ default_factory=lambda: os.getenv(
297
+ "AGENT_LEARNING_STDLIB_SCORE_DIR",
298
+ "./data/agent-learning/stdlib-scores",
299
+ )
300
+ )
301
+ pass_threshold: float = field(
302
+ default_factory=lambda: _env_float(
303
+ "AGENT_LEARNING_STDLIB_PASS_THRESHOLD", 0.5
304
+ )
305
+ )
306
+ feature_dim: int = field(
307
+ default_factory=lambda: _env_int(
308
+ "AGENT_LEARNING_STDLIB_FEATURE_DIM", 1024
309
+ )
310
+ )
311
+
312
+
313
+ @dataclass
314
+ class NlpTextScoreConfig:
315
+ """Configuration for the Tier 2 NLP text scorers.
316
+
317
+ Tier 2 wraps a TF-IDF vectorizer + scikit-learn logistic
318
+ regression around the response text. The fitted vectorizer and
319
+ classifier are persisted to ``{snapshot_dir}/{name}.nlp_text.joblib``
320
+ with a sibling JSON header at ``{name}.nlp_text.json``. When no
321
+ snapshot is present the scorers fall back to a rule-engine signal
322
+ only (adherence, completion) or the pass threshold (intent).
323
+
324
+ Requires the ``[nlp]`` extra (``pip install
325
+ agent-learning[nlp]``).
326
+ """
327
+
328
+ snapshot_dir: str = field(
329
+ default_factory=lambda: os.getenv(
330
+ "AGENT_LEARNING_NLP_TEXT_SCORE_DIR",
331
+ "./data/agent-learning/nlp-text-scores",
332
+ )
333
+ )
334
+ pass_threshold: float = field(
335
+ default_factory=lambda: _env_float(
336
+ "AGENT_LEARNING_NLP_TEXT_PASS_THRESHOLD", 0.5
337
+ )
338
+ )
339
+ max_features: int = field(
340
+ default_factory=lambda: _env_int(
341
+ "AGENT_LEARNING_NLP_TEXT_MAX_FEATURES", 20000
342
+ )
343
+ )
344
+ ngram_min: int = field(
345
+ default_factory=lambda: _env_int("AGENT_LEARNING_NLP_TEXT_NGRAM_MIN", 1)
346
+ )
347
+ ngram_max: int = field(
348
+ default_factory=lambda: _env_int("AGENT_LEARNING_NLP_TEXT_NGRAM_MAX", 2)
349
+ )
350
+
351
+
352
+ @dataclass
353
+ class SlmScoreConfig:
354
+ """Configuration for the Tier 3 small-language-model scorers.
355
+
356
+ Tier 3 wraps a locally-hosted instance of Microsoft
357
+ Phi-4-mini-instruct (3.8 B parameters, 4-bit ONNX) via
358
+ ``onnxruntime-genai``. The model is loaded from ``model_dir`` on
359
+ first use and reused across the three scorers in the same process.
360
+
361
+ Requires the ``[slm]`` extra (``pip install
362
+ agent-learning[slm]``). The default ``model_dir``
363
+ expects a Phi-4-mini-instruct INT4 ONNX bundle laid out under
364
+ ``./models/phi-4-mini-instruct-int4-onnx``; set
365
+ ``AGENT_LEARNING_SLM_MODEL_DIR`` to point at any local path.
366
+ """
367
+
368
+ model_dir: str = field(
369
+ default_factory=lambda: os.getenv(
370
+ "AGENT_LEARNING_SLM_MODEL_DIR",
371
+ "./models/phi-4-mini-instruct-int4-onnx",
372
+ )
373
+ )
374
+ pass_threshold: float = field(
375
+ default_factory=lambda: _env_float(
376
+ "AGENT_LEARNING_SLM_PASS_THRESHOLD", 0.5
377
+ )
378
+ )
379
+ max_new_tokens: int = field(
380
+ default_factory=lambda: _env_int(
381
+ "AGENT_LEARNING_SLM_MAX_NEW_TOKENS", 64
382
+ )
383
+ )
384
+ temperature: float = field(
385
+ default_factory=lambda: _env_float(
386
+ "AGENT_LEARNING_SLM_TEMPERATURE", 0.0
387
+ )
388
+ )
389
+
390
+
391
+ @dataclass
392
+ class ScoreRuntimeConfig:
393
+ """Top-level switch across the four scoring tiers.
394
+
395
+ Two selectors exist, in priority order:
396
+
397
+ 1. ``tier`` (preferred). One of ``"stdlib"`` (Tier 1, zero deps),
398
+ ``"nlp"`` (Tier 2, scikit-learn + rapidfuzz, ``[nlp]`` extra),
399
+ ``"slm"`` (Tier 3, Phi-4-mini-instruct ONNX, ``[slm]`` extra),
400
+ or ``"llm"`` (Tier 4, azure-ai-evaluation, ``[llm]`` extra).
401
+ 2. ``mode`` (legacy). One of ``"nlp"`` (the existing phi+action_id
402
+ binary scoring stack) or ``"llm"`` (Azure AI Evaluation).
403
+ Used when ``tier`` is ``None``.
404
+
405
+ The default ``mode`` stays at ``"llm"`` for backwards compatibility
406
+ with callers already wired through azure-ai-evaluation. Set
407
+ ``AGENT_LEARNING_SCORE_TIER=stdlib`` to opt in to the new Tier 1
408
+ text-based scorers.
409
+ """
410
+
411
+ mode: ScoreMode = field(default_factory=_env_score_mode)
412
+ tier: Optional[ScoreTier] = field(default_factory=_env_score_tier)
413
+ stdlib: StdlibScoreConfig = field(default_factory=StdlibScoreConfig)
414
+ nlp: NlpScoreConfig = field(default_factory=NlpScoreConfig)
415
+ nlp_text: NlpTextScoreConfig = field(default_factory=NlpTextScoreConfig)
416
+ slm: SlmScoreConfig = field(default_factory=SlmScoreConfig)
417
+ llm: ScoreConfig = field(default_factory=ScoreConfig)
418
+
419
+
420
+ # ---------------------------------------------------------------------------
421
+ # Capture / shaping
422
+ # ---------------------------------------------------------------------------
423
+
424
+
425
+ @dataclass
426
+ class CaptureConfig:
427
+ """Configuration for episode capture."""
428
+
429
+ enabled: bool = field(default_factory=lambda: _env_bool("AGENT_LEARNING_ENABLE_CAPTURE", False))
430
+ agent_id: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_AGENT_ID", "default"))
431
+ agent_name: Optional[str] = field(
432
+ default_factory=lambda: os.getenv("AGENT_LEARNING_AGENT_NAME") or None
433
+ )
434
+ task_id: str = field(default_factory=lambda: os.getenv("AGENT_LEARNING_TASK_ID", "default"))
435
+ task_name: Optional[str] = field(
436
+ default_factory=lambda: os.getenv("AGENT_LEARNING_TASK_NAME") or None
437
+ )
438
+ local_fallback_dir: str = field(
439
+ default_factory=lambda: os.getenv("AGENT_LEARNING_DATA_DIR", "./data/agent-learning")
440
+ )
441
+ max_output_length: int = field(default_factory=lambda: _env_int("AGENT_LEARNING_MAX_OUTPUT_LEN", 10000))
442
+ redact_secrets: bool = field(default_factory=lambda: _env_bool("AGENT_LEARNING_REDACT_SECRETS", True))
443
+
444
+
445
+ @dataclass
446
+ class ShapingConfig:
447
+ """Weights and penalties used to combine metric scores into a scalar reward.
448
+
449
+ Each weight is multiplied against the metric's normalized score (in
450
+ [0, 1]) before summing. Weights need not sum to 1 — the shaper will
451
+ return the raw weighted sum, clamped to [-1, 1].
452
+
453
+ The defaults bias the credit-assignment toward task completion (the
454
+ metric the action template directly controls) and away from intent
455
+ resolution (typically computed by an upstream classifier the policy
456
+ cannot influence).
457
+ """
458
+
459
+ intent_resolution_weight: float = field(
460
+ default_factory=lambda: _env_float("AGENT_LEARNING_W_INTENT", 0.10)
461
+ )
462
+ task_adherence_weight: float = field(
463
+ default_factory=lambda: _env_float("AGENT_LEARNING_W_ADHERENCE", 0.20)
464
+ )
465
+ task_completion_weight: float = field(
466
+ default_factory=lambda: _env_float("AGENT_LEARNING_W_COMPLETION", 0.50)
467
+ )
468
+ latency_penalty_threshold_ms: int = field(
469
+ default_factory=lambda: _env_int("AGENT_LEARNING_LATENCY_THRESHOLD_MS", 15000)
470
+ )
471
+ latency_penalty_value: float = field(
472
+ default_factory=lambda: _env_float("AGENT_LEARNING_LATENCY_PENALTY", -0.1)
473
+ )
474
+ # Routing terms. Computed by the calling system and passed to
475
+ # ``RewardShaper.shape(..., routing_correct=..., hallucinated_class=...)``.
476
+ # The values below are the magnitudes used when the relevant condition
477
+ # fires.
478
+ route_correct_reward: float = field(
479
+ default_factory=lambda: _env_float("AGENT_LEARNING_R_ROUTE_CORRECT", 0.20)
480
+ )
481
+ route_wrong_penalty: float = field(
482
+ default_factory=lambda: _env_float("AGENT_LEARNING_P_ROUTE_WRONG", -0.30)
483
+ )
484
+ hallucinated_class_penalty: float = field(
485
+ default_factory=lambda: _env_float("AGENT_LEARNING_P_HALLU", -0.25)
486
+ )
487
+
488
+
489
+ # ---------------------------------------------------------------------------
490
+ # Learner
491
+ # ---------------------------------------------------------------------------
492
+
493
+
494
+ @dataclass
495
+ class LearnerConfig:
496
+ """Hyperparameters for the REINFORCE-with-baseline learner."""
497
+
498
+ learning_rate: float = field(default_factory=lambda: _env_float("AGENT_LEARNING_LR", 0.05))
499
+ baseline_decay: float = field(default_factory=lambda: _env_float("AGENT_LEARNING_BASELINE_DECAY", 0.9))
500
+ entropy_bonus: float = field(default_factory=lambda: _env_float("AGENT_LEARNING_ENTROPY_BONUS", 0.01))
501
+ max_logit_abs: float = field(default_factory=lambda: _env_float("AGENT_LEARNING_MAX_LOGIT", 10.0))
502
+ importance_clip: float = field(
503
+ default_factory=lambda: _env_float("AGENT_LEARNING_IMPORTANCE_CLIP", 5.0)
504
+ )
505
+ min_train_episodes: int = field(
506
+ default_factory=lambda: _env_int("AGENT_LEARNING_MIN_TRAIN_EPISODES", 5)
507
+ )
508
+
509
+
510
+ __all__ = [
511
+ "CaptureConfig",
512
+ "CosmosConfig",
513
+ "CredentialMode",
514
+ "ScoreConfig",
515
+ "ScoreMode",
516
+ "ScoreRuntimeConfig",
517
+ "ScoreTier",
518
+ "LearnerConfig",
519
+ "NlpScoreConfig",
520
+ "NlpTextScoreConfig",
521
+ "ShapingConfig",
522
+ "SlmScoreConfig",
523
+ "StdlibScoreConfig",
524
+ ]
@@ -0,0 +1,6 @@
1
+ """Learning algorithms operating over policies + episode/reward records."""
2
+
3
+ from .base import Learner, LearnerResult
4
+ from .reinforce import ReinforceLearner
5
+
6
+ __all__ = ["Learner", "LearnerResult", "ReinforceLearner"]
@@ -0,0 +1,38 @@
1
+ """Learner interface."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from abc import ABC, abstractmethod
6
+ from dataclasses import dataclass, field
7
+ from typing import Any, Dict, Iterable, List
8
+
9
+ from ..policy.base import Policy
10
+ from ..types import Episode, Reward
11
+
12
+
13
+ @dataclass
14
+ class LearnerResult:
15
+ """Summary returned by :meth:`Learner.update`."""
16
+
17
+ episodes_used: int
18
+ mean_reward: float
19
+ baseline_before: float
20
+ baseline_after: float
21
+ logit_deltas: Dict[str, float] = field(default_factory=dict)
22
+ extra: Dict[str, Any] = field(default_factory=dict)
23
+
24
+
25
+ class Learner(ABC):
26
+ """Abstract base class for an agent-learning algorithm."""
27
+
28
+ @abstractmethod
29
+ def update(
30
+ self,
31
+ policy: Policy,
32
+ episodes: Iterable[Episode],
33
+ rewards: Iterable[Reward],
34
+ ) -> LearnerResult:
35
+ """Apply one update to ``policy`` using the supplied data."""
36
+
37
+
38
+ __all__ = ["Learner", "LearnerResult"]