agent-learning 0.4.1__tar.gz → 0.4.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. {agent_learning-0.4.1/src/agent_learning.egg-info → agent_learning-0.4.2}/PKG-INFO +2 -2
  2. {agent_learning-0.4.1 → agent_learning-0.4.2}/PYPI.md +1 -1
  3. {agent_learning-0.4.1 → agent_learning-0.4.2}/README.md +13 -18
  4. {agent_learning-0.4.1 → agent_learning-0.4.2}/pyproject.toml +1 -1
  5. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/__init__.py +3 -3
  6. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/_version.py +1 -1
  7. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/cli.py +4 -4
  8. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/learners/reinforce.py +7 -6
  9. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/metrics/__init__.py +4 -3
  10. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/metrics/base.py +8 -0
  11. agent_learning-0.4.2/src/agent_learning/metrics/local.py +151 -0
  12. agent_learning-0.4.2/src/agent_learning/metrics/registry.py +59 -0
  13. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/rewards/writer.py +4 -1
  14. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/training/runner.py +21 -3
  15. {agent_learning-0.4.1 → agent_learning-0.4.2/src/agent_learning.egg-info}/PKG-INFO +2 -2
  16. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning.egg-info/SOURCES.txt +2 -0
  17. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_cli.py +90 -0
  18. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_learner.py +27 -0
  19. agent_learning-0.4.2/tests/test_metrics_local.py +60 -0
  20. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_shaping.py +23 -1
  21. agent_learning-0.4.1/src/agent_learning/metrics/registry.py +0 -42
  22. {agent_learning-0.4.1 → agent_learning-0.4.2}/LICENSE +0 -0
  23. {agent_learning-0.4.1 → agent_learning-0.4.2}/setup.cfg +0 -0
  24. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/capture.py +0 -0
  25. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/__init__.py +0 -0
  26. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/base.py +0 -0
  27. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/router.py +0 -0
  28. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/scorers/__init__.py +0 -0
  29. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/scorers/_base.py +0 -0
  30. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/scorers/adherence.py +0 -0
  31. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/scorers/completion.py +0 -0
  32. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/scorers/intent.py +0 -0
  33. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/config.py +0 -0
  34. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/learners/__init__.py +0 -0
  35. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/learners/base.py +0 -0
  36. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/metrics/intent_resolution.py +0 -0
  37. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/metrics/task_adherence.py +0 -0
  38. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/metrics/task_completion.py +0 -0
  39. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/policy/__init__.py +0 -0
  40. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/policy/base.py +0 -0
  41. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/policy/contextual_softmax.py +0 -0
  42. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/policy/softmax_bandit.py +0 -0
  43. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/py.typed +0 -0
  44. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/rewards/__init__.py +0 -0
  45. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/rewards/shaping.py +0 -0
  46. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/__init__.py +0 -0
  47. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/base.py +0 -0
  48. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/llm/__init__.py +0 -0
  49. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/llm/_base.py +0 -0
  50. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/llm/adherence.py +0 -0
  51. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/llm/completion.py +0 -0
  52. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/llm/intent.py +0 -0
  53. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp/__init__.py +0 -0
  54. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp/_base.py +0 -0
  55. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp/adherence.py +0 -0
  56. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp/completion.py +0 -0
  57. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp/intent.py +0 -0
  58. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp_text/__init__.py +0 -0
  59. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp_text/_base.py +0 -0
  60. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp_text/adherence.py +0 -0
  61. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp_text/completion.py +0 -0
  62. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp_text/intent.py +0 -0
  63. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/slm/__init__.py +0 -0
  64. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/slm/_base.py +0 -0
  65. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/slm/adherence.py +0 -0
  66. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/slm/completion.py +0 -0
  67. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/slm/intent.py +0 -0
  68. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/stdlib/__init__.py +0 -0
  69. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/stdlib/_text.py +0 -0
  70. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/stdlib/adherence.py +0 -0
  71. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/stdlib/completion.py +0 -0
  72. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/stdlib/intent.py +0 -0
  73. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/storage/__init__.py +0 -0
  74. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/storage/base.py +0 -0
  75. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/storage/cosmos.py +0 -0
  76. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/storage/local.py +0 -0
  77. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/storage/memory.py +0 -0
  78. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/training/__init__.py +0 -0
  79. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/types.py +0 -0
  80. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning.egg-info/dependency_links.txt +0 -0
  81. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning.egg-info/entry_points.txt +0 -0
  82. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning.egg-info/requires.txt +0 -0
  83. {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning.egg-info/top_level.txt +0 -0
  84. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_capture.py +0 -0
  85. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_contextual_policy.py +0 -0
  86. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_default_store.py +0 -0
  87. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_end_to_end.py +0 -0
  88. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_policy.py +0 -0
  89. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_scorers.py +0 -0
  90. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_scorers_llm.py +0 -0
  91. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_scorers_nlp_text.py +0 -0
  92. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_scorers_slm.py +0 -0
  93. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_scorers_stdlib.py +0 -0
  94. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_storage_local.py +0 -0
  95. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_storage_memory.py +0 -0
  96. {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_types.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agent-learning
3
- Version: 0.4.1
3
+ Version: 0.4.2
4
4
  Summary: Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default.
5
5
  Author: Chris Tava
6
6
  License: MIT License
@@ -75,7 +75,7 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
75
75
 
76
76
  1. The **policy** is a softmax distribution over `N` discrete actions (e.g., "take action A", "take action B", "take action C"). It lives in Python and updates in milliseconds.
77
77
 
78
- 2. Each episode is **evaluated** by three AI Evaluation evaluators — `IntentResolutionEvaluator`, `TaskAdherenceEvaluator`, and `TaskCompletionEvaluator` — whose scores are combined into a single scalar reward.
78
+ 2. Each episode is **evaluated** on-device by three stdlib scorers for intent resolution, task adherence, and task completion. Their scores are combined into a single scalar reward with no scoring endpoint or environment variables required. Azure AI evaluators remain available as an opt-in.
79
79
 
80
80
  3. A **REINFORCE-with-baseline** learner updates the policy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
81
81
 
@@ -8,7 +8,7 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
8
8
 
9
9
  1. The **policy** is a softmax distribution over `N` discrete actions (e.g., "take action A", "take action B", "take action C"). It lives in Python and updates in milliseconds.
10
10
 
11
- 2. Each episode is **evaluated** by three AI Evaluation evaluators — `IntentResolutionEvaluator`, `TaskAdherenceEvaluator`, and `TaskCompletionEvaluator` — whose scores are combined into a single scalar reward.
11
+ 2. Each episode is **evaluated** on-device by three stdlib scorers for intent resolution, task adherence, and task completion. Their scores are combined into a single scalar reward with no scoring endpoint or environment variables required. Azure AI evaluators remain available as an opt-in.
12
12
 
13
13
  3. A **REINFORCE-with-baseline** learner updates the policy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
14
14
 
@@ -1,8 +1,8 @@
1
1
  # agent-learning
2
2
 
3
3
  Native reinforcement learning SDK for AI agents. An in-process
4
- learner optimizes a small, interpretable policy over discrete agent choices (e.g., "take action A", "take action B", "take action C") using AI Evaluation scores as the reward
5
- signal.
4
+ learner optimizes a small, interpretable policy over discrete agent choices (e.g., "take action A", "take action B", "take action C") using on-device evaluation scores as the reward
5
+ signal by default.
6
6
 
7
7
  <p align="center">
8
8
  <img src="images/agent-learning-loop.svg" alt="Animated loop: Policy chooses an action, Score evaluates the episode, and Learner updates the policy" width="960" style="max-width:100%; height:auto;" />
@@ -17,9 +17,10 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
17
17
 
18
18
  <img src="images/0f85e08d0c47cd01.png" alt="Policy selects one of N discrete actions" width="360" style="max-width:100%; height:auto;" />
19
19
 
20
- 2. Each episode is **evaluated** by three AI Evaluation
21
- evaluators — `IntentResolutionEvaluator`, `TaskAdherenceEvaluator`,
22
- and `TaskCompletionEvaluator` — whose scores are combined into a single scalar reward.
20
+ 2. Each episode is **evaluated** locally by three stdlib scorers for intent
21
+ resolution, task adherence, and task completion. Their scores are combined
22
+ into one scalar reward. No scoring endpoint or environment variable is
23
+ required. Configured Azure AI evaluators remain available as an opt-in.
23
24
 
24
25
  <img src="images/246d112f995b785a.png" alt="Three evaluator scores feed a single scalar reward" width="360" style="max-width:100%; height:auto;" />
25
26
 
@@ -54,14 +55,16 @@ agent-learn.exe --help
54
55
  Released versions are published to PyPI:
55
56
  <https://pypi.org/project/agent-learning/>.
56
57
 
58
+ ## Functional Testing
59
+
60
+ A good way to see the SDK in action is to run the
61
+ interactive capture scenario first, followed by the offline batch update:
62
+
57
63
  ```powershell
58
- py -m pip install agent-learning
59
- agent-learn.exe --help
64
+ python tests/functional_cli_interactive.py
65
+ python tests/functional_cli_batch.py
60
66
  ```
61
67
 
62
- `pip` installs `agent-learn.exe` into the active Python environment's
63
- `Scripts` directory.
64
-
65
68
  ## Usage
66
69
 
67
70
  The `agent-learn` CLI provides the current task-learning-loop operations:
@@ -77,11 +80,3 @@ agent-learn score --agent-id <agent_id> [--task-id <task_id>] [--limit <1-500>]
77
80
  agent-learn train --agent-id <agent_id> [--task-id <task_id>] [--limit <1-500>] [--start-date <date>] [--end-date <date>] [--skip-scoring]
78
81
  agent-learn task-policy --agent-id <agent_id> --task-id <task_id>
79
82
  ```
80
-
81
- The subprocess-level functional workflow uses an isolated local store. Run the
82
- interactive capture scenario first, followed by the offline batch update:
83
-
84
- ```powershell
85
- python tests/functional_cli_interactive.py
86
- python tests/functional_cli_batch.py
87
- ```
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "agent-learning"
7
- version = "0.4.1"
7
+ version = "0.4.2"
8
8
  description = "Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default."
9
9
  readme = "PYPI.md"
10
10
  requires-python = ">=3.10"
@@ -7,9 +7,9 @@ native, in-process learner. The SDK is organised into five layers:
7
7
  ``Reward``, ``PolicySnapshot``, ...).
8
8
  - ``agent_learning.storage`` - pluggable persistence (Cosmos DB,
9
9
  local file system, and in-memory).
10
- - ``agent_learning.metrics`` - score-based metrics that wrap the
11
- Azure AI Evaluation evaluators for Intent Resolution, Task
12
- Adherence, and Task Completion.
10
+ - ``agent_learning.metrics`` - on-device stdlib metrics by default,
11
+ with optional Azure AI evaluators for Intent Resolution, Task
12
+ Adherence, and Task Completion.
13
13
  - ``agent_learning.rewards`` - reward shaping + persistence.
14
14
  - ``agent_learning.policy`` - discrete softmax bandit policy.
15
15
  - ``agent_learning.learners`` - REINFORCE-with-baseline learner.
@@ -1,3 +1,3 @@
1
1
  """Package version."""
2
2
 
3
- __version__ = "0.4.1"
3
+ __version__ = "0.4.2"
@@ -216,11 +216,11 @@ def _cmd_score(args: argparse.Namespace) -> int:
216
216
  )
217
217
  scored = 0
218
218
  for episode in episodes:
219
- existing = store.get_rewards_for_episode(episode.id, args.agent_id)
220
- if existing:
219
+ if runner.has_usable_reward(episode):
221
220
  continue
222
- runner.score_and_record(episode)
223
- scored += 1
221
+ rewards = runner.score_and_record(episode)
222
+ if any(reward.source == RewardSource.AGGREGATE for reward in rewards):
223
+ scored += 1
224
224
  print(json.dumps({"episodes_seen": len(episodes), "newly_scored": scored}, indent=2))
225
225
  return 0
226
226
 
@@ -49,14 +49,14 @@ class ReinforceLearner(Learner):
49
49
 
50
50
  # Index rewards by episode id - we only consume aggregate rewards
51
51
  episode_list = list(episodes)
52
- aggregate_rewards: Dict[str, float] = {}
52
+ aggregate_rewards: Dict[str, Reward] = {}
53
53
  for r in rewards:
54
54
  if r.source != RewardSource.AGGREGATE:
55
55
  continue
56
- # Keep the most recent aggregate per episode (sorted by created_at desc)
56
+ # Rescoring may append a replacement aggregate. Keep the newest.
57
57
  current = aggregate_rewards.get(r.episode_id)
58
- if current is None:
59
- aggregate_rewards[r.episode_id] = r.value
58
+ if current is None or r.created_at > current.created_at:
59
+ aggregate_rewards[r.episode_id] = r
60
60
 
61
61
  snapshot = policy.snapshot()
62
62
  baseline_before = snapshot.baseline
@@ -72,9 +72,10 @@ class ReinforceLearner(Learner):
72
72
  for episode in episode_list:
73
73
  if episode.action_id is None or episode.action_id not in action_index:
74
74
  continue
75
- reward_value = aggregate_rewards.get(episode.id)
76
- if reward_value is None:
75
+ reward = aggregate_rewards.get(episode.id)
76
+ if reward is None:
77
77
  continue
78
+ reward_value = reward.value
78
79
  advantage = reward_value - baseline_before
79
80
 
80
81
  # Importance sampling weight when the episode was logged under a
@@ -1,18 +1,19 @@
1
1
  """Score-based evaluation metrics for native RL reward shaping.
2
2
 
3
- Each metric is a thin wrapper around an evaluator from
4
- ``azure-ai-evaluation``. The wrapper normalises the raw evaluator score
5
- into the ``[0, 1]`` range expected by the reward shaper.
3
+ Metrics use on-device stdlib scorers by default. Configured Azure AI
4
+ evaluators remain available for remote LLM scoring.
6
5
  """
7
6
 
8
7
  from .base import MetricEvaluator, MetricRequest
9
8
  from .intent_resolution import IntentResolutionMetric
9
+ from .local import LocalScorerMetric
10
10
  from .task_adherence import TaskAdherenceMetric
11
11
  from .task_completion import TaskCompletionMetric
12
12
  from .registry import default_metrics, evaluate_all
13
13
 
14
14
  __all__ = [
15
15
  "IntentResolutionMetric",
16
+ "LocalScorerMetric",
16
17
  "MetricEvaluator",
17
18
  "MetricRequest",
18
19
  "TaskAdherenceMetric",
@@ -42,6 +42,14 @@ class MetricRequest:
42
42
  system_message=episode.system_message,
43
43
  tool_calls=_format_tool_calls(episode),
44
44
  tool_definitions=episode.metadata.get("tool_definitions"),
45
+ extra={
46
+ "action_id": episode.action_id,
47
+ "context_features": episode.context_features,
48
+ "execution_status": episode.execution_status,
49
+ "expected_outcome": episode.expected_outcome,
50
+ "metadata": episode.metadata,
51
+ "result_summary": episode.result_summary,
52
+ },
45
53
  )
46
54
 
47
55
 
@@ -0,0 +1,151 @@
1
+ """Metric adapters for on-device scoring backends."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from collections.abc import Sequence
7
+ from typing import Any
8
+
9
+ from ..scorers.base import Scorer, ScoreResult
10
+ from ..scorers.stdlib._text import tokenize
11
+ from ..types import MetricName, MetricResult
12
+ from .base import MetricEvaluator, MetricRequest
13
+
14
+ logger = logging.getLogger(__name__)
15
+
16
+
17
+ class LocalScorerMetric(MetricEvaluator):
18
+ """Project a local :class:`Scorer` onto the metric interface."""
19
+
20
+ NAME = MetricName.INTENT_RESOLUTION
21
+
22
+ def __init__(self, metric: MetricName, scorer: Scorer) -> None:
23
+ super().__init__(evaluator=scorer)
24
+ self.NAME = metric
25
+ self._scorer = scorer
26
+
27
+ def _build_evaluator(self) -> Any: # pragma: no cover - injected in __init__
28
+ return self._scorer
29
+
30
+ def _build_kwargs(self, request: MetricRequest) -> dict[str, Any]:
31
+ metadata = request.extra.get("metadata")
32
+ if not isinstance(metadata, dict):
33
+ metadata = {}
34
+ context_features = request.extra.get("context_features")
35
+ if not isinstance(context_features, dict):
36
+ context_features = {}
37
+ contract = metadata.get("score_contract") or metadata.get("adherence_contract")
38
+ if not isinstance(contract, dict):
39
+ contract = {}
40
+ expected_tokens = metadata.get("expected_tokens")
41
+ if not isinstance(expected_tokens, Sequence) or isinstance(
42
+ expected_tokens, (str, bytes)
43
+ ):
44
+ expected_tokens = tokenize(str(request.extra.get("expected_outcome") or ""))
45
+ return {
46
+ "query": request.query,
47
+ "response": request.response,
48
+ "system_message": request.system_message,
49
+ "tool_calls": request.tool_calls,
50
+ "action_id": request.extra.get("action_id"),
51
+ "phi": context_features.get("phi"),
52
+ "contract": contract,
53
+ "expected_tokens": expected_tokens,
54
+ }
55
+
56
+ def _normalize(self, raw: dict[str, Any]) -> float | None:
57
+ value = raw.get("normalized")
58
+ return float(value) if value is not None else None
59
+
60
+ def evaluate(self, request: MetricRequest) -> MetricResult:
61
+ if not (request.query and request.response):
62
+ return MetricResult(
63
+ metric=self.NAME,
64
+ score=None,
65
+ normalized=None,
66
+ status="skipped",
67
+ reason="query or response is empty",
68
+ evaluator=f"local:{self._scorer.name}",
69
+ )
70
+
71
+ authoritative = self._authoritative_completion(request)
72
+ if authoritative is not None:
73
+ value, reason = authoritative
74
+ return MetricResult(
75
+ metric=self.NAME,
76
+ score=value,
77
+ normalized=value,
78
+ status="completed",
79
+ reason=reason,
80
+ evaluator="local:episode-outcome",
81
+ )
82
+
83
+ try:
84
+ result = self._scorer.score(**self._build_kwargs(request))
85
+ except Exception as exc: # noqa: BLE001 # pragma: no cover
86
+ logger.warning("Local metric %s failed: %s", self.NAME.value, exc)
87
+ return MetricResult(
88
+ metric=self.NAME,
89
+ score=None,
90
+ normalized=None,
91
+ status="skipped",
92
+ reason=f"local scorer error: {exc}",
93
+ evaluator=f"local:{self._scorer.name}",
94
+ )
95
+ return _to_metric_result(self.NAME, self._scorer.name, result)
96
+
97
+ def _authoritative_completion(
98
+ self, request: MetricRequest
99
+ ) -> tuple[float, str] | None:
100
+ if self.NAME != MetricName.TASK_COMPLETION:
101
+ return None
102
+ metadata = request.extra.get("metadata")
103
+ if not isinstance(metadata, dict):
104
+ metadata = {}
105
+ action_id = request.extra.get("action_id")
106
+ correct_action_id = metadata.get("correct_action_id")
107
+ if action_id and correct_action_id:
108
+ return (
109
+ float(action_id == correct_action_id),
110
+ "derived from action_id and metadata.correct_action_id",
111
+ )
112
+ completed = metadata.get("task_completed")
113
+ if isinstance(completed, bool):
114
+ return float(completed), "derived from metadata.task_completed"
115
+ status = str(request.extra.get("execution_status") or "").lower()
116
+ if status == "completed":
117
+ return 1.0, "derived from execution_status=completed"
118
+ if status == "failed":
119
+ return 0.0, "derived from execution_status=failed"
120
+ if status == "partial":
121
+ return 0.5, "derived from execution_status=partial"
122
+ return None
123
+
124
+
125
+ def _to_metric_result(
126
+ metric: MetricName, scorer_name: str, result: ScoreResult
127
+ ) -> MetricResult:
128
+ normalized = max(0.0, min(1.0, float(result.normalized)))
129
+ return MetricResult(
130
+ metric=metric,
131
+ score=normalized,
132
+ normalized=normalized,
133
+ status="completed",
134
+ reason=f"local {result.label}",
135
+ properties=dict(result.features),
136
+ evaluator=f"local:{scorer_name}",
137
+ metadata={"label": result.label, "confidence": result.confidence},
138
+ )
139
+
140
+
141
+ def local_metrics(scorers: tuple[Scorer, Scorer, Scorer]) -> list[MetricEvaluator]:
142
+ """Return local metric adapters in reward-shaping order."""
143
+ intent, adherence, completion = scorers
144
+ return [
145
+ LocalScorerMetric(MetricName.INTENT_RESOLUTION, intent),
146
+ LocalScorerMetric(MetricName.TASK_ADHERENCE, adherence),
147
+ LocalScorerMetric(MetricName.TASK_COMPLETION, completion),
148
+ ]
149
+
150
+
151
+ __all__ = ["LocalScorerMetric", "local_metrics"]
@@ -0,0 +1,59 @@
1
+ """Convenience helpers to evaluate an episode across all default metrics."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Iterable, List, Optional
6
+
7
+ from ..config import ScoreConfig, ScoreRuntimeConfig
8
+ from ..scorers import build_scorers
9
+ from ..types import Episode, MetricResult
10
+ from .base import MetricEvaluator, MetricRequest
11
+ from .intent_resolution import IntentResolutionMetric
12
+ from .local import local_metrics
13
+ from .task_adherence import TaskAdherenceMetric
14
+ from .task_completion import TaskCompletionMetric
15
+
16
+
17
+ def default_metrics(
18
+ score_config: Optional[ScoreConfig] = None,
19
+ score_runtime_config: Optional[ScoreRuntimeConfig] = None,
20
+ ) -> List[MetricEvaluator]:
21
+ """Return local metrics by default, or Azure metrics when configured."""
22
+ runtime = score_runtime_config or ScoreRuntimeConfig()
23
+ llm = score_config or runtime.llm
24
+ if score_config is not None or runtime.tier == "llm" or (
25
+ runtime.tier is None and llm.enabled
26
+ ):
27
+ return [
28
+ IntentResolutionMetric(llm),
29
+ TaskAdherenceMetric(llm),
30
+ TaskCompletionMetric(llm),
31
+ ]
32
+ if runtime.tier is None:
33
+ runtime.tier = "stdlib"
34
+ return local_metrics(build_scorers(runtime))
35
+
36
+
37
+ def evaluate_all(
38
+ episode: Episode,
39
+ metrics: Optional[Iterable[MetricEvaluator]] = None,
40
+ *,
41
+ score_config: Optional[ScoreConfig] = None,
42
+ score_runtime_config: Optional[ScoreRuntimeConfig] = None,
43
+ ) -> List[MetricResult]:
44
+ """Evaluate one episode against every supplied metric.
45
+
46
+ Each metric is given its own try/except inside :meth:`evaluate`,
47
+ so a single failing scorer will not prevent the others from
48
+ producing scores.
49
+ """
50
+ metric_list = (
51
+ list(metrics)
52
+ if metrics is not None
53
+ else default_metrics(score_config, score_runtime_config)
54
+ )
55
+ request = MetricRequest.from_episode(episode)
56
+ return [m.evaluate(request) for m in metric_list]
57
+
58
+
59
+ __all__ = ["default_metrics", "evaluate_all"]
@@ -98,7 +98,10 @@ class RewardWriter:
98
98
  except Exception as exc: # pragma: no cover
99
99
  logger.warning("Failed to persist %s penalty for %s: %s", kind, episode.id, exc)
100
100
 
101
- # 4) Aggregate reward consumed by the learner
101
+ # 4) Aggregate reward consumed by the learner. Do not turn an
102
+ # all-skipped evaluation into a misleading neutral reward.
103
+ if not shaped.metric_contributions and not shaped.penalties:
104
+ return stored
102
105
  aggregate = Reward(
103
106
  episode_id=episode.id,
104
107
  agent_id=episode.agent_id,
@@ -19,7 +19,7 @@ import logging
19
19
  from datetime import datetime, timezone
20
20
  from typing import Dict, Iterable, List, Optional
21
21
 
22
- from ..config import LearnerConfig, ScoreConfig, ShapingConfig
22
+ from ..config import LearnerConfig, ScoreConfig, ScoreRuntimeConfig, ShapingConfig
23
23
  from ..learners.base import Learner, LearnerResult
24
24
  from ..learners.reinforce import ReinforceLearner
25
25
  from ..metrics.base import MetricEvaluator
@@ -47,12 +47,17 @@ class LearningRunner:
47
47
  writer: Optional[RewardWriter] = None,
48
48
  learner: Optional[Learner] = None,
49
49
  score_config: Optional[ScoreConfig] = None,
50
+ score_runtime_config: Optional[ScoreRuntimeConfig] = None,
50
51
  learner_config: Optional[LearnerConfig] = None,
51
52
  shaping_config: Optional[ShapingConfig] = None,
52
53
  ) -> None:
53
54
  self._store = store or get_default_store()
54
55
  self._policy = policy
55
- self._metrics = list(metrics) if metrics is not None else default_metrics(score_config)
56
+ self._metrics = (
57
+ list(metrics)
58
+ if metrics is not None
59
+ else default_metrics(score_config, score_runtime_config)
60
+ )
56
61
  self._shaper = shaper or RewardShaper(shaping_config)
57
62
  self._writer = writer or RewardWriter(self._store)
58
63
  self._learner = learner or ReinforceLearner(learner_config)
@@ -137,6 +142,19 @@ class LearningRunner:
137
142
  # Helpers
138
143
  # ------------------------------------------------------------------
139
144
 
145
+ def has_usable_reward(self, episode: Episode) -> bool:
146
+ """Return whether an episode has an aggregate backed by valid metrics."""
147
+ rewards = self._store.get_rewards_for_episode(episode.id, episode.agent_id)
148
+ if not any(reward.source.value == "aggregate" for reward in rewards):
149
+ return False
150
+ metrics = self._store.get_metric_results(episode.id, episode.agent_id)
151
+ if not metrics:
152
+ return True
153
+ return any(
154
+ result.status == "completed" and result.normalized is not None
155
+ for result in metrics
156
+ )
157
+
140
158
  def _collect_rewards(
141
159
  self,
142
160
  agent_id: str,
@@ -147,7 +165,7 @@ class LearningRunner:
147
165
  rewards: List[Reward] = []
148
166
  for episode in episodes:
149
167
  existing = self._store.get_rewards_for_episode(episode.id, agent_id)
150
- if existing:
168
+ if self.has_usable_reward(episode):
151
169
  rewards.extend(existing)
152
170
  continue
153
171
  if not score_missing:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agent-learning
3
- Version: 0.4.1
3
+ Version: 0.4.2
4
4
  Summary: Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default.
5
5
  Author: Chris Tava
6
6
  License: MIT License
@@ -75,7 +75,7 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
75
75
 
76
76
  1. The **policy** is a softmax distribution over `N` discrete actions (e.g., "take action A", "take action B", "take action C"). It lives in Python and updates in milliseconds.
77
77
 
78
- 2. Each episode is **evaluated** by three AI Evaluation evaluators — `IntentResolutionEvaluator`, `TaskAdherenceEvaluator`, and `TaskCompletionEvaluator` — whose scores are combined into a single scalar reward.
78
+ 2. Each episode is **evaluated** on-device by three stdlib scorers for intent resolution, task adherence, and task completion. Their scores are combined into a single scalar reward with no scoring endpoint or environment variables required. Azure AI evaluators remain available as an opt-in.
79
79
 
80
80
  3. A **REINFORCE-with-baseline** learner updates the policy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
81
81
 
@@ -29,6 +29,7 @@ src/agent_learning/learners/reinforce.py
29
29
  src/agent_learning/metrics/__init__.py
30
30
  src/agent_learning/metrics/base.py
31
31
  src/agent_learning/metrics/intent_resolution.py
32
+ src/agent_learning/metrics/local.py
32
33
  src/agent_learning/metrics/registry.py
33
34
  src/agent_learning/metrics/task_adherence.py
34
35
  src/agent_learning/metrics/task_completion.py
@@ -79,6 +80,7 @@ tests/test_contextual_policy.py
79
80
  tests/test_default_store.py
80
81
  tests/test_end_to_end.py
81
82
  tests/test_learner.py
83
+ tests/test_metrics_local.py
82
84
  tests/test_policy.py
83
85
  tests/test_scorers.py
84
86
  tests/test_scorers_llm.py
@@ -167,6 +167,96 @@ def test_task_episode_register_persists_full_episode(
167
167
  assert episode.is_full
168
168
 
169
169
 
170
+ def test_score_uses_local_stdlib_without_configuration(
171
+ monkeypatch, capsys
172
+ ) -> None:
173
+ for name in (
174
+ "AGENT_LEARNING_SCORE_ENDPOINT",
175
+ "AGENT_LEARNING_SCORE_DEPLOYMENT",
176
+ "AGENT_LEARNING_SCORE_TIER",
177
+ ):
178
+ monkeypatch.delenv(name, raising=False)
179
+ store = InMemoryStore()
180
+ monkeypatch.setattr(cli, "get_default_store", lambda: store)
181
+ episode = Episode(
182
+ agent_id="scout",
183
+ task_id="context-window",
184
+ user_input="What is the context window?",
185
+ assistant_output="The context window is 922,000 tokens.",
186
+ intent_summary="Report the context window",
187
+ action_id="inspect_context",
188
+ expected_outcome="Return the live context limit",
189
+ execution_status="completed",
190
+ result_summary="Returned the live limit",
191
+ metadata={
192
+ "correct_action_id": "inspect_context",
193
+ "task_completed": True,
194
+ },
195
+ )
196
+ store.store_episode(episode)
197
+
198
+ assert cli.main(["score", "--agent-id", "scout"]) == 0
199
+ result = json.loads(capsys.readouterr().out)
200
+ assert result == {"episodes_seen": 1, "newly_scored": 1}
201
+ metrics = store.get_metric_results(episode.id, episode.agent_id)
202
+ assert len(metrics) == 3
203
+ assert all(metric.status == "completed" for metric in metrics)
204
+ assert all((metric.evaluator or "").startswith("local:") for metric in metrics)
205
+ rewards = store.get_rewards_for_episode(episode.id, episode.agent_id)
206
+ aggregate = next(reward for reward in rewards if reward.source == RewardSource.AGGREGATE)
207
+ assert aggregate.value > 0.0
208
+
209
+
210
+ def test_score_replaces_skipped_only_evaluation(monkeypatch, capsys) -> None:
211
+ store = InMemoryStore()
212
+ monkeypatch.setattr(cli, "get_default_store", lambda: store)
213
+ episode = Episode(
214
+ agent_id="scout",
215
+ task_id="context-window",
216
+ user_input="What is the context window?",
217
+ assistant_output="The context window is 922,000 tokens.",
218
+ action_id="inspect_context",
219
+ execution_status="completed",
220
+ metadata={"task_completed": True},
221
+ )
222
+ store.store_episode(episode)
223
+ store.store_metric_results(
224
+ episode.id,
225
+ episode.agent_id,
226
+ [
227
+ MetricResult(
228
+ metric=metric,
229
+ score=None,
230
+ normalized=None,
231
+ status="skipped",
232
+ reason="remote scorer was not configured",
233
+ )
234
+ for metric in MetricName
235
+ ],
236
+ )
237
+ store.store_reward(
238
+ Reward(
239
+ episode_id=episode.id,
240
+ agent_id=episode.agent_id,
241
+ source=RewardSource.AGGREGATE,
242
+ value=0.0,
243
+ created_at="2026-08-09T00:00:00+00:00",
244
+ )
245
+ )
246
+
247
+ assert cli.main(["score", "--agent-id", "scout"]) == 0
248
+ assert json.loads(capsys.readouterr().out)["newly_scored"] == 1
249
+ metrics = store.get_metric_results(episode.id, episode.agent_id)
250
+ assert sum(metric.status == "completed" for metric in metrics) == 3
251
+ aggregates = [
252
+ reward
253
+ for reward in store.get_rewards_for_episode(episode.id, episode.agent_id)
254
+ if reward.source == RewardSource.AGGREGATE
255
+ ]
256
+ assert len(aggregates) == 2
257
+ assert max(aggregates, key=lambda reward: reward.created_at).value > 0.0
258
+
259
+
170
260
  def test_agent_training_uses_one_limit_and_preserves_task_policy_history(
171
261
  monkeypatch, capsys
172
262
  ) -> None:
@@ -89,3 +89,30 @@ def test_unknown_action_is_skipped() -> None:
89
89
  ]
90
90
  result = learner.update(policy, [ep], rewards)
91
91
  assert result.episodes_used == 0
92
+
93
+
94
+ def test_most_recent_aggregate_reward_wins() -> None:
95
+ policy = SoftmaxPolicy.from_actions([Action(id="a"), Action(id="b")])
96
+ learner = ReinforceLearner(LearnerConfig(learning_rate=0.5, entropy_bonus=0.0))
97
+ episode = _make_episode("default", "a")
98
+ rewards = [
99
+ Reward(
100
+ episode_id=episode.id,
101
+ agent_id=episode.agent_id,
102
+ source=RewardSource.AGGREGATE,
103
+ value=0.0,
104
+ created_at="2026-08-09T00:00:00+00:00",
105
+ ),
106
+ Reward(
107
+ episode_id=episode.id,
108
+ agent_id=episode.agent_id,
109
+ source=RewardSource.AGGREGATE,
110
+ value=0.8,
111
+ created_at="2026-08-09T00:00:01+00:00",
112
+ ),
113
+ ]
114
+
115
+ result = learner.update(policy, [episode], rewards)
116
+
117
+ assert result.mean_reward == 0.8
118
+ assert policy.snapshot().logits["a"] > 0.0
@@ -0,0 +1,60 @@
1
+ """Tests for default on-device metric routing."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from agent_learning.config import ScoreConfig
6
+ from agent_learning.metrics import (
7
+ IntentResolutionMetric,
8
+ LocalScorerMetric,
9
+ default_metrics,
10
+ evaluate_all,
11
+ )
12
+ from agent_learning.types import Episode, MetricName
13
+
14
+
15
+ def test_default_metrics_are_local_without_azure_configuration(monkeypatch) -> None:
16
+ for name in (
17
+ "AGENT_LEARNING_SCORE_ENDPOINT",
18
+ "AGENT_LEARNING_SCORE_DEPLOYMENT",
19
+ "AGENT_LEARNING_SCORE_TIER",
20
+ ):
21
+ monkeypatch.delenv(name, raising=False)
22
+
23
+ metrics = default_metrics()
24
+
25
+ assert len(metrics) == 3
26
+ assert all(isinstance(metric, LocalScorerMetric) for metric in metrics)
27
+
28
+
29
+ def test_explicit_score_config_preserves_azure_metrics() -> None:
30
+ metrics = default_metrics(
31
+ ScoreConfig(
32
+ azure_endpoint="https://example.openai.azure.com",
33
+ azure_deployment="grader",
34
+ credential_mode="none",
35
+ )
36
+ )
37
+
38
+ assert isinstance(metrics[0], IntentResolutionMetric)
39
+
40
+
41
+ def test_correct_action_id_overrides_contradictory_completion_flag(
42
+ monkeypatch,
43
+ ) -> None:
44
+ monkeypatch.delenv("AGENT_LEARNING_SCORE_ENDPOINT", raising=False)
45
+ monkeypatch.delenv("AGENT_LEARNING_SCORE_DEPLOYMENT", raising=False)
46
+ episode = Episode(
47
+ user_input="Answer the task",
48
+ assistant_output="Used the wrong action",
49
+ action_id="wrong",
50
+ execution_status="completed",
51
+ metadata={"correct_action_id": "right", "task_completed": True},
52
+ )
53
+
54
+ results = evaluate_all(episode)
55
+
56
+ completion = next(
57
+ result for result in results if result.metric == MetricName.TASK_COMPLETION
58
+ )
59
+ assert completion.normalized == 0.0
60
+ assert completion.reason == "derived from action_id and metadata.correct_action_id"
@@ -3,8 +3,10 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  from agent_learning.config import ShapingConfig
6
+ from agent_learning.rewards import RewardWriter
6
7
  from agent_learning.rewards.shaping import RewardShaper
7
- from agent_learning.types import MetricName, MetricResult
8
+ from agent_learning.storage import InMemoryStore
9
+ from agent_learning.types import Episode, MetricName, MetricResult, RewardSource
8
10
 
9
11
 
10
12
  def _result(metric: MetricName, normalized: float) -> MetricResult:
@@ -71,6 +73,26 @@ def test_skipped_metrics_are_ignored() -> None:
71
73
  assert abs(shaped.value - 0.7) < 1e-9
72
74
 
73
75
 
76
+ def test_all_skipped_metrics_do_not_persist_aggregate() -> None:
77
+ store = InMemoryStore()
78
+ episode = Episode(agent_id="scout")
79
+ results = [
80
+ MetricResult(
81
+ metric=metric,
82
+ score=None,
83
+ normalized=None,
84
+ status="skipped",
85
+ )
86
+ for metric in MetricName
87
+ ]
88
+ shaped = RewardShaper().shape(results)
89
+
90
+ rewards = RewardWriter(store).write(episode, results, shaped)
91
+
92
+ assert not any(reward.source == RewardSource.AGGREGATE for reward in rewards)
93
+ assert store.get_rewards_for_episode(episode.id, episode.agent_id) == []
94
+
95
+
74
96
  def test_latency_penalty_applied() -> None:
75
97
  cfg = ShapingConfig(
76
98
  intent_resolution_weight=0.0,
@@ -1,42 +0,0 @@
1
- """Convenience helpers to evaluate an episode across all default metrics."""
2
-
3
- from __future__ import annotations
4
-
5
- from typing import Iterable, List, Optional
6
-
7
- from ..config import ScoreConfig
8
- from ..types import Episode, MetricResult
9
- from .base import MetricEvaluator, MetricRequest
10
- from .intent_resolution import IntentResolutionMetric
11
- from .task_adherence import TaskAdherenceMetric
12
- from .task_completion import TaskCompletionMetric
13
-
14
-
15
- def default_metrics(score_config: Optional[ScoreConfig] = None) -> List[MetricEvaluator]:
16
- """Return the three native metrics wired with the same score config."""
17
- cfg = score_config or ScoreConfig()
18
- return [
19
- IntentResolutionMetric(cfg),
20
- TaskAdherenceMetric(cfg),
21
- TaskCompletionMetric(cfg),
22
- ]
23
-
24
-
25
- def evaluate_all(
26
- episode: Episode,
27
- metrics: Optional[Iterable[MetricEvaluator]] = None,
28
- *,
29
- score_config: Optional[ScoreConfig] = None,
30
- ) -> List[MetricResult]:
31
- """Evaluate one episode against every supplied metric.
32
-
33
- Each metric is given its own try/except inside :meth:`evaluate`,
34
- so a single failing scorer will not prevent the others from
35
- producing scores.
36
- """
37
- metric_list = list(metrics) if metrics is not None else default_metrics(score_config)
38
- request = MetricRequest.from_episode(episode)
39
- return [m.evaluate(request) for m in metric_list]
40
-
41
-
42
- __all__ = ["default_metrics", "evaluate_all"]
File without changes
File without changes