agent-learning 0.4.1__tar.gz → 0.4.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agent_learning-0.4.1/src/agent_learning.egg-info → agent_learning-0.4.2}/PKG-INFO +2 -2
- {agent_learning-0.4.1 → agent_learning-0.4.2}/PYPI.md +1 -1
- {agent_learning-0.4.1 → agent_learning-0.4.2}/README.md +13 -18
- {agent_learning-0.4.1 → agent_learning-0.4.2}/pyproject.toml +1 -1
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/__init__.py +3 -3
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/_version.py +1 -1
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/cli.py +4 -4
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/learners/reinforce.py +7 -6
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/metrics/__init__.py +4 -3
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/metrics/base.py +8 -0
- agent_learning-0.4.2/src/agent_learning/metrics/local.py +151 -0
- agent_learning-0.4.2/src/agent_learning/metrics/registry.py +59 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/rewards/writer.py +4 -1
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/training/runner.py +21 -3
- {agent_learning-0.4.1 → agent_learning-0.4.2/src/agent_learning.egg-info}/PKG-INFO +2 -2
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning.egg-info/SOURCES.txt +2 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_cli.py +90 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_learner.py +27 -0
- agent_learning-0.4.2/tests/test_metrics_local.py +60 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_shaping.py +23 -1
- agent_learning-0.4.1/src/agent_learning/metrics/registry.py +0 -42
- {agent_learning-0.4.1 → agent_learning-0.4.2}/LICENSE +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/setup.cfg +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/capture.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/router.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/scorers/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/scorers/_base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/scorers/adherence.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/scorers/completion.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/scorers/intent.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/config.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/learners/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/learners/base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/metrics/intent_resolution.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/metrics/task_adherence.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/metrics/task_completion.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/policy/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/policy/base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/policy/contextual_softmax.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/policy/softmax_bandit.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/py.typed +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/rewards/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/rewards/shaping.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/llm/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/llm/_base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/llm/adherence.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/llm/completion.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/llm/intent.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp/_base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp/adherence.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp/completion.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp/intent.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp_text/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp_text/_base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp_text/adherence.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp_text/completion.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp_text/intent.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/slm/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/slm/_base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/slm/adherence.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/slm/completion.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/slm/intent.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/stdlib/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/stdlib/_text.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/stdlib/adherence.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/stdlib/completion.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/stdlib/intent.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/storage/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/storage/base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/storage/cosmos.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/storage/local.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/storage/memory.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/training/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/types.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning.egg-info/dependency_links.txt +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning.egg-info/entry_points.txt +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning.egg-info/requires.txt +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning.egg-info/top_level.txt +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_capture.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_contextual_policy.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_default_store.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_end_to_end.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_policy.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_scorers.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_scorers_llm.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_scorers_nlp_text.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_scorers_slm.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_scorers_stdlib.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_storage_local.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_storage_memory.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.2}/tests/test_types.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: agent-learning
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.2
|
|
4
4
|
Summary: Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default.
|
|
5
5
|
Author: Chris Tava
|
|
6
6
|
License: MIT License
|
|
@@ -75,7 +75,7 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
|
|
|
75
75
|
|
|
76
76
|
1. The **policy** is a softmax distribution over `N` discrete actions (e.g., "take action A", "take action B", "take action C"). It lives in Python and updates in milliseconds.
|
|
77
77
|
|
|
78
|
-
2. Each episode is **evaluated** by three
|
|
78
|
+
2. Each episode is **evaluated** on-device by three stdlib scorers for intent resolution, task adherence, and task completion. Their scores are combined into a single scalar reward with no scoring endpoint or environment variables required. Azure AI evaluators remain available as an opt-in.
|
|
79
79
|
|
|
80
80
|
3. A **REINFORCE-with-baseline** learner updates the policy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
|
|
81
81
|
|
|
@@ -8,7 +8,7 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
|
|
|
8
8
|
|
|
9
9
|
1. The **policy** is a softmax distribution over `N` discrete actions (e.g., "take action A", "take action B", "take action C"). It lives in Python and updates in milliseconds.
|
|
10
10
|
|
|
11
|
-
2. Each episode is **evaluated** by three
|
|
11
|
+
2. Each episode is **evaluated** on-device by three stdlib scorers for intent resolution, task adherence, and task completion. Their scores are combined into a single scalar reward with no scoring endpoint or environment variables required. Azure AI evaluators remain available as an opt-in.
|
|
12
12
|
|
|
13
13
|
3. A **REINFORCE-with-baseline** learner updates the policy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
|
|
14
14
|
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
# agent-learning
|
|
2
2
|
|
|
3
3
|
Native reinforcement learning SDK for AI agents. An in-process
|
|
4
|
-
learner optimizes a small, interpretable policy over discrete agent choices (e.g., "take action A", "take action B", "take action C") using
|
|
5
|
-
signal.
|
|
4
|
+
learner optimizes a small, interpretable policy over discrete agent choices (e.g., "take action A", "take action B", "take action C") using on-device evaluation scores as the reward
|
|
5
|
+
signal by default.
|
|
6
6
|
|
|
7
7
|
<p align="center">
|
|
8
8
|
<img src="images/agent-learning-loop.svg" alt="Animated loop: Policy chooses an action, Score evaluates the episode, and Learner updates the policy" width="960" style="max-width:100%; height:auto;" />
|
|
@@ -17,9 +17,10 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
|
|
|
17
17
|
|
|
18
18
|
<img src="images/0f85e08d0c47cd01.png" alt="Policy selects one of N discrete actions" width="360" style="max-width:100%; height:auto;" />
|
|
19
19
|
|
|
20
|
-
2. Each episode is **evaluated** by three
|
|
21
|
-
|
|
22
|
-
|
|
20
|
+
2. Each episode is **evaluated** locally by three stdlib scorers for intent
|
|
21
|
+
resolution, task adherence, and task completion. Their scores are combined
|
|
22
|
+
into one scalar reward. No scoring endpoint or environment variable is
|
|
23
|
+
required. Configured Azure AI evaluators remain available as an opt-in.
|
|
23
24
|
|
|
24
25
|
<img src="images/246d112f995b785a.png" alt="Three evaluator scores feed a single scalar reward" width="360" style="max-width:100%; height:auto;" />
|
|
25
26
|
|
|
@@ -54,14 +55,16 @@ agent-learn.exe --help
|
|
|
54
55
|
Released versions are published to PyPI:
|
|
55
56
|
<https://pypi.org/project/agent-learning/>.
|
|
56
57
|
|
|
58
|
+
## Functional Testing
|
|
59
|
+
|
|
60
|
+
A good way to see the SDK in action is to run the
|
|
61
|
+
interactive capture scenario first, followed by the offline batch update:
|
|
62
|
+
|
|
57
63
|
```powershell
|
|
58
|
-
py
|
|
59
|
-
|
|
64
|
+
python tests/functional_cli_interactive.py
|
|
65
|
+
python tests/functional_cli_batch.py
|
|
60
66
|
```
|
|
61
67
|
|
|
62
|
-
`pip` installs `agent-learn.exe` into the active Python environment's
|
|
63
|
-
`Scripts` directory.
|
|
64
|
-
|
|
65
68
|
## Usage
|
|
66
69
|
|
|
67
70
|
The `agent-learn` CLI provides the current task-learning-loop operations:
|
|
@@ -77,11 +80,3 @@ agent-learn score --agent-id <agent_id> [--task-id <task_id>] [--limit <1-500>]
|
|
|
77
80
|
agent-learn train --agent-id <agent_id> [--task-id <task_id>] [--limit <1-500>] [--start-date <date>] [--end-date <date>] [--skip-scoring]
|
|
78
81
|
agent-learn task-policy --agent-id <agent_id> --task-id <task_id>
|
|
79
82
|
```
|
|
80
|
-
|
|
81
|
-
The subprocess-level functional workflow uses an isolated local store. Run the
|
|
82
|
-
interactive capture scenario first, followed by the offline batch update:
|
|
83
|
-
|
|
84
|
-
```powershell
|
|
85
|
-
python tests/functional_cli_interactive.py
|
|
86
|
-
python tests/functional_cli_batch.py
|
|
87
|
-
```
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "agent-learning"
|
|
7
|
-
version = "0.4.
|
|
7
|
+
version = "0.4.2"
|
|
8
8
|
description = "Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default."
|
|
9
9
|
readme = "PYPI.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -7,9 +7,9 @@ native, in-process learner. The SDK is organised into five layers:
|
|
|
7
7
|
``Reward``, ``PolicySnapshot``, ...).
|
|
8
8
|
- ``agent_learning.storage`` - pluggable persistence (Cosmos DB,
|
|
9
9
|
local file system, and in-memory).
|
|
10
|
-
- ``agent_learning.metrics`` -
|
|
11
|
-
|
|
12
|
-
|
|
10
|
+
- ``agent_learning.metrics`` - on-device stdlib metrics by default,
|
|
11
|
+
with optional Azure AI evaluators for Intent Resolution, Task
|
|
12
|
+
Adherence, and Task Completion.
|
|
13
13
|
- ``agent_learning.rewards`` - reward shaping + persistence.
|
|
14
14
|
- ``agent_learning.policy`` - discrete softmax bandit policy.
|
|
15
15
|
- ``agent_learning.learners`` - REINFORCE-with-baseline learner.
|
|
@@ -216,11 +216,11 @@ def _cmd_score(args: argparse.Namespace) -> int:
|
|
|
216
216
|
)
|
|
217
217
|
scored = 0
|
|
218
218
|
for episode in episodes:
|
|
219
|
-
|
|
220
|
-
if existing:
|
|
219
|
+
if runner.has_usable_reward(episode):
|
|
221
220
|
continue
|
|
222
|
-
runner.score_and_record(episode)
|
|
223
|
-
|
|
221
|
+
rewards = runner.score_and_record(episode)
|
|
222
|
+
if any(reward.source == RewardSource.AGGREGATE for reward in rewards):
|
|
223
|
+
scored += 1
|
|
224
224
|
print(json.dumps({"episodes_seen": len(episodes), "newly_scored": scored}, indent=2))
|
|
225
225
|
return 0
|
|
226
226
|
|
|
@@ -49,14 +49,14 @@ class ReinforceLearner(Learner):
|
|
|
49
49
|
|
|
50
50
|
# Index rewards by episode id - we only consume aggregate rewards
|
|
51
51
|
episode_list = list(episodes)
|
|
52
|
-
aggregate_rewards: Dict[str,
|
|
52
|
+
aggregate_rewards: Dict[str, Reward] = {}
|
|
53
53
|
for r in rewards:
|
|
54
54
|
if r.source != RewardSource.AGGREGATE:
|
|
55
55
|
continue
|
|
56
|
-
#
|
|
56
|
+
# Rescoring may append a replacement aggregate. Keep the newest.
|
|
57
57
|
current = aggregate_rewards.get(r.episode_id)
|
|
58
|
-
if current is None:
|
|
59
|
-
aggregate_rewards[r.episode_id] = r
|
|
58
|
+
if current is None or r.created_at > current.created_at:
|
|
59
|
+
aggregate_rewards[r.episode_id] = r
|
|
60
60
|
|
|
61
61
|
snapshot = policy.snapshot()
|
|
62
62
|
baseline_before = snapshot.baseline
|
|
@@ -72,9 +72,10 @@ class ReinforceLearner(Learner):
|
|
|
72
72
|
for episode in episode_list:
|
|
73
73
|
if episode.action_id is None or episode.action_id not in action_index:
|
|
74
74
|
continue
|
|
75
|
-
|
|
76
|
-
if
|
|
75
|
+
reward = aggregate_rewards.get(episode.id)
|
|
76
|
+
if reward is None:
|
|
77
77
|
continue
|
|
78
|
+
reward_value = reward.value
|
|
78
79
|
advantage = reward_value - baseline_before
|
|
79
80
|
|
|
80
81
|
# Importance sampling weight when the episode was logged under a
|
|
@@ -1,18 +1,19 @@
|
|
|
1
1
|
"""Score-based evaluation metrics for native RL reward shaping.
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
into the ``[0, 1]`` range expected by the reward shaper.
|
|
3
|
+
Metrics use on-device stdlib scorers by default. Configured Azure AI
|
|
4
|
+
evaluators remain available for remote LLM scoring.
|
|
6
5
|
"""
|
|
7
6
|
|
|
8
7
|
from .base import MetricEvaluator, MetricRequest
|
|
9
8
|
from .intent_resolution import IntentResolutionMetric
|
|
9
|
+
from .local import LocalScorerMetric
|
|
10
10
|
from .task_adherence import TaskAdherenceMetric
|
|
11
11
|
from .task_completion import TaskCompletionMetric
|
|
12
12
|
from .registry import default_metrics, evaluate_all
|
|
13
13
|
|
|
14
14
|
__all__ = [
|
|
15
15
|
"IntentResolutionMetric",
|
|
16
|
+
"LocalScorerMetric",
|
|
16
17
|
"MetricEvaluator",
|
|
17
18
|
"MetricRequest",
|
|
18
19
|
"TaskAdherenceMetric",
|
|
@@ -42,6 +42,14 @@ class MetricRequest:
|
|
|
42
42
|
system_message=episode.system_message,
|
|
43
43
|
tool_calls=_format_tool_calls(episode),
|
|
44
44
|
tool_definitions=episode.metadata.get("tool_definitions"),
|
|
45
|
+
extra={
|
|
46
|
+
"action_id": episode.action_id,
|
|
47
|
+
"context_features": episode.context_features,
|
|
48
|
+
"execution_status": episode.execution_status,
|
|
49
|
+
"expected_outcome": episode.expected_outcome,
|
|
50
|
+
"metadata": episode.metadata,
|
|
51
|
+
"result_summary": episode.result_summary,
|
|
52
|
+
},
|
|
45
53
|
)
|
|
46
54
|
|
|
47
55
|
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
"""Metric adapters for on-device scoring backends."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from collections.abc import Sequence
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from ..scorers.base import Scorer, ScoreResult
|
|
10
|
+
from ..scorers.stdlib._text import tokenize
|
|
11
|
+
from ..types import MetricName, MetricResult
|
|
12
|
+
from .base import MetricEvaluator, MetricRequest
|
|
13
|
+
|
|
14
|
+
logger = logging.getLogger(__name__)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class LocalScorerMetric(MetricEvaluator):
|
|
18
|
+
"""Project a local :class:`Scorer` onto the metric interface."""
|
|
19
|
+
|
|
20
|
+
NAME = MetricName.INTENT_RESOLUTION
|
|
21
|
+
|
|
22
|
+
def __init__(self, metric: MetricName, scorer: Scorer) -> None:
|
|
23
|
+
super().__init__(evaluator=scorer)
|
|
24
|
+
self.NAME = metric
|
|
25
|
+
self._scorer = scorer
|
|
26
|
+
|
|
27
|
+
def _build_evaluator(self) -> Any: # pragma: no cover - injected in __init__
|
|
28
|
+
return self._scorer
|
|
29
|
+
|
|
30
|
+
def _build_kwargs(self, request: MetricRequest) -> dict[str, Any]:
|
|
31
|
+
metadata = request.extra.get("metadata")
|
|
32
|
+
if not isinstance(metadata, dict):
|
|
33
|
+
metadata = {}
|
|
34
|
+
context_features = request.extra.get("context_features")
|
|
35
|
+
if not isinstance(context_features, dict):
|
|
36
|
+
context_features = {}
|
|
37
|
+
contract = metadata.get("score_contract") or metadata.get("adherence_contract")
|
|
38
|
+
if not isinstance(contract, dict):
|
|
39
|
+
contract = {}
|
|
40
|
+
expected_tokens = metadata.get("expected_tokens")
|
|
41
|
+
if not isinstance(expected_tokens, Sequence) or isinstance(
|
|
42
|
+
expected_tokens, (str, bytes)
|
|
43
|
+
):
|
|
44
|
+
expected_tokens = tokenize(str(request.extra.get("expected_outcome") or ""))
|
|
45
|
+
return {
|
|
46
|
+
"query": request.query,
|
|
47
|
+
"response": request.response,
|
|
48
|
+
"system_message": request.system_message,
|
|
49
|
+
"tool_calls": request.tool_calls,
|
|
50
|
+
"action_id": request.extra.get("action_id"),
|
|
51
|
+
"phi": context_features.get("phi"),
|
|
52
|
+
"contract": contract,
|
|
53
|
+
"expected_tokens": expected_tokens,
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
def _normalize(self, raw: dict[str, Any]) -> float | None:
|
|
57
|
+
value = raw.get("normalized")
|
|
58
|
+
return float(value) if value is not None else None
|
|
59
|
+
|
|
60
|
+
def evaluate(self, request: MetricRequest) -> MetricResult:
|
|
61
|
+
if not (request.query and request.response):
|
|
62
|
+
return MetricResult(
|
|
63
|
+
metric=self.NAME,
|
|
64
|
+
score=None,
|
|
65
|
+
normalized=None,
|
|
66
|
+
status="skipped",
|
|
67
|
+
reason="query or response is empty",
|
|
68
|
+
evaluator=f"local:{self._scorer.name}",
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
authoritative = self._authoritative_completion(request)
|
|
72
|
+
if authoritative is not None:
|
|
73
|
+
value, reason = authoritative
|
|
74
|
+
return MetricResult(
|
|
75
|
+
metric=self.NAME,
|
|
76
|
+
score=value,
|
|
77
|
+
normalized=value,
|
|
78
|
+
status="completed",
|
|
79
|
+
reason=reason,
|
|
80
|
+
evaluator="local:episode-outcome",
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
try:
|
|
84
|
+
result = self._scorer.score(**self._build_kwargs(request))
|
|
85
|
+
except Exception as exc: # noqa: BLE001 # pragma: no cover
|
|
86
|
+
logger.warning("Local metric %s failed: %s", self.NAME.value, exc)
|
|
87
|
+
return MetricResult(
|
|
88
|
+
metric=self.NAME,
|
|
89
|
+
score=None,
|
|
90
|
+
normalized=None,
|
|
91
|
+
status="skipped",
|
|
92
|
+
reason=f"local scorer error: {exc}",
|
|
93
|
+
evaluator=f"local:{self._scorer.name}",
|
|
94
|
+
)
|
|
95
|
+
return _to_metric_result(self.NAME, self._scorer.name, result)
|
|
96
|
+
|
|
97
|
+
def _authoritative_completion(
|
|
98
|
+
self, request: MetricRequest
|
|
99
|
+
) -> tuple[float, str] | None:
|
|
100
|
+
if self.NAME != MetricName.TASK_COMPLETION:
|
|
101
|
+
return None
|
|
102
|
+
metadata = request.extra.get("metadata")
|
|
103
|
+
if not isinstance(metadata, dict):
|
|
104
|
+
metadata = {}
|
|
105
|
+
action_id = request.extra.get("action_id")
|
|
106
|
+
correct_action_id = metadata.get("correct_action_id")
|
|
107
|
+
if action_id and correct_action_id:
|
|
108
|
+
return (
|
|
109
|
+
float(action_id == correct_action_id),
|
|
110
|
+
"derived from action_id and metadata.correct_action_id",
|
|
111
|
+
)
|
|
112
|
+
completed = metadata.get("task_completed")
|
|
113
|
+
if isinstance(completed, bool):
|
|
114
|
+
return float(completed), "derived from metadata.task_completed"
|
|
115
|
+
status = str(request.extra.get("execution_status") or "").lower()
|
|
116
|
+
if status == "completed":
|
|
117
|
+
return 1.0, "derived from execution_status=completed"
|
|
118
|
+
if status == "failed":
|
|
119
|
+
return 0.0, "derived from execution_status=failed"
|
|
120
|
+
if status == "partial":
|
|
121
|
+
return 0.5, "derived from execution_status=partial"
|
|
122
|
+
return None
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _to_metric_result(
|
|
126
|
+
metric: MetricName, scorer_name: str, result: ScoreResult
|
|
127
|
+
) -> MetricResult:
|
|
128
|
+
normalized = max(0.0, min(1.0, float(result.normalized)))
|
|
129
|
+
return MetricResult(
|
|
130
|
+
metric=metric,
|
|
131
|
+
score=normalized,
|
|
132
|
+
normalized=normalized,
|
|
133
|
+
status="completed",
|
|
134
|
+
reason=f"local {result.label}",
|
|
135
|
+
properties=dict(result.features),
|
|
136
|
+
evaluator=f"local:{scorer_name}",
|
|
137
|
+
metadata={"label": result.label, "confidence": result.confidence},
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def local_metrics(scorers: tuple[Scorer, Scorer, Scorer]) -> list[MetricEvaluator]:
|
|
142
|
+
"""Return local metric adapters in reward-shaping order."""
|
|
143
|
+
intent, adherence, completion = scorers
|
|
144
|
+
return [
|
|
145
|
+
LocalScorerMetric(MetricName.INTENT_RESOLUTION, intent),
|
|
146
|
+
LocalScorerMetric(MetricName.TASK_ADHERENCE, adherence),
|
|
147
|
+
LocalScorerMetric(MetricName.TASK_COMPLETION, completion),
|
|
148
|
+
]
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
__all__ = ["LocalScorerMetric", "local_metrics"]
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""Convenience helpers to evaluate an episode across all default metrics."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Iterable, List, Optional
|
|
6
|
+
|
|
7
|
+
from ..config import ScoreConfig, ScoreRuntimeConfig
|
|
8
|
+
from ..scorers import build_scorers
|
|
9
|
+
from ..types import Episode, MetricResult
|
|
10
|
+
from .base import MetricEvaluator, MetricRequest
|
|
11
|
+
from .intent_resolution import IntentResolutionMetric
|
|
12
|
+
from .local import local_metrics
|
|
13
|
+
from .task_adherence import TaskAdherenceMetric
|
|
14
|
+
from .task_completion import TaskCompletionMetric
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def default_metrics(
|
|
18
|
+
score_config: Optional[ScoreConfig] = None,
|
|
19
|
+
score_runtime_config: Optional[ScoreRuntimeConfig] = None,
|
|
20
|
+
) -> List[MetricEvaluator]:
|
|
21
|
+
"""Return local metrics by default, or Azure metrics when configured."""
|
|
22
|
+
runtime = score_runtime_config or ScoreRuntimeConfig()
|
|
23
|
+
llm = score_config or runtime.llm
|
|
24
|
+
if score_config is not None or runtime.tier == "llm" or (
|
|
25
|
+
runtime.tier is None and llm.enabled
|
|
26
|
+
):
|
|
27
|
+
return [
|
|
28
|
+
IntentResolutionMetric(llm),
|
|
29
|
+
TaskAdherenceMetric(llm),
|
|
30
|
+
TaskCompletionMetric(llm),
|
|
31
|
+
]
|
|
32
|
+
if runtime.tier is None:
|
|
33
|
+
runtime.tier = "stdlib"
|
|
34
|
+
return local_metrics(build_scorers(runtime))
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def evaluate_all(
|
|
38
|
+
episode: Episode,
|
|
39
|
+
metrics: Optional[Iterable[MetricEvaluator]] = None,
|
|
40
|
+
*,
|
|
41
|
+
score_config: Optional[ScoreConfig] = None,
|
|
42
|
+
score_runtime_config: Optional[ScoreRuntimeConfig] = None,
|
|
43
|
+
) -> List[MetricResult]:
|
|
44
|
+
"""Evaluate one episode against every supplied metric.
|
|
45
|
+
|
|
46
|
+
Each metric is given its own try/except inside :meth:`evaluate`,
|
|
47
|
+
so a single failing scorer will not prevent the others from
|
|
48
|
+
producing scores.
|
|
49
|
+
"""
|
|
50
|
+
metric_list = (
|
|
51
|
+
list(metrics)
|
|
52
|
+
if metrics is not None
|
|
53
|
+
else default_metrics(score_config, score_runtime_config)
|
|
54
|
+
)
|
|
55
|
+
request = MetricRequest.from_episode(episode)
|
|
56
|
+
return [m.evaluate(request) for m in metric_list]
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
__all__ = ["default_metrics", "evaluate_all"]
|
|
@@ -98,7 +98,10 @@ class RewardWriter:
|
|
|
98
98
|
except Exception as exc: # pragma: no cover
|
|
99
99
|
logger.warning("Failed to persist %s penalty for %s: %s", kind, episode.id, exc)
|
|
100
100
|
|
|
101
|
-
# 4) Aggregate reward consumed by the learner
|
|
101
|
+
# 4) Aggregate reward consumed by the learner. Do not turn an
|
|
102
|
+
# all-skipped evaluation into a misleading neutral reward.
|
|
103
|
+
if not shaped.metric_contributions and not shaped.penalties:
|
|
104
|
+
return stored
|
|
102
105
|
aggregate = Reward(
|
|
103
106
|
episode_id=episode.id,
|
|
104
107
|
agent_id=episode.agent_id,
|
|
@@ -19,7 +19,7 @@ import logging
|
|
|
19
19
|
from datetime import datetime, timezone
|
|
20
20
|
from typing import Dict, Iterable, List, Optional
|
|
21
21
|
|
|
22
|
-
from ..config import LearnerConfig, ScoreConfig, ShapingConfig
|
|
22
|
+
from ..config import LearnerConfig, ScoreConfig, ScoreRuntimeConfig, ShapingConfig
|
|
23
23
|
from ..learners.base import Learner, LearnerResult
|
|
24
24
|
from ..learners.reinforce import ReinforceLearner
|
|
25
25
|
from ..metrics.base import MetricEvaluator
|
|
@@ -47,12 +47,17 @@ class LearningRunner:
|
|
|
47
47
|
writer: Optional[RewardWriter] = None,
|
|
48
48
|
learner: Optional[Learner] = None,
|
|
49
49
|
score_config: Optional[ScoreConfig] = None,
|
|
50
|
+
score_runtime_config: Optional[ScoreRuntimeConfig] = None,
|
|
50
51
|
learner_config: Optional[LearnerConfig] = None,
|
|
51
52
|
shaping_config: Optional[ShapingConfig] = None,
|
|
52
53
|
) -> None:
|
|
53
54
|
self._store = store or get_default_store()
|
|
54
55
|
self._policy = policy
|
|
55
|
-
self._metrics =
|
|
56
|
+
self._metrics = (
|
|
57
|
+
list(metrics)
|
|
58
|
+
if metrics is not None
|
|
59
|
+
else default_metrics(score_config, score_runtime_config)
|
|
60
|
+
)
|
|
56
61
|
self._shaper = shaper or RewardShaper(shaping_config)
|
|
57
62
|
self._writer = writer or RewardWriter(self._store)
|
|
58
63
|
self._learner = learner or ReinforceLearner(learner_config)
|
|
@@ -137,6 +142,19 @@ class LearningRunner:
|
|
|
137
142
|
# Helpers
|
|
138
143
|
# ------------------------------------------------------------------
|
|
139
144
|
|
|
145
|
+
def has_usable_reward(self, episode: Episode) -> bool:
|
|
146
|
+
"""Return whether an episode has an aggregate backed by valid metrics."""
|
|
147
|
+
rewards = self._store.get_rewards_for_episode(episode.id, episode.agent_id)
|
|
148
|
+
if not any(reward.source.value == "aggregate" for reward in rewards):
|
|
149
|
+
return False
|
|
150
|
+
metrics = self._store.get_metric_results(episode.id, episode.agent_id)
|
|
151
|
+
if not metrics:
|
|
152
|
+
return True
|
|
153
|
+
return any(
|
|
154
|
+
result.status == "completed" and result.normalized is not None
|
|
155
|
+
for result in metrics
|
|
156
|
+
)
|
|
157
|
+
|
|
140
158
|
def _collect_rewards(
|
|
141
159
|
self,
|
|
142
160
|
agent_id: str,
|
|
@@ -147,7 +165,7 @@ class LearningRunner:
|
|
|
147
165
|
rewards: List[Reward] = []
|
|
148
166
|
for episode in episodes:
|
|
149
167
|
existing = self._store.get_rewards_for_episode(episode.id, agent_id)
|
|
150
|
-
if
|
|
168
|
+
if self.has_usable_reward(episode):
|
|
151
169
|
rewards.extend(existing)
|
|
152
170
|
continue
|
|
153
171
|
if not score_missing:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: agent-learning
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.2
|
|
4
4
|
Summary: Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default.
|
|
5
5
|
Author: Chris Tava
|
|
6
6
|
License: MIT License
|
|
@@ -75,7 +75,7 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
|
|
|
75
75
|
|
|
76
76
|
1. The **policy** is a softmax distribution over `N` discrete actions (e.g., "take action A", "take action B", "take action C"). It lives in Python and updates in milliseconds.
|
|
77
77
|
|
|
78
|
-
2. Each episode is **evaluated** by three
|
|
78
|
+
2. Each episode is **evaluated** on-device by three stdlib scorers for intent resolution, task adherence, and task completion. Their scores are combined into a single scalar reward with no scoring endpoint or environment variables required. Azure AI evaluators remain available as an opt-in.
|
|
79
79
|
|
|
80
80
|
3. A **REINFORCE-with-baseline** learner updates the policy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
|
|
81
81
|
|
|
@@ -29,6 +29,7 @@ src/agent_learning/learners/reinforce.py
|
|
|
29
29
|
src/agent_learning/metrics/__init__.py
|
|
30
30
|
src/agent_learning/metrics/base.py
|
|
31
31
|
src/agent_learning/metrics/intent_resolution.py
|
|
32
|
+
src/agent_learning/metrics/local.py
|
|
32
33
|
src/agent_learning/metrics/registry.py
|
|
33
34
|
src/agent_learning/metrics/task_adherence.py
|
|
34
35
|
src/agent_learning/metrics/task_completion.py
|
|
@@ -79,6 +80,7 @@ tests/test_contextual_policy.py
|
|
|
79
80
|
tests/test_default_store.py
|
|
80
81
|
tests/test_end_to_end.py
|
|
81
82
|
tests/test_learner.py
|
|
83
|
+
tests/test_metrics_local.py
|
|
82
84
|
tests/test_policy.py
|
|
83
85
|
tests/test_scorers.py
|
|
84
86
|
tests/test_scorers_llm.py
|
|
@@ -167,6 +167,96 @@ def test_task_episode_register_persists_full_episode(
|
|
|
167
167
|
assert episode.is_full
|
|
168
168
|
|
|
169
169
|
|
|
170
|
+
def test_score_uses_local_stdlib_without_configuration(
|
|
171
|
+
monkeypatch, capsys
|
|
172
|
+
) -> None:
|
|
173
|
+
for name in (
|
|
174
|
+
"AGENT_LEARNING_SCORE_ENDPOINT",
|
|
175
|
+
"AGENT_LEARNING_SCORE_DEPLOYMENT",
|
|
176
|
+
"AGENT_LEARNING_SCORE_TIER",
|
|
177
|
+
):
|
|
178
|
+
monkeypatch.delenv(name, raising=False)
|
|
179
|
+
store = InMemoryStore()
|
|
180
|
+
monkeypatch.setattr(cli, "get_default_store", lambda: store)
|
|
181
|
+
episode = Episode(
|
|
182
|
+
agent_id="scout",
|
|
183
|
+
task_id="context-window",
|
|
184
|
+
user_input="What is the context window?",
|
|
185
|
+
assistant_output="The context window is 922,000 tokens.",
|
|
186
|
+
intent_summary="Report the context window",
|
|
187
|
+
action_id="inspect_context",
|
|
188
|
+
expected_outcome="Return the live context limit",
|
|
189
|
+
execution_status="completed",
|
|
190
|
+
result_summary="Returned the live limit",
|
|
191
|
+
metadata={
|
|
192
|
+
"correct_action_id": "inspect_context",
|
|
193
|
+
"task_completed": True,
|
|
194
|
+
},
|
|
195
|
+
)
|
|
196
|
+
store.store_episode(episode)
|
|
197
|
+
|
|
198
|
+
assert cli.main(["score", "--agent-id", "scout"]) == 0
|
|
199
|
+
result = json.loads(capsys.readouterr().out)
|
|
200
|
+
assert result == {"episodes_seen": 1, "newly_scored": 1}
|
|
201
|
+
metrics = store.get_metric_results(episode.id, episode.agent_id)
|
|
202
|
+
assert len(metrics) == 3
|
|
203
|
+
assert all(metric.status == "completed" for metric in metrics)
|
|
204
|
+
assert all((metric.evaluator or "").startswith("local:") for metric in metrics)
|
|
205
|
+
rewards = store.get_rewards_for_episode(episode.id, episode.agent_id)
|
|
206
|
+
aggregate = next(reward for reward in rewards if reward.source == RewardSource.AGGREGATE)
|
|
207
|
+
assert aggregate.value > 0.0
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def test_score_replaces_skipped_only_evaluation(monkeypatch, capsys) -> None:
|
|
211
|
+
store = InMemoryStore()
|
|
212
|
+
monkeypatch.setattr(cli, "get_default_store", lambda: store)
|
|
213
|
+
episode = Episode(
|
|
214
|
+
agent_id="scout",
|
|
215
|
+
task_id="context-window",
|
|
216
|
+
user_input="What is the context window?",
|
|
217
|
+
assistant_output="The context window is 922,000 tokens.",
|
|
218
|
+
action_id="inspect_context",
|
|
219
|
+
execution_status="completed",
|
|
220
|
+
metadata={"task_completed": True},
|
|
221
|
+
)
|
|
222
|
+
store.store_episode(episode)
|
|
223
|
+
store.store_metric_results(
|
|
224
|
+
episode.id,
|
|
225
|
+
episode.agent_id,
|
|
226
|
+
[
|
|
227
|
+
MetricResult(
|
|
228
|
+
metric=metric,
|
|
229
|
+
score=None,
|
|
230
|
+
normalized=None,
|
|
231
|
+
status="skipped",
|
|
232
|
+
reason="remote scorer was not configured",
|
|
233
|
+
)
|
|
234
|
+
for metric in MetricName
|
|
235
|
+
],
|
|
236
|
+
)
|
|
237
|
+
store.store_reward(
|
|
238
|
+
Reward(
|
|
239
|
+
episode_id=episode.id,
|
|
240
|
+
agent_id=episode.agent_id,
|
|
241
|
+
source=RewardSource.AGGREGATE,
|
|
242
|
+
value=0.0,
|
|
243
|
+
created_at="2026-08-09T00:00:00+00:00",
|
|
244
|
+
)
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
assert cli.main(["score", "--agent-id", "scout"]) == 0
|
|
248
|
+
assert json.loads(capsys.readouterr().out)["newly_scored"] == 1
|
|
249
|
+
metrics = store.get_metric_results(episode.id, episode.agent_id)
|
|
250
|
+
assert sum(metric.status == "completed" for metric in metrics) == 3
|
|
251
|
+
aggregates = [
|
|
252
|
+
reward
|
|
253
|
+
for reward in store.get_rewards_for_episode(episode.id, episode.agent_id)
|
|
254
|
+
if reward.source == RewardSource.AGGREGATE
|
|
255
|
+
]
|
|
256
|
+
assert len(aggregates) == 2
|
|
257
|
+
assert max(aggregates, key=lambda reward: reward.created_at).value > 0.0
|
|
258
|
+
|
|
259
|
+
|
|
170
260
|
def test_agent_training_uses_one_limit_and_preserves_task_policy_history(
|
|
171
261
|
monkeypatch, capsys
|
|
172
262
|
) -> None:
|
|
@@ -89,3 +89,30 @@ def test_unknown_action_is_skipped() -> None:
|
|
|
89
89
|
]
|
|
90
90
|
result = learner.update(policy, [ep], rewards)
|
|
91
91
|
assert result.episodes_used == 0
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def test_most_recent_aggregate_reward_wins() -> None:
|
|
95
|
+
policy = SoftmaxPolicy.from_actions([Action(id="a"), Action(id="b")])
|
|
96
|
+
learner = ReinforceLearner(LearnerConfig(learning_rate=0.5, entropy_bonus=0.0))
|
|
97
|
+
episode = _make_episode("default", "a")
|
|
98
|
+
rewards = [
|
|
99
|
+
Reward(
|
|
100
|
+
episode_id=episode.id,
|
|
101
|
+
agent_id=episode.agent_id,
|
|
102
|
+
source=RewardSource.AGGREGATE,
|
|
103
|
+
value=0.0,
|
|
104
|
+
created_at="2026-08-09T00:00:00+00:00",
|
|
105
|
+
),
|
|
106
|
+
Reward(
|
|
107
|
+
episode_id=episode.id,
|
|
108
|
+
agent_id=episode.agent_id,
|
|
109
|
+
source=RewardSource.AGGREGATE,
|
|
110
|
+
value=0.8,
|
|
111
|
+
created_at="2026-08-09T00:00:01+00:00",
|
|
112
|
+
),
|
|
113
|
+
]
|
|
114
|
+
|
|
115
|
+
result = learner.update(policy, [episode], rewards)
|
|
116
|
+
|
|
117
|
+
assert result.mean_reward == 0.8
|
|
118
|
+
assert policy.snapshot().logits["a"] > 0.0
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""Tests for default on-device metric routing."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from agent_learning.config import ScoreConfig
|
|
6
|
+
from agent_learning.metrics import (
|
|
7
|
+
IntentResolutionMetric,
|
|
8
|
+
LocalScorerMetric,
|
|
9
|
+
default_metrics,
|
|
10
|
+
evaluate_all,
|
|
11
|
+
)
|
|
12
|
+
from agent_learning.types import Episode, MetricName
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_default_metrics_are_local_without_azure_configuration(monkeypatch) -> None:
|
|
16
|
+
for name in (
|
|
17
|
+
"AGENT_LEARNING_SCORE_ENDPOINT",
|
|
18
|
+
"AGENT_LEARNING_SCORE_DEPLOYMENT",
|
|
19
|
+
"AGENT_LEARNING_SCORE_TIER",
|
|
20
|
+
):
|
|
21
|
+
monkeypatch.delenv(name, raising=False)
|
|
22
|
+
|
|
23
|
+
metrics = default_metrics()
|
|
24
|
+
|
|
25
|
+
assert len(metrics) == 3
|
|
26
|
+
assert all(isinstance(metric, LocalScorerMetric) for metric in metrics)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def test_explicit_score_config_preserves_azure_metrics() -> None:
|
|
30
|
+
metrics = default_metrics(
|
|
31
|
+
ScoreConfig(
|
|
32
|
+
azure_endpoint="https://example.openai.azure.com",
|
|
33
|
+
azure_deployment="grader",
|
|
34
|
+
credential_mode="none",
|
|
35
|
+
)
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
assert isinstance(metrics[0], IntentResolutionMetric)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def test_correct_action_id_overrides_contradictory_completion_flag(
|
|
42
|
+
monkeypatch,
|
|
43
|
+
) -> None:
|
|
44
|
+
monkeypatch.delenv("AGENT_LEARNING_SCORE_ENDPOINT", raising=False)
|
|
45
|
+
monkeypatch.delenv("AGENT_LEARNING_SCORE_DEPLOYMENT", raising=False)
|
|
46
|
+
episode = Episode(
|
|
47
|
+
user_input="Answer the task",
|
|
48
|
+
assistant_output="Used the wrong action",
|
|
49
|
+
action_id="wrong",
|
|
50
|
+
execution_status="completed",
|
|
51
|
+
metadata={"correct_action_id": "right", "task_completed": True},
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
results = evaluate_all(episode)
|
|
55
|
+
|
|
56
|
+
completion = next(
|
|
57
|
+
result for result in results if result.metric == MetricName.TASK_COMPLETION
|
|
58
|
+
)
|
|
59
|
+
assert completion.normalized == 0.0
|
|
60
|
+
assert completion.reason == "derived from action_id and metadata.correct_action_id"
|
|
@@ -3,8 +3,10 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
from agent_learning.config import ShapingConfig
|
|
6
|
+
from agent_learning.rewards import RewardWriter
|
|
6
7
|
from agent_learning.rewards.shaping import RewardShaper
|
|
7
|
-
from agent_learning.
|
|
8
|
+
from agent_learning.storage import InMemoryStore
|
|
9
|
+
from agent_learning.types import Episode, MetricName, MetricResult, RewardSource
|
|
8
10
|
|
|
9
11
|
|
|
10
12
|
def _result(metric: MetricName, normalized: float) -> MetricResult:
|
|
@@ -71,6 +73,26 @@ def test_skipped_metrics_are_ignored() -> None:
|
|
|
71
73
|
assert abs(shaped.value - 0.7) < 1e-9
|
|
72
74
|
|
|
73
75
|
|
|
76
|
+
def test_all_skipped_metrics_do_not_persist_aggregate() -> None:
|
|
77
|
+
store = InMemoryStore()
|
|
78
|
+
episode = Episode(agent_id="scout")
|
|
79
|
+
results = [
|
|
80
|
+
MetricResult(
|
|
81
|
+
metric=metric,
|
|
82
|
+
score=None,
|
|
83
|
+
normalized=None,
|
|
84
|
+
status="skipped",
|
|
85
|
+
)
|
|
86
|
+
for metric in MetricName
|
|
87
|
+
]
|
|
88
|
+
shaped = RewardShaper().shape(results)
|
|
89
|
+
|
|
90
|
+
rewards = RewardWriter(store).write(episode, results, shaped)
|
|
91
|
+
|
|
92
|
+
assert not any(reward.source == RewardSource.AGGREGATE for reward in rewards)
|
|
93
|
+
assert store.get_rewards_for_episode(episode.id, episode.agent_id) == []
|
|
94
|
+
|
|
95
|
+
|
|
74
96
|
def test_latency_penalty_applied() -> None:
|
|
75
97
|
cfg = ShapingConfig(
|
|
76
98
|
intent_resolution_weight=0.0,
|
|
@@ -1,42 +0,0 @@
|
|
|
1
|
-
"""Convenience helpers to evaluate an episode across all default metrics."""
|
|
2
|
-
|
|
3
|
-
from __future__ import annotations
|
|
4
|
-
|
|
5
|
-
from typing import Iterable, List, Optional
|
|
6
|
-
|
|
7
|
-
from ..config import ScoreConfig
|
|
8
|
-
from ..types import Episode, MetricResult
|
|
9
|
-
from .base import MetricEvaluator, MetricRequest
|
|
10
|
-
from .intent_resolution import IntentResolutionMetric
|
|
11
|
-
from .task_adherence import TaskAdherenceMetric
|
|
12
|
-
from .task_completion import TaskCompletionMetric
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
def default_metrics(score_config: Optional[ScoreConfig] = None) -> List[MetricEvaluator]:
|
|
16
|
-
"""Return the three native metrics wired with the same score config."""
|
|
17
|
-
cfg = score_config or ScoreConfig()
|
|
18
|
-
return [
|
|
19
|
-
IntentResolutionMetric(cfg),
|
|
20
|
-
TaskAdherenceMetric(cfg),
|
|
21
|
-
TaskCompletionMetric(cfg),
|
|
22
|
-
]
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
def evaluate_all(
|
|
26
|
-
episode: Episode,
|
|
27
|
-
metrics: Optional[Iterable[MetricEvaluator]] = None,
|
|
28
|
-
*,
|
|
29
|
-
score_config: Optional[ScoreConfig] = None,
|
|
30
|
-
) -> List[MetricResult]:
|
|
31
|
-
"""Evaluate one episode against every supplied metric.
|
|
32
|
-
|
|
33
|
-
Each metric is given its own try/except inside :meth:`evaluate`,
|
|
34
|
-
so a single failing scorer will not prevent the others from
|
|
35
|
-
producing scores.
|
|
36
|
-
"""
|
|
37
|
-
metric_list = list(metrics) if metrics is not None else default_metrics(score_config)
|
|
38
|
-
request = MetricRequest.from_episode(episode)
|
|
39
|
-
return [m.evaluate(request) for m in metric_list]
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
__all__ = ["default_metrics", "evaluate_all"]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/scorers/__init__.py
RENAMED
|
File without changes
|
{agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/scorers/_base.py
RENAMED
|
File without changes
|
{agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/scorers/adherence.py
RENAMED
|
File without changes
|
{agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/scorers/completion.py
RENAMED
|
File without changes
|
{agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/classifiers/scorers/intent.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/metrics/intent_resolution.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/policy/contextual_softmax.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp_text/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp_text/adherence.py
RENAMED
|
File without changes
|
{agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/nlp_text/completion.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/stdlib/adherence.py
RENAMED
|
File without changes
|
{agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning/scorers/stdlib/completion.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agent_learning-0.4.1 → agent_learning-0.4.2}/src/agent_learning.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|