agent-learning 0.4.1__tar.gz → 0.4.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agent_learning-0.4.1/src/agent_learning.egg-info → agent_learning-0.4.3}/PKG-INFO +5 -5
- agent_learning-0.4.3/PYPI.md +15 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/README.md +20 -25
- {agent_learning-0.4.1 → agent_learning-0.4.3}/pyproject.toml +1 -1
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/__init__.py +3 -3
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/_version.py +1 -1
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/cli.py +37 -6
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/learners/reinforce.py +7 -6
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/metrics/__init__.py +4 -3
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/metrics/base.py +8 -0
- agent_learning-0.4.3/src/agent_learning/metrics/local.py +151 -0
- agent_learning-0.4.3/src/agent_learning/metrics/registry.py +59 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/rewards/writer.py +4 -1
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/storage/base.py +3 -1
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/storage/cosmos.py +8 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/storage/local.py +4 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/storage/memory.py +4 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/training/runner.py +21 -3
- {agent_learning-0.4.1 → agent_learning-0.4.3/src/agent_learning.egg-info}/PKG-INFO +5 -5
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning.egg-info/SOURCES.txt +2 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_cli.py +180 -1
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_learner.py +27 -0
- agent_learning-0.4.3/tests/test_metrics_local.py +60 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_shaping.py +23 -1
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_storage_local.py +3 -0
- agent_learning-0.4.1/PYPI.md +0 -15
- agent_learning-0.4.1/src/agent_learning/metrics/registry.py +0 -42
- {agent_learning-0.4.1 → agent_learning-0.4.3}/LICENSE +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/setup.cfg +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/capture.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/router.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/scorers/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/scorers/_base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/scorers/adherence.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/scorers/completion.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/scorers/intent.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/config.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/learners/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/learners/base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/metrics/intent_resolution.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/metrics/task_adherence.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/metrics/task_completion.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/policy/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/policy/base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/policy/contextual_softmax.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/policy/softmax_bandit.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/py.typed +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/rewards/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/rewards/shaping.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/llm/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/llm/_base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/llm/adherence.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/llm/completion.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/llm/intent.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp/_base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp/adherence.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp/completion.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp/intent.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp_text/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp_text/_base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp_text/adherence.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp_text/completion.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp_text/intent.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/slm/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/slm/_base.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/slm/adherence.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/slm/completion.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/slm/intent.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/stdlib/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/stdlib/_text.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/stdlib/adherence.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/stdlib/completion.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/stdlib/intent.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/storage/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/training/__init__.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/types.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning.egg-info/dependency_links.txt +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning.egg-info/entry_points.txt +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning.egg-info/requires.txt +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning.egg-info/top_level.txt +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_capture.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_contextual_policy.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_default_store.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_end_to_end.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_policy.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_scorers.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_scorers_llm.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_scorers_nlp_text.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_scorers_slm.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_scorers_stdlib.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_storage_memory.py +0 -0
- {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_types.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: agent-learning
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.3
|
|
4
4
|
Summary: Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default.
|
|
5
5
|
Author: Chris Tava
|
|
6
6
|
License: MIT License
|
|
@@ -67,16 +67,16 @@ Dynamic: license-file
|
|
|
67
67
|
|
|
68
68
|
# agent-learning
|
|
69
69
|
|
|
70
|
-
Native reinforcement learning SDK for AI agents. An in-process
|
|
70
|
+
Native reinforcement learning SDK for AI agents. An in-process Learner optimizes a small, interpretable TaskPolicy over discrete agent choices (understand intent and complete task by choosing the right outcome).
|
|
71
71
|
|
|
72
72
|
## How it works
|
|
73
73
|
|
|
74
74
|
The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tune jobs and no opaque update cycles — just three pieces that run in your existing Python process:
|
|
75
75
|
|
|
76
|
-
1.
|
|
76
|
+
1. **TaskPolicy** is a softmax distribution over `N` discrete actions (e.g., "take action A", "take action B", "take action C"). It lives in Python and updates in milliseconds.
|
|
77
77
|
|
|
78
|
-
2.
|
|
78
|
+
2. **Score** evaluates each episode on-device with three stdlib scorers for intent resolution, task adherence, and task completion. Their scores are combined into a single scalar reward with no scoring endpoint or environment variables required. Azure AI evaluators remain available as an opt-in.
|
|
79
79
|
|
|
80
|
-
3.
|
|
80
|
+
3. **Learner** applies REINFORCE-with-baseline to update TaskPolicy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
|
|
81
81
|
|
|
82
82
|
Every episode, reward, run, and deployment is captured by the configured store — in-memory or local files by default, or Azure Cosmos DB — giving you a complete lineage and audit trail of how the policy evolved over time.
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# agent-learning
|
|
2
|
+
|
|
3
|
+
Native reinforcement learning SDK for AI agents. An in-process Learner optimizes a small, interpretable TaskPolicy over discrete agent choices (understand intent and complete task by choosing the right outcome).
|
|
4
|
+
|
|
5
|
+
## How it works
|
|
6
|
+
|
|
7
|
+
The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tune jobs and no opaque update cycles — just three pieces that run in your existing Python process:
|
|
8
|
+
|
|
9
|
+
1. **TaskPolicy** is a softmax distribution over `N` discrete actions (e.g., "take action A", "take action B", "take action C"). It lives in Python and updates in milliseconds.
|
|
10
|
+
|
|
11
|
+
2. **Score** evaluates each episode on-device with three stdlib scorers for intent resolution, task adherence, and task completion. Their scores are combined into a single scalar reward with no scoring endpoint or environment variables required. Azure AI evaluators remain available as an opt-in.
|
|
12
|
+
|
|
13
|
+
3. **Learner** applies REINFORCE-with-baseline to update TaskPolicy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
|
|
14
|
+
|
|
15
|
+
Every episode, reward, run, and deployment is captured by the configured store — in-memory or local files by default, or Azure Cosmos DB — giving you a complete lineage and audit trail of how the policy evolved over time.
|
|
@@ -1,29 +1,30 @@
|
|
|
1
1
|
# agent-learning
|
|
2
2
|
|
|
3
3
|
Native reinforcement learning SDK for AI agents. An in-process
|
|
4
|
-
|
|
5
|
-
signal.
|
|
4
|
+
Learner optimizes a small, interpretable TaskPolicy over discrete agent choices (e.g., "take action A", "take action B", "take action C") using on-device evaluation scores as the reward
|
|
5
|
+
signal by default.
|
|
6
6
|
|
|
7
7
|
<p align="center">
|
|
8
|
-
<img src="images/agent-learning-loop.svg" alt="Animated loop:
|
|
8
|
+
<img src="images/agent-learning-loop.svg" alt="Animated TaskPolicy Score Learner loop: TaskPolicy chooses a task action, Score evaluates the episode, and Learner updates TaskPolicy" width="960" style="max-width:100%; height:auto;" />
|
|
9
9
|
</p>
|
|
10
10
|
|
|
11
11
|
## How it works
|
|
12
12
|
|
|
13
13
|
The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tune jobs and no opaque update cycles — just three pieces that run in your existing Python process:
|
|
14
14
|
|
|
15
|
-
1.
|
|
15
|
+
1. **TaskPolicy** is a softmax distribution over `N` discrete
|
|
16
16
|
actions (e.g., "take action A", "take action B", "take action C"). It lives in Python and updates in milliseconds.
|
|
17
17
|
|
|
18
|
-
<img src="images/0f85e08d0c47cd01.png" alt="
|
|
18
|
+
<img src="images/0f85e08d0c47cd01.png" alt="TaskPolicy selects one of N discrete actions" width="360" style="max-width:100%; height:auto;" />
|
|
19
19
|
|
|
20
|
-
2.
|
|
21
|
-
|
|
22
|
-
|
|
20
|
+
2. **Score** evaluates each episode locally with three stdlib scorers for intent
|
|
21
|
+
resolution, task adherence, and task completion. Their scores are combined
|
|
22
|
+
into one scalar reward. No scoring endpoint or environment variable is
|
|
23
|
+
required. Configured Azure AI evaluators remain available as an opt-in.
|
|
23
24
|
|
|
24
25
|
<img src="images/246d112f995b785a.png" alt="Three evaluator scores feed a single scalar reward" width="360" style="max-width:100%; height:auto;" />
|
|
25
26
|
|
|
26
|
-
3.
|
|
27
|
+
3. **Learner** applies REINFORCE-with-baseline to update TaskPolicy logits
|
|
27
28
|
directly from stored episodes. Updates are tiny gradient steps
|
|
28
29
|
that run on local compute and persist through a pluggable store — in-memory
|
|
29
30
|
or local files by default, with Azure Cosmos DB optional.
|
|
@@ -54,14 +55,16 @@ agent-learn.exe --help
|
|
|
54
55
|
Released versions are published to PyPI:
|
|
55
56
|
<https://pypi.org/project/agent-learning/>.
|
|
56
57
|
|
|
58
|
+
## Functional Testing
|
|
59
|
+
|
|
60
|
+
A good way to see the SDK in action is to run the
|
|
61
|
+
interactive capture scenario first, followed by the offline batch update:
|
|
62
|
+
|
|
57
63
|
```powershell
|
|
58
|
-
py
|
|
59
|
-
|
|
64
|
+
python tests/functional_cli_interactive.py
|
|
65
|
+
python tests/functional_cli_batch.py
|
|
60
66
|
```
|
|
61
67
|
|
|
62
|
-
`pip` installs `agent-learn.exe` into the active Python environment's
|
|
63
|
-
`Scripts` directory.
|
|
64
|
-
|
|
65
68
|
## Usage
|
|
66
69
|
|
|
67
70
|
The `agent-learn` CLI provides the current task-learning-loop operations:
|
|
@@ -69,19 +72,11 @@ The `agent-learn` CLI provides the current task-learning-loop operations:
|
|
|
69
72
|
```text
|
|
70
73
|
agent-learn list
|
|
71
74
|
agent-learn tasks-list <agent_id>
|
|
72
|
-
agent-learn task-episodes-count <agent_id> [--task-id <task_id>]
|
|
73
|
-
agent-learn task-episodes-list <agent_id> [--task-id <task_id>] [--limit <1-500>] [--include-incomplete]
|
|
75
|
+
agent-learn task-episodes-count <agent_id> [--task-id <task_id>] [--start-date <date>] [--end-date <date>]
|
|
76
|
+
agent-learn task-episodes-list <agent_id> [--task-id <task_id>] [--limit <1-500>] [--include-incomplete] [--start-date <date>] [--end-date <date>]
|
|
74
77
|
agent-learn task-policy-init --agent-id <agent_id> --task-id <task_id> --actions ./actions.json
|
|
75
78
|
agent-learn task-episode-register --agent-id <agent_id> --task-id <task_id> --episode ./episode.json
|
|
76
79
|
agent-learn score --agent-id <agent_id> [--task-id <task_id>] [--limit <1-500>]
|
|
77
|
-
agent-learn train --agent-id <agent_id> [--task-id <task_id>] [--limit <1-500>] [--start-date <date>] [--end-date <date>] [--skip-scoring]
|
|
80
|
+
agent-learn train --agent-id <agent_id> [--task-id <task_id>] [--limit <1-500>] [--min-episodes <1-500>] [--start-date <date>] [--end-date <date>] [--skip-scoring]
|
|
78
81
|
agent-learn task-policy --agent-id <agent_id> --task-id <task_id>
|
|
79
82
|
```
|
|
80
|
-
|
|
81
|
-
The subprocess-level functional workflow uses an isolated local store. Run the
|
|
82
|
-
interactive capture scenario first, followed by the offline batch update:
|
|
83
|
-
|
|
84
|
-
```powershell
|
|
85
|
-
python tests/functional_cli_interactive.py
|
|
86
|
-
python tests/functional_cli_batch.py
|
|
87
|
-
```
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "agent-learning"
|
|
7
|
-
version = "0.4.
|
|
7
|
+
version = "0.4.3"
|
|
8
8
|
description = "Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default."
|
|
9
9
|
readme = "PYPI.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -7,9 +7,9 @@ native, in-process learner. The SDK is organised into five layers:
|
|
|
7
7
|
``Reward``, ``PolicySnapshot``, ...).
|
|
8
8
|
- ``agent_learning.storage`` - pluggable persistence (Cosmos DB,
|
|
9
9
|
local file system, and in-memory).
|
|
10
|
-
- ``agent_learning.metrics`` -
|
|
11
|
-
|
|
12
|
-
|
|
10
|
+
- ``agent_learning.metrics`` - on-device stdlib metrics by default,
|
|
11
|
+
with optional Azure AI evaluators for Intent Resolution, Task
|
|
12
|
+
Adherence, and Task Completion.
|
|
13
13
|
- ``agent_learning.rewards`` - reward shaping + persistence.
|
|
14
14
|
- ``agent_learning.policy`` - discrete softmax bandit policy.
|
|
15
15
|
- ``agent_learning.learners`` - REINFORCE-with-baseline learner.
|
|
@@ -8,6 +8,7 @@ import logging
|
|
|
8
8
|
import sys
|
|
9
9
|
import uuid
|
|
10
10
|
from collections import Counter
|
|
11
|
+
from datetime import datetime, timezone
|
|
11
12
|
from typing import Any
|
|
12
13
|
|
|
13
14
|
from .policy.softmax_bandit import SoftmaxPolicy
|
|
@@ -27,6 +28,16 @@ def _episode_limit(value: str) -> int:
|
|
|
27
28
|
return limit
|
|
28
29
|
|
|
29
30
|
|
|
31
|
+
def _iso_date(value: str) -> str:
|
|
32
|
+
try:
|
|
33
|
+
parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
|
|
34
|
+
except ValueError as exc:
|
|
35
|
+
raise argparse.ArgumentTypeError(f"invalid ISO 8601 date: {value!r}") from exc
|
|
36
|
+
if parsed.tzinfo is None:
|
|
37
|
+
parsed = parsed.replace(tzinfo=timezone.utc)
|
|
38
|
+
return parsed.astimezone(timezone.utc).isoformat()
|
|
39
|
+
|
|
40
|
+
|
|
30
41
|
def _build_arg_parser() -> argparse.ArgumentParser:
|
|
31
42
|
parser = argparse.ArgumentParser(prog="agent-learn", description="Native RL CLI for AI agents.")
|
|
32
43
|
sub = parser.add_subparsers(dest="command", required=True)
|
|
@@ -42,6 +53,8 @@ def _build_arg_parser() -> argparse.ArgumentParser:
|
|
|
42
53
|
)
|
|
43
54
|
count.add_argument("agent_id")
|
|
44
55
|
count.add_argument("--task-id")
|
|
56
|
+
count.add_argument("--start-date", type=_iso_date)
|
|
57
|
+
count.add_argument("--end-date", type=_iso_date)
|
|
45
58
|
|
|
46
59
|
episodes = sub.add_parser(
|
|
47
60
|
"task-episodes-list",
|
|
@@ -51,13 +64,16 @@ def _build_arg_parser() -> argparse.ArgumentParser:
|
|
|
51
64
|
episodes.add_argument("--task-id")
|
|
52
65
|
episodes.add_argument("--limit", type=_episode_limit, default=_MAX_EPISODES)
|
|
53
66
|
episodes.add_argument("--include-incomplete", action="store_true")
|
|
67
|
+
episodes.add_argument("--start-date", type=_iso_date)
|
|
68
|
+
episodes.add_argument("--end-date", type=_iso_date)
|
|
54
69
|
|
|
55
70
|
train = sub.add_parser("train", help="Run one offline learning batch.")
|
|
56
71
|
train.add_argument("--agent-id", required=True)
|
|
57
72
|
train.add_argument("--task-id")
|
|
58
73
|
train.add_argument("--limit", type=_episode_limit, default=200)
|
|
59
|
-
train.add_argument("--
|
|
60
|
-
train.add_argument("--
|
|
74
|
+
train.add_argument("--min-episodes", type=_episode_limit, default=1)
|
|
75
|
+
train.add_argument("--start-date", type=_iso_date)
|
|
76
|
+
train.add_argument("--end-date", type=_iso_date)
|
|
61
77
|
train.add_argument(
|
|
62
78
|
"--skip-scoring",
|
|
63
79
|
action="store_true",
|
|
@@ -118,6 +134,8 @@ def _cmd_agents_episodes_count(args: argparse.Namespace) -> int:
|
|
|
118
134
|
args.agent_id,
|
|
119
135
|
task_id=args.task_id,
|
|
120
136
|
full_only=True,
|
|
137
|
+
start_date=args.start_date,
|
|
138
|
+
end_date=args.end_date,
|
|
121
139
|
)
|
|
122
140
|
print(count)
|
|
123
141
|
return 0
|
|
@@ -129,6 +147,8 @@ def _cmd_agents_episodes_list(args: argparse.Namespace) -> int:
|
|
|
129
147
|
args.agent_id,
|
|
130
148
|
task_id=args.task_id,
|
|
131
149
|
limit=_MAX_EPISODES,
|
|
150
|
+
start_date=args.start_date,
|
|
151
|
+
end_date=args.end_date,
|
|
132
152
|
)
|
|
133
153
|
if not args.include_incomplete:
|
|
134
154
|
episodes = [episode for episode in episodes if episode.is_full]
|
|
@@ -183,6 +203,17 @@ def _cmd_train(args: argparse.Namespace) -> int:
|
|
|
183
203
|
if episode_limit == 0:
|
|
184
204
|
skipped.append({"task_id": task_id, "reason": "no episodes in selected batch"})
|
|
185
205
|
continue
|
|
206
|
+
if episode_limit < args.min_episodes:
|
|
207
|
+
skipped.append(
|
|
208
|
+
{
|
|
209
|
+
"task_id": task_id,
|
|
210
|
+
"reason": (
|
|
211
|
+
f"selected batch has {episode_limit} episodes; "
|
|
212
|
+
f"minimum is {args.min_episodes}"
|
|
213
|
+
),
|
|
214
|
+
}
|
|
215
|
+
)
|
|
216
|
+
continue
|
|
186
217
|
policy = SoftmaxPolicy.from_snapshot(snapshot)
|
|
187
218
|
runner = LearningRunner(store=store, policy=policy)
|
|
188
219
|
run = runner.run_offline_batch(
|
|
@@ -216,11 +247,11 @@ def _cmd_score(args: argparse.Namespace) -> int:
|
|
|
216
247
|
)
|
|
217
248
|
scored = 0
|
|
218
249
|
for episode in episodes:
|
|
219
|
-
|
|
220
|
-
if existing:
|
|
250
|
+
if runner.has_usable_reward(episode):
|
|
221
251
|
continue
|
|
222
|
-
runner.score_and_record(episode)
|
|
223
|
-
|
|
252
|
+
rewards = runner.score_and_record(episode)
|
|
253
|
+
if any(reward.source == RewardSource.AGGREGATE for reward in rewards):
|
|
254
|
+
scored += 1
|
|
224
255
|
print(json.dumps({"episodes_seen": len(episodes), "newly_scored": scored}, indent=2))
|
|
225
256
|
return 0
|
|
226
257
|
|
|
@@ -49,14 +49,14 @@ class ReinforceLearner(Learner):
|
|
|
49
49
|
|
|
50
50
|
# Index rewards by episode id - we only consume aggregate rewards
|
|
51
51
|
episode_list = list(episodes)
|
|
52
|
-
aggregate_rewards: Dict[str,
|
|
52
|
+
aggregate_rewards: Dict[str, Reward] = {}
|
|
53
53
|
for r in rewards:
|
|
54
54
|
if r.source != RewardSource.AGGREGATE:
|
|
55
55
|
continue
|
|
56
|
-
#
|
|
56
|
+
# Rescoring may append a replacement aggregate. Keep the newest.
|
|
57
57
|
current = aggregate_rewards.get(r.episode_id)
|
|
58
|
-
if current is None:
|
|
59
|
-
aggregate_rewards[r.episode_id] = r
|
|
58
|
+
if current is None or r.created_at > current.created_at:
|
|
59
|
+
aggregate_rewards[r.episode_id] = r
|
|
60
60
|
|
|
61
61
|
snapshot = policy.snapshot()
|
|
62
62
|
baseline_before = snapshot.baseline
|
|
@@ -72,9 +72,10 @@ class ReinforceLearner(Learner):
|
|
|
72
72
|
for episode in episode_list:
|
|
73
73
|
if episode.action_id is None or episode.action_id not in action_index:
|
|
74
74
|
continue
|
|
75
|
-
|
|
76
|
-
if
|
|
75
|
+
reward = aggregate_rewards.get(episode.id)
|
|
76
|
+
if reward is None:
|
|
77
77
|
continue
|
|
78
|
+
reward_value = reward.value
|
|
78
79
|
advantage = reward_value - baseline_before
|
|
79
80
|
|
|
80
81
|
# Importance sampling weight when the episode was logged under a
|
|
@@ -1,18 +1,19 @@
|
|
|
1
1
|
"""Score-based evaluation metrics for native RL reward shaping.
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
into the ``[0, 1]`` range expected by the reward shaper.
|
|
3
|
+
Metrics use on-device stdlib scorers by default. Configured Azure AI
|
|
4
|
+
evaluators remain available for remote LLM scoring.
|
|
6
5
|
"""
|
|
7
6
|
|
|
8
7
|
from .base import MetricEvaluator, MetricRequest
|
|
9
8
|
from .intent_resolution import IntentResolutionMetric
|
|
9
|
+
from .local import LocalScorerMetric
|
|
10
10
|
from .task_adherence import TaskAdherenceMetric
|
|
11
11
|
from .task_completion import TaskCompletionMetric
|
|
12
12
|
from .registry import default_metrics, evaluate_all
|
|
13
13
|
|
|
14
14
|
__all__ = [
|
|
15
15
|
"IntentResolutionMetric",
|
|
16
|
+
"LocalScorerMetric",
|
|
16
17
|
"MetricEvaluator",
|
|
17
18
|
"MetricRequest",
|
|
18
19
|
"TaskAdherenceMetric",
|
|
@@ -42,6 +42,14 @@ class MetricRequest:
|
|
|
42
42
|
system_message=episode.system_message,
|
|
43
43
|
tool_calls=_format_tool_calls(episode),
|
|
44
44
|
tool_definitions=episode.metadata.get("tool_definitions"),
|
|
45
|
+
extra={
|
|
46
|
+
"action_id": episode.action_id,
|
|
47
|
+
"context_features": episode.context_features,
|
|
48
|
+
"execution_status": episode.execution_status,
|
|
49
|
+
"expected_outcome": episode.expected_outcome,
|
|
50
|
+
"metadata": episode.metadata,
|
|
51
|
+
"result_summary": episode.result_summary,
|
|
52
|
+
},
|
|
45
53
|
)
|
|
46
54
|
|
|
47
55
|
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
"""Metric adapters for on-device scoring backends."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from collections.abc import Sequence
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from ..scorers.base import Scorer, ScoreResult
|
|
10
|
+
from ..scorers.stdlib._text import tokenize
|
|
11
|
+
from ..types import MetricName, MetricResult
|
|
12
|
+
from .base import MetricEvaluator, MetricRequest
|
|
13
|
+
|
|
14
|
+
logger = logging.getLogger(__name__)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class LocalScorerMetric(MetricEvaluator):
|
|
18
|
+
"""Project a local :class:`Scorer` onto the metric interface."""
|
|
19
|
+
|
|
20
|
+
NAME = MetricName.INTENT_RESOLUTION
|
|
21
|
+
|
|
22
|
+
def __init__(self, metric: MetricName, scorer: Scorer) -> None:
|
|
23
|
+
super().__init__(evaluator=scorer)
|
|
24
|
+
self.NAME = metric
|
|
25
|
+
self._scorer = scorer
|
|
26
|
+
|
|
27
|
+
def _build_evaluator(self) -> Any: # pragma: no cover - injected in __init__
|
|
28
|
+
return self._scorer
|
|
29
|
+
|
|
30
|
+
def _build_kwargs(self, request: MetricRequest) -> dict[str, Any]:
|
|
31
|
+
metadata = request.extra.get("metadata")
|
|
32
|
+
if not isinstance(metadata, dict):
|
|
33
|
+
metadata = {}
|
|
34
|
+
context_features = request.extra.get("context_features")
|
|
35
|
+
if not isinstance(context_features, dict):
|
|
36
|
+
context_features = {}
|
|
37
|
+
contract = metadata.get("score_contract") or metadata.get("adherence_contract")
|
|
38
|
+
if not isinstance(contract, dict):
|
|
39
|
+
contract = {}
|
|
40
|
+
expected_tokens = metadata.get("expected_tokens")
|
|
41
|
+
if not isinstance(expected_tokens, Sequence) or isinstance(
|
|
42
|
+
expected_tokens, (str, bytes)
|
|
43
|
+
):
|
|
44
|
+
expected_tokens = tokenize(str(request.extra.get("expected_outcome") or ""))
|
|
45
|
+
return {
|
|
46
|
+
"query": request.query,
|
|
47
|
+
"response": request.response,
|
|
48
|
+
"system_message": request.system_message,
|
|
49
|
+
"tool_calls": request.tool_calls,
|
|
50
|
+
"action_id": request.extra.get("action_id"),
|
|
51
|
+
"phi": context_features.get("phi"),
|
|
52
|
+
"contract": contract,
|
|
53
|
+
"expected_tokens": expected_tokens,
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
def _normalize(self, raw: dict[str, Any]) -> float | None:
|
|
57
|
+
value = raw.get("normalized")
|
|
58
|
+
return float(value) if value is not None else None
|
|
59
|
+
|
|
60
|
+
def evaluate(self, request: MetricRequest) -> MetricResult:
|
|
61
|
+
if not (request.query and request.response):
|
|
62
|
+
return MetricResult(
|
|
63
|
+
metric=self.NAME,
|
|
64
|
+
score=None,
|
|
65
|
+
normalized=None,
|
|
66
|
+
status="skipped",
|
|
67
|
+
reason="query or response is empty",
|
|
68
|
+
evaluator=f"local:{self._scorer.name}",
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
authoritative = self._authoritative_completion(request)
|
|
72
|
+
if authoritative is not None:
|
|
73
|
+
value, reason = authoritative
|
|
74
|
+
return MetricResult(
|
|
75
|
+
metric=self.NAME,
|
|
76
|
+
score=value,
|
|
77
|
+
normalized=value,
|
|
78
|
+
status="completed",
|
|
79
|
+
reason=reason,
|
|
80
|
+
evaluator="local:episode-outcome",
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
try:
|
|
84
|
+
result = self._scorer.score(**self._build_kwargs(request))
|
|
85
|
+
except Exception as exc: # noqa: BLE001 # pragma: no cover
|
|
86
|
+
logger.warning("Local metric %s failed: %s", self.NAME.value, exc)
|
|
87
|
+
return MetricResult(
|
|
88
|
+
metric=self.NAME,
|
|
89
|
+
score=None,
|
|
90
|
+
normalized=None,
|
|
91
|
+
status="skipped",
|
|
92
|
+
reason=f"local scorer error: {exc}",
|
|
93
|
+
evaluator=f"local:{self._scorer.name}",
|
|
94
|
+
)
|
|
95
|
+
return _to_metric_result(self.NAME, self._scorer.name, result)
|
|
96
|
+
|
|
97
|
+
def _authoritative_completion(
|
|
98
|
+
self, request: MetricRequest
|
|
99
|
+
) -> tuple[float, str] | None:
|
|
100
|
+
if self.NAME != MetricName.TASK_COMPLETION:
|
|
101
|
+
return None
|
|
102
|
+
metadata = request.extra.get("metadata")
|
|
103
|
+
if not isinstance(metadata, dict):
|
|
104
|
+
metadata = {}
|
|
105
|
+
action_id = request.extra.get("action_id")
|
|
106
|
+
correct_action_id = metadata.get("correct_action_id")
|
|
107
|
+
if action_id and correct_action_id:
|
|
108
|
+
return (
|
|
109
|
+
float(action_id == correct_action_id),
|
|
110
|
+
"derived from action_id and metadata.correct_action_id",
|
|
111
|
+
)
|
|
112
|
+
completed = metadata.get("task_completed")
|
|
113
|
+
if isinstance(completed, bool):
|
|
114
|
+
return float(completed), "derived from metadata.task_completed"
|
|
115
|
+
status = str(request.extra.get("execution_status") or "").lower()
|
|
116
|
+
if status == "completed":
|
|
117
|
+
return 1.0, "derived from execution_status=completed"
|
|
118
|
+
if status == "failed":
|
|
119
|
+
return 0.0, "derived from execution_status=failed"
|
|
120
|
+
if status == "partial":
|
|
121
|
+
return 0.5, "derived from execution_status=partial"
|
|
122
|
+
return None
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _to_metric_result(
|
|
126
|
+
metric: MetricName, scorer_name: str, result: ScoreResult
|
|
127
|
+
) -> MetricResult:
|
|
128
|
+
normalized = max(0.0, min(1.0, float(result.normalized)))
|
|
129
|
+
return MetricResult(
|
|
130
|
+
metric=metric,
|
|
131
|
+
score=normalized,
|
|
132
|
+
normalized=normalized,
|
|
133
|
+
status="completed",
|
|
134
|
+
reason=f"local {result.label}",
|
|
135
|
+
properties=dict(result.features),
|
|
136
|
+
evaluator=f"local:{scorer_name}",
|
|
137
|
+
metadata={"label": result.label, "confidence": result.confidence},
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def local_metrics(scorers: tuple[Scorer, Scorer, Scorer]) -> list[MetricEvaluator]:
|
|
142
|
+
"""Return local metric adapters in reward-shaping order."""
|
|
143
|
+
intent, adherence, completion = scorers
|
|
144
|
+
return [
|
|
145
|
+
LocalScorerMetric(MetricName.INTENT_RESOLUTION, intent),
|
|
146
|
+
LocalScorerMetric(MetricName.TASK_ADHERENCE, adherence),
|
|
147
|
+
LocalScorerMetric(MetricName.TASK_COMPLETION, completion),
|
|
148
|
+
]
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
__all__ = ["LocalScorerMetric", "local_metrics"]
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""Convenience helpers to evaluate an episode across all default metrics."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Iterable, List, Optional
|
|
6
|
+
|
|
7
|
+
from ..config import ScoreConfig, ScoreRuntimeConfig
|
|
8
|
+
from ..scorers import build_scorers
|
|
9
|
+
from ..types import Episode, MetricResult
|
|
10
|
+
from .base import MetricEvaluator, MetricRequest
|
|
11
|
+
from .intent_resolution import IntentResolutionMetric
|
|
12
|
+
from .local import local_metrics
|
|
13
|
+
from .task_adherence import TaskAdherenceMetric
|
|
14
|
+
from .task_completion import TaskCompletionMetric
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def default_metrics(
|
|
18
|
+
score_config: Optional[ScoreConfig] = None,
|
|
19
|
+
score_runtime_config: Optional[ScoreRuntimeConfig] = None,
|
|
20
|
+
) -> List[MetricEvaluator]:
|
|
21
|
+
"""Return local metrics by default, or Azure metrics when configured."""
|
|
22
|
+
runtime = score_runtime_config or ScoreRuntimeConfig()
|
|
23
|
+
llm = score_config or runtime.llm
|
|
24
|
+
if score_config is not None or runtime.tier == "llm" or (
|
|
25
|
+
runtime.tier is None and llm.enabled
|
|
26
|
+
):
|
|
27
|
+
return [
|
|
28
|
+
IntentResolutionMetric(llm),
|
|
29
|
+
TaskAdherenceMetric(llm),
|
|
30
|
+
TaskCompletionMetric(llm),
|
|
31
|
+
]
|
|
32
|
+
if runtime.tier is None:
|
|
33
|
+
runtime.tier = "stdlib"
|
|
34
|
+
return local_metrics(build_scorers(runtime))
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def evaluate_all(
|
|
38
|
+
episode: Episode,
|
|
39
|
+
metrics: Optional[Iterable[MetricEvaluator]] = None,
|
|
40
|
+
*,
|
|
41
|
+
score_config: Optional[ScoreConfig] = None,
|
|
42
|
+
score_runtime_config: Optional[ScoreRuntimeConfig] = None,
|
|
43
|
+
) -> List[MetricResult]:
|
|
44
|
+
"""Evaluate one episode against every supplied metric.
|
|
45
|
+
|
|
46
|
+
Each metric is given its own try/except inside :meth:`evaluate`,
|
|
47
|
+
so a single failing scorer will not prevent the others from
|
|
48
|
+
producing scores.
|
|
49
|
+
"""
|
|
50
|
+
metric_list = (
|
|
51
|
+
list(metrics)
|
|
52
|
+
if metrics is not None
|
|
53
|
+
else default_metrics(score_config, score_runtime_config)
|
|
54
|
+
)
|
|
55
|
+
request = MetricRequest.from_episode(episode)
|
|
56
|
+
return [m.evaluate(request) for m in metric_list]
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
__all__ = ["default_metrics", "evaluate_all"]
|
|
@@ -98,7 +98,10 @@ class RewardWriter:
|
|
|
98
98
|
except Exception as exc: # pragma: no cover
|
|
99
99
|
logger.warning("Failed to persist %s penalty for %s: %s", kind, episode.id, exc)
|
|
100
100
|
|
|
101
|
-
# 4) Aggregate reward consumed by the learner
|
|
101
|
+
# 4) Aggregate reward consumed by the learner. Do not turn an
|
|
102
|
+
# all-skipped evaluation into a misleading neutral reward.
|
|
103
|
+
if not shaped.metric_contributions and not shaped.penalties:
|
|
104
|
+
return stored
|
|
102
105
|
aggregate = Reward(
|
|
103
106
|
episode_id=episode.id,
|
|
104
107
|
agent_id=episode.agent_id,
|
|
@@ -66,8 +66,10 @@ class LearningStore(ABC):
|
|
|
66
66
|
*,
|
|
67
67
|
task_id: Optional[str] = None,
|
|
68
68
|
full_only: bool = False,
|
|
69
|
+
start_date: Optional[str] = None,
|
|
70
|
+
end_date: Optional[str] = None,
|
|
69
71
|
) -> int:
|
|
70
|
-
"""Count episodes
|
|
72
|
+
"""Count episodes filtered by task, completeness, or time window."""
|
|
71
73
|
|
|
72
74
|
# ---- Metric results -------------------------------------------
|
|
73
75
|
|
|
@@ -272,6 +272,8 @@ class CosmosStore(LearningStore):
|
|
|
272
272
|
*,
|
|
273
273
|
task_id: Optional[str] = None,
|
|
274
274
|
full_only: bool = False,
|
|
275
|
+
start_date: Optional[str] = None,
|
|
276
|
+
end_date: Optional[str] = None,
|
|
275
277
|
) -> int:
|
|
276
278
|
clauses = ["c.agent_id = @agent_id"]
|
|
277
279
|
params: List[Dict[str, Any]] = [{"name": "@agent_id", "value": agent_id}]
|
|
@@ -281,6 +283,12 @@ class CosmosStore(LearningStore):
|
|
|
281
283
|
else:
|
|
282
284
|
clauses.append("c.task_id = @task_id")
|
|
283
285
|
params.append({"name": "@task_id", "value": task_id})
|
|
286
|
+
if start_date:
|
|
287
|
+
clauses.append("c.created_at >= @start_date")
|
|
288
|
+
params.append({"name": "@start_date", "value": start_date})
|
|
289
|
+
if end_date:
|
|
290
|
+
clauses.append("c.created_at <= @end_date")
|
|
291
|
+
params.append({"name": "@end_date", "value": end_date})
|
|
284
292
|
if full_only:
|
|
285
293
|
clauses.extend(
|
|
286
294
|
[
|
|
@@ -212,12 +212,16 @@ class LocalFileStore(LearningStore):
|
|
|
212
212
|
*,
|
|
213
213
|
task_id: Optional[str] = None,
|
|
214
214
|
full_only: bool = False,
|
|
215
|
+
start_date: Optional[str] = None,
|
|
216
|
+
end_date: Optional[str] = None,
|
|
215
217
|
) -> int:
|
|
216
218
|
return sum(
|
|
217
219
|
1
|
|
218
220
|
for doc in self._read_dir_docs("episodes", agent_id)
|
|
219
221
|
if (task_id is None or doc.get("task_id", "default") == task_id)
|
|
220
222
|
and (not full_only or Episode.from_dict(doc).is_full)
|
|
223
|
+
and (start_date is None or doc.get("created_at", "") >= start_date)
|
|
224
|
+
and (end_date is None or doc.get("created_at", "") <= end_date)
|
|
221
225
|
)
|
|
222
226
|
|
|
223
227
|
# ------------------------------------------------------------------
|
|
@@ -93,6 +93,8 @@ class InMemoryStore(LearningStore):
|
|
|
93
93
|
*,
|
|
94
94
|
task_id: Optional[str] = None,
|
|
95
95
|
full_only: bool = False,
|
|
96
|
+
start_date: Optional[str] = None,
|
|
97
|
+
end_date: Optional[str] = None,
|
|
96
98
|
) -> int:
|
|
97
99
|
return sum(
|
|
98
100
|
1
|
|
@@ -100,6 +102,8 @@ class InMemoryStore(LearningStore):
|
|
|
100
102
|
if episode.agent_id == agent_id
|
|
101
103
|
and (task_id is None or episode.task_id == task_id)
|
|
102
104
|
and (not full_only or episode.is_full)
|
|
105
|
+
and (start_date is None or episode.created_at >= start_date)
|
|
106
|
+
and (end_date is None or episode.created_at <= end_date)
|
|
103
107
|
)
|
|
104
108
|
|
|
105
109
|
# ---- Metric results -------------------------------------------
|