agent-learning 0.4.1__tar.gz → 0.4.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. {agent_learning-0.4.1/src/agent_learning.egg-info → agent_learning-0.4.3}/PKG-INFO +5 -5
  2. agent_learning-0.4.3/PYPI.md +15 -0
  3. {agent_learning-0.4.1 → agent_learning-0.4.3}/README.md +20 -25
  4. {agent_learning-0.4.1 → agent_learning-0.4.3}/pyproject.toml +1 -1
  5. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/__init__.py +3 -3
  6. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/_version.py +1 -1
  7. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/cli.py +37 -6
  8. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/learners/reinforce.py +7 -6
  9. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/metrics/__init__.py +4 -3
  10. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/metrics/base.py +8 -0
  11. agent_learning-0.4.3/src/agent_learning/metrics/local.py +151 -0
  12. agent_learning-0.4.3/src/agent_learning/metrics/registry.py +59 -0
  13. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/rewards/writer.py +4 -1
  14. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/storage/base.py +3 -1
  15. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/storage/cosmos.py +8 -0
  16. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/storage/local.py +4 -0
  17. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/storage/memory.py +4 -0
  18. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/training/runner.py +21 -3
  19. {agent_learning-0.4.1 → agent_learning-0.4.3/src/agent_learning.egg-info}/PKG-INFO +5 -5
  20. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning.egg-info/SOURCES.txt +2 -0
  21. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_cli.py +180 -1
  22. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_learner.py +27 -0
  23. agent_learning-0.4.3/tests/test_metrics_local.py +60 -0
  24. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_shaping.py +23 -1
  25. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_storage_local.py +3 -0
  26. agent_learning-0.4.1/PYPI.md +0 -15
  27. agent_learning-0.4.1/src/agent_learning/metrics/registry.py +0 -42
  28. {agent_learning-0.4.1 → agent_learning-0.4.3}/LICENSE +0 -0
  29. {agent_learning-0.4.1 → agent_learning-0.4.3}/setup.cfg +0 -0
  30. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/capture.py +0 -0
  31. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/__init__.py +0 -0
  32. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/base.py +0 -0
  33. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/router.py +0 -0
  34. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/scorers/__init__.py +0 -0
  35. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/scorers/_base.py +0 -0
  36. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/scorers/adherence.py +0 -0
  37. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/scorers/completion.py +0 -0
  38. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/classifiers/scorers/intent.py +0 -0
  39. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/config.py +0 -0
  40. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/learners/__init__.py +0 -0
  41. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/learners/base.py +0 -0
  42. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/metrics/intent_resolution.py +0 -0
  43. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/metrics/task_adherence.py +0 -0
  44. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/metrics/task_completion.py +0 -0
  45. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/policy/__init__.py +0 -0
  46. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/policy/base.py +0 -0
  47. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/policy/contextual_softmax.py +0 -0
  48. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/policy/softmax_bandit.py +0 -0
  49. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/py.typed +0 -0
  50. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/rewards/__init__.py +0 -0
  51. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/rewards/shaping.py +0 -0
  52. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/__init__.py +0 -0
  53. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/base.py +0 -0
  54. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/llm/__init__.py +0 -0
  55. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/llm/_base.py +0 -0
  56. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/llm/adherence.py +0 -0
  57. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/llm/completion.py +0 -0
  58. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/llm/intent.py +0 -0
  59. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp/__init__.py +0 -0
  60. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp/_base.py +0 -0
  61. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp/adherence.py +0 -0
  62. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp/completion.py +0 -0
  63. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp/intent.py +0 -0
  64. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp_text/__init__.py +0 -0
  65. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp_text/_base.py +0 -0
  66. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp_text/adherence.py +0 -0
  67. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp_text/completion.py +0 -0
  68. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/nlp_text/intent.py +0 -0
  69. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/slm/__init__.py +0 -0
  70. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/slm/_base.py +0 -0
  71. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/slm/adherence.py +0 -0
  72. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/slm/completion.py +0 -0
  73. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/slm/intent.py +0 -0
  74. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/stdlib/__init__.py +0 -0
  75. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/stdlib/_text.py +0 -0
  76. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/stdlib/adherence.py +0 -0
  77. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/stdlib/completion.py +0 -0
  78. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/scorers/stdlib/intent.py +0 -0
  79. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/storage/__init__.py +0 -0
  80. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/training/__init__.py +0 -0
  81. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning/types.py +0 -0
  82. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning.egg-info/dependency_links.txt +0 -0
  83. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning.egg-info/entry_points.txt +0 -0
  84. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning.egg-info/requires.txt +0 -0
  85. {agent_learning-0.4.1 → agent_learning-0.4.3}/src/agent_learning.egg-info/top_level.txt +0 -0
  86. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_capture.py +0 -0
  87. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_contextual_policy.py +0 -0
  88. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_default_store.py +0 -0
  89. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_end_to_end.py +0 -0
  90. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_policy.py +0 -0
  91. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_scorers.py +0 -0
  92. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_scorers_llm.py +0 -0
  93. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_scorers_nlp_text.py +0 -0
  94. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_scorers_slm.py +0 -0
  95. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_scorers_stdlib.py +0 -0
  96. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_storage_memory.py +0 -0
  97. {agent_learning-0.4.1 → agent_learning-0.4.3}/tests/test_types.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agent-learning
3
- Version: 0.4.1
3
+ Version: 0.4.3
4
4
  Summary: Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default.
5
5
  Author: Chris Tava
6
6
  License: MIT License
@@ -67,16 +67,16 @@ Dynamic: license-file
67
67
 
68
68
  # agent-learning
69
69
 
70
- Native reinforcement learning SDK for AI agents. An in-process learner optimizes a small, interpretable policy over discrete agent choices (understand intent and complete task by choosing the right outcome).
70
+ Native reinforcement learning SDK for AI agents. An in-process Learner optimizes a small, interpretable TaskPolicy over discrete agent choices (understand intent and complete task by choosing the right outcome).
71
71
 
72
72
  ## How it works
73
73
 
74
74
  The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tune jobs and no opaque update cycles — just three pieces that run in your existing Python process:
75
75
 
76
- 1. The **policy** is a softmax distribution over `N` discrete actions (e.g., "take action A", "take action B", "take action C"). It lives in Python and updates in milliseconds.
76
+ 1. **TaskPolicy** is a softmax distribution over `N` discrete actions (e.g., "take action A", "take action B", "take action C"). It lives in Python and updates in milliseconds.
77
77
 
78
- 2. Each episode is **evaluated** by three AI Evaluation evaluators — `IntentResolutionEvaluator`, `TaskAdherenceEvaluator`, and `TaskCompletionEvaluator` — whose scores are combined into a single scalar reward.
78
+ 2. **Score** evaluates each episode on-device with three stdlib scorers for intent resolution, task adherence, and task completion. Their scores are combined into a single scalar reward with no scoring endpoint or environment variables required. Azure AI evaluators remain available as an opt-in.
79
79
 
80
- 3. A **REINFORCE-with-baseline** learner updates the policy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
80
+ 3. **Learner** applies REINFORCE-with-baseline to update TaskPolicy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
81
81
 
82
82
  Every episode, reward, run, and deployment is captured by the configured store — in-memory or local files by default, or Azure Cosmos DB — giving you a complete lineage and audit trail of how the policy evolved over time.
@@ -0,0 +1,15 @@
1
+ # agent-learning
2
+
3
+ Native reinforcement learning SDK for AI agents. An in-process Learner optimizes a small, interpretable TaskPolicy over discrete agent choices (understand intent and complete task by choosing the right outcome).
4
+
5
+ ## How it works
6
+
7
+ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tune jobs and no opaque update cycles — just three pieces that run in your existing Python process:
8
+
9
+ 1. **TaskPolicy** is a softmax distribution over `N` discrete actions (e.g., "take action A", "take action B", "take action C"). It lives in Python and updates in milliseconds.
10
+
11
+ 2. **Score** evaluates each episode on-device with three stdlib scorers for intent resolution, task adherence, and task completion. Their scores are combined into a single scalar reward with no scoring endpoint or environment variables required. Azure AI evaluators remain available as an opt-in.
12
+
13
+ 3. **Learner** applies REINFORCE-with-baseline to update TaskPolicy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
14
+
15
+ Every episode, reward, run, and deployment is captured by the configured store — in-memory or local files by default, or Azure Cosmos DB — giving you a complete lineage and audit trail of how the policy evolved over time.
@@ -1,29 +1,30 @@
1
1
  # agent-learning
2
2
 
3
3
  Native reinforcement learning SDK for AI agents. An in-process
4
- learner optimizes a small, interpretable policy over discrete agent choices (e.g., "take action A", "take action B", "take action C") using AI Evaluation scores as the reward
5
- signal.
4
+ Learner optimizes a small, interpretable TaskPolicy over discrete agent choices (e.g., "take action A", "take action B", "take action C") using on-device evaluation scores as the reward
5
+ signal by default.
6
6
 
7
7
  <p align="center">
8
- <img src="images/agent-learning-loop.svg" alt="Animated loop: Policy chooses an action, Score evaluates the episode, and Learner updates the policy" width="960" style="max-width:100%; height:auto;" />
8
+ <img src="images/agent-learning-loop.svg" alt="Animated TaskPolicy Score Learner loop: TaskPolicy chooses a task action, Score evaluates the episode, and Learner updates TaskPolicy" width="960" style="max-width:100%; height:auto;" />
9
9
  </p>
10
10
 
11
11
  ## How it works
12
12
 
13
13
  The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tune jobs and no opaque update cycles — just three pieces that run in your existing Python process:
14
14
 
15
- 1. The **policy** is a softmax distribution over `N` discrete
15
+ 1. **TaskPolicy** is a softmax distribution over `N` discrete
16
16
  actions (e.g., "take action A", "take action B", "take action C"). It lives in Python and updates in milliseconds.
17
17
 
18
- <img src="images/0f85e08d0c47cd01.png" alt="Policy selects one of N discrete actions" width="360" style="max-width:100%; height:auto;" />
18
+ <img src="images/0f85e08d0c47cd01.png" alt="TaskPolicy selects one of N discrete actions" width="360" style="max-width:100%; height:auto;" />
19
19
 
20
- 2. Each episode is **evaluated** by three AI Evaluation
21
- evaluators — `IntentResolutionEvaluator`, `TaskAdherenceEvaluator`,
22
- and `TaskCompletionEvaluator` — whose scores are combined into a single scalar reward.
20
+ 2. **Score** evaluates each episode locally with three stdlib scorers for intent
21
+ resolution, task adherence, and task completion. Their scores are combined
22
+ into one scalar reward. No scoring endpoint or environment variable is
23
+ required. Configured Azure AI evaluators remain available as an opt-in.
23
24
 
24
25
  <img src="images/246d112f995b785a.png" alt="Three evaluator scores feed a single scalar reward" width="360" style="max-width:100%; height:auto;" />
25
26
 
26
- 3. A **Reinforce-with-baseline** learner updates the policy logits
27
+ 3. **Learner** applies REINFORCE-with-baseline to update TaskPolicy logits
27
28
  directly from stored episodes. Updates are tiny gradient steps
28
29
  that run on local compute and persist through a pluggable store — in-memory
29
30
  or local files by default, with Azure Cosmos DB optional.
@@ -54,14 +55,16 @@ agent-learn.exe --help
54
55
  Released versions are published to PyPI:
55
56
  <https://pypi.org/project/agent-learning/>.
56
57
 
58
+ ## Functional Testing
59
+
60
+ A good way to see the SDK in action is to run the
61
+ interactive capture scenario first, followed by the offline batch update:
62
+
57
63
  ```powershell
58
- py -m pip install agent-learning
59
- agent-learn.exe --help
64
+ python tests/functional_cli_interactive.py
65
+ python tests/functional_cli_batch.py
60
66
  ```
61
67
 
62
- `pip` installs `agent-learn.exe` into the active Python environment's
63
- `Scripts` directory.
64
-
65
68
  ## Usage
66
69
 
67
70
  The `agent-learn` CLI provides the current task-learning-loop operations:
@@ -69,19 +72,11 @@ The `agent-learn` CLI provides the current task-learning-loop operations:
69
72
  ```text
70
73
  agent-learn list
71
74
  agent-learn tasks-list <agent_id>
72
- agent-learn task-episodes-count <agent_id> [--task-id <task_id>]
73
- agent-learn task-episodes-list <agent_id> [--task-id <task_id>] [--limit <1-500>] [--include-incomplete]
75
+ agent-learn task-episodes-count <agent_id> [--task-id <task_id>] [--start-date <date>] [--end-date <date>]
76
+ agent-learn task-episodes-list <agent_id> [--task-id <task_id>] [--limit <1-500>] [--include-incomplete] [--start-date <date>] [--end-date <date>]
74
77
  agent-learn task-policy-init --agent-id <agent_id> --task-id <task_id> --actions ./actions.json
75
78
  agent-learn task-episode-register --agent-id <agent_id> --task-id <task_id> --episode ./episode.json
76
79
  agent-learn score --agent-id <agent_id> [--task-id <task_id>] [--limit <1-500>]
77
- agent-learn train --agent-id <agent_id> [--task-id <task_id>] [--limit <1-500>] [--start-date <date>] [--end-date <date>] [--skip-scoring]
80
+ agent-learn train --agent-id <agent_id> [--task-id <task_id>] [--limit <1-500>] [--min-episodes <1-500>] [--start-date <date>] [--end-date <date>] [--skip-scoring]
78
81
  agent-learn task-policy --agent-id <agent_id> --task-id <task_id>
79
82
  ```
80
-
81
- The subprocess-level functional workflow uses an isolated local store. Run the
82
- interactive capture scenario first, followed by the offline batch update:
83
-
84
- ```powershell
85
- python tests/functional_cli_interactive.py
86
- python tests/functional_cli_batch.py
87
- ```
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "agent-learning"
7
- version = "0.4.1"
7
+ version = "0.4.3"
8
8
  description = "Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default."
9
9
  readme = "PYPI.md"
10
10
  requires-python = ">=3.10"
@@ -7,9 +7,9 @@ native, in-process learner. The SDK is organised into five layers:
7
7
  ``Reward``, ``PolicySnapshot``, ...).
8
8
  - ``agent_learning.storage`` - pluggable persistence (Cosmos DB,
9
9
  local file system, and in-memory).
10
- - ``agent_learning.metrics`` - score-based metrics that wrap the
11
- Azure AI Evaluation evaluators for Intent Resolution, Task
12
- Adherence, and Task Completion.
10
+ - ``agent_learning.metrics`` - on-device stdlib metrics by default,
11
+ with optional Azure AI evaluators for Intent Resolution, Task
12
+ Adherence, and Task Completion.
13
13
  - ``agent_learning.rewards`` - reward shaping + persistence.
14
14
  - ``agent_learning.policy`` - discrete softmax bandit policy.
15
15
  - ``agent_learning.learners`` - REINFORCE-with-baseline learner.
@@ -1,3 +1,3 @@
1
1
  """Package version."""
2
2
 
3
- __version__ = "0.4.1"
3
+ __version__ = "0.4.3"
@@ -8,6 +8,7 @@ import logging
8
8
  import sys
9
9
  import uuid
10
10
  from collections import Counter
11
+ from datetime import datetime, timezone
11
12
  from typing import Any
12
13
 
13
14
  from .policy.softmax_bandit import SoftmaxPolicy
@@ -27,6 +28,16 @@ def _episode_limit(value: str) -> int:
27
28
  return limit
28
29
 
29
30
 
31
+ def _iso_date(value: str) -> str:
32
+ try:
33
+ parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
34
+ except ValueError as exc:
35
+ raise argparse.ArgumentTypeError(f"invalid ISO 8601 date: {value!r}") from exc
36
+ if parsed.tzinfo is None:
37
+ parsed = parsed.replace(tzinfo=timezone.utc)
38
+ return parsed.astimezone(timezone.utc).isoformat()
39
+
40
+
30
41
  def _build_arg_parser() -> argparse.ArgumentParser:
31
42
  parser = argparse.ArgumentParser(prog="agent-learn", description="Native RL CLI for AI agents.")
32
43
  sub = parser.add_subparsers(dest="command", required=True)
@@ -42,6 +53,8 @@ def _build_arg_parser() -> argparse.ArgumentParser:
42
53
  )
43
54
  count.add_argument("agent_id")
44
55
  count.add_argument("--task-id")
56
+ count.add_argument("--start-date", type=_iso_date)
57
+ count.add_argument("--end-date", type=_iso_date)
45
58
 
46
59
  episodes = sub.add_parser(
47
60
  "task-episodes-list",
@@ -51,13 +64,16 @@ def _build_arg_parser() -> argparse.ArgumentParser:
51
64
  episodes.add_argument("--task-id")
52
65
  episodes.add_argument("--limit", type=_episode_limit, default=_MAX_EPISODES)
53
66
  episodes.add_argument("--include-incomplete", action="store_true")
67
+ episodes.add_argument("--start-date", type=_iso_date)
68
+ episodes.add_argument("--end-date", type=_iso_date)
54
69
 
55
70
  train = sub.add_parser("train", help="Run one offline learning batch.")
56
71
  train.add_argument("--agent-id", required=True)
57
72
  train.add_argument("--task-id")
58
73
  train.add_argument("--limit", type=_episode_limit, default=200)
59
- train.add_argument("--start-date")
60
- train.add_argument("--end-date")
74
+ train.add_argument("--min-episodes", type=_episode_limit, default=1)
75
+ train.add_argument("--start-date", type=_iso_date)
76
+ train.add_argument("--end-date", type=_iso_date)
61
77
  train.add_argument(
62
78
  "--skip-scoring",
63
79
  action="store_true",
@@ -118,6 +134,8 @@ def _cmd_agents_episodes_count(args: argparse.Namespace) -> int:
118
134
  args.agent_id,
119
135
  task_id=args.task_id,
120
136
  full_only=True,
137
+ start_date=args.start_date,
138
+ end_date=args.end_date,
121
139
  )
122
140
  print(count)
123
141
  return 0
@@ -129,6 +147,8 @@ def _cmd_agents_episodes_list(args: argparse.Namespace) -> int:
129
147
  args.agent_id,
130
148
  task_id=args.task_id,
131
149
  limit=_MAX_EPISODES,
150
+ start_date=args.start_date,
151
+ end_date=args.end_date,
132
152
  )
133
153
  if not args.include_incomplete:
134
154
  episodes = [episode for episode in episodes if episode.is_full]
@@ -183,6 +203,17 @@ def _cmd_train(args: argparse.Namespace) -> int:
183
203
  if episode_limit == 0:
184
204
  skipped.append({"task_id": task_id, "reason": "no episodes in selected batch"})
185
205
  continue
206
+ if episode_limit < args.min_episodes:
207
+ skipped.append(
208
+ {
209
+ "task_id": task_id,
210
+ "reason": (
211
+ f"selected batch has {episode_limit} episodes; "
212
+ f"minimum is {args.min_episodes}"
213
+ ),
214
+ }
215
+ )
216
+ continue
186
217
  policy = SoftmaxPolicy.from_snapshot(snapshot)
187
218
  runner = LearningRunner(store=store, policy=policy)
188
219
  run = runner.run_offline_batch(
@@ -216,11 +247,11 @@ def _cmd_score(args: argparse.Namespace) -> int:
216
247
  )
217
248
  scored = 0
218
249
  for episode in episodes:
219
- existing = store.get_rewards_for_episode(episode.id, args.agent_id)
220
- if existing:
250
+ if runner.has_usable_reward(episode):
221
251
  continue
222
- runner.score_and_record(episode)
223
- scored += 1
252
+ rewards = runner.score_and_record(episode)
253
+ if any(reward.source == RewardSource.AGGREGATE for reward in rewards):
254
+ scored += 1
224
255
  print(json.dumps({"episodes_seen": len(episodes), "newly_scored": scored}, indent=2))
225
256
  return 0
226
257
 
@@ -49,14 +49,14 @@ class ReinforceLearner(Learner):
49
49
 
50
50
  # Index rewards by episode id - we only consume aggregate rewards
51
51
  episode_list = list(episodes)
52
- aggregate_rewards: Dict[str, float] = {}
52
+ aggregate_rewards: Dict[str, Reward] = {}
53
53
  for r in rewards:
54
54
  if r.source != RewardSource.AGGREGATE:
55
55
  continue
56
- # Keep the most recent aggregate per episode (sorted by created_at desc)
56
+ # Rescoring may append a replacement aggregate. Keep the newest.
57
57
  current = aggregate_rewards.get(r.episode_id)
58
- if current is None:
59
- aggregate_rewards[r.episode_id] = r.value
58
+ if current is None or r.created_at > current.created_at:
59
+ aggregate_rewards[r.episode_id] = r
60
60
 
61
61
  snapshot = policy.snapshot()
62
62
  baseline_before = snapshot.baseline
@@ -72,9 +72,10 @@ class ReinforceLearner(Learner):
72
72
  for episode in episode_list:
73
73
  if episode.action_id is None or episode.action_id not in action_index:
74
74
  continue
75
- reward_value = aggregate_rewards.get(episode.id)
76
- if reward_value is None:
75
+ reward = aggregate_rewards.get(episode.id)
76
+ if reward is None:
77
77
  continue
78
+ reward_value = reward.value
78
79
  advantage = reward_value - baseline_before
79
80
 
80
81
  # Importance sampling weight when the episode was logged under a
@@ -1,18 +1,19 @@
1
1
  """Score-based evaluation metrics for native RL reward shaping.
2
2
 
3
- Each metric is a thin wrapper around an evaluator from
4
- ``azure-ai-evaluation``. The wrapper normalises the raw evaluator score
5
- into the ``[0, 1]`` range expected by the reward shaper.
3
+ Metrics use on-device stdlib scorers by default. Configured Azure AI
4
+ evaluators remain available for remote LLM scoring.
6
5
  """
7
6
 
8
7
  from .base import MetricEvaluator, MetricRequest
9
8
  from .intent_resolution import IntentResolutionMetric
9
+ from .local import LocalScorerMetric
10
10
  from .task_adherence import TaskAdherenceMetric
11
11
  from .task_completion import TaskCompletionMetric
12
12
  from .registry import default_metrics, evaluate_all
13
13
 
14
14
  __all__ = [
15
15
  "IntentResolutionMetric",
16
+ "LocalScorerMetric",
16
17
  "MetricEvaluator",
17
18
  "MetricRequest",
18
19
  "TaskAdherenceMetric",
@@ -42,6 +42,14 @@ class MetricRequest:
42
42
  system_message=episode.system_message,
43
43
  tool_calls=_format_tool_calls(episode),
44
44
  tool_definitions=episode.metadata.get("tool_definitions"),
45
+ extra={
46
+ "action_id": episode.action_id,
47
+ "context_features": episode.context_features,
48
+ "execution_status": episode.execution_status,
49
+ "expected_outcome": episode.expected_outcome,
50
+ "metadata": episode.metadata,
51
+ "result_summary": episode.result_summary,
52
+ },
45
53
  )
46
54
 
47
55
 
@@ -0,0 +1,151 @@
1
+ """Metric adapters for on-device scoring backends."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from collections.abc import Sequence
7
+ from typing import Any
8
+
9
+ from ..scorers.base import Scorer, ScoreResult
10
+ from ..scorers.stdlib._text import tokenize
11
+ from ..types import MetricName, MetricResult
12
+ from .base import MetricEvaluator, MetricRequest
13
+
14
+ logger = logging.getLogger(__name__)
15
+
16
+
17
+ class LocalScorerMetric(MetricEvaluator):
18
+ """Project a local :class:`Scorer` onto the metric interface."""
19
+
20
+ NAME = MetricName.INTENT_RESOLUTION
21
+
22
+ def __init__(self, metric: MetricName, scorer: Scorer) -> None:
23
+ super().__init__(evaluator=scorer)
24
+ self.NAME = metric
25
+ self._scorer = scorer
26
+
27
+ def _build_evaluator(self) -> Any: # pragma: no cover - injected in __init__
28
+ return self._scorer
29
+
30
+ def _build_kwargs(self, request: MetricRequest) -> dict[str, Any]:
31
+ metadata = request.extra.get("metadata")
32
+ if not isinstance(metadata, dict):
33
+ metadata = {}
34
+ context_features = request.extra.get("context_features")
35
+ if not isinstance(context_features, dict):
36
+ context_features = {}
37
+ contract = metadata.get("score_contract") or metadata.get("adherence_contract")
38
+ if not isinstance(contract, dict):
39
+ contract = {}
40
+ expected_tokens = metadata.get("expected_tokens")
41
+ if not isinstance(expected_tokens, Sequence) or isinstance(
42
+ expected_tokens, (str, bytes)
43
+ ):
44
+ expected_tokens = tokenize(str(request.extra.get("expected_outcome") or ""))
45
+ return {
46
+ "query": request.query,
47
+ "response": request.response,
48
+ "system_message": request.system_message,
49
+ "tool_calls": request.tool_calls,
50
+ "action_id": request.extra.get("action_id"),
51
+ "phi": context_features.get("phi"),
52
+ "contract": contract,
53
+ "expected_tokens": expected_tokens,
54
+ }
55
+
56
+ def _normalize(self, raw: dict[str, Any]) -> float | None:
57
+ value = raw.get("normalized")
58
+ return float(value) if value is not None else None
59
+
60
+ def evaluate(self, request: MetricRequest) -> MetricResult:
61
+ if not (request.query and request.response):
62
+ return MetricResult(
63
+ metric=self.NAME,
64
+ score=None,
65
+ normalized=None,
66
+ status="skipped",
67
+ reason="query or response is empty",
68
+ evaluator=f"local:{self._scorer.name}",
69
+ )
70
+
71
+ authoritative = self._authoritative_completion(request)
72
+ if authoritative is not None:
73
+ value, reason = authoritative
74
+ return MetricResult(
75
+ metric=self.NAME,
76
+ score=value,
77
+ normalized=value,
78
+ status="completed",
79
+ reason=reason,
80
+ evaluator="local:episode-outcome",
81
+ )
82
+
83
+ try:
84
+ result = self._scorer.score(**self._build_kwargs(request))
85
+ except Exception as exc: # noqa: BLE001 # pragma: no cover
86
+ logger.warning("Local metric %s failed: %s", self.NAME.value, exc)
87
+ return MetricResult(
88
+ metric=self.NAME,
89
+ score=None,
90
+ normalized=None,
91
+ status="skipped",
92
+ reason=f"local scorer error: {exc}",
93
+ evaluator=f"local:{self._scorer.name}",
94
+ )
95
+ return _to_metric_result(self.NAME, self._scorer.name, result)
96
+
97
+ def _authoritative_completion(
98
+ self, request: MetricRequest
99
+ ) -> tuple[float, str] | None:
100
+ if self.NAME != MetricName.TASK_COMPLETION:
101
+ return None
102
+ metadata = request.extra.get("metadata")
103
+ if not isinstance(metadata, dict):
104
+ metadata = {}
105
+ action_id = request.extra.get("action_id")
106
+ correct_action_id = metadata.get("correct_action_id")
107
+ if action_id and correct_action_id:
108
+ return (
109
+ float(action_id == correct_action_id),
110
+ "derived from action_id and metadata.correct_action_id",
111
+ )
112
+ completed = metadata.get("task_completed")
113
+ if isinstance(completed, bool):
114
+ return float(completed), "derived from metadata.task_completed"
115
+ status = str(request.extra.get("execution_status") or "").lower()
116
+ if status == "completed":
117
+ return 1.0, "derived from execution_status=completed"
118
+ if status == "failed":
119
+ return 0.0, "derived from execution_status=failed"
120
+ if status == "partial":
121
+ return 0.5, "derived from execution_status=partial"
122
+ return None
123
+
124
+
125
+ def _to_metric_result(
126
+ metric: MetricName, scorer_name: str, result: ScoreResult
127
+ ) -> MetricResult:
128
+ normalized = max(0.0, min(1.0, float(result.normalized)))
129
+ return MetricResult(
130
+ metric=metric,
131
+ score=normalized,
132
+ normalized=normalized,
133
+ status="completed",
134
+ reason=f"local {result.label}",
135
+ properties=dict(result.features),
136
+ evaluator=f"local:{scorer_name}",
137
+ metadata={"label": result.label, "confidence": result.confidence},
138
+ )
139
+
140
+
141
+ def local_metrics(scorers: tuple[Scorer, Scorer, Scorer]) -> list[MetricEvaluator]:
142
+ """Return local metric adapters in reward-shaping order."""
143
+ intent, adherence, completion = scorers
144
+ return [
145
+ LocalScorerMetric(MetricName.INTENT_RESOLUTION, intent),
146
+ LocalScorerMetric(MetricName.TASK_ADHERENCE, adherence),
147
+ LocalScorerMetric(MetricName.TASK_COMPLETION, completion),
148
+ ]
149
+
150
+
151
+ __all__ = ["LocalScorerMetric", "local_metrics"]
@@ -0,0 +1,59 @@
1
+ """Convenience helpers to evaluate an episode across all default metrics."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Iterable, List, Optional
6
+
7
+ from ..config import ScoreConfig, ScoreRuntimeConfig
8
+ from ..scorers import build_scorers
9
+ from ..types import Episode, MetricResult
10
+ from .base import MetricEvaluator, MetricRequest
11
+ from .intent_resolution import IntentResolutionMetric
12
+ from .local import local_metrics
13
+ from .task_adherence import TaskAdherenceMetric
14
+ from .task_completion import TaskCompletionMetric
15
+
16
+
17
+ def default_metrics(
18
+ score_config: Optional[ScoreConfig] = None,
19
+ score_runtime_config: Optional[ScoreRuntimeConfig] = None,
20
+ ) -> List[MetricEvaluator]:
21
+ """Return local metrics by default, or Azure metrics when configured."""
22
+ runtime = score_runtime_config or ScoreRuntimeConfig()
23
+ llm = score_config or runtime.llm
24
+ if score_config is not None or runtime.tier == "llm" or (
25
+ runtime.tier is None and llm.enabled
26
+ ):
27
+ return [
28
+ IntentResolutionMetric(llm),
29
+ TaskAdherenceMetric(llm),
30
+ TaskCompletionMetric(llm),
31
+ ]
32
+ if runtime.tier is None:
33
+ runtime.tier = "stdlib"
34
+ return local_metrics(build_scorers(runtime))
35
+
36
+
37
+ def evaluate_all(
38
+ episode: Episode,
39
+ metrics: Optional[Iterable[MetricEvaluator]] = None,
40
+ *,
41
+ score_config: Optional[ScoreConfig] = None,
42
+ score_runtime_config: Optional[ScoreRuntimeConfig] = None,
43
+ ) -> List[MetricResult]:
44
+ """Evaluate one episode against every supplied metric.
45
+
46
+ Each metric is given its own try/except inside :meth:`evaluate`,
47
+ so a single failing scorer will not prevent the others from
48
+ producing scores.
49
+ """
50
+ metric_list = (
51
+ list(metrics)
52
+ if metrics is not None
53
+ else default_metrics(score_config, score_runtime_config)
54
+ )
55
+ request = MetricRequest.from_episode(episode)
56
+ return [m.evaluate(request) for m in metric_list]
57
+
58
+
59
+ __all__ = ["default_metrics", "evaluate_all"]
@@ -98,7 +98,10 @@ class RewardWriter:
98
98
  except Exception as exc: # pragma: no cover
99
99
  logger.warning("Failed to persist %s penalty for %s: %s", kind, episode.id, exc)
100
100
 
101
- # 4) Aggregate reward consumed by the learner
101
+ # 4) Aggregate reward consumed by the learner. Do not turn an
102
+ # all-skipped evaluation into a misleading neutral reward.
103
+ if not shaped.metric_contributions and not shaped.penalties:
104
+ return stored
102
105
  aggregate = Reward(
103
106
  episode_id=episode.id,
104
107
  agent_id=episode.agent_id,
@@ -66,8 +66,10 @@ class LearningStore(ABC):
66
66
  *,
67
67
  task_id: Optional[str] = None,
68
68
  full_only: bool = False,
69
+ start_date: Optional[str] = None,
70
+ end_date: Optional[str] = None,
69
71
  ) -> int:
70
- """Count episodes, optionally limiting the count to full episodes."""
72
+ """Count episodes filtered by task, completeness, or time window."""
71
73
 
72
74
  # ---- Metric results -------------------------------------------
73
75
 
@@ -272,6 +272,8 @@ class CosmosStore(LearningStore):
272
272
  *,
273
273
  task_id: Optional[str] = None,
274
274
  full_only: bool = False,
275
+ start_date: Optional[str] = None,
276
+ end_date: Optional[str] = None,
275
277
  ) -> int:
276
278
  clauses = ["c.agent_id = @agent_id"]
277
279
  params: List[Dict[str, Any]] = [{"name": "@agent_id", "value": agent_id}]
@@ -281,6 +283,12 @@ class CosmosStore(LearningStore):
281
283
  else:
282
284
  clauses.append("c.task_id = @task_id")
283
285
  params.append({"name": "@task_id", "value": task_id})
286
+ if start_date:
287
+ clauses.append("c.created_at >= @start_date")
288
+ params.append({"name": "@start_date", "value": start_date})
289
+ if end_date:
290
+ clauses.append("c.created_at <= @end_date")
291
+ params.append({"name": "@end_date", "value": end_date})
284
292
  if full_only:
285
293
  clauses.extend(
286
294
  [
@@ -212,12 +212,16 @@ class LocalFileStore(LearningStore):
212
212
  *,
213
213
  task_id: Optional[str] = None,
214
214
  full_only: bool = False,
215
+ start_date: Optional[str] = None,
216
+ end_date: Optional[str] = None,
215
217
  ) -> int:
216
218
  return sum(
217
219
  1
218
220
  for doc in self._read_dir_docs("episodes", agent_id)
219
221
  if (task_id is None or doc.get("task_id", "default") == task_id)
220
222
  and (not full_only or Episode.from_dict(doc).is_full)
223
+ and (start_date is None or doc.get("created_at", "") >= start_date)
224
+ and (end_date is None or doc.get("created_at", "") <= end_date)
221
225
  )
222
226
 
223
227
  # ------------------------------------------------------------------
@@ -93,6 +93,8 @@ class InMemoryStore(LearningStore):
93
93
  *,
94
94
  task_id: Optional[str] = None,
95
95
  full_only: bool = False,
96
+ start_date: Optional[str] = None,
97
+ end_date: Optional[str] = None,
96
98
  ) -> int:
97
99
  return sum(
98
100
  1
@@ -100,6 +102,8 @@ class InMemoryStore(LearningStore):
100
102
  if episode.agent_id == agent_id
101
103
  and (task_id is None or episode.task_id == task_id)
102
104
  and (not full_only or episode.is_full)
105
+ and (start_date is None or episode.created_at >= start_date)
106
+ and (end_date is None or episode.created_at <= end_date)
103
107
  )
104
108
 
105
109
  # ---- Metric results -------------------------------------------