agent-learning 0.4.3__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. {agent_learning-0.4.3/src/agent_learning.egg-info → agent_learning-0.5.0}/PKG-INFO +9 -1
  2. {agent_learning-0.4.3 → agent_learning-0.5.0}/PYPI.md +8 -0
  3. {agent_learning-0.4.3 → agent_learning-0.5.0}/README.md +15 -4
  4. {agent_learning-0.4.3 → agent_learning-0.5.0}/pyproject.toml +1 -1
  5. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/_version.py +1 -1
  6. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/cli.py +231 -4
  7. {agent_learning-0.4.3 → agent_learning-0.5.0/src/agent_learning.egg-info}/PKG-INFO +9 -1
  8. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_cli.py +241 -2
  9. {agent_learning-0.4.3 → agent_learning-0.5.0}/LICENSE +0 -0
  10. {agent_learning-0.4.3 → agent_learning-0.5.0}/setup.cfg +0 -0
  11. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/__init__.py +0 -0
  12. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/capture.py +0 -0
  13. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/__init__.py +0 -0
  14. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/base.py +0 -0
  15. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/router.py +0 -0
  16. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/scorers/__init__.py +0 -0
  17. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/scorers/_base.py +0 -0
  18. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/scorers/adherence.py +0 -0
  19. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/scorers/completion.py +0 -0
  20. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/scorers/intent.py +0 -0
  21. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/config.py +0 -0
  22. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/learners/__init__.py +0 -0
  23. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/learners/base.py +0 -0
  24. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/learners/reinforce.py +0 -0
  25. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/metrics/__init__.py +0 -0
  26. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/metrics/base.py +0 -0
  27. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/metrics/intent_resolution.py +0 -0
  28. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/metrics/local.py +0 -0
  29. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/metrics/registry.py +0 -0
  30. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/metrics/task_adherence.py +0 -0
  31. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/metrics/task_completion.py +0 -0
  32. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/policy/__init__.py +0 -0
  33. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/policy/base.py +0 -0
  34. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/policy/contextual_softmax.py +0 -0
  35. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/policy/softmax_bandit.py +0 -0
  36. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/py.typed +0 -0
  37. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/rewards/__init__.py +0 -0
  38. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/rewards/shaping.py +0 -0
  39. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/rewards/writer.py +0 -0
  40. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/__init__.py +0 -0
  41. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/base.py +0 -0
  42. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/llm/__init__.py +0 -0
  43. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/llm/_base.py +0 -0
  44. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/llm/adherence.py +0 -0
  45. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/llm/completion.py +0 -0
  46. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/llm/intent.py +0 -0
  47. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp/__init__.py +0 -0
  48. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp/_base.py +0 -0
  49. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp/adherence.py +0 -0
  50. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp/completion.py +0 -0
  51. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp/intent.py +0 -0
  52. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp_text/__init__.py +0 -0
  53. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp_text/_base.py +0 -0
  54. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp_text/adherence.py +0 -0
  55. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp_text/completion.py +0 -0
  56. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp_text/intent.py +0 -0
  57. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/slm/__init__.py +0 -0
  58. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/slm/_base.py +0 -0
  59. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/slm/adherence.py +0 -0
  60. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/slm/completion.py +0 -0
  61. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/slm/intent.py +0 -0
  62. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/stdlib/__init__.py +0 -0
  63. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/stdlib/_text.py +0 -0
  64. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/stdlib/adherence.py +0 -0
  65. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/stdlib/completion.py +0 -0
  66. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/stdlib/intent.py +0 -0
  67. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/storage/__init__.py +0 -0
  68. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/storage/base.py +0 -0
  69. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/storage/cosmos.py +0 -0
  70. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/storage/local.py +0 -0
  71. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/storage/memory.py +0 -0
  72. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/training/__init__.py +0 -0
  73. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/training/runner.py +0 -0
  74. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/types.py +0 -0
  75. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning.egg-info/SOURCES.txt +0 -0
  76. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning.egg-info/dependency_links.txt +0 -0
  77. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning.egg-info/entry_points.txt +0 -0
  78. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning.egg-info/requires.txt +0 -0
  79. {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning.egg-info/top_level.txt +0 -0
  80. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_capture.py +0 -0
  81. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_contextual_policy.py +0 -0
  82. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_default_store.py +0 -0
  83. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_end_to_end.py +0 -0
  84. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_learner.py +0 -0
  85. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_metrics_local.py +0 -0
  86. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_policy.py +0 -0
  87. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_scorers.py +0 -0
  88. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_scorers_llm.py +0 -0
  89. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_scorers_nlp_text.py +0 -0
  90. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_scorers_slm.py +0 -0
  91. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_scorers_stdlib.py +0 -0
  92. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_shaping.py +0 -0
  93. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_storage_local.py +0 -0
  94. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_storage_memory.py +0 -0
  95. {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_types.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agent-learning
3
- Version: 0.4.3
3
+ Version: 0.5.0
4
4
  Summary: Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default.
5
5
  Author: Chris Tava
6
6
  License: MIT License
@@ -69,6 +69,10 @@ Dynamic: license-file
69
69
 
70
70
  Native reinforcement learning SDK for AI agents. An in-process Learner optimizes a small, interpretable TaskPolicy over discrete agent choices (understand intent and complete task by choosing the right outcome).
71
71
 
72
+ TaskPolicies model reusable decisions among executable alternatives such as
73
+ models, skills, tools, workflows, or workloads. Factual questions, ordinary
74
+ chat, reporting, and learning automation are not policy tasks.
75
+
72
76
  ## How it works
73
77
 
74
78
  The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tune jobs and no opaque update cycles — just three pieces that run in your existing Python process:
@@ -79,4 +83,8 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
79
83
 
80
84
  3. **Learner** applies REINFORCE-with-baseline to update TaskPolicy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
81
85
 
86
+ `task-policy-decide` closes the loop at execution time by returning the selected
87
+ action plus historical correctness, reward, result summaries, and per-metric
88
+ quality feedback for the agent to use on its next delegated decision.
89
+
82
90
  Every episode, reward, run, and deployment is captured by the configured store — in-memory or local files by default, or Azure Cosmos DB — giving you a complete lineage and audit trail of how the policy evolved over time.
@@ -2,6 +2,10 @@
2
2
 
3
3
  Native reinforcement learning SDK for AI agents. An in-process Learner optimizes a small, interpretable TaskPolicy over discrete agent choices (understand intent and complete task by choosing the right outcome).
4
4
 
5
+ TaskPolicies model reusable decisions among executable alternatives such as
6
+ models, skills, tools, workflows, or workloads. Factual questions, ordinary
7
+ chat, reporting, and learning automation are not policy tasks.
8
+
5
9
  ## How it works
6
10
 
7
11
  The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tune jobs and no opaque update cycles — just three pieces that run in your existing Python process:
@@ -12,4 +16,8 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
12
16
 
13
17
  3. **Learner** applies REINFORCE-with-baseline to update TaskPolicy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
14
18
 
19
+ `task-policy-decide` closes the loop at execution time by returning the selected
20
+ action plus historical correctness, reward, result summaries, and per-metric
21
+ quality feedback for the agent to use on its next delegated decision.
22
+
15
23
  Every episode, reward, run, and deployment is captured by the configured store — in-memory or local files by default, or Azure Cosmos DB — giving you a complete lineage and audit trail of how the policy evolved over time.
@@ -4,6 +4,10 @@ Native reinforcement learning SDK for AI agents. An in-process
4
4
  Learner optimizes a small, interpretable TaskPolicy over discrete agent choices (e.g., "take action A", "take action B", "take action C") using on-device evaluation scores as the reward
5
5
  signal by default.
6
6
 
7
+ TaskPolicies represent reusable **decisions among executable alternatives**.
8
+ They are not conversation logs: factual questions, ordinary chat, reporting,
9
+ and agent-learning automation are not policy tasks.
10
+
7
11
  <p align="center">
8
12
  <img src="images/agent-learning-loop.svg" alt="Animated TaskPolicy Score Learner loop: TaskPolicy chooses a task action, Score evaluates the episode, and Learner updates TaskPolicy" width="960" style="max-width:100%; height:auto;" />
9
13
  </p>
@@ -31,6 +35,11 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
31
35
 
32
36
  <img src="images/cc970c453583c982.png" alt="Policy quality improves with every batch of episodes" width="360" style="max-width:100%; height:auto;" />
33
37
 
38
+ Before the next delegated execution, `task-policy-decide` samples the learned
39
+ policy and returns the selected action together with correctness rate, mean
40
+ reward, recent result summaries, and intent/adherence/completion scores. Agents
41
+ consume that feedback rather than training a policy that is never used.
42
+
34
43
  Every episode, reward, run, and deployment is captured by the
35
44
  configured store — in-memory or local files by default, or Azure Cosmos DB —
36
45
  giving you a complete lineage and audit trail of how the policy
@@ -71,12 +80,14 @@ The `agent-learn` CLI provides the current task-learning-loop operations:
71
80
 
72
81
  ```text
73
82
  agent-learn list
74
- agent-learn tasks-list <agent_id>
83
+ agent-learn --version
84
+ agent-learn tasks-list <agent_id> [--decision-only]
75
85
  agent-learn task-episodes-count <agent_id> [--task-id <task_id>] [--start-date <date>] [--end-date <date>]
76
86
  agent-learn task-episodes-list <agent_id> [--task-id <task_id>] [--limit <1-500>] [--include-incomplete] [--start-date <date>] [--end-date <date>]
77
- agent-learn task-policy-init --agent-id <agent_id> --task-id <task_id> --actions ./actions.json
78
- agent-learn task-episode-register --agent-id <agent_id> --task-id <task_id> --episode ./episode.json
87
+ agent-learn task-policy-init --agent-id <agent_id> --task-id <task_id> --decision-context <context> --actions ./actions.json
88
+ agent-learn task-policy-decide --agent-id <agent_id> --task-id <task_id> [--history-limit <1-500>] [--greedy] [--seed <integer>]
89
+ agent-learn task-episode-register --agent-id <agent_id> --task-id <task_id> --episode ./episode.json [--require-decision-policy]
79
90
  agent-learn score --agent-id <agent_id> [--task-id <task_id>] [--limit <1-500>]
80
- agent-learn train --agent-id <agent_id> [--task-id <task_id>] [--limit <1-500>] [--min-episodes <1-500>] [--start-date <date>] [--end-date <date>] [--skip-scoring]
91
+ agent-learn train --agent-id <agent_id> [--task-id <task_id>] [--decision-only] [--limit <1-500>] [--min-episodes <1-500>] [--start-date <date>] [--end-date <date>] [--skip-scoring]
81
92
  agent-learn task-policy --agent-id <agent_id> --task-id <task_id>
82
93
  ```
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "agent-learning"
7
- version = "0.4.3"
7
+ version = "0.5.0"
8
8
  description = "Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default."
9
9
  readme = "PYPI.md"
10
10
  requires-python = ">=3.10"
@@ -1,3 +1,3 @@
1
1
  """Package version."""
2
2
 
3
- __version__ = "0.4.3"
3
+ __version__ = "0.5.0"
@@ -5,12 +5,15 @@ from __future__ import annotations
5
5
  import argparse
6
6
  import json
7
7
  import logging
8
+ import math
9
+ import random
8
10
  import sys
9
11
  import uuid
10
12
  from collections import Counter
11
13
  from datetime import datetime, timezone
12
14
  from typing import Any
13
15
 
16
+ from ._version import __version__
14
17
  from .policy.softmax_bandit import SoftmaxPolicy
15
18
  from .storage.cosmos import get_default_store
16
19
  from .training.runner import LearningRunner
@@ -19,6 +22,7 @@ from .types import Action, Episode, MetricName, PolicySnapshot, RewardSource
19
22
  logger = logging.getLogger(__name__)
20
23
 
21
24
  _MAX_EPISODES = 500
25
+ _DECISION_POLICY_SCOPE = "delegated_decision"
22
26
 
23
27
 
24
28
  def _episode_limit(value: str) -> int:
@@ -39,13 +43,18 @@ def _iso_date(value: str) -> str:
39
43
 
40
44
 
41
45
  def _build_arg_parser() -> argparse.ArgumentParser:
42
- parser = argparse.ArgumentParser(prog="agent-learn", description="Native RL CLI for AI agents.")
46
+ parser = argparse.ArgumentParser(
47
+ prog="agent-learn",
48
+ description=f"Native RL CLI for AI agents. SDK version {__version__}.",
49
+ )
50
+ parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
43
51
  sub = parser.add_subparsers(dest="command", required=True)
44
52
 
45
53
  sub.add_parser("list", help="List discovered agent ids and names.")
46
54
 
47
55
  tasks = sub.add_parser("tasks-list", help="List tasks for an agent.")
48
56
  tasks.add_argument("agent_id")
57
+ tasks.add_argument("--decision-only", action="store_true")
49
58
 
50
59
  count = sub.add_parser(
51
60
  "task-episodes-count",
@@ -72,6 +81,7 @@ def _build_arg_parser() -> argparse.ArgumentParser:
72
81
  train.add_argument("--task-id")
73
82
  train.add_argument("--limit", type=_episode_limit, default=200)
74
83
  train.add_argument("--min-episodes", type=_episode_limit, default=1)
84
+ train.add_argument("--decision-only", action="store_true")
75
85
  train.add_argument("--start-date", type=_iso_date)
76
86
  train.add_argument("--end-date", type=_iso_date)
77
87
  train.add_argument(
@@ -89,12 +99,27 @@ def _build_arg_parser() -> argparse.ArgumentParser:
89
99
  show.add_argument("--agent-id", required=True)
90
100
  show.add_argument("--task-id", required=True)
91
101
 
102
+ decide = sub.add_parser(
103
+ "task-policy-decide",
104
+ help="Choose a delegated decision action and return learned feedback.",
105
+ )
106
+ decide.add_argument("--agent-id", required=True)
107
+ decide.add_argument("--task-id", required=True)
108
+ decide.add_argument("--history-limit", type=_episode_limit, default=100)
109
+ decide.add_argument("--greedy", action="store_true")
110
+ decide.add_argument("--seed", type=int)
111
+
92
112
  init = sub.add_parser(
93
113
  "task-policy-init",
94
114
  help="Create and activate the initial policy for an agent task.",
95
115
  )
96
116
  init.add_argument("--agent-id", required=True)
97
117
  init.add_argument("--task-id", required=True)
118
+ init.add_argument(
119
+ "--decision-context",
120
+ required=True,
121
+ help="Stable description of the delegated choice this policy controls.",
122
+ )
98
123
  init.add_argument(
99
124
  "--actions",
100
125
  required=True,
@@ -107,6 +132,7 @@ def _build_arg_parser() -> argparse.ArgumentParser:
107
132
  )
108
133
  register.add_argument("--agent-id", required=True)
109
134
  register.add_argument("--task-id", required=True)
135
+ register.add_argument("--require-decision-policy", action="store_true")
110
136
  register.add_argument(
111
137
  "--episode",
112
138
  required=True,
@@ -124,7 +150,14 @@ def _cmd_agents_list(args: argparse.Namespace) -> int:
124
150
 
125
151
 
126
152
  def _cmd_agent_tasks_list(args: argparse.Namespace) -> int:
127
- tasks = get_default_store().list_agent_tasks(args.agent_id)
153
+ store = get_default_store()
154
+ tasks = store.list_agent_tasks(args.agent_id)
155
+ if args.decision_only:
156
+ tasks = [
157
+ task
158
+ for task in tasks
159
+ if _is_decision_policy(store.get_active_policy(args.agent_id, task.id))
160
+ ]
128
161
  print(json.dumps([{"id": task.id, "name": task.name} for task in tasks], indent=2))
129
162
  return 0
130
163
 
@@ -199,6 +232,9 @@ def _cmd_train(args: argparse.Namespace) -> int:
199
232
  if snapshot is None:
200
233
  skipped.append({"task_id": task_id, "reason": "no active policy"})
201
234
  continue
235
+ if args.decision_only and not _is_decision_policy(snapshot):
236
+ skipped.append({"task_id": task_id, "reason": "not a delegated decision policy"})
237
+ continue
202
238
  episode_limit = episode_limits.get(task_id, 0)
203
239
  if episode_limit == 0:
204
240
  skipped.append({"task_id": task_id, "reason": "no episodes in selected batch"})
@@ -266,6 +302,89 @@ def _policy_payload(snapshot: PolicySnapshot) -> dict[str, Any]:
266
302
  return payload
267
303
 
268
304
 
305
+ def _is_decision_policy(snapshot: PolicySnapshot | None) -> bool:
306
+ return bool(
307
+ snapshot
308
+ and snapshot.metadata.get("policy_scope") == _DECISION_POLICY_SCOPE
309
+ )
310
+
311
+
312
+ def _latest_aggregate(store: Any, episode: Episode) -> float | None:
313
+ rewards = [
314
+ reward
315
+ for reward in store.get_rewards_for_episode(episode.id, episode.agent_id)
316
+ if reward.source == RewardSource.AGGREGATE
317
+ ]
318
+ if not rewards:
319
+ return None
320
+ return max(rewards, key=lambda reward: reward.created_at).value
321
+
322
+
323
+ def _decision_feedback(
324
+ store: Any, snapshot: PolicySnapshot, history_limit: int
325
+ ) -> dict[str, Any]:
326
+ stats = {
327
+ action.id: {
328
+ "attempts": 0,
329
+ "correctness_evaluated": 0,
330
+ "correct": 0,
331
+ "correctness_rate": None,
332
+ "rewarded_episodes": 0,
333
+ "mean_reward": None,
334
+ "recent_outcomes": [],
335
+ }
336
+ for action in snapshot.actions
337
+ }
338
+ reward_totals = {action.id: 0.0 for action in snapshot.actions}
339
+ episodes = store.query_episodes(
340
+ snapshot.agent_id,
341
+ task_id=snapshot.task_id,
342
+ limit=history_limit,
343
+ )
344
+ for episode in episodes:
345
+ action_id = episode.action_id
346
+ if action_id not in stats:
347
+ continue
348
+ action_stats = stats[action_id]
349
+ action_stats["attempts"] += 1
350
+ correct_action_id = episode.metadata.get("correct_action_id")
351
+ was_correct = None
352
+ if correct_action_id:
353
+ was_correct = action_id == correct_action_id
354
+ action_stats["correctness_evaluated"] += 1
355
+ action_stats["correct"] += int(was_correct)
356
+ reward = _latest_aggregate(store, episode)
357
+ if reward is not None:
358
+ action_stats["rewarded_episodes"] += 1
359
+ reward_totals[action_id] += reward
360
+ if len(action_stats["recent_outcomes"]) < 3:
361
+ score_breakdown = {}
362
+ for result in store.get_metric_results(episode.id, episode.agent_id):
363
+ score_breakdown[result.metric.value] = {
364
+ "normalized": result.normalized,
365
+ "status": result.status,
366
+ "reason": result.reason,
367
+ }
368
+ action_stats["recent_outcomes"].append(
369
+ {
370
+ "created_at": episode.created_at,
371
+ "was_correct": was_correct,
372
+ "reward": reward,
373
+ "execution_status": episode.execution_status,
374
+ "result_summary": episode.result_summary,
375
+ "score_breakdown": score_breakdown,
376
+ }
377
+ )
378
+ for action_id, action_stats in stats.items():
379
+ evaluated = action_stats["correctness_evaluated"]
380
+ rewarded = action_stats["rewarded_episodes"]
381
+ if evaluated:
382
+ action_stats["correctness_rate"] = action_stats["correct"] / evaluated
383
+ if rewarded:
384
+ action_stats["mean_reward"] = reward_totals[action_id] / rewarded
385
+ return {"episodes_reviewed": len(episodes), "actions": stats}
386
+
387
+
269
388
  def _policy_difference(
270
389
  current: PolicySnapshot, previous: PolicySnapshot | None
271
390
  ) -> dict[str, Any] | None:
@@ -326,6 +445,78 @@ def _cmd_show_task_policy(args: argparse.Namespace) -> int:
326
445
  return 0
327
446
 
328
447
 
448
+ def _cmd_decide_task_policy(args: argparse.Namespace) -> int:
449
+ store = get_default_store()
450
+ snapshot = store.get_active_policy(args.agent_id, args.task_id)
451
+ if snapshot is None:
452
+ print(
453
+ f"No active policy found for agent_id={args.agent_id!r}, "
454
+ f"task_id={args.task_id!r}.",
455
+ file=sys.stderr,
456
+ )
457
+ return 2
458
+ if not _is_decision_policy(snapshot):
459
+ print(
460
+ "The active policy is not marked as a delegated decision policy. "
461
+ "Questions, reporting tasks, and agent-learning automation are not eligible.",
462
+ file=sys.stderr,
463
+ )
464
+ return 2
465
+ rng = random.Random(args.seed) if args.seed is not None else None
466
+ policy = SoftmaxPolicy.from_snapshot(snapshot, rng=rng)
467
+ probabilities = policy.probabilities()
468
+ recommended_index = max(range(len(probabilities)), key=probabilities.__getitem__)
469
+ if args.greedy:
470
+ selected_index = recommended_index
471
+ selected_action = snapshot.actions[selected_index]
472
+ selected_probability = probabilities[selected_index]
473
+ logprob = math.log(max(selected_probability, 1e-12))
474
+ mode = "greedy"
475
+ else:
476
+ decision = policy.choose()
477
+ selected_action = decision.action
478
+ selected_index = next(
479
+ index
480
+ for index, action in enumerate(snapshot.actions)
481
+ if action.id == selected_action.id
482
+ )
483
+ selected_probability = probabilities[selected_index]
484
+ logprob = decision.logprob
485
+ mode = "sampled"
486
+ feedback = _decision_feedback(store, snapshot, args.history_limit)
487
+ selected_stats = feedback["actions"][selected_action.id]
488
+ recommendation = snapshot.actions[recommended_index]
489
+ print(
490
+ json.dumps(
491
+ {
492
+ "agent_id": snapshot.agent_id,
493
+ "task_id": snapshot.task_id,
494
+ "decision_context": snapshot.metadata.get("decision_context"),
495
+ "policy_id": snapshot.id,
496
+ "policy_version": snapshot.version,
497
+ "selection_mode": mode,
498
+ "selected_action": {
499
+ **selected_action.to_dict(),
500
+ "probability": selected_probability,
501
+ "logprob": logprob,
502
+ },
503
+ "recommended_action": {
504
+ **recommendation.to_dict(),
505
+ "probability": probabilities[recommended_index],
506
+ },
507
+ "action_probabilities": {
508
+ action.id: probability
509
+ for action, probability in zip(snapshot.actions, probabilities)
510
+ },
511
+ "selected_action_feedback": selected_stats,
512
+ "historical_feedback": feedback,
513
+ },
514
+ indent=2,
515
+ )
516
+ )
517
+ return 0
518
+
519
+
329
520
  def _cmd_init_task_policy(args: argparse.Namespace) -> int:
330
521
  store = get_default_store()
331
522
  if store.get_active_policy(args.agent_id, args.task_id) is not None:
@@ -341,20 +532,35 @@ def _cmd_init_task_policy(args: argparse.Namespace) -> int:
341
532
  except (OSError, json.JSONDecodeError) as exc:
342
533
  print(f"Unable to read --actions file: {exc}", file=sys.stderr)
343
534
  return 2
344
- if not isinstance(action_payloads, list) or not action_payloads:
345
- print("--actions file must contain a non-empty JSON list", file=sys.stderr)
535
+ if not isinstance(action_payloads, list) or len(action_payloads) < 2:
536
+ print(
537
+ "--actions file must contain at least two delegated decision actions",
538
+ file=sys.stderr,
539
+ )
346
540
  return 2
347
541
  try:
348
542
  actions = [Action.from_dict(item) for item in action_payloads]
349
543
  except (KeyError, TypeError, ValueError) as exc:
350
544
  print(f"Invalid action definition: {exc}", file=sys.stderr)
351
545
  return 2
546
+ action_ids = [action.id for action in actions]
547
+ if any(not action_id.strip() for action_id in action_ids) or len(set(action_ids)) != len(
548
+ action_ids
549
+ ):
550
+ print("Decision action ids must be non-empty and unique", file=sys.stderr)
551
+ return 2
352
552
  policy = SoftmaxPolicy.from_actions(
353
553
  actions,
354
554
  agent_id=args.agent_id,
355
555
  task_id=args.task_id,
356
556
  )
357
557
  snapshot = policy.snapshot()
558
+ snapshot.metadata.update(
559
+ {
560
+ "policy_scope": _DECISION_POLICY_SCOPE,
561
+ "decision_context": args.decision_context,
562
+ }
563
+ )
358
564
  store.store_policy(snapshot)
359
565
  print(json.dumps(_policy_payload(snapshot), indent=2))
360
566
  return 0
@@ -389,6 +595,26 @@ def _cmd_register_task_episode(args: argparse.Namespace) -> int:
389
595
  print(f"Invalid episode definition: {exc}", file=sys.stderr)
390
596
  return 2
391
597
 
598
+ if args.require_decision_policy:
599
+ policy = get_default_store().get_policy(episode.policy_id or "", args.agent_id)
600
+ if not _is_decision_policy(policy) or policy.task_id != args.task_id:
601
+ print(
602
+ "Episode registration requires a delegated decision policy and its policy_id.",
603
+ file=sys.stderr,
604
+ )
605
+ return 2
606
+ action_ids = {action.id for action in policy.actions}
607
+ if episode.action_id not in action_ids:
608
+ print("Episode action_id is not in the delegated decision policy.", file=sys.stderr)
609
+ return 2
610
+ correct_action_id = episode.metadata.get("correct_action_id")
611
+ if correct_action_id is not None and correct_action_id not in action_ids:
612
+ print(
613
+ "Episode metadata.correct_action_id is not in the delegated decision policy.",
614
+ file=sys.stderr,
615
+ )
616
+ return 2
617
+
392
618
  get_default_store().store_episode(episode)
393
619
  print(json.dumps(episode.to_dict(), indent=2))
394
620
  return 0
@@ -406,6 +632,7 @@ def main(argv: list[str] | None = None) -> int:
406
632
  "train": _cmd_train,
407
633
  "score": _cmd_score,
408
634
  "task-policy": _cmd_show_task_policy,
635
+ "task-policy-decide": _cmd_decide_task_policy,
409
636
  "task-policy-init": _cmd_init_task_policy,
410
637
  "task-episode-register": _cmd_register_task_episode,
411
638
  }
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agent-learning
3
- Version: 0.4.3
3
+ Version: 0.5.0
4
4
  Summary: Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default.
5
5
  Author: Chris Tava
6
6
  License: MIT License
@@ -69,6 +69,10 @@ Dynamic: license-file
69
69
 
70
70
  Native reinforcement learning SDK for AI agents. An in-process Learner optimizes a small, interpretable TaskPolicy over discrete agent choices (understand intent and complete task by choosing the right outcome).
71
71
 
72
+ TaskPolicies model reusable decisions among executable alternatives such as
73
+ models, skills, tools, workflows, or workloads. Factual questions, ordinary
74
+ chat, reporting, and learning automation are not policy tasks.
75
+
72
76
  ## How it works
73
77
 
74
78
  The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tune jobs and no opaque update cycles — just three pieces that run in your existing Python process:
@@ -79,4 +83,8 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
79
83
 
80
84
  3. **Learner** applies REINFORCE-with-baseline to update TaskPolicy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
81
85
 
86
+ `task-policy-decide` closes the loop at execution time by returning the selected
87
+ action plus historical correctness, reward, result summaries, and per-metric
88
+ quality feedback for the agent to use on its next delegated decision.
89
+
82
90
  Every episode, reward, run, and deployment is captured by the configured store — in-memory or local files by default, or Azure Cosmos DB — giving you a complete lineage and audit trail of how the policy evolved over time.
@@ -7,7 +7,7 @@ from pathlib import Path
7
7
 
8
8
  import pytest
9
9
 
10
- from agent_learning import cli
10
+ from agent_learning import __version__, cli
11
11
  from agent_learning.policy import SoftmaxPolicy
12
12
  from agent_learning.storage import InMemoryStore
13
13
  from agent_learning.types import (
@@ -35,6 +35,14 @@ def _full_episode() -> Episode:
35
35
  )
36
36
 
37
37
 
38
+ def test_help_and_version_print_sdk_version(capsys) -> None:
39
+ assert f"SDK version {__version__}" in cli._build_arg_parser().format_help()
40
+ with pytest.raises(SystemExit) as exit_info:
41
+ cli.main(["--version"])
42
+ assert exit_info.value.code == 0
43
+ assert capsys.readouterr().out.strip() == f"agent-learn {__version__}"
44
+
45
+
38
46
  def test_discovery_and_full_episode_count(monkeypatch, capsys) -> None:
39
47
  store = InMemoryStore()
40
48
  store.store_episode(_full_episode())
@@ -134,7 +142,12 @@ def test_task_policy_init_and_inspection(monkeypatch, capsys, tmp_path: Path) ->
134
142
  monkeypatch.setattr(cli, "get_default_store", lambda: store)
135
143
  actions_path = tmp_path / "actions.json"
136
144
  actions_path.write_text(
137
- json.dumps([{"id": "respond", "description": "Respond directly"}]),
145
+ json.dumps(
146
+ [
147
+ {"id": "respond", "description": "Respond directly"},
148
+ {"id": "delegate", "description": "Delegate the response"},
149
+ ]
150
+ ),
138
151
  encoding="utf-8",
139
152
  )
140
153
 
@@ -146,6 +159,8 @@ def test_task_policy_init_and_inspection(monkeypatch, capsys, tmp_path: Path) ->
146
159
  "agent-1",
147
160
  "--task-id",
148
161
  "chat",
162
+ "--decision-context",
163
+ "Choose how the agent should respond to a chat request",
149
164
  "--actions",
150
165
  str(actions_path),
151
166
  ]
@@ -154,6 +169,10 @@ def test_task_policy_init_and_inspection(monkeypatch, capsys, tmp_path: Path) ->
154
169
  )
155
170
  initialized = json.loads(capsys.readouterr().out)
156
171
  assert initialized["task_id"] == "chat"
172
+ assert initialized["metadata"] == {
173
+ "policy_scope": "delegated_decision",
174
+ "decision_context": "Choose how the agent should respond to a chat request",
175
+ }
157
176
 
158
177
  assert (
159
178
  cli.main(
@@ -166,6 +185,33 @@ def test_task_policy_init_and_inspection(monkeypatch, capsys, tmp_path: Path) ->
166
185
  assert inspected["previous_policy"] is None
167
186
 
168
187
 
188
+ def test_task_policy_init_requires_two_unique_decision_actions(
189
+ monkeypatch, capsys, tmp_path: Path
190
+ ) -> None:
191
+ store = InMemoryStore()
192
+ monkeypatch.setattr(cli, "get_default_store", lambda: store)
193
+ actions_path = tmp_path / "actions.json"
194
+ actions_path.write_text(json.dumps([{"id": "only"}]), encoding="utf-8")
195
+
196
+ result = cli.main(
197
+ [
198
+ "task-policy-init",
199
+ "--agent-id",
200
+ "scout",
201
+ "--task-id",
202
+ "not-a-decision",
203
+ "--decision-context",
204
+ "There is only one action",
205
+ "--actions",
206
+ str(actions_path),
207
+ ]
208
+ )
209
+
210
+ assert result == 2
211
+ assert "at least two" in capsys.readouterr().err
212
+ assert store.get_active_policy("scout", "not-a-decision") is None
213
+
214
+
169
215
  def test_task_episode_register_persists_full_episode(
170
216
  monkeypatch, capsys, tmp_path: Path
171
217
  ) -> None:
@@ -209,6 +255,199 @@ def test_task_episode_register_persists_full_episode(
209
255
  assert episode.is_full
210
256
 
211
257
 
258
+ def test_task_policy_decide_returns_learned_feedback(monkeypatch, capsys) -> None:
259
+ store = InMemoryStore()
260
+ monkeypatch.setattr(cli, "get_default_store", lambda: store)
261
+ policy = SoftmaxPolicy.from_actions(
262
+ [Action(id="use_skill"), Action(id="use_model")],
263
+ agent_id="scout",
264
+ task_id="choose-delegation",
265
+ initial_logits={"use_skill": 1.0, "use_model": 0.0},
266
+ )
267
+ snapshot = policy.snapshot()
268
+ snapshot.metadata = {
269
+ "policy_scope": "delegated_decision",
270
+ "decision_context": "Choose whether to delegate to a skill or language model",
271
+ }
272
+ store.store_policy(snapshot)
273
+ outcomes = [
274
+ ("correct", "use_skill", 0.8),
275
+ ("incorrect", "use_model", -0.3),
276
+ ]
277
+ for label, correct_action_id, reward_value in outcomes:
278
+ episode = Episode(
279
+ id=label,
280
+ agent_id="scout",
281
+ task_id="choose-delegation",
282
+ policy_id=snapshot.id,
283
+ action_id="use_skill",
284
+ execution_status="completed",
285
+ result_summary=label,
286
+ metadata={"correct_action_id": correct_action_id},
287
+ )
288
+ store.store_episode(episode)
289
+ store.store_metric_results(
290
+ episode.id,
291
+ episode.agent_id,
292
+ [
293
+ MetricResult(
294
+ metric=MetricName.TASK_COMPLETION,
295
+ score=1.0 if label == "correct" else 0.0,
296
+ normalized=1.0 if label == "correct" else 0.0,
297
+ status="completed",
298
+ reason=label,
299
+ )
300
+ ],
301
+ )
302
+ store.store_reward(
303
+ Reward(
304
+ episode_id=episode.id,
305
+ agent_id=episode.agent_id,
306
+ source=RewardSource.AGGREGATE,
307
+ value=reward_value,
308
+ )
309
+ )
310
+
311
+ assert (
312
+ cli.main(
313
+ [
314
+ "task-policy-decide",
315
+ "--agent-id",
316
+ "scout",
317
+ "--task-id",
318
+ "choose-delegation",
319
+ "--greedy",
320
+ ]
321
+ )
322
+ == 0
323
+ )
324
+ result = json.loads(capsys.readouterr().out)
325
+
326
+ assert result["selected_action"]["id"] == "use_skill"
327
+ assert result["selected_action"]["probability"] > 0.5
328
+ assert result["recommended_action"]["id"] == "use_skill"
329
+ feedback = result["selected_action_feedback"]
330
+ assert feedback["attempts"] == 2
331
+ assert feedback["correctness_rate"] == 0.5
332
+ assert feedback["mean_reward"] == pytest.approx(0.25)
333
+ assert {item["was_correct"] for item in feedback["recent_outcomes"]} == {
334
+ True,
335
+ False,
336
+ }
337
+ assert {
338
+ item["score_breakdown"]["task_completion"]["normalized"]
339
+ for item in feedback["recent_outcomes"]
340
+ } == {0.0, 1.0}
341
+
342
+
343
+ def test_decision_only_excludes_unmarked_question_policy(monkeypatch, capsys) -> None:
344
+ store = InMemoryStore()
345
+ monkeypatch.setattr(cli, "get_default_store", lambda: store)
346
+ question = SoftmaxPolicy.from_actions(
347
+ [Action(id="answer")], agent_id="scout", task_id="answer-question"
348
+ ).snapshot()
349
+ decision = SoftmaxPolicy.from_actions(
350
+ [Action(id="delegate")], agent_id="scout", task_id="choose-delegation"
351
+ ).snapshot()
352
+ decision.metadata = {
353
+ "policy_scope": "delegated_decision",
354
+ "decision_context": "Choose a delegate",
355
+ }
356
+ store.store_policy(question)
357
+ store.store_policy(decision)
358
+
359
+ assert cli.main(["tasks-list", "scout", "--decision-only"]) == 0
360
+ assert json.loads(capsys.readouterr().out) == [
361
+ {"id": "choose-delegation", "name": "choose-delegation"}
362
+ ]
363
+ assert (
364
+ cli.main(
365
+ [
366
+ "train",
367
+ "--agent-id",
368
+ "scout",
369
+ "--task-id",
370
+ "answer-question",
371
+ "--decision-only",
372
+ ]
373
+ )
374
+ == 2
375
+ )
376
+ result = json.loads(capsys.readouterr().out)
377
+ assert result["skipped"] == [
378
+ {"task_id": "answer-question", "reason": "not a delegated decision policy"}
379
+ ]
380
+
381
+
382
+ def test_decision_episode_registration_requires_marked_policy(
383
+ monkeypatch, capsys, tmp_path: Path
384
+ ) -> None:
385
+ store = InMemoryStore()
386
+ monkeypatch.setattr(cli, "get_default_store", lambda: store)
387
+ policy = SoftmaxPolicy.from_actions(
388
+ [Action(id="delegate")], agent_id="scout", task_id="choose-delegation"
389
+ ).snapshot()
390
+ policy.metadata = {
391
+ "policy_scope": "delegated_decision",
392
+ "decision_context": "Choose a delegate",
393
+ }
394
+ store.store_policy(policy)
395
+ episode_path = tmp_path / "decision.json"
396
+ episode_path.write_text(
397
+ json.dumps(
398
+ {
399
+ "policy_id": policy.id,
400
+ "policy_version": policy.version,
401
+ "action_id": "delegate",
402
+ "intent_summary": "Choose a delegate",
403
+ "expected_outcome": "Use the best delegate",
404
+ "execution_status": "completed",
405
+ "result_summary": "Delegation completed",
406
+ }
407
+ ),
408
+ encoding="utf-8",
409
+ )
410
+
411
+ assert (
412
+ cli.main(
413
+ [
414
+ "task-episode-register",
415
+ "--agent-id",
416
+ "scout",
417
+ "--task-id",
418
+ "choose-delegation",
419
+ "--episode",
420
+ str(episode_path),
421
+ "--require-decision-policy",
422
+ ]
423
+ )
424
+ == 0
425
+ )
426
+ registered = json.loads(capsys.readouterr().out)
427
+ assert store.get_episode(registered["id"], "scout") is not None
428
+
429
+ invalid_path = tmp_path / "invalid-decision.json"
430
+ invalid_payload = json.loads(episode_path.read_text(encoding="utf-8"))
431
+ invalid_payload["metadata"] = {"correct_action_id": "outside-policy"}
432
+ invalid_path.write_text(json.dumps(invalid_payload), encoding="utf-8")
433
+ assert (
434
+ cli.main(
435
+ [
436
+ "task-episode-register",
437
+ "--agent-id",
438
+ "scout",
439
+ "--task-id",
440
+ "choose-delegation",
441
+ "--episode",
442
+ str(invalid_path),
443
+ "--require-decision-policy",
444
+ ]
445
+ )
446
+ == 2
447
+ )
448
+ assert "correct_action_id" in capsys.readouterr().err
449
+
450
+
212
451
  def test_score_uses_local_stdlib_without_configuration(
213
452
  monkeypatch, capsys
214
453
  ) -> None:
File without changes
File without changes