agent-learning 0.4.3__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agent_learning-0.4.3/src/agent_learning.egg-info → agent_learning-0.5.0}/PKG-INFO +9 -1
- {agent_learning-0.4.3 → agent_learning-0.5.0}/PYPI.md +8 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/README.md +15 -4
- {agent_learning-0.4.3 → agent_learning-0.5.0}/pyproject.toml +1 -1
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/_version.py +1 -1
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/cli.py +231 -4
- {agent_learning-0.4.3 → agent_learning-0.5.0/src/agent_learning.egg-info}/PKG-INFO +9 -1
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_cli.py +241 -2
- {agent_learning-0.4.3 → agent_learning-0.5.0}/LICENSE +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/setup.cfg +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/__init__.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/capture.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/__init__.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/base.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/router.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/scorers/__init__.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/scorers/_base.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/scorers/adherence.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/scorers/completion.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/scorers/intent.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/config.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/learners/__init__.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/learners/base.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/learners/reinforce.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/metrics/__init__.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/metrics/base.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/metrics/intent_resolution.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/metrics/local.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/metrics/registry.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/metrics/task_adherence.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/metrics/task_completion.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/policy/__init__.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/policy/base.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/policy/contextual_softmax.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/policy/softmax_bandit.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/py.typed +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/rewards/__init__.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/rewards/shaping.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/rewards/writer.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/__init__.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/base.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/llm/__init__.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/llm/_base.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/llm/adherence.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/llm/completion.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/llm/intent.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp/__init__.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp/_base.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp/adherence.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp/completion.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp/intent.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp_text/__init__.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp_text/_base.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp_text/adherence.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp_text/completion.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp_text/intent.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/slm/__init__.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/slm/_base.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/slm/adherence.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/slm/completion.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/slm/intent.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/stdlib/__init__.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/stdlib/_text.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/stdlib/adherence.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/stdlib/completion.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/stdlib/intent.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/storage/__init__.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/storage/base.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/storage/cosmos.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/storage/local.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/storage/memory.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/training/__init__.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/training/runner.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/types.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning.egg-info/SOURCES.txt +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning.egg-info/dependency_links.txt +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning.egg-info/entry_points.txt +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning.egg-info/requires.txt +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning.egg-info/top_level.txt +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_capture.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_contextual_policy.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_default_store.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_end_to_end.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_learner.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_metrics_local.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_policy.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_scorers.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_scorers_llm.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_scorers_nlp_text.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_scorers_slm.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_scorers_stdlib.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_shaping.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_storage_local.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_storage_memory.py +0 -0
- {agent_learning-0.4.3 → agent_learning-0.5.0}/tests/test_types.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: agent-learning
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default.
|
|
5
5
|
Author: Chris Tava
|
|
6
6
|
License: MIT License
|
|
@@ -69,6 +69,10 @@ Dynamic: license-file
|
|
|
69
69
|
|
|
70
70
|
Native reinforcement learning SDK for AI agents. An in-process Learner optimizes a small, interpretable TaskPolicy over discrete agent choices (understand intent and complete task by choosing the right outcome).
|
|
71
71
|
|
|
72
|
+
TaskPolicies model reusable decisions among executable alternatives such as
|
|
73
|
+
models, skills, tools, workflows, or workloads. Factual questions, ordinary
|
|
74
|
+
chat, reporting, and learning automation are not policy tasks.
|
|
75
|
+
|
|
72
76
|
## How it works
|
|
73
77
|
|
|
74
78
|
The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tune jobs and no opaque update cycles — just three pieces that run in your existing Python process:
|
|
@@ -79,4 +83,8 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
|
|
|
79
83
|
|
|
80
84
|
3. **Learner** applies REINFORCE-with-baseline to update TaskPolicy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
|
|
81
85
|
|
|
86
|
+
`task-policy-decide` closes the loop at execution time by returning the selected
|
|
87
|
+
action plus historical correctness, reward, result summaries, and per-metric
|
|
88
|
+
quality feedback for the agent to use on its next delegated decision.
|
|
89
|
+
|
|
82
90
|
Every episode, reward, run, and deployment is captured by the configured store — in-memory or local files by default, or Azure Cosmos DB — giving you a complete lineage and audit trail of how the policy evolved over time.
|
|
@@ -2,6 +2,10 @@
|
|
|
2
2
|
|
|
3
3
|
Native reinforcement learning SDK for AI agents. An in-process Learner optimizes a small, interpretable TaskPolicy over discrete agent choices (understand intent and complete task by choosing the right outcome).
|
|
4
4
|
|
|
5
|
+
TaskPolicies model reusable decisions among executable alternatives such as
|
|
6
|
+
models, skills, tools, workflows, or workloads. Factual questions, ordinary
|
|
7
|
+
chat, reporting, and learning automation are not policy tasks.
|
|
8
|
+
|
|
5
9
|
## How it works
|
|
6
10
|
|
|
7
11
|
The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tune jobs and no opaque update cycles — just three pieces that run in your existing Python process:
|
|
@@ -12,4 +16,8 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
|
|
|
12
16
|
|
|
13
17
|
3. **Learner** applies REINFORCE-with-baseline to update TaskPolicy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
|
|
14
18
|
|
|
19
|
+
`task-policy-decide` closes the loop at execution time by returning the selected
|
|
20
|
+
action plus historical correctness, reward, result summaries, and per-metric
|
|
21
|
+
quality feedback for the agent to use on its next delegated decision.
|
|
22
|
+
|
|
15
23
|
Every episode, reward, run, and deployment is captured by the configured store — in-memory or local files by default, or Azure Cosmos DB — giving you a complete lineage and audit trail of how the policy evolved over time.
|
|
@@ -4,6 +4,10 @@ Native reinforcement learning SDK for AI agents. An in-process
|
|
|
4
4
|
Learner optimizes a small, interpretable TaskPolicy over discrete agent choices (e.g., "take action A", "take action B", "take action C") using on-device evaluation scores as the reward
|
|
5
5
|
signal by default.
|
|
6
6
|
|
|
7
|
+
TaskPolicies represent reusable **decisions among executable alternatives**.
|
|
8
|
+
They are not conversation logs: factual questions, ordinary chat, reporting,
|
|
9
|
+
and agent-learning automation are not policy tasks.
|
|
10
|
+
|
|
7
11
|
<p align="center">
|
|
8
12
|
<img src="images/agent-learning-loop.svg" alt="Animated TaskPolicy Score Learner loop: TaskPolicy chooses a task action, Score evaluates the episode, and Learner updates TaskPolicy" width="960" style="max-width:100%; height:auto;" />
|
|
9
13
|
</p>
|
|
@@ -31,6 +35,11 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
|
|
|
31
35
|
|
|
32
36
|
<img src="images/cc970c453583c982.png" alt="Policy quality improves with every batch of episodes" width="360" style="max-width:100%; height:auto;" />
|
|
33
37
|
|
|
38
|
+
Before the next delegated execution, `task-policy-decide` samples the learned
|
|
39
|
+
policy and returns the selected action together with correctness rate, mean
|
|
40
|
+
reward, recent result summaries, and intent/adherence/completion scores. Agents
|
|
41
|
+
consume that feedback rather than training a policy that is never used.
|
|
42
|
+
|
|
34
43
|
Every episode, reward, run, and deployment is captured by the
|
|
35
44
|
configured store — in-memory or local files by default, or Azure Cosmos DB —
|
|
36
45
|
giving you a complete lineage and audit trail of how the policy
|
|
@@ -71,12 +80,14 @@ The `agent-learn` CLI provides the current task-learning-loop operations:
|
|
|
71
80
|
|
|
72
81
|
```text
|
|
73
82
|
agent-learn list
|
|
74
|
-
agent-learn
|
|
83
|
+
agent-learn --version
|
|
84
|
+
agent-learn tasks-list <agent_id> [--decision-only]
|
|
75
85
|
agent-learn task-episodes-count <agent_id> [--task-id <task_id>] [--start-date <date>] [--end-date <date>]
|
|
76
86
|
agent-learn task-episodes-list <agent_id> [--task-id <task_id>] [--limit <1-500>] [--include-incomplete] [--start-date <date>] [--end-date <date>]
|
|
77
|
-
agent-learn task-policy-init --agent-id <agent_id> --task-id <task_id> --actions ./actions.json
|
|
78
|
-
agent-learn task-
|
|
87
|
+
agent-learn task-policy-init --agent-id <agent_id> --task-id <task_id> --decision-context <context> --actions ./actions.json
|
|
88
|
+
agent-learn task-policy-decide --agent-id <agent_id> --task-id <task_id> [--history-limit <1-500>] [--greedy] [--seed <integer>]
|
|
89
|
+
agent-learn task-episode-register --agent-id <agent_id> --task-id <task_id> --episode ./episode.json [--require-decision-policy]
|
|
79
90
|
agent-learn score --agent-id <agent_id> [--task-id <task_id>] [--limit <1-500>]
|
|
80
|
-
agent-learn train --agent-id <agent_id> [--task-id <task_id>] [--limit <1-500>] [--min-episodes <1-500>] [--start-date <date>] [--end-date <date>] [--skip-scoring]
|
|
91
|
+
agent-learn train --agent-id <agent_id> [--task-id <task_id>] [--decision-only] [--limit <1-500>] [--min-episodes <1-500>] [--start-date <date>] [--end-date <date>] [--skip-scoring]
|
|
81
92
|
agent-learn task-policy --agent-id <agent_id> --task-id <task_id>
|
|
82
93
|
```
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "agent-learning"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.5.0"
|
|
8
8
|
description = "Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default."
|
|
9
9
|
readme = "PYPI.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -5,12 +5,15 @@ from __future__ import annotations
|
|
|
5
5
|
import argparse
|
|
6
6
|
import json
|
|
7
7
|
import logging
|
|
8
|
+
import math
|
|
9
|
+
import random
|
|
8
10
|
import sys
|
|
9
11
|
import uuid
|
|
10
12
|
from collections import Counter
|
|
11
13
|
from datetime import datetime, timezone
|
|
12
14
|
from typing import Any
|
|
13
15
|
|
|
16
|
+
from ._version import __version__
|
|
14
17
|
from .policy.softmax_bandit import SoftmaxPolicy
|
|
15
18
|
from .storage.cosmos import get_default_store
|
|
16
19
|
from .training.runner import LearningRunner
|
|
@@ -19,6 +22,7 @@ from .types import Action, Episode, MetricName, PolicySnapshot, RewardSource
|
|
|
19
22
|
logger = logging.getLogger(__name__)
|
|
20
23
|
|
|
21
24
|
_MAX_EPISODES = 500
|
|
25
|
+
_DECISION_POLICY_SCOPE = "delegated_decision"
|
|
22
26
|
|
|
23
27
|
|
|
24
28
|
def _episode_limit(value: str) -> int:
|
|
@@ -39,13 +43,18 @@ def _iso_date(value: str) -> str:
|
|
|
39
43
|
|
|
40
44
|
|
|
41
45
|
def _build_arg_parser() -> argparse.ArgumentParser:
|
|
42
|
-
parser = argparse.ArgumentParser(
|
|
46
|
+
parser = argparse.ArgumentParser(
|
|
47
|
+
prog="agent-learn",
|
|
48
|
+
description=f"Native RL CLI for AI agents. SDK version {__version__}.",
|
|
49
|
+
)
|
|
50
|
+
parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
|
|
43
51
|
sub = parser.add_subparsers(dest="command", required=True)
|
|
44
52
|
|
|
45
53
|
sub.add_parser("list", help="List discovered agent ids and names.")
|
|
46
54
|
|
|
47
55
|
tasks = sub.add_parser("tasks-list", help="List tasks for an agent.")
|
|
48
56
|
tasks.add_argument("agent_id")
|
|
57
|
+
tasks.add_argument("--decision-only", action="store_true")
|
|
49
58
|
|
|
50
59
|
count = sub.add_parser(
|
|
51
60
|
"task-episodes-count",
|
|
@@ -72,6 +81,7 @@ def _build_arg_parser() -> argparse.ArgumentParser:
|
|
|
72
81
|
train.add_argument("--task-id")
|
|
73
82
|
train.add_argument("--limit", type=_episode_limit, default=200)
|
|
74
83
|
train.add_argument("--min-episodes", type=_episode_limit, default=1)
|
|
84
|
+
train.add_argument("--decision-only", action="store_true")
|
|
75
85
|
train.add_argument("--start-date", type=_iso_date)
|
|
76
86
|
train.add_argument("--end-date", type=_iso_date)
|
|
77
87
|
train.add_argument(
|
|
@@ -89,12 +99,27 @@ def _build_arg_parser() -> argparse.ArgumentParser:
|
|
|
89
99
|
show.add_argument("--agent-id", required=True)
|
|
90
100
|
show.add_argument("--task-id", required=True)
|
|
91
101
|
|
|
102
|
+
decide = sub.add_parser(
|
|
103
|
+
"task-policy-decide",
|
|
104
|
+
help="Choose a delegated decision action and return learned feedback.",
|
|
105
|
+
)
|
|
106
|
+
decide.add_argument("--agent-id", required=True)
|
|
107
|
+
decide.add_argument("--task-id", required=True)
|
|
108
|
+
decide.add_argument("--history-limit", type=_episode_limit, default=100)
|
|
109
|
+
decide.add_argument("--greedy", action="store_true")
|
|
110
|
+
decide.add_argument("--seed", type=int)
|
|
111
|
+
|
|
92
112
|
init = sub.add_parser(
|
|
93
113
|
"task-policy-init",
|
|
94
114
|
help="Create and activate the initial policy for an agent task.",
|
|
95
115
|
)
|
|
96
116
|
init.add_argument("--agent-id", required=True)
|
|
97
117
|
init.add_argument("--task-id", required=True)
|
|
118
|
+
init.add_argument(
|
|
119
|
+
"--decision-context",
|
|
120
|
+
required=True,
|
|
121
|
+
help="Stable description of the delegated choice this policy controls.",
|
|
122
|
+
)
|
|
98
123
|
init.add_argument(
|
|
99
124
|
"--actions",
|
|
100
125
|
required=True,
|
|
@@ -107,6 +132,7 @@ def _build_arg_parser() -> argparse.ArgumentParser:
|
|
|
107
132
|
)
|
|
108
133
|
register.add_argument("--agent-id", required=True)
|
|
109
134
|
register.add_argument("--task-id", required=True)
|
|
135
|
+
register.add_argument("--require-decision-policy", action="store_true")
|
|
110
136
|
register.add_argument(
|
|
111
137
|
"--episode",
|
|
112
138
|
required=True,
|
|
@@ -124,7 +150,14 @@ def _cmd_agents_list(args: argparse.Namespace) -> int:
|
|
|
124
150
|
|
|
125
151
|
|
|
126
152
|
def _cmd_agent_tasks_list(args: argparse.Namespace) -> int:
|
|
127
|
-
|
|
153
|
+
store = get_default_store()
|
|
154
|
+
tasks = store.list_agent_tasks(args.agent_id)
|
|
155
|
+
if args.decision_only:
|
|
156
|
+
tasks = [
|
|
157
|
+
task
|
|
158
|
+
for task in tasks
|
|
159
|
+
if _is_decision_policy(store.get_active_policy(args.agent_id, task.id))
|
|
160
|
+
]
|
|
128
161
|
print(json.dumps([{"id": task.id, "name": task.name} for task in tasks], indent=2))
|
|
129
162
|
return 0
|
|
130
163
|
|
|
@@ -199,6 +232,9 @@ def _cmd_train(args: argparse.Namespace) -> int:
|
|
|
199
232
|
if snapshot is None:
|
|
200
233
|
skipped.append({"task_id": task_id, "reason": "no active policy"})
|
|
201
234
|
continue
|
|
235
|
+
if args.decision_only and not _is_decision_policy(snapshot):
|
|
236
|
+
skipped.append({"task_id": task_id, "reason": "not a delegated decision policy"})
|
|
237
|
+
continue
|
|
202
238
|
episode_limit = episode_limits.get(task_id, 0)
|
|
203
239
|
if episode_limit == 0:
|
|
204
240
|
skipped.append({"task_id": task_id, "reason": "no episodes in selected batch"})
|
|
@@ -266,6 +302,89 @@ def _policy_payload(snapshot: PolicySnapshot) -> dict[str, Any]:
|
|
|
266
302
|
return payload
|
|
267
303
|
|
|
268
304
|
|
|
305
|
+
def _is_decision_policy(snapshot: PolicySnapshot | None) -> bool:
|
|
306
|
+
return bool(
|
|
307
|
+
snapshot
|
|
308
|
+
and snapshot.metadata.get("policy_scope") == _DECISION_POLICY_SCOPE
|
|
309
|
+
)
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def _latest_aggregate(store: Any, episode: Episode) -> float | None:
|
|
313
|
+
rewards = [
|
|
314
|
+
reward
|
|
315
|
+
for reward in store.get_rewards_for_episode(episode.id, episode.agent_id)
|
|
316
|
+
if reward.source == RewardSource.AGGREGATE
|
|
317
|
+
]
|
|
318
|
+
if not rewards:
|
|
319
|
+
return None
|
|
320
|
+
return max(rewards, key=lambda reward: reward.created_at).value
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _decision_feedback(
|
|
324
|
+
store: Any, snapshot: PolicySnapshot, history_limit: int
|
|
325
|
+
) -> dict[str, Any]:
|
|
326
|
+
stats = {
|
|
327
|
+
action.id: {
|
|
328
|
+
"attempts": 0,
|
|
329
|
+
"correctness_evaluated": 0,
|
|
330
|
+
"correct": 0,
|
|
331
|
+
"correctness_rate": None,
|
|
332
|
+
"rewarded_episodes": 0,
|
|
333
|
+
"mean_reward": None,
|
|
334
|
+
"recent_outcomes": [],
|
|
335
|
+
}
|
|
336
|
+
for action in snapshot.actions
|
|
337
|
+
}
|
|
338
|
+
reward_totals = {action.id: 0.0 for action in snapshot.actions}
|
|
339
|
+
episodes = store.query_episodes(
|
|
340
|
+
snapshot.agent_id,
|
|
341
|
+
task_id=snapshot.task_id,
|
|
342
|
+
limit=history_limit,
|
|
343
|
+
)
|
|
344
|
+
for episode in episodes:
|
|
345
|
+
action_id = episode.action_id
|
|
346
|
+
if action_id not in stats:
|
|
347
|
+
continue
|
|
348
|
+
action_stats = stats[action_id]
|
|
349
|
+
action_stats["attempts"] += 1
|
|
350
|
+
correct_action_id = episode.metadata.get("correct_action_id")
|
|
351
|
+
was_correct = None
|
|
352
|
+
if correct_action_id:
|
|
353
|
+
was_correct = action_id == correct_action_id
|
|
354
|
+
action_stats["correctness_evaluated"] += 1
|
|
355
|
+
action_stats["correct"] += int(was_correct)
|
|
356
|
+
reward = _latest_aggregate(store, episode)
|
|
357
|
+
if reward is not None:
|
|
358
|
+
action_stats["rewarded_episodes"] += 1
|
|
359
|
+
reward_totals[action_id] += reward
|
|
360
|
+
if len(action_stats["recent_outcomes"]) < 3:
|
|
361
|
+
score_breakdown = {}
|
|
362
|
+
for result in store.get_metric_results(episode.id, episode.agent_id):
|
|
363
|
+
score_breakdown[result.metric.value] = {
|
|
364
|
+
"normalized": result.normalized,
|
|
365
|
+
"status": result.status,
|
|
366
|
+
"reason": result.reason,
|
|
367
|
+
}
|
|
368
|
+
action_stats["recent_outcomes"].append(
|
|
369
|
+
{
|
|
370
|
+
"created_at": episode.created_at,
|
|
371
|
+
"was_correct": was_correct,
|
|
372
|
+
"reward": reward,
|
|
373
|
+
"execution_status": episode.execution_status,
|
|
374
|
+
"result_summary": episode.result_summary,
|
|
375
|
+
"score_breakdown": score_breakdown,
|
|
376
|
+
}
|
|
377
|
+
)
|
|
378
|
+
for action_id, action_stats in stats.items():
|
|
379
|
+
evaluated = action_stats["correctness_evaluated"]
|
|
380
|
+
rewarded = action_stats["rewarded_episodes"]
|
|
381
|
+
if evaluated:
|
|
382
|
+
action_stats["correctness_rate"] = action_stats["correct"] / evaluated
|
|
383
|
+
if rewarded:
|
|
384
|
+
action_stats["mean_reward"] = reward_totals[action_id] / rewarded
|
|
385
|
+
return {"episodes_reviewed": len(episodes), "actions": stats}
|
|
386
|
+
|
|
387
|
+
|
|
269
388
|
def _policy_difference(
|
|
270
389
|
current: PolicySnapshot, previous: PolicySnapshot | None
|
|
271
390
|
) -> dict[str, Any] | None:
|
|
@@ -326,6 +445,78 @@ def _cmd_show_task_policy(args: argparse.Namespace) -> int:
|
|
|
326
445
|
return 0
|
|
327
446
|
|
|
328
447
|
|
|
448
|
+
def _cmd_decide_task_policy(args: argparse.Namespace) -> int:
|
|
449
|
+
store = get_default_store()
|
|
450
|
+
snapshot = store.get_active_policy(args.agent_id, args.task_id)
|
|
451
|
+
if snapshot is None:
|
|
452
|
+
print(
|
|
453
|
+
f"No active policy found for agent_id={args.agent_id!r}, "
|
|
454
|
+
f"task_id={args.task_id!r}.",
|
|
455
|
+
file=sys.stderr,
|
|
456
|
+
)
|
|
457
|
+
return 2
|
|
458
|
+
if not _is_decision_policy(snapshot):
|
|
459
|
+
print(
|
|
460
|
+
"The active policy is not marked as a delegated decision policy. "
|
|
461
|
+
"Questions, reporting tasks, and agent-learning automation are not eligible.",
|
|
462
|
+
file=sys.stderr,
|
|
463
|
+
)
|
|
464
|
+
return 2
|
|
465
|
+
rng = random.Random(args.seed) if args.seed is not None else None
|
|
466
|
+
policy = SoftmaxPolicy.from_snapshot(snapshot, rng=rng)
|
|
467
|
+
probabilities = policy.probabilities()
|
|
468
|
+
recommended_index = max(range(len(probabilities)), key=probabilities.__getitem__)
|
|
469
|
+
if args.greedy:
|
|
470
|
+
selected_index = recommended_index
|
|
471
|
+
selected_action = snapshot.actions[selected_index]
|
|
472
|
+
selected_probability = probabilities[selected_index]
|
|
473
|
+
logprob = math.log(max(selected_probability, 1e-12))
|
|
474
|
+
mode = "greedy"
|
|
475
|
+
else:
|
|
476
|
+
decision = policy.choose()
|
|
477
|
+
selected_action = decision.action
|
|
478
|
+
selected_index = next(
|
|
479
|
+
index
|
|
480
|
+
for index, action in enumerate(snapshot.actions)
|
|
481
|
+
if action.id == selected_action.id
|
|
482
|
+
)
|
|
483
|
+
selected_probability = probabilities[selected_index]
|
|
484
|
+
logprob = decision.logprob
|
|
485
|
+
mode = "sampled"
|
|
486
|
+
feedback = _decision_feedback(store, snapshot, args.history_limit)
|
|
487
|
+
selected_stats = feedback["actions"][selected_action.id]
|
|
488
|
+
recommendation = snapshot.actions[recommended_index]
|
|
489
|
+
print(
|
|
490
|
+
json.dumps(
|
|
491
|
+
{
|
|
492
|
+
"agent_id": snapshot.agent_id,
|
|
493
|
+
"task_id": snapshot.task_id,
|
|
494
|
+
"decision_context": snapshot.metadata.get("decision_context"),
|
|
495
|
+
"policy_id": snapshot.id,
|
|
496
|
+
"policy_version": snapshot.version,
|
|
497
|
+
"selection_mode": mode,
|
|
498
|
+
"selected_action": {
|
|
499
|
+
**selected_action.to_dict(),
|
|
500
|
+
"probability": selected_probability,
|
|
501
|
+
"logprob": logprob,
|
|
502
|
+
},
|
|
503
|
+
"recommended_action": {
|
|
504
|
+
**recommendation.to_dict(),
|
|
505
|
+
"probability": probabilities[recommended_index],
|
|
506
|
+
},
|
|
507
|
+
"action_probabilities": {
|
|
508
|
+
action.id: probability
|
|
509
|
+
for action, probability in zip(snapshot.actions, probabilities)
|
|
510
|
+
},
|
|
511
|
+
"selected_action_feedback": selected_stats,
|
|
512
|
+
"historical_feedback": feedback,
|
|
513
|
+
},
|
|
514
|
+
indent=2,
|
|
515
|
+
)
|
|
516
|
+
)
|
|
517
|
+
return 0
|
|
518
|
+
|
|
519
|
+
|
|
329
520
|
def _cmd_init_task_policy(args: argparse.Namespace) -> int:
|
|
330
521
|
store = get_default_store()
|
|
331
522
|
if store.get_active_policy(args.agent_id, args.task_id) is not None:
|
|
@@ -341,20 +532,35 @@ def _cmd_init_task_policy(args: argparse.Namespace) -> int:
|
|
|
341
532
|
except (OSError, json.JSONDecodeError) as exc:
|
|
342
533
|
print(f"Unable to read --actions file: {exc}", file=sys.stderr)
|
|
343
534
|
return 2
|
|
344
|
-
if not isinstance(action_payloads, list) or
|
|
345
|
-
print(
|
|
535
|
+
if not isinstance(action_payloads, list) or len(action_payloads) < 2:
|
|
536
|
+
print(
|
|
537
|
+
"--actions file must contain at least two delegated decision actions",
|
|
538
|
+
file=sys.stderr,
|
|
539
|
+
)
|
|
346
540
|
return 2
|
|
347
541
|
try:
|
|
348
542
|
actions = [Action.from_dict(item) for item in action_payloads]
|
|
349
543
|
except (KeyError, TypeError, ValueError) as exc:
|
|
350
544
|
print(f"Invalid action definition: {exc}", file=sys.stderr)
|
|
351
545
|
return 2
|
|
546
|
+
action_ids = [action.id for action in actions]
|
|
547
|
+
if any(not action_id.strip() for action_id in action_ids) or len(set(action_ids)) != len(
|
|
548
|
+
action_ids
|
|
549
|
+
):
|
|
550
|
+
print("Decision action ids must be non-empty and unique", file=sys.stderr)
|
|
551
|
+
return 2
|
|
352
552
|
policy = SoftmaxPolicy.from_actions(
|
|
353
553
|
actions,
|
|
354
554
|
agent_id=args.agent_id,
|
|
355
555
|
task_id=args.task_id,
|
|
356
556
|
)
|
|
357
557
|
snapshot = policy.snapshot()
|
|
558
|
+
snapshot.metadata.update(
|
|
559
|
+
{
|
|
560
|
+
"policy_scope": _DECISION_POLICY_SCOPE,
|
|
561
|
+
"decision_context": args.decision_context,
|
|
562
|
+
}
|
|
563
|
+
)
|
|
358
564
|
store.store_policy(snapshot)
|
|
359
565
|
print(json.dumps(_policy_payload(snapshot), indent=2))
|
|
360
566
|
return 0
|
|
@@ -389,6 +595,26 @@ def _cmd_register_task_episode(args: argparse.Namespace) -> int:
|
|
|
389
595
|
print(f"Invalid episode definition: {exc}", file=sys.stderr)
|
|
390
596
|
return 2
|
|
391
597
|
|
|
598
|
+
if args.require_decision_policy:
|
|
599
|
+
policy = get_default_store().get_policy(episode.policy_id or "", args.agent_id)
|
|
600
|
+
if not _is_decision_policy(policy) or policy.task_id != args.task_id:
|
|
601
|
+
print(
|
|
602
|
+
"Episode registration requires a delegated decision policy and its policy_id.",
|
|
603
|
+
file=sys.stderr,
|
|
604
|
+
)
|
|
605
|
+
return 2
|
|
606
|
+
action_ids = {action.id for action in policy.actions}
|
|
607
|
+
if episode.action_id not in action_ids:
|
|
608
|
+
print("Episode action_id is not in the delegated decision policy.", file=sys.stderr)
|
|
609
|
+
return 2
|
|
610
|
+
correct_action_id = episode.metadata.get("correct_action_id")
|
|
611
|
+
if correct_action_id is not None and correct_action_id not in action_ids:
|
|
612
|
+
print(
|
|
613
|
+
"Episode metadata.correct_action_id is not in the delegated decision policy.",
|
|
614
|
+
file=sys.stderr,
|
|
615
|
+
)
|
|
616
|
+
return 2
|
|
617
|
+
|
|
392
618
|
get_default_store().store_episode(episode)
|
|
393
619
|
print(json.dumps(episode.to_dict(), indent=2))
|
|
394
620
|
return 0
|
|
@@ -406,6 +632,7 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
406
632
|
"train": _cmd_train,
|
|
407
633
|
"score": _cmd_score,
|
|
408
634
|
"task-policy": _cmd_show_task_policy,
|
|
635
|
+
"task-policy-decide": _cmd_decide_task_policy,
|
|
409
636
|
"task-policy-init": _cmd_init_task_policy,
|
|
410
637
|
"task-episode-register": _cmd_register_task_episode,
|
|
411
638
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: agent-learning
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: Native, in-process reinforcement learning SDK for AI agents — in-memory and local-file storage by default.
|
|
5
5
|
Author: Chris Tava
|
|
6
6
|
License: MIT License
|
|
@@ -69,6 +69,10 @@ Dynamic: license-file
|
|
|
69
69
|
|
|
70
70
|
Native reinforcement learning SDK for AI agents. An in-process Learner optimizes a small, interpretable TaskPolicy over discrete agent choices (understand intent and complete task by choosing the right outcome).
|
|
71
71
|
|
|
72
|
+
TaskPolicies model reusable decisions among executable alternatives such as
|
|
73
|
+
models, skills, tools, workflows, or workloads. Factual questions, ordinary
|
|
74
|
+
chat, reporting, and learning automation are not policy tasks.
|
|
75
|
+
|
|
72
76
|
## How it works
|
|
73
77
|
|
|
74
78
|
The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tune jobs and no opaque update cycles — just three pieces that run in your existing Python process:
|
|
@@ -79,4 +83,8 @@ The SDK improves agents without LLM weight fine-tuning. There are no GPU fine-tu
|
|
|
79
83
|
|
|
80
84
|
3. **Learner** applies REINFORCE-with-baseline to update TaskPolicy logits directly from logged episodes. Updates are tiny gradient steps that run on local compute and persist through a pluggable store — in-memory or local files by default, with Azure Cosmos DB optional.
|
|
81
85
|
|
|
86
|
+
`task-policy-decide` closes the loop at execution time by returning the selected
|
|
87
|
+
action plus historical correctness, reward, result summaries, and per-metric
|
|
88
|
+
quality feedback for the agent to use on its next delegated decision.
|
|
89
|
+
|
|
82
90
|
Every episode, reward, run, and deployment is captured by the configured store — in-memory or local files by default, or Azure Cosmos DB — giving you a complete lineage and audit trail of how the policy evolved over time.
|
|
@@ -7,7 +7,7 @@ from pathlib import Path
|
|
|
7
7
|
|
|
8
8
|
import pytest
|
|
9
9
|
|
|
10
|
-
from agent_learning import cli
|
|
10
|
+
from agent_learning import __version__, cli
|
|
11
11
|
from agent_learning.policy import SoftmaxPolicy
|
|
12
12
|
from agent_learning.storage import InMemoryStore
|
|
13
13
|
from agent_learning.types import (
|
|
@@ -35,6 +35,14 @@ def _full_episode() -> Episode:
|
|
|
35
35
|
)
|
|
36
36
|
|
|
37
37
|
|
|
38
|
+
def test_help_and_version_print_sdk_version(capsys) -> None:
|
|
39
|
+
assert f"SDK version {__version__}" in cli._build_arg_parser().format_help()
|
|
40
|
+
with pytest.raises(SystemExit) as exit_info:
|
|
41
|
+
cli.main(["--version"])
|
|
42
|
+
assert exit_info.value.code == 0
|
|
43
|
+
assert capsys.readouterr().out.strip() == f"agent-learn {__version__}"
|
|
44
|
+
|
|
45
|
+
|
|
38
46
|
def test_discovery_and_full_episode_count(monkeypatch, capsys) -> None:
|
|
39
47
|
store = InMemoryStore()
|
|
40
48
|
store.store_episode(_full_episode())
|
|
@@ -134,7 +142,12 @@ def test_task_policy_init_and_inspection(monkeypatch, capsys, tmp_path: Path) ->
|
|
|
134
142
|
monkeypatch.setattr(cli, "get_default_store", lambda: store)
|
|
135
143
|
actions_path = tmp_path / "actions.json"
|
|
136
144
|
actions_path.write_text(
|
|
137
|
-
json.dumps(
|
|
145
|
+
json.dumps(
|
|
146
|
+
[
|
|
147
|
+
{"id": "respond", "description": "Respond directly"},
|
|
148
|
+
{"id": "delegate", "description": "Delegate the response"},
|
|
149
|
+
]
|
|
150
|
+
),
|
|
138
151
|
encoding="utf-8",
|
|
139
152
|
)
|
|
140
153
|
|
|
@@ -146,6 +159,8 @@ def test_task_policy_init_and_inspection(monkeypatch, capsys, tmp_path: Path) ->
|
|
|
146
159
|
"agent-1",
|
|
147
160
|
"--task-id",
|
|
148
161
|
"chat",
|
|
162
|
+
"--decision-context",
|
|
163
|
+
"Choose how the agent should respond to a chat request",
|
|
149
164
|
"--actions",
|
|
150
165
|
str(actions_path),
|
|
151
166
|
]
|
|
@@ -154,6 +169,10 @@ def test_task_policy_init_and_inspection(monkeypatch, capsys, tmp_path: Path) ->
|
|
|
154
169
|
)
|
|
155
170
|
initialized = json.loads(capsys.readouterr().out)
|
|
156
171
|
assert initialized["task_id"] == "chat"
|
|
172
|
+
assert initialized["metadata"] == {
|
|
173
|
+
"policy_scope": "delegated_decision",
|
|
174
|
+
"decision_context": "Choose how the agent should respond to a chat request",
|
|
175
|
+
}
|
|
157
176
|
|
|
158
177
|
assert (
|
|
159
178
|
cli.main(
|
|
@@ -166,6 +185,33 @@ def test_task_policy_init_and_inspection(monkeypatch, capsys, tmp_path: Path) ->
|
|
|
166
185
|
assert inspected["previous_policy"] is None
|
|
167
186
|
|
|
168
187
|
|
|
188
|
+
def test_task_policy_init_requires_two_unique_decision_actions(
|
|
189
|
+
monkeypatch, capsys, tmp_path: Path
|
|
190
|
+
) -> None:
|
|
191
|
+
store = InMemoryStore()
|
|
192
|
+
monkeypatch.setattr(cli, "get_default_store", lambda: store)
|
|
193
|
+
actions_path = tmp_path / "actions.json"
|
|
194
|
+
actions_path.write_text(json.dumps([{"id": "only"}]), encoding="utf-8")
|
|
195
|
+
|
|
196
|
+
result = cli.main(
|
|
197
|
+
[
|
|
198
|
+
"task-policy-init",
|
|
199
|
+
"--agent-id",
|
|
200
|
+
"scout",
|
|
201
|
+
"--task-id",
|
|
202
|
+
"not-a-decision",
|
|
203
|
+
"--decision-context",
|
|
204
|
+
"There is only one action",
|
|
205
|
+
"--actions",
|
|
206
|
+
str(actions_path),
|
|
207
|
+
]
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
assert result == 2
|
|
211
|
+
assert "at least two" in capsys.readouterr().err
|
|
212
|
+
assert store.get_active_policy("scout", "not-a-decision") is None
|
|
213
|
+
|
|
214
|
+
|
|
169
215
|
def test_task_episode_register_persists_full_episode(
|
|
170
216
|
monkeypatch, capsys, tmp_path: Path
|
|
171
217
|
) -> None:
|
|
@@ -209,6 +255,199 @@ def test_task_episode_register_persists_full_episode(
|
|
|
209
255
|
assert episode.is_full
|
|
210
256
|
|
|
211
257
|
|
|
258
|
+
def test_task_policy_decide_returns_learned_feedback(monkeypatch, capsys) -> None:
|
|
259
|
+
store = InMemoryStore()
|
|
260
|
+
monkeypatch.setattr(cli, "get_default_store", lambda: store)
|
|
261
|
+
policy = SoftmaxPolicy.from_actions(
|
|
262
|
+
[Action(id="use_skill"), Action(id="use_model")],
|
|
263
|
+
agent_id="scout",
|
|
264
|
+
task_id="choose-delegation",
|
|
265
|
+
initial_logits={"use_skill": 1.0, "use_model": 0.0},
|
|
266
|
+
)
|
|
267
|
+
snapshot = policy.snapshot()
|
|
268
|
+
snapshot.metadata = {
|
|
269
|
+
"policy_scope": "delegated_decision",
|
|
270
|
+
"decision_context": "Choose whether to delegate to a skill or language model",
|
|
271
|
+
}
|
|
272
|
+
store.store_policy(snapshot)
|
|
273
|
+
outcomes = [
|
|
274
|
+
("correct", "use_skill", 0.8),
|
|
275
|
+
("incorrect", "use_model", -0.3),
|
|
276
|
+
]
|
|
277
|
+
for label, correct_action_id, reward_value in outcomes:
|
|
278
|
+
episode = Episode(
|
|
279
|
+
id=label,
|
|
280
|
+
agent_id="scout",
|
|
281
|
+
task_id="choose-delegation",
|
|
282
|
+
policy_id=snapshot.id,
|
|
283
|
+
action_id="use_skill",
|
|
284
|
+
execution_status="completed",
|
|
285
|
+
result_summary=label,
|
|
286
|
+
metadata={"correct_action_id": correct_action_id},
|
|
287
|
+
)
|
|
288
|
+
store.store_episode(episode)
|
|
289
|
+
store.store_metric_results(
|
|
290
|
+
episode.id,
|
|
291
|
+
episode.agent_id,
|
|
292
|
+
[
|
|
293
|
+
MetricResult(
|
|
294
|
+
metric=MetricName.TASK_COMPLETION,
|
|
295
|
+
score=1.0 if label == "correct" else 0.0,
|
|
296
|
+
normalized=1.0 if label == "correct" else 0.0,
|
|
297
|
+
status="completed",
|
|
298
|
+
reason=label,
|
|
299
|
+
)
|
|
300
|
+
],
|
|
301
|
+
)
|
|
302
|
+
store.store_reward(
|
|
303
|
+
Reward(
|
|
304
|
+
episode_id=episode.id,
|
|
305
|
+
agent_id=episode.agent_id,
|
|
306
|
+
source=RewardSource.AGGREGATE,
|
|
307
|
+
value=reward_value,
|
|
308
|
+
)
|
|
309
|
+
)
|
|
310
|
+
|
|
311
|
+
assert (
|
|
312
|
+
cli.main(
|
|
313
|
+
[
|
|
314
|
+
"task-policy-decide",
|
|
315
|
+
"--agent-id",
|
|
316
|
+
"scout",
|
|
317
|
+
"--task-id",
|
|
318
|
+
"choose-delegation",
|
|
319
|
+
"--greedy",
|
|
320
|
+
]
|
|
321
|
+
)
|
|
322
|
+
== 0
|
|
323
|
+
)
|
|
324
|
+
result = json.loads(capsys.readouterr().out)
|
|
325
|
+
|
|
326
|
+
assert result["selected_action"]["id"] == "use_skill"
|
|
327
|
+
assert result["selected_action"]["probability"] > 0.5
|
|
328
|
+
assert result["recommended_action"]["id"] == "use_skill"
|
|
329
|
+
feedback = result["selected_action_feedback"]
|
|
330
|
+
assert feedback["attempts"] == 2
|
|
331
|
+
assert feedback["correctness_rate"] == 0.5
|
|
332
|
+
assert feedback["mean_reward"] == pytest.approx(0.25)
|
|
333
|
+
assert {item["was_correct"] for item in feedback["recent_outcomes"]} == {
|
|
334
|
+
True,
|
|
335
|
+
False,
|
|
336
|
+
}
|
|
337
|
+
assert {
|
|
338
|
+
item["score_breakdown"]["task_completion"]["normalized"]
|
|
339
|
+
for item in feedback["recent_outcomes"]
|
|
340
|
+
} == {0.0, 1.0}
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def test_decision_only_excludes_unmarked_question_policy(monkeypatch, capsys) -> None:
|
|
344
|
+
store = InMemoryStore()
|
|
345
|
+
monkeypatch.setattr(cli, "get_default_store", lambda: store)
|
|
346
|
+
question = SoftmaxPolicy.from_actions(
|
|
347
|
+
[Action(id="answer")], agent_id="scout", task_id="answer-question"
|
|
348
|
+
).snapshot()
|
|
349
|
+
decision = SoftmaxPolicy.from_actions(
|
|
350
|
+
[Action(id="delegate")], agent_id="scout", task_id="choose-delegation"
|
|
351
|
+
).snapshot()
|
|
352
|
+
decision.metadata = {
|
|
353
|
+
"policy_scope": "delegated_decision",
|
|
354
|
+
"decision_context": "Choose a delegate",
|
|
355
|
+
}
|
|
356
|
+
store.store_policy(question)
|
|
357
|
+
store.store_policy(decision)
|
|
358
|
+
|
|
359
|
+
assert cli.main(["tasks-list", "scout", "--decision-only"]) == 0
|
|
360
|
+
assert json.loads(capsys.readouterr().out) == [
|
|
361
|
+
{"id": "choose-delegation", "name": "choose-delegation"}
|
|
362
|
+
]
|
|
363
|
+
assert (
|
|
364
|
+
cli.main(
|
|
365
|
+
[
|
|
366
|
+
"train",
|
|
367
|
+
"--agent-id",
|
|
368
|
+
"scout",
|
|
369
|
+
"--task-id",
|
|
370
|
+
"answer-question",
|
|
371
|
+
"--decision-only",
|
|
372
|
+
]
|
|
373
|
+
)
|
|
374
|
+
== 2
|
|
375
|
+
)
|
|
376
|
+
result = json.loads(capsys.readouterr().out)
|
|
377
|
+
assert result["skipped"] == [
|
|
378
|
+
{"task_id": "answer-question", "reason": "not a delegated decision policy"}
|
|
379
|
+
]
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
def test_decision_episode_registration_requires_marked_policy(
|
|
383
|
+
monkeypatch, capsys, tmp_path: Path
|
|
384
|
+
) -> None:
|
|
385
|
+
store = InMemoryStore()
|
|
386
|
+
monkeypatch.setattr(cli, "get_default_store", lambda: store)
|
|
387
|
+
policy = SoftmaxPolicy.from_actions(
|
|
388
|
+
[Action(id="delegate")], agent_id="scout", task_id="choose-delegation"
|
|
389
|
+
).snapshot()
|
|
390
|
+
policy.metadata = {
|
|
391
|
+
"policy_scope": "delegated_decision",
|
|
392
|
+
"decision_context": "Choose a delegate",
|
|
393
|
+
}
|
|
394
|
+
store.store_policy(policy)
|
|
395
|
+
episode_path = tmp_path / "decision.json"
|
|
396
|
+
episode_path.write_text(
|
|
397
|
+
json.dumps(
|
|
398
|
+
{
|
|
399
|
+
"policy_id": policy.id,
|
|
400
|
+
"policy_version": policy.version,
|
|
401
|
+
"action_id": "delegate",
|
|
402
|
+
"intent_summary": "Choose a delegate",
|
|
403
|
+
"expected_outcome": "Use the best delegate",
|
|
404
|
+
"execution_status": "completed",
|
|
405
|
+
"result_summary": "Delegation completed",
|
|
406
|
+
}
|
|
407
|
+
),
|
|
408
|
+
encoding="utf-8",
|
|
409
|
+
)
|
|
410
|
+
|
|
411
|
+
assert (
|
|
412
|
+
cli.main(
|
|
413
|
+
[
|
|
414
|
+
"task-episode-register",
|
|
415
|
+
"--agent-id",
|
|
416
|
+
"scout",
|
|
417
|
+
"--task-id",
|
|
418
|
+
"choose-delegation",
|
|
419
|
+
"--episode",
|
|
420
|
+
str(episode_path),
|
|
421
|
+
"--require-decision-policy",
|
|
422
|
+
]
|
|
423
|
+
)
|
|
424
|
+
== 0
|
|
425
|
+
)
|
|
426
|
+
registered = json.loads(capsys.readouterr().out)
|
|
427
|
+
assert store.get_episode(registered["id"], "scout") is not None
|
|
428
|
+
|
|
429
|
+
invalid_path = tmp_path / "invalid-decision.json"
|
|
430
|
+
invalid_payload = json.loads(episode_path.read_text(encoding="utf-8"))
|
|
431
|
+
invalid_payload["metadata"] = {"correct_action_id": "outside-policy"}
|
|
432
|
+
invalid_path.write_text(json.dumps(invalid_payload), encoding="utf-8")
|
|
433
|
+
assert (
|
|
434
|
+
cli.main(
|
|
435
|
+
[
|
|
436
|
+
"task-episode-register",
|
|
437
|
+
"--agent-id",
|
|
438
|
+
"scout",
|
|
439
|
+
"--task-id",
|
|
440
|
+
"choose-delegation",
|
|
441
|
+
"--episode",
|
|
442
|
+
str(invalid_path),
|
|
443
|
+
"--require-decision-policy",
|
|
444
|
+
]
|
|
445
|
+
)
|
|
446
|
+
== 2
|
|
447
|
+
)
|
|
448
|
+
assert "correct_action_id" in capsys.readouterr().err
|
|
449
|
+
|
|
450
|
+
|
|
212
451
|
def test_score_uses_local_stdlib_without_configuration(
|
|
213
452
|
monkeypatch, capsys
|
|
214
453
|
) -> None:
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/scorers/__init__.py
RENAMED
|
File without changes
|
{agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/scorers/_base.py
RENAMED
|
File without changes
|
{agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/scorers/adherence.py
RENAMED
|
File without changes
|
{agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/scorers/completion.py
RENAMED
|
File without changes
|
{agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/classifiers/scorers/intent.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/metrics/intent_resolution.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/policy/contextual_softmax.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp_text/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp_text/adherence.py
RENAMED
|
File without changes
|
{agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/nlp_text/completion.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/stdlib/adherence.py
RENAMED
|
File without changes
|
{agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning/scorers/stdlib/completion.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{agent_learning-0.4.3 → agent_learning-0.5.0}/src/agent_learning.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|