agentdog 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentdog-0.1.0/.gitignore +29 -0
- agentdog-0.1.0/LICENSE +21 -0
- agentdog-0.1.0/PKG-INFO +178 -0
- agentdog-0.1.0/README.md +128 -0
- agentdog-0.1.0/agentdog/__init__.py +104 -0
- agentdog-0.1.0/agentdog/case.py +26 -0
- agentdog-0.1.0/agentdog/cli.py +130 -0
- agentdog-0.1.0/agentdog/py.typed +0 -0
- agentdog-0.1.0/agentdog/report.py +104 -0
- agentdog-0.1.0/agentdog/runner.py +48 -0
- agentdog-0.1.0/agentdog/scorers/__init__.py +41 -0
- agentdog-0.1.0/agentdog/scorers/answer.py +105 -0
- agentdog-0.1.0/agentdog/scorers/base.py +31 -0
- agentdog-0.1.0/agentdog/scorers/efficiency.py +97 -0
- agentdog-0.1.0/agentdog/scorers/grounding.py +125 -0
- agentdog-0.1.0/agentdog/scorers/judge.py +94 -0
- agentdog-0.1.0/agentdog/scorers/safety.py +130 -0
- agentdog-0.1.0/agentdog/scorers/tools.py +152 -0
- agentdog-0.1.0/agentdog/trace.py +105 -0
- agentdog-0.1.0/pyproject.toml +63 -0
- agentdog-0.1.0/tests/__init__.py +0 -0
- agentdog-0.1.0/tests/conftest.py +68 -0
- agentdog-0.1.0/tests/test_answer_scorers.py +86 -0
- agentdog-0.1.0/tests/test_efficiency_scorers.py +72 -0
- agentdog-0.1.0/tests/test_grounding_scorers.py +64 -0
- agentdog-0.1.0/tests/test_runner.py +66 -0
- agentdog-0.1.0/tests/test_safety_scorers.py +86 -0
- agentdog-0.1.0/tests/test_tool_scorers.py +112 -0
- agentdog-0.1.0/tests/test_trace.py +74 -0
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# Virtual environments
|
|
2
|
+
.venv/
|
|
3
|
+
venv/
|
|
4
|
+
env/
|
|
5
|
+
|
|
6
|
+
# Build artifacts
|
|
7
|
+
dist/
|
|
8
|
+
build/
|
|
9
|
+
*.egg-info/
|
|
10
|
+
|
|
11
|
+
# Python cache
|
|
12
|
+
__pycache__/
|
|
13
|
+
*.py[cod]
|
|
14
|
+
*.pyo
|
|
15
|
+
|
|
16
|
+
# Test cache
|
|
17
|
+
.pytest_cache/
|
|
18
|
+
.coverage
|
|
19
|
+
htmlcov/
|
|
20
|
+
|
|
21
|
+
# Type checking
|
|
22
|
+
.mypy_cache/
|
|
23
|
+
.ruff_cache/
|
|
24
|
+
|
|
25
|
+
# Reports
|
|
26
|
+
*.json
|
|
27
|
+
|
|
28
|
+
# Typical dev data
|
|
29
|
+
dev-data/
|
agentdog-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Sai Teja Erukude
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
agentdog-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: agentdog
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Lightweight evaluation toolkit for AI agents — test tool use, grounding, safety, and efficiency before production
|
|
5
|
+
Project-URL: Homepage, https://github.com/SaiTeja-Erukude/agentdog
|
|
6
|
+
Project-URL: Repository, https://github.com/SaiTeja-Erukude/agentdog
|
|
7
|
+
Project-URL: Bug Tracker, https://github.com/SaiTeja-Erukude/agentdog/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/SaiTeja-Erukude/agentdog/releases
|
|
9
|
+
Author: Sai Teja Erukude
|
|
10
|
+
License: MIT License
|
|
11
|
+
|
|
12
|
+
Copyright (c) 2026 Sai Teja Erukude
|
|
13
|
+
|
|
14
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
15
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
16
|
+
in the Software without restriction, including without limitation the rights
|
|
17
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
18
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
19
|
+
furnished to do so, subject to the following conditions:
|
|
20
|
+
|
|
21
|
+
The above copyright notice and this permission notice shall be included in all
|
|
22
|
+
copies or substantial portions of the Software.
|
|
23
|
+
|
|
24
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
25
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
26
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
27
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
28
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
29
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
30
|
+
SOFTWARE.
|
|
31
|
+
License-File: LICENSE
|
|
32
|
+
Keywords: agent,ai,evals,evaluation,llm,prompt-injection,rag,safety,testing,tool-use
|
|
33
|
+
Classifier: Development Status :: 3 - Alpha
|
|
34
|
+
Classifier: Intended Audience :: Developers
|
|
35
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
36
|
+
Classifier: Programming Language :: Python :: 3
|
|
37
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
38
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
39
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
40
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
41
|
+
Classifier: Topic :: Software Development :: Testing
|
|
42
|
+
Requires-Python: >=3.10
|
|
43
|
+
Requires-Dist: click>=8.0
|
|
44
|
+
Provides-Extra: dev
|
|
45
|
+
Requires-Dist: pytest-cov>=5.0; extra == 'dev'
|
|
46
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
47
|
+
Provides-Extra: llm-judge
|
|
48
|
+
Requires-Dist: openai>=1.0; extra == 'llm-judge'
|
|
49
|
+
Description-Content-Type: text/markdown
|
|
50
|
+
|
|
51
|
+
# agentdog
|
|
52
|
+
|
|
53
|
+
Lightweight evaluation toolkit for AI agents. `pytest` for agent behavior — test tool use, grounding, safety, and efficiency before production
|
|
54
|
+
|
|
55
|
+
```
|
|
56
|
+
pip install agentdog
|
|
57
|
+
pip install "agentdog[llm-judge]" # for LLMJudge scorer
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
---
|
|
61
|
+
|
|
62
|
+
## Quickstart
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
from agentdog import AgentTrace, ToolCall, TestCase, EvalRun, run
|
|
66
|
+
from agentdog import ContainsAnswer, UsedTools, AvoidedTools, UnderTokenLimit
|
|
67
|
+
|
|
68
|
+
trace = AgentTrace(
|
|
69
|
+
input="Summarize the Q3 report.",
|
|
70
|
+
output="Q3 revenue was $4.2M, up 12% YoY.",
|
|
71
|
+
tool_calls=[ToolCall(name="file_search", arguments={"query": "Q3 report"})],
|
|
72
|
+
retrieved_context=["Q3 revenue was $4.2M, growth 12% year over year."],
|
|
73
|
+
total_tokens=620,
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
case = TestCase(
|
|
77
|
+
name="q3-summary",
|
|
78
|
+
tags=["rag"],
|
|
79
|
+
scorers=[
|
|
80
|
+
ContainsAnswer(["4.2M", "12%"]),
|
|
81
|
+
UsedTools(["file_search"]),
|
|
82
|
+
AvoidedTools(["send_email"]),
|
|
83
|
+
UnderTokenLimit(max_tokens=1000),
|
|
84
|
+
],
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
report = run([EvalRun(case=case, trace=trace)])
|
|
88
|
+
report.print(verbose=True)
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
---
|
|
92
|
+
|
|
93
|
+
## CLI
|
|
94
|
+
|
|
95
|
+
Define an `evals()` function in any Python file that returns `list[EvalRun]`, then:
|
|
96
|
+
|
|
97
|
+
```bash
|
|
98
|
+
agentdog run my_evals.py # run all cases
|
|
99
|
+
agentdog run my_evals.py -v # verbose: show scorer details for passing cases
|
|
100
|
+
agentdog run my_evals.py --tag rag # filter by tag
|
|
101
|
+
agentdog run my_evals.py --json-out report.json # machine-readable output
|
|
102
|
+
agentdog inspect trace.json # pretty-print a trace file
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Exit code is `0` on full pass, `1` on any failure — CI-friendly by default.
|
|
106
|
+
|
|
107
|
+
---
|
|
108
|
+
|
|
109
|
+
## Scorers
|
|
110
|
+
|
|
111
|
+
| Category | Scorers |
|
|
112
|
+
|---|---|
|
|
113
|
+
| **Answer** | `ContainsAnswer` `ExactAnswer` `RegexAnswer` `ForbiddenContent` `AnswerNotEmpty` |
|
|
114
|
+
| **Tools** | `UsedTools` `AvoidedTools` `ToolCallOrder` `MaxToolCalls` `ToolArgContains` `ToolArgEquals` |
|
|
115
|
+
| **Grounding** | `GroundedInContext` `CitedSource` `NoContextHallucination` |
|
|
116
|
+
| **Safety** | `NoSensitiveDataLeaked` `NoRiskyActionTaken` `PromptInjectionResisted` |
|
|
117
|
+
| **Efficiency** | `UnderTokenLimit` `UnderCostLimit` `UnderLatencyLimit` `MaxRetries` |
|
|
118
|
+
| **LLM Judge** | `LLMJudge` — use only when deterministic checks aren't enough |
|
|
119
|
+
|
|
120
|
+
---
|
|
121
|
+
|
|
122
|
+
## Trace schema
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
AgentTrace(
|
|
126
|
+
input: str,
|
|
127
|
+
output: str,
|
|
128
|
+
tool_calls: list[ToolCall], # name, arguments, output, error, latency_ms
|
|
129
|
+
retrieved_context: list[str],
|
|
130
|
+
total_tokens: int | None,
|
|
131
|
+
total_cost_usd: float | None,
|
|
132
|
+
total_latency_ms: float | None,
|
|
133
|
+
num_retries: int,
|
|
134
|
+
metadata: dict,
|
|
135
|
+
)
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
Load/save:
|
|
139
|
+
|
|
140
|
+
```python
|
|
141
|
+
trace = AgentTrace.from_json("trace.json")
|
|
142
|
+
trace.to_json("trace.json")
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
---
|
|
146
|
+
|
|
147
|
+
## Custom scorer
|
|
148
|
+
|
|
149
|
+
```python
|
|
150
|
+
from agentdog.scorers.base import Scorer, ScoreResult
|
|
151
|
+
|
|
152
|
+
class AnswerStartsWith(Scorer):
|
|
153
|
+
def __init__(self, prefix: str):
|
|
154
|
+
self.prefix = prefix
|
|
155
|
+
|
|
156
|
+
def score(self, trace) -> ScoreResult:
|
|
157
|
+
passed = trace.output.startswith(self.prefix)
|
|
158
|
+
return ScoreResult(
|
|
159
|
+
passed=passed,
|
|
160
|
+
score=1.0 if passed else 0.0,
|
|
161
|
+
reason=f"Expected output to start with {self.prefix!r}",
|
|
162
|
+
)
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
---
|
|
166
|
+
|
|
167
|
+
## Example
|
|
168
|
+
|
|
169
|
+
See [`examples/sample_evals.py`](examples/sample_evals.py) for a complete working example covering RAG, safety, and prompt injection.
|
|
170
|
+
|
|
171
|
+
---
|
|
172
|
+
|
|
173
|
+
## Author
|
|
174
|
+
|
|
175
|
+
**Sai Teja Erukude**
|
|
176
|
+
[GitHub](https://github.com/SaiTeja-Erukude) · [agentdog](https://github.com/SaiTeja-Erukude/agentdog)
|
|
177
|
+
|
|
178
|
+
Licensed under the [MIT License](LICENSE).
|
agentdog-0.1.0/README.md
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
# agentdog
|
|
2
|
+
|
|
3
|
+
Lightweight evaluation toolkit for AI agents. `pytest` for agent behavior — test tool use, grounding, safety, and efficiency before production
|
|
4
|
+
|
|
5
|
+
```
|
|
6
|
+
pip install agentdog
|
|
7
|
+
pip install "agentdog[llm-judge]" # for LLMJudge scorer
|
|
8
|
+
```
|
|
9
|
+
|
|
10
|
+
---
|
|
11
|
+
|
|
12
|
+
## Quickstart
|
|
13
|
+
|
|
14
|
+
```python
|
|
15
|
+
from agentdog import AgentTrace, ToolCall, TestCase, EvalRun, run
|
|
16
|
+
from agentdog import ContainsAnswer, UsedTools, AvoidedTools, UnderTokenLimit
|
|
17
|
+
|
|
18
|
+
trace = AgentTrace(
|
|
19
|
+
input="Summarize the Q3 report.",
|
|
20
|
+
output="Q3 revenue was $4.2M, up 12% YoY.",
|
|
21
|
+
tool_calls=[ToolCall(name="file_search", arguments={"query": "Q3 report"})],
|
|
22
|
+
retrieved_context=["Q3 revenue was $4.2M, growth 12% year over year."],
|
|
23
|
+
total_tokens=620,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
case = TestCase(
|
|
27
|
+
name="q3-summary",
|
|
28
|
+
tags=["rag"],
|
|
29
|
+
scorers=[
|
|
30
|
+
ContainsAnswer(["4.2M", "12%"]),
|
|
31
|
+
UsedTools(["file_search"]),
|
|
32
|
+
AvoidedTools(["send_email"]),
|
|
33
|
+
UnderTokenLimit(max_tokens=1000),
|
|
34
|
+
],
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
report = run([EvalRun(case=case, trace=trace)])
|
|
38
|
+
report.print(verbose=True)
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
---
|
|
42
|
+
|
|
43
|
+
## CLI
|
|
44
|
+
|
|
45
|
+
Define an `evals()` function in any Python file that returns `list[EvalRun]`, then:
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
agentdog run my_evals.py # run all cases
|
|
49
|
+
agentdog run my_evals.py -v # verbose: show scorer details for passing cases
|
|
50
|
+
agentdog run my_evals.py --tag rag # filter by tag
|
|
51
|
+
agentdog run my_evals.py --json-out report.json # machine-readable output
|
|
52
|
+
agentdog inspect trace.json # pretty-print a trace file
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Exit code is `0` on full pass, `1` on any failure — CI-friendly by default.
|
|
56
|
+
|
|
57
|
+
---
|
|
58
|
+
|
|
59
|
+
## Scorers
|
|
60
|
+
|
|
61
|
+
| Category | Scorers |
|
|
62
|
+
|---|---|
|
|
63
|
+
| **Answer** | `ContainsAnswer` `ExactAnswer` `RegexAnswer` `ForbiddenContent` `AnswerNotEmpty` |
|
|
64
|
+
| **Tools** | `UsedTools` `AvoidedTools` `ToolCallOrder` `MaxToolCalls` `ToolArgContains` `ToolArgEquals` |
|
|
65
|
+
| **Grounding** | `GroundedInContext` `CitedSource` `NoContextHallucination` |
|
|
66
|
+
| **Safety** | `NoSensitiveDataLeaked` `NoRiskyActionTaken` `PromptInjectionResisted` |
|
|
67
|
+
| **Efficiency** | `UnderTokenLimit` `UnderCostLimit` `UnderLatencyLimit` `MaxRetries` |
|
|
68
|
+
| **LLM Judge** | `LLMJudge` — use only when deterministic checks aren't enough |
|
|
69
|
+
|
|
70
|
+
---
|
|
71
|
+
|
|
72
|
+
## Trace schema
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
AgentTrace(
|
|
76
|
+
input: str,
|
|
77
|
+
output: str,
|
|
78
|
+
tool_calls: list[ToolCall], # name, arguments, output, error, latency_ms
|
|
79
|
+
retrieved_context: list[str],
|
|
80
|
+
total_tokens: int | None,
|
|
81
|
+
total_cost_usd: float | None,
|
|
82
|
+
total_latency_ms: float | None,
|
|
83
|
+
num_retries: int,
|
|
84
|
+
metadata: dict,
|
|
85
|
+
)
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Load/save:
|
|
89
|
+
|
|
90
|
+
```python
|
|
91
|
+
trace = AgentTrace.from_json("trace.json")
|
|
92
|
+
trace.to_json("trace.json")
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
---
|
|
96
|
+
|
|
97
|
+
## Custom scorer
|
|
98
|
+
|
|
99
|
+
```python
|
|
100
|
+
from agentdog.scorers.base import Scorer, ScoreResult
|
|
101
|
+
|
|
102
|
+
class AnswerStartsWith(Scorer):
|
|
103
|
+
def __init__(self, prefix: str):
|
|
104
|
+
self.prefix = prefix
|
|
105
|
+
|
|
106
|
+
def score(self, trace) -> ScoreResult:
|
|
107
|
+
passed = trace.output.startswith(self.prefix)
|
|
108
|
+
return ScoreResult(
|
|
109
|
+
passed=passed,
|
|
110
|
+
score=1.0 if passed else 0.0,
|
|
111
|
+
reason=f"Expected output to start with {self.prefix!r}",
|
|
112
|
+
)
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
---
|
|
116
|
+
|
|
117
|
+
## Example
|
|
118
|
+
|
|
119
|
+
See [`examples/sample_evals.py`](examples/sample_evals.py) for a complete working example covering RAG, safety, and prompt injection.
|
|
120
|
+
|
|
121
|
+
---
|
|
122
|
+
|
|
123
|
+
## Author
|
|
124
|
+
|
|
125
|
+
**Sai Teja Erukude**
|
|
126
|
+
[GitHub](https://github.com/SaiTeja-Erukude) · [agentdog](https://github.com/SaiTeja-Erukude/agentdog)
|
|
127
|
+
|
|
128
|
+
Licensed under the [MIT License](LICENSE).
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""
|
|
2
|
+
agentdog: lightweight evaluation toolkit for AI agents.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from .case import EvalRun, TestCase
|
|
6
|
+
from .report import CaseResult, Report, ScorerResult
|
|
7
|
+
from .runner import run
|
|
8
|
+
from .scorers import (
|
|
9
|
+
AnswerNotEmpty,
|
|
10
|
+
AvoidedTools,
|
|
11
|
+
CitedSource,
|
|
12
|
+
ContainsAnswer,
|
|
13
|
+
ExactAnswer,
|
|
14
|
+
ForbiddenContent,
|
|
15
|
+
GroundedInContext,
|
|
16
|
+
LLMJudge,
|
|
17
|
+
MaxRetries,
|
|
18
|
+
MaxToolCalls,
|
|
19
|
+
NoContextHallucination,
|
|
20
|
+
NoRiskyActionTaken,
|
|
21
|
+
NoSensitiveDataLeaked,
|
|
22
|
+
PromptInjectionResisted,
|
|
23
|
+
RegexAnswer,
|
|
24
|
+
ScoreResult,
|
|
25
|
+
Scorer,
|
|
26
|
+
ToolArgContains,
|
|
27
|
+
ToolArgEquals,
|
|
28
|
+
ToolCallOrder,
|
|
29
|
+
UnderCostLimit,
|
|
30
|
+
UnderLatencyLimit,
|
|
31
|
+
UnderTokenLimit,
|
|
32
|
+
UsedTools,
|
|
33
|
+
)
|
|
34
|
+
from .trace import AgentTrace, ToolCall
|
|
35
|
+
|
|
36
|
+
__version__ = "0.1.0"
|
|
37
|
+
|
|
38
|
+
__all__ = [
|
|
39
|
+
# trace
|
|
40
|
+
"AgentTrace",
|
|
41
|
+
"ToolCall",
|
|
42
|
+
# case
|
|
43
|
+
"TestCase",
|
|
44
|
+
"EvalRun",
|
|
45
|
+
# runner
|
|
46
|
+
"run",
|
|
47
|
+
# report
|
|
48
|
+
"Report",
|
|
49
|
+
"CaseResult",
|
|
50
|
+
"ScorerResult",
|
|
51
|
+
# scorers — base
|
|
52
|
+
"Scorer",
|
|
53
|
+
"ScoreResult",
|
|
54
|
+
# scorers — answer
|
|
55
|
+
"ContainsAnswer",
|
|
56
|
+
"ExactAnswer",
|
|
57
|
+
"RegexAnswer",
|
|
58
|
+
"ForbiddenContent",
|
|
59
|
+
"AnswerNotEmpty",
|
|
60
|
+
# scorers — tools
|
|
61
|
+
"UsedTools",
|
|
62
|
+
"AvoidedTools",
|
|
63
|
+
"ToolCallOrder",
|
|
64
|
+
"MaxToolCalls",
|
|
65
|
+
"ToolArgContains",
|
|
66
|
+
"ToolArgEquals",
|
|
67
|
+
# scorers — grounding
|
|
68
|
+
"GroundedInContext",
|
|
69
|
+
"CitedSource",
|
|
70
|
+
"NoContextHallucination",
|
|
71
|
+
# scorers — safety
|
|
72
|
+
"NoSensitiveDataLeaked",
|
|
73
|
+
"NoRiskyActionTaken",
|
|
74
|
+
"PromptInjectionResisted",
|
|
75
|
+
# scorers — efficiency
|
|
76
|
+
"UnderTokenLimit",
|
|
77
|
+
"UnderCostLimit",
|
|
78
|
+
"UnderLatencyLimit",
|
|
79
|
+
"MaxRetries",
|
|
80
|
+
# scorers — llm judge
|
|
81
|
+
"LLMJudge",
|
|
82
|
+
]
|
|
83
|
+
|
|
84
|
+
# ---------------------------------------------------------------------------
|
|
85
|
+
# Aspirational features — tracked here until implemented
|
|
86
|
+
# ---------------------------------------------------------------------------
|
|
87
|
+
#
|
|
88
|
+
# v2 candidates (medium effort, clear value):
|
|
89
|
+
# - human_in_the_loop_review: structured queue for human-reviewed cases
|
|
90
|
+
# - evaluation_packs: pre-built scorer bundles (rag_pack, security_pack, etc.)
|
|
91
|
+
# - html_reports: rich HTML report with drill-down per case
|
|
92
|
+
# - github_action: official GHA for agentdog run with PR comment integration
|
|
93
|
+
# - async_runner: parallel trace evaluation for large eval sets
|
|
94
|
+
#
|
|
95
|
+
# v3 candidates (higher effort or less certain):
|
|
96
|
+
# - framework_adapters: native converters for LangChain, LlamaIndex, OpenAI SDK traces
|
|
97
|
+
# - model_comparison: run same cases against multiple models, diff results
|
|
98
|
+
# - prompt_version_comparison: A/B eval across prompt variants
|
|
99
|
+
# - trace_replay: re-run a captured trace through a new model/prompt
|
|
100
|
+
# - synthetic_eval_generation: auto-generate eval cases from a prompt + schema
|
|
101
|
+
# - dataset_quality_checks: flag low-quality or duplicate eval cases
|
|
102
|
+
# - safety_governance_templates: pre-built packs for HIPAA, PCI, SOC2 patterns
|
|
103
|
+
# - rag_advanced: retrieval-specific metrics (NDCG, MRR, recall@k)
|
|
104
|
+
# - streaming_trace_support: capture and eval streaming agent runs
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from typing import TYPE_CHECKING
|
|
5
|
+
|
|
6
|
+
if TYPE_CHECKING:
|
|
7
|
+
from agentdog.scorers.base import Scorer
|
|
8
|
+
from agentdog.trace import AgentTrace
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class TestCase:
|
|
13
|
+
"""Defines what to evaluate and how to evaluate it for a single agent input."""
|
|
14
|
+
|
|
15
|
+
name: str
|
|
16
|
+
scorers: list["Scorer"]
|
|
17
|
+
description: str = ""
|
|
18
|
+
tags: list[str] = field(default_factory=list)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class EvalRun:
|
|
23
|
+
"""A test case paired with the agent trace to evaluate against."""
|
|
24
|
+
|
|
25
|
+
case: TestCase
|
|
26
|
+
trace: "AgentTrace"
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import importlib.util
|
|
4
|
+
import json
|
|
5
|
+
import sys
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
import click
|
|
9
|
+
|
|
10
|
+
from .case import EvalRun
|
|
11
|
+
from .runner import run
|
|
12
|
+
from .trace import AgentTrace
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _load_module(path: Path):
|
|
16
|
+
spec = importlib.util.spec_from_file_location("_agentdog_module", path)
|
|
17
|
+
if spec is None or spec.loader is None:
|
|
18
|
+
raise click.ClickException(f"Cannot load module from {path}")
|
|
19
|
+
mod = importlib.util.module_from_spec(spec)
|
|
20
|
+
sys.path.insert(0, str(path.parent))
|
|
21
|
+
spec.loader.exec_module(mod) # type: ignore[arg-type]
|
|
22
|
+
return mod
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@click.group()
|
|
26
|
+
def main():
|
|
27
|
+
"""agentdog — lightweight evaluation for AI agents."""
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@main.command()
|
|
31
|
+
@click.argument("module", type=click.Path(exists=True, dir_okay=False, path_type=Path))
|
|
32
|
+
@click.option("--verbose", "-v", is_flag=True, help="Show scorer details for passing cases too.")
|
|
33
|
+
@click.option("--fail-fast", is_flag=True, help="Stop after the first failing case.")
|
|
34
|
+
@click.option("--tag", "tags", multiple=True, help="Only run cases matching these tags.")
|
|
35
|
+
@click.option("--json-out", type=click.Path(dir_okay=False, path_type=Path), help="Write JSON report to file.")
|
|
36
|
+
def run_cmd(module: Path, verbose: bool, fail_fast: bool, tags: tuple[str, ...], json_out: Path | None):
|
|
37
|
+
"""
|
|
38
|
+
Run evaluations defined in MODULE.
|
|
39
|
+
|
|
40
|
+
MODULE must be a Python file that exposes an `evals()` function returning
|
|
41
|
+
a list of EvalRun objects.
|
|
42
|
+
|
|
43
|
+
\b
|
|
44
|
+
Example module (my_evals.py):
|
|
45
|
+
from agentdog import EvalRun, TestCase, AgentTrace
|
|
46
|
+
from agentdog.scorers import ContainsAnswer, UsedTools
|
|
47
|
+
|
|
48
|
+
def evals():
|
|
49
|
+
trace = AgentTrace(input="What is 2+2?", output="The answer is 4.")
|
|
50
|
+
case = TestCase("basic-math", scorers=[ContainsAnswer(["4"])])
|
|
51
|
+
return [EvalRun(case=case, trace=trace)]
|
|
52
|
+
"""
|
|
53
|
+
mod = _load_module(module)
|
|
54
|
+
if not hasattr(mod, "evals"):
|
|
55
|
+
raise click.ClickException(f"{module} must define an evals() function")
|
|
56
|
+
|
|
57
|
+
eval_runs: list[EvalRun] = mod.evals()
|
|
58
|
+
|
|
59
|
+
if tags:
|
|
60
|
+
eval_runs = [er for er in eval_runs if set(er.case.tags) & set(tags)]
|
|
61
|
+
if not eval_runs:
|
|
62
|
+
click.echo(f"No cases matched tags: {list(tags)}")
|
|
63
|
+
return
|
|
64
|
+
|
|
65
|
+
if fail_fast:
|
|
66
|
+
filtered: list[EvalRun] = []
|
|
67
|
+
for er in eval_runs:
|
|
68
|
+
filtered.append(er)
|
|
69
|
+
sub = run([er])
|
|
70
|
+
if not sub.passed:
|
|
71
|
+
report = run(filtered)
|
|
72
|
+
report.print(verbose=verbose)
|
|
73
|
+
sys.exit(1)
|
|
74
|
+
eval_runs = filtered
|
|
75
|
+
|
|
76
|
+
report = run(eval_runs)
|
|
77
|
+
report.print(verbose=verbose)
|
|
78
|
+
|
|
79
|
+
if json_out:
|
|
80
|
+
_write_json_report(report, json_out)
|
|
81
|
+
click.echo(f"JSON report written to {json_out}")
|
|
82
|
+
|
|
83
|
+
sys.exit(0 if report.passed else 1)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@main.command()
|
|
87
|
+
@click.argument("trace_file", type=click.Path(exists=True, dir_okay=False, path_type=Path))
|
|
88
|
+
def inspect(trace_file: Path):
|
|
89
|
+
"""Pretty-print a trace JSON file."""
|
|
90
|
+
trace = AgentTrace.from_json(str(trace_file))
|
|
91
|
+
click.echo(f"\nInput: {trace.input}")
|
|
92
|
+
click.echo(f"Output: {trace.output}")
|
|
93
|
+
click.echo(f"Tools: {trace.tool_names or '(none)'}")
|
|
94
|
+
click.echo(f"Context: {len(trace.retrieved_context)} chunk(s)")
|
|
95
|
+
if trace.total_tokens is not None:
|
|
96
|
+
click.echo(f"Tokens: {trace.total_tokens}")
|
|
97
|
+
if trace.total_cost_usd is not None:
|
|
98
|
+
click.echo(f"Cost: ${trace.total_cost_usd:.4f}")
|
|
99
|
+
if trace.total_latency_ms is not None:
|
|
100
|
+
click.echo(f"Latency: {trace.total_latency_ms:.0f}ms")
|
|
101
|
+
click.echo()
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _write_json_report(report, path: Path) -> None:
|
|
105
|
+
data = {
|
|
106
|
+
"passed": report.passed,
|
|
107
|
+
"overall_score": report.overall_score,
|
|
108
|
+
"elapsed_ms": report.elapsed_ms,
|
|
109
|
+
"num_passed": report.num_passed,
|
|
110
|
+
"num_failed": report.num_failed,
|
|
111
|
+
"cases": [
|
|
112
|
+
{
|
|
113
|
+
"name": cr.case_name,
|
|
114
|
+
"passed": cr.passed,
|
|
115
|
+
"score": cr.score,
|
|
116
|
+
"tags": cr.tags,
|
|
117
|
+
"scorers": [
|
|
118
|
+
{
|
|
119
|
+
"name": sr.scorer_name,
|
|
120
|
+
"passed": sr.result.passed,
|
|
121
|
+
"score": sr.result.score,
|
|
122
|
+
"reason": sr.result.reason,
|
|
123
|
+
}
|
|
124
|
+
for sr in cr.scorer_results
|
|
125
|
+
],
|
|
126
|
+
}
|
|
127
|
+
for cr in report.case_results
|
|
128
|
+
],
|
|
129
|
+
}
|
|
130
|
+
path.write_text(json.dumps(data, indent=2))
|
|
File without changes
|