agentdog 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentdog/__init__.py +104 -0
- agentdog/case.py +26 -0
- agentdog/cli.py +130 -0
- agentdog/py.typed +0 -0
- agentdog/report.py +104 -0
- agentdog/runner.py +48 -0
- agentdog/scorers/__init__.py +41 -0
- agentdog/scorers/answer.py +105 -0
- agentdog/scorers/base.py +31 -0
- agentdog/scorers/efficiency.py +97 -0
- agentdog/scorers/grounding.py +125 -0
- agentdog/scorers/judge.py +94 -0
- agentdog/scorers/safety.py +130 -0
- agentdog/scorers/tools.py +152 -0
- agentdog/trace.py +105 -0
- agentdog-0.1.0.dist-info/METADATA +178 -0
- agentdog-0.1.0.dist-info/RECORD +20 -0
- agentdog-0.1.0.dist-info/WHEEL +4 -0
- agentdog-0.1.0.dist-info/entry_points.txt +2 -0
- agentdog-0.1.0.dist-info/licenses/LICENSE +21 -0
agentdog/__init__.py
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""
|
|
2
|
+
agentdog: lightweight evaluation toolkit for AI agents.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from .case import EvalRun, TestCase
|
|
6
|
+
from .report import CaseResult, Report, ScorerResult
|
|
7
|
+
from .runner import run
|
|
8
|
+
from .scorers import (
|
|
9
|
+
AnswerNotEmpty,
|
|
10
|
+
AvoidedTools,
|
|
11
|
+
CitedSource,
|
|
12
|
+
ContainsAnswer,
|
|
13
|
+
ExactAnswer,
|
|
14
|
+
ForbiddenContent,
|
|
15
|
+
GroundedInContext,
|
|
16
|
+
LLMJudge,
|
|
17
|
+
MaxRetries,
|
|
18
|
+
MaxToolCalls,
|
|
19
|
+
NoContextHallucination,
|
|
20
|
+
NoRiskyActionTaken,
|
|
21
|
+
NoSensitiveDataLeaked,
|
|
22
|
+
PromptInjectionResisted,
|
|
23
|
+
RegexAnswer,
|
|
24
|
+
ScoreResult,
|
|
25
|
+
Scorer,
|
|
26
|
+
ToolArgContains,
|
|
27
|
+
ToolArgEquals,
|
|
28
|
+
ToolCallOrder,
|
|
29
|
+
UnderCostLimit,
|
|
30
|
+
UnderLatencyLimit,
|
|
31
|
+
UnderTokenLimit,
|
|
32
|
+
UsedTools,
|
|
33
|
+
)
|
|
34
|
+
from .trace import AgentTrace, ToolCall
|
|
35
|
+
|
|
36
|
+
__version__ = "0.1.0"
|
|
37
|
+
|
|
38
|
+
__all__ = [
|
|
39
|
+
# trace
|
|
40
|
+
"AgentTrace",
|
|
41
|
+
"ToolCall",
|
|
42
|
+
# case
|
|
43
|
+
"TestCase",
|
|
44
|
+
"EvalRun",
|
|
45
|
+
# runner
|
|
46
|
+
"run",
|
|
47
|
+
# report
|
|
48
|
+
"Report",
|
|
49
|
+
"CaseResult",
|
|
50
|
+
"ScorerResult",
|
|
51
|
+
# scorers — base
|
|
52
|
+
"Scorer",
|
|
53
|
+
"ScoreResult",
|
|
54
|
+
# scorers — answer
|
|
55
|
+
"ContainsAnswer",
|
|
56
|
+
"ExactAnswer",
|
|
57
|
+
"RegexAnswer",
|
|
58
|
+
"ForbiddenContent",
|
|
59
|
+
"AnswerNotEmpty",
|
|
60
|
+
# scorers — tools
|
|
61
|
+
"UsedTools",
|
|
62
|
+
"AvoidedTools",
|
|
63
|
+
"ToolCallOrder",
|
|
64
|
+
"MaxToolCalls",
|
|
65
|
+
"ToolArgContains",
|
|
66
|
+
"ToolArgEquals",
|
|
67
|
+
# scorers — grounding
|
|
68
|
+
"GroundedInContext",
|
|
69
|
+
"CitedSource",
|
|
70
|
+
"NoContextHallucination",
|
|
71
|
+
# scorers — safety
|
|
72
|
+
"NoSensitiveDataLeaked",
|
|
73
|
+
"NoRiskyActionTaken",
|
|
74
|
+
"PromptInjectionResisted",
|
|
75
|
+
# scorers — efficiency
|
|
76
|
+
"UnderTokenLimit",
|
|
77
|
+
"UnderCostLimit",
|
|
78
|
+
"UnderLatencyLimit",
|
|
79
|
+
"MaxRetries",
|
|
80
|
+
# scorers — llm judge
|
|
81
|
+
"LLMJudge",
|
|
82
|
+
]
|
|
83
|
+
|
|
84
|
+
# ---------------------------------------------------------------------------
|
|
85
|
+
# Aspirational features — tracked here until implemented
|
|
86
|
+
# ---------------------------------------------------------------------------
|
|
87
|
+
#
|
|
88
|
+
# v2 candidates (medium effort, clear value):
|
|
89
|
+
# - human_in_the_loop_review: structured queue for human-reviewed cases
|
|
90
|
+
# - evaluation_packs: pre-built scorer bundles (rag_pack, security_pack, etc.)
|
|
91
|
+
# - html_reports: rich HTML report with drill-down per case
|
|
92
|
+
# - github_action: official GHA for agentdog run with PR comment integration
|
|
93
|
+
# - async_runner: parallel trace evaluation for large eval sets
|
|
94
|
+
#
|
|
95
|
+
# v3 candidates (higher effort or less certain):
|
|
96
|
+
# - framework_adapters: native converters for LangChain, LlamaIndex, OpenAI SDK traces
|
|
97
|
+
# - model_comparison: run same cases against multiple models, diff results
|
|
98
|
+
# - prompt_version_comparison: A/B eval across prompt variants
|
|
99
|
+
# - trace_replay: re-run a captured trace through a new model/prompt
|
|
100
|
+
# - synthetic_eval_generation: auto-generate eval cases from a prompt + schema
|
|
101
|
+
# - dataset_quality_checks: flag low-quality or duplicate eval cases
|
|
102
|
+
# - safety_governance_templates: pre-built packs for HIPAA, PCI, SOC2 patterns
|
|
103
|
+
# - rag_advanced: retrieval-specific metrics (NDCG, MRR, recall@k)
|
|
104
|
+
# - streaming_trace_support: capture and eval streaming agent runs
|
agentdog/case.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from typing import TYPE_CHECKING
|
|
5
|
+
|
|
6
|
+
if TYPE_CHECKING:
|
|
7
|
+
from agentdog.scorers.base import Scorer
|
|
8
|
+
from agentdog.trace import AgentTrace
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class TestCase:
|
|
13
|
+
"""Defines what to evaluate and how to evaluate it for a single agent input."""
|
|
14
|
+
|
|
15
|
+
name: str
|
|
16
|
+
scorers: list["Scorer"]
|
|
17
|
+
description: str = ""
|
|
18
|
+
tags: list[str] = field(default_factory=list)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class EvalRun:
|
|
23
|
+
"""A test case paired with the agent trace to evaluate against."""
|
|
24
|
+
|
|
25
|
+
case: TestCase
|
|
26
|
+
trace: "AgentTrace"
|
agentdog/cli.py
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import importlib.util
|
|
4
|
+
import json
|
|
5
|
+
import sys
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
import click
|
|
9
|
+
|
|
10
|
+
from .case import EvalRun
|
|
11
|
+
from .runner import run
|
|
12
|
+
from .trace import AgentTrace
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _load_module(path: Path):
|
|
16
|
+
spec = importlib.util.spec_from_file_location("_agentdog_module", path)
|
|
17
|
+
if spec is None or spec.loader is None:
|
|
18
|
+
raise click.ClickException(f"Cannot load module from {path}")
|
|
19
|
+
mod = importlib.util.module_from_spec(spec)
|
|
20
|
+
sys.path.insert(0, str(path.parent))
|
|
21
|
+
spec.loader.exec_module(mod) # type: ignore[arg-type]
|
|
22
|
+
return mod
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@click.group()
|
|
26
|
+
def main():
|
|
27
|
+
"""agentdog — lightweight evaluation for AI agents."""
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@main.command()
|
|
31
|
+
@click.argument("module", type=click.Path(exists=True, dir_okay=False, path_type=Path))
|
|
32
|
+
@click.option("--verbose", "-v", is_flag=True, help="Show scorer details for passing cases too.")
|
|
33
|
+
@click.option("--fail-fast", is_flag=True, help="Stop after the first failing case.")
|
|
34
|
+
@click.option("--tag", "tags", multiple=True, help="Only run cases matching these tags.")
|
|
35
|
+
@click.option("--json-out", type=click.Path(dir_okay=False, path_type=Path), help="Write JSON report to file.")
|
|
36
|
+
def run_cmd(module: Path, verbose: bool, fail_fast: bool, tags: tuple[str, ...], json_out: Path | None):
|
|
37
|
+
"""
|
|
38
|
+
Run evaluations defined in MODULE.
|
|
39
|
+
|
|
40
|
+
MODULE must be a Python file that exposes an `evals()` function returning
|
|
41
|
+
a list of EvalRun objects.
|
|
42
|
+
|
|
43
|
+
\b
|
|
44
|
+
Example module (my_evals.py):
|
|
45
|
+
from agentdog import EvalRun, TestCase, AgentTrace
|
|
46
|
+
from agentdog.scorers import ContainsAnswer, UsedTools
|
|
47
|
+
|
|
48
|
+
def evals():
|
|
49
|
+
trace = AgentTrace(input="What is 2+2?", output="The answer is 4.")
|
|
50
|
+
case = TestCase("basic-math", scorers=[ContainsAnswer(["4"])])
|
|
51
|
+
return [EvalRun(case=case, trace=trace)]
|
|
52
|
+
"""
|
|
53
|
+
mod = _load_module(module)
|
|
54
|
+
if not hasattr(mod, "evals"):
|
|
55
|
+
raise click.ClickException(f"{module} must define an evals() function")
|
|
56
|
+
|
|
57
|
+
eval_runs: list[EvalRun] = mod.evals()
|
|
58
|
+
|
|
59
|
+
if tags:
|
|
60
|
+
eval_runs = [er for er in eval_runs if set(er.case.tags) & set(tags)]
|
|
61
|
+
if not eval_runs:
|
|
62
|
+
click.echo(f"No cases matched tags: {list(tags)}")
|
|
63
|
+
return
|
|
64
|
+
|
|
65
|
+
if fail_fast:
|
|
66
|
+
filtered: list[EvalRun] = []
|
|
67
|
+
for er in eval_runs:
|
|
68
|
+
filtered.append(er)
|
|
69
|
+
sub = run([er])
|
|
70
|
+
if not sub.passed:
|
|
71
|
+
report = run(filtered)
|
|
72
|
+
report.print(verbose=verbose)
|
|
73
|
+
sys.exit(1)
|
|
74
|
+
eval_runs = filtered
|
|
75
|
+
|
|
76
|
+
report = run(eval_runs)
|
|
77
|
+
report.print(verbose=verbose)
|
|
78
|
+
|
|
79
|
+
if json_out:
|
|
80
|
+
_write_json_report(report, json_out)
|
|
81
|
+
click.echo(f"JSON report written to {json_out}")
|
|
82
|
+
|
|
83
|
+
sys.exit(0 if report.passed else 1)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@main.command()
|
|
87
|
+
@click.argument("trace_file", type=click.Path(exists=True, dir_okay=False, path_type=Path))
|
|
88
|
+
def inspect(trace_file: Path):
|
|
89
|
+
"""Pretty-print a trace JSON file."""
|
|
90
|
+
trace = AgentTrace.from_json(str(trace_file))
|
|
91
|
+
click.echo(f"\nInput: {trace.input}")
|
|
92
|
+
click.echo(f"Output: {trace.output}")
|
|
93
|
+
click.echo(f"Tools: {trace.tool_names or '(none)'}")
|
|
94
|
+
click.echo(f"Context: {len(trace.retrieved_context)} chunk(s)")
|
|
95
|
+
if trace.total_tokens is not None:
|
|
96
|
+
click.echo(f"Tokens: {trace.total_tokens}")
|
|
97
|
+
if trace.total_cost_usd is not None:
|
|
98
|
+
click.echo(f"Cost: ${trace.total_cost_usd:.4f}")
|
|
99
|
+
if trace.total_latency_ms is not None:
|
|
100
|
+
click.echo(f"Latency: {trace.total_latency_ms:.0f}ms")
|
|
101
|
+
click.echo()
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _write_json_report(report, path: Path) -> None:
|
|
105
|
+
data = {
|
|
106
|
+
"passed": report.passed,
|
|
107
|
+
"overall_score": report.overall_score,
|
|
108
|
+
"elapsed_ms": report.elapsed_ms,
|
|
109
|
+
"num_passed": report.num_passed,
|
|
110
|
+
"num_failed": report.num_failed,
|
|
111
|
+
"cases": [
|
|
112
|
+
{
|
|
113
|
+
"name": cr.case_name,
|
|
114
|
+
"passed": cr.passed,
|
|
115
|
+
"score": cr.score,
|
|
116
|
+
"tags": cr.tags,
|
|
117
|
+
"scorers": [
|
|
118
|
+
{
|
|
119
|
+
"name": sr.scorer_name,
|
|
120
|
+
"passed": sr.result.passed,
|
|
121
|
+
"score": sr.result.score,
|
|
122
|
+
"reason": sr.result.reason,
|
|
123
|
+
}
|
|
124
|
+
for sr in cr.scorer_results
|
|
125
|
+
],
|
|
126
|
+
}
|
|
127
|
+
for cr in report.case_results
|
|
128
|
+
],
|
|
129
|
+
}
|
|
130
|
+
path.write_text(json.dumps(data, indent=2))
|
agentdog/py.typed
ADDED
|
File without changes
|
agentdog/report.py
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import time
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
|
|
6
|
+
import click
|
|
7
|
+
|
|
8
|
+
from .scorers.base import ScoreResult
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class ScorerResult:
|
|
13
|
+
scorer_name: str
|
|
14
|
+
result: ScoreResult
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass
|
|
18
|
+
class CaseResult:
|
|
19
|
+
case_name: str
|
|
20
|
+
scorer_results: list[ScorerResult] = field(default_factory=list)
|
|
21
|
+
tags: list[str] = field(default_factory=list)
|
|
22
|
+
|
|
23
|
+
@property
|
|
24
|
+
def passed(self) -> bool:
|
|
25
|
+
return all(sr.result.passed for sr in self.scorer_results)
|
|
26
|
+
|
|
27
|
+
@property
|
|
28
|
+
def score(self) -> float:
|
|
29
|
+
if not self.scorer_results:
|
|
30
|
+
return 1.0
|
|
31
|
+
return sum(sr.result.score for sr in self.scorer_results) / len(self.scorer_results)
|
|
32
|
+
|
|
33
|
+
@property
|
|
34
|
+
def failures(self) -> list[ScorerResult]:
|
|
35
|
+
return [sr for sr in self.scorer_results if not sr.result.passed]
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class Report:
|
|
40
|
+
case_results: list[CaseResult] = field(default_factory=list)
|
|
41
|
+
elapsed_ms: float = 0.0
|
|
42
|
+
|
|
43
|
+
@property
|
|
44
|
+
def passed(self) -> bool:
|
|
45
|
+
return all(cr.passed for cr in self.case_results)
|
|
46
|
+
|
|
47
|
+
@property
|
|
48
|
+
def num_passed(self) -> int:
|
|
49
|
+
return sum(1 for cr in self.case_results if cr.passed)
|
|
50
|
+
|
|
51
|
+
@property
|
|
52
|
+
def num_failed(self) -> int:
|
|
53
|
+
return len(self.case_results) - self.num_passed
|
|
54
|
+
|
|
55
|
+
@property
|
|
56
|
+
def overall_score(self) -> float:
|
|
57
|
+
if not self.case_results:
|
|
58
|
+
return 1.0
|
|
59
|
+
return sum(cr.score for cr in self.case_results) / len(self.case_results)
|
|
60
|
+
|
|
61
|
+
def print(self, verbose: bool = False) -> None:
|
|
62
|
+
_print_report(self, verbose=verbose)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
# ── Rendering ────────────────────────────────────────────────────────────────
|
|
66
|
+
|
|
67
|
+
_PASS = click.style("PASS", fg="green", bold=True)
|
|
68
|
+
_FAIL = click.style("FAIL", fg="red", bold=True)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _print_report(report: Report, verbose: bool = False) -> None:
|
|
72
|
+
click.echo()
|
|
73
|
+
click.echo(click.style("=" * 60, fg="bright_black"))
|
|
74
|
+
click.echo(click.style(" agentdog results", bold=True))
|
|
75
|
+
click.echo(click.style("=" * 60, fg="bright_black"))
|
|
76
|
+
|
|
77
|
+
for cr in report.case_results:
|
|
78
|
+
status = _PASS if cr.passed else _FAIL
|
|
79
|
+
tag_str = f" [{', '.join(cr.tags)}]" if cr.tags else ""
|
|
80
|
+
click.echo(f"\n {status} {cr.case_name}{tag_str} (score: {cr.score:.2f})")
|
|
81
|
+
|
|
82
|
+
if verbose or not cr.passed:
|
|
83
|
+
for sr in cr.scorer_results:
|
|
84
|
+
icon = click.style("ok", fg="green") if sr.result.passed else click.style("!!", fg="red")
|
|
85
|
+
line = f" [{icon}] {sr.scorer_name}"
|
|
86
|
+
if sr.result.reason:
|
|
87
|
+
line += f" - {sr.result.reason}"
|
|
88
|
+
click.echo(line)
|
|
89
|
+
|
|
90
|
+
click.echo()
|
|
91
|
+
click.echo(click.style("-" * 60, fg="bright_black"))
|
|
92
|
+
|
|
93
|
+
summary_color = "green" if report.passed else "red"
|
|
94
|
+
click.echo(
|
|
95
|
+
click.style(
|
|
96
|
+
f" {report.num_passed}/{len(report.case_results)} cases passed"
|
|
97
|
+
f" | overall score: {report.overall_score:.2f}"
|
|
98
|
+
f" | {report.elapsed_ms:.0f}ms",
|
|
99
|
+
fg=summary_color,
|
|
100
|
+
bold=True,
|
|
101
|
+
)
|
|
102
|
+
)
|
|
103
|
+
click.echo(click.style("=" * 60, fg="bright_black"))
|
|
104
|
+
click.echo()
|
agentdog/runner.py
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import time
|
|
4
|
+
from typing import Sequence
|
|
5
|
+
|
|
6
|
+
from .case import EvalRun
|
|
7
|
+
from .report import CaseResult, Report, ScorerResult
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def run(eval_runs: Sequence[EvalRun], verbose: bool = False) -> Report:
|
|
11
|
+
"""
|
|
12
|
+
Execute all scorers for each EvalRun and return a Report.
|
|
13
|
+
|
|
14
|
+
Args:
|
|
15
|
+
eval_runs: Sequence of (TestCase, AgentTrace) pairs to evaluate.
|
|
16
|
+
verbose: If True, print detailed scorer output for passing cases too.
|
|
17
|
+
|
|
18
|
+
Returns:
|
|
19
|
+
Report with per-case and aggregate results.
|
|
20
|
+
"""
|
|
21
|
+
start = time.perf_counter()
|
|
22
|
+
case_results: list[CaseResult] = []
|
|
23
|
+
|
|
24
|
+
for er in eval_runs:
|
|
25
|
+
scorer_results: list[ScorerResult] = []
|
|
26
|
+
for scorer in er.case.scorers:
|
|
27
|
+
try:
|
|
28
|
+
result = scorer.score(er.trace)
|
|
29
|
+
except Exception as exc:
|
|
30
|
+
from .scorers.base import ScoreResult
|
|
31
|
+
|
|
32
|
+
result = ScoreResult(
|
|
33
|
+
passed=False,
|
|
34
|
+
score=0.0,
|
|
35
|
+
reason=f"Scorer raised exception: {exc}",
|
|
36
|
+
)
|
|
37
|
+
scorer_results.append(ScorerResult(scorer_name=scorer.name, result=result))
|
|
38
|
+
|
|
39
|
+
case_results.append(
|
|
40
|
+
CaseResult(
|
|
41
|
+
case_name=er.case.name,
|
|
42
|
+
scorer_results=scorer_results,
|
|
43
|
+
tags=er.case.tags,
|
|
44
|
+
)
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
elapsed_ms = (time.perf_counter() - start) * 1000
|
|
48
|
+
return Report(case_results=case_results, elapsed_ms=elapsed_ms)
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
from .answer import AnswerNotEmpty, ContainsAnswer, ExactAnswer, ForbiddenContent, RegexAnswer
|
|
2
|
+
from .base import ScoreResult, Scorer
|
|
3
|
+
from .efficiency import MaxRetries, UnderCostLimit, UnderLatencyLimit, UnderTokenLimit
|
|
4
|
+
from .grounding import CitedSource, GroundedInContext, NoContextHallucination
|
|
5
|
+
from .judge import LLMJudge
|
|
6
|
+
from .safety import NoRiskyActionTaken, NoSensitiveDataLeaked, PromptInjectionResisted
|
|
7
|
+
from .tools import AvoidedTools, MaxToolCalls, ToolArgContains, ToolArgEquals, ToolCallOrder, UsedTools
|
|
8
|
+
|
|
9
|
+
__all__ = [
|
|
10
|
+
# base
|
|
11
|
+
"Scorer",
|
|
12
|
+
"ScoreResult",
|
|
13
|
+
# answer
|
|
14
|
+
"ContainsAnswer",
|
|
15
|
+
"ExactAnswer",
|
|
16
|
+
"RegexAnswer",
|
|
17
|
+
"ForbiddenContent",
|
|
18
|
+
"AnswerNotEmpty",
|
|
19
|
+
# tools
|
|
20
|
+
"UsedTools",
|
|
21
|
+
"AvoidedTools",
|
|
22
|
+
"ToolCallOrder",
|
|
23
|
+
"MaxToolCalls",
|
|
24
|
+
"ToolArgContains",
|
|
25
|
+
"ToolArgEquals",
|
|
26
|
+
# grounding
|
|
27
|
+
"GroundedInContext",
|
|
28
|
+
"CitedSource",
|
|
29
|
+
"NoContextHallucination",
|
|
30
|
+
# safety
|
|
31
|
+
"NoSensitiveDataLeaked",
|
|
32
|
+
"NoRiskyActionTaken",
|
|
33
|
+
"PromptInjectionResisted",
|
|
34
|
+
# efficiency
|
|
35
|
+
"UnderTokenLimit",
|
|
36
|
+
"UnderCostLimit",
|
|
37
|
+
"UnderLatencyLimit",
|
|
38
|
+
"MaxRetries",
|
|
39
|
+
# llm judge
|
|
40
|
+
"LLMJudge",
|
|
41
|
+
]
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from typing import TYPE_CHECKING
|
|
5
|
+
|
|
6
|
+
from .base import ScoreResult, Scorer
|
|
7
|
+
|
|
8
|
+
if TYPE_CHECKING:
|
|
9
|
+
from agentdog.trace import AgentTrace
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class ContainsAnswer(Scorer):
|
|
13
|
+
"""Pass if the output contains all required substrings."""
|
|
14
|
+
|
|
15
|
+
def __init__(self, required: list[str], case_sensitive: bool = False):
|
|
16
|
+
self.required = required
|
|
17
|
+
self.case_sensitive = case_sensitive
|
|
18
|
+
|
|
19
|
+
def score(self, trace: "AgentTrace") -> ScoreResult:
|
|
20
|
+
output = trace.output if self.case_sensitive else trace.output.lower()
|
|
21
|
+
missing = []
|
|
22
|
+
for phrase in self.required:
|
|
23
|
+
needle = phrase if self.case_sensitive else phrase.lower()
|
|
24
|
+
if needle not in output:
|
|
25
|
+
missing.append(phrase)
|
|
26
|
+
passed = len(missing) == 0
|
|
27
|
+
return ScoreResult(
|
|
28
|
+
passed=passed,
|
|
29
|
+
score=1.0 - len(missing) / len(self.required),
|
|
30
|
+
reason="" if passed else f"Missing from output: {missing}",
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class ExactAnswer(Scorer):
|
|
35
|
+
"""Pass if the output exactly equals the expected string (after optional strip)."""
|
|
36
|
+
|
|
37
|
+
def __init__(self, expected: str, strip: bool = True):
|
|
38
|
+
self.expected = expected
|
|
39
|
+
self.strip = strip
|
|
40
|
+
|
|
41
|
+
def score(self, trace: "AgentTrace") -> ScoreResult:
|
|
42
|
+
actual = trace.output.strip() if self.strip else trace.output
|
|
43
|
+
expected = self.expected.strip() if self.strip else self.expected
|
|
44
|
+
passed = actual == expected
|
|
45
|
+
return ScoreResult(
|
|
46
|
+
passed=passed,
|
|
47
|
+
score=1.0 if passed else 0.0,
|
|
48
|
+
reason="" if passed else f"Expected {expected!r}, got {actual!r}",
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class RegexAnswer(Scorer):
|
|
53
|
+
"""Pass if the output matches the given regex pattern."""
|
|
54
|
+
|
|
55
|
+
def __init__(self, pattern: str, flags: int = re.IGNORECASE):
|
|
56
|
+
self.pattern = pattern
|
|
57
|
+
self.flags = flags
|
|
58
|
+
self._re = re.compile(pattern, flags)
|
|
59
|
+
|
|
60
|
+
def score(self, trace: "AgentTrace") -> ScoreResult:
|
|
61
|
+
match = self._re.search(trace.output)
|
|
62
|
+
passed = match is not None
|
|
63
|
+
return ScoreResult(
|
|
64
|
+
passed=passed,
|
|
65
|
+
score=1.0 if passed else 0.0,
|
|
66
|
+
reason="" if passed else f"Pattern {self.pattern!r} not found in output",
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class ForbiddenContent(Scorer):
|
|
71
|
+
"""Pass if the output contains none of the forbidden substrings."""
|
|
72
|
+
|
|
73
|
+
def __init__(self, forbidden: list[str], case_sensitive: bool = False):
|
|
74
|
+
self.forbidden = forbidden
|
|
75
|
+
self.case_sensitive = case_sensitive
|
|
76
|
+
|
|
77
|
+
def score(self, trace: "AgentTrace") -> ScoreResult:
|
|
78
|
+
output = trace.output if self.case_sensitive else trace.output.lower()
|
|
79
|
+
found = []
|
|
80
|
+
for phrase in self.forbidden:
|
|
81
|
+
needle = phrase if self.case_sensitive else phrase.lower()
|
|
82
|
+
if needle in output:
|
|
83
|
+
found.append(phrase)
|
|
84
|
+
passed = len(found) == 0
|
|
85
|
+
return ScoreResult(
|
|
86
|
+
passed=passed,
|
|
87
|
+
score=1.0 - len(found) / len(self.forbidden),
|
|
88
|
+
reason="" if passed else f"Forbidden content found: {found}",
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class AnswerNotEmpty(Scorer):
|
|
93
|
+
"""Pass if the output is non-empty after stripping whitespace."""
|
|
94
|
+
|
|
95
|
+
def __init__(self, min_chars: int = 1):
|
|
96
|
+
self.min_chars = min_chars
|
|
97
|
+
|
|
98
|
+
def score(self, trace: "AgentTrace") -> ScoreResult:
|
|
99
|
+
length = len(trace.output.strip())
|
|
100
|
+
passed = length >= self.min_chars
|
|
101
|
+
return ScoreResult(
|
|
102
|
+
passed=passed,
|
|
103
|
+
score=1.0 if passed else 0.0,
|
|
104
|
+
reason="" if passed else f"Output too short: {length} chars (min {self.min_chars})",
|
|
105
|
+
)
|
agentdog/scorers/base.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from abc import ABC, abstractmethod
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
from typing import TYPE_CHECKING
|
|
6
|
+
|
|
7
|
+
if TYPE_CHECKING:
|
|
8
|
+
from agentdog.trace import AgentTrace
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class ScoreResult:
|
|
13
|
+
passed: bool
|
|
14
|
+
score: float # 0.0 = complete failure, 1.0 = full pass
|
|
15
|
+
reason: str = ""
|
|
16
|
+
details: dict = field(default_factory=dict)
|
|
17
|
+
|
|
18
|
+
def __bool__(self) -> bool:
|
|
19
|
+
return self.passed
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class Scorer(ABC):
|
|
23
|
+
"""Base class for all scorers."""
|
|
24
|
+
|
|
25
|
+
@property
|
|
26
|
+
def name(self) -> str:
|
|
27
|
+
return self.__class__.__name__
|
|
28
|
+
|
|
29
|
+
@abstractmethod
|
|
30
|
+
def score(self, trace: "AgentTrace") -> ScoreResult:
|
|
31
|
+
...
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import TYPE_CHECKING
|
|
4
|
+
|
|
5
|
+
from .base import ScoreResult, Scorer
|
|
6
|
+
|
|
7
|
+
if TYPE_CHECKING:
|
|
8
|
+
from agentdog.trace import AgentTrace
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class UnderTokenLimit(Scorer):
|
|
12
|
+
"""Pass if total token usage is at or below the limit."""
|
|
13
|
+
|
|
14
|
+
def __init__(self, max_tokens: int):
|
|
15
|
+
self.max_tokens = max_tokens
|
|
16
|
+
|
|
17
|
+
def score(self, trace: "AgentTrace") -> ScoreResult:
|
|
18
|
+
if trace.total_tokens is None:
|
|
19
|
+
return ScoreResult(
|
|
20
|
+
passed=True,
|
|
21
|
+
score=1.0,
|
|
22
|
+
reason="Token count not available in trace (skipped)",
|
|
23
|
+
)
|
|
24
|
+
passed = trace.total_tokens <= self.max_tokens
|
|
25
|
+
ratio = trace.total_tokens / self.max_tokens
|
|
26
|
+
return ScoreResult(
|
|
27
|
+
passed=passed,
|
|
28
|
+
score=max(0.0, 1.0 - (ratio - 1.0)) if not passed else 1.0,
|
|
29
|
+
reason="" if passed else f"Token usage {trace.total_tokens} exceeds limit {self.max_tokens}",
|
|
30
|
+
details={"actual": trace.total_tokens, "max": self.max_tokens},
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class UnderCostLimit(Scorer):
|
|
35
|
+
"""Pass if total cost is at or below the USD limit."""
|
|
36
|
+
|
|
37
|
+
def __init__(self, max_cost_usd: float):
|
|
38
|
+
self.max_cost_usd = max_cost_usd
|
|
39
|
+
|
|
40
|
+
def score(self, trace: "AgentTrace") -> ScoreResult:
|
|
41
|
+
if trace.total_cost_usd is None:
|
|
42
|
+
return ScoreResult(
|
|
43
|
+
passed=True,
|
|
44
|
+
score=1.0,
|
|
45
|
+
reason="Cost not available in trace (skipped)",
|
|
46
|
+
)
|
|
47
|
+
passed = trace.total_cost_usd <= self.max_cost_usd
|
|
48
|
+
ratio = trace.total_cost_usd / self.max_cost_usd if self.max_cost_usd else float("inf")
|
|
49
|
+
return ScoreResult(
|
|
50
|
+
passed=passed,
|
|
51
|
+
score=max(0.0, 1.0 - (ratio - 1.0)) if not passed else 1.0,
|
|
52
|
+
reason="" if passed else (
|
|
53
|
+
f"Cost ${trace.total_cost_usd:.4f} exceeds limit ${self.max_cost_usd:.4f}"
|
|
54
|
+
),
|
|
55
|
+
details={"actual_usd": trace.total_cost_usd, "max_usd": self.max_cost_usd},
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class UnderLatencyLimit(Scorer):
|
|
60
|
+
"""Pass if total latency is at or below the limit in milliseconds."""
|
|
61
|
+
|
|
62
|
+
def __init__(self, max_latency_ms: float):
|
|
63
|
+
self.max_latency_ms = max_latency_ms
|
|
64
|
+
|
|
65
|
+
def score(self, trace: "AgentTrace") -> ScoreResult:
|
|
66
|
+
if trace.total_latency_ms is None:
|
|
67
|
+
return ScoreResult(
|
|
68
|
+
passed=True,
|
|
69
|
+
score=1.0,
|
|
70
|
+
reason="Latency not available in trace (skipped)",
|
|
71
|
+
)
|
|
72
|
+
passed = trace.total_latency_ms <= self.max_latency_ms
|
|
73
|
+
ratio = trace.total_latency_ms / self.max_latency_ms
|
|
74
|
+
return ScoreResult(
|
|
75
|
+
passed=passed,
|
|
76
|
+
score=max(0.0, 1.0 - (ratio - 1.0)) if not passed else 1.0,
|
|
77
|
+
reason="" if passed else (
|
|
78
|
+
f"Latency {trace.total_latency_ms:.0f}ms exceeds limit {self.max_latency_ms:.0f}ms"
|
|
79
|
+
),
|
|
80
|
+
details={"actual_ms": trace.total_latency_ms, "max_ms": self.max_latency_ms},
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
class MaxRetries(Scorer):
|
|
85
|
+
"""Pass if the number of retries is at or below the limit."""
|
|
86
|
+
|
|
87
|
+
def __init__(self, max_retries: int):
|
|
88
|
+
self.max_retries = max_retries
|
|
89
|
+
|
|
90
|
+
def score(self, trace: "AgentTrace") -> ScoreResult:
|
|
91
|
+
passed = trace.num_retries <= self.max_retries
|
|
92
|
+
return ScoreResult(
|
|
93
|
+
passed=passed,
|
|
94
|
+
score=1.0 if passed else max(0.0, 1.0 - (trace.num_retries - self.max_retries) / (self.max_retries + 1)),
|
|
95
|
+
reason="" if passed else f"Retries {trace.num_retries} exceeds max {self.max_retries}",
|
|
96
|
+
details={"actual": trace.num_retries, "max": self.max_retries},
|
|
97
|
+
)
|