agentdog 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
agentdog/__init__.py ADDED
@@ -0,0 +1,104 @@
1
+ """
2
+ agentdog: lightweight evaluation toolkit for AI agents.
3
+ """
4
+
5
+ from .case import EvalRun, TestCase
6
+ from .report import CaseResult, Report, ScorerResult
7
+ from .runner import run
8
+ from .scorers import (
9
+ AnswerNotEmpty,
10
+ AvoidedTools,
11
+ CitedSource,
12
+ ContainsAnswer,
13
+ ExactAnswer,
14
+ ForbiddenContent,
15
+ GroundedInContext,
16
+ LLMJudge,
17
+ MaxRetries,
18
+ MaxToolCalls,
19
+ NoContextHallucination,
20
+ NoRiskyActionTaken,
21
+ NoSensitiveDataLeaked,
22
+ PromptInjectionResisted,
23
+ RegexAnswer,
24
+ ScoreResult,
25
+ Scorer,
26
+ ToolArgContains,
27
+ ToolArgEquals,
28
+ ToolCallOrder,
29
+ UnderCostLimit,
30
+ UnderLatencyLimit,
31
+ UnderTokenLimit,
32
+ UsedTools,
33
+ )
34
+ from .trace import AgentTrace, ToolCall
35
+
36
+ __version__ = "0.1.0"
37
+
38
+ __all__ = [
39
+ # trace
40
+ "AgentTrace",
41
+ "ToolCall",
42
+ # case
43
+ "TestCase",
44
+ "EvalRun",
45
+ # runner
46
+ "run",
47
+ # report
48
+ "Report",
49
+ "CaseResult",
50
+ "ScorerResult",
51
+ # scorers — base
52
+ "Scorer",
53
+ "ScoreResult",
54
+ # scorers — answer
55
+ "ContainsAnswer",
56
+ "ExactAnswer",
57
+ "RegexAnswer",
58
+ "ForbiddenContent",
59
+ "AnswerNotEmpty",
60
+ # scorers — tools
61
+ "UsedTools",
62
+ "AvoidedTools",
63
+ "ToolCallOrder",
64
+ "MaxToolCalls",
65
+ "ToolArgContains",
66
+ "ToolArgEquals",
67
+ # scorers — grounding
68
+ "GroundedInContext",
69
+ "CitedSource",
70
+ "NoContextHallucination",
71
+ # scorers — safety
72
+ "NoSensitiveDataLeaked",
73
+ "NoRiskyActionTaken",
74
+ "PromptInjectionResisted",
75
+ # scorers — efficiency
76
+ "UnderTokenLimit",
77
+ "UnderCostLimit",
78
+ "UnderLatencyLimit",
79
+ "MaxRetries",
80
+ # scorers — llm judge
81
+ "LLMJudge",
82
+ ]
83
+
84
+ # ---------------------------------------------------------------------------
85
+ # Aspirational features — tracked here until implemented
86
+ # ---------------------------------------------------------------------------
87
+ #
88
+ # v2 candidates (medium effort, clear value):
89
+ # - human_in_the_loop_review: structured queue for human-reviewed cases
90
+ # - evaluation_packs: pre-built scorer bundles (rag_pack, security_pack, etc.)
91
+ # - html_reports: rich HTML report with drill-down per case
92
+ # - github_action: official GHA for agentdog run with PR comment integration
93
+ # - async_runner: parallel trace evaluation for large eval sets
94
+ #
95
+ # v3 candidates (higher effort or less certain):
96
+ # - framework_adapters: native converters for LangChain, LlamaIndex, OpenAI SDK traces
97
+ # - model_comparison: run same cases against multiple models, diff results
98
+ # - prompt_version_comparison: A/B eval across prompt variants
99
+ # - trace_replay: re-run a captured trace through a new model/prompt
100
+ # - synthetic_eval_generation: auto-generate eval cases from a prompt + schema
101
+ # - dataset_quality_checks: flag low-quality or duplicate eval cases
102
+ # - safety_governance_templates: pre-built packs for HIPAA, PCI, SOC2 patterns
103
+ # - rag_advanced: retrieval-specific metrics (NDCG, MRR, recall@k)
104
+ # - streaming_trace_support: capture and eval streaming agent runs
agentdog/case.py ADDED
@@ -0,0 +1,26 @@
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import dataclass, field
4
+ from typing import TYPE_CHECKING
5
+
6
+ if TYPE_CHECKING:
7
+ from agentdog.scorers.base import Scorer
8
+ from agentdog.trace import AgentTrace
9
+
10
+
11
+ @dataclass
12
+ class TestCase:
13
+ """Defines what to evaluate and how to evaluate it for a single agent input."""
14
+
15
+ name: str
16
+ scorers: list["Scorer"]
17
+ description: str = ""
18
+ tags: list[str] = field(default_factory=list)
19
+
20
+
21
+ @dataclass
22
+ class EvalRun:
23
+ """A test case paired with the agent trace to evaluate against."""
24
+
25
+ case: TestCase
26
+ trace: "AgentTrace"
agentdog/cli.py ADDED
@@ -0,0 +1,130 @@
1
+ from __future__ import annotations
2
+
3
+ import importlib.util
4
+ import json
5
+ import sys
6
+ from pathlib import Path
7
+
8
+ import click
9
+
10
+ from .case import EvalRun
11
+ from .runner import run
12
+ from .trace import AgentTrace
13
+
14
+
15
+ def _load_module(path: Path):
16
+ spec = importlib.util.spec_from_file_location("_agentdog_module", path)
17
+ if spec is None or spec.loader is None:
18
+ raise click.ClickException(f"Cannot load module from {path}")
19
+ mod = importlib.util.module_from_spec(spec)
20
+ sys.path.insert(0, str(path.parent))
21
+ spec.loader.exec_module(mod) # type: ignore[arg-type]
22
+ return mod
23
+
24
+
25
+ @click.group()
26
+ def main():
27
+ """agentdog — lightweight evaluation for AI agents."""
28
+
29
+
30
+ @main.command()
31
+ @click.argument("module", type=click.Path(exists=True, dir_okay=False, path_type=Path))
32
+ @click.option("--verbose", "-v", is_flag=True, help="Show scorer details for passing cases too.")
33
+ @click.option("--fail-fast", is_flag=True, help="Stop after the first failing case.")
34
+ @click.option("--tag", "tags", multiple=True, help="Only run cases matching these tags.")
35
+ @click.option("--json-out", type=click.Path(dir_okay=False, path_type=Path), help="Write JSON report to file.")
36
+ def run_cmd(module: Path, verbose: bool, fail_fast: bool, tags: tuple[str, ...], json_out: Path | None):
37
+ """
38
+ Run evaluations defined in MODULE.
39
+
40
+ MODULE must be a Python file that exposes an `evals()` function returning
41
+ a list of EvalRun objects.
42
+
43
+ \b
44
+ Example module (my_evals.py):
45
+ from agentdog import EvalRun, TestCase, AgentTrace
46
+ from agentdog.scorers import ContainsAnswer, UsedTools
47
+
48
+ def evals():
49
+ trace = AgentTrace(input="What is 2+2?", output="The answer is 4.")
50
+ case = TestCase("basic-math", scorers=[ContainsAnswer(["4"])])
51
+ return [EvalRun(case=case, trace=trace)]
52
+ """
53
+ mod = _load_module(module)
54
+ if not hasattr(mod, "evals"):
55
+ raise click.ClickException(f"{module} must define an evals() function")
56
+
57
+ eval_runs: list[EvalRun] = mod.evals()
58
+
59
+ if tags:
60
+ eval_runs = [er for er in eval_runs if set(er.case.tags) & set(tags)]
61
+ if not eval_runs:
62
+ click.echo(f"No cases matched tags: {list(tags)}")
63
+ return
64
+
65
+ if fail_fast:
66
+ filtered: list[EvalRun] = []
67
+ for er in eval_runs:
68
+ filtered.append(er)
69
+ sub = run([er])
70
+ if not sub.passed:
71
+ report = run(filtered)
72
+ report.print(verbose=verbose)
73
+ sys.exit(1)
74
+ eval_runs = filtered
75
+
76
+ report = run(eval_runs)
77
+ report.print(verbose=verbose)
78
+
79
+ if json_out:
80
+ _write_json_report(report, json_out)
81
+ click.echo(f"JSON report written to {json_out}")
82
+
83
+ sys.exit(0 if report.passed else 1)
84
+
85
+
86
+ @main.command()
87
+ @click.argument("trace_file", type=click.Path(exists=True, dir_okay=False, path_type=Path))
88
+ def inspect(trace_file: Path):
89
+ """Pretty-print a trace JSON file."""
90
+ trace = AgentTrace.from_json(str(trace_file))
91
+ click.echo(f"\nInput: {trace.input}")
92
+ click.echo(f"Output: {trace.output}")
93
+ click.echo(f"Tools: {trace.tool_names or '(none)'}")
94
+ click.echo(f"Context: {len(trace.retrieved_context)} chunk(s)")
95
+ if trace.total_tokens is not None:
96
+ click.echo(f"Tokens: {trace.total_tokens}")
97
+ if trace.total_cost_usd is not None:
98
+ click.echo(f"Cost: ${trace.total_cost_usd:.4f}")
99
+ if trace.total_latency_ms is not None:
100
+ click.echo(f"Latency: {trace.total_latency_ms:.0f}ms")
101
+ click.echo()
102
+
103
+
104
+ def _write_json_report(report, path: Path) -> None:
105
+ data = {
106
+ "passed": report.passed,
107
+ "overall_score": report.overall_score,
108
+ "elapsed_ms": report.elapsed_ms,
109
+ "num_passed": report.num_passed,
110
+ "num_failed": report.num_failed,
111
+ "cases": [
112
+ {
113
+ "name": cr.case_name,
114
+ "passed": cr.passed,
115
+ "score": cr.score,
116
+ "tags": cr.tags,
117
+ "scorers": [
118
+ {
119
+ "name": sr.scorer_name,
120
+ "passed": sr.result.passed,
121
+ "score": sr.result.score,
122
+ "reason": sr.result.reason,
123
+ }
124
+ for sr in cr.scorer_results
125
+ ],
126
+ }
127
+ for cr in report.case_results
128
+ ],
129
+ }
130
+ path.write_text(json.dumps(data, indent=2))
agentdog/py.typed ADDED
File without changes
agentdog/report.py ADDED
@@ -0,0 +1,104 @@
1
+ from __future__ import annotations
2
+
3
+ import time
4
+ from dataclasses import dataclass, field
5
+
6
+ import click
7
+
8
+ from .scorers.base import ScoreResult
9
+
10
+
11
+ @dataclass
12
+ class ScorerResult:
13
+ scorer_name: str
14
+ result: ScoreResult
15
+
16
+
17
+ @dataclass
18
+ class CaseResult:
19
+ case_name: str
20
+ scorer_results: list[ScorerResult] = field(default_factory=list)
21
+ tags: list[str] = field(default_factory=list)
22
+
23
+ @property
24
+ def passed(self) -> bool:
25
+ return all(sr.result.passed for sr in self.scorer_results)
26
+
27
+ @property
28
+ def score(self) -> float:
29
+ if not self.scorer_results:
30
+ return 1.0
31
+ return sum(sr.result.score for sr in self.scorer_results) / len(self.scorer_results)
32
+
33
+ @property
34
+ def failures(self) -> list[ScorerResult]:
35
+ return [sr for sr in self.scorer_results if not sr.result.passed]
36
+
37
+
38
+ @dataclass
39
+ class Report:
40
+ case_results: list[CaseResult] = field(default_factory=list)
41
+ elapsed_ms: float = 0.0
42
+
43
+ @property
44
+ def passed(self) -> bool:
45
+ return all(cr.passed for cr in self.case_results)
46
+
47
+ @property
48
+ def num_passed(self) -> int:
49
+ return sum(1 for cr in self.case_results if cr.passed)
50
+
51
+ @property
52
+ def num_failed(self) -> int:
53
+ return len(self.case_results) - self.num_passed
54
+
55
+ @property
56
+ def overall_score(self) -> float:
57
+ if not self.case_results:
58
+ return 1.0
59
+ return sum(cr.score for cr in self.case_results) / len(self.case_results)
60
+
61
+ def print(self, verbose: bool = False) -> None:
62
+ _print_report(self, verbose=verbose)
63
+
64
+
65
+ # ── Rendering ────────────────────────────────────────────────────────────────
66
+
67
+ _PASS = click.style("PASS", fg="green", bold=True)
68
+ _FAIL = click.style("FAIL", fg="red", bold=True)
69
+
70
+
71
+ def _print_report(report: Report, verbose: bool = False) -> None:
72
+ click.echo()
73
+ click.echo(click.style("=" * 60, fg="bright_black"))
74
+ click.echo(click.style(" agentdog results", bold=True))
75
+ click.echo(click.style("=" * 60, fg="bright_black"))
76
+
77
+ for cr in report.case_results:
78
+ status = _PASS if cr.passed else _FAIL
79
+ tag_str = f" [{', '.join(cr.tags)}]" if cr.tags else ""
80
+ click.echo(f"\n {status} {cr.case_name}{tag_str} (score: {cr.score:.2f})")
81
+
82
+ if verbose or not cr.passed:
83
+ for sr in cr.scorer_results:
84
+ icon = click.style("ok", fg="green") if sr.result.passed else click.style("!!", fg="red")
85
+ line = f" [{icon}] {sr.scorer_name}"
86
+ if sr.result.reason:
87
+ line += f" - {sr.result.reason}"
88
+ click.echo(line)
89
+
90
+ click.echo()
91
+ click.echo(click.style("-" * 60, fg="bright_black"))
92
+
93
+ summary_color = "green" if report.passed else "red"
94
+ click.echo(
95
+ click.style(
96
+ f" {report.num_passed}/{len(report.case_results)} cases passed"
97
+ f" | overall score: {report.overall_score:.2f}"
98
+ f" | {report.elapsed_ms:.0f}ms",
99
+ fg=summary_color,
100
+ bold=True,
101
+ )
102
+ )
103
+ click.echo(click.style("=" * 60, fg="bright_black"))
104
+ click.echo()
agentdog/runner.py ADDED
@@ -0,0 +1,48 @@
1
+ from __future__ import annotations
2
+
3
+ import time
4
+ from typing import Sequence
5
+
6
+ from .case import EvalRun
7
+ from .report import CaseResult, Report, ScorerResult
8
+
9
+
10
+ def run(eval_runs: Sequence[EvalRun], verbose: bool = False) -> Report:
11
+ """
12
+ Execute all scorers for each EvalRun and return a Report.
13
+
14
+ Args:
15
+ eval_runs: Sequence of (TestCase, AgentTrace) pairs to evaluate.
16
+ verbose: If True, print detailed scorer output for passing cases too.
17
+
18
+ Returns:
19
+ Report with per-case and aggregate results.
20
+ """
21
+ start = time.perf_counter()
22
+ case_results: list[CaseResult] = []
23
+
24
+ for er in eval_runs:
25
+ scorer_results: list[ScorerResult] = []
26
+ for scorer in er.case.scorers:
27
+ try:
28
+ result = scorer.score(er.trace)
29
+ except Exception as exc:
30
+ from .scorers.base import ScoreResult
31
+
32
+ result = ScoreResult(
33
+ passed=False,
34
+ score=0.0,
35
+ reason=f"Scorer raised exception: {exc}",
36
+ )
37
+ scorer_results.append(ScorerResult(scorer_name=scorer.name, result=result))
38
+
39
+ case_results.append(
40
+ CaseResult(
41
+ case_name=er.case.name,
42
+ scorer_results=scorer_results,
43
+ tags=er.case.tags,
44
+ )
45
+ )
46
+
47
+ elapsed_ms = (time.perf_counter() - start) * 1000
48
+ return Report(case_results=case_results, elapsed_ms=elapsed_ms)
@@ -0,0 +1,41 @@
1
+ from .answer import AnswerNotEmpty, ContainsAnswer, ExactAnswer, ForbiddenContent, RegexAnswer
2
+ from .base import ScoreResult, Scorer
3
+ from .efficiency import MaxRetries, UnderCostLimit, UnderLatencyLimit, UnderTokenLimit
4
+ from .grounding import CitedSource, GroundedInContext, NoContextHallucination
5
+ from .judge import LLMJudge
6
+ from .safety import NoRiskyActionTaken, NoSensitiveDataLeaked, PromptInjectionResisted
7
+ from .tools import AvoidedTools, MaxToolCalls, ToolArgContains, ToolArgEquals, ToolCallOrder, UsedTools
8
+
9
+ __all__ = [
10
+ # base
11
+ "Scorer",
12
+ "ScoreResult",
13
+ # answer
14
+ "ContainsAnswer",
15
+ "ExactAnswer",
16
+ "RegexAnswer",
17
+ "ForbiddenContent",
18
+ "AnswerNotEmpty",
19
+ # tools
20
+ "UsedTools",
21
+ "AvoidedTools",
22
+ "ToolCallOrder",
23
+ "MaxToolCalls",
24
+ "ToolArgContains",
25
+ "ToolArgEquals",
26
+ # grounding
27
+ "GroundedInContext",
28
+ "CitedSource",
29
+ "NoContextHallucination",
30
+ # safety
31
+ "NoSensitiveDataLeaked",
32
+ "NoRiskyActionTaken",
33
+ "PromptInjectionResisted",
34
+ # efficiency
35
+ "UnderTokenLimit",
36
+ "UnderCostLimit",
37
+ "UnderLatencyLimit",
38
+ "MaxRetries",
39
+ # llm judge
40
+ "LLMJudge",
41
+ ]
@@ -0,0 +1,105 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from typing import TYPE_CHECKING
5
+
6
+ from .base import ScoreResult, Scorer
7
+
8
+ if TYPE_CHECKING:
9
+ from agentdog.trace import AgentTrace
10
+
11
+
12
+ class ContainsAnswer(Scorer):
13
+ """Pass if the output contains all required substrings."""
14
+
15
+ def __init__(self, required: list[str], case_sensitive: bool = False):
16
+ self.required = required
17
+ self.case_sensitive = case_sensitive
18
+
19
+ def score(self, trace: "AgentTrace") -> ScoreResult:
20
+ output = trace.output if self.case_sensitive else trace.output.lower()
21
+ missing = []
22
+ for phrase in self.required:
23
+ needle = phrase if self.case_sensitive else phrase.lower()
24
+ if needle not in output:
25
+ missing.append(phrase)
26
+ passed = len(missing) == 0
27
+ return ScoreResult(
28
+ passed=passed,
29
+ score=1.0 - len(missing) / len(self.required),
30
+ reason="" if passed else f"Missing from output: {missing}",
31
+ )
32
+
33
+
34
+ class ExactAnswer(Scorer):
35
+ """Pass if the output exactly equals the expected string (after optional strip)."""
36
+
37
+ def __init__(self, expected: str, strip: bool = True):
38
+ self.expected = expected
39
+ self.strip = strip
40
+
41
+ def score(self, trace: "AgentTrace") -> ScoreResult:
42
+ actual = trace.output.strip() if self.strip else trace.output
43
+ expected = self.expected.strip() if self.strip else self.expected
44
+ passed = actual == expected
45
+ return ScoreResult(
46
+ passed=passed,
47
+ score=1.0 if passed else 0.0,
48
+ reason="" if passed else f"Expected {expected!r}, got {actual!r}",
49
+ )
50
+
51
+
52
+ class RegexAnswer(Scorer):
53
+ """Pass if the output matches the given regex pattern."""
54
+
55
+ def __init__(self, pattern: str, flags: int = re.IGNORECASE):
56
+ self.pattern = pattern
57
+ self.flags = flags
58
+ self._re = re.compile(pattern, flags)
59
+
60
+ def score(self, trace: "AgentTrace") -> ScoreResult:
61
+ match = self._re.search(trace.output)
62
+ passed = match is not None
63
+ return ScoreResult(
64
+ passed=passed,
65
+ score=1.0 if passed else 0.0,
66
+ reason="" if passed else f"Pattern {self.pattern!r} not found in output",
67
+ )
68
+
69
+
70
+ class ForbiddenContent(Scorer):
71
+ """Pass if the output contains none of the forbidden substrings."""
72
+
73
+ def __init__(self, forbidden: list[str], case_sensitive: bool = False):
74
+ self.forbidden = forbidden
75
+ self.case_sensitive = case_sensitive
76
+
77
+ def score(self, trace: "AgentTrace") -> ScoreResult:
78
+ output = trace.output if self.case_sensitive else trace.output.lower()
79
+ found = []
80
+ for phrase in self.forbidden:
81
+ needle = phrase if self.case_sensitive else phrase.lower()
82
+ if needle in output:
83
+ found.append(phrase)
84
+ passed = len(found) == 0
85
+ return ScoreResult(
86
+ passed=passed,
87
+ score=1.0 - len(found) / len(self.forbidden),
88
+ reason="" if passed else f"Forbidden content found: {found}",
89
+ )
90
+
91
+
92
+ class AnswerNotEmpty(Scorer):
93
+ """Pass if the output is non-empty after stripping whitespace."""
94
+
95
+ def __init__(self, min_chars: int = 1):
96
+ self.min_chars = min_chars
97
+
98
+ def score(self, trace: "AgentTrace") -> ScoreResult:
99
+ length = len(trace.output.strip())
100
+ passed = length >= self.min_chars
101
+ return ScoreResult(
102
+ passed=passed,
103
+ score=1.0 if passed else 0.0,
104
+ reason="" if passed else f"Output too short: {length} chars (min {self.min_chars})",
105
+ )
@@ -0,0 +1,31 @@
1
+ from __future__ import annotations
2
+
3
+ from abc import ABC, abstractmethod
4
+ from dataclasses import dataclass, field
5
+ from typing import TYPE_CHECKING
6
+
7
+ if TYPE_CHECKING:
8
+ from agentdog.trace import AgentTrace
9
+
10
+
11
+ @dataclass
12
+ class ScoreResult:
13
+ passed: bool
14
+ score: float # 0.0 = complete failure, 1.0 = full pass
15
+ reason: str = ""
16
+ details: dict = field(default_factory=dict)
17
+
18
+ def __bool__(self) -> bool:
19
+ return self.passed
20
+
21
+
22
+ class Scorer(ABC):
23
+ """Base class for all scorers."""
24
+
25
+ @property
26
+ def name(self) -> str:
27
+ return self.__class__.__name__
28
+
29
+ @abstractmethod
30
+ def score(self, trace: "AgentTrace") -> ScoreResult:
31
+ ...
@@ -0,0 +1,97 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import TYPE_CHECKING
4
+
5
+ from .base import ScoreResult, Scorer
6
+
7
+ if TYPE_CHECKING:
8
+ from agentdog.trace import AgentTrace
9
+
10
+
11
+ class UnderTokenLimit(Scorer):
12
+ """Pass if total token usage is at or below the limit."""
13
+
14
+ def __init__(self, max_tokens: int):
15
+ self.max_tokens = max_tokens
16
+
17
+ def score(self, trace: "AgentTrace") -> ScoreResult:
18
+ if trace.total_tokens is None:
19
+ return ScoreResult(
20
+ passed=True,
21
+ score=1.0,
22
+ reason="Token count not available in trace (skipped)",
23
+ )
24
+ passed = trace.total_tokens <= self.max_tokens
25
+ ratio = trace.total_tokens / self.max_tokens
26
+ return ScoreResult(
27
+ passed=passed,
28
+ score=max(0.0, 1.0 - (ratio - 1.0)) if not passed else 1.0,
29
+ reason="" if passed else f"Token usage {trace.total_tokens} exceeds limit {self.max_tokens}",
30
+ details={"actual": trace.total_tokens, "max": self.max_tokens},
31
+ )
32
+
33
+
34
+ class UnderCostLimit(Scorer):
35
+ """Pass if total cost is at or below the USD limit."""
36
+
37
+ def __init__(self, max_cost_usd: float):
38
+ self.max_cost_usd = max_cost_usd
39
+
40
+ def score(self, trace: "AgentTrace") -> ScoreResult:
41
+ if trace.total_cost_usd is None:
42
+ return ScoreResult(
43
+ passed=True,
44
+ score=1.0,
45
+ reason="Cost not available in trace (skipped)",
46
+ )
47
+ passed = trace.total_cost_usd <= self.max_cost_usd
48
+ ratio = trace.total_cost_usd / self.max_cost_usd if self.max_cost_usd else float("inf")
49
+ return ScoreResult(
50
+ passed=passed,
51
+ score=max(0.0, 1.0 - (ratio - 1.0)) if not passed else 1.0,
52
+ reason="" if passed else (
53
+ f"Cost ${trace.total_cost_usd:.4f} exceeds limit ${self.max_cost_usd:.4f}"
54
+ ),
55
+ details={"actual_usd": trace.total_cost_usd, "max_usd": self.max_cost_usd},
56
+ )
57
+
58
+
59
+ class UnderLatencyLimit(Scorer):
60
+ """Pass if total latency is at or below the limit in milliseconds."""
61
+
62
+ def __init__(self, max_latency_ms: float):
63
+ self.max_latency_ms = max_latency_ms
64
+
65
+ def score(self, trace: "AgentTrace") -> ScoreResult:
66
+ if trace.total_latency_ms is None:
67
+ return ScoreResult(
68
+ passed=True,
69
+ score=1.0,
70
+ reason="Latency not available in trace (skipped)",
71
+ )
72
+ passed = trace.total_latency_ms <= self.max_latency_ms
73
+ ratio = trace.total_latency_ms / self.max_latency_ms
74
+ return ScoreResult(
75
+ passed=passed,
76
+ score=max(0.0, 1.0 - (ratio - 1.0)) if not passed else 1.0,
77
+ reason="" if passed else (
78
+ f"Latency {trace.total_latency_ms:.0f}ms exceeds limit {self.max_latency_ms:.0f}ms"
79
+ ),
80
+ details={"actual_ms": trace.total_latency_ms, "max_ms": self.max_latency_ms},
81
+ )
82
+
83
+
84
+ class MaxRetries(Scorer):
85
+ """Pass if the number of retries is at or below the limit."""
86
+
87
+ def __init__(self, max_retries: int):
88
+ self.max_retries = max_retries
89
+
90
+ def score(self, trace: "AgentTrace") -> ScoreResult:
91
+ passed = trace.num_retries <= self.max_retries
92
+ return ScoreResult(
93
+ passed=passed,
94
+ score=1.0 if passed else max(0.0, 1.0 - (trace.num_retries - self.max_retries) / (self.max_retries + 1)),
95
+ reason="" if passed else f"Retries {trace.num_retries} exceeds max {self.max_retries}",
96
+ details={"actual": trace.num_retries, "max": self.max_retries},
97
+ )