evalkeep 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalkeep/__init__.py +12 -0
- evalkeep/__main__.py +6 -0
- evalkeep/adapters/__init__.py +45 -0
- evalkeep/adapters/base.py +92 -0
- evalkeep/adapters/jsonl.py +164 -0
- evalkeep/adapters/langsmith.py +436 -0
- evalkeep/adapters/otlp.py +442 -0
- evalkeep/adapters/semconv.py +208 -0
- evalkeep/analysis.py +174 -0
- evalkeep/analysis_run.py +160 -0
- evalkeep/analyzers/__init__.py +52 -0
- evalkeep/analyzers/anthropic.py +145 -0
- evalkeep/analyzers/stub.py +34 -0
- evalkeep/cache.py +122 -0
- evalkeep/cli.py +1933 -0
- evalkeep/clustering.py +383 -0
- evalkeep/clusters.py +101 -0
- evalkeep/commands/__init__.py +1 -0
- evalkeep/commands/analyze_cmd.py +100 -0
- evalkeep/commands/compare_cmd.py +169 -0
- evalkeep/commands/dataset_cmd.py +182 -0
- evalkeep/commands/detect_cmd.py +154 -0
- evalkeep/commands/discover_cmd.py +274 -0
- evalkeep/commands/ingest_cmd.py +50 -0
- evalkeep/commands/init_cmd.py +151 -0
- evalkeep/commands/pipeline_cmd.py +156 -0
- evalkeep/commands/review_cmd.py +141 -0
- evalkeep/commands/run_cmd.py +131 -0
- evalkeep/commands/target_cmd.py +109 -0
- evalkeep/commands/trace_cmd.py +58 -0
- evalkeep/comparison.py +432 -0
- evalkeep/config.py +209 -0
- evalkeep/detection.py +94 -0
- evalkeep/detectors.py +182 -0
- evalkeep/discovery.py +208 -0
- evalkeep/embeddings/__init__.py +31 -0
- evalkeep/embeddings/base.py +32 -0
- evalkeep/embeddings/hashing.py +98 -0
- evalkeep/errors.py +42 -0
- evalkeep/examples/__init__.py +37 -0
- evalkeep/examples/langsmith/runs.jsonl +18 -0
- evalkeep/examples/opentelemetry/spans.json +898 -0
- evalkeep/examples/refund-agent/agents/baseline.py +66 -0
- evalkeep/examples/refund-agent/agents/candidate.py +66 -0
- evalkeep/examples/refund-agent/traces.jsonl +5 -0
- evalkeep/examples/tau-bench/prepare.py +230 -0
- evalkeep/exporters/__init__.py +45 -0
- evalkeep/exporters/generic.py +31 -0
- evalkeep/exporters/promptfoo.py +219 -0
- evalkeep/failures.py +95 -0
- evalkeep/generation.py +303 -0
- evalkeep/hashing.py +56 -0
- evalkeep/ingest.py +257 -0
- evalkeep/prompts.py +127 -0
- evalkeep/pseudonyms.py +82 -0
- evalkeep/py.typed +0 -0
- evalkeep/redaction.py +333 -0
- evalkeep/regression.py +409 -0
- evalkeep/review.py +309 -0
- evalkeep/runner.py +302 -0
- evalkeep/runs.py +185 -0
- evalkeep/storage/__init__.py +37 -0
- evalkeep/storage/clusters.py +163 -0
- evalkeep/storage/failures.py +254 -0
- evalkeep/storage/migrations.py +370 -0
- evalkeep/storage/regression.py +136 -0
- evalkeep/storage/runs.py +223 -0
- evalkeep/storage/store.py +429 -0
- evalkeep/targets.py +205 -0
- evalkeep/trace.py +238 -0
- evalkeep-0.1.0.dist-info/METADATA +221 -0
- evalkeep-0.1.0.dist-info/RECORD +75 -0
- evalkeep-0.1.0.dist-info/WHEEL +4 -0
- evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
- evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/targets.py
ADDED
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
"""Agent targets: how to reach the thing under test, without its credentials.
|
|
2
|
+
|
|
3
|
+
A target says where an agent lives and how to read its answer. It is committed
|
|
4
|
+
to Git, so it must never contain a secret -- and that is enforced, not merely
|
|
5
|
+
requested: saving a target whose configuration contains something that looks
|
|
6
|
+
like a credential is refused, and the same detectors that redact traces do the
|
|
7
|
+
looking. Secrets are referenced as ``${ENV_VAR}`` and resolved by the runner
|
|
8
|
+
from the environment at run time.
|
|
9
|
+
|
|
10
|
+
Every provider kind is normalized to one response shape::
|
|
11
|
+
|
|
12
|
+
{"text": "...", "toolCalls": [{"tool": "...", "arguments": {...}}, ...]}
|
|
13
|
+
|
|
14
|
+
so an expectation means the same thing whether it runs against an HTTP endpoint,
|
|
15
|
+
a local script or a model provider.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import os
|
|
21
|
+
import re
|
|
22
|
+
from enum import StrEnum
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
from typing import Any
|
|
25
|
+
|
|
26
|
+
import yaml
|
|
27
|
+
from pydantic import BaseModel, ConfigDict, Field, ValidationError
|
|
28
|
+
|
|
29
|
+
from evalkeep.errors import CommandError
|
|
30
|
+
from evalkeep.redaction import TOKEN_PATTERNS, is_secret_field
|
|
31
|
+
|
|
32
|
+
TARGETS_FILENAME = "targets.yaml"
|
|
33
|
+
TARGETS_VERSION = 1
|
|
34
|
+
|
|
35
|
+
BASELINE = "baseline"
|
|
36
|
+
CANDIDATE = "candidate"
|
|
37
|
+
|
|
38
|
+
#: The only way a secret may appear in a committed target.
|
|
39
|
+
ENV_REFERENCE = re.compile(r"^\$\{([A-Z][A-Z0-9_]*)\}$")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class TargetKind(StrEnum):
|
|
43
|
+
HTTP = "http"
|
|
44
|
+
PYTHON = "python"
|
|
45
|
+
JAVASCRIPT = "javascript"
|
|
46
|
+
#: A model provider called directly, for testing tool selection rather than
|
|
47
|
+
#: a whole application.
|
|
48
|
+
MODEL = "model"
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class Extraction(BaseModel):
|
|
52
|
+
"""Where the answer lives in the target's response.
|
|
53
|
+
|
|
54
|
+
Expressions are evaluated over the parsed response body, so ``json.reply``
|
|
55
|
+
reads ``{"reply": ...}``. Only HTTP targets need these; a script or model
|
|
56
|
+
provider is adapted by the generated configuration.
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
model_config = ConfigDict(extra="forbid")
|
|
60
|
+
|
|
61
|
+
output: str = "json.output"
|
|
62
|
+
tool_calls: str | None = "json.toolCalls"
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class Target(BaseModel):
|
|
66
|
+
model_config = ConfigDict(extra="forbid")
|
|
67
|
+
|
|
68
|
+
target_id: str
|
|
69
|
+
kind: TargetKind
|
|
70
|
+
description: str | None = None
|
|
71
|
+
|
|
72
|
+
# HTTP
|
|
73
|
+
url: str | None = None
|
|
74
|
+
method: str = "POST"
|
|
75
|
+
headers: dict[str, str] = Field(default_factory=dict)
|
|
76
|
+
body: dict[str, Any] = Field(default_factory=dict)
|
|
77
|
+
extract: Extraction = Field(default_factory=Extraction)
|
|
78
|
+
|
|
79
|
+
# Script targets
|
|
80
|
+
path: str | None = None
|
|
81
|
+
function: str | None = None
|
|
82
|
+
|
|
83
|
+
# Direct model provider
|
|
84
|
+
provider: str | None = None
|
|
85
|
+
|
|
86
|
+
def validate_shape(self) -> None:
|
|
87
|
+
"""Check the fields this kind of target actually needs."""
|
|
88
|
+
match self.kind:
|
|
89
|
+
case TargetKind.HTTP:
|
|
90
|
+
if not self.url:
|
|
91
|
+
raise CommandError("An http target needs a url.")
|
|
92
|
+
if not self.body:
|
|
93
|
+
raise CommandError(
|
|
94
|
+
"An http target needs a body.",
|
|
95
|
+
hint="Use {{input}} where the test input should go.",
|
|
96
|
+
)
|
|
97
|
+
case TargetKind.PYTHON | TargetKind.JAVASCRIPT:
|
|
98
|
+
if not self.path:
|
|
99
|
+
raise CommandError(f"A {self.kind.value} target needs a path.")
|
|
100
|
+
case TargetKind.MODEL:
|
|
101
|
+
if not self.provider:
|
|
102
|
+
raise CommandError(
|
|
103
|
+
"A model target needs a provider.",
|
|
104
|
+
hint="For example: anthropic:messages:claude-opus-5",
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class TargetFile(BaseModel):
|
|
109
|
+
"""The contents of ``targets.yaml``."""
|
|
110
|
+
|
|
111
|
+
model_config = ConfigDict(extra="forbid")
|
|
112
|
+
|
|
113
|
+
version: int = TARGETS_VERSION
|
|
114
|
+
targets: dict[str, Target] = Field(default_factory=dict)
|
|
115
|
+
|
|
116
|
+
def to_yaml(self) -> str:
|
|
117
|
+
payload = self.model_dump(mode="json", exclude_none=True)
|
|
118
|
+
for target in payload["targets"].values():
|
|
119
|
+
target.pop("target_id", None)
|
|
120
|
+
return yaml.safe_dump(payload, sort_keys=False, default_flow_style=False)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def find_secrets(target: Target) -> list[str]:
|
|
124
|
+
"""Every place this target holds a literal credential.
|
|
125
|
+
|
|
126
|
+
Two rules, both borrowed from redaction so the definitions cannot drift: a
|
|
127
|
+
value that *looks* like a known token, and any value under a
|
|
128
|
+
credential-shaped key that is not an ``${ENV_VAR}`` reference.
|
|
129
|
+
"""
|
|
130
|
+
problems: list[str] = []
|
|
131
|
+
for path, value in _walk(target.model_dump(mode="json")):
|
|
132
|
+
if not isinstance(value, str) or ENV_REFERENCE.match(value.strip()):
|
|
133
|
+
continue
|
|
134
|
+
key = path.rsplit(".", 1)[-1]
|
|
135
|
+
if any(pattern.search(value) for pattern in TOKEN_PATTERNS):
|
|
136
|
+
problems.append(f"{path} looks like a credential")
|
|
137
|
+
elif is_secret_field(key) and value.strip():
|
|
138
|
+
problems.append(
|
|
139
|
+
f"{path} is a credential field with a literal value; use ${{ENV_VAR}} instead"
|
|
140
|
+
)
|
|
141
|
+
return problems
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def referenced_environment(target: Target) -> dict[str, bool]:
|
|
145
|
+
"""Which environment variables this target needs, and whether each is set."""
|
|
146
|
+
referenced: dict[str, bool] = {}
|
|
147
|
+
for _, value in _walk(target.model_dump(mode="json")):
|
|
148
|
+
if isinstance(value, str) and (match := ENV_REFERENCE.match(value.strip())):
|
|
149
|
+
name = match.group(1)
|
|
150
|
+
referenced[name] = bool(os.environ.get(name))
|
|
151
|
+
return referenced
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def load_targets(root: Path) -> TargetFile:
|
|
155
|
+
path = root / TARGETS_FILENAME
|
|
156
|
+
if not path.is_file():
|
|
157
|
+
return TargetFile()
|
|
158
|
+
try:
|
|
159
|
+
raw = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
|
|
160
|
+
except yaml.YAMLError as exc:
|
|
161
|
+
raise CommandError(f"Could not parse {path}: {exc}") from exc
|
|
162
|
+
if not isinstance(raw, dict):
|
|
163
|
+
raise CommandError(f"Expected a mapping in {path}.")
|
|
164
|
+
|
|
165
|
+
for target_id, entry in (raw.get("targets") or {}).items():
|
|
166
|
+
if isinstance(entry, dict):
|
|
167
|
+
entry.setdefault("target_id", target_id)
|
|
168
|
+
try:
|
|
169
|
+
return TargetFile.model_validate(raw)
|
|
170
|
+
except ValidationError as exc:
|
|
171
|
+
raise CommandError(f"Invalid target configuration in {path}:\n{exc}") from exc
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def save_targets(root: Path, targets: TargetFile) -> Path:
|
|
175
|
+
path = root / TARGETS_FILENAME
|
|
176
|
+
try:
|
|
177
|
+
path.write_text(targets.to_yaml(), encoding="utf-8")
|
|
178
|
+
except OSError as exc:
|
|
179
|
+
raise CommandError(f"Could not write {path}: {exc}") from exc
|
|
180
|
+
return path
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def get_target(root: Path, target_id: str) -> Target:
|
|
184
|
+
targets = load_targets(root)
|
|
185
|
+
target = targets.targets.get(target_id.strip())
|
|
186
|
+
if target is None:
|
|
187
|
+
known = ", ".join(sorted(targets.targets)) or "none configured"
|
|
188
|
+
raise CommandError(
|
|
189
|
+
f"No target named {target_id.strip()!r}.",
|
|
190
|
+
hint=f"Known targets: {known}. Add one with 'evalkeep targets add'.",
|
|
191
|
+
)
|
|
192
|
+
return target
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _walk(value: Any, prefix: str = "") -> list[tuple[str, Any]]:
|
|
196
|
+
found: list[tuple[str, Any]] = []
|
|
197
|
+
if isinstance(value, dict):
|
|
198
|
+
for key, child in value.items():
|
|
199
|
+
found.extend(_walk(child, f"{prefix}{key}."))
|
|
200
|
+
elif isinstance(value, list):
|
|
201
|
+
for index, child in enumerate(value):
|
|
202
|
+
found.extend(_walk(child, f"{prefix}{index}."))
|
|
203
|
+
else:
|
|
204
|
+
found.append((prefix.rstrip("."), value))
|
|
205
|
+
return found
|
evalkeep/trace.py
ADDED
|
@@ -0,0 +1,238 @@
|
|
|
1
|
+
"""The normalized trace schema.
|
|
2
|
+
|
|
3
|
+
Every adapter converts a provider's format into these models, so the rest of
|
|
4
|
+
Evalkeep -- redaction, failure detection, clustering, test generation -- only
|
|
5
|
+
ever sees one shape.
|
|
6
|
+
|
|
7
|
+
Two deliberate strictnesses:
|
|
8
|
+
|
|
9
|
+
* Structural models reject unknown fields. A typo like ``trace-id`` is a defect
|
|
10
|
+
worth reporting, not data worth keeping, and every field that survives here
|
|
11
|
+
must be one the redactor knows how to scrub. Arbitrary provider data belongs
|
|
12
|
+
under ``metadata.extra``, which redaction walks recursively.
|
|
13
|
+
* Identifiers are stripped and must be non-empty, so a whitespace-only
|
|
14
|
+
``trace_id`` fails validation rather than becoming a silent duplicate.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from datetime import UTC, datetime
|
|
20
|
+
from enum import StrEnum
|
|
21
|
+
from typing import Annotated, Any, Literal
|
|
22
|
+
|
|
23
|
+
from pydantic import (
|
|
24
|
+
BaseModel,
|
|
25
|
+
ConfigDict,
|
|
26
|
+
Field,
|
|
27
|
+
StringConstraints,
|
|
28
|
+
field_validator,
|
|
29
|
+
model_validator,
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
#: Bumped when the trace schema changes in a way stored traces must migrate for.
|
|
33
|
+
SCHEMA_VERSION = 1
|
|
34
|
+
|
|
35
|
+
MAX_IDENTIFIER_LENGTH = 256
|
|
36
|
+
MAX_TOOL_NAME_LENGTH = 128
|
|
37
|
+
|
|
38
|
+
#: Identifiers are trimmed and must survive trimming.
|
|
39
|
+
Identifier = Annotated[
|
|
40
|
+
str,
|
|
41
|
+
StringConstraints(strip_whitespace=True, min_length=1, max_length=MAX_IDENTIFIER_LENGTH),
|
|
42
|
+
]
|
|
43
|
+
|
|
44
|
+
#: Tool names must look like callable identifiers: no spaces, no path separators,
|
|
45
|
+
#: nothing that could be mistaken for a shell fragment when exported to a runner.
|
|
46
|
+
ToolName = Annotated[
|
|
47
|
+
str,
|
|
48
|
+
StringConstraints(
|
|
49
|
+
strip_whitespace=True,
|
|
50
|
+
min_length=1,
|
|
51
|
+
max_length=MAX_TOOL_NAME_LENGTH,
|
|
52
|
+
pattern=r"^[A-Za-z_][A-Za-z0-9_.-]*$",
|
|
53
|
+
),
|
|
54
|
+
]
|
|
55
|
+
|
|
56
|
+
NonEmptyStr = Annotated[str, StringConstraints(strip_whitespace=True, min_length=1)]
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class Role(StrEnum):
|
|
60
|
+
USER = "user"
|
|
61
|
+
ASSISTANT = "assistant"
|
|
62
|
+
SYSTEM = "system"
|
|
63
|
+
TOOL = "tool"
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class OutcomeStatus(StrEnum):
|
|
67
|
+
SUCCESS = "success"
|
|
68
|
+
FAILURE = "failure"
|
|
69
|
+
ERROR = "error"
|
|
70
|
+
UNKNOWN = "unknown"
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
class _Strict(BaseModel):
|
|
74
|
+
"""Base for every trace model: unknown fields are a validation error."""
|
|
75
|
+
|
|
76
|
+
model_config = ConfigDict(extra="forbid")
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _as_utc(value: datetime | None) -> datetime | None:
|
|
80
|
+
"""Treat a naive timestamp as UTC so orderings stay comparable."""
|
|
81
|
+
if value is not None and value.tzinfo is None:
|
|
82
|
+
return value.replace(tzinfo=UTC)
|
|
83
|
+
return value
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class Message(_Strict):
|
|
87
|
+
role: Role
|
|
88
|
+
content: str
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
class TraceInput(_Strict):
|
|
92
|
+
"""What the agent was asked to do."""
|
|
93
|
+
|
|
94
|
+
text: str | None = None
|
|
95
|
+
messages: list[Message] = Field(default_factory=list)
|
|
96
|
+
|
|
97
|
+
@model_validator(mode="after")
|
|
98
|
+
def _require_content(self) -> TraceInput:
|
|
99
|
+
if not (self.text or "").strip() and not self.messages:
|
|
100
|
+
raise ValueError("input must provide non-empty 'text' or at least one message")
|
|
101
|
+
return self
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
class TraceOutput(_Strict):
|
|
105
|
+
"""What the agent produced. Absent for traces captured before a response."""
|
|
106
|
+
|
|
107
|
+
text: str | None = None
|
|
108
|
+
messages: list[Message] = Field(default_factory=list)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class _BaseEvent(_Strict):
|
|
112
|
+
event_id: Identifier
|
|
113
|
+
timestamp: datetime | None = None
|
|
114
|
+
|
|
115
|
+
_normalize_timestamp = field_validator("timestamp")(_as_utc)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
class MessageEvent(_BaseEvent):
|
|
119
|
+
type: Literal["message"] = "message"
|
|
120
|
+
role: Role
|
|
121
|
+
content: str
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
class ToolCallEvent(_BaseEvent):
|
|
125
|
+
type: Literal["tool_call"] = "tool_call"
|
|
126
|
+
tool: ToolName
|
|
127
|
+
arguments: dict[str, Any] = Field(default_factory=dict)
|
|
128
|
+
call_id: Identifier | None = None
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
class ToolResultEvent(_BaseEvent):
|
|
132
|
+
type: Literal["tool_result"] = "tool_result"
|
|
133
|
+
tool: ToolName
|
|
134
|
+
call_id: Identifier | None = None
|
|
135
|
+
result: Any = None
|
|
136
|
+
error: str | None = None
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class EvaluationEvent(_BaseEvent):
|
|
140
|
+
"""An evaluator's verdict recorded alongside the interaction."""
|
|
141
|
+
|
|
142
|
+
type: Literal["evaluation"] = "evaluation"
|
|
143
|
+
name: NonEmptyStr
|
|
144
|
+
passed: bool | None = None
|
|
145
|
+
score: float | None = None
|
|
146
|
+
reason: str | None = None
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
Event = Annotated[
|
|
150
|
+
MessageEvent | ToolCallEvent | ToolResultEvent | EvaluationEvent,
|
|
151
|
+
Field(discriminator="type"),
|
|
152
|
+
]
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
class Feedback(_Strict):
|
|
156
|
+
"""Human or automated feedback attached to the interaction."""
|
|
157
|
+
|
|
158
|
+
rating: Literal["positive", "negative"] | None = None
|
|
159
|
+
score: float | None = None
|
|
160
|
+
comment: str | None = None
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
class Evaluation(_Strict):
|
|
164
|
+
name: NonEmptyStr
|
|
165
|
+
passed: bool | None = None
|
|
166
|
+
score: float | None = None
|
|
167
|
+
reason: str | None = None
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
class Outcome(_Strict):
|
|
171
|
+
"""The three evidence sources the failure detectors read in guide 8D."""
|
|
172
|
+
|
|
173
|
+
status: OutcomeStatus = OutcomeStatus.UNKNOWN
|
|
174
|
+
feedback: Feedback | None = None
|
|
175
|
+
evaluations: list[Evaluation] = Field(default_factory=list)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
class TraceMetadata(_Strict):
|
|
179
|
+
recorded_at: datetime | None = None
|
|
180
|
+
source: str | None = None
|
|
181
|
+
agent: str | None = None
|
|
182
|
+
model: str | None = None
|
|
183
|
+
tags: list[str] = Field(default_factory=list)
|
|
184
|
+
#: The one open field. Provider-specific data goes here and is redacted
|
|
185
|
+
#: recursively; nothing else in the schema accepts unknown keys.
|
|
186
|
+
extra: dict[str, Any] = Field(default_factory=dict)
|
|
187
|
+
|
|
188
|
+
_normalize_recorded_at = field_validator("recorded_at")(_as_utc)
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
class NormalizedTrace(_Strict):
|
|
192
|
+
"""One recorded interaction, in the only shape Evalkeep stores."""
|
|
193
|
+
|
|
194
|
+
schema_version: int = SCHEMA_VERSION
|
|
195
|
+
trace_id: Identifier
|
|
196
|
+
input: TraceInput
|
|
197
|
+
output: TraceOutput | None = None
|
|
198
|
+
events: list[Event] = Field(default_factory=list)
|
|
199
|
+
outcome: Outcome = Field(default_factory=Outcome)
|
|
200
|
+
metadata: TraceMetadata = Field(default_factory=TraceMetadata)
|
|
201
|
+
|
|
202
|
+
@model_validator(mode="after")
|
|
203
|
+
def _validate_event_sequence(self) -> NormalizedTrace:
|
|
204
|
+
"""Events must be uniquely identified, ordered, and internally consistent."""
|
|
205
|
+
seen_event_ids: set[str] = set()
|
|
206
|
+
open_calls: set[str] = set()
|
|
207
|
+
previous: datetime | None = None
|
|
208
|
+
|
|
209
|
+
for index, event in enumerate(self.events):
|
|
210
|
+
where = f"events[{index}]"
|
|
211
|
+
if event.event_id in seen_event_ids:
|
|
212
|
+
raise ValueError(f"{where}: duplicate event_id {event.event_id!r}")
|
|
213
|
+
seen_event_ids.add(event.event_id)
|
|
214
|
+
|
|
215
|
+
if event.timestamp is not None:
|
|
216
|
+
if previous is not None and event.timestamp < previous:
|
|
217
|
+
raise ValueError(
|
|
218
|
+
f"{where}: timestamp {event.timestamp.isoformat()} is earlier than "
|
|
219
|
+
f"the preceding event's {previous.isoformat()}; events must be ordered"
|
|
220
|
+
)
|
|
221
|
+
previous = event.timestamp
|
|
222
|
+
|
|
223
|
+
if isinstance(event, ToolCallEvent) and event.call_id is not None:
|
|
224
|
+
if event.call_id in open_calls:
|
|
225
|
+
raise ValueError(f"{where}: duplicate call_id {event.call_id!r}")
|
|
226
|
+
open_calls.add(event.call_id)
|
|
227
|
+
elif isinstance(event, ToolResultEvent) and event.call_id is not None:
|
|
228
|
+
if event.call_id not in open_calls:
|
|
229
|
+
raise ValueError(
|
|
230
|
+
f"{where}: call_id {event.call_id!r} does not match an earlier tool_call"
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
return self
|
|
234
|
+
|
|
235
|
+
@property
|
|
236
|
+
def tool_calls(self) -> list[ToolCallEvent]:
|
|
237
|
+
"""Observed tool calls, in order. Used by detection and test generation."""
|
|
238
|
+
return [event for event in self.events if isinstance(event, ToolCallEvent)]
|
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: evalkeep
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Turn production agent failures into a small, reviewed regression suite.
|
|
5
|
+
Keywords: llm,evals,evaluation,regression-testing,ai-agents,testing,promptfoo
|
|
6
|
+
Author: Rakshita Devurkar
|
|
7
|
+
Author-email: Rakshita Devurkar <13130544+rakshita-devurkar@users.noreply.github.com>
|
|
8
|
+
License-Expression: Apache-2.0
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Environment :: Console
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
19
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
20
|
+
Classifier: Topic :: Software Development :: Testing
|
|
21
|
+
Classifier: Typing :: Typed
|
|
22
|
+
Requires-Dist: click>=8.5.0
|
|
23
|
+
Requires-Dist: numpy>=2.4.6
|
|
24
|
+
Requires-Dist: pydantic>=2.13.5
|
|
25
|
+
Requires-Dist: pyyaml>=6.0.3
|
|
26
|
+
Requires-Dist: rich>=15.0.0
|
|
27
|
+
Requires-Dist: typer>=0.27.2
|
|
28
|
+
Requires-Dist: anthropic>=1.0.0 ; extra == 'anthropic'
|
|
29
|
+
Requires-Python: >=3.11
|
|
30
|
+
Project-URL: Homepage, https://github.com/rakshita-devurkar/evalkeep
|
|
31
|
+
Project-URL: Repository, https://github.com/rakshita-devurkar/evalkeep
|
|
32
|
+
Project-URL: Issues, https://github.com/rakshita-devurkar/evalkeep/issues
|
|
33
|
+
Project-URL: Changelog, https://github.com/rakshita-devurkar/evalkeep/blob/main/CHANGELOG.md
|
|
34
|
+
Project-URL: Documentation, https://github.com/rakshita-devurkar/evalkeep/blob/main/docs/pipeline.md
|
|
35
|
+
Provides-Extra: anthropic
|
|
36
|
+
Description-Content-Type: text/markdown
|
|
37
|
+
|
|
38
|
+
# Evalkeep
|
|
39
|
+
|
|
40
|
+
**Stop fixing the same agent bug twice.** Evalkeep turns production failures
|
|
41
|
+
into a small, reviewed regression suite, and tells you whether a fix held — or
|
|
42
|
+
that the evidence is too thin to say.
|
|
43
|
+
|
|
44
|
+
Existing eval tools *execute* tests. The hard part is deciding which of
|
|
45
|
+
thousands of production traces deserve permanent coverage. Evalkeep owns that
|
|
46
|
+
decision:
|
|
47
|
+
|
|
48
|
+
```
|
|
49
|
+
trace → failure evidence → failure family → representative case
|
|
50
|
+
→ reviewed regression test → runner execution → trustworthy comparison
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
**Evalkeep does not run your agent and is not an eval framework.** It generates
|
|
54
|
+
tests, delegates execution to [Promptfoo](https://promptfoo.dev), and compares
|
|
55
|
+
baseline against candidate. It sits upstream of your eval runner, not next to it.
|
|
56
|
+
|
|
57
|
+
## Install
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
git clone https://github.com/rakshita-devurkar/evalkeep && cd evalkeep
|
|
61
|
+
uv sync
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Node.js is needed only for `evalkeep run`, which shells out to Promptfoo.
|
|
65
|
+
|
|
66
|
+
## Quick start
|
|
67
|
+
|
|
68
|
+
One command takes a trace file to a review queue. Offline, no API key:
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
uv run evalkeep demo .
|
|
72
|
+
uv run evalkeep init
|
|
73
|
+
uv run evalkeep from-traces refund-agent/traces.jsonl
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
```
|
|
77
|
+
5 trace(s) ingested
|
|
78
|
+
3 failure(s) found explicit_status x3, failed_evaluator x1, negative_feedback x2
|
|
79
|
+
2 failure famil(ies)
|
|
80
|
+
3 with enough evidence for a regression test
|
|
81
|
+
3 of them only forbid the mistake that was observed; say what should have happened at review.
|
|
82
|
+
|
|
83
|
+
Review them: evalkeep review (3 pending)
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
That second qualifier is the honest state of a first run: nothing in a trace
|
|
87
|
+
says what the agent *should* have done. Describing the failures closes it, by
|
|
88
|
+
hand at review or with an analyzer configured. It stops at review on purpose —
|
|
89
|
+
approving a test is a judgement.
|
|
90
|
+
|
|
91
|
+
## On real agent data
|
|
92
|
+
|
|
93
|
+
The example above is five traces. Here is the same pipeline on
|
|
94
|
+
[tau-bench trajectories](https://huggingface.co/datasets/AgentSuite/tau-bench-trajectories):
|
|
95
|
+
165 retail and airline customer-service tasks per model, each scored by
|
|
96
|
+
comparing the final database state against the expected one. Real tool calls,
|
|
97
|
+
and an independent verdict — the two halves a regression suite needs.
|
|
98
|
+
|
|
99
|
+
```bash
|
|
100
|
+
python tau-bench/prepare.py # ~8 MB, two models
|
|
101
|
+
uv run evalkeep from-traces tau-bench/Qwen3-235B-A22B-FP8.traces.jsonl
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
```
|
|
105
|
+
165 trace(s) ingested
|
|
106
|
+
90 failure(s) found explicit_status x90, failed_evaluator x90
|
|
107
|
+
4 failure famil(ies)
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Four families, from 90 failures: retail exchange and refund flows, airline
|
|
111
|
+
reservation changes, and two shapes of giving up and escalating to a human.
|
|
112
|
+
Build a test per failure, approve them, and run two recorded models against the
|
|
113
|
+
suite one of them produced:
|
|
114
|
+
|
|
115
|
+
```bash
|
|
116
|
+
uv run evalkeep dataset build --all
|
|
117
|
+
uv run evalkeep review
|
|
118
|
+
uv run evalkeep targets add baseline --type python --function call_api \
|
|
119
|
+
--path tau-bench/replay_Qwen3_235B_A22B_FP8.py
|
|
120
|
+
uv run evalkeep targets add candidate --type python --function call_api \
|
|
121
|
+
--path tau-bench/replay_claude_4_5_sonnet_thinking_off.py
|
|
122
|
+
uv run evalkeep run --target baseline && uv run evalkeep run --target candidate
|
|
123
|
+
uv run evalkeep compare
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
```
|
|
127
|
+
compared 89
|
|
128
|
+
baseline pass rate 4.5%
|
|
129
|
+
candidate pass rate 42.7%
|
|
130
|
+
difference +38.2%
|
|
131
|
+
p-value 0.0000
|
|
132
|
+
95% interval +27.2% to +49.2%
|
|
133
|
+
McNemar's exact test: the change is unlikely to be chance.
|
|
134
|
+
|
|
135
|
+
1 test(s) excluded and not counted in any rate above.
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
Baseline scoring 4.5% is the control: the tests came from its own failures, so
|
|
139
|
+
it should fail nearly all of them. The four it passes are the documented
|
|
140
|
+
weakness of deriving a test with nothing describing the failure — assertions
|
|
141
|
+
target the last tool call, which is sometimes a harmless lookup. The excluded
|
|
142
|
+
test is one whose target raised rather than answered, and it is kept out of
|
|
143
|
+
every rate rather than counted as a failure.
|
|
144
|
+
|
|
145
|
+
## Doing it stage by stage
|
|
146
|
+
|
|
147
|
+
`from-traces` runs five commands in order. Each is a real decision with its own
|
|
148
|
+
evidence, and you will want them separately once you are tuning a suite:
|
|
149
|
+
|
|
150
|
+
```bash
|
|
151
|
+
uv run evalkeep ingest traces.jsonl # validate, redact, store
|
|
152
|
+
uv run evalkeep detect # evidence-backed failures
|
|
153
|
+
uv run evalkeep analyze # describe them (or: failures label)
|
|
154
|
+
uv run evalkeep discover # embed, cluster, pick representatives
|
|
155
|
+
uv run evalkeep dataset build # draft a test per representative
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
Re-running `from-traces` is safe: it skips traces it already has and rebuilds
|
|
159
|
+
drafts, but never touches a test you have reviewed.
|
|
160
|
+
|
|
161
|
+
## Bring your own traces
|
|
162
|
+
|
|
163
|
+
```bash
|
|
164
|
+
uv run evalkeep ingest spans.json --format otlp # OpenTelemetry / OpenInference
|
|
165
|
+
uv run evalkeep ingest runs.jsonl --format langsmith # LangSmith
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
Adapters read files, never APIs — no credentials, any vendor tier. OpenTelemetry
|
|
169
|
+
covers the most ground, since Langfuse, Braintrust and Phoenix all ingest OTLP.
|
|
170
|
+
`evalkeep demo` writes an example export in each format.
|
|
171
|
+
|
|
172
|
+
## Commands
|
|
173
|
+
|
|
174
|
+
| Stage | Commands |
|
|
175
|
+
| --- | --- |
|
|
176
|
+
| Set up | `init`, `targets add/list/show/remove` |
|
|
177
|
+
| Ingest | `ingest`, `trace list/show` |
|
|
178
|
+
| Detect | `detect`, `failures list/show/confirm/dismiss/add` |
|
|
179
|
+
| Analyze | `analyze`, `failures label` |
|
|
180
|
+
| Group | `discover`, `clusters list/show/rename/merge/split/dismiss/restore` |
|
|
181
|
+
| Build | `dataset build/list/show` |
|
|
182
|
+
| Review | `review`, `dataset approve/reject/edit` |
|
|
183
|
+
| Run | `export`, `run --target ...`, `runs list/show` |
|
|
184
|
+
| Compare | `compare`, `baseline promote/show` |
|
|
185
|
+
|
|
186
|
+
## What it guarantees
|
|
187
|
+
|
|
188
|
+
- **Values are redacted before storage** — in memory, with no path around it.
|
|
189
|
+
Identifiers can be [pseudonymized](docs/security.md#identifiers) too.
|
|
190
|
+
- **Automation never overwrites human judgement.** Re-running any stage
|
|
191
|
+
refreshes derived data and leaves your reviews, labels and edits alone.
|
|
192
|
+
- **Nothing is exported without approval.** Generated tests are drafts.
|
|
193
|
+
- **A test that never ran is not a test that failed.** Timeouts and crashed
|
|
194
|
+
providers are excluded from every rate, so an outage cannot read as a regression.
|
|
195
|
+
- **One lucky pass is not a fix.** `run --repetitions N` reports a per-case
|
|
196
|
+
verdict; a case that only sometimes passes is flaky, never passing.
|
|
197
|
+
- **Score changes are not overclaimed.** McNemar's exact test, and no confidence
|
|
198
|
+
interval when the sample cannot support one.
|
|
199
|
+
|
|
200
|
+
## More
|
|
201
|
+
|
|
202
|
+
[How it works](docs/pipeline.md) · [Privacy and security](docs/security.md) ·
|
|
203
|
+
[Roadmap](docs/roadmap.md) · [Contributing](CONTRIBUTING.md) ·
|
|
204
|
+
[Changelog](CHANGELOG.md)
|
|
205
|
+
|
|
206
|
+
```bash
|
|
207
|
+
uv sync && uv run pytest # 985 tests
|
|
208
|
+
uv run ruff check . && uv run mypy # lint and strict types
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
`EVALKEEP_E2E=1 uv run pytest` also runs the suite against real Promptfoo.
|
|
212
|
+
Exit codes: `0` success, `1` ran but some records were rejected, `2` could not
|
|
213
|
+
run.
|
|
214
|
+
|
|
215
|
+
0.1 is feature-complete and not yet released to PyPI. Multi-turn replay and
|
|
216
|
+
longitudinal failure history are still open; see the
|
|
217
|
+
[roadmap](docs/roadmap.md).
|
|
218
|
+
|
|
219
|
+
## License
|
|
220
|
+
|
|
221
|
+
Apache-2.0. See [LICENSE](LICENSE).
|