evalkeep 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. evalkeep/__init__.py +12 -0
  2. evalkeep/__main__.py +6 -0
  3. evalkeep/adapters/__init__.py +45 -0
  4. evalkeep/adapters/base.py +92 -0
  5. evalkeep/adapters/jsonl.py +164 -0
  6. evalkeep/adapters/langsmith.py +436 -0
  7. evalkeep/adapters/otlp.py +442 -0
  8. evalkeep/adapters/semconv.py +208 -0
  9. evalkeep/analysis.py +174 -0
  10. evalkeep/analysis_run.py +160 -0
  11. evalkeep/analyzers/__init__.py +52 -0
  12. evalkeep/analyzers/anthropic.py +145 -0
  13. evalkeep/analyzers/stub.py +34 -0
  14. evalkeep/cache.py +122 -0
  15. evalkeep/cli.py +1933 -0
  16. evalkeep/clustering.py +383 -0
  17. evalkeep/clusters.py +101 -0
  18. evalkeep/commands/__init__.py +1 -0
  19. evalkeep/commands/analyze_cmd.py +100 -0
  20. evalkeep/commands/compare_cmd.py +169 -0
  21. evalkeep/commands/dataset_cmd.py +182 -0
  22. evalkeep/commands/detect_cmd.py +154 -0
  23. evalkeep/commands/discover_cmd.py +274 -0
  24. evalkeep/commands/ingest_cmd.py +50 -0
  25. evalkeep/commands/init_cmd.py +151 -0
  26. evalkeep/commands/pipeline_cmd.py +156 -0
  27. evalkeep/commands/review_cmd.py +141 -0
  28. evalkeep/commands/run_cmd.py +131 -0
  29. evalkeep/commands/target_cmd.py +109 -0
  30. evalkeep/commands/trace_cmd.py +58 -0
  31. evalkeep/comparison.py +432 -0
  32. evalkeep/config.py +209 -0
  33. evalkeep/detection.py +94 -0
  34. evalkeep/detectors.py +182 -0
  35. evalkeep/discovery.py +208 -0
  36. evalkeep/embeddings/__init__.py +31 -0
  37. evalkeep/embeddings/base.py +32 -0
  38. evalkeep/embeddings/hashing.py +98 -0
  39. evalkeep/errors.py +42 -0
  40. evalkeep/examples/__init__.py +37 -0
  41. evalkeep/examples/langsmith/runs.jsonl +18 -0
  42. evalkeep/examples/opentelemetry/spans.json +898 -0
  43. evalkeep/examples/refund-agent/agents/baseline.py +66 -0
  44. evalkeep/examples/refund-agent/agents/candidate.py +66 -0
  45. evalkeep/examples/refund-agent/traces.jsonl +5 -0
  46. evalkeep/examples/tau-bench/prepare.py +230 -0
  47. evalkeep/exporters/__init__.py +45 -0
  48. evalkeep/exporters/generic.py +31 -0
  49. evalkeep/exporters/promptfoo.py +219 -0
  50. evalkeep/failures.py +95 -0
  51. evalkeep/generation.py +303 -0
  52. evalkeep/hashing.py +56 -0
  53. evalkeep/ingest.py +257 -0
  54. evalkeep/prompts.py +127 -0
  55. evalkeep/pseudonyms.py +82 -0
  56. evalkeep/py.typed +0 -0
  57. evalkeep/redaction.py +333 -0
  58. evalkeep/regression.py +409 -0
  59. evalkeep/review.py +309 -0
  60. evalkeep/runner.py +302 -0
  61. evalkeep/runs.py +185 -0
  62. evalkeep/storage/__init__.py +37 -0
  63. evalkeep/storage/clusters.py +163 -0
  64. evalkeep/storage/failures.py +254 -0
  65. evalkeep/storage/migrations.py +370 -0
  66. evalkeep/storage/regression.py +136 -0
  67. evalkeep/storage/runs.py +223 -0
  68. evalkeep/storage/store.py +429 -0
  69. evalkeep/targets.py +205 -0
  70. evalkeep/trace.py +238 -0
  71. evalkeep-0.1.0.dist-info/METADATA +221 -0
  72. evalkeep-0.1.0.dist-info/RECORD +75 -0
  73. evalkeep-0.1.0.dist-info/WHEEL +4 -0
  74. evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
  75. evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/targets.py ADDED
@@ -0,0 +1,205 @@
1
+ """Agent targets: how to reach the thing under test, without its credentials.
2
+
3
+ A target says where an agent lives and how to read its answer. It is committed
4
+ to Git, so it must never contain a secret -- and that is enforced, not merely
5
+ requested: saving a target whose configuration contains something that looks
6
+ like a credential is refused, and the same detectors that redact traces do the
7
+ looking. Secrets are referenced as ``${ENV_VAR}`` and resolved by the runner
8
+ from the environment at run time.
9
+
10
+ Every provider kind is normalized to one response shape::
11
+
12
+ {"text": "...", "toolCalls": [{"tool": "...", "arguments": {...}}, ...]}
13
+
14
+ so an expectation means the same thing whether it runs against an HTTP endpoint,
15
+ a local script or a model provider.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import os
21
+ import re
22
+ from enum import StrEnum
23
+ from pathlib import Path
24
+ from typing import Any
25
+
26
+ import yaml
27
+ from pydantic import BaseModel, ConfigDict, Field, ValidationError
28
+
29
+ from evalkeep.errors import CommandError
30
+ from evalkeep.redaction import TOKEN_PATTERNS, is_secret_field
31
+
32
+ TARGETS_FILENAME = "targets.yaml"
33
+ TARGETS_VERSION = 1
34
+
35
+ BASELINE = "baseline"
36
+ CANDIDATE = "candidate"
37
+
38
+ #: The only way a secret may appear in a committed target.
39
+ ENV_REFERENCE = re.compile(r"^\$\{([A-Z][A-Z0-9_]*)\}$")
40
+
41
+
42
+ class TargetKind(StrEnum):
43
+ HTTP = "http"
44
+ PYTHON = "python"
45
+ JAVASCRIPT = "javascript"
46
+ #: A model provider called directly, for testing tool selection rather than
47
+ #: a whole application.
48
+ MODEL = "model"
49
+
50
+
51
+ class Extraction(BaseModel):
52
+ """Where the answer lives in the target's response.
53
+
54
+ Expressions are evaluated over the parsed response body, so ``json.reply``
55
+ reads ``{"reply": ...}``. Only HTTP targets need these; a script or model
56
+ provider is adapted by the generated configuration.
57
+ """
58
+
59
+ model_config = ConfigDict(extra="forbid")
60
+
61
+ output: str = "json.output"
62
+ tool_calls: str | None = "json.toolCalls"
63
+
64
+
65
+ class Target(BaseModel):
66
+ model_config = ConfigDict(extra="forbid")
67
+
68
+ target_id: str
69
+ kind: TargetKind
70
+ description: str | None = None
71
+
72
+ # HTTP
73
+ url: str | None = None
74
+ method: str = "POST"
75
+ headers: dict[str, str] = Field(default_factory=dict)
76
+ body: dict[str, Any] = Field(default_factory=dict)
77
+ extract: Extraction = Field(default_factory=Extraction)
78
+
79
+ # Script targets
80
+ path: str | None = None
81
+ function: str | None = None
82
+
83
+ # Direct model provider
84
+ provider: str | None = None
85
+
86
+ def validate_shape(self) -> None:
87
+ """Check the fields this kind of target actually needs."""
88
+ match self.kind:
89
+ case TargetKind.HTTP:
90
+ if not self.url:
91
+ raise CommandError("An http target needs a url.")
92
+ if not self.body:
93
+ raise CommandError(
94
+ "An http target needs a body.",
95
+ hint="Use {{input}} where the test input should go.",
96
+ )
97
+ case TargetKind.PYTHON | TargetKind.JAVASCRIPT:
98
+ if not self.path:
99
+ raise CommandError(f"A {self.kind.value} target needs a path.")
100
+ case TargetKind.MODEL:
101
+ if not self.provider:
102
+ raise CommandError(
103
+ "A model target needs a provider.",
104
+ hint="For example: anthropic:messages:claude-opus-5",
105
+ )
106
+
107
+
108
+ class TargetFile(BaseModel):
109
+ """The contents of ``targets.yaml``."""
110
+
111
+ model_config = ConfigDict(extra="forbid")
112
+
113
+ version: int = TARGETS_VERSION
114
+ targets: dict[str, Target] = Field(default_factory=dict)
115
+
116
+ def to_yaml(self) -> str:
117
+ payload = self.model_dump(mode="json", exclude_none=True)
118
+ for target in payload["targets"].values():
119
+ target.pop("target_id", None)
120
+ return yaml.safe_dump(payload, sort_keys=False, default_flow_style=False)
121
+
122
+
123
+ def find_secrets(target: Target) -> list[str]:
124
+ """Every place this target holds a literal credential.
125
+
126
+ Two rules, both borrowed from redaction so the definitions cannot drift: a
127
+ value that *looks* like a known token, and any value under a
128
+ credential-shaped key that is not an ``${ENV_VAR}`` reference.
129
+ """
130
+ problems: list[str] = []
131
+ for path, value in _walk(target.model_dump(mode="json")):
132
+ if not isinstance(value, str) or ENV_REFERENCE.match(value.strip()):
133
+ continue
134
+ key = path.rsplit(".", 1)[-1]
135
+ if any(pattern.search(value) for pattern in TOKEN_PATTERNS):
136
+ problems.append(f"{path} looks like a credential")
137
+ elif is_secret_field(key) and value.strip():
138
+ problems.append(
139
+ f"{path} is a credential field with a literal value; use ${{ENV_VAR}} instead"
140
+ )
141
+ return problems
142
+
143
+
144
+ def referenced_environment(target: Target) -> dict[str, bool]:
145
+ """Which environment variables this target needs, and whether each is set."""
146
+ referenced: dict[str, bool] = {}
147
+ for _, value in _walk(target.model_dump(mode="json")):
148
+ if isinstance(value, str) and (match := ENV_REFERENCE.match(value.strip())):
149
+ name = match.group(1)
150
+ referenced[name] = bool(os.environ.get(name))
151
+ return referenced
152
+
153
+
154
+ def load_targets(root: Path) -> TargetFile:
155
+ path = root / TARGETS_FILENAME
156
+ if not path.is_file():
157
+ return TargetFile()
158
+ try:
159
+ raw = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
160
+ except yaml.YAMLError as exc:
161
+ raise CommandError(f"Could not parse {path}: {exc}") from exc
162
+ if not isinstance(raw, dict):
163
+ raise CommandError(f"Expected a mapping in {path}.")
164
+
165
+ for target_id, entry in (raw.get("targets") or {}).items():
166
+ if isinstance(entry, dict):
167
+ entry.setdefault("target_id", target_id)
168
+ try:
169
+ return TargetFile.model_validate(raw)
170
+ except ValidationError as exc:
171
+ raise CommandError(f"Invalid target configuration in {path}:\n{exc}") from exc
172
+
173
+
174
+ def save_targets(root: Path, targets: TargetFile) -> Path:
175
+ path = root / TARGETS_FILENAME
176
+ try:
177
+ path.write_text(targets.to_yaml(), encoding="utf-8")
178
+ except OSError as exc:
179
+ raise CommandError(f"Could not write {path}: {exc}") from exc
180
+ return path
181
+
182
+
183
+ def get_target(root: Path, target_id: str) -> Target:
184
+ targets = load_targets(root)
185
+ target = targets.targets.get(target_id.strip())
186
+ if target is None:
187
+ known = ", ".join(sorted(targets.targets)) or "none configured"
188
+ raise CommandError(
189
+ f"No target named {target_id.strip()!r}.",
190
+ hint=f"Known targets: {known}. Add one with 'evalkeep targets add'.",
191
+ )
192
+ return target
193
+
194
+
195
+ def _walk(value: Any, prefix: str = "") -> list[tuple[str, Any]]:
196
+ found: list[tuple[str, Any]] = []
197
+ if isinstance(value, dict):
198
+ for key, child in value.items():
199
+ found.extend(_walk(child, f"{prefix}{key}."))
200
+ elif isinstance(value, list):
201
+ for index, child in enumerate(value):
202
+ found.extend(_walk(child, f"{prefix}{index}."))
203
+ else:
204
+ found.append((prefix.rstrip("."), value))
205
+ return found
evalkeep/trace.py ADDED
@@ -0,0 +1,238 @@
1
+ """The normalized trace schema.
2
+
3
+ Every adapter converts a provider's format into these models, so the rest of
4
+ Evalkeep -- redaction, failure detection, clustering, test generation -- only
5
+ ever sees one shape.
6
+
7
+ Two deliberate strictnesses:
8
+
9
+ * Structural models reject unknown fields. A typo like ``trace-id`` is a defect
10
+ worth reporting, not data worth keeping, and every field that survives here
11
+ must be one the redactor knows how to scrub. Arbitrary provider data belongs
12
+ under ``metadata.extra``, which redaction walks recursively.
13
+ * Identifiers are stripped and must be non-empty, so a whitespace-only
14
+ ``trace_id`` fails validation rather than becoming a silent duplicate.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from datetime import UTC, datetime
20
+ from enum import StrEnum
21
+ from typing import Annotated, Any, Literal
22
+
23
+ from pydantic import (
24
+ BaseModel,
25
+ ConfigDict,
26
+ Field,
27
+ StringConstraints,
28
+ field_validator,
29
+ model_validator,
30
+ )
31
+
32
+ #: Bumped when the trace schema changes in a way stored traces must migrate for.
33
+ SCHEMA_VERSION = 1
34
+
35
+ MAX_IDENTIFIER_LENGTH = 256
36
+ MAX_TOOL_NAME_LENGTH = 128
37
+
38
+ #: Identifiers are trimmed and must survive trimming.
39
+ Identifier = Annotated[
40
+ str,
41
+ StringConstraints(strip_whitespace=True, min_length=1, max_length=MAX_IDENTIFIER_LENGTH),
42
+ ]
43
+
44
+ #: Tool names must look like callable identifiers: no spaces, no path separators,
45
+ #: nothing that could be mistaken for a shell fragment when exported to a runner.
46
+ ToolName = Annotated[
47
+ str,
48
+ StringConstraints(
49
+ strip_whitespace=True,
50
+ min_length=1,
51
+ max_length=MAX_TOOL_NAME_LENGTH,
52
+ pattern=r"^[A-Za-z_][A-Za-z0-9_.-]*$",
53
+ ),
54
+ ]
55
+
56
+ NonEmptyStr = Annotated[str, StringConstraints(strip_whitespace=True, min_length=1)]
57
+
58
+
59
+ class Role(StrEnum):
60
+ USER = "user"
61
+ ASSISTANT = "assistant"
62
+ SYSTEM = "system"
63
+ TOOL = "tool"
64
+
65
+
66
+ class OutcomeStatus(StrEnum):
67
+ SUCCESS = "success"
68
+ FAILURE = "failure"
69
+ ERROR = "error"
70
+ UNKNOWN = "unknown"
71
+
72
+
73
+ class _Strict(BaseModel):
74
+ """Base for every trace model: unknown fields are a validation error."""
75
+
76
+ model_config = ConfigDict(extra="forbid")
77
+
78
+
79
+ def _as_utc(value: datetime | None) -> datetime | None:
80
+ """Treat a naive timestamp as UTC so orderings stay comparable."""
81
+ if value is not None and value.tzinfo is None:
82
+ return value.replace(tzinfo=UTC)
83
+ return value
84
+
85
+
86
+ class Message(_Strict):
87
+ role: Role
88
+ content: str
89
+
90
+
91
+ class TraceInput(_Strict):
92
+ """What the agent was asked to do."""
93
+
94
+ text: str | None = None
95
+ messages: list[Message] = Field(default_factory=list)
96
+
97
+ @model_validator(mode="after")
98
+ def _require_content(self) -> TraceInput:
99
+ if not (self.text or "").strip() and not self.messages:
100
+ raise ValueError("input must provide non-empty 'text' or at least one message")
101
+ return self
102
+
103
+
104
+ class TraceOutput(_Strict):
105
+ """What the agent produced. Absent for traces captured before a response."""
106
+
107
+ text: str | None = None
108
+ messages: list[Message] = Field(default_factory=list)
109
+
110
+
111
+ class _BaseEvent(_Strict):
112
+ event_id: Identifier
113
+ timestamp: datetime | None = None
114
+
115
+ _normalize_timestamp = field_validator("timestamp")(_as_utc)
116
+
117
+
118
+ class MessageEvent(_BaseEvent):
119
+ type: Literal["message"] = "message"
120
+ role: Role
121
+ content: str
122
+
123
+
124
+ class ToolCallEvent(_BaseEvent):
125
+ type: Literal["tool_call"] = "tool_call"
126
+ tool: ToolName
127
+ arguments: dict[str, Any] = Field(default_factory=dict)
128
+ call_id: Identifier | None = None
129
+
130
+
131
+ class ToolResultEvent(_BaseEvent):
132
+ type: Literal["tool_result"] = "tool_result"
133
+ tool: ToolName
134
+ call_id: Identifier | None = None
135
+ result: Any = None
136
+ error: str | None = None
137
+
138
+
139
+ class EvaluationEvent(_BaseEvent):
140
+ """An evaluator's verdict recorded alongside the interaction."""
141
+
142
+ type: Literal["evaluation"] = "evaluation"
143
+ name: NonEmptyStr
144
+ passed: bool | None = None
145
+ score: float | None = None
146
+ reason: str | None = None
147
+
148
+
149
+ Event = Annotated[
150
+ MessageEvent | ToolCallEvent | ToolResultEvent | EvaluationEvent,
151
+ Field(discriminator="type"),
152
+ ]
153
+
154
+
155
+ class Feedback(_Strict):
156
+ """Human or automated feedback attached to the interaction."""
157
+
158
+ rating: Literal["positive", "negative"] | None = None
159
+ score: float | None = None
160
+ comment: str | None = None
161
+
162
+
163
+ class Evaluation(_Strict):
164
+ name: NonEmptyStr
165
+ passed: bool | None = None
166
+ score: float | None = None
167
+ reason: str | None = None
168
+
169
+
170
+ class Outcome(_Strict):
171
+ """The three evidence sources the failure detectors read in guide 8D."""
172
+
173
+ status: OutcomeStatus = OutcomeStatus.UNKNOWN
174
+ feedback: Feedback | None = None
175
+ evaluations: list[Evaluation] = Field(default_factory=list)
176
+
177
+
178
+ class TraceMetadata(_Strict):
179
+ recorded_at: datetime | None = None
180
+ source: str | None = None
181
+ agent: str | None = None
182
+ model: str | None = None
183
+ tags: list[str] = Field(default_factory=list)
184
+ #: The one open field. Provider-specific data goes here and is redacted
185
+ #: recursively; nothing else in the schema accepts unknown keys.
186
+ extra: dict[str, Any] = Field(default_factory=dict)
187
+
188
+ _normalize_recorded_at = field_validator("recorded_at")(_as_utc)
189
+
190
+
191
+ class NormalizedTrace(_Strict):
192
+ """One recorded interaction, in the only shape Evalkeep stores."""
193
+
194
+ schema_version: int = SCHEMA_VERSION
195
+ trace_id: Identifier
196
+ input: TraceInput
197
+ output: TraceOutput | None = None
198
+ events: list[Event] = Field(default_factory=list)
199
+ outcome: Outcome = Field(default_factory=Outcome)
200
+ metadata: TraceMetadata = Field(default_factory=TraceMetadata)
201
+
202
+ @model_validator(mode="after")
203
+ def _validate_event_sequence(self) -> NormalizedTrace:
204
+ """Events must be uniquely identified, ordered, and internally consistent."""
205
+ seen_event_ids: set[str] = set()
206
+ open_calls: set[str] = set()
207
+ previous: datetime | None = None
208
+
209
+ for index, event in enumerate(self.events):
210
+ where = f"events[{index}]"
211
+ if event.event_id in seen_event_ids:
212
+ raise ValueError(f"{where}: duplicate event_id {event.event_id!r}")
213
+ seen_event_ids.add(event.event_id)
214
+
215
+ if event.timestamp is not None:
216
+ if previous is not None and event.timestamp < previous:
217
+ raise ValueError(
218
+ f"{where}: timestamp {event.timestamp.isoformat()} is earlier than "
219
+ f"the preceding event's {previous.isoformat()}; events must be ordered"
220
+ )
221
+ previous = event.timestamp
222
+
223
+ if isinstance(event, ToolCallEvent) and event.call_id is not None:
224
+ if event.call_id in open_calls:
225
+ raise ValueError(f"{where}: duplicate call_id {event.call_id!r}")
226
+ open_calls.add(event.call_id)
227
+ elif isinstance(event, ToolResultEvent) and event.call_id is not None:
228
+ if event.call_id not in open_calls:
229
+ raise ValueError(
230
+ f"{where}: call_id {event.call_id!r} does not match an earlier tool_call"
231
+ )
232
+
233
+ return self
234
+
235
+ @property
236
+ def tool_calls(self) -> list[ToolCallEvent]:
237
+ """Observed tool calls, in order. Used by detection and test generation."""
238
+ return [event for event in self.events if isinstance(event, ToolCallEvent)]
@@ -0,0 +1,221 @@
1
+ Metadata-Version: 2.4
2
+ Name: evalkeep
3
+ Version: 0.1.0
4
+ Summary: Turn production agent failures into a small, reviewed regression suite.
5
+ Keywords: llm,evals,evaluation,regression-testing,ai-agents,testing,promptfoo
6
+ Author: Rakshita Devurkar
7
+ Author-email: Rakshita Devurkar <13130544+rakshita-devurkar@users.noreply.github.com>
8
+ License-Expression: Apache-2.0
9
+ License-File: LICENSE
10
+ Classifier: Development Status :: 3 - Alpha
11
+ Classifier: Environment :: Console
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Operating System :: OS Independent
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.11
16
+ Classifier: Programming Language :: Python :: 3.12
17
+ Classifier: Programming Language :: Python :: 3.13
18
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
19
+ Classifier: Topic :: Software Development :: Quality Assurance
20
+ Classifier: Topic :: Software Development :: Testing
21
+ Classifier: Typing :: Typed
22
+ Requires-Dist: click>=8.5.0
23
+ Requires-Dist: numpy>=2.4.6
24
+ Requires-Dist: pydantic>=2.13.5
25
+ Requires-Dist: pyyaml>=6.0.3
26
+ Requires-Dist: rich>=15.0.0
27
+ Requires-Dist: typer>=0.27.2
28
+ Requires-Dist: anthropic>=1.0.0 ; extra == 'anthropic'
29
+ Requires-Python: >=3.11
30
+ Project-URL: Homepage, https://github.com/rakshita-devurkar/evalkeep
31
+ Project-URL: Repository, https://github.com/rakshita-devurkar/evalkeep
32
+ Project-URL: Issues, https://github.com/rakshita-devurkar/evalkeep/issues
33
+ Project-URL: Changelog, https://github.com/rakshita-devurkar/evalkeep/blob/main/CHANGELOG.md
34
+ Project-URL: Documentation, https://github.com/rakshita-devurkar/evalkeep/blob/main/docs/pipeline.md
35
+ Provides-Extra: anthropic
36
+ Description-Content-Type: text/markdown
37
+
38
+ # Evalkeep
39
+
40
+ **Stop fixing the same agent bug twice.** Evalkeep turns production failures
41
+ into a small, reviewed regression suite, and tells you whether a fix held — or
42
+ that the evidence is too thin to say.
43
+
44
+ Existing eval tools *execute* tests. The hard part is deciding which of
45
+ thousands of production traces deserve permanent coverage. Evalkeep owns that
46
+ decision:
47
+
48
+ ```
49
+ trace → failure evidence → failure family → representative case
50
+ → reviewed regression test → runner execution → trustworthy comparison
51
+ ```
52
+
53
+ **Evalkeep does not run your agent and is not an eval framework.** It generates
54
+ tests, delegates execution to [Promptfoo](https://promptfoo.dev), and compares
55
+ baseline against candidate. It sits upstream of your eval runner, not next to it.
56
+
57
+ ## Install
58
+
59
+ ```bash
60
+ git clone https://github.com/rakshita-devurkar/evalkeep && cd evalkeep
61
+ uv sync
62
+ ```
63
+
64
+ Node.js is needed only for `evalkeep run`, which shells out to Promptfoo.
65
+
66
+ ## Quick start
67
+
68
+ One command takes a trace file to a review queue. Offline, no API key:
69
+
70
+ ```bash
71
+ uv run evalkeep demo .
72
+ uv run evalkeep init
73
+ uv run evalkeep from-traces refund-agent/traces.jsonl
74
+ ```
75
+
76
+ ```
77
+ 5 trace(s) ingested
78
+ 3 failure(s) found explicit_status x3, failed_evaluator x1, negative_feedback x2
79
+ 2 failure famil(ies)
80
+ 3 with enough evidence for a regression test
81
+ 3 of them only forbid the mistake that was observed; say what should have happened at review.
82
+
83
+ Review them: evalkeep review (3 pending)
84
+ ```
85
+
86
+ That second qualifier is the honest state of a first run: nothing in a trace
87
+ says what the agent *should* have done. Describing the failures closes it, by
88
+ hand at review or with an analyzer configured. It stops at review on purpose —
89
+ approving a test is a judgement.
90
+
91
+ ## On real agent data
92
+
93
+ The example above is five traces. Here is the same pipeline on
94
+ [tau-bench trajectories](https://huggingface.co/datasets/AgentSuite/tau-bench-trajectories):
95
+ 165 retail and airline customer-service tasks per model, each scored by
96
+ comparing the final database state against the expected one. Real tool calls,
97
+ and an independent verdict — the two halves a regression suite needs.
98
+
99
+ ```bash
100
+ python tau-bench/prepare.py # ~8 MB, two models
101
+ uv run evalkeep from-traces tau-bench/Qwen3-235B-A22B-FP8.traces.jsonl
102
+ ```
103
+
104
+ ```
105
+ 165 trace(s) ingested
106
+ 90 failure(s) found explicit_status x90, failed_evaluator x90
107
+ 4 failure famil(ies)
108
+ ```
109
+
110
+ Four families, from 90 failures: retail exchange and refund flows, airline
111
+ reservation changes, and two shapes of giving up and escalating to a human.
112
+ Build a test per failure, approve them, and run two recorded models against the
113
+ suite one of them produced:
114
+
115
+ ```bash
116
+ uv run evalkeep dataset build --all
117
+ uv run evalkeep review
118
+ uv run evalkeep targets add baseline --type python --function call_api \
119
+ --path tau-bench/replay_Qwen3_235B_A22B_FP8.py
120
+ uv run evalkeep targets add candidate --type python --function call_api \
121
+ --path tau-bench/replay_claude_4_5_sonnet_thinking_off.py
122
+ uv run evalkeep run --target baseline && uv run evalkeep run --target candidate
123
+ uv run evalkeep compare
124
+ ```
125
+
126
+ ```
127
+ compared 89
128
+ baseline pass rate 4.5%
129
+ candidate pass rate 42.7%
130
+ difference +38.2%
131
+ p-value 0.0000
132
+ 95% interval +27.2% to +49.2%
133
+ McNemar's exact test: the change is unlikely to be chance.
134
+
135
+ 1 test(s) excluded and not counted in any rate above.
136
+ ```
137
+
138
+ Baseline scoring 4.5% is the control: the tests came from its own failures, so
139
+ it should fail nearly all of them. The four it passes are the documented
140
+ weakness of deriving a test with nothing describing the failure — assertions
141
+ target the last tool call, which is sometimes a harmless lookup. The excluded
142
+ test is one whose target raised rather than answered, and it is kept out of
143
+ every rate rather than counted as a failure.
144
+
145
+ ## Doing it stage by stage
146
+
147
+ `from-traces` runs five commands in order. Each is a real decision with its own
148
+ evidence, and you will want them separately once you are tuning a suite:
149
+
150
+ ```bash
151
+ uv run evalkeep ingest traces.jsonl # validate, redact, store
152
+ uv run evalkeep detect # evidence-backed failures
153
+ uv run evalkeep analyze # describe them (or: failures label)
154
+ uv run evalkeep discover # embed, cluster, pick representatives
155
+ uv run evalkeep dataset build # draft a test per representative
156
+ ```
157
+
158
+ Re-running `from-traces` is safe: it skips traces it already has and rebuilds
159
+ drafts, but never touches a test you have reviewed.
160
+
161
+ ## Bring your own traces
162
+
163
+ ```bash
164
+ uv run evalkeep ingest spans.json --format otlp # OpenTelemetry / OpenInference
165
+ uv run evalkeep ingest runs.jsonl --format langsmith # LangSmith
166
+ ```
167
+
168
+ Adapters read files, never APIs — no credentials, any vendor tier. OpenTelemetry
169
+ covers the most ground, since Langfuse, Braintrust and Phoenix all ingest OTLP.
170
+ `evalkeep demo` writes an example export in each format.
171
+
172
+ ## Commands
173
+
174
+ | Stage | Commands |
175
+ | --- | --- |
176
+ | Set up | `init`, `targets add/list/show/remove` |
177
+ | Ingest | `ingest`, `trace list/show` |
178
+ | Detect | `detect`, `failures list/show/confirm/dismiss/add` |
179
+ | Analyze | `analyze`, `failures label` |
180
+ | Group | `discover`, `clusters list/show/rename/merge/split/dismiss/restore` |
181
+ | Build | `dataset build/list/show` |
182
+ | Review | `review`, `dataset approve/reject/edit` |
183
+ | Run | `export`, `run --target ...`, `runs list/show` |
184
+ | Compare | `compare`, `baseline promote/show` |
185
+
186
+ ## What it guarantees
187
+
188
+ - **Values are redacted before storage** — in memory, with no path around it.
189
+ Identifiers can be [pseudonymized](docs/security.md#identifiers) too.
190
+ - **Automation never overwrites human judgement.** Re-running any stage
191
+ refreshes derived data and leaves your reviews, labels and edits alone.
192
+ - **Nothing is exported without approval.** Generated tests are drafts.
193
+ - **A test that never ran is not a test that failed.** Timeouts and crashed
194
+ providers are excluded from every rate, so an outage cannot read as a regression.
195
+ - **One lucky pass is not a fix.** `run --repetitions N` reports a per-case
196
+ verdict; a case that only sometimes passes is flaky, never passing.
197
+ - **Score changes are not overclaimed.** McNemar's exact test, and no confidence
198
+ interval when the sample cannot support one.
199
+
200
+ ## More
201
+
202
+ [How it works](docs/pipeline.md) · [Privacy and security](docs/security.md) ·
203
+ [Roadmap](docs/roadmap.md) · [Contributing](CONTRIBUTING.md) ·
204
+ [Changelog](CHANGELOG.md)
205
+
206
+ ```bash
207
+ uv sync && uv run pytest # 985 tests
208
+ uv run ruff check . && uv run mypy # lint and strict types
209
+ ```
210
+
211
+ `EVALKEEP_E2E=1 uv run pytest` also runs the suite against real Promptfoo.
212
+ Exit codes: `0` success, `1` ran but some records were rejected, `2` could not
213
+ run.
214
+
215
+ 0.1 is feature-complete and not yet released to PyPI. Multi-turn replay and
216
+ longitudinal failure history are still open; see the
217
+ [roadmap](docs/roadmap.md).
218
+
219
+ ## License
220
+
221
+ Apache-2.0. See [LICENSE](LICENSE).