evalkeep 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. evalkeep/__init__.py +12 -0
  2. evalkeep/__main__.py +6 -0
  3. evalkeep/adapters/__init__.py +45 -0
  4. evalkeep/adapters/base.py +92 -0
  5. evalkeep/adapters/jsonl.py +164 -0
  6. evalkeep/adapters/langsmith.py +436 -0
  7. evalkeep/adapters/otlp.py +442 -0
  8. evalkeep/adapters/semconv.py +208 -0
  9. evalkeep/analysis.py +174 -0
  10. evalkeep/analysis_run.py +160 -0
  11. evalkeep/analyzers/__init__.py +52 -0
  12. evalkeep/analyzers/anthropic.py +145 -0
  13. evalkeep/analyzers/stub.py +34 -0
  14. evalkeep/cache.py +122 -0
  15. evalkeep/cli.py +1933 -0
  16. evalkeep/clustering.py +383 -0
  17. evalkeep/clusters.py +101 -0
  18. evalkeep/commands/__init__.py +1 -0
  19. evalkeep/commands/analyze_cmd.py +100 -0
  20. evalkeep/commands/compare_cmd.py +169 -0
  21. evalkeep/commands/dataset_cmd.py +182 -0
  22. evalkeep/commands/detect_cmd.py +154 -0
  23. evalkeep/commands/discover_cmd.py +274 -0
  24. evalkeep/commands/ingest_cmd.py +50 -0
  25. evalkeep/commands/init_cmd.py +151 -0
  26. evalkeep/commands/pipeline_cmd.py +156 -0
  27. evalkeep/commands/review_cmd.py +141 -0
  28. evalkeep/commands/run_cmd.py +131 -0
  29. evalkeep/commands/target_cmd.py +109 -0
  30. evalkeep/commands/trace_cmd.py +58 -0
  31. evalkeep/comparison.py +432 -0
  32. evalkeep/config.py +209 -0
  33. evalkeep/detection.py +94 -0
  34. evalkeep/detectors.py +182 -0
  35. evalkeep/discovery.py +208 -0
  36. evalkeep/embeddings/__init__.py +31 -0
  37. evalkeep/embeddings/base.py +32 -0
  38. evalkeep/embeddings/hashing.py +98 -0
  39. evalkeep/errors.py +42 -0
  40. evalkeep/examples/__init__.py +37 -0
  41. evalkeep/examples/langsmith/runs.jsonl +18 -0
  42. evalkeep/examples/opentelemetry/spans.json +898 -0
  43. evalkeep/examples/refund-agent/agents/baseline.py +66 -0
  44. evalkeep/examples/refund-agent/agents/candidate.py +66 -0
  45. evalkeep/examples/refund-agent/traces.jsonl +5 -0
  46. evalkeep/examples/tau-bench/prepare.py +230 -0
  47. evalkeep/exporters/__init__.py +45 -0
  48. evalkeep/exporters/generic.py +31 -0
  49. evalkeep/exporters/promptfoo.py +219 -0
  50. evalkeep/failures.py +95 -0
  51. evalkeep/generation.py +303 -0
  52. evalkeep/hashing.py +56 -0
  53. evalkeep/ingest.py +257 -0
  54. evalkeep/prompts.py +127 -0
  55. evalkeep/pseudonyms.py +82 -0
  56. evalkeep/py.typed +0 -0
  57. evalkeep/redaction.py +333 -0
  58. evalkeep/regression.py +409 -0
  59. evalkeep/review.py +309 -0
  60. evalkeep/runner.py +302 -0
  61. evalkeep/runs.py +185 -0
  62. evalkeep/storage/__init__.py +37 -0
  63. evalkeep/storage/clusters.py +163 -0
  64. evalkeep/storage/failures.py +254 -0
  65. evalkeep/storage/migrations.py +370 -0
  66. evalkeep/storage/regression.py +136 -0
  67. evalkeep/storage/runs.py +223 -0
  68. evalkeep/storage/store.py +429 -0
  69. evalkeep/targets.py +205 -0
  70. evalkeep/trace.py +238 -0
  71. evalkeep-0.1.0.dist-info/METADATA +221 -0
  72. evalkeep-0.1.0.dist-info/RECORD +75 -0
  73. evalkeep-0.1.0.dist-info/WHEEL +4 -0
  74. evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
  75. evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/regression.py ADDED
@@ -0,0 +1,409 @@
1
+ """Regression tests: what a failure must never do again.
2
+
3
+ A regression test preserves an observed failure as a permanent check. The shape
4
+ of one follows from what a trace can and cannot tell us:
5
+
6
+ * A trace shows exactly **what the agent did wrong**, so the forbidding half of
7
+ a test -- "do not refund an older order" -- is derivable and deterministic.
8
+ * A trace does **not** show what the agent should have done instead. "Refund
9
+ exactly the newest order" is a judgement about intent that no amount of
10
+ reading the trace can supply.
11
+
12
+ So generation produces the forbidding half automatically and marks the test as
13
+ needing the positive half from a reviewer. That split is why drafts are never
14
+ exported without approval.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import hashlib
20
+ import re
21
+ from dataclasses import dataclass, field
22
+ from datetime import UTC, datetime
23
+ from enum import StrEnum
24
+ from itertools import combinations
25
+ from typing import Any
26
+
27
+ MAX_SLUG_WORDS = 6
28
+ MAX_SLUG_LENGTH = 48
29
+ _TEST_ID_DIGEST_LENGTH = 8
30
+
31
+ _WORD_PATTERN = re.compile(r"[a-z0-9]+")
32
+ _REDACTION_PATTERN = re.compile(r"\[REDACTED:[a-z_]+\]", re.IGNORECASE)
33
+
34
+
35
+ class ExpectationType(StrEnum):
36
+ OUTPUT_CONTAINS = "output_contains"
37
+ OUTPUT_NOT_CONTAINS = "output_not_contains"
38
+ OUTPUT_MATCHES = "output_matches"
39
+ TOOL_CALLED = "tool_called"
40
+ TOOL_NOT_CALLED = "tool_not_called"
41
+ TOOL_ARGUMENT_EQUALS = "tool_argument_equals"
42
+ TOOL_ARGUMENT_NOT_EQUALS = "tool_argument_not_equals"
43
+ MAX_TOOL_CALLS = "max_tool_calls"
44
+ #: The escape hatch for cases with no deterministic check. Costs an LLM
45
+ #: judge at run time, so it is used only where nothing else applies.
46
+ HUMAN_RUBRIC = "human_rubric"
47
+
48
+
49
+ #: Expectations that say what the agent *should* do. A test made only of the
50
+ #: others records a prohibition without an intent, which is half a test.
51
+ POSITIVE_TYPES: frozenset[ExpectationType] = frozenset(
52
+ {
53
+ ExpectationType.OUTPUT_CONTAINS,
54
+ ExpectationType.OUTPUT_MATCHES,
55
+ ExpectationType.TOOL_CALLED,
56
+ ExpectationType.TOOL_ARGUMENT_EQUALS,
57
+ ExpectationType.HUMAN_RUBRIC,
58
+ }
59
+ )
60
+
61
+ #: Everything except the rubric can be checked without a model.
62
+ DETERMINISTIC_TYPES: frozenset[ExpectationType] = frozenset(
63
+ set(ExpectationType) - {ExpectationType.HUMAN_RUBRIC}
64
+ )
65
+
66
+
67
+ class ReviewStatus(StrEnum):
68
+ """The review lifecycle. Generation only ever writes ``DRAFT``."""
69
+
70
+ DRAFT = "draft"
71
+ APPROVED = "approved"
72
+ REJECTED = "rejected"
73
+
74
+
75
+ @dataclass(frozen=True)
76
+ class Expectation:
77
+ type: ExpectationType
78
+ value: Any = None
79
+ tool: str | None = None
80
+ #: Argument path within a tool call, dotted for nested values.
81
+ path: str | None = None
82
+
83
+ @property
84
+ def deterministic(self) -> bool:
85
+ return self.type in DETERMINISTIC_TYPES
86
+
87
+ @property
88
+ def positive(self) -> bool:
89
+ return self.type in POSITIVE_TYPES
90
+
91
+ def describe(self) -> str:
92
+ match self.type:
93
+ case ExpectationType.TOOL_ARGUMENT_EQUALS:
94
+ return f"{self.tool}.{self.path} == {self.value!r}"
95
+ case ExpectationType.TOOL_ARGUMENT_NOT_EQUALS:
96
+ return f"{self.tool}.{self.path} != {self.value!r}"
97
+ case ExpectationType.TOOL_CALLED:
98
+ return f"calls {self.tool}"
99
+ case ExpectationType.TOOL_NOT_CALLED:
100
+ return f"never calls {self.tool}"
101
+ case ExpectationType.MAX_TOOL_CALLS:
102
+ target = self.tool or "any tool"
103
+ return f"at most {self.value} calls to {target}"
104
+ case ExpectationType.HUMAN_RUBRIC:
105
+ return f"rubric: {self.value}"
106
+ case _:
107
+ return f"{self.type.value} {self.value!r}"
108
+
109
+ def to_dict(self) -> dict[str, Any]:
110
+ payload: dict[str, Any] = {"type": self.type.value, "value": self.value}
111
+ if self.tool is not None:
112
+ payload["tool"] = self.tool
113
+ if self.path is not None:
114
+ payload["path"] = self.path
115
+ return payload
116
+
117
+ @classmethod
118
+ def from_dict(cls, payload: dict[str, Any]) -> Expectation:
119
+ return cls(
120
+ type=ExpectationType(payload["type"]),
121
+ value=payload.get("value"),
122
+ tool=payload.get("tool"),
123
+ path=payload.get("path"),
124
+ )
125
+
126
+
127
+ @dataclass(frozen=True)
128
+ class Contradiction:
129
+ """Two expectations that cannot both hold."""
130
+
131
+ first: Expectation
132
+ second: Expectation
133
+ reason: str
134
+
135
+ def describe(self) -> str:
136
+ return f"{self.first.describe()} vs {self.second.describe()}: {self.reason}"
137
+
138
+
139
+ def validate_expectation(expectation: Expectation) -> str | None:
140
+ """Reasons an expectation could never be evaluated, or ``None`` if it can."""
141
+ match expectation.type:
142
+ case ExpectationType.TOOL_ARGUMENT_EQUALS | ExpectationType.TOOL_ARGUMENT_NOT_EQUALS:
143
+ if not expectation.tool or not expectation.path:
144
+ return "an argument expectation needs both a tool and a path"
145
+ case ExpectationType.TOOL_CALLED | ExpectationType.TOOL_NOT_CALLED:
146
+ if not expectation.tool:
147
+ return "a tool expectation needs a tool"
148
+ case ExpectationType.MAX_TOOL_CALLS:
149
+ if not isinstance(expectation.value, int) or isinstance(expectation.value, bool):
150
+ return "max_tool_calls needs a whole number"
151
+ if expectation.value < 0:
152
+ return "max_tool_calls cannot be negative"
153
+ case ExpectationType.OUTPUT_MATCHES:
154
+ if not isinstance(expectation.value, str):
155
+ return "output_matches needs a pattern"
156
+ try:
157
+ re.compile(expectation.value)
158
+ except re.error as exc:
159
+ return f"output_matches pattern does not compile: {exc}"
160
+ case ExpectationType.OUTPUT_CONTAINS | ExpectationType.OUTPUT_NOT_CONTAINS:
161
+ if not isinstance(expectation.value, str) or not expectation.value.strip():
162
+ return f"{expectation.type.value} needs non-empty text"
163
+ case ExpectationType.HUMAN_RUBRIC:
164
+ if not isinstance(expectation.value, str) or not expectation.value.strip():
165
+ return "human_rubric needs a description of what should happen"
166
+ return None
167
+
168
+
169
+ def find_contradictions(expectations: list[Expectation]) -> list[Contradiction]:
170
+ """Every pair of expectations that cannot both be satisfied.
171
+
172
+ Run at generation time so a draft is never saved self-defeating, and again
173
+ at review time so an edit cannot introduce one. A contradictory test fails
174
+ on every agent, including a correct one, which makes it worse than no test:
175
+ it reports a regression that is really a bug in the suite.
176
+ """
177
+ found: list[Contradiction] = []
178
+ for first, second in combinations(expectations, 2):
179
+ reason = _conflict(first, second) or _conflict(second, first)
180
+ if reason is not None:
181
+ found.append(Contradiction(first=first, second=second, reason=reason))
182
+ return found
183
+
184
+
185
+ def _conflict(one: Expectation, two: Expectation) -> str | None:
186
+ if (
187
+ one.type is ExpectationType.TOOL_CALLED
188
+ and two.type is ExpectationType.TOOL_NOT_CALLED
189
+ and one.tool == two.tool
190
+ ):
191
+ return "the same tool cannot be both required and forbidden"
192
+
193
+ if (
194
+ one.type is ExpectationType.TOOL_NOT_CALLED
195
+ and two.type
196
+ in {ExpectationType.TOOL_ARGUMENT_EQUALS, ExpectationType.TOOL_ARGUMENT_NOT_EQUALS}
197
+ and one.tool == two.tool
198
+ ):
199
+ return "an argument of a forbidden tool can never be checked"
200
+
201
+ if (
202
+ one.type is ExpectationType.TOOL_CALLED
203
+ and two.type is ExpectationType.MAX_TOOL_CALLS
204
+ and two.value == 0
205
+ and (two.tool is None or two.tool == one.tool)
206
+ ):
207
+ return "a required tool cannot also be capped at zero calls"
208
+
209
+ if (
210
+ one.type is ExpectationType.TOOL_ARGUMENT_EQUALS
211
+ and two.type is ExpectationType.TOOL_ARGUMENT_NOT_EQUALS
212
+ and (one.tool, one.path) == (two.tool, two.path)
213
+ and one.value == two.value
214
+ ):
215
+ return "the same argument cannot be required and forbidden to equal one value"
216
+
217
+ if (
218
+ one.type is two.type is ExpectationType.TOOL_ARGUMENT_EQUALS
219
+ and (one.tool, one.path) == (two.tool, two.path)
220
+ and one.value != two.value
221
+ ):
222
+ return "one argument cannot equal two different values"
223
+
224
+ if (
225
+ one.type is ExpectationType.OUTPUT_CONTAINS
226
+ and two.type is ExpectationType.OUTPUT_NOT_CONTAINS
227
+ and one.value == two.value
228
+ ):
229
+ return "the output cannot both contain and not contain the same text"
230
+
231
+ if (
232
+ one.type is two.type is ExpectationType.MAX_TOOL_CALLS
233
+ and one.tool == two.tool
234
+ and one.value != two.value
235
+ ):
236
+ return "two different call limits for the same tool"
237
+
238
+ return None
239
+
240
+
241
+ @dataclass
242
+ class CaseInput:
243
+ """What the test sends to the agent."""
244
+
245
+ text: str | None = None
246
+ messages: list[dict[str, str]] = field(default_factory=list)
247
+
248
+ def to_dict(self) -> dict[str, Any]:
249
+ return {"text": self.text, "messages": self.messages}
250
+
251
+ @classmethod
252
+ def from_dict(cls, payload: dict[str, Any]) -> CaseInput:
253
+ return cls(text=payload.get("text"), messages=payload.get("messages", []))
254
+
255
+
256
+ @dataclass
257
+ class Fixture:
258
+ """A tool result the original agent saw, so a replay can reproduce it."""
259
+
260
+ tool: str
261
+ arguments: dict[str, Any] = field(default_factory=dict)
262
+ result: Any = None
263
+ error: str | None = None
264
+ call_id: str | None = None
265
+
266
+ def to_dict(self) -> dict[str, Any]:
267
+ return {
268
+ "tool": self.tool,
269
+ "arguments": self.arguments,
270
+ "result": self.result,
271
+ "error": self.error,
272
+ "call_id": self.call_id,
273
+ }
274
+
275
+ def for_replay(self) -> dict[str, Any]:
276
+ """What a target needs to reproduce this call.
277
+
278
+ ``call_id`` is left out: it identifies the original recording, not the
279
+ interaction, and a replaying target has no use for it.
280
+ """
281
+ return {
282
+ "tool": self.tool,
283
+ "arguments": self.arguments,
284
+ "result": self.result,
285
+ "error": self.error,
286
+ }
287
+
288
+ @classmethod
289
+ def from_dict(cls, payload: dict[str, Any]) -> Fixture:
290
+ return cls(
291
+ tool=payload["tool"],
292
+ arguments=payload.get("arguments", {}),
293
+ result=payload.get("result"),
294
+ error=payload.get("error"),
295
+ call_id=payload.get("call_id"),
296
+ )
297
+
298
+
299
+ @dataclass
300
+ class Provenance:
301
+ """Where a test came from, in enough detail to defend or discard it."""
302
+
303
+ trace_id: str
304
+ failure_id: str
305
+ content_hash: str
306
+ #: Recorded, but never part of the test ID: clusters are rebuilt and
307
+ #: relabelled, and an ID that moved with them would not be stable.
308
+ cluster_id: str | None = None
309
+ cluster_label: str | None = None
310
+ representative_roles: list[str] = field(default_factory=list)
311
+ failure_type: str | None = None
312
+ severity: str | None = None
313
+ analyzer: str | None = None
314
+ analysis_summary: str | None = None
315
+ evidence: list[str] = field(default_factory=list)
316
+ generator_version: int = 1
317
+ generated_at: datetime = field(default_factory=lambda: datetime.now(UTC))
318
+
319
+ def to_dict(self) -> dict[str, Any]:
320
+ payload = {
321
+ key: getattr(self, key)
322
+ for key in (
323
+ "trace_id",
324
+ "failure_id",
325
+ "content_hash",
326
+ "cluster_id",
327
+ "cluster_label",
328
+ "representative_roles",
329
+ "failure_type",
330
+ "severity",
331
+ "analyzer",
332
+ "analysis_summary",
333
+ "evidence",
334
+ "generator_version",
335
+ )
336
+ }
337
+ payload["generated_at"] = self.generated_at.isoformat()
338
+ return payload
339
+
340
+ @classmethod
341
+ def from_dict(cls, payload: dict[str, Any]) -> Provenance:
342
+ data = dict(payload)
343
+ generated_at = data.pop("generated_at", None)
344
+ return cls(
345
+ **data,
346
+ generated_at=(
347
+ datetime.fromisoformat(generated_at) if generated_at else datetime.now(UTC)
348
+ ),
349
+ )
350
+
351
+
352
+ @dataclass
353
+ class RegressionTest:
354
+ test_id: str
355
+ failure_id: str
356
+ input: CaseInput
357
+ provenance: Provenance
358
+ status: ReviewStatus = ReviewStatus.DRAFT
359
+ fixtures: list[Fixture] = field(default_factory=list)
360
+ expectations: list[Expectation] = field(default_factory=list)
361
+ #: Things a reviewer must decide, recorded rather than guessed at.
362
+ warnings: list[str] = field(default_factory=list)
363
+ reviewer: str | None = None
364
+ review_reason: str | None = None
365
+ reviewed_at: datetime | None = None
366
+ #: True once a person changed the generated content, not merely approved it.
367
+ edited: bool = False
368
+ edited_by: str | None = None
369
+ created_at: datetime = field(default_factory=lambda: datetime.now(UTC))
370
+ updated_at: datetime = field(default_factory=lambda: datetime.now(UTC))
371
+
372
+ @property
373
+ def reviewed(self) -> bool:
374
+ return self.status is not ReviewStatus.DRAFT
375
+
376
+ @property
377
+ def deterministic_expectations(self) -> list[Expectation]:
378
+ return [item for item in self.expectations if item.deterministic]
379
+
380
+ @property
381
+ def has_positive_expectation(self) -> bool:
382
+ return any(item.positive for item in self.expectations)
383
+
384
+ @property
385
+ def contradictions(self) -> list[Contradiction]:
386
+ return find_contradictions(self.expectations)
387
+
388
+
389
+ def slugify(text: str) -> str:
390
+ """A short, readable, stable stem for a test ID."""
391
+ cleaned = _REDACTION_PATTERN.sub(" ", text or "").lower()
392
+ words = _WORD_PATTERN.findall(cleaned)[:MAX_SLUG_WORDS]
393
+ return "_".join(words)[:MAX_SLUG_LENGTH].strip("_")
394
+
395
+
396
+ def make_test_id(trace_id: str, input_text: str) -> str:
397
+ """A readable, stable test ID derived only from immutable facts.
398
+
399
+ Deliberately *not* derived from the cluster label or the analysis: both are
400
+ mutable. A reviewer renaming a cluster, or a re-analysis changing a failure
401
+ type, must not rename a test that is already committed to Git and referenced
402
+ by past run results.
403
+
404
+ The stem comes from the trace's own input, which never changes once stored,
405
+ and the suffix from the trace ID, which guarantees uniqueness.
406
+ """
407
+ digest = hashlib.sha256(trace_id.encode("utf-8")).hexdigest()[:_TEST_ID_DIGEST_LENGTH]
408
+ stem = slugify(input_text)
409
+ return f"{stem}_{digest}" if stem else f"test_{digest}"