evalkeep 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. evalkeep/__init__.py +12 -0
  2. evalkeep/__main__.py +6 -0
  3. evalkeep/adapters/__init__.py +45 -0
  4. evalkeep/adapters/base.py +92 -0
  5. evalkeep/adapters/jsonl.py +164 -0
  6. evalkeep/adapters/langsmith.py +436 -0
  7. evalkeep/adapters/otlp.py +442 -0
  8. evalkeep/adapters/semconv.py +208 -0
  9. evalkeep/analysis.py +174 -0
  10. evalkeep/analysis_run.py +160 -0
  11. evalkeep/analyzers/__init__.py +52 -0
  12. evalkeep/analyzers/anthropic.py +145 -0
  13. evalkeep/analyzers/stub.py +34 -0
  14. evalkeep/cache.py +122 -0
  15. evalkeep/cli.py +1933 -0
  16. evalkeep/clustering.py +383 -0
  17. evalkeep/clusters.py +101 -0
  18. evalkeep/commands/__init__.py +1 -0
  19. evalkeep/commands/analyze_cmd.py +100 -0
  20. evalkeep/commands/compare_cmd.py +169 -0
  21. evalkeep/commands/dataset_cmd.py +182 -0
  22. evalkeep/commands/detect_cmd.py +154 -0
  23. evalkeep/commands/discover_cmd.py +274 -0
  24. evalkeep/commands/ingest_cmd.py +50 -0
  25. evalkeep/commands/init_cmd.py +151 -0
  26. evalkeep/commands/pipeline_cmd.py +156 -0
  27. evalkeep/commands/review_cmd.py +141 -0
  28. evalkeep/commands/run_cmd.py +131 -0
  29. evalkeep/commands/target_cmd.py +109 -0
  30. evalkeep/commands/trace_cmd.py +58 -0
  31. evalkeep/comparison.py +432 -0
  32. evalkeep/config.py +209 -0
  33. evalkeep/detection.py +94 -0
  34. evalkeep/detectors.py +182 -0
  35. evalkeep/discovery.py +208 -0
  36. evalkeep/embeddings/__init__.py +31 -0
  37. evalkeep/embeddings/base.py +32 -0
  38. evalkeep/embeddings/hashing.py +98 -0
  39. evalkeep/errors.py +42 -0
  40. evalkeep/examples/__init__.py +37 -0
  41. evalkeep/examples/langsmith/runs.jsonl +18 -0
  42. evalkeep/examples/opentelemetry/spans.json +898 -0
  43. evalkeep/examples/refund-agent/agents/baseline.py +66 -0
  44. evalkeep/examples/refund-agent/agents/candidate.py +66 -0
  45. evalkeep/examples/refund-agent/traces.jsonl +5 -0
  46. evalkeep/examples/tau-bench/prepare.py +230 -0
  47. evalkeep/exporters/__init__.py +45 -0
  48. evalkeep/exporters/generic.py +31 -0
  49. evalkeep/exporters/promptfoo.py +219 -0
  50. evalkeep/failures.py +95 -0
  51. evalkeep/generation.py +303 -0
  52. evalkeep/hashing.py +56 -0
  53. evalkeep/ingest.py +257 -0
  54. evalkeep/prompts.py +127 -0
  55. evalkeep/pseudonyms.py +82 -0
  56. evalkeep/py.typed +0 -0
  57. evalkeep/redaction.py +333 -0
  58. evalkeep/regression.py +409 -0
  59. evalkeep/review.py +309 -0
  60. evalkeep/runner.py +302 -0
  61. evalkeep/runs.py +185 -0
  62. evalkeep/storage/__init__.py +37 -0
  63. evalkeep/storage/clusters.py +163 -0
  64. evalkeep/storage/failures.py +254 -0
  65. evalkeep/storage/migrations.py +370 -0
  66. evalkeep/storage/regression.py +136 -0
  67. evalkeep/storage/runs.py +223 -0
  68. evalkeep/storage/store.py +429 -0
  69. evalkeep/targets.py +205 -0
  70. evalkeep/trace.py +238 -0
  71. evalkeep-0.1.0.dist-info/METADATA +221 -0
  72. evalkeep-0.1.0.dist-info/RECORD +75 -0
  73. evalkeep-0.1.0.dist-info/WHEEL +4 -0
  74. evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
  75. evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/redaction.py ADDED
@@ -0,0 +1,333 @@
1
+ """Deterministic redaction, applied in memory before anything reaches storage.
2
+
3
+ Redaction runs on the parsed trace and returns a new one; the raw trace is never
4
+ written to the database, exported, or sent to an analyzer. The rules err toward
5
+ over-redaction -- a redacted order total is an inconvenience, a stored API key is
6
+ an incident -- with two exceptions that keep the data usable:
7
+
8
+ * Identity fields (``trace_id``, ``event_id``, ``call_id``, ``tool`` and the
9
+ other structural keys in :data:`NEVER_REDACTED`) are never rewritten. Scrubbing
10
+ an identifier would break the very links the pipeline is built on.
11
+ * Secret *field names* only redact string and container values. ``token_count:
12
+ 512`` is a number, not a credential, and redacting it would lose real signal.
13
+
14
+ Everything is deterministic: the same input always produces the same output, with
15
+ fixed placeholders rather than hashes of the original. Two customers' email
16
+ addresses collapse to the same placeholder, which is what clustering wants.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import re
22
+ from collections.abc import Iterable
23
+ from dataclasses import dataclass, field
24
+ from enum import StrEnum
25
+ from typing import Any
26
+
27
+ from evalkeep.config import RedactionConfig
28
+ from evalkeep.pseudonyms import PREFIXES, Pseudonymizer
29
+ from evalkeep.trace import NormalizedTrace
30
+
31
+
32
+ class RedactionRule(StrEnum):
33
+ EMAIL = "email"
34
+ PHONE = "phone"
35
+ PAYMENT_CARD = "payment_card"
36
+ TOKEN = "token"
37
+ SECRET_FIELD = "secret_field"
38
+ #: An identifier replaced by a per-project token rather than a placeholder,
39
+ #: so the links it carries survive.
40
+ PSEUDONYM = "pseudonym"
41
+
42
+
43
+ def placeholder(rule: RedactionRule) -> str:
44
+ return f"[REDACTED:{rule.value}]"
45
+
46
+
47
+ #: Structural and identity keys whose values are never rewritten.
48
+ NEVER_REDACTED: frozenset[str] = frozenset(
49
+ {
50
+ "trace_id",
51
+ "event_id",
52
+ "call_id",
53
+ "schema_version",
54
+ "type",
55
+ "role",
56
+ "status",
57
+ "tool",
58
+ "rating",
59
+ "passed",
60
+ "score",
61
+ "name",
62
+ }
63
+ )
64
+
65
+ #: A key segment equal to one of these marks a credential.
66
+ SECRET_SEGMENTS: frozenset[str] = frozenset(
67
+ {
68
+ "password",
69
+ "passwd",
70
+ "secret",
71
+ "token",
72
+ "credential",
73
+ "credentials",
74
+ "apikey",
75
+ "cvv",
76
+ "cvc",
77
+ "ssn",
78
+ "authorization",
79
+ "pin",
80
+ }
81
+ )
82
+
83
+ #: A normalized key containing one of these marks a credential.
84
+ SECRET_PHRASES: tuple[str, ...] = (
85
+ "apikey",
86
+ "accesstoken",
87
+ "refreshtoken",
88
+ "idtoken",
89
+ "authtoken",
90
+ "bearertoken",
91
+ "privatekey",
92
+ "secretkey",
93
+ "clientsecret",
94
+ "sessionkey",
95
+ "creditcard",
96
+ "cardnumber",
97
+ "socialsecurity",
98
+ )
99
+
100
+ EMAIL_PATTERN = re.compile(r"[A-Za-z0-9._%+\-]+@[A-Za-z0-9.\-]+\.[A-Za-z]{2,}")
101
+
102
+ #: Requires separators, so order numbers and totals are not mistaken for phones.
103
+ PHONE_PATTERN = re.compile(
104
+ r"""(?<![\w-])
105
+ (?:\+\d{1,3}[\s.\-]?)?
106
+ (?:\(\d{3}\)\s?|\d{3}[\s.\-])
107
+ \d{3}[\s.\-]\d{4}
108
+ (?![\w-])""",
109
+ re.VERBOSE,
110
+ )
111
+ E164_PATTERN = re.compile(r"(?<![\w-])\+\d{10,15}(?![\w-])")
112
+
113
+ #: A candidate only; :func:`_luhn_ok` decides whether it is really a card. The
114
+ #: boundaries exclude only adjacent digits, so a card embedded in an identifier
115
+ #: is still caught -- over-redacting an ID beats storing a card number.
116
+ CARD_CANDIDATE_PATTERN = re.compile(r"(?<!\d)(?:\d[ \-]?){12,18}\d(?!\d)")
117
+
118
+ TOKEN_PATTERNS: tuple[re.Pattern[str], ...] = (
119
+ re.compile(r"\bsk-[A-Za-z0-9_\-]{16,}"),
120
+ re.compile(r"\bpk-[A-Za-z0-9_\-]{16,}"),
121
+ re.compile(r"\bgh[pousr]_[A-Za-z0-9]{20,}"),
122
+ re.compile(r"\bxox[baprs]-[A-Za-z0-9\-]{10,}"),
123
+ re.compile(r"\bAKIA[0-9A-Z]{16}\b"),
124
+ re.compile(r"\bASIA[0-9A-Z]{16}\b"),
125
+ re.compile(r"\bAIza[0-9A-Za-z_\-]{35}\b"),
126
+ re.compile(r"\beyJ[A-Za-z0-9_\-]{8,}\.[A-Za-z0-9_\-]{8,}\.[A-Za-z0-9_\-]{8,}"),
127
+ re.compile(r"(?i)\bbearer\s+[A-Za-z0-9._\-]{20,}"),
128
+ )
129
+
130
+
131
+ @dataclass
132
+ class RedactionSummary:
133
+ """How many values each rule replaced. Stored alongside the trace."""
134
+
135
+ counts: dict[RedactionRule, int] = field(default_factory=dict)
136
+
137
+ def record(self, rule: RedactionRule, times: int = 1) -> None:
138
+ if times:
139
+ self.counts[rule] = self.counts.get(rule, 0) + times
140
+
141
+ @property
142
+ def total(self) -> int:
143
+ return sum(self.counts.values())
144
+
145
+ def to_dict(self) -> dict[str, int]:
146
+ return {rule.value: count for rule, count in sorted(self.counts.items())}
147
+
148
+ def merge(self, other: RedactionSummary) -> None:
149
+ for rule, count in other.counts.items():
150
+ self.record(rule, count)
151
+
152
+
153
+ class Redactor:
154
+ """Applies the configured rules to a trace, in memory, before storage."""
155
+
156
+ def __init__(
157
+ self,
158
+ config: RedactionConfig | None = None,
159
+ *,
160
+ pseudonymizer: Pseudonymizer | None = None,
161
+ ) -> None:
162
+ self.config = config or RedactionConfig()
163
+ self._pseudonymizer = pseudonymizer
164
+
165
+ @property
166
+ def pseudonymizing(self) -> bool:
167
+ return self._pseudonymizer is not None
168
+
169
+ def redact(self, trace: NormalizedTrace) -> tuple[NormalizedTrace, RedactionSummary]:
170
+ """Return a redacted copy of ``trace`` and what was replaced."""
171
+ summary = RedactionSummary()
172
+ payload = trace.model_dump(mode="json")
173
+ cleaned = self._walk(payload, summary, key=None)
174
+ # Re-validating proves redaction produced a trace the rest of the
175
+ # pipeline can still read, rather than something merely dict-shaped.
176
+ return NormalizedTrace.model_validate(cleaned), summary
177
+
178
+ def redact_text(self, text: str, summary: RedactionSummary) -> str:
179
+ """Apply every enabled pattern rule to one string."""
180
+ if self.config.token_prefixes:
181
+ for pattern in TOKEN_PATTERNS:
182
+ text, hits = pattern.subn(placeholder(RedactionRule.TOKEN), text)
183
+ summary.record(RedactionRule.TOKEN, hits)
184
+ # Phones before cards, deliberately. A phone number sitting next to a
185
+ # card forms one digit run, and Luhn alone cannot say which digits
186
+ # belong to which; replacing the phone first removes the ambiguity
187
+ # rather than guessing. Phone patterns require separators in positions
188
+ # a card's groups never produce, so they cannot match inside a card.
189
+ if self.config.phone_numbers:
190
+ for pattern in (PHONE_PATTERN, E164_PATTERN):
191
+ text, hits = pattern.subn(placeholder(RedactionRule.PHONE), text)
192
+ summary.record(RedactionRule.PHONE, hits)
193
+ if self.config.payment_cards:
194
+ text = self._redact_cards(text, summary)
195
+ if self.config.emails:
196
+ text, hits = EMAIL_PATTERN.subn(placeholder(RedactionRule.EMAIL), text)
197
+ summary.record(RedactionRule.EMAIL, hits)
198
+ return text
199
+
200
+ def _redact_cards(self, text: str, summary: RedactionSummary) -> str:
201
+ """Replace only digit runs that pass Luhn, so totals and IDs survive.
202
+
203
+ Candidates are matched as whole separator-delimited groups, and a
204
+ candidate that fails Luhn is retried over its sub-runs of groups. Both
205
+ matter: a phone number sitting next to a card forms one long digit run
206
+ that fails Luhn as a whole, and searching groups rather than arbitrary
207
+ offsets finds the card inside it without inventing Luhn-valid windows
208
+ that were never a card.
209
+ """
210
+
211
+ def replace(match: re.Match[str]) -> str:
212
+ span = match.group(0)
213
+ groups = [(m.start(), m.end()) for m in re.finditer(r"\d+", span)]
214
+ for start in range(len(groups)):
215
+ for end in range(len(groups), start, -1):
216
+ digits = "".join(span[a:b] for a, b in groups[start:end])
217
+ if not 13 <= len(digits) <= 19 or not _luhn_ok(digits):
218
+ continue
219
+ summary.record(RedactionRule.PAYMENT_CARD)
220
+ low, high = groups[start][0], groups[end - 1][1]
221
+ tail = self._redact_cards(span[high:], summary)
222
+ return span[:low] + placeholder(RedactionRule.PAYMENT_CARD) + tail
223
+ return span
224
+
225
+ return CARD_CANDIDATE_PATTERN.sub(replace, text)
226
+
227
+ def _walk(self, value: Any, summary: RedactionSummary, *, key: str | None) -> Any:
228
+ if isinstance(value, dict):
229
+ cleaned: dict[str, Any] = {}
230
+ for child_key, child in value.items():
231
+ if self._is_secret_value(child_key, child):
232
+ summary.record(RedactionRule.SECRET_FIELD)
233
+ cleaned[child_key] = placeholder(RedactionRule.SECRET_FIELD)
234
+ else:
235
+ cleaned[child_key] = self._walk(child, summary, key=child_key)
236
+ return cleaned
237
+ if isinstance(value, list):
238
+ return [self._walk(item, summary, key=key) for item in value]
239
+ if isinstance(value, str):
240
+ if key is not None and key in NEVER_REDACTED:
241
+ return self._identifier(key, value, summary)
242
+ return self.redact_text(value, summary)
243
+ return value
244
+
245
+ def _identifier(self, key: str, value: str, summary: RedactionSummary) -> str:
246
+ """Identifiers survive verbatim, or become a stable per-project token.
247
+
248
+ Never a placeholder: two traces collapsing to `[REDACTED:email]` would
249
+ collide, and the pipeline is built on these values being distinct.
250
+ """
251
+ if self._pseudonymizer is None or key not in PREFIXES or not value:
252
+ return value
253
+ summary.record(RedactionRule.PSEUDONYM)
254
+ return self._pseudonymizer.token(value, field=key)
255
+
256
+ def _is_secret_value(self, key: str, value: Any) -> bool:
257
+ """A credential is a string or a container, never a count or a flag."""
258
+ if not self.config.secret_field_names:
259
+ return False
260
+ if not isinstance(value, str | dict | list):
261
+ return False
262
+ return is_secret_field(key)
263
+
264
+
265
+ def _luhn_ok(digits: str) -> bool:
266
+ total = 0
267
+ for index, character in enumerate(reversed(digits)):
268
+ digit = int(character)
269
+ if index % 2 == 1:
270
+ digit *= 2
271
+ if digit > 9:
272
+ digit -= 9
273
+ total += digit
274
+ return total % 10 == 0
275
+
276
+
277
+ def is_secret_field(key: str) -> bool:
278
+ """True when a field name marks its value as a credential."""
279
+ if key in NEVER_REDACTED:
280
+ return False
281
+ segments = list(_ordered_segments(key))
282
+ if set(segments) & SECRET_SEGMENTS:
283
+ return True
284
+ joined = "".join(segments)
285
+ return any(phrase in joined for phrase in SECRET_PHRASES)
286
+
287
+
288
+ def _ordered_segments(key: str) -> Iterable[str]:
289
+ parts = re.split(r"[^A-Za-z0-9]+", key)
290
+ for part in parts:
291
+ for piece in re.findall(r"[A-Z]+(?![a-z])|[A-Z][a-z0-9]*|[a-z0-9]+", part):
292
+ yield piece.lower()
293
+
294
+
295
+ #: The patterns that make an identifier suspicious. Used to warn when
296
+ #: pseudonymization is off and an ID looks like it carries customer data.
297
+ _IDENTIFIER_RISKS: tuple[tuple[str, re.Pattern[str]], ...] = (
298
+ ("an email address", EMAIL_PATTERN),
299
+ ("a phone number", PHONE_PATTERN),
300
+ ("a phone number", E164_PATTERN),
301
+ )
302
+
303
+
304
+ def risky_identifiers(trace: NormalizedTrace) -> list[str]:
305
+ """Identifiers in this trace that appear to contain personal data.
306
+
307
+ Reported rather than rewritten. Rewriting them without pseudonymization
308
+ would break the links; staying silent would let the documented guarantee
309
+ quietly overstate what happened.
310
+ """
311
+ found: list[str] = []
312
+ candidates: list[tuple[str, str]] = [("trace_id", trace.trace_id)]
313
+ for event in trace.events:
314
+ candidates.append(("event_id", event.event_id))
315
+ call_id = getattr(event, "call_id", None)
316
+ if isinstance(call_id, str):
317
+ candidates.append(("call_id", call_id))
318
+
319
+ for name, value in candidates:
320
+ # Identifiers are usually `prefix-value`, and the patterns treat a
321
+ # hyphen as part of a word so that order numbers are not mistaken for
322
+ # phone numbers. Scanning a separator-normalized copy as well catches
323
+ # `customer-(415) 555-2671`, which is exactly the shape that matters
324
+ # here and would otherwise slip through.
325
+ forms = {value, re.sub(r"[-_]+", " ", value)}
326
+ for description, pattern in _IDENTIFIER_RISKS:
327
+ if any(pattern.search(form) for form in forms):
328
+ found.append(f"{name} contains what looks like {description}")
329
+ break
330
+ else:
331
+ if any(token.search(form) for form in forms for token in TOKEN_PATTERNS):
332
+ found.append(f"{name} contains what looks like a credential")
333
+ return found