evalkeep 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalkeep/__init__.py +12 -0
- evalkeep/__main__.py +6 -0
- evalkeep/adapters/__init__.py +45 -0
- evalkeep/adapters/base.py +92 -0
- evalkeep/adapters/jsonl.py +164 -0
- evalkeep/adapters/langsmith.py +436 -0
- evalkeep/adapters/otlp.py +442 -0
- evalkeep/adapters/semconv.py +208 -0
- evalkeep/analysis.py +174 -0
- evalkeep/analysis_run.py +160 -0
- evalkeep/analyzers/__init__.py +52 -0
- evalkeep/analyzers/anthropic.py +145 -0
- evalkeep/analyzers/stub.py +34 -0
- evalkeep/cache.py +122 -0
- evalkeep/cli.py +1933 -0
- evalkeep/clustering.py +383 -0
- evalkeep/clusters.py +101 -0
- evalkeep/commands/__init__.py +1 -0
- evalkeep/commands/analyze_cmd.py +100 -0
- evalkeep/commands/compare_cmd.py +169 -0
- evalkeep/commands/dataset_cmd.py +182 -0
- evalkeep/commands/detect_cmd.py +154 -0
- evalkeep/commands/discover_cmd.py +274 -0
- evalkeep/commands/ingest_cmd.py +50 -0
- evalkeep/commands/init_cmd.py +151 -0
- evalkeep/commands/pipeline_cmd.py +156 -0
- evalkeep/commands/review_cmd.py +141 -0
- evalkeep/commands/run_cmd.py +131 -0
- evalkeep/commands/target_cmd.py +109 -0
- evalkeep/commands/trace_cmd.py +58 -0
- evalkeep/comparison.py +432 -0
- evalkeep/config.py +209 -0
- evalkeep/detection.py +94 -0
- evalkeep/detectors.py +182 -0
- evalkeep/discovery.py +208 -0
- evalkeep/embeddings/__init__.py +31 -0
- evalkeep/embeddings/base.py +32 -0
- evalkeep/embeddings/hashing.py +98 -0
- evalkeep/errors.py +42 -0
- evalkeep/examples/__init__.py +37 -0
- evalkeep/examples/langsmith/runs.jsonl +18 -0
- evalkeep/examples/opentelemetry/spans.json +898 -0
- evalkeep/examples/refund-agent/agents/baseline.py +66 -0
- evalkeep/examples/refund-agent/agents/candidate.py +66 -0
- evalkeep/examples/refund-agent/traces.jsonl +5 -0
- evalkeep/examples/tau-bench/prepare.py +230 -0
- evalkeep/exporters/__init__.py +45 -0
- evalkeep/exporters/generic.py +31 -0
- evalkeep/exporters/promptfoo.py +219 -0
- evalkeep/failures.py +95 -0
- evalkeep/generation.py +303 -0
- evalkeep/hashing.py +56 -0
- evalkeep/ingest.py +257 -0
- evalkeep/prompts.py +127 -0
- evalkeep/pseudonyms.py +82 -0
- evalkeep/py.typed +0 -0
- evalkeep/redaction.py +333 -0
- evalkeep/regression.py +409 -0
- evalkeep/review.py +309 -0
- evalkeep/runner.py +302 -0
- evalkeep/runs.py +185 -0
- evalkeep/storage/__init__.py +37 -0
- evalkeep/storage/clusters.py +163 -0
- evalkeep/storage/failures.py +254 -0
- evalkeep/storage/migrations.py +370 -0
- evalkeep/storage/regression.py +136 -0
- evalkeep/storage/runs.py +223 -0
- evalkeep/storage/store.py +429 -0
- evalkeep/targets.py +205 -0
- evalkeep/trace.py +238 -0
- evalkeep-0.1.0.dist-info/METADATA +221 -0
- evalkeep-0.1.0.dist-info/RECORD +75 -0
- evalkeep-0.1.0.dist-info/WHEEL +4 -0
- evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
- evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/redaction.py
ADDED
|
@@ -0,0 +1,333 @@
|
|
|
1
|
+
"""Deterministic redaction, applied in memory before anything reaches storage.
|
|
2
|
+
|
|
3
|
+
Redaction runs on the parsed trace and returns a new one; the raw trace is never
|
|
4
|
+
written to the database, exported, or sent to an analyzer. The rules err toward
|
|
5
|
+
over-redaction -- a redacted order total is an inconvenience, a stored API key is
|
|
6
|
+
an incident -- with two exceptions that keep the data usable:
|
|
7
|
+
|
|
8
|
+
* Identity fields (``trace_id``, ``event_id``, ``call_id``, ``tool`` and the
|
|
9
|
+
other structural keys in :data:`NEVER_REDACTED`) are never rewritten. Scrubbing
|
|
10
|
+
an identifier would break the very links the pipeline is built on.
|
|
11
|
+
* Secret *field names* only redact string and container values. ``token_count:
|
|
12
|
+
512`` is a number, not a credential, and redacting it would lose real signal.
|
|
13
|
+
|
|
14
|
+
Everything is deterministic: the same input always produces the same output, with
|
|
15
|
+
fixed placeholders rather than hashes of the original. Two customers' email
|
|
16
|
+
addresses collapse to the same placeholder, which is what clustering wants.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import re
|
|
22
|
+
from collections.abc import Iterable
|
|
23
|
+
from dataclasses import dataclass, field
|
|
24
|
+
from enum import StrEnum
|
|
25
|
+
from typing import Any
|
|
26
|
+
|
|
27
|
+
from evalkeep.config import RedactionConfig
|
|
28
|
+
from evalkeep.pseudonyms import PREFIXES, Pseudonymizer
|
|
29
|
+
from evalkeep.trace import NormalizedTrace
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class RedactionRule(StrEnum):
|
|
33
|
+
EMAIL = "email"
|
|
34
|
+
PHONE = "phone"
|
|
35
|
+
PAYMENT_CARD = "payment_card"
|
|
36
|
+
TOKEN = "token"
|
|
37
|
+
SECRET_FIELD = "secret_field"
|
|
38
|
+
#: An identifier replaced by a per-project token rather than a placeholder,
|
|
39
|
+
#: so the links it carries survive.
|
|
40
|
+
PSEUDONYM = "pseudonym"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def placeholder(rule: RedactionRule) -> str:
|
|
44
|
+
return f"[REDACTED:{rule.value}]"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
#: Structural and identity keys whose values are never rewritten.
|
|
48
|
+
NEVER_REDACTED: frozenset[str] = frozenset(
|
|
49
|
+
{
|
|
50
|
+
"trace_id",
|
|
51
|
+
"event_id",
|
|
52
|
+
"call_id",
|
|
53
|
+
"schema_version",
|
|
54
|
+
"type",
|
|
55
|
+
"role",
|
|
56
|
+
"status",
|
|
57
|
+
"tool",
|
|
58
|
+
"rating",
|
|
59
|
+
"passed",
|
|
60
|
+
"score",
|
|
61
|
+
"name",
|
|
62
|
+
}
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
#: A key segment equal to one of these marks a credential.
|
|
66
|
+
SECRET_SEGMENTS: frozenset[str] = frozenset(
|
|
67
|
+
{
|
|
68
|
+
"password",
|
|
69
|
+
"passwd",
|
|
70
|
+
"secret",
|
|
71
|
+
"token",
|
|
72
|
+
"credential",
|
|
73
|
+
"credentials",
|
|
74
|
+
"apikey",
|
|
75
|
+
"cvv",
|
|
76
|
+
"cvc",
|
|
77
|
+
"ssn",
|
|
78
|
+
"authorization",
|
|
79
|
+
"pin",
|
|
80
|
+
}
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
#: A normalized key containing one of these marks a credential.
|
|
84
|
+
SECRET_PHRASES: tuple[str, ...] = (
|
|
85
|
+
"apikey",
|
|
86
|
+
"accesstoken",
|
|
87
|
+
"refreshtoken",
|
|
88
|
+
"idtoken",
|
|
89
|
+
"authtoken",
|
|
90
|
+
"bearertoken",
|
|
91
|
+
"privatekey",
|
|
92
|
+
"secretkey",
|
|
93
|
+
"clientsecret",
|
|
94
|
+
"sessionkey",
|
|
95
|
+
"creditcard",
|
|
96
|
+
"cardnumber",
|
|
97
|
+
"socialsecurity",
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
EMAIL_PATTERN = re.compile(r"[A-Za-z0-9._%+\-]+@[A-Za-z0-9.\-]+\.[A-Za-z]{2,}")
|
|
101
|
+
|
|
102
|
+
#: Requires separators, so order numbers and totals are not mistaken for phones.
|
|
103
|
+
PHONE_PATTERN = re.compile(
|
|
104
|
+
r"""(?<![\w-])
|
|
105
|
+
(?:\+\d{1,3}[\s.\-]?)?
|
|
106
|
+
(?:\(\d{3}\)\s?|\d{3}[\s.\-])
|
|
107
|
+
\d{3}[\s.\-]\d{4}
|
|
108
|
+
(?![\w-])""",
|
|
109
|
+
re.VERBOSE,
|
|
110
|
+
)
|
|
111
|
+
E164_PATTERN = re.compile(r"(?<![\w-])\+\d{10,15}(?![\w-])")
|
|
112
|
+
|
|
113
|
+
#: A candidate only; :func:`_luhn_ok` decides whether it is really a card. The
|
|
114
|
+
#: boundaries exclude only adjacent digits, so a card embedded in an identifier
|
|
115
|
+
#: is still caught -- over-redacting an ID beats storing a card number.
|
|
116
|
+
CARD_CANDIDATE_PATTERN = re.compile(r"(?<!\d)(?:\d[ \-]?){12,18}\d(?!\d)")
|
|
117
|
+
|
|
118
|
+
TOKEN_PATTERNS: tuple[re.Pattern[str], ...] = (
|
|
119
|
+
re.compile(r"\bsk-[A-Za-z0-9_\-]{16,}"),
|
|
120
|
+
re.compile(r"\bpk-[A-Za-z0-9_\-]{16,}"),
|
|
121
|
+
re.compile(r"\bgh[pousr]_[A-Za-z0-9]{20,}"),
|
|
122
|
+
re.compile(r"\bxox[baprs]-[A-Za-z0-9\-]{10,}"),
|
|
123
|
+
re.compile(r"\bAKIA[0-9A-Z]{16}\b"),
|
|
124
|
+
re.compile(r"\bASIA[0-9A-Z]{16}\b"),
|
|
125
|
+
re.compile(r"\bAIza[0-9A-Za-z_\-]{35}\b"),
|
|
126
|
+
re.compile(r"\beyJ[A-Za-z0-9_\-]{8,}\.[A-Za-z0-9_\-]{8,}\.[A-Za-z0-9_\-]{8,}"),
|
|
127
|
+
re.compile(r"(?i)\bbearer\s+[A-Za-z0-9._\-]{20,}"),
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
@dataclass
|
|
132
|
+
class RedactionSummary:
|
|
133
|
+
"""How many values each rule replaced. Stored alongside the trace."""
|
|
134
|
+
|
|
135
|
+
counts: dict[RedactionRule, int] = field(default_factory=dict)
|
|
136
|
+
|
|
137
|
+
def record(self, rule: RedactionRule, times: int = 1) -> None:
|
|
138
|
+
if times:
|
|
139
|
+
self.counts[rule] = self.counts.get(rule, 0) + times
|
|
140
|
+
|
|
141
|
+
@property
|
|
142
|
+
def total(self) -> int:
|
|
143
|
+
return sum(self.counts.values())
|
|
144
|
+
|
|
145
|
+
def to_dict(self) -> dict[str, int]:
|
|
146
|
+
return {rule.value: count for rule, count in sorted(self.counts.items())}
|
|
147
|
+
|
|
148
|
+
def merge(self, other: RedactionSummary) -> None:
|
|
149
|
+
for rule, count in other.counts.items():
|
|
150
|
+
self.record(rule, count)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
class Redactor:
|
|
154
|
+
"""Applies the configured rules to a trace, in memory, before storage."""
|
|
155
|
+
|
|
156
|
+
def __init__(
|
|
157
|
+
self,
|
|
158
|
+
config: RedactionConfig | None = None,
|
|
159
|
+
*,
|
|
160
|
+
pseudonymizer: Pseudonymizer | None = None,
|
|
161
|
+
) -> None:
|
|
162
|
+
self.config = config or RedactionConfig()
|
|
163
|
+
self._pseudonymizer = pseudonymizer
|
|
164
|
+
|
|
165
|
+
@property
|
|
166
|
+
def pseudonymizing(self) -> bool:
|
|
167
|
+
return self._pseudonymizer is not None
|
|
168
|
+
|
|
169
|
+
def redact(self, trace: NormalizedTrace) -> tuple[NormalizedTrace, RedactionSummary]:
|
|
170
|
+
"""Return a redacted copy of ``trace`` and what was replaced."""
|
|
171
|
+
summary = RedactionSummary()
|
|
172
|
+
payload = trace.model_dump(mode="json")
|
|
173
|
+
cleaned = self._walk(payload, summary, key=None)
|
|
174
|
+
# Re-validating proves redaction produced a trace the rest of the
|
|
175
|
+
# pipeline can still read, rather than something merely dict-shaped.
|
|
176
|
+
return NormalizedTrace.model_validate(cleaned), summary
|
|
177
|
+
|
|
178
|
+
def redact_text(self, text: str, summary: RedactionSummary) -> str:
|
|
179
|
+
"""Apply every enabled pattern rule to one string."""
|
|
180
|
+
if self.config.token_prefixes:
|
|
181
|
+
for pattern in TOKEN_PATTERNS:
|
|
182
|
+
text, hits = pattern.subn(placeholder(RedactionRule.TOKEN), text)
|
|
183
|
+
summary.record(RedactionRule.TOKEN, hits)
|
|
184
|
+
# Phones before cards, deliberately. A phone number sitting next to a
|
|
185
|
+
# card forms one digit run, and Luhn alone cannot say which digits
|
|
186
|
+
# belong to which; replacing the phone first removes the ambiguity
|
|
187
|
+
# rather than guessing. Phone patterns require separators in positions
|
|
188
|
+
# a card's groups never produce, so they cannot match inside a card.
|
|
189
|
+
if self.config.phone_numbers:
|
|
190
|
+
for pattern in (PHONE_PATTERN, E164_PATTERN):
|
|
191
|
+
text, hits = pattern.subn(placeholder(RedactionRule.PHONE), text)
|
|
192
|
+
summary.record(RedactionRule.PHONE, hits)
|
|
193
|
+
if self.config.payment_cards:
|
|
194
|
+
text = self._redact_cards(text, summary)
|
|
195
|
+
if self.config.emails:
|
|
196
|
+
text, hits = EMAIL_PATTERN.subn(placeholder(RedactionRule.EMAIL), text)
|
|
197
|
+
summary.record(RedactionRule.EMAIL, hits)
|
|
198
|
+
return text
|
|
199
|
+
|
|
200
|
+
def _redact_cards(self, text: str, summary: RedactionSummary) -> str:
|
|
201
|
+
"""Replace only digit runs that pass Luhn, so totals and IDs survive.
|
|
202
|
+
|
|
203
|
+
Candidates are matched as whole separator-delimited groups, and a
|
|
204
|
+
candidate that fails Luhn is retried over its sub-runs of groups. Both
|
|
205
|
+
matter: a phone number sitting next to a card forms one long digit run
|
|
206
|
+
that fails Luhn as a whole, and searching groups rather than arbitrary
|
|
207
|
+
offsets finds the card inside it without inventing Luhn-valid windows
|
|
208
|
+
that were never a card.
|
|
209
|
+
"""
|
|
210
|
+
|
|
211
|
+
def replace(match: re.Match[str]) -> str:
|
|
212
|
+
span = match.group(0)
|
|
213
|
+
groups = [(m.start(), m.end()) for m in re.finditer(r"\d+", span)]
|
|
214
|
+
for start in range(len(groups)):
|
|
215
|
+
for end in range(len(groups), start, -1):
|
|
216
|
+
digits = "".join(span[a:b] for a, b in groups[start:end])
|
|
217
|
+
if not 13 <= len(digits) <= 19 or not _luhn_ok(digits):
|
|
218
|
+
continue
|
|
219
|
+
summary.record(RedactionRule.PAYMENT_CARD)
|
|
220
|
+
low, high = groups[start][0], groups[end - 1][1]
|
|
221
|
+
tail = self._redact_cards(span[high:], summary)
|
|
222
|
+
return span[:low] + placeholder(RedactionRule.PAYMENT_CARD) + tail
|
|
223
|
+
return span
|
|
224
|
+
|
|
225
|
+
return CARD_CANDIDATE_PATTERN.sub(replace, text)
|
|
226
|
+
|
|
227
|
+
def _walk(self, value: Any, summary: RedactionSummary, *, key: str | None) -> Any:
|
|
228
|
+
if isinstance(value, dict):
|
|
229
|
+
cleaned: dict[str, Any] = {}
|
|
230
|
+
for child_key, child in value.items():
|
|
231
|
+
if self._is_secret_value(child_key, child):
|
|
232
|
+
summary.record(RedactionRule.SECRET_FIELD)
|
|
233
|
+
cleaned[child_key] = placeholder(RedactionRule.SECRET_FIELD)
|
|
234
|
+
else:
|
|
235
|
+
cleaned[child_key] = self._walk(child, summary, key=child_key)
|
|
236
|
+
return cleaned
|
|
237
|
+
if isinstance(value, list):
|
|
238
|
+
return [self._walk(item, summary, key=key) for item in value]
|
|
239
|
+
if isinstance(value, str):
|
|
240
|
+
if key is not None and key in NEVER_REDACTED:
|
|
241
|
+
return self._identifier(key, value, summary)
|
|
242
|
+
return self.redact_text(value, summary)
|
|
243
|
+
return value
|
|
244
|
+
|
|
245
|
+
def _identifier(self, key: str, value: str, summary: RedactionSummary) -> str:
|
|
246
|
+
"""Identifiers survive verbatim, or become a stable per-project token.
|
|
247
|
+
|
|
248
|
+
Never a placeholder: two traces collapsing to `[REDACTED:email]` would
|
|
249
|
+
collide, and the pipeline is built on these values being distinct.
|
|
250
|
+
"""
|
|
251
|
+
if self._pseudonymizer is None or key not in PREFIXES or not value:
|
|
252
|
+
return value
|
|
253
|
+
summary.record(RedactionRule.PSEUDONYM)
|
|
254
|
+
return self._pseudonymizer.token(value, field=key)
|
|
255
|
+
|
|
256
|
+
def _is_secret_value(self, key: str, value: Any) -> bool:
|
|
257
|
+
"""A credential is a string or a container, never a count or a flag."""
|
|
258
|
+
if not self.config.secret_field_names:
|
|
259
|
+
return False
|
|
260
|
+
if not isinstance(value, str | dict | list):
|
|
261
|
+
return False
|
|
262
|
+
return is_secret_field(key)
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def _luhn_ok(digits: str) -> bool:
|
|
266
|
+
total = 0
|
|
267
|
+
for index, character in enumerate(reversed(digits)):
|
|
268
|
+
digit = int(character)
|
|
269
|
+
if index % 2 == 1:
|
|
270
|
+
digit *= 2
|
|
271
|
+
if digit > 9:
|
|
272
|
+
digit -= 9
|
|
273
|
+
total += digit
|
|
274
|
+
return total % 10 == 0
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def is_secret_field(key: str) -> bool:
|
|
278
|
+
"""True when a field name marks its value as a credential."""
|
|
279
|
+
if key in NEVER_REDACTED:
|
|
280
|
+
return False
|
|
281
|
+
segments = list(_ordered_segments(key))
|
|
282
|
+
if set(segments) & SECRET_SEGMENTS:
|
|
283
|
+
return True
|
|
284
|
+
joined = "".join(segments)
|
|
285
|
+
return any(phrase in joined for phrase in SECRET_PHRASES)
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def _ordered_segments(key: str) -> Iterable[str]:
|
|
289
|
+
parts = re.split(r"[^A-Za-z0-9]+", key)
|
|
290
|
+
for part in parts:
|
|
291
|
+
for piece in re.findall(r"[A-Z]+(?![a-z])|[A-Z][a-z0-9]*|[a-z0-9]+", part):
|
|
292
|
+
yield piece.lower()
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
#: The patterns that make an identifier suspicious. Used to warn when
|
|
296
|
+
#: pseudonymization is off and an ID looks like it carries customer data.
|
|
297
|
+
_IDENTIFIER_RISKS: tuple[tuple[str, re.Pattern[str]], ...] = (
|
|
298
|
+
("an email address", EMAIL_PATTERN),
|
|
299
|
+
("a phone number", PHONE_PATTERN),
|
|
300
|
+
("a phone number", E164_PATTERN),
|
|
301
|
+
)
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def risky_identifiers(trace: NormalizedTrace) -> list[str]:
|
|
305
|
+
"""Identifiers in this trace that appear to contain personal data.
|
|
306
|
+
|
|
307
|
+
Reported rather than rewritten. Rewriting them without pseudonymization
|
|
308
|
+
would break the links; staying silent would let the documented guarantee
|
|
309
|
+
quietly overstate what happened.
|
|
310
|
+
"""
|
|
311
|
+
found: list[str] = []
|
|
312
|
+
candidates: list[tuple[str, str]] = [("trace_id", trace.trace_id)]
|
|
313
|
+
for event in trace.events:
|
|
314
|
+
candidates.append(("event_id", event.event_id))
|
|
315
|
+
call_id = getattr(event, "call_id", None)
|
|
316
|
+
if isinstance(call_id, str):
|
|
317
|
+
candidates.append(("call_id", call_id))
|
|
318
|
+
|
|
319
|
+
for name, value in candidates:
|
|
320
|
+
# Identifiers are usually `prefix-value`, and the patterns treat a
|
|
321
|
+
# hyphen as part of a word so that order numbers are not mistaken for
|
|
322
|
+
# phone numbers. Scanning a separator-normalized copy as well catches
|
|
323
|
+
# `customer-(415) 555-2671`, which is exactly the shape that matters
|
|
324
|
+
# here and would otherwise slip through.
|
|
325
|
+
forms = {value, re.sub(r"[-_]+", " ", value)}
|
|
326
|
+
for description, pattern in _IDENTIFIER_RISKS:
|
|
327
|
+
if any(pattern.search(form) for form in forms):
|
|
328
|
+
found.append(f"{name} contains what looks like {description}")
|
|
329
|
+
break
|
|
330
|
+
else:
|
|
331
|
+
if any(token.search(form) for form in forms for token in TOKEN_PATTERNS):
|
|
332
|
+
found.append(f"{name} contains what looks like a credential")
|
|
333
|
+
return found
|