evalkeep 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. evalkeep/__init__.py +12 -0
  2. evalkeep/__main__.py +6 -0
  3. evalkeep/adapters/__init__.py +45 -0
  4. evalkeep/adapters/base.py +92 -0
  5. evalkeep/adapters/jsonl.py +164 -0
  6. evalkeep/adapters/langsmith.py +436 -0
  7. evalkeep/adapters/otlp.py +442 -0
  8. evalkeep/adapters/semconv.py +208 -0
  9. evalkeep/analysis.py +174 -0
  10. evalkeep/analysis_run.py +160 -0
  11. evalkeep/analyzers/__init__.py +52 -0
  12. evalkeep/analyzers/anthropic.py +145 -0
  13. evalkeep/analyzers/stub.py +34 -0
  14. evalkeep/cache.py +122 -0
  15. evalkeep/cli.py +1933 -0
  16. evalkeep/clustering.py +383 -0
  17. evalkeep/clusters.py +101 -0
  18. evalkeep/commands/__init__.py +1 -0
  19. evalkeep/commands/analyze_cmd.py +100 -0
  20. evalkeep/commands/compare_cmd.py +169 -0
  21. evalkeep/commands/dataset_cmd.py +182 -0
  22. evalkeep/commands/detect_cmd.py +154 -0
  23. evalkeep/commands/discover_cmd.py +274 -0
  24. evalkeep/commands/ingest_cmd.py +50 -0
  25. evalkeep/commands/init_cmd.py +151 -0
  26. evalkeep/commands/pipeline_cmd.py +156 -0
  27. evalkeep/commands/review_cmd.py +141 -0
  28. evalkeep/commands/run_cmd.py +131 -0
  29. evalkeep/commands/target_cmd.py +109 -0
  30. evalkeep/commands/trace_cmd.py +58 -0
  31. evalkeep/comparison.py +432 -0
  32. evalkeep/config.py +209 -0
  33. evalkeep/detection.py +94 -0
  34. evalkeep/detectors.py +182 -0
  35. evalkeep/discovery.py +208 -0
  36. evalkeep/embeddings/__init__.py +31 -0
  37. evalkeep/embeddings/base.py +32 -0
  38. evalkeep/embeddings/hashing.py +98 -0
  39. evalkeep/errors.py +42 -0
  40. evalkeep/examples/__init__.py +37 -0
  41. evalkeep/examples/langsmith/runs.jsonl +18 -0
  42. evalkeep/examples/opentelemetry/spans.json +898 -0
  43. evalkeep/examples/refund-agent/agents/baseline.py +66 -0
  44. evalkeep/examples/refund-agent/agents/candidate.py +66 -0
  45. evalkeep/examples/refund-agent/traces.jsonl +5 -0
  46. evalkeep/examples/tau-bench/prepare.py +230 -0
  47. evalkeep/exporters/__init__.py +45 -0
  48. evalkeep/exporters/generic.py +31 -0
  49. evalkeep/exporters/promptfoo.py +219 -0
  50. evalkeep/failures.py +95 -0
  51. evalkeep/generation.py +303 -0
  52. evalkeep/hashing.py +56 -0
  53. evalkeep/ingest.py +257 -0
  54. evalkeep/prompts.py +127 -0
  55. evalkeep/pseudonyms.py +82 -0
  56. evalkeep/py.typed +0 -0
  57. evalkeep/redaction.py +333 -0
  58. evalkeep/regression.py +409 -0
  59. evalkeep/review.py +309 -0
  60. evalkeep/runner.py +302 -0
  61. evalkeep/runs.py +185 -0
  62. evalkeep/storage/__init__.py +37 -0
  63. evalkeep/storage/clusters.py +163 -0
  64. evalkeep/storage/failures.py +254 -0
  65. evalkeep/storage/migrations.py +370 -0
  66. evalkeep/storage/regression.py +136 -0
  67. evalkeep/storage/runs.py +223 -0
  68. evalkeep/storage/store.py +429 -0
  69. evalkeep/targets.py +205 -0
  70. evalkeep/trace.py +238 -0
  71. evalkeep-0.1.0.dist-info/METADATA +221 -0
  72. evalkeep-0.1.0.dist-info/RECORD +75 -0
  73. evalkeep-0.1.0.dist-info/WHEEL +4 -0
  74. evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
  75. evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/ingest.py ADDED
@@ -0,0 +1,257 @@
1
+ """The ingest pipeline: validate, redact, hash, store -- in that order.
2
+
3
+ Redaction happens between validation and storage, in memory, so a raw value
4
+ never reaches the database. The pipeline streams: issues are written to the
5
+ error JSONL as they are found and only a bounded sample is kept for display, so
6
+ validating 100k traces costs the set of trace IDs seen so far, not the file.
7
+
8
+ Three modes share one implementation:
9
+
10
+ * **validate** (no store) -- parse and check only. Needs no project.
11
+ * **dry run** -- redact and ask the store what *would* happen, writing nothing.
12
+ * **ingest** -- the same, then commit.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import json
18
+ from collections.abc import Iterator
19
+ from contextlib import ExitStack
20
+ from dataclasses import dataclass, field
21
+ from enum import StrEnum
22
+ from pathlib import Path
23
+ from typing import TextIO
24
+
25
+ from evalkeep.adapters import AdapterRecord, IssueKind, TraceAdapter, TraceIssue
26
+ from evalkeep.errors import CommandError, ExitCode
27
+ from evalkeep.redaction import RedactionSummary, Redactor, risky_identifiers
28
+ from evalkeep.storage import StoreOutcome, StoreResult, TraceStore
29
+
30
+ #: Issues kept for terminal display; the rest go to the error JSONL.
31
+ DEFAULT_SAMPLE_LIMIT = 20
32
+
33
+
34
+ class IngestMode(StrEnum):
35
+ VALIDATE = "validate"
36
+ DRY_RUN = "dry-run"
37
+ STORE = "ingest"
38
+
39
+
40
+ @dataclass
41
+ class IngestReport:
42
+ """What a pass over a trace file found. Counts are exact; ``sample`` is bounded."""
43
+
44
+ path: Path
45
+ adapter: str
46
+ mode: IngestMode = IngestMode.VALIDATE
47
+
48
+ # Validation
49
+ records: int = 0
50
+ valid: int = 0
51
+ invalid: int = 0
52
+ issue_count: int = 0
53
+ duplicate_ids: int = 0
54
+
55
+ # Storage
56
+ stored: int = 0
57
+ already_stored: int = 0
58
+ content_duplicates: int = 0
59
+ id_conflicts: int = 0
60
+ #: New sightings recorded. An interaction seen again is not stored twice,
61
+ #: but the fact that it happened again is kept.
62
+ occurrences: int = 0
63
+
64
+ # Redaction
65
+ redactions: int = 0
66
+ redacted_traces: int = 0
67
+ #: Traces whose identifiers look like they carry personal data, counted
68
+ #: only when pseudonymization is off and they were therefore stored as-is.
69
+ identifier_risks: int = 0
70
+ redaction_summary: RedactionSummary = field(default_factory=RedactionSummary)
71
+
72
+ error_path: Path | None = None
73
+ sample: list[TraceIssue] = field(default_factory=list)
74
+ #: Things worth telling the user that are not record errors.
75
+ notices: list[str] = field(default_factory=list)
76
+
77
+ @property
78
+ def ok(self) -> bool:
79
+ """Records that could not be handled at all -- duplicates are not failures."""
80
+ return self.invalid == 0 and self.id_conflicts == 0
81
+
82
+ @property
83
+ def truncated(self) -> int:
84
+ return max(0, self.issue_count - len(self.sample))
85
+
86
+ @property
87
+ def skipped(self) -> int:
88
+ """Valid traces the store already knew about."""
89
+ return self.already_stored + self.content_duplicates
90
+
91
+ @property
92
+ def exit_code(self) -> ExitCode:
93
+ return ExitCode.OK if self.ok else ExitCode.RECORD_ERRORS
94
+
95
+
96
+ def iter_records(path: Path, adapter: TraceAdapter) -> Iterator[AdapterRecord]:
97
+ """Adapter records plus the duplicate-ID check the adapter cannot make.
98
+
99
+ An adapter sees one record at a time and so cannot know that a trace ID has
100
+ already appeared in this file. The second occurrence is rejected rather than
101
+ merged: accepting it would decide, in the wrong place, which copy wins.
102
+ """
103
+ seen: set[str] = set()
104
+ for record in adapter.read(path):
105
+ if record.trace is None:
106
+ yield record
107
+ continue
108
+ trace_id = record.trace.trace_id
109
+ if trace_id in seen:
110
+ yield AdapterRecord.rejected(
111
+ record.line,
112
+ TraceIssue(
113
+ line=record.line,
114
+ kind=IssueKind.DUPLICATE_ID,
115
+ message=f"trace_id {trace_id!r} already appeared earlier in this file",
116
+ trace_id=trace_id,
117
+ hint="Trace IDs must be unique; the first occurrence is kept.",
118
+ ),
119
+ )
120
+ continue
121
+ seen.add(trace_id)
122
+ yield record
123
+
124
+
125
+ def ingest_file(
126
+ path: Path,
127
+ adapter: TraceAdapter,
128
+ *,
129
+ store: TraceStore | None = None,
130
+ redactor: Redactor | None = None,
131
+ dry_run: bool = False,
132
+ error_path: Path | None = None,
133
+ sample_limit: int = DEFAULT_SAMPLE_LIMIT,
134
+ ) -> IngestReport:
135
+ """Run the pipeline over ``path``. Without ``store``, validation only."""
136
+ _check_readable_file(path)
137
+ mode = (
138
+ IngestMode.VALIDATE
139
+ if store is None
140
+ else (IngestMode.DRY_RUN if dry_run else IngestMode.STORE)
141
+ )
142
+ report = IngestReport(path=path, adapter=adapter.name, mode=mode, error_path=error_path)
143
+ redactor = redactor or Redactor()
144
+
145
+ with ExitStack() as stack:
146
+ errors: TextIO | None = None
147
+ if error_path is not None:
148
+ errors = stack.enter_context(_open_error_file(error_path))
149
+
150
+ def report_issue(issue: TraceIssue) -> None:
151
+ report.issue_count += 1
152
+ if issue.kind is IssueKind.DUPLICATE_ID:
153
+ report.duplicate_ids += 1
154
+ if len(report.sample) < sample_limit:
155
+ report.sample.append(issue)
156
+ if errors is not None:
157
+ errors.write(json.dumps(issue.to_dict(), sort_keys=True) + "\n")
158
+
159
+ for record in iter_records(path, adapter):
160
+ report.records += 1
161
+ if record.trace is None:
162
+ report.invalid += 1
163
+ for issue in record.issues:
164
+ report_issue(issue)
165
+ continue
166
+
167
+ report.valid += 1
168
+ if store is None:
169
+ continue
170
+
171
+ # Redaction sits here on purpose: between a trace being valid and it
172
+ # touching the database, with no path around it.
173
+ if not redactor.pseudonymizing:
174
+ risks = risky_identifiers(record.trace)
175
+ if risks:
176
+ report.identifier_risks += 1
177
+ for risk in risks:
178
+ if risk not in report.notices and len(report.notices) < 5:
179
+ report.notices.append(risk)
180
+
181
+ redacted, summary = redactor.redact(record.trace)
182
+ report.redactions += summary.total
183
+ report.redacted_traces += 1 if summary.total else 0
184
+ report.redaction_summary.merge(summary)
185
+
186
+ outcome = (
187
+ store.classify(redacted) if dry_run else store.add(redacted, redaction=summary)
188
+ )
189
+ _count_outcome(report, outcome.result)
190
+
191
+ # Every accepted sighting is recorded, whether or not the
192
+ # interaction itself was new. Deduplication belongs to the test
193
+ # suite; the evidence keeps its count.
194
+ canonical = _canonical_id(outcome)
195
+ if (
196
+ not dry_run
197
+ and canonical is not None
198
+ and store.record_occurrence(
199
+ redacted, canonical_trace_id=canonical, digest=outcome.content_hash
200
+ )
201
+ ):
202
+ report.occurrences += 1
203
+ if outcome.result is StoreResult.ID_CONFLICT:
204
+ report_issue(
205
+ TraceIssue(
206
+ line=record.line,
207
+ kind=IssueKind.ID_CONFLICT,
208
+ message=(
209
+ f"trace_id {redacted.trace_id!r} is already stored with "
210
+ "different content"
211
+ ),
212
+ trace_id=redacted.trace_id,
213
+ hint="Give the new trace a different ID, or remove the stored one.",
214
+ )
215
+ )
216
+
217
+ return report
218
+
219
+
220
+ def _canonical_id(outcome: StoreOutcome) -> str | None:
221
+ """Which stored trace this sighting belongs to, or None if it was refused."""
222
+ match outcome.result:
223
+ case StoreResult.STORED | StoreResult.ALREADY_STORED:
224
+ return outcome.trace_id
225
+ case StoreResult.CONTENT_DUPLICATE:
226
+ return outcome.existing_trace_id
227
+ case _:
228
+ return None
229
+
230
+
231
+ def _count_outcome(report: IngestReport, result: StoreResult) -> None:
232
+ match result:
233
+ case StoreResult.STORED:
234
+ report.stored += 1
235
+ case StoreResult.ALREADY_STORED:
236
+ report.already_stored += 1
237
+ case StoreResult.CONTENT_DUPLICATE:
238
+ report.content_duplicates += 1
239
+ case StoreResult.ID_CONFLICT:
240
+ report.id_conflicts += 1
241
+
242
+
243
+ def _open_error_file(error_path: Path) -> TextIO:
244
+ try:
245
+ error_path.parent.mkdir(parents=True, exist_ok=True)
246
+ return error_path.open("w", encoding="utf-8")
247
+ except OSError as exc:
248
+ raise CommandError(f"Could not write the error file {error_path}: {exc}") from exc
249
+
250
+
251
+ def _check_readable_file(path: Path) -> None:
252
+ if not path.exists():
253
+ raise CommandError(f"{path} does not exist.")
254
+ if path.is_dir():
255
+ raise CommandError(f"{path} is a directory, not a trace file.")
256
+ if not path.is_file():
257
+ raise CommandError(f"{path} is not a regular file.")
evalkeep/prompts.py ADDED
@@ -0,0 +1,127 @@
1
+ """Versioned prompts and the strict JSON schema their answers must satisfy.
2
+
3
+ The version is part of every cache key and is stored with every analysis, so a
4
+ prompt change never silently mixes old and new labels in one dataset. Editing
5
+ the prompt text without bumping :data:`FAILURE_ANALYSIS_PROMPT_VERSION` is a
6
+ bug: it would serve stale cached answers for a question you no longer ask.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ from typing import Any
13
+
14
+ from evalkeep.analysis import Component, FailureType, Severity
15
+ from evalkeep.detectors import Signal
16
+ from evalkeep.trace import (
17
+ EvaluationEvent,
18
+ MessageEvent,
19
+ NormalizedTrace,
20
+ ToolCallEvent,
21
+ ToolResultEvent,
22
+ )
23
+
24
+ FAILURE_ANALYSIS_PROMPT_VERSION = 1
25
+
26
+ MAX_SUMMARY_LENGTH = 240
27
+
28
+ FAILURE_ANALYSIS_SCHEMA: dict[str, Any] = {
29
+ "type": "object",
30
+ "properties": {
31
+ "failure_type": {
32
+ "type": "string",
33
+ "enum": [member.value for member in FailureType],
34
+ "description": "The kind of failure, from the fixed list.",
35
+ },
36
+ "component": {
37
+ "type": "string",
38
+ "enum": [member.value for member in Component],
39
+ "description": "Where in the agent the failure originated.",
40
+ },
41
+ "severity": {
42
+ "type": "string",
43
+ "enum": [member.value for member in Severity],
44
+ "description": "How much damage this failure does to a user.",
45
+ },
46
+ "summary": {
47
+ "type": "string",
48
+ "description": (
49
+ "One sentence naming the specific mistake, in terms that would "
50
+ "match other traces with the same underlying problem."
51
+ ),
52
+ },
53
+ },
54
+ "required": ["failure_type", "component", "severity", "summary"],
55
+ "additionalProperties": False,
56
+ }
57
+
58
+ FAILURE_ANALYSIS_SYSTEM = """\
59
+ You classify failures in recorded AI-agent interactions so that similar failures \
60
+ can be grouped together.
61
+
62
+ Rules:
63
+ - Describe only what the recorded interaction shows. Do not speculate about code \
64
+ you cannot see, and do not propose fixes.
65
+ - Write the summary so that two traces with the same underlying problem would \
66
+ receive near-identical summaries. Name the mistake, not the customer, the order \
67
+ or the wording of this particular request.
68
+ - Values shown as [REDACTED:...] were removed before you saw them. Treat them as \
69
+ opaque; never guess what they were.
70
+ - Choose the most specific failure_type that fits. Use "other" only when nothing \
71
+ else applies.
72
+ - Severity is about user impact: critical means money, data or safety; low means \
73
+ cosmetic or easily noticed.\
74
+ """
75
+
76
+
77
+ def failure_analysis_prompt(trace: NormalizedTrace, signals: list[Signal]) -> str:
78
+ """The user-turn prompt describing one failing interaction."""
79
+ sections = [
80
+ "Here is a recorded interaction that has been marked as a failure.",
81
+ "",
82
+ "## Evidence that it failed",
83
+ ]
84
+ sections.extend(f"- ({signal.kind.value}) {signal.summary}" for signal in signals)
85
+ sections += ["", "## The interaction", _render_trace(trace)]
86
+ sections += [
87
+ "",
88
+ "Classify this failure. Respond with a JSON object matching the required schema.",
89
+ ]
90
+ return "\n".join(sections)
91
+
92
+
93
+ def _render_trace(trace: NormalizedTrace) -> str:
94
+ """A compact, stable rendering. Stable matters: it feeds the cache key."""
95
+ lines: list[str] = []
96
+ if trace.input.text:
97
+ lines.append(f"user: {trace.input.text}")
98
+ for message in trace.input.messages:
99
+ lines.append(f"{message.role.value}: {message.content}")
100
+
101
+ for event in trace.events:
102
+ if isinstance(event, ToolCallEvent):
103
+ arguments = json.dumps(event.arguments, sort_keys=True)
104
+ lines.append(f"tool_call: {event.tool}({arguments})")
105
+ elif isinstance(event, ToolResultEvent):
106
+ result = json.dumps(event.result, sort_keys=True, default=str)
107
+ suffix = f" error={event.error}" if event.error else ""
108
+ lines.append(f"tool_result: {event.tool} -> {result}{suffix}")
109
+ elif isinstance(event, MessageEvent):
110
+ lines.append(f"{event.role.value}: {event.content}")
111
+ elif isinstance(event, EvaluationEvent):
112
+ verdict = {True: "pass", False: "fail", None: "unrecorded"}[event.passed]
113
+ lines.append(f"evaluation: {event.name} {verdict}")
114
+
115
+ if trace.output is not None and trace.output.text:
116
+ lines.append(f"assistant: {trace.output.text}")
117
+ for message in trace.output.messages if trace.output else []:
118
+ lines.append(f"{message.role.value}: {message.content}")
119
+
120
+ if trace.outcome.feedback is not None and trace.outcome.feedback.comment:
121
+ lines.append(f"feedback: {trace.outcome.feedback.comment}")
122
+ for evaluation in trace.outcome.evaluations:
123
+ if evaluation.passed is False:
124
+ reason = f": {evaluation.reason}" if evaluation.reason else ""
125
+ lines.append(f"failed evaluation: {evaluation.name}{reason}")
126
+
127
+ return "\n".join(lines)
evalkeep/pseudonyms.py ADDED
@@ -0,0 +1,82 @@
1
+ """Deterministic pseudonyms for identifiers that may carry customer data.
2
+
3
+ Redaction deliberately leaves identifiers alone, because rewriting them would
4
+ break the links the pipeline runs on. That is fine when a `trace_id` is a UUID
5
+ and dangerous when it is `order-jane@example.com-2026-06-01`.
6
+
7
+ Pseudonymization resolves the tension instead of trading one problem for the
8
+ other. Each identifier becomes a token derived from a per-project secret salt:
9
+
10
+ * **Deterministic**, so the same original always yields the same token and every
11
+ link in the pipeline survives.
12
+ * **Not reversible from the database**, because the original is never stored --
13
+ only the token is.
14
+ * **Still usable by hand**, because a lookup can hash whatever the user typed
15
+ and search for that. You keep using the IDs your own systems know.
16
+ * **Scoped to one project**, because the salt is per-project and never
17
+ committed. Two projects produce different tokens for the same original, so a
18
+ shared export leaks nothing about another project's data.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import hashlib
24
+ import secrets
25
+ from pathlib import Path
26
+
27
+ from evalkeep.errors import CommandError
28
+
29
+ SALT_FILENAME = "salt"
30
+ SALT_BYTES = 32
31
+ TOKEN_LENGTH = 12
32
+
33
+ #: Which identifier gets which readable prefix, so a pseudonym still looks like
34
+ #: the kind of thing it replaced.
35
+ PREFIXES: dict[str, str] = {
36
+ "trace_id": "trace",
37
+ "event_id": "event",
38
+ "call_id": "call",
39
+ }
40
+
41
+
42
+ class Pseudonymizer:
43
+ """Turns an identifier into a stable token, given a project's salt."""
44
+
45
+ def __init__(self, salt: bytes) -> None:
46
+ if len(salt) < 16:
47
+ raise ValueError("the salt must be at least 16 bytes")
48
+ self._salt = salt
49
+
50
+ def token(self, value: str, *, field: str) -> str:
51
+ """A stable pseudonym for ``value``, prefixed by the kind of ID it is."""
52
+ digest = hashlib.blake2b(
53
+ value.encode("utf-8"), digest_size=TOKEN_LENGTH // 2, key=self._salt
54
+ ).hexdigest()
55
+ return f"{PREFIXES.get(field, 'id')}-{digest}"
56
+
57
+ @classmethod
58
+ def load(cls, path: Path) -> Pseudonymizer:
59
+ """Read a project's salt, creating one on first use."""
60
+ if not path.is_file():
61
+ return cls(_create_salt(path))
62
+ try:
63
+ salt = bytes.fromhex(path.read_text(encoding="utf-8").strip())
64
+ except (OSError, ValueError) as exc:
65
+ raise CommandError(
66
+ f"Could not read the pseudonymization salt at {path}: {exc}.",
67
+ hint="If it was lost, previously stored identifiers cannot be "
68
+ "matched again; re-ingest into a fresh project.",
69
+ ) from exc
70
+ return cls(salt)
71
+
72
+
73
+ def _create_salt(path: Path) -> bytes:
74
+ salt = secrets.token_bytes(SALT_BYTES)
75
+ try:
76
+ path.parent.mkdir(parents=True, exist_ok=True)
77
+ path.write_text(salt.hex(), encoding="utf-8")
78
+ # The salt is what makes the tokens unguessable; treat it like a key.
79
+ path.chmod(0o600)
80
+ except OSError as exc:
81
+ raise CommandError(f"Could not write the salt to {path}: {exc}") from exc
82
+ return salt
evalkeep/py.typed ADDED
File without changes