evalkeep 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalkeep/__init__.py +12 -0
- evalkeep/__main__.py +6 -0
- evalkeep/adapters/__init__.py +45 -0
- evalkeep/adapters/base.py +92 -0
- evalkeep/adapters/jsonl.py +164 -0
- evalkeep/adapters/langsmith.py +436 -0
- evalkeep/adapters/otlp.py +442 -0
- evalkeep/adapters/semconv.py +208 -0
- evalkeep/analysis.py +174 -0
- evalkeep/analysis_run.py +160 -0
- evalkeep/analyzers/__init__.py +52 -0
- evalkeep/analyzers/anthropic.py +145 -0
- evalkeep/analyzers/stub.py +34 -0
- evalkeep/cache.py +122 -0
- evalkeep/cli.py +1933 -0
- evalkeep/clustering.py +383 -0
- evalkeep/clusters.py +101 -0
- evalkeep/commands/__init__.py +1 -0
- evalkeep/commands/analyze_cmd.py +100 -0
- evalkeep/commands/compare_cmd.py +169 -0
- evalkeep/commands/dataset_cmd.py +182 -0
- evalkeep/commands/detect_cmd.py +154 -0
- evalkeep/commands/discover_cmd.py +274 -0
- evalkeep/commands/ingest_cmd.py +50 -0
- evalkeep/commands/init_cmd.py +151 -0
- evalkeep/commands/pipeline_cmd.py +156 -0
- evalkeep/commands/review_cmd.py +141 -0
- evalkeep/commands/run_cmd.py +131 -0
- evalkeep/commands/target_cmd.py +109 -0
- evalkeep/commands/trace_cmd.py +58 -0
- evalkeep/comparison.py +432 -0
- evalkeep/config.py +209 -0
- evalkeep/detection.py +94 -0
- evalkeep/detectors.py +182 -0
- evalkeep/discovery.py +208 -0
- evalkeep/embeddings/__init__.py +31 -0
- evalkeep/embeddings/base.py +32 -0
- evalkeep/embeddings/hashing.py +98 -0
- evalkeep/errors.py +42 -0
- evalkeep/examples/__init__.py +37 -0
- evalkeep/examples/langsmith/runs.jsonl +18 -0
- evalkeep/examples/opentelemetry/spans.json +898 -0
- evalkeep/examples/refund-agent/agents/baseline.py +66 -0
- evalkeep/examples/refund-agent/agents/candidate.py +66 -0
- evalkeep/examples/refund-agent/traces.jsonl +5 -0
- evalkeep/examples/tau-bench/prepare.py +230 -0
- evalkeep/exporters/__init__.py +45 -0
- evalkeep/exporters/generic.py +31 -0
- evalkeep/exporters/promptfoo.py +219 -0
- evalkeep/failures.py +95 -0
- evalkeep/generation.py +303 -0
- evalkeep/hashing.py +56 -0
- evalkeep/ingest.py +257 -0
- evalkeep/prompts.py +127 -0
- evalkeep/pseudonyms.py +82 -0
- evalkeep/py.typed +0 -0
- evalkeep/redaction.py +333 -0
- evalkeep/regression.py +409 -0
- evalkeep/review.py +309 -0
- evalkeep/runner.py +302 -0
- evalkeep/runs.py +185 -0
- evalkeep/storage/__init__.py +37 -0
- evalkeep/storage/clusters.py +163 -0
- evalkeep/storage/failures.py +254 -0
- evalkeep/storage/migrations.py +370 -0
- evalkeep/storage/regression.py +136 -0
- evalkeep/storage/runs.py +223 -0
- evalkeep/storage/store.py +429 -0
- evalkeep/targets.py +205 -0
- evalkeep/trace.py +238 -0
- evalkeep-0.1.0.dist-info/METADATA +221 -0
- evalkeep-0.1.0.dist-info/RECORD +75 -0
- evalkeep-0.1.0.dist-info/WHEEL +4 -0
- evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
- evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/ingest.py
ADDED
|
@@ -0,0 +1,257 @@
|
|
|
1
|
+
"""The ingest pipeline: validate, redact, hash, store -- in that order.
|
|
2
|
+
|
|
3
|
+
Redaction happens between validation and storage, in memory, so a raw value
|
|
4
|
+
never reaches the database. The pipeline streams: issues are written to the
|
|
5
|
+
error JSONL as they are found and only a bounded sample is kept for display, so
|
|
6
|
+
validating 100k traces costs the set of trace IDs seen so far, not the file.
|
|
7
|
+
|
|
8
|
+
Three modes share one implementation:
|
|
9
|
+
|
|
10
|
+
* **validate** (no store) -- parse and check only. Needs no project.
|
|
11
|
+
* **dry run** -- redact and ask the store what *would* happen, writing nothing.
|
|
12
|
+
* **ingest** -- the same, then commit.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
from collections.abc import Iterator
|
|
19
|
+
from contextlib import ExitStack
|
|
20
|
+
from dataclasses import dataclass, field
|
|
21
|
+
from enum import StrEnum
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import TextIO
|
|
24
|
+
|
|
25
|
+
from evalkeep.adapters import AdapterRecord, IssueKind, TraceAdapter, TraceIssue
|
|
26
|
+
from evalkeep.errors import CommandError, ExitCode
|
|
27
|
+
from evalkeep.redaction import RedactionSummary, Redactor, risky_identifiers
|
|
28
|
+
from evalkeep.storage import StoreOutcome, StoreResult, TraceStore
|
|
29
|
+
|
|
30
|
+
#: Issues kept for terminal display; the rest go to the error JSONL.
|
|
31
|
+
DEFAULT_SAMPLE_LIMIT = 20
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class IngestMode(StrEnum):
|
|
35
|
+
VALIDATE = "validate"
|
|
36
|
+
DRY_RUN = "dry-run"
|
|
37
|
+
STORE = "ingest"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass
|
|
41
|
+
class IngestReport:
|
|
42
|
+
"""What a pass over a trace file found. Counts are exact; ``sample`` is bounded."""
|
|
43
|
+
|
|
44
|
+
path: Path
|
|
45
|
+
adapter: str
|
|
46
|
+
mode: IngestMode = IngestMode.VALIDATE
|
|
47
|
+
|
|
48
|
+
# Validation
|
|
49
|
+
records: int = 0
|
|
50
|
+
valid: int = 0
|
|
51
|
+
invalid: int = 0
|
|
52
|
+
issue_count: int = 0
|
|
53
|
+
duplicate_ids: int = 0
|
|
54
|
+
|
|
55
|
+
# Storage
|
|
56
|
+
stored: int = 0
|
|
57
|
+
already_stored: int = 0
|
|
58
|
+
content_duplicates: int = 0
|
|
59
|
+
id_conflicts: int = 0
|
|
60
|
+
#: New sightings recorded. An interaction seen again is not stored twice,
|
|
61
|
+
#: but the fact that it happened again is kept.
|
|
62
|
+
occurrences: int = 0
|
|
63
|
+
|
|
64
|
+
# Redaction
|
|
65
|
+
redactions: int = 0
|
|
66
|
+
redacted_traces: int = 0
|
|
67
|
+
#: Traces whose identifiers look like they carry personal data, counted
|
|
68
|
+
#: only when pseudonymization is off and they were therefore stored as-is.
|
|
69
|
+
identifier_risks: int = 0
|
|
70
|
+
redaction_summary: RedactionSummary = field(default_factory=RedactionSummary)
|
|
71
|
+
|
|
72
|
+
error_path: Path | None = None
|
|
73
|
+
sample: list[TraceIssue] = field(default_factory=list)
|
|
74
|
+
#: Things worth telling the user that are not record errors.
|
|
75
|
+
notices: list[str] = field(default_factory=list)
|
|
76
|
+
|
|
77
|
+
@property
|
|
78
|
+
def ok(self) -> bool:
|
|
79
|
+
"""Records that could not be handled at all -- duplicates are not failures."""
|
|
80
|
+
return self.invalid == 0 and self.id_conflicts == 0
|
|
81
|
+
|
|
82
|
+
@property
|
|
83
|
+
def truncated(self) -> int:
|
|
84
|
+
return max(0, self.issue_count - len(self.sample))
|
|
85
|
+
|
|
86
|
+
@property
|
|
87
|
+
def skipped(self) -> int:
|
|
88
|
+
"""Valid traces the store already knew about."""
|
|
89
|
+
return self.already_stored + self.content_duplicates
|
|
90
|
+
|
|
91
|
+
@property
|
|
92
|
+
def exit_code(self) -> ExitCode:
|
|
93
|
+
return ExitCode.OK if self.ok else ExitCode.RECORD_ERRORS
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def iter_records(path: Path, adapter: TraceAdapter) -> Iterator[AdapterRecord]:
|
|
97
|
+
"""Adapter records plus the duplicate-ID check the adapter cannot make.
|
|
98
|
+
|
|
99
|
+
An adapter sees one record at a time and so cannot know that a trace ID has
|
|
100
|
+
already appeared in this file. The second occurrence is rejected rather than
|
|
101
|
+
merged: accepting it would decide, in the wrong place, which copy wins.
|
|
102
|
+
"""
|
|
103
|
+
seen: set[str] = set()
|
|
104
|
+
for record in adapter.read(path):
|
|
105
|
+
if record.trace is None:
|
|
106
|
+
yield record
|
|
107
|
+
continue
|
|
108
|
+
trace_id = record.trace.trace_id
|
|
109
|
+
if trace_id in seen:
|
|
110
|
+
yield AdapterRecord.rejected(
|
|
111
|
+
record.line,
|
|
112
|
+
TraceIssue(
|
|
113
|
+
line=record.line,
|
|
114
|
+
kind=IssueKind.DUPLICATE_ID,
|
|
115
|
+
message=f"trace_id {trace_id!r} already appeared earlier in this file",
|
|
116
|
+
trace_id=trace_id,
|
|
117
|
+
hint="Trace IDs must be unique; the first occurrence is kept.",
|
|
118
|
+
),
|
|
119
|
+
)
|
|
120
|
+
continue
|
|
121
|
+
seen.add(trace_id)
|
|
122
|
+
yield record
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def ingest_file(
|
|
126
|
+
path: Path,
|
|
127
|
+
adapter: TraceAdapter,
|
|
128
|
+
*,
|
|
129
|
+
store: TraceStore | None = None,
|
|
130
|
+
redactor: Redactor | None = None,
|
|
131
|
+
dry_run: bool = False,
|
|
132
|
+
error_path: Path | None = None,
|
|
133
|
+
sample_limit: int = DEFAULT_SAMPLE_LIMIT,
|
|
134
|
+
) -> IngestReport:
|
|
135
|
+
"""Run the pipeline over ``path``. Without ``store``, validation only."""
|
|
136
|
+
_check_readable_file(path)
|
|
137
|
+
mode = (
|
|
138
|
+
IngestMode.VALIDATE
|
|
139
|
+
if store is None
|
|
140
|
+
else (IngestMode.DRY_RUN if dry_run else IngestMode.STORE)
|
|
141
|
+
)
|
|
142
|
+
report = IngestReport(path=path, adapter=adapter.name, mode=mode, error_path=error_path)
|
|
143
|
+
redactor = redactor or Redactor()
|
|
144
|
+
|
|
145
|
+
with ExitStack() as stack:
|
|
146
|
+
errors: TextIO | None = None
|
|
147
|
+
if error_path is not None:
|
|
148
|
+
errors = stack.enter_context(_open_error_file(error_path))
|
|
149
|
+
|
|
150
|
+
def report_issue(issue: TraceIssue) -> None:
|
|
151
|
+
report.issue_count += 1
|
|
152
|
+
if issue.kind is IssueKind.DUPLICATE_ID:
|
|
153
|
+
report.duplicate_ids += 1
|
|
154
|
+
if len(report.sample) < sample_limit:
|
|
155
|
+
report.sample.append(issue)
|
|
156
|
+
if errors is not None:
|
|
157
|
+
errors.write(json.dumps(issue.to_dict(), sort_keys=True) + "\n")
|
|
158
|
+
|
|
159
|
+
for record in iter_records(path, adapter):
|
|
160
|
+
report.records += 1
|
|
161
|
+
if record.trace is None:
|
|
162
|
+
report.invalid += 1
|
|
163
|
+
for issue in record.issues:
|
|
164
|
+
report_issue(issue)
|
|
165
|
+
continue
|
|
166
|
+
|
|
167
|
+
report.valid += 1
|
|
168
|
+
if store is None:
|
|
169
|
+
continue
|
|
170
|
+
|
|
171
|
+
# Redaction sits here on purpose: between a trace being valid and it
|
|
172
|
+
# touching the database, with no path around it.
|
|
173
|
+
if not redactor.pseudonymizing:
|
|
174
|
+
risks = risky_identifiers(record.trace)
|
|
175
|
+
if risks:
|
|
176
|
+
report.identifier_risks += 1
|
|
177
|
+
for risk in risks:
|
|
178
|
+
if risk not in report.notices and len(report.notices) < 5:
|
|
179
|
+
report.notices.append(risk)
|
|
180
|
+
|
|
181
|
+
redacted, summary = redactor.redact(record.trace)
|
|
182
|
+
report.redactions += summary.total
|
|
183
|
+
report.redacted_traces += 1 if summary.total else 0
|
|
184
|
+
report.redaction_summary.merge(summary)
|
|
185
|
+
|
|
186
|
+
outcome = (
|
|
187
|
+
store.classify(redacted) if dry_run else store.add(redacted, redaction=summary)
|
|
188
|
+
)
|
|
189
|
+
_count_outcome(report, outcome.result)
|
|
190
|
+
|
|
191
|
+
# Every accepted sighting is recorded, whether or not the
|
|
192
|
+
# interaction itself was new. Deduplication belongs to the test
|
|
193
|
+
# suite; the evidence keeps its count.
|
|
194
|
+
canonical = _canonical_id(outcome)
|
|
195
|
+
if (
|
|
196
|
+
not dry_run
|
|
197
|
+
and canonical is not None
|
|
198
|
+
and store.record_occurrence(
|
|
199
|
+
redacted, canonical_trace_id=canonical, digest=outcome.content_hash
|
|
200
|
+
)
|
|
201
|
+
):
|
|
202
|
+
report.occurrences += 1
|
|
203
|
+
if outcome.result is StoreResult.ID_CONFLICT:
|
|
204
|
+
report_issue(
|
|
205
|
+
TraceIssue(
|
|
206
|
+
line=record.line,
|
|
207
|
+
kind=IssueKind.ID_CONFLICT,
|
|
208
|
+
message=(
|
|
209
|
+
f"trace_id {redacted.trace_id!r} is already stored with "
|
|
210
|
+
"different content"
|
|
211
|
+
),
|
|
212
|
+
trace_id=redacted.trace_id,
|
|
213
|
+
hint="Give the new trace a different ID, or remove the stored one.",
|
|
214
|
+
)
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
return report
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _canonical_id(outcome: StoreOutcome) -> str | None:
|
|
221
|
+
"""Which stored trace this sighting belongs to, or None if it was refused."""
|
|
222
|
+
match outcome.result:
|
|
223
|
+
case StoreResult.STORED | StoreResult.ALREADY_STORED:
|
|
224
|
+
return outcome.trace_id
|
|
225
|
+
case StoreResult.CONTENT_DUPLICATE:
|
|
226
|
+
return outcome.existing_trace_id
|
|
227
|
+
case _:
|
|
228
|
+
return None
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _count_outcome(report: IngestReport, result: StoreResult) -> None:
|
|
232
|
+
match result:
|
|
233
|
+
case StoreResult.STORED:
|
|
234
|
+
report.stored += 1
|
|
235
|
+
case StoreResult.ALREADY_STORED:
|
|
236
|
+
report.already_stored += 1
|
|
237
|
+
case StoreResult.CONTENT_DUPLICATE:
|
|
238
|
+
report.content_duplicates += 1
|
|
239
|
+
case StoreResult.ID_CONFLICT:
|
|
240
|
+
report.id_conflicts += 1
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def _open_error_file(error_path: Path) -> TextIO:
|
|
244
|
+
try:
|
|
245
|
+
error_path.parent.mkdir(parents=True, exist_ok=True)
|
|
246
|
+
return error_path.open("w", encoding="utf-8")
|
|
247
|
+
except OSError as exc:
|
|
248
|
+
raise CommandError(f"Could not write the error file {error_path}: {exc}") from exc
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _check_readable_file(path: Path) -> None:
|
|
252
|
+
if not path.exists():
|
|
253
|
+
raise CommandError(f"{path} does not exist.")
|
|
254
|
+
if path.is_dir():
|
|
255
|
+
raise CommandError(f"{path} is a directory, not a trace file.")
|
|
256
|
+
if not path.is_file():
|
|
257
|
+
raise CommandError(f"{path} is not a regular file.")
|
evalkeep/prompts.py
ADDED
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
"""Versioned prompts and the strict JSON schema their answers must satisfy.
|
|
2
|
+
|
|
3
|
+
The version is part of every cache key and is stored with every analysis, so a
|
|
4
|
+
prompt change never silently mixes old and new labels in one dataset. Editing
|
|
5
|
+
the prompt text without bumping :data:`FAILURE_ANALYSIS_PROMPT_VERSION` is a
|
|
6
|
+
bug: it would serve stale cached answers for a question you no longer ask.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from evalkeep.analysis import Component, FailureType, Severity
|
|
15
|
+
from evalkeep.detectors import Signal
|
|
16
|
+
from evalkeep.trace import (
|
|
17
|
+
EvaluationEvent,
|
|
18
|
+
MessageEvent,
|
|
19
|
+
NormalizedTrace,
|
|
20
|
+
ToolCallEvent,
|
|
21
|
+
ToolResultEvent,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
FAILURE_ANALYSIS_PROMPT_VERSION = 1
|
|
25
|
+
|
|
26
|
+
MAX_SUMMARY_LENGTH = 240
|
|
27
|
+
|
|
28
|
+
FAILURE_ANALYSIS_SCHEMA: dict[str, Any] = {
|
|
29
|
+
"type": "object",
|
|
30
|
+
"properties": {
|
|
31
|
+
"failure_type": {
|
|
32
|
+
"type": "string",
|
|
33
|
+
"enum": [member.value for member in FailureType],
|
|
34
|
+
"description": "The kind of failure, from the fixed list.",
|
|
35
|
+
},
|
|
36
|
+
"component": {
|
|
37
|
+
"type": "string",
|
|
38
|
+
"enum": [member.value for member in Component],
|
|
39
|
+
"description": "Where in the agent the failure originated.",
|
|
40
|
+
},
|
|
41
|
+
"severity": {
|
|
42
|
+
"type": "string",
|
|
43
|
+
"enum": [member.value for member in Severity],
|
|
44
|
+
"description": "How much damage this failure does to a user.",
|
|
45
|
+
},
|
|
46
|
+
"summary": {
|
|
47
|
+
"type": "string",
|
|
48
|
+
"description": (
|
|
49
|
+
"One sentence naming the specific mistake, in terms that would "
|
|
50
|
+
"match other traces with the same underlying problem."
|
|
51
|
+
),
|
|
52
|
+
},
|
|
53
|
+
},
|
|
54
|
+
"required": ["failure_type", "component", "severity", "summary"],
|
|
55
|
+
"additionalProperties": False,
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
FAILURE_ANALYSIS_SYSTEM = """\
|
|
59
|
+
You classify failures in recorded AI-agent interactions so that similar failures \
|
|
60
|
+
can be grouped together.
|
|
61
|
+
|
|
62
|
+
Rules:
|
|
63
|
+
- Describe only what the recorded interaction shows. Do not speculate about code \
|
|
64
|
+
you cannot see, and do not propose fixes.
|
|
65
|
+
- Write the summary so that two traces with the same underlying problem would \
|
|
66
|
+
receive near-identical summaries. Name the mistake, not the customer, the order \
|
|
67
|
+
or the wording of this particular request.
|
|
68
|
+
- Values shown as [REDACTED:...] were removed before you saw them. Treat them as \
|
|
69
|
+
opaque; never guess what they were.
|
|
70
|
+
- Choose the most specific failure_type that fits. Use "other" only when nothing \
|
|
71
|
+
else applies.
|
|
72
|
+
- Severity is about user impact: critical means money, data or safety; low means \
|
|
73
|
+
cosmetic or easily noticed.\
|
|
74
|
+
"""
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def failure_analysis_prompt(trace: NormalizedTrace, signals: list[Signal]) -> str:
|
|
78
|
+
"""The user-turn prompt describing one failing interaction."""
|
|
79
|
+
sections = [
|
|
80
|
+
"Here is a recorded interaction that has been marked as a failure.",
|
|
81
|
+
"",
|
|
82
|
+
"## Evidence that it failed",
|
|
83
|
+
]
|
|
84
|
+
sections.extend(f"- ({signal.kind.value}) {signal.summary}" for signal in signals)
|
|
85
|
+
sections += ["", "## The interaction", _render_trace(trace)]
|
|
86
|
+
sections += [
|
|
87
|
+
"",
|
|
88
|
+
"Classify this failure. Respond with a JSON object matching the required schema.",
|
|
89
|
+
]
|
|
90
|
+
return "\n".join(sections)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _render_trace(trace: NormalizedTrace) -> str:
|
|
94
|
+
"""A compact, stable rendering. Stable matters: it feeds the cache key."""
|
|
95
|
+
lines: list[str] = []
|
|
96
|
+
if trace.input.text:
|
|
97
|
+
lines.append(f"user: {trace.input.text}")
|
|
98
|
+
for message in trace.input.messages:
|
|
99
|
+
lines.append(f"{message.role.value}: {message.content}")
|
|
100
|
+
|
|
101
|
+
for event in trace.events:
|
|
102
|
+
if isinstance(event, ToolCallEvent):
|
|
103
|
+
arguments = json.dumps(event.arguments, sort_keys=True)
|
|
104
|
+
lines.append(f"tool_call: {event.tool}({arguments})")
|
|
105
|
+
elif isinstance(event, ToolResultEvent):
|
|
106
|
+
result = json.dumps(event.result, sort_keys=True, default=str)
|
|
107
|
+
suffix = f" error={event.error}" if event.error else ""
|
|
108
|
+
lines.append(f"tool_result: {event.tool} -> {result}{suffix}")
|
|
109
|
+
elif isinstance(event, MessageEvent):
|
|
110
|
+
lines.append(f"{event.role.value}: {event.content}")
|
|
111
|
+
elif isinstance(event, EvaluationEvent):
|
|
112
|
+
verdict = {True: "pass", False: "fail", None: "unrecorded"}[event.passed]
|
|
113
|
+
lines.append(f"evaluation: {event.name} {verdict}")
|
|
114
|
+
|
|
115
|
+
if trace.output is not None and trace.output.text:
|
|
116
|
+
lines.append(f"assistant: {trace.output.text}")
|
|
117
|
+
for message in trace.output.messages if trace.output else []:
|
|
118
|
+
lines.append(f"{message.role.value}: {message.content}")
|
|
119
|
+
|
|
120
|
+
if trace.outcome.feedback is not None and trace.outcome.feedback.comment:
|
|
121
|
+
lines.append(f"feedback: {trace.outcome.feedback.comment}")
|
|
122
|
+
for evaluation in trace.outcome.evaluations:
|
|
123
|
+
if evaluation.passed is False:
|
|
124
|
+
reason = f": {evaluation.reason}" if evaluation.reason else ""
|
|
125
|
+
lines.append(f"failed evaluation: {evaluation.name}{reason}")
|
|
126
|
+
|
|
127
|
+
return "\n".join(lines)
|
evalkeep/pseudonyms.py
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""Deterministic pseudonyms for identifiers that may carry customer data.
|
|
2
|
+
|
|
3
|
+
Redaction deliberately leaves identifiers alone, because rewriting them would
|
|
4
|
+
break the links the pipeline runs on. That is fine when a `trace_id` is a UUID
|
|
5
|
+
and dangerous when it is `order-jane@example.com-2026-06-01`.
|
|
6
|
+
|
|
7
|
+
Pseudonymization resolves the tension instead of trading one problem for the
|
|
8
|
+
other. Each identifier becomes a token derived from a per-project secret salt:
|
|
9
|
+
|
|
10
|
+
* **Deterministic**, so the same original always yields the same token and every
|
|
11
|
+
link in the pipeline survives.
|
|
12
|
+
* **Not reversible from the database**, because the original is never stored --
|
|
13
|
+
only the token is.
|
|
14
|
+
* **Still usable by hand**, because a lookup can hash whatever the user typed
|
|
15
|
+
and search for that. You keep using the IDs your own systems know.
|
|
16
|
+
* **Scoped to one project**, because the salt is per-project and never
|
|
17
|
+
committed. Two projects produce different tokens for the same original, so a
|
|
18
|
+
shared export leaks nothing about another project's data.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import hashlib
|
|
24
|
+
import secrets
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
|
|
27
|
+
from evalkeep.errors import CommandError
|
|
28
|
+
|
|
29
|
+
SALT_FILENAME = "salt"
|
|
30
|
+
SALT_BYTES = 32
|
|
31
|
+
TOKEN_LENGTH = 12
|
|
32
|
+
|
|
33
|
+
#: Which identifier gets which readable prefix, so a pseudonym still looks like
|
|
34
|
+
#: the kind of thing it replaced.
|
|
35
|
+
PREFIXES: dict[str, str] = {
|
|
36
|
+
"trace_id": "trace",
|
|
37
|
+
"event_id": "event",
|
|
38
|
+
"call_id": "call",
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class Pseudonymizer:
|
|
43
|
+
"""Turns an identifier into a stable token, given a project's salt."""
|
|
44
|
+
|
|
45
|
+
def __init__(self, salt: bytes) -> None:
|
|
46
|
+
if len(salt) < 16:
|
|
47
|
+
raise ValueError("the salt must be at least 16 bytes")
|
|
48
|
+
self._salt = salt
|
|
49
|
+
|
|
50
|
+
def token(self, value: str, *, field: str) -> str:
|
|
51
|
+
"""A stable pseudonym for ``value``, prefixed by the kind of ID it is."""
|
|
52
|
+
digest = hashlib.blake2b(
|
|
53
|
+
value.encode("utf-8"), digest_size=TOKEN_LENGTH // 2, key=self._salt
|
|
54
|
+
).hexdigest()
|
|
55
|
+
return f"{PREFIXES.get(field, 'id')}-{digest}"
|
|
56
|
+
|
|
57
|
+
@classmethod
|
|
58
|
+
def load(cls, path: Path) -> Pseudonymizer:
|
|
59
|
+
"""Read a project's salt, creating one on first use."""
|
|
60
|
+
if not path.is_file():
|
|
61
|
+
return cls(_create_salt(path))
|
|
62
|
+
try:
|
|
63
|
+
salt = bytes.fromhex(path.read_text(encoding="utf-8").strip())
|
|
64
|
+
except (OSError, ValueError) as exc:
|
|
65
|
+
raise CommandError(
|
|
66
|
+
f"Could not read the pseudonymization salt at {path}: {exc}.",
|
|
67
|
+
hint="If it was lost, previously stored identifiers cannot be "
|
|
68
|
+
"matched again; re-ingest into a fresh project.",
|
|
69
|
+
) from exc
|
|
70
|
+
return cls(salt)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _create_salt(path: Path) -> bytes:
|
|
74
|
+
salt = secrets.token_bytes(SALT_BYTES)
|
|
75
|
+
try:
|
|
76
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
77
|
+
path.write_text(salt.hex(), encoding="utf-8")
|
|
78
|
+
# The salt is what makes the tokens unguessable; treat it like a key.
|
|
79
|
+
path.chmod(0o600)
|
|
80
|
+
except OSError as exc:
|
|
81
|
+
raise CommandError(f"Could not write the salt to {path}: {exc}") from exc
|
|
82
|
+
return salt
|
evalkeep/py.typed
ADDED
|
File without changes
|