evalkeep 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalkeep/__init__.py +12 -0
- evalkeep/__main__.py +6 -0
- evalkeep/adapters/__init__.py +45 -0
- evalkeep/adapters/base.py +92 -0
- evalkeep/adapters/jsonl.py +164 -0
- evalkeep/adapters/langsmith.py +436 -0
- evalkeep/adapters/otlp.py +442 -0
- evalkeep/adapters/semconv.py +208 -0
- evalkeep/analysis.py +174 -0
- evalkeep/analysis_run.py +160 -0
- evalkeep/analyzers/__init__.py +52 -0
- evalkeep/analyzers/anthropic.py +145 -0
- evalkeep/analyzers/stub.py +34 -0
- evalkeep/cache.py +122 -0
- evalkeep/cli.py +1933 -0
- evalkeep/clustering.py +383 -0
- evalkeep/clusters.py +101 -0
- evalkeep/commands/__init__.py +1 -0
- evalkeep/commands/analyze_cmd.py +100 -0
- evalkeep/commands/compare_cmd.py +169 -0
- evalkeep/commands/dataset_cmd.py +182 -0
- evalkeep/commands/detect_cmd.py +154 -0
- evalkeep/commands/discover_cmd.py +274 -0
- evalkeep/commands/ingest_cmd.py +50 -0
- evalkeep/commands/init_cmd.py +151 -0
- evalkeep/commands/pipeline_cmd.py +156 -0
- evalkeep/commands/review_cmd.py +141 -0
- evalkeep/commands/run_cmd.py +131 -0
- evalkeep/commands/target_cmd.py +109 -0
- evalkeep/commands/trace_cmd.py +58 -0
- evalkeep/comparison.py +432 -0
- evalkeep/config.py +209 -0
- evalkeep/detection.py +94 -0
- evalkeep/detectors.py +182 -0
- evalkeep/discovery.py +208 -0
- evalkeep/embeddings/__init__.py +31 -0
- evalkeep/embeddings/base.py +32 -0
- evalkeep/embeddings/hashing.py +98 -0
- evalkeep/errors.py +42 -0
- evalkeep/examples/__init__.py +37 -0
- evalkeep/examples/langsmith/runs.jsonl +18 -0
- evalkeep/examples/opentelemetry/spans.json +898 -0
- evalkeep/examples/refund-agent/agents/baseline.py +66 -0
- evalkeep/examples/refund-agent/agents/candidate.py +66 -0
- evalkeep/examples/refund-agent/traces.jsonl +5 -0
- evalkeep/examples/tau-bench/prepare.py +230 -0
- evalkeep/exporters/__init__.py +45 -0
- evalkeep/exporters/generic.py +31 -0
- evalkeep/exporters/promptfoo.py +219 -0
- evalkeep/failures.py +95 -0
- evalkeep/generation.py +303 -0
- evalkeep/hashing.py +56 -0
- evalkeep/ingest.py +257 -0
- evalkeep/prompts.py +127 -0
- evalkeep/pseudonyms.py +82 -0
- evalkeep/py.typed +0 -0
- evalkeep/redaction.py +333 -0
- evalkeep/regression.py +409 -0
- evalkeep/review.py +309 -0
- evalkeep/runner.py +302 -0
- evalkeep/runs.py +185 -0
- evalkeep/storage/__init__.py +37 -0
- evalkeep/storage/clusters.py +163 -0
- evalkeep/storage/failures.py +254 -0
- evalkeep/storage/migrations.py +370 -0
- evalkeep/storage/regression.py +136 -0
- evalkeep/storage/runs.py +223 -0
- evalkeep/storage/store.py +429 -0
- evalkeep/targets.py +205 -0
- evalkeep/trace.py +238 -0
- evalkeep-0.1.0.dist-info/METADATA +221 -0
- evalkeep-0.1.0.dist-info/RECORD +75 -0
- evalkeep-0.1.0.dist-info/WHEEL +4 -0
- evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
- evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/comparison.py
ADDED
|
@@ -0,0 +1,432 @@
|
|
|
1
|
+
"""Comparing two runs, and saying only what the numbers support.
|
|
2
|
+
|
|
3
|
+
The whole pipeline exists to answer one question -- did this change make the
|
|
4
|
+
agent better or worse -- and this is where that answer is produced. Three rules
|
|
5
|
+
shape it, all of them about not overclaiming:
|
|
6
|
+
|
|
7
|
+
* **A test that errored is not a data point.** An error says the harness or the
|
|
8
|
+
target broke, not that the agent got the answer wrong. Errored pairs are
|
|
9
|
+
excluded from every count and reported separately, because letting an outage
|
|
10
|
+
read as a regression is the single most damaging mistake this tool could make.
|
|
11
|
+
* **Two runs are only comparable if they answered the same questions.** Runs
|
|
12
|
+
carry a suite hash; comparing across different suites is refused unless the
|
|
13
|
+
caller explicitly asks for the intersection.
|
|
14
|
+
* **A confidence interval is only reported when it means something.** With a
|
|
15
|
+
handful of discordant pairs the normal approximation is not trustworthy, so
|
|
16
|
+
the interval is withheld and the reason is printed instead of a number that
|
|
17
|
+
would look authoritative and be wrong.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
from dataclasses import dataclass, field
|
|
23
|
+
from enum import StrEnum
|
|
24
|
+
from math import comb
|
|
25
|
+
|
|
26
|
+
from evalkeep.runs import (
|
|
27
|
+
CaseResult,
|
|
28
|
+
CaseSummary,
|
|
29
|
+
EvaluationRun,
|
|
30
|
+
Outcome,
|
|
31
|
+
Verdict,
|
|
32
|
+
summarize,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
#: Below this many discordant pairs the normal approximation behind the interval
|
|
36
|
+
#: is not trustworthy, so no interval is reported. A common rule of thumb, and
|
|
37
|
+
#: chosen here because being silent is better than being confidently wrong.
|
|
38
|
+
MIN_DISCORDANT_FOR_INTERVAL = 10
|
|
39
|
+
|
|
40
|
+
#: 95% two-sided normal quantile.
|
|
41
|
+
_Z = 1.959963984540054
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class Classification(StrEnum):
|
|
45
|
+
"""Guide 9.1's truth table, with its two error rows made explicit."""
|
|
46
|
+
|
|
47
|
+
UNCHANGED_PASS = "unchanged_pass"
|
|
48
|
+
FIXED = "fixed"
|
|
49
|
+
#: Improved and now passes most of the time, but not every time. Only
|
|
50
|
+
#: reachable with repetitions -- a single execution cannot tell the
|
|
51
|
+
#: difference between this and `fixed`, which is the whole reason to repeat.
|
|
52
|
+
LIKELY_FIXED = "likely_fixed"
|
|
53
|
+
REGRESSION = "regression"
|
|
54
|
+
UNCHANGED_FAILURE = "unchanged_failure"
|
|
55
|
+
#: One side never ran. Excluded from the counts, reported on its own.
|
|
56
|
+
NOT_COMPARABLE = "not_comparable"
|
|
57
|
+
#: Present in one run and absent from the other.
|
|
58
|
+
MISSING = "missing"
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
#: The four classifications that say something about the agent.
|
|
62
|
+
COMPARABLE = (
|
|
63
|
+
Classification.UNCHANGED_PASS,
|
|
64
|
+
Classification.FIXED,
|
|
65
|
+
Classification.LIKELY_FIXED,
|
|
66
|
+
Classification.REGRESSION,
|
|
67
|
+
Classification.UNCHANGED_FAILURE,
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
#: Classifications that count as the candidate passing, for the paired test.
|
|
71
|
+
#: A case that only sometimes passes is not counted as passing: the point of
|
|
72
|
+
#: repeating was to stop calling that a fix.
|
|
73
|
+
_PASSING = (Classification.UNCHANGED_PASS, Classification.FIXED)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
@dataclass(frozen=True)
|
|
77
|
+
class CaseComparison:
|
|
78
|
+
test_id: str
|
|
79
|
+
classification: Classification
|
|
80
|
+
baseline: CaseResult | None = None
|
|
81
|
+
candidate: CaseResult | None = None
|
|
82
|
+
baseline_summary: CaseSummary | None = None
|
|
83
|
+
candidate_summary: CaseSummary | None = None
|
|
84
|
+
|
|
85
|
+
@property
|
|
86
|
+
def flaky(self) -> bool:
|
|
87
|
+
"""Whether either side was inconsistent across its repetitions.
|
|
88
|
+
|
|
89
|
+
Orthogonal to the classification: a case can be both a regression and
|
|
90
|
+
flaky, and hiding one behind the other would lose a real finding.
|
|
91
|
+
"""
|
|
92
|
+
return any(
|
|
93
|
+
summary is not None and summary.flaky
|
|
94
|
+
for summary in (self.baseline_summary, self.candidate_summary)
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
@property
|
|
98
|
+
def confidence(self) -> tuple[float, float] | None:
|
|
99
|
+
"""How reliably the candidate passes this case, if it was repeated."""
|
|
100
|
+
if self.candidate_summary is None:
|
|
101
|
+
return None
|
|
102
|
+
return self.candidate_summary.confidence
|
|
103
|
+
|
|
104
|
+
@property
|
|
105
|
+
def reason(self) -> str:
|
|
106
|
+
"""Why this pair is not comparable, when it is not."""
|
|
107
|
+
if self.classification is Classification.MISSING:
|
|
108
|
+
side = "candidate" if self.baseline_summary is not None else "baseline"
|
|
109
|
+
return f"absent from the {side} run"
|
|
110
|
+
for label, summary, result in (
|
|
111
|
+
("baseline", self.baseline_summary, self.baseline),
|
|
112
|
+
("candidate", self.candidate_summary, self.candidate),
|
|
113
|
+
):
|
|
114
|
+
if summary is not None and summary.verdict is Verdict.ERROR:
|
|
115
|
+
kind = (
|
|
116
|
+
result.error_kind.value if result is not None and result.error_kind else "error"
|
|
117
|
+
)
|
|
118
|
+
return f"{label} {kind}"
|
|
119
|
+
return ""
|
|
120
|
+
|
|
121
|
+
@property
|
|
122
|
+
def rates(self) -> str:
|
|
123
|
+
"""How often each side passed, when either was repeated."""
|
|
124
|
+
parts = []
|
|
125
|
+
for label, summary in (
|
|
126
|
+
("before", self.baseline_summary),
|
|
127
|
+
("after", self.candidate_summary),
|
|
128
|
+
):
|
|
129
|
+
if summary is not None and summary.repetitions > 1:
|
|
130
|
+
parts.append(f"{label} {summary.passed}/{summary.evaluated}")
|
|
131
|
+
return ", ".join(parts)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
@dataclass
|
|
135
|
+
class PairedStatistics:
|
|
136
|
+
"""Paired analysis over the tests both runs actually evaluated."""
|
|
137
|
+
|
|
138
|
+
pairs: int
|
|
139
|
+
fixed: int
|
|
140
|
+
regressions: int
|
|
141
|
+
difference: float
|
|
142
|
+
p_value: float
|
|
143
|
+
interval: tuple[float, float] | None = None
|
|
144
|
+
interval_method: str | None = None
|
|
145
|
+
#: Why an interval was withheld, when it was.
|
|
146
|
+
note: str | None = None
|
|
147
|
+
|
|
148
|
+
@property
|
|
149
|
+
def discordant(self) -> int:
|
|
150
|
+
"""Pairs where the two runs disagreed. All the information is here."""
|
|
151
|
+
return self.fixed + self.regressions
|
|
152
|
+
|
|
153
|
+
@property
|
|
154
|
+
def significant(self) -> bool:
|
|
155
|
+
return self.p_value < 0.05
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
@dataclass
|
|
159
|
+
class ComparisonReport:
|
|
160
|
+
baseline_run: EvaluationRun
|
|
161
|
+
candidate_run: EvaluationRun
|
|
162
|
+
comparisons: list[CaseComparison] = field(default_factory=list)
|
|
163
|
+
suite_compatible: bool = True
|
|
164
|
+
|
|
165
|
+
@property
|
|
166
|
+
def counts(self) -> dict[Classification, int]:
|
|
167
|
+
tally: dict[Classification, int] = {}
|
|
168
|
+
for comparison in self.comparisons:
|
|
169
|
+
tally[comparison.classification] = tally.get(comparison.classification, 0) + 1
|
|
170
|
+
return tally
|
|
171
|
+
|
|
172
|
+
@property
|
|
173
|
+
def comparable(self) -> list[CaseComparison]:
|
|
174
|
+
return [c for c in self.comparisons if c.classification in COMPARABLE]
|
|
175
|
+
|
|
176
|
+
@property
|
|
177
|
+
def excluded(self) -> list[CaseComparison]:
|
|
178
|
+
return [c for c in self.comparisons if c.classification not in COMPARABLE]
|
|
179
|
+
|
|
180
|
+
@property
|
|
181
|
+
def regressions(self) -> list[CaseComparison]:
|
|
182
|
+
return [c for c in self.comparisons if c.classification is Classification.REGRESSION]
|
|
183
|
+
|
|
184
|
+
@property
|
|
185
|
+
def fixes(self) -> list[CaseComparison]:
|
|
186
|
+
return [c for c in self.comparisons if c.classification is Classification.FIXED]
|
|
187
|
+
|
|
188
|
+
@property
|
|
189
|
+
def baseline_pass_rate(self) -> float | None:
|
|
190
|
+
return _rate(_passing(self.comparable, before=True), len(self.comparable))
|
|
191
|
+
|
|
192
|
+
@property
|
|
193
|
+
def candidate_pass_rate(self) -> float | None:
|
|
194
|
+
"""The share of cases that pass *reliably*.
|
|
195
|
+
|
|
196
|
+
A case that passes nine times in ten is not counted as passing here.
|
|
197
|
+
Counting it would reintroduce exactly the overclaim that repeating the
|
|
198
|
+
run was meant to remove.
|
|
199
|
+
"""
|
|
200
|
+
return _rate(_passing(self.comparable, before=False), len(self.comparable))
|
|
201
|
+
|
|
202
|
+
@property
|
|
203
|
+
def flaky(self) -> list[CaseComparison]:
|
|
204
|
+
return [c for c in self.comparisons if c.flaky]
|
|
205
|
+
|
|
206
|
+
@property
|
|
207
|
+
def repeated(self) -> bool:
|
|
208
|
+
return max(self.baseline_run.repetitions, self.candidate_run.repetitions) > 1
|
|
209
|
+
|
|
210
|
+
@property
|
|
211
|
+
def statistics(self) -> PairedStatistics | None:
|
|
212
|
+
return paired_statistics(self.comparable)
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def compare_results(
|
|
216
|
+
baseline_run: EvaluationRun,
|
|
217
|
+
baseline_results: list[CaseResult],
|
|
218
|
+
candidate_run: EvaluationRun,
|
|
219
|
+
candidate_results: list[CaseResult],
|
|
220
|
+
) -> ComparisonReport:
|
|
221
|
+
"""Align two runs by stable test ID and classify every case."""
|
|
222
|
+
baseline_summaries = summarize(baseline_results)
|
|
223
|
+
candidate_summaries = summarize(candidate_results)
|
|
224
|
+
baseline_first = _first_by_case(baseline_results)
|
|
225
|
+
candidate_first = _first_by_case(candidate_results)
|
|
226
|
+
|
|
227
|
+
comparisons = [
|
|
228
|
+
_classify(
|
|
229
|
+
test_id,
|
|
230
|
+
baseline_summaries.get(test_id),
|
|
231
|
+
candidate_summaries.get(test_id),
|
|
232
|
+
baseline_first.get(test_id),
|
|
233
|
+
candidate_first.get(test_id),
|
|
234
|
+
)
|
|
235
|
+
for test_id in sorted(set(baseline_summaries) | set(candidate_summaries))
|
|
236
|
+
]
|
|
237
|
+
return ComparisonReport(
|
|
238
|
+
baseline_run=baseline_run,
|
|
239
|
+
candidate_run=candidate_run,
|
|
240
|
+
comparisons=comparisons,
|
|
241
|
+
suite_compatible=baseline_run.suite_hash == candidate_run.suite_hash,
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _first_by_case(results: list[CaseResult]) -> dict[str, CaseResult]:
|
|
246
|
+
"""One representative execution per case, for showing a failure reason."""
|
|
247
|
+
first: dict[str, CaseResult] = {}
|
|
248
|
+
for result in results:
|
|
249
|
+
first.setdefault(result.test_id, result)
|
|
250
|
+
if result.outcome is Outcome.FAIL:
|
|
251
|
+
first[result.test_id] = result
|
|
252
|
+
return first
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _classify(
|
|
256
|
+
test_id: str,
|
|
257
|
+
baseline: CaseSummary | None,
|
|
258
|
+
candidate: CaseSummary | None,
|
|
259
|
+
baseline_result: CaseResult | None,
|
|
260
|
+
candidate_result: CaseResult | None,
|
|
261
|
+
) -> CaseComparison:
|
|
262
|
+
"""Extend the truth table from single outcomes to repeated ones.
|
|
263
|
+
|
|
264
|
+
With one repetition each side is either PASS or FAIL and this reduces
|
|
265
|
+
exactly to the original four rows. With more, a third verdict appears --
|
|
266
|
+
FLAKY -- and it is what stops a single lucky pass being reported as a fix.
|
|
267
|
+
"""
|
|
268
|
+
if baseline is None or candidate is None:
|
|
269
|
+
classification = Classification.MISSING
|
|
270
|
+
elif baseline.verdict is Verdict.ERROR or candidate.verdict is Verdict.ERROR:
|
|
271
|
+
classification = Classification.NOT_COMPARABLE
|
|
272
|
+
else:
|
|
273
|
+
classification = _direction(baseline, candidate)
|
|
274
|
+
|
|
275
|
+
return CaseComparison(
|
|
276
|
+
test_id=test_id,
|
|
277
|
+
classification=classification,
|
|
278
|
+
baseline=baseline_result,
|
|
279
|
+
candidate=candidate_result,
|
|
280
|
+
baseline_summary=baseline,
|
|
281
|
+
candidate_summary=candidate,
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def _direction(baseline: CaseSummary, candidate: CaseSummary) -> Classification:
|
|
286
|
+
before, after = baseline.verdict, candidate.verdict
|
|
287
|
+
|
|
288
|
+
if after is Verdict.PASS:
|
|
289
|
+
# Reliably passing now. It is a fix unless it was already reliable.
|
|
290
|
+
return Classification.UNCHANGED_PASS if before is Verdict.PASS else Classification.FIXED
|
|
291
|
+
|
|
292
|
+
if after is Verdict.FAIL:
|
|
293
|
+
# Never passes now. A regression only if it used to pass at all.
|
|
294
|
+
return (
|
|
295
|
+
Classification.UNCHANGED_FAILURE
|
|
296
|
+
if before is Verdict.FAIL
|
|
297
|
+
else Classification.REGRESSION
|
|
298
|
+
)
|
|
299
|
+
|
|
300
|
+
# The candidate is flaky. Whether that is progress depends on what it was.
|
|
301
|
+
if before is Verdict.PASS:
|
|
302
|
+
# It used to always pass and now sometimes does not. That is worse,
|
|
303
|
+
# whatever the rate says.
|
|
304
|
+
return Classification.REGRESSION
|
|
305
|
+
if before is Verdict.FAIL:
|
|
306
|
+
return (
|
|
307
|
+
Classification.LIKELY_FIXED
|
|
308
|
+
if _passes_more_often_than_not(candidate)
|
|
309
|
+
else Classification.UNCHANGED_FAILURE
|
|
310
|
+
)
|
|
311
|
+
|
|
312
|
+
# Flaky before and flaky after: compare how often, not whether.
|
|
313
|
+
before_rate = baseline.pass_rate or 0.0
|
|
314
|
+
after_rate = candidate.pass_rate or 0.0
|
|
315
|
+
if after_rate > before_rate:
|
|
316
|
+
return (
|
|
317
|
+
Classification.LIKELY_FIXED
|
|
318
|
+
if _passes_more_often_than_not(candidate)
|
|
319
|
+
else Classification.UNCHANGED_FAILURE
|
|
320
|
+
)
|
|
321
|
+
if after_rate < before_rate:
|
|
322
|
+
return Classification.REGRESSION
|
|
323
|
+
return Classification.UNCHANGED_FAILURE
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def _passes_more_often_than_not(summary: CaseSummary) -> bool:
|
|
327
|
+
"""True only when the sample supports the claim, not merely suggests it.
|
|
328
|
+
|
|
329
|
+
Two passes out of three looks like a majority and is not evidence of one;
|
|
330
|
+
the lower bound of the interval is what decides.
|
|
331
|
+
"""
|
|
332
|
+
interval = summary.confidence
|
|
333
|
+
return interval is not None and interval[0] > 0.5
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def paired_statistics(comparable: list[CaseComparison]) -> PairedStatistics | None:
|
|
337
|
+
"""McNemar's exact test over the pairs, with an interval only when earned.
|
|
338
|
+
|
|
339
|
+
The test is exact rather than the chi-square approximation: suites here are
|
|
340
|
+
often small, and the approximation is unreliable exactly where these suites
|
|
341
|
+
live. Only discordant pairs carry information -- a test both runs passed
|
|
342
|
+
says nothing about whether anything changed -- so the test is a two-sided
|
|
343
|
+
binomial on fixes versus regressions.
|
|
344
|
+
"""
|
|
345
|
+
pairs = len(comparable)
|
|
346
|
+
if pairs == 0:
|
|
347
|
+
return None
|
|
348
|
+
|
|
349
|
+
fixed = sum(1 for c in comparable if c.classification is Classification.FIXED)
|
|
350
|
+
regressions = sum(1 for c in comparable if c.classification is Classification.REGRESSION)
|
|
351
|
+
difference = (fixed - regressions) / pairs
|
|
352
|
+
|
|
353
|
+
discordant = fixed + regressions
|
|
354
|
+
if discordant == 0:
|
|
355
|
+
# Nothing changed on any test. There is no evidence of a difference,
|
|
356
|
+
# which is not the same as evidence of no difference.
|
|
357
|
+
return PairedStatistics(
|
|
358
|
+
pairs=pairs,
|
|
359
|
+
fixed=0,
|
|
360
|
+
regressions=0,
|
|
361
|
+
difference=0.0,
|
|
362
|
+
p_value=1.0,
|
|
363
|
+
note="No test changed outcome, so there is nothing to test.",
|
|
364
|
+
)
|
|
365
|
+
|
|
366
|
+
p_value = exact_binomial_two_sided(fixed, discordant)
|
|
367
|
+
|
|
368
|
+
statistics = PairedStatistics(
|
|
369
|
+
pairs=pairs,
|
|
370
|
+
fixed=fixed,
|
|
371
|
+
regressions=regressions,
|
|
372
|
+
difference=difference,
|
|
373
|
+
p_value=p_value,
|
|
374
|
+
)
|
|
375
|
+
|
|
376
|
+
if discordant < MIN_DISCORDANT_FOR_INTERVAL:
|
|
377
|
+
statistics.note = (
|
|
378
|
+
f"Only {discordant} test(s) changed outcome; that is too few for a "
|
|
379
|
+
"trustworthy interval, so none is given."
|
|
380
|
+
)
|
|
381
|
+
return statistics
|
|
382
|
+
|
|
383
|
+
# Wald interval for the paired difference in proportions.
|
|
384
|
+
variance = (fixed + regressions - (fixed - regressions) ** 2 / pairs) / pairs**2
|
|
385
|
+
margin = _Z * (variance**0.5)
|
|
386
|
+
statistics.interval = (
|
|
387
|
+
max(-1.0, difference - margin),
|
|
388
|
+
min(1.0, difference + margin),
|
|
389
|
+
)
|
|
390
|
+
statistics.interval_method = "paired Wald, 95%"
|
|
391
|
+
return statistics
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
def exact_binomial_two_sided(successes: int, trials: int) -> float:
|
|
395
|
+
"""The exact two-sided binomial p-value at p = 0.5.
|
|
396
|
+
|
|
397
|
+
This is McNemar's exact test: under the null, each discordant pair is a coin
|
|
398
|
+
flip, so the question is how surprising this split would be. At p = 0.5 the
|
|
399
|
+
distribution is symmetric, which makes the two-sided value simply both tails
|
|
400
|
+
of the more extreme side -- no approximation, and no reason to carry SciPy's
|
|
401
|
+
82 MB for one call. Checked against `scipy.stats.binomtest` for every split
|
|
402
|
+
up to sixty trials before that dependency was removed; the largest
|
|
403
|
+
disagreement was 5.6e-16, which is floating-point noise.
|
|
404
|
+
"""
|
|
405
|
+
if trials <= 0:
|
|
406
|
+
return 1.0
|
|
407
|
+
smaller = min(successes, trials - successes)
|
|
408
|
+
tail = sum(comb(trials, k) for k in range(smaller + 1)) / 2**trials
|
|
409
|
+
return float(min(1.0, 2 * tail))
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def _reliably_passing(summary: CaseSummary | None) -> bool:
|
|
413
|
+
return summary is not None and summary.verdict is Verdict.PASS
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def _passing(comparisons: list[CaseComparison], *, before: bool) -> int:
|
|
417
|
+
return sum(
|
|
418
|
+
1
|
|
419
|
+
for c in comparisons
|
|
420
|
+
if _reliably_passing(c.baseline_summary if before else c.candidate_summary)
|
|
421
|
+
)
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
def _became(comparison: CaseComparison, *, passing: bool) -> bool:
|
|
425
|
+
"""Whether this case crossed the pass/not-pass line in the given direction."""
|
|
426
|
+
was = _reliably_passing(comparison.baseline_summary)
|
|
427
|
+
now = _reliably_passing(comparison.candidate_summary)
|
|
428
|
+
return (now and not was) if passing else (was and not now)
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def _rate(passed: int, total: int) -> float | None:
|
|
432
|
+
return passed / total if total else None
|
evalkeep/config.py
ADDED
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
"""Project configuration and the on-disk layout created by ``evalkeep init``.
|
|
2
|
+
|
|
3
|
+
Everything Evalkeep writes lives under a single state directory
|
|
4
|
+
(``.evalkeep/`` by default) so that a project can be inspected, backed up or
|
|
5
|
+
deleted in one step. Only ``evalkeep.yaml`` is meant to be committed; the
|
|
6
|
+
state directory holds raw traces, the database, caches and run outputs, all of
|
|
7
|
+
which stay out of Git (see the generated ``.gitignore`` entries).
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
import yaml
|
|
16
|
+
from pydantic import BaseModel, Field, ValidationError
|
|
17
|
+
|
|
18
|
+
from evalkeep.errors import CommandError
|
|
19
|
+
from evalkeep.pseudonyms import SALT_FILENAME, Pseudonymizer
|
|
20
|
+
|
|
21
|
+
CONFIG_FILENAME = "evalkeep.yaml"
|
|
22
|
+
STATE_DIRNAME = ".evalkeep"
|
|
23
|
+
|
|
24
|
+
#: Bumped when the on-disk layout changes in a way that needs a migration.
|
|
25
|
+
CONFIG_VERSION = 1
|
|
26
|
+
|
|
27
|
+
#: Subdirectories of the state directory, with the purpose of each.
|
|
28
|
+
STATE_SUBDIRS: dict[str, str] = {
|
|
29
|
+
"data": "Redacted trace payloads and intermediate artifacts.",
|
|
30
|
+
"cache": "Analyzer and embedding caches, keyed by content hash.",
|
|
31
|
+
"runs": "Raw Promptfoo run outputs, one directory per run.",
|
|
32
|
+
"exports": "Generated export files (generic JSONL, Promptfoo YAML).",
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
#: Paths excluded from Git. Approved tests and config are committed; traces,
|
|
36
|
+
#: the database, caches and run outputs are not.
|
|
37
|
+
GITIGNORE_ENTRIES: tuple[str, ...] = (
|
|
38
|
+
".env",
|
|
39
|
+
f"{STATE_DIRNAME}/database.db",
|
|
40
|
+
# The salt is what makes pseudonyms unguessable.
|
|
41
|
+
f"{STATE_DIRNAME}/salt",
|
|
42
|
+
f"{STATE_DIRNAME}/data/",
|
|
43
|
+
f"{STATE_DIRNAME}/cache/",
|
|
44
|
+
f"{STATE_DIRNAME}/runs/",
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
GITIGNORE_HEADER = "# evalkeep"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class RedactionConfig(BaseModel):
|
|
51
|
+
"""Which built-in redactors run before anything is written to storage."""
|
|
52
|
+
|
|
53
|
+
emails: bool = True
|
|
54
|
+
phone_numbers: bool = True
|
|
55
|
+
payment_cards: bool = True
|
|
56
|
+
token_prefixes: bool = True
|
|
57
|
+
secret_field_names: bool = True
|
|
58
|
+
#: Replace trace, event and call IDs with per-project tokens. Off by
|
|
59
|
+
#: default because it changes the IDs you see; turn it on when your own
|
|
60
|
+
#: identifiers embed customer data. Lookups keep accepting the originals.
|
|
61
|
+
pseudonymize_identifiers: bool = False
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class AnalyzerConfig(BaseModel):
|
|
65
|
+
"""Which provider describes failures, if any.
|
|
66
|
+
|
|
67
|
+
``manual`` is the default: Evalkeep runs fully offline and failures are
|
|
68
|
+
labelled by hand until a provider is configured.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
provider: str = "manual"
|
|
72
|
+
model: str = "claude-opus-5"
|
|
73
|
+
effort: str = "medium"
|
|
74
|
+
max_tokens: int = 16000
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class ClusteringConfig(BaseModel):
|
|
78
|
+
"""How failures are embedded and grouped.
|
|
79
|
+
|
|
80
|
+
Every field is stored with the clustering run that used it, so a grouping
|
|
81
|
+
can always be reproduced or explained.
|
|
82
|
+
"""
|
|
83
|
+
|
|
84
|
+
embedder: str = "hashing"
|
|
85
|
+
dimensions: int = 512
|
|
86
|
+
#: Recorded and applied even where the algorithm is deterministic, so that
|
|
87
|
+
#: swapping in a randomized one later cannot quietly break reproducibility.
|
|
88
|
+
seed: int = 0
|
|
89
|
+
algorithm: str = "agglomerative"
|
|
90
|
+
metric: str = "cosine"
|
|
91
|
+
linkage: str = "average"
|
|
92
|
+
#: Cosine distance above which two failures are different families.
|
|
93
|
+
threshold: float = 0.55
|
|
94
|
+
#: The same, for failures nobody has described yet. Lower because the text
|
|
95
|
+
#: being compared is different in kind -- short, structured, and repetitive
|
|
96
|
+
#: rather than a written sentence -- so the distances it produces sit on a
|
|
97
|
+
#: tighter scale. Measured stable anywhere from 0.35 to 0.50.
|
|
98
|
+
undescribed_threshold: float = 0.45
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
class RunnerConfig(BaseModel):
|
|
102
|
+
"""How to invoke the execution engine.
|
|
103
|
+
|
|
104
|
+
``command`` is an argument list, never a string: it is passed straight to
|
|
105
|
+
the process without a shell, so nothing in a trace can be interpreted as a
|
|
106
|
+
shell metacharacter.
|
|
107
|
+
"""
|
|
108
|
+
|
|
109
|
+
command: list[str] = Field(default_factory=lambda: ["npx", "--yes", "promptfoo@0.122.2"])
|
|
110
|
+
timeout_seconds: int = 900
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class ProjectConfig(BaseModel):
|
|
114
|
+
"""The contents of ``evalkeep.yaml``."""
|
|
115
|
+
|
|
116
|
+
version: int = CONFIG_VERSION
|
|
117
|
+
project_name: str = "evalkeep-project"
|
|
118
|
+
state_dir: str = STATE_DIRNAME
|
|
119
|
+
redaction: RedactionConfig = Field(default_factory=RedactionConfig)
|
|
120
|
+
analyzer: AnalyzerConfig = Field(default_factory=AnalyzerConfig)
|
|
121
|
+
clustering: ClusteringConfig = Field(default_factory=ClusteringConfig)
|
|
122
|
+
runner: RunnerConfig = Field(default_factory=RunnerConfig)
|
|
123
|
+
|
|
124
|
+
def to_yaml(self) -> str:
|
|
125
|
+
return yaml.safe_dump(
|
|
126
|
+
self.model_dump(mode="json"), sort_keys=False, default_flow_style=False
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
@classmethod
|
|
130
|
+
def from_yaml(cls, text: str, *, source: Path | None = None) -> ProjectConfig:
|
|
131
|
+
where = f" in {source}" if source is not None else ""
|
|
132
|
+
try:
|
|
133
|
+
raw: Any = yaml.safe_load(text)
|
|
134
|
+
except yaml.YAMLError as exc:
|
|
135
|
+
raise CommandError(f"Could not parse YAML{where}: {exc}") from exc
|
|
136
|
+
if raw is None:
|
|
137
|
+
raw = {}
|
|
138
|
+
if not isinstance(raw, dict):
|
|
139
|
+
raise CommandError(f"Expected a YAML mapping{where}, got {type(raw).__name__}.")
|
|
140
|
+
try:
|
|
141
|
+
return cls.model_validate(raw)
|
|
142
|
+
except ValidationError as exc:
|
|
143
|
+
raise CommandError(f"Invalid configuration{where}:\n{exc}") from exc
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
class Project:
|
|
147
|
+
"""Resolved paths for one Evalkeep project rooted at ``root``."""
|
|
148
|
+
|
|
149
|
+
def __init__(self, root: Path, config: ProjectConfig) -> None:
|
|
150
|
+
self.root = root
|
|
151
|
+
self.config = config
|
|
152
|
+
|
|
153
|
+
@property
|
|
154
|
+
def config_path(self) -> Path:
|
|
155
|
+
return self.root / CONFIG_FILENAME
|
|
156
|
+
|
|
157
|
+
@property
|
|
158
|
+
def state_dir(self) -> Path:
|
|
159
|
+
return self.root / self.config.state_dir
|
|
160
|
+
|
|
161
|
+
@property
|
|
162
|
+
def database_path(self) -> Path:
|
|
163
|
+
return self.state_dir / "database.db"
|
|
164
|
+
|
|
165
|
+
@property
|
|
166
|
+
def salt_path(self) -> Path:
|
|
167
|
+
return self.state_dir / SALT_FILENAME
|
|
168
|
+
|
|
169
|
+
def pseudonymizer(self) -> Pseudonymizer | None:
|
|
170
|
+
"""The project's pseudonymizer, or ``None`` when the feature is off."""
|
|
171
|
+
if not self.config.redaction.pseudonymize_identifiers:
|
|
172
|
+
return None
|
|
173
|
+
return Pseudonymizer.load(self.salt_path)
|
|
174
|
+
|
|
175
|
+
def identify(self, value: str) -> list[str]:
|
|
176
|
+
"""Every stored ID a user-supplied identifier could mean.
|
|
177
|
+
|
|
178
|
+
With pseudonymization on, someone will sometimes paste an ID from their
|
|
179
|
+
own systems and sometimes one Evalkeep printed. Both should work, and
|
|
180
|
+
neither requires storing the original.
|
|
181
|
+
"""
|
|
182
|
+
cleaned = value.strip()
|
|
183
|
+
pseudonymizer = self.pseudonymizer()
|
|
184
|
+
if pseudonymizer is None:
|
|
185
|
+
return [cleaned]
|
|
186
|
+
return [cleaned, pseudonymizer.token(cleaned, field="trace_id")]
|
|
187
|
+
|
|
188
|
+
def subdir(self, name: str) -> Path:
|
|
189
|
+
return self.state_dir / name
|
|
190
|
+
|
|
191
|
+
@classmethod
|
|
192
|
+
def load(cls, root: Path) -> Project:
|
|
193
|
+
"""Load an initialized project, or explain how to create one."""
|
|
194
|
+
config_path = root / CONFIG_FILENAME
|
|
195
|
+
if not config_path.is_file():
|
|
196
|
+
raise CommandError(
|
|
197
|
+
f"No {CONFIG_FILENAME} found in {root}.",
|
|
198
|
+
hint="Run 'evalkeep init' first.",
|
|
199
|
+
)
|
|
200
|
+
config = ProjectConfig.from_yaml(
|
|
201
|
+
config_path.read_text(encoding="utf-8"), source=config_path
|
|
202
|
+
)
|
|
203
|
+
if config.version > CONFIG_VERSION:
|
|
204
|
+
raise CommandError(
|
|
205
|
+
f"{config_path} was written by a newer Evalkeep "
|
|
206
|
+
f"(config version {config.version}, this build understands {CONFIG_VERSION}).",
|
|
207
|
+
hint="Upgrade evalkeep.",
|
|
208
|
+
)
|
|
209
|
+
return cls(root, config)
|