evalkeep 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. evalkeep/__init__.py +12 -0
  2. evalkeep/__main__.py +6 -0
  3. evalkeep/adapters/__init__.py +45 -0
  4. evalkeep/adapters/base.py +92 -0
  5. evalkeep/adapters/jsonl.py +164 -0
  6. evalkeep/adapters/langsmith.py +436 -0
  7. evalkeep/adapters/otlp.py +442 -0
  8. evalkeep/adapters/semconv.py +208 -0
  9. evalkeep/analysis.py +174 -0
  10. evalkeep/analysis_run.py +160 -0
  11. evalkeep/analyzers/__init__.py +52 -0
  12. evalkeep/analyzers/anthropic.py +145 -0
  13. evalkeep/analyzers/stub.py +34 -0
  14. evalkeep/cache.py +122 -0
  15. evalkeep/cli.py +1933 -0
  16. evalkeep/clustering.py +383 -0
  17. evalkeep/clusters.py +101 -0
  18. evalkeep/commands/__init__.py +1 -0
  19. evalkeep/commands/analyze_cmd.py +100 -0
  20. evalkeep/commands/compare_cmd.py +169 -0
  21. evalkeep/commands/dataset_cmd.py +182 -0
  22. evalkeep/commands/detect_cmd.py +154 -0
  23. evalkeep/commands/discover_cmd.py +274 -0
  24. evalkeep/commands/ingest_cmd.py +50 -0
  25. evalkeep/commands/init_cmd.py +151 -0
  26. evalkeep/commands/pipeline_cmd.py +156 -0
  27. evalkeep/commands/review_cmd.py +141 -0
  28. evalkeep/commands/run_cmd.py +131 -0
  29. evalkeep/commands/target_cmd.py +109 -0
  30. evalkeep/commands/trace_cmd.py +58 -0
  31. evalkeep/comparison.py +432 -0
  32. evalkeep/config.py +209 -0
  33. evalkeep/detection.py +94 -0
  34. evalkeep/detectors.py +182 -0
  35. evalkeep/discovery.py +208 -0
  36. evalkeep/embeddings/__init__.py +31 -0
  37. evalkeep/embeddings/base.py +32 -0
  38. evalkeep/embeddings/hashing.py +98 -0
  39. evalkeep/errors.py +42 -0
  40. evalkeep/examples/__init__.py +37 -0
  41. evalkeep/examples/langsmith/runs.jsonl +18 -0
  42. evalkeep/examples/opentelemetry/spans.json +898 -0
  43. evalkeep/examples/refund-agent/agents/baseline.py +66 -0
  44. evalkeep/examples/refund-agent/agents/candidate.py +66 -0
  45. evalkeep/examples/refund-agent/traces.jsonl +5 -0
  46. evalkeep/examples/tau-bench/prepare.py +230 -0
  47. evalkeep/exporters/__init__.py +45 -0
  48. evalkeep/exporters/generic.py +31 -0
  49. evalkeep/exporters/promptfoo.py +219 -0
  50. evalkeep/failures.py +95 -0
  51. evalkeep/generation.py +303 -0
  52. evalkeep/hashing.py +56 -0
  53. evalkeep/ingest.py +257 -0
  54. evalkeep/prompts.py +127 -0
  55. evalkeep/pseudonyms.py +82 -0
  56. evalkeep/py.typed +0 -0
  57. evalkeep/redaction.py +333 -0
  58. evalkeep/regression.py +409 -0
  59. evalkeep/review.py +309 -0
  60. evalkeep/runner.py +302 -0
  61. evalkeep/runs.py +185 -0
  62. evalkeep/storage/__init__.py +37 -0
  63. evalkeep/storage/clusters.py +163 -0
  64. evalkeep/storage/failures.py +254 -0
  65. evalkeep/storage/migrations.py +370 -0
  66. evalkeep/storage/regression.py +136 -0
  67. evalkeep/storage/runs.py +223 -0
  68. evalkeep/storage/store.py +429 -0
  69. evalkeep/targets.py +205 -0
  70. evalkeep/trace.py +238 -0
  71. evalkeep-0.1.0.dist-info/METADATA +221 -0
  72. evalkeep-0.1.0.dist-info/RECORD +75 -0
  73. evalkeep-0.1.0.dist-info/WHEEL +4 -0
  74. evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
  75. evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/comparison.py ADDED
@@ -0,0 +1,432 @@
1
+ """Comparing two runs, and saying only what the numbers support.
2
+
3
+ The whole pipeline exists to answer one question -- did this change make the
4
+ agent better or worse -- and this is where that answer is produced. Three rules
5
+ shape it, all of them about not overclaiming:
6
+
7
+ * **A test that errored is not a data point.** An error says the harness or the
8
+ target broke, not that the agent got the answer wrong. Errored pairs are
9
+ excluded from every count and reported separately, because letting an outage
10
+ read as a regression is the single most damaging mistake this tool could make.
11
+ * **Two runs are only comparable if they answered the same questions.** Runs
12
+ carry a suite hash; comparing across different suites is refused unless the
13
+ caller explicitly asks for the intersection.
14
+ * **A confidence interval is only reported when it means something.** With a
15
+ handful of discordant pairs the normal approximation is not trustworthy, so
16
+ the interval is withheld and the reason is printed instead of a number that
17
+ would look authoritative and be wrong.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ from dataclasses import dataclass, field
23
+ from enum import StrEnum
24
+ from math import comb
25
+
26
+ from evalkeep.runs import (
27
+ CaseResult,
28
+ CaseSummary,
29
+ EvaluationRun,
30
+ Outcome,
31
+ Verdict,
32
+ summarize,
33
+ )
34
+
35
+ #: Below this many discordant pairs the normal approximation behind the interval
36
+ #: is not trustworthy, so no interval is reported. A common rule of thumb, and
37
+ #: chosen here because being silent is better than being confidently wrong.
38
+ MIN_DISCORDANT_FOR_INTERVAL = 10
39
+
40
+ #: 95% two-sided normal quantile.
41
+ _Z = 1.959963984540054
42
+
43
+
44
+ class Classification(StrEnum):
45
+ """Guide 9.1's truth table, with its two error rows made explicit."""
46
+
47
+ UNCHANGED_PASS = "unchanged_pass"
48
+ FIXED = "fixed"
49
+ #: Improved and now passes most of the time, but not every time. Only
50
+ #: reachable with repetitions -- a single execution cannot tell the
51
+ #: difference between this and `fixed`, which is the whole reason to repeat.
52
+ LIKELY_FIXED = "likely_fixed"
53
+ REGRESSION = "regression"
54
+ UNCHANGED_FAILURE = "unchanged_failure"
55
+ #: One side never ran. Excluded from the counts, reported on its own.
56
+ NOT_COMPARABLE = "not_comparable"
57
+ #: Present in one run and absent from the other.
58
+ MISSING = "missing"
59
+
60
+
61
+ #: The four classifications that say something about the agent.
62
+ COMPARABLE = (
63
+ Classification.UNCHANGED_PASS,
64
+ Classification.FIXED,
65
+ Classification.LIKELY_FIXED,
66
+ Classification.REGRESSION,
67
+ Classification.UNCHANGED_FAILURE,
68
+ )
69
+
70
+ #: Classifications that count as the candidate passing, for the paired test.
71
+ #: A case that only sometimes passes is not counted as passing: the point of
72
+ #: repeating was to stop calling that a fix.
73
+ _PASSING = (Classification.UNCHANGED_PASS, Classification.FIXED)
74
+
75
+
76
+ @dataclass(frozen=True)
77
+ class CaseComparison:
78
+ test_id: str
79
+ classification: Classification
80
+ baseline: CaseResult | None = None
81
+ candidate: CaseResult | None = None
82
+ baseline_summary: CaseSummary | None = None
83
+ candidate_summary: CaseSummary | None = None
84
+
85
+ @property
86
+ def flaky(self) -> bool:
87
+ """Whether either side was inconsistent across its repetitions.
88
+
89
+ Orthogonal to the classification: a case can be both a regression and
90
+ flaky, and hiding one behind the other would lose a real finding.
91
+ """
92
+ return any(
93
+ summary is not None and summary.flaky
94
+ for summary in (self.baseline_summary, self.candidate_summary)
95
+ )
96
+
97
+ @property
98
+ def confidence(self) -> tuple[float, float] | None:
99
+ """How reliably the candidate passes this case, if it was repeated."""
100
+ if self.candidate_summary is None:
101
+ return None
102
+ return self.candidate_summary.confidence
103
+
104
+ @property
105
+ def reason(self) -> str:
106
+ """Why this pair is not comparable, when it is not."""
107
+ if self.classification is Classification.MISSING:
108
+ side = "candidate" if self.baseline_summary is not None else "baseline"
109
+ return f"absent from the {side} run"
110
+ for label, summary, result in (
111
+ ("baseline", self.baseline_summary, self.baseline),
112
+ ("candidate", self.candidate_summary, self.candidate),
113
+ ):
114
+ if summary is not None and summary.verdict is Verdict.ERROR:
115
+ kind = (
116
+ result.error_kind.value if result is not None and result.error_kind else "error"
117
+ )
118
+ return f"{label} {kind}"
119
+ return ""
120
+
121
+ @property
122
+ def rates(self) -> str:
123
+ """How often each side passed, when either was repeated."""
124
+ parts = []
125
+ for label, summary in (
126
+ ("before", self.baseline_summary),
127
+ ("after", self.candidate_summary),
128
+ ):
129
+ if summary is not None and summary.repetitions > 1:
130
+ parts.append(f"{label} {summary.passed}/{summary.evaluated}")
131
+ return ", ".join(parts)
132
+
133
+
134
+ @dataclass
135
+ class PairedStatistics:
136
+ """Paired analysis over the tests both runs actually evaluated."""
137
+
138
+ pairs: int
139
+ fixed: int
140
+ regressions: int
141
+ difference: float
142
+ p_value: float
143
+ interval: tuple[float, float] | None = None
144
+ interval_method: str | None = None
145
+ #: Why an interval was withheld, when it was.
146
+ note: str | None = None
147
+
148
+ @property
149
+ def discordant(self) -> int:
150
+ """Pairs where the two runs disagreed. All the information is here."""
151
+ return self.fixed + self.regressions
152
+
153
+ @property
154
+ def significant(self) -> bool:
155
+ return self.p_value < 0.05
156
+
157
+
158
+ @dataclass
159
+ class ComparisonReport:
160
+ baseline_run: EvaluationRun
161
+ candidate_run: EvaluationRun
162
+ comparisons: list[CaseComparison] = field(default_factory=list)
163
+ suite_compatible: bool = True
164
+
165
+ @property
166
+ def counts(self) -> dict[Classification, int]:
167
+ tally: dict[Classification, int] = {}
168
+ for comparison in self.comparisons:
169
+ tally[comparison.classification] = tally.get(comparison.classification, 0) + 1
170
+ return tally
171
+
172
+ @property
173
+ def comparable(self) -> list[CaseComparison]:
174
+ return [c for c in self.comparisons if c.classification in COMPARABLE]
175
+
176
+ @property
177
+ def excluded(self) -> list[CaseComparison]:
178
+ return [c for c in self.comparisons if c.classification not in COMPARABLE]
179
+
180
+ @property
181
+ def regressions(self) -> list[CaseComparison]:
182
+ return [c for c in self.comparisons if c.classification is Classification.REGRESSION]
183
+
184
+ @property
185
+ def fixes(self) -> list[CaseComparison]:
186
+ return [c for c in self.comparisons if c.classification is Classification.FIXED]
187
+
188
+ @property
189
+ def baseline_pass_rate(self) -> float | None:
190
+ return _rate(_passing(self.comparable, before=True), len(self.comparable))
191
+
192
+ @property
193
+ def candidate_pass_rate(self) -> float | None:
194
+ """The share of cases that pass *reliably*.
195
+
196
+ A case that passes nine times in ten is not counted as passing here.
197
+ Counting it would reintroduce exactly the overclaim that repeating the
198
+ run was meant to remove.
199
+ """
200
+ return _rate(_passing(self.comparable, before=False), len(self.comparable))
201
+
202
+ @property
203
+ def flaky(self) -> list[CaseComparison]:
204
+ return [c for c in self.comparisons if c.flaky]
205
+
206
+ @property
207
+ def repeated(self) -> bool:
208
+ return max(self.baseline_run.repetitions, self.candidate_run.repetitions) > 1
209
+
210
+ @property
211
+ def statistics(self) -> PairedStatistics | None:
212
+ return paired_statistics(self.comparable)
213
+
214
+
215
+ def compare_results(
216
+ baseline_run: EvaluationRun,
217
+ baseline_results: list[CaseResult],
218
+ candidate_run: EvaluationRun,
219
+ candidate_results: list[CaseResult],
220
+ ) -> ComparisonReport:
221
+ """Align two runs by stable test ID and classify every case."""
222
+ baseline_summaries = summarize(baseline_results)
223
+ candidate_summaries = summarize(candidate_results)
224
+ baseline_first = _first_by_case(baseline_results)
225
+ candidate_first = _first_by_case(candidate_results)
226
+
227
+ comparisons = [
228
+ _classify(
229
+ test_id,
230
+ baseline_summaries.get(test_id),
231
+ candidate_summaries.get(test_id),
232
+ baseline_first.get(test_id),
233
+ candidate_first.get(test_id),
234
+ )
235
+ for test_id in sorted(set(baseline_summaries) | set(candidate_summaries))
236
+ ]
237
+ return ComparisonReport(
238
+ baseline_run=baseline_run,
239
+ candidate_run=candidate_run,
240
+ comparisons=comparisons,
241
+ suite_compatible=baseline_run.suite_hash == candidate_run.suite_hash,
242
+ )
243
+
244
+
245
+ def _first_by_case(results: list[CaseResult]) -> dict[str, CaseResult]:
246
+ """One representative execution per case, for showing a failure reason."""
247
+ first: dict[str, CaseResult] = {}
248
+ for result in results:
249
+ first.setdefault(result.test_id, result)
250
+ if result.outcome is Outcome.FAIL:
251
+ first[result.test_id] = result
252
+ return first
253
+
254
+
255
+ def _classify(
256
+ test_id: str,
257
+ baseline: CaseSummary | None,
258
+ candidate: CaseSummary | None,
259
+ baseline_result: CaseResult | None,
260
+ candidate_result: CaseResult | None,
261
+ ) -> CaseComparison:
262
+ """Extend the truth table from single outcomes to repeated ones.
263
+
264
+ With one repetition each side is either PASS or FAIL and this reduces
265
+ exactly to the original four rows. With more, a third verdict appears --
266
+ FLAKY -- and it is what stops a single lucky pass being reported as a fix.
267
+ """
268
+ if baseline is None or candidate is None:
269
+ classification = Classification.MISSING
270
+ elif baseline.verdict is Verdict.ERROR or candidate.verdict is Verdict.ERROR:
271
+ classification = Classification.NOT_COMPARABLE
272
+ else:
273
+ classification = _direction(baseline, candidate)
274
+
275
+ return CaseComparison(
276
+ test_id=test_id,
277
+ classification=classification,
278
+ baseline=baseline_result,
279
+ candidate=candidate_result,
280
+ baseline_summary=baseline,
281
+ candidate_summary=candidate,
282
+ )
283
+
284
+
285
+ def _direction(baseline: CaseSummary, candidate: CaseSummary) -> Classification:
286
+ before, after = baseline.verdict, candidate.verdict
287
+
288
+ if after is Verdict.PASS:
289
+ # Reliably passing now. It is a fix unless it was already reliable.
290
+ return Classification.UNCHANGED_PASS if before is Verdict.PASS else Classification.FIXED
291
+
292
+ if after is Verdict.FAIL:
293
+ # Never passes now. A regression only if it used to pass at all.
294
+ return (
295
+ Classification.UNCHANGED_FAILURE
296
+ if before is Verdict.FAIL
297
+ else Classification.REGRESSION
298
+ )
299
+
300
+ # The candidate is flaky. Whether that is progress depends on what it was.
301
+ if before is Verdict.PASS:
302
+ # It used to always pass and now sometimes does not. That is worse,
303
+ # whatever the rate says.
304
+ return Classification.REGRESSION
305
+ if before is Verdict.FAIL:
306
+ return (
307
+ Classification.LIKELY_FIXED
308
+ if _passes_more_often_than_not(candidate)
309
+ else Classification.UNCHANGED_FAILURE
310
+ )
311
+
312
+ # Flaky before and flaky after: compare how often, not whether.
313
+ before_rate = baseline.pass_rate or 0.0
314
+ after_rate = candidate.pass_rate or 0.0
315
+ if after_rate > before_rate:
316
+ return (
317
+ Classification.LIKELY_FIXED
318
+ if _passes_more_often_than_not(candidate)
319
+ else Classification.UNCHANGED_FAILURE
320
+ )
321
+ if after_rate < before_rate:
322
+ return Classification.REGRESSION
323
+ return Classification.UNCHANGED_FAILURE
324
+
325
+
326
+ def _passes_more_often_than_not(summary: CaseSummary) -> bool:
327
+ """True only when the sample supports the claim, not merely suggests it.
328
+
329
+ Two passes out of three looks like a majority and is not evidence of one;
330
+ the lower bound of the interval is what decides.
331
+ """
332
+ interval = summary.confidence
333
+ return interval is not None and interval[0] > 0.5
334
+
335
+
336
+ def paired_statistics(comparable: list[CaseComparison]) -> PairedStatistics | None:
337
+ """McNemar's exact test over the pairs, with an interval only when earned.
338
+
339
+ The test is exact rather than the chi-square approximation: suites here are
340
+ often small, and the approximation is unreliable exactly where these suites
341
+ live. Only discordant pairs carry information -- a test both runs passed
342
+ says nothing about whether anything changed -- so the test is a two-sided
343
+ binomial on fixes versus regressions.
344
+ """
345
+ pairs = len(comparable)
346
+ if pairs == 0:
347
+ return None
348
+
349
+ fixed = sum(1 for c in comparable if c.classification is Classification.FIXED)
350
+ regressions = sum(1 for c in comparable if c.classification is Classification.REGRESSION)
351
+ difference = (fixed - regressions) / pairs
352
+
353
+ discordant = fixed + regressions
354
+ if discordant == 0:
355
+ # Nothing changed on any test. There is no evidence of a difference,
356
+ # which is not the same as evidence of no difference.
357
+ return PairedStatistics(
358
+ pairs=pairs,
359
+ fixed=0,
360
+ regressions=0,
361
+ difference=0.0,
362
+ p_value=1.0,
363
+ note="No test changed outcome, so there is nothing to test.",
364
+ )
365
+
366
+ p_value = exact_binomial_two_sided(fixed, discordant)
367
+
368
+ statistics = PairedStatistics(
369
+ pairs=pairs,
370
+ fixed=fixed,
371
+ regressions=regressions,
372
+ difference=difference,
373
+ p_value=p_value,
374
+ )
375
+
376
+ if discordant < MIN_DISCORDANT_FOR_INTERVAL:
377
+ statistics.note = (
378
+ f"Only {discordant} test(s) changed outcome; that is too few for a "
379
+ "trustworthy interval, so none is given."
380
+ )
381
+ return statistics
382
+
383
+ # Wald interval for the paired difference in proportions.
384
+ variance = (fixed + regressions - (fixed - regressions) ** 2 / pairs) / pairs**2
385
+ margin = _Z * (variance**0.5)
386
+ statistics.interval = (
387
+ max(-1.0, difference - margin),
388
+ min(1.0, difference + margin),
389
+ )
390
+ statistics.interval_method = "paired Wald, 95%"
391
+ return statistics
392
+
393
+
394
+ def exact_binomial_two_sided(successes: int, trials: int) -> float:
395
+ """The exact two-sided binomial p-value at p = 0.5.
396
+
397
+ This is McNemar's exact test: under the null, each discordant pair is a coin
398
+ flip, so the question is how surprising this split would be. At p = 0.5 the
399
+ distribution is symmetric, which makes the two-sided value simply both tails
400
+ of the more extreme side -- no approximation, and no reason to carry SciPy's
401
+ 82 MB for one call. Checked against `scipy.stats.binomtest` for every split
402
+ up to sixty trials before that dependency was removed; the largest
403
+ disagreement was 5.6e-16, which is floating-point noise.
404
+ """
405
+ if trials <= 0:
406
+ return 1.0
407
+ smaller = min(successes, trials - successes)
408
+ tail = sum(comb(trials, k) for k in range(smaller + 1)) / 2**trials
409
+ return float(min(1.0, 2 * tail))
410
+
411
+
412
+ def _reliably_passing(summary: CaseSummary | None) -> bool:
413
+ return summary is not None and summary.verdict is Verdict.PASS
414
+
415
+
416
+ def _passing(comparisons: list[CaseComparison], *, before: bool) -> int:
417
+ return sum(
418
+ 1
419
+ for c in comparisons
420
+ if _reliably_passing(c.baseline_summary if before else c.candidate_summary)
421
+ )
422
+
423
+
424
+ def _became(comparison: CaseComparison, *, passing: bool) -> bool:
425
+ """Whether this case crossed the pass/not-pass line in the given direction."""
426
+ was = _reliably_passing(comparison.baseline_summary)
427
+ now = _reliably_passing(comparison.candidate_summary)
428
+ return (now and not was) if passing else (was and not now)
429
+
430
+
431
+ def _rate(passed: int, total: int) -> float | None:
432
+ return passed / total if total else None
evalkeep/config.py ADDED
@@ -0,0 +1,209 @@
1
+ """Project configuration and the on-disk layout created by ``evalkeep init``.
2
+
3
+ Everything Evalkeep writes lives under a single state directory
4
+ (``.evalkeep/`` by default) so that a project can be inspected, backed up or
5
+ deleted in one step. Only ``evalkeep.yaml`` is meant to be committed; the
6
+ state directory holds raw traces, the database, caches and run outputs, all of
7
+ which stay out of Git (see the generated ``.gitignore`` entries).
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from pathlib import Path
13
+ from typing import Any
14
+
15
+ import yaml
16
+ from pydantic import BaseModel, Field, ValidationError
17
+
18
+ from evalkeep.errors import CommandError
19
+ from evalkeep.pseudonyms import SALT_FILENAME, Pseudonymizer
20
+
21
+ CONFIG_FILENAME = "evalkeep.yaml"
22
+ STATE_DIRNAME = ".evalkeep"
23
+
24
+ #: Bumped when the on-disk layout changes in a way that needs a migration.
25
+ CONFIG_VERSION = 1
26
+
27
+ #: Subdirectories of the state directory, with the purpose of each.
28
+ STATE_SUBDIRS: dict[str, str] = {
29
+ "data": "Redacted trace payloads and intermediate artifacts.",
30
+ "cache": "Analyzer and embedding caches, keyed by content hash.",
31
+ "runs": "Raw Promptfoo run outputs, one directory per run.",
32
+ "exports": "Generated export files (generic JSONL, Promptfoo YAML).",
33
+ }
34
+
35
+ #: Paths excluded from Git. Approved tests and config are committed; traces,
36
+ #: the database, caches and run outputs are not.
37
+ GITIGNORE_ENTRIES: tuple[str, ...] = (
38
+ ".env",
39
+ f"{STATE_DIRNAME}/database.db",
40
+ # The salt is what makes pseudonyms unguessable.
41
+ f"{STATE_DIRNAME}/salt",
42
+ f"{STATE_DIRNAME}/data/",
43
+ f"{STATE_DIRNAME}/cache/",
44
+ f"{STATE_DIRNAME}/runs/",
45
+ )
46
+
47
+ GITIGNORE_HEADER = "# evalkeep"
48
+
49
+
50
+ class RedactionConfig(BaseModel):
51
+ """Which built-in redactors run before anything is written to storage."""
52
+
53
+ emails: bool = True
54
+ phone_numbers: bool = True
55
+ payment_cards: bool = True
56
+ token_prefixes: bool = True
57
+ secret_field_names: bool = True
58
+ #: Replace trace, event and call IDs with per-project tokens. Off by
59
+ #: default because it changes the IDs you see; turn it on when your own
60
+ #: identifiers embed customer data. Lookups keep accepting the originals.
61
+ pseudonymize_identifiers: bool = False
62
+
63
+
64
+ class AnalyzerConfig(BaseModel):
65
+ """Which provider describes failures, if any.
66
+
67
+ ``manual`` is the default: Evalkeep runs fully offline and failures are
68
+ labelled by hand until a provider is configured.
69
+ """
70
+
71
+ provider: str = "manual"
72
+ model: str = "claude-opus-5"
73
+ effort: str = "medium"
74
+ max_tokens: int = 16000
75
+
76
+
77
+ class ClusteringConfig(BaseModel):
78
+ """How failures are embedded and grouped.
79
+
80
+ Every field is stored with the clustering run that used it, so a grouping
81
+ can always be reproduced or explained.
82
+ """
83
+
84
+ embedder: str = "hashing"
85
+ dimensions: int = 512
86
+ #: Recorded and applied even where the algorithm is deterministic, so that
87
+ #: swapping in a randomized one later cannot quietly break reproducibility.
88
+ seed: int = 0
89
+ algorithm: str = "agglomerative"
90
+ metric: str = "cosine"
91
+ linkage: str = "average"
92
+ #: Cosine distance above which two failures are different families.
93
+ threshold: float = 0.55
94
+ #: The same, for failures nobody has described yet. Lower because the text
95
+ #: being compared is different in kind -- short, structured, and repetitive
96
+ #: rather than a written sentence -- so the distances it produces sit on a
97
+ #: tighter scale. Measured stable anywhere from 0.35 to 0.50.
98
+ undescribed_threshold: float = 0.45
99
+
100
+
101
+ class RunnerConfig(BaseModel):
102
+ """How to invoke the execution engine.
103
+
104
+ ``command`` is an argument list, never a string: it is passed straight to
105
+ the process without a shell, so nothing in a trace can be interpreted as a
106
+ shell metacharacter.
107
+ """
108
+
109
+ command: list[str] = Field(default_factory=lambda: ["npx", "--yes", "promptfoo@0.122.2"])
110
+ timeout_seconds: int = 900
111
+
112
+
113
+ class ProjectConfig(BaseModel):
114
+ """The contents of ``evalkeep.yaml``."""
115
+
116
+ version: int = CONFIG_VERSION
117
+ project_name: str = "evalkeep-project"
118
+ state_dir: str = STATE_DIRNAME
119
+ redaction: RedactionConfig = Field(default_factory=RedactionConfig)
120
+ analyzer: AnalyzerConfig = Field(default_factory=AnalyzerConfig)
121
+ clustering: ClusteringConfig = Field(default_factory=ClusteringConfig)
122
+ runner: RunnerConfig = Field(default_factory=RunnerConfig)
123
+
124
+ def to_yaml(self) -> str:
125
+ return yaml.safe_dump(
126
+ self.model_dump(mode="json"), sort_keys=False, default_flow_style=False
127
+ )
128
+
129
+ @classmethod
130
+ def from_yaml(cls, text: str, *, source: Path | None = None) -> ProjectConfig:
131
+ where = f" in {source}" if source is not None else ""
132
+ try:
133
+ raw: Any = yaml.safe_load(text)
134
+ except yaml.YAMLError as exc:
135
+ raise CommandError(f"Could not parse YAML{where}: {exc}") from exc
136
+ if raw is None:
137
+ raw = {}
138
+ if not isinstance(raw, dict):
139
+ raise CommandError(f"Expected a YAML mapping{where}, got {type(raw).__name__}.")
140
+ try:
141
+ return cls.model_validate(raw)
142
+ except ValidationError as exc:
143
+ raise CommandError(f"Invalid configuration{where}:\n{exc}") from exc
144
+
145
+
146
+ class Project:
147
+ """Resolved paths for one Evalkeep project rooted at ``root``."""
148
+
149
+ def __init__(self, root: Path, config: ProjectConfig) -> None:
150
+ self.root = root
151
+ self.config = config
152
+
153
+ @property
154
+ def config_path(self) -> Path:
155
+ return self.root / CONFIG_FILENAME
156
+
157
+ @property
158
+ def state_dir(self) -> Path:
159
+ return self.root / self.config.state_dir
160
+
161
+ @property
162
+ def database_path(self) -> Path:
163
+ return self.state_dir / "database.db"
164
+
165
+ @property
166
+ def salt_path(self) -> Path:
167
+ return self.state_dir / SALT_FILENAME
168
+
169
+ def pseudonymizer(self) -> Pseudonymizer | None:
170
+ """The project's pseudonymizer, or ``None`` when the feature is off."""
171
+ if not self.config.redaction.pseudonymize_identifiers:
172
+ return None
173
+ return Pseudonymizer.load(self.salt_path)
174
+
175
+ def identify(self, value: str) -> list[str]:
176
+ """Every stored ID a user-supplied identifier could mean.
177
+
178
+ With pseudonymization on, someone will sometimes paste an ID from their
179
+ own systems and sometimes one Evalkeep printed. Both should work, and
180
+ neither requires storing the original.
181
+ """
182
+ cleaned = value.strip()
183
+ pseudonymizer = self.pseudonymizer()
184
+ if pseudonymizer is None:
185
+ return [cleaned]
186
+ return [cleaned, pseudonymizer.token(cleaned, field="trace_id")]
187
+
188
+ def subdir(self, name: str) -> Path:
189
+ return self.state_dir / name
190
+
191
+ @classmethod
192
+ def load(cls, root: Path) -> Project:
193
+ """Load an initialized project, or explain how to create one."""
194
+ config_path = root / CONFIG_FILENAME
195
+ if not config_path.is_file():
196
+ raise CommandError(
197
+ f"No {CONFIG_FILENAME} found in {root}.",
198
+ hint="Run 'evalkeep init' first.",
199
+ )
200
+ config = ProjectConfig.from_yaml(
201
+ config_path.read_text(encoding="utf-8"), source=config_path
202
+ )
203
+ if config.version > CONFIG_VERSION:
204
+ raise CommandError(
205
+ f"{config_path} was written by a newer Evalkeep "
206
+ f"(config version {config.version}, this build understands {CONFIG_VERSION}).",
207
+ hint="Upgrade evalkeep.",
208
+ )
209
+ return cls(root, config)