evalkeep 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. evalkeep/__init__.py +12 -0
  2. evalkeep/__main__.py +6 -0
  3. evalkeep/adapters/__init__.py +45 -0
  4. evalkeep/adapters/base.py +92 -0
  5. evalkeep/adapters/jsonl.py +164 -0
  6. evalkeep/adapters/langsmith.py +436 -0
  7. evalkeep/adapters/otlp.py +442 -0
  8. evalkeep/adapters/semconv.py +208 -0
  9. evalkeep/analysis.py +174 -0
  10. evalkeep/analysis_run.py +160 -0
  11. evalkeep/analyzers/__init__.py +52 -0
  12. evalkeep/analyzers/anthropic.py +145 -0
  13. evalkeep/analyzers/stub.py +34 -0
  14. evalkeep/cache.py +122 -0
  15. evalkeep/cli.py +1933 -0
  16. evalkeep/clustering.py +383 -0
  17. evalkeep/clusters.py +101 -0
  18. evalkeep/commands/__init__.py +1 -0
  19. evalkeep/commands/analyze_cmd.py +100 -0
  20. evalkeep/commands/compare_cmd.py +169 -0
  21. evalkeep/commands/dataset_cmd.py +182 -0
  22. evalkeep/commands/detect_cmd.py +154 -0
  23. evalkeep/commands/discover_cmd.py +274 -0
  24. evalkeep/commands/ingest_cmd.py +50 -0
  25. evalkeep/commands/init_cmd.py +151 -0
  26. evalkeep/commands/pipeline_cmd.py +156 -0
  27. evalkeep/commands/review_cmd.py +141 -0
  28. evalkeep/commands/run_cmd.py +131 -0
  29. evalkeep/commands/target_cmd.py +109 -0
  30. evalkeep/commands/trace_cmd.py +58 -0
  31. evalkeep/comparison.py +432 -0
  32. evalkeep/config.py +209 -0
  33. evalkeep/detection.py +94 -0
  34. evalkeep/detectors.py +182 -0
  35. evalkeep/discovery.py +208 -0
  36. evalkeep/embeddings/__init__.py +31 -0
  37. evalkeep/embeddings/base.py +32 -0
  38. evalkeep/embeddings/hashing.py +98 -0
  39. evalkeep/errors.py +42 -0
  40. evalkeep/examples/__init__.py +37 -0
  41. evalkeep/examples/langsmith/runs.jsonl +18 -0
  42. evalkeep/examples/opentelemetry/spans.json +898 -0
  43. evalkeep/examples/refund-agent/agents/baseline.py +66 -0
  44. evalkeep/examples/refund-agent/agents/candidate.py +66 -0
  45. evalkeep/examples/refund-agent/traces.jsonl +5 -0
  46. evalkeep/examples/tau-bench/prepare.py +230 -0
  47. evalkeep/exporters/__init__.py +45 -0
  48. evalkeep/exporters/generic.py +31 -0
  49. evalkeep/exporters/promptfoo.py +219 -0
  50. evalkeep/failures.py +95 -0
  51. evalkeep/generation.py +303 -0
  52. evalkeep/hashing.py +56 -0
  53. evalkeep/ingest.py +257 -0
  54. evalkeep/prompts.py +127 -0
  55. evalkeep/pseudonyms.py +82 -0
  56. evalkeep/py.typed +0 -0
  57. evalkeep/redaction.py +333 -0
  58. evalkeep/regression.py +409 -0
  59. evalkeep/review.py +309 -0
  60. evalkeep/runner.py +302 -0
  61. evalkeep/runs.py +185 -0
  62. evalkeep/storage/__init__.py +37 -0
  63. evalkeep/storage/clusters.py +163 -0
  64. evalkeep/storage/failures.py +254 -0
  65. evalkeep/storage/migrations.py +370 -0
  66. evalkeep/storage/regression.py +136 -0
  67. evalkeep/storage/runs.py +223 -0
  68. evalkeep/storage/store.py +429 -0
  69. evalkeep/targets.py +205 -0
  70. evalkeep/trace.py +238 -0
  71. evalkeep-0.1.0.dist-info/METADATA +221 -0
  72. evalkeep-0.1.0.dist-info/RECORD +75 -0
  73. evalkeep-0.1.0.dist-info/WHEEL +4 -0
  74. evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
  75. evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/clustering.py ADDED
@@ -0,0 +1,383 @@
1
+ """Grouping failures into families, and choosing who represents each family.
2
+
3
+ The algorithm is average-linkage agglomerative clustering over cosine distance,
4
+ cut at a configured distance threshold. It was chosen for three properties that
5
+ matter more here than raw clustering quality:
6
+
7
+ * **It is deterministic.** No initialisation, no random restarts: the same
8
+ vectors and the same threshold always produce the same grouping. A seed is
9
+ still recorded with every run, so swapping in a randomized algorithm later
10
+ cannot quietly break reproducibility.
11
+ * **It does not need the number of clusters up front.** Nobody knows how many
12
+ failure families a trace file contains.
13
+ * **The threshold means something.** It is a cosine distance, so it can be
14
+ explained, tuned and written down, rather than being an opaque knob.
15
+
16
+ Average linkage rather than single linkage on purpose: single linkage chains, so
17
+ one ambiguous failure sitting between two families would merge both.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import re
23
+ from dataclasses import dataclass
24
+ from typing import Any
25
+
26
+ import numpy as np
27
+
28
+ from evalkeep.analysis import SEVERITY_ORDER, FailureAnalysis, Severity
29
+ from evalkeep.clusters import Cluster, ClusterMember, MemberRole
30
+ from evalkeep.config import ClusteringConfig
31
+ from evalkeep.detectors import SignalKind
32
+ from evalkeep.errors import CommandError
33
+
34
+
35
+ @dataclass(frozen=True)
36
+ class ClusterInput:
37
+ """One failure, as clustering sees it.
38
+
39
+ A failure is normally grouped by its *description* -- the structured
40
+ analysis someone or something wrote for it. Before anyone has described
41
+ them, it can still be grouped by what was *observed*: which tools ran and
42
+ what the evidence said. That is weaker, and the difference is tracked here
43
+ rather than hidden, so a report can say which it did.
44
+ """
45
+
46
+ failure_id: str
47
+ text: str
48
+ severity: Severity | None = None
49
+ failure_type: str | None = None
50
+ component: str | None = None
51
+ #: The tools this failure used, for naming a family nobody has described.
52
+ behaviour: str | None = None
53
+ #: False when this was grouped by observed behaviour rather than a description.
54
+ described: bool = True
55
+
56
+ @classmethod
57
+ def from_analysis(cls, failure_id: str, analysis: FailureAnalysis) -> ClusterInput:
58
+ return cls(
59
+ failure_id=failure_id,
60
+ text=cluster_text(analysis),
61
+ severity=analysis.severity,
62
+ failure_type=analysis.failure_type.value,
63
+ component=analysis.component.value,
64
+ )
65
+
66
+ @classmethod
67
+ def from_observation(
68
+ cls, failure_id: str, text: str, *, behaviour: str | None = None
69
+ ) -> ClusterInput:
70
+ return cls(failure_id=failure_id, text=text, behaviour=behaviour, described=False)
71
+
72
+
73
+ def observation_text(tools: list[str], kinds: list[str], evidence: list[str]) -> str:
74
+ """What a failure looks like before anyone has described it.
75
+
76
+ Built from what the agent *did* -- which tools ran, what kind of evidence
77
+ caught it -- and only then from the evidence's own words. Deliberately not
78
+ from the request: two customers asking the same thing in different words are
79
+ the same failure, and their phrasing would scatter them.
80
+
81
+ The tool list is repeated because in a bag-of-words representation
82
+ repetition *is* weight, and behaviour is the reliable signal here. Two
83
+ reports of one bug are worded differently while the calls the agent made
84
+ stay the same, so letting the wording dominate loses the family. Sublinear
85
+ term weighting damps the repetition, so this is a nudge rather than a
86
+ override.
87
+
88
+ Boilerplate evidence is dropped: "explicitly marked as failed" appears on
89
+ every explicit failure and so distinguishes none of them.
90
+ """
91
+ listed = ", ".join(sorted(set(tools)))
92
+ informative = [line for line in evidence if line and not _boilerplate(line)]
93
+ weighted = " ".join([listed] * 3) if listed else ""
94
+ return f"{weighted} | {' '.join(sorted(set(kinds)))} | {' '.join(informative)}".strip(" |")
95
+
96
+
97
+ _BOILERPLATE = (
98
+ "explicitly marked as failed",
99
+ "explicitly marked as errored",
100
+ "feedback was rated negative",
101
+ "reported a failure",
102
+ )
103
+
104
+
105
+ def _boilerplate(line: str) -> bool:
106
+ lowered = line.lower()
107
+ return any(phrase in lowered for phrase in _BOILERPLATE)
108
+
109
+
110
+ def cluster_text(analysis: FailureAnalysis) -> str:
111
+ """The text representation a failure is embedded from.
112
+
113
+ The structured labels lead, then the summary. Including the type and
114
+ component means two failures sharing a family agree on those tokens before
115
+ a single word of prose is compared, which is what keeps a well-labelled
116
+ dataset grouping tightly even with a purely lexical embedder.
117
+ """
118
+ return f"{analysis.failure_type.value} in {analysis.component.value}: {analysis.summary}"
119
+
120
+
121
+ def clustering_parameters(config: ClusteringConfig) -> dict[str, Any]:
122
+ """Everything needed to reproduce a grouping, stored with the run."""
123
+ return {
124
+ "algorithm": config.algorithm,
125
+ "metric": config.metric,
126
+ "linkage": config.linkage,
127
+ "threshold": config.threshold,
128
+ "seed": config.seed,
129
+ "embedder": config.embedder,
130
+ "dimensions": config.dimensions,
131
+ }
132
+
133
+
134
+ def build_clusters(
135
+ inputs: list[ClusterInput], vectors: list[list[float]], config: ClusteringConfig
136
+ ) -> list[Cluster]:
137
+ """Group ``inputs`` and choose representatives for each group."""
138
+ if not inputs:
139
+ return []
140
+ if len(inputs) != len(vectors): # pragma: no cover - callers pair these
141
+ raise ValueError("inputs and vectors must be the same length")
142
+
143
+ matrix = np.asarray(vectors, dtype=np.float64)
144
+
145
+ # A described failure is embedded from someone's account of what went wrong;
146
+ # an undescribed one from the tools it called and the evidence that caught
147
+ # it. Those are different kinds of text, so their distances sit on different
148
+ # scales and no single threshold cuts both correctly -- and a distance
149
+ # measured *between* the two populations does not mean anything at all.
150
+ # So each is clustered against its own kind, at its own threshold, and the
151
+ # groups are concatenated. Two records of one bug, one described and one
152
+ # not, therefore land in separate families: that is the honest answer, since
153
+ # nothing yet establishes they are the same.
154
+ clusters: list[Cluster] = []
155
+ for described in (True, False):
156
+ indices = [i for i, item in enumerate(inputs) if item.described is described]
157
+ if not indices:
158
+ continue
159
+ threshold = config.threshold if described else config.undescribed_threshold
160
+ assignments = _assign(matrix[indices], config, threshold)
161
+ for group in sorted(set(assignments)):
162
+ rows = [
163
+ index for index, label in zip(indices, assignments, strict=True) if label == group
164
+ ]
165
+ clusters.append(_build_one([inputs[i] for i in rows], matrix[rows]))
166
+
167
+ # Largest first, then by ID: a stable order for humans and for tests.
168
+ clusters.sort(key=lambda cluster: (-cluster.size, cluster.cluster_id))
169
+ return clusters
170
+
171
+
172
+ def _assign(matrix: np.ndarray[Any, Any], config: ClusteringConfig, threshold: float) -> list[int]:
173
+ if config.linkage != "average" or config.metric != "cosine":
174
+ raise CommandError(
175
+ f"Only average linkage over cosine distance is implemented, not "
176
+ f"{config.linkage!r} over {config.metric!r}.",
177
+ hint="Change clustering.linkage and clustering.metric in evalkeep.yaml.",
178
+ )
179
+ return average_linkage(matrix, threshold)
180
+
181
+
182
+ def average_linkage(vectors: np.ndarray[Any, Any], threshold: float) -> list[int]:
183
+ """Average-linkage agglomerative clustering over cosine distance.
184
+
185
+ Implemented here rather than pulled from scikit-learn, which would bring
186
+ scipy with it -- 119 MB of install for one class, in a tool whose clustering
187
+ is a few hundred vectors of lexical similarity. Verified against
188
+ scikit-learn's implementation across 240 random datasets before that
189
+ dependency was removed, and roughly thirty times faster at two thousand
190
+ points, because a general implementation does far more than this one case
191
+ needs.
192
+
193
+ Cluster distances are updated by the Lance-Williams rule, and each row keeps
194
+ its nearest neighbour so a merge costs a scan rather than a full search.
195
+ """
196
+ count = len(vectors)
197
+ if count <= 1:
198
+ return [0] * count
199
+
200
+ # Vectors are L2-normalized, so cosine distance is 1 - the dot product.
201
+ distances = np.clip(1.0 - vectors @ vectors.T, 0.0, 2.0)
202
+ np.fill_diagonal(distances, np.inf)
203
+
204
+ sizes = np.ones(count)
205
+ alive = np.ones(count, dtype=bool)
206
+ members: list[list[int]] = [[index] for index in range(count)]
207
+ nearest = distances.argmin(axis=1)
208
+ best = distances[np.arange(count), nearest]
209
+
210
+ for _ in range(count - 1):
211
+ candidates = np.where(alive, best, np.inf)
212
+ first = int(candidates.argmin())
213
+ if candidates[first] >= threshold:
214
+ break
215
+ second = int(nearest[first])
216
+ if first > second:
217
+ first, second = second, first
218
+
219
+ total = sizes[first] + sizes[second]
220
+ merged = (sizes[first] * distances[first] + sizes[second] * distances[second]) / total
221
+ distances[first] = merged
222
+ distances[:, first] = merged
223
+ distances[first, first] = np.inf
224
+ distances[second, :] = np.inf
225
+ distances[:, second] = np.inf
226
+
227
+ alive[second] = False
228
+ sizes[first] = total
229
+ members[first] += members[second]
230
+
231
+ # Only rows whose nearest neighbour was one of the merged pair can have
232
+ # changed, so the rest of the cache stays valid.
233
+ stale = np.where(alive & ((nearest == first) | (nearest == second)))[0]
234
+ for row in np.union1d(stale, [first]):
235
+ if alive[row]:
236
+ nearest[row] = int(distances[row].argmin())
237
+ best[row] = distances[row, nearest[row]]
238
+
239
+ labels = [0] * count
240
+ for label, index in enumerate(np.where(alive)[0]):
241
+ for point in members[index]:
242
+ labels[point] = label
243
+ return labels
244
+
245
+
246
+ def _build_one(members: list[ClusterInput], vectors: np.ndarray[Any, Any]) -> Cluster:
247
+ centroid = _centroid(vectors)
248
+ # Vectors are L2-normalized, so a dot product is the cosine similarity.
249
+ # Clamped because floating point can push a dot product just past 1.0,
250
+ # which would surface as a distance of -0.00 in the member listing.
251
+ distances = [min(2.0, max(0.0, float(1.0 - np.dot(vector, centroid)))) for vector in vectors]
252
+
253
+ cluster_members = [
254
+ ClusterMember(failure_id=item.failure_id, distance=distance)
255
+ for item, distance in zip(members, distances, strict=True)
256
+ ]
257
+ assign_roles(
258
+ cluster_members,
259
+ {item.failure_id: item.severity for item in members if item.severity is not None},
260
+ )
261
+ return Cluster.build(label=derive_label(members), members=cluster_members)
262
+
263
+
264
+ def _centroid(vectors: np.ndarray[Any, Any]) -> np.ndarray[Any, Any]:
265
+ centroid: np.ndarray[Any, Any] = vectors.mean(axis=0)
266
+ magnitude = float(np.linalg.norm(centroid))
267
+ return centroid / magnitude if magnitude else centroid
268
+
269
+
270
+ def assign_roles(
271
+ members: list[ClusterMember], severities: dict[str, Severity]
272
+ ) -> list[ClusterMember]:
273
+ """Mark the central, boundary and worst-case members, in place.
274
+
275
+ The three roles answer three different questions -- what this family
276
+ typically looks like, how far it stretches, and how bad it gets -- so one
277
+ failure can hold several. In a cluster of one it holds all three; roles
278
+ accumulate on a member rather than partitioning the cluster.
279
+
280
+ Shared with the editing commands on purpose: a cluster that a reviewer
281
+ merged or split must end up with the same kind of representatives as one
282
+ the algorithm produced, or the selection would silently differ depending on
283
+ how the cluster came to exist.
284
+ """
285
+ for member in members:
286
+ member.roles.clear()
287
+
288
+ order = sorted(range(len(members)), key=lambda i: (members[i].distance, i))
289
+ members[order[0]].roles.append(MemberRole.CENTRAL)
290
+ if len(members) > 1:
291
+ members[order[-1]].roles.append(MemberRole.BOUNDARY)
292
+
293
+ if severities:
294
+ worst = min(
295
+ range(len(members)),
296
+ key=lambda i: (
297
+ _severity_rank(severities, members[i].failure_id),
298
+ members[i].distance,
299
+ i,
300
+ ),
301
+ )
302
+ if MemberRole.HIGH_SEVERITY not in members[worst].roles:
303
+ members[worst].roles.append(MemberRole.HIGH_SEVERITY)
304
+ return members
305
+
306
+
307
+ def _severity_rank(severities: dict[str, Severity], failure_id: str) -> int:
308
+ severity = severities.get(failure_id)
309
+ # An unlabelled member cannot be the worst case; sort it last.
310
+ return SEVERITY_ORDER.index(severity) if severity is not None else len(SEVERITY_ORDER)
311
+
312
+
313
+ def derive_label(inputs: list[ClusterInput]) -> str:
314
+ """Name a family after what its members have in common.
315
+
316
+ Derived rather than generated: it needs no provider, it is reproducible, and
317
+ a reviewer can rename it. Guide 8G deliberately keeps this label out of test
318
+ IDs for exactly that reason -- it is mutable.
319
+
320
+ An undescribed family is named for what it did, and says so, because a label
321
+ that reads like an analysis when nobody analysed anything would be a lie.
322
+ """
323
+ described = [item for item in inputs if item.described and item.failure_type]
324
+ if not described:
325
+ behaviours = [item.behaviour for item in inputs if item.behaviour]
326
+ if behaviours:
327
+ return f"undescribed: {_most_common(iter(behaviours))}"
328
+ # No tool calls to name it after -- a ledger of outcomes rather than a
329
+ # trace of actions. Fall back to the words its members share, because a
330
+ # listing where every family reads "undescribed failures" tells a
331
+ # reviewer nothing about which one to open.
332
+ shared = _shared_terms(inputs)
333
+ return f"undescribed: {shared}" if shared else "undescribed failures"
334
+ types = _most_common(item.failure_type or "" for item in described)
335
+ components = _most_common(item.component or "" for item in described)
336
+ return f"{types} in {components}"
337
+
338
+
339
+ #: Present on every family, so they name none of them.
340
+ _UNINFORMATIVE = frozenset({kind.value for kind in SignalKind} | {"none", "null", "true", "false"})
341
+
342
+
343
+ def _shared_terms(inputs: list[ClusterInput], limit: int = 3) -> str:
344
+ """The words most of a family has in common, as a name for it.
345
+
346
+ Document frequency within the family, not raw count: a term repeated many
347
+ times in one long member says nothing about the family, while a term
348
+ present in most members is what they share. Terms carrying digits are
349
+ dropped -- "150" and "600" are what makes two instances of one contract
350
+ breach different, not what makes them the same.
351
+ """
352
+ documents = [frozenset(_terms(item.text)) for item in inputs]
353
+ documents = [document for document in documents if document]
354
+ if not documents:
355
+ return ""
356
+ counts: dict[str, int] = {}
357
+ for document in documents:
358
+ for term in document:
359
+ counts[term] = counts.get(term, 0) + 1
360
+ needed = max(1, (len(documents) + 1) // 2)
361
+ ranked = sorted(
362
+ ((term, n) for term, n in counts.items() if n >= needed),
363
+ key=lambda pair: (-pair[1], pair[0]),
364
+ )
365
+ return ", ".join(term for term, _ in ranked[:limit])
366
+
367
+
368
+ def _terms(text: str) -> list[str]:
369
+ return [
370
+ token
371
+ for token in re.findall(r"\w+", text.lower())
372
+ if len(token) > 1
373
+ and not any(char.isdigit() for char in token)
374
+ and token not in _UNINFORMATIVE
375
+ ]
376
+
377
+
378
+ def _most_common(values: Any) -> str:
379
+ counts: dict[str, int] = {}
380
+ for value in values:
381
+ counts[value] = counts.get(value, 0) + 1
382
+ # Ties break alphabetically so the label is a function of the members alone.
383
+ return sorted(counts.items(), key=lambda item: (-item[1], item[0]))[0][0]
evalkeep/clusters.py ADDED
@@ -0,0 +1,101 @@
1
+ """Clusters: families of failures, and the representatives chosen from them.
2
+
3
+ A cluster's identity is derived from its members, so an unchanged clustering
4
+ re-computes to the same IDs and any labels you gave it survive. Change the
5
+ membership and the identity changes -- which is honest, because it is no longer
6
+ the same group.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import hashlib
12
+ from dataclasses import dataclass, field
13
+ from datetime import UTC, datetime
14
+ from enum import StrEnum
15
+ from typing import Any
16
+
17
+ CLUSTER_ID_PREFIX = "cl-"
18
+ _CLUSTER_ID_LENGTH = 12
19
+
20
+
21
+ class MemberRole(StrEnum):
22
+ """Why a member was selected as a representative, if it was.
23
+
24
+ The three roles answer three different questions about a failure family:
25
+ what it typically looks like, how far it stretches, and how bad it gets.
26
+ """
27
+
28
+ #: Closest to the centroid: the most typical example of this family.
29
+ CENTRAL = "central"
30
+ #: Furthest from the centroid: the edge case the family still contains.
31
+ BOUNDARY = "boundary"
32
+ #: The worst outcome in the family, whether or not it is typical.
33
+ HIGH_SEVERITY = "high_severity"
34
+
35
+
36
+ @dataclass
37
+ class ClusterMember:
38
+ failure_id: str
39
+ #: Cosine distance to the cluster centroid.
40
+ distance: float
41
+ roles: list[MemberRole] = field(default_factory=list)
42
+
43
+ @property
44
+ def representative(self) -> bool:
45
+ return bool(self.roles)
46
+
47
+
48
+ def cluster_id_for(failure_ids: list[str]) -> str:
49
+ """A stable ID derived from membership, so re-clustering is idempotent."""
50
+ material = "\n".join(sorted(failure_ids))
51
+ digest = hashlib.sha256(material.encode("utf-8")).hexdigest()[:_CLUSTER_ID_LENGTH]
52
+ return f"{CLUSTER_ID_PREFIX}{digest}"
53
+
54
+
55
+ @dataclass
56
+ class Cluster:
57
+ cluster_id: str
58
+ label: str
59
+ members: list[ClusterMember] = field(default_factory=list)
60
+ #: Set when a person renamed it, so re-clustering knows not to relabel.
61
+ labelled_by: str | None = None
62
+ #: A family the reviewer decided is not worth regression coverage.
63
+ dismissed: bool = False
64
+ created_at: datetime = field(default_factory=lambda: datetime.now(UTC))
65
+
66
+ @classmethod
67
+ def build(cls, label: str, members: list[ClusterMember]) -> Cluster:
68
+ return cls(
69
+ cluster_id=cluster_id_for([member.failure_id for member in members]),
70
+ label=label,
71
+ members=members,
72
+ )
73
+
74
+ @property
75
+ def size(self) -> int:
76
+ return len(self.members)
77
+
78
+ @property
79
+ def failure_ids(self) -> list[str]:
80
+ return [member.failure_id for member in self.members]
81
+
82
+ @property
83
+ def representatives(self) -> list[ClusterMember]:
84
+ return [member for member in self.members if member.representative]
85
+
86
+ @property
87
+ def edited(self) -> bool:
88
+ """True once a person has changed something automation would overwrite."""
89
+ return self.labelled_by is not None or self.dismissed
90
+
91
+
92
+ @dataclass
93
+ class ClusteringRun:
94
+ """One clustering, with everything needed to reproduce it."""
95
+
96
+ run_id: str
97
+ embedder: str
98
+ dimensions: int
99
+ parameters: dict[str, Any]
100
+ failures: int = 0
101
+ created_at: datetime = field(default_factory=lambda: datetime.now(UTC))
@@ -0,0 +1 @@
1
+ """Command implementations, kept free of Typer/Rich so they stay testable."""
@@ -0,0 +1,100 @@
1
+ """``evalkeep analyze`` and ``evalkeep failures label`` -- describe failures.
2
+
3
+ Manual labelling is a first-class path, not a fallback. With no provider
4
+ configured Evalkeep still produces a fully labelled dataset; it just asks a
5
+ person for the labels instead of a model.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from pathlib import Path
11
+
12
+ from evalkeep.analysis import Component, FailureAnalysis, FailureType, Severity
13
+ from evalkeep.analysis_run import AnalysisReport, analyze_failures
14
+ from evalkeep.analyzers import MANUAL_PROVIDER, get_analyzer
15
+ from evalkeep.cache import AnalysisCache
16
+ from evalkeep.commands.detect_cmd import default_reviewer, resolve_failure
17
+ from evalkeep.config import Project
18
+ from evalkeep.errors import CommandError
19
+ from evalkeep.redaction import RedactionSummary, Redactor
20
+ from evalkeep.storage import TraceStore
21
+
22
+ #: Hand-written labels answer no prompt, so they carry no prompt version.
23
+ MANUAL_PROMPT_VERSION = 0
24
+
25
+ NO_PROVIDER_MESSAGE = (
26
+ "No analyzer provider is configured, so there is nothing to run automatically."
27
+ )
28
+ NO_PROVIDER_HINT = (
29
+ "Label failures by hand with 'evalkeep failures label <id> --type ... "
30
+ "--component ... --severity ... --summary ...', or set analyzer.provider "
31
+ "in evalkeep.yaml."
32
+ )
33
+
34
+
35
+ def run_analysis(
36
+ *,
37
+ project_root: Path = Path(),
38
+ reanalyze: bool = False,
39
+ overwrite_manual: bool = False,
40
+ limit: int | None = None,
41
+ use_cache: bool = True,
42
+ ) -> AnalysisReport:
43
+ """Analyze failures with the configured provider."""
44
+ project = Project.load(project_root.expanduser().resolve())
45
+ provider = get_analyzer(project.config.analyzer)
46
+ if provider is None:
47
+ raise CommandError(NO_PROVIDER_MESSAGE, hint=NO_PROVIDER_HINT)
48
+
49
+ cache = AnalysisCache(project.subdir("cache"), enabled=use_cache)
50
+ with TraceStore.open(project.database_path) as store:
51
+ if store.failures.count() == 0:
52
+ raise CommandError(
53
+ "No failure candidates to analyze.",
54
+ hint="Run 'evalkeep detect' first.",
55
+ )
56
+ return analyze_failures(
57
+ store,
58
+ provider,
59
+ cache,
60
+ redactor=Redactor(project.config.redaction),
61
+ reanalyze=reanalyze,
62
+ overwrite_manual=overwrite_manual,
63
+ limit=limit,
64
+ )
65
+
66
+
67
+ def label_failure(
68
+ identifier: str,
69
+ *,
70
+ failure_type: FailureType,
71
+ component: Component,
72
+ severity: Severity,
73
+ summary: str,
74
+ project_root: Path = Path(),
75
+ labeler: str | None = None,
76
+ ) -> FailureAnalysis:
77
+ """Record a hand-written analysis for one failure."""
78
+ cleaned = summary.strip()
79
+ if not cleaned:
80
+ raise CommandError("A label needs a non-empty --summary.")
81
+
82
+ project = Project.load(project_root.expanduser().resolve())
83
+ who = labeler or default_reviewer()
84
+ redactor = Redactor(project.config.redaction)
85
+
86
+ with TraceStore.open(project.database_path) as store:
87
+ failure = resolve_failure(store, identifier)
88
+ analysis = FailureAnalysis(
89
+ failure_type=failure_type,
90
+ component=component,
91
+ severity=severity,
92
+ # A person can type a real email address into a summary; redact it
93
+ # for the same reason the trace itself was redacted.
94
+ summary=redactor.redact_text(cleaned, RedactionSummary()),
95
+ analyzer=f"{MANUAL_PROVIDER}:{who}",
96
+ prompt_version=MANUAL_PROMPT_VERSION,
97
+ labeler=who,
98
+ )
99
+ store.failures.save_analysis(failure.failure_id, analysis)
100
+ return analysis