evalkeep 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalkeep/__init__.py +12 -0
- evalkeep/__main__.py +6 -0
- evalkeep/adapters/__init__.py +45 -0
- evalkeep/adapters/base.py +92 -0
- evalkeep/adapters/jsonl.py +164 -0
- evalkeep/adapters/langsmith.py +436 -0
- evalkeep/adapters/otlp.py +442 -0
- evalkeep/adapters/semconv.py +208 -0
- evalkeep/analysis.py +174 -0
- evalkeep/analysis_run.py +160 -0
- evalkeep/analyzers/__init__.py +52 -0
- evalkeep/analyzers/anthropic.py +145 -0
- evalkeep/analyzers/stub.py +34 -0
- evalkeep/cache.py +122 -0
- evalkeep/cli.py +1933 -0
- evalkeep/clustering.py +383 -0
- evalkeep/clusters.py +101 -0
- evalkeep/commands/__init__.py +1 -0
- evalkeep/commands/analyze_cmd.py +100 -0
- evalkeep/commands/compare_cmd.py +169 -0
- evalkeep/commands/dataset_cmd.py +182 -0
- evalkeep/commands/detect_cmd.py +154 -0
- evalkeep/commands/discover_cmd.py +274 -0
- evalkeep/commands/ingest_cmd.py +50 -0
- evalkeep/commands/init_cmd.py +151 -0
- evalkeep/commands/pipeline_cmd.py +156 -0
- evalkeep/commands/review_cmd.py +141 -0
- evalkeep/commands/run_cmd.py +131 -0
- evalkeep/commands/target_cmd.py +109 -0
- evalkeep/commands/trace_cmd.py +58 -0
- evalkeep/comparison.py +432 -0
- evalkeep/config.py +209 -0
- evalkeep/detection.py +94 -0
- evalkeep/detectors.py +182 -0
- evalkeep/discovery.py +208 -0
- evalkeep/embeddings/__init__.py +31 -0
- evalkeep/embeddings/base.py +32 -0
- evalkeep/embeddings/hashing.py +98 -0
- evalkeep/errors.py +42 -0
- evalkeep/examples/__init__.py +37 -0
- evalkeep/examples/langsmith/runs.jsonl +18 -0
- evalkeep/examples/opentelemetry/spans.json +898 -0
- evalkeep/examples/refund-agent/agents/baseline.py +66 -0
- evalkeep/examples/refund-agent/agents/candidate.py +66 -0
- evalkeep/examples/refund-agent/traces.jsonl +5 -0
- evalkeep/examples/tau-bench/prepare.py +230 -0
- evalkeep/exporters/__init__.py +45 -0
- evalkeep/exporters/generic.py +31 -0
- evalkeep/exporters/promptfoo.py +219 -0
- evalkeep/failures.py +95 -0
- evalkeep/generation.py +303 -0
- evalkeep/hashing.py +56 -0
- evalkeep/ingest.py +257 -0
- evalkeep/prompts.py +127 -0
- evalkeep/pseudonyms.py +82 -0
- evalkeep/py.typed +0 -0
- evalkeep/redaction.py +333 -0
- evalkeep/regression.py +409 -0
- evalkeep/review.py +309 -0
- evalkeep/runner.py +302 -0
- evalkeep/runs.py +185 -0
- evalkeep/storage/__init__.py +37 -0
- evalkeep/storage/clusters.py +163 -0
- evalkeep/storage/failures.py +254 -0
- evalkeep/storage/migrations.py +370 -0
- evalkeep/storage/regression.py +136 -0
- evalkeep/storage/runs.py +223 -0
- evalkeep/storage/store.py +429 -0
- evalkeep/targets.py +205 -0
- evalkeep/trace.py +238 -0
- evalkeep-0.1.0.dist-info/METADATA +221 -0
- evalkeep-0.1.0.dist-info/RECORD +75 -0
- evalkeep-0.1.0.dist-info/WHEEL +4 -0
- evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
- evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/clustering.py
ADDED
|
@@ -0,0 +1,383 @@
|
|
|
1
|
+
"""Grouping failures into families, and choosing who represents each family.
|
|
2
|
+
|
|
3
|
+
The algorithm is average-linkage agglomerative clustering over cosine distance,
|
|
4
|
+
cut at a configured distance threshold. It was chosen for three properties that
|
|
5
|
+
matter more here than raw clustering quality:
|
|
6
|
+
|
|
7
|
+
* **It is deterministic.** No initialisation, no random restarts: the same
|
|
8
|
+
vectors and the same threshold always produce the same grouping. A seed is
|
|
9
|
+
still recorded with every run, so swapping in a randomized algorithm later
|
|
10
|
+
cannot quietly break reproducibility.
|
|
11
|
+
* **It does not need the number of clusters up front.** Nobody knows how many
|
|
12
|
+
failure families a trace file contains.
|
|
13
|
+
* **The threshold means something.** It is a cosine distance, so it can be
|
|
14
|
+
explained, tuned and written down, rather than being an opaque knob.
|
|
15
|
+
|
|
16
|
+
Average linkage rather than single linkage on purpose: single linkage chains, so
|
|
17
|
+
one ambiguous failure sitting between two families would merge both.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import re
|
|
23
|
+
from dataclasses import dataclass
|
|
24
|
+
from typing import Any
|
|
25
|
+
|
|
26
|
+
import numpy as np
|
|
27
|
+
|
|
28
|
+
from evalkeep.analysis import SEVERITY_ORDER, FailureAnalysis, Severity
|
|
29
|
+
from evalkeep.clusters import Cluster, ClusterMember, MemberRole
|
|
30
|
+
from evalkeep.config import ClusteringConfig
|
|
31
|
+
from evalkeep.detectors import SignalKind
|
|
32
|
+
from evalkeep.errors import CommandError
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True)
|
|
36
|
+
class ClusterInput:
|
|
37
|
+
"""One failure, as clustering sees it.
|
|
38
|
+
|
|
39
|
+
A failure is normally grouped by its *description* -- the structured
|
|
40
|
+
analysis someone or something wrote for it. Before anyone has described
|
|
41
|
+
them, it can still be grouped by what was *observed*: which tools ran and
|
|
42
|
+
what the evidence said. That is weaker, and the difference is tracked here
|
|
43
|
+
rather than hidden, so a report can say which it did.
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
failure_id: str
|
|
47
|
+
text: str
|
|
48
|
+
severity: Severity | None = None
|
|
49
|
+
failure_type: str | None = None
|
|
50
|
+
component: str | None = None
|
|
51
|
+
#: The tools this failure used, for naming a family nobody has described.
|
|
52
|
+
behaviour: str | None = None
|
|
53
|
+
#: False when this was grouped by observed behaviour rather than a description.
|
|
54
|
+
described: bool = True
|
|
55
|
+
|
|
56
|
+
@classmethod
|
|
57
|
+
def from_analysis(cls, failure_id: str, analysis: FailureAnalysis) -> ClusterInput:
|
|
58
|
+
return cls(
|
|
59
|
+
failure_id=failure_id,
|
|
60
|
+
text=cluster_text(analysis),
|
|
61
|
+
severity=analysis.severity,
|
|
62
|
+
failure_type=analysis.failure_type.value,
|
|
63
|
+
component=analysis.component.value,
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
@classmethod
|
|
67
|
+
def from_observation(
|
|
68
|
+
cls, failure_id: str, text: str, *, behaviour: str | None = None
|
|
69
|
+
) -> ClusterInput:
|
|
70
|
+
return cls(failure_id=failure_id, text=text, behaviour=behaviour, described=False)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def observation_text(tools: list[str], kinds: list[str], evidence: list[str]) -> str:
|
|
74
|
+
"""What a failure looks like before anyone has described it.
|
|
75
|
+
|
|
76
|
+
Built from what the agent *did* -- which tools ran, what kind of evidence
|
|
77
|
+
caught it -- and only then from the evidence's own words. Deliberately not
|
|
78
|
+
from the request: two customers asking the same thing in different words are
|
|
79
|
+
the same failure, and their phrasing would scatter them.
|
|
80
|
+
|
|
81
|
+
The tool list is repeated because in a bag-of-words representation
|
|
82
|
+
repetition *is* weight, and behaviour is the reliable signal here. Two
|
|
83
|
+
reports of one bug are worded differently while the calls the agent made
|
|
84
|
+
stay the same, so letting the wording dominate loses the family. Sublinear
|
|
85
|
+
term weighting damps the repetition, so this is a nudge rather than a
|
|
86
|
+
override.
|
|
87
|
+
|
|
88
|
+
Boilerplate evidence is dropped: "explicitly marked as failed" appears on
|
|
89
|
+
every explicit failure and so distinguishes none of them.
|
|
90
|
+
"""
|
|
91
|
+
listed = ", ".join(sorted(set(tools)))
|
|
92
|
+
informative = [line for line in evidence if line and not _boilerplate(line)]
|
|
93
|
+
weighted = " ".join([listed] * 3) if listed else ""
|
|
94
|
+
return f"{weighted} | {' '.join(sorted(set(kinds)))} | {' '.join(informative)}".strip(" |")
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
_BOILERPLATE = (
|
|
98
|
+
"explicitly marked as failed",
|
|
99
|
+
"explicitly marked as errored",
|
|
100
|
+
"feedback was rated negative",
|
|
101
|
+
"reported a failure",
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _boilerplate(line: str) -> bool:
|
|
106
|
+
lowered = line.lower()
|
|
107
|
+
return any(phrase in lowered for phrase in _BOILERPLATE)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def cluster_text(analysis: FailureAnalysis) -> str:
|
|
111
|
+
"""The text representation a failure is embedded from.
|
|
112
|
+
|
|
113
|
+
The structured labels lead, then the summary. Including the type and
|
|
114
|
+
component means two failures sharing a family agree on those tokens before
|
|
115
|
+
a single word of prose is compared, which is what keeps a well-labelled
|
|
116
|
+
dataset grouping tightly even with a purely lexical embedder.
|
|
117
|
+
"""
|
|
118
|
+
return f"{analysis.failure_type.value} in {analysis.component.value}: {analysis.summary}"
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def clustering_parameters(config: ClusteringConfig) -> dict[str, Any]:
|
|
122
|
+
"""Everything needed to reproduce a grouping, stored with the run."""
|
|
123
|
+
return {
|
|
124
|
+
"algorithm": config.algorithm,
|
|
125
|
+
"metric": config.metric,
|
|
126
|
+
"linkage": config.linkage,
|
|
127
|
+
"threshold": config.threshold,
|
|
128
|
+
"seed": config.seed,
|
|
129
|
+
"embedder": config.embedder,
|
|
130
|
+
"dimensions": config.dimensions,
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def build_clusters(
|
|
135
|
+
inputs: list[ClusterInput], vectors: list[list[float]], config: ClusteringConfig
|
|
136
|
+
) -> list[Cluster]:
|
|
137
|
+
"""Group ``inputs`` and choose representatives for each group."""
|
|
138
|
+
if not inputs:
|
|
139
|
+
return []
|
|
140
|
+
if len(inputs) != len(vectors): # pragma: no cover - callers pair these
|
|
141
|
+
raise ValueError("inputs and vectors must be the same length")
|
|
142
|
+
|
|
143
|
+
matrix = np.asarray(vectors, dtype=np.float64)
|
|
144
|
+
|
|
145
|
+
# A described failure is embedded from someone's account of what went wrong;
|
|
146
|
+
# an undescribed one from the tools it called and the evidence that caught
|
|
147
|
+
# it. Those are different kinds of text, so their distances sit on different
|
|
148
|
+
# scales and no single threshold cuts both correctly -- and a distance
|
|
149
|
+
# measured *between* the two populations does not mean anything at all.
|
|
150
|
+
# So each is clustered against its own kind, at its own threshold, and the
|
|
151
|
+
# groups are concatenated. Two records of one bug, one described and one
|
|
152
|
+
# not, therefore land in separate families: that is the honest answer, since
|
|
153
|
+
# nothing yet establishes they are the same.
|
|
154
|
+
clusters: list[Cluster] = []
|
|
155
|
+
for described in (True, False):
|
|
156
|
+
indices = [i for i, item in enumerate(inputs) if item.described is described]
|
|
157
|
+
if not indices:
|
|
158
|
+
continue
|
|
159
|
+
threshold = config.threshold if described else config.undescribed_threshold
|
|
160
|
+
assignments = _assign(matrix[indices], config, threshold)
|
|
161
|
+
for group in sorted(set(assignments)):
|
|
162
|
+
rows = [
|
|
163
|
+
index for index, label in zip(indices, assignments, strict=True) if label == group
|
|
164
|
+
]
|
|
165
|
+
clusters.append(_build_one([inputs[i] for i in rows], matrix[rows]))
|
|
166
|
+
|
|
167
|
+
# Largest first, then by ID: a stable order for humans and for tests.
|
|
168
|
+
clusters.sort(key=lambda cluster: (-cluster.size, cluster.cluster_id))
|
|
169
|
+
return clusters
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _assign(matrix: np.ndarray[Any, Any], config: ClusteringConfig, threshold: float) -> list[int]:
|
|
173
|
+
if config.linkage != "average" or config.metric != "cosine":
|
|
174
|
+
raise CommandError(
|
|
175
|
+
f"Only average linkage over cosine distance is implemented, not "
|
|
176
|
+
f"{config.linkage!r} over {config.metric!r}.",
|
|
177
|
+
hint="Change clustering.linkage and clustering.metric in evalkeep.yaml.",
|
|
178
|
+
)
|
|
179
|
+
return average_linkage(matrix, threshold)
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def average_linkage(vectors: np.ndarray[Any, Any], threshold: float) -> list[int]:
|
|
183
|
+
"""Average-linkage agglomerative clustering over cosine distance.
|
|
184
|
+
|
|
185
|
+
Implemented here rather than pulled from scikit-learn, which would bring
|
|
186
|
+
scipy with it -- 119 MB of install for one class, in a tool whose clustering
|
|
187
|
+
is a few hundred vectors of lexical similarity. Verified against
|
|
188
|
+
scikit-learn's implementation across 240 random datasets before that
|
|
189
|
+
dependency was removed, and roughly thirty times faster at two thousand
|
|
190
|
+
points, because a general implementation does far more than this one case
|
|
191
|
+
needs.
|
|
192
|
+
|
|
193
|
+
Cluster distances are updated by the Lance-Williams rule, and each row keeps
|
|
194
|
+
its nearest neighbour so a merge costs a scan rather than a full search.
|
|
195
|
+
"""
|
|
196
|
+
count = len(vectors)
|
|
197
|
+
if count <= 1:
|
|
198
|
+
return [0] * count
|
|
199
|
+
|
|
200
|
+
# Vectors are L2-normalized, so cosine distance is 1 - the dot product.
|
|
201
|
+
distances = np.clip(1.0 - vectors @ vectors.T, 0.0, 2.0)
|
|
202
|
+
np.fill_diagonal(distances, np.inf)
|
|
203
|
+
|
|
204
|
+
sizes = np.ones(count)
|
|
205
|
+
alive = np.ones(count, dtype=bool)
|
|
206
|
+
members: list[list[int]] = [[index] for index in range(count)]
|
|
207
|
+
nearest = distances.argmin(axis=1)
|
|
208
|
+
best = distances[np.arange(count), nearest]
|
|
209
|
+
|
|
210
|
+
for _ in range(count - 1):
|
|
211
|
+
candidates = np.where(alive, best, np.inf)
|
|
212
|
+
first = int(candidates.argmin())
|
|
213
|
+
if candidates[first] >= threshold:
|
|
214
|
+
break
|
|
215
|
+
second = int(nearest[first])
|
|
216
|
+
if first > second:
|
|
217
|
+
first, second = second, first
|
|
218
|
+
|
|
219
|
+
total = sizes[first] + sizes[second]
|
|
220
|
+
merged = (sizes[first] * distances[first] + sizes[second] * distances[second]) / total
|
|
221
|
+
distances[first] = merged
|
|
222
|
+
distances[:, first] = merged
|
|
223
|
+
distances[first, first] = np.inf
|
|
224
|
+
distances[second, :] = np.inf
|
|
225
|
+
distances[:, second] = np.inf
|
|
226
|
+
|
|
227
|
+
alive[second] = False
|
|
228
|
+
sizes[first] = total
|
|
229
|
+
members[first] += members[second]
|
|
230
|
+
|
|
231
|
+
# Only rows whose nearest neighbour was one of the merged pair can have
|
|
232
|
+
# changed, so the rest of the cache stays valid.
|
|
233
|
+
stale = np.where(alive & ((nearest == first) | (nearest == second)))[0]
|
|
234
|
+
for row in np.union1d(stale, [first]):
|
|
235
|
+
if alive[row]:
|
|
236
|
+
nearest[row] = int(distances[row].argmin())
|
|
237
|
+
best[row] = distances[row, nearest[row]]
|
|
238
|
+
|
|
239
|
+
labels = [0] * count
|
|
240
|
+
for label, index in enumerate(np.where(alive)[0]):
|
|
241
|
+
for point in members[index]:
|
|
242
|
+
labels[point] = label
|
|
243
|
+
return labels
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _build_one(members: list[ClusterInput], vectors: np.ndarray[Any, Any]) -> Cluster:
|
|
247
|
+
centroid = _centroid(vectors)
|
|
248
|
+
# Vectors are L2-normalized, so a dot product is the cosine similarity.
|
|
249
|
+
# Clamped because floating point can push a dot product just past 1.0,
|
|
250
|
+
# which would surface as a distance of -0.00 in the member listing.
|
|
251
|
+
distances = [min(2.0, max(0.0, float(1.0 - np.dot(vector, centroid)))) for vector in vectors]
|
|
252
|
+
|
|
253
|
+
cluster_members = [
|
|
254
|
+
ClusterMember(failure_id=item.failure_id, distance=distance)
|
|
255
|
+
for item, distance in zip(members, distances, strict=True)
|
|
256
|
+
]
|
|
257
|
+
assign_roles(
|
|
258
|
+
cluster_members,
|
|
259
|
+
{item.failure_id: item.severity for item in members if item.severity is not None},
|
|
260
|
+
)
|
|
261
|
+
return Cluster.build(label=derive_label(members), members=cluster_members)
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def _centroid(vectors: np.ndarray[Any, Any]) -> np.ndarray[Any, Any]:
|
|
265
|
+
centroid: np.ndarray[Any, Any] = vectors.mean(axis=0)
|
|
266
|
+
magnitude = float(np.linalg.norm(centroid))
|
|
267
|
+
return centroid / magnitude if magnitude else centroid
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def assign_roles(
|
|
271
|
+
members: list[ClusterMember], severities: dict[str, Severity]
|
|
272
|
+
) -> list[ClusterMember]:
|
|
273
|
+
"""Mark the central, boundary and worst-case members, in place.
|
|
274
|
+
|
|
275
|
+
The three roles answer three different questions -- what this family
|
|
276
|
+
typically looks like, how far it stretches, and how bad it gets -- so one
|
|
277
|
+
failure can hold several. In a cluster of one it holds all three; roles
|
|
278
|
+
accumulate on a member rather than partitioning the cluster.
|
|
279
|
+
|
|
280
|
+
Shared with the editing commands on purpose: a cluster that a reviewer
|
|
281
|
+
merged or split must end up with the same kind of representatives as one
|
|
282
|
+
the algorithm produced, or the selection would silently differ depending on
|
|
283
|
+
how the cluster came to exist.
|
|
284
|
+
"""
|
|
285
|
+
for member in members:
|
|
286
|
+
member.roles.clear()
|
|
287
|
+
|
|
288
|
+
order = sorted(range(len(members)), key=lambda i: (members[i].distance, i))
|
|
289
|
+
members[order[0]].roles.append(MemberRole.CENTRAL)
|
|
290
|
+
if len(members) > 1:
|
|
291
|
+
members[order[-1]].roles.append(MemberRole.BOUNDARY)
|
|
292
|
+
|
|
293
|
+
if severities:
|
|
294
|
+
worst = min(
|
|
295
|
+
range(len(members)),
|
|
296
|
+
key=lambda i: (
|
|
297
|
+
_severity_rank(severities, members[i].failure_id),
|
|
298
|
+
members[i].distance,
|
|
299
|
+
i,
|
|
300
|
+
),
|
|
301
|
+
)
|
|
302
|
+
if MemberRole.HIGH_SEVERITY not in members[worst].roles:
|
|
303
|
+
members[worst].roles.append(MemberRole.HIGH_SEVERITY)
|
|
304
|
+
return members
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def _severity_rank(severities: dict[str, Severity], failure_id: str) -> int:
|
|
308
|
+
severity = severities.get(failure_id)
|
|
309
|
+
# An unlabelled member cannot be the worst case; sort it last.
|
|
310
|
+
return SEVERITY_ORDER.index(severity) if severity is not None else len(SEVERITY_ORDER)
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def derive_label(inputs: list[ClusterInput]) -> str:
|
|
314
|
+
"""Name a family after what its members have in common.
|
|
315
|
+
|
|
316
|
+
Derived rather than generated: it needs no provider, it is reproducible, and
|
|
317
|
+
a reviewer can rename it. Guide 8G deliberately keeps this label out of test
|
|
318
|
+
IDs for exactly that reason -- it is mutable.
|
|
319
|
+
|
|
320
|
+
An undescribed family is named for what it did, and says so, because a label
|
|
321
|
+
that reads like an analysis when nobody analysed anything would be a lie.
|
|
322
|
+
"""
|
|
323
|
+
described = [item for item in inputs if item.described and item.failure_type]
|
|
324
|
+
if not described:
|
|
325
|
+
behaviours = [item.behaviour for item in inputs if item.behaviour]
|
|
326
|
+
if behaviours:
|
|
327
|
+
return f"undescribed: {_most_common(iter(behaviours))}"
|
|
328
|
+
# No tool calls to name it after -- a ledger of outcomes rather than a
|
|
329
|
+
# trace of actions. Fall back to the words its members share, because a
|
|
330
|
+
# listing where every family reads "undescribed failures" tells a
|
|
331
|
+
# reviewer nothing about which one to open.
|
|
332
|
+
shared = _shared_terms(inputs)
|
|
333
|
+
return f"undescribed: {shared}" if shared else "undescribed failures"
|
|
334
|
+
types = _most_common(item.failure_type or "" for item in described)
|
|
335
|
+
components = _most_common(item.component or "" for item in described)
|
|
336
|
+
return f"{types} in {components}"
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
#: Present on every family, so they name none of them.
|
|
340
|
+
_UNINFORMATIVE = frozenset({kind.value for kind in SignalKind} | {"none", "null", "true", "false"})
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def _shared_terms(inputs: list[ClusterInput], limit: int = 3) -> str:
|
|
344
|
+
"""The words most of a family has in common, as a name for it.
|
|
345
|
+
|
|
346
|
+
Document frequency within the family, not raw count: a term repeated many
|
|
347
|
+
times in one long member says nothing about the family, while a term
|
|
348
|
+
present in most members is what they share. Terms carrying digits are
|
|
349
|
+
dropped -- "150" and "600" are what makes two instances of one contract
|
|
350
|
+
breach different, not what makes them the same.
|
|
351
|
+
"""
|
|
352
|
+
documents = [frozenset(_terms(item.text)) for item in inputs]
|
|
353
|
+
documents = [document for document in documents if document]
|
|
354
|
+
if not documents:
|
|
355
|
+
return ""
|
|
356
|
+
counts: dict[str, int] = {}
|
|
357
|
+
for document in documents:
|
|
358
|
+
for term in document:
|
|
359
|
+
counts[term] = counts.get(term, 0) + 1
|
|
360
|
+
needed = max(1, (len(documents) + 1) // 2)
|
|
361
|
+
ranked = sorted(
|
|
362
|
+
((term, n) for term, n in counts.items() if n >= needed),
|
|
363
|
+
key=lambda pair: (-pair[1], pair[0]),
|
|
364
|
+
)
|
|
365
|
+
return ", ".join(term for term, _ in ranked[:limit])
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
def _terms(text: str) -> list[str]:
|
|
369
|
+
return [
|
|
370
|
+
token
|
|
371
|
+
for token in re.findall(r"\w+", text.lower())
|
|
372
|
+
if len(token) > 1
|
|
373
|
+
and not any(char.isdigit() for char in token)
|
|
374
|
+
and token not in _UNINFORMATIVE
|
|
375
|
+
]
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
def _most_common(values: Any) -> str:
|
|
379
|
+
counts: dict[str, int] = {}
|
|
380
|
+
for value in values:
|
|
381
|
+
counts[value] = counts.get(value, 0) + 1
|
|
382
|
+
# Ties break alphabetically so the label is a function of the members alone.
|
|
383
|
+
return sorted(counts.items(), key=lambda item: (-item[1], item[0]))[0][0]
|
evalkeep/clusters.py
ADDED
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""Clusters: families of failures, and the representatives chosen from them.
|
|
2
|
+
|
|
3
|
+
A cluster's identity is derived from its members, so an unchanged clustering
|
|
4
|
+
re-computes to the same IDs and any labels you gave it survive. Change the
|
|
5
|
+
membership and the identity changes -- which is honest, because it is no longer
|
|
6
|
+
the same group.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import hashlib
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
from datetime import UTC, datetime
|
|
14
|
+
from enum import StrEnum
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
CLUSTER_ID_PREFIX = "cl-"
|
|
18
|
+
_CLUSTER_ID_LENGTH = 12
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class MemberRole(StrEnum):
|
|
22
|
+
"""Why a member was selected as a representative, if it was.
|
|
23
|
+
|
|
24
|
+
The three roles answer three different questions about a failure family:
|
|
25
|
+
what it typically looks like, how far it stretches, and how bad it gets.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
#: Closest to the centroid: the most typical example of this family.
|
|
29
|
+
CENTRAL = "central"
|
|
30
|
+
#: Furthest from the centroid: the edge case the family still contains.
|
|
31
|
+
BOUNDARY = "boundary"
|
|
32
|
+
#: The worst outcome in the family, whether or not it is typical.
|
|
33
|
+
HIGH_SEVERITY = "high_severity"
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class ClusterMember:
|
|
38
|
+
failure_id: str
|
|
39
|
+
#: Cosine distance to the cluster centroid.
|
|
40
|
+
distance: float
|
|
41
|
+
roles: list[MemberRole] = field(default_factory=list)
|
|
42
|
+
|
|
43
|
+
@property
|
|
44
|
+
def representative(self) -> bool:
|
|
45
|
+
return bool(self.roles)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def cluster_id_for(failure_ids: list[str]) -> str:
|
|
49
|
+
"""A stable ID derived from membership, so re-clustering is idempotent."""
|
|
50
|
+
material = "\n".join(sorted(failure_ids))
|
|
51
|
+
digest = hashlib.sha256(material.encode("utf-8")).hexdigest()[:_CLUSTER_ID_LENGTH]
|
|
52
|
+
return f"{CLUSTER_ID_PREFIX}{digest}"
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass
|
|
56
|
+
class Cluster:
|
|
57
|
+
cluster_id: str
|
|
58
|
+
label: str
|
|
59
|
+
members: list[ClusterMember] = field(default_factory=list)
|
|
60
|
+
#: Set when a person renamed it, so re-clustering knows not to relabel.
|
|
61
|
+
labelled_by: str | None = None
|
|
62
|
+
#: A family the reviewer decided is not worth regression coverage.
|
|
63
|
+
dismissed: bool = False
|
|
64
|
+
created_at: datetime = field(default_factory=lambda: datetime.now(UTC))
|
|
65
|
+
|
|
66
|
+
@classmethod
|
|
67
|
+
def build(cls, label: str, members: list[ClusterMember]) -> Cluster:
|
|
68
|
+
return cls(
|
|
69
|
+
cluster_id=cluster_id_for([member.failure_id for member in members]),
|
|
70
|
+
label=label,
|
|
71
|
+
members=members,
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
@property
|
|
75
|
+
def size(self) -> int:
|
|
76
|
+
return len(self.members)
|
|
77
|
+
|
|
78
|
+
@property
|
|
79
|
+
def failure_ids(self) -> list[str]:
|
|
80
|
+
return [member.failure_id for member in self.members]
|
|
81
|
+
|
|
82
|
+
@property
|
|
83
|
+
def representatives(self) -> list[ClusterMember]:
|
|
84
|
+
return [member for member in self.members if member.representative]
|
|
85
|
+
|
|
86
|
+
@property
|
|
87
|
+
def edited(self) -> bool:
|
|
88
|
+
"""True once a person has changed something automation would overwrite."""
|
|
89
|
+
return self.labelled_by is not None or self.dismissed
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@dataclass
|
|
93
|
+
class ClusteringRun:
|
|
94
|
+
"""One clustering, with everything needed to reproduce it."""
|
|
95
|
+
|
|
96
|
+
run_id: str
|
|
97
|
+
embedder: str
|
|
98
|
+
dimensions: int
|
|
99
|
+
parameters: dict[str, Any]
|
|
100
|
+
failures: int = 0
|
|
101
|
+
created_at: datetime = field(default_factory=lambda: datetime.now(UTC))
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Command implementations, kept free of Typer/Rich so they stay testable."""
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""``evalkeep analyze`` and ``evalkeep failures label`` -- describe failures.
|
|
2
|
+
|
|
3
|
+
Manual labelling is a first-class path, not a fallback. With no provider
|
|
4
|
+
configured Evalkeep still produces a fully labelled dataset; it just asks a
|
|
5
|
+
person for the labels instead of a model.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from evalkeep.analysis import Component, FailureAnalysis, FailureType, Severity
|
|
13
|
+
from evalkeep.analysis_run import AnalysisReport, analyze_failures
|
|
14
|
+
from evalkeep.analyzers import MANUAL_PROVIDER, get_analyzer
|
|
15
|
+
from evalkeep.cache import AnalysisCache
|
|
16
|
+
from evalkeep.commands.detect_cmd import default_reviewer, resolve_failure
|
|
17
|
+
from evalkeep.config import Project
|
|
18
|
+
from evalkeep.errors import CommandError
|
|
19
|
+
from evalkeep.redaction import RedactionSummary, Redactor
|
|
20
|
+
from evalkeep.storage import TraceStore
|
|
21
|
+
|
|
22
|
+
#: Hand-written labels answer no prompt, so they carry no prompt version.
|
|
23
|
+
MANUAL_PROMPT_VERSION = 0
|
|
24
|
+
|
|
25
|
+
NO_PROVIDER_MESSAGE = (
|
|
26
|
+
"No analyzer provider is configured, so there is nothing to run automatically."
|
|
27
|
+
)
|
|
28
|
+
NO_PROVIDER_HINT = (
|
|
29
|
+
"Label failures by hand with 'evalkeep failures label <id> --type ... "
|
|
30
|
+
"--component ... --severity ... --summary ...', or set analyzer.provider "
|
|
31
|
+
"in evalkeep.yaml."
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def run_analysis(
|
|
36
|
+
*,
|
|
37
|
+
project_root: Path = Path(),
|
|
38
|
+
reanalyze: bool = False,
|
|
39
|
+
overwrite_manual: bool = False,
|
|
40
|
+
limit: int | None = None,
|
|
41
|
+
use_cache: bool = True,
|
|
42
|
+
) -> AnalysisReport:
|
|
43
|
+
"""Analyze failures with the configured provider."""
|
|
44
|
+
project = Project.load(project_root.expanduser().resolve())
|
|
45
|
+
provider = get_analyzer(project.config.analyzer)
|
|
46
|
+
if provider is None:
|
|
47
|
+
raise CommandError(NO_PROVIDER_MESSAGE, hint=NO_PROVIDER_HINT)
|
|
48
|
+
|
|
49
|
+
cache = AnalysisCache(project.subdir("cache"), enabled=use_cache)
|
|
50
|
+
with TraceStore.open(project.database_path) as store:
|
|
51
|
+
if store.failures.count() == 0:
|
|
52
|
+
raise CommandError(
|
|
53
|
+
"No failure candidates to analyze.",
|
|
54
|
+
hint="Run 'evalkeep detect' first.",
|
|
55
|
+
)
|
|
56
|
+
return analyze_failures(
|
|
57
|
+
store,
|
|
58
|
+
provider,
|
|
59
|
+
cache,
|
|
60
|
+
redactor=Redactor(project.config.redaction),
|
|
61
|
+
reanalyze=reanalyze,
|
|
62
|
+
overwrite_manual=overwrite_manual,
|
|
63
|
+
limit=limit,
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def label_failure(
|
|
68
|
+
identifier: str,
|
|
69
|
+
*,
|
|
70
|
+
failure_type: FailureType,
|
|
71
|
+
component: Component,
|
|
72
|
+
severity: Severity,
|
|
73
|
+
summary: str,
|
|
74
|
+
project_root: Path = Path(),
|
|
75
|
+
labeler: str | None = None,
|
|
76
|
+
) -> FailureAnalysis:
|
|
77
|
+
"""Record a hand-written analysis for one failure."""
|
|
78
|
+
cleaned = summary.strip()
|
|
79
|
+
if not cleaned:
|
|
80
|
+
raise CommandError("A label needs a non-empty --summary.")
|
|
81
|
+
|
|
82
|
+
project = Project.load(project_root.expanduser().resolve())
|
|
83
|
+
who = labeler or default_reviewer()
|
|
84
|
+
redactor = Redactor(project.config.redaction)
|
|
85
|
+
|
|
86
|
+
with TraceStore.open(project.database_path) as store:
|
|
87
|
+
failure = resolve_failure(store, identifier)
|
|
88
|
+
analysis = FailureAnalysis(
|
|
89
|
+
failure_type=failure_type,
|
|
90
|
+
component=component,
|
|
91
|
+
severity=severity,
|
|
92
|
+
# A person can type a real email address into a summary; redact it
|
|
93
|
+
# for the same reason the trace itself was redacted.
|
|
94
|
+
summary=redactor.redact_text(cleaned, RedactionSummary()),
|
|
95
|
+
analyzer=f"{MANUAL_PROVIDER}:{who}",
|
|
96
|
+
prompt_version=MANUAL_PROMPT_VERSION,
|
|
97
|
+
labeler=who,
|
|
98
|
+
)
|
|
99
|
+
store.failures.save_analysis(failure.failure_id, analysis)
|
|
100
|
+
return analysis
|