evalkeep 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalkeep/__init__.py +12 -0
- evalkeep/__main__.py +6 -0
- evalkeep/adapters/__init__.py +45 -0
- evalkeep/adapters/base.py +92 -0
- evalkeep/adapters/jsonl.py +164 -0
- evalkeep/adapters/langsmith.py +436 -0
- evalkeep/adapters/otlp.py +442 -0
- evalkeep/adapters/semconv.py +208 -0
- evalkeep/analysis.py +174 -0
- evalkeep/analysis_run.py +160 -0
- evalkeep/analyzers/__init__.py +52 -0
- evalkeep/analyzers/anthropic.py +145 -0
- evalkeep/analyzers/stub.py +34 -0
- evalkeep/cache.py +122 -0
- evalkeep/cli.py +1933 -0
- evalkeep/clustering.py +383 -0
- evalkeep/clusters.py +101 -0
- evalkeep/commands/__init__.py +1 -0
- evalkeep/commands/analyze_cmd.py +100 -0
- evalkeep/commands/compare_cmd.py +169 -0
- evalkeep/commands/dataset_cmd.py +182 -0
- evalkeep/commands/detect_cmd.py +154 -0
- evalkeep/commands/discover_cmd.py +274 -0
- evalkeep/commands/ingest_cmd.py +50 -0
- evalkeep/commands/init_cmd.py +151 -0
- evalkeep/commands/pipeline_cmd.py +156 -0
- evalkeep/commands/review_cmd.py +141 -0
- evalkeep/commands/run_cmd.py +131 -0
- evalkeep/commands/target_cmd.py +109 -0
- evalkeep/commands/trace_cmd.py +58 -0
- evalkeep/comparison.py +432 -0
- evalkeep/config.py +209 -0
- evalkeep/detection.py +94 -0
- evalkeep/detectors.py +182 -0
- evalkeep/discovery.py +208 -0
- evalkeep/embeddings/__init__.py +31 -0
- evalkeep/embeddings/base.py +32 -0
- evalkeep/embeddings/hashing.py +98 -0
- evalkeep/errors.py +42 -0
- evalkeep/examples/__init__.py +37 -0
- evalkeep/examples/langsmith/runs.jsonl +18 -0
- evalkeep/examples/opentelemetry/spans.json +898 -0
- evalkeep/examples/refund-agent/agents/baseline.py +66 -0
- evalkeep/examples/refund-agent/agents/candidate.py +66 -0
- evalkeep/examples/refund-agent/traces.jsonl +5 -0
- evalkeep/examples/tau-bench/prepare.py +230 -0
- evalkeep/exporters/__init__.py +45 -0
- evalkeep/exporters/generic.py +31 -0
- evalkeep/exporters/promptfoo.py +219 -0
- evalkeep/failures.py +95 -0
- evalkeep/generation.py +303 -0
- evalkeep/hashing.py +56 -0
- evalkeep/ingest.py +257 -0
- evalkeep/prompts.py +127 -0
- evalkeep/pseudonyms.py +82 -0
- evalkeep/py.typed +0 -0
- evalkeep/redaction.py +333 -0
- evalkeep/regression.py +409 -0
- evalkeep/review.py +309 -0
- evalkeep/runner.py +302 -0
- evalkeep/runs.py +185 -0
- evalkeep/storage/__init__.py +37 -0
- evalkeep/storage/clusters.py +163 -0
- evalkeep/storage/failures.py +254 -0
- evalkeep/storage/migrations.py +370 -0
- evalkeep/storage/regression.py +136 -0
- evalkeep/storage/runs.py +223 -0
- evalkeep/storage/store.py +429 -0
- evalkeep/targets.py +205 -0
- evalkeep/trace.py +238 -0
- evalkeep-0.1.0.dist-info/METADATA +221 -0
- evalkeep-0.1.0.dist-info/RECORD +75 -0
- evalkeep-0.1.0.dist-info/WHEEL +4 -0
- evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
- evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/detection.py
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""The detection pass: turn stored traces into evidence-backed failure candidates.
|
|
2
|
+
|
|
3
|
+
Detection is idempotent, and the rules that make it so are the whole design:
|
|
4
|
+
|
|
5
|
+
* A failure's ID is a pure function of its trace ID, so a second pass addresses
|
|
6
|
+
the same row instead of creating another one.
|
|
7
|
+
* Signals are *derived*. Every pass recomputes and replaces them, so changing or
|
|
8
|
+
adding a detector updates the evidence on existing failures.
|
|
9
|
+
* A review is *not* derived. Once a person confirms or dismisses a failure,
|
|
10
|
+
detection updates its signals and leaves the decision, reviewer and reason
|
|
11
|
+
alone -- automated analysis never overwrites a human judgement.
|
|
12
|
+
* A candidate nobody has reviewed, whose evidence has since disappeared (a
|
|
13
|
+
detector was removed or narrowed), is withdrawn. Nothing human is lost,
|
|
14
|
+
because nothing human was there. Manually added failures are never withdrawn.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from collections.abc import Iterable
|
|
20
|
+
from dataclasses import dataclass, field
|
|
21
|
+
|
|
22
|
+
from evalkeep.detectors import DETECTORS, FailureDetector, SignalKind, detect_signals
|
|
23
|
+
from evalkeep.failures import Failure, FailureOrigin
|
|
24
|
+
from evalkeep.storage import TraceStore
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class DetectionReport:
|
|
29
|
+
"""What one detection pass did."""
|
|
30
|
+
|
|
31
|
+
traces: int = 0
|
|
32
|
+
created: int = 0
|
|
33
|
+
updated: int = 0
|
|
34
|
+
unchanged: int = 0
|
|
35
|
+
withdrawn: int = 0
|
|
36
|
+
preserved_reviews: int = 0
|
|
37
|
+
signals: int = 0
|
|
38
|
+
by_kind: dict[SignalKind, int] = field(default_factory=dict)
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def failures(self) -> int:
|
|
42
|
+
"""Traces that came out of this pass carrying evidence."""
|
|
43
|
+
return self.created + self.updated + self.unchanged
|
|
44
|
+
|
|
45
|
+
@property
|
|
46
|
+
def changed(self) -> bool:
|
|
47
|
+
return bool(self.created or self.updated or self.withdrawn)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def detect_failures(
|
|
51
|
+
store: TraceStore, *, detectors: Iterable[FailureDetector] = DETECTORS
|
|
52
|
+
) -> DetectionReport:
|
|
53
|
+
"""Run every detector over every stored trace and persist the result."""
|
|
54
|
+
detectors = tuple(detectors)
|
|
55
|
+
report = DetectionReport()
|
|
56
|
+
failures = store.failures
|
|
57
|
+
|
|
58
|
+
for trace in store.iter_traces():
|
|
59
|
+
report.traces += 1
|
|
60
|
+
signals = detect_signals(trace, detectors)
|
|
61
|
+
existing = failures.get_by_trace(trace.trace_id)
|
|
62
|
+
|
|
63
|
+
if not signals:
|
|
64
|
+
# No evidence. Withdraw an unreviewed detector candidate; leave
|
|
65
|
+
# anything a person touched or created.
|
|
66
|
+
if (
|
|
67
|
+
existing is not None
|
|
68
|
+
and not existing.reviewed
|
|
69
|
+
and existing.origin is FailureOrigin.DETECTOR
|
|
70
|
+
):
|
|
71
|
+
failures.delete(existing.failure_id)
|
|
72
|
+
report.withdrawn += 1
|
|
73
|
+
continue
|
|
74
|
+
|
|
75
|
+
report.signals += len(signals)
|
|
76
|
+
for signal in signals:
|
|
77
|
+
report.by_kind[signal.kind] = report.by_kind.get(signal.kind, 0) + 1
|
|
78
|
+
|
|
79
|
+
if existing is None:
|
|
80
|
+
failures.save(Failure.from_signals(trace.trace_id, signals))
|
|
81
|
+
report.created += 1
|
|
82
|
+
continue
|
|
83
|
+
|
|
84
|
+
if existing.reviewed:
|
|
85
|
+
report.preserved_reviews += 1
|
|
86
|
+
if [s.to_dict() for s in existing.signals] == [s.to_dict() for s in signals]:
|
|
87
|
+
report.unchanged += 1
|
|
88
|
+
continue
|
|
89
|
+
|
|
90
|
+
existing.signals = signals
|
|
91
|
+
failures.save(existing)
|
|
92
|
+
report.updated += 1
|
|
93
|
+
|
|
94
|
+
return report
|
evalkeep/detectors.py
ADDED
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
"""Failure detectors: evidence that something went wrong, never a guess.
|
|
2
|
+
|
|
3
|
+
A detector reads one stored (already redacted) trace and emits a
|
|
4
|
+
:class:`Signal` for each piece of explicit evidence it finds. Detectors do not
|
|
5
|
+
infer failure from the shape of an interaction, and they do not score their
|
|
6
|
+
confidence -- a signal either points at something a person recorded, or it does
|
|
7
|
+
not exist. Combining signals is therefore counting evidence, not adding
|
|
8
|
+
probabilities: a trace with three signals is better documented than one with a
|
|
9
|
+
single signal, not "more likely" to be a failure.
|
|
10
|
+
|
|
11
|
+
Adding a detector means implementing :class:`FailureDetector` and registering
|
|
12
|
+
it; nothing else in the pipeline changes.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from collections.abc import Iterable, Iterator
|
|
18
|
+
from dataclasses import dataclass, field
|
|
19
|
+
from enum import StrEnum
|
|
20
|
+
from typing import Any, ClassVar, Protocol, runtime_checkable
|
|
21
|
+
|
|
22
|
+
from evalkeep.trace import EvaluationEvent, NormalizedTrace, OutcomeStatus
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class SignalKind(StrEnum):
|
|
26
|
+
"""The category of evidence, independent of which detector found it."""
|
|
27
|
+
|
|
28
|
+
EXPLICIT_STATUS = "explicit_status"
|
|
29
|
+
NEGATIVE_FEEDBACK = "negative_feedback"
|
|
30
|
+
FAILED_EVALUATOR = "failed_evaluator"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(frozen=True)
|
|
34
|
+
class Signal:
|
|
35
|
+
"""One piece of evidence, with a pointer back to where it came from."""
|
|
36
|
+
|
|
37
|
+
detector: str
|
|
38
|
+
kind: SignalKind
|
|
39
|
+
#: Path into the trace, so a reviewer can check the claim themselves.
|
|
40
|
+
source: str
|
|
41
|
+
summary: str
|
|
42
|
+
evidence: dict[str, Any] = field(default_factory=dict)
|
|
43
|
+
|
|
44
|
+
def to_dict(self) -> dict[str, Any]:
|
|
45
|
+
return {
|
|
46
|
+
"detector": self.detector,
|
|
47
|
+
"kind": self.kind.value,
|
|
48
|
+
"source": self.source,
|
|
49
|
+
"summary": self.summary,
|
|
50
|
+
"evidence": self.evidence,
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
@classmethod
|
|
54
|
+
def from_dict(cls, payload: dict[str, Any]) -> Signal:
|
|
55
|
+
return cls(
|
|
56
|
+
detector=payload["detector"],
|
|
57
|
+
kind=SignalKind(payload["kind"]),
|
|
58
|
+
source=payload["source"],
|
|
59
|
+
summary=payload["summary"],
|
|
60
|
+
evidence=payload.get("evidence", {}),
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@runtime_checkable
|
|
65
|
+
class FailureDetector(Protocol):
|
|
66
|
+
"""Reads a trace, yields evidence. Never raises on well-formed input."""
|
|
67
|
+
|
|
68
|
+
name: ClassVar[str]
|
|
69
|
+
description: ClassVar[str]
|
|
70
|
+
|
|
71
|
+
def detect(self, trace: NormalizedTrace) -> Iterable[Signal]:
|
|
72
|
+
"""Yield one signal per piece of evidence found in ``trace``."""
|
|
73
|
+
...
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
class ExplicitStatusDetector:
|
|
77
|
+
"""The recorded outcome says the interaction failed."""
|
|
78
|
+
|
|
79
|
+
name: ClassVar[str] = "explicit_status"
|
|
80
|
+
description: ClassVar[str] = "outcome.status is 'failure' or 'error'"
|
|
81
|
+
|
|
82
|
+
_FAILING: ClassVar[dict[OutcomeStatus, str]] = {
|
|
83
|
+
OutcomeStatus.FAILURE: "the trace is explicitly marked as failed",
|
|
84
|
+
OutcomeStatus.ERROR: "the trace is explicitly marked as errored",
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
def detect(self, trace: NormalizedTrace) -> Iterator[Signal]:
|
|
88
|
+
summary = self._FAILING.get(trace.outcome.status)
|
|
89
|
+
if summary is None:
|
|
90
|
+
return
|
|
91
|
+
yield Signal(
|
|
92
|
+
detector=self.name,
|
|
93
|
+
kind=SignalKind.EXPLICIT_STATUS,
|
|
94
|
+
source="outcome.status",
|
|
95
|
+
summary=summary,
|
|
96
|
+
evidence={"status": trace.outcome.status.value},
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class NegativeFeedbackDetector:
|
|
101
|
+
"""Structured feedback records a negative rating.
|
|
102
|
+
|
|
103
|
+
Only an explicit ``rating`` of ``negative`` counts. A bare numeric score
|
|
104
|
+
arrives without its scale -- 2 could be poor out of 5 or good out of 3 --
|
|
105
|
+
and inventing a threshold would be exactly the kind of blind scoring this
|
|
106
|
+
pipeline is meant to avoid.
|
|
107
|
+
"""
|
|
108
|
+
|
|
109
|
+
name: ClassVar[str] = "negative_feedback"
|
|
110
|
+
description: ClassVar[str] = "outcome.feedback.rating is 'negative'"
|
|
111
|
+
|
|
112
|
+
def detect(self, trace: NormalizedTrace) -> Iterator[Signal]:
|
|
113
|
+
feedback = trace.outcome.feedback
|
|
114
|
+
if feedback is None or feedback.rating != "negative":
|
|
115
|
+
return
|
|
116
|
+
evidence: dict[str, Any] = {"rating": feedback.rating}
|
|
117
|
+
if feedback.comment:
|
|
118
|
+
evidence["comment"] = feedback.comment
|
|
119
|
+
if feedback.score is not None:
|
|
120
|
+
evidence["score"] = feedback.score
|
|
121
|
+
yield Signal(
|
|
122
|
+
detector=self.name,
|
|
123
|
+
kind=SignalKind.NEGATIVE_FEEDBACK,
|
|
124
|
+
source="outcome.feedback",
|
|
125
|
+
summary=feedback.comment or "feedback was rated negative",
|
|
126
|
+
evidence=evidence,
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
class FailedEvaluatorDetector:
|
|
131
|
+
"""An evaluator recorded a failing verdict, in the outcome or in an event."""
|
|
132
|
+
|
|
133
|
+
name: ClassVar[str] = "failed_evaluator"
|
|
134
|
+
description: ClassVar[str] = "an evaluation recorded passed=false"
|
|
135
|
+
|
|
136
|
+
def detect(self, trace: NormalizedTrace) -> Iterator[Signal]:
|
|
137
|
+
for index, evaluation in enumerate(trace.outcome.evaluations):
|
|
138
|
+
if evaluation.passed is False:
|
|
139
|
+
yield self._signal(
|
|
140
|
+
source=f"outcome.evaluations.{index}",
|
|
141
|
+
name=evaluation.name,
|
|
142
|
+
reason=evaluation.reason,
|
|
143
|
+
score=evaluation.score,
|
|
144
|
+
)
|
|
145
|
+
for index, event in enumerate(trace.events):
|
|
146
|
+
if isinstance(event, EvaluationEvent) and event.passed is False:
|
|
147
|
+
yield self._signal(
|
|
148
|
+
source=f"events.{index}",
|
|
149
|
+
name=event.name,
|
|
150
|
+
reason=event.reason,
|
|
151
|
+
score=event.score,
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
def _signal(self, *, source: str, name: str, reason: str | None, score: float | None) -> Signal:
|
|
155
|
+
evidence: dict[str, Any] = {"evaluator": name, "passed": False}
|
|
156
|
+
if reason:
|
|
157
|
+
evidence["reason"] = reason
|
|
158
|
+
if score is not None:
|
|
159
|
+
evidence["score"] = score
|
|
160
|
+
return Signal(
|
|
161
|
+
detector=self.name,
|
|
162
|
+
kind=SignalKind.FAILED_EVALUATOR,
|
|
163
|
+
source=source,
|
|
164
|
+
summary=reason or f"evaluator {name!r} reported a failure",
|
|
165
|
+
evidence=evidence,
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
#: The detectors run by ``evalkeep detect``, in a fixed order so that the
|
|
170
|
+
#: signals persisted for a trace are reproducible.
|
|
171
|
+
DETECTORS: tuple[FailureDetector, ...] = (
|
|
172
|
+
ExplicitStatusDetector(),
|
|
173
|
+
NegativeFeedbackDetector(),
|
|
174
|
+
FailedEvaluatorDetector(),
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def detect_signals(
|
|
179
|
+
trace: NormalizedTrace, detectors: Iterable[FailureDetector] = DETECTORS
|
|
180
|
+
) -> list[Signal]:
|
|
181
|
+
"""Every signal every detector finds in ``trace``, in detector order."""
|
|
182
|
+
return [signal for detector in detectors for signal in detector.detect(trace)]
|
evalkeep/discovery.py
ADDED
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
"""The discovery pass: embed analyzed failures, group them, pick representatives.
|
|
2
|
+
|
|
3
|
+
Discovery is derived from analysis, so it can always be recomputed. What cannot
|
|
4
|
+
be recomputed is what a reviewer did to the result -- renaming a family,
|
|
5
|
+
dismissing one, merging two that the distance metric kept apart. Those edits are
|
|
6
|
+
preserved where the group survives unchanged, and re-clustering refuses to
|
|
7
|
+
discard them without an explicit ``--force``.
|
|
8
|
+
|
|
9
|
+
Cluster identity is derived from membership, which is what makes that possible:
|
|
10
|
+
re-running on unchanged data produces the same cluster IDs, so labels stay
|
|
11
|
+
attached to the families they were written for.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import uuid
|
|
17
|
+
from dataclasses import dataclass, field
|
|
18
|
+
|
|
19
|
+
from evalkeep.cache import EmbeddingCache, embedding_key
|
|
20
|
+
from evalkeep.clustering import (
|
|
21
|
+
ClusterInput,
|
|
22
|
+
build_clusters,
|
|
23
|
+
clustering_parameters,
|
|
24
|
+
observation_text,
|
|
25
|
+
)
|
|
26
|
+
from evalkeep.clusters import Cluster, ClusteringRun
|
|
27
|
+
from evalkeep.config import ClusteringConfig
|
|
28
|
+
from evalkeep.embeddings import EmbeddingProvider
|
|
29
|
+
from evalkeep.failures import FailureStatus
|
|
30
|
+
from evalkeep.storage import TraceStore
|
|
31
|
+
|
|
32
|
+
#: Dismissed failures are not part of any family worth covering.
|
|
33
|
+
CLUSTERABLE = (FailureStatus.CANDIDATE, FailureStatus.CONFIRMED)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class DiscoveryReport:
|
|
38
|
+
"""What one discovery pass did."""
|
|
39
|
+
|
|
40
|
+
embedder: str
|
|
41
|
+
considered: int = 0
|
|
42
|
+
unanalyzed: int = 0
|
|
43
|
+
clusters: int = 0
|
|
44
|
+
singletons: int = 0
|
|
45
|
+
representatives: int = 0
|
|
46
|
+
embedded: int = 0
|
|
47
|
+
from_cache: int = 0
|
|
48
|
+
kept_labels: int = 0
|
|
49
|
+
discarded_edits: int = 0
|
|
50
|
+
#: Failures grouped by observed behaviour because nobody described them.
|
|
51
|
+
grouped_undescribed: int = 0
|
|
52
|
+
largest: int = 0
|
|
53
|
+
parameters: dict[str, object] = field(default_factory=dict)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class ClusterEditsWouldBeLost(Exception):
|
|
57
|
+
"""Re-clustering would discard reviewer edits that cannot be carried over."""
|
|
58
|
+
|
|
59
|
+
def __init__(self, clusters: list[Cluster]) -> None:
|
|
60
|
+
self.clusters = clusters
|
|
61
|
+
super().__init__(f"{len(clusters)} edited clusters would be discarded")
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def discover(
|
|
65
|
+
store: TraceStore,
|
|
66
|
+
embedder: EmbeddingProvider,
|
|
67
|
+
cache: EmbeddingCache,
|
|
68
|
+
config: ClusteringConfig,
|
|
69
|
+
*,
|
|
70
|
+
force: bool = False,
|
|
71
|
+
group_undescribed: bool = False,
|
|
72
|
+
) -> DiscoveryReport:
|
|
73
|
+
"""Cluster analyzed failures and select representatives."""
|
|
74
|
+
report = DiscoveryReport(embedder=embedder.identity)
|
|
75
|
+
report.parameters = dict(clustering_parameters(config))
|
|
76
|
+
|
|
77
|
+
inputs = _gather(store, report, group_undescribed=group_undescribed)
|
|
78
|
+
vectors = _embed(inputs, embedder, cache, report)
|
|
79
|
+
clusters = build_clusters(inputs, vectors, config)
|
|
80
|
+
|
|
81
|
+
_carry_over_edits(store, clusters, report, force=force)
|
|
82
|
+
|
|
83
|
+
run = ClusteringRun(
|
|
84
|
+
run_id=uuid.uuid4().hex,
|
|
85
|
+
embedder=embedder.identity,
|
|
86
|
+
dimensions=embedder.dimensions,
|
|
87
|
+
parameters=report.parameters,
|
|
88
|
+
failures=len(inputs),
|
|
89
|
+
)
|
|
90
|
+
store.clusters.replace_run(run, clusters)
|
|
91
|
+
|
|
92
|
+
report.clusters = len(clusters)
|
|
93
|
+
report.singletons = sum(1 for cluster in clusters if cluster.size == 1)
|
|
94
|
+
report.representatives = sum(len(cluster.representatives) for cluster in clusters)
|
|
95
|
+
report.largest = max((cluster.size for cluster in clusters), default=0)
|
|
96
|
+
return report
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _gather(
|
|
100
|
+
store: TraceStore, report: DiscoveryReport, *, group_undescribed: bool = False
|
|
101
|
+
) -> list[ClusterInput]:
|
|
102
|
+
"""Undismissed failures, grouped by description where one exists.
|
|
103
|
+
|
|
104
|
+
An undescribed failure is skipped by default: grouping by observed behaviour
|
|
105
|
+
is weaker than grouping by what a failure *is*, and silently mixing the two
|
|
106
|
+
would make a report mean two different things. ``group_undescribed`` opts
|
|
107
|
+
in, and the counts say which was used.
|
|
108
|
+
"""
|
|
109
|
+
inputs: list[ClusterInput] = []
|
|
110
|
+
for failure in store.failures.iter_all():
|
|
111
|
+
if failure.status not in CLUSTERABLE:
|
|
112
|
+
continue
|
|
113
|
+
report.considered += 1
|
|
114
|
+
analysis = store.failures.get_analysis(failure.failure_id)
|
|
115
|
+
if analysis is not None:
|
|
116
|
+
inputs.append(ClusterInput.from_analysis(failure.failure_id, analysis))
|
|
117
|
+
continue
|
|
118
|
+
|
|
119
|
+
report.unanalyzed += 1
|
|
120
|
+
if not group_undescribed:
|
|
121
|
+
continue
|
|
122
|
+
stored = store.get(failure.trace_id)
|
|
123
|
+
if stored is None: # pragma: no cover - the foreign key prevents this
|
|
124
|
+
continue
|
|
125
|
+
tools = sorted({call.tool for call in stored.trace.tool_calls})
|
|
126
|
+
inputs.append(
|
|
127
|
+
ClusterInput.from_observation(
|
|
128
|
+
failure.failure_id,
|
|
129
|
+
behaviour=", ".join(tools) or None,
|
|
130
|
+
text=observation_text(
|
|
131
|
+
[call.tool for call in stored.trace.tool_calls],
|
|
132
|
+
[signal.kind.value for signal in failure.signals],
|
|
133
|
+
[signal.summary for signal in failure.signals],
|
|
134
|
+
),
|
|
135
|
+
)
|
|
136
|
+
)
|
|
137
|
+
report.grouped_undescribed += 1
|
|
138
|
+
# A stable input order keeps the clustering reproducible.
|
|
139
|
+
inputs.sort(key=lambda item: item.failure_id)
|
|
140
|
+
return inputs
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _embed(
|
|
144
|
+
inputs: list[ClusterInput],
|
|
145
|
+
embedder: EmbeddingProvider,
|
|
146
|
+
cache: EmbeddingCache,
|
|
147
|
+
report: DiscoveryReport,
|
|
148
|
+
) -> list[list[float]]:
|
|
149
|
+
"""Embed, reusing cached vectors for text that has not changed."""
|
|
150
|
+
vectors: list[list[float] | None] = []
|
|
151
|
+
pending: list[int] = []
|
|
152
|
+
|
|
153
|
+
for index, item in enumerate(inputs):
|
|
154
|
+
cached = cache.get_vector(embedding_key(item.text, embedder.identity))
|
|
155
|
+
vectors.append(cached)
|
|
156
|
+
if cached is None:
|
|
157
|
+
pending.append(index)
|
|
158
|
+
|
|
159
|
+
if pending:
|
|
160
|
+
fresh = embedder.embed([inputs[index].text for index in pending])
|
|
161
|
+
for index, vector in zip(pending, fresh, strict=True):
|
|
162
|
+
vectors[index] = vector
|
|
163
|
+
cache.put_vector(embedding_key(inputs[index].text, embedder.identity), vector)
|
|
164
|
+
|
|
165
|
+
report.embedded = len(pending)
|
|
166
|
+
report.from_cache = len(inputs) - len(pending)
|
|
167
|
+
return [vector for vector in vectors if vector is not None]
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _carry_over_edits(
|
|
171
|
+
store: TraceStore,
|
|
172
|
+
clusters: list[Cluster],
|
|
173
|
+
report: DiscoveryReport,
|
|
174
|
+
*,
|
|
175
|
+
force: bool,
|
|
176
|
+
) -> None:
|
|
177
|
+
"""Reattach reviewer edits to families that came back unchanged.
|
|
178
|
+
|
|
179
|
+
A cluster ID is a function of its members, so a family whose membership did
|
|
180
|
+
not change keeps its ID -- and with it, its name and its dismissal. A family
|
|
181
|
+
whose membership *did* change is genuinely a different group, and its edits
|
|
182
|
+
cannot be carried over honestly. Rather than dropping them quietly, the pass
|
|
183
|
+
refuses until the caller says to proceed.
|
|
184
|
+
"""
|
|
185
|
+
existing = {cluster.cluster_id: cluster for cluster in store.clusters.list()}
|
|
186
|
+
if not existing:
|
|
187
|
+
return
|
|
188
|
+
|
|
189
|
+
rebuilt = {cluster.cluster_id: cluster for cluster in clusters}
|
|
190
|
+
for cluster_id, previous in existing.items():
|
|
191
|
+
if not previous.edited:
|
|
192
|
+
continue
|
|
193
|
+
survivor = rebuilt.get(cluster_id)
|
|
194
|
+
if survivor is None:
|
|
195
|
+
report.discarded_edits += 1
|
|
196
|
+
continue
|
|
197
|
+
survivor.label = previous.label
|
|
198
|
+
survivor.labelled_by = previous.labelled_by
|
|
199
|
+
survivor.dismissed = previous.dismissed
|
|
200
|
+
report.kept_labels += 1
|
|
201
|
+
|
|
202
|
+
if report.discarded_edits and not force:
|
|
203
|
+
lost = [
|
|
204
|
+
cluster
|
|
205
|
+
for cluster_id, cluster in existing.items()
|
|
206
|
+
if cluster.edited and cluster_id not in rebuilt
|
|
207
|
+
]
|
|
208
|
+
raise ClusterEditsWouldBeLost(lost)
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""Embedding providers and the registry the project configuration resolves against."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from evalkeep.config import ClusteringConfig
|
|
6
|
+
from evalkeep.embeddings.base import EmbeddingProvider
|
|
7
|
+
from evalkeep.embeddings.hashing import HashingEmbedder
|
|
8
|
+
from evalkeep.errors import CommandError
|
|
9
|
+
|
|
10
|
+
KNOWN_EMBEDDERS: dict[str, str] = {HashingEmbedder.name: HashingEmbedder.description}
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def get_embedder(config: ClusteringConfig) -> EmbeddingProvider:
|
|
14
|
+
"""Build the configured embedding provider.
|
|
15
|
+
|
|
16
|
+
A hosted embedding model plugs in here: implement
|
|
17
|
+
:class:`~evalkeep.embeddings.base.EmbeddingProvider`, give it an identity
|
|
18
|
+
that names the model, and register it below. Nothing downstream changes --
|
|
19
|
+
the cache and the stored clustering parameters key off ``identity``, so old
|
|
20
|
+
and new vectors can never be silently mixed.
|
|
21
|
+
"""
|
|
22
|
+
if config.embedder == HashingEmbedder.name:
|
|
23
|
+
return HashingEmbedder(dimensions=config.dimensions, seed=config.seed)
|
|
24
|
+
known = ", ".join(sorted(KNOWN_EMBEDDERS))
|
|
25
|
+
raise CommandError(
|
|
26
|
+
f"Unknown embedding provider {config.embedder!r}.",
|
|
27
|
+
hint=f"Known providers: {known}.",
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
__all__ = ["KNOWN_EMBEDDERS", "EmbeddingProvider", "HashingEmbedder", "get_embedder"]
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""The embedding provider contract.
|
|
2
|
+
|
|
3
|
+
Embeddings turn failure descriptions into vectors so that similar failures land
|
|
4
|
+
near each other. The interface is deliberately tiny -- one method over a batch
|
|
5
|
+
of strings -- so a hosted embedding model can replace the built-in one without
|
|
6
|
+
anything else in the pipeline changing.
|
|
7
|
+
|
|
8
|
+
``identity`` is part of every cache key and is stored with every clustering run.
|
|
9
|
+
It must change whenever the vectors would change: a different model, a different
|
|
10
|
+
dimensionality or a different seed is a different embedding space, and mixing
|
|
11
|
+
two of them silently would produce meaningless distances.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from typing import ClassVar, Protocol, runtime_checkable
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@runtime_checkable
|
|
20
|
+
class EmbeddingProvider(Protocol):
|
|
21
|
+
name: ClassVar[str]
|
|
22
|
+
description: ClassVar[str]
|
|
23
|
+
|
|
24
|
+
@property
|
|
25
|
+
def identity(self) -> str: ...
|
|
26
|
+
|
|
27
|
+
@property
|
|
28
|
+
def dimensions(self) -> int: ...
|
|
29
|
+
|
|
30
|
+
def embed(self, texts: list[str]) -> list[list[float]]:
|
|
31
|
+
"""Embed a batch of texts, returning one L2-normalized vector each."""
|
|
32
|
+
...
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
"""A deterministic, offline embedding provider using the hashing trick.
|
|
2
|
+
|
|
3
|
+
Tokens (and adjacent token pairs) are hashed to fixed positions in a vector with
|
|
4
|
+
a signed contribution, then the vector is L2-normalized so cosine similarity is
|
|
5
|
+
a dot product. This is feature hashing, a real technique -- not a stub -- but it
|
|
6
|
+
is worth being precise about what it does and does not capture:
|
|
7
|
+
|
|
8
|
+
* It captures **lexical** overlap. Two summaries describing the same mistake in
|
|
9
|
+
similar words land close together. Bigrams mean word order carries some
|
|
10
|
+
weight, so "refunded the oldest order" and "ordered the oldest refund" differ.
|
|
11
|
+
* It does **not** capture meaning. "refunded the wrong order" and "issued a
|
|
12
|
+
credit for the incorrect purchase" share almost no tokens and will not group,
|
|
13
|
+
where a trained embedding model would place them together.
|
|
14
|
+
|
|
15
|
+
That tradeoff is acceptable here because the text being embedded is not free
|
|
16
|
+
prose: it is a structured analysis whose failure type and component are part of
|
|
17
|
+
the string, written under a prompt that explicitly asks for wording that repeats
|
|
18
|
+
across the same failure family. Lexical similarity does real work on that input.
|
|
19
|
+
|
|
20
|
+
Determinism is the point. ``hash()`` is randomized per process in Python and
|
|
21
|
+
must never be used; BLAKE2b keyed by the configured seed gives the same vector
|
|
22
|
+
on every machine and every run, which is what makes a clustering reproducible.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import hashlib
|
|
28
|
+
import math
|
|
29
|
+
import re
|
|
30
|
+
from itertools import pairwise
|
|
31
|
+
from typing import ClassVar
|
|
32
|
+
|
|
33
|
+
DEFAULT_DIMENSIONS = 512
|
|
34
|
+
DEFAULT_SEED = 0
|
|
35
|
+
|
|
36
|
+
_TOKEN_PATTERN = re.compile(r"[a-z0-9]+")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class HashingEmbedder:
|
|
40
|
+
name: ClassVar[str] = "hashing"
|
|
41
|
+
description: ClassVar[str] = "Deterministic offline feature hashing (lexical similarity)"
|
|
42
|
+
|
|
43
|
+
def __init__(self, *, dimensions: int = DEFAULT_DIMENSIONS, seed: int = DEFAULT_SEED) -> None:
|
|
44
|
+
if dimensions < 8:
|
|
45
|
+
raise ValueError("dimensions must be at least 8")
|
|
46
|
+
self._dimensions = dimensions
|
|
47
|
+
self.seed = seed
|
|
48
|
+
|
|
49
|
+
@property
|
|
50
|
+
def identity(self) -> str:
|
|
51
|
+
return f"{self.name}:{self._dimensions}:{self.seed}"
|
|
52
|
+
|
|
53
|
+
@property
|
|
54
|
+
def dimensions(self) -> int:
|
|
55
|
+
return self._dimensions
|
|
56
|
+
|
|
57
|
+
def embed(self, texts: list[str]) -> list[list[float]]:
|
|
58
|
+
return [self._embed_one(text) for text in texts]
|
|
59
|
+
|
|
60
|
+
def _embed_one(self, text: str) -> list[float]:
|
|
61
|
+
vector = [0.0] * self._dimensions
|
|
62
|
+
for feature, weight in _features(text).items():
|
|
63
|
+
index, sign = self._position(feature)
|
|
64
|
+
vector[index] += sign * weight
|
|
65
|
+
return _normalize(vector)
|
|
66
|
+
|
|
67
|
+
def _position(self, feature: str) -> tuple[int, int]:
|
|
68
|
+
"""Hash a feature to a bucket and a sign.
|
|
69
|
+
|
|
70
|
+
The sign halves the bias from collisions: two unrelated features landing
|
|
71
|
+
in the same bucket are as likely to cancel as to reinforce.
|
|
72
|
+
"""
|
|
73
|
+
digest = hashlib.blake2b(
|
|
74
|
+
feature.encode("utf-8"), digest_size=9, key=str(self.seed).encode("utf-8")
|
|
75
|
+
).digest()
|
|
76
|
+
value = int.from_bytes(digest, "big")
|
|
77
|
+
return value % self._dimensions, 1 if (value >> 63) & 1 else -1
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _features(text: str) -> dict[str, float]:
|
|
81
|
+
"""Unigrams and bigrams, with sublinear term weighting."""
|
|
82
|
+
tokens = _TOKEN_PATTERN.findall(text.lower())
|
|
83
|
+
counts: dict[str, int] = {}
|
|
84
|
+
for token in tokens:
|
|
85
|
+
counts[token] = counts.get(token, 0) + 1
|
|
86
|
+
for first, second in pairwise(tokens):
|
|
87
|
+
bigram = f"{first}_{second}"
|
|
88
|
+
counts[bigram] = counts.get(bigram, 0) + 1
|
|
89
|
+
# 1 + log(count): a word repeated ten times matters more than once, but not
|
|
90
|
+
# ten times more, so one long quoted string cannot dominate a summary.
|
|
91
|
+
return {feature: 1.0 + math.log(count) for feature, count in counts.items()}
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _normalize(vector: list[float]) -> list[float]:
|
|
95
|
+
magnitude = math.sqrt(sum(value * value for value in vector))
|
|
96
|
+
if magnitude == 0.0:
|
|
97
|
+
return vector
|
|
98
|
+
return [value / magnitude for value in vector]
|
evalkeep/errors.py
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""Stable exit codes and the error type that carries them.
|
|
2
|
+
|
|
3
|
+
The codes are part of the CLI contract and are relied on by scripts and CI:
|
|
4
|
+
|
|
5
|
+
* ``0`` -- the command succeeded.
|
|
6
|
+
* ``1`` -- the command ran, but some records were invalid or rejected.
|
|
7
|
+
* ``2`` -- the command itself could not run (bad usage, missing file,
|
|
8
|
+
unwritable directory, uninitialized project).
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from enum import IntEnum
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class ExitCode(IntEnum):
|
|
17
|
+
OK = 0
|
|
18
|
+
RECORD_ERRORS = 1
|
|
19
|
+
COMMAND_ERROR = 2
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class EvalkeepError(Exception):
|
|
23
|
+
"""An error that maps to a stable exit code and a readable message."""
|
|
24
|
+
|
|
25
|
+
exit_code: ExitCode = ExitCode.COMMAND_ERROR
|
|
26
|
+
|
|
27
|
+
def __init__(self, message: str, *, hint: str | None = None) -> None:
|
|
28
|
+
super().__init__(message)
|
|
29
|
+
self.message = message
|
|
30
|
+
self.hint = hint
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class CommandError(EvalkeepError):
|
|
34
|
+
"""The command could not run at all."""
|
|
35
|
+
|
|
36
|
+
exit_code = ExitCode.COMMAND_ERROR
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class RecordError(EvalkeepError):
|
|
40
|
+
"""The command ran but some input records were rejected."""
|
|
41
|
+
|
|
42
|
+
exit_code = ExitCode.RECORD_ERRORS
|