evalkeep 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. evalkeep/__init__.py +12 -0
  2. evalkeep/__main__.py +6 -0
  3. evalkeep/adapters/__init__.py +45 -0
  4. evalkeep/adapters/base.py +92 -0
  5. evalkeep/adapters/jsonl.py +164 -0
  6. evalkeep/adapters/langsmith.py +436 -0
  7. evalkeep/adapters/otlp.py +442 -0
  8. evalkeep/adapters/semconv.py +208 -0
  9. evalkeep/analysis.py +174 -0
  10. evalkeep/analysis_run.py +160 -0
  11. evalkeep/analyzers/__init__.py +52 -0
  12. evalkeep/analyzers/anthropic.py +145 -0
  13. evalkeep/analyzers/stub.py +34 -0
  14. evalkeep/cache.py +122 -0
  15. evalkeep/cli.py +1933 -0
  16. evalkeep/clustering.py +383 -0
  17. evalkeep/clusters.py +101 -0
  18. evalkeep/commands/__init__.py +1 -0
  19. evalkeep/commands/analyze_cmd.py +100 -0
  20. evalkeep/commands/compare_cmd.py +169 -0
  21. evalkeep/commands/dataset_cmd.py +182 -0
  22. evalkeep/commands/detect_cmd.py +154 -0
  23. evalkeep/commands/discover_cmd.py +274 -0
  24. evalkeep/commands/ingest_cmd.py +50 -0
  25. evalkeep/commands/init_cmd.py +151 -0
  26. evalkeep/commands/pipeline_cmd.py +156 -0
  27. evalkeep/commands/review_cmd.py +141 -0
  28. evalkeep/commands/run_cmd.py +131 -0
  29. evalkeep/commands/target_cmd.py +109 -0
  30. evalkeep/commands/trace_cmd.py +58 -0
  31. evalkeep/comparison.py +432 -0
  32. evalkeep/config.py +209 -0
  33. evalkeep/detection.py +94 -0
  34. evalkeep/detectors.py +182 -0
  35. evalkeep/discovery.py +208 -0
  36. evalkeep/embeddings/__init__.py +31 -0
  37. evalkeep/embeddings/base.py +32 -0
  38. evalkeep/embeddings/hashing.py +98 -0
  39. evalkeep/errors.py +42 -0
  40. evalkeep/examples/__init__.py +37 -0
  41. evalkeep/examples/langsmith/runs.jsonl +18 -0
  42. evalkeep/examples/opentelemetry/spans.json +898 -0
  43. evalkeep/examples/refund-agent/agents/baseline.py +66 -0
  44. evalkeep/examples/refund-agent/agents/candidate.py +66 -0
  45. evalkeep/examples/refund-agent/traces.jsonl +5 -0
  46. evalkeep/examples/tau-bench/prepare.py +230 -0
  47. evalkeep/exporters/__init__.py +45 -0
  48. evalkeep/exporters/generic.py +31 -0
  49. evalkeep/exporters/promptfoo.py +219 -0
  50. evalkeep/failures.py +95 -0
  51. evalkeep/generation.py +303 -0
  52. evalkeep/hashing.py +56 -0
  53. evalkeep/ingest.py +257 -0
  54. evalkeep/prompts.py +127 -0
  55. evalkeep/pseudonyms.py +82 -0
  56. evalkeep/py.typed +0 -0
  57. evalkeep/redaction.py +333 -0
  58. evalkeep/regression.py +409 -0
  59. evalkeep/review.py +309 -0
  60. evalkeep/runner.py +302 -0
  61. evalkeep/runs.py +185 -0
  62. evalkeep/storage/__init__.py +37 -0
  63. evalkeep/storage/clusters.py +163 -0
  64. evalkeep/storage/failures.py +254 -0
  65. evalkeep/storage/migrations.py +370 -0
  66. evalkeep/storage/regression.py +136 -0
  67. evalkeep/storage/runs.py +223 -0
  68. evalkeep/storage/store.py +429 -0
  69. evalkeep/targets.py +205 -0
  70. evalkeep/trace.py +238 -0
  71. evalkeep-0.1.0.dist-info/METADATA +221 -0
  72. evalkeep-0.1.0.dist-info/RECORD +75 -0
  73. evalkeep-0.1.0.dist-info/WHEEL +4 -0
  74. evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
  75. evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/detection.py ADDED
@@ -0,0 +1,94 @@
1
+ """The detection pass: turn stored traces into evidence-backed failure candidates.
2
+
3
+ Detection is idempotent, and the rules that make it so are the whole design:
4
+
5
+ * A failure's ID is a pure function of its trace ID, so a second pass addresses
6
+ the same row instead of creating another one.
7
+ * Signals are *derived*. Every pass recomputes and replaces them, so changing or
8
+ adding a detector updates the evidence on existing failures.
9
+ * A review is *not* derived. Once a person confirms or dismisses a failure,
10
+ detection updates its signals and leaves the decision, reviewer and reason
11
+ alone -- automated analysis never overwrites a human judgement.
12
+ * A candidate nobody has reviewed, whose evidence has since disappeared (a
13
+ detector was removed or narrowed), is withdrawn. Nothing human is lost,
14
+ because nothing human was there. Manually added failures are never withdrawn.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from collections.abc import Iterable
20
+ from dataclasses import dataclass, field
21
+
22
+ from evalkeep.detectors import DETECTORS, FailureDetector, SignalKind, detect_signals
23
+ from evalkeep.failures import Failure, FailureOrigin
24
+ from evalkeep.storage import TraceStore
25
+
26
+
27
+ @dataclass
28
+ class DetectionReport:
29
+ """What one detection pass did."""
30
+
31
+ traces: int = 0
32
+ created: int = 0
33
+ updated: int = 0
34
+ unchanged: int = 0
35
+ withdrawn: int = 0
36
+ preserved_reviews: int = 0
37
+ signals: int = 0
38
+ by_kind: dict[SignalKind, int] = field(default_factory=dict)
39
+
40
+ @property
41
+ def failures(self) -> int:
42
+ """Traces that came out of this pass carrying evidence."""
43
+ return self.created + self.updated + self.unchanged
44
+
45
+ @property
46
+ def changed(self) -> bool:
47
+ return bool(self.created or self.updated or self.withdrawn)
48
+
49
+
50
+ def detect_failures(
51
+ store: TraceStore, *, detectors: Iterable[FailureDetector] = DETECTORS
52
+ ) -> DetectionReport:
53
+ """Run every detector over every stored trace and persist the result."""
54
+ detectors = tuple(detectors)
55
+ report = DetectionReport()
56
+ failures = store.failures
57
+
58
+ for trace in store.iter_traces():
59
+ report.traces += 1
60
+ signals = detect_signals(trace, detectors)
61
+ existing = failures.get_by_trace(trace.trace_id)
62
+
63
+ if not signals:
64
+ # No evidence. Withdraw an unreviewed detector candidate; leave
65
+ # anything a person touched or created.
66
+ if (
67
+ existing is not None
68
+ and not existing.reviewed
69
+ and existing.origin is FailureOrigin.DETECTOR
70
+ ):
71
+ failures.delete(existing.failure_id)
72
+ report.withdrawn += 1
73
+ continue
74
+
75
+ report.signals += len(signals)
76
+ for signal in signals:
77
+ report.by_kind[signal.kind] = report.by_kind.get(signal.kind, 0) + 1
78
+
79
+ if existing is None:
80
+ failures.save(Failure.from_signals(trace.trace_id, signals))
81
+ report.created += 1
82
+ continue
83
+
84
+ if existing.reviewed:
85
+ report.preserved_reviews += 1
86
+ if [s.to_dict() for s in existing.signals] == [s.to_dict() for s in signals]:
87
+ report.unchanged += 1
88
+ continue
89
+
90
+ existing.signals = signals
91
+ failures.save(existing)
92
+ report.updated += 1
93
+
94
+ return report
evalkeep/detectors.py ADDED
@@ -0,0 +1,182 @@
1
+ """Failure detectors: evidence that something went wrong, never a guess.
2
+
3
+ A detector reads one stored (already redacted) trace and emits a
4
+ :class:`Signal` for each piece of explicit evidence it finds. Detectors do not
5
+ infer failure from the shape of an interaction, and they do not score their
6
+ confidence -- a signal either points at something a person recorded, or it does
7
+ not exist. Combining signals is therefore counting evidence, not adding
8
+ probabilities: a trace with three signals is better documented than one with a
9
+ single signal, not "more likely" to be a failure.
10
+
11
+ Adding a detector means implementing :class:`FailureDetector` and registering
12
+ it; nothing else in the pipeline changes.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from collections.abc import Iterable, Iterator
18
+ from dataclasses import dataclass, field
19
+ from enum import StrEnum
20
+ from typing import Any, ClassVar, Protocol, runtime_checkable
21
+
22
+ from evalkeep.trace import EvaluationEvent, NormalizedTrace, OutcomeStatus
23
+
24
+
25
+ class SignalKind(StrEnum):
26
+ """The category of evidence, independent of which detector found it."""
27
+
28
+ EXPLICIT_STATUS = "explicit_status"
29
+ NEGATIVE_FEEDBACK = "negative_feedback"
30
+ FAILED_EVALUATOR = "failed_evaluator"
31
+
32
+
33
+ @dataclass(frozen=True)
34
+ class Signal:
35
+ """One piece of evidence, with a pointer back to where it came from."""
36
+
37
+ detector: str
38
+ kind: SignalKind
39
+ #: Path into the trace, so a reviewer can check the claim themselves.
40
+ source: str
41
+ summary: str
42
+ evidence: dict[str, Any] = field(default_factory=dict)
43
+
44
+ def to_dict(self) -> dict[str, Any]:
45
+ return {
46
+ "detector": self.detector,
47
+ "kind": self.kind.value,
48
+ "source": self.source,
49
+ "summary": self.summary,
50
+ "evidence": self.evidence,
51
+ }
52
+
53
+ @classmethod
54
+ def from_dict(cls, payload: dict[str, Any]) -> Signal:
55
+ return cls(
56
+ detector=payload["detector"],
57
+ kind=SignalKind(payload["kind"]),
58
+ source=payload["source"],
59
+ summary=payload["summary"],
60
+ evidence=payload.get("evidence", {}),
61
+ )
62
+
63
+
64
+ @runtime_checkable
65
+ class FailureDetector(Protocol):
66
+ """Reads a trace, yields evidence. Never raises on well-formed input."""
67
+
68
+ name: ClassVar[str]
69
+ description: ClassVar[str]
70
+
71
+ def detect(self, trace: NormalizedTrace) -> Iterable[Signal]:
72
+ """Yield one signal per piece of evidence found in ``trace``."""
73
+ ...
74
+
75
+
76
+ class ExplicitStatusDetector:
77
+ """The recorded outcome says the interaction failed."""
78
+
79
+ name: ClassVar[str] = "explicit_status"
80
+ description: ClassVar[str] = "outcome.status is 'failure' or 'error'"
81
+
82
+ _FAILING: ClassVar[dict[OutcomeStatus, str]] = {
83
+ OutcomeStatus.FAILURE: "the trace is explicitly marked as failed",
84
+ OutcomeStatus.ERROR: "the trace is explicitly marked as errored",
85
+ }
86
+
87
+ def detect(self, trace: NormalizedTrace) -> Iterator[Signal]:
88
+ summary = self._FAILING.get(trace.outcome.status)
89
+ if summary is None:
90
+ return
91
+ yield Signal(
92
+ detector=self.name,
93
+ kind=SignalKind.EXPLICIT_STATUS,
94
+ source="outcome.status",
95
+ summary=summary,
96
+ evidence={"status": trace.outcome.status.value},
97
+ )
98
+
99
+
100
+ class NegativeFeedbackDetector:
101
+ """Structured feedback records a negative rating.
102
+
103
+ Only an explicit ``rating`` of ``negative`` counts. A bare numeric score
104
+ arrives without its scale -- 2 could be poor out of 5 or good out of 3 --
105
+ and inventing a threshold would be exactly the kind of blind scoring this
106
+ pipeline is meant to avoid.
107
+ """
108
+
109
+ name: ClassVar[str] = "negative_feedback"
110
+ description: ClassVar[str] = "outcome.feedback.rating is 'negative'"
111
+
112
+ def detect(self, trace: NormalizedTrace) -> Iterator[Signal]:
113
+ feedback = trace.outcome.feedback
114
+ if feedback is None or feedback.rating != "negative":
115
+ return
116
+ evidence: dict[str, Any] = {"rating": feedback.rating}
117
+ if feedback.comment:
118
+ evidence["comment"] = feedback.comment
119
+ if feedback.score is not None:
120
+ evidence["score"] = feedback.score
121
+ yield Signal(
122
+ detector=self.name,
123
+ kind=SignalKind.NEGATIVE_FEEDBACK,
124
+ source="outcome.feedback",
125
+ summary=feedback.comment or "feedback was rated negative",
126
+ evidence=evidence,
127
+ )
128
+
129
+
130
+ class FailedEvaluatorDetector:
131
+ """An evaluator recorded a failing verdict, in the outcome or in an event."""
132
+
133
+ name: ClassVar[str] = "failed_evaluator"
134
+ description: ClassVar[str] = "an evaluation recorded passed=false"
135
+
136
+ def detect(self, trace: NormalizedTrace) -> Iterator[Signal]:
137
+ for index, evaluation in enumerate(trace.outcome.evaluations):
138
+ if evaluation.passed is False:
139
+ yield self._signal(
140
+ source=f"outcome.evaluations.{index}",
141
+ name=evaluation.name,
142
+ reason=evaluation.reason,
143
+ score=evaluation.score,
144
+ )
145
+ for index, event in enumerate(trace.events):
146
+ if isinstance(event, EvaluationEvent) and event.passed is False:
147
+ yield self._signal(
148
+ source=f"events.{index}",
149
+ name=event.name,
150
+ reason=event.reason,
151
+ score=event.score,
152
+ )
153
+
154
+ def _signal(self, *, source: str, name: str, reason: str | None, score: float | None) -> Signal:
155
+ evidence: dict[str, Any] = {"evaluator": name, "passed": False}
156
+ if reason:
157
+ evidence["reason"] = reason
158
+ if score is not None:
159
+ evidence["score"] = score
160
+ return Signal(
161
+ detector=self.name,
162
+ kind=SignalKind.FAILED_EVALUATOR,
163
+ source=source,
164
+ summary=reason or f"evaluator {name!r} reported a failure",
165
+ evidence=evidence,
166
+ )
167
+
168
+
169
+ #: The detectors run by ``evalkeep detect``, in a fixed order so that the
170
+ #: signals persisted for a trace are reproducible.
171
+ DETECTORS: tuple[FailureDetector, ...] = (
172
+ ExplicitStatusDetector(),
173
+ NegativeFeedbackDetector(),
174
+ FailedEvaluatorDetector(),
175
+ )
176
+
177
+
178
+ def detect_signals(
179
+ trace: NormalizedTrace, detectors: Iterable[FailureDetector] = DETECTORS
180
+ ) -> list[Signal]:
181
+ """Every signal every detector finds in ``trace``, in detector order."""
182
+ return [signal for detector in detectors for signal in detector.detect(trace)]
evalkeep/discovery.py ADDED
@@ -0,0 +1,208 @@
1
+ """The discovery pass: embed analyzed failures, group them, pick representatives.
2
+
3
+ Discovery is derived from analysis, so it can always be recomputed. What cannot
4
+ be recomputed is what a reviewer did to the result -- renaming a family,
5
+ dismissing one, merging two that the distance metric kept apart. Those edits are
6
+ preserved where the group survives unchanged, and re-clustering refuses to
7
+ discard them without an explicit ``--force``.
8
+
9
+ Cluster identity is derived from membership, which is what makes that possible:
10
+ re-running on unchanged data produces the same cluster IDs, so labels stay
11
+ attached to the families they were written for.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import uuid
17
+ from dataclasses import dataclass, field
18
+
19
+ from evalkeep.cache import EmbeddingCache, embedding_key
20
+ from evalkeep.clustering import (
21
+ ClusterInput,
22
+ build_clusters,
23
+ clustering_parameters,
24
+ observation_text,
25
+ )
26
+ from evalkeep.clusters import Cluster, ClusteringRun
27
+ from evalkeep.config import ClusteringConfig
28
+ from evalkeep.embeddings import EmbeddingProvider
29
+ from evalkeep.failures import FailureStatus
30
+ from evalkeep.storage import TraceStore
31
+
32
+ #: Dismissed failures are not part of any family worth covering.
33
+ CLUSTERABLE = (FailureStatus.CANDIDATE, FailureStatus.CONFIRMED)
34
+
35
+
36
+ @dataclass
37
+ class DiscoveryReport:
38
+ """What one discovery pass did."""
39
+
40
+ embedder: str
41
+ considered: int = 0
42
+ unanalyzed: int = 0
43
+ clusters: int = 0
44
+ singletons: int = 0
45
+ representatives: int = 0
46
+ embedded: int = 0
47
+ from_cache: int = 0
48
+ kept_labels: int = 0
49
+ discarded_edits: int = 0
50
+ #: Failures grouped by observed behaviour because nobody described them.
51
+ grouped_undescribed: int = 0
52
+ largest: int = 0
53
+ parameters: dict[str, object] = field(default_factory=dict)
54
+
55
+
56
+ class ClusterEditsWouldBeLost(Exception):
57
+ """Re-clustering would discard reviewer edits that cannot be carried over."""
58
+
59
+ def __init__(self, clusters: list[Cluster]) -> None:
60
+ self.clusters = clusters
61
+ super().__init__(f"{len(clusters)} edited clusters would be discarded")
62
+
63
+
64
+ def discover(
65
+ store: TraceStore,
66
+ embedder: EmbeddingProvider,
67
+ cache: EmbeddingCache,
68
+ config: ClusteringConfig,
69
+ *,
70
+ force: bool = False,
71
+ group_undescribed: bool = False,
72
+ ) -> DiscoveryReport:
73
+ """Cluster analyzed failures and select representatives."""
74
+ report = DiscoveryReport(embedder=embedder.identity)
75
+ report.parameters = dict(clustering_parameters(config))
76
+
77
+ inputs = _gather(store, report, group_undescribed=group_undescribed)
78
+ vectors = _embed(inputs, embedder, cache, report)
79
+ clusters = build_clusters(inputs, vectors, config)
80
+
81
+ _carry_over_edits(store, clusters, report, force=force)
82
+
83
+ run = ClusteringRun(
84
+ run_id=uuid.uuid4().hex,
85
+ embedder=embedder.identity,
86
+ dimensions=embedder.dimensions,
87
+ parameters=report.parameters,
88
+ failures=len(inputs),
89
+ )
90
+ store.clusters.replace_run(run, clusters)
91
+
92
+ report.clusters = len(clusters)
93
+ report.singletons = sum(1 for cluster in clusters if cluster.size == 1)
94
+ report.representatives = sum(len(cluster.representatives) for cluster in clusters)
95
+ report.largest = max((cluster.size for cluster in clusters), default=0)
96
+ return report
97
+
98
+
99
+ def _gather(
100
+ store: TraceStore, report: DiscoveryReport, *, group_undescribed: bool = False
101
+ ) -> list[ClusterInput]:
102
+ """Undismissed failures, grouped by description where one exists.
103
+
104
+ An undescribed failure is skipped by default: grouping by observed behaviour
105
+ is weaker than grouping by what a failure *is*, and silently mixing the two
106
+ would make a report mean two different things. ``group_undescribed`` opts
107
+ in, and the counts say which was used.
108
+ """
109
+ inputs: list[ClusterInput] = []
110
+ for failure in store.failures.iter_all():
111
+ if failure.status not in CLUSTERABLE:
112
+ continue
113
+ report.considered += 1
114
+ analysis = store.failures.get_analysis(failure.failure_id)
115
+ if analysis is not None:
116
+ inputs.append(ClusterInput.from_analysis(failure.failure_id, analysis))
117
+ continue
118
+
119
+ report.unanalyzed += 1
120
+ if not group_undescribed:
121
+ continue
122
+ stored = store.get(failure.trace_id)
123
+ if stored is None: # pragma: no cover - the foreign key prevents this
124
+ continue
125
+ tools = sorted({call.tool for call in stored.trace.tool_calls})
126
+ inputs.append(
127
+ ClusterInput.from_observation(
128
+ failure.failure_id,
129
+ behaviour=", ".join(tools) or None,
130
+ text=observation_text(
131
+ [call.tool for call in stored.trace.tool_calls],
132
+ [signal.kind.value for signal in failure.signals],
133
+ [signal.summary for signal in failure.signals],
134
+ ),
135
+ )
136
+ )
137
+ report.grouped_undescribed += 1
138
+ # A stable input order keeps the clustering reproducible.
139
+ inputs.sort(key=lambda item: item.failure_id)
140
+ return inputs
141
+
142
+
143
+ def _embed(
144
+ inputs: list[ClusterInput],
145
+ embedder: EmbeddingProvider,
146
+ cache: EmbeddingCache,
147
+ report: DiscoveryReport,
148
+ ) -> list[list[float]]:
149
+ """Embed, reusing cached vectors for text that has not changed."""
150
+ vectors: list[list[float] | None] = []
151
+ pending: list[int] = []
152
+
153
+ for index, item in enumerate(inputs):
154
+ cached = cache.get_vector(embedding_key(item.text, embedder.identity))
155
+ vectors.append(cached)
156
+ if cached is None:
157
+ pending.append(index)
158
+
159
+ if pending:
160
+ fresh = embedder.embed([inputs[index].text for index in pending])
161
+ for index, vector in zip(pending, fresh, strict=True):
162
+ vectors[index] = vector
163
+ cache.put_vector(embedding_key(inputs[index].text, embedder.identity), vector)
164
+
165
+ report.embedded = len(pending)
166
+ report.from_cache = len(inputs) - len(pending)
167
+ return [vector for vector in vectors if vector is not None]
168
+
169
+
170
+ def _carry_over_edits(
171
+ store: TraceStore,
172
+ clusters: list[Cluster],
173
+ report: DiscoveryReport,
174
+ *,
175
+ force: bool,
176
+ ) -> None:
177
+ """Reattach reviewer edits to families that came back unchanged.
178
+
179
+ A cluster ID is a function of its members, so a family whose membership did
180
+ not change keeps its ID -- and with it, its name and its dismissal. A family
181
+ whose membership *did* change is genuinely a different group, and its edits
182
+ cannot be carried over honestly. Rather than dropping them quietly, the pass
183
+ refuses until the caller says to proceed.
184
+ """
185
+ existing = {cluster.cluster_id: cluster for cluster in store.clusters.list()}
186
+ if not existing:
187
+ return
188
+
189
+ rebuilt = {cluster.cluster_id: cluster for cluster in clusters}
190
+ for cluster_id, previous in existing.items():
191
+ if not previous.edited:
192
+ continue
193
+ survivor = rebuilt.get(cluster_id)
194
+ if survivor is None:
195
+ report.discarded_edits += 1
196
+ continue
197
+ survivor.label = previous.label
198
+ survivor.labelled_by = previous.labelled_by
199
+ survivor.dismissed = previous.dismissed
200
+ report.kept_labels += 1
201
+
202
+ if report.discarded_edits and not force:
203
+ lost = [
204
+ cluster
205
+ for cluster_id, cluster in existing.items()
206
+ if cluster.edited and cluster_id not in rebuilt
207
+ ]
208
+ raise ClusterEditsWouldBeLost(lost)
@@ -0,0 +1,31 @@
1
+ """Embedding providers and the registry the project configuration resolves against."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from evalkeep.config import ClusteringConfig
6
+ from evalkeep.embeddings.base import EmbeddingProvider
7
+ from evalkeep.embeddings.hashing import HashingEmbedder
8
+ from evalkeep.errors import CommandError
9
+
10
+ KNOWN_EMBEDDERS: dict[str, str] = {HashingEmbedder.name: HashingEmbedder.description}
11
+
12
+
13
+ def get_embedder(config: ClusteringConfig) -> EmbeddingProvider:
14
+ """Build the configured embedding provider.
15
+
16
+ A hosted embedding model plugs in here: implement
17
+ :class:`~evalkeep.embeddings.base.EmbeddingProvider`, give it an identity
18
+ that names the model, and register it below. Nothing downstream changes --
19
+ the cache and the stored clustering parameters key off ``identity``, so old
20
+ and new vectors can never be silently mixed.
21
+ """
22
+ if config.embedder == HashingEmbedder.name:
23
+ return HashingEmbedder(dimensions=config.dimensions, seed=config.seed)
24
+ known = ", ".join(sorted(KNOWN_EMBEDDERS))
25
+ raise CommandError(
26
+ f"Unknown embedding provider {config.embedder!r}.",
27
+ hint=f"Known providers: {known}.",
28
+ )
29
+
30
+
31
+ __all__ = ["KNOWN_EMBEDDERS", "EmbeddingProvider", "HashingEmbedder", "get_embedder"]
@@ -0,0 +1,32 @@
1
+ """The embedding provider contract.
2
+
3
+ Embeddings turn failure descriptions into vectors so that similar failures land
4
+ near each other. The interface is deliberately tiny -- one method over a batch
5
+ of strings -- so a hosted embedding model can replace the built-in one without
6
+ anything else in the pipeline changing.
7
+
8
+ ``identity`` is part of every cache key and is stored with every clustering run.
9
+ It must change whenever the vectors would change: a different model, a different
10
+ dimensionality or a different seed is a different embedding space, and mixing
11
+ two of them silently would produce meaningless distances.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ from typing import ClassVar, Protocol, runtime_checkable
17
+
18
+
19
+ @runtime_checkable
20
+ class EmbeddingProvider(Protocol):
21
+ name: ClassVar[str]
22
+ description: ClassVar[str]
23
+
24
+ @property
25
+ def identity(self) -> str: ...
26
+
27
+ @property
28
+ def dimensions(self) -> int: ...
29
+
30
+ def embed(self, texts: list[str]) -> list[list[float]]:
31
+ """Embed a batch of texts, returning one L2-normalized vector each."""
32
+ ...
@@ -0,0 +1,98 @@
1
+ """A deterministic, offline embedding provider using the hashing trick.
2
+
3
+ Tokens (and adjacent token pairs) are hashed to fixed positions in a vector with
4
+ a signed contribution, then the vector is L2-normalized so cosine similarity is
5
+ a dot product. This is feature hashing, a real technique -- not a stub -- but it
6
+ is worth being precise about what it does and does not capture:
7
+
8
+ * It captures **lexical** overlap. Two summaries describing the same mistake in
9
+ similar words land close together. Bigrams mean word order carries some
10
+ weight, so "refunded the oldest order" and "ordered the oldest refund" differ.
11
+ * It does **not** capture meaning. "refunded the wrong order" and "issued a
12
+ credit for the incorrect purchase" share almost no tokens and will not group,
13
+ where a trained embedding model would place them together.
14
+
15
+ That tradeoff is acceptable here because the text being embedded is not free
16
+ prose: it is a structured analysis whose failure type and component are part of
17
+ the string, written under a prompt that explicitly asks for wording that repeats
18
+ across the same failure family. Lexical similarity does real work on that input.
19
+
20
+ Determinism is the point. ``hash()`` is randomized per process in Python and
21
+ must never be used; BLAKE2b keyed by the configured seed gives the same vector
22
+ on every machine and every run, which is what makes a clustering reproducible.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import hashlib
28
+ import math
29
+ import re
30
+ from itertools import pairwise
31
+ from typing import ClassVar
32
+
33
+ DEFAULT_DIMENSIONS = 512
34
+ DEFAULT_SEED = 0
35
+
36
+ _TOKEN_PATTERN = re.compile(r"[a-z0-9]+")
37
+
38
+
39
+ class HashingEmbedder:
40
+ name: ClassVar[str] = "hashing"
41
+ description: ClassVar[str] = "Deterministic offline feature hashing (lexical similarity)"
42
+
43
+ def __init__(self, *, dimensions: int = DEFAULT_DIMENSIONS, seed: int = DEFAULT_SEED) -> None:
44
+ if dimensions < 8:
45
+ raise ValueError("dimensions must be at least 8")
46
+ self._dimensions = dimensions
47
+ self.seed = seed
48
+
49
+ @property
50
+ def identity(self) -> str:
51
+ return f"{self.name}:{self._dimensions}:{self.seed}"
52
+
53
+ @property
54
+ def dimensions(self) -> int:
55
+ return self._dimensions
56
+
57
+ def embed(self, texts: list[str]) -> list[list[float]]:
58
+ return [self._embed_one(text) for text in texts]
59
+
60
+ def _embed_one(self, text: str) -> list[float]:
61
+ vector = [0.0] * self._dimensions
62
+ for feature, weight in _features(text).items():
63
+ index, sign = self._position(feature)
64
+ vector[index] += sign * weight
65
+ return _normalize(vector)
66
+
67
+ def _position(self, feature: str) -> tuple[int, int]:
68
+ """Hash a feature to a bucket and a sign.
69
+
70
+ The sign halves the bias from collisions: two unrelated features landing
71
+ in the same bucket are as likely to cancel as to reinforce.
72
+ """
73
+ digest = hashlib.blake2b(
74
+ feature.encode("utf-8"), digest_size=9, key=str(self.seed).encode("utf-8")
75
+ ).digest()
76
+ value = int.from_bytes(digest, "big")
77
+ return value % self._dimensions, 1 if (value >> 63) & 1 else -1
78
+
79
+
80
+ def _features(text: str) -> dict[str, float]:
81
+ """Unigrams and bigrams, with sublinear term weighting."""
82
+ tokens = _TOKEN_PATTERN.findall(text.lower())
83
+ counts: dict[str, int] = {}
84
+ for token in tokens:
85
+ counts[token] = counts.get(token, 0) + 1
86
+ for first, second in pairwise(tokens):
87
+ bigram = f"{first}_{second}"
88
+ counts[bigram] = counts.get(bigram, 0) + 1
89
+ # 1 + log(count): a word repeated ten times matters more than once, but not
90
+ # ten times more, so one long quoted string cannot dominate a summary.
91
+ return {feature: 1.0 + math.log(count) for feature, count in counts.items()}
92
+
93
+
94
+ def _normalize(vector: list[float]) -> list[float]:
95
+ magnitude = math.sqrt(sum(value * value for value in vector))
96
+ if magnitude == 0.0:
97
+ return vector
98
+ return [value / magnitude for value in vector]
evalkeep/errors.py ADDED
@@ -0,0 +1,42 @@
1
+ """Stable exit codes and the error type that carries them.
2
+
3
+ The codes are part of the CLI contract and are relied on by scripts and CI:
4
+
5
+ * ``0`` -- the command succeeded.
6
+ * ``1`` -- the command ran, but some records were invalid or rejected.
7
+ * ``2`` -- the command itself could not run (bad usage, missing file,
8
+ unwritable directory, uninitialized project).
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from enum import IntEnum
14
+
15
+
16
+ class ExitCode(IntEnum):
17
+ OK = 0
18
+ RECORD_ERRORS = 1
19
+ COMMAND_ERROR = 2
20
+
21
+
22
+ class EvalkeepError(Exception):
23
+ """An error that maps to a stable exit code and a readable message."""
24
+
25
+ exit_code: ExitCode = ExitCode.COMMAND_ERROR
26
+
27
+ def __init__(self, message: str, *, hint: str | None = None) -> None:
28
+ super().__init__(message)
29
+ self.message = message
30
+ self.hint = hint
31
+
32
+
33
+ class CommandError(EvalkeepError):
34
+ """The command could not run at all."""
35
+
36
+ exit_code = ExitCode.COMMAND_ERROR
37
+
38
+
39
+ class RecordError(EvalkeepError):
40
+ """The command ran but some input records were rejected."""
41
+
42
+ exit_code = ExitCode.RECORD_ERRORS