evalkeep 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalkeep/__init__.py +12 -0
- evalkeep/__main__.py +6 -0
- evalkeep/adapters/__init__.py +45 -0
- evalkeep/adapters/base.py +92 -0
- evalkeep/adapters/jsonl.py +164 -0
- evalkeep/adapters/langsmith.py +436 -0
- evalkeep/adapters/otlp.py +442 -0
- evalkeep/adapters/semconv.py +208 -0
- evalkeep/analysis.py +174 -0
- evalkeep/analysis_run.py +160 -0
- evalkeep/analyzers/__init__.py +52 -0
- evalkeep/analyzers/anthropic.py +145 -0
- evalkeep/analyzers/stub.py +34 -0
- evalkeep/cache.py +122 -0
- evalkeep/cli.py +1933 -0
- evalkeep/clustering.py +383 -0
- evalkeep/clusters.py +101 -0
- evalkeep/commands/__init__.py +1 -0
- evalkeep/commands/analyze_cmd.py +100 -0
- evalkeep/commands/compare_cmd.py +169 -0
- evalkeep/commands/dataset_cmd.py +182 -0
- evalkeep/commands/detect_cmd.py +154 -0
- evalkeep/commands/discover_cmd.py +274 -0
- evalkeep/commands/ingest_cmd.py +50 -0
- evalkeep/commands/init_cmd.py +151 -0
- evalkeep/commands/pipeline_cmd.py +156 -0
- evalkeep/commands/review_cmd.py +141 -0
- evalkeep/commands/run_cmd.py +131 -0
- evalkeep/commands/target_cmd.py +109 -0
- evalkeep/commands/trace_cmd.py +58 -0
- evalkeep/comparison.py +432 -0
- evalkeep/config.py +209 -0
- evalkeep/detection.py +94 -0
- evalkeep/detectors.py +182 -0
- evalkeep/discovery.py +208 -0
- evalkeep/embeddings/__init__.py +31 -0
- evalkeep/embeddings/base.py +32 -0
- evalkeep/embeddings/hashing.py +98 -0
- evalkeep/errors.py +42 -0
- evalkeep/examples/__init__.py +37 -0
- evalkeep/examples/langsmith/runs.jsonl +18 -0
- evalkeep/examples/opentelemetry/spans.json +898 -0
- evalkeep/examples/refund-agent/agents/baseline.py +66 -0
- evalkeep/examples/refund-agent/agents/candidate.py +66 -0
- evalkeep/examples/refund-agent/traces.jsonl +5 -0
- evalkeep/examples/tau-bench/prepare.py +230 -0
- evalkeep/exporters/__init__.py +45 -0
- evalkeep/exporters/generic.py +31 -0
- evalkeep/exporters/promptfoo.py +219 -0
- evalkeep/failures.py +95 -0
- evalkeep/generation.py +303 -0
- evalkeep/hashing.py +56 -0
- evalkeep/ingest.py +257 -0
- evalkeep/prompts.py +127 -0
- evalkeep/pseudonyms.py +82 -0
- evalkeep/py.typed +0 -0
- evalkeep/redaction.py +333 -0
- evalkeep/regression.py +409 -0
- evalkeep/review.py +309 -0
- evalkeep/runner.py +302 -0
- evalkeep/runs.py +185 -0
- evalkeep/storage/__init__.py +37 -0
- evalkeep/storage/clusters.py +163 -0
- evalkeep/storage/failures.py +254 -0
- evalkeep/storage/migrations.py +370 -0
- evalkeep/storage/regression.py +136 -0
- evalkeep/storage/runs.py +223 -0
- evalkeep/storage/store.py +429 -0
- evalkeep/targets.py +205 -0
- evalkeep/trace.py +238 -0
- evalkeep-0.1.0.dist-info/METADATA +221 -0
- evalkeep-0.1.0.dist-info/RECORD +75 -0
- evalkeep-0.1.0.dist-info/WHEEL +4 -0
- evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
- evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/regression.py
ADDED
|
@@ -0,0 +1,409 @@
|
|
|
1
|
+
"""Regression tests: what a failure must never do again.
|
|
2
|
+
|
|
3
|
+
A regression test preserves an observed failure as a permanent check. The shape
|
|
4
|
+
of one follows from what a trace can and cannot tell us:
|
|
5
|
+
|
|
6
|
+
* A trace shows exactly **what the agent did wrong**, so the forbidding half of
|
|
7
|
+
a test -- "do not refund an older order" -- is derivable and deterministic.
|
|
8
|
+
* A trace does **not** show what the agent should have done instead. "Refund
|
|
9
|
+
exactly the newest order" is a judgement about intent that no amount of
|
|
10
|
+
reading the trace can supply.
|
|
11
|
+
|
|
12
|
+
So generation produces the forbidding half automatically and marks the test as
|
|
13
|
+
needing the positive half from a reviewer. That split is why drafts are never
|
|
14
|
+
exported without approval.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import hashlib
|
|
20
|
+
import re
|
|
21
|
+
from dataclasses import dataclass, field
|
|
22
|
+
from datetime import UTC, datetime
|
|
23
|
+
from enum import StrEnum
|
|
24
|
+
from itertools import combinations
|
|
25
|
+
from typing import Any
|
|
26
|
+
|
|
27
|
+
MAX_SLUG_WORDS = 6
|
|
28
|
+
MAX_SLUG_LENGTH = 48
|
|
29
|
+
_TEST_ID_DIGEST_LENGTH = 8
|
|
30
|
+
|
|
31
|
+
_WORD_PATTERN = re.compile(r"[a-z0-9]+")
|
|
32
|
+
_REDACTION_PATTERN = re.compile(r"\[REDACTED:[a-z_]+\]", re.IGNORECASE)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class ExpectationType(StrEnum):
|
|
36
|
+
OUTPUT_CONTAINS = "output_contains"
|
|
37
|
+
OUTPUT_NOT_CONTAINS = "output_not_contains"
|
|
38
|
+
OUTPUT_MATCHES = "output_matches"
|
|
39
|
+
TOOL_CALLED = "tool_called"
|
|
40
|
+
TOOL_NOT_CALLED = "tool_not_called"
|
|
41
|
+
TOOL_ARGUMENT_EQUALS = "tool_argument_equals"
|
|
42
|
+
TOOL_ARGUMENT_NOT_EQUALS = "tool_argument_not_equals"
|
|
43
|
+
MAX_TOOL_CALLS = "max_tool_calls"
|
|
44
|
+
#: The escape hatch for cases with no deterministic check. Costs an LLM
|
|
45
|
+
#: judge at run time, so it is used only where nothing else applies.
|
|
46
|
+
HUMAN_RUBRIC = "human_rubric"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
#: Expectations that say what the agent *should* do. A test made only of the
|
|
50
|
+
#: others records a prohibition without an intent, which is half a test.
|
|
51
|
+
POSITIVE_TYPES: frozenset[ExpectationType] = frozenset(
|
|
52
|
+
{
|
|
53
|
+
ExpectationType.OUTPUT_CONTAINS,
|
|
54
|
+
ExpectationType.OUTPUT_MATCHES,
|
|
55
|
+
ExpectationType.TOOL_CALLED,
|
|
56
|
+
ExpectationType.TOOL_ARGUMENT_EQUALS,
|
|
57
|
+
ExpectationType.HUMAN_RUBRIC,
|
|
58
|
+
}
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
#: Everything except the rubric can be checked without a model.
|
|
62
|
+
DETERMINISTIC_TYPES: frozenset[ExpectationType] = frozenset(
|
|
63
|
+
set(ExpectationType) - {ExpectationType.HUMAN_RUBRIC}
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class ReviewStatus(StrEnum):
|
|
68
|
+
"""The review lifecycle. Generation only ever writes ``DRAFT``."""
|
|
69
|
+
|
|
70
|
+
DRAFT = "draft"
|
|
71
|
+
APPROVED = "approved"
|
|
72
|
+
REJECTED = "rejected"
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass(frozen=True)
|
|
76
|
+
class Expectation:
|
|
77
|
+
type: ExpectationType
|
|
78
|
+
value: Any = None
|
|
79
|
+
tool: str | None = None
|
|
80
|
+
#: Argument path within a tool call, dotted for nested values.
|
|
81
|
+
path: str | None = None
|
|
82
|
+
|
|
83
|
+
@property
|
|
84
|
+
def deterministic(self) -> bool:
|
|
85
|
+
return self.type in DETERMINISTIC_TYPES
|
|
86
|
+
|
|
87
|
+
@property
|
|
88
|
+
def positive(self) -> bool:
|
|
89
|
+
return self.type in POSITIVE_TYPES
|
|
90
|
+
|
|
91
|
+
def describe(self) -> str:
|
|
92
|
+
match self.type:
|
|
93
|
+
case ExpectationType.TOOL_ARGUMENT_EQUALS:
|
|
94
|
+
return f"{self.tool}.{self.path} == {self.value!r}"
|
|
95
|
+
case ExpectationType.TOOL_ARGUMENT_NOT_EQUALS:
|
|
96
|
+
return f"{self.tool}.{self.path} != {self.value!r}"
|
|
97
|
+
case ExpectationType.TOOL_CALLED:
|
|
98
|
+
return f"calls {self.tool}"
|
|
99
|
+
case ExpectationType.TOOL_NOT_CALLED:
|
|
100
|
+
return f"never calls {self.tool}"
|
|
101
|
+
case ExpectationType.MAX_TOOL_CALLS:
|
|
102
|
+
target = self.tool or "any tool"
|
|
103
|
+
return f"at most {self.value} calls to {target}"
|
|
104
|
+
case ExpectationType.HUMAN_RUBRIC:
|
|
105
|
+
return f"rubric: {self.value}"
|
|
106
|
+
case _:
|
|
107
|
+
return f"{self.type.value} {self.value!r}"
|
|
108
|
+
|
|
109
|
+
def to_dict(self) -> dict[str, Any]:
|
|
110
|
+
payload: dict[str, Any] = {"type": self.type.value, "value": self.value}
|
|
111
|
+
if self.tool is not None:
|
|
112
|
+
payload["tool"] = self.tool
|
|
113
|
+
if self.path is not None:
|
|
114
|
+
payload["path"] = self.path
|
|
115
|
+
return payload
|
|
116
|
+
|
|
117
|
+
@classmethod
|
|
118
|
+
def from_dict(cls, payload: dict[str, Any]) -> Expectation:
|
|
119
|
+
return cls(
|
|
120
|
+
type=ExpectationType(payload["type"]),
|
|
121
|
+
value=payload.get("value"),
|
|
122
|
+
tool=payload.get("tool"),
|
|
123
|
+
path=payload.get("path"),
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
@dataclass(frozen=True)
|
|
128
|
+
class Contradiction:
|
|
129
|
+
"""Two expectations that cannot both hold."""
|
|
130
|
+
|
|
131
|
+
first: Expectation
|
|
132
|
+
second: Expectation
|
|
133
|
+
reason: str
|
|
134
|
+
|
|
135
|
+
def describe(self) -> str:
|
|
136
|
+
return f"{self.first.describe()} vs {self.second.describe()}: {self.reason}"
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def validate_expectation(expectation: Expectation) -> str | None:
|
|
140
|
+
"""Reasons an expectation could never be evaluated, or ``None`` if it can."""
|
|
141
|
+
match expectation.type:
|
|
142
|
+
case ExpectationType.TOOL_ARGUMENT_EQUALS | ExpectationType.TOOL_ARGUMENT_NOT_EQUALS:
|
|
143
|
+
if not expectation.tool or not expectation.path:
|
|
144
|
+
return "an argument expectation needs both a tool and a path"
|
|
145
|
+
case ExpectationType.TOOL_CALLED | ExpectationType.TOOL_NOT_CALLED:
|
|
146
|
+
if not expectation.tool:
|
|
147
|
+
return "a tool expectation needs a tool"
|
|
148
|
+
case ExpectationType.MAX_TOOL_CALLS:
|
|
149
|
+
if not isinstance(expectation.value, int) or isinstance(expectation.value, bool):
|
|
150
|
+
return "max_tool_calls needs a whole number"
|
|
151
|
+
if expectation.value < 0:
|
|
152
|
+
return "max_tool_calls cannot be negative"
|
|
153
|
+
case ExpectationType.OUTPUT_MATCHES:
|
|
154
|
+
if not isinstance(expectation.value, str):
|
|
155
|
+
return "output_matches needs a pattern"
|
|
156
|
+
try:
|
|
157
|
+
re.compile(expectation.value)
|
|
158
|
+
except re.error as exc:
|
|
159
|
+
return f"output_matches pattern does not compile: {exc}"
|
|
160
|
+
case ExpectationType.OUTPUT_CONTAINS | ExpectationType.OUTPUT_NOT_CONTAINS:
|
|
161
|
+
if not isinstance(expectation.value, str) or not expectation.value.strip():
|
|
162
|
+
return f"{expectation.type.value} needs non-empty text"
|
|
163
|
+
case ExpectationType.HUMAN_RUBRIC:
|
|
164
|
+
if not isinstance(expectation.value, str) or not expectation.value.strip():
|
|
165
|
+
return "human_rubric needs a description of what should happen"
|
|
166
|
+
return None
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def find_contradictions(expectations: list[Expectation]) -> list[Contradiction]:
|
|
170
|
+
"""Every pair of expectations that cannot both be satisfied.
|
|
171
|
+
|
|
172
|
+
Run at generation time so a draft is never saved self-defeating, and again
|
|
173
|
+
at review time so an edit cannot introduce one. A contradictory test fails
|
|
174
|
+
on every agent, including a correct one, which makes it worse than no test:
|
|
175
|
+
it reports a regression that is really a bug in the suite.
|
|
176
|
+
"""
|
|
177
|
+
found: list[Contradiction] = []
|
|
178
|
+
for first, second in combinations(expectations, 2):
|
|
179
|
+
reason = _conflict(first, second) or _conflict(second, first)
|
|
180
|
+
if reason is not None:
|
|
181
|
+
found.append(Contradiction(first=first, second=second, reason=reason))
|
|
182
|
+
return found
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _conflict(one: Expectation, two: Expectation) -> str | None:
|
|
186
|
+
if (
|
|
187
|
+
one.type is ExpectationType.TOOL_CALLED
|
|
188
|
+
and two.type is ExpectationType.TOOL_NOT_CALLED
|
|
189
|
+
and one.tool == two.tool
|
|
190
|
+
):
|
|
191
|
+
return "the same tool cannot be both required and forbidden"
|
|
192
|
+
|
|
193
|
+
if (
|
|
194
|
+
one.type is ExpectationType.TOOL_NOT_CALLED
|
|
195
|
+
and two.type
|
|
196
|
+
in {ExpectationType.TOOL_ARGUMENT_EQUALS, ExpectationType.TOOL_ARGUMENT_NOT_EQUALS}
|
|
197
|
+
and one.tool == two.tool
|
|
198
|
+
):
|
|
199
|
+
return "an argument of a forbidden tool can never be checked"
|
|
200
|
+
|
|
201
|
+
if (
|
|
202
|
+
one.type is ExpectationType.TOOL_CALLED
|
|
203
|
+
and two.type is ExpectationType.MAX_TOOL_CALLS
|
|
204
|
+
and two.value == 0
|
|
205
|
+
and (two.tool is None or two.tool == one.tool)
|
|
206
|
+
):
|
|
207
|
+
return "a required tool cannot also be capped at zero calls"
|
|
208
|
+
|
|
209
|
+
if (
|
|
210
|
+
one.type is ExpectationType.TOOL_ARGUMENT_EQUALS
|
|
211
|
+
and two.type is ExpectationType.TOOL_ARGUMENT_NOT_EQUALS
|
|
212
|
+
and (one.tool, one.path) == (two.tool, two.path)
|
|
213
|
+
and one.value == two.value
|
|
214
|
+
):
|
|
215
|
+
return "the same argument cannot be required and forbidden to equal one value"
|
|
216
|
+
|
|
217
|
+
if (
|
|
218
|
+
one.type is two.type is ExpectationType.TOOL_ARGUMENT_EQUALS
|
|
219
|
+
and (one.tool, one.path) == (two.tool, two.path)
|
|
220
|
+
and one.value != two.value
|
|
221
|
+
):
|
|
222
|
+
return "one argument cannot equal two different values"
|
|
223
|
+
|
|
224
|
+
if (
|
|
225
|
+
one.type is ExpectationType.OUTPUT_CONTAINS
|
|
226
|
+
and two.type is ExpectationType.OUTPUT_NOT_CONTAINS
|
|
227
|
+
and one.value == two.value
|
|
228
|
+
):
|
|
229
|
+
return "the output cannot both contain and not contain the same text"
|
|
230
|
+
|
|
231
|
+
if (
|
|
232
|
+
one.type is two.type is ExpectationType.MAX_TOOL_CALLS
|
|
233
|
+
and one.tool == two.tool
|
|
234
|
+
and one.value != two.value
|
|
235
|
+
):
|
|
236
|
+
return "two different call limits for the same tool"
|
|
237
|
+
|
|
238
|
+
return None
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
@dataclass
|
|
242
|
+
class CaseInput:
|
|
243
|
+
"""What the test sends to the agent."""
|
|
244
|
+
|
|
245
|
+
text: str | None = None
|
|
246
|
+
messages: list[dict[str, str]] = field(default_factory=list)
|
|
247
|
+
|
|
248
|
+
def to_dict(self) -> dict[str, Any]:
|
|
249
|
+
return {"text": self.text, "messages": self.messages}
|
|
250
|
+
|
|
251
|
+
@classmethod
|
|
252
|
+
def from_dict(cls, payload: dict[str, Any]) -> CaseInput:
|
|
253
|
+
return cls(text=payload.get("text"), messages=payload.get("messages", []))
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
@dataclass
|
|
257
|
+
class Fixture:
|
|
258
|
+
"""A tool result the original agent saw, so a replay can reproduce it."""
|
|
259
|
+
|
|
260
|
+
tool: str
|
|
261
|
+
arguments: dict[str, Any] = field(default_factory=dict)
|
|
262
|
+
result: Any = None
|
|
263
|
+
error: str | None = None
|
|
264
|
+
call_id: str | None = None
|
|
265
|
+
|
|
266
|
+
def to_dict(self) -> dict[str, Any]:
|
|
267
|
+
return {
|
|
268
|
+
"tool": self.tool,
|
|
269
|
+
"arguments": self.arguments,
|
|
270
|
+
"result": self.result,
|
|
271
|
+
"error": self.error,
|
|
272
|
+
"call_id": self.call_id,
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
def for_replay(self) -> dict[str, Any]:
|
|
276
|
+
"""What a target needs to reproduce this call.
|
|
277
|
+
|
|
278
|
+
``call_id`` is left out: it identifies the original recording, not the
|
|
279
|
+
interaction, and a replaying target has no use for it.
|
|
280
|
+
"""
|
|
281
|
+
return {
|
|
282
|
+
"tool": self.tool,
|
|
283
|
+
"arguments": self.arguments,
|
|
284
|
+
"result": self.result,
|
|
285
|
+
"error": self.error,
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
@classmethod
|
|
289
|
+
def from_dict(cls, payload: dict[str, Any]) -> Fixture:
|
|
290
|
+
return cls(
|
|
291
|
+
tool=payload["tool"],
|
|
292
|
+
arguments=payload.get("arguments", {}),
|
|
293
|
+
result=payload.get("result"),
|
|
294
|
+
error=payload.get("error"),
|
|
295
|
+
call_id=payload.get("call_id"),
|
|
296
|
+
)
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
@dataclass
|
|
300
|
+
class Provenance:
|
|
301
|
+
"""Where a test came from, in enough detail to defend or discard it."""
|
|
302
|
+
|
|
303
|
+
trace_id: str
|
|
304
|
+
failure_id: str
|
|
305
|
+
content_hash: str
|
|
306
|
+
#: Recorded, but never part of the test ID: clusters are rebuilt and
|
|
307
|
+
#: relabelled, and an ID that moved with them would not be stable.
|
|
308
|
+
cluster_id: str | None = None
|
|
309
|
+
cluster_label: str | None = None
|
|
310
|
+
representative_roles: list[str] = field(default_factory=list)
|
|
311
|
+
failure_type: str | None = None
|
|
312
|
+
severity: str | None = None
|
|
313
|
+
analyzer: str | None = None
|
|
314
|
+
analysis_summary: str | None = None
|
|
315
|
+
evidence: list[str] = field(default_factory=list)
|
|
316
|
+
generator_version: int = 1
|
|
317
|
+
generated_at: datetime = field(default_factory=lambda: datetime.now(UTC))
|
|
318
|
+
|
|
319
|
+
def to_dict(self) -> dict[str, Any]:
|
|
320
|
+
payload = {
|
|
321
|
+
key: getattr(self, key)
|
|
322
|
+
for key in (
|
|
323
|
+
"trace_id",
|
|
324
|
+
"failure_id",
|
|
325
|
+
"content_hash",
|
|
326
|
+
"cluster_id",
|
|
327
|
+
"cluster_label",
|
|
328
|
+
"representative_roles",
|
|
329
|
+
"failure_type",
|
|
330
|
+
"severity",
|
|
331
|
+
"analyzer",
|
|
332
|
+
"analysis_summary",
|
|
333
|
+
"evidence",
|
|
334
|
+
"generator_version",
|
|
335
|
+
)
|
|
336
|
+
}
|
|
337
|
+
payload["generated_at"] = self.generated_at.isoformat()
|
|
338
|
+
return payload
|
|
339
|
+
|
|
340
|
+
@classmethod
|
|
341
|
+
def from_dict(cls, payload: dict[str, Any]) -> Provenance:
|
|
342
|
+
data = dict(payload)
|
|
343
|
+
generated_at = data.pop("generated_at", None)
|
|
344
|
+
return cls(
|
|
345
|
+
**data,
|
|
346
|
+
generated_at=(
|
|
347
|
+
datetime.fromisoformat(generated_at) if generated_at else datetime.now(UTC)
|
|
348
|
+
),
|
|
349
|
+
)
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
@dataclass
|
|
353
|
+
class RegressionTest:
|
|
354
|
+
test_id: str
|
|
355
|
+
failure_id: str
|
|
356
|
+
input: CaseInput
|
|
357
|
+
provenance: Provenance
|
|
358
|
+
status: ReviewStatus = ReviewStatus.DRAFT
|
|
359
|
+
fixtures: list[Fixture] = field(default_factory=list)
|
|
360
|
+
expectations: list[Expectation] = field(default_factory=list)
|
|
361
|
+
#: Things a reviewer must decide, recorded rather than guessed at.
|
|
362
|
+
warnings: list[str] = field(default_factory=list)
|
|
363
|
+
reviewer: str | None = None
|
|
364
|
+
review_reason: str | None = None
|
|
365
|
+
reviewed_at: datetime | None = None
|
|
366
|
+
#: True once a person changed the generated content, not merely approved it.
|
|
367
|
+
edited: bool = False
|
|
368
|
+
edited_by: str | None = None
|
|
369
|
+
created_at: datetime = field(default_factory=lambda: datetime.now(UTC))
|
|
370
|
+
updated_at: datetime = field(default_factory=lambda: datetime.now(UTC))
|
|
371
|
+
|
|
372
|
+
@property
|
|
373
|
+
def reviewed(self) -> bool:
|
|
374
|
+
return self.status is not ReviewStatus.DRAFT
|
|
375
|
+
|
|
376
|
+
@property
|
|
377
|
+
def deterministic_expectations(self) -> list[Expectation]:
|
|
378
|
+
return [item for item in self.expectations if item.deterministic]
|
|
379
|
+
|
|
380
|
+
@property
|
|
381
|
+
def has_positive_expectation(self) -> bool:
|
|
382
|
+
return any(item.positive for item in self.expectations)
|
|
383
|
+
|
|
384
|
+
@property
|
|
385
|
+
def contradictions(self) -> list[Contradiction]:
|
|
386
|
+
return find_contradictions(self.expectations)
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
def slugify(text: str) -> str:
|
|
390
|
+
"""A short, readable, stable stem for a test ID."""
|
|
391
|
+
cleaned = _REDACTION_PATTERN.sub(" ", text or "").lower()
|
|
392
|
+
words = _WORD_PATTERN.findall(cleaned)[:MAX_SLUG_WORDS]
|
|
393
|
+
return "_".join(words)[:MAX_SLUG_LENGTH].strip("_")
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
def make_test_id(trace_id: str, input_text: str) -> str:
|
|
397
|
+
"""A readable, stable test ID derived only from immutable facts.
|
|
398
|
+
|
|
399
|
+
Deliberately *not* derived from the cluster label or the analysis: both are
|
|
400
|
+
mutable. A reviewer renaming a cluster, or a re-analysis changing a failure
|
|
401
|
+
type, must not rename a test that is already committed to Git and referenced
|
|
402
|
+
by past run results.
|
|
403
|
+
|
|
404
|
+
The stem comes from the trace's own input, which never changes once stored,
|
|
405
|
+
and the suffix from the trace ID, which guarantees uniqueness.
|
|
406
|
+
"""
|
|
407
|
+
digest = hashlib.sha256(trace_id.encode("utf-8")).hexdigest()[:_TEST_ID_DIGEST_LENGTH]
|
|
408
|
+
stem = slugify(input_text)
|
|
409
|
+
return f"{stem}_{digest}" if stem else f"test_{digest}"
|