evalkeep 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalkeep/__init__.py +12 -0
- evalkeep/__main__.py +6 -0
- evalkeep/adapters/__init__.py +45 -0
- evalkeep/adapters/base.py +92 -0
- evalkeep/adapters/jsonl.py +164 -0
- evalkeep/adapters/langsmith.py +436 -0
- evalkeep/adapters/otlp.py +442 -0
- evalkeep/adapters/semconv.py +208 -0
- evalkeep/analysis.py +174 -0
- evalkeep/analysis_run.py +160 -0
- evalkeep/analyzers/__init__.py +52 -0
- evalkeep/analyzers/anthropic.py +145 -0
- evalkeep/analyzers/stub.py +34 -0
- evalkeep/cache.py +122 -0
- evalkeep/cli.py +1933 -0
- evalkeep/clustering.py +383 -0
- evalkeep/clusters.py +101 -0
- evalkeep/commands/__init__.py +1 -0
- evalkeep/commands/analyze_cmd.py +100 -0
- evalkeep/commands/compare_cmd.py +169 -0
- evalkeep/commands/dataset_cmd.py +182 -0
- evalkeep/commands/detect_cmd.py +154 -0
- evalkeep/commands/discover_cmd.py +274 -0
- evalkeep/commands/ingest_cmd.py +50 -0
- evalkeep/commands/init_cmd.py +151 -0
- evalkeep/commands/pipeline_cmd.py +156 -0
- evalkeep/commands/review_cmd.py +141 -0
- evalkeep/commands/run_cmd.py +131 -0
- evalkeep/commands/target_cmd.py +109 -0
- evalkeep/commands/trace_cmd.py +58 -0
- evalkeep/comparison.py +432 -0
- evalkeep/config.py +209 -0
- evalkeep/detection.py +94 -0
- evalkeep/detectors.py +182 -0
- evalkeep/discovery.py +208 -0
- evalkeep/embeddings/__init__.py +31 -0
- evalkeep/embeddings/base.py +32 -0
- evalkeep/embeddings/hashing.py +98 -0
- evalkeep/errors.py +42 -0
- evalkeep/examples/__init__.py +37 -0
- evalkeep/examples/langsmith/runs.jsonl +18 -0
- evalkeep/examples/opentelemetry/spans.json +898 -0
- evalkeep/examples/refund-agent/agents/baseline.py +66 -0
- evalkeep/examples/refund-agent/agents/candidate.py +66 -0
- evalkeep/examples/refund-agent/traces.jsonl +5 -0
- evalkeep/examples/tau-bench/prepare.py +230 -0
- evalkeep/exporters/__init__.py +45 -0
- evalkeep/exporters/generic.py +31 -0
- evalkeep/exporters/promptfoo.py +219 -0
- evalkeep/failures.py +95 -0
- evalkeep/generation.py +303 -0
- evalkeep/hashing.py +56 -0
- evalkeep/ingest.py +257 -0
- evalkeep/prompts.py +127 -0
- evalkeep/pseudonyms.py +82 -0
- evalkeep/py.typed +0 -0
- evalkeep/redaction.py +333 -0
- evalkeep/regression.py +409 -0
- evalkeep/review.py +309 -0
- evalkeep/runner.py +302 -0
- evalkeep/runs.py +185 -0
- evalkeep/storage/__init__.py +37 -0
- evalkeep/storage/clusters.py +163 -0
- evalkeep/storage/failures.py +254 -0
- evalkeep/storage/migrations.py +370 -0
- evalkeep/storage/regression.py +136 -0
- evalkeep/storage/runs.py +223 -0
- evalkeep/storage/store.py +429 -0
- evalkeep/targets.py +205 -0
- evalkeep/trace.py +238 -0
- evalkeep-0.1.0.dist-info/METADATA +221 -0
- evalkeep-0.1.0.dist-info/RECORD +75 -0
- evalkeep-0.1.0.dist-info/WHEEL +4 -0
- evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
- evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,370 @@
|
|
|
1
|
+
"""SQLite migrations, applied in order and recorded in ``schema_migrations``.
|
|
2
|
+
|
|
3
|
+
Each migration runs inside one transaction together with the row that records
|
|
4
|
+
it, so a database is never left half-migrated. Migrations are append-only: to
|
|
5
|
+
change the schema, add a new one rather than editing an applied one.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import sqlite3
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from datetime import UTC, datetime
|
|
13
|
+
|
|
14
|
+
from evalkeep.errors import CommandError
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(frozen=True)
|
|
18
|
+
class Migration:
|
|
19
|
+
version: int
|
|
20
|
+
name: str
|
|
21
|
+
statements: tuple[str, ...]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
MIGRATIONS: tuple[Migration, ...] = (
|
|
25
|
+
Migration(
|
|
26
|
+
version=1,
|
|
27
|
+
name="traces and events",
|
|
28
|
+
statements=(
|
|
29
|
+
"""
|
|
30
|
+
CREATE TABLE traces (
|
|
31
|
+
trace_id TEXT PRIMARY KEY,
|
|
32
|
+
content_hash TEXT NOT NULL,
|
|
33
|
+
schema_version INTEGER NOT NULL,
|
|
34
|
+
status TEXT NOT NULL,
|
|
35
|
+
source TEXT,
|
|
36
|
+
recorded_at TEXT,
|
|
37
|
+
ingested_at TEXT NOT NULL,
|
|
38
|
+
redactions INTEGER NOT NULL DEFAULT 0,
|
|
39
|
+
redaction_summary TEXT NOT NULL DEFAULT '{}',
|
|
40
|
+
payload TEXT NOT NULL
|
|
41
|
+
)
|
|
42
|
+
""",
|
|
43
|
+
"CREATE INDEX traces_content_hash ON traces(content_hash)",
|
|
44
|
+
"CREATE INDEX traces_status ON traces(status)",
|
|
45
|
+
"""
|
|
46
|
+
CREATE TABLE events (
|
|
47
|
+
trace_id TEXT NOT NULL
|
|
48
|
+
REFERENCES traces(trace_id) ON DELETE CASCADE,
|
|
49
|
+
position INTEGER NOT NULL,
|
|
50
|
+
event_id TEXT NOT NULL,
|
|
51
|
+
type TEXT NOT NULL,
|
|
52
|
+
tool TEXT,
|
|
53
|
+
call_id TEXT,
|
|
54
|
+
timestamp TEXT,
|
|
55
|
+
payload TEXT NOT NULL,
|
|
56
|
+
PRIMARY KEY (trace_id, position)
|
|
57
|
+
)
|
|
58
|
+
""",
|
|
59
|
+
"CREATE INDEX events_tool ON events(tool)",
|
|
60
|
+
),
|
|
61
|
+
),
|
|
62
|
+
Migration(
|
|
63
|
+
version=2,
|
|
64
|
+
name="failure candidates and signals",
|
|
65
|
+
statements=(
|
|
66
|
+
# UNIQUE on trace_id is the schema-level form of "one failure
|
|
67
|
+
# candidate per trace"; detection cannot create a second one.
|
|
68
|
+
"""
|
|
69
|
+
CREATE TABLE failures (
|
|
70
|
+
failure_id TEXT PRIMARY KEY,
|
|
71
|
+
trace_id TEXT NOT NULL UNIQUE
|
|
72
|
+
REFERENCES traces(trace_id) ON DELETE CASCADE,
|
|
73
|
+
status TEXT NOT NULL,
|
|
74
|
+
origin TEXT NOT NULL,
|
|
75
|
+
detected_at TEXT NOT NULL,
|
|
76
|
+
updated_at TEXT NOT NULL,
|
|
77
|
+
reviewer TEXT,
|
|
78
|
+
reason TEXT
|
|
79
|
+
)
|
|
80
|
+
""",
|
|
81
|
+
"CREATE INDEX failures_status ON failures(status)",
|
|
82
|
+
"""
|
|
83
|
+
CREATE TABLE failure_signals (
|
|
84
|
+
failure_id TEXT NOT NULL
|
|
85
|
+
REFERENCES failures(failure_id) ON DELETE CASCADE,
|
|
86
|
+
position INTEGER NOT NULL,
|
|
87
|
+
detector TEXT NOT NULL,
|
|
88
|
+
kind TEXT NOT NULL,
|
|
89
|
+
source TEXT NOT NULL,
|
|
90
|
+
summary TEXT NOT NULL,
|
|
91
|
+
evidence TEXT NOT NULL,
|
|
92
|
+
PRIMARY KEY (failure_id, position)
|
|
93
|
+
)
|
|
94
|
+
""",
|
|
95
|
+
"CREATE INDEX failure_signals_kind ON failure_signals(kind)",
|
|
96
|
+
),
|
|
97
|
+
),
|
|
98
|
+
Migration(
|
|
99
|
+
version=3,
|
|
100
|
+
name="failure analysis",
|
|
101
|
+
statements=(
|
|
102
|
+
# One analysis per failure: the latest replaces the previous one,
|
|
103
|
+
# and the provenance columns say who produced it.
|
|
104
|
+
"""
|
|
105
|
+
CREATE TABLE failure_analyses (
|
|
106
|
+
failure_id TEXT PRIMARY KEY
|
|
107
|
+
REFERENCES failures(failure_id) ON DELETE CASCADE,
|
|
108
|
+
failure_type TEXT NOT NULL,
|
|
109
|
+
component TEXT NOT NULL,
|
|
110
|
+
severity TEXT NOT NULL,
|
|
111
|
+
summary TEXT NOT NULL,
|
|
112
|
+
analyzer TEXT NOT NULL,
|
|
113
|
+
prompt_version INTEGER NOT NULL,
|
|
114
|
+
analyzed_at TEXT NOT NULL,
|
|
115
|
+
labeler TEXT,
|
|
116
|
+
raw_response TEXT
|
|
117
|
+
)
|
|
118
|
+
""",
|
|
119
|
+
"CREATE INDEX failure_analyses_type ON failure_analyses(failure_type)",
|
|
120
|
+
"CREATE INDEX failure_analyses_severity ON failure_analyses(severity)",
|
|
121
|
+
),
|
|
122
|
+
),
|
|
123
|
+
Migration(
|
|
124
|
+
version=4,
|
|
125
|
+
name="clusters and representatives",
|
|
126
|
+
statements=(
|
|
127
|
+
# One row per `discover`, holding everything needed to reproduce
|
|
128
|
+
# the grouping it produced.
|
|
129
|
+
"""
|
|
130
|
+
CREATE TABLE clustering_runs (
|
|
131
|
+
run_id TEXT PRIMARY KEY,
|
|
132
|
+
created_at TEXT NOT NULL,
|
|
133
|
+
embedder TEXT NOT NULL,
|
|
134
|
+
dimensions INTEGER NOT NULL,
|
|
135
|
+
parameters TEXT NOT NULL,
|
|
136
|
+
failures INTEGER NOT NULL
|
|
137
|
+
)
|
|
138
|
+
""",
|
|
139
|
+
"""
|
|
140
|
+
CREATE TABLE clusters (
|
|
141
|
+
cluster_id TEXT PRIMARY KEY,
|
|
142
|
+
run_id TEXT NOT NULL
|
|
143
|
+
REFERENCES clustering_runs(run_id) ON DELETE CASCADE,
|
|
144
|
+
label TEXT NOT NULL,
|
|
145
|
+
labelled_by TEXT,
|
|
146
|
+
dismissed INTEGER NOT NULL DEFAULT 0,
|
|
147
|
+
created_at TEXT NOT NULL
|
|
148
|
+
)
|
|
149
|
+
""",
|
|
150
|
+
"CREATE INDEX clusters_run ON clusters(run_id)",
|
|
151
|
+
"""
|
|
152
|
+
CREATE TABLE cluster_members (
|
|
153
|
+
cluster_id TEXT NOT NULL
|
|
154
|
+
REFERENCES clusters(cluster_id) ON DELETE CASCADE,
|
|
155
|
+
failure_id TEXT NOT NULL
|
|
156
|
+
REFERENCES failures(failure_id) ON DELETE CASCADE,
|
|
157
|
+
distance REAL NOT NULL,
|
|
158
|
+
roles TEXT NOT NULL DEFAULT '[]',
|
|
159
|
+
PRIMARY KEY (cluster_id, failure_id)
|
|
160
|
+
)
|
|
161
|
+
""",
|
|
162
|
+
"CREATE INDEX cluster_members_failure ON cluster_members(failure_id)",
|
|
163
|
+
),
|
|
164
|
+
),
|
|
165
|
+
Migration(
|
|
166
|
+
version=5,
|
|
167
|
+
name="regression test drafts",
|
|
168
|
+
statements=(
|
|
169
|
+
# cluster_id is a plain column, not a foreign key: clusters are
|
|
170
|
+
# rebuilt by every `discover`, and a test must outlive the grouping
|
|
171
|
+
# that suggested it.
|
|
172
|
+
"""
|
|
173
|
+
CREATE TABLE regression_tests (
|
|
174
|
+
test_id TEXT PRIMARY KEY,
|
|
175
|
+
failure_id TEXT NOT NULL UNIQUE
|
|
176
|
+
REFERENCES failures(failure_id) ON DELETE CASCADE,
|
|
177
|
+
cluster_id TEXT,
|
|
178
|
+
status TEXT NOT NULL,
|
|
179
|
+
input TEXT NOT NULL,
|
|
180
|
+
fixtures TEXT NOT NULL,
|
|
181
|
+
expectations TEXT NOT NULL,
|
|
182
|
+
warnings TEXT NOT NULL,
|
|
183
|
+
provenance TEXT NOT NULL,
|
|
184
|
+
reviewer TEXT,
|
|
185
|
+
review_reason TEXT,
|
|
186
|
+
created_at TEXT NOT NULL,
|
|
187
|
+
updated_at TEXT NOT NULL
|
|
188
|
+
)
|
|
189
|
+
""",
|
|
190
|
+
"CREATE INDEX regression_tests_status ON regression_tests(status)",
|
|
191
|
+
"CREATE INDEX regression_tests_cluster ON regression_tests(cluster_id)",
|
|
192
|
+
),
|
|
193
|
+
),
|
|
194
|
+
Migration(
|
|
195
|
+
version=6,
|
|
196
|
+
name="review audit trail",
|
|
197
|
+
statements=(
|
|
198
|
+
"ALTER TABLE regression_tests ADD COLUMN reviewed_at TEXT",
|
|
199
|
+
"ALTER TABLE regression_tests ADD COLUMN edited INTEGER NOT NULL DEFAULT 0",
|
|
200
|
+
"ALTER TABLE regression_tests ADD COLUMN edited_by TEXT",
|
|
201
|
+
),
|
|
202
|
+
),
|
|
203
|
+
Migration(
|
|
204
|
+
version=7,
|
|
205
|
+
name="evaluation runs and results",
|
|
206
|
+
statements=(
|
|
207
|
+
"""
|
|
208
|
+
CREATE TABLE evaluation_runs (
|
|
209
|
+
run_id TEXT PRIMARY KEY,
|
|
210
|
+
target_id TEXT NOT NULL,
|
|
211
|
+
suite_hash TEXT NOT NULL,
|
|
212
|
+
tests INTEGER NOT NULL,
|
|
213
|
+
status TEXT NOT NULL,
|
|
214
|
+
runner TEXT,
|
|
215
|
+
environment TEXT NOT NULL DEFAULT '{}',
|
|
216
|
+
started_at TEXT NOT NULL,
|
|
217
|
+
finished_at TEXT,
|
|
218
|
+
output_dir TEXT
|
|
219
|
+
)
|
|
220
|
+
""",
|
|
221
|
+
"CREATE INDEX evaluation_runs_target ON evaluation_runs(target_id)",
|
|
222
|
+
"""
|
|
223
|
+
CREATE TABLE test_results (
|
|
224
|
+
run_id TEXT NOT NULL
|
|
225
|
+
REFERENCES evaluation_runs(run_id) ON DELETE CASCADE,
|
|
226
|
+
test_id TEXT NOT NULL,
|
|
227
|
+
outcome TEXT NOT NULL,
|
|
228
|
+
error_kind TEXT,
|
|
229
|
+
error TEXT,
|
|
230
|
+
latency_ms INTEGER,
|
|
231
|
+
observation TEXT,
|
|
232
|
+
failed_assertions TEXT NOT NULL DEFAULT '[]',
|
|
233
|
+
PRIMARY KEY (run_id, test_id)
|
|
234
|
+
)
|
|
235
|
+
""",
|
|
236
|
+
"CREATE INDEX test_results_outcome ON test_results(outcome)",
|
|
237
|
+
),
|
|
238
|
+
),
|
|
239
|
+
Migration(
|
|
240
|
+
version=8,
|
|
241
|
+
name="baseline promotions",
|
|
242
|
+
statements=(
|
|
243
|
+
# An append-only record rather than a flag: which run was the
|
|
244
|
+
# baseline, when, and who decided, is exactly the kind of history a
|
|
245
|
+
# regression argument later depends on.
|
|
246
|
+
"""
|
|
247
|
+
CREATE TABLE baseline_promotions (
|
|
248
|
+
promotion_id TEXT PRIMARY KEY,
|
|
249
|
+
run_id TEXT NOT NULL
|
|
250
|
+
REFERENCES evaluation_runs(run_id) ON DELETE CASCADE,
|
|
251
|
+
target_id TEXT NOT NULL,
|
|
252
|
+
promoted_at TEXT NOT NULL,
|
|
253
|
+
reviewer TEXT NOT NULL,
|
|
254
|
+
reason TEXT
|
|
255
|
+
)
|
|
256
|
+
""",
|
|
257
|
+
"CREATE INDEX baseline_promotions_time ON baseline_promotions(promoted_at)",
|
|
258
|
+
),
|
|
259
|
+
),
|
|
260
|
+
Migration(
|
|
261
|
+
version=9,
|
|
262
|
+
name="trace occurrences",
|
|
263
|
+
statements=(
|
|
264
|
+
# One row per *sighting* of an interaction, while `traces` keeps one
|
|
265
|
+
# row per distinct interaction. Deduplication is right for the test
|
|
266
|
+
# suite and wrong for the evidence: how often a failure happens, and
|
|
267
|
+
# which versions it affects, is what a severity judgement rests on.
|
|
268
|
+
#
|
|
269
|
+
# occurrence_id is derived from the sighting's own content, so
|
|
270
|
+
# re-ingesting a file records nothing new rather than inflating the
|
|
271
|
+
# counts.
|
|
272
|
+
"""
|
|
273
|
+
CREATE TABLE trace_occurrences (
|
|
274
|
+
occurrence_id TEXT PRIMARY KEY,
|
|
275
|
+
canonical_trace_id TEXT NOT NULL
|
|
276
|
+
REFERENCES traces(trace_id) ON DELETE CASCADE,
|
|
277
|
+
content_hash TEXT NOT NULL,
|
|
278
|
+
trace_id TEXT NOT NULL,
|
|
279
|
+
source TEXT,
|
|
280
|
+
agent TEXT,
|
|
281
|
+
model TEXT,
|
|
282
|
+
recorded_at TEXT,
|
|
283
|
+
ingested_at TEXT NOT NULL
|
|
284
|
+
)
|
|
285
|
+
""",
|
|
286
|
+
"CREATE INDEX trace_occurrences_canonical ON trace_occurrences(canonical_trace_id)",
|
|
287
|
+
"CREATE INDEX trace_occurrences_hash ON trace_occurrences(content_hash)",
|
|
288
|
+
),
|
|
289
|
+
),
|
|
290
|
+
Migration(
|
|
291
|
+
version=10,
|
|
292
|
+
name="repeated execution",
|
|
293
|
+
statements=(
|
|
294
|
+
"ALTER TABLE evaluation_runs ADD COLUMN repetitions INTEGER NOT NULL DEFAULT 1",
|
|
295
|
+
# test_results needs `repetition` in its primary key, and SQLite
|
|
296
|
+
# cannot alter one, so the table is rebuilt. Existing rows become
|
|
297
|
+
# repetition 0, which is exactly what a single-execution run was.
|
|
298
|
+
"""
|
|
299
|
+
CREATE TABLE test_results_rebuilt (
|
|
300
|
+
run_id TEXT NOT NULL
|
|
301
|
+
REFERENCES evaluation_runs(run_id) ON DELETE CASCADE,
|
|
302
|
+
test_id TEXT NOT NULL,
|
|
303
|
+
repetition INTEGER NOT NULL DEFAULT 0,
|
|
304
|
+
outcome TEXT NOT NULL,
|
|
305
|
+
error_kind TEXT,
|
|
306
|
+
error TEXT,
|
|
307
|
+
latency_ms INTEGER,
|
|
308
|
+
observation TEXT,
|
|
309
|
+
failed_assertions TEXT NOT NULL DEFAULT '[]',
|
|
310
|
+
PRIMARY KEY (run_id, test_id, repetition)
|
|
311
|
+
)
|
|
312
|
+
""",
|
|
313
|
+
"""
|
|
314
|
+
INSERT INTO test_results_rebuilt (
|
|
315
|
+
run_id, test_id, repetition, outcome, error_kind, error,
|
|
316
|
+
latency_ms, observation, failed_assertions
|
|
317
|
+
)
|
|
318
|
+
SELECT run_id, test_id, 0, outcome, error_kind, error,
|
|
319
|
+
latency_ms, observation, failed_assertions
|
|
320
|
+
FROM test_results
|
|
321
|
+
""",
|
|
322
|
+
"DROP TABLE test_results",
|
|
323
|
+
"ALTER TABLE test_results_rebuilt RENAME TO test_results",
|
|
324
|
+
"CREATE INDEX test_results_outcome ON test_results(outcome)",
|
|
325
|
+
"CREATE INDEX test_results_case ON test_results(run_id, test_id)",
|
|
326
|
+
),
|
|
327
|
+
),
|
|
328
|
+
)
|
|
329
|
+
|
|
330
|
+
LATEST_VERSION = max(migration.version for migration in MIGRATIONS)
|
|
331
|
+
|
|
332
|
+
_SCHEMA_MIGRATIONS_TABLE = """
|
|
333
|
+
CREATE TABLE IF NOT EXISTS schema_migrations (
|
|
334
|
+
version INTEGER PRIMARY KEY,
|
|
335
|
+
name TEXT NOT NULL,
|
|
336
|
+
applied_at TEXT NOT NULL
|
|
337
|
+
)
|
|
338
|
+
"""
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def applied_version(connection: sqlite3.Connection) -> int:
|
|
342
|
+
"""The highest migration recorded in this database, or 0 for a new one."""
|
|
343
|
+
connection.execute(_SCHEMA_MIGRATIONS_TABLE)
|
|
344
|
+
row = connection.execute("SELECT MAX(version) FROM schema_migrations").fetchone()
|
|
345
|
+
return int(row[0]) if row and row[0] is not None else 0
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
def apply_migrations(connection: sqlite3.Connection) -> list[Migration]:
|
|
349
|
+
"""Bring the database up to :data:`LATEST_VERSION`. Returns what it ran."""
|
|
350
|
+
current = applied_version(connection)
|
|
351
|
+
if current > LATEST_VERSION:
|
|
352
|
+
raise CommandError(
|
|
353
|
+
f"The database is at schema version {current}, but this build only "
|
|
354
|
+
f"understands {LATEST_VERSION}.",
|
|
355
|
+
hint="Upgrade evalkeep.",
|
|
356
|
+
)
|
|
357
|
+
|
|
358
|
+
applied: list[Migration] = []
|
|
359
|
+
for migration in MIGRATIONS:
|
|
360
|
+
if migration.version <= current:
|
|
361
|
+
continue
|
|
362
|
+
with connection: # one transaction per migration, including its record
|
|
363
|
+
for statement in migration.statements:
|
|
364
|
+
connection.execute(statement)
|
|
365
|
+
connection.execute(
|
|
366
|
+
"INSERT INTO schema_migrations (version, name, applied_at) VALUES (?, ?, ?)",
|
|
367
|
+
(migration.version, migration.name, datetime.now(UTC).isoformat()),
|
|
368
|
+
)
|
|
369
|
+
applied.append(migration)
|
|
370
|
+
return applied
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
"""Persistence for regression-test drafts."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import sqlite3
|
|
7
|
+
from collections.abc import Iterator
|
|
8
|
+
from datetime import datetime
|
|
9
|
+
|
|
10
|
+
from evalkeep.regression import (
|
|
11
|
+
CaseInput,
|
|
12
|
+
Expectation,
|
|
13
|
+
Fixture,
|
|
14
|
+
Provenance,
|
|
15
|
+
RegressionTest,
|
|
16
|
+
ReviewStatus,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class RegressionStore:
|
|
21
|
+
"""Read/write access to the regression-test table."""
|
|
22
|
+
|
|
23
|
+
def __init__(self, connection: sqlite3.Connection) -> None:
|
|
24
|
+
self._connection = connection
|
|
25
|
+
|
|
26
|
+
def save(self, test: RegressionTest) -> None:
|
|
27
|
+
with self._connection:
|
|
28
|
+
self._connection.execute(
|
|
29
|
+
"""
|
|
30
|
+
INSERT INTO regression_tests (
|
|
31
|
+
test_id, failure_id, cluster_id, status, input, fixtures,
|
|
32
|
+
expectations, warnings, provenance, reviewer, review_reason,
|
|
33
|
+
reviewed_at, edited, edited_by, created_at, updated_at
|
|
34
|
+
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
35
|
+
ON CONFLICT(test_id) DO UPDATE SET
|
|
36
|
+
cluster_id = excluded.cluster_id,
|
|
37
|
+
status = excluded.status,
|
|
38
|
+
input = excluded.input,
|
|
39
|
+
fixtures = excluded.fixtures,
|
|
40
|
+
expectations = excluded.expectations,
|
|
41
|
+
warnings = excluded.warnings,
|
|
42
|
+
provenance = excluded.provenance,
|
|
43
|
+
reviewer = excluded.reviewer,
|
|
44
|
+
review_reason = excluded.review_reason,
|
|
45
|
+
reviewed_at = excluded.reviewed_at,
|
|
46
|
+
edited = excluded.edited,
|
|
47
|
+
edited_by = excluded.edited_by,
|
|
48
|
+
updated_at = excluded.updated_at
|
|
49
|
+
""",
|
|
50
|
+
(
|
|
51
|
+
test.test_id,
|
|
52
|
+
test.failure_id,
|
|
53
|
+
test.provenance.cluster_id,
|
|
54
|
+
test.status.value,
|
|
55
|
+
json.dumps(test.input.to_dict(), sort_keys=True),
|
|
56
|
+
json.dumps([f.to_dict() for f in test.fixtures], sort_keys=True),
|
|
57
|
+
json.dumps([e.to_dict() for e in test.expectations], sort_keys=True),
|
|
58
|
+
json.dumps(test.warnings),
|
|
59
|
+
json.dumps(test.provenance.to_dict(), sort_keys=True),
|
|
60
|
+
test.reviewer,
|
|
61
|
+
test.review_reason,
|
|
62
|
+
test.reviewed_at.isoformat() if test.reviewed_at else None,
|
|
63
|
+
int(test.edited),
|
|
64
|
+
test.edited_by,
|
|
65
|
+
test.created_at.isoformat(),
|
|
66
|
+
test.updated_at.isoformat(),
|
|
67
|
+
),
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
def get(self, test_id: str) -> RegressionTest | None:
|
|
71
|
+
row = self._connection.execute(
|
|
72
|
+
"SELECT * FROM regression_tests WHERE test_id = ?", (test_id.strip(),)
|
|
73
|
+
).fetchone()
|
|
74
|
+
return _build(row) if row is not None else None
|
|
75
|
+
|
|
76
|
+
def get_by_failure(self, failure_id: str) -> RegressionTest | None:
|
|
77
|
+
row = self._connection.execute(
|
|
78
|
+
"SELECT * FROM regression_tests WHERE failure_id = ?", (failure_id.strip(),)
|
|
79
|
+
).fetchone()
|
|
80
|
+
return _build(row) if row is not None else None
|
|
81
|
+
|
|
82
|
+
def list(
|
|
83
|
+
self, *, status: ReviewStatus | None = None, limit: int = 50, offset: int = 0
|
|
84
|
+
) -> list[RegressionTest]:
|
|
85
|
+
query = "SELECT * FROM regression_tests"
|
|
86
|
+
parameters: list[object] = []
|
|
87
|
+
if status is not None:
|
|
88
|
+
query += " WHERE status = ?"
|
|
89
|
+
parameters.append(status.value)
|
|
90
|
+
query += " ORDER BY created_at, test_id LIMIT ? OFFSET ?"
|
|
91
|
+
parameters += [limit, offset]
|
|
92
|
+
return [_build(row) for row in self._connection.execute(query, parameters)]
|
|
93
|
+
|
|
94
|
+
def iter_all(self) -> Iterator[RegressionTest]:
|
|
95
|
+
for row in self._connection.execute("SELECT * FROM regression_tests ORDER BY test_id"):
|
|
96
|
+
yield _build(row)
|
|
97
|
+
|
|
98
|
+
def count(self, *, status: ReviewStatus | None = None) -> int:
|
|
99
|
+
if status is None:
|
|
100
|
+
row = self._connection.execute("SELECT COUNT(*) AS n FROM regression_tests").fetchone()
|
|
101
|
+
else:
|
|
102
|
+
row = self._connection.execute(
|
|
103
|
+
"SELECT COUNT(*) AS n FROM regression_tests WHERE status = ?",
|
|
104
|
+
(status.value,),
|
|
105
|
+
).fetchone()
|
|
106
|
+
return int(row["n"])
|
|
107
|
+
|
|
108
|
+
def counts_by_status(self) -> dict[ReviewStatus, int]:
|
|
109
|
+
rows = self._connection.execute(
|
|
110
|
+
"SELECT status, COUNT(*) AS n FROM regression_tests GROUP BY status"
|
|
111
|
+
).fetchall()
|
|
112
|
+
return {ReviewStatus(row["status"]): int(row["n"]) for row in rows}
|
|
113
|
+
|
|
114
|
+
def delete(self, test_id: str) -> None:
|
|
115
|
+
with self._connection:
|
|
116
|
+
self._connection.execute("DELETE FROM regression_tests WHERE test_id = ?", (test_id,))
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _build(row: sqlite3.Row) -> RegressionTest:
|
|
120
|
+
return RegressionTest(
|
|
121
|
+
test_id=row["test_id"],
|
|
122
|
+
failure_id=row["failure_id"],
|
|
123
|
+
status=ReviewStatus(row["status"]),
|
|
124
|
+
input=CaseInput.from_dict(json.loads(row["input"])),
|
|
125
|
+
fixtures=[Fixture.from_dict(item) for item in json.loads(row["fixtures"])],
|
|
126
|
+
expectations=[Expectation.from_dict(item) for item in json.loads(row["expectations"])],
|
|
127
|
+
warnings=list(json.loads(row["warnings"])),
|
|
128
|
+
provenance=Provenance.from_dict(json.loads(row["provenance"])),
|
|
129
|
+
reviewer=row["reviewer"],
|
|
130
|
+
review_reason=row["review_reason"],
|
|
131
|
+
reviewed_at=(datetime.fromisoformat(row["reviewed_at"]) if row["reviewed_at"] else None),
|
|
132
|
+
edited=bool(row["edited"]),
|
|
133
|
+
edited_by=row["edited_by"],
|
|
134
|
+
created_at=datetime.fromisoformat(row["created_at"]),
|
|
135
|
+
updated_at=datetime.fromisoformat(row["updated_at"]),
|
|
136
|
+
)
|