crapkit 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crapkit/__init__.py +2 -0
- crapkit/__main__.py +5 -0
- crapkit/_pygdefer.py +86 -0
- crapkit/analyze.py +375 -0
- crapkit/cache.py +58 -0
- crapkit/churn.py +113 -0
- crapkit/churn_cache.py +108 -0
- crapkit/churn_log.py +286 -0
- crapkit/cli/__init__.py +316 -0
- crapkit/cli/_shared.py +130 -0
- crapkit/cli/admin.py +650 -0
- crapkit/cli/analyses.py +144 -0
- crapkit/cli/parser.py +384 -0
- crapkit/cli/queue.py +926 -0
- crapkit/cli/ratchet_cmds.py +172 -0
- crapkit/cli/reports.py +459 -0
- crapkit/cli/scoring.py +500 -0
- crapkit/cli/verifying.py +580 -0
- crapkit/config.py +289 -0
- crapkit/coupling.py +89 -0
- crapkit/coverage_istanbul.py +225 -0
- crapkit/coverage_py.py +87 -0
- crapkit/covstream.py +320 -0
- crapkit/diffparse.py +98 -0
- crapkit/digest.py +191 -0
- crapkit/discover.py +365 -0
- crapkit/doctor.py +308 -0
- crapkit/dup.py +179 -0
- crapkit/errors.py +18 -0
- crapkit/gitio.py +504 -0
- crapkit/hook.py +167 -0
- crapkit/junitparse.py +87 -0
- crapkit/lanes.py +373 -0
- crapkit/lizardcognitive.py +238 -0
- crapkit/mcp_server.py +167 -0
- crapkit/merge.py +77 -0
- crapkit/mutate.py +96 -0
- crapkit/mutate_pool.py +152 -0
- crapkit/override.py +94 -0
- crapkit/packet.py +343 -0
- crapkit/ratchet.py +236 -0
- crapkit/ratchet_report.py +135 -0
- crapkit/sarif.py +82 -0
- crapkit/sarifio.py +49 -0
- crapkit/scaffold.py +361 -0
- crapkit/score.py +255 -0
- crapkit/snapshot.py +51 -0
- crapkit/store.py +1066 -0
- crapkit/uncovered.py +131 -0
- crapkit/universe.py +157 -0
- crapkit/verify.py +194 -0
- crapkit/watch.py +112 -0
- crapkit/worklist.py +290 -0
- crapkit-0.2.0.dist-info/METADATA +802 -0
- crapkit-0.2.0.dist-info/RECORD +59 -0
- crapkit-0.2.0.dist-info/WHEEL +5 -0
- crapkit-0.2.0.dist-info/entry_points.txt +2 -0
- crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
- crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/store.py
ADDED
|
@@ -0,0 +1,1066 @@
|
|
|
1
|
+
"""Append-only SQLite snapshot store.
|
|
2
|
+
|
|
3
|
+
Run metadata (commit, tool versions, lane provenance) lives on the run row;
|
|
4
|
+
scored rows are pure data keyed by run. Nothing here is ever updated in
|
|
5
|
+
place — a rebuild is a new run. Coverage columns are NULL on inventory-only
|
|
6
|
+
runs and populated on scored runs.
|
|
7
|
+
|
|
8
|
+
A function's identity — scope, path, long_name — lives once, in `identities`,
|
|
9
|
+
and every run's rows point at it by id. Storing the three strings on the row
|
|
10
|
+
rewrote a hundred thousand identities on every run; the flagship consumer's
|
|
11
|
+
store reached 246 MB that way. The reads join them back, so nothing above this
|
|
12
|
+
module can tell: read_rows and read_scored return the same values in the same
|
|
13
|
+
order, and identity ids never reach a sort key.
|
|
14
|
+
|
|
15
|
+
`runs prune` is the one exception to append-only, and it deletes whole runs
|
|
16
|
+
rather than editing any row: see prune_keep_set for what it may never take.
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import json
|
|
21
|
+
import sqlite3
|
|
22
|
+
import sys
|
|
23
|
+
import zlib
|
|
24
|
+
from itertools import takewhile
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
from typing import NamedTuple
|
|
27
|
+
|
|
28
|
+
from .packet import bare_name, handle_ordinal
|
|
29
|
+
from .snapshot import InventoryRow
|
|
30
|
+
|
|
31
|
+
# {table} so the migration can build the same shape under a temp name and swap
|
|
32
|
+
# it in last: the live table is never dropped until its replacement is filled.
|
|
33
|
+
#
|
|
34
|
+
# flag and remedy are INTEGER codes into `flags` and `remedies`. The two columns
|
|
35
|
+
# held 14.7 MB of repeated short strings on the flagship consumer's store; the
|
|
36
|
+
# codes never leave this module, so every read still hands back "measured" and
|
|
37
|
+
# "add-tests".
|
|
38
|
+
_FUNCTIONS_DDL = """CREATE TABLE IF NOT EXISTS {table} (
|
|
39
|
+
run_id INTEGER NOT NULL REFERENCES runs(id),
|
|
40
|
+
identity_id INTEGER NOT NULL REFERENCES identities(id),
|
|
41
|
+
start INTEGER NOT NULL, end INTEGER NOT NULL,
|
|
42
|
+
ccn_std INTEGER NOT NULL, ccn_mod INTEGER NOT NULL, ccn INTEGER NOT NULL,
|
|
43
|
+
nloc INTEGER NOT NULL, params INTEGER NOT NULL, nesting INTEGER NOT NULL,
|
|
44
|
+
cov REAL, flag INTEGER, crap REAL, remedy INTEGER,
|
|
45
|
+
cognitive INTEGER NOT NULL DEFAULT 0
|
|
46
|
+
)"""
|
|
47
|
+
|
|
48
|
+
# The UNIQUE leads with the path, and that ordering IS the index the path-scoped
|
|
49
|
+
# reads seek: brief, explain and function_span all ask (path, long_name). A key
|
|
50
|
+
# led by scope answers none of them, which is why it needed a second index on
|
|
51
|
+
# (path, long_name) carried beside it.
|
|
52
|
+
_IDENTITY_KEY = ("path", "long_name", "scope")
|
|
53
|
+
_IDENTITIES_DDL = """CREATE TABLE IF NOT EXISTS {table} (
|
|
54
|
+
id INTEGER PRIMARY KEY,
|
|
55
|
+
scope TEXT NOT NULL, path TEXT NOT NULL, long_name TEXT NOT NULL,
|
|
56
|
+
UNIQUE(path, long_name, scope)
|
|
57
|
+
)"""
|
|
58
|
+
|
|
59
|
+
_CODE_DDL = """CREATE TABLE IF NOT EXISTS {table} (
|
|
60
|
+
id INTEGER PRIMARY KEY, name TEXT NOT NULL UNIQUE
|
|
61
|
+
)"""
|
|
62
|
+
|
|
63
|
+
# crapkit's own verdict vocabulary, at FIXED codes: insertion order would let two
|
|
64
|
+
# stores that met the same names in a different order hold different integers,
|
|
65
|
+
# and a store is a file people copy between machines. A name from outside this
|
|
66
|
+
# list is still stored, at a code minted after these.
|
|
67
|
+
_CODE_SEEDS = {"flags": ("measured", "untested", "no-lane", "cc-only"),
|
|
68
|
+
"remedies": ("ok", "add-tests", "decompose")}
|
|
69
|
+
|
|
70
|
+
_SCHEMA = f"""
|
|
71
|
+
CREATE TABLE IF NOT EXISTS runs (
|
|
72
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
73
|
+
commit_sha TEXT NOT NULL,
|
|
74
|
+
tool_versions TEXT NOT NULL,
|
|
75
|
+
lanes BLOB NOT NULL DEFAULT '{{}}',
|
|
76
|
+
kind TEXT NOT NULL DEFAULT 'coverage',
|
|
77
|
+
verdict_ok INTEGER,
|
|
78
|
+
findings INTEGER NOT NULL DEFAULT 0,
|
|
79
|
+
created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ', 'now'))
|
|
80
|
+
);
|
|
81
|
+
{_IDENTITIES_DDL.format(table="identities")};
|
|
82
|
+
{_CODE_DDL.format(table="flags")};
|
|
83
|
+
{_CODE_DDL.format(table="remedies")};
|
|
84
|
+
{_FUNCTIONS_DDL.format(table="functions")};
|
|
85
|
+
CREATE TABLE IF NOT EXISTS overrides (
|
|
86
|
+
run_id INTEGER NOT NULL REFERENCES runs(id),
|
|
87
|
+
path TEXT NOT NULL, long_name TEXT NOT NULL,
|
|
88
|
+
crap REAL NOT NULL, reason TEXT NOT NULL,
|
|
89
|
+
created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ', 'now'))
|
|
90
|
+
);
|
|
91
|
+
CREATE TABLE IF NOT EXISTS attempts (
|
|
92
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
93
|
+
path TEXT NOT NULL, long_name TEXT NOT NULL,
|
|
94
|
+
commit_sha TEXT NOT NULL,
|
|
95
|
+
created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ', 'now')),
|
|
96
|
+
closed_at TEXT,
|
|
97
|
+
handle TEXT
|
|
98
|
+
);
|
|
99
|
+
"""
|
|
100
|
+
|
|
101
|
+
# Indexes run after the migration, never with it: on a store still in the old
|
|
102
|
+
# shape none of these columns exists yet.
|
|
103
|
+
#
|
|
104
|
+
# Two per-row indexes on functions, not three. A run-keyed seek and an
|
|
105
|
+
# identity-keyed seek are the only two shapes any read asks for; a third index
|
|
106
|
+
# keyed (run_id, identity_id) answered neither of them better and cost 10.8 MB
|
|
107
|
+
# and a fifth of the insert time on the flagship consumer's store.
|
|
108
|
+
_INDEXES = """
|
|
109
|
+
CREATE INDEX IF NOT EXISTS idx_functions_run ON functions(run_id);
|
|
110
|
+
CREATE INDEX IF NOT EXISTS idx_functions_identity ON functions(identity_id, run_id);
|
|
111
|
+
CREATE INDEX IF NOT EXISTS idx_attempts_open ON attempts(closed_at);
|
|
112
|
+
"""
|
|
113
|
+
|
|
114
|
+
# Indexes an earlier shape carried that nothing reads now. idx_identities_path
|
|
115
|
+
# is what the reordered UNIQUE replaced; both are dropped on open.
|
|
116
|
+
_DEAD_INDEXES = ("idx_functions_run_path", "idx_identities_path")
|
|
117
|
+
|
|
118
|
+
_JOINED = "FROM functions f JOIN identities i ON i.id = f.identity_id"
|
|
119
|
+
# Path-scoped reads, identities first. CROSS JOIN is SQLite's documented way to
|
|
120
|
+
# pin the outer table, and pinning it is the whole difference: a path names a
|
|
121
|
+
# handful of identities, a run names a hundred thousand rows, and with no
|
|
122
|
+
# table statistics the planner picks the run and scans it.
|
|
123
|
+
_BY_PATH = "FROM identities i CROSS JOIN functions f ON f.identity_id = i.id"
|
|
124
|
+
_ID_COLS = "i.scope, i.path, i.long_name"
|
|
125
|
+
_METRIC_COLS = "f.start, f.end, f.ccn_std, f.ccn_mod, f.ccn, f.nloc, f.params, f.nesting"
|
|
126
|
+
_INV_COLS = f"{_ID_COLS}, {_METRIC_COLS}, f.cognitive"
|
|
127
|
+
_ALL_COLS = f"{_ID_COLS}, {_METRIC_COLS}, f.cov, f.flag, f.crap, f.remedy, f.cognitive"
|
|
128
|
+
_CRAP_COLS = f"{_ID_COLS}, f.crap"
|
|
129
|
+
# what write_run binds per row: everything but the three identity strings
|
|
130
|
+
_WRITE_COLS = ("start, end, ccn_std, ccn_mod, ccn, nloc, params, nesting, "
|
|
131
|
+
"cov, flag, crap, remedy, cognitive")
|
|
132
|
+
_N_COLS = _WRITE_COLS.count(",") + 3 # every _WRITE_COLS column plus run_id and identity_id
|
|
133
|
+
# the identity strings, never the ids: a row's place in an export may not depend
|
|
134
|
+
# on when its identity was first seen
|
|
135
|
+
_ROW_ORDER = "ORDER BY i.scope, i.path, f.start, f.end, i.long_name"
|
|
136
|
+
|
|
137
|
+
def _selected(**substitutions: str) -> str:
|
|
138
|
+
"""_WRITE_COLS as a SELECT list off alias f, with named columns replaced.
|
|
139
|
+
|
|
140
|
+
The migrations read the live table column for column; the two verdict
|
|
141
|
+
columns arrive from their lookup tables instead.
|
|
142
|
+
"""
|
|
143
|
+
return ", ".join(substitutions.get(col, f"f.{col}")
|
|
144
|
+
for col in _WRITE_COLS.replace(" ", "").split(","))
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
_CODED_COLS = _selected(flag="fl.id", remedy="rm.id")
|
|
148
|
+
_CODE_JOIN = "LEFT JOIN flags fl ON fl.name = f.flag LEFT JOIN remedies rm ON rm.name = f.remedy"
|
|
149
|
+
|
|
150
|
+
# Names this crapkit does not know still get a code rather than a NULL: the
|
|
151
|
+
# LEFT JOIN below would silently drop a verdict a newer scorer invented.
|
|
152
|
+
_HARVEST = tuple(
|
|
153
|
+
f"INSERT OR IGNORE INTO {table} (name) SELECT DISTINCT {column} FROM functions "
|
|
154
|
+
f"WHERE {column} IS NOT NULL AND typeof({column}) = 'text'"
|
|
155
|
+
for table, column in (("flags", "flag"), ("remedies", "remedy")))
|
|
156
|
+
|
|
157
|
+
# The old-shape rewrite, in order, run as one transaction. The live table is
|
|
158
|
+
# read until the second-to-last statement and dropped only once its replacement
|
|
159
|
+
# is full, so an interrupt anywhere rolls back to a database the old code reads.
|
|
160
|
+
# It lands rows in the CURRENT shape, codes and all, so a pre-identity store
|
|
161
|
+
# needs one rewrite rather than two.
|
|
162
|
+
_IDENTITY_MIGRATION = (
|
|
163
|
+
# a killed process leaves no temp table behind — SQLite rolls the whole
|
|
164
|
+
# transaction back — but a retry must not trip over one either way
|
|
165
|
+
"DROP TABLE IF EXISTS functions_mig",
|
|
166
|
+
"DROP TABLE IF EXISTS identities", # the empty one _SCHEMA just created
|
|
167
|
+
_IDENTITIES_DDL.format(table="identities"),
|
|
168
|
+
"INSERT INTO identities (scope, path, long_name) "
|
|
169
|
+
"SELECT DISTINCT scope, path, long_name FROM functions ORDER BY scope, path, long_name",
|
|
170
|
+
*_HARVEST,
|
|
171
|
+
_FUNCTIONS_DDL.format(table="functions_mig"),
|
|
172
|
+
f"INSERT INTO functions_mig (run_id, identity_id, {_WRITE_COLS}) "
|
|
173
|
+
f"SELECT f.run_id, i.id, {_CODED_COLS} FROM functions f JOIN identities i "
|
|
174
|
+
"ON i.scope = f.scope AND i.path = f.path AND i.long_name = f.long_name "
|
|
175
|
+
f"{_CODE_JOIN}",
|
|
176
|
+
"DROP TABLE functions",
|
|
177
|
+
"ALTER TABLE functions_mig RENAME TO functions",
|
|
178
|
+
)
|
|
179
|
+
|
|
180
|
+
# Re-key the identity table. Nothing moves but the UNIQUE, so the ids on every
|
|
181
|
+
# functions row keep pointing at the same identity.
|
|
182
|
+
_REKEY_IDENTITIES = (
|
|
183
|
+
"DROP TABLE IF EXISTS identities_mig",
|
|
184
|
+
_IDENTITIES_DDL.format(table="identities_mig"),
|
|
185
|
+
"INSERT INTO identities_mig (id, scope, path, long_name) "
|
|
186
|
+
"SELECT id, scope, path, long_name FROM identities",
|
|
187
|
+
"DROP TABLE identities",
|
|
188
|
+
"ALTER TABLE identities_mig RENAME TO identities",
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
# Verdict strings to codes, same build-and-swap discipline.
|
|
192
|
+
_CODE_MIGRATION = (
|
|
193
|
+
"DROP TABLE IF EXISTS functions_mig",
|
|
194
|
+
*_HARVEST,
|
|
195
|
+
_FUNCTIONS_DDL.format(table="functions_mig"),
|
|
196
|
+
f"INSERT INTO functions_mig (run_id, identity_id, {_WRITE_COLS}) "
|
|
197
|
+
f"SELECT f.run_id, f.identity_id, {_CODED_COLS} FROM functions f {_CODE_JOIN}",
|
|
198
|
+
"DROP TABLE functions",
|
|
199
|
+
"ALTER TABLE functions_mig RENAME TO functions",
|
|
200
|
+
)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
class CrapRow(NamedTuple):
|
|
204
|
+
"""A scored function reduced to what a comparison between two runs needs.
|
|
205
|
+
|
|
206
|
+
These four fields are the whole of what build_digest reads: the key it pairs
|
|
207
|
+
functions on, the number it compares, and the scope whose ceiling decides
|
|
208
|
+
over-target. The other twelve columns of a ScoredRow are 140,000 rows of
|
|
209
|
+
dead weight per run, twice per digest.
|
|
210
|
+
"""
|
|
211
|
+
scope: str
|
|
212
|
+
path: str
|
|
213
|
+
long_name: str
|
|
214
|
+
crap: float
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
class _Codes(NamedTuple):
|
|
218
|
+
"""One lookup table, both ways: names on the way in, back out on the way out."""
|
|
219
|
+
ids: dict
|
|
220
|
+
names: dict
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _read_codes(conn, table: str) -> _Codes:
|
|
224
|
+
rows = conn.execute(f"SELECT id, name FROM {table}").fetchall()
|
|
225
|
+
return _Codes({name: code for code, name in rows}, dict(rows))
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _code(ids: dict, name):
|
|
229
|
+
"""The stored code for a verdict string. None stays None: an inventory row
|
|
230
|
+
has no verdict, and a store written before coverage existed holds NULLs."""
|
|
231
|
+
return None if name is None else ids[name]
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _name(names: dict, code):
|
|
235
|
+
"""The verdict string a stored code stands for."""
|
|
236
|
+
return None if code is None else names[code]
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _deflate(text: str) -> bytes:
|
|
240
|
+
"""A run's lane record, compressed. It is by far the largest thing on the run
|
|
241
|
+
row — a failing lane records every failure by name — and JSON of that shape
|
|
242
|
+
goes to a fifth of its bytes."""
|
|
243
|
+
return zlib.compress(text.encode("utf-8"), 6)
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _inflate(stored) -> str:
|
|
247
|
+
"""The lane record back as JSON text. A row written before the column was
|
|
248
|
+
deflated holds the text itself, so both storage classes read the same."""
|
|
249
|
+
return zlib.decompress(stored).decode("utf-8") if isinstance(stored, bytes) else stored
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _verdict_names(rows: list) -> tuple[set, set]:
|
|
253
|
+
"""The flag and remedy strings this batch stores. Inventory rows carry
|
|
254
|
+
neither, so they name nothing."""
|
|
255
|
+
scored = [row for row in rows if len(row) == 16]
|
|
256
|
+
return ({row[12] for row in scored} - {None}, {row[14] for row in scored} - {None})
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _writable(row, flags: dict, remedies: dict):
|
|
260
|
+
"""A row's metric columns in _WRITE_COLS order, identity dropped, verdict coded.
|
|
261
|
+
|
|
262
|
+
Scored rows already carry the four coverage columns; inventory rows carry
|
|
263
|
+
cognitive last, so the four unscored columns slot in before it.
|
|
264
|
+
"""
|
|
265
|
+
if len(row) == 16:
|
|
266
|
+
return (*row[3:12], _code(flags, row[12]), row[13], _code(remedies, row[14]), row[15])
|
|
267
|
+
return (*row[3:11], None, None, None, None, row[11])
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def _own_ceilings(target: int, scope_targets: dict[str, int] | None) -> list[tuple[str, int]]:
|
|
271
|
+
"""The scopes whose ceiling is not the repo's, sorted.
|
|
272
|
+
|
|
273
|
+
A scope that declares the repo target declares nothing: its branch and the
|
|
274
|
+
ELSE say the same number. Dropping it is what lets a config with per-scope
|
|
275
|
+
blocks but no per-scope targets compare against one bound parameter over a
|
|
276
|
+
million rows of history.
|
|
277
|
+
"""
|
|
278
|
+
return sorted((scope, ceiling) for scope, ceiling in (scope_targets or {}).items()
|
|
279
|
+
if ceiling != target)
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
class _Ceiling(NamedTuple):
|
|
283
|
+
"""The CRAP ceiling a row is compared against, as SQL.
|
|
284
|
+
|
|
285
|
+
`per_scope` says whether the expression reads i.scope. It decides whether a
|
|
286
|
+
whole-run total has to join the identity table at all, which is the only
|
|
287
|
+
reason that query ever joined it.
|
|
288
|
+
"""
|
|
289
|
+
expr: str
|
|
290
|
+
params: list
|
|
291
|
+
per_scope: bool
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def _ceiling_expr(target: int, scope_targets: dict[str, int] | None) -> _Ceiling:
|
|
295
|
+
"""The per-scope CRAP ceiling as a parameterized CASE, so an over-target count
|
|
296
|
+
decided in SQL is decided exactly the way digest._over_count decides it."""
|
|
297
|
+
own = _own_ceilings(target, scope_targets)
|
|
298
|
+
params: list = []
|
|
299
|
+
for scope, ceiling in own:
|
|
300
|
+
params.extend((scope, ceiling))
|
|
301
|
+
params.append(target)
|
|
302
|
+
if not own:
|
|
303
|
+
return _Ceiling("?", params, False) # a CASE with no WHEN is not SQL
|
|
304
|
+
return _Ceiling(f"CASE i.scope {' '.join('WHEN ? THEN ?' for _ in own)} ELSE ? END",
|
|
305
|
+
params, True)
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def _scored_rows(cur, flags: dict, remedies: dict) -> list:
|
|
309
|
+
"""A cursor over _ALL_COLS, streamed into ScoredRows with identities interned.
|
|
310
|
+
|
|
311
|
+
Rows stream off the cursor rather than through a fetchall list, and the
|
|
312
|
+
identity strings are interned process-wide, so a second read reuses the
|
|
313
|
+
first one's strings. The verdict columns need no interning: every row of a
|
|
314
|
+
run shares the one string its code names.
|
|
315
|
+
"""
|
|
316
|
+
from .score import ScoredRow
|
|
317
|
+
si = sys.intern
|
|
318
|
+
return [ScoredRow(si(scope), si(path), si(name), a, b, c, d, e, f, g, h, cov,
|
|
319
|
+
_name(flags, flag), crap, _name(remedies, remedy), cog)
|
|
320
|
+
for scope, path, name, a, b, c, d, e, f, g, h, cov, flag, crap, remedy, cog in cur]
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _scope_clause(scopes, *, keyword: str = "IN") -> tuple[str, list]:
|
|
324
|
+
"""A parameterized `AND i.scope IN (...)`, empty when no scope was named.
|
|
325
|
+
|
|
326
|
+
Deduplicated and sorted so the same request always builds the same SQL text,
|
|
327
|
+
which is what lets SQLite reuse the prepared statement across calls. NOT IN
|
|
328
|
+
is the same cut the other way: what a scope-blind read must leave behind.
|
|
329
|
+
"""
|
|
330
|
+
if not scopes:
|
|
331
|
+
return "", []
|
|
332
|
+
names = sorted(set(scopes))
|
|
333
|
+
return f"AND i.scope {keyword} ({','.join('?' * len(names))})", names
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
class SnapshotStore:
|
|
337
|
+
def __init__(self, path: Path | str):
|
|
338
|
+
self._path = Path(path)
|
|
339
|
+
self._identities: dict[tuple[str, str, str], int] = {}
|
|
340
|
+
self._conn = sqlite3.connect(str(path))
|
|
341
|
+
self._conn.executescript(_SCHEMA)
|
|
342
|
+
self._seed_codes()
|
|
343
|
+
self._migrate()
|
|
344
|
+
# after the migration, never with it: on a store still in the old shape
|
|
345
|
+
# these index columns do not exist yet
|
|
346
|
+
self._conn.executescript(_INDEXES)
|
|
347
|
+
self._conn.commit()
|
|
348
|
+
# last, because a migration can mint codes for names it met on the way
|
|
349
|
+
self._codes = {table: _read_codes(self._conn, table) for table in _CODE_SEEDS}
|
|
350
|
+
|
|
351
|
+
def _seed_codes(self) -> None:
|
|
352
|
+
"""The known verdict names, at their fixed codes. Before any migration:
|
|
353
|
+
the rewrites resolve every stored string through these tables."""
|
|
354
|
+
for table, names in _CODE_SEEDS.items():
|
|
355
|
+
self._conn.executemany(
|
|
356
|
+
f"INSERT OR IGNORE INTO {table} (id, name) VALUES (?, ?)",
|
|
357
|
+
list(enumerate(names, 1)))
|
|
358
|
+
|
|
359
|
+
def _migrate(self) -> None:
|
|
360
|
+
# _SCHEMA and _INDEXES are migration paths in themselves: every statement
|
|
361
|
+
# in them is IF NOT EXISTS and runs on every open, so a database written
|
|
362
|
+
# before a table or an index existed grows it the next time it is opened.
|
|
363
|
+
# What is left here is what CREATE cannot express — a column added to a
|
|
364
|
+
# table that already exists, and the two rewrites.
|
|
365
|
+
self._add_coverage_columns()
|
|
366
|
+
self._add_cognitive_column()
|
|
367
|
+
self._add_run_provenance_columns()
|
|
368
|
+
self._add_claim_handle_column()
|
|
369
|
+
self._conn.commit()
|
|
370
|
+
self._normalize_identities()
|
|
371
|
+
self._restack()
|
|
372
|
+
|
|
373
|
+
def size_bytes(self) -> int:
|
|
374
|
+
"""The file as the OS sees it: what a prune has to move to mean anything."""
|
|
375
|
+
return self._path.stat().st_size if self._path.is_file() else 0
|
|
376
|
+
|
|
377
|
+
def _declared_types(self, table: str) -> dict[str, str]:
|
|
378
|
+
return {row[1]: row[2].upper() for row in self._conn.execute(f"PRAGMA table_info({table})")}
|
|
379
|
+
|
|
380
|
+
def _existing_columns(self, table: str) -> set[str]:
|
|
381
|
+
return set(self._declared_types(table))
|
|
382
|
+
|
|
383
|
+
def _add_coverage_columns(self) -> None:
|
|
384
|
+
have = self._existing_columns("functions")
|
|
385
|
+
for col, decl in (("cov", "REAL"), ("flag", "INTEGER"),
|
|
386
|
+
("crap", "REAL"), ("remedy", "INTEGER")):
|
|
387
|
+
if col not in have:
|
|
388
|
+
self._conn.execute(f"ALTER TABLE functions ADD COLUMN {col} {decl}")
|
|
389
|
+
|
|
390
|
+
def _add_cognitive_column(self) -> None:
|
|
391
|
+
if "cognitive" not in self._existing_columns("functions"):
|
|
392
|
+
self._conn.execute(
|
|
393
|
+
"ALTER TABLE functions ADD COLUMN cognitive INTEGER NOT NULL DEFAULT 0")
|
|
394
|
+
|
|
395
|
+
def _add_run_provenance_columns(self) -> None:
|
|
396
|
+
run_cols = self._existing_columns("runs")
|
|
397
|
+
if "lanes" not in run_cols:
|
|
398
|
+
self._conn.execute("ALTER TABLE runs ADD COLUMN lanes TEXT NOT NULL DEFAULT '{}'")
|
|
399
|
+
if "kind" not in run_cols:
|
|
400
|
+
self._conn.execute("ALTER TABLE runs ADD COLUMN kind TEXT NOT NULL DEFAULT 'coverage'")
|
|
401
|
+
# rows that predate the column have unknown provenance; the DEFAULT
|
|
402
|
+
# must not promote them to baseline-grade 'coverage'
|
|
403
|
+
self._conn.execute("UPDATE runs SET kind = 'legacy'")
|
|
404
|
+
if "verdict_ok" not in run_cols:
|
|
405
|
+
self._conn.execute("ALTER TABLE runs ADD COLUMN verdict_ok INTEGER")
|
|
406
|
+
if "findings" not in run_cols:
|
|
407
|
+
self._conn.execute(
|
|
408
|
+
"ALTER TABLE runs ADD COLUMN findings INTEGER NOT NULL DEFAULT 0")
|
|
409
|
+
|
|
410
|
+
def _add_claim_handle_column(self) -> None:
|
|
411
|
+
"""The name a claim was taken under, beside the long_name it was taken on.
|
|
412
|
+
|
|
413
|
+
An anonymous function's long_name is `(anonymous)` for every anonymous
|
|
414
|
+
function in its file, so a release by long_name closes whichever claim
|
|
415
|
+
sorts first. The handle is the ordinal that tells them apart, and it is
|
|
416
|
+
stored rather than recomputed: the whole point is that it survives the
|
|
417
|
+
line shifts the session's own edit makes.
|
|
418
|
+
"""
|
|
419
|
+
if "handle" not in self._existing_columns("attempts"):
|
|
420
|
+
self._conn.execute("ALTER TABLE attempts ADD COLUMN handle TEXT")
|
|
421
|
+
|
|
422
|
+
def _normalize_identities(self) -> None:
|
|
423
|
+
"""Move scope, path and long_name out of every functions row.
|
|
424
|
+
|
|
425
|
+
The old shape is the one that still has a scope column. The rewrite runs
|
|
426
|
+
as one transaction and swaps the table in last, so an interrupt leaves
|
|
427
|
+
the old table whole, rows and all, and the next open retries from there.
|
|
428
|
+
Old databases migrate silently; `runs prune` reclaims the freed pages.
|
|
429
|
+
"""
|
|
430
|
+
if "scope" not in self._existing_columns("functions"):
|
|
431
|
+
return
|
|
432
|
+
self._conn.execute("BEGIN") # DDL does not open one, and this must be atomic
|
|
433
|
+
try:
|
|
434
|
+
for statement in _IDENTITY_MIGRATION:
|
|
435
|
+
self._conn.execute(statement)
|
|
436
|
+
self._conn.commit()
|
|
437
|
+
except Exception:
|
|
438
|
+
self._conn.rollback()
|
|
439
|
+
raise
|
|
440
|
+
|
|
441
|
+
def _identity_key(self) -> tuple:
|
|
442
|
+
"""The columns of the identity UNIQUE, in key order."""
|
|
443
|
+
unique = [row[1] for row in self._conn.execute("PRAGMA index_list(identities)") if row[2]]
|
|
444
|
+
if not unique:
|
|
445
|
+
return ()
|
|
446
|
+
return tuple(row[2] for row in self._conn.execute(f'PRAGMA index_info("{unique[0]}")'))
|
|
447
|
+
|
|
448
|
+
def _restack_steps(self) -> tuple:
|
|
449
|
+
"""The rewrites this store still owes, in the order they must run.
|
|
450
|
+
|
|
451
|
+
Both are version-detected off the schema itself rather than a stamp: a
|
|
452
|
+
store carries its own shape, and a stamp is one more thing to get wrong.
|
|
453
|
+
"""
|
|
454
|
+
steps = tuple(f"DROP INDEX IF EXISTS {name}" for name in _DEAD_INDEXES)
|
|
455
|
+
if self._identity_key() != _IDENTITY_KEY:
|
|
456
|
+
steps += _REKEY_IDENTITIES
|
|
457
|
+
if self._declared_types("functions").get("flag") == "TEXT":
|
|
458
|
+
steps += _CODE_MIGRATION
|
|
459
|
+
return steps
|
|
460
|
+
|
|
461
|
+
def _deflate_lanes(self) -> None:
|
|
462
|
+
"""Compress the lane records still held as text. Never conditional on the
|
|
463
|
+
rewrites: a store this code restacked can still be written by an older
|
|
464
|
+
crapkit, and the next open should take those rows too."""
|
|
465
|
+
rows = self._conn.execute(
|
|
466
|
+
"SELECT id, lanes FROM runs WHERE typeof(lanes) = 'text'").fetchall()
|
|
467
|
+
self._conn.executemany("UPDATE runs SET lanes = ? WHERE id = ?",
|
|
468
|
+
[(_deflate(text), rid) for rid, text in rows])
|
|
469
|
+
|
|
470
|
+
def _restack(self) -> None:
|
|
471
|
+
"""Re-key the identities, code the verdicts, deflate the lane records.
|
|
472
|
+
|
|
473
|
+
Together these took the flagship consumer's store from 131.4 to 92.3 MB
|
|
474
|
+
with every timed read faster. One transaction, tables built under temp
|
|
475
|
+
names and swapped in last, so a process killed anywhere in here leaves a
|
|
476
|
+
database the previous code still reads and the next open retries.
|
|
477
|
+
"""
|
|
478
|
+
self._conn.execute("BEGIN") # DDL does not open one, and this must be atomic
|
|
479
|
+
try:
|
|
480
|
+
for statement in self._restack_steps():
|
|
481
|
+
self._conn.execute(statement)
|
|
482
|
+
self._deflate_lanes()
|
|
483
|
+
self._conn.commit()
|
|
484
|
+
except Exception:
|
|
485
|
+
self._conn.rollback()
|
|
486
|
+
raise
|
|
487
|
+
|
|
488
|
+
def _cache_identities(self) -> None:
|
|
489
|
+
"""Every identity in the store, keyed by triple.
|
|
490
|
+
|
|
491
|
+
One scan, and the strings are interned, so the cache shares them with
|
|
492
|
+
the rows a read of the same store already built.
|
|
493
|
+
"""
|
|
494
|
+
si = sys.intern
|
|
495
|
+
self._identities = {
|
|
496
|
+
(si(scope), si(path), si(name)): rid
|
|
497
|
+
for rid, scope, path, name in self._conn.execute(
|
|
498
|
+
"SELECT id, scope, path, long_name FROM identities")}
|
|
499
|
+
|
|
500
|
+
def _identity_ids(self, rows: list) -> dict[tuple[str, str, str], int]:
|
|
501
|
+
"""(scope, path, long_name) -> identities.id, covering every row.
|
|
502
|
+
|
|
503
|
+
One pass over the rows and at most two over the identity table, never a
|
|
504
|
+
lookup per row: a rebuild rewrites the same hundred thousand identities
|
|
505
|
+
every run. The reload after the insert is what makes INSERT OR IGNORE
|
|
506
|
+
safe — a triple a concurrent writer got in first is read back rather
|
|
507
|
+
than left unresolved.
|
|
508
|
+
"""
|
|
509
|
+
if not self._identities:
|
|
510
|
+
self._cache_identities()
|
|
511
|
+
fresh = sorted({row[:3] for row in rows} - self._identities.keys())
|
|
512
|
+
if fresh:
|
|
513
|
+
self._conn.executemany(
|
|
514
|
+
"INSERT OR IGNORE INTO identities (scope, path, long_name) VALUES (?, ?, ?)",
|
|
515
|
+
fresh)
|
|
516
|
+
self._cache_identities()
|
|
517
|
+
return self._identities
|
|
518
|
+
|
|
519
|
+
def _code_ids(self, table: str, names: set) -> dict:
|
|
520
|
+
"""name -> code, covering every name this batch will store.
|
|
521
|
+
|
|
522
|
+
The seeded vocabulary answers every name a crapkit run produces, so this
|
|
523
|
+
inserts nothing in practice. A name from somewhere else is minted a code
|
|
524
|
+
rather than dropped, and the reload after the insert is what makes
|
|
525
|
+
INSERT OR IGNORE safe against a concurrent writer.
|
|
526
|
+
"""
|
|
527
|
+
fresh = sorted(names - self._codes[table].ids.keys())
|
|
528
|
+
if fresh:
|
|
529
|
+
self._conn.executemany(f"INSERT OR IGNORE INTO {table} (name) VALUES (?)",
|
|
530
|
+
[(name,) for name in fresh])
|
|
531
|
+
self._codes[table] = _read_codes(self._conn, table)
|
|
532
|
+
return self._codes[table].ids
|
|
533
|
+
|
|
534
|
+
def write_run(self, *, commit: str, tool_versions: dict[str, str], rows: list,
|
|
535
|
+
lanes: dict | None = None, kind: str = "coverage") -> int:
|
|
536
|
+
flag_names, remedy_names = _verdict_names(rows)
|
|
537
|
+
with self._conn:
|
|
538
|
+
cur = self._conn.execute(
|
|
539
|
+
"INSERT INTO runs (commit_sha, tool_versions, lanes, kind) VALUES (?, ?, ?, ?)",
|
|
540
|
+
(commit, json.dumps(tool_versions, sort_keys=True),
|
|
541
|
+
_deflate(json.dumps(lanes or {}, sort_keys=True)), kind),
|
|
542
|
+
)
|
|
543
|
+
run_id = cur.lastrowid
|
|
544
|
+
ids = self._identity_ids(rows)
|
|
545
|
+
flags = self._code_ids("flags", flag_names)
|
|
546
|
+
remedies = self._code_ids("remedies", remedy_names)
|
|
547
|
+
# a generator, not two lists: executemany consumes any iterator, and
|
|
548
|
+
# materializing the padded copy doubled the rows in memory alongside
|
|
549
|
+
# the caller's own list for the length of the insert
|
|
550
|
+
self._conn.executemany(
|
|
551
|
+
f"INSERT INTO functions (run_id, identity_id, {_WRITE_COLS}) "
|
|
552
|
+
f"VALUES ({','.join('?' * _N_COLS)})",
|
|
553
|
+
((run_id, ids[row[:3]], *_writable(row, flags, remedies)) for row in rows),
|
|
554
|
+
)
|
|
555
|
+
return run_id
|
|
556
|
+
|
|
557
|
+
def read_rows(self, run_id: int, *, min_ccn: int = 0,
|
|
558
|
+
scopes: list[str] | None = None) -> list[InventoryRow]:
|
|
559
|
+
"""Inventory rows for one run, in export order.
|
|
560
|
+
|
|
561
|
+
min_ccn is the caller's admission floor pushed into the scan: a row a
|
|
562
|
+
worklist could never admit costs nothing to leave in SQLite. scopes is
|
|
563
|
+
the same idea for a scope-shaped cut, decided on the joined identity.
|
|
564
|
+
|
|
565
|
+
Rows stream off the cursor rather than through a fetchall list, and the
|
|
566
|
+
identity strings are interned. A run holds thousands of distinct paths
|
|
567
|
+
repeated across a hundred thousand rows, and interning is process-wide,
|
|
568
|
+
so a second run read in the same process reuses the first one's strings.
|
|
569
|
+
"""
|
|
570
|
+
clause, names = _scope_clause(scopes)
|
|
571
|
+
cur = self._conn.execute(
|
|
572
|
+
f"SELECT {_INV_COLS} {_JOINED} WHERE f.run_id = ? AND f.ccn >= ? "
|
|
573
|
+
f"{clause} {_ROW_ORDER}",
|
|
574
|
+
(run_id, min_ccn, *names),
|
|
575
|
+
)
|
|
576
|
+
si = sys.intern
|
|
577
|
+
return [InventoryRow(si(scope), si(path), si(name), a, b, c, d, e, f, g, h, cog)
|
|
578
|
+
for scope, path, name, a, b, c, d, e, f, g, h, cog in cur]
|
|
579
|
+
|
|
580
|
+
def read_scored(self, run_id: int, *, min_ccn: int = 0,
|
|
581
|
+
scopes: list[str] | None = None) -> list:
|
|
582
|
+
"""Scored rows for one run, same streaming and interning as read_rows.
|
|
583
|
+
|
|
584
|
+
This is where interning pays: a digest holds two runs at once, and the
|
|
585
|
+
second one costs almost nothing for paths the first already interned.
|
|
586
|
+
"""
|
|
587
|
+
clause, names = _scope_clause(scopes)
|
|
588
|
+
return self._scored(self._conn.execute(
|
|
589
|
+
f"SELECT {_ALL_COLS} {_JOINED} WHERE f.run_id = ? AND f.crap IS NOT NULL "
|
|
590
|
+
f"AND f.ccn >= ? {clause} {_ROW_ORDER}",
|
|
591
|
+
(run_id, min_ccn, *names),
|
|
592
|
+
))
|
|
593
|
+
|
|
594
|
+
def _scored(self, cur) -> list:
|
|
595
|
+
return _scored_rows(cur, self._codes["flags"].names, self._codes["remedies"].names)
|
|
596
|
+
|
|
597
|
+
def read_crap(self, run_id: int) -> list[CrapRow]:
|
|
598
|
+
"""One run's scored functions as (scope, path, long_name, crap).
|
|
599
|
+
|
|
600
|
+
What `digest` compares. It holds two whole runs at once and reads four
|
|
601
|
+
of the sixteen fields, so the twelve it does not read are paid for twice
|
|
602
|
+
over — 1,328 ms of a digest on the flagship consumer's store.
|
|
603
|
+
"""
|
|
604
|
+
cur = self._conn.execute(
|
|
605
|
+
f"SELECT {_CRAP_COLS} {_JOINED} WHERE f.run_id = ? AND f.crap IS NOT NULL "
|
|
606
|
+
f"{_ROW_ORDER}", (run_id,))
|
|
607
|
+
si = sys.intern
|
|
608
|
+
return [CrapRow(si(scope), si(path), si(name), crap)
|
|
609
|
+
for scope, path, name, crap in cur]
|
|
610
|
+
|
|
611
|
+
def read_scored_file(self, run_id: int, path: str) -> list:
|
|
612
|
+
"""One file's scored rows: the path seeks identities, the ids seek the run.
|
|
613
|
+
|
|
614
|
+
A brief asks about a single function; reading the whole run to find it
|
|
615
|
+
materializes every other row of a hundred-thousand-function repo.
|
|
616
|
+
"""
|
|
617
|
+
return self._scored(self._conn.execute(
|
|
618
|
+
f"SELECT {_ALL_COLS} {_BY_PATH} WHERE i.path = ? AND f.run_id = ? "
|
|
619
|
+
f"AND f.crap IS NOT NULL {_ROW_ORDER}",
|
|
620
|
+
(path, run_id),
|
|
621
|
+
))
|
|
622
|
+
|
|
623
|
+
def read_marks(self, run_id: int, *, min_ccn: int = 0,
|
|
624
|
+
scopes: list[str] | None = None) -> dict[tuple[str, str], tuple[str, str]]:
|
|
625
|
+
"""(path, long_name) -> (flag, remedy) for every row this run scored.
|
|
626
|
+
|
|
627
|
+
Two columns, not the row: the worklist reads inventory rows and needs
|
|
628
|
+
only the verdict half — which rows a floor must not hide, which the
|
|
629
|
+
queue will never offer, which are finished — so building a ScoredRow per
|
|
630
|
+
function would double the cost of the command. Twins collapse to the
|
|
631
|
+
WORST of them, the same rule the ratchet and the verdict use, because
|
|
632
|
+
the ORDER BY lets the highest-CRAP twin overwrite its siblings. An
|
|
633
|
+
inventory-only run answers nothing: its remedy is NULL.
|
|
634
|
+
"""
|
|
635
|
+
clause, names = _scope_clause(scopes)
|
|
636
|
+
cur = self._conn.execute(
|
|
637
|
+
f"SELECT i.path, i.long_name, f.flag, f.remedy {_JOINED} WHERE f.run_id = ? "
|
|
638
|
+
f"AND f.remedy IS NOT NULL AND f.ccn >= ? {clause} ORDER BY f.crap",
|
|
639
|
+
(run_id, min_ccn, *names))
|
|
640
|
+
flags, remedies = self._codes["flags"].names, self._codes["remedies"].names
|
|
641
|
+
return {(path, name): (_name(flags, flag), _name(remedies, remedy))
|
|
642
|
+
for path, name, flag, remedy in cur}
|
|
643
|
+
|
|
644
|
+
def count_by_path(self, run_id: int, *, flag: str,
|
|
645
|
+
skip_scopes=frozenset()) -> list[tuple[str, int, int]]:
|
|
646
|
+
"""Per path in one run: (path, functions, others), path order.
|
|
647
|
+
|
|
648
|
+
`others` counts the rows whose flag is NOT the named one, which is all
|
|
649
|
+
doctor asks of a run: a directory is a measurement gap only when nothing
|
|
650
|
+
in it carries any other verdict. A run holds a hundred thousand rows and
|
|
651
|
+
a few thousand paths, so the grouping belongs where the rows are — and
|
|
652
|
+
the scopes to leave out belong in the WHERE, not in a Python filter that
|
|
653
|
+
reads them first.
|
|
654
|
+
"""
|
|
655
|
+
clause, names = _scope_clause(skip_scopes, keyword="NOT IN")
|
|
656
|
+
cur = self._conn.execute(
|
|
657
|
+
f"SELECT i.path, COUNT(*), SUM(f.flag IS NOT ?) {_JOINED} "
|
|
658
|
+
f"WHERE f.run_id = ? AND f.crap IS NOT NULL {clause} GROUP BY i.path "
|
|
659
|
+
"ORDER BY i.path",
|
|
660
|
+
(_code(self._codes["flags"].ids, flag), run_id, *names))
|
|
661
|
+
return cur.fetchall()
|
|
662
|
+
|
|
663
|
+
def count_scored_below(self, run_id: int, min_ccn: int,
|
|
664
|
+
scopes: list[str] | None = None) -> int:
|
|
665
|
+
"""The rows read_scored(min_ccn=...) skipped. An empty queue still has to
|
|
666
|
+
report them, and one COUNT is cheaper than reading them to count them.
|
|
667
|
+
|
|
668
|
+
Same scopes the read took, or a scoped queue reports rows it was never
|
|
669
|
+
going to offer.
|
|
670
|
+
"""
|
|
671
|
+
clause, names = _scope_clause(scopes)
|
|
672
|
+
cur = self._conn.execute(
|
|
673
|
+
f"SELECT COUNT(*) {_JOINED} WHERE f.run_id = ? AND f.crap IS NOT NULL "
|
|
674
|
+
f"AND f.ccn < ? {clause}",
|
|
675
|
+
(run_id, min_ccn, *names))
|
|
676
|
+
return cur.fetchone()[0]
|
|
677
|
+
|
|
678
|
+
def run_totals(self, *, target: int,
|
|
679
|
+
scope_targets: dict[str, int] | None = None) -> dict[int, tuple]:
|
|
680
|
+
"""Per-run (functions, over_target, crap_load) summed inside the scan.
|
|
681
|
+
|
|
682
|
+
trend used to build every ScoredRow of every trusted run to add up three
|
|
683
|
+
numbers and throw the rows away. The identity table is joined only when
|
|
684
|
+
the ceiling really is per-scope: a repo whose scopes all take the repo
|
|
685
|
+
target reads the rows and skips 1.12 M index seeks.
|
|
686
|
+
"""
|
|
687
|
+
ceiling = _ceiling_expr(target, scope_targets)
|
|
688
|
+
source = _JOINED if ceiling.per_scope else "FROM functions f"
|
|
689
|
+
cur = self._conn.execute(
|
|
690
|
+
f"SELECT f.run_id, COUNT(*), SUM(f.crap > {ceiling.expr}), SUM(f.crap) {source} "
|
|
691
|
+
"WHERE f.crap IS NOT NULL GROUP BY f.run_id", ceiling.params)
|
|
692
|
+
return {run_id: (n, over, load) for run_id, n, over, load in cur}
|
|
693
|
+
|
|
694
|
+
def run_scope_totals(self, *, target: int,
|
|
695
|
+
scope_targets: dict[str, int] | None = None) -> dict[int, dict[str, tuple]]:
|
|
696
|
+
"""run_totals cut one level finer: (functions, over_target, crap_load) per
|
|
697
|
+
(run, scope), summed inside the same scan rather than by reading rows.
|
|
698
|
+
|
|
699
|
+
This one groups BY the scope, so it joins the identity whatever shape the
|
|
700
|
+
ceiling takes.
|
|
701
|
+
"""
|
|
702
|
+
ceiling = _ceiling_expr(target, scope_targets)
|
|
703
|
+
cur = self._conn.execute(
|
|
704
|
+
f"SELECT f.run_id, i.scope, COUNT(*), SUM(f.crap > {ceiling.expr}), "
|
|
705
|
+
f"SUM(f.crap) {_JOINED} WHERE f.crap IS NOT NULL "
|
|
706
|
+
"GROUP BY f.run_id, i.scope ORDER BY f.run_id, i.scope",
|
|
707
|
+
ceiling.params)
|
|
708
|
+
out: dict[int, dict[str, tuple]] = {}
|
|
709
|
+
for run_id, scope, n, over, load in cur:
|
|
710
|
+
out.setdefault(run_id, {})[scope] = (n, over, load)
|
|
711
|
+
return out
|
|
712
|
+
|
|
713
|
+
def function_span(self, run_id: int, path: str, long_name: str) -> tuple | None:
|
|
714
|
+
"""One function's (start, end) in one run, off the identity path index.
|
|
715
|
+
|
|
716
|
+
Same tie-break as read_rows, which this replaced: path and long_name are
|
|
717
|
+
pinned by the WHERE, so export order reduces to scope, start, end.
|
|
718
|
+
"""
|
|
719
|
+
cur = self._conn.execute(
|
|
720
|
+
f"SELECT f.start, f.end {_BY_PATH} WHERE i.path = ? AND i.long_name = ? "
|
|
721
|
+
"AND f.run_id = ? ORDER BY i.scope, f.start, f.end LIMIT 1",
|
|
722
|
+
(path, long_name, run_id))
|
|
723
|
+
return cur.fetchone()
|
|
724
|
+
|
|
725
|
+
def set_verdict_ok(self, run_id: int, ok: bool, *, findings: int = 0) -> None:
|
|
726
|
+
"""Stamp a verdict on a run, with how many findings it carried.
|
|
727
|
+
|
|
728
|
+
The count is what a later refusal quotes back: a baseline that skips
|
|
729
|
+
this run has to say what it is protecting, and re-deriving it would mean
|
|
730
|
+
rerunning the lanes on a tree that has moved on.
|
|
731
|
+
"""
|
|
732
|
+
with self._conn:
|
|
733
|
+
self._conn.execute("UPDATE runs SET verdict_ok = ?, findings = ? WHERE id = ?",
|
|
734
|
+
(1 if ok else 0, findings, run_id))
|
|
735
|
+
|
|
736
|
+
def list_runs(self) -> list[dict]:
|
|
737
|
+
cur = self._conn.execute(
|
|
738
|
+
"SELECT id, commit_sha, tool_versions, lanes, kind, verdict_ok, findings, "
|
|
739
|
+
"created_at FROM runs ORDER BY id")
|
|
740
|
+
return [
|
|
741
|
+
{"id": rid, "commit": sha, "tool_versions": json.loads(tv),
|
|
742
|
+
"lanes": json.loads(_inflate(lanes)), "kind": kind,
|
|
743
|
+
"verdict_ok": None if ok is None else bool(ok), "findings": findings,
|
|
744
|
+
"created_at": ts}
|
|
745
|
+
for rid, sha, tv, lanes, kind, ok, findings, ts in cur.fetchall()
|
|
746
|
+
]
|
|
747
|
+
|
|
748
|
+
def write_overrides(self, run_id: int, rows: list[tuple[str, str, float, str]]) -> None:
|
|
749
|
+
with self._conn:
|
|
750
|
+
self._conn.executemany(
|
|
751
|
+
"INSERT INTO overrides (run_id, path, long_name, crap, reason) VALUES (?,?,?,?,?)",
|
|
752
|
+
[(run_id, *r) for r in rows],
|
|
753
|
+
)
|
|
754
|
+
|
|
755
|
+
def read_overrides(self, run_id: int) -> list[tuple[str, str, float, str]]:
|
|
756
|
+
cur = self._conn.execute(
|
|
757
|
+
"SELECT path, long_name, crap, reason FROM overrides WHERE run_id = ? ORDER BY path, long_name",
|
|
758
|
+
(run_id,))
|
|
759
|
+
return list(cur.fetchall())
|
|
760
|
+
|
|
761
|
+
def record_claim(self, *, path: str, long_name: str, commit: str,
|
|
762
|
+
handle: str | None = None) -> int:
|
|
763
|
+
"""Take a claim on one function. Opt-in: nothing writes here unless a
|
|
764
|
+
session asked for it, so a store with no claims answers every query the
|
|
765
|
+
way it did before claims existed.
|
|
766
|
+
|
|
767
|
+
`handle` is the name the claim was handed out under. None is the honest
|
|
768
|
+
answer for a caller that never had one, and reads back as null.
|
|
769
|
+
"""
|
|
770
|
+
with self._conn:
|
|
771
|
+
cur = self._conn.execute(
|
|
772
|
+
"INSERT INTO attempts (path, long_name, commit_sha, handle) "
|
|
773
|
+
"VALUES (?, ?, ?, ?)", (path, long_name, commit, handle))
|
|
774
|
+
return cur.lastrowid
|
|
775
|
+
|
|
776
|
+
def open_claims(self) -> list[dict]:
|
|
777
|
+
cur = self._conn.execute(
|
|
778
|
+
"SELECT id, path, long_name, commit_sha, created_at, handle FROM attempts "
|
|
779
|
+
"WHERE closed_at IS NULL ORDER BY id")
|
|
780
|
+
return [{"id": cid, "path": p, "long_name": n, "commit": sha,
|
|
781
|
+
"created_at": ts, "handle": handle}
|
|
782
|
+
for cid, p, n, sha, ts, handle in cur]
|
|
783
|
+
|
|
784
|
+
def attempts_for(self, keys) -> dict[tuple[str, str], list[dict]]:
|
|
785
|
+
"""Every claim ever taken on each named function, oldest first.
|
|
786
|
+
|
|
787
|
+
One query for the whole batch, filtered on the indexed path and paired
|
|
788
|
+
up here: a packet run asks about N functions, and a query apiece is N
|
|
789
|
+
round trips for what one path filter already returns. Every requested
|
|
790
|
+
key is in the answer, so a function nobody ever claimed reads as [].
|
|
791
|
+
"""
|
|
792
|
+
wanted = list(dict.fromkeys(keys))
|
|
793
|
+
found: dict[tuple[str, str], list[dict]] = {key: [] for key in wanted}
|
|
794
|
+
if not wanted:
|
|
795
|
+
return found
|
|
796
|
+
paths = sorted({path for path, _ in wanted})
|
|
797
|
+
cur = self._conn.execute(
|
|
798
|
+
"SELECT path, long_name, created_at, closed_at FROM attempts "
|
|
799
|
+
f"WHERE path IN ({','.join('?' * len(paths))}) ORDER BY id", paths)
|
|
800
|
+
for path, long_name, opened, closed in cur:
|
|
801
|
+
if (path, long_name) in found:
|
|
802
|
+
found[(path, long_name)].append({"opened": opened, "closed": closed})
|
|
803
|
+
return found
|
|
804
|
+
|
|
805
|
+
def close_claims(self, claim_ids) -> int:
|
|
806
|
+
"""Stamp the named claims closed; already-closed ones are left alone, so
|
|
807
|
+
a verify that runs twice closes the same claim once."""
|
|
808
|
+
ids = sorted(claim_ids)
|
|
809
|
+
if not ids:
|
|
810
|
+
return 0
|
|
811
|
+
with self._conn:
|
|
812
|
+
cur = self._conn.execute(
|
|
813
|
+
"UPDATE attempts SET closed_at = strftime('%Y-%m-%dT%H:%M:%SZ', 'now') "
|
|
814
|
+
f"WHERE closed_at IS NULL AND id IN ({','.join('?' * len(ids))})", ids)
|
|
815
|
+
return cur.rowcount
|
|
816
|
+
|
|
817
|
+
def _oldest_kept_at(self, keep_ids: set[int]) -> str | None:
|
|
818
|
+
ids = sorted(keep_ids)
|
|
819
|
+
if not ids:
|
|
820
|
+
return None
|
|
821
|
+
cur = self._conn.execute(
|
|
822
|
+
f"SELECT MIN(created_at) FROM runs WHERE id IN ({','.join('?' * len(ids))})", ids)
|
|
823
|
+
return cur.fetchone()[0]
|
|
824
|
+
|
|
825
|
+
def prune_claims(self, keep_ids: set[int]) -> int:
|
|
826
|
+
"""Drop claims older than the oldest run a prune keeps.
|
|
827
|
+
|
|
828
|
+
Retention is one decision: a claim taken before the surviving history
|
|
829
|
+
names a function nothing left will score, so no verify can ever close it
|
|
830
|
+
and it would hide that function from every future queue.
|
|
831
|
+
"""
|
|
832
|
+
floor = self._oldest_kept_at(keep_ids)
|
|
833
|
+
if floor is None:
|
|
834
|
+
return 0
|
|
835
|
+
with self._conn:
|
|
836
|
+
cur = self._conn.execute("DELETE FROM attempts WHERE created_at < ?", (floor,))
|
|
837
|
+
return cur.rowcount
|
|
838
|
+
|
|
839
|
+
def latest_run(self, *, commit: str) -> int | None:
|
|
840
|
+
cur = self._conn.execute("SELECT MAX(id) FROM runs WHERE commit_sha = ?", (commit,))
|
|
841
|
+
(rid,) = cur.fetchone()
|
|
842
|
+
return rid
|
|
843
|
+
|
|
844
|
+
|
|
845
|
+
def read_overrides_all(self) -> list[tuple]:
|
|
846
|
+
"""The full audit trail: (run_id, path, long_name, crap, reason, created_at, commit)."""
|
|
847
|
+
cur = self._conn.execute(
|
|
848
|
+
"""SELECT o.run_id, o.path, o.long_name, o.crap, o.reason, r.created_at, r.commit_sha
|
|
849
|
+
FROM overrides o JOIN runs r ON r.id = o.run_id ORDER BY o.run_id, o.path""")
|
|
850
|
+
return list(cur.fetchall())
|
|
851
|
+
|
|
852
|
+
def find_functions(self, path: str, name_fragment: str) -> list[str]:
|
|
853
|
+
"""Distinct long_names in a path containing the fragment, across all runs.
|
|
854
|
+
|
|
855
|
+
Joined to functions rather than read off identities alone: a prune drops
|
|
856
|
+
the rows of a run and leaves its identities behind, and a name no
|
|
857
|
+
surviving run scored is a name `brief` cannot resolve.
|
|
858
|
+
|
|
859
|
+
`(anonymous)#N` is resolved by position instead: an anonymous function
|
|
860
|
+
carries no text for a LIKE to match, and the fragment would otherwise
|
|
861
|
+
hunt for a `#` no long_name has.
|
|
862
|
+
"""
|
|
863
|
+
ordinal = handle_ordinal(name_fragment)
|
|
864
|
+
if ordinal is not None:
|
|
865
|
+
return self._nth_anonymous(path, ordinal)
|
|
866
|
+
cur = self._conn.execute(
|
|
867
|
+
f"SELECT DISTINCT i.long_name {_BY_PATH} "
|
|
868
|
+
"WHERE i.path = ? AND i.long_name LIKE ? ORDER BY i.long_name",
|
|
869
|
+
(path, f"%{name_fragment}%"))
|
|
870
|
+
return [n for (n,) in cur]
|
|
871
|
+
|
|
872
|
+
def _nth_anonymous(self, path: str, ordinal: int) -> list[str]:
|
|
873
|
+
"""The path's Nth anonymous function, or nothing when there is no Nth.
|
|
874
|
+
|
|
875
|
+
Read off the newest run that scored the path, because the ordinal names
|
|
876
|
+
a position in the file as it stands now; an older run held other
|
|
877
|
+
positions. Nothing found is a list, not an error: the caller already
|
|
878
|
+
reports a name that matched no function.
|
|
879
|
+
"""
|
|
880
|
+
names = self._anonymous_names(path)
|
|
881
|
+
return names[ordinal - 1:ordinal] if ordinal >= 1 else []
|
|
882
|
+
|
|
883
|
+
def _anonymous_names(self, path: str) -> list[str]:
|
|
884
|
+
"""The long_names of the path's anonymous functions, in file order."""
|
|
885
|
+
run_id = self._newest_run_for(path)
|
|
886
|
+
if run_id is None:
|
|
887
|
+
return []
|
|
888
|
+
cur = self._conn.execute(
|
|
889
|
+
f"SELECT i.long_name {_BY_PATH} WHERE i.path = ? AND f.run_id = ? "
|
|
890
|
+
"ORDER BY f.start", (path, run_id))
|
|
891
|
+
return [n for (n,) in cur if not bare_name(n)]
|
|
892
|
+
|
|
893
|
+
def _newest_run_for(self, path: str) -> int | None:
|
|
894
|
+
"""The last run that holds a row for this path. Hook runs carry no rows,
|
|
895
|
+
so they never win it."""
|
|
896
|
+
cur = self._conn.execute(f"SELECT MAX(f.run_id) {_BY_PATH} WHERE i.path = ?",
|
|
897
|
+
(path,))
|
|
898
|
+
return cur.fetchone()[0]
|
|
899
|
+
|
|
900
|
+
def function_history(self, path: str, long_name: str) -> list[dict]:
|
|
901
|
+
"""One row per run this function appears in: the trajectory behind a verdict.
|
|
902
|
+
|
|
903
|
+
The path seeks identities once; (identity_id, run_id) then hands back
|
|
904
|
+
every run that scored it, already in run order.
|
|
905
|
+
"""
|
|
906
|
+
cur = self._conn.execute(
|
|
907
|
+
f"""SELECT f.run_id, r.commit_sha, r.kind, r.created_at, f.ccn, f.cov, f.flag, f.crap
|
|
908
|
+
{_BY_PATH} JOIN runs r ON r.id = f.run_id
|
|
909
|
+
WHERE i.path = ? AND i.long_name = ? ORDER BY f.run_id""",
|
|
910
|
+
(path, long_name))
|
|
911
|
+
flags = self._codes["flags"].names
|
|
912
|
+
return [{"run_id": rid, "commit": sha, "kind": kind, "created_at": ts,
|
|
913
|
+
"ccn": ccn, "cov": cov, "flag": _name(flags, flag), "crap": crap}
|
|
914
|
+
for rid, sha, kind, ts, ccn, cov, flag, crap in cur]
|
|
915
|
+
|
|
916
|
+
def override_run_ids(self) -> set[int]:
|
|
917
|
+
"""Runs an override record names. Deleting one deletes an audit row."""
|
|
918
|
+
return {rid for (rid,) in self._conn.execute("SELECT DISTINCT run_id FROM overrides")}
|
|
919
|
+
|
|
920
|
+
def _doomed_ids(self, keep_ids: set[int]) -> list[tuple]:
|
|
921
|
+
cur = self._conn.execute("SELECT id FROM runs ORDER BY id")
|
|
922
|
+
return [(rid,) for (rid,) in cur if rid not in keep_ids]
|
|
923
|
+
|
|
924
|
+
def prune_runs(self, keep_ids: set[int]) -> int:
|
|
925
|
+
"""Delete every run outside keep_ids, rows and metadata together.
|
|
926
|
+
|
|
927
|
+
Whole runs, never rows within a run: a run row that outlives its
|
|
928
|
+
functions reads as a real run that scored zero, which is how a prune
|
|
929
|
+
turns a silent digest into a false alarm and a trend into fiction.
|
|
930
|
+
"""
|
|
931
|
+
doomed = self._doomed_ids(keep_ids)
|
|
932
|
+
with self._conn:
|
|
933
|
+
self._conn.executemany("DELETE FROM functions WHERE run_id = ?", doomed)
|
|
934
|
+
self._conn.executemany("DELETE FROM runs WHERE id = ?", doomed)
|
|
935
|
+
return len(doomed)
|
|
936
|
+
|
|
937
|
+
def vacuum(self) -> None:
|
|
938
|
+
"""Hand the freed pages back to the OS. A DELETE alone frees none of
|
|
939
|
+
them: it moves the pages to the freelist and the file never shrinks."""
|
|
940
|
+
self._conn.commit() # VACUUM cannot run inside a transaction
|
|
941
|
+
self._conn.execute("VACUUM")
|
|
942
|
+
|
|
943
|
+
|
|
944
|
+
def default_baseline(store: SnapshotStore) -> dict | None:
|
|
945
|
+
"""The run a verify compares against: the newest TRUSTED scored run.
|
|
946
|
+
|
|
947
|
+
Trusted = a coverage run, or a verify run whose verdict passed. A failed
|
|
948
|
+
verify must never become the next baseline (rerunning verify on a broken
|
|
949
|
+
tree would launder its own failures), and hook-override anchor runs carry
|
|
950
|
+
no scored rows at all.
|
|
951
|
+
"""
|
|
952
|
+
eligible = trusted_runs(store)
|
|
953
|
+
return eligible[-1] if eligible else None
|
|
954
|
+
|
|
955
|
+
|
|
956
|
+
class BaselinePick(NamedTuple):
|
|
957
|
+
"""What `verify` measures against by default, and what the taint rule refused.
|
|
958
|
+
|
|
959
|
+
`skipped` and `blocker` are both None on the ordinary path. When they are
|
|
960
|
+
not, they are the message: the newest trusted run the rule passed over, and
|
|
961
|
+
the failed verify it passed it over for.
|
|
962
|
+
"""
|
|
963
|
+
run: dict | None
|
|
964
|
+
skipped: dict | None
|
|
965
|
+
blocker: dict | None
|
|
966
|
+
|
|
967
|
+
|
|
968
|
+
def _verdict_verifies(runs: list[dict], through_id: int) -> list[dict]:
|
|
969
|
+
"""Verify runs at or below `through_id` that actually recorded a verdict.
|
|
970
|
+
|
|
971
|
+
A crashed verify leaves verdict_ok NULL. It found nothing, so it may not
|
|
972
|
+
block a baseline, and it proved nothing, so it may not clear one either.
|
|
973
|
+
"""
|
|
974
|
+
return [r for r in runs if r["kind"] == "verify"
|
|
975
|
+
and r["verdict_ok"] is not None and r["id"] <= through_id]
|
|
976
|
+
|
|
977
|
+
|
|
978
|
+
def _blocking_verify(runs: list[dict], candidate_id: int) -> dict | None:
|
|
979
|
+
"""The failed verify standing in front of `candidate_id`, if one does.
|
|
980
|
+
|
|
981
|
+
A failed verify recorded findings against a tree. Any later run that becomes
|
|
982
|
+
the baseline moves the comparison point past them: the functions it flagged
|
|
983
|
+
stop being touched, and no verify ever looks at them again. Only a PASSING
|
|
984
|
+
verify clears it, because passing is the proof the findings were answered.
|
|
985
|
+
"""
|
|
986
|
+
blocker = None
|
|
987
|
+
for r in _verdict_verifies(runs, candidate_id):
|
|
988
|
+
blocker = None if r["verdict_ok"] else r
|
|
989
|
+
return blocker
|
|
990
|
+
|
|
991
|
+
|
|
992
|
+
def _is_refused(candidate: tuple) -> bool:
|
|
993
|
+
return candidate[1] is not None
|
|
994
|
+
|
|
995
|
+
|
|
996
|
+
def _baseline_candidates(runs: list[dict]) -> list[tuple]:
|
|
997
|
+
"""Every trusted run newest first, each paired with the verify blocking it."""
|
|
998
|
+
return [(r, _blocking_verify(runs, r["id"]))
|
|
999
|
+
for r in reversed([r for r in runs if is_trusted(r)])]
|
|
1000
|
+
|
|
1001
|
+
|
|
1002
|
+
def pick_baseline(runs: list[dict]) -> BaselinePick:
|
|
1003
|
+
"""The newest trusted run no unanswered failed verify stands in front of.
|
|
1004
|
+
|
|
1005
|
+
Run id is the order and the walk is newest first, so everything above the
|
|
1006
|
+
first clean candidate was refused. `verify --baseline ID` skips this
|
|
1007
|
+
entirely, which is the deliberate, auditable way to accept a newer run.
|
|
1008
|
+
"""
|
|
1009
|
+
newest_first = _baseline_candidates(runs)
|
|
1010
|
+
refused = list(takewhile(_is_refused, newest_first))
|
|
1011
|
+
kept = newest_first[len(refused):]
|
|
1012
|
+
skipped, blocker = refused[0] if refused else (None, None)
|
|
1013
|
+
return BaselinePick(kept[0][0] if kept else None, skipped, blocker)
|
|
1014
|
+
|
|
1015
|
+
|
|
1016
|
+
def is_trusted(r: dict) -> bool:
|
|
1017
|
+
"""Trusted = a coverage run, or a verify run whose verdict passed."""
|
|
1018
|
+
if not r["lanes"] or r["kind"] == "hook":
|
|
1019
|
+
return False
|
|
1020
|
+
if r["kind"] in ("coverage", "legacy", None):
|
|
1021
|
+
return True
|
|
1022
|
+
return r["kind"] == "verify" and r["verdict_ok"] is True
|
|
1023
|
+
|
|
1024
|
+
|
|
1025
|
+
def trusted_runs(store: SnapshotStore) -> list[dict]:
|
|
1026
|
+
return [r for r in store.list_runs() if is_trusted(r)]
|
|
1027
|
+
|
|
1028
|
+
|
|
1029
|
+
def _digest_pair_ids(trusted: list[dict]) -> set[int]:
|
|
1030
|
+
"""Both halves of the pair `crapkit digest` compares.
|
|
1031
|
+
|
|
1032
|
+
Losing either half is the loudest way a prune can go wrong: the digest
|
|
1033
|
+
would read the surviving run as a codebase that appeared from nothing and
|
|
1034
|
+
alert every over-target function in the repo as new.
|
|
1035
|
+
"""
|
|
1036
|
+
from .digest import latest_comparable_pair
|
|
1037
|
+
|
|
1038
|
+
pair = latest_comparable_pair(trusted)
|
|
1039
|
+
return {r["id"] for r in pair} if pair else set()
|
|
1040
|
+
|
|
1041
|
+
|
|
1042
|
+
def _passing_verify_ids(runs: list[dict]) -> set[int]:
|
|
1043
|
+
"""Every run `verify --baseline ID` can still legitimately name."""
|
|
1044
|
+
return {r["id"] for r in runs if r["kind"] == "verify" and r["verdict_ok"] is True}
|
|
1045
|
+
|
|
1046
|
+
|
|
1047
|
+
def _newest_non_hook_id(runs: list[dict]) -> set[int]:
|
|
1048
|
+
"""worklist and duplication read the newest non-hook run, trusted or not."""
|
|
1049
|
+
ids = [r["id"] for r in runs if r["kind"] != "hook"]
|
|
1050
|
+
return {ids[-1]} if ids else set()
|
|
1051
|
+
|
|
1052
|
+
|
|
1053
|
+
def prune_keep_set(runs: list[dict], override_run_ids, *, keep: int) -> set[int]:
|
|
1054
|
+
"""The runs a prune may never delete.
|
|
1055
|
+
|
|
1056
|
+
Retention counts trusted runs, but recency alone is not the contract: a
|
|
1057
|
+
prune that keeps N and nothing else re-arms the digest, drops a baseline
|
|
1058
|
+
someone can still name, and orphans an override record whose audit trail
|
|
1059
|
+
joins through the run row.
|
|
1060
|
+
"""
|
|
1061
|
+
if keep < 1:
|
|
1062
|
+
raise ValueError(f"keep must be >= 1, got {keep}")
|
|
1063
|
+
trusted = [r for r in runs if is_trusted(r)]
|
|
1064
|
+
return ({r["id"] for r in trusted[-keep:]}
|
|
1065
|
+
| _digest_pair_ids(trusted) | _passing_verify_ids(runs)
|
|
1066
|
+
| _newest_non_hook_id(runs) | set(override_run_ids))
|