crapkit 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. crapkit/__init__.py +2 -0
  2. crapkit/__main__.py +5 -0
  3. crapkit/_pygdefer.py +86 -0
  4. crapkit/analyze.py +375 -0
  5. crapkit/cache.py +58 -0
  6. crapkit/churn.py +113 -0
  7. crapkit/churn_cache.py +108 -0
  8. crapkit/churn_log.py +286 -0
  9. crapkit/cli/__init__.py +316 -0
  10. crapkit/cli/_shared.py +130 -0
  11. crapkit/cli/admin.py +650 -0
  12. crapkit/cli/analyses.py +144 -0
  13. crapkit/cli/parser.py +384 -0
  14. crapkit/cli/queue.py +926 -0
  15. crapkit/cli/ratchet_cmds.py +172 -0
  16. crapkit/cli/reports.py +459 -0
  17. crapkit/cli/scoring.py +500 -0
  18. crapkit/cli/verifying.py +580 -0
  19. crapkit/config.py +289 -0
  20. crapkit/coupling.py +89 -0
  21. crapkit/coverage_istanbul.py +225 -0
  22. crapkit/coverage_py.py +87 -0
  23. crapkit/covstream.py +320 -0
  24. crapkit/diffparse.py +98 -0
  25. crapkit/digest.py +191 -0
  26. crapkit/discover.py +365 -0
  27. crapkit/doctor.py +308 -0
  28. crapkit/dup.py +179 -0
  29. crapkit/errors.py +18 -0
  30. crapkit/gitio.py +504 -0
  31. crapkit/hook.py +167 -0
  32. crapkit/junitparse.py +87 -0
  33. crapkit/lanes.py +373 -0
  34. crapkit/lizardcognitive.py +238 -0
  35. crapkit/mcp_server.py +167 -0
  36. crapkit/merge.py +77 -0
  37. crapkit/mutate.py +96 -0
  38. crapkit/mutate_pool.py +152 -0
  39. crapkit/override.py +94 -0
  40. crapkit/packet.py +343 -0
  41. crapkit/ratchet.py +236 -0
  42. crapkit/ratchet_report.py +135 -0
  43. crapkit/sarif.py +82 -0
  44. crapkit/sarifio.py +49 -0
  45. crapkit/scaffold.py +361 -0
  46. crapkit/score.py +255 -0
  47. crapkit/snapshot.py +51 -0
  48. crapkit/store.py +1066 -0
  49. crapkit/uncovered.py +131 -0
  50. crapkit/universe.py +157 -0
  51. crapkit/verify.py +194 -0
  52. crapkit/watch.py +112 -0
  53. crapkit/worklist.py +290 -0
  54. crapkit-0.2.0.dist-info/METADATA +802 -0
  55. crapkit-0.2.0.dist-info/RECORD +59 -0
  56. crapkit-0.2.0.dist-info/WHEEL +5 -0
  57. crapkit-0.2.0.dist-info/entry_points.txt +2 -0
  58. crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
  59. crapkit-0.2.0.dist-info/top_level.txt +1 -0
crapkit/store.py ADDED
@@ -0,0 +1,1066 @@
1
+ """Append-only SQLite snapshot store.
2
+
3
+ Run metadata (commit, tool versions, lane provenance) lives on the run row;
4
+ scored rows are pure data keyed by run. Nothing here is ever updated in
5
+ place — a rebuild is a new run. Coverage columns are NULL on inventory-only
6
+ runs and populated on scored runs.
7
+
8
+ A function's identity — scope, path, long_name — lives once, in `identities`,
9
+ and every run's rows point at it by id. Storing the three strings on the row
10
+ rewrote a hundred thousand identities on every run; the flagship consumer's
11
+ store reached 246 MB that way. The reads join them back, so nothing above this
12
+ module can tell: read_rows and read_scored return the same values in the same
13
+ order, and identity ids never reach a sort key.
14
+
15
+ `runs prune` is the one exception to append-only, and it deletes whole runs
16
+ rather than editing any row: see prune_keep_set for what it may never take.
17
+ """
18
+ from __future__ import annotations
19
+
20
+ import json
21
+ import sqlite3
22
+ import sys
23
+ import zlib
24
+ from itertools import takewhile
25
+ from pathlib import Path
26
+ from typing import NamedTuple
27
+
28
+ from .packet import bare_name, handle_ordinal
29
+ from .snapshot import InventoryRow
30
+
31
+ # {table} so the migration can build the same shape under a temp name and swap
32
+ # it in last: the live table is never dropped until its replacement is filled.
33
+ #
34
+ # flag and remedy are INTEGER codes into `flags` and `remedies`. The two columns
35
+ # held 14.7 MB of repeated short strings on the flagship consumer's store; the
36
+ # codes never leave this module, so every read still hands back "measured" and
37
+ # "add-tests".
38
+ _FUNCTIONS_DDL = """CREATE TABLE IF NOT EXISTS {table} (
39
+ run_id INTEGER NOT NULL REFERENCES runs(id),
40
+ identity_id INTEGER NOT NULL REFERENCES identities(id),
41
+ start INTEGER NOT NULL, end INTEGER NOT NULL,
42
+ ccn_std INTEGER NOT NULL, ccn_mod INTEGER NOT NULL, ccn INTEGER NOT NULL,
43
+ nloc INTEGER NOT NULL, params INTEGER NOT NULL, nesting INTEGER NOT NULL,
44
+ cov REAL, flag INTEGER, crap REAL, remedy INTEGER,
45
+ cognitive INTEGER NOT NULL DEFAULT 0
46
+ )"""
47
+
48
+ # The UNIQUE leads with the path, and that ordering IS the index the path-scoped
49
+ # reads seek: brief, explain and function_span all ask (path, long_name). A key
50
+ # led by scope answers none of them, which is why it needed a second index on
51
+ # (path, long_name) carried beside it.
52
+ _IDENTITY_KEY = ("path", "long_name", "scope")
53
+ _IDENTITIES_DDL = """CREATE TABLE IF NOT EXISTS {table} (
54
+ id INTEGER PRIMARY KEY,
55
+ scope TEXT NOT NULL, path TEXT NOT NULL, long_name TEXT NOT NULL,
56
+ UNIQUE(path, long_name, scope)
57
+ )"""
58
+
59
+ _CODE_DDL = """CREATE TABLE IF NOT EXISTS {table} (
60
+ id INTEGER PRIMARY KEY, name TEXT NOT NULL UNIQUE
61
+ )"""
62
+
63
+ # crapkit's own verdict vocabulary, at FIXED codes: insertion order would let two
64
+ # stores that met the same names in a different order hold different integers,
65
+ # and a store is a file people copy between machines. A name from outside this
66
+ # list is still stored, at a code minted after these.
67
+ _CODE_SEEDS = {"flags": ("measured", "untested", "no-lane", "cc-only"),
68
+ "remedies": ("ok", "add-tests", "decompose")}
69
+
70
+ _SCHEMA = f"""
71
+ CREATE TABLE IF NOT EXISTS runs (
72
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
73
+ commit_sha TEXT NOT NULL,
74
+ tool_versions TEXT NOT NULL,
75
+ lanes BLOB NOT NULL DEFAULT '{{}}',
76
+ kind TEXT NOT NULL DEFAULT 'coverage',
77
+ verdict_ok INTEGER,
78
+ findings INTEGER NOT NULL DEFAULT 0,
79
+ created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ', 'now'))
80
+ );
81
+ {_IDENTITIES_DDL.format(table="identities")};
82
+ {_CODE_DDL.format(table="flags")};
83
+ {_CODE_DDL.format(table="remedies")};
84
+ {_FUNCTIONS_DDL.format(table="functions")};
85
+ CREATE TABLE IF NOT EXISTS overrides (
86
+ run_id INTEGER NOT NULL REFERENCES runs(id),
87
+ path TEXT NOT NULL, long_name TEXT NOT NULL,
88
+ crap REAL NOT NULL, reason TEXT NOT NULL,
89
+ created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ', 'now'))
90
+ );
91
+ CREATE TABLE IF NOT EXISTS attempts (
92
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
93
+ path TEXT NOT NULL, long_name TEXT NOT NULL,
94
+ commit_sha TEXT NOT NULL,
95
+ created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ', 'now')),
96
+ closed_at TEXT,
97
+ handle TEXT
98
+ );
99
+ """
100
+
101
+ # Indexes run after the migration, never with it: on a store still in the old
102
+ # shape none of these columns exists yet.
103
+ #
104
+ # Two per-row indexes on functions, not three. A run-keyed seek and an
105
+ # identity-keyed seek are the only two shapes any read asks for; a third index
106
+ # keyed (run_id, identity_id) answered neither of them better and cost 10.8 MB
107
+ # and a fifth of the insert time on the flagship consumer's store.
108
+ _INDEXES = """
109
+ CREATE INDEX IF NOT EXISTS idx_functions_run ON functions(run_id);
110
+ CREATE INDEX IF NOT EXISTS idx_functions_identity ON functions(identity_id, run_id);
111
+ CREATE INDEX IF NOT EXISTS idx_attempts_open ON attempts(closed_at);
112
+ """
113
+
114
+ # Indexes an earlier shape carried that nothing reads now. idx_identities_path
115
+ # is what the reordered UNIQUE replaced; both are dropped on open.
116
+ _DEAD_INDEXES = ("idx_functions_run_path", "idx_identities_path")
117
+
118
+ _JOINED = "FROM functions f JOIN identities i ON i.id = f.identity_id"
119
+ # Path-scoped reads, identities first. CROSS JOIN is SQLite's documented way to
120
+ # pin the outer table, and pinning it is the whole difference: a path names a
121
+ # handful of identities, a run names a hundred thousand rows, and with no
122
+ # table statistics the planner picks the run and scans it.
123
+ _BY_PATH = "FROM identities i CROSS JOIN functions f ON f.identity_id = i.id"
124
+ _ID_COLS = "i.scope, i.path, i.long_name"
125
+ _METRIC_COLS = "f.start, f.end, f.ccn_std, f.ccn_mod, f.ccn, f.nloc, f.params, f.nesting"
126
+ _INV_COLS = f"{_ID_COLS}, {_METRIC_COLS}, f.cognitive"
127
+ _ALL_COLS = f"{_ID_COLS}, {_METRIC_COLS}, f.cov, f.flag, f.crap, f.remedy, f.cognitive"
128
+ _CRAP_COLS = f"{_ID_COLS}, f.crap"
129
+ # what write_run binds per row: everything but the three identity strings
130
+ _WRITE_COLS = ("start, end, ccn_std, ccn_mod, ccn, nloc, params, nesting, "
131
+ "cov, flag, crap, remedy, cognitive")
132
+ _N_COLS = _WRITE_COLS.count(",") + 3 # every _WRITE_COLS column plus run_id and identity_id
133
+ # the identity strings, never the ids: a row's place in an export may not depend
134
+ # on when its identity was first seen
135
+ _ROW_ORDER = "ORDER BY i.scope, i.path, f.start, f.end, i.long_name"
136
+
137
+ def _selected(**substitutions: str) -> str:
138
+ """_WRITE_COLS as a SELECT list off alias f, with named columns replaced.
139
+
140
+ The migrations read the live table column for column; the two verdict
141
+ columns arrive from their lookup tables instead.
142
+ """
143
+ return ", ".join(substitutions.get(col, f"f.{col}")
144
+ for col in _WRITE_COLS.replace(" ", "").split(","))
145
+
146
+
147
+ _CODED_COLS = _selected(flag="fl.id", remedy="rm.id")
148
+ _CODE_JOIN = "LEFT JOIN flags fl ON fl.name = f.flag LEFT JOIN remedies rm ON rm.name = f.remedy"
149
+
150
+ # Names this crapkit does not know still get a code rather than a NULL: the
151
+ # LEFT JOIN below would silently drop a verdict a newer scorer invented.
152
+ _HARVEST = tuple(
153
+ f"INSERT OR IGNORE INTO {table} (name) SELECT DISTINCT {column} FROM functions "
154
+ f"WHERE {column} IS NOT NULL AND typeof({column}) = 'text'"
155
+ for table, column in (("flags", "flag"), ("remedies", "remedy")))
156
+
157
+ # The old-shape rewrite, in order, run as one transaction. The live table is
158
+ # read until the second-to-last statement and dropped only once its replacement
159
+ # is full, so an interrupt anywhere rolls back to a database the old code reads.
160
+ # It lands rows in the CURRENT shape, codes and all, so a pre-identity store
161
+ # needs one rewrite rather than two.
162
+ _IDENTITY_MIGRATION = (
163
+ # a killed process leaves no temp table behind — SQLite rolls the whole
164
+ # transaction back — but a retry must not trip over one either way
165
+ "DROP TABLE IF EXISTS functions_mig",
166
+ "DROP TABLE IF EXISTS identities", # the empty one _SCHEMA just created
167
+ _IDENTITIES_DDL.format(table="identities"),
168
+ "INSERT INTO identities (scope, path, long_name) "
169
+ "SELECT DISTINCT scope, path, long_name FROM functions ORDER BY scope, path, long_name",
170
+ *_HARVEST,
171
+ _FUNCTIONS_DDL.format(table="functions_mig"),
172
+ f"INSERT INTO functions_mig (run_id, identity_id, {_WRITE_COLS}) "
173
+ f"SELECT f.run_id, i.id, {_CODED_COLS} FROM functions f JOIN identities i "
174
+ "ON i.scope = f.scope AND i.path = f.path AND i.long_name = f.long_name "
175
+ f"{_CODE_JOIN}",
176
+ "DROP TABLE functions",
177
+ "ALTER TABLE functions_mig RENAME TO functions",
178
+ )
179
+
180
+ # Re-key the identity table. Nothing moves but the UNIQUE, so the ids on every
181
+ # functions row keep pointing at the same identity.
182
+ _REKEY_IDENTITIES = (
183
+ "DROP TABLE IF EXISTS identities_mig",
184
+ _IDENTITIES_DDL.format(table="identities_mig"),
185
+ "INSERT INTO identities_mig (id, scope, path, long_name) "
186
+ "SELECT id, scope, path, long_name FROM identities",
187
+ "DROP TABLE identities",
188
+ "ALTER TABLE identities_mig RENAME TO identities",
189
+ )
190
+
191
+ # Verdict strings to codes, same build-and-swap discipline.
192
+ _CODE_MIGRATION = (
193
+ "DROP TABLE IF EXISTS functions_mig",
194
+ *_HARVEST,
195
+ _FUNCTIONS_DDL.format(table="functions_mig"),
196
+ f"INSERT INTO functions_mig (run_id, identity_id, {_WRITE_COLS}) "
197
+ f"SELECT f.run_id, f.identity_id, {_CODED_COLS} FROM functions f {_CODE_JOIN}",
198
+ "DROP TABLE functions",
199
+ "ALTER TABLE functions_mig RENAME TO functions",
200
+ )
201
+
202
+
203
+ class CrapRow(NamedTuple):
204
+ """A scored function reduced to what a comparison between two runs needs.
205
+
206
+ These four fields are the whole of what build_digest reads: the key it pairs
207
+ functions on, the number it compares, and the scope whose ceiling decides
208
+ over-target. The other twelve columns of a ScoredRow are 140,000 rows of
209
+ dead weight per run, twice per digest.
210
+ """
211
+ scope: str
212
+ path: str
213
+ long_name: str
214
+ crap: float
215
+
216
+
217
+ class _Codes(NamedTuple):
218
+ """One lookup table, both ways: names on the way in, back out on the way out."""
219
+ ids: dict
220
+ names: dict
221
+
222
+
223
+ def _read_codes(conn, table: str) -> _Codes:
224
+ rows = conn.execute(f"SELECT id, name FROM {table}").fetchall()
225
+ return _Codes({name: code for code, name in rows}, dict(rows))
226
+
227
+
228
+ def _code(ids: dict, name):
229
+ """The stored code for a verdict string. None stays None: an inventory row
230
+ has no verdict, and a store written before coverage existed holds NULLs."""
231
+ return None if name is None else ids[name]
232
+
233
+
234
+ def _name(names: dict, code):
235
+ """The verdict string a stored code stands for."""
236
+ return None if code is None else names[code]
237
+
238
+
239
+ def _deflate(text: str) -> bytes:
240
+ """A run's lane record, compressed. It is by far the largest thing on the run
241
+ row — a failing lane records every failure by name — and JSON of that shape
242
+ goes to a fifth of its bytes."""
243
+ return zlib.compress(text.encode("utf-8"), 6)
244
+
245
+
246
+ def _inflate(stored) -> str:
247
+ """The lane record back as JSON text. A row written before the column was
248
+ deflated holds the text itself, so both storage classes read the same."""
249
+ return zlib.decompress(stored).decode("utf-8") if isinstance(stored, bytes) else stored
250
+
251
+
252
+ def _verdict_names(rows: list) -> tuple[set, set]:
253
+ """The flag and remedy strings this batch stores. Inventory rows carry
254
+ neither, so they name nothing."""
255
+ scored = [row for row in rows if len(row) == 16]
256
+ return ({row[12] for row in scored} - {None}, {row[14] for row in scored} - {None})
257
+
258
+
259
+ def _writable(row, flags: dict, remedies: dict):
260
+ """A row's metric columns in _WRITE_COLS order, identity dropped, verdict coded.
261
+
262
+ Scored rows already carry the four coverage columns; inventory rows carry
263
+ cognitive last, so the four unscored columns slot in before it.
264
+ """
265
+ if len(row) == 16:
266
+ return (*row[3:12], _code(flags, row[12]), row[13], _code(remedies, row[14]), row[15])
267
+ return (*row[3:11], None, None, None, None, row[11])
268
+
269
+
270
+ def _own_ceilings(target: int, scope_targets: dict[str, int] | None) -> list[tuple[str, int]]:
271
+ """The scopes whose ceiling is not the repo's, sorted.
272
+
273
+ A scope that declares the repo target declares nothing: its branch and the
274
+ ELSE say the same number. Dropping it is what lets a config with per-scope
275
+ blocks but no per-scope targets compare against one bound parameter over a
276
+ million rows of history.
277
+ """
278
+ return sorted((scope, ceiling) for scope, ceiling in (scope_targets or {}).items()
279
+ if ceiling != target)
280
+
281
+
282
+ class _Ceiling(NamedTuple):
283
+ """The CRAP ceiling a row is compared against, as SQL.
284
+
285
+ `per_scope` says whether the expression reads i.scope. It decides whether a
286
+ whole-run total has to join the identity table at all, which is the only
287
+ reason that query ever joined it.
288
+ """
289
+ expr: str
290
+ params: list
291
+ per_scope: bool
292
+
293
+
294
+ def _ceiling_expr(target: int, scope_targets: dict[str, int] | None) -> _Ceiling:
295
+ """The per-scope CRAP ceiling as a parameterized CASE, so an over-target count
296
+ decided in SQL is decided exactly the way digest._over_count decides it."""
297
+ own = _own_ceilings(target, scope_targets)
298
+ params: list = []
299
+ for scope, ceiling in own:
300
+ params.extend((scope, ceiling))
301
+ params.append(target)
302
+ if not own:
303
+ return _Ceiling("?", params, False) # a CASE with no WHEN is not SQL
304
+ return _Ceiling(f"CASE i.scope {' '.join('WHEN ? THEN ?' for _ in own)} ELSE ? END",
305
+ params, True)
306
+
307
+
308
+ def _scored_rows(cur, flags: dict, remedies: dict) -> list:
309
+ """A cursor over _ALL_COLS, streamed into ScoredRows with identities interned.
310
+
311
+ Rows stream off the cursor rather than through a fetchall list, and the
312
+ identity strings are interned process-wide, so a second read reuses the
313
+ first one's strings. The verdict columns need no interning: every row of a
314
+ run shares the one string its code names.
315
+ """
316
+ from .score import ScoredRow
317
+ si = sys.intern
318
+ return [ScoredRow(si(scope), si(path), si(name), a, b, c, d, e, f, g, h, cov,
319
+ _name(flags, flag), crap, _name(remedies, remedy), cog)
320
+ for scope, path, name, a, b, c, d, e, f, g, h, cov, flag, crap, remedy, cog in cur]
321
+
322
+
323
+ def _scope_clause(scopes, *, keyword: str = "IN") -> tuple[str, list]:
324
+ """A parameterized `AND i.scope IN (...)`, empty when no scope was named.
325
+
326
+ Deduplicated and sorted so the same request always builds the same SQL text,
327
+ which is what lets SQLite reuse the prepared statement across calls. NOT IN
328
+ is the same cut the other way: what a scope-blind read must leave behind.
329
+ """
330
+ if not scopes:
331
+ return "", []
332
+ names = sorted(set(scopes))
333
+ return f"AND i.scope {keyword} ({','.join('?' * len(names))})", names
334
+
335
+
336
+ class SnapshotStore:
337
+ def __init__(self, path: Path | str):
338
+ self._path = Path(path)
339
+ self._identities: dict[tuple[str, str, str], int] = {}
340
+ self._conn = sqlite3.connect(str(path))
341
+ self._conn.executescript(_SCHEMA)
342
+ self._seed_codes()
343
+ self._migrate()
344
+ # after the migration, never with it: on a store still in the old shape
345
+ # these index columns do not exist yet
346
+ self._conn.executescript(_INDEXES)
347
+ self._conn.commit()
348
+ # last, because a migration can mint codes for names it met on the way
349
+ self._codes = {table: _read_codes(self._conn, table) for table in _CODE_SEEDS}
350
+
351
+ def _seed_codes(self) -> None:
352
+ """The known verdict names, at their fixed codes. Before any migration:
353
+ the rewrites resolve every stored string through these tables."""
354
+ for table, names in _CODE_SEEDS.items():
355
+ self._conn.executemany(
356
+ f"INSERT OR IGNORE INTO {table} (id, name) VALUES (?, ?)",
357
+ list(enumerate(names, 1)))
358
+
359
+ def _migrate(self) -> None:
360
+ # _SCHEMA and _INDEXES are migration paths in themselves: every statement
361
+ # in them is IF NOT EXISTS and runs on every open, so a database written
362
+ # before a table or an index existed grows it the next time it is opened.
363
+ # What is left here is what CREATE cannot express — a column added to a
364
+ # table that already exists, and the two rewrites.
365
+ self._add_coverage_columns()
366
+ self._add_cognitive_column()
367
+ self._add_run_provenance_columns()
368
+ self._add_claim_handle_column()
369
+ self._conn.commit()
370
+ self._normalize_identities()
371
+ self._restack()
372
+
373
+ def size_bytes(self) -> int:
374
+ """The file as the OS sees it: what a prune has to move to mean anything."""
375
+ return self._path.stat().st_size if self._path.is_file() else 0
376
+
377
+ def _declared_types(self, table: str) -> dict[str, str]:
378
+ return {row[1]: row[2].upper() for row in self._conn.execute(f"PRAGMA table_info({table})")}
379
+
380
+ def _existing_columns(self, table: str) -> set[str]:
381
+ return set(self._declared_types(table))
382
+
383
+ def _add_coverage_columns(self) -> None:
384
+ have = self._existing_columns("functions")
385
+ for col, decl in (("cov", "REAL"), ("flag", "INTEGER"),
386
+ ("crap", "REAL"), ("remedy", "INTEGER")):
387
+ if col not in have:
388
+ self._conn.execute(f"ALTER TABLE functions ADD COLUMN {col} {decl}")
389
+
390
+ def _add_cognitive_column(self) -> None:
391
+ if "cognitive" not in self._existing_columns("functions"):
392
+ self._conn.execute(
393
+ "ALTER TABLE functions ADD COLUMN cognitive INTEGER NOT NULL DEFAULT 0")
394
+
395
+ def _add_run_provenance_columns(self) -> None:
396
+ run_cols = self._existing_columns("runs")
397
+ if "lanes" not in run_cols:
398
+ self._conn.execute("ALTER TABLE runs ADD COLUMN lanes TEXT NOT NULL DEFAULT '{}'")
399
+ if "kind" not in run_cols:
400
+ self._conn.execute("ALTER TABLE runs ADD COLUMN kind TEXT NOT NULL DEFAULT 'coverage'")
401
+ # rows that predate the column have unknown provenance; the DEFAULT
402
+ # must not promote them to baseline-grade 'coverage'
403
+ self._conn.execute("UPDATE runs SET kind = 'legacy'")
404
+ if "verdict_ok" not in run_cols:
405
+ self._conn.execute("ALTER TABLE runs ADD COLUMN verdict_ok INTEGER")
406
+ if "findings" not in run_cols:
407
+ self._conn.execute(
408
+ "ALTER TABLE runs ADD COLUMN findings INTEGER NOT NULL DEFAULT 0")
409
+
410
+ def _add_claim_handle_column(self) -> None:
411
+ """The name a claim was taken under, beside the long_name it was taken on.
412
+
413
+ An anonymous function's long_name is `(anonymous)` for every anonymous
414
+ function in its file, so a release by long_name closes whichever claim
415
+ sorts first. The handle is the ordinal that tells them apart, and it is
416
+ stored rather than recomputed: the whole point is that it survives the
417
+ line shifts the session's own edit makes.
418
+ """
419
+ if "handle" not in self._existing_columns("attempts"):
420
+ self._conn.execute("ALTER TABLE attempts ADD COLUMN handle TEXT")
421
+
422
+ def _normalize_identities(self) -> None:
423
+ """Move scope, path and long_name out of every functions row.
424
+
425
+ The old shape is the one that still has a scope column. The rewrite runs
426
+ as one transaction and swaps the table in last, so an interrupt leaves
427
+ the old table whole, rows and all, and the next open retries from there.
428
+ Old databases migrate silently; `runs prune` reclaims the freed pages.
429
+ """
430
+ if "scope" not in self._existing_columns("functions"):
431
+ return
432
+ self._conn.execute("BEGIN") # DDL does not open one, and this must be atomic
433
+ try:
434
+ for statement in _IDENTITY_MIGRATION:
435
+ self._conn.execute(statement)
436
+ self._conn.commit()
437
+ except Exception:
438
+ self._conn.rollback()
439
+ raise
440
+
441
+ def _identity_key(self) -> tuple:
442
+ """The columns of the identity UNIQUE, in key order."""
443
+ unique = [row[1] for row in self._conn.execute("PRAGMA index_list(identities)") if row[2]]
444
+ if not unique:
445
+ return ()
446
+ return tuple(row[2] for row in self._conn.execute(f'PRAGMA index_info("{unique[0]}")'))
447
+
448
+ def _restack_steps(self) -> tuple:
449
+ """The rewrites this store still owes, in the order they must run.
450
+
451
+ Both are version-detected off the schema itself rather than a stamp: a
452
+ store carries its own shape, and a stamp is one more thing to get wrong.
453
+ """
454
+ steps = tuple(f"DROP INDEX IF EXISTS {name}" for name in _DEAD_INDEXES)
455
+ if self._identity_key() != _IDENTITY_KEY:
456
+ steps += _REKEY_IDENTITIES
457
+ if self._declared_types("functions").get("flag") == "TEXT":
458
+ steps += _CODE_MIGRATION
459
+ return steps
460
+
461
+ def _deflate_lanes(self) -> None:
462
+ """Compress the lane records still held as text. Never conditional on the
463
+ rewrites: a store this code restacked can still be written by an older
464
+ crapkit, and the next open should take those rows too."""
465
+ rows = self._conn.execute(
466
+ "SELECT id, lanes FROM runs WHERE typeof(lanes) = 'text'").fetchall()
467
+ self._conn.executemany("UPDATE runs SET lanes = ? WHERE id = ?",
468
+ [(_deflate(text), rid) for rid, text in rows])
469
+
470
+ def _restack(self) -> None:
471
+ """Re-key the identities, code the verdicts, deflate the lane records.
472
+
473
+ Together these took the flagship consumer's store from 131.4 to 92.3 MB
474
+ with every timed read faster. One transaction, tables built under temp
475
+ names and swapped in last, so a process killed anywhere in here leaves a
476
+ database the previous code still reads and the next open retries.
477
+ """
478
+ self._conn.execute("BEGIN") # DDL does not open one, and this must be atomic
479
+ try:
480
+ for statement in self._restack_steps():
481
+ self._conn.execute(statement)
482
+ self._deflate_lanes()
483
+ self._conn.commit()
484
+ except Exception:
485
+ self._conn.rollback()
486
+ raise
487
+
488
+ def _cache_identities(self) -> None:
489
+ """Every identity in the store, keyed by triple.
490
+
491
+ One scan, and the strings are interned, so the cache shares them with
492
+ the rows a read of the same store already built.
493
+ """
494
+ si = sys.intern
495
+ self._identities = {
496
+ (si(scope), si(path), si(name)): rid
497
+ for rid, scope, path, name in self._conn.execute(
498
+ "SELECT id, scope, path, long_name FROM identities")}
499
+
500
+ def _identity_ids(self, rows: list) -> dict[tuple[str, str, str], int]:
501
+ """(scope, path, long_name) -> identities.id, covering every row.
502
+
503
+ One pass over the rows and at most two over the identity table, never a
504
+ lookup per row: a rebuild rewrites the same hundred thousand identities
505
+ every run. The reload after the insert is what makes INSERT OR IGNORE
506
+ safe — a triple a concurrent writer got in first is read back rather
507
+ than left unresolved.
508
+ """
509
+ if not self._identities:
510
+ self._cache_identities()
511
+ fresh = sorted({row[:3] for row in rows} - self._identities.keys())
512
+ if fresh:
513
+ self._conn.executemany(
514
+ "INSERT OR IGNORE INTO identities (scope, path, long_name) VALUES (?, ?, ?)",
515
+ fresh)
516
+ self._cache_identities()
517
+ return self._identities
518
+
519
+ def _code_ids(self, table: str, names: set) -> dict:
520
+ """name -> code, covering every name this batch will store.
521
+
522
+ The seeded vocabulary answers every name a crapkit run produces, so this
523
+ inserts nothing in practice. A name from somewhere else is minted a code
524
+ rather than dropped, and the reload after the insert is what makes
525
+ INSERT OR IGNORE safe against a concurrent writer.
526
+ """
527
+ fresh = sorted(names - self._codes[table].ids.keys())
528
+ if fresh:
529
+ self._conn.executemany(f"INSERT OR IGNORE INTO {table} (name) VALUES (?)",
530
+ [(name,) for name in fresh])
531
+ self._codes[table] = _read_codes(self._conn, table)
532
+ return self._codes[table].ids
533
+
534
+ def write_run(self, *, commit: str, tool_versions: dict[str, str], rows: list,
535
+ lanes: dict | None = None, kind: str = "coverage") -> int:
536
+ flag_names, remedy_names = _verdict_names(rows)
537
+ with self._conn:
538
+ cur = self._conn.execute(
539
+ "INSERT INTO runs (commit_sha, tool_versions, lanes, kind) VALUES (?, ?, ?, ?)",
540
+ (commit, json.dumps(tool_versions, sort_keys=True),
541
+ _deflate(json.dumps(lanes or {}, sort_keys=True)), kind),
542
+ )
543
+ run_id = cur.lastrowid
544
+ ids = self._identity_ids(rows)
545
+ flags = self._code_ids("flags", flag_names)
546
+ remedies = self._code_ids("remedies", remedy_names)
547
+ # a generator, not two lists: executemany consumes any iterator, and
548
+ # materializing the padded copy doubled the rows in memory alongside
549
+ # the caller's own list for the length of the insert
550
+ self._conn.executemany(
551
+ f"INSERT INTO functions (run_id, identity_id, {_WRITE_COLS}) "
552
+ f"VALUES ({','.join('?' * _N_COLS)})",
553
+ ((run_id, ids[row[:3]], *_writable(row, flags, remedies)) for row in rows),
554
+ )
555
+ return run_id
556
+
557
+ def read_rows(self, run_id: int, *, min_ccn: int = 0,
558
+ scopes: list[str] | None = None) -> list[InventoryRow]:
559
+ """Inventory rows for one run, in export order.
560
+
561
+ min_ccn is the caller's admission floor pushed into the scan: a row a
562
+ worklist could never admit costs nothing to leave in SQLite. scopes is
563
+ the same idea for a scope-shaped cut, decided on the joined identity.
564
+
565
+ Rows stream off the cursor rather than through a fetchall list, and the
566
+ identity strings are interned. A run holds thousands of distinct paths
567
+ repeated across a hundred thousand rows, and interning is process-wide,
568
+ so a second run read in the same process reuses the first one's strings.
569
+ """
570
+ clause, names = _scope_clause(scopes)
571
+ cur = self._conn.execute(
572
+ f"SELECT {_INV_COLS} {_JOINED} WHERE f.run_id = ? AND f.ccn >= ? "
573
+ f"{clause} {_ROW_ORDER}",
574
+ (run_id, min_ccn, *names),
575
+ )
576
+ si = sys.intern
577
+ return [InventoryRow(si(scope), si(path), si(name), a, b, c, d, e, f, g, h, cog)
578
+ for scope, path, name, a, b, c, d, e, f, g, h, cog in cur]
579
+
580
+ def read_scored(self, run_id: int, *, min_ccn: int = 0,
581
+ scopes: list[str] | None = None) -> list:
582
+ """Scored rows for one run, same streaming and interning as read_rows.
583
+
584
+ This is where interning pays: a digest holds two runs at once, and the
585
+ second one costs almost nothing for paths the first already interned.
586
+ """
587
+ clause, names = _scope_clause(scopes)
588
+ return self._scored(self._conn.execute(
589
+ f"SELECT {_ALL_COLS} {_JOINED} WHERE f.run_id = ? AND f.crap IS NOT NULL "
590
+ f"AND f.ccn >= ? {clause} {_ROW_ORDER}",
591
+ (run_id, min_ccn, *names),
592
+ ))
593
+
594
+ def _scored(self, cur) -> list:
595
+ return _scored_rows(cur, self._codes["flags"].names, self._codes["remedies"].names)
596
+
597
+ def read_crap(self, run_id: int) -> list[CrapRow]:
598
+ """One run's scored functions as (scope, path, long_name, crap).
599
+
600
+ What `digest` compares. It holds two whole runs at once and reads four
601
+ of the sixteen fields, so the twelve it does not read are paid for twice
602
+ over — 1,328 ms of a digest on the flagship consumer's store.
603
+ """
604
+ cur = self._conn.execute(
605
+ f"SELECT {_CRAP_COLS} {_JOINED} WHERE f.run_id = ? AND f.crap IS NOT NULL "
606
+ f"{_ROW_ORDER}", (run_id,))
607
+ si = sys.intern
608
+ return [CrapRow(si(scope), si(path), si(name), crap)
609
+ for scope, path, name, crap in cur]
610
+
611
+ def read_scored_file(self, run_id: int, path: str) -> list:
612
+ """One file's scored rows: the path seeks identities, the ids seek the run.
613
+
614
+ A brief asks about a single function; reading the whole run to find it
615
+ materializes every other row of a hundred-thousand-function repo.
616
+ """
617
+ return self._scored(self._conn.execute(
618
+ f"SELECT {_ALL_COLS} {_BY_PATH} WHERE i.path = ? AND f.run_id = ? "
619
+ f"AND f.crap IS NOT NULL {_ROW_ORDER}",
620
+ (path, run_id),
621
+ ))
622
+
623
+ def read_marks(self, run_id: int, *, min_ccn: int = 0,
624
+ scopes: list[str] | None = None) -> dict[tuple[str, str], tuple[str, str]]:
625
+ """(path, long_name) -> (flag, remedy) for every row this run scored.
626
+
627
+ Two columns, not the row: the worklist reads inventory rows and needs
628
+ only the verdict half — which rows a floor must not hide, which the
629
+ queue will never offer, which are finished — so building a ScoredRow per
630
+ function would double the cost of the command. Twins collapse to the
631
+ WORST of them, the same rule the ratchet and the verdict use, because
632
+ the ORDER BY lets the highest-CRAP twin overwrite its siblings. An
633
+ inventory-only run answers nothing: its remedy is NULL.
634
+ """
635
+ clause, names = _scope_clause(scopes)
636
+ cur = self._conn.execute(
637
+ f"SELECT i.path, i.long_name, f.flag, f.remedy {_JOINED} WHERE f.run_id = ? "
638
+ f"AND f.remedy IS NOT NULL AND f.ccn >= ? {clause} ORDER BY f.crap",
639
+ (run_id, min_ccn, *names))
640
+ flags, remedies = self._codes["flags"].names, self._codes["remedies"].names
641
+ return {(path, name): (_name(flags, flag), _name(remedies, remedy))
642
+ for path, name, flag, remedy in cur}
643
+
644
+ def count_by_path(self, run_id: int, *, flag: str,
645
+ skip_scopes=frozenset()) -> list[tuple[str, int, int]]:
646
+ """Per path in one run: (path, functions, others), path order.
647
+
648
+ `others` counts the rows whose flag is NOT the named one, which is all
649
+ doctor asks of a run: a directory is a measurement gap only when nothing
650
+ in it carries any other verdict. A run holds a hundred thousand rows and
651
+ a few thousand paths, so the grouping belongs where the rows are — and
652
+ the scopes to leave out belong in the WHERE, not in a Python filter that
653
+ reads them first.
654
+ """
655
+ clause, names = _scope_clause(skip_scopes, keyword="NOT IN")
656
+ cur = self._conn.execute(
657
+ f"SELECT i.path, COUNT(*), SUM(f.flag IS NOT ?) {_JOINED} "
658
+ f"WHERE f.run_id = ? AND f.crap IS NOT NULL {clause} GROUP BY i.path "
659
+ "ORDER BY i.path",
660
+ (_code(self._codes["flags"].ids, flag), run_id, *names))
661
+ return cur.fetchall()
662
+
663
+ def count_scored_below(self, run_id: int, min_ccn: int,
664
+ scopes: list[str] | None = None) -> int:
665
+ """The rows read_scored(min_ccn=...) skipped. An empty queue still has to
666
+ report them, and one COUNT is cheaper than reading them to count them.
667
+
668
+ Same scopes the read took, or a scoped queue reports rows it was never
669
+ going to offer.
670
+ """
671
+ clause, names = _scope_clause(scopes)
672
+ cur = self._conn.execute(
673
+ f"SELECT COUNT(*) {_JOINED} WHERE f.run_id = ? AND f.crap IS NOT NULL "
674
+ f"AND f.ccn < ? {clause}",
675
+ (run_id, min_ccn, *names))
676
+ return cur.fetchone()[0]
677
+
678
+ def run_totals(self, *, target: int,
679
+ scope_targets: dict[str, int] | None = None) -> dict[int, tuple]:
680
+ """Per-run (functions, over_target, crap_load) summed inside the scan.
681
+
682
+ trend used to build every ScoredRow of every trusted run to add up three
683
+ numbers and throw the rows away. The identity table is joined only when
684
+ the ceiling really is per-scope: a repo whose scopes all take the repo
685
+ target reads the rows and skips 1.12 M index seeks.
686
+ """
687
+ ceiling = _ceiling_expr(target, scope_targets)
688
+ source = _JOINED if ceiling.per_scope else "FROM functions f"
689
+ cur = self._conn.execute(
690
+ f"SELECT f.run_id, COUNT(*), SUM(f.crap > {ceiling.expr}), SUM(f.crap) {source} "
691
+ "WHERE f.crap IS NOT NULL GROUP BY f.run_id", ceiling.params)
692
+ return {run_id: (n, over, load) for run_id, n, over, load in cur}
693
+
694
+ def run_scope_totals(self, *, target: int,
695
+ scope_targets: dict[str, int] | None = None) -> dict[int, dict[str, tuple]]:
696
+ """run_totals cut one level finer: (functions, over_target, crap_load) per
697
+ (run, scope), summed inside the same scan rather than by reading rows.
698
+
699
+ This one groups BY the scope, so it joins the identity whatever shape the
700
+ ceiling takes.
701
+ """
702
+ ceiling = _ceiling_expr(target, scope_targets)
703
+ cur = self._conn.execute(
704
+ f"SELECT f.run_id, i.scope, COUNT(*), SUM(f.crap > {ceiling.expr}), "
705
+ f"SUM(f.crap) {_JOINED} WHERE f.crap IS NOT NULL "
706
+ "GROUP BY f.run_id, i.scope ORDER BY f.run_id, i.scope",
707
+ ceiling.params)
708
+ out: dict[int, dict[str, tuple]] = {}
709
+ for run_id, scope, n, over, load in cur:
710
+ out.setdefault(run_id, {})[scope] = (n, over, load)
711
+ return out
712
+
713
+ def function_span(self, run_id: int, path: str, long_name: str) -> tuple | None:
714
+ """One function's (start, end) in one run, off the identity path index.
715
+
716
+ Same tie-break as read_rows, which this replaced: path and long_name are
717
+ pinned by the WHERE, so export order reduces to scope, start, end.
718
+ """
719
+ cur = self._conn.execute(
720
+ f"SELECT f.start, f.end {_BY_PATH} WHERE i.path = ? AND i.long_name = ? "
721
+ "AND f.run_id = ? ORDER BY i.scope, f.start, f.end LIMIT 1",
722
+ (path, long_name, run_id))
723
+ return cur.fetchone()
724
+
725
+ def set_verdict_ok(self, run_id: int, ok: bool, *, findings: int = 0) -> None:
726
+ """Stamp a verdict on a run, with how many findings it carried.
727
+
728
+ The count is what a later refusal quotes back: a baseline that skips
729
+ this run has to say what it is protecting, and re-deriving it would mean
730
+ rerunning the lanes on a tree that has moved on.
731
+ """
732
+ with self._conn:
733
+ self._conn.execute("UPDATE runs SET verdict_ok = ?, findings = ? WHERE id = ?",
734
+ (1 if ok else 0, findings, run_id))
735
+
736
+ def list_runs(self) -> list[dict]:
737
+ cur = self._conn.execute(
738
+ "SELECT id, commit_sha, tool_versions, lanes, kind, verdict_ok, findings, "
739
+ "created_at FROM runs ORDER BY id")
740
+ return [
741
+ {"id": rid, "commit": sha, "tool_versions": json.loads(tv),
742
+ "lanes": json.loads(_inflate(lanes)), "kind": kind,
743
+ "verdict_ok": None if ok is None else bool(ok), "findings": findings,
744
+ "created_at": ts}
745
+ for rid, sha, tv, lanes, kind, ok, findings, ts in cur.fetchall()
746
+ ]
747
+
748
+ def write_overrides(self, run_id: int, rows: list[tuple[str, str, float, str]]) -> None:
749
+ with self._conn:
750
+ self._conn.executemany(
751
+ "INSERT INTO overrides (run_id, path, long_name, crap, reason) VALUES (?,?,?,?,?)",
752
+ [(run_id, *r) for r in rows],
753
+ )
754
+
755
+ def read_overrides(self, run_id: int) -> list[tuple[str, str, float, str]]:
756
+ cur = self._conn.execute(
757
+ "SELECT path, long_name, crap, reason FROM overrides WHERE run_id = ? ORDER BY path, long_name",
758
+ (run_id,))
759
+ return list(cur.fetchall())
760
+
761
+ def record_claim(self, *, path: str, long_name: str, commit: str,
762
+ handle: str | None = None) -> int:
763
+ """Take a claim on one function. Opt-in: nothing writes here unless a
764
+ session asked for it, so a store with no claims answers every query the
765
+ way it did before claims existed.
766
+
767
+ `handle` is the name the claim was handed out under. None is the honest
768
+ answer for a caller that never had one, and reads back as null.
769
+ """
770
+ with self._conn:
771
+ cur = self._conn.execute(
772
+ "INSERT INTO attempts (path, long_name, commit_sha, handle) "
773
+ "VALUES (?, ?, ?, ?)", (path, long_name, commit, handle))
774
+ return cur.lastrowid
775
+
776
+ def open_claims(self) -> list[dict]:
777
+ cur = self._conn.execute(
778
+ "SELECT id, path, long_name, commit_sha, created_at, handle FROM attempts "
779
+ "WHERE closed_at IS NULL ORDER BY id")
780
+ return [{"id": cid, "path": p, "long_name": n, "commit": sha,
781
+ "created_at": ts, "handle": handle}
782
+ for cid, p, n, sha, ts, handle in cur]
783
+
784
+ def attempts_for(self, keys) -> dict[tuple[str, str], list[dict]]:
785
+ """Every claim ever taken on each named function, oldest first.
786
+
787
+ One query for the whole batch, filtered on the indexed path and paired
788
+ up here: a packet run asks about N functions, and a query apiece is N
789
+ round trips for what one path filter already returns. Every requested
790
+ key is in the answer, so a function nobody ever claimed reads as [].
791
+ """
792
+ wanted = list(dict.fromkeys(keys))
793
+ found: dict[tuple[str, str], list[dict]] = {key: [] for key in wanted}
794
+ if not wanted:
795
+ return found
796
+ paths = sorted({path for path, _ in wanted})
797
+ cur = self._conn.execute(
798
+ "SELECT path, long_name, created_at, closed_at FROM attempts "
799
+ f"WHERE path IN ({','.join('?' * len(paths))}) ORDER BY id", paths)
800
+ for path, long_name, opened, closed in cur:
801
+ if (path, long_name) in found:
802
+ found[(path, long_name)].append({"opened": opened, "closed": closed})
803
+ return found
804
+
805
+ def close_claims(self, claim_ids) -> int:
806
+ """Stamp the named claims closed; already-closed ones are left alone, so
807
+ a verify that runs twice closes the same claim once."""
808
+ ids = sorted(claim_ids)
809
+ if not ids:
810
+ return 0
811
+ with self._conn:
812
+ cur = self._conn.execute(
813
+ "UPDATE attempts SET closed_at = strftime('%Y-%m-%dT%H:%M:%SZ', 'now') "
814
+ f"WHERE closed_at IS NULL AND id IN ({','.join('?' * len(ids))})", ids)
815
+ return cur.rowcount
816
+
817
+ def _oldest_kept_at(self, keep_ids: set[int]) -> str | None:
818
+ ids = sorted(keep_ids)
819
+ if not ids:
820
+ return None
821
+ cur = self._conn.execute(
822
+ f"SELECT MIN(created_at) FROM runs WHERE id IN ({','.join('?' * len(ids))})", ids)
823
+ return cur.fetchone()[0]
824
+
825
+ def prune_claims(self, keep_ids: set[int]) -> int:
826
+ """Drop claims older than the oldest run a prune keeps.
827
+
828
+ Retention is one decision: a claim taken before the surviving history
829
+ names a function nothing left will score, so no verify can ever close it
830
+ and it would hide that function from every future queue.
831
+ """
832
+ floor = self._oldest_kept_at(keep_ids)
833
+ if floor is None:
834
+ return 0
835
+ with self._conn:
836
+ cur = self._conn.execute("DELETE FROM attempts WHERE created_at < ?", (floor,))
837
+ return cur.rowcount
838
+
839
+ def latest_run(self, *, commit: str) -> int | None:
840
+ cur = self._conn.execute("SELECT MAX(id) FROM runs WHERE commit_sha = ?", (commit,))
841
+ (rid,) = cur.fetchone()
842
+ return rid
843
+
844
+
845
+ def read_overrides_all(self) -> list[tuple]:
846
+ """The full audit trail: (run_id, path, long_name, crap, reason, created_at, commit)."""
847
+ cur = self._conn.execute(
848
+ """SELECT o.run_id, o.path, o.long_name, o.crap, o.reason, r.created_at, r.commit_sha
849
+ FROM overrides o JOIN runs r ON r.id = o.run_id ORDER BY o.run_id, o.path""")
850
+ return list(cur.fetchall())
851
+
852
+ def find_functions(self, path: str, name_fragment: str) -> list[str]:
853
+ """Distinct long_names in a path containing the fragment, across all runs.
854
+
855
+ Joined to functions rather than read off identities alone: a prune drops
856
+ the rows of a run and leaves its identities behind, and a name no
857
+ surviving run scored is a name `brief` cannot resolve.
858
+
859
+ `(anonymous)#N` is resolved by position instead: an anonymous function
860
+ carries no text for a LIKE to match, and the fragment would otherwise
861
+ hunt for a `#` no long_name has.
862
+ """
863
+ ordinal = handle_ordinal(name_fragment)
864
+ if ordinal is not None:
865
+ return self._nth_anonymous(path, ordinal)
866
+ cur = self._conn.execute(
867
+ f"SELECT DISTINCT i.long_name {_BY_PATH} "
868
+ "WHERE i.path = ? AND i.long_name LIKE ? ORDER BY i.long_name",
869
+ (path, f"%{name_fragment}%"))
870
+ return [n for (n,) in cur]
871
+
872
+ def _nth_anonymous(self, path: str, ordinal: int) -> list[str]:
873
+ """The path's Nth anonymous function, or nothing when there is no Nth.
874
+
875
+ Read off the newest run that scored the path, because the ordinal names
876
+ a position in the file as it stands now; an older run held other
877
+ positions. Nothing found is a list, not an error: the caller already
878
+ reports a name that matched no function.
879
+ """
880
+ names = self._anonymous_names(path)
881
+ return names[ordinal - 1:ordinal] if ordinal >= 1 else []
882
+
883
+ def _anonymous_names(self, path: str) -> list[str]:
884
+ """The long_names of the path's anonymous functions, in file order."""
885
+ run_id = self._newest_run_for(path)
886
+ if run_id is None:
887
+ return []
888
+ cur = self._conn.execute(
889
+ f"SELECT i.long_name {_BY_PATH} WHERE i.path = ? AND f.run_id = ? "
890
+ "ORDER BY f.start", (path, run_id))
891
+ return [n for (n,) in cur if not bare_name(n)]
892
+
893
+ def _newest_run_for(self, path: str) -> int | None:
894
+ """The last run that holds a row for this path. Hook runs carry no rows,
895
+ so they never win it."""
896
+ cur = self._conn.execute(f"SELECT MAX(f.run_id) {_BY_PATH} WHERE i.path = ?",
897
+ (path,))
898
+ return cur.fetchone()[0]
899
+
900
+ def function_history(self, path: str, long_name: str) -> list[dict]:
901
+ """One row per run this function appears in: the trajectory behind a verdict.
902
+
903
+ The path seeks identities once; (identity_id, run_id) then hands back
904
+ every run that scored it, already in run order.
905
+ """
906
+ cur = self._conn.execute(
907
+ f"""SELECT f.run_id, r.commit_sha, r.kind, r.created_at, f.ccn, f.cov, f.flag, f.crap
908
+ {_BY_PATH} JOIN runs r ON r.id = f.run_id
909
+ WHERE i.path = ? AND i.long_name = ? ORDER BY f.run_id""",
910
+ (path, long_name))
911
+ flags = self._codes["flags"].names
912
+ return [{"run_id": rid, "commit": sha, "kind": kind, "created_at": ts,
913
+ "ccn": ccn, "cov": cov, "flag": _name(flags, flag), "crap": crap}
914
+ for rid, sha, kind, ts, ccn, cov, flag, crap in cur]
915
+
916
+ def override_run_ids(self) -> set[int]:
917
+ """Runs an override record names. Deleting one deletes an audit row."""
918
+ return {rid for (rid,) in self._conn.execute("SELECT DISTINCT run_id FROM overrides")}
919
+
920
+ def _doomed_ids(self, keep_ids: set[int]) -> list[tuple]:
921
+ cur = self._conn.execute("SELECT id FROM runs ORDER BY id")
922
+ return [(rid,) for (rid,) in cur if rid not in keep_ids]
923
+
924
+ def prune_runs(self, keep_ids: set[int]) -> int:
925
+ """Delete every run outside keep_ids, rows and metadata together.
926
+
927
+ Whole runs, never rows within a run: a run row that outlives its
928
+ functions reads as a real run that scored zero, which is how a prune
929
+ turns a silent digest into a false alarm and a trend into fiction.
930
+ """
931
+ doomed = self._doomed_ids(keep_ids)
932
+ with self._conn:
933
+ self._conn.executemany("DELETE FROM functions WHERE run_id = ?", doomed)
934
+ self._conn.executemany("DELETE FROM runs WHERE id = ?", doomed)
935
+ return len(doomed)
936
+
937
+ def vacuum(self) -> None:
938
+ """Hand the freed pages back to the OS. A DELETE alone frees none of
939
+ them: it moves the pages to the freelist and the file never shrinks."""
940
+ self._conn.commit() # VACUUM cannot run inside a transaction
941
+ self._conn.execute("VACUUM")
942
+
943
+
944
+ def default_baseline(store: SnapshotStore) -> dict | None:
945
+ """The run a verify compares against: the newest TRUSTED scored run.
946
+
947
+ Trusted = a coverage run, or a verify run whose verdict passed. A failed
948
+ verify must never become the next baseline (rerunning verify on a broken
949
+ tree would launder its own failures), and hook-override anchor runs carry
950
+ no scored rows at all.
951
+ """
952
+ eligible = trusted_runs(store)
953
+ return eligible[-1] if eligible else None
954
+
955
+
956
+ class BaselinePick(NamedTuple):
957
+ """What `verify` measures against by default, and what the taint rule refused.
958
+
959
+ `skipped` and `blocker` are both None on the ordinary path. When they are
960
+ not, they are the message: the newest trusted run the rule passed over, and
961
+ the failed verify it passed it over for.
962
+ """
963
+ run: dict | None
964
+ skipped: dict | None
965
+ blocker: dict | None
966
+
967
+
968
+ def _verdict_verifies(runs: list[dict], through_id: int) -> list[dict]:
969
+ """Verify runs at or below `through_id` that actually recorded a verdict.
970
+
971
+ A crashed verify leaves verdict_ok NULL. It found nothing, so it may not
972
+ block a baseline, and it proved nothing, so it may not clear one either.
973
+ """
974
+ return [r for r in runs if r["kind"] == "verify"
975
+ and r["verdict_ok"] is not None and r["id"] <= through_id]
976
+
977
+
978
+ def _blocking_verify(runs: list[dict], candidate_id: int) -> dict | None:
979
+ """The failed verify standing in front of `candidate_id`, if one does.
980
+
981
+ A failed verify recorded findings against a tree. Any later run that becomes
982
+ the baseline moves the comparison point past them: the functions it flagged
983
+ stop being touched, and no verify ever looks at them again. Only a PASSING
984
+ verify clears it, because passing is the proof the findings were answered.
985
+ """
986
+ blocker = None
987
+ for r in _verdict_verifies(runs, candidate_id):
988
+ blocker = None if r["verdict_ok"] else r
989
+ return blocker
990
+
991
+
992
+ def _is_refused(candidate: tuple) -> bool:
993
+ return candidate[1] is not None
994
+
995
+
996
+ def _baseline_candidates(runs: list[dict]) -> list[tuple]:
997
+ """Every trusted run newest first, each paired with the verify blocking it."""
998
+ return [(r, _blocking_verify(runs, r["id"]))
999
+ for r in reversed([r for r in runs if is_trusted(r)])]
1000
+
1001
+
1002
+ def pick_baseline(runs: list[dict]) -> BaselinePick:
1003
+ """The newest trusted run no unanswered failed verify stands in front of.
1004
+
1005
+ Run id is the order and the walk is newest first, so everything above the
1006
+ first clean candidate was refused. `verify --baseline ID` skips this
1007
+ entirely, which is the deliberate, auditable way to accept a newer run.
1008
+ """
1009
+ newest_first = _baseline_candidates(runs)
1010
+ refused = list(takewhile(_is_refused, newest_first))
1011
+ kept = newest_first[len(refused):]
1012
+ skipped, blocker = refused[0] if refused else (None, None)
1013
+ return BaselinePick(kept[0][0] if kept else None, skipped, blocker)
1014
+
1015
+
1016
+ def is_trusted(r: dict) -> bool:
1017
+ """Trusted = a coverage run, or a verify run whose verdict passed."""
1018
+ if not r["lanes"] or r["kind"] == "hook":
1019
+ return False
1020
+ if r["kind"] in ("coverage", "legacy", None):
1021
+ return True
1022
+ return r["kind"] == "verify" and r["verdict_ok"] is True
1023
+
1024
+
1025
+ def trusted_runs(store: SnapshotStore) -> list[dict]:
1026
+ return [r for r in store.list_runs() if is_trusted(r)]
1027
+
1028
+
1029
+ def _digest_pair_ids(trusted: list[dict]) -> set[int]:
1030
+ """Both halves of the pair `crapkit digest` compares.
1031
+
1032
+ Losing either half is the loudest way a prune can go wrong: the digest
1033
+ would read the surviving run as a codebase that appeared from nothing and
1034
+ alert every over-target function in the repo as new.
1035
+ """
1036
+ from .digest import latest_comparable_pair
1037
+
1038
+ pair = latest_comparable_pair(trusted)
1039
+ return {r["id"] for r in pair} if pair else set()
1040
+
1041
+
1042
+ def _passing_verify_ids(runs: list[dict]) -> set[int]:
1043
+ """Every run `verify --baseline ID` can still legitimately name."""
1044
+ return {r["id"] for r in runs if r["kind"] == "verify" and r["verdict_ok"] is True}
1045
+
1046
+
1047
+ def _newest_non_hook_id(runs: list[dict]) -> set[int]:
1048
+ """worklist and duplication read the newest non-hook run, trusted or not."""
1049
+ ids = [r["id"] for r in runs if r["kind"] != "hook"]
1050
+ return {ids[-1]} if ids else set()
1051
+
1052
+
1053
+ def prune_keep_set(runs: list[dict], override_run_ids, *, keep: int) -> set[int]:
1054
+ """The runs a prune may never delete.
1055
+
1056
+ Retention counts trusted runs, but recency alone is not the contract: a
1057
+ prune that keeps N and nothing else re-arms the digest, drops a baseline
1058
+ someone can still name, and orphans an override record whose audit trail
1059
+ joins through the run row.
1060
+ """
1061
+ if keep < 1:
1062
+ raise ValueError(f"keep must be >= 1, got {keep}")
1063
+ trusted = [r for r in runs if is_trusted(r)]
1064
+ return ({r["id"] for r in trusted[-keep:]}
1065
+ | _digest_pair_ids(trusted) | _passing_verify_ids(runs)
1066
+ | _newest_non_hook_id(runs) | set(override_run_ids))