cctally 1.91.0 → 1.92.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/CHANGELOG.md +51 -0
  2. package/README.md +4 -2
  3. package/bin/_cctally_cache.py +903 -74
  4. package/bin/_cctally_config.py +57 -0
  5. package/bin/_cctally_core.py +94 -14
  6. package/bin/_cctally_dashboard.py +217 -19
  7. package/bin/_cctally_dashboard_conversation.py +170 -20
  8. package/bin/_cctally_dashboard_envelope.py +2 -0
  9. package/bin/_cctally_db.py +481 -19
  10. package/bin/_cctally_doctor.py +18 -1
  11. package/bin/_cctally_journal.py +1156 -21
  12. package/bin/_cctally_journal_repair.py +6 -0
  13. package/bin/_cctally_parser.py +26 -0
  14. package/bin/_cctally_quota.py +171 -55
  15. package/bin/_cctally_record.py +13 -1
  16. package/bin/_cctally_rederive.py +4 -0
  17. package/bin/_cctally_statusline.py +6 -6
  18. package/bin/_cctally_store.py +1061 -40
  19. package/bin/_cctally_transcript.py +32 -2
  20. package/bin/_cctally_tui.py +54 -6
  21. package/bin/_lib_cache_report.py +8 -3
  22. package/bin/_lib_codex_conversation.py +851 -81
  23. package/bin/_lib_codex_conversation_query.py +2031 -96
  24. package/bin/_lib_codex_find_projection.py +517 -0
  25. package/bin/_lib_codex_harness_preamble.py +176 -0
  26. package/bin/_lib_codex_hooks.py +5 -3
  27. package/bin/_lib_codex_js_scan.py +254 -0
  28. package/bin/_lib_codex_landmarks.py +309 -0
  29. package/bin/_lib_codex_title_clean.py +116 -0
  30. package/bin/_lib_conversation_dispatch.py +168 -22
  31. package/bin/_lib_conversation_query.py +62 -2
  32. package/bin/_lib_conversation_watch.py +4 -2
  33. package/bin/_lib_doctor.py +64 -0
  34. package/bin/_lib_quota_alert_axes.py +31 -34
  35. package/bin/_lib_stats_damage.py +523 -0
  36. package/bin/_lib_stats_publish.py +243 -0
  37. package/bin/cctally +17 -3
  38. package/dashboard/static/assets/index-Dat-mza6.js +97 -0
  39. package/dashboard/static/assets/{index-Dwirao3Y.css → index-DnWdv8um.css} +1 -1
  40. package/dashboard/static/dashboard.html +2 -2
  41. package/package.json +8 -1
  42. package/dashboard/static/assets/index-CILAoEja.js +0 -90
@@ -23,9 +23,10 @@ in ``quota_window_snapshots`` moving at all:
23
23
  and becomes eligible when wall time passes it with no mutation to observe.
24
24
  Persisting that boundary and treating ``now >= boundary`` as dirty is what
25
25
  closes it. Unlike axes 2 and 3 this one fires on WALL CLOCK rather than on a
26
- configuration change, which is why the hook path defers it
27
- (``defer_scheduled``) instead of paying an unannounced whole-history pass on
28
- a blocking tick.
26
+ configuration change. An ownership schedule lets the hook evaluate only the
27
+ complete roots whose deadlines matured; scalar-only legacy state still
28
+ defers rather than paying an unannounced whole-history pass on a blocking
29
+ tick.
29
30
  5. **Durable lifecycle state** — the existing arming rows and terminal events,
30
31
  unchanged. Represented here only as the fingerprints axis 2 compares.
31
32
 
@@ -80,6 +81,7 @@ def alert_dirty_scope(
80
81
  gate_after: bool,
81
82
  now: dt.datetime,
82
83
  next_evaluation_at: "dt.datetime | None",
84
+ scheduled_roots: "Iterable[str] | None" = None,
83
85
  defer_scheduled: bool = False,
84
86
  ) -> AlertDirtyScope:
85
87
  """Resolve the five axes into one decision.
@@ -89,14 +91,11 @@ def alert_dirty_scope(
89
91
  observed_slot, window_minutes)``; the ROOT is element 1, which is what an
90
92
  exact-rule change is scoped to.
91
93
 
92
- ``defer_scheduled`` is the hook path's (``full_pass="defer"``). Axes 2 and 3
93
- are driven by a configuration change the user just made, so widening for
94
- them is bounded and expected; axis 4 is driven by WALL CLOCK, which makes it
95
- the one route into a whole-history pass that can land on a blocking hook
96
- tick with nothing to have predicted it. Under this flag it is recorded as
97
- ``REASON_SCHEDULED_DEFERRED`` and does NOT strengthen the scope — and the
98
- caller owes the stored boundary a carry-through, because a deferral that
99
- lets the boundary be recomputed is a silent drop.
94
+ ``scheduled_roots`` is the validated ownership retained with axis 4. When
95
+ present, a matured instant scopes to those roots even on the hook path.
96
+ ``None`` is the legacy/unavailable-ownership shape; only that shape needs
97
+ ``defer_scheduled`` to avoid an unannounced whole-history hook pass, and the
98
+ caller then owes the scalar boundary a carry-through.
100
99
  """
101
100
  reasons: list[str] = []
102
101
  if not gate_after:
@@ -132,11 +131,19 @@ def alert_dirty_scope(
132
131
  reasons.append("rule_changed")
133
132
 
134
133
  if next_evaluation_at is not None and now >= next_evaluation_at:
135
- # A future-clocked observation just became eligible. Which identity it
136
- # belongs to is not recorded — only the instant — so the honest scope is
137
- # everything, and on the hook path "everything" is precisely what may
138
- # not run.
139
- if defer_scheduled:
134
+ # Epoch 1007 records the roots owning each scheduled instant. A complete
135
+ # semantic pass over those roots is bounded enough for the hook path and
136
+ # is all axis 4 needs. ``None`` means legacy/unavailable ownership, where
137
+ # the only honest scope remains everything (and therefore deferral on a
138
+ # hook tick). An empty known set means the owning roots are not lifecycle
139
+ # eligible on this tick; the stored axis remains due for a later tick.
140
+ if scheduled_roots is not None:
141
+ due_roots = {str(root) for root in scheduled_roots if str(root)}
142
+ if due_roots:
143
+ scope = _strongest(scope, SCOPE_ROOTS)
144
+ roots |= due_roots
145
+ reasons.append("scheduled")
146
+ elif defer_scheduled:
140
147
  reasons.append(REASON_SCHEDULED_DEFERRED)
141
148
  else:
142
149
  scope = _strongest(scope, SCOPE_ALL)
@@ -154,12 +161,12 @@ def next_evaluation_boundary(
154
161
  ) -> "dt.datetime | None":
155
162
  """The earliest still-future capture the projector must come back for.
156
163
 
157
- A bounded pass only sees the dirty windows, so the STORED boundary is
164
+ This is the legacy scalar helper. A bounded pass only sees dirty windows, so
165
+ the STORED boundary is
158
166
  retained whenever it is still in the future: dropping it would forget a
159
167
  future-clocked observation sitting in a window this pass never loaded. Once
160
- wall time passes it the axis fires, the pass widens to everything, and the
161
- boundary is recomputed from complete evidence — so a retained value can only
162
- ever cost one extra pass, never a missed one.
168
+ wall time passes it the axis fires and the caller decides whether it has
169
+ enough ownership to scope the pass.
163
170
 
164
171
  ``retain_due`` keeps a boundary that is ALREADY due, which is the case where
165
172
  "recomputed from complete evidence" is a lie: a reporting-only pass never
@@ -167,20 +174,10 @@ def next_evaluation_boundary(
167
174
  the widening deliberately did not look. Either would otherwise retire the
168
175
  axis on behalf of an evaluation nobody performed.
169
176
 
170
- A due value sorts before every future candidate, so it stays until a pass
171
- that genuinely looked at everything retires it — in practice a hook tick
172
- that widened to whole-history for axis 2 or 3, since carrying alert
173
- eligibility is what separates such a pass from a reporting-only one and the
174
- hook is the only production caller that carries it.
175
-
176
- It does NOT stay "until a pass that can act on it does", and that gap is
177
- open rather than closed: on a hook-only install with a steady enabled gate,
178
- unchanged rules and a quiet ledger, no qualifying pass ever runs and the
179
- instant is retained indefinitely. The cost is bounded — the tick stays
180
- bounded and fast, and the window is re-evaluated as soon as it goes
181
- ledger-dirty again, which for a live window is continuous — so the exposure
182
- is a future-clocked capture in a window that then goes permanently quiet
183
- never qualifying a threshold. Under-alerting, never a stall or a burst.
177
+ Epoch 1007's per-root map is maintained by the projector rather than this
178
+ helper. It closes the quiet-window gap by letting a hook tick replace only
179
+ the roots it evaluated; scalar-only legacy state still uses ``retain_due``
180
+ and the conservative full/deferred path.
184
181
  """
185
182
  candidates = [value for value in capture_times if value > now]
186
183
  if stored is not None and (retain_due or stored > now):
@@ -0,0 +1,523 @@
1
+ """Pure damage-characterization kernel for a corrupt SQLite index (#496 S1 F8).
2
+
3
+ Two independent sources feed one structured description.
4
+
5
+ `parse_integrity_rows` converts the row forms `PRAGMA integrity_check` emits
6
+ into typed findings. That covers only the minority of incidents: in 54 of the
7
+ 74 retained production forensics bundles the pragma RAISED before producing any
8
+ row, so the bundle holds the plain string ``error: database disk image is
9
+ malformed`` and nothing can be derived from it.
10
+
11
+ `scan_sqlite_file` covers the rest by reading the file itself — the 100-byte
12
+ header, then the `sqlite_schema` b-tree rooted at page 1 — and probing the type
13
+ byte of every root page that schema names. It opens no SQLite connection, so it
14
+ still describes a file SQLite refuses to open.
15
+
16
+ `shape_token` normalizes either source into a short equality-comparable token
17
+ with page numbers, cell indices and rowids removed, so a recurring damage class
18
+ is detectable by comparison rather than by reading prose.
19
+
20
+ Nothing in this module raises. A rebuild must never fail because diagnostic
21
+ enrichment failed, so every failure path returns an ``unavailable`` method with
22
+ a bounded reason string.
23
+ """
24
+ from __future__ import annotations
25
+
26
+ import hashlib
27
+ import pathlib
28
+ import re
29
+
30
+ SCHEMA_VERSION = 1
31
+
32
+ #: Finding keys are always all present, so consumers never need ``.get``.
33
+ _FINDING_KEYS = ("kind", "table", "index", "column", "page", "cell", "rowid", "raw")
34
+
35
+ _SQLITE_MAGIC = b"SQLite format 3\x00"
36
+
37
+ #: Leaf/interior table (0x0d/0x05) and leaf/interior index (0x0a/0x02) pages.
38
+ _TABLE_PAGE_TYPES = frozenset({0x0D, 0x05})
39
+ _INDEX_PAGE_TYPES = frozenset({0x0A, 0x02})
40
+
41
+ #: Derived, never a separate literal: `observed` below reads "not a table page"
42
+ #: as "an index page", which is only sound while these three agree.
43
+ _VALID_PAGE_TYPES = _TABLE_PAGE_TYPES | _INDEX_PAGE_TYPES
44
+
45
+ #: Bounds the `sqlite_schema` walk so a corrupt child pointer cannot make the
46
+ #: scan read the whole file.
47
+ _MAX_SCHEMA_PAGES = 4096
48
+
49
+ _REASON_MAX = 200
50
+
51
+
52
+ class _ScanError(Exception):
53
+ """Internal: the raw scan cannot proceed. Never escapes this module."""
54
+
55
+
56
+ def _finding(kind: str, raw: str, **fields) -> dict:
57
+ out = {key: None for key in _FINDING_KEYS}
58
+ out["kind"] = kind
59
+ out["raw"] = raw
60
+ out.update(fields)
61
+ return out
62
+
63
+
64
+ def _reason(exc: BaseException) -> str:
65
+ text = f"{type(exc).__name__}: {exc}"
66
+ if len(text) <= _REASON_MAX:
67
+ return text
68
+ return text[: _REASON_MAX - 1] + "…"
69
+
70
+
71
+ # ==========================================================================
72
+ # integrity_check row parsing
73
+ # ==========================================================================
74
+
75
+ # The closed set of forms the production corpus actually contains. Anything
76
+ # else is retained verbatim as `unparsed` rather than discarded.
77
+ _RE_INDEX_ENTRY_COUNT = re.compile(
78
+ r"^wrong # of entries in index (?P<index>\S+)$"
79
+ )
80
+ _RE_ROW_MISSING = re.compile(
81
+ r"^row (?P<rowid>\d+) missing from index (?P<index>\S+)$"
82
+ )
83
+ _RE_NON_UNIQUE = re.compile(
84
+ r"^non-unique entry in index (?P<index>\S+)$"
85
+ )
86
+ _RE_COLUMN_VALUE = re.compile(
87
+ r"^(?P<what>NULL|NUMERIC|TEXT|BLOB|REAL|INTEGER) value in "
88
+ r"(?P<table>[^.\s]+)\.(?P<column>\S+)$"
89
+ )
90
+ _RE_TREE_CELL = re.compile(
91
+ r"^Tree (?P<tree>\d+) page (?P<page>\d+) cell (?P<cell>\d+): (?P<detail>.+)$"
92
+ )
93
+ _RE_TREE_PAGE = re.compile(
94
+ r"^Tree (?P<tree>\d+) page (?P<page>\d+): (?P<detail>.+)$"
95
+ )
96
+ _RE_PAGE_NEVER_USED = re.compile(r"^Page (?P<page>\d+): never used$")
97
+ _RE_PAGE_DETAIL = re.compile(r"^Page (?P<page>\d+): (?P<detail>.+)$")
98
+ _RE_DATABASE_BANNER = re.compile(r"^\*\*\* in database \S+ \*\*\*$")
99
+
100
+
101
+ def parse_integrity_rows(rows) -> list:
102
+ """Convert `PRAGMA integrity_check` output into typed findings.
103
+
104
+ ``rows`` is the value the forensics bundle stores: a list of strings when
105
+ the pragma returned rows, the captured error string when it raised, or
106
+ ``None`` when it never ran. Only the list form can yield findings.
107
+ """
108
+ if not isinstance(rows, (list, tuple)):
109
+ return []
110
+ findings = []
111
+ for row in rows:
112
+ text = str(row).strip()
113
+ if not text:
114
+ continue
115
+ if text.casefold() == "ok" or _RE_DATABASE_BANNER.match(text):
116
+ continue
117
+
118
+ match = _RE_INDEX_ENTRY_COUNT.match(text)
119
+ if match:
120
+ findings.append(
121
+ _finding("index_entry_count", text, index=match["index"])
122
+ )
123
+ continue
124
+
125
+ match = _RE_ROW_MISSING.match(text)
126
+ if match:
127
+ findings.append(
128
+ _finding(
129
+ "row_missing_from_index",
130
+ text,
131
+ index=match["index"],
132
+ rowid=int(match["rowid"]),
133
+ )
134
+ )
135
+ continue
136
+
137
+ match = _RE_NON_UNIQUE.match(text)
138
+ if match:
139
+ findings.append(
140
+ _finding("index_non_unique", text, index=match["index"])
141
+ )
142
+ continue
143
+
144
+ match = _RE_COLUMN_VALUE.match(text)
145
+ if match:
146
+ kind = "null_value" if match["what"] == "NULL" else "type_mismatch"
147
+ findings.append(
148
+ _finding(
149
+ kind, text, table=match["table"], column=match["column"],
150
+ )
151
+ )
152
+ continue
153
+
154
+ match = _RE_TREE_CELL.match(text)
155
+ if match:
156
+ findings.append(
157
+ _finding(
158
+ "tree_cell",
159
+ text,
160
+ page=int(match["page"]),
161
+ cell=int(match["cell"]),
162
+ )
163
+ )
164
+ continue
165
+
166
+ match = _RE_TREE_PAGE.match(text)
167
+ if match:
168
+ kind = (
169
+ "btree_init_error"
170
+ if "btreeInitPage()" in match["detail"]
171
+ else "tree_page"
172
+ )
173
+ findings.append(_finding(kind, text, page=int(match["page"])))
174
+ continue
175
+
176
+ match = _RE_PAGE_NEVER_USED.match(text)
177
+ if match:
178
+ findings.append(
179
+ _finding("page_never_used", text, page=int(match["page"]))
180
+ )
181
+ continue
182
+
183
+ match = _RE_PAGE_DETAIL.match(text)
184
+ if match:
185
+ findings.append(
186
+ _finding("page_detail", text, page=int(match["page"]))
187
+ )
188
+ continue
189
+
190
+ findings.append(_finding("unparsed", text))
191
+ return findings
192
+
193
+
194
+ # ==========================================================================
195
+ # raw file scan
196
+ # ==========================================================================
197
+
198
+ def _varint(buf: bytes, pos: int) -> tuple:
199
+ value = 0
200
+ for index in range(9):
201
+ if pos + index >= len(buf):
202
+ raise _ScanError("varint runs past the end of the page")
203
+ byte = buf[pos + index]
204
+ if index == 8:
205
+ return ((value << 8) | byte), pos + 9
206
+ value = (value << 7) | (byte & 0x7F)
207
+ if not byte & 0x80:
208
+ return value, pos + index + 1
209
+ raise _ScanError("malformed varint")
210
+
211
+
212
+ def _serial_size(serial: int) -> int:
213
+ if serial in (0, 8, 9, 10, 11):
214
+ return 0
215
+ if serial <= 4:
216
+ return serial
217
+ if serial == 5:
218
+ return 6
219
+ if serial == 6 or serial == 7:
220
+ return 8
221
+ return (serial - 12) // 2
222
+
223
+
224
+ def _serial_int(buf: bytes, offset: int, serial: int) -> "int | None":
225
+ if serial == 8:
226
+ return 0
227
+ if serial == 9:
228
+ return 1
229
+ size = _serial_size(serial)
230
+ if serial > 6 or size == 0:
231
+ return None
232
+ if offset + size > len(buf):
233
+ raise _ScanError("integer column runs past the end of the page")
234
+ return int.from_bytes(buf[offset:offset + size], "big", signed=True)
235
+
236
+
237
+ def _serial_text(buf: bytes, offset: int, serial: int) -> "str | None":
238
+ if serial < 13 or serial % 2 == 0:
239
+ return None
240
+ size = _serial_size(serial)
241
+ if offset + size > len(buf):
242
+ raise _ScanError("text column runs past the end of the page")
243
+ return buf[offset:offset + size].decode("utf-8", "replace")
244
+
245
+
246
+ def _read_page(handle, page_number: int, page_size: int) -> bytes:
247
+ handle.seek((page_number - 1) * page_size)
248
+ data = handle.read(page_size)
249
+ if len(data) != page_size:
250
+ raise _ScanError(f"page {page_number} is short or absent")
251
+ return data
252
+
253
+
254
+ def _parse_schema_cell(page: bytes, offset: int, objects: dict) -> None:
255
+ """Read `(type, name, rootpage, sql)` out of one `sqlite_schema` leaf cell.
256
+
257
+ The `sql` text is needed only to recognise a `WITHOUT ROWID` table, whose
258
+ root legitimately IS an index b-tree. It is the last column, so a payload
259
+ that overflows the page yields `None` and the caller declines to judge that
260
+ object's page kind rather than reporting a healthy table as damaged.
261
+ """
262
+ _payload_size, pos = _varint(page, offset)
263
+ _rowid, pos = _varint(page, pos)
264
+ header_size, header_pos = _varint(page, pos)
265
+ header_end = pos + header_size
266
+ serials = []
267
+ cursor = header_pos
268
+ while cursor < header_end and len(serials) < 5:
269
+ serial, cursor = _varint(page, cursor)
270
+ serials.append(serial)
271
+ if len(serials) < 4:
272
+ raise _ScanError("sqlite_schema record has fewer than four columns")
273
+ value_offsets = []
274
+ body = header_end
275
+ for serial in serials:
276
+ value_offsets.append(body)
277
+ body += _serial_size(serial)
278
+ obj_type = _serial_text(page, value_offsets[0], serials[0])
279
+ name = _serial_text(page, value_offsets[1], serials[1])
280
+ rootpage = _serial_int(page, value_offsets[3], serials[3])
281
+ sql = None
282
+ if len(serials) == 5:
283
+ try:
284
+ sql = _serial_text(page, value_offsets[4], serials[4])
285
+ except (_ScanError, IndexError, ValueError):
286
+ sql = None
287
+ if not name or not obj_type or not rootpage or rootpage < 1:
288
+ return
289
+ objects[name] = (obj_type, int(rootpage), sql)
290
+
291
+
292
+ def _walk_schema_btree(
293
+ handle, page_number: int, page_size: int, objects: dict, visited: set,
294
+ ) -> None:
295
+ if page_number in visited:
296
+ raise _ScanError(f"sqlite_schema b-tree revisits page {page_number}")
297
+ visited.add(page_number)
298
+ if len(visited) > _MAX_SCHEMA_PAGES:
299
+ raise _ScanError("sqlite_schema b-tree exceeds the page budget")
300
+ page = _read_page(handle, page_number, page_size)
301
+ base = 100 if page_number == 1 else 0
302
+ if base >= len(page):
303
+ raise _ScanError("page 1 is shorter than the file header")
304
+ page_type = page[base]
305
+ cell_count = int.from_bytes(page[base + 3:base + 5], "big")
306
+ if page_type == 0x0D:
307
+ pointer_base = base + 8
308
+ for index in range(cell_count):
309
+ at = pointer_base + 2 * index
310
+ if at + 2 > len(page):
311
+ raise _ScanError("cell pointer array runs past the page")
312
+ cell_offset = int.from_bytes(page[at:at + 2], "big")
313
+ # One unreadable row must not discard the whole map.
314
+ try:
315
+ _parse_schema_cell(page, cell_offset, objects)
316
+ except (_ScanError, IndexError, ValueError):
317
+ continue
318
+ elif page_type == 0x05:
319
+ pointer_base = base + 12
320
+ children = []
321
+ for index in range(cell_count):
322
+ at = pointer_base + 2 * index
323
+ if at + 2 > len(page):
324
+ raise _ScanError("cell pointer array runs past the page")
325
+ cell_offset = int.from_bytes(page[at:at + 2], "big")
326
+ if cell_offset + 4 > len(page):
327
+ raise _ScanError("interior cell runs past the page")
328
+ children.append(
329
+ int.from_bytes(page[cell_offset:cell_offset + 4], "big")
330
+ )
331
+ children.append(int.from_bytes(page[base + 8:base + 12], "big"))
332
+ for child in children:
333
+ if child < 1:
334
+ raise _ScanError("interior page names child page 0")
335
+ _walk_schema_btree(handle, child, page_size, objects, visited)
336
+ else:
337
+ raise _ScanError(
338
+ f"page {page_number} is not a table b-tree page "
339
+ f"(type byte 0x{page_type:02x})"
340
+ )
341
+
342
+
343
+ def _unavailable(reason: str) -> dict:
344
+ return {"method": "unavailable", "findings": [], "reason": reason}
345
+
346
+
347
+ #: `WITHOUT ROWID` as a table-option, i.e. after the column list closes.
348
+ _WITHOUT_ROWID_TAIL = re.compile(r"WITHOUT\s+ROWID", re.IGNORECASE)
349
+
350
+
351
+ def _is_without_rowid(sql) -> bool:
352
+ """True only when the table-options tail declares `WITHOUT ROWID`.
353
+
354
+ Searching the whole statement would exempt an ordinary rowid table whose
355
+ declaration merely MENTIONS the phrase — in a comment, a CHECK literal or a
356
+ DEFAULT literal — and a table exempted that way is reported as healthy no
357
+ matter how damaged it is, with no signal anywhere. Anchoring past the final
358
+ closing parenthesis costs the `) WITHOUT /*c*/ ROWID` form, which fails
359
+ toward declining to judge rather than toward a false clean bill.
360
+ """
361
+ if not sql:
362
+ return False
363
+ text = str(sql)
364
+ close = text.rfind(")")
365
+ if close < 0:
366
+ return False
367
+ return _WITHOUT_ROWID_TAIL.search(text[close + 1:]) is not None
368
+
369
+
370
+ def scan_sqlite_file(path) -> dict:
371
+ """Describe a SQLite file by reading its own bytes. Never raises.
372
+
373
+ Returns ``{"method": "raw_scan"|"unavailable", "findings": [...],
374
+ "reason": str | None}``. Pages are read individually by seek; the file is
375
+ never loaded.
376
+ """
377
+ if path is None:
378
+ return _unavailable("ValueError: no path was supplied")
379
+ try:
380
+ target = pathlib.Path(path)
381
+ with target.open("rb") as handle:
382
+ header = handle.read(100)
383
+ if len(header) < 100:
384
+ raise _ScanError("file is shorter than the 100-byte header")
385
+ if not header.startswith(_SQLITE_MAGIC):
386
+ raise _ScanError("file does not carry the SQLite header magic")
387
+ raw_page_size = int.from_bytes(header[16:18], "big")
388
+ page_size = 65536 if raw_page_size == 1 else raw_page_size
389
+ if page_size < 512 or (page_size & (page_size - 1)) != 0:
390
+ raise _ScanError(f"header declares an invalid page size {page_size}")
391
+ size = target.stat().st_size
392
+ if size % page_size != 0:
393
+ raise _ScanError(
394
+ "file size is not a whole number of "
395
+ f"{page_size}-byte pages"
396
+ )
397
+
398
+ objects: dict = {}
399
+ _walk_schema_btree(handle, 1, page_size, objects, set())
400
+
401
+ findings = []
402
+ for name in sorted(objects):
403
+ obj_type, rootpage, sql = objects[name]
404
+ if obj_type not in ("table", "index"):
405
+ continue
406
+ handle.seek((rootpage - 1) * page_size + (100 if rootpage == 1 else 0))
407
+ probe = handle.read(1)
408
+ if len(probe) != 1:
409
+ raise _ScanError(
410
+ f"root page {rootpage} of {obj_type} {name} is absent"
411
+ )
412
+ page_type = probe[0]
413
+ if page_type not in _VALID_PAGE_TYPES:
414
+ findings.append(
415
+ _finding(
416
+ "bad_root_page_type",
417
+ f"root page {rootpage} of {obj_type} {name} has type "
418
+ f"byte 0x{page_type:02x}",
419
+ table=name if obj_type == "table" else None,
420
+ index=name if obj_type == "index" else None,
421
+ page=rootpage,
422
+ )
423
+ )
424
+ continue
425
+ if obj_type == "table" and sql is None:
426
+ # Only a TABLE needs its declaration, and only to rule out
427
+ # WITHOUT ROWID. Gating this on every object type would skip
428
+ # every `sqlite_autoindex_*` entry, whose `sql` is NULL by
429
+ # definition — including the automatic index the recurring
430
+ # production signature names.
431
+ continue
432
+ if obj_type == "index" or _is_without_rowid(sql):
433
+ expected = _INDEX_PAGE_TYPES
434
+ else:
435
+ expected = _TABLE_PAGE_TYPES
436
+ if page_type in expected:
437
+ continue
438
+ observed = "index" if page_type in _INDEX_PAGE_TYPES else "table"
439
+ article = "an" if observed == "index" else "a"
440
+ findings.append(
441
+ _finding(
442
+ "root_page_kind_mismatch",
443
+ f"root page {rootpage} of {obj_type} {name} holds "
444
+ f"{article} {observed} b-tree page "
445
+ f"(type byte 0x{page_type:02x})",
446
+ table=name if obj_type == "table" else None,
447
+ index=name if obj_type == "index" else None,
448
+ page=rootpage,
449
+ )
450
+ )
451
+ return {"method": "raw_scan", "findings": findings, "reason": None}
452
+ except Exception as exc: # noqa: BLE001 — characterization never raises
453
+ return _unavailable(_reason(exc))
454
+
455
+
456
+ # ==========================================================================
457
+ # normalized shape token
458
+ # ==========================================================================
459
+
460
+ def shape_token(findings) -> str:
461
+ """A short equality-comparable token for one damage class.
462
+
463
+ Derived from the SORTED SET of ``(kind, table, index)`` triples, so page
464
+ numbers, cell indices and rowids — which differ between two instances of
465
+ the same fault — cannot make one recurrence look like two.
466
+ """
467
+ try:
468
+ triples = sorted(
469
+ {
470
+ (
471
+ str(finding.get("kind") or ""),
472
+ str(finding.get("table") or ""),
473
+ str(finding.get("index") or ""),
474
+ )
475
+ for finding in (findings or ())
476
+ }
477
+ )
478
+ except Exception: # noqa: BLE001 — characterization never raises
479
+ return "none"
480
+ if not triples:
481
+ return "none"
482
+ payload = "|".join(":".join(triple) for triple in triples)
483
+ return hashlib.sha256(payload.encode("utf-8")).hexdigest()[:16]
484
+
485
+
486
+ def describe_damage(*, integrity_rows, path) -> dict:
487
+ """One structured description from whichever sources are available.
488
+
489
+ ``method`` is ``integrity_rows`` when only the pragma produced findings,
490
+ ``raw_scan`` when only the file scan did, ``both`` when each did, and
491
+ ``unavailable`` when neither source could say anything.
492
+ """
493
+ try:
494
+ parsed = parse_integrity_rows(integrity_rows)
495
+ scan = scan_sqlite_file(path)
496
+ scanned = list(scan.get("findings") or ())
497
+ scan_available = scan.get("method") == "raw_scan"
498
+
499
+ if parsed and scanned:
500
+ method = "both"
501
+ elif parsed:
502
+ method = "integrity_rows"
503
+ elif scan_available:
504
+ method = "raw_scan"
505
+ else:
506
+ method = "unavailable"
507
+
508
+ findings = parsed + scanned
509
+ return {
510
+ "schemaVersion": SCHEMA_VERSION,
511
+ "method": method,
512
+ "findings": findings,
513
+ "shapeToken": shape_token(findings),
514
+ "reason": None if scan_available else scan.get("reason"),
515
+ }
516
+ except Exception as exc: # noqa: BLE001 — characterization never raises
517
+ return {
518
+ "schemaVersion": SCHEMA_VERSION,
519
+ "method": "unavailable",
520
+ "findings": [],
521
+ "shapeToken": "none",
522
+ "reason": _reason(exc),
523
+ }