cctally 1.91.0 → 1.92.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +51 -0
- package/README.md +4 -2
- package/bin/_cctally_cache.py +903 -74
- package/bin/_cctally_config.py +57 -0
- package/bin/_cctally_core.py +94 -14
- package/bin/_cctally_dashboard.py +217 -19
- package/bin/_cctally_dashboard_conversation.py +170 -20
- package/bin/_cctally_dashboard_envelope.py +2 -0
- package/bin/_cctally_db.py +481 -19
- package/bin/_cctally_doctor.py +18 -1
- package/bin/_cctally_journal.py +1156 -21
- package/bin/_cctally_journal_repair.py +6 -0
- package/bin/_cctally_parser.py +26 -0
- package/bin/_cctally_quota.py +171 -55
- package/bin/_cctally_record.py +13 -1
- package/bin/_cctally_rederive.py +4 -0
- package/bin/_cctally_statusline.py +6 -6
- package/bin/_cctally_store.py +1061 -40
- package/bin/_cctally_transcript.py +32 -2
- package/bin/_cctally_tui.py +54 -6
- package/bin/_lib_cache_report.py +8 -3
- package/bin/_lib_codex_conversation.py +851 -81
- package/bin/_lib_codex_conversation_query.py +2031 -96
- package/bin/_lib_codex_find_projection.py +517 -0
- package/bin/_lib_codex_harness_preamble.py +176 -0
- package/bin/_lib_codex_hooks.py +5 -3
- package/bin/_lib_codex_js_scan.py +254 -0
- package/bin/_lib_codex_landmarks.py +309 -0
- package/bin/_lib_codex_title_clean.py +116 -0
- package/bin/_lib_conversation_dispatch.py +168 -22
- package/bin/_lib_conversation_query.py +62 -2
- package/bin/_lib_conversation_watch.py +4 -2
- package/bin/_lib_doctor.py +64 -0
- package/bin/_lib_quota_alert_axes.py +31 -34
- package/bin/_lib_stats_damage.py +523 -0
- package/bin/_lib_stats_publish.py +243 -0
- package/bin/cctally +17 -3
- package/dashboard/static/assets/index-Dat-mza6.js +97 -0
- package/dashboard/static/assets/{index-Dwirao3Y.css → index-DnWdv8um.css} +1 -1
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +8 -1
- package/dashboard/static/assets/index-CILAoEja.js +0 -90
|
@@ -23,9 +23,10 @@ in ``quota_window_snapshots`` moving at all:
|
|
|
23
23
|
and becomes eligible when wall time passes it with no mutation to observe.
|
|
24
24
|
Persisting that boundary and treating ``now >= boundary`` as dirty is what
|
|
25
25
|
closes it. Unlike axes 2 and 3 this one fires on WALL CLOCK rather than on a
|
|
26
|
-
configuration change
|
|
27
|
-
|
|
28
|
-
a blocking
|
|
26
|
+
configuration change. An ownership schedule lets the hook evaluate only the
|
|
27
|
+
complete roots whose deadlines matured; scalar-only legacy state still
|
|
28
|
+
defers rather than paying an unannounced whole-history pass on a blocking
|
|
29
|
+
tick.
|
|
29
30
|
5. **Durable lifecycle state** — the existing arming rows and terminal events,
|
|
30
31
|
unchanged. Represented here only as the fingerprints axis 2 compares.
|
|
31
32
|
|
|
@@ -80,6 +81,7 @@ def alert_dirty_scope(
|
|
|
80
81
|
gate_after: bool,
|
|
81
82
|
now: dt.datetime,
|
|
82
83
|
next_evaluation_at: "dt.datetime | None",
|
|
84
|
+
scheduled_roots: "Iterable[str] | None" = None,
|
|
83
85
|
defer_scheduled: bool = False,
|
|
84
86
|
) -> AlertDirtyScope:
|
|
85
87
|
"""Resolve the five axes into one decision.
|
|
@@ -89,14 +91,11 @@ def alert_dirty_scope(
|
|
|
89
91
|
observed_slot, window_minutes)``; the ROOT is element 1, which is what an
|
|
90
92
|
exact-rule change is scoped to.
|
|
91
93
|
|
|
92
|
-
``
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
``REASON_SCHEDULED_DEFERRED`` and does NOT strengthen the scope — and the
|
|
98
|
-
caller owes the stored boundary a carry-through, because a deferral that
|
|
99
|
-
lets the boundary be recomputed is a silent drop.
|
|
94
|
+
``scheduled_roots`` is the validated ownership retained with axis 4. When
|
|
95
|
+
present, a matured instant scopes to those roots even on the hook path.
|
|
96
|
+
``None`` is the legacy/unavailable-ownership shape; only that shape needs
|
|
97
|
+
``defer_scheduled`` to avoid an unannounced whole-history hook pass, and the
|
|
98
|
+
caller then owes the scalar boundary a carry-through.
|
|
100
99
|
"""
|
|
101
100
|
reasons: list[str] = []
|
|
102
101
|
if not gate_after:
|
|
@@ -132,11 +131,19 @@ def alert_dirty_scope(
|
|
|
132
131
|
reasons.append("rule_changed")
|
|
133
132
|
|
|
134
133
|
if next_evaluation_at is not None and now >= next_evaluation_at:
|
|
135
|
-
#
|
|
136
|
-
#
|
|
137
|
-
#
|
|
138
|
-
#
|
|
139
|
-
|
|
134
|
+
# Epoch 1007 records the roots owning each scheduled instant. A complete
|
|
135
|
+
# semantic pass over those roots is bounded enough for the hook path and
|
|
136
|
+
# is all axis 4 needs. ``None`` means legacy/unavailable ownership, where
|
|
137
|
+
# the only honest scope remains everything (and therefore deferral on a
|
|
138
|
+
# hook tick). An empty known set means the owning roots are not lifecycle
|
|
139
|
+
# eligible on this tick; the stored axis remains due for a later tick.
|
|
140
|
+
if scheduled_roots is not None:
|
|
141
|
+
due_roots = {str(root) for root in scheduled_roots if str(root)}
|
|
142
|
+
if due_roots:
|
|
143
|
+
scope = _strongest(scope, SCOPE_ROOTS)
|
|
144
|
+
roots |= due_roots
|
|
145
|
+
reasons.append("scheduled")
|
|
146
|
+
elif defer_scheduled:
|
|
140
147
|
reasons.append(REASON_SCHEDULED_DEFERRED)
|
|
141
148
|
else:
|
|
142
149
|
scope = _strongest(scope, SCOPE_ALL)
|
|
@@ -154,12 +161,12 @@ def next_evaluation_boundary(
|
|
|
154
161
|
) -> "dt.datetime | None":
|
|
155
162
|
"""The earliest still-future capture the projector must come back for.
|
|
156
163
|
|
|
157
|
-
A bounded pass only sees
|
|
164
|
+
This is the legacy scalar helper. A bounded pass only sees dirty windows, so
|
|
165
|
+
the STORED boundary is
|
|
158
166
|
retained whenever it is still in the future: dropping it would forget a
|
|
159
167
|
future-clocked observation sitting in a window this pass never loaded. Once
|
|
160
|
-
wall time passes it the axis fires
|
|
161
|
-
|
|
162
|
-
ever cost one extra pass, never a missed one.
|
|
168
|
+
wall time passes it the axis fires and the caller decides whether it has
|
|
169
|
+
enough ownership to scope the pass.
|
|
163
170
|
|
|
164
171
|
``retain_due`` keeps a boundary that is ALREADY due, which is the case where
|
|
165
172
|
"recomputed from complete evidence" is a lie: a reporting-only pass never
|
|
@@ -167,20 +174,10 @@ def next_evaluation_boundary(
|
|
|
167
174
|
the widening deliberately did not look. Either would otherwise retire the
|
|
168
175
|
axis on behalf of an evaluation nobody performed.
|
|
169
176
|
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
hook is the only production caller that carries it.
|
|
175
|
-
|
|
176
|
-
It does NOT stay "until a pass that can act on it does", and that gap is
|
|
177
|
-
open rather than closed: on a hook-only install with a steady enabled gate,
|
|
178
|
-
unchanged rules and a quiet ledger, no qualifying pass ever runs and the
|
|
179
|
-
instant is retained indefinitely. The cost is bounded — the tick stays
|
|
180
|
-
bounded and fast, and the window is re-evaluated as soon as it goes
|
|
181
|
-
ledger-dirty again, which for a live window is continuous — so the exposure
|
|
182
|
-
is a future-clocked capture in a window that then goes permanently quiet
|
|
183
|
-
never qualifying a threshold. Under-alerting, never a stall or a burst.
|
|
177
|
+
Epoch 1007's per-root map is maintained by the projector rather than this
|
|
178
|
+
helper. It closes the quiet-window gap by letting a hook tick replace only
|
|
179
|
+
the roots it evaluated; scalar-only legacy state still uses ``retain_due``
|
|
180
|
+
and the conservative full/deferred path.
|
|
184
181
|
"""
|
|
185
182
|
candidates = [value for value in capture_times if value > now]
|
|
186
183
|
if stored is not None and (retain_due or stored > now):
|
|
@@ -0,0 +1,523 @@
|
|
|
1
|
+
"""Pure damage-characterization kernel for a corrupt SQLite index (#496 S1 F8).
|
|
2
|
+
|
|
3
|
+
Two independent sources feed one structured description.
|
|
4
|
+
|
|
5
|
+
`parse_integrity_rows` converts the row forms `PRAGMA integrity_check` emits
|
|
6
|
+
into typed findings. That covers only the minority of incidents: in 54 of the
|
|
7
|
+
74 retained production forensics bundles the pragma RAISED before producing any
|
|
8
|
+
row, so the bundle holds the plain string ``error: database disk image is
|
|
9
|
+
malformed`` and nothing can be derived from it.
|
|
10
|
+
|
|
11
|
+
`scan_sqlite_file` covers the rest by reading the file itself — the 100-byte
|
|
12
|
+
header, then the `sqlite_schema` b-tree rooted at page 1 — and probing the type
|
|
13
|
+
byte of every root page that schema names. It opens no SQLite connection, so it
|
|
14
|
+
still describes a file SQLite refuses to open.
|
|
15
|
+
|
|
16
|
+
`shape_token` normalizes either source into a short equality-comparable token
|
|
17
|
+
with page numbers, cell indices and rowids removed, so a recurring damage class
|
|
18
|
+
is detectable by comparison rather than by reading prose.
|
|
19
|
+
|
|
20
|
+
Nothing in this module raises. A rebuild must never fail because diagnostic
|
|
21
|
+
enrichment failed, so every failure path returns an ``unavailable`` method with
|
|
22
|
+
a bounded reason string.
|
|
23
|
+
"""
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import hashlib
|
|
27
|
+
import pathlib
|
|
28
|
+
import re
|
|
29
|
+
|
|
30
|
+
SCHEMA_VERSION = 1
|
|
31
|
+
|
|
32
|
+
#: Finding keys are always all present, so consumers never need ``.get``.
|
|
33
|
+
_FINDING_KEYS = ("kind", "table", "index", "column", "page", "cell", "rowid", "raw")
|
|
34
|
+
|
|
35
|
+
_SQLITE_MAGIC = b"SQLite format 3\x00"
|
|
36
|
+
|
|
37
|
+
#: Leaf/interior table (0x0d/0x05) and leaf/interior index (0x0a/0x02) pages.
|
|
38
|
+
_TABLE_PAGE_TYPES = frozenset({0x0D, 0x05})
|
|
39
|
+
_INDEX_PAGE_TYPES = frozenset({0x0A, 0x02})
|
|
40
|
+
|
|
41
|
+
#: Derived, never a separate literal: `observed` below reads "not a table page"
|
|
42
|
+
#: as "an index page", which is only sound while these three agree.
|
|
43
|
+
_VALID_PAGE_TYPES = _TABLE_PAGE_TYPES | _INDEX_PAGE_TYPES
|
|
44
|
+
|
|
45
|
+
#: Bounds the `sqlite_schema` walk so a corrupt child pointer cannot make the
|
|
46
|
+
#: scan read the whole file.
|
|
47
|
+
_MAX_SCHEMA_PAGES = 4096
|
|
48
|
+
|
|
49
|
+
_REASON_MAX = 200
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class _ScanError(Exception):
|
|
53
|
+
"""Internal: the raw scan cannot proceed. Never escapes this module."""
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _finding(kind: str, raw: str, **fields) -> dict:
|
|
57
|
+
out = {key: None for key in _FINDING_KEYS}
|
|
58
|
+
out["kind"] = kind
|
|
59
|
+
out["raw"] = raw
|
|
60
|
+
out.update(fields)
|
|
61
|
+
return out
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _reason(exc: BaseException) -> str:
|
|
65
|
+
text = f"{type(exc).__name__}: {exc}"
|
|
66
|
+
if len(text) <= _REASON_MAX:
|
|
67
|
+
return text
|
|
68
|
+
return text[: _REASON_MAX - 1] + "…"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
# ==========================================================================
|
|
72
|
+
# integrity_check row parsing
|
|
73
|
+
# ==========================================================================
|
|
74
|
+
|
|
75
|
+
# The closed set of forms the production corpus actually contains. Anything
|
|
76
|
+
# else is retained verbatim as `unparsed` rather than discarded.
|
|
77
|
+
_RE_INDEX_ENTRY_COUNT = re.compile(
|
|
78
|
+
r"^wrong # of entries in index (?P<index>\S+)$"
|
|
79
|
+
)
|
|
80
|
+
_RE_ROW_MISSING = re.compile(
|
|
81
|
+
r"^row (?P<rowid>\d+) missing from index (?P<index>\S+)$"
|
|
82
|
+
)
|
|
83
|
+
_RE_NON_UNIQUE = re.compile(
|
|
84
|
+
r"^non-unique entry in index (?P<index>\S+)$"
|
|
85
|
+
)
|
|
86
|
+
_RE_COLUMN_VALUE = re.compile(
|
|
87
|
+
r"^(?P<what>NULL|NUMERIC|TEXT|BLOB|REAL|INTEGER) value in "
|
|
88
|
+
r"(?P<table>[^.\s]+)\.(?P<column>\S+)$"
|
|
89
|
+
)
|
|
90
|
+
_RE_TREE_CELL = re.compile(
|
|
91
|
+
r"^Tree (?P<tree>\d+) page (?P<page>\d+) cell (?P<cell>\d+): (?P<detail>.+)$"
|
|
92
|
+
)
|
|
93
|
+
_RE_TREE_PAGE = re.compile(
|
|
94
|
+
r"^Tree (?P<tree>\d+) page (?P<page>\d+): (?P<detail>.+)$"
|
|
95
|
+
)
|
|
96
|
+
_RE_PAGE_NEVER_USED = re.compile(r"^Page (?P<page>\d+): never used$")
|
|
97
|
+
_RE_PAGE_DETAIL = re.compile(r"^Page (?P<page>\d+): (?P<detail>.+)$")
|
|
98
|
+
_RE_DATABASE_BANNER = re.compile(r"^\*\*\* in database \S+ \*\*\*$")
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def parse_integrity_rows(rows) -> list:
|
|
102
|
+
"""Convert `PRAGMA integrity_check` output into typed findings.
|
|
103
|
+
|
|
104
|
+
``rows`` is the value the forensics bundle stores: a list of strings when
|
|
105
|
+
the pragma returned rows, the captured error string when it raised, or
|
|
106
|
+
``None`` when it never ran. Only the list form can yield findings.
|
|
107
|
+
"""
|
|
108
|
+
if not isinstance(rows, (list, tuple)):
|
|
109
|
+
return []
|
|
110
|
+
findings = []
|
|
111
|
+
for row in rows:
|
|
112
|
+
text = str(row).strip()
|
|
113
|
+
if not text:
|
|
114
|
+
continue
|
|
115
|
+
if text.casefold() == "ok" or _RE_DATABASE_BANNER.match(text):
|
|
116
|
+
continue
|
|
117
|
+
|
|
118
|
+
match = _RE_INDEX_ENTRY_COUNT.match(text)
|
|
119
|
+
if match:
|
|
120
|
+
findings.append(
|
|
121
|
+
_finding("index_entry_count", text, index=match["index"])
|
|
122
|
+
)
|
|
123
|
+
continue
|
|
124
|
+
|
|
125
|
+
match = _RE_ROW_MISSING.match(text)
|
|
126
|
+
if match:
|
|
127
|
+
findings.append(
|
|
128
|
+
_finding(
|
|
129
|
+
"row_missing_from_index",
|
|
130
|
+
text,
|
|
131
|
+
index=match["index"],
|
|
132
|
+
rowid=int(match["rowid"]),
|
|
133
|
+
)
|
|
134
|
+
)
|
|
135
|
+
continue
|
|
136
|
+
|
|
137
|
+
match = _RE_NON_UNIQUE.match(text)
|
|
138
|
+
if match:
|
|
139
|
+
findings.append(
|
|
140
|
+
_finding("index_non_unique", text, index=match["index"])
|
|
141
|
+
)
|
|
142
|
+
continue
|
|
143
|
+
|
|
144
|
+
match = _RE_COLUMN_VALUE.match(text)
|
|
145
|
+
if match:
|
|
146
|
+
kind = "null_value" if match["what"] == "NULL" else "type_mismatch"
|
|
147
|
+
findings.append(
|
|
148
|
+
_finding(
|
|
149
|
+
kind, text, table=match["table"], column=match["column"],
|
|
150
|
+
)
|
|
151
|
+
)
|
|
152
|
+
continue
|
|
153
|
+
|
|
154
|
+
match = _RE_TREE_CELL.match(text)
|
|
155
|
+
if match:
|
|
156
|
+
findings.append(
|
|
157
|
+
_finding(
|
|
158
|
+
"tree_cell",
|
|
159
|
+
text,
|
|
160
|
+
page=int(match["page"]),
|
|
161
|
+
cell=int(match["cell"]),
|
|
162
|
+
)
|
|
163
|
+
)
|
|
164
|
+
continue
|
|
165
|
+
|
|
166
|
+
match = _RE_TREE_PAGE.match(text)
|
|
167
|
+
if match:
|
|
168
|
+
kind = (
|
|
169
|
+
"btree_init_error"
|
|
170
|
+
if "btreeInitPage()" in match["detail"]
|
|
171
|
+
else "tree_page"
|
|
172
|
+
)
|
|
173
|
+
findings.append(_finding(kind, text, page=int(match["page"])))
|
|
174
|
+
continue
|
|
175
|
+
|
|
176
|
+
match = _RE_PAGE_NEVER_USED.match(text)
|
|
177
|
+
if match:
|
|
178
|
+
findings.append(
|
|
179
|
+
_finding("page_never_used", text, page=int(match["page"]))
|
|
180
|
+
)
|
|
181
|
+
continue
|
|
182
|
+
|
|
183
|
+
match = _RE_PAGE_DETAIL.match(text)
|
|
184
|
+
if match:
|
|
185
|
+
findings.append(
|
|
186
|
+
_finding("page_detail", text, page=int(match["page"]))
|
|
187
|
+
)
|
|
188
|
+
continue
|
|
189
|
+
|
|
190
|
+
findings.append(_finding("unparsed", text))
|
|
191
|
+
return findings
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
# ==========================================================================
|
|
195
|
+
# raw file scan
|
|
196
|
+
# ==========================================================================
|
|
197
|
+
|
|
198
|
+
def _varint(buf: bytes, pos: int) -> tuple:
|
|
199
|
+
value = 0
|
|
200
|
+
for index in range(9):
|
|
201
|
+
if pos + index >= len(buf):
|
|
202
|
+
raise _ScanError("varint runs past the end of the page")
|
|
203
|
+
byte = buf[pos + index]
|
|
204
|
+
if index == 8:
|
|
205
|
+
return ((value << 8) | byte), pos + 9
|
|
206
|
+
value = (value << 7) | (byte & 0x7F)
|
|
207
|
+
if not byte & 0x80:
|
|
208
|
+
return value, pos + index + 1
|
|
209
|
+
raise _ScanError("malformed varint")
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _serial_size(serial: int) -> int:
|
|
213
|
+
if serial in (0, 8, 9, 10, 11):
|
|
214
|
+
return 0
|
|
215
|
+
if serial <= 4:
|
|
216
|
+
return serial
|
|
217
|
+
if serial == 5:
|
|
218
|
+
return 6
|
|
219
|
+
if serial == 6 or serial == 7:
|
|
220
|
+
return 8
|
|
221
|
+
return (serial - 12) // 2
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _serial_int(buf: bytes, offset: int, serial: int) -> "int | None":
|
|
225
|
+
if serial == 8:
|
|
226
|
+
return 0
|
|
227
|
+
if serial == 9:
|
|
228
|
+
return 1
|
|
229
|
+
size = _serial_size(serial)
|
|
230
|
+
if serial > 6 or size == 0:
|
|
231
|
+
return None
|
|
232
|
+
if offset + size > len(buf):
|
|
233
|
+
raise _ScanError("integer column runs past the end of the page")
|
|
234
|
+
return int.from_bytes(buf[offset:offset + size], "big", signed=True)
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def _serial_text(buf: bytes, offset: int, serial: int) -> "str | None":
|
|
238
|
+
if serial < 13 or serial % 2 == 0:
|
|
239
|
+
return None
|
|
240
|
+
size = _serial_size(serial)
|
|
241
|
+
if offset + size > len(buf):
|
|
242
|
+
raise _ScanError("text column runs past the end of the page")
|
|
243
|
+
return buf[offset:offset + size].decode("utf-8", "replace")
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _read_page(handle, page_number: int, page_size: int) -> bytes:
|
|
247
|
+
handle.seek((page_number - 1) * page_size)
|
|
248
|
+
data = handle.read(page_size)
|
|
249
|
+
if len(data) != page_size:
|
|
250
|
+
raise _ScanError(f"page {page_number} is short or absent")
|
|
251
|
+
return data
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _parse_schema_cell(page: bytes, offset: int, objects: dict) -> None:
|
|
255
|
+
"""Read `(type, name, rootpage, sql)` out of one `sqlite_schema` leaf cell.
|
|
256
|
+
|
|
257
|
+
The `sql` text is needed only to recognise a `WITHOUT ROWID` table, whose
|
|
258
|
+
root legitimately IS an index b-tree. It is the last column, so a payload
|
|
259
|
+
that overflows the page yields `None` and the caller declines to judge that
|
|
260
|
+
object's page kind rather than reporting a healthy table as damaged.
|
|
261
|
+
"""
|
|
262
|
+
_payload_size, pos = _varint(page, offset)
|
|
263
|
+
_rowid, pos = _varint(page, pos)
|
|
264
|
+
header_size, header_pos = _varint(page, pos)
|
|
265
|
+
header_end = pos + header_size
|
|
266
|
+
serials = []
|
|
267
|
+
cursor = header_pos
|
|
268
|
+
while cursor < header_end and len(serials) < 5:
|
|
269
|
+
serial, cursor = _varint(page, cursor)
|
|
270
|
+
serials.append(serial)
|
|
271
|
+
if len(serials) < 4:
|
|
272
|
+
raise _ScanError("sqlite_schema record has fewer than four columns")
|
|
273
|
+
value_offsets = []
|
|
274
|
+
body = header_end
|
|
275
|
+
for serial in serials:
|
|
276
|
+
value_offsets.append(body)
|
|
277
|
+
body += _serial_size(serial)
|
|
278
|
+
obj_type = _serial_text(page, value_offsets[0], serials[0])
|
|
279
|
+
name = _serial_text(page, value_offsets[1], serials[1])
|
|
280
|
+
rootpage = _serial_int(page, value_offsets[3], serials[3])
|
|
281
|
+
sql = None
|
|
282
|
+
if len(serials) == 5:
|
|
283
|
+
try:
|
|
284
|
+
sql = _serial_text(page, value_offsets[4], serials[4])
|
|
285
|
+
except (_ScanError, IndexError, ValueError):
|
|
286
|
+
sql = None
|
|
287
|
+
if not name or not obj_type or not rootpage or rootpage < 1:
|
|
288
|
+
return
|
|
289
|
+
objects[name] = (obj_type, int(rootpage), sql)
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def _walk_schema_btree(
|
|
293
|
+
handle, page_number: int, page_size: int, objects: dict, visited: set,
|
|
294
|
+
) -> None:
|
|
295
|
+
if page_number in visited:
|
|
296
|
+
raise _ScanError(f"sqlite_schema b-tree revisits page {page_number}")
|
|
297
|
+
visited.add(page_number)
|
|
298
|
+
if len(visited) > _MAX_SCHEMA_PAGES:
|
|
299
|
+
raise _ScanError("sqlite_schema b-tree exceeds the page budget")
|
|
300
|
+
page = _read_page(handle, page_number, page_size)
|
|
301
|
+
base = 100 if page_number == 1 else 0
|
|
302
|
+
if base >= len(page):
|
|
303
|
+
raise _ScanError("page 1 is shorter than the file header")
|
|
304
|
+
page_type = page[base]
|
|
305
|
+
cell_count = int.from_bytes(page[base + 3:base + 5], "big")
|
|
306
|
+
if page_type == 0x0D:
|
|
307
|
+
pointer_base = base + 8
|
|
308
|
+
for index in range(cell_count):
|
|
309
|
+
at = pointer_base + 2 * index
|
|
310
|
+
if at + 2 > len(page):
|
|
311
|
+
raise _ScanError("cell pointer array runs past the page")
|
|
312
|
+
cell_offset = int.from_bytes(page[at:at + 2], "big")
|
|
313
|
+
# One unreadable row must not discard the whole map.
|
|
314
|
+
try:
|
|
315
|
+
_parse_schema_cell(page, cell_offset, objects)
|
|
316
|
+
except (_ScanError, IndexError, ValueError):
|
|
317
|
+
continue
|
|
318
|
+
elif page_type == 0x05:
|
|
319
|
+
pointer_base = base + 12
|
|
320
|
+
children = []
|
|
321
|
+
for index in range(cell_count):
|
|
322
|
+
at = pointer_base + 2 * index
|
|
323
|
+
if at + 2 > len(page):
|
|
324
|
+
raise _ScanError("cell pointer array runs past the page")
|
|
325
|
+
cell_offset = int.from_bytes(page[at:at + 2], "big")
|
|
326
|
+
if cell_offset + 4 > len(page):
|
|
327
|
+
raise _ScanError("interior cell runs past the page")
|
|
328
|
+
children.append(
|
|
329
|
+
int.from_bytes(page[cell_offset:cell_offset + 4], "big")
|
|
330
|
+
)
|
|
331
|
+
children.append(int.from_bytes(page[base + 8:base + 12], "big"))
|
|
332
|
+
for child in children:
|
|
333
|
+
if child < 1:
|
|
334
|
+
raise _ScanError("interior page names child page 0")
|
|
335
|
+
_walk_schema_btree(handle, child, page_size, objects, visited)
|
|
336
|
+
else:
|
|
337
|
+
raise _ScanError(
|
|
338
|
+
f"page {page_number} is not a table b-tree page "
|
|
339
|
+
f"(type byte 0x{page_type:02x})"
|
|
340
|
+
)
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def _unavailable(reason: str) -> dict:
|
|
344
|
+
return {"method": "unavailable", "findings": [], "reason": reason}
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
#: `WITHOUT ROWID` as a table-option, i.e. after the column list closes.
|
|
348
|
+
_WITHOUT_ROWID_TAIL = re.compile(r"WITHOUT\s+ROWID", re.IGNORECASE)
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def _is_without_rowid(sql) -> bool:
|
|
352
|
+
"""True only when the table-options tail declares `WITHOUT ROWID`.
|
|
353
|
+
|
|
354
|
+
Searching the whole statement would exempt an ordinary rowid table whose
|
|
355
|
+
declaration merely MENTIONS the phrase — in a comment, a CHECK literal or a
|
|
356
|
+
DEFAULT literal — and a table exempted that way is reported as healthy no
|
|
357
|
+
matter how damaged it is, with no signal anywhere. Anchoring past the final
|
|
358
|
+
closing parenthesis costs the `) WITHOUT /*c*/ ROWID` form, which fails
|
|
359
|
+
toward declining to judge rather than toward a false clean bill.
|
|
360
|
+
"""
|
|
361
|
+
if not sql:
|
|
362
|
+
return False
|
|
363
|
+
text = str(sql)
|
|
364
|
+
close = text.rfind(")")
|
|
365
|
+
if close < 0:
|
|
366
|
+
return False
|
|
367
|
+
return _WITHOUT_ROWID_TAIL.search(text[close + 1:]) is not None
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def scan_sqlite_file(path) -> dict:
|
|
371
|
+
"""Describe a SQLite file by reading its own bytes. Never raises.
|
|
372
|
+
|
|
373
|
+
Returns ``{"method": "raw_scan"|"unavailable", "findings": [...],
|
|
374
|
+
"reason": str | None}``. Pages are read individually by seek; the file is
|
|
375
|
+
never loaded.
|
|
376
|
+
"""
|
|
377
|
+
if path is None:
|
|
378
|
+
return _unavailable("ValueError: no path was supplied")
|
|
379
|
+
try:
|
|
380
|
+
target = pathlib.Path(path)
|
|
381
|
+
with target.open("rb") as handle:
|
|
382
|
+
header = handle.read(100)
|
|
383
|
+
if len(header) < 100:
|
|
384
|
+
raise _ScanError("file is shorter than the 100-byte header")
|
|
385
|
+
if not header.startswith(_SQLITE_MAGIC):
|
|
386
|
+
raise _ScanError("file does not carry the SQLite header magic")
|
|
387
|
+
raw_page_size = int.from_bytes(header[16:18], "big")
|
|
388
|
+
page_size = 65536 if raw_page_size == 1 else raw_page_size
|
|
389
|
+
if page_size < 512 or (page_size & (page_size - 1)) != 0:
|
|
390
|
+
raise _ScanError(f"header declares an invalid page size {page_size}")
|
|
391
|
+
size = target.stat().st_size
|
|
392
|
+
if size % page_size != 0:
|
|
393
|
+
raise _ScanError(
|
|
394
|
+
"file size is not a whole number of "
|
|
395
|
+
f"{page_size}-byte pages"
|
|
396
|
+
)
|
|
397
|
+
|
|
398
|
+
objects: dict = {}
|
|
399
|
+
_walk_schema_btree(handle, 1, page_size, objects, set())
|
|
400
|
+
|
|
401
|
+
findings = []
|
|
402
|
+
for name in sorted(objects):
|
|
403
|
+
obj_type, rootpage, sql = objects[name]
|
|
404
|
+
if obj_type not in ("table", "index"):
|
|
405
|
+
continue
|
|
406
|
+
handle.seek((rootpage - 1) * page_size + (100 if rootpage == 1 else 0))
|
|
407
|
+
probe = handle.read(1)
|
|
408
|
+
if len(probe) != 1:
|
|
409
|
+
raise _ScanError(
|
|
410
|
+
f"root page {rootpage} of {obj_type} {name} is absent"
|
|
411
|
+
)
|
|
412
|
+
page_type = probe[0]
|
|
413
|
+
if page_type not in _VALID_PAGE_TYPES:
|
|
414
|
+
findings.append(
|
|
415
|
+
_finding(
|
|
416
|
+
"bad_root_page_type",
|
|
417
|
+
f"root page {rootpage} of {obj_type} {name} has type "
|
|
418
|
+
f"byte 0x{page_type:02x}",
|
|
419
|
+
table=name if obj_type == "table" else None,
|
|
420
|
+
index=name if obj_type == "index" else None,
|
|
421
|
+
page=rootpage,
|
|
422
|
+
)
|
|
423
|
+
)
|
|
424
|
+
continue
|
|
425
|
+
if obj_type == "table" and sql is None:
|
|
426
|
+
# Only a TABLE needs its declaration, and only to rule out
|
|
427
|
+
# WITHOUT ROWID. Gating this on every object type would skip
|
|
428
|
+
# every `sqlite_autoindex_*` entry, whose `sql` is NULL by
|
|
429
|
+
# definition — including the automatic index the recurring
|
|
430
|
+
# production signature names.
|
|
431
|
+
continue
|
|
432
|
+
if obj_type == "index" or _is_without_rowid(sql):
|
|
433
|
+
expected = _INDEX_PAGE_TYPES
|
|
434
|
+
else:
|
|
435
|
+
expected = _TABLE_PAGE_TYPES
|
|
436
|
+
if page_type in expected:
|
|
437
|
+
continue
|
|
438
|
+
observed = "index" if page_type in _INDEX_PAGE_TYPES else "table"
|
|
439
|
+
article = "an" if observed == "index" else "a"
|
|
440
|
+
findings.append(
|
|
441
|
+
_finding(
|
|
442
|
+
"root_page_kind_mismatch",
|
|
443
|
+
f"root page {rootpage} of {obj_type} {name} holds "
|
|
444
|
+
f"{article} {observed} b-tree page "
|
|
445
|
+
f"(type byte 0x{page_type:02x})",
|
|
446
|
+
table=name if obj_type == "table" else None,
|
|
447
|
+
index=name if obj_type == "index" else None,
|
|
448
|
+
page=rootpage,
|
|
449
|
+
)
|
|
450
|
+
)
|
|
451
|
+
return {"method": "raw_scan", "findings": findings, "reason": None}
|
|
452
|
+
except Exception as exc: # noqa: BLE001 — characterization never raises
|
|
453
|
+
return _unavailable(_reason(exc))
|
|
454
|
+
|
|
455
|
+
|
|
456
|
+
# ==========================================================================
|
|
457
|
+
# normalized shape token
|
|
458
|
+
# ==========================================================================
|
|
459
|
+
|
|
460
|
+
def shape_token(findings) -> str:
|
|
461
|
+
"""A short equality-comparable token for one damage class.
|
|
462
|
+
|
|
463
|
+
Derived from the SORTED SET of ``(kind, table, index)`` triples, so page
|
|
464
|
+
numbers, cell indices and rowids — which differ between two instances of
|
|
465
|
+
the same fault — cannot make one recurrence look like two.
|
|
466
|
+
"""
|
|
467
|
+
try:
|
|
468
|
+
triples = sorted(
|
|
469
|
+
{
|
|
470
|
+
(
|
|
471
|
+
str(finding.get("kind") or ""),
|
|
472
|
+
str(finding.get("table") or ""),
|
|
473
|
+
str(finding.get("index") or ""),
|
|
474
|
+
)
|
|
475
|
+
for finding in (findings or ())
|
|
476
|
+
}
|
|
477
|
+
)
|
|
478
|
+
except Exception: # noqa: BLE001 — characterization never raises
|
|
479
|
+
return "none"
|
|
480
|
+
if not triples:
|
|
481
|
+
return "none"
|
|
482
|
+
payload = "|".join(":".join(triple) for triple in triples)
|
|
483
|
+
return hashlib.sha256(payload.encode("utf-8")).hexdigest()[:16]
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
def describe_damage(*, integrity_rows, path) -> dict:
|
|
487
|
+
"""One structured description from whichever sources are available.
|
|
488
|
+
|
|
489
|
+
``method`` is ``integrity_rows`` when only the pragma produced findings,
|
|
490
|
+
``raw_scan`` when only the file scan did, ``both`` when each did, and
|
|
491
|
+
``unavailable`` when neither source could say anything.
|
|
492
|
+
"""
|
|
493
|
+
try:
|
|
494
|
+
parsed = parse_integrity_rows(integrity_rows)
|
|
495
|
+
scan = scan_sqlite_file(path)
|
|
496
|
+
scanned = list(scan.get("findings") or ())
|
|
497
|
+
scan_available = scan.get("method") == "raw_scan"
|
|
498
|
+
|
|
499
|
+
if parsed and scanned:
|
|
500
|
+
method = "both"
|
|
501
|
+
elif parsed:
|
|
502
|
+
method = "integrity_rows"
|
|
503
|
+
elif scan_available:
|
|
504
|
+
method = "raw_scan"
|
|
505
|
+
else:
|
|
506
|
+
method = "unavailable"
|
|
507
|
+
|
|
508
|
+
findings = parsed + scanned
|
|
509
|
+
return {
|
|
510
|
+
"schemaVersion": SCHEMA_VERSION,
|
|
511
|
+
"method": method,
|
|
512
|
+
"findings": findings,
|
|
513
|
+
"shapeToken": shape_token(findings),
|
|
514
|
+
"reason": None if scan_available else scan.get("reason"),
|
|
515
|
+
}
|
|
516
|
+
except Exception as exc: # noqa: BLE001 — characterization never raises
|
|
517
|
+
return {
|
|
518
|
+
"schemaVersion": SCHEMA_VERSION,
|
|
519
|
+
"method": "unavailable",
|
|
520
|
+
"findings": [],
|
|
521
|
+
"shapeToken": "none",
|
|
522
|
+
"reason": _reason(exc),
|
|
523
|
+
}
|