cctally 1.93.0 → 1.94.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +48 -0
- package/bin/_cctally_cache.py +55 -6
- package/bin/_cctally_codex.py +35 -3
- package/bin/_cctally_config.py +108 -2
- package/bin/_cctally_core.py +37 -8
- package/bin/_cctally_dashboard.py +6 -0
- package/bin/_cctally_dashboard_share.py +744 -137
- package/bin/_cctally_dashboard_sources.py +9 -0
- package/bin/_cctally_db.py +315 -40
- package/bin/_cctally_doctor.py +188 -3
- package/bin/_cctally_five_hour.py +39 -12
- package/bin/_cctally_forecast.py +29 -47
- package/bin/_cctally_journal.py +79 -7
- package/bin/_cctally_parser.py +71 -1
- package/bin/_cctally_project.py +0 -2
- package/bin/_cctally_record.py +18 -0
- package/bin/_cctally_reporting.py +0 -9
- package/bin/_cctally_retention.py +2890 -0
- package/bin/_cctally_share.py +97 -92
- package/bin/_cctally_source_analytics.py +44 -16
- package/bin/_cctally_store.py +194 -14
- package/bin/_lib_artifact_retention.py +1645 -0
- package/bin/_lib_display_tz.py +18 -0
- package/bin/_lib_doctor.py +378 -24
- package/bin/_lib_share.py +1359 -183
- package/bin/_lib_share_templates.py +181 -69
- package/bin/_lib_view_models.py +12 -0
- package/bin/cctally +36 -4
- package/dashboard/static/assets/{index-HlIK7k8Q.js → index-CaUQziq_.js} +48 -48
- package/dashboard/static/assets/index-DUQjFSX7.css +1 -0
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +3 -1
- package/dashboard/static/assets/index-DwWJOYxd.css +0 -1
|
@@ -2770,6 +2770,15 @@ def _build_codex_native_weekly_view(
|
|
|
2770
2770
|
total_tokens=sum(row.total_tokens for row in display_rows),
|
|
2771
2771
|
period_start=(periods[0].start_at if periods else None),
|
|
2772
2772
|
period_end=now_utc,
|
|
2773
|
+
# A DISPLAY LABEL, deliberately not resolved here (#503 S2 review
|
|
2774
|
+
# F9). It can be an abbreviation such as `EDT`, which
|
|
2775
|
+
# `period_civil_dates` cannot load — but this value is the
|
|
2776
|
+
# dashboard envelope's, and the envelope's shape is pinned by
|
|
2777
|
+
# oracle tests. Resolution happens at the two SHARE boundaries
|
|
2778
|
+
# that turn it into a `PeriodSpec.display_tz`:
|
|
2779
|
+
# `_cctally_dashboard_share._share_resolved_display_tz` and
|
|
2780
|
+
# `_cctally_codex._build_codex_share_snapshot`, both of which
|
|
2781
|
+
# call `resolve_display_tz_name` unconditionally.
|
|
2773
2782
|
display_tz_label=display_tz_name or str(dt.datetime.now().astimezone().tzinfo),
|
|
2774
2783
|
)
|
|
2775
2784
|
|
package/bin/_cctally_db.py
CHANGED
|
@@ -270,9 +270,15 @@ class StatsDbCorruptError(sqlite3.DatabaseError):
|
|
|
270
270
|
dashboard/TUI background threads, the 5h-anchor fallback) keep treating it
|
|
271
271
|
as a DB failure exactly as before — but command-level handlers that map DB
|
|
272
272
|
errors to OTHER exit codes must re-raise it so the global staged diagnosis
|
|
273
|
-
wins (``cmd_record_credit`` does; its documented DB-error exit is 3).
|
|
274
|
-
|
|
275
|
-
|
|
273
|
+
wins (``cmd_record_credit`` does; its documented DB-error exit is 3).
|
|
274
|
+
|
|
275
|
+
stats.db is never auto-recreated the way cache.db is, and the reason is
|
|
276
|
+
NOT that it is undrivable. With retained journal data it is a disposable
|
|
277
|
+
index and the classifier-gated heal rebuilds it from the journal, losing
|
|
278
|
+
nothing. The refusal exists for the pre-cutover install with no retained
|
|
279
|
+
journal data, whose stats.db may be the only copy of its recorded history:
|
|
280
|
+
an empty recreate there would destroy it silently, so the guided repair
|
|
281
|
+
path runs instead.
|
|
276
282
|
"""
|
|
277
283
|
|
|
278
284
|
|
|
@@ -280,7 +286,7 @@ class StatsPublicationFailedError(StatsDbCorruptError):
|
|
|
280
286
|
"""A replacement stats index was published and then FAILED validation.
|
|
281
287
|
|
|
282
288
|
Distinct from ``StatsDbCorruptError``'s ordinary case because replacement
|
|
283
|
-
already occurred, so the inherited "
|
|
289
|
+
already occurred, so the inherited "never auto-recreated" wording would be
|
|
284
290
|
false. Subclasses it deliberately: every graceful-degrade site and the CLI
|
|
285
291
|
boundary's staged exit 3 keep applying unchanged (#496 S1 F1).
|
|
286
292
|
"""
|
|
@@ -1131,6 +1137,95 @@ def write_corruption_forensics(
|
|
|
1131
1137
|
return result if return_result else out
|
|
1132
1138
|
|
|
1133
1139
|
|
|
1140
|
+
def _binary_version() -> "str | None":
|
|
1141
|
+
"""The running binary's released version, or None when it cannot be read."""
|
|
1142
|
+
try:
|
|
1143
|
+
import _lib_changelog
|
|
1144
|
+
|
|
1145
|
+
value = _lib_changelog._read_latest_changelog_version()
|
|
1146
|
+
except Exception: # pragma: no cover — a missing CHANGELOG is not fatal
|
|
1147
|
+
return None
|
|
1148
|
+
return value[0] if value else None
|
|
1149
|
+
|
|
1150
|
+
|
|
1151
|
+
@dataclass(frozen=True)
|
|
1152
|
+
class QuarantineContext:
|
|
1153
|
+
"""What a direct producer knows about the recovery it is quarantining for.
|
|
1154
|
+
|
|
1155
|
+
#496 S6 §4.2. Supplied when a NEW pending record is created and persisted
|
|
1156
|
+
into it; the three resume call sites read it back from that record and pass
|
|
1157
|
+
none of their own, so a resuming process never invents a trigger it did not
|
|
1158
|
+
observe. An incident finalized without a context stays `schemaVersion: 1`,
|
|
1159
|
+
which the retention planner treats as unclassified and therefore protected
|
|
1160
|
+
— the safe direction.
|
|
1161
|
+
"""
|
|
1162
|
+
|
|
1163
|
+
trigger: str
|
|
1164
|
+
trigger_error: "str | None" = None
|
|
1165
|
+
forensics_path: "str | None" = None
|
|
1166
|
+
binary_version: "str | None" = None
|
|
1167
|
+
|
|
1168
|
+
def to_record(self) -> "dict[str, object]":
|
|
1169
|
+
return {
|
|
1170
|
+
"trigger": self.trigger,
|
|
1171
|
+
"triggerError": self.trigger_error,
|
|
1172
|
+
"forensicsPath": self.forensics_path,
|
|
1173
|
+
"binaryVersion": self.binary_version,
|
|
1174
|
+
}
|
|
1175
|
+
|
|
1176
|
+
|
|
1177
|
+
def quarantine_context(
|
|
1178
|
+
*,
|
|
1179
|
+
trigger: str,
|
|
1180
|
+
trigger_error: object = None,
|
|
1181
|
+
forensics_path: object = None,
|
|
1182
|
+
) -> QuarantineContext:
|
|
1183
|
+
"""Build a `QuarantineContext`, bounding the error text and stamping the
|
|
1184
|
+
binary version. `trigger_error` may be an exception; it is rendered and
|
|
1185
|
+
truncated to `_FORENSICS_EXCEPTION_MESSAGE_MAX` so a pathological SQLite
|
|
1186
|
+
message cannot bloat the incident manifest."""
|
|
1187
|
+
return QuarantineContext(
|
|
1188
|
+
trigger=_bounded_forensics_text(trigger, _FORENSICS_ORIGIN_MAX),
|
|
1189
|
+
trigger_error=(
|
|
1190
|
+
None
|
|
1191
|
+
if trigger_error is None
|
|
1192
|
+
else _bounded_forensics_text(
|
|
1193
|
+
trigger_error, _FORENSICS_EXCEPTION_MESSAGE_MAX
|
|
1194
|
+
)
|
|
1195
|
+
),
|
|
1196
|
+
forensics_path=(
|
|
1197
|
+
None if forensics_path is None else str(forensics_path)
|
|
1198
|
+
),
|
|
1199
|
+
binary_version=_binary_version(),
|
|
1200
|
+
)
|
|
1201
|
+
|
|
1202
|
+
|
|
1203
|
+
def _context_from_record(raw: object) -> "QuarantineContext | None":
|
|
1204
|
+
"""Read a persisted context back, or None when it cannot classify.
|
|
1205
|
+
|
|
1206
|
+
Deliberately lenient about the optional fields and strict about `trigger`:
|
|
1207
|
+
a record whose trigger is missing or not a non-empty string cannot produce
|
|
1208
|
+
a self-classifying manifest, so the incident finalizes as v1 and stays
|
|
1209
|
+
protected rather than failing the resume outright. Refusing to resume over
|
|
1210
|
+
a metadata defect would turn a recoverable quarantine into an outage.
|
|
1211
|
+
"""
|
|
1212
|
+
if not isinstance(raw, dict):
|
|
1213
|
+
return None
|
|
1214
|
+
trigger = raw.get("trigger")
|
|
1215
|
+
if not isinstance(trigger, str) or not trigger:
|
|
1216
|
+
return None
|
|
1217
|
+
|
|
1218
|
+
def _text(value: object) -> "str | None":
|
|
1219
|
+
return value if isinstance(value, str) and value else None
|
|
1220
|
+
|
|
1221
|
+
return QuarantineContext(
|
|
1222
|
+
trigger=trigger,
|
|
1223
|
+
trigger_error=_text(raw.get("triggerError")),
|
|
1224
|
+
forensics_path=_text(raw.get("forensicsPath")),
|
|
1225
|
+
binary_version=_text(raw.get("binaryVersion")),
|
|
1226
|
+
)
|
|
1227
|
+
|
|
1228
|
+
|
|
1134
1229
|
def _quarantine_pending_path(db_path: pathlib.Path) -> pathlib.Path:
|
|
1135
1230
|
return db_path.with_name(f"{db_path.name}.quarantine-pending.json")
|
|
1136
1231
|
|
|
@@ -1200,8 +1295,37 @@ def _load_pending_quarantine(db_path: pathlib.Path) -> "dict[str, Any] | None":
|
|
|
1200
1295
|
return state
|
|
1201
1296
|
|
|
1202
1297
|
|
|
1298
|
+
_QUARANTINE_MISSING_CONTEXT_WARNED = False # one-shot warn flag
|
|
1299
|
+
|
|
1300
|
+
|
|
1301
|
+
def _warn_quarantine_created_without_context(incident: pathlib.Path) -> None:
|
|
1302
|
+
"""Report a creation that supplied no `QuarantineContext` (#496 S6 §4.2).
|
|
1303
|
+
|
|
1304
|
+
The incident is still finalized, as a v1 manifest. Raising instead would
|
|
1305
|
+
turn a metadata defect into a FAILED corruption recovery, which trades a
|
|
1306
|
+
permanently protected incident for lost evidence — the wrong direction. But
|
|
1307
|
+
the two adjacent programming errors (a context on a resume, a context with
|
|
1308
|
+
``strict=False``) both raise, so this one must not be the single path that
|
|
1309
|
+
degrades with nothing said: a v1 incident is unclassifiable and therefore
|
|
1310
|
+
protected forever, and nothing else would ever explain why.
|
|
1311
|
+
"""
|
|
1312
|
+
global _QUARANTINE_MISSING_CONTEXT_WARNED
|
|
1313
|
+
if _QUARANTINE_MISSING_CONTEXT_WARNED:
|
|
1314
|
+
return
|
|
1315
|
+
_QUARANTINE_MISSING_CONTEXT_WARNED = True
|
|
1316
|
+
eprint(
|
|
1317
|
+
f"[quarantine] created {incident} without a QuarantineContext; its "
|
|
1318
|
+
"manifest stays schemaVersion 1, which the retention planner treats "
|
|
1319
|
+
"as unclassified and never reclaims. Every production producer "
|
|
1320
|
+
"supplies one — report this."
|
|
1321
|
+
)
|
|
1322
|
+
|
|
1323
|
+
|
|
1203
1324
|
def _quarantine_db_family_strict(
|
|
1204
|
-
db_path: pathlib.Path,
|
|
1325
|
+
db_path: pathlib.Path,
|
|
1326
|
+
*,
|
|
1327
|
+
ts: "str | None" = None,
|
|
1328
|
+
context: "QuarantineContext | None" = None,
|
|
1205
1329
|
) -> pathlib.Path:
|
|
1206
1330
|
"""Resumably move every snapshotted family member or fail closed.
|
|
1207
1331
|
|
|
@@ -1209,10 +1333,20 @@ def _quarantine_db_family_strict(
|
|
|
1209
1333
|
owner or individual rename failure therefore leaves enough information for
|
|
1210
1334
|
the next maintenance-exclusive opener to complete the exact same incident;
|
|
1211
1335
|
recreation is forbidden until every member is present at its destination.
|
|
1336
|
+
|
|
1337
|
+
``context`` (#496 S6 §4.2) is supplied only when this call CREATES the
|
|
1338
|
+
pending record. It is persisted into that record and read back on every
|
|
1339
|
+
resume, so the manifest describes what the process that observed the
|
|
1340
|
+
corruption knew rather than what a later resumer guessed.
|
|
1212
1341
|
"""
|
|
1213
1342
|
db_path = pathlib.Path(db_path)
|
|
1214
1343
|
pending = _quarantine_pending_path(db_path)
|
|
1215
1344
|
state = _load_pending_quarantine(db_path)
|
|
1345
|
+
if state is not None and context is not None:
|
|
1346
|
+
raise ValueError(
|
|
1347
|
+
"a resume of a pending quarantine must not supply its own "
|
|
1348
|
+
f"context: {pending}"
|
|
1349
|
+
)
|
|
1216
1350
|
if state is None:
|
|
1217
1351
|
ts = ts or _db_backup_timestamp()
|
|
1218
1352
|
root = _cctally_core.APP_DIR / "quarantine"
|
|
@@ -1244,6 +1378,13 @@ def _quarantine_db_family_strict(
|
|
|
1244
1378
|
"members": members,
|
|
1245
1379
|
"createdAtUtc": _forensics_iso(dt.datetime.now(dt.timezone.utc)),
|
|
1246
1380
|
}
|
|
1381
|
+
# Additive under the SAME record schemaVersion: an older binary reading
|
|
1382
|
+
# this record ignores the key and finalizes a v1 manifest rather than
|
|
1383
|
+
# refusing to resume, which a version bump would have caused.
|
|
1384
|
+
if context is not None:
|
|
1385
|
+
state["context"] = context.to_record()
|
|
1386
|
+
else:
|
|
1387
|
+
_warn_quarantine_created_without_context(incident)
|
|
1247
1388
|
_atomic_write_private_json(pending, state)
|
|
1248
1389
|
else:
|
|
1249
1390
|
incident = pathlib.Path(state["incidentPath"])
|
|
@@ -1273,30 +1414,57 @@ def _quarantine_db_family_strict(
|
|
|
1273
1414
|
pass
|
|
1274
1415
|
moved.append(name)
|
|
1275
1416
|
|
|
1276
|
-
manifest = {
|
|
1417
|
+
manifest: "dict[str, Any]" = {
|
|
1277
1418
|
"schemaVersion": 1,
|
|
1278
1419
|
"quarantinedAtUtc": _forensics_iso(dt.datetime.now(dt.timezone.utc)),
|
|
1279
1420
|
"originalPath": str(db_path),
|
|
1280
1421
|
"movedFiles": moved,
|
|
1281
1422
|
"complete": True,
|
|
1282
1423
|
}
|
|
1424
|
+
# #496 S6 §4.2: the additive v2 manifest. Every key above keeps its v1 name
|
|
1425
|
+
# and meaning; the bump only records that the four classification fields
|
|
1426
|
+
# are present. A manifest without a truthy trigger stays v1 and is treated
|
|
1427
|
+
# as unclassified — and therefore protected — by the retention planner.
|
|
1428
|
+
persisted = _context_from_record(state.get("context"))
|
|
1429
|
+
if persisted is not None:
|
|
1430
|
+
manifest.update({
|
|
1431
|
+
"schemaVersion": 2,
|
|
1432
|
+
"trigger": persisted.trigger,
|
|
1433
|
+
"triggerError": persisted.trigger_error,
|
|
1434
|
+
"forensicsPath": persisted.forensics_path,
|
|
1435
|
+
"binaryVersion": persisted.binary_version,
|
|
1436
|
+
})
|
|
1283
1437
|
_atomic_write_private_json(incident / "manifest.json", manifest)
|
|
1438
|
+
_fsync_directory(incident)
|
|
1284
1439
|
pending.unlink()
|
|
1285
1440
|
_fsync_directory(pending.parent)
|
|
1286
1441
|
return incident
|
|
1287
1442
|
|
|
1288
1443
|
|
|
1289
1444
|
def quarantine_db_family(
|
|
1290
|
-
db_path,
|
|
1445
|
+
db_path,
|
|
1446
|
+
*,
|
|
1447
|
+
ts: "str | None" = None,
|
|
1448
|
+
strict: bool = False,
|
|
1449
|
+
context: "QuarantineContext | None" = None,
|
|
1291
1450
|
) -> pathlib.Path:
|
|
1292
1451
|
"""Move a damaged DB + its ``-wal``/``-shm`` sidecars into a single
|
|
1293
1452
|
timestamped incident directory under ``quarantine/`` with a manifest (spec
|
|
1294
1453
|
§6.3). NEVER deletes evidence — three renames under the caller's exclusion
|
|
1295
1454
|
locks, not pretending to be one atomic op. ``0o700`` dir / ``0o600`` files.
|
|
1296
|
-
Returns the incident directory (which may be empty if nothing was present).
|
|
1455
|
+
Returns the incident directory (which may be empty if nothing was present).
|
|
1456
|
+
|
|
1457
|
+
``context`` (#496 S6 §4.2) is required when CREATING a new pending record
|
|
1458
|
+
and must be omitted on a resume, where it is read back from that record.
|
|
1459
|
+
Only the strict path persists it; ``strict=False`` has no production caller.
|
|
1460
|
+
"""
|
|
1297
1461
|
db_path = pathlib.Path(db_path)
|
|
1298
1462
|
if strict:
|
|
1299
|
-
return _quarantine_db_family_strict(db_path, ts=ts)
|
|
1463
|
+
return _quarantine_db_family_strict(db_path, ts=ts, context=context)
|
|
1464
|
+
if context is not None:
|
|
1465
|
+
raise ValueError(
|
|
1466
|
+
"quarantine_db_family(strict=False) does not persist a context"
|
|
1467
|
+
)
|
|
1300
1468
|
ts = ts or _db_backup_timestamp()
|
|
1301
1469
|
root = _cctally_core.APP_DIR / "quarantine"
|
|
1302
1470
|
incident = root / f"{db_path.name}-{ts}"
|
|
@@ -1328,8 +1496,11 @@ def quarantine_db_family(
|
|
|
1328
1496
|
"movedFiles": moved,
|
|
1329
1497
|
}
|
|
1330
1498
|
try:
|
|
1331
|
-
|
|
1332
|
-
|
|
1499
|
+
# The same durable private writer the strict path uses: a bare
|
|
1500
|
+
# write_text left the manifest at the process umask (0644 observed) and
|
|
1501
|
+
# unfsynced, so the incident could survive a crash while the record
|
|
1502
|
+
# describing it did not.
|
|
1503
|
+
_atomic_write_private_json(incident / "manifest.json", manifest)
|
|
1333
1504
|
except OSError:
|
|
1334
1505
|
pass
|
|
1335
1506
|
return incident
|
|
@@ -1419,26 +1590,34 @@ def cmd_db_rebuild(args: argparse.Namespace) -> int:
|
|
|
1419
1590
|
"journal.ingest.lock. Retry shortly."
|
|
1420
1591
|
)
|
|
1421
1592
|
return 3
|
|
1593
|
+
# #496 S6 §5.3: SHARED from before the forensics bundle through the
|
|
1594
|
+
# final rebuild record. `rebuild_stats_index` takes it again for its own
|
|
1595
|
+
# preservation span; the hold is refcounted, so that is a no-op here.
|
|
1596
|
+
import _cctally_retention
|
|
1597
|
+
|
|
1422
1598
|
try:
|
|
1423
|
-
|
|
1424
|
-
|
|
1425
|
-
|
|
1426
|
-
|
|
1427
|
-
|
|
1428
|
-
|
|
1429
|
-
# authorizer-armed `open_db(_target_path=...)` connection, and we
|
|
1430
|
-
# hold maintenance exclusive, which is what spec §3.1 sanctions.
|
|
1431
|
-
with _cctally_store.stats_write_scope("maintenance-rebuild"):
|
|
1432
|
-
result = _cctally_journal.rebuild_stats_index(
|
|
1433
|
-
context=_cctally_journal.RebuildContext(
|
|
1434
|
-
trigger="db-rebuild",
|
|
1435
|
-
trigger_error=None,
|
|
1436
|
-
forensics_path=(
|
|
1437
|
-
str(forensics) if forensics is not None else None
|
|
1438
|
-
),
|
|
1599
|
+
with _cctally_retention.retention_shared(label="db rebuild"):
|
|
1600
|
+
if path.exists():
|
|
1601
|
+
# Forensics FIRST. The common cutover performs the final
|
|
1602
|
+
# drain check only after the scratch index is complete.
|
|
1603
|
+
forensics = write_corruption_forensics(
|
|
1604
|
+
path, db_label="stats"
|
|
1439
1605
|
)
|
|
1440
|
-
|
|
1441
|
-
|
|
1606
|
+
# #386: declare the sanctioned maintenance regime around the
|
|
1607
|
+
# replacement — the rebuild writes its scratch index through an
|
|
1608
|
+
# authorizer-armed `open_db(_target_path=...)` connection, and
|
|
1609
|
+
# we hold maintenance exclusive, which spec §3.1 sanctions.
|
|
1610
|
+
with _cctally_store.stats_write_scope("maintenance-rebuild"):
|
|
1611
|
+
result = _cctally_journal.rebuild_stats_index(
|
|
1612
|
+
context=_cctally_journal.RebuildContext(
|
|
1613
|
+
trigger="db-rebuild",
|
|
1614
|
+
trigger_error=None,
|
|
1615
|
+
forensics_path=(
|
|
1616
|
+
str(forensics) if forensics is not None else None
|
|
1617
|
+
),
|
|
1618
|
+
)
|
|
1619
|
+
)
|
|
1620
|
+
incident = result.quarantine_dir
|
|
1442
1621
|
except Exception as exc:
|
|
1443
1622
|
eprint(f"cctally: stats.db rebuild failed: {exc}")
|
|
1444
1623
|
return 3
|
|
@@ -8951,15 +9130,24 @@ def cmd_db_unskip(args: argparse.Namespace) -> int:
|
|
|
8951
9130
|
|
|
8952
9131
|
|
|
8953
9132
|
def cmd_db_recover(args: argparse.Namespace) -> int:
|
|
8954
|
-
"""Revert a version-ahead
|
|
9133
|
+
"""Revert a version-ahead cache.db to this binary's known schema head (#145).
|
|
8955
9134
|
|
|
8956
9135
|
cache.db is fully re-derivable, so `--db cache` heals without --yes.
|
|
8957
|
-
|
|
8958
|
-
|
|
8959
|
-
|
|
8960
|
-
stats.db
|
|
8961
|
-
|
|
8962
|
-
|
|
9136
|
+
|
|
9137
|
+
**`--db stats` is retired and this function never repairs it.** The body
|
|
9138
|
+
below exits 2 and points at `db rebuild --db stats`, so the two sentences
|
|
9139
|
+
this docstring used to carry — that stats.db holds rows nothing can
|
|
9140
|
+
re-derive, and that `--db stats` requires an explicit --yes and honors the
|
|
9141
|
+
#146 prod guard — described a code path that no longer exists. Both are
|
|
9142
|
+
replaced here rather than corrected, because neither is true of the
|
|
9143
|
+
shipped command: with retained journal data stats.db is a disposable
|
|
9144
|
+
index that a version mismatch self-heals by rebuilding from the journal,
|
|
9145
|
+
and on a pre-cutover install with no retained journal data — whose
|
|
9146
|
+
stats.db may be the only copy of its recorded history — trim-and-revert
|
|
9147
|
+
was never the safe answer either; `db repair --db stats --yes` is.
|
|
9148
|
+
|
|
9149
|
+
Bypasses open_db()/open_cache_db() (raw connect) so it never re-triggers
|
|
9150
|
+
the dispatcher. Idempotent: a no-op when the DB is not ahead.
|
|
8963
9151
|
"""
|
|
8964
9152
|
which = args.db # "cache" | "stats"
|
|
8965
9153
|
if which == "stats":
|
|
@@ -9335,6 +9523,66 @@ def _copy_db_family(
|
|
|
9335
9523
|
_fsync_file(dst)
|
|
9336
9524
|
|
|
9337
9525
|
|
|
9526
|
+
#: The classification sidecar's own schema version, matching the incident
|
|
9527
|
+
#: `classification.json` the correlator writes.
|
|
9528
|
+
_BACKUP_CLASSIFICATION_SCHEMA_VERSION = 1
|
|
9529
|
+
|
|
9530
|
+
|
|
9531
|
+
def _backup_classification_path(stem: pathlib.Path) -> pathlib.Path:
|
|
9532
|
+
return stem.with_name(f"{stem.name}.classification.json")
|
|
9533
|
+
|
|
9534
|
+
|
|
9535
|
+
def _write_backup_classification_sidecar(
|
|
9536
|
+
stem: pathlib.Path,
|
|
9537
|
+
*,
|
|
9538
|
+
trigger: str = "db-repair",
|
|
9539
|
+
method: str = "db-repair",
|
|
9540
|
+
forensics_path: "str | None" = None,
|
|
9541
|
+
) -> pathlib.Path:
|
|
9542
|
+
"""Classify a machine-written backup family for retention (#496 S6 §3.7).
|
|
9543
|
+
|
|
9544
|
+
A `.bak-*` family has no incident manifest, so §3.3 classifies it by this
|
|
9545
|
+
sidecar instead — and only while the member identities it records still
|
|
9546
|
+
match the family on disk. Recording device and inode is what stops a
|
|
9547
|
+
REPLACEMENT family written at the same machine-shaped stem from inheriting
|
|
9548
|
+
this `exact` verdict and becoming deletable without ever having been
|
|
9549
|
+
classified.
|
|
9550
|
+
|
|
9551
|
+
The caller writes this only AFTER the copied family and its directory are
|
|
9552
|
+
durable, and before it releases the shared retention lock. A crash in that
|
|
9553
|
+
window leaves the backup unclassified and therefore protected, which is the
|
|
9554
|
+
safe direction.
|
|
9555
|
+
"""
|
|
9556
|
+
members = []
|
|
9557
|
+
for suffix in ("", "-wal", "-shm"):
|
|
9558
|
+
member = pathlib.Path(str(stem) + suffix)
|
|
9559
|
+
try:
|
|
9560
|
+
info = member.stat()
|
|
9561
|
+
except OSError:
|
|
9562
|
+
continue
|
|
9563
|
+
members.append({
|
|
9564
|
+
"name": member.name,
|
|
9565
|
+
"size": int(info.st_size),
|
|
9566
|
+
"mtime": float(info.st_mtime),
|
|
9567
|
+
"device": int(info.st_dev),
|
|
9568
|
+
"inode": int(info.st_ino),
|
|
9569
|
+
})
|
|
9570
|
+
payload = {
|
|
9571
|
+
"schemaVersion": _BACKUP_CLASSIFICATION_SCHEMA_VERSION,
|
|
9572
|
+
"method": method,
|
|
9573
|
+
"confidence": "exact",
|
|
9574
|
+
"trigger": trigger,
|
|
9575
|
+
"binaryVersion": _binary_version(),
|
|
9576
|
+
"backupStem": stem.name,
|
|
9577
|
+
"members": members,
|
|
9578
|
+
"forensicsPath": forensics_path,
|
|
9579
|
+
"classifiedAtUtc": _forensics_iso(dt.datetime.now(dt.timezone.utc)),
|
|
9580
|
+
}
|
|
9581
|
+
path = _backup_classification_path(stem)
|
|
9582
|
+
_atomic_write_private_json(path, payload)
|
|
9583
|
+
return path
|
|
9584
|
+
|
|
9585
|
+
|
|
9338
9586
|
def _read_user_version_header(path: pathlib.Path) -> "int | None":
|
|
9339
9587
|
"""Read SQLite's big-endian user_version field without opening pages."""
|
|
9340
9588
|
try:
|
|
@@ -9460,6 +9708,13 @@ def _repair_preflight_and_copy(
|
|
|
9460
9708
|
# repair marker, and the caller has already proved no old handle exists.
|
|
9461
9709
|
_copy_db_family(path, backup)
|
|
9462
9710
|
_fsync_directory(path.parent)
|
|
9711
|
+
# #496 S6 §3.7. Written here, immediately after the family and its
|
|
9712
|
+
# directory are durable, so EVERY backup this function creates is
|
|
9713
|
+
# classified — including one left behind by a repair that then fails
|
|
9714
|
+
# its WAL checkpoint. `_atomic_write_private_json` fsyncs the file and
|
|
9715
|
+
# the directory entry, and the caller still holds the shared retention
|
|
9716
|
+
# lock (§5.3), so the sidecar is durable before reclamation can see it.
|
|
9717
|
+
_write_backup_classification_sidecar(backup)
|
|
9463
9718
|
conn.rollback()
|
|
9464
9719
|
|
|
9465
9720
|
try:
|
|
@@ -9507,9 +9762,13 @@ def cmd_db_repair(args: argparse.Namespace) -> int:
|
|
|
9507
9762
|
return 0
|
|
9508
9763
|
if not getattr(args, "yes", False):
|
|
9509
9764
|
eprint(
|
|
9510
|
-
"cctally: repairing stats.db replaces the live
|
|
9511
|
-
"
|
|
9512
|
-
"
|
|
9765
|
+
"cctally: repairing stats.db replaces the live database after "
|
|
9766
|
+
"preserving the corrupt original. This command is for a "
|
|
9767
|
+
"pre-cutover install with no retained journal data, whose stats.db "
|
|
9768
|
+
"may be the only copy of its recorded history; with retained "
|
|
9769
|
+
"journal data stats.db is a disposable index and `cctally db "
|
|
9770
|
+
"rebuild --db stats` is the command to use. Re-run with --yes "
|
|
9771
|
+
"after stopping the dashboard and other cctally processes."
|
|
9513
9772
|
)
|
|
9514
9773
|
return 2
|
|
9515
9774
|
|
|
@@ -9599,8 +9858,24 @@ def _cmd_db_repair_locked(args: argparse.Namespace, path: pathlib.Path) -> int:
|
|
|
9599
9858
|
|
|
9600
9859
|
|
|
9601
9860
|
def _cmd_db_repair_claimed(args: argparse.Namespace, path: pathlib.Path) -> int:
|
|
9602
|
-
"""Repair body; caller owns the marker and has proved no old handles.
|
|
9861
|
+
"""Repair body; caller owns the marker and has proved no old handles.
|
|
9862
|
+
|
|
9863
|
+
#496 S6 §5.3: the shared retention hold is taken HERE — after the
|
|
9864
|
+
maintenance flock its caller holds, and above `_repair_preflight_and_copy`,
|
|
9865
|
+
whose `_copy_db_family` runs inside a `BEGIN IMMEDIATE`. Acquiring around
|
|
9866
|
+
the copy instead would put a lock acquisition inside a SQLite transaction
|
|
9867
|
+
and invert the lock order. It is released after the classification sidecar
|
|
9868
|
+
for the backup family is durable.
|
|
9869
|
+
"""
|
|
9870
|
+
import _cctally_retention
|
|
9871
|
+
|
|
9872
|
+
with _cctally_retention.retention_shared(label="db repair"):
|
|
9873
|
+
return _cmd_db_repair_under_retention(args, path)
|
|
9874
|
+
|
|
9603
9875
|
|
|
9876
|
+
def _cmd_db_repair_under_retention(
|
|
9877
|
+
args: argparse.Namespace, path: pathlib.Path,
|
|
9878
|
+
) -> int:
|
|
9604
9879
|
timeout_ms = int(getattr(args, "busy_timeout_ms", 250) or 250)
|
|
9605
9880
|
sqlite_binary = (
|
|
9606
9881
|
getattr(args, "sqlite3_binary", None) or shutil.which("sqlite3")
|