cctally 1.82.1 → 1.83.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +58 -0
- package/README.md +12 -5
- package/bin/_cctally_alerts.py +8 -1
- package/bin/_cctally_cache.py +912 -149
- package/bin/_cctally_config.py +43 -4
- package/bin/_cctally_core.py +933 -759
- package/bin/_cctally_dashboard.py +157 -47
- package/bin/_cctally_dashboard_cache_report.py +13 -6
- package/bin/_cctally_dashboard_conversation.py +1 -0
- package/bin/_cctally_dashboard_envelope.py +116 -8
- package/bin/_cctally_dashboard_share.py +50 -19
- package/bin/_cctally_dashboard_sources.py +223 -48
- package/bin/_cctally_db.py +605 -128
- package/bin/_cctally_doctor.py +413 -28
- package/bin/_cctally_five_hour.py +12 -5
- package/bin/_cctally_journal.py +2050 -156
- package/bin/_cctally_journal_repair.py +519 -0
- package/bin/_cctally_milestone_history.py +142 -56
- package/bin/_cctally_milestones.py +179 -111
- package/bin/_cctally_parser.py +42 -0
- package/bin/_cctally_project.py +24 -18
- package/bin/_cctally_quota.py +139 -25
- package/bin/_cctally_record.py +279 -108
- package/bin/_cctally_rederive.py +1052 -0
- package/bin/_cctally_reporting.py +58 -53
- package/bin/_cctally_setup.py +1 -0
- package/bin/_cctally_source_analytics.py +4 -1
- package/bin/_cctally_statusline.py +11 -11
- package/bin/_cctally_store.py +1039 -31
- package/bin/_cctally_sync_week.py +17 -8
- package/bin/_cctally_tui.py +350 -44
- package/bin/_cctally_update.py +133 -8
- package/bin/_cctally_weekrefs.py +14 -0
- package/bin/_lib_aggregators.py +10 -6
- package/bin/_lib_cache_report.py +101 -9
- package/bin/_lib_codex_pools.py +82 -0
- package/bin/_lib_conversation_query.py +81 -33
- package/bin/_lib_dashboard_sources.py +75 -0
- package/bin/_lib_diff_kernel.py +28 -15
- package/bin/_lib_doctor.py +342 -4
- package/bin/_lib_journal.py +924 -2
- package/bin/_lib_jsonl.py +43 -14
- package/bin/_lib_pricing.py +140 -21
- package/bin/_lib_rederive.py +395 -0
- package/bin/_lib_share.py +58 -2
- package/bin/cctally +56 -8
- package/dashboard/static/assets/{index-BKM43pxK.js → index-3bgCMVHb.js} +52 -52
- package/dashboard/static/assets/index-D27EIHEI.css +1 -0
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +5 -1
- package/dashboard/static/assets/index-Dk1nplOz.css +0 -1
package/bin/_cctally_journal.py
CHANGED
|
@@ -27,9 +27,11 @@ from __future__ import annotations
|
|
|
27
27
|
|
|
28
28
|
import datetime as dt
|
|
29
29
|
import fcntl
|
|
30
|
+
import hashlib
|
|
30
31
|
import json
|
|
31
32
|
import os
|
|
32
33
|
import pathlib
|
|
34
|
+
import signal
|
|
33
35
|
import sqlite3
|
|
34
36
|
import sys
|
|
35
37
|
import time
|
|
@@ -55,12 +57,43 @@ _QUOTA_DEDUP_INDEX_NAME = ".quota-observation-keys"
|
|
|
55
57
|
_QUOTA_DEDUP_DIR: str | None = None
|
|
56
58
|
_QUOTA_DEDUP_KEYS: set[str] = set()
|
|
57
59
|
_QUOTA_DEDUP_LOADED = False
|
|
60
|
+
_HIGH_WATER_UNSET = object()
|
|
58
61
|
|
|
59
62
|
|
|
60
63
|
class JournalError(Exception):
|
|
61
64
|
"""A structural journal-append failure (line too long, unrepairable tail)."""
|
|
62
65
|
|
|
63
66
|
|
|
67
|
+
class CorrectionRebuildRequired(JournalError):
|
|
68
|
+
"""A completed correction cannot be applied incrementally to a live index.
|
|
69
|
+
|
|
70
|
+
The recovery boundary needs more than a message: it must rebuild through
|
|
71
|
+
the exact completed-batch commit that triggered the mismatch, then
|
|
72
|
+
revalidate the exact effective metadata under exclusive ownership.
|
|
73
|
+
"""
|
|
74
|
+
|
|
75
|
+
def __init__(
|
|
76
|
+
self,
|
|
77
|
+
message,
|
|
78
|
+
*,
|
|
79
|
+
batch_id=None,
|
|
80
|
+
event_id=None,
|
|
81
|
+
high_water=None,
|
|
82
|
+
expected_metadata=None,
|
|
83
|
+
recovery_eligible=False,
|
|
84
|
+
):
|
|
85
|
+
super().__init__(message)
|
|
86
|
+
self.batch_id = batch_id
|
|
87
|
+
self.event_id = event_id
|
|
88
|
+
self.high_water = high_water
|
|
89
|
+
self.expected_metadata = expected_metadata
|
|
90
|
+
self.recovery_eligible = recovery_eligible
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
class CorrectionRecoveryError(JournalError):
|
|
94
|
+
"""Bounded correction recovery could not safely replace the live index."""
|
|
95
|
+
|
|
96
|
+
|
|
64
97
|
# --------------------------------------------------------------------------
|
|
65
98
|
# leaf lock
|
|
66
99
|
# --------------------------------------------------------------------------
|
|
@@ -338,6 +371,92 @@ def append_record(
|
|
|
338
371
|
_release_leaf_lock(lock_fd)
|
|
339
372
|
|
|
340
373
|
|
|
374
|
+
def append_records(
|
|
375
|
+
records: list[dict],
|
|
376
|
+
*,
|
|
377
|
+
now_utc: dt.datetime | None = None,
|
|
378
|
+
expected_high_water=_HIGH_WATER_UNSET,
|
|
379
|
+
line_hook=None,
|
|
380
|
+
) -> tuple[str, int]:
|
|
381
|
+
"""Append one ordered record group under a single leaf-lock hold.
|
|
382
|
+
|
|
383
|
+
The group is not transactionally atomic across a power loss: a crash can
|
|
384
|
+
leave complete prefix lines plus one torn final line, exactly like the
|
|
385
|
+
single-record appender. It *is* non-interleavable with other appenders, so a
|
|
386
|
+
correction batch remains physically ordered. ``expected_high_water`` is
|
|
387
|
+
checked while holding the same leaf lock that performs the append, closing
|
|
388
|
+
the plan/revalidate/append race.
|
|
389
|
+
"""
|
|
390
|
+
if not isinstance(records, list) or not records:
|
|
391
|
+
raise ValueError("journal record group must be a non-empty list")
|
|
392
|
+
if now_utc is None:
|
|
393
|
+
now_utc = dt.datetime.now(dt.timezone.utc)
|
|
394
|
+
encoded = []
|
|
395
|
+
for record in records:
|
|
396
|
+
data = _lib_journal.encode_line(record)
|
|
397
|
+
if len(data) > _MAX_LINE_BYTES:
|
|
398
|
+
raise JournalError(
|
|
399
|
+
f"journal line is {len(data)} bytes, exceeds the "
|
|
400
|
+
f"{_MAX_LINE_BYTES}-byte limit (spec §4.3)"
|
|
401
|
+
)
|
|
402
|
+
encoded.append(data)
|
|
403
|
+
|
|
404
|
+
journal_dir = _cctally_core.JOURNAL_DIR
|
|
405
|
+
seg_name = _lib_journal.segment_name(now_utc)
|
|
406
|
+
seg_path = journal_dir / seg_name
|
|
407
|
+
dir_created = not journal_dir.exists()
|
|
408
|
+
journal_dir.mkdir(parents=True, exist_ok=True)
|
|
409
|
+
if dir_created:
|
|
410
|
+
try:
|
|
411
|
+
os.chmod(journal_dir, 0o700)
|
|
412
|
+
except OSError:
|
|
413
|
+
pass
|
|
414
|
+
|
|
415
|
+
lock_fd = _acquire_leaf_lock()
|
|
416
|
+
try:
|
|
417
|
+
segments = list_segments()
|
|
418
|
+
actual_high_water = None
|
|
419
|
+
if segments:
|
|
420
|
+
latest = segments[-1]
|
|
421
|
+
actual_high_water = (
|
|
422
|
+
latest,
|
|
423
|
+
os.stat(journal_dir / latest).st_size,
|
|
424
|
+
)
|
|
425
|
+
if (
|
|
426
|
+
expected_high_water is not _HIGH_WATER_UNSET
|
|
427
|
+
and actual_high_water != expected_high_water
|
|
428
|
+
):
|
|
429
|
+
raise JournalError(
|
|
430
|
+
"journal high-water changed before correction append "
|
|
431
|
+
f"(expected {expected_high_water!r}, found {actual_high_water!r})"
|
|
432
|
+
)
|
|
433
|
+
|
|
434
|
+
seg_created = not seg_path.exists()
|
|
435
|
+
fd = os.open(str(seg_path), os.O_RDWR | os.O_APPEND | os.O_CREAT, 0o600)
|
|
436
|
+
try:
|
|
437
|
+
if seg_created:
|
|
438
|
+
try:
|
|
439
|
+
os.fchmod(fd, 0o600)
|
|
440
|
+
except OSError:
|
|
441
|
+
pass
|
|
442
|
+
_repair_torn_tail(fd)
|
|
443
|
+
for index, data in enumerate(encoded, start=1):
|
|
444
|
+
_write_all(fd, data)
|
|
445
|
+
if line_hook is not None:
|
|
446
|
+
line_hook(index)
|
|
447
|
+
os.fsync(fd)
|
|
448
|
+
end_offset = os.fstat(fd).st_size
|
|
449
|
+
finally:
|
|
450
|
+
os.close(fd)
|
|
451
|
+
if seg_created:
|
|
452
|
+
_fsync_dir(journal_dir)
|
|
453
|
+
if dir_created:
|
|
454
|
+
_fsync_dir(journal_dir.parent)
|
|
455
|
+
return (seg_name, end_offset)
|
|
456
|
+
finally:
|
|
457
|
+
_release_leaf_lock(lock_fd)
|
|
458
|
+
|
|
459
|
+
|
|
341
460
|
def list_segments() -> list[str]:
|
|
342
461
|
"""Journal segment basenames in canonical order (spec §4.1): bootstrap
|
|
343
462
|
segments first, then observation segments, each class lexicographic.
|
|
@@ -358,25 +477,47 @@ def list_segments() -> list[str]:
|
|
|
358
477
|
return sorted(names, key=_lib_journal.segment_sort_key)
|
|
359
478
|
|
|
360
479
|
|
|
361
|
-
def
|
|
362
|
-
"""
|
|
480
|
+
def _has_retained_journal_bytes(segment_sizes) -> bool:
|
|
481
|
+
"""Whether any canonical journal segment retains replayable bytes."""
|
|
482
|
+
return any(int(size) > 0 for size in segment_sizes)
|
|
363
483
|
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
484
|
+
|
|
485
|
+
def _journal_rebuild_snapshot() -> tuple[tuple[str, int] | None, bool]:
|
|
486
|
+
"""Snapshot the canonical high-water and retained-byte truth together.
|
|
487
|
+
|
|
488
|
+
A freshly created newest segment can legitimately be empty while older
|
|
489
|
+
immutable segments still contain the durable rebuild source. Holding the
|
|
490
|
+
leaf lock across both facts keeps the epoch resolver from making its
|
|
491
|
+
fail-closed decision against two different journal states.
|
|
492
|
+
"""
|
|
368
493
|
lock_fd = _acquire_leaf_lock()
|
|
369
494
|
try:
|
|
370
495
|
segments = list_segments()
|
|
371
496
|
if not segments:
|
|
372
|
-
return None
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
497
|
+
return None, False
|
|
498
|
+
sizes = [
|
|
499
|
+
os.stat(_cctally_core.JOURNAL_DIR / segment).st_size
|
|
500
|
+
for segment in segments
|
|
501
|
+
]
|
|
502
|
+
return (
|
|
503
|
+
(segments[-1], sizes[-1]),
|
|
504
|
+
_has_retained_journal_bytes(sizes),
|
|
505
|
+
)
|
|
376
506
|
finally:
|
|
377
507
|
_release_leaf_lock(lock_fd)
|
|
378
508
|
|
|
379
509
|
|
|
510
|
+
def journal_high_water() -> tuple[str, int] | None:
|
|
511
|
+
"""Snapshot ``(latest segment basename, size)`` under a µs leaf-lock hold.
|
|
512
|
+
|
|
513
|
+
"Latest" is the canonically-last segment (spec §4.1 order). The ingest
|
|
514
|
+
cycle takes this snapshot and consumes only ``cursor → HW`` so a line
|
|
515
|
+
appended after the snapshot belongs to the next cycle (spec §5.2.1).
|
|
516
|
+
Returns ``None`` when no segment exists yet."""
|
|
517
|
+
high_water, _has_bytes = _journal_rebuild_snapshot()
|
|
518
|
+
return high_water
|
|
519
|
+
|
|
520
|
+
|
|
380
521
|
# ==========================================================================
|
|
381
522
|
# Single-flight ingest cycle (spec §5.1 / §5.2, revision 3)
|
|
382
523
|
# ==========================================================================
|
|
@@ -449,6 +590,9 @@ class IngestResult:
|
|
|
449
590
|
malformed: int # lines in range that failed to decode (spec §4.4)
|
|
450
591
|
events_emitted: int # evt lines emitted this cycle (Model-A + harvest)
|
|
451
592
|
alerts: list # alert payloads dispatched post-commit (step 6)
|
|
593
|
+
# #374: same-revision divergences handled this cycle — emissions withheld at
|
|
594
|
+
# the write boundary plus journal evts the preflight reader quarantined.
|
|
595
|
+
conflicts_dropped: int = 0
|
|
452
596
|
# Exception discipline (6b-gate P2): the exception that aborted the cycle on
|
|
453
597
|
# an OPPORTUNISTIC ingest — the txn rolled back, the cursor did NOT advance
|
|
454
598
|
# (invariant ii), and `run_stats_ingest` logged it loudly and returned
|
|
@@ -490,11 +634,38 @@ class IngestContext:
|
|
|
490
634
|
# (reset INSERT OR IGNORE rowcount == 1), so a crash-replayed reset never
|
|
491
635
|
# re-suppresses with a divergent list.
|
|
492
636
|
suppression_map: dict = field(default_factory=dict)
|
|
637
|
+
# Task B rederive seam. Normal ingest leaves both defaults unchanged.
|
|
638
|
+
# A scratch planner supplies an in-memory sink so derived events are captured
|
|
639
|
+
# instead of appended to the durable journal, and disables projection-file
|
|
640
|
+
# writes while still exercising the same SQLite derivation/fold code.
|
|
641
|
+
event_sink: "list | None" = None
|
|
642
|
+
projection_writes: bool = True
|
|
643
|
+
# Scratch replay reconstructs ephemeral marker state in memory. A planner
|
|
644
|
+
# shares this dict across its per-record contexts.
|
|
645
|
+
projection_state: dict = field(default_factory=dict)
|
|
646
|
+
# #374 write boundary: emissions WITHHELD this cycle because they would have
|
|
647
|
+
# violated the same-revision rule. Each entry is a `DroppedConflict`; the row
|
|
648
|
+
# was converged from the already-journaled effective event instead.
|
|
649
|
+
conflicts_dropped: list = field(default_factory=list)
|
|
493
650
|
|
|
494
651
|
def as_of_for(self, record: dict) -> str:
|
|
495
652
|
return record["at"]
|
|
496
653
|
|
|
497
654
|
|
|
655
|
+
@dataclass(frozen=True)
|
|
656
|
+
class DroppedConflict:
|
|
657
|
+
"""One live emission withheld at the write boundary (#374 §6).
|
|
658
|
+
|
|
659
|
+
The journal is append-only, so a divergent line can never be un-written —
|
|
660
|
+
the only durable defence is to never append it. The row is converged from
|
|
661
|
+
the effective event instead, and the rejected content is reported here.
|
|
662
|
+
"""
|
|
663
|
+
|
|
664
|
+
event_id: str
|
|
665
|
+
rev: int
|
|
666
|
+
rejected_hash: str
|
|
667
|
+
|
|
668
|
+
|
|
498
669
|
@dataclass(frozen=True)
|
|
499
670
|
class _EvtSpec:
|
|
500
671
|
"""How to fold one evt `kind` into its target table (step 4a replay + the
|
|
@@ -615,27 +786,161 @@ def _release_ingest_lock(fd: int) -> None:
|
|
|
615
786
|
os.close(fd)
|
|
616
787
|
|
|
617
788
|
|
|
789
|
+
def _acquire_maintenance_shared(mode: str, timeout_s: float) -> int | None:
|
|
790
|
+
"""Acquire the stats maintenance lock shared, before the ingest lock."""
|
|
791
|
+
_cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
|
|
792
|
+
fd = os.open(
|
|
793
|
+
str(_cctally_core.STATS_LOCK_MAINTENANCE_PATH),
|
|
794
|
+
os.O_RDWR | os.O_CREAT,
|
|
795
|
+
0o600,
|
|
796
|
+
)
|
|
797
|
+
if mode == "opportunistic":
|
|
798
|
+
try:
|
|
799
|
+
fcntl.flock(fd, fcntl.LOCK_SH | fcntl.LOCK_NB)
|
|
800
|
+
_cctally_core.note_stats_maintenance_acquired()
|
|
801
|
+
return fd
|
|
802
|
+
except (BlockingIOError, OSError):
|
|
803
|
+
os.close(fd)
|
|
804
|
+
return None
|
|
805
|
+
deadline = time.monotonic() + timeout_s
|
|
806
|
+
try:
|
|
807
|
+
while True:
|
|
808
|
+
try:
|
|
809
|
+
fcntl.flock(fd, fcntl.LOCK_SH | fcntl.LOCK_NB)
|
|
810
|
+
_cctally_core.note_stats_maintenance_acquired()
|
|
811
|
+
return fd
|
|
812
|
+
except (BlockingIOError, OSError):
|
|
813
|
+
if time.monotonic() >= deadline:
|
|
814
|
+
os.close(fd)
|
|
815
|
+
return None
|
|
816
|
+
time.sleep(0.02)
|
|
817
|
+
except BaseException:
|
|
818
|
+
os.close(fd)
|
|
819
|
+
raise
|
|
820
|
+
|
|
821
|
+
|
|
822
|
+
def _acquire_maintenance_exclusive(mode: str, timeout_s: float) -> int | None:
|
|
823
|
+
"""Acquire the stats maintenance lock exclusively for legacy cutover."""
|
|
824
|
+
_cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
|
|
825
|
+
fd = os.open(
|
|
826
|
+
str(_cctally_core.STATS_LOCK_MAINTENANCE_PATH),
|
|
827
|
+
os.O_RDWR | os.O_CREAT,
|
|
828
|
+
0o600,
|
|
829
|
+
)
|
|
830
|
+
if mode == "opportunistic":
|
|
831
|
+
try:
|
|
832
|
+
fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
833
|
+
_cctally_core.note_stats_maintenance_acquired()
|
|
834
|
+
return fd
|
|
835
|
+
except (BlockingIOError, OSError):
|
|
836
|
+
os.close(fd)
|
|
837
|
+
return None
|
|
838
|
+
deadline = time.monotonic() + timeout_s
|
|
839
|
+
try:
|
|
840
|
+
while True:
|
|
841
|
+
try:
|
|
842
|
+
fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
843
|
+
_cctally_core.note_stats_maintenance_acquired()
|
|
844
|
+
return fd
|
|
845
|
+
except (BlockingIOError, OSError):
|
|
846
|
+
if time.monotonic() >= deadline:
|
|
847
|
+
os.close(fd)
|
|
848
|
+
return None
|
|
849
|
+
time.sleep(0.02)
|
|
850
|
+
except BaseException:
|
|
851
|
+
os.close(fd)
|
|
852
|
+
raise
|
|
853
|
+
|
|
854
|
+
|
|
855
|
+
def _release_maintenance_shared(fd: int) -> None:
|
|
856
|
+
try:
|
|
857
|
+
fcntl.flock(fd, fcntl.LOCK_UN)
|
|
858
|
+
finally:
|
|
859
|
+
# #386: paired with the note in both acquire helpers. `open_db()` takes
|
|
860
|
+
# this same lock SHARED on a fresh fd, and flock conflicts are
|
|
861
|
+
# process-wide across descriptions, so the legacy/fresh ingest branch
|
|
862
|
+
# (which holds it EXCLUSIVE across its `open_db()`) would self-deadlock
|
|
863
|
+
# without the re-entrancy signal.
|
|
864
|
+
_cctally_core.note_stats_maintenance_released()
|
|
865
|
+
os.close(fd)
|
|
866
|
+
|
|
867
|
+
|
|
868
|
+
def _downgrade_maintenance_shared(fd: int) -> None:
|
|
869
|
+
"""Atomically downgrade a held maintenance lock from EX to SH."""
|
|
870
|
+
fcntl.flock(fd, fcntl.LOCK_SH)
|
|
871
|
+
|
|
872
|
+
|
|
873
|
+
def _stats_db_identity():
|
|
874
|
+
"""Return the current stats main-file identity, or ``None`` if absent."""
|
|
875
|
+
try:
|
|
876
|
+
stat = os.stat(_cctally_core.DB_PATH)
|
|
877
|
+
except OSError:
|
|
878
|
+
return None
|
|
879
|
+
return (stat.st_dev, stat.st_ino)
|
|
880
|
+
|
|
881
|
+
|
|
882
|
+
def _stats_db_user_version() -> int | None:
|
|
883
|
+
"""Read the main file's raw epoch without invoking schema or heal paths."""
|
|
884
|
+
try:
|
|
885
|
+
conn = sqlite3.connect(
|
|
886
|
+
f"file:{_cctally_core.DB_PATH}?mode=ro",
|
|
887
|
+
uri=True,
|
|
888
|
+
)
|
|
889
|
+
except sqlite3.Error:
|
|
890
|
+
return None
|
|
891
|
+
try:
|
|
892
|
+
return int(conn.execute("PRAGMA user_version").fetchone()[0])
|
|
893
|
+
except sqlite3.Error:
|
|
894
|
+
return None
|
|
895
|
+
finally:
|
|
896
|
+
conn.close()
|
|
897
|
+
|
|
898
|
+
|
|
618
899
|
# --------------------------------------------------------------------------
|
|
619
900
|
# cursor (spec §5.2.2: segment-aware, prior-month tails covered)
|
|
620
901
|
# --------------------------------------------------------------------------
|
|
621
902
|
|
|
622
903
|
def _read_cursor(conn: sqlite3.Connection) -> tuple[str, int] | None:
|
|
623
904
|
"""Return `(segment_basename, offset)` from `journal_cursor`, or None when
|
|
624
|
-
nothing has been consumed yet (start of the first segment).
|
|
905
|
+
nothing has been consumed yet (start of the first segment).
|
|
906
|
+
|
|
907
|
+
``applied_segment`` / ``applied_offset`` are the trusted duplicate written
|
|
908
|
+
in the same stats transaction as every materialized row (#410 Task B). A
|
|
909
|
+
cursor-only hand edit can therefore no longer skip durable events and make
|
|
910
|
+
their natural keys appear new: on disagreement, resume from the last
|
|
911
|
+
atomically applied prefix and let the normal replay heal both pairs."""
|
|
625
912
|
row = conn.execute(
|
|
626
|
-
"SELECT segment, offset
|
|
913
|
+
"SELECT segment, offset, applied_segment, applied_offset "
|
|
914
|
+
"FROM journal_cursor WHERE id = 1"
|
|
627
915
|
).fetchone()
|
|
628
916
|
if row is None:
|
|
629
917
|
return None
|
|
630
|
-
|
|
918
|
+
public = (str(row[0]), int(row[1]))
|
|
919
|
+
if row[2] is None or row[3] is None:
|
|
920
|
+
raise JournalError(
|
|
921
|
+
"journal cursor applied-prefix guard is incomplete; "
|
|
922
|
+
"run cctally db rebuild --db stats"
|
|
923
|
+
)
|
|
924
|
+
applied = (str(row[2]), int(row[3]))
|
|
925
|
+
if public != applied:
|
|
926
|
+
print(
|
|
927
|
+
f"[journal] cursor-only advancement detected: public={public!r}, "
|
|
928
|
+
f"applied={applied!r}; replaying from the applied prefix",
|
|
929
|
+
file=sys.stderr,
|
|
930
|
+
)
|
|
931
|
+
return applied
|
|
631
932
|
|
|
632
933
|
|
|
633
934
|
def _write_cursor(conn: sqlite3.Connection, segment: str, offset: int) -> None:
|
|
634
935
|
conn.execute(
|
|
635
|
-
"INSERT INTO journal_cursor
|
|
936
|
+
"INSERT INTO journal_cursor "
|
|
937
|
+
"(id, segment, offset, applied_segment, applied_offset) "
|
|
938
|
+
"VALUES (1, ?, ?, ?, ?) "
|
|
636
939
|
"ON CONFLICT(id) DO UPDATE SET segment = excluded.segment, "
|
|
637
|
-
"offset = excluded.offset"
|
|
638
|
-
|
|
940
|
+
"offset = excluded.offset, "
|
|
941
|
+
"applied_segment = excluded.applied_segment, "
|
|
942
|
+
"applied_offset = excluded.applied_offset",
|
|
943
|
+
(segment, offset, segment, offset),
|
|
639
944
|
)
|
|
640
945
|
|
|
641
946
|
|
|
@@ -690,6 +995,49 @@ def _read_range(cursor, hw) -> list[tuple[str, int, bytes]]:
|
|
|
690
995
|
return lines
|
|
691
996
|
|
|
692
997
|
|
|
998
|
+
def journal_prefix_hash(high_water) -> "str | None":
|
|
999
|
+
"""Hash exact raw segment bytes through one canonical high-water."""
|
|
1000
|
+
if high_water is None:
|
|
1001
|
+
return None
|
|
1002
|
+
digest = hashlib.sha256()
|
|
1003
|
+
found = False
|
|
1004
|
+
for segment in list_segments():
|
|
1005
|
+
path = _cctally_core.JOURNAL_DIR / segment
|
|
1006
|
+
size = high_water[1] if segment == high_water[0] else path.stat().st_size
|
|
1007
|
+
data = path.read_bytes()[:size]
|
|
1008
|
+
if len(data) != size:
|
|
1009
|
+
raise OSError(f"journal segment changed while reading: {segment}")
|
|
1010
|
+
name = segment.encode("utf-8")
|
|
1011
|
+
digest.update(len(name).to_bytes(4, "big"))
|
|
1012
|
+
digest.update(name)
|
|
1013
|
+
digest.update(len(data).to_bytes(8, "big"))
|
|
1014
|
+
digest.update(data)
|
|
1015
|
+
if segment == high_water[0]:
|
|
1016
|
+
found = True
|
|
1017
|
+
break
|
|
1018
|
+
if not found:
|
|
1019
|
+
raise OSError(
|
|
1020
|
+
f"journal high-water segment is unavailable: {high_water[0]}"
|
|
1021
|
+
)
|
|
1022
|
+
return "sha256:" + digest.hexdigest()
|
|
1023
|
+
|
|
1024
|
+
|
|
1025
|
+
def _capture_protocol_prefix_evidence(record, prior_high_water, evidence) -> None:
|
|
1026
|
+
"""Capture the actual raw prefix immediately preceding one audit record."""
|
|
1027
|
+
if (
|
|
1028
|
+
record.get("t") == "op"
|
|
1029
|
+
and isinstance(record.get("payload"), dict)
|
|
1030
|
+
and record["payload"].get("kind")
|
|
1031
|
+
== _lib_journal._PROTOCOL_RESOLUTION_KIND
|
|
1032
|
+
):
|
|
1033
|
+
evidence.append(
|
|
1034
|
+
(
|
|
1035
|
+
prior_high_water,
|
|
1036
|
+
journal_prefix_hash(prior_high_water),
|
|
1037
|
+
)
|
|
1038
|
+
)
|
|
1039
|
+
|
|
1040
|
+
|
|
693
1041
|
# --------------------------------------------------------------------------
|
|
694
1042
|
# cache leg — Codex quota obs -> cache.db quota_window_snapshots (spec §5.2
|
|
695
1043
|
# step 3, Task 7 Item 2). Runs BEFORE the stats BEGIN IMMEDIATE, under the
|
|
@@ -832,14 +1180,18 @@ def _resolve_ref(conn: sqlite3.Connection, table: str, logical_id) -> int | None
|
|
|
832
1180
|
return int(row[0])
|
|
833
1181
|
|
|
834
1182
|
|
|
835
|
-
def _insert_or_ignore(
|
|
1183
|
+
def _insert_or_ignore(
|
|
1184
|
+
conn: sqlite3.Connection, table: str, cols: dict, *, strict: bool = False
|
|
1185
|
+
):
|
|
836
1186
|
keys = list(cols.keys())
|
|
837
1187
|
colnames = ", ".join(keys)
|
|
838
1188
|
placeholders = ", ".join("?" for _ in keys)
|
|
839
|
-
|
|
840
|
-
f"INSERT OR IGNORE INTO {table} ({colnames}) VALUES ({placeholders})"
|
|
841
|
-
tuple(cols[k] for k in keys),
|
|
1189
|
+
statement = (
|
|
1190
|
+
f"INSERT OR IGNORE INTO {table} ({colnames}) VALUES ({placeholders})"
|
|
842
1191
|
)
|
|
1192
|
+
if strict:
|
|
1193
|
+
statement = statement.replace("INSERT OR IGNORE", "INSERT", 1)
|
|
1194
|
+
return conn.execute(statement, tuple(cols[k] for k in keys))
|
|
843
1195
|
|
|
844
1196
|
|
|
845
1197
|
def _reverse_ref(conn: sqlite3.Connection, ref_table: str, rowid) -> "str | None":
|
|
@@ -1253,6 +1605,34 @@ _BLOCK_CHILDREN = (
|
|
|
1253
1605
|
_BLOCK_CHILD_KEYS = frozenset(k for k, _t in _BLOCK_CHILDREN)
|
|
1254
1606
|
|
|
1255
1607
|
|
|
1608
|
+
def _replace_block_children(
|
|
1609
|
+
conn, block_id, parent_account, parent_window, children
|
|
1610
|
+
) -> None:
|
|
1611
|
+
"""Materialize one frozen block's child sets exactly.
|
|
1612
|
+
|
|
1613
|
+
A close event owns the complete model/project membership, so replay and
|
|
1614
|
+
convergence replace both sets rather than relying on natural-key
|
|
1615
|
+
``INSERT OR IGNORE``. This removes stale children and restores missing
|
|
1616
|
+
children while preserving the parent rowid used by milestone FKs.
|
|
1617
|
+
"""
|
|
1618
|
+
for payload_key, child_table in _BLOCK_CHILDREN:
|
|
1619
|
+
if parent_account is None:
|
|
1620
|
+
predicate = "five_hour_window_key = ?"
|
|
1621
|
+
params = (int(parent_window),)
|
|
1622
|
+
else:
|
|
1623
|
+
predicate = "account_key = ? AND five_hour_window_key = ?"
|
|
1624
|
+
params = (parent_account, int(parent_window))
|
|
1625
|
+
conn.execute(
|
|
1626
|
+
f"DELETE FROM {child_table} WHERE {predicate}", params
|
|
1627
|
+
)
|
|
1628
|
+
for child in children.get(payload_key, []):
|
|
1629
|
+
cols = dict(child)
|
|
1630
|
+
cols["block_id"] = int(block_id)
|
|
1631
|
+
if parent_account is not None:
|
|
1632
|
+
cols["account_key"] = parent_account
|
|
1633
|
+
_insert_or_ignore(conn, child_table, cols, strict=True)
|
|
1634
|
+
|
|
1635
|
+
|
|
1256
1636
|
def _apply_generic_evt(conn, evt):
|
|
1257
1637
|
"""Fold an evt line into its target table (spec §5.3), returning the sqlite
|
|
1258
1638
|
cursor of the `INSERT OR IGNORE`.
|
|
@@ -1278,28 +1658,37 @@ def _apply_generic_evt(conn, evt):
|
|
|
1278
1658
|
cols[key] = value
|
|
1279
1659
|
# Re-derive any projection-FK columns from a journaled natural-key column
|
|
1280
1660
|
# (spec §5.3 — e.g. five_hour_milestones.block_id from five_hour_window_key,
|
|
1281
|
-
# since the open block is a projection with no logical id).
|
|
1282
|
-
# (account_key, <lookup_col>) when the row carries an account (#341, review
|
|
1283
|
-
# finding 3): a shared physical 5h window resolves THIS account's block, so a
|
|
1284
|
-
# milestone child never attaches to another account's block. 0 when absent.
|
|
1661
|
+
# since the open block is a projection with no logical id).
|
|
1285
1662
|
acct = cols.get("account_key")
|
|
1286
1663
|
for column, (ref_table, lookup_col) in spec.derived_fk.items():
|
|
1287
|
-
|
|
1288
|
-
|
|
1289
|
-
f"SELECT id FROM {ref_table} "
|
|
1290
|
-
f"WHERE {lookup_col} = ? AND account_key = ?",
|
|
1291
|
-
(cols.get(lookup_col), acct),
|
|
1292
|
-
).fetchone()
|
|
1293
|
-
else:
|
|
1294
|
-
row = conn.execute(
|
|
1295
|
-
f"SELECT id FROM {ref_table} WHERE {lookup_col} = ?",
|
|
1296
|
-
(cols.get(lookup_col),),
|
|
1297
|
-
).fetchone()
|
|
1298
|
-
cols[column] = int(row[0]) if row is not None else 0
|
|
1664
|
+
cols[column] = _derived_fk_value(
|
|
1665
|
+
conn, ref_table, lookup_col, cols.get(lookup_col), acct)
|
|
1299
1666
|
return _insert_or_ignore(conn, spec.table, cols)
|
|
1300
1667
|
|
|
1301
1668
|
|
|
1302
|
-
def
|
|
1669
|
+
def _derived_fk_value(conn, ref_table, lookup_col, lookup_value, account_key):
|
|
1670
|
+
"""Resolve one derived (re-derived-at-fold) FK column (spec §5.3).
|
|
1671
|
+
|
|
1672
|
+
Composite `(account_key, <lookup_col>)` when the row carries an account
|
|
1673
|
+
(#341, review finding 3): a shared physical 5h window resolves THIS
|
|
1674
|
+
account's block, so a milestone never attaches to another account's block.
|
|
1675
|
+
0 when unresolvable. The SINGLE home of this rule — the fold applier and the
|
|
1676
|
+
#374 duplicate-path validation must agree by construction, not by copy."""
|
|
1677
|
+
if account_key is not None:
|
|
1678
|
+
row = conn.execute(
|
|
1679
|
+
f"SELECT id FROM {ref_table} "
|
|
1680
|
+
f"WHERE {lookup_col} = ? AND account_key = ?",
|
|
1681
|
+
(lookup_value, account_key),
|
|
1682
|
+
).fetchone()
|
|
1683
|
+
else:
|
|
1684
|
+
row = conn.execute(
|
|
1685
|
+
f"SELECT id FROM {ref_table} WHERE {lookup_col} = ?",
|
|
1686
|
+
(lookup_value,),
|
|
1687
|
+
).fetchone()
|
|
1688
|
+
return int(row[0]) if row is not None else 0
|
|
1689
|
+
|
|
1690
|
+
|
|
1691
|
+
def _apply_weekly_credit_effects(conn, evt, *, projection_writes=True):
|
|
1303
1692
|
"""Apply a `weekly_credit_effects` evt (spec §5.3 event+effects). The
|
|
1304
1693
|
same-window sub-25pp credit writes NO reset row, so its DESTRUCTIVE effects
|
|
1305
1694
|
ride this vehicle: delete the stale-replica snapshots by their logical
|
|
@@ -1326,7 +1715,7 @@ def _apply_weekly_credit_effects(conn, evt):
|
|
|
1326
1715
|
conn.execute(
|
|
1327
1716
|
"DELETE FROM weekly_credit_floors WHERE journal_id = ?", (logical_id,))
|
|
1328
1717
|
floor = payload.get("hwm_floor")
|
|
1329
|
-
if floor:
|
|
1718
|
+
if floor and projection_writes:
|
|
1330
1719
|
try:
|
|
1331
1720
|
(_cctally_core.APP_DIR / "hwm-7d").write_text(
|
|
1332
1721
|
f"{floor['week_start_date']} {floor['weekly_percent']}\n"
|
|
@@ -1340,17 +1729,30 @@ def _apply_quota_alert_arming(conn, evt):
|
|
|
1340
1729
|
"""Fold a `quota_alert_arming` evt (spec §5.3 "state", Task 7 Item 5). The
|
|
1341
1730
|
quota-alert arming boundary is journaled state — its `activated_at_utc` is a
|
|
1342
1731
|
forward-only alert boundary that MUST survive a stats.db rebuild so the
|
|
1343
|
-
reconcile honors it (no historical re-fires).
|
|
1344
|
-
|
|
1345
|
-
|
|
1346
|
-
|
|
1347
|
-
|
|
1732
|
+
reconcile honors it (no historical re-fires). Activation records UPSERT the
|
|
1733
|
+
natural key; explicit disarm records DELETE that same account-qualified key.
|
|
1734
|
+
Canonical replay therefore leaves the latest retained state in force, and
|
|
1735
|
+
re-applying either transition is a clean no-op. `quota_alert_arming` has no
|
|
1736
|
+
`journal_id` column (it is not in the Task-4 additive list); idempotence is
|
|
1737
|
+
the natural-key upsert/delete, not a journal_id INSERT OR IGNORE."""
|
|
1348
1738
|
p = evt.get("payload") or {}
|
|
1349
1739
|
# account_key (#341) is part of the arming identity/UNIQUE. A live-emitted evt
|
|
1350
1740
|
# carries payload.account_key; a legacy (pre-#341) cutover-exported arming has
|
|
1351
1741
|
# none -> normalise to the sentinel (Codex legacy -> unattributed) so the
|
|
1352
1742
|
# NOT NULL column always receives a value.
|
|
1353
1743
|
account_key = p.get("account_key") or _lib_accounts.UNATTRIBUTED
|
|
1744
|
+
if p.get("state") == "disarmed":
|
|
1745
|
+
conn.execute(
|
|
1746
|
+
"DELETE FROM quota_alert_arming "
|
|
1747
|
+
"WHERE source=? AND source_root_key=? AND account_key=? "
|
|
1748
|
+
"AND logical_limit_key=? AND observed_slot=? AND window_minutes=?",
|
|
1749
|
+
(
|
|
1750
|
+
p.get("source"), p.get("source_root_key"), account_key,
|
|
1751
|
+
p.get("logical_limit_key"), p.get("observed_slot"),
|
|
1752
|
+
p.get("window_minutes"),
|
|
1753
|
+
),
|
|
1754
|
+
)
|
|
1755
|
+
return None
|
|
1354
1756
|
conn.execute(
|
|
1355
1757
|
"INSERT INTO quota_alert_arming "
|
|
1356
1758
|
"(source, source_root_key, logical_limit_key, observed_slot, "
|
|
@@ -1368,12 +1770,14 @@ def _apply_quota_alert_arming(conn, evt):
|
|
|
1368
1770
|
|
|
1369
1771
|
|
|
1370
1772
|
def _apply_block_close(conn, evt):
|
|
1371
|
-
"""Fold
|
|
1372
|
-
|
|
1373
|
-
|
|
1374
|
-
(
|
|
1375
|
-
|
|
1376
|
-
|
|
1773
|
+
"""Fold one authoritative frozen-block fact.
|
|
1774
|
+
|
|
1775
|
+
A replay may meet an existing open projection under the same
|
|
1776
|
+
``(account_key, five_hour_window_key)`` natural key. ``INSERT OR IGNORE``
|
|
1777
|
+
alone would silently leave that mutable row and its children in place, so
|
|
1778
|
+
the event now converges the existing parent in place and replaces both
|
|
1779
|
+
child sets exactly. The parent rowid is preserved for milestone FKs.
|
|
1780
|
+
"""
|
|
1377
1781
|
payload = evt.get("payload") or {}
|
|
1378
1782
|
parent = {"journal_id": evt["id"]}
|
|
1379
1783
|
children = {}
|
|
@@ -1385,40 +1789,39 @@ def _apply_block_close(conn, evt):
|
|
|
1385
1789
|
continue
|
|
1386
1790
|
parent[key] = value
|
|
1387
1791
|
_insert_or_ignore(conn, "five_hour_blocks", parent)
|
|
1388
|
-
|
|
1389
|
-
# review finding 3): a shared physical window resolves THIS account's block
|
|
1390
|
-
# so its rollup children attach to the right parent.
|
|
1792
|
+
|
|
1391
1793
|
p_acct = parent.get("account_key")
|
|
1392
1794
|
if p_acct is not None:
|
|
1393
1795
|
prow = conn.execute(
|
|
1394
|
-
"SELECT id
|
|
1796
|
+
"SELECT id, journal_id, account_key, five_hour_window_key "
|
|
1797
|
+
"FROM five_hour_blocks "
|
|
1395
1798
|
"WHERE five_hour_window_key = ? AND account_key = ?",
|
|
1396
1799
|
(parent.get("five_hour_window_key"), p_acct),
|
|
1397
1800
|
).fetchone()
|
|
1398
1801
|
else:
|
|
1399
1802
|
prow = conn.execute(
|
|
1400
|
-
"SELECT id
|
|
1803
|
+
"SELECT id, journal_id, account_key, five_hour_window_key "
|
|
1804
|
+
"FROM five_hour_blocks "
|
|
1805
|
+
"WHERE five_hour_window_key = ?",
|
|
1401
1806
|
(parent.get("five_hour_window_key"),),
|
|
1402
1807
|
).fetchone()
|
|
1403
1808
|
if prow is None:
|
|
1404
|
-
|
|
1809
|
+
raise JournalError(
|
|
1810
|
+
f"five_hour_block_close {evt['id']} did not materialize its parent"
|
|
1811
|
+
)
|
|
1405
1812
|
block_id = int(prow[0])
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
|
|
1409
|
-
|
|
1410
|
-
|
|
1411
|
-
|
|
1412
|
-
|
|
1413
|
-
|
|
1414
|
-
|
|
1415
|
-
|
|
1416
|
-
|
|
1417
|
-
|
|
1418
|
-
# cutover mapping) leaves the DEFAULT untouched.
|
|
1419
|
-
if p_acct is not None:
|
|
1420
|
-
cols["account_key"] = p_acct
|
|
1421
|
-
_insert_or_ignore(conn, child_table, cols)
|
|
1813
|
+
existing_journal_id = prow[1]
|
|
1814
|
+
if existing_journal_id not in (None, evt["id"]):
|
|
1815
|
+
raise JournalError(
|
|
1816
|
+
f"five_hour_block_close {evt['id']} collided with "
|
|
1817
|
+
f"{existing_journal_id} on its parent natural key"
|
|
1818
|
+
)
|
|
1819
|
+
assignments = ", ".join(f"{name} = ?" for name in parent)
|
|
1820
|
+
conn.execute(
|
|
1821
|
+
f"UPDATE five_hour_blocks SET {assignments} WHERE id = ?",
|
|
1822
|
+
(*parent.values(), block_id),
|
|
1823
|
+
)
|
|
1824
|
+
_replace_block_children(conn, block_id, prow[2], prow[3], children)
|
|
1422
1825
|
return None
|
|
1423
1826
|
|
|
1424
1827
|
|
|
@@ -1452,13 +1855,16 @@ def _apply_reset_with_suppression(conn, evt):
|
|
|
1452
1855
|
return None
|
|
1453
1856
|
|
|
1454
1857
|
|
|
1455
|
-
def _apply_evt(conn, evt):
|
|
1858
|
+
def _apply_evt(conn, evt, *, projection_writes=True):
|
|
1456
1859
|
"""Dispatch one evt line to its fold applier by `payload.kind` (step 4a
|
|
1457
1860
|
replay + the emit_model_a apply path). A kind with a bespoke `applier`
|
|
1458
1861
|
(weekly_credit_effects, five_hour_block_close) uses it; everything else
|
|
1459
1862
|
goes through the generic column-map fold. Apply-only: NO alert dispatch,
|
|
1460
1863
|
NO ctx — replay is structurally unable to fire alerts (spec §5.2 step 4a)."""
|
|
1461
1864
|
spec = _EVT_SPECS.get((evt.get("payload") or {}).get("kind"))
|
|
1865
|
+
if spec is not None and spec.applier is _apply_weekly_credit_effects:
|
|
1866
|
+
return spec.applier(
|
|
1867
|
+
conn, evt, projection_writes=projection_writes)
|
|
1462
1868
|
if spec is not None and spec.applier is not None:
|
|
1463
1869
|
return spec.applier(conn, evt)
|
|
1464
1870
|
return _apply_generic_evt(conn, evt)
|
|
@@ -1594,9 +2000,34 @@ def emit_model_a(ctx, *, kind, evt_id, table, columns, refs=None, at=None):
|
|
|
1594
2000
|
payload.update(refs)
|
|
1595
2001
|
evt = _lib_journal.make_evt(kind=kind, id=evt_id, at=(at or _now_iso()),
|
|
1596
2002
|
payload=payload)
|
|
1597
|
-
|
|
1598
|
-
|
|
1599
|
-
|
|
2003
|
+
if ctx.event_sink is not None:
|
|
2004
|
+
# SCRATCH planning (`db rederive`, spec §6): capture EVERY derived
|
|
2005
|
+
# candidate — never classify, never drop, or the very divergence the
|
|
2006
|
+
# planner exists to correct would be filtered out of `desired_events`
|
|
2007
|
+
# and the diff would report a false no-op. Model-A events are still
|
|
2008
|
+
# APPLIED to the private scratch projection because callers consume the
|
|
2009
|
+
# returned rowid immediately (`snapshot_accept` stores it before
|
|
2010
|
+
# milestone/block derivation). Live effective metadata is never read or
|
|
2011
|
+
# written. Discriminated on `is not None` — the sink is `list | None`
|
|
2012
|
+
# and an EMPTY sink list is falsy.
|
|
2013
|
+
ctx.event_sink.append(evt)
|
|
2014
|
+
ctx.events_emitted += 1
|
|
2015
|
+
_apply_evt(ctx.conn, evt, projection_writes=ctx.projection_writes)
|
|
2016
|
+
else:
|
|
2017
|
+
decision = _classify_live_effective_event(ctx.conn, evt)
|
|
2018
|
+
if decision == CLASSIFY_CONFLICT:
|
|
2019
|
+
# No append, no metadata mutation — converge the row instead.
|
|
2020
|
+
_record_dropped_conflict(ctx, evt)
|
|
2021
|
+
_converge_row_from_effective(ctx.conn, evt_id, table=table)
|
|
2022
|
+
else:
|
|
2023
|
+
# `new` AND `duplicate` both still append: crash-replay convergence
|
|
2024
|
+
# and two-bootstrap idempotency are built on that.
|
|
2025
|
+
append_record(evt)
|
|
2026
|
+
ctx.events_emitted += 1
|
|
2027
|
+
if decision == CLASSIFY_NEW:
|
|
2028
|
+
_record_new_effective_event(ctx.conn, evt)
|
|
2029
|
+
_apply_evt(
|
|
2030
|
+
ctx.conn, evt, projection_writes=ctx.projection_writes)
|
|
1600
2031
|
if table is None:
|
|
1601
2032
|
return None
|
|
1602
2033
|
row = ctx.conn.execute(
|
|
@@ -1657,6 +2088,71 @@ def _build_harvest_evt(ctx, spec, row):
|
|
|
1657
2088
|
return _lib_journal.make_evt(kind=spec.kind, id=eid, at=at, payload=payload)
|
|
1658
2089
|
|
|
1659
2090
|
|
|
2091
|
+
def _emit_harvest_row(ctx, spec, row):
|
|
2092
|
+
"""Journal and stamp one already-selected natural-keyed row."""
|
|
2093
|
+
conn = ctx.conn
|
|
2094
|
+
evt = _build_harvest_evt(ctx, spec, row)
|
|
2095
|
+
if ctx.event_sink is not None:
|
|
2096
|
+
# Scratch planning: capture + stamp the PRIVATE projection so a later
|
|
2097
|
+
# raw record does not re-harvest the same row and its downstream FKs
|
|
2098
|
+
# still resolve. Never classify, drop, or touch live metadata.
|
|
2099
|
+
ctx.event_sink.append(evt)
|
|
2100
|
+
ctx.events_emitted += 1
|
|
2101
|
+
conn.execute(
|
|
2102
|
+
f"UPDATE {spec.table} SET journal_id = ? WHERE id = ?",
|
|
2103
|
+
(evt["id"], row["id"]),
|
|
2104
|
+
)
|
|
2105
|
+
return evt
|
|
2106
|
+
|
|
2107
|
+
decision = _classify_live_effective_event(conn, evt)
|
|
2108
|
+
if decision == CLASSIFY_CONFLICT:
|
|
2109
|
+
_record_dropped_conflict(ctx, evt)
|
|
2110
|
+
_converge_row_from_effective(
|
|
2111
|
+
conn, evt["id"], table=spec.table, rowid=row["id"]
|
|
2112
|
+
)
|
|
2113
|
+
return evt
|
|
2114
|
+
|
|
2115
|
+
append_record(evt)
|
|
2116
|
+
ctx.events_emitted += 1
|
|
2117
|
+
if decision == CLASSIFY_NEW:
|
|
2118
|
+
_record_new_effective_event(conn, evt)
|
|
2119
|
+
else:
|
|
2120
|
+
# An exact crash-retry duplicate still validates excluded derived FKs
|
|
2121
|
+
# before it stamps the physical row.
|
|
2122
|
+
_validate_excluded_derived_fks(conn, spec, row)
|
|
2123
|
+
conn.execute(
|
|
2124
|
+
f"UPDATE {spec.table} SET journal_id = ? WHERE id = ?",
|
|
2125
|
+
(evt["id"], row["id"]),
|
|
2126
|
+
)
|
|
2127
|
+
return evt
|
|
2128
|
+
|
|
2129
|
+
|
|
2130
|
+
def freeze_five_hour_block_close(ctx, block_id: int):
|
|
2131
|
+
"""Freeze one closed block immediately as a complete replayable fact.
|
|
2132
|
+
|
|
2133
|
+
Unlike the end-of-cycle table scan, this surface also accepts an already
|
|
2134
|
+
stamped row so a lost-commit retry can re-emit the exact duplicate selected
|
|
2135
|
+
by its retained closure trigger. It never derives from cache state itself;
|
|
2136
|
+
the parent and both child sets present at this call are the durable boundary.
|
|
2137
|
+
"""
|
|
2138
|
+
spec = next(
|
|
2139
|
+
item for item in _HARVEST_SPECS
|
|
2140
|
+
if item.kind == "five_hour_block_close"
|
|
2141
|
+
)
|
|
2142
|
+
row = ctx.conn.execute(
|
|
2143
|
+
"SELECT * FROM five_hour_blocks WHERE id = ?", (int(block_id),)
|
|
2144
|
+
).fetchone()
|
|
2145
|
+
if row is None:
|
|
2146
|
+
raise JournalError(
|
|
2147
|
+
f"cannot freeze five_hour_block_close: missing block {block_id}"
|
|
2148
|
+
)
|
|
2149
|
+
if int(row["is_closed"]) != 1:
|
|
2150
|
+
raise JournalError(
|
|
2151
|
+
f"cannot freeze five_hour_block_close: block {block_id} is open"
|
|
2152
|
+
)
|
|
2153
|
+
return _emit_harvest_row(ctx, spec, row)
|
|
2154
|
+
|
|
2155
|
+
|
|
1660
2156
|
def _harvest(ctx) -> None:
|
|
1661
2157
|
"""Step 4c: journal + stamp every natural-keyed row inserted this cycle
|
|
1662
2158
|
(`journal_id IS NULL`). Families harvest in dependency order so a referenced
|
|
@@ -1672,13 +2168,7 @@ def _harvest(ctx) -> None:
|
|
|
1672
2168
|
f"SELECT * FROM {spec.table} WHERE {where} ORDER BY id"
|
|
1673
2169
|
).fetchall()
|
|
1674
2170
|
for row in rows:
|
|
1675
|
-
|
|
1676
|
-
append_record(evt)
|
|
1677
|
-
ctx.events_emitted += 1
|
|
1678
|
-
conn.execute(
|
|
1679
|
-
f"UPDATE {spec.table} SET journal_id = ? WHERE id = ?",
|
|
1680
|
-
(evt["id"], row["id"]),
|
|
1681
|
-
)
|
|
2171
|
+
_emit_harvest_row(ctx, spec, row)
|
|
1682
2172
|
|
|
1683
2173
|
|
|
1684
2174
|
# --------------------------------------------------------------------------
|
|
@@ -1853,14 +2343,551 @@ def _fold_order(evt) -> int:
|
|
|
1853
2343
|
return (_EVT_SPECS.get(kind) or _UNKNOWN_EVT_SPEC).order
|
|
1854
2344
|
|
|
1855
2345
|
|
|
1856
|
-
|
|
1857
|
-
|
|
1858
|
-
|
|
2346
|
+
def _write_effective_metadata(conn, selection) -> None:
|
|
2347
|
+
"""Replace the disposable effective-event summary from a pure selection."""
|
|
2348
|
+
conn.execute("DELETE FROM journal_effective_events")
|
|
2349
|
+
for event_id, selected in selection.by_id.items():
|
|
2350
|
+
event_json = None
|
|
2351
|
+
if selected.record is not None:
|
|
2352
|
+
event_json = (
|
|
2353
|
+
_lib_journal.encode_line(selected.record)
|
|
2354
|
+
.decode("utf-8")
|
|
2355
|
+
.rstrip("\n")
|
|
2356
|
+
)
|
|
2357
|
+
conn.execute(
|
|
2358
|
+
"INSERT INTO journal_effective_events "
|
|
2359
|
+
"(event_id, rev, status, content_hash, batch_id, event_json) "
|
|
2360
|
+
"VALUES (?, ?, ?, ?, ?, ?)",
|
|
2361
|
+
(
|
|
2362
|
+
event_id,
|
|
2363
|
+
selected.rev,
|
|
2364
|
+
selected.status,
|
|
2365
|
+
selected.content_hash,
|
|
2366
|
+
selected.batch_id,
|
|
2367
|
+
event_json,
|
|
2368
|
+
),
|
|
2369
|
+
)
|
|
2370
|
+
_write_protocol_violations(
|
|
2371
|
+
conn,
|
|
2372
|
+
selection.protocol_violations,
|
|
2373
|
+
selection.acknowledged_protocol_violations,
|
|
2374
|
+
)
|
|
1859
2375
|
|
|
1860
|
-
|
|
1861
|
-
|
|
1862
|
-
|
|
1863
|
-
|
|
2376
|
+
|
|
2377
|
+
def _write_protocol_violations(conn, violations, acknowledged=()) -> None:
|
|
2378
|
+
"""Replace the disposable structural-violation summary."""
|
|
2379
|
+
conn.execute("DELETE FROM journal_protocol_violations")
|
|
2380
|
+
rows = [*violations, *acknowledged]
|
|
2381
|
+
rows.sort(
|
|
2382
|
+
key=lambda violation: (
|
|
2383
|
+
violation.batch_id,
|
|
2384
|
+
violation.kind,
|
|
2385
|
+
violation.fingerprint,
|
|
2386
|
+
)
|
|
2387
|
+
)
|
|
2388
|
+
for violation in rows:
|
|
2389
|
+
conn.execute(
|
|
2390
|
+
"INSERT INTO journal_protocol_violations "
|
|
2391
|
+
"(fingerprint, batch_id, kind, violation_json) "
|
|
2392
|
+
"VALUES (?, ?, ?, ?)",
|
|
2393
|
+
(
|
|
2394
|
+
violation.fingerprint,
|
|
2395
|
+
violation.batch_id,
|
|
2396
|
+
violation.kind,
|
|
2397
|
+
json.dumps(
|
|
2398
|
+
violation.to_dict(),
|
|
2399
|
+
sort_keys=True,
|
|
2400
|
+
separators=(",", ":"),
|
|
2401
|
+
),
|
|
2402
|
+
),
|
|
2403
|
+
)
|
|
2404
|
+
|
|
2405
|
+
|
|
2406
|
+
def _metadata_row(conn, event_id):
|
|
2407
|
+
return conn.execute(
|
|
2408
|
+
"SELECT rev, status, content_hash, batch_id "
|
|
2409
|
+
"FROM journal_effective_events WHERE event_id = ?",
|
|
2410
|
+
(event_id,),
|
|
2411
|
+
).fetchone()
|
|
2412
|
+
|
|
2413
|
+
|
|
2414
|
+
def _metadata_event_record(conn, event_id):
|
|
2415
|
+
row = conn.execute(
|
|
2416
|
+
"SELECT event_json FROM journal_effective_events WHERE event_id = ?",
|
|
2417
|
+
(event_id,),
|
|
2418
|
+
).fetchone()
|
|
2419
|
+
if row is None or row[0] is None:
|
|
2420
|
+
return None
|
|
2421
|
+
return _lib_journal.decode_line(row[0].encode("utf-8"))
|
|
2422
|
+
|
|
2423
|
+
|
|
2424
|
+
def _legacy_qaa_can_advance(conn, selected) -> bool:
|
|
2425
|
+
return (
|
|
2426
|
+
_lib_journal.is_legacy_quota_arming_record(selected.record)
|
|
2427
|
+
and _lib_journal.is_legacy_quota_arming_record(
|
|
2428
|
+
_metadata_event_record(conn, selected.event_id)
|
|
2429
|
+
)
|
|
2430
|
+
)
|
|
2431
|
+
|
|
2432
|
+
|
|
2433
|
+
def _insert_effective_metadata(conn, selected) -> None:
|
|
2434
|
+
event_json = None
|
|
2435
|
+
if selected.record is not None:
|
|
2436
|
+
event_json = (
|
|
2437
|
+
_lib_journal.encode_line(selected.record).decode("utf-8").rstrip("\n")
|
|
2438
|
+
)
|
|
2439
|
+
conn.execute(
|
|
2440
|
+
"INSERT INTO journal_effective_events "
|
|
2441
|
+
"(event_id, rev, status, content_hash, batch_id, event_json) "
|
|
2442
|
+
"VALUES (?, ?, ?, ?, ?, ?)",
|
|
2443
|
+
(
|
|
2444
|
+
selected.event_id,
|
|
2445
|
+
selected.rev,
|
|
2446
|
+
selected.status,
|
|
2447
|
+
selected.content_hash,
|
|
2448
|
+
selected.batch_id,
|
|
2449
|
+
event_json,
|
|
2450
|
+
),
|
|
2451
|
+
)
|
|
2452
|
+
|
|
2453
|
+
|
|
2454
|
+
CLASSIFY_NEW = "new"
|
|
2455
|
+
CLASSIFY_DUPLICATE = "duplicate"
|
|
2456
|
+
CLASSIFY_CONFLICT = "conflict"
|
|
2457
|
+
|
|
2458
|
+
|
|
2459
|
+
def _classify_live_effective_event(conn, evt) -> str:
|
|
2460
|
+
"""Decide what a freshly derived evt means against the live effective
|
|
2461
|
+
metadata — WITHOUT mutating anything (#374 §6).
|
|
2462
|
+
|
|
2463
|
+
The whole point of the split is ordering: both emit paths call this BEFORE
|
|
2464
|
+
`append_record`, so a conflicting emission is never written. Previously the
|
|
2465
|
+
append ran first and the check raised afterwards, so the divergent line
|
|
2466
|
+
landed in the append-only journal and the rollback could not take it back —
|
|
2467
|
+
poisoning every subsequent rebuild.
|
|
2468
|
+
|
|
2469
|
+
Returns `CLASSIFY_NEW` (no prior effective event, or a legacy `qaa` state
|
|
2470
|
+
stream that may advance), `CLASSIFY_DUPLICATE` (byte-identical to the prior
|
|
2471
|
+
effective event) or `CLASSIFY_CONFLICT` (same revision, different content).
|
|
2472
|
+
Raises `CorrectionRebuildRequired` on a revision mismatch — a completed
|
|
2473
|
+
correction batch outranks any live emission and stays FATAL.
|
|
2474
|
+
"""
|
|
2475
|
+
selected = _lib_journal.resolve_effective_events([evt]).by_id[evt["id"]]
|
|
2476
|
+
prior = _metadata_row(conn, selected.event_id)
|
|
2477
|
+
if prior is None:
|
|
2478
|
+
return CLASSIFY_NEW
|
|
2479
|
+
prior_rev, prior_status, prior_hash, prior_batch = prior
|
|
2480
|
+
if int(prior_rev) == selected.rev:
|
|
2481
|
+
if prior_status != selected.status or prior_hash != selected.content_hash:
|
|
2482
|
+
if _legacy_qaa_can_advance(conn, selected):
|
|
2483
|
+
return CLASSIFY_NEW
|
|
2484
|
+
return CLASSIFY_CONFLICT
|
|
2485
|
+
return CLASSIFY_DUPLICATE
|
|
2486
|
+
raise CorrectionRebuildRequired(
|
|
2487
|
+
f"event {selected.event_id} rev {selected.rev} conflicts with effective "
|
|
2488
|
+
f"rev {prior_rev} from {prior_batch or 'base journal'}",
|
|
2489
|
+
batch_id=prior_batch,
|
|
2490
|
+
event_id=selected.event_id,
|
|
2491
|
+
high_water=_correction_commit_high_water(prior_batch),
|
|
2492
|
+
expected_metadata=(
|
|
2493
|
+
int(prior_rev),
|
|
2494
|
+
prior_status,
|
|
2495
|
+
prior_hash,
|
|
2496
|
+
prior_batch,
|
|
2497
|
+
),
|
|
2498
|
+
)
|
|
2499
|
+
|
|
2500
|
+
|
|
2501
|
+
def _record_new_effective_event(conn, evt) -> None:
|
|
2502
|
+
"""Write the effective-metadata row for a `CLASSIFY_NEW` emission. The
|
|
2503
|
+
DELETE covers the legacy `qaa` advance (the only case where a prior row is
|
|
2504
|
+
replaced rather than absent)."""
|
|
2505
|
+
selected = _lib_journal.resolve_effective_events([evt]).by_id[evt["id"]]
|
|
2506
|
+
if _metadata_row(conn, selected.event_id) is not None:
|
|
2507
|
+
conn.execute(
|
|
2508
|
+
"DELETE FROM journal_effective_events WHERE event_id = ?",
|
|
2509
|
+
(selected.event_id,),
|
|
2510
|
+
)
|
|
2511
|
+
_insert_effective_metadata(conn, selected)
|
|
2512
|
+
|
|
2513
|
+
|
|
2514
|
+
def _record_live_effective_event(conn, evt) -> bool:
|
|
2515
|
+
"""Record a newly folded base event; return False when the caller must NOT
|
|
2516
|
+
apply it — an exact duplicate, or a quarantined same-revision conflict.
|
|
2517
|
+
|
|
2518
|
+
Retained for the step-4a replay site, whose conflicts the preflight reader
|
|
2519
|
+
has already dropped. The emit paths use the classifier directly so they can
|
|
2520
|
+
withhold the append and converge the row."""
|
|
2521
|
+
decision = _classify_live_effective_event(conn, evt)
|
|
2522
|
+
if decision == CLASSIFY_NEW:
|
|
2523
|
+
_record_new_effective_event(conn, evt)
|
|
2524
|
+
return True
|
|
2525
|
+
return False
|
|
2526
|
+
|
|
2527
|
+
|
|
2528
|
+
def _record_dropped_conflict(ctx, evt) -> None:
|
|
2529
|
+
"""Count + report one withheld emission (spec §8: a one-line stderr note per
|
|
2530
|
+
dropped emission, and a count on the cycle summary)."""
|
|
2531
|
+
selected = _lib_journal.resolve_effective_events([evt]).by_id[evt["id"]]
|
|
2532
|
+
ctx.conflicts_dropped.append(
|
|
2533
|
+
DroppedConflict(
|
|
2534
|
+
event_id=selected.event_id,
|
|
2535
|
+
rev=selected.rev,
|
|
2536
|
+
rejected_hash=selected.content_hash,
|
|
2537
|
+
)
|
|
2538
|
+
)
|
|
2539
|
+
print(
|
|
2540
|
+
f"[journal] withheld a divergent emission for {selected.event_id} "
|
|
2541
|
+
f"rev {selected.rev}; converged the row from the journaled event",
|
|
2542
|
+
file=sys.stderr,
|
|
2543
|
+
)
|
|
2544
|
+
|
|
2545
|
+
|
|
2546
|
+
def _effective_event_for_convergence(conn, event_id) -> dict:
|
|
2547
|
+
"""The ACTIVE, validated journal record the live row must converge to.
|
|
2548
|
+
|
|
2549
|
+
Fails closed (#374 §6): `decode_line` only checks that the value is an object
|
|
2550
|
+
with a string `t`, and a tombstoned selection deliberately stores
|
|
2551
|
+
`event_json` as NULL — so a same-revision active-vs-tombstone conflict has no
|
|
2552
|
+
record to converge from. Missing, tombstoned, or hash-mismatched metadata
|
|
2553
|
+
raises here and the caller therefore NEVER stamps."""
|
|
2554
|
+
row = conn.execute(
|
|
2555
|
+
"SELECT rev, status, content_hash, event_json "
|
|
2556
|
+
"FROM journal_effective_events WHERE event_id = ?",
|
|
2557
|
+
(event_id,),
|
|
2558
|
+
).fetchone()
|
|
2559
|
+
if row is None:
|
|
2560
|
+
raise _lib_journal.JournalProtocolError(
|
|
2561
|
+
f"cannot converge {event_id}: no effective metadata")
|
|
2562
|
+
_rev, status, content_hash, event_json = row
|
|
2563
|
+
if status != "active" or event_json is None:
|
|
2564
|
+
raise _lib_journal.JournalProtocolError(
|
|
2565
|
+
f"cannot converge {event_id}: effective selection is {status!r} "
|
|
2566
|
+
"with no retained record")
|
|
2567
|
+
record = _lib_journal.decode_line(event_json.encode("utf-8"))
|
|
2568
|
+
if (
|
|
2569
|
+
record is None
|
|
2570
|
+
or record.get("t") != "evt"
|
|
2571
|
+
or record.get("id") != event_id
|
|
2572
|
+
or not isinstance(record.get("payload"), dict)
|
|
2573
|
+
):
|
|
2574
|
+
raise _lib_journal.JournalProtocolError(
|
|
2575
|
+
f"cannot converge {event_id}: retained record is not a matching evt")
|
|
2576
|
+
if _lib_journal._sha256_canonical(record) != content_hash:
|
|
2577
|
+
raise _lib_journal.JournalProtocolError(
|
|
2578
|
+
f"cannot converge {event_id}: retained record hash mismatch")
|
|
2579
|
+
return record
|
|
2580
|
+
|
|
2581
|
+
|
|
2582
|
+
# Effect keys that ride an evt payload but are NOT target-table columns.
|
|
2583
|
+
_EVT_EFFECT_KEYS = frozenset(
|
|
2584
|
+
{"kind", "suppression", "suppression_table", "floor_suppression", "hwm_floor"}
|
|
2585
|
+
)
|
|
2586
|
+
|
|
2587
|
+
|
|
2588
|
+
def _evt_target_columns(conn, evt, spec) -> tuple:
|
|
2589
|
+
"""Decode one evt payload into `(columns, children)` for its target row —
|
|
2590
|
+
the same mapping `_apply_generic_evt` performs, but WITHOUT inserting, so a
|
|
2591
|
+
convergence can UPDATE an existing physical row."""
|
|
2592
|
+
payload = evt.get("payload") or {}
|
|
2593
|
+
cols = {"journal_id": evt["id"]}
|
|
2594
|
+
children: dict = {}
|
|
2595
|
+
for key, value in payload.items():
|
|
2596
|
+
if key in _EVT_EFFECT_KEYS:
|
|
2597
|
+
continue
|
|
2598
|
+
if key in _BLOCK_CHILD_KEYS:
|
|
2599
|
+
children[key] = value or []
|
|
2600
|
+
continue
|
|
2601
|
+
if key in spec.fk_refs:
|
|
2602
|
+
column, ref_table = spec.fk_refs[key]
|
|
2603
|
+
cols[column] = _resolve_ref(conn, ref_table, value)
|
|
2604
|
+
else:
|
|
2605
|
+
cols[key] = value
|
|
2606
|
+
acct = cols.get("account_key")
|
|
2607
|
+
for column, (ref_table, lookup_col) in spec.derived_fk.items():
|
|
2608
|
+
cols[column] = _derived_fk_value(
|
|
2609
|
+
conn, ref_table, lookup_col, cols.get(lookup_col), acct)
|
|
2610
|
+
return cols, children
|
|
2611
|
+
|
|
2612
|
+
|
|
2613
|
+
CONVERGE_DROPPED = "dropped"
|
|
2614
|
+
CONVERGE_APPLIED = "converged"
|
|
2615
|
+
|
|
2616
|
+
|
|
2617
|
+
def _converge_row_from_effective(conn, event_id, *, table=None, rowid=None) -> str:
|
|
2618
|
+
"""Bring the live physical row into agreement with the already-journaled
|
|
2619
|
+
effective event, and stamp `journal_id` in the SAME operation (#374 §6).
|
|
2620
|
+
|
|
2621
|
+
This is an EXPLICIT row-convergence operation, deliberately NOT a generic
|
|
2622
|
+
re-run of an arbitrary fold applier: `_apply_generic_evt` ends in
|
|
2623
|
+
`INSERT OR IGNORE`, and effect-bearing appliers cannot safely run out of
|
|
2624
|
+
canonical order. `five_hour_block_close` is the strengthened exception:
|
|
2625
|
+
its ordinary fold also converges the parent and exact child sets so orphan
|
|
2626
|
+
replay freezes an existing projection before raw derivation. Keeping the
|
|
2627
|
+
explicit convergence path still avoids invoking destructive effects and
|
|
2628
|
+
supports row-targeted conflict repair.
|
|
2629
|
+
|
|
2630
|
+
Effect-bearing families are NOT converged. `event_json` is authoritative row
|
|
2631
|
+
*data*, never permission to invoke every applier: `_apply_weekly_credit_
|
|
2632
|
+
effects` performs destructive deletes and writes a non-transactional HWM
|
|
2633
|
+
projection, and the reset appliers replay suppression deletes. Replaying an
|
|
2634
|
+
older effective event at the CURRENT execution point is not equivalent to
|
|
2635
|
+
folding it at its canonical journal position. So an effects-only family
|
|
2636
|
+
(`spec.table is None`) returns `CONVERGE_DROPPED` and nothing is replayed.
|
|
2637
|
+
|
|
2638
|
+
A family WITH a table but no physical row carrying the event id is also
|
|
2639
|
+
`CONVERGE_DROPPED`: convergence updates what exists, it never materialises a
|
|
2640
|
+
row (see the `rowid is None` branch).
|
|
2641
|
+
"""
|
|
2642
|
+
record = _effective_event_for_convergence(conn, event_id)
|
|
2643
|
+
spec = _EVT_SPECS.get((record.get("payload") or {}).get("kind"))
|
|
2644
|
+
if spec is None or spec.table is None:
|
|
2645
|
+
return CONVERGE_DROPPED
|
|
2646
|
+
target = spec.table
|
|
2647
|
+
if table is not None and table != target:
|
|
2648
|
+
raise JournalError(
|
|
2649
|
+
f"convergence target mismatch for {event_id}: {table} != {target}")
|
|
2650
|
+
cols, children = _evt_target_columns(conn, record, spec)
|
|
2651
|
+
if rowid is None:
|
|
2652
|
+
row = conn.execute(
|
|
2653
|
+
f"SELECT id FROM {target} WHERE journal_id = ?", (event_id,)
|
|
2654
|
+
).fetchone()
|
|
2655
|
+
rowid = int(row[0]) if row is not None else None
|
|
2656
|
+
if rowid is None:
|
|
2657
|
+
# NOTHING to converge — and materializing a row here would be wrong twice
|
|
2658
|
+
# over (#374 review). A row absent because a suppression effect
|
|
2659
|
+
# deliberately DELETED it would be resurrected, so the live index and a
|
|
2660
|
+
# rebuild would diverge — the very contract convergence exists to hold.
|
|
2661
|
+
# And an insert swallowed by a natural-key UNIQUE would leave the
|
|
2662
|
+
# follow-up lookup empty and abort the whole cycle on a `JournalError`.
|
|
2663
|
+
# Drop and report; the emission was already withheld by the caller.
|
|
2664
|
+
print(
|
|
2665
|
+
f"[journal] no live row for {event_id} in {target}; "
|
|
2666
|
+
"dropped the divergent emission without materializing one",
|
|
2667
|
+
file=sys.stderr,
|
|
2668
|
+
)
|
|
2669
|
+
return CONVERGE_DROPPED
|
|
2670
|
+
assignments = ", ".join(f"{name} = ?" for name in cols)
|
|
2671
|
+
conn.execute(
|
|
2672
|
+
f"UPDATE {target} SET {assignments} WHERE id = ?",
|
|
2673
|
+
(*cols.values(), int(rowid)),
|
|
2674
|
+
)
|
|
2675
|
+
if spec.applier is _apply_block_close:
|
|
2676
|
+
identity = conn.execute(
|
|
2677
|
+
"SELECT account_key, five_hour_window_key "
|
|
2678
|
+
"FROM five_hour_blocks WHERE id = ?",
|
|
2679
|
+
(int(rowid),),
|
|
2680
|
+
).fetchone()
|
|
2681
|
+
if identity is None:
|
|
2682
|
+
raise JournalError(
|
|
2683
|
+
f"convergence target vanished for {event_id}"
|
|
2684
|
+
)
|
|
2685
|
+
_replace_block_children(
|
|
2686
|
+
conn, int(rowid), identity[0], identity[1], children
|
|
2687
|
+
)
|
|
2688
|
+
return CONVERGE_APPLIED
|
|
2689
|
+
|
|
2690
|
+
|
|
2691
|
+
def _validate_excluded_derived_fks(conn, spec, row) -> None:
|
|
2692
|
+
"""Validate every column the harvest EXCLUDES from the evt, before the
|
|
2693
|
+
duplicate path stamps the row (#374 §6 / acceptance 8).
|
|
2694
|
+
|
|
2695
|
+
Byte identity of the emitted event proves the JOURNALED columns match. It
|
|
2696
|
+
proves nothing about the excluded ones: `_build_harvest_evt` omits
|
|
2697
|
+
`journal_id`, physical ids and derived-FK columns, and
|
|
2698
|
+
`five_hour_milestones.block_id` is deliberately derived rather than
|
|
2699
|
+
journaled — so a milestone pointing at the WRONG block can emit an otherwise
|
|
2700
|
+
byte-identical event. The canonical logical dump also excludes `block_id`
|
|
2701
|
+
and would not catch it.
|
|
2702
|
+
|
|
2703
|
+
An unresolvable reference is NOT an error on its own. `_derived_fk_value`
|
|
2704
|
+
returns 0 as the "no such parent" sentinel and the fold appliers store that
|
|
2705
|
+
same 0, so a legitimately parentless row — e.g. a `five_hour_milestones` row
|
|
2706
|
+
whose `five_hour_blocks` replica the 5h-credit stale-replica DELETE removed —
|
|
2707
|
+
carries `actual == expected == 0` and is in agreement. Raising on that shape
|
|
2708
|
+
escaped `_harvest`, rolled the cycle back, left the row unstamped and made
|
|
2709
|
+
every later cycle repeat it (#374 review). ONLY disagreement is fatal."""
|
|
2710
|
+
if not spec.derived_fk:
|
|
2711
|
+
return
|
|
2712
|
+
keys = set(row.keys())
|
|
2713
|
+
account_key = row["account_key"] if "account_key" in keys else None
|
|
2714
|
+
for column, (ref_table, lookup_col) in spec.derived_fk.items():
|
|
2715
|
+
expected = _derived_fk_value(
|
|
2716
|
+
conn, ref_table, lookup_col, row[lookup_col], account_key)
|
|
2717
|
+
actual = row[column]
|
|
2718
|
+
if int(actual) != expected:
|
|
2719
|
+
raise JournalError(
|
|
2720
|
+
f"harvest {spec.kind}: derived FK {column}={actual!r} does not "
|
|
2721
|
+
f"resolve to {ref_table}.{lookup_col}={row[lookup_col]!r} "
|
|
2722
|
+
f"(re-derived {expected})")
|
|
2723
|
+
|
|
2724
|
+
|
|
2725
|
+
def _full_effective_selection(hw):
|
|
2726
|
+
records = []
|
|
2727
|
+
evidence = []
|
|
2728
|
+
prior_high_water = None
|
|
2729
|
+
if hw is not None:
|
|
2730
|
+
for segment, offset, raw in _read_range(None, hw):
|
|
2731
|
+
record = _lib_journal.decode_line(raw)
|
|
2732
|
+
if record is not None:
|
|
2733
|
+
_capture_protocol_prefix_evidence(
|
|
2734
|
+
record,
|
|
2735
|
+
prior_high_water,
|
|
2736
|
+
evidence,
|
|
2737
|
+
)
|
|
2738
|
+
records.append(record)
|
|
2739
|
+
prior_high_water = (segment, offset + len(raw) + 1)
|
|
2740
|
+
cutover_claude = resolve_cutover_claude_account()
|
|
2741
|
+
for record in records:
|
|
2742
|
+
_normalize_legacy_account_stamp(record, cutover_claude)
|
|
2743
|
+
return _lib_journal.resolve_effective_events(
|
|
2744
|
+
records,
|
|
2745
|
+
protocol_prefix_evidence=evidence,
|
|
2746
|
+
)
|
|
2747
|
+
|
|
2748
|
+
|
|
2749
|
+
def _correction_commit_high_water(batch_id, hw=None):
|
|
2750
|
+
"""Return the exact end offset of one completed-batch commit marker.
|
|
2751
|
+
|
|
2752
|
+
The batch was already structurally validated either by the full effective
|
|
2753
|
+
selector or by the live metadata row that names it. The earliest matching
|
|
2754
|
+
commit is the narrowest complete prefix and remains stable even when later
|
|
2755
|
+
journal bytes or crash-replayed duplicate markers exist.
|
|
2756
|
+
"""
|
|
2757
|
+
if not batch_id:
|
|
2758
|
+
return None
|
|
2759
|
+
if hw is None:
|
|
2760
|
+
hw = journal_high_water()
|
|
2761
|
+
if hw is None:
|
|
2762
|
+
return None
|
|
2763
|
+
for segment, offset, raw in _read_range(None, hw):
|
|
2764
|
+
record = _lib_journal.decode_line(raw)
|
|
2765
|
+
if (
|
|
2766
|
+
record is not None
|
|
2767
|
+
and record.get("t") == "correction_batch"
|
|
2768
|
+
and record.get("phase") == "commit"
|
|
2769
|
+
and record.get("id") == batch_id
|
|
2770
|
+
):
|
|
2771
|
+
return (segment, offset + len(raw) + 1)
|
|
2772
|
+
return None
|
|
2773
|
+
|
|
2774
|
+
|
|
2775
|
+
def _preflight_live_events(
|
|
2776
|
+
conn, records, hw, conflicts=None, protocol_scan=None
|
|
2777
|
+
):
|
|
2778
|
+
"""Validate unread evt/correction records before the stats transaction.
|
|
2779
|
+
|
|
2780
|
+
A READER (#374 §6): the evt records it inspects are already durably in the
|
|
2781
|
+
journal, so same-revision divergence must NOT raise here — that raise wedged
|
|
2782
|
+
every cycle over an already-poisoned journal, exactly like the rebuild. The
|
|
2783
|
+
divergent evt is DROPPED from the apply set, the prior effective event
|
|
2784
|
+
stands, and the group is appended to `conflicts` when a sink is supplied.
|
|
2785
|
+
`CorrectionRebuildRequired` stays fatal."""
|
|
2786
|
+
event_records = [record for record in records if record.get("t") == "evt"]
|
|
2787
|
+
selected_new = _lib_journal.resolve_effective_events(event_records)
|
|
2788
|
+
if conflicts is not None:
|
|
2789
|
+
conflicts.extend(selected_new.conflicts)
|
|
2790
|
+
to_apply = []
|
|
2791
|
+
for evt in selected_new.active:
|
|
2792
|
+
selected = selected_new.by_id[evt["id"]]
|
|
2793
|
+
prior = _metadata_row(conn, selected.event_id)
|
|
2794
|
+
if prior is None:
|
|
2795
|
+
to_apply.append(evt)
|
|
2796
|
+
continue
|
|
2797
|
+
prior_rev, prior_status, prior_hash, prior_batch = prior
|
|
2798
|
+
if int(prior_rev) == selected.rev:
|
|
2799
|
+
if prior_status != selected.status or prior_hash != selected.content_hash:
|
|
2800
|
+
if _legacy_qaa_can_advance(conn, selected):
|
|
2801
|
+
to_apply.append(evt)
|
|
2802
|
+
continue
|
|
2803
|
+
if conflicts is not None:
|
|
2804
|
+
conflicts.append(
|
|
2805
|
+
_lib_journal.EventConflict(
|
|
2806
|
+
event_id=selected.event_id,
|
|
2807
|
+
rev=selected.rev,
|
|
2808
|
+
content_hashes=tuple(
|
|
2809
|
+
sorted({prior_hash, selected.content_hash})),
|
|
2810
|
+
selected_hash=prior_hash,
|
|
2811
|
+
)
|
|
2812
|
+
)
|
|
2813
|
+
print(
|
|
2814
|
+
f"[journal] quarantined a divergent journal event for "
|
|
2815
|
+
f"{selected.event_id} rev {selected.rev}; the prior "
|
|
2816
|
+
"effective event stands",
|
|
2817
|
+
file=sys.stderr,
|
|
2818
|
+
)
|
|
2819
|
+
continue
|
|
2820
|
+
continue
|
|
2821
|
+
raise CorrectionRebuildRequired(
|
|
2822
|
+
f"event {selected.event_id} rev {selected.rev} conflicts with "
|
|
2823
|
+
f"effective rev {prior_rev} from {prior_batch or 'base journal'}",
|
|
2824
|
+
batch_id=prior_batch,
|
|
2825
|
+
event_id=selected.event_id,
|
|
2826
|
+
high_water=_correction_commit_high_water(prior_batch, hw),
|
|
2827
|
+
expected_metadata=(
|
|
2828
|
+
int(prior_rev),
|
|
2829
|
+
prior_status,
|
|
2830
|
+
prior_hash,
|
|
2831
|
+
prior_batch,
|
|
2832
|
+
),
|
|
2833
|
+
)
|
|
2834
|
+
|
|
2835
|
+
if any(
|
|
2836
|
+
record.get("t") in {"correction", "correction_batch"}
|
|
2837
|
+
or (
|
|
2838
|
+
record.get("t") == "op"
|
|
2839
|
+
and isinstance(record.get("payload"), dict)
|
|
2840
|
+
and record["payload"].get("kind")
|
|
2841
|
+
== _lib_journal._PROTOCOL_RESOLUTION_KIND
|
|
2842
|
+
)
|
|
2843
|
+
for record in records
|
|
2844
|
+
):
|
|
2845
|
+
full = _full_effective_selection(hw)
|
|
2846
|
+
if protocol_scan is not None:
|
|
2847
|
+
protocol_scan["scanned"] = True
|
|
2848
|
+
protocol_scan["violations"] = full.protocol_violations
|
|
2849
|
+
protocol_scan["acknowledged"] = (
|
|
2850
|
+
full.acknowledged_protocol_violations
|
|
2851
|
+
)
|
|
2852
|
+
for selected in full.by_id.values():
|
|
2853
|
+
if selected.batch_id is None:
|
|
2854
|
+
continue
|
|
2855
|
+
prior = _metadata_row(conn, selected.event_id)
|
|
2856
|
+
if prior is not None:
|
|
2857
|
+
prior_tuple = (int(prior[0]), prior[1], prior[2], prior[3])
|
|
2858
|
+
selected_tuple = (
|
|
2859
|
+
selected.rev,
|
|
2860
|
+
selected.status,
|
|
2861
|
+
selected.content_hash,
|
|
2862
|
+
selected.batch_id,
|
|
2863
|
+
)
|
|
2864
|
+
if prior_tuple == selected_tuple:
|
|
2865
|
+
continue
|
|
2866
|
+
raise CorrectionRebuildRequired(
|
|
2867
|
+
f"completed correction batch {selected.batch_id} requires "
|
|
2868
|
+
"stats index rebuild",
|
|
2869
|
+
batch_id=selected.batch_id,
|
|
2870
|
+
event_id=selected.event_id,
|
|
2871
|
+
high_water=_correction_commit_high_water(selected.batch_id, hw),
|
|
2872
|
+
expected_metadata=(
|
|
2873
|
+
selected.rev,
|
|
2874
|
+
selected.status,
|
|
2875
|
+
selected.content_hash,
|
|
2876
|
+
selected.batch_id,
|
|
2877
|
+
),
|
|
2878
|
+
recovery_eligible=True,
|
|
2879
|
+
)
|
|
2880
|
+
return to_apply
|
|
2881
|
+
|
|
2882
|
+
|
|
2883
|
+
# --------------------------------------------------------------------------
|
|
2884
|
+
# the cycle (spec §5.2, revision 3)
|
|
2885
|
+
# --------------------------------------------------------------------------
|
|
2886
|
+
|
|
2887
|
+
def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
|
|
2888
|
+
codex_apply=None, post_commit=None) -> IngestResult:
|
|
2889
|
+
# Step 1: HW snapshot (leaf lock, µs). Lines appended after this — by other
|
|
2890
|
+
# processes OR by this cycle's own evt emission — are past HW and belong to
|
|
1864
2891
|
# the next cycle (§5.2.1, closes the skipped-append race).
|
|
1865
2892
|
hw = journal_high_water()
|
|
1866
2893
|
# An empty journal (no segments yet) has nothing to consume. Normally that is
|
|
@@ -1909,7 +2936,17 @@ def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
|
|
|
1909
2936
|
|
|
1910
2937
|
records = [r for (r, _s, _o) in decoded]
|
|
1911
2938
|
batch = [r for r in records if r.get("t") in ("obs", "op")]
|
|
1912
|
-
|
|
2939
|
+
# #374: the preflight reader quarantines same-revision divergence instead of
|
|
2940
|
+
# raising; the groups it drops are counted on the cycle summary.
|
|
2941
|
+
preflight_conflicts: list = []
|
|
2942
|
+
protocol_scan: dict = {}
|
|
2943
|
+
journal_evts = _preflight_live_events(
|
|
2944
|
+
conn,
|
|
2945
|
+
records,
|
|
2946
|
+
cursor_target,
|
|
2947
|
+
conflicts=preflight_conflicts,
|
|
2948
|
+
protocol_scan=protocol_scan,
|
|
2949
|
+
)
|
|
1913
2950
|
|
|
1914
2951
|
# Step 4: ONE BEGIN IMMEDIATE — replay + pipeline + derived-fact journaling +
|
|
1915
2952
|
# cursor advance, atomic (§5.2 crash boundary). A crash before COMMIT rolls
|
|
@@ -1925,7 +2962,8 @@ def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
|
|
|
1925
2962
|
# before a referencing one (milestones); NO ctx, so replay is
|
|
1926
2963
|
# structurally unable to fire an alert (§5.2 step 4a).
|
|
1927
2964
|
for evt in sorted(journal_evts, key=_fold_order):
|
|
1928
|
-
|
|
2965
|
+
if _record_live_effective_event(conn, evt):
|
|
2966
|
+
_apply_evt(conn, evt)
|
|
1929
2967
|
# 4b. Per-record sequential pipeline over obs/op in canonical order —
|
|
1930
2968
|
# sequential is REQUIRED (reset/credit detection precedes the same
|
|
1931
2969
|
# record's snapshot-accept; a reset-spanning batch needs prior records'
|
|
@@ -1969,6 +3007,16 @@ def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
|
|
|
1969
3007
|
# no-op when the batch carries no account stamps (byte-stable on a
|
|
1970
3008
|
# pre-multi-account single-account install).
|
|
1971
3009
|
_derive_account_last_seen(conn, records)
|
|
3010
|
+
# A full correction-prefix preflight is authoritative for the
|
|
3011
|
+
# disposable protocol summary. Replace it in the same transaction as
|
|
3012
|
+
# the cursor so shallow Doctor paths observe either the old complete
|
|
3013
|
+
# result or the new complete result, never an in-between state.
|
|
3014
|
+
if protocol_scan.get("scanned"):
|
|
3015
|
+
_write_protocol_violations(
|
|
3016
|
+
conn,
|
|
3017
|
+
protocol_scan.get("violations", ()),
|
|
3018
|
+
protocol_scan.get("acknowledged", ()),
|
|
3019
|
+
)
|
|
1972
3020
|
# 4d. Advance the cursor (to HW, or to the cache-leg prefix boundary).
|
|
1973
3021
|
# `cursor_target is None` ONLY on a reconcile-only cycle over a still-
|
|
1974
3022
|
# empty journal (§5.2 above): there are no consumed lines to advance
|
|
@@ -2000,14 +3048,21 @@ def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
|
|
|
2000
3048
|
(ALERT_DISPATCHER or _dispatch_pending_alerts)(alerts)
|
|
2001
3049
|
|
|
2002
3050
|
return IngestResult(ran=True, consumed=len(records), malformed=malformed,
|
|
2003
|
-
events_emitted=ctx.events_emitted, alerts=alerts
|
|
3051
|
+
events_emitted=ctx.events_emitted, alerts=alerts,
|
|
3052
|
+
conflicts_dropped=(len(ctx.conflicts_dropped)
|
|
3053
|
+
+ len(preflight_conflicts)))
|
|
2004
3054
|
|
|
2005
3055
|
|
|
2006
|
-
def
|
|
2007
|
-
|
|
2008
|
-
|
|
2009
|
-
|
|
2010
|
-
|
|
3056
|
+
def _run_stats_ingest_once(
|
|
3057
|
+
*,
|
|
3058
|
+
mode: str = "opportunistic",
|
|
3059
|
+
timeout_s: float = 10.0,
|
|
3060
|
+
conn: sqlite3.Connection | None = None,
|
|
3061
|
+
reconcile_config=None,
|
|
3062
|
+
codex_apply=None,
|
|
3063
|
+
post_commit=None,
|
|
3064
|
+
) -> IngestResult:
|
|
3065
|
+
"""Run one single-flight attempt, without correction-recovery orchestration.
|
|
2011
3066
|
|
|
2012
3067
|
`mode="opportunistic"` takes the ingest lock non-blocking (busy → `ran=False`;
|
|
2013
3068
|
the current holder consumes the lines). `mode="authoritative"` waits up to
|
|
@@ -2040,17 +3095,112 @@ def run_stats_ingest(*, mode: str = "opportunistic", timeout_s: float = 10.0,
|
|
|
2040
3095
|
never broken; an AUTHORITATIVE ingest re-raises so its caller (record-usage,
|
|
2041
3096
|
record-credit, sync-week, statusline publication) sees the failure.
|
|
2042
3097
|
"""
|
|
2043
|
-
lock_fd = _acquire_ingest_lock(mode, timeout_s)
|
|
2044
|
-
if lock_fd is None:
|
|
2045
|
-
return IngestResult(ran=False, consumed=0, malformed=0,
|
|
2046
|
-
events_emitted=0, alerts=[])
|
|
2047
3098
|
own_conn = conn is None
|
|
3099
|
+
maintenance_fd = None
|
|
3100
|
+
lock_fd = None
|
|
2048
3101
|
try:
|
|
3102
|
+
# Let open_db resolve an epoch mismatch or classified corruption before
|
|
3103
|
+
# this caller owns any lock. It can therefore take maintenance EX ->
|
|
3104
|
+
# ingest in the required order. A fresh/legacy DB is different: its
|
|
3105
|
+
# open runs the one-time schema/cutover path, so serialize that whole
|
|
3106
|
+
# path under maintenance EX and downgrade to SH before taking ingest.
|
|
3107
|
+
# For a current/mismatched epoch, open first, then take maintenance SH
|
|
3108
|
+
# and verify the main-file identity did not change across the open; if
|
|
3109
|
+
# a sibling rebuilt in that gap, discard the stale handle and retry.
|
|
2049
3110
|
if own_conn:
|
|
2050
|
-
|
|
3111
|
+
while True:
|
|
3112
|
+
raw_epoch = _stats_db_user_version()
|
|
3113
|
+
if (
|
|
3114
|
+
raw_epoch is None
|
|
3115
|
+
or raw_epoch <= _cctally_core.LEGACY_STATS_HEAD
|
|
3116
|
+
):
|
|
3117
|
+
maintenance_fd = _acquire_maintenance_exclusive(
|
|
3118
|
+
mode, timeout_s
|
|
3119
|
+
)
|
|
3120
|
+
if maintenance_fd is None:
|
|
3121
|
+
return IngestResult(
|
|
3122
|
+
ran=False,
|
|
3123
|
+
consumed=0,
|
|
3124
|
+
malformed=0,
|
|
3125
|
+
events_emitted=0,
|
|
3126
|
+
alerts=[],
|
|
3127
|
+
)
|
|
3128
|
+
identity_before = _stats_db_identity()
|
|
3129
|
+
conn = _cctally_core.open_db()
|
|
3130
|
+
_downgrade_maintenance_shared(maintenance_fd)
|
|
3131
|
+
else:
|
|
3132
|
+
identity_before = _stats_db_identity()
|
|
3133
|
+
conn = _cctally_core.open_db()
|
|
3134
|
+
maintenance_fd = _acquire_maintenance_shared(
|
|
3135
|
+
mode, timeout_s
|
|
3136
|
+
)
|
|
3137
|
+
if maintenance_fd is None:
|
|
3138
|
+
if conn is not None:
|
|
3139
|
+
conn.close()
|
|
3140
|
+
conn = None
|
|
3141
|
+
return IngestResult(
|
|
3142
|
+
ran=False,
|
|
3143
|
+
consumed=0,
|
|
3144
|
+
malformed=0,
|
|
3145
|
+
events_emitted=0,
|
|
3146
|
+
alerts=[],
|
|
3147
|
+
)
|
|
3148
|
+
identity_after = _stats_db_identity()
|
|
3149
|
+
opened_epoch = conn.execute("PRAGMA user_version").fetchone()[0]
|
|
3150
|
+
epoch_ok = (
|
|
3151
|
+
opened_epoch <= _cctally_core.LEGACY_STATS_HEAD
|
|
3152
|
+
or opened_epoch == _cctally_core.STATS_INDEX_EPOCH
|
|
3153
|
+
)
|
|
3154
|
+
if (
|
|
3155
|
+
identity_before == identity_after
|
|
3156
|
+
and epoch_ok
|
|
3157
|
+
):
|
|
3158
|
+
break
|
|
3159
|
+
_release_maintenance_shared(maintenance_fd)
|
|
3160
|
+
maintenance_fd = None
|
|
3161
|
+
conn.close()
|
|
3162
|
+
conn = None
|
|
3163
|
+
else:
|
|
3164
|
+
maintenance_fd = _acquire_maintenance_shared(mode, timeout_s)
|
|
3165
|
+
if maintenance_fd is None:
|
|
3166
|
+
return IngestResult(
|
|
3167
|
+
ran=False,
|
|
3168
|
+
consumed=0,
|
|
3169
|
+
malformed=0,
|
|
3170
|
+
events_emitted=0,
|
|
3171
|
+
alerts=[],
|
|
3172
|
+
)
|
|
3173
|
+
|
|
3174
|
+
lock_fd = _acquire_ingest_lock(mode, timeout_s)
|
|
3175
|
+
if lock_fd is None:
|
|
3176
|
+
if own_conn and conn is not None:
|
|
3177
|
+
conn.close()
|
|
3178
|
+
conn = None
|
|
3179
|
+
return IngestResult(
|
|
3180
|
+
ran=False,
|
|
3181
|
+
consumed=0,
|
|
3182
|
+
malformed=0,
|
|
3183
|
+
events_emitted=0,
|
|
3184
|
+
alerts=[],
|
|
3185
|
+
)
|
|
2051
3186
|
try:
|
|
2052
|
-
|
|
2053
|
-
|
|
3187
|
+
# #386: declare the sanctioned steady-state write regime for the
|
|
3188
|
+
# duration of the cycle. Two consumers: the Stage 3 authorizer, and
|
|
3189
|
+
# `holds_ingest_lock()` — a corruption surfacing from INSIDE the
|
|
3190
|
+
# cycle (via a nested `open_db()`, e.g. the cross-DB stats read on
|
|
3191
|
+
# the quota leg) reaches the heal hook while this process already
|
|
3192
|
+
# owns journal.ingest.lock, and the heal must recognise itself as
|
|
3193
|
+
# the serialized writer rather than wait 5s for a lock it holds.
|
|
3194
|
+
import _cctally_store
|
|
3195
|
+
with _cctally_store.stats_write_scope("ingest", ingest_lock=True):
|
|
3196
|
+
return _run_cycle(conn, reconcile_config=reconcile_config,
|
|
3197
|
+
codex_apply=codex_apply,
|
|
3198
|
+
post_commit=post_commit)
|
|
3199
|
+
except CorrectionRebuildRequired:
|
|
3200
|
+
# The public boundary must unwind its transaction, internally owned
|
|
3201
|
+
# connection, ingest lock, and maintenance-shared lock before it can
|
|
3202
|
+
# seek maintenance EXCLUSIVE in total order.
|
|
3203
|
+
raise
|
|
2054
3204
|
except Exception as exc:
|
|
2055
3205
|
if mode == "authoritative":
|
|
2056
3206
|
raise
|
|
@@ -2065,7 +3215,242 @@ def run_stats_ingest(*, mode: str = "opportunistic", timeout_s: float = 10.0,
|
|
|
2065
3215
|
if own_conn and conn is not None:
|
|
2066
3216
|
conn.close()
|
|
2067
3217
|
finally:
|
|
2068
|
-
|
|
3218
|
+
if lock_fd is not None:
|
|
3219
|
+
_release_ingest_lock(lock_fd)
|
|
3220
|
+
if maintenance_fd is not None:
|
|
3221
|
+
_release_maintenance_shared(maintenance_fd)
|
|
3222
|
+
|
|
3223
|
+
|
|
3224
|
+
def _correction_recovery_guidance(cause) -> str:
|
|
3225
|
+
detail = str(cause)
|
|
3226
|
+
lower = detail.lower()
|
|
3227
|
+
holder = (
|
|
3228
|
+
"open handle" in lower
|
|
3229
|
+
or "family is still open" in lower
|
|
3230
|
+
or "open in process" in lower
|
|
3231
|
+
)
|
|
3232
|
+
prefix = ""
|
|
3233
|
+
if holder:
|
|
3234
|
+
prefix = (
|
|
3235
|
+
"stop the dashboard or other process holding stats.db open, then "
|
|
3236
|
+
)
|
|
3237
|
+
return (
|
|
3238
|
+
f"{detail}; {prefix}run `cctally db rebuild --db stats` and retry"
|
|
3239
|
+
)
|
|
3240
|
+
|
|
3241
|
+
|
|
3242
|
+
def _correction_error_result(error) -> IngestResult:
|
|
3243
|
+
print(
|
|
3244
|
+
f"[ingest] correction recovery declined, cursor unmoved: {error}",
|
|
3245
|
+
file=sys.stderr,
|
|
3246
|
+
)
|
|
3247
|
+
return IngestResult(
|
|
3248
|
+
ran=True,
|
|
3249
|
+
consumed=0,
|
|
3250
|
+
malformed=0,
|
|
3251
|
+
events_emitted=0,
|
|
3252
|
+
alerts=[],
|
|
3253
|
+
error=error,
|
|
3254
|
+
)
|
|
3255
|
+
|
|
3256
|
+
|
|
3257
|
+
def _correction_index_converged(signal: CorrectionRebuildRequired) -> bool:
|
|
3258
|
+
"""Revalidate the triggering effective row without open-time mutation."""
|
|
3259
|
+
if signal.event_id is None or signal.expected_metadata is None:
|
|
3260
|
+
return False
|
|
3261
|
+
path = pathlib.Path(_cctally_core.DB_PATH)
|
|
3262
|
+
if not path.exists():
|
|
3263
|
+
return False
|
|
3264
|
+
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True, timeout=5.0)
|
|
3265
|
+
try:
|
|
3266
|
+
row = conn.execute(
|
|
3267
|
+
"SELECT rev, status, content_hash, batch_id "
|
|
3268
|
+
"FROM journal_effective_events WHERE event_id = ?",
|
|
3269
|
+
(signal.event_id,),
|
|
3270
|
+
).fetchone()
|
|
3271
|
+
if row is None:
|
|
3272
|
+
return False
|
|
3273
|
+
return (
|
|
3274
|
+
int(row[0]),
|
|
3275
|
+
row[1],
|
|
3276
|
+
row[2],
|
|
3277
|
+
row[3],
|
|
3278
|
+
) == tuple(signal.expected_metadata)
|
|
3279
|
+
finally:
|
|
3280
|
+
conn.close()
|
|
3281
|
+
|
|
3282
|
+
|
|
3283
|
+
def _correction_scratch_mains() -> set[pathlib.Path]:
|
|
3284
|
+
path = pathlib.Path(_cctally_core.DB_PATH)
|
|
3285
|
+
prefix = path.name + ".rebuilding-"
|
|
3286
|
+
return {
|
|
3287
|
+
member
|
|
3288
|
+
for member in path.parent.glob(prefix + "*")
|
|
3289
|
+
if not member.name.endswith(("-wal", "-shm"))
|
|
3290
|
+
}
|
|
3291
|
+
|
|
3292
|
+
|
|
3293
|
+
def _cleanup_new_correction_scratches(before: set[pathlib.Path]) -> None:
|
|
3294
|
+
for scratch in _correction_scratch_mains() - before:
|
|
3295
|
+
try:
|
|
3296
|
+
_remove_db_family(scratch)
|
|
3297
|
+
except OSError:
|
|
3298
|
+
pass
|
|
3299
|
+
|
|
3300
|
+
|
|
3301
|
+
def _recover_completed_correction(
|
|
3302
|
+
signal: CorrectionRebuildRequired,
|
|
3303
|
+
*,
|
|
3304
|
+
mode: str,
|
|
3305
|
+
timeout_s: float,
|
|
3306
|
+
) -> None:
|
|
3307
|
+
"""Revalidate and, when still needed, replace through the trigger prefix."""
|
|
3308
|
+
if (
|
|
3309
|
+
signal.batch_id is None
|
|
3310
|
+
or signal.event_id is None
|
|
3311
|
+
or signal.high_water is None
|
|
3312
|
+
or signal.expected_metadata is None
|
|
3313
|
+
):
|
|
3314
|
+
raise CorrectionRecoveryError(
|
|
3315
|
+
_correction_recovery_guidance(
|
|
3316
|
+
"completed correction lacks an exact validated commit high-water"
|
|
3317
|
+
)
|
|
3318
|
+
)
|
|
3319
|
+
|
|
3320
|
+
maintenance_fd = _acquire_maintenance_exclusive(mode, timeout_s)
|
|
3321
|
+
if maintenance_fd is None:
|
|
3322
|
+
raise CorrectionRecoveryError(
|
|
3323
|
+
_correction_recovery_guidance(
|
|
3324
|
+
"stats maintenance lock is busy"
|
|
3325
|
+
)
|
|
3326
|
+
)
|
|
3327
|
+
ingest_fd = None
|
|
3328
|
+
try:
|
|
3329
|
+
ingest_fd = _acquire_ingest_lock(mode, timeout_s)
|
|
3330
|
+
if ingest_fd is None:
|
|
3331
|
+
raise CorrectionRecoveryError(
|
|
3332
|
+
_correction_recovery_guidance(
|
|
3333
|
+
"another ingest holds journal.ingest.lock"
|
|
3334
|
+
)
|
|
3335
|
+
)
|
|
3336
|
+
|
|
3337
|
+
# A sibling may have rebuilt after the original attempt unwound. The
|
|
3338
|
+
# locked re-check prevents redundant preservation/publication.
|
|
3339
|
+
try:
|
|
3340
|
+
converged = _correction_index_converged(signal)
|
|
3341
|
+
except Exception as exc:
|
|
3342
|
+
raise CorrectionRecoveryError(
|
|
3343
|
+
_correction_recovery_guidance(
|
|
3344
|
+
f"correction revalidation failed: {exc}"
|
|
3345
|
+
)
|
|
3346
|
+
) from exc
|
|
3347
|
+
if converged:
|
|
3348
|
+
return
|
|
3349
|
+
|
|
3350
|
+
import _cctally_db
|
|
3351
|
+
if _cctally_db._would_block_prod_stats(_cctally_core.DB_PATH):
|
|
3352
|
+
raise CorrectionRecoveryError(
|
|
3353
|
+
_correction_recovery_guidance(
|
|
3354
|
+
"refusing to rebuild the prod stats.db from a dev checkout"
|
|
3355
|
+
)
|
|
3356
|
+
)
|
|
3357
|
+
|
|
3358
|
+
import _cctally_store
|
|
3359
|
+
scratches_before = _correction_scratch_mains()
|
|
3360
|
+
try:
|
|
3361
|
+
with _cctally_store.stats_write_scope(
|
|
3362
|
+
"maintenance-correction-rebuild",
|
|
3363
|
+
ingest_lock=True,
|
|
3364
|
+
):
|
|
3365
|
+
rebuild_stats_index(high_water=signal.high_water)
|
|
3366
|
+
except BaseException as exc:
|
|
3367
|
+
_cleanup_new_correction_scratches(scratches_before)
|
|
3368
|
+
if isinstance(exc, (KeyboardInterrupt, SystemExit)):
|
|
3369
|
+
raise
|
|
3370
|
+
raise CorrectionRecoveryError(
|
|
3371
|
+
_correction_recovery_guidance(exc)
|
|
3372
|
+
) from exc
|
|
3373
|
+
finally:
|
|
3374
|
+
if ingest_fd is not None:
|
|
3375
|
+
_release_ingest_lock(ingest_fd)
|
|
3376
|
+
_release_maintenance_shared(maintenance_fd)
|
|
3377
|
+
|
|
3378
|
+
|
|
3379
|
+
def run_stats_ingest(
|
|
3380
|
+
*,
|
|
3381
|
+
mode: str = "opportunistic",
|
|
3382
|
+
timeout_s: float = 10.0,
|
|
3383
|
+
conn: sqlite3.Connection | None = None,
|
|
3384
|
+
reconcile_config=None,
|
|
3385
|
+
codex_apply=None,
|
|
3386
|
+
post_commit=None,
|
|
3387
|
+
) -> IngestResult:
|
|
3388
|
+
"""Run one cycle, healing one completed-correction mismatch when safe.
|
|
3389
|
+
|
|
3390
|
+
The initial attempt fully unwinds before recovery seeks maintenance
|
|
3391
|
+
EXCLUSIVE then ingest. Recovery revalidates, rebuilds through the exact
|
|
3392
|
+
triggering commit, releases both locks, and retries once on a freshly opened
|
|
3393
|
+
current-family connection. Caller-owned connections are never closed or
|
|
3394
|
+
replaced. A second correction signal is surfaced with the manual remedy.
|
|
3395
|
+
"""
|
|
3396
|
+
kwargs = {
|
|
3397
|
+
"mode": mode,
|
|
3398
|
+
"timeout_s": timeout_s,
|
|
3399
|
+
"conn": conn,
|
|
3400
|
+
"reconcile_config": reconcile_config,
|
|
3401
|
+
"codex_apply": codex_apply,
|
|
3402
|
+
"post_commit": post_commit,
|
|
3403
|
+
}
|
|
3404
|
+
try:
|
|
3405
|
+
return _run_stats_ingest_once(**kwargs)
|
|
3406
|
+
except CorrectionRebuildRequired as signal:
|
|
3407
|
+
if not signal.recovery_eligible:
|
|
3408
|
+
raise
|
|
3409
|
+
if conn is not None:
|
|
3410
|
+
raise CorrectionRebuildRequired(
|
|
3411
|
+
_correction_recovery_guidance(
|
|
3412
|
+
"automatic correction recovery cannot replace a "
|
|
3413
|
+
"caller-owned stats.db connection"
|
|
3414
|
+
),
|
|
3415
|
+
batch_id=signal.batch_id,
|
|
3416
|
+
event_id=signal.event_id,
|
|
3417
|
+
high_water=signal.high_water,
|
|
3418
|
+
expected_metadata=signal.expected_metadata,
|
|
3419
|
+
recovery_eligible=True,
|
|
3420
|
+
) from signal
|
|
3421
|
+
|
|
3422
|
+
try:
|
|
3423
|
+
_recover_completed_correction(
|
|
3424
|
+
signal,
|
|
3425
|
+
mode=mode,
|
|
3426
|
+
timeout_s=timeout_s,
|
|
3427
|
+
)
|
|
3428
|
+
except CorrectionRecoveryError as exc:
|
|
3429
|
+
if mode == "authoritative":
|
|
3430
|
+
raise
|
|
3431
|
+
return _correction_error_result(exc)
|
|
3432
|
+
|
|
3433
|
+
retry_kwargs = dict(kwargs)
|
|
3434
|
+
retry_kwargs["conn"] = None
|
|
3435
|
+
try:
|
|
3436
|
+
result = _run_stats_ingest_once(**retry_kwargs)
|
|
3437
|
+
except Exception as exc:
|
|
3438
|
+
wrapped = CorrectionRecoveryError(
|
|
3439
|
+
_correction_recovery_guidance(
|
|
3440
|
+
f"single correction-recovery retry failed: {exc}"
|
|
3441
|
+
)
|
|
3442
|
+
)
|
|
3443
|
+
if mode == "authoritative":
|
|
3444
|
+
raise wrapped from exc
|
|
3445
|
+
return _correction_error_result(wrapped)
|
|
3446
|
+
if result.error is not None:
|
|
3447
|
+
wrapped = CorrectionRecoveryError(
|
|
3448
|
+
_correction_recovery_guidance(
|
|
3449
|
+
f"single correction-recovery retry failed: {result.error}"
|
|
3450
|
+
)
|
|
3451
|
+
)
|
|
3452
|
+
return _correction_error_result(wrapped)
|
|
3453
|
+
return result
|
|
2069
3454
|
|
|
2070
3455
|
|
|
2071
3456
|
# ==========================================================================
|
|
@@ -2133,14 +3518,29 @@ class RebuildResult:
|
|
|
2133
3518
|
duration_s: float # wall time of the whole rebuild
|
|
2134
3519
|
segments_read: int # journal segments folded
|
|
2135
3520
|
lines_folded: int # op + evt lines applied (obs are rederive input)
|
|
2136
|
-
|
|
2137
|
-
|
|
2138
|
-
|
|
3521
|
+
# #374: divergent same-revision groups quarantined behind a lowest-sequence
|
|
3522
|
+
# provisional winner. The rebuild COMPLETES and exits 0 — reporting them is
|
|
3523
|
+
# how we refuse to assert that a guessed winner is authoritative.
|
|
3524
|
+
conflicts: tuple = ()
|
|
3525
|
+
# #402 Task A: whole correction batches omitted after one of the seven
|
|
3526
|
+
# enumerated structural violations. The usable index still publishes, but
|
|
3527
|
+
# every operator surface must report that intended corrections were omitted.
|
|
3528
|
+
protocol_violations: tuple = ()
|
|
3529
|
+
# #402 Task B: exact violations the operator acknowledged as omitted. The
|
|
3530
|
+
# batches remain tainted; this is diagnostic/audit state, never validity.
|
|
3531
|
+
acknowledged_protocol_violations: tuple = ()
|
|
3532
|
+
quarantine_dir: "pathlib.Path | None" = None
|
|
3533
|
+
|
|
3534
|
+
|
|
3535
|
+
def _remove_db_sidecars_strict(path) -> None:
|
|
3536
|
+
"""Remove both sidecars or fail before publishing a replacement main file."""
|
|
2139
3537
|
for suffix in ("-wal", "-shm"):
|
|
3538
|
+
candidate = pathlib.Path(str(path) + suffix)
|
|
2140
3539
|
try:
|
|
2141
|
-
|
|
2142
|
-
except
|
|
3540
|
+
candidate.unlink()
|
|
3541
|
+
except FileNotFoundError:
|
|
2143
3542
|
pass
|
|
3543
|
+
_fsync_dir(pathlib.Path(path).parent)
|
|
2144
3544
|
|
|
2145
3545
|
|
|
2146
3546
|
def _remove_db_family(path) -> None:
|
|
@@ -2151,6 +3551,426 @@ def _remove_db_family(path) -> None:
|
|
|
2151
3551
|
pass
|
|
2152
3552
|
|
|
2153
3553
|
|
|
3554
|
+
def _stats_rebuild_test_pause(point: str) -> None:
|
|
3555
|
+
"""Private process-control seam for the #388 interrupted-rebuild tests."""
|
|
3556
|
+
if os.environ.get("CCTALLY_TEST_STATS_REBUILD_PAUSE_AT") != point:
|
|
3557
|
+
return
|
|
3558
|
+
marker = os.environ.get("CCTALLY_TEST_STATS_REBUILD_MARKER")
|
|
3559
|
+
if not marker:
|
|
3560
|
+
return
|
|
3561
|
+
pathlib.Path(marker).write_text(f"{os.getpid()}\n")
|
|
3562
|
+
os.kill(os.getpid(), signal.SIGSTOP)
|
|
3563
|
+
|
|
3564
|
+
|
|
3565
|
+
_REBUILD_REQUIRED_TABLES = frozenset(
|
|
3566
|
+
{
|
|
3567
|
+
"accounts",
|
|
3568
|
+
"budget_milestones",
|
|
3569
|
+
"five_hour_block_models",
|
|
3570
|
+
"five_hour_block_projects",
|
|
3571
|
+
"five_hour_blocks",
|
|
3572
|
+
"five_hour_milestones",
|
|
3573
|
+
"five_hour_reset_events",
|
|
3574
|
+
"journal_cursor",
|
|
3575
|
+
"journal_effective_events",
|
|
3576
|
+
"journal_protocol_violations",
|
|
3577
|
+
"percent_milestones",
|
|
3578
|
+
"project_budget_milestones",
|
|
3579
|
+
"projected_milestones",
|
|
3580
|
+
"quota_alert_arming",
|
|
3581
|
+
"quota_percent_milestones",
|
|
3582
|
+
"quota_projection_state",
|
|
3583
|
+
"quota_threshold_events",
|
|
3584
|
+
"quota_window_blocks",
|
|
3585
|
+
"schema_migrations",
|
|
3586
|
+
"schema_migrations_skipped",
|
|
3587
|
+
"stats_open_fixups",
|
|
3588
|
+
"week_reset_events",
|
|
3589
|
+
"weekly_cost_snapshots",
|
|
3590
|
+
"weekly_credit_floors",
|
|
3591
|
+
"weekly_usage_snapshots",
|
|
3592
|
+
}
|
|
3593
|
+
)
|
|
3594
|
+
_REBUILD_REQUIRED_INDEXES = frozenset(
|
|
3595
|
+
{
|
|
3596
|
+
"idx_budget_milestones_journal_id",
|
|
3597
|
+
"idx_budget_milestones_journal_id_null",
|
|
3598
|
+
"idx_cost_week_start_at_time",
|
|
3599
|
+
"idx_cost_week_time",
|
|
3600
|
+
"idx_five_hour_block_models_block",
|
|
3601
|
+
"idx_five_hour_block_models_window",
|
|
3602
|
+
"idx_five_hour_block_projects_block",
|
|
3603
|
+
"idx_five_hour_block_projects_window",
|
|
3604
|
+
"idx_five_hour_blocks_block_start",
|
|
3605
|
+
"idx_five_hour_blocks_journal_id",
|
|
3606
|
+
"idx_five_hour_blocks_journal_id_null",
|
|
3607
|
+
"idx_five_hour_milestones_block",
|
|
3608
|
+
"idx_five_hour_milestones_journal_id",
|
|
3609
|
+
"idx_five_hour_milestones_journal_id_null",
|
|
3610
|
+
"idx_five_hour_reset_events_journal_id",
|
|
3611
|
+
"idx_five_hour_reset_events_journal_id_null",
|
|
3612
|
+
"idx_percent_milestones_journal_id",
|
|
3613
|
+
"idx_percent_milestones_journal_id_null",
|
|
3614
|
+
"idx_project_budget_milestones_journal_id",
|
|
3615
|
+
"idx_project_budget_milestones_journal_id_null",
|
|
3616
|
+
"idx_projected_milestones_journal_id",
|
|
3617
|
+
"idx_projected_milestones_journal_id_null",
|
|
3618
|
+
"idx_quota_blocks_active",
|
|
3619
|
+
"idx_quota_milestones_active",
|
|
3620
|
+
"idx_quota_threshold_events_active",
|
|
3621
|
+
"idx_usage_week_start_at_time",
|
|
3622
|
+
"idx_usage_week_time",
|
|
3623
|
+
"idx_week_reset_events_journal_id",
|
|
3624
|
+
"idx_week_reset_events_journal_id_null",
|
|
3625
|
+
"idx_weekly_cost_snapshots_journal_id",
|
|
3626
|
+
"idx_weekly_credit_floors_journal_id",
|
|
3627
|
+
"idx_weekly_usage_snapshots_5h_window_key",
|
|
3628
|
+
"idx_weekly_usage_snapshots_journal_id",
|
|
3629
|
+
}
|
|
3630
|
+
)
|
|
3631
|
+
# SHA-256 of the current epoch's non-internal table/index sqlite_schema rows,
|
|
3632
|
+
# ordered by (type, name). Unlike table-name checks, this catches a silently
|
|
3633
|
+
# omitted column, constraint, partial predicate, or index definition. An epoch
|
|
3634
|
+
# schema change must update this contract alongside STATS_INDEX_EPOCH.
|
|
3635
|
+
_REBUILD_SCHEMA_FINGERPRINT = (
|
|
3636
|
+
"3e0ec46a965c9fa10ac827cfd1656c66ecae50ed289e3aa1c34d2cf6a3e5c4a3"
|
|
3637
|
+
)
|
|
3638
|
+
|
|
3639
|
+
|
|
3640
|
+
def _stats_schema_fingerprint(conn: sqlite3.Connection) -> str:
|
|
3641
|
+
rows = [
|
|
3642
|
+
tuple(row)
|
|
3643
|
+
for row in conn.execute(
|
|
3644
|
+
"SELECT type, name, tbl_name, sql FROM sqlite_schema "
|
|
3645
|
+
"WHERE type IN ('table', 'index') "
|
|
3646
|
+
"AND name NOT LIKE 'sqlite_%' ORDER BY type, name"
|
|
3647
|
+
)
|
|
3648
|
+
]
|
|
3649
|
+
payload = json.dumps(rows, ensure_ascii=True, separators=(",", ":"))
|
|
3650
|
+
return hashlib.sha256(payload.encode("utf-8")).hexdigest()
|
|
3651
|
+
|
|
3652
|
+
|
|
3653
|
+
def _validate_rebuilt_stats_index(
|
|
3654
|
+
conn: sqlite3.Connection, high_water: "tuple[str, int] | None"
|
|
3655
|
+
) -> None:
|
|
3656
|
+
"""Validate the scratch index before it is eligible for publication."""
|
|
3657
|
+
integrity = [str(row[0]) for row in conn.execute("PRAGMA integrity_check")]
|
|
3658
|
+
if integrity != ["ok"]:
|
|
3659
|
+
raise JournalError(
|
|
3660
|
+
"rebuilt stats index failed integrity_check: " + "; ".join(integrity)
|
|
3661
|
+
)
|
|
3662
|
+
|
|
3663
|
+
epoch = int(conn.execute("PRAGMA user_version").fetchone()[0])
|
|
3664
|
+
if epoch != _cctally_core.STATS_INDEX_EPOCH:
|
|
3665
|
+
raise JournalError(
|
|
3666
|
+
f"rebuilt stats index has epoch {epoch}, expected "
|
|
3667
|
+
f"{_cctally_core.STATS_INDEX_EPOCH}"
|
|
3668
|
+
)
|
|
3669
|
+
|
|
3670
|
+
tables = {
|
|
3671
|
+
str(row[0])
|
|
3672
|
+
for row in conn.execute(
|
|
3673
|
+
"SELECT name FROM sqlite_schema "
|
|
3674
|
+
"WHERE type = 'table' AND name NOT LIKE 'sqlite_%'"
|
|
3675
|
+
)
|
|
3676
|
+
}
|
|
3677
|
+
missing_tables = sorted(_REBUILD_REQUIRED_TABLES - tables)
|
|
3678
|
+
unexpected_tables = sorted(tables - _REBUILD_REQUIRED_TABLES)
|
|
3679
|
+
if missing_tables or unexpected_tables:
|
|
3680
|
+
raise JournalError(
|
|
3681
|
+
"rebuilt stats index table contract mismatch"
|
|
3682
|
+
f"; missing={missing_tables!r}; unexpected={unexpected_tables!r}"
|
|
3683
|
+
)
|
|
3684
|
+
|
|
3685
|
+
indexes = {
|
|
3686
|
+
str(row[0])
|
|
3687
|
+
for row in conn.execute(
|
|
3688
|
+
"SELECT name FROM sqlite_schema "
|
|
3689
|
+
"WHERE type = 'index' AND name NOT LIKE 'sqlite_%'"
|
|
3690
|
+
)
|
|
3691
|
+
}
|
|
3692
|
+
missing_indexes = sorted(_REBUILD_REQUIRED_INDEXES - indexes)
|
|
3693
|
+
unexpected_indexes = sorted(indexes - _REBUILD_REQUIRED_INDEXES)
|
|
3694
|
+
if missing_indexes or unexpected_indexes:
|
|
3695
|
+
raise JournalError(
|
|
3696
|
+
"rebuilt stats index index contract mismatch"
|
|
3697
|
+
f"; missing={missing_indexes!r}; unexpected={unexpected_indexes!r}"
|
|
3698
|
+
)
|
|
3699
|
+
|
|
3700
|
+
schema_fingerprint = _stats_schema_fingerprint(conn)
|
|
3701
|
+
if schema_fingerprint != _REBUILD_SCHEMA_FINGERPRINT:
|
|
3702
|
+
raise JournalError(
|
|
3703
|
+
"rebuilt stats index schema definition mismatch: "
|
|
3704
|
+
f"{schema_fingerprint}, expected {_REBUILD_SCHEMA_FINGERPRINT}"
|
|
3705
|
+
)
|
|
3706
|
+
|
|
3707
|
+
# Force representative table and cursor B-tree reads. Header readability
|
|
3708
|
+
# and a constant-only SELECT do not establish that the index is usable.
|
|
3709
|
+
conn.execute(
|
|
3710
|
+
"SELECT id, journal_id FROM weekly_usage_snapshots "
|
|
3711
|
+
"ORDER BY id DESC LIMIT 1"
|
|
3712
|
+
).fetchall()
|
|
3713
|
+
cursor_row = conn.execute(
|
|
3714
|
+
"SELECT segment, offset, applied_segment, applied_offset "
|
|
3715
|
+
"FROM journal_cursor WHERE id = 1"
|
|
3716
|
+
).fetchone()
|
|
3717
|
+
actual_cursor = (
|
|
3718
|
+
(str(cursor_row[0]), int(cursor_row[1]))
|
|
3719
|
+
if cursor_row is not None
|
|
3720
|
+
else None
|
|
3721
|
+
)
|
|
3722
|
+
applied_cursor = (
|
|
3723
|
+
(str(cursor_row[2]), int(cursor_row[3]))
|
|
3724
|
+
if cursor_row is not None
|
|
3725
|
+
and cursor_row[2] is not None
|
|
3726
|
+
and cursor_row[3] is not None
|
|
3727
|
+
else None
|
|
3728
|
+
)
|
|
3729
|
+
if actual_cursor != high_water or applied_cursor != high_water:
|
|
3730
|
+
raise JournalError(
|
|
3731
|
+
"rebuilt stats index cursor contract "
|
|
3732
|
+
f"(public={actual_cursor!r}, applied={applied_cursor!r}) "
|
|
3733
|
+
f"does not match pinned journal high-water {high_water!r}"
|
|
3734
|
+
)
|
|
3735
|
+
|
|
3736
|
+
|
|
3737
|
+
def stats_index_matches_journal_prefix(
|
|
3738
|
+
path: pathlib.Path, high_water: "tuple[str, int] | None"
|
|
3739
|
+
) -> bool:
|
|
3740
|
+
"""Whether ``path`` is a fully valid materialization of ``high_water``.
|
|
3741
|
+
|
|
3742
|
+
This is intentionally stronger than "the index has rows": it validates the
|
|
3743
|
+
full Task A publication contract and compares the disposable effective-event
|
|
3744
|
+
summary with the canonical journal selection. A legitimate empty index
|
|
3745
|
+
therefore matches an empty selection, while a valid-looking empty/partial
|
|
3746
|
+
index over data-bearing journal events does not.
|
|
3747
|
+
"""
|
|
3748
|
+
if not pathlib.Path(path).exists():
|
|
3749
|
+
return False
|
|
3750
|
+
try:
|
|
3751
|
+
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
|
|
3752
|
+
try:
|
|
3753
|
+
_validate_rebuilt_stats_index(conn, high_water)
|
|
3754
|
+
decoded: list[dict] = []
|
|
3755
|
+
protocol_evidence = []
|
|
3756
|
+
prior_high_water = None
|
|
3757
|
+
if high_water is not None:
|
|
3758
|
+
for segment, offset, raw in _read_range(None, high_water):
|
|
3759
|
+
record = _lib_journal.decode_line(raw)
|
|
3760
|
+
if record is not None:
|
|
3761
|
+
_capture_protocol_prefix_evidence(
|
|
3762
|
+
record,
|
|
3763
|
+
prior_high_water,
|
|
3764
|
+
protocol_evidence,
|
|
3765
|
+
)
|
|
3766
|
+
decoded.append(record)
|
|
3767
|
+
prior_high_water = (
|
|
3768
|
+
segment,
|
|
3769
|
+
offset + len(raw) + 1,
|
|
3770
|
+
)
|
|
3771
|
+
cutover_claude = resolve_cutover_claude_account()
|
|
3772
|
+
for record in decoded:
|
|
3773
|
+
_normalize_legacy_account_stamp(record, cutover_claude)
|
|
3774
|
+
selection = _lib_journal.resolve_effective_events(
|
|
3775
|
+
decoded,
|
|
3776
|
+
protocol_prefix_evidence=protocol_evidence,
|
|
3777
|
+
)
|
|
3778
|
+
expected = []
|
|
3779
|
+
for event_id, selected in selection.by_id.items():
|
|
3780
|
+
event_json = None
|
|
3781
|
+
if selected.record is not None:
|
|
3782
|
+
event_json = (
|
|
3783
|
+
_lib_journal.encode_line(selected.record)
|
|
3784
|
+
.decode("utf-8")
|
|
3785
|
+
.rstrip("\n")
|
|
3786
|
+
)
|
|
3787
|
+
expected.append(
|
|
3788
|
+
(
|
|
3789
|
+
event_id,
|
|
3790
|
+
selected.rev,
|
|
3791
|
+
selected.status,
|
|
3792
|
+
selected.content_hash,
|
|
3793
|
+
selected.batch_id,
|
|
3794
|
+
event_json,
|
|
3795
|
+
)
|
|
3796
|
+
)
|
|
3797
|
+
expected.sort(key=lambda row: row[0])
|
|
3798
|
+
actual = [
|
|
3799
|
+
tuple(row)
|
|
3800
|
+
for row in conn.execute(
|
|
3801
|
+
"SELECT event_id, rev, status, content_hash, batch_id, "
|
|
3802
|
+
"event_json FROM journal_effective_events ORDER BY event_id"
|
|
3803
|
+
)
|
|
3804
|
+
]
|
|
3805
|
+
if actual != expected:
|
|
3806
|
+
return False
|
|
3807
|
+
expected_violation_rows = [
|
|
3808
|
+
*selection.protocol_violations,
|
|
3809
|
+
*selection.acknowledged_protocol_violations,
|
|
3810
|
+
]
|
|
3811
|
+
expected_violation_rows.sort(
|
|
3812
|
+
key=lambda violation: (
|
|
3813
|
+
violation.batch_id,
|
|
3814
|
+
violation.kind,
|
|
3815
|
+
violation.fingerprint,
|
|
3816
|
+
)
|
|
3817
|
+
)
|
|
3818
|
+
expected_violations = [
|
|
3819
|
+
json.dumps(
|
|
3820
|
+
violation.to_dict(),
|
|
3821
|
+
sort_keys=True,
|
|
3822
|
+
separators=(",", ":"),
|
|
3823
|
+
)
|
|
3824
|
+
for violation in expected_violation_rows
|
|
3825
|
+
]
|
|
3826
|
+
actual_violations = [
|
|
3827
|
+
str(row[0])
|
|
3828
|
+
for row in conn.execute(
|
|
3829
|
+
"SELECT violation_json FROM journal_protocol_violations "
|
|
3830
|
+
"ORDER BY batch_id, kind, fingerprint"
|
|
3831
|
+
)
|
|
3832
|
+
]
|
|
3833
|
+
if actual_violations != expected_violations:
|
|
3834
|
+
return False
|
|
3835
|
+
for record in selection.active:
|
|
3836
|
+
if record.get("t") != "evt":
|
|
3837
|
+
continue
|
|
3838
|
+
spec = _EVT_SPECS.get((record.get("payload") or {}).get("kind"))
|
|
3839
|
+
if spec is None or spec.table is None:
|
|
3840
|
+
continue
|
|
3841
|
+
row = conn.execute(
|
|
3842
|
+
f"SELECT id FROM {spec.table} WHERE journal_id = ?",
|
|
3843
|
+
(record["id"],),
|
|
3844
|
+
).fetchone()
|
|
3845
|
+
if row is None:
|
|
3846
|
+
return False
|
|
3847
|
+
if spec.applier is _apply_block_close:
|
|
3848
|
+
block_id = int(row[0])
|
|
3849
|
+
payload = record.get("payload") or {}
|
|
3850
|
+
for payload_key, child_table in _BLOCK_CHILDREN:
|
|
3851
|
+
child_count = conn.execute(
|
|
3852
|
+
f"SELECT COUNT(*) FROM {child_table} WHERE block_id = ?",
|
|
3853
|
+
(block_id,),
|
|
3854
|
+
).fetchone()[0]
|
|
3855
|
+
if int(child_count) != len(payload.get(payload_key) or ()):
|
|
3856
|
+
return False
|
|
3857
|
+
return True
|
|
3858
|
+
finally:
|
|
3859
|
+
conn.close()
|
|
3860
|
+
except (
|
|
3861
|
+
OSError,
|
|
3862
|
+
sqlite3.DatabaseError,
|
|
3863
|
+
JournalError,
|
|
3864
|
+
_lib_journal.JournalProtocolError,
|
|
3865
|
+
):
|
|
3866
|
+
return False
|
|
3867
|
+
|
|
3868
|
+
|
|
3869
|
+
def _prepare_existing_stats_for_cutover(path: pathlib.Path) -> None:
|
|
3870
|
+
"""Checkpoint a readable old index so removing its sidecars is kill-safe."""
|
|
3871
|
+
import _cctally_db
|
|
3872
|
+
|
|
3873
|
+
try:
|
|
3874
|
+
conn = sqlite3.connect(str(path), timeout=15.0)
|
|
3875
|
+
try:
|
|
3876
|
+
conn.execute("PRAGMA schema_version").fetchone()
|
|
3877
|
+
checkpoint = conn.execute("PRAGMA wal_checkpoint(TRUNCATE)").fetchone()
|
|
3878
|
+
if checkpoint is not None and int(checkpoint[0]) != 0:
|
|
3879
|
+
raise JournalError(
|
|
3880
|
+
"old stats index WAL could not be drained before cutover"
|
|
3881
|
+
)
|
|
3882
|
+
finally:
|
|
3883
|
+
conn.close()
|
|
3884
|
+
except sqlite3.DatabaseError as exc:
|
|
3885
|
+
# Auto-heal necessarily starts from an unreadable family. Preserve its
|
|
3886
|
+
# exact bytes below, then publish the already-validated replacement.
|
|
3887
|
+
if _cctally_db._is_sqlite_corruption_error(exc):
|
|
3888
|
+
return
|
|
3889
|
+
raise
|
|
3890
|
+
|
|
3891
|
+
|
|
3892
|
+
def _preserve_stats_family_for_cutover(path: pathlib.Path) -> pathlib.Path:
|
|
3893
|
+
"""Durably copy the old family into quarantine without removing the main."""
|
|
3894
|
+
import _cctally_db
|
|
3895
|
+
|
|
3896
|
+
root = _cctally_core.APP_DIR / "quarantine"
|
|
3897
|
+
root.mkdir(parents=True, exist_ok=True)
|
|
3898
|
+
# The quarantine entry itself must survive power loss before any old
|
|
3899
|
+
# sidecar can be removed. fsyncing only the new root/incident cannot make
|
|
3900
|
+
# the root's directory entry durable in APP_DIR.
|
|
3901
|
+
_fsync_dir(_cctally_core.APP_DIR)
|
|
3902
|
+
try:
|
|
3903
|
+
os.chmod(root, 0o700)
|
|
3904
|
+
except OSError:
|
|
3905
|
+
pass
|
|
3906
|
+
stamp = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%S_%f")
|
|
3907
|
+
incident = root / f"{path.name}-{stamp}"
|
|
3908
|
+
incident.mkdir(mode=0o700)
|
|
3909
|
+
destination = incident / path.name
|
|
3910
|
+
members = [
|
|
3911
|
+
pathlib.Path(str(path) + suffix).name
|
|
3912
|
+
for suffix in ("", "-wal", "-shm")
|
|
3913
|
+
if pathlib.Path(str(path) + suffix).exists()
|
|
3914
|
+
]
|
|
3915
|
+
if not members:
|
|
3916
|
+
raise OSError(f"no database family exists to preserve at {path}")
|
|
3917
|
+
_cctally_db._copy_db_family(path, destination)
|
|
3918
|
+
manifest = {
|
|
3919
|
+
"schemaVersion": 1,
|
|
3920
|
+
"quarantinedAtUtc": dt.datetime.now(dt.timezone.utc).isoformat(
|
|
3921
|
+
timespec="seconds"
|
|
3922
|
+
).replace("+00:00", "Z"),
|
|
3923
|
+
"originalPath": str(path),
|
|
3924
|
+
"movedFiles": members,
|
|
3925
|
+
"complete": True,
|
|
3926
|
+
"cutoverProtocol": "preserve-then-atomic-replace-v1",
|
|
3927
|
+
}
|
|
3928
|
+
_cctally_db._atomic_write_private_json(incident / "manifest.json", manifest)
|
|
3929
|
+
_fsync_dir(incident)
|
|
3930
|
+
_fsync_dir(root)
|
|
3931
|
+
return incident
|
|
3932
|
+
|
|
3933
|
+
|
|
3934
|
+
def _publish_rebuilt_stats_index(
|
|
3935
|
+
*,
|
|
3936
|
+
scratch: pathlib.Path,
|
|
3937
|
+
destination: pathlib.Path,
|
|
3938
|
+
preserve_existing: bool,
|
|
3939
|
+
before_swap=None,
|
|
3940
|
+
) -> "pathlib.Path | None":
|
|
3941
|
+
"""Publish one validated, closed, sidecar-free scratch index atomically."""
|
|
3942
|
+
import _cctally_store
|
|
3943
|
+
|
|
3944
|
+
family_exists = any(
|
|
3945
|
+
pathlib.Path(str(destination) + suffix).exists()
|
|
3946
|
+
for suffix in ("", "-wal", "-shm")
|
|
3947
|
+
)
|
|
3948
|
+
incident = None
|
|
3949
|
+
if family_exists:
|
|
3950
|
+
blocked = _cctally_store._stats_family_drained(destination)
|
|
3951
|
+
if blocked is not None:
|
|
3952
|
+
raise JournalError(f"stats.db cutover declined: {blocked}")
|
|
3953
|
+
_cctally_store._stats_storm_test_pause("stats_replace_drained")
|
|
3954
|
+
if preserve_existing:
|
|
3955
|
+
# Preserve the exact pre-cutover family, including a committed WAL
|
|
3956
|
+
# and SHM, before checkpointing mutates or removes those sidecars.
|
|
3957
|
+
incident = _preserve_stats_family_for_cutover(destination)
|
|
3958
|
+
if destination.exists():
|
|
3959
|
+
_prepare_existing_stats_for_cutover(destination)
|
|
3960
|
+
# The old main stays present and, when it was readable, fully
|
|
3961
|
+
# checkpointed. A kill from here until os.replace therefore still
|
|
3962
|
+
# leaves a usable old destination while preventing stale sidecars from
|
|
3963
|
+
# being paired with the replacement main.
|
|
3964
|
+
_remove_db_sidecars_strict(destination)
|
|
3965
|
+
|
|
3966
|
+
if before_swap is not None:
|
|
3967
|
+
before_swap()
|
|
3968
|
+
_stats_rebuild_test_pause("rebuild_before_cutover")
|
|
3969
|
+
os.replace(str(scratch), str(destination))
|
|
3970
|
+
_fsync_dir(destination.parent)
|
|
3971
|
+
return incident
|
|
3972
|
+
|
|
3973
|
+
|
|
2154
3974
|
def _rebuild_quota_cache_leg(records) -> None:
|
|
2155
3975
|
"""Re-materialize cache.db `quota_window_snapshots` from the journal's Codex
|
|
2156
3976
|
quota obs (spec §5.4). The journal obs are the DURABLE source (§1 latent
|
|
@@ -2209,7 +4029,13 @@ def _rebuild_quota_cache_leg(records) -> None:
|
|
|
2209
4029
|
release_cache_writer_flocks(held)
|
|
2210
4030
|
|
|
2211
4031
|
|
|
2212
|
-
def rebuild_stats_index(
|
|
4032
|
+
def rebuild_stats_index(
|
|
4033
|
+
*,
|
|
4034
|
+
target_path=None,
|
|
4035
|
+
high_water: "tuple[str, int] | None" = None,
|
|
4036
|
+
update_quota_cache: bool = True,
|
|
4037
|
+
before_swap=None,
|
|
4038
|
+
) -> RebuildResult:
|
|
2213
4039
|
"""Build a FRESH stats index from the journal alone (spec §5.4).
|
|
2214
4040
|
|
|
2215
4041
|
Replays every segment in canonical `(segment, offset)` order into a fresh
|
|
@@ -2219,10 +4045,14 @@ def rebuild_stats_index(*, target_path=None) -> RebuildResult:
|
|
|
2219
4045
|
no alerts, no `reconcile_config` (see the module note above). Post-rebuild the
|
|
2220
4046
|
cursor equals the journal high-water.
|
|
2221
4047
|
|
|
2222
|
-
`target_path` selects the destination (default `DB_PATH`).
|
|
2223
|
-
|
|
2224
|
-
|
|
2225
|
-
|
|
4048
|
+
`target_path` selects the destination (default `DB_PATH`). `high_water`
|
|
4049
|
+
optionally pins the exact inclusive journal prefix; later bytes stay beyond
|
|
4050
|
+
the rebuilt cursor. `update_quota_cache=False` is the Task-C Claude-only
|
|
4051
|
+
path whose caller already holds a stable cache exclusion. The common
|
|
4052
|
+
cutover keeps a live destination in place until its validated replacement
|
|
4053
|
+
is ready, preserves the old family, detaches old sidecars, and atomically
|
|
4054
|
+
replaces the main file. A `target_path` build uses the same atomic
|
|
4055
|
+
publication but does not create a live-family quarantine incident.
|
|
2226
4056
|
"""
|
|
2227
4057
|
start = time.monotonic()
|
|
2228
4058
|
dest = (pathlib.Path(target_path) if target_path is not None
|
|
@@ -2231,8 +4061,19 @@ def rebuild_stats_index(*, target_path=None) -> RebuildResult:
|
|
|
2231
4061
|
# HW snapshot at the START — lines appended during the rebuild are past HW
|
|
2232
4062
|
# and belong to the next ingest cycle (they replay idempotently); mirrors the
|
|
2233
4063
|
# live cycle's §5.2.1 HW-prefix rule.
|
|
2234
|
-
hw = journal_high_water()
|
|
4064
|
+
hw = high_water if high_water is not None else journal_high_water()
|
|
2235
4065
|
segments = list_segments()
|
|
4066
|
+
if hw is not None:
|
|
4067
|
+
if hw[0] not in segments:
|
|
4068
|
+
raise JournalError(
|
|
4069
|
+
f"rebuild high-water segment is missing: {hw[0]}"
|
|
4070
|
+
)
|
|
4071
|
+
current_size = os.path.getsize(_cctally_core.JOURNAL_DIR / hw[0])
|
|
4072
|
+
if hw[1] < 0 or hw[1] > current_size:
|
|
4073
|
+
raise JournalError(
|
|
4074
|
+
f"rebuild high-water offset is invalid for {hw[0]}: {hw[1]}"
|
|
4075
|
+
)
|
|
4076
|
+
segments = segments[:segments.index(hw[0]) + 1]
|
|
2236
4077
|
|
|
2237
4078
|
stamp = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%S_%f")
|
|
2238
4079
|
scratch = dest.with_name(dest.name + f".rebuilding-{stamp}")
|
|
@@ -2246,13 +4087,28 @@ def rebuild_stats_index(*, target_path=None) -> RebuildResult:
|
|
|
2246
4087
|
lines_folded = 0
|
|
2247
4088
|
try:
|
|
2248
4089
|
decoded: list = []
|
|
4090
|
+
protocol_evidence = []
|
|
4091
|
+
prior_high_water = None
|
|
2249
4092
|
if hw is not None:
|
|
2250
|
-
for
|
|
4093
|
+
for segment, offset, raw in _read_range(None, hw):
|
|
2251
4094
|
rec = _lib_journal.decode_line(raw)
|
|
2252
4095
|
if rec is None:
|
|
2253
4096
|
malformed += 1
|
|
4097
|
+
prior_high_water = (
|
|
4098
|
+
segment,
|
|
4099
|
+
offset + len(raw) + 1,
|
|
4100
|
+
)
|
|
2254
4101
|
continue
|
|
4102
|
+
_capture_protocol_prefix_evidence(
|
|
4103
|
+
rec,
|
|
4104
|
+
prior_high_water,
|
|
4105
|
+
protocol_evidence,
|
|
4106
|
+
)
|
|
2255
4107
|
decoded.append(rec)
|
|
4108
|
+
prior_high_water = (
|
|
4109
|
+
segment,
|
|
4110
|
+
offset + len(raw) + 1,
|
|
4111
|
+
)
|
|
2256
4112
|
|
|
2257
4113
|
# Legacy account normalisation (#341, spec §2 / handoff item 2): a
|
|
2258
4114
|
# pre-#341 real-account line lacks an account stamp — inject the cutover
|
|
@@ -2265,9 +4121,18 @@ def rebuild_stats_index(*, target_path=None) -> RebuildResult:
|
|
|
2265
4121
|
for rec in decoded:
|
|
2266
4122
|
_normalize_legacy_account_stamp(rec, cutover_claude)
|
|
2267
4123
|
|
|
4124
|
+
# Resolve corrections BEFORE either disposable index is mutated. A
|
|
4125
|
+
# malformed revision, divergent same-revision candidate, or invalid
|
|
4126
|
+
# committed manifest leaves the existing destination untouched.
|
|
4127
|
+
effective = _lib_journal.resolve_effective_events(
|
|
4128
|
+
decoded,
|
|
4129
|
+
protocol_prefix_evidence=protocol_evidence,
|
|
4130
|
+
)
|
|
4131
|
+
|
|
2268
4132
|
# Cache leg BEFORE any stats txn (provider-flock lock-order): journal
|
|
2269
4133
|
# Codex quota obs -> cache.db quota_window_snapshots.
|
|
2270
|
-
|
|
4134
|
+
if update_quota_cache:
|
|
4135
|
+
_rebuild_quota_cache_leg(decoded)
|
|
2271
4136
|
|
|
2272
4137
|
# One ordered fold stream: op-folds (order 5) + evts, keyed by
|
|
2273
4138
|
# (fold_order, canonical seq) so referenced families resolve before
|
|
@@ -2278,16 +4143,19 @@ def rebuild_stats_index(*, target_path=None) -> RebuildResult:
|
|
|
2278
4143
|
kind = (rec.get("payload") or {}).get("kind")
|
|
2279
4144
|
if t == "op" and kind in FOLD_APPLIERS:
|
|
2280
4145
|
stream.append((_OP_FOLD_ORDER, seq, "op", rec))
|
|
2281
|
-
|
|
2282
|
-
|
|
4146
|
+
for seq, rec in enumerate(effective.active):
|
|
4147
|
+
stream.append((_fold_order(rec), seq, "evt", rec))
|
|
2283
4148
|
stream.sort(key=lambda x: (x[0], x[1]))
|
|
2284
4149
|
structural = [s for s in stream if s[0] < _REBUILD_MILESTONE_ORDER]
|
|
2285
4150
|
tail = [s for s in stream if s[0] >= _REBUILD_MILESTONE_ORDER]
|
|
2286
4151
|
|
|
4152
|
+
_stats_rebuild_test_pause("rebuild_fold_started")
|
|
4153
|
+
|
|
2287
4154
|
# Phase 1 (txn A) — structural folds: op floors, snapshot_accept, cost
|
|
2288
4155
|
# snapshots, resets+suppression, block_close, arming, credit effects.
|
|
2289
4156
|
conn.execute("BEGIN IMMEDIATE")
|
|
2290
4157
|
try:
|
|
4158
|
+
_write_effective_metadata(conn, effective)
|
|
2291
4159
|
for _order, _seq, kind, rec in structural:
|
|
2292
4160
|
if kind == "op":
|
|
2293
4161
|
FOLD_APPLIERS[(rec.get("payload") or {}).get("kind")](conn, rec)
|
|
@@ -2350,21 +4218,35 @@ def rebuild_stats_index(*, target_path=None) -> RebuildResult:
|
|
|
2350
4218
|
except sqlite3.Error:
|
|
2351
4219
|
rows_by_table[tbl] = 0
|
|
2352
4220
|
# Drain the WAL into the main file so the atomic rename carries all data.
|
|
2353
|
-
conn.execute("PRAGMA wal_checkpoint(TRUNCATE)")
|
|
4221
|
+
checkpoint = conn.execute("PRAGMA wal_checkpoint(TRUNCATE)").fetchone()
|
|
4222
|
+
if checkpoint is not None and int(checkpoint[0]) != 0:
|
|
4223
|
+
raise JournalError("rebuilt stats index WAL could not be drained")
|
|
4224
|
+
_validate_rebuilt_stats_index(conn, hw)
|
|
4225
|
+
_stats_rebuild_test_pause("rebuild_scratch_complete")
|
|
2354
4226
|
finally:
|
|
2355
4227
|
conn.close()
|
|
2356
4228
|
|
|
2357
|
-
#
|
|
2358
|
-
|
|
2359
|
-
|
|
2360
|
-
|
|
2361
|
-
|
|
2362
|
-
|
|
4229
|
+
# Closed, drained, validated, and durable before the old family is touched.
|
|
4230
|
+
_remove_db_sidecars_strict(scratch)
|
|
4231
|
+
with scratch.open("rb") as handle:
|
|
4232
|
+
os.fsync(handle.fileno())
|
|
4233
|
+
_fsync_dir(scratch.parent)
|
|
4234
|
+
incident = _publish_rebuilt_stats_index(
|
|
4235
|
+
scratch=scratch,
|
|
4236
|
+
destination=dest,
|
|
4237
|
+
preserve_existing=target_path is None,
|
|
4238
|
+
before_swap=before_swap,
|
|
4239
|
+
)
|
|
2363
4240
|
|
|
2364
4241
|
return RebuildResult(
|
|
2365
4242
|
rows_by_table=rows_by_table, malformed=malformed,
|
|
2366
4243
|
duration_s=time.monotonic() - start, segments_read=len(segments),
|
|
2367
|
-
lines_folded=lines_folded,
|
|
4244
|
+
lines_folded=lines_folded, conflicts=effective.conflicts,
|
|
4245
|
+
protocol_violations=effective.protocol_violations,
|
|
4246
|
+
acknowledged_protocol_violations=(
|
|
4247
|
+
effective.acknowledged_protocol_violations
|
|
4248
|
+
),
|
|
4249
|
+
quarantine_dir=incident,
|
|
2368
4250
|
)
|
|
2369
4251
|
|
|
2370
4252
|
|
|
@@ -2396,10 +4278,26 @@ def rebuild_stats_index(*, target_path=None) -> RebuildResult:
|
|
|
2396
4278
|
# the stamping runs; a re-run after any crash re-exports byte-identical lines
|
|
2397
4279
|
# (ids are `b:<table>:<rowid>`, independent of the retry's timestamp), so a
|
|
2398
4280
|
# duplicate/leftover bootstrap folds idempotently (`INSERT OR IGNORE`). The
|
|
2399
|
-
# cutover does NOT take the ingest lock
|
|
2400
|
-
#
|
|
2401
|
-
#
|
|
2402
|
-
#
|
|
4281
|
+
# cutover does NOT take the ingest lock. **The conclusion is right; the reason
|
|
4282
|
+
# once written here was false and is corrected (#386).** It was: "open_db
|
|
4283
|
+
# reaches it from INSIDE run_stats_ingest's own ingest-lock hold — re-acquiring
|
|
4284
|
+
# would self-deadlock." It does not: `run_stats_ingest` takes maintenance
|
|
4285
|
+
# EXCLUSIVE *before* `open_db()` on the legacy/fresh branch and only acquires
|
|
4286
|
+
# `journal.ingest.lock` after `open_db` has returned, so cutover never runs
|
|
4287
|
+
# under an ingest hold from that path.
|
|
4288
|
+
#
|
|
4289
|
+
# The real reason is that cutover runs under maintenance-EXCLUSIVE — since #386,
|
|
4290
|
+
# unconditionally, via `_cctally_store.stats_open_time_guard()` around
|
|
4291
|
+
# `open_db`'s whole open-time mutation region, whichever command reached it.
|
|
4292
|
+
# Maintenance-exclusive already serializes it against ingest, so the ingest lock
|
|
4293
|
+
# would add nothing; single-flight of the STAMP is provided by `BEGIN
|
|
4294
|
+
# IMMEDIATE`, and concurrent cutovers converge by id.
|
|
4295
|
+
#
|
|
4296
|
+
# DO NOT "fix" this by making cutover take the ingest lock. `holds_ingest_lock()`
|
|
4297
|
+
# now exists, and the deleted sentence reads like an invitation to add the
|
|
4298
|
+
# acquire with a re-entrancy check. Taking maintenance then ingest here would be
|
|
4299
|
+
# in lock order and would not deadlock — it would simply be a second, redundant
|
|
4300
|
+
# lock on a path that already holds the stronger one.
|
|
2403
4301
|
|
|
2404
4302
|
|
|
2405
4303
|
def _cutover_iso(dt_utc: dt.datetime) -> str:
|
|
@@ -2434,9 +4332,8 @@ class _CutoverSpec:
|
|
|
2434
4332
|
# (quota_alert_arming): its fold applier converges by NATURAL-KEY upsert, so
|
|
2435
4333
|
# there is nothing to stamp back and it is excluded from the no-NULL-survivors
|
|
2436
4334
|
# invariant (§8). When `natural_key_id` is set, the exported evt id is the
|
|
2437
|
-
#
|
|
2438
|
-
# emission
|
|
2439
|
-
# one id) instead of the `b:<table>:<rowid>` bootstrap id.
|
|
4335
|
+
# state-instance form (`<natural_key_prefix>:<col>:<col>…`) matching the LIVE
|
|
4336
|
+
# emission, instead of the `b:<table>:<rowid>` bootstrap id.
|
|
2440
4337
|
stamp: bool = True
|
|
2441
4338
|
natural_key_prefix: str = "" # evt_id kind prefix (e.g. "qaa")
|
|
2442
4339
|
natural_key_id: tuple = () # columns forming the natural-key evt id
|
|
@@ -2480,23 +4377,18 @@ _CUTOVER_SPECS = (
|
|
|
2480
4377
|
# forward-only alert clock (`activated_at_utc`) that must survive rebuild so
|
|
2481
4378
|
# the reconcile honors it (no historical re-fires). No journal_id column →
|
|
2482
4379
|
# NOT stamped; the fold applier upserts by natural key. The evt id is the
|
|
2483
|
-
# `qaa:`
|
|
2484
|
-
# `_cctally_quota._codex_leg._emit_arming`)
|
|
2485
|
-
#
|
|
4380
|
+
# `qaa:` state-instance form (matching the live emission in
|
|
4381
|
+
# `_cctally_quota._codex_leg._emit_arming`): the natural row key is followed
|
|
4382
|
+
# by fingerprint + activation boundary so distinct state transitions never
|
|
4383
|
+
# collide at rev 0, while exact re-emission of one state converges.
|
|
2486
4384
|
_CutoverSpec("quota_alert_arming", "quota_alert_arming", "evt",
|
|
2487
4385
|
"activated_at_utc", stamp=False, natural_key_prefix="qaa",
|
|
2488
4386
|
natural_key_id=("source", "source_root_key", "account_key",
|
|
2489
4387
|
"logical_limit_key", "observed_slot",
|
|
2490
|
-
"window_minutes"
|
|
4388
|
+
"window_minutes", "rule_fingerprint",
|
|
4389
|
+
"activated_at_utc")),
|
|
2491
4390
|
)
|
|
2492
4391
|
|
|
2493
|
-
# Journal-covered stats tables whose rows get a `journal_id` stamp at cutover.
|
|
2494
|
-
# five_hour_blocks stamps only its CLOSED rows (open blocks stay NULL — they are
|
|
2495
|
-
# re-materialized projections). `stamp=False` families (quota_alert_arming: no
|
|
2496
|
-
# journal_id column) are excluded — they converge by natural-key upsert.
|
|
2497
|
-
_CUTOVER_STAMP_TABLES = tuple(s.table for s in _CUTOVER_SPECS if s.stamp)
|
|
2498
|
-
|
|
2499
|
-
|
|
2500
4392
|
def _export_stats_table(conn, spec) -> list:
|
|
2501
4393
|
"""Return `[(line_record, rowid), ...]` for every row of `spec.table`
|
|
2502
4394
|
(closed rows only when `spec.closed_only`). Bootstrap id = b:<table>:<rowid>;
|
|
@@ -2528,13 +4420,13 @@ def _export_stats_table(conn, spec) -> list:
|
|
|
2528
4420
|
payload[payload_key] = [
|
|
2529
4421
|
{k: cr[k] for k in cr.keys() if k not in ("id", "block_id")}
|
|
2530
4422
|
for cr in child_rows]
|
|
4423
|
+
if spec.kind == "quota_alert_arming":
|
|
4424
|
+
payload["journal_identity_version"] = 2
|
|
2531
4425
|
if spec.natural_key_id:
|
|
2532
|
-
# §5.3 "state" family: the evt id is the
|
|
2533
|
-
# the live emission), NOT the b:<table>:<rowid> bootstrap id. A
|
|
2534
|
-
# (pre-#341) stats.db has no `account_key` column, so
|
|
2535
|
-
# component
|
|
2536
|
-
# qaa id becomes `qaa:...:unattributed:...`, matching what a live
|
|
2537
|
-
# re-emission for the unattributed identity would build.
|
|
4426
|
+
# §5.3 "state" family: the evt id is the state-instance form (matching
|
|
4427
|
+
# the live emission), NOT the b:<table>:<rowid> bootstrap id. A
|
|
4428
|
+
# legacy (pre-#341) stats.db has no `account_key` column, so that
|
|
4429
|
+
# missing component is the sentinel (#341).
|
|
2538
4430
|
row_cols = set(row.keys())
|
|
2539
4431
|
bid = _lib_journal.evt_id(
|
|
2540
4432
|
spec.natural_key_prefix,
|
|
@@ -2704,9 +4596,10 @@ def run_cutover(conn, *, now_utc: dt.datetime | None = None) -> "str | None":
|
|
|
2704
4596
|
# Account epoch-transition coordinator (#341, spec §2)
|
|
2705
4597
|
# ==========================================================================
|
|
2706
4598
|
#
|
|
2707
|
-
#
|
|
2708
|
-
#
|
|
2709
|
-
#
|
|
4599
|
+
# The epoch-transition coordinator was introduced when 1000 -> 1001 added the
|
|
4600
|
+
# account dimension. Later disposable-index epoch bumps reuse it idempotently:
|
|
4601
|
+
# `resolve_stats_epoch_mismatch` runs this coordinator BEFORE the rebuild, in
|
|
4602
|
+
# exact order (spec §2, review finding 1):
|
|
2710
4603
|
# (1) resolve the cutover identity WITHOUT opening stats.db — a stable-read of
|
|
2711
4604
|
# ~/.claude.json; stably-absent / torn -> `unattributed` (never a guess);
|
|
2712
4605
|
# (2) atomically check/append the canonical cutover op (stable semantic id
|
|
@@ -2796,7 +4689,8 @@ def run_epoch_transition(*, claude_json_path=None) -> str:
|
|
|
2796
4689
|
identity, check/append the canonical cutover op, THEN rebuild — in that exact
|
|
2797
4690
|
order, so the op is inside the rebuild's input. Returns the resolved
|
|
2798
4691
|
``claude_legacy_account``. Exposed for tests; the epoch-mismatch path calls
|
|
2799
|
-
it
|
|
4692
|
+
it under the maintenance + ingest locks, and the rebuild's common cutover
|
|
4693
|
+
preserves the old index only after the replacement is validated."""
|
|
2800
4694
|
claude_key = _resolve_claude_cutover_identity(claude_json_path)
|
|
2801
4695
|
recorded = append_accounts_cutover_op(claude_key)
|
|
2802
4696
|
rebuild_stats_index()
|