cctally 1.82.1 → 1.83.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/CHANGELOG.md +58 -0
  2. package/README.md +12 -5
  3. package/bin/_cctally_alerts.py +8 -1
  4. package/bin/_cctally_cache.py +912 -149
  5. package/bin/_cctally_config.py +43 -4
  6. package/bin/_cctally_core.py +933 -759
  7. package/bin/_cctally_dashboard.py +157 -47
  8. package/bin/_cctally_dashboard_cache_report.py +13 -6
  9. package/bin/_cctally_dashboard_conversation.py +1 -0
  10. package/bin/_cctally_dashboard_envelope.py +116 -8
  11. package/bin/_cctally_dashboard_share.py +50 -19
  12. package/bin/_cctally_dashboard_sources.py +223 -48
  13. package/bin/_cctally_db.py +605 -128
  14. package/bin/_cctally_doctor.py +413 -28
  15. package/bin/_cctally_five_hour.py +12 -5
  16. package/bin/_cctally_journal.py +2050 -156
  17. package/bin/_cctally_journal_repair.py +519 -0
  18. package/bin/_cctally_milestone_history.py +142 -56
  19. package/bin/_cctally_milestones.py +179 -111
  20. package/bin/_cctally_parser.py +42 -0
  21. package/bin/_cctally_project.py +24 -18
  22. package/bin/_cctally_quota.py +139 -25
  23. package/bin/_cctally_record.py +279 -108
  24. package/bin/_cctally_rederive.py +1052 -0
  25. package/bin/_cctally_reporting.py +58 -53
  26. package/bin/_cctally_setup.py +1 -0
  27. package/bin/_cctally_source_analytics.py +4 -1
  28. package/bin/_cctally_statusline.py +11 -11
  29. package/bin/_cctally_store.py +1039 -31
  30. package/bin/_cctally_sync_week.py +17 -8
  31. package/bin/_cctally_tui.py +350 -44
  32. package/bin/_cctally_update.py +133 -8
  33. package/bin/_cctally_weekrefs.py +14 -0
  34. package/bin/_lib_aggregators.py +10 -6
  35. package/bin/_lib_cache_report.py +101 -9
  36. package/bin/_lib_codex_pools.py +82 -0
  37. package/bin/_lib_conversation_query.py +81 -33
  38. package/bin/_lib_dashboard_sources.py +75 -0
  39. package/bin/_lib_diff_kernel.py +28 -15
  40. package/bin/_lib_doctor.py +342 -4
  41. package/bin/_lib_journal.py +924 -2
  42. package/bin/_lib_jsonl.py +43 -14
  43. package/bin/_lib_pricing.py +140 -21
  44. package/bin/_lib_rederive.py +395 -0
  45. package/bin/_lib_share.py +58 -2
  46. package/bin/cctally +56 -8
  47. package/dashboard/static/assets/{index-BKM43pxK.js → index-3bgCMVHb.js} +52 -52
  48. package/dashboard/static/assets/index-D27EIHEI.css +1 -0
  49. package/dashboard/static/dashboard.html +2 -2
  50. package/package.json +5 -1
  51. package/dashboard/static/assets/index-Dk1nplOz.css +0 -1
@@ -27,9 +27,11 @@ from __future__ import annotations
27
27
 
28
28
  import datetime as dt
29
29
  import fcntl
30
+ import hashlib
30
31
  import json
31
32
  import os
32
33
  import pathlib
34
+ import signal
33
35
  import sqlite3
34
36
  import sys
35
37
  import time
@@ -55,12 +57,43 @@ _QUOTA_DEDUP_INDEX_NAME = ".quota-observation-keys"
55
57
  _QUOTA_DEDUP_DIR: str | None = None
56
58
  _QUOTA_DEDUP_KEYS: set[str] = set()
57
59
  _QUOTA_DEDUP_LOADED = False
60
+ _HIGH_WATER_UNSET = object()
58
61
 
59
62
 
60
63
  class JournalError(Exception):
61
64
  """A structural journal-append failure (line too long, unrepairable tail)."""
62
65
 
63
66
 
67
+ class CorrectionRebuildRequired(JournalError):
68
+ """A completed correction cannot be applied incrementally to a live index.
69
+
70
+ The recovery boundary needs more than a message: it must rebuild through
71
+ the exact completed-batch commit that triggered the mismatch, then
72
+ revalidate the exact effective metadata under exclusive ownership.
73
+ """
74
+
75
+ def __init__(
76
+ self,
77
+ message,
78
+ *,
79
+ batch_id=None,
80
+ event_id=None,
81
+ high_water=None,
82
+ expected_metadata=None,
83
+ recovery_eligible=False,
84
+ ):
85
+ super().__init__(message)
86
+ self.batch_id = batch_id
87
+ self.event_id = event_id
88
+ self.high_water = high_water
89
+ self.expected_metadata = expected_metadata
90
+ self.recovery_eligible = recovery_eligible
91
+
92
+
93
+ class CorrectionRecoveryError(JournalError):
94
+ """Bounded correction recovery could not safely replace the live index."""
95
+
96
+
64
97
  # --------------------------------------------------------------------------
65
98
  # leaf lock
66
99
  # --------------------------------------------------------------------------
@@ -338,6 +371,92 @@ def append_record(
338
371
  _release_leaf_lock(lock_fd)
339
372
 
340
373
 
374
+ def append_records(
375
+ records: list[dict],
376
+ *,
377
+ now_utc: dt.datetime | None = None,
378
+ expected_high_water=_HIGH_WATER_UNSET,
379
+ line_hook=None,
380
+ ) -> tuple[str, int]:
381
+ """Append one ordered record group under a single leaf-lock hold.
382
+
383
+ The group is not transactionally atomic across a power loss: a crash can
384
+ leave complete prefix lines plus one torn final line, exactly like the
385
+ single-record appender. It *is* non-interleavable with other appenders, so a
386
+ correction batch remains physically ordered. ``expected_high_water`` is
387
+ checked while holding the same leaf lock that performs the append, closing
388
+ the plan/revalidate/append race.
389
+ """
390
+ if not isinstance(records, list) or not records:
391
+ raise ValueError("journal record group must be a non-empty list")
392
+ if now_utc is None:
393
+ now_utc = dt.datetime.now(dt.timezone.utc)
394
+ encoded = []
395
+ for record in records:
396
+ data = _lib_journal.encode_line(record)
397
+ if len(data) > _MAX_LINE_BYTES:
398
+ raise JournalError(
399
+ f"journal line is {len(data)} bytes, exceeds the "
400
+ f"{_MAX_LINE_BYTES}-byte limit (spec §4.3)"
401
+ )
402
+ encoded.append(data)
403
+
404
+ journal_dir = _cctally_core.JOURNAL_DIR
405
+ seg_name = _lib_journal.segment_name(now_utc)
406
+ seg_path = journal_dir / seg_name
407
+ dir_created = not journal_dir.exists()
408
+ journal_dir.mkdir(parents=True, exist_ok=True)
409
+ if dir_created:
410
+ try:
411
+ os.chmod(journal_dir, 0o700)
412
+ except OSError:
413
+ pass
414
+
415
+ lock_fd = _acquire_leaf_lock()
416
+ try:
417
+ segments = list_segments()
418
+ actual_high_water = None
419
+ if segments:
420
+ latest = segments[-1]
421
+ actual_high_water = (
422
+ latest,
423
+ os.stat(journal_dir / latest).st_size,
424
+ )
425
+ if (
426
+ expected_high_water is not _HIGH_WATER_UNSET
427
+ and actual_high_water != expected_high_water
428
+ ):
429
+ raise JournalError(
430
+ "journal high-water changed before correction append "
431
+ f"(expected {expected_high_water!r}, found {actual_high_water!r})"
432
+ )
433
+
434
+ seg_created = not seg_path.exists()
435
+ fd = os.open(str(seg_path), os.O_RDWR | os.O_APPEND | os.O_CREAT, 0o600)
436
+ try:
437
+ if seg_created:
438
+ try:
439
+ os.fchmod(fd, 0o600)
440
+ except OSError:
441
+ pass
442
+ _repair_torn_tail(fd)
443
+ for index, data in enumerate(encoded, start=1):
444
+ _write_all(fd, data)
445
+ if line_hook is not None:
446
+ line_hook(index)
447
+ os.fsync(fd)
448
+ end_offset = os.fstat(fd).st_size
449
+ finally:
450
+ os.close(fd)
451
+ if seg_created:
452
+ _fsync_dir(journal_dir)
453
+ if dir_created:
454
+ _fsync_dir(journal_dir.parent)
455
+ return (seg_name, end_offset)
456
+ finally:
457
+ _release_leaf_lock(lock_fd)
458
+
459
+
341
460
  def list_segments() -> list[str]:
342
461
  """Journal segment basenames in canonical order (spec §4.1): bootstrap
343
462
  segments first, then observation segments, each class lexicographic.
@@ -358,25 +477,47 @@ def list_segments() -> list[str]:
358
477
  return sorted(names, key=_lib_journal.segment_sort_key)
359
478
 
360
479
 
361
- def journal_high_water() -> tuple[str, int] | None:
362
- """Snapshot ``(latest segment basename, size)`` under a µs leaf-lock hold.
480
+ def _has_retained_journal_bytes(segment_sizes) -> bool:
481
+ """Whether any canonical journal segment retains replayable bytes."""
482
+ return any(int(size) > 0 for size in segment_sizes)
363
483
 
364
- "Latest" is the canonically-last segment (spec §4.1 order). The ingest
365
- cycle takes this snapshot and consumes only ``cursor → HW`` so a line
366
- appended after the snapshot belongs to the next cycle (spec §5.2.1).
367
- Returns ``None`` when no segment exists yet."""
484
+
485
+ def _journal_rebuild_snapshot() -> tuple[tuple[str, int] | None, bool]:
486
+ """Snapshot the canonical high-water and retained-byte truth together.
487
+
488
+ A freshly created newest segment can legitimately be empty while older
489
+ immutable segments still contain the durable rebuild source. Holding the
490
+ leaf lock across both facts keeps the epoch resolver from making its
491
+ fail-closed decision against two different journal states.
492
+ """
368
493
  lock_fd = _acquire_leaf_lock()
369
494
  try:
370
495
  segments = list_segments()
371
496
  if not segments:
372
- return None
373
- latest = segments[-1]
374
- size = os.stat(_cctally_core.JOURNAL_DIR / latest).st_size
375
- return (latest, size)
497
+ return None, False
498
+ sizes = [
499
+ os.stat(_cctally_core.JOURNAL_DIR / segment).st_size
500
+ for segment in segments
501
+ ]
502
+ return (
503
+ (segments[-1], sizes[-1]),
504
+ _has_retained_journal_bytes(sizes),
505
+ )
376
506
  finally:
377
507
  _release_leaf_lock(lock_fd)
378
508
 
379
509
 
510
+ def journal_high_water() -> tuple[str, int] | None:
511
+ """Snapshot ``(latest segment basename, size)`` under a µs leaf-lock hold.
512
+
513
+ "Latest" is the canonically-last segment (spec §4.1 order). The ingest
514
+ cycle takes this snapshot and consumes only ``cursor → HW`` so a line
515
+ appended after the snapshot belongs to the next cycle (spec §5.2.1).
516
+ Returns ``None`` when no segment exists yet."""
517
+ high_water, _has_bytes = _journal_rebuild_snapshot()
518
+ return high_water
519
+
520
+
380
521
  # ==========================================================================
381
522
  # Single-flight ingest cycle (spec §5.1 / §5.2, revision 3)
382
523
  # ==========================================================================
@@ -449,6 +590,9 @@ class IngestResult:
449
590
  malformed: int # lines in range that failed to decode (spec §4.4)
450
591
  events_emitted: int # evt lines emitted this cycle (Model-A + harvest)
451
592
  alerts: list # alert payloads dispatched post-commit (step 6)
593
+ # #374: same-revision divergences handled this cycle — emissions withheld at
594
+ # the write boundary plus journal evts the preflight reader quarantined.
595
+ conflicts_dropped: int = 0
452
596
  # Exception discipline (6b-gate P2): the exception that aborted the cycle on
453
597
  # an OPPORTUNISTIC ingest — the txn rolled back, the cursor did NOT advance
454
598
  # (invariant ii), and `run_stats_ingest` logged it loudly and returned
@@ -490,11 +634,38 @@ class IngestContext:
490
634
  # (reset INSERT OR IGNORE rowcount == 1), so a crash-replayed reset never
491
635
  # re-suppresses with a divergent list.
492
636
  suppression_map: dict = field(default_factory=dict)
637
+ # Task B rederive seam. Normal ingest leaves both defaults unchanged.
638
+ # A scratch planner supplies an in-memory sink so derived events are captured
639
+ # instead of appended to the durable journal, and disables projection-file
640
+ # writes while still exercising the same SQLite derivation/fold code.
641
+ event_sink: "list | None" = None
642
+ projection_writes: bool = True
643
+ # Scratch replay reconstructs ephemeral marker state in memory. A planner
644
+ # shares this dict across its per-record contexts.
645
+ projection_state: dict = field(default_factory=dict)
646
+ # #374 write boundary: emissions WITHHELD this cycle because they would have
647
+ # violated the same-revision rule. Each entry is a `DroppedConflict`; the row
648
+ # was converged from the already-journaled effective event instead.
649
+ conflicts_dropped: list = field(default_factory=list)
493
650
 
494
651
  def as_of_for(self, record: dict) -> str:
495
652
  return record["at"]
496
653
 
497
654
 
655
+ @dataclass(frozen=True)
656
+ class DroppedConflict:
657
+ """One live emission withheld at the write boundary (#374 §6).
658
+
659
+ The journal is append-only, so a divergent line can never be un-written —
660
+ the only durable defence is to never append it. The row is converged from
661
+ the effective event instead, and the rejected content is reported here.
662
+ """
663
+
664
+ event_id: str
665
+ rev: int
666
+ rejected_hash: str
667
+
668
+
498
669
  @dataclass(frozen=True)
499
670
  class _EvtSpec:
500
671
  """How to fold one evt `kind` into its target table (step 4a replay + the
@@ -615,27 +786,161 @@ def _release_ingest_lock(fd: int) -> None:
615
786
  os.close(fd)
616
787
 
617
788
 
789
+ def _acquire_maintenance_shared(mode: str, timeout_s: float) -> int | None:
790
+ """Acquire the stats maintenance lock shared, before the ingest lock."""
791
+ _cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
792
+ fd = os.open(
793
+ str(_cctally_core.STATS_LOCK_MAINTENANCE_PATH),
794
+ os.O_RDWR | os.O_CREAT,
795
+ 0o600,
796
+ )
797
+ if mode == "opportunistic":
798
+ try:
799
+ fcntl.flock(fd, fcntl.LOCK_SH | fcntl.LOCK_NB)
800
+ _cctally_core.note_stats_maintenance_acquired()
801
+ return fd
802
+ except (BlockingIOError, OSError):
803
+ os.close(fd)
804
+ return None
805
+ deadline = time.monotonic() + timeout_s
806
+ try:
807
+ while True:
808
+ try:
809
+ fcntl.flock(fd, fcntl.LOCK_SH | fcntl.LOCK_NB)
810
+ _cctally_core.note_stats_maintenance_acquired()
811
+ return fd
812
+ except (BlockingIOError, OSError):
813
+ if time.monotonic() >= deadline:
814
+ os.close(fd)
815
+ return None
816
+ time.sleep(0.02)
817
+ except BaseException:
818
+ os.close(fd)
819
+ raise
820
+
821
+
822
+ def _acquire_maintenance_exclusive(mode: str, timeout_s: float) -> int | None:
823
+ """Acquire the stats maintenance lock exclusively for legacy cutover."""
824
+ _cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
825
+ fd = os.open(
826
+ str(_cctally_core.STATS_LOCK_MAINTENANCE_PATH),
827
+ os.O_RDWR | os.O_CREAT,
828
+ 0o600,
829
+ )
830
+ if mode == "opportunistic":
831
+ try:
832
+ fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
833
+ _cctally_core.note_stats_maintenance_acquired()
834
+ return fd
835
+ except (BlockingIOError, OSError):
836
+ os.close(fd)
837
+ return None
838
+ deadline = time.monotonic() + timeout_s
839
+ try:
840
+ while True:
841
+ try:
842
+ fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
843
+ _cctally_core.note_stats_maintenance_acquired()
844
+ return fd
845
+ except (BlockingIOError, OSError):
846
+ if time.monotonic() >= deadline:
847
+ os.close(fd)
848
+ return None
849
+ time.sleep(0.02)
850
+ except BaseException:
851
+ os.close(fd)
852
+ raise
853
+
854
+
855
+ def _release_maintenance_shared(fd: int) -> None:
856
+ try:
857
+ fcntl.flock(fd, fcntl.LOCK_UN)
858
+ finally:
859
+ # #386: paired with the note in both acquire helpers. `open_db()` takes
860
+ # this same lock SHARED on a fresh fd, and flock conflicts are
861
+ # process-wide across descriptions, so the legacy/fresh ingest branch
862
+ # (which holds it EXCLUSIVE across its `open_db()`) would self-deadlock
863
+ # without the re-entrancy signal.
864
+ _cctally_core.note_stats_maintenance_released()
865
+ os.close(fd)
866
+
867
+
868
+ def _downgrade_maintenance_shared(fd: int) -> None:
869
+ """Atomically downgrade a held maintenance lock from EX to SH."""
870
+ fcntl.flock(fd, fcntl.LOCK_SH)
871
+
872
+
873
+ def _stats_db_identity():
874
+ """Return the current stats main-file identity, or ``None`` if absent."""
875
+ try:
876
+ stat = os.stat(_cctally_core.DB_PATH)
877
+ except OSError:
878
+ return None
879
+ return (stat.st_dev, stat.st_ino)
880
+
881
+
882
+ def _stats_db_user_version() -> int | None:
883
+ """Read the main file's raw epoch without invoking schema or heal paths."""
884
+ try:
885
+ conn = sqlite3.connect(
886
+ f"file:{_cctally_core.DB_PATH}?mode=ro",
887
+ uri=True,
888
+ )
889
+ except sqlite3.Error:
890
+ return None
891
+ try:
892
+ return int(conn.execute("PRAGMA user_version").fetchone()[0])
893
+ except sqlite3.Error:
894
+ return None
895
+ finally:
896
+ conn.close()
897
+
898
+
618
899
  # --------------------------------------------------------------------------
619
900
  # cursor (spec §5.2.2: segment-aware, prior-month tails covered)
620
901
  # --------------------------------------------------------------------------
621
902
 
622
903
  def _read_cursor(conn: sqlite3.Connection) -> tuple[str, int] | None:
623
904
  """Return `(segment_basename, offset)` from `journal_cursor`, or None when
624
- nothing has been consumed yet (start of the first segment)."""
905
+ nothing has been consumed yet (start of the first segment).
906
+
907
+ ``applied_segment`` / ``applied_offset`` are the trusted duplicate written
908
+ in the same stats transaction as every materialized row (#410 Task B). A
909
+ cursor-only hand edit can therefore no longer skip durable events and make
910
+ their natural keys appear new: on disagreement, resume from the last
911
+ atomically applied prefix and let the normal replay heal both pairs."""
625
912
  row = conn.execute(
626
- "SELECT segment, offset FROM journal_cursor WHERE id = 1"
913
+ "SELECT segment, offset, applied_segment, applied_offset "
914
+ "FROM journal_cursor WHERE id = 1"
627
915
  ).fetchone()
628
916
  if row is None:
629
917
  return None
630
- return (row[0], int(row[1]))
918
+ public = (str(row[0]), int(row[1]))
919
+ if row[2] is None or row[3] is None:
920
+ raise JournalError(
921
+ "journal cursor applied-prefix guard is incomplete; "
922
+ "run cctally db rebuild --db stats"
923
+ )
924
+ applied = (str(row[2]), int(row[3]))
925
+ if public != applied:
926
+ print(
927
+ f"[journal] cursor-only advancement detected: public={public!r}, "
928
+ f"applied={applied!r}; replaying from the applied prefix",
929
+ file=sys.stderr,
930
+ )
931
+ return applied
631
932
 
632
933
 
633
934
  def _write_cursor(conn: sqlite3.Connection, segment: str, offset: int) -> None:
634
935
  conn.execute(
635
- "INSERT INTO journal_cursor (id, segment, offset) VALUES (1, ?, ?) "
936
+ "INSERT INTO journal_cursor "
937
+ "(id, segment, offset, applied_segment, applied_offset) "
938
+ "VALUES (1, ?, ?, ?, ?) "
636
939
  "ON CONFLICT(id) DO UPDATE SET segment = excluded.segment, "
637
- "offset = excluded.offset",
638
- (segment, offset),
940
+ "offset = excluded.offset, "
941
+ "applied_segment = excluded.applied_segment, "
942
+ "applied_offset = excluded.applied_offset",
943
+ (segment, offset, segment, offset),
639
944
  )
640
945
 
641
946
 
@@ -690,6 +995,49 @@ def _read_range(cursor, hw) -> list[tuple[str, int, bytes]]:
690
995
  return lines
691
996
 
692
997
 
998
+ def journal_prefix_hash(high_water) -> "str | None":
999
+ """Hash exact raw segment bytes through one canonical high-water."""
1000
+ if high_water is None:
1001
+ return None
1002
+ digest = hashlib.sha256()
1003
+ found = False
1004
+ for segment in list_segments():
1005
+ path = _cctally_core.JOURNAL_DIR / segment
1006
+ size = high_water[1] if segment == high_water[0] else path.stat().st_size
1007
+ data = path.read_bytes()[:size]
1008
+ if len(data) != size:
1009
+ raise OSError(f"journal segment changed while reading: {segment}")
1010
+ name = segment.encode("utf-8")
1011
+ digest.update(len(name).to_bytes(4, "big"))
1012
+ digest.update(name)
1013
+ digest.update(len(data).to_bytes(8, "big"))
1014
+ digest.update(data)
1015
+ if segment == high_water[0]:
1016
+ found = True
1017
+ break
1018
+ if not found:
1019
+ raise OSError(
1020
+ f"journal high-water segment is unavailable: {high_water[0]}"
1021
+ )
1022
+ return "sha256:" + digest.hexdigest()
1023
+
1024
+
1025
+ def _capture_protocol_prefix_evidence(record, prior_high_water, evidence) -> None:
1026
+ """Capture the actual raw prefix immediately preceding one audit record."""
1027
+ if (
1028
+ record.get("t") == "op"
1029
+ and isinstance(record.get("payload"), dict)
1030
+ and record["payload"].get("kind")
1031
+ == _lib_journal._PROTOCOL_RESOLUTION_KIND
1032
+ ):
1033
+ evidence.append(
1034
+ (
1035
+ prior_high_water,
1036
+ journal_prefix_hash(prior_high_water),
1037
+ )
1038
+ )
1039
+
1040
+
693
1041
  # --------------------------------------------------------------------------
694
1042
  # cache leg — Codex quota obs -> cache.db quota_window_snapshots (spec §5.2
695
1043
  # step 3, Task 7 Item 2). Runs BEFORE the stats BEGIN IMMEDIATE, under the
@@ -832,14 +1180,18 @@ def _resolve_ref(conn: sqlite3.Connection, table: str, logical_id) -> int | None
832
1180
  return int(row[0])
833
1181
 
834
1182
 
835
- def _insert_or_ignore(conn: sqlite3.Connection, table: str, cols: dict):
1183
+ def _insert_or_ignore(
1184
+ conn: sqlite3.Connection, table: str, cols: dict, *, strict: bool = False
1185
+ ):
836
1186
  keys = list(cols.keys())
837
1187
  colnames = ", ".join(keys)
838
1188
  placeholders = ", ".join("?" for _ in keys)
839
- return conn.execute(
840
- f"INSERT OR IGNORE INTO {table} ({colnames}) VALUES ({placeholders})",
841
- tuple(cols[k] for k in keys),
1189
+ statement = (
1190
+ f"INSERT OR IGNORE INTO {table} ({colnames}) VALUES ({placeholders})"
842
1191
  )
1192
+ if strict:
1193
+ statement = statement.replace("INSERT OR IGNORE", "INSERT", 1)
1194
+ return conn.execute(statement, tuple(cols[k] for k in keys))
843
1195
 
844
1196
 
845
1197
  def _reverse_ref(conn: sqlite3.Connection, ref_table: str, rowid) -> "str | None":
@@ -1253,6 +1605,34 @@ _BLOCK_CHILDREN = (
1253
1605
  _BLOCK_CHILD_KEYS = frozenset(k for k, _t in _BLOCK_CHILDREN)
1254
1606
 
1255
1607
 
1608
+ def _replace_block_children(
1609
+ conn, block_id, parent_account, parent_window, children
1610
+ ) -> None:
1611
+ """Materialize one frozen block's child sets exactly.
1612
+
1613
+ A close event owns the complete model/project membership, so replay and
1614
+ convergence replace both sets rather than relying on natural-key
1615
+ ``INSERT OR IGNORE``. This removes stale children and restores missing
1616
+ children while preserving the parent rowid used by milestone FKs.
1617
+ """
1618
+ for payload_key, child_table in _BLOCK_CHILDREN:
1619
+ if parent_account is None:
1620
+ predicate = "five_hour_window_key = ?"
1621
+ params = (int(parent_window),)
1622
+ else:
1623
+ predicate = "account_key = ? AND five_hour_window_key = ?"
1624
+ params = (parent_account, int(parent_window))
1625
+ conn.execute(
1626
+ f"DELETE FROM {child_table} WHERE {predicate}", params
1627
+ )
1628
+ for child in children.get(payload_key, []):
1629
+ cols = dict(child)
1630
+ cols["block_id"] = int(block_id)
1631
+ if parent_account is not None:
1632
+ cols["account_key"] = parent_account
1633
+ _insert_or_ignore(conn, child_table, cols, strict=True)
1634
+
1635
+
1256
1636
  def _apply_generic_evt(conn, evt):
1257
1637
  """Fold an evt line into its target table (spec §5.3), returning the sqlite
1258
1638
  cursor of the `INSERT OR IGNORE`.
@@ -1278,28 +1658,37 @@ def _apply_generic_evt(conn, evt):
1278
1658
  cols[key] = value
1279
1659
  # Re-derive any projection-FK columns from a journaled natural-key column
1280
1660
  # (spec §5.3 — e.g. five_hour_milestones.block_id from five_hour_window_key,
1281
- # since the open block is a projection with no logical id). Composite
1282
- # (account_key, <lookup_col>) when the row carries an account (#341, review
1283
- # finding 3): a shared physical 5h window resolves THIS account's block, so a
1284
- # milestone child never attaches to another account's block. 0 when absent.
1661
+ # since the open block is a projection with no logical id).
1285
1662
  acct = cols.get("account_key")
1286
1663
  for column, (ref_table, lookup_col) in spec.derived_fk.items():
1287
- if acct is not None:
1288
- row = conn.execute(
1289
- f"SELECT id FROM {ref_table} "
1290
- f"WHERE {lookup_col} = ? AND account_key = ?",
1291
- (cols.get(lookup_col), acct),
1292
- ).fetchone()
1293
- else:
1294
- row = conn.execute(
1295
- f"SELECT id FROM {ref_table} WHERE {lookup_col} = ?",
1296
- (cols.get(lookup_col),),
1297
- ).fetchone()
1298
- cols[column] = int(row[0]) if row is not None else 0
1664
+ cols[column] = _derived_fk_value(
1665
+ conn, ref_table, lookup_col, cols.get(lookup_col), acct)
1299
1666
  return _insert_or_ignore(conn, spec.table, cols)
1300
1667
 
1301
1668
 
1302
- def _apply_weekly_credit_effects(conn, evt):
1669
+ def _derived_fk_value(conn, ref_table, lookup_col, lookup_value, account_key):
1670
+ """Resolve one derived (re-derived-at-fold) FK column (spec §5.3).
1671
+
1672
+ Composite `(account_key, <lookup_col>)` when the row carries an account
1673
+ (#341, review finding 3): a shared physical 5h window resolves THIS
1674
+ account's block, so a milestone never attaches to another account's block.
1675
+ 0 when unresolvable. The SINGLE home of this rule — the fold applier and the
1676
+ #374 duplicate-path validation must agree by construction, not by copy."""
1677
+ if account_key is not None:
1678
+ row = conn.execute(
1679
+ f"SELECT id FROM {ref_table} "
1680
+ f"WHERE {lookup_col} = ? AND account_key = ?",
1681
+ (lookup_value, account_key),
1682
+ ).fetchone()
1683
+ else:
1684
+ row = conn.execute(
1685
+ f"SELECT id FROM {ref_table} WHERE {lookup_col} = ?",
1686
+ (lookup_value,),
1687
+ ).fetchone()
1688
+ return int(row[0]) if row is not None else 0
1689
+
1690
+
1691
+ def _apply_weekly_credit_effects(conn, evt, *, projection_writes=True):
1303
1692
  """Apply a `weekly_credit_effects` evt (spec §5.3 event+effects). The
1304
1693
  same-window sub-25pp credit writes NO reset row, so its DESTRUCTIVE effects
1305
1694
  ride this vehicle: delete the stale-replica snapshots by their logical
@@ -1326,7 +1715,7 @@ def _apply_weekly_credit_effects(conn, evt):
1326
1715
  conn.execute(
1327
1716
  "DELETE FROM weekly_credit_floors WHERE journal_id = ?", (logical_id,))
1328
1717
  floor = payload.get("hwm_floor")
1329
- if floor:
1718
+ if floor and projection_writes:
1330
1719
  try:
1331
1720
  (_cctally_core.APP_DIR / "hwm-7d").write_text(
1332
1721
  f"{floor['week_start_date']} {floor['weekly_percent']}\n"
@@ -1340,17 +1729,30 @@ def _apply_quota_alert_arming(conn, evt):
1340
1729
  """Fold a `quota_alert_arming` evt (spec §5.3 "state", Task 7 Item 5). The
1341
1730
  quota-alert arming boundary is journaled state — its `activated_at_utc` is a
1342
1731
  forward-only alert boundary that MUST survive a stats.db rebuild so the
1343
- reconcile honors it (no historical re-fires). Applied as an UPSERT on the
1344
- arming natural key, in canonical order, so the latest state per key wins and
1345
- re-applying an already-present evt is a clean no-op. `quota_alert_arming` has
1346
- no `journal_id` column (it is not in the Task-4 additive list); idempotence
1347
- is the natural-key upsert, not a journal_id INSERT OR IGNORE."""
1732
+ reconcile honors it (no historical re-fires). Activation records UPSERT the
1733
+ natural key; explicit disarm records DELETE that same account-qualified key.
1734
+ Canonical replay therefore leaves the latest retained state in force, and
1735
+ re-applying either transition is a clean no-op. `quota_alert_arming` has no
1736
+ `journal_id` column (it is not in the Task-4 additive list); idempotence is
1737
+ the natural-key upsert/delete, not a journal_id INSERT OR IGNORE."""
1348
1738
  p = evt.get("payload") or {}
1349
1739
  # account_key (#341) is part of the arming identity/UNIQUE. A live-emitted evt
1350
1740
  # carries payload.account_key; a legacy (pre-#341) cutover-exported arming has
1351
1741
  # none -> normalise to the sentinel (Codex legacy -> unattributed) so the
1352
1742
  # NOT NULL column always receives a value.
1353
1743
  account_key = p.get("account_key") or _lib_accounts.UNATTRIBUTED
1744
+ if p.get("state") == "disarmed":
1745
+ conn.execute(
1746
+ "DELETE FROM quota_alert_arming "
1747
+ "WHERE source=? AND source_root_key=? AND account_key=? "
1748
+ "AND logical_limit_key=? AND observed_slot=? AND window_minutes=?",
1749
+ (
1750
+ p.get("source"), p.get("source_root_key"), account_key,
1751
+ p.get("logical_limit_key"), p.get("observed_slot"),
1752
+ p.get("window_minutes"),
1753
+ ),
1754
+ )
1755
+ return None
1354
1756
  conn.execute(
1355
1757
  "INSERT INTO quota_alert_arming "
1356
1758
  "(source, source_root_key, logical_limit_key, observed_slot, "
@@ -1368,12 +1770,14 @@ def _apply_quota_alert_arming(conn, evt):
1368
1770
 
1369
1771
 
1370
1772
  def _apply_block_close(conn, evt):
1371
- """Fold a `five_hour_block_close` evt (spec §5.3): insert the frozen parent
1372
- block (`INSERT OR IGNORE` on window-key / journal_id), then its embedded
1373
- `_models`/`_projects` rollup children under the resolved parent block_id
1374
- (each idempotent on its own natural key). Live harvest only STAMPS the
1375
- parent's journal_id — 6b's close hook already wrote parent+children — so this
1376
- insert path runs for replay/rebuild."""
1773
+ """Fold one authoritative frozen-block fact.
1774
+
1775
+ A replay may meet an existing open projection under the same
1776
+ ``(account_key, five_hour_window_key)`` natural key. ``INSERT OR IGNORE``
1777
+ alone would silently leave that mutable row and its children in place, so
1778
+ the event now converges the existing parent in place and replaces both
1779
+ child sets exactly. The parent rowid is preserved for milestone FKs.
1780
+ """
1377
1781
  payload = evt.get("payload") or {}
1378
1782
  parent = {"journal_id": evt["id"]}
1379
1783
  children = {}
@@ -1385,40 +1789,39 @@ def _apply_block_close(conn, evt):
1385
1789
  continue
1386
1790
  parent[key] = value
1387
1791
  _insert_or_ignore(conn, "five_hour_blocks", parent)
1388
- # Composite (account_key, five_hour_window_key) parent resolution (#341,
1389
- # review finding 3): a shared physical window resolves THIS account's block
1390
- # so its rollup children attach to the right parent.
1792
+
1391
1793
  p_acct = parent.get("account_key")
1392
1794
  if p_acct is not None:
1393
1795
  prow = conn.execute(
1394
- "SELECT id FROM five_hour_blocks "
1796
+ "SELECT id, journal_id, account_key, five_hour_window_key "
1797
+ "FROM five_hour_blocks "
1395
1798
  "WHERE five_hour_window_key = ? AND account_key = ?",
1396
1799
  (parent.get("five_hour_window_key"), p_acct),
1397
1800
  ).fetchone()
1398
1801
  else:
1399
1802
  prow = conn.execute(
1400
- "SELECT id FROM five_hour_blocks WHERE five_hour_window_key = ?",
1803
+ "SELECT id, journal_id, account_key, five_hour_window_key "
1804
+ "FROM five_hour_blocks "
1805
+ "WHERE five_hour_window_key = ?",
1401
1806
  (parent.get("five_hour_window_key"),),
1402
1807
  ).fetchone()
1403
1808
  if prow is None:
1404
- return None
1809
+ raise JournalError(
1810
+ f"five_hour_block_close {evt['id']} did not materialize its parent"
1811
+ )
1405
1812
  block_id = int(prow[0])
1406
- for payload_key, child_table in _BLOCK_CHILDREN:
1407
- for child in children.get(payload_key, []):
1408
- cols = dict(child)
1409
- cols["block_id"] = block_id
1410
- # Force each child under the PARENT's account (#341 P2-2, 8a review):
1411
- # a no-op on the live/already-stamped path (children already agree),
1412
- # but on the legacy rebuild path _normalize_legacy_account_stamp
1413
- # re-derives ONLY the parent's payload.account_key — the embedded
1414
- # _models/_projects children stay unstamped and would otherwise take
1415
- # the schema DEFAULT 'unattributed', mismatching their parent and
1416
- # splitting the composite (account_key, window, model/project) UNIQUE
1417
- # partition. Guarded on p_acct so a truly account-less rebuild (no
1418
- # cutover mapping) leaves the DEFAULT untouched.
1419
- if p_acct is not None:
1420
- cols["account_key"] = p_acct
1421
- _insert_or_ignore(conn, child_table, cols)
1813
+ existing_journal_id = prow[1]
1814
+ if existing_journal_id not in (None, evt["id"]):
1815
+ raise JournalError(
1816
+ f"five_hour_block_close {evt['id']} collided with "
1817
+ f"{existing_journal_id} on its parent natural key"
1818
+ )
1819
+ assignments = ", ".join(f"{name} = ?" for name in parent)
1820
+ conn.execute(
1821
+ f"UPDATE five_hour_blocks SET {assignments} WHERE id = ?",
1822
+ (*parent.values(), block_id),
1823
+ )
1824
+ _replace_block_children(conn, block_id, prow[2], prow[3], children)
1422
1825
  return None
1423
1826
 
1424
1827
 
@@ -1452,13 +1855,16 @@ def _apply_reset_with_suppression(conn, evt):
1452
1855
  return None
1453
1856
 
1454
1857
 
1455
- def _apply_evt(conn, evt):
1858
+ def _apply_evt(conn, evt, *, projection_writes=True):
1456
1859
  """Dispatch one evt line to its fold applier by `payload.kind` (step 4a
1457
1860
  replay + the emit_model_a apply path). A kind with a bespoke `applier`
1458
1861
  (weekly_credit_effects, five_hour_block_close) uses it; everything else
1459
1862
  goes through the generic column-map fold. Apply-only: NO alert dispatch,
1460
1863
  NO ctx — replay is structurally unable to fire alerts (spec §5.2 step 4a)."""
1461
1864
  spec = _EVT_SPECS.get((evt.get("payload") or {}).get("kind"))
1865
+ if spec is not None and spec.applier is _apply_weekly_credit_effects:
1866
+ return spec.applier(
1867
+ conn, evt, projection_writes=projection_writes)
1462
1868
  if spec is not None and spec.applier is not None:
1463
1869
  return spec.applier(conn, evt)
1464
1870
  return _apply_generic_evt(conn, evt)
@@ -1594,9 +2000,34 @@ def emit_model_a(ctx, *, kind, evt_id, table, columns, refs=None, at=None):
1594
2000
  payload.update(refs)
1595
2001
  evt = _lib_journal.make_evt(kind=kind, id=evt_id, at=(at or _now_iso()),
1596
2002
  payload=payload)
1597
- append_record(evt)
1598
- ctx.events_emitted += 1
1599
- _apply_evt(ctx.conn, evt)
2003
+ if ctx.event_sink is not None:
2004
+ # SCRATCH planning (`db rederive`, spec §6): capture EVERY derived
2005
+ # candidate — never classify, never drop, or the very divergence the
2006
+ # planner exists to correct would be filtered out of `desired_events`
2007
+ # and the diff would report a false no-op. Model-A events are still
2008
+ # APPLIED to the private scratch projection because callers consume the
2009
+ # returned rowid immediately (`snapshot_accept` stores it before
2010
+ # milestone/block derivation). Live effective metadata is never read or
2011
+ # written. Discriminated on `is not None` — the sink is `list | None`
2012
+ # and an EMPTY sink list is falsy.
2013
+ ctx.event_sink.append(evt)
2014
+ ctx.events_emitted += 1
2015
+ _apply_evt(ctx.conn, evt, projection_writes=ctx.projection_writes)
2016
+ else:
2017
+ decision = _classify_live_effective_event(ctx.conn, evt)
2018
+ if decision == CLASSIFY_CONFLICT:
2019
+ # No append, no metadata mutation — converge the row instead.
2020
+ _record_dropped_conflict(ctx, evt)
2021
+ _converge_row_from_effective(ctx.conn, evt_id, table=table)
2022
+ else:
2023
+ # `new` AND `duplicate` both still append: crash-replay convergence
2024
+ # and two-bootstrap idempotency are built on that.
2025
+ append_record(evt)
2026
+ ctx.events_emitted += 1
2027
+ if decision == CLASSIFY_NEW:
2028
+ _record_new_effective_event(ctx.conn, evt)
2029
+ _apply_evt(
2030
+ ctx.conn, evt, projection_writes=ctx.projection_writes)
1600
2031
  if table is None:
1601
2032
  return None
1602
2033
  row = ctx.conn.execute(
@@ -1657,6 +2088,71 @@ def _build_harvest_evt(ctx, spec, row):
1657
2088
  return _lib_journal.make_evt(kind=spec.kind, id=eid, at=at, payload=payload)
1658
2089
 
1659
2090
 
2091
+ def _emit_harvest_row(ctx, spec, row):
2092
+ """Journal and stamp one already-selected natural-keyed row."""
2093
+ conn = ctx.conn
2094
+ evt = _build_harvest_evt(ctx, spec, row)
2095
+ if ctx.event_sink is not None:
2096
+ # Scratch planning: capture + stamp the PRIVATE projection so a later
2097
+ # raw record does not re-harvest the same row and its downstream FKs
2098
+ # still resolve. Never classify, drop, or touch live metadata.
2099
+ ctx.event_sink.append(evt)
2100
+ ctx.events_emitted += 1
2101
+ conn.execute(
2102
+ f"UPDATE {spec.table} SET journal_id = ? WHERE id = ?",
2103
+ (evt["id"], row["id"]),
2104
+ )
2105
+ return evt
2106
+
2107
+ decision = _classify_live_effective_event(conn, evt)
2108
+ if decision == CLASSIFY_CONFLICT:
2109
+ _record_dropped_conflict(ctx, evt)
2110
+ _converge_row_from_effective(
2111
+ conn, evt["id"], table=spec.table, rowid=row["id"]
2112
+ )
2113
+ return evt
2114
+
2115
+ append_record(evt)
2116
+ ctx.events_emitted += 1
2117
+ if decision == CLASSIFY_NEW:
2118
+ _record_new_effective_event(conn, evt)
2119
+ else:
2120
+ # An exact crash-retry duplicate still validates excluded derived FKs
2121
+ # before it stamps the physical row.
2122
+ _validate_excluded_derived_fks(conn, spec, row)
2123
+ conn.execute(
2124
+ f"UPDATE {spec.table} SET journal_id = ? WHERE id = ?",
2125
+ (evt["id"], row["id"]),
2126
+ )
2127
+ return evt
2128
+
2129
+
2130
+ def freeze_five_hour_block_close(ctx, block_id: int):
2131
+ """Freeze one closed block immediately as a complete replayable fact.
2132
+
2133
+ Unlike the end-of-cycle table scan, this surface also accepts an already
2134
+ stamped row so a lost-commit retry can re-emit the exact duplicate selected
2135
+ by its retained closure trigger. It never derives from cache state itself;
2136
+ the parent and both child sets present at this call are the durable boundary.
2137
+ """
2138
+ spec = next(
2139
+ item for item in _HARVEST_SPECS
2140
+ if item.kind == "five_hour_block_close"
2141
+ )
2142
+ row = ctx.conn.execute(
2143
+ "SELECT * FROM five_hour_blocks WHERE id = ?", (int(block_id),)
2144
+ ).fetchone()
2145
+ if row is None:
2146
+ raise JournalError(
2147
+ f"cannot freeze five_hour_block_close: missing block {block_id}"
2148
+ )
2149
+ if int(row["is_closed"]) != 1:
2150
+ raise JournalError(
2151
+ f"cannot freeze five_hour_block_close: block {block_id} is open"
2152
+ )
2153
+ return _emit_harvest_row(ctx, spec, row)
2154
+
2155
+
1660
2156
  def _harvest(ctx) -> None:
1661
2157
  """Step 4c: journal + stamp every natural-keyed row inserted this cycle
1662
2158
  (`journal_id IS NULL`). Families harvest in dependency order so a referenced
@@ -1672,13 +2168,7 @@ def _harvest(ctx) -> None:
1672
2168
  f"SELECT * FROM {spec.table} WHERE {where} ORDER BY id"
1673
2169
  ).fetchall()
1674
2170
  for row in rows:
1675
- evt = _build_harvest_evt(ctx, spec, row)
1676
- append_record(evt)
1677
- ctx.events_emitted += 1
1678
- conn.execute(
1679
- f"UPDATE {spec.table} SET journal_id = ? WHERE id = ?",
1680
- (evt["id"], row["id"]),
1681
- )
2171
+ _emit_harvest_row(ctx, spec, row)
1682
2172
 
1683
2173
 
1684
2174
  # --------------------------------------------------------------------------
@@ -1853,14 +2343,551 @@ def _fold_order(evt) -> int:
1853
2343
  return (_EVT_SPECS.get(kind) or _UNKNOWN_EVT_SPEC).order
1854
2344
 
1855
2345
 
1856
- # --------------------------------------------------------------------------
1857
- # the cycle (spec §5.2, revision 3)
1858
- # --------------------------------------------------------------------------
2346
+ def _write_effective_metadata(conn, selection) -> None:
2347
+ """Replace the disposable effective-event summary from a pure selection."""
2348
+ conn.execute("DELETE FROM journal_effective_events")
2349
+ for event_id, selected in selection.by_id.items():
2350
+ event_json = None
2351
+ if selected.record is not None:
2352
+ event_json = (
2353
+ _lib_journal.encode_line(selected.record)
2354
+ .decode("utf-8")
2355
+ .rstrip("\n")
2356
+ )
2357
+ conn.execute(
2358
+ "INSERT INTO journal_effective_events "
2359
+ "(event_id, rev, status, content_hash, batch_id, event_json) "
2360
+ "VALUES (?, ?, ?, ?, ?, ?)",
2361
+ (
2362
+ event_id,
2363
+ selected.rev,
2364
+ selected.status,
2365
+ selected.content_hash,
2366
+ selected.batch_id,
2367
+ event_json,
2368
+ ),
2369
+ )
2370
+ _write_protocol_violations(
2371
+ conn,
2372
+ selection.protocol_violations,
2373
+ selection.acknowledged_protocol_violations,
2374
+ )
1859
2375
 
1860
- def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
1861
- codex_apply=None, post_commit=None) -> IngestResult:
1862
- # Step 1: HW snapshot (leaf lock, µs). Lines appended after this — by other
1863
- # processes OR by this cycle's own evt emission — are past HW and belong to
2376
+
2377
+ def _write_protocol_violations(conn, violations, acknowledged=()) -> None:
2378
+ """Replace the disposable structural-violation summary."""
2379
+ conn.execute("DELETE FROM journal_protocol_violations")
2380
+ rows = [*violations, *acknowledged]
2381
+ rows.sort(
2382
+ key=lambda violation: (
2383
+ violation.batch_id,
2384
+ violation.kind,
2385
+ violation.fingerprint,
2386
+ )
2387
+ )
2388
+ for violation in rows:
2389
+ conn.execute(
2390
+ "INSERT INTO journal_protocol_violations "
2391
+ "(fingerprint, batch_id, kind, violation_json) "
2392
+ "VALUES (?, ?, ?, ?)",
2393
+ (
2394
+ violation.fingerprint,
2395
+ violation.batch_id,
2396
+ violation.kind,
2397
+ json.dumps(
2398
+ violation.to_dict(),
2399
+ sort_keys=True,
2400
+ separators=(",", ":"),
2401
+ ),
2402
+ ),
2403
+ )
2404
+
2405
+
2406
+ def _metadata_row(conn, event_id):
2407
+ return conn.execute(
2408
+ "SELECT rev, status, content_hash, batch_id "
2409
+ "FROM journal_effective_events WHERE event_id = ?",
2410
+ (event_id,),
2411
+ ).fetchone()
2412
+
2413
+
2414
+ def _metadata_event_record(conn, event_id):
2415
+ row = conn.execute(
2416
+ "SELECT event_json FROM journal_effective_events WHERE event_id = ?",
2417
+ (event_id,),
2418
+ ).fetchone()
2419
+ if row is None or row[0] is None:
2420
+ return None
2421
+ return _lib_journal.decode_line(row[0].encode("utf-8"))
2422
+
2423
+
2424
+ def _legacy_qaa_can_advance(conn, selected) -> bool:
2425
+ return (
2426
+ _lib_journal.is_legacy_quota_arming_record(selected.record)
2427
+ and _lib_journal.is_legacy_quota_arming_record(
2428
+ _metadata_event_record(conn, selected.event_id)
2429
+ )
2430
+ )
2431
+
2432
+
2433
+ def _insert_effective_metadata(conn, selected) -> None:
2434
+ event_json = None
2435
+ if selected.record is not None:
2436
+ event_json = (
2437
+ _lib_journal.encode_line(selected.record).decode("utf-8").rstrip("\n")
2438
+ )
2439
+ conn.execute(
2440
+ "INSERT INTO journal_effective_events "
2441
+ "(event_id, rev, status, content_hash, batch_id, event_json) "
2442
+ "VALUES (?, ?, ?, ?, ?, ?)",
2443
+ (
2444
+ selected.event_id,
2445
+ selected.rev,
2446
+ selected.status,
2447
+ selected.content_hash,
2448
+ selected.batch_id,
2449
+ event_json,
2450
+ ),
2451
+ )
2452
+
2453
+
2454
+ CLASSIFY_NEW = "new"
2455
+ CLASSIFY_DUPLICATE = "duplicate"
2456
+ CLASSIFY_CONFLICT = "conflict"
2457
+
2458
+
2459
+ def _classify_live_effective_event(conn, evt) -> str:
2460
+ """Decide what a freshly derived evt means against the live effective
2461
+ metadata — WITHOUT mutating anything (#374 §6).
2462
+
2463
+ The whole point of the split is ordering: both emit paths call this BEFORE
2464
+ `append_record`, so a conflicting emission is never written. Previously the
2465
+ append ran first and the check raised afterwards, so the divergent line
2466
+ landed in the append-only journal and the rollback could not take it back —
2467
+ poisoning every subsequent rebuild.
2468
+
2469
+ Returns `CLASSIFY_NEW` (no prior effective event, or a legacy `qaa` state
2470
+ stream that may advance), `CLASSIFY_DUPLICATE` (byte-identical to the prior
2471
+ effective event) or `CLASSIFY_CONFLICT` (same revision, different content).
2472
+ Raises `CorrectionRebuildRequired` on a revision mismatch — a completed
2473
+ correction batch outranks any live emission and stays FATAL.
2474
+ """
2475
+ selected = _lib_journal.resolve_effective_events([evt]).by_id[evt["id"]]
2476
+ prior = _metadata_row(conn, selected.event_id)
2477
+ if prior is None:
2478
+ return CLASSIFY_NEW
2479
+ prior_rev, prior_status, prior_hash, prior_batch = prior
2480
+ if int(prior_rev) == selected.rev:
2481
+ if prior_status != selected.status or prior_hash != selected.content_hash:
2482
+ if _legacy_qaa_can_advance(conn, selected):
2483
+ return CLASSIFY_NEW
2484
+ return CLASSIFY_CONFLICT
2485
+ return CLASSIFY_DUPLICATE
2486
+ raise CorrectionRebuildRequired(
2487
+ f"event {selected.event_id} rev {selected.rev} conflicts with effective "
2488
+ f"rev {prior_rev} from {prior_batch or 'base journal'}",
2489
+ batch_id=prior_batch,
2490
+ event_id=selected.event_id,
2491
+ high_water=_correction_commit_high_water(prior_batch),
2492
+ expected_metadata=(
2493
+ int(prior_rev),
2494
+ prior_status,
2495
+ prior_hash,
2496
+ prior_batch,
2497
+ ),
2498
+ )
2499
+
2500
+
2501
+ def _record_new_effective_event(conn, evt) -> None:
2502
+ """Write the effective-metadata row for a `CLASSIFY_NEW` emission. The
2503
+ DELETE covers the legacy `qaa` advance (the only case where a prior row is
2504
+ replaced rather than absent)."""
2505
+ selected = _lib_journal.resolve_effective_events([evt]).by_id[evt["id"]]
2506
+ if _metadata_row(conn, selected.event_id) is not None:
2507
+ conn.execute(
2508
+ "DELETE FROM journal_effective_events WHERE event_id = ?",
2509
+ (selected.event_id,),
2510
+ )
2511
+ _insert_effective_metadata(conn, selected)
2512
+
2513
+
2514
+ def _record_live_effective_event(conn, evt) -> bool:
2515
+ """Record a newly folded base event; return False when the caller must NOT
2516
+ apply it — an exact duplicate, or a quarantined same-revision conflict.
2517
+
2518
+ Retained for the step-4a replay site, whose conflicts the preflight reader
2519
+ has already dropped. The emit paths use the classifier directly so they can
2520
+ withhold the append and converge the row."""
2521
+ decision = _classify_live_effective_event(conn, evt)
2522
+ if decision == CLASSIFY_NEW:
2523
+ _record_new_effective_event(conn, evt)
2524
+ return True
2525
+ return False
2526
+
2527
+
2528
+ def _record_dropped_conflict(ctx, evt) -> None:
2529
+ """Count + report one withheld emission (spec §8: a one-line stderr note per
2530
+ dropped emission, and a count on the cycle summary)."""
2531
+ selected = _lib_journal.resolve_effective_events([evt]).by_id[evt["id"]]
2532
+ ctx.conflicts_dropped.append(
2533
+ DroppedConflict(
2534
+ event_id=selected.event_id,
2535
+ rev=selected.rev,
2536
+ rejected_hash=selected.content_hash,
2537
+ )
2538
+ )
2539
+ print(
2540
+ f"[journal] withheld a divergent emission for {selected.event_id} "
2541
+ f"rev {selected.rev}; converged the row from the journaled event",
2542
+ file=sys.stderr,
2543
+ )
2544
+
2545
+
2546
+ def _effective_event_for_convergence(conn, event_id) -> dict:
2547
+ """The ACTIVE, validated journal record the live row must converge to.
2548
+
2549
+ Fails closed (#374 §6): `decode_line` only checks that the value is an object
2550
+ with a string `t`, and a tombstoned selection deliberately stores
2551
+ `event_json` as NULL — so a same-revision active-vs-tombstone conflict has no
2552
+ record to converge from. Missing, tombstoned, or hash-mismatched metadata
2553
+ raises here and the caller therefore NEVER stamps."""
2554
+ row = conn.execute(
2555
+ "SELECT rev, status, content_hash, event_json "
2556
+ "FROM journal_effective_events WHERE event_id = ?",
2557
+ (event_id,),
2558
+ ).fetchone()
2559
+ if row is None:
2560
+ raise _lib_journal.JournalProtocolError(
2561
+ f"cannot converge {event_id}: no effective metadata")
2562
+ _rev, status, content_hash, event_json = row
2563
+ if status != "active" or event_json is None:
2564
+ raise _lib_journal.JournalProtocolError(
2565
+ f"cannot converge {event_id}: effective selection is {status!r} "
2566
+ "with no retained record")
2567
+ record = _lib_journal.decode_line(event_json.encode("utf-8"))
2568
+ if (
2569
+ record is None
2570
+ or record.get("t") != "evt"
2571
+ or record.get("id") != event_id
2572
+ or not isinstance(record.get("payload"), dict)
2573
+ ):
2574
+ raise _lib_journal.JournalProtocolError(
2575
+ f"cannot converge {event_id}: retained record is not a matching evt")
2576
+ if _lib_journal._sha256_canonical(record) != content_hash:
2577
+ raise _lib_journal.JournalProtocolError(
2578
+ f"cannot converge {event_id}: retained record hash mismatch")
2579
+ return record
2580
+
2581
+
2582
+ # Effect keys that ride an evt payload but are NOT target-table columns.
2583
+ _EVT_EFFECT_KEYS = frozenset(
2584
+ {"kind", "suppression", "suppression_table", "floor_suppression", "hwm_floor"}
2585
+ )
2586
+
2587
+
2588
+ def _evt_target_columns(conn, evt, spec) -> tuple:
2589
+ """Decode one evt payload into `(columns, children)` for its target row —
2590
+ the same mapping `_apply_generic_evt` performs, but WITHOUT inserting, so a
2591
+ convergence can UPDATE an existing physical row."""
2592
+ payload = evt.get("payload") or {}
2593
+ cols = {"journal_id": evt["id"]}
2594
+ children: dict = {}
2595
+ for key, value in payload.items():
2596
+ if key in _EVT_EFFECT_KEYS:
2597
+ continue
2598
+ if key in _BLOCK_CHILD_KEYS:
2599
+ children[key] = value or []
2600
+ continue
2601
+ if key in spec.fk_refs:
2602
+ column, ref_table = spec.fk_refs[key]
2603
+ cols[column] = _resolve_ref(conn, ref_table, value)
2604
+ else:
2605
+ cols[key] = value
2606
+ acct = cols.get("account_key")
2607
+ for column, (ref_table, lookup_col) in spec.derived_fk.items():
2608
+ cols[column] = _derived_fk_value(
2609
+ conn, ref_table, lookup_col, cols.get(lookup_col), acct)
2610
+ return cols, children
2611
+
2612
+
2613
+ CONVERGE_DROPPED = "dropped"
2614
+ CONVERGE_APPLIED = "converged"
2615
+
2616
+
2617
+ def _converge_row_from_effective(conn, event_id, *, table=None, rowid=None) -> str:
2618
+ """Bring the live physical row into agreement with the already-journaled
2619
+ effective event, and stamp `journal_id` in the SAME operation (#374 §6).
2620
+
2621
+ This is an EXPLICIT row-convergence operation, deliberately NOT a generic
2622
+ re-run of an arbitrary fold applier: `_apply_generic_evt` ends in
2623
+ `INSERT OR IGNORE`, and effect-bearing appliers cannot safely run out of
2624
+ canonical order. `five_hour_block_close` is the strengthened exception:
2625
+ its ordinary fold also converges the parent and exact child sets so orphan
2626
+ replay freezes an existing projection before raw derivation. Keeping the
2627
+ explicit convergence path still avoids invoking destructive effects and
2628
+ supports row-targeted conflict repair.
2629
+
2630
+ Effect-bearing families are NOT converged. `event_json` is authoritative row
2631
+ *data*, never permission to invoke every applier: `_apply_weekly_credit_
2632
+ effects` performs destructive deletes and writes a non-transactional HWM
2633
+ projection, and the reset appliers replay suppression deletes. Replaying an
2634
+ older effective event at the CURRENT execution point is not equivalent to
2635
+ folding it at its canonical journal position. So an effects-only family
2636
+ (`spec.table is None`) returns `CONVERGE_DROPPED` and nothing is replayed.
2637
+
2638
+ A family WITH a table but no physical row carrying the event id is also
2639
+ `CONVERGE_DROPPED`: convergence updates what exists, it never materialises a
2640
+ row (see the `rowid is None` branch).
2641
+ """
2642
+ record = _effective_event_for_convergence(conn, event_id)
2643
+ spec = _EVT_SPECS.get((record.get("payload") or {}).get("kind"))
2644
+ if spec is None or spec.table is None:
2645
+ return CONVERGE_DROPPED
2646
+ target = spec.table
2647
+ if table is not None and table != target:
2648
+ raise JournalError(
2649
+ f"convergence target mismatch for {event_id}: {table} != {target}")
2650
+ cols, children = _evt_target_columns(conn, record, spec)
2651
+ if rowid is None:
2652
+ row = conn.execute(
2653
+ f"SELECT id FROM {target} WHERE journal_id = ?", (event_id,)
2654
+ ).fetchone()
2655
+ rowid = int(row[0]) if row is not None else None
2656
+ if rowid is None:
2657
+ # NOTHING to converge — and materializing a row here would be wrong twice
2658
+ # over (#374 review). A row absent because a suppression effect
2659
+ # deliberately DELETED it would be resurrected, so the live index and a
2660
+ # rebuild would diverge — the very contract convergence exists to hold.
2661
+ # And an insert swallowed by a natural-key UNIQUE would leave the
2662
+ # follow-up lookup empty and abort the whole cycle on a `JournalError`.
2663
+ # Drop and report; the emission was already withheld by the caller.
2664
+ print(
2665
+ f"[journal] no live row for {event_id} in {target}; "
2666
+ "dropped the divergent emission without materializing one",
2667
+ file=sys.stderr,
2668
+ )
2669
+ return CONVERGE_DROPPED
2670
+ assignments = ", ".join(f"{name} = ?" for name in cols)
2671
+ conn.execute(
2672
+ f"UPDATE {target} SET {assignments} WHERE id = ?",
2673
+ (*cols.values(), int(rowid)),
2674
+ )
2675
+ if spec.applier is _apply_block_close:
2676
+ identity = conn.execute(
2677
+ "SELECT account_key, five_hour_window_key "
2678
+ "FROM five_hour_blocks WHERE id = ?",
2679
+ (int(rowid),),
2680
+ ).fetchone()
2681
+ if identity is None:
2682
+ raise JournalError(
2683
+ f"convergence target vanished for {event_id}"
2684
+ )
2685
+ _replace_block_children(
2686
+ conn, int(rowid), identity[0], identity[1], children
2687
+ )
2688
+ return CONVERGE_APPLIED
2689
+
2690
+
2691
+ def _validate_excluded_derived_fks(conn, spec, row) -> None:
2692
+ """Validate every column the harvest EXCLUDES from the evt, before the
2693
+ duplicate path stamps the row (#374 §6 / acceptance 8).
2694
+
2695
+ Byte identity of the emitted event proves the JOURNALED columns match. It
2696
+ proves nothing about the excluded ones: `_build_harvest_evt` omits
2697
+ `journal_id`, physical ids and derived-FK columns, and
2698
+ `five_hour_milestones.block_id` is deliberately derived rather than
2699
+ journaled — so a milestone pointing at the WRONG block can emit an otherwise
2700
+ byte-identical event. The canonical logical dump also excludes `block_id`
2701
+ and would not catch it.
2702
+
2703
+ An unresolvable reference is NOT an error on its own. `_derived_fk_value`
2704
+ returns 0 as the "no such parent" sentinel and the fold appliers store that
2705
+ same 0, so a legitimately parentless row — e.g. a `five_hour_milestones` row
2706
+ whose `five_hour_blocks` replica the 5h-credit stale-replica DELETE removed —
2707
+ carries `actual == expected == 0` and is in agreement. Raising on that shape
2708
+ escaped `_harvest`, rolled the cycle back, left the row unstamped and made
2709
+ every later cycle repeat it (#374 review). ONLY disagreement is fatal."""
2710
+ if not spec.derived_fk:
2711
+ return
2712
+ keys = set(row.keys())
2713
+ account_key = row["account_key"] if "account_key" in keys else None
2714
+ for column, (ref_table, lookup_col) in spec.derived_fk.items():
2715
+ expected = _derived_fk_value(
2716
+ conn, ref_table, lookup_col, row[lookup_col], account_key)
2717
+ actual = row[column]
2718
+ if int(actual) != expected:
2719
+ raise JournalError(
2720
+ f"harvest {spec.kind}: derived FK {column}={actual!r} does not "
2721
+ f"resolve to {ref_table}.{lookup_col}={row[lookup_col]!r} "
2722
+ f"(re-derived {expected})")
2723
+
2724
+
2725
+ def _full_effective_selection(hw):
2726
+ records = []
2727
+ evidence = []
2728
+ prior_high_water = None
2729
+ if hw is not None:
2730
+ for segment, offset, raw in _read_range(None, hw):
2731
+ record = _lib_journal.decode_line(raw)
2732
+ if record is not None:
2733
+ _capture_protocol_prefix_evidence(
2734
+ record,
2735
+ prior_high_water,
2736
+ evidence,
2737
+ )
2738
+ records.append(record)
2739
+ prior_high_water = (segment, offset + len(raw) + 1)
2740
+ cutover_claude = resolve_cutover_claude_account()
2741
+ for record in records:
2742
+ _normalize_legacy_account_stamp(record, cutover_claude)
2743
+ return _lib_journal.resolve_effective_events(
2744
+ records,
2745
+ protocol_prefix_evidence=evidence,
2746
+ )
2747
+
2748
+
2749
+ def _correction_commit_high_water(batch_id, hw=None):
2750
+ """Return the exact end offset of one completed-batch commit marker.
2751
+
2752
+ The batch was already structurally validated either by the full effective
2753
+ selector or by the live metadata row that names it. The earliest matching
2754
+ commit is the narrowest complete prefix and remains stable even when later
2755
+ journal bytes or crash-replayed duplicate markers exist.
2756
+ """
2757
+ if not batch_id:
2758
+ return None
2759
+ if hw is None:
2760
+ hw = journal_high_water()
2761
+ if hw is None:
2762
+ return None
2763
+ for segment, offset, raw in _read_range(None, hw):
2764
+ record = _lib_journal.decode_line(raw)
2765
+ if (
2766
+ record is not None
2767
+ and record.get("t") == "correction_batch"
2768
+ and record.get("phase") == "commit"
2769
+ and record.get("id") == batch_id
2770
+ ):
2771
+ return (segment, offset + len(raw) + 1)
2772
+ return None
2773
+
2774
+
2775
+ def _preflight_live_events(
2776
+ conn, records, hw, conflicts=None, protocol_scan=None
2777
+ ):
2778
+ """Validate unread evt/correction records before the stats transaction.
2779
+
2780
+ A READER (#374 §6): the evt records it inspects are already durably in the
2781
+ journal, so same-revision divergence must NOT raise here — that raise wedged
2782
+ every cycle over an already-poisoned journal, exactly like the rebuild. The
2783
+ divergent evt is DROPPED from the apply set, the prior effective event
2784
+ stands, and the group is appended to `conflicts` when a sink is supplied.
2785
+ `CorrectionRebuildRequired` stays fatal."""
2786
+ event_records = [record for record in records if record.get("t") == "evt"]
2787
+ selected_new = _lib_journal.resolve_effective_events(event_records)
2788
+ if conflicts is not None:
2789
+ conflicts.extend(selected_new.conflicts)
2790
+ to_apply = []
2791
+ for evt in selected_new.active:
2792
+ selected = selected_new.by_id[evt["id"]]
2793
+ prior = _metadata_row(conn, selected.event_id)
2794
+ if prior is None:
2795
+ to_apply.append(evt)
2796
+ continue
2797
+ prior_rev, prior_status, prior_hash, prior_batch = prior
2798
+ if int(prior_rev) == selected.rev:
2799
+ if prior_status != selected.status or prior_hash != selected.content_hash:
2800
+ if _legacy_qaa_can_advance(conn, selected):
2801
+ to_apply.append(evt)
2802
+ continue
2803
+ if conflicts is not None:
2804
+ conflicts.append(
2805
+ _lib_journal.EventConflict(
2806
+ event_id=selected.event_id,
2807
+ rev=selected.rev,
2808
+ content_hashes=tuple(
2809
+ sorted({prior_hash, selected.content_hash})),
2810
+ selected_hash=prior_hash,
2811
+ )
2812
+ )
2813
+ print(
2814
+ f"[journal] quarantined a divergent journal event for "
2815
+ f"{selected.event_id} rev {selected.rev}; the prior "
2816
+ "effective event stands",
2817
+ file=sys.stderr,
2818
+ )
2819
+ continue
2820
+ continue
2821
+ raise CorrectionRebuildRequired(
2822
+ f"event {selected.event_id} rev {selected.rev} conflicts with "
2823
+ f"effective rev {prior_rev} from {prior_batch or 'base journal'}",
2824
+ batch_id=prior_batch,
2825
+ event_id=selected.event_id,
2826
+ high_water=_correction_commit_high_water(prior_batch, hw),
2827
+ expected_metadata=(
2828
+ int(prior_rev),
2829
+ prior_status,
2830
+ prior_hash,
2831
+ prior_batch,
2832
+ ),
2833
+ )
2834
+
2835
+ if any(
2836
+ record.get("t") in {"correction", "correction_batch"}
2837
+ or (
2838
+ record.get("t") == "op"
2839
+ and isinstance(record.get("payload"), dict)
2840
+ and record["payload"].get("kind")
2841
+ == _lib_journal._PROTOCOL_RESOLUTION_KIND
2842
+ )
2843
+ for record in records
2844
+ ):
2845
+ full = _full_effective_selection(hw)
2846
+ if protocol_scan is not None:
2847
+ protocol_scan["scanned"] = True
2848
+ protocol_scan["violations"] = full.protocol_violations
2849
+ protocol_scan["acknowledged"] = (
2850
+ full.acknowledged_protocol_violations
2851
+ )
2852
+ for selected in full.by_id.values():
2853
+ if selected.batch_id is None:
2854
+ continue
2855
+ prior = _metadata_row(conn, selected.event_id)
2856
+ if prior is not None:
2857
+ prior_tuple = (int(prior[0]), prior[1], prior[2], prior[3])
2858
+ selected_tuple = (
2859
+ selected.rev,
2860
+ selected.status,
2861
+ selected.content_hash,
2862
+ selected.batch_id,
2863
+ )
2864
+ if prior_tuple == selected_tuple:
2865
+ continue
2866
+ raise CorrectionRebuildRequired(
2867
+ f"completed correction batch {selected.batch_id} requires "
2868
+ "stats index rebuild",
2869
+ batch_id=selected.batch_id,
2870
+ event_id=selected.event_id,
2871
+ high_water=_correction_commit_high_water(selected.batch_id, hw),
2872
+ expected_metadata=(
2873
+ selected.rev,
2874
+ selected.status,
2875
+ selected.content_hash,
2876
+ selected.batch_id,
2877
+ ),
2878
+ recovery_eligible=True,
2879
+ )
2880
+ return to_apply
2881
+
2882
+
2883
+ # --------------------------------------------------------------------------
2884
+ # the cycle (spec §5.2, revision 3)
2885
+ # --------------------------------------------------------------------------
2886
+
2887
+ def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
2888
+ codex_apply=None, post_commit=None) -> IngestResult:
2889
+ # Step 1: HW snapshot (leaf lock, µs). Lines appended after this — by other
2890
+ # processes OR by this cycle's own evt emission — are past HW and belong to
1864
2891
  # the next cycle (§5.2.1, closes the skipped-append race).
1865
2892
  hw = journal_high_water()
1866
2893
  # An empty journal (no segments yet) has nothing to consume. Normally that is
@@ -1909,7 +2936,17 @@ def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
1909
2936
 
1910
2937
  records = [r for (r, _s, _o) in decoded]
1911
2938
  batch = [r for r in records if r.get("t") in ("obs", "op")]
1912
- journal_evts = [r for r in records if r.get("t") == "evt"]
2939
+ # #374: the preflight reader quarantines same-revision divergence instead of
2940
+ # raising; the groups it drops are counted on the cycle summary.
2941
+ preflight_conflicts: list = []
2942
+ protocol_scan: dict = {}
2943
+ journal_evts = _preflight_live_events(
2944
+ conn,
2945
+ records,
2946
+ cursor_target,
2947
+ conflicts=preflight_conflicts,
2948
+ protocol_scan=protocol_scan,
2949
+ )
1913
2950
 
1914
2951
  # Step 4: ONE BEGIN IMMEDIATE — replay + pipeline + derived-fact journaling +
1915
2952
  # cursor advance, atomic (§5.2 crash boundary). A crash before COMMIT rolls
@@ -1925,7 +2962,8 @@ def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
1925
2962
  # before a referencing one (milestones); NO ctx, so replay is
1926
2963
  # structurally unable to fire an alert (§5.2 step 4a).
1927
2964
  for evt in sorted(journal_evts, key=_fold_order):
1928
- _apply_evt(conn, evt)
2965
+ if _record_live_effective_event(conn, evt):
2966
+ _apply_evt(conn, evt)
1929
2967
  # 4b. Per-record sequential pipeline over obs/op in canonical order —
1930
2968
  # sequential is REQUIRED (reset/credit detection precedes the same
1931
2969
  # record's snapshot-accept; a reset-spanning batch needs prior records'
@@ -1969,6 +3007,16 @@ def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
1969
3007
  # no-op when the batch carries no account stamps (byte-stable on a
1970
3008
  # pre-multi-account single-account install).
1971
3009
  _derive_account_last_seen(conn, records)
3010
+ # A full correction-prefix preflight is authoritative for the
3011
+ # disposable protocol summary. Replace it in the same transaction as
3012
+ # the cursor so shallow Doctor paths observe either the old complete
3013
+ # result or the new complete result, never an in-between state.
3014
+ if protocol_scan.get("scanned"):
3015
+ _write_protocol_violations(
3016
+ conn,
3017
+ protocol_scan.get("violations", ()),
3018
+ protocol_scan.get("acknowledged", ()),
3019
+ )
1972
3020
  # 4d. Advance the cursor (to HW, or to the cache-leg prefix boundary).
1973
3021
  # `cursor_target is None` ONLY on a reconcile-only cycle over a still-
1974
3022
  # empty journal (§5.2 above): there are no consumed lines to advance
@@ -2000,14 +3048,21 @@ def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
2000
3048
  (ALERT_DISPATCHER or _dispatch_pending_alerts)(alerts)
2001
3049
 
2002
3050
  return IngestResult(ran=True, consumed=len(records), malformed=malformed,
2003
- events_emitted=ctx.events_emitted, alerts=alerts)
3051
+ events_emitted=ctx.events_emitted, alerts=alerts,
3052
+ conflicts_dropped=(len(ctx.conflicts_dropped)
3053
+ + len(preflight_conflicts)))
2004
3054
 
2005
3055
 
2006
- def run_stats_ingest(*, mode: str = "opportunistic", timeout_s: float = 10.0,
2007
- conn: sqlite3.Connection | None = None,
2008
- reconcile_config=None, codex_apply=None,
2009
- post_commit=None) -> IngestResult:
2010
- """Run one ingest cycle as the single-flight stats.db writer (spec §5.1/§5.2).
3056
+ def _run_stats_ingest_once(
3057
+ *,
3058
+ mode: str = "opportunistic",
3059
+ timeout_s: float = 10.0,
3060
+ conn: sqlite3.Connection | None = None,
3061
+ reconcile_config=None,
3062
+ codex_apply=None,
3063
+ post_commit=None,
3064
+ ) -> IngestResult:
3065
+ """Run one single-flight attempt, without correction-recovery orchestration.
2011
3066
 
2012
3067
  `mode="opportunistic"` takes the ingest lock non-blocking (busy → `ran=False`;
2013
3068
  the current holder consumes the lines). `mode="authoritative"` waits up to
@@ -2040,17 +3095,112 @@ def run_stats_ingest(*, mode: str = "opportunistic", timeout_s: float = 10.0,
2040
3095
  never broken; an AUTHORITATIVE ingest re-raises so its caller (record-usage,
2041
3096
  record-credit, sync-week, statusline publication) sees the failure.
2042
3097
  """
2043
- lock_fd = _acquire_ingest_lock(mode, timeout_s)
2044
- if lock_fd is None:
2045
- return IngestResult(ran=False, consumed=0, malformed=0,
2046
- events_emitted=0, alerts=[])
2047
3098
  own_conn = conn is None
3099
+ maintenance_fd = None
3100
+ lock_fd = None
2048
3101
  try:
3102
+ # Let open_db resolve an epoch mismatch or classified corruption before
3103
+ # this caller owns any lock. It can therefore take maintenance EX ->
3104
+ # ingest in the required order. A fresh/legacy DB is different: its
3105
+ # open runs the one-time schema/cutover path, so serialize that whole
3106
+ # path under maintenance EX and downgrade to SH before taking ingest.
3107
+ # For a current/mismatched epoch, open first, then take maintenance SH
3108
+ # and verify the main-file identity did not change across the open; if
3109
+ # a sibling rebuilt in that gap, discard the stale handle and retry.
2049
3110
  if own_conn:
2050
- conn = _cctally_core.open_db()
3111
+ while True:
3112
+ raw_epoch = _stats_db_user_version()
3113
+ if (
3114
+ raw_epoch is None
3115
+ or raw_epoch <= _cctally_core.LEGACY_STATS_HEAD
3116
+ ):
3117
+ maintenance_fd = _acquire_maintenance_exclusive(
3118
+ mode, timeout_s
3119
+ )
3120
+ if maintenance_fd is None:
3121
+ return IngestResult(
3122
+ ran=False,
3123
+ consumed=0,
3124
+ malformed=0,
3125
+ events_emitted=0,
3126
+ alerts=[],
3127
+ )
3128
+ identity_before = _stats_db_identity()
3129
+ conn = _cctally_core.open_db()
3130
+ _downgrade_maintenance_shared(maintenance_fd)
3131
+ else:
3132
+ identity_before = _stats_db_identity()
3133
+ conn = _cctally_core.open_db()
3134
+ maintenance_fd = _acquire_maintenance_shared(
3135
+ mode, timeout_s
3136
+ )
3137
+ if maintenance_fd is None:
3138
+ if conn is not None:
3139
+ conn.close()
3140
+ conn = None
3141
+ return IngestResult(
3142
+ ran=False,
3143
+ consumed=0,
3144
+ malformed=0,
3145
+ events_emitted=0,
3146
+ alerts=[],
3147
+ )
3148
+ identity_after = _stats_db_identity()
3149
+ opened_epoch = conn.execute("PRAGMA user_version").fetchone()[0]
3150
+ epoch_ok = (
3151
+ opened_epoch <= _cctally_core.LEGACY_STATS_HEAD
3152
+ or opened_epoch == _cctally_core.STATS_INDEX_EPOCH
3153
+ )
3154
+ if (
3155
+ identity_before == identity_after
3156
+ and epoch_ok
3157
+ ):
3158
+ break
3159
+ _release_maintenance_shared(maintenance_fd)
3160
+ maintenance_fd = None
3161
+ conn.close()
3162
+ conn = None
3163
+ else:
3164
+ maintenance_fd = _acquire_maintenance_shared(mode, timeout_s)
3165
+ if maintenance_fd is None:
3166
+ return IngestResult(
3167
+ ran=False,
3168
+ consumed=0,
3169
+ malformed=0,
3170
+ events_emitted=0,
3171
+ alerts=[],
3172
+ )
3173
+
3174
+ lock_fd = _acquire_ingest_lock(mode, timeout_s)
3175
+ if lock_fd is None:
3176
+ if own_conn and conn is not None:
3177
+ conn.close()
3178
+ conn = None
3179
+ return IngestResult(
3180
+ ran=False,
3181
+ consumed=0,
3182
+ malformed=0,
3183
+ events_emitted=0,
3184
+ alerts=[],
3185
+ )
2051
3186
  try:
2052
- return _run_cycle(conn, reconcile_config=reconcile_config,
2053
- codex_apply=codex_apply, post_commit=post_commit)
3187
+ # #386: declare the sanctioned steady-state write regime for the
3188
+ # duration of the cycle. Two consumers: the Stage 3 authorizer, and
3189
+ # `holds_ingest_lock()` — a corruption surfacing from INSIDE the
3190
+ # cycle (via a nested `open_db()`, e.g. the cross-DB stats read on
3191
+ # the quota leg) reaches the heal hook while this process already
3192
+ # owns journal.ingest.lock, and the heal must recognise itself as
3193
+ # the serialized writer rather than wait 5s for a lock it holds.
3194
+ import _cctally_store
3195
+ with _cctally_store.stats_write_scope("ingest", ingest_lock=True):
3196
+ return _run_cycle(conn, reconcile_config=reconcile_config,
3197
+ codex_apply=codex_apply,
3198
+ post_commit=post_commit)
3199
+ except CorrectionRebuildRequired:
3200
+ # The public boundary must unwind its transaction, internally owned
3201
+ # connection, ingest lock, and maintenance-shared lock before it can
3202
+ # seek maintenance EXCLUSIVE in total order.
3203
+ raise
2054
3204
  except Exception as exc:
2055
3205
  if mode == "authoritative":
2056
3206
  raise
@@ -2065,7 +3215,242 @@ def run_stats_ingest(*, mode: str = "opportunistic", timeout_s: float = 10.0,
2065
3215
  if own_conn and conn is not None:
2066
3216
  conn.close()
2067
3217
  finally:
2068
- _release_ingest_lock(lock_fd)
3218
+ if lock_fd is not None:
3219
+ _release_ingest_lock(lock_fd)
3220
+ if maintenance_fd is not None:
3221
+ _release_maintenance_shared(maintenance_fd)
3222
+
3223
+
3224
+ def _correction_recovery_guidance(cause) -> str:
3225
+ detail = str(cause)
3226
+ lower = detail.lower()
3227
+ holder = (
3228
+ "open handle" in lower
3229
+ or "family is still open" in lower
3230
+ or "open in process" in lower
3231
+ )
3232
+ prefix = ""
3233
+ if holder:
3234
+ prefix = (
3235
+ "stop the dashboard or other process holding stats.db open, then "
3236
+ )
3237
+ return (
3238
+ f"{detail}; {prefix}run `cctally db rebuild --db stats` and retry"
3239
+ )
3240
+
3241
+
3242
+ def _correction_error_result(error) -> IngestResult:
3243
+ print(
3244
+ f"[ingest] correction recovery declined, cursor unmoved: {error}",
3245
+ file=sys.stderr,
3246
+ )
3247
+ return IngestResult(
3248
+ ran=True,
3249
+ consumed=0,
3250
+ malformed=0,
3251
+ events_emitted=0,
3252
+ alerts=[],
3253
+ error=error,
3254
+ )
3255
+
3256
+
3257
+ def _correction_index_converged(signal: CorrectionRebuildRequired) -> bool:
3258
+ """Revalidate the triggering effective row without open-time mutation."""
3259
+ if signal.event_id is None or signal.expected_metadata is None:
3260
+ return False
3261
+ path = pathlib.Path(_cctally_core.DB_PATH)
3262
+ if not path.exists():
3263
+ return False
3264
+ conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True, timeout=5.0)
3265
+ try:
3266
+ row = conn.execute(
3267
+ "SELECT rev, status, content_hash, batch_id "
3268
+ "FROM journal_effective_events WHERE event_id = ?",
3269
+ (signal.event_id,),
3270
+ ).fetchone()
3271
+ if row is None:
3272
+ return False
3273
+ return (
3274
+ int(row[0]),
3275
+ row[1],
3276
+ row[2],
3277
+ row[3],
3278
+ ) == tuple(signal.expected_metadata)
3279
+ finally:
3280
+ conn.close()
3281
+
3282
+
3283
+ def _correction_scratch_mains() -> set[pathlib.Path]:
3284
+ path = pathlib.Path(_cctally_core.DB_PATH)
3285
+ prefix = path.name + ".rebuilding-"
3286
+ return {
3287
+ member
3288
+ for member in path.parent.glob(prefix + "*")
3289
+ if not member.name.endswith(("-wal", "-shm"))
3290
+ }
3291
+
3292
+
3293
+ def _cleanup_new_correction_scratches(before: set[pathlib.Path]) -> None:
3294
+ for scratch in _correction_scratch_mains() - before:
3295
+ try:
3296
+ _remove_db_family(scratch)
3297
+ except OSError:
3298
+ pass
3299
+
3300
+
3301
+ def _recover_completed_correction(
3302
+ signal: CorrectionRebuildRequired,
3303
+ *,
3304
+ mode: str,
3305
+ timeout_s: float,
3306
+ ) -> None:
3307
+ """Revalidate and, when still needed, replace through the trigger prefix."""
3308
+ if (
3309
+ signal.batch_id is None
3310
+ or signal.event_id is None
3311
+ or signal.high_water is None
3312
+ or signal.expected_metadata is None
3313
+ ):
3314
+ raise CorrectionRecoveryError(
3315
+ _correction_recovery_guidance(
3316
+ "completed correction lacks an exact validated commit high-water"
3317
+ )
3318
+ )
3319
+
3320
+ maintenance_fd = _acquire_maintenance_exclusive(mode, timeout_s)
3321
+ if maintenance_fd is None:
3322
+ raise CorrectionRecoveryError(
3323
+ _correction_recovery_guidance(
3324
+ "stats maintenance lock is busy"
3325
+ )
3326
+ )
3327
+ ingest_fd = None
3328
+ try:
3329
+ ingest_fd = _acquire_ingest_lock(mode, timeout_s)
3330
+ if ingest_fd is None:
3331
+ raise CorrectionRecoveryError(
3332
+ _correction_recovery_guidance(
3333
+ "another ingest holds journal.ingest.lock"
3334
+ )
3335
+ )
3336
+
3337
+ # A sibling may have rebuilt after the original attempt unwound. The
3338
+ # locked re-check prevents redundant preservation/publication.
3339
+ try:
3340
+ converged = _correction_index_converged(signal)
3341
+ except Exception as exc:
3342
+ raise CorrectionRecoveryError(
3343
+ _correction_recovery_guidance(
3344
+ f"correction revalidation failed: {exc}"
3345
+ )
3346
+ ) from exc
3347
+ if converged:
3348
+ return
3349
+
3350
+ import _cctally_db
3351
+ if _cctally_db._would_block_prod_stats(_cctally_core.DB_PATH):
3352
+ raise CorrectionRecoveryError(
3353
+ _correction_recovery_guidance(
3354
+ "refusing to rebuild the prod stats.db from a dev checkout"
3355
+ )
3356
+ )
3357
+
3358
+ import _cctally_store
3359
+ scratches_before = _correction_scratch_mains()
3360
+ try:
3361
+ with _cctally_store.stats_write_scope(
3362
+ "maintenance-correction-rebuild",
3363
+ ingest_lock=True,
3364
+ ):
3365
+ rebuild_stats_index(high_water=signal.high_water)
3366
+ except BaseException as exc:
3367
+ _cleanup_new_correction_scratches(scratches_before)
3368
+ if isinstance(exc, (KeyboardInterrupt, SystemExit)):
3369
+ raise
3370
+ raise CorrectionRecoveryError(
3371
+ _correction_recovery_guidance(exc)
3372
+ ) from exc
3373
+ finally:
3374
+ if ingest_fd is not None:
3375
+ _release_ingest_lock(ingest_fd)
3376
+ _release_maintenance_shared(maintenance_fd)
3377
+
3378
+
3379
+ def run_stats_ingest(
3380
+ *,
3381
+ mode: str = "opportunistic",
3382
+ timeout_s: float = 10.0,
3383
+ conn: sqlite3.Connection | None = None,
3384
+ reconcile_config=None,
3385
+ codex_apply=None,
3386
+ post_commit=None,
3387
+ ) -> IngestResult:
3388
+ """Run one cycle, healing one completed-correction mismatch when safe.
3389
+
3390
+ The initial attempt fully unwinds before recovery seeks maintenance
3391
+ EXCLUSIVE then ingest. Recovery revalidates, rebuilds through the exact
3392
+ triggering commit, releases both locks, and retries once on a freshly opened
3393
+ current-family connection. Caller-owned connections are never closed or
3394
+ replaced. A second correction signal is surfaced with the manual remedy.
3395
+ """
3396
+ kwargs = {
3397
+ "mode": mode,
3398
+ "timeout_s": timeout_s,
3399
+ "conn": conn,
3400
+ "reconcile_config": reconcile_config,
3401
+ "codex_apply": codex_apply,
3402
+ "post_commit": post_commit,
3403
+ }
3404
+ try:
3405
+ return _run_stats_ingest_once(**kwargs)
3406
+ except CorrectionRebuildRequired as signal:
3407
+ if not signal.recovery_eligible:
3408
+ raise
3409
+ if conn is not None:
3410
+ raise CorrectionRebuildRequired(
3411
+ _correction_recovery_guidance(
3412
+ "automatic correction recovery cannot replace a "
3413
+ "caller-owned stats.db connection"
3414
+ ),
3415
+ batch_id=signal.batch_id,
3416
+ event_id=signal.event_id,
3417
+ high_water=signal.high_water,
3418
+ expected_metadata=signal.expected_metadata,
3419
+ recovery_eligible=True,
3420
+ ) from signal
3421
+
3422
+ try:
3423
+ _recover_completed_correction(
3424
+ signal,
3425
+ mode=mode,
3426
+ timeout_s=timeout_s,
3427
+ )
3428
+ except CorrectionRecoveryError as exc:
3429
+ if mode == "authoritative":
3430
+ raise
3431
+ return _correction_error_result(exc)
3432
+
3433
+ retry_kwargs = dict(kwargs)
3434
+ retry_kwargs["conn"] = None
3435
+ try:
3436
+ result = _run_stats_ingest_once(**retry_kwargs)
3437
+ except Exception as exc:
3438
+ wrapped = CorrectionRecoveryError(
3439
+ _correction_recovery_guidance(
3440
+ f"single correction-recovery retry failed: {exc}"
3441
+ )
3442
+ )
3443
+ if mode == "authoritative":
3444
+ raise wrapped from exc
3445
+ return _correction_error_result(wrapped)
3446
+ if result.error is not None:
3447
+ wrapped = CorrectionRecoveryError(
3448
+ _correction_recovery_guidance(
3449
+ f"single correction-recovery retry failed: {result.error}"
3450
+ )
3451
+ )
3452
+ return _correction_error_result(wrapped)
3453
+ return result
2069
3454
 
2070
3455
 
2071
3456
  # ==========================================================================
@@ -2133,14 +3518,29 @@ class RebuildResult:
2133
3518
  duration_s: float # wall time of the whole rebuild
2134
3519
  segments_read: int # journal segments folded
2135
3520
  lines_folded: int # op + evt lines applied (obs are rederive input)
2136
-
2137
-
2138
- def _remove_db_sidecars(path) -> None:
3521
+ # #374: divergent same-revision groups quarantined behind a lowest-sequence
3522
+ # provisional winner. The rebuild COMPLETES and exits 0 — reporting them is
3523
+ # how we refuse to assert that a guessed winner is authoritative.
3524
+ conflicts: tuple = ()
3525
+ # #402 Task A: whole correction batches omitted after one of the seven
3526
+ # enumerated structural violations. The usable index still publishes, but
3527
+ # every operator surface must report that intended corrections were omitted.
3528
+ protocol_violations: tuple = ()
3529
+ # #402 Task B: exact violations the operator acknowledged as omitted. The
3530
+ # batches remain tainted; this is diagnostic/audit state, never validity.
3531
+ acknowledged_protocol_violations: tuple = ()
3532
+ quarantine_dir: "pathlib.Path | None" = None
3533
+
3534
+
3535
+ def _remove_db_sidecars_strict(path) -> None:
3536
+ """Remove both sidecars or fail before publishing a replacement main file."""
2139
3537
  for suffix in ("-wal", "-shm"):
3538
+ candidate = pathlib.Path(str(path) + suffix)
2140
3539
  try:
2141
- pathlib.Path(str(path) + suffix).unlink()
2142
- except OSError:
3540
+ candidate.unlink()
3541
+ except FileNotFoundError:
2143
3542
  pass
3543
+ _fsync_dir(pathlib.Path(path).parent)
2144
3544
 
2145
3545
 
2146
3546
  def _remove_db_family(path) -> None:
@@ -2151,6 +3551,426 @@ def _remove_db_family(path) -> None:
2151
3551
  pass
2152
3552
 
2153
3553
 
3554
+ def _stats_rebuild_test_pause(point: str) -> None:
3555
+ """Private process-control seam for the #388 interrupted-rebuild tests."""
3556
+ if os.environ.get("CCTALLY_TEST_STATS_REBUILD_PAUSE_AT") != point:
3557
+ return
3558
+ marker = os.environ.get("CCTALLY_TEST_STATS_REBUILD_MARKER")
3559
+ if not marker:
3560
+ return
3561
+ pathlib.Path(marker).write_text(f"{os.getpid()}\n")
3562
+ os.kill(os.getpid(), signal.SIGSTOP)
3563
+
3564
+
3565
+ _REBUILD_REQUIRED_TABLES = frozenset(
3566
+ {
3567
+ "accounts",
3568
+ "budget_milestones",
3569
+ "five_hour_block_models",
3570
+ "five_hour_block_projects",
3571
+ "five_hour_blocks",
3572
+ "five_hour_milestones",
3573
+ "five_hour_reset_events",
3574
+ "journal_cursor",
3575
+ "journal_effective_events",
3576
+ "journal_protocol_violations",
3577
+ "percent_milestones",
3578
+ "project_budget_milestones",
3579
+ "projected_milestones",
3580
+ "quota_alert_arming",
3581
+ "quota_percent_milestones",
3582
+ "quota_projection_state",
3583
+ "quota_threshold_events",
3584
+ "quota_window_blocks",
3585
+ "schema_migrations",
3586
+ "schema_migrations_skipped",
3587
+ "stats_open_fixups",
3588
+ "week_reset_events",
3589
+ "weekly_cost_snapshots",
3590
+ "weekly_credit_floors",
3591
+ "weekly_usage_snapshots",
3592
+ }
3593
+ )
3594
+ _REBUILD_REQUIRED_INDEXES = frozenset(
3595
+ {
3596
+ "idx_budget_milestones_journal_id",
3597
+ "idx_budget_milestones_journal_id_null",
3598
+ "idx_cost_week_start_at_time",
3599
+ "idx_cost_week_time",
3600
+ "idx_five_hour_block_models_block",
3601
+ "idx_five_hour_block_models_window",
3602
+ "idx_five_hour_block_projects_block",
3603
+ "idx_five_hour_block_projects_window",
3604
+ "idx_five_hour_blocks_block_start",
3605
+ "idx_five_hour_blocks_journal_id",
3606
+ "idx_five_hour_blocks_journal_id_null",
3607
+ "idx_five_hour_milestones_block",
3608
+ "idx_five_hour_milestones_journal_id",
3609
+ "idx_five_hour_milestones_journal_id_null",
3610
+ "idx_five_hour_reset_events_journal_id",
3611
+ "idx_five_hour_reset_events_journal_id_null",
3612
+ "idx_percent_milestones_journal_id",
3613
+ "idx_percent_milestones_journal_id_null",
3614
+ "idx_project_budget_milestones_journal_id",
3615
+ "idx_project_budget_milestones_journal_id_null",
3616
+ "idx_projected_milestones_journal_id",
3617
+ "idx_projected_milestones_journal_id_null",
3618
+ "idx_quota_blocks_active",
3619
+ "idx_quota_milestones_active",
3620
+ "idx_quota_threshold_events_active",
3621
+ "idx_usage_week_start_at_time",
3622
+ "idx_usage_week_time",
3623
+ "idx_week_reset_events_journal_id",
3624
+ "idx_week_reset_events_journal_id_null",
3625
+ "idx_weekly_cost_snapshots_journal_id",
3626
+ "idx_weekly_credit_floors_journal_id",
3627
+ "idx_weekly_usage_snapshots_5h_window_key",
3628
+ "idx_weekly_usage_snapshots_journal_id",
3629
+ }
3630
+ )
3631
+ # SHA-256 of the current epoch's non-internal table/index sqlite_schema rows,
3632
+ # ordered by (type, name). Unlike table-name checks, this catches a silently
3633
+ # omitted column, constraint, partial predicate, or index definition. An epoch
3634
+ # schema change must update this contract alongside STATS_INDEX_EPOCH.
3635
+ _REBUILD_SCHEMA_FINGERPRINT = (
3636
+ "3e0ec46a965c9fa10ac827cfd1656c66ecae50ed289e3aa1c34d2cf6a3e5c4a3"
3637
+ )
3638
+
3639
+
3640
+ def _stats_schema_fingerprint(conn: sqlite3.Connection) -> str:
3641
+ rows = [
3642
+ tuple(row)
3643
+ for row in conn.execute(
3644
+ "SELECT type, name, tbl_name, sql FROM sqlite_schema "
3645
+ "WHERE type IN ('table', 'index') "
3646
+ "AND name NOT LIKE 'sqlite_%' ORDER BY type, name"
3647
+ )
3648
+ ]
3649
+ payload = json.dumps(rows, ensure_ascii=True, separators=(",", ":"))
3650
+ return hashlib.sha256(payload.encode("utf-8")).hexdigest()
3651
+
3652
+
3653
+ def _validate_rebuilt_stats_index(
3654
+ conn: sqlite3.Connection, high_water: "tuple[str, int] | None"
3655
+ ) -> None:
3656
+ """Validate the scratch index before it is eligible for publication."""
3657
+ integrity = [str(row[0]) for row in conn.execute("PRAGMA integrity_check")]
3658
+ if integrity != ["ok"]:
3659
+ raise JournalError(
3660
+ "rebuilt stats index failed integrity_check: " + "; ".join(integrity)
3661
+ )
3662
+
3663
+ epoch = int(conn.execute("PRAGMA user_version").fetchone()[0])
3664
+ if epoch != _cctally_core.STATS_INDEX_EPOCH:
3665
+ raise JournalError(
3666
+ f"rebuilt stats index has epoch {epoch}, expected "
3667
+ f"{_cctally_core.STATS_INDEX_EPOCH}"
3668
+ )
3669
+
3670
+ tables = {
3671
+ str(row[0])
3672
+ for row in conn.execute(
3673
+ "SELECT name FROM sqlite_schema "
3674
+ "WHERE type = 'table' AND name NOT LIKE 'sqlite_%'"
3675
+ )
3676
+ }
3677
+ missing_tables = sorted(_REBUILD_REQUIRED_TABLES - tables)
3678
+ unexpected_tables = sorted(tables - _REBUILD_REQUIRED_TABLES)
3679
+ if missing_tables or unexpected_tables:
3680
+ raise JournalError(
3681
+ "rebuilt stats index table contract mismatch"
3682
+ f"; missing={missing_tables!r}; unexpected={unexpected_tables!r}"
3683
+ )
3684
+
3685
+ indexes = {
3686
+ str(row[0])
3687
+ for row in conn.execute(
3688
+ "SELECT name FROM sqlite_schema "
3689
+ "WHERE type = 'index' AND name NOT LIKE 'sqlite_%'"
3690
+ )
3691
+ }
3692
+ missing_indexes = sorted(_REBUILD_REQUIRED_INDEXES - indexes)
3693
+ unexpected_indexes = sorted(indexes - _REBUILD_REQUIRED_INDEXES)
3694
+ if missing_indexes or unexpected_indexes:
3695
+ raise JournalError(
3696
+ "rebuilt stats index index contract mismatch"
3697
+ f"; missing={missing_indexes!r}; unexpected={unexpected_indexes!r}"
3698
+ )
3699
+
3700
+ schema_fingerprint = _stats_schema_fingerprint(conn)
3701
+ if schema_fingerprint != _REBUILD_SCHEMA_FINGERPRINT:
3702
+ raise JournalError(
3703
+ "rebuilt stats index schema definition mismatch: "
3704
+ f"{schema_fingerprint}, expected {_REBUILD_SCHEMA_FINGERPRINT}"
3705
+ )
3706
+
3707
+ # Force representative table and cursor B-tree reads. Header readability
3708
+ # and a constant-only SELECT do not establish that the index is usable.
3709
+ conn.execute(
3710
+ "SELECT id, journal_id FROM weekly_usage_snapshots "
3711
+ "ORDER BY id DESC LIMIT 1"
3712
+ ).fetchall()
3713
+ cursor_row = conn.execute(
3714
+ "SELECT segment, offset, applied_segment, applied_offset "
3715
+ "FROM journal_cursor WHERE id = 1"
3716
+ ).fetchone()
3717
+ actual_cursor = (
3718
+ (str(cursor_row[0]), int(cursor_row[1]))
3719
+ if cursor_row is not None
3720
+ else None
3721
+ )
3722
+ applied_cursor = (
3723
+ (str(cursor_row[2]), int(cursor_row[3]))
3724
+ if cursor_row is not None
3725
+ and cursor_row[2] is not None
3726
+ and cursor_row[3] is not None
3727
+ else None
3728
+ )
3729
+ if actual_cursor != high_water or applied_cursor != high_water:
3730
+ raise JournalError(
3731
+ "rebuilt stats index cursor contract "
3732
+ f"(public={actual_cursor!r}, applied={applied_cursor!r}) "
3733
+ f"does not match pinned journal high-water {high_water!r}"
3734
+ )
3735
+
3736
+
3737
+ def stats_index_matches_journal_prefix(
3738
+ path: pathlib.Path, high_water: "tuple[str, int] | None"
3739
+ ) -> bool:
3740
+ """Whether ``path`` is a fully valid materialization of ``high_water``.
3741
+
3742
+ This is intentionally stronger than "the index has rows": it validates the
3743
+ full Task A publication contract and compares the disposable effective-event
3744
+ summary with the canonical journal selection. A legitimate empty index
3745
+ therefore matches an empty selection, while a valid-looking empty/partial
3746
+ index over data-bearing journal events does not.
3747
+ """
3748
+ if not pathlib.Path(path).exists():
3749
+ return False
3750
+ try:
3751
+ conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
3752
+ try:
3753
+ _validate_rebuilt_stats_index(conn, high_water)
3754
+ decoded: list[dict] = []
3755
+ protocol_evidence = []
3756
+ prior_high_water = None
3757
+ if high_water is not None:
3758
+ for segment, offset, raw in _read_range(None, high_water):
3759
+ record = _lib_journal.decode_line(raw)
3760
+ if record is not None:
3761
+ _capture_protocol_prefix_evidence(
3762
+ record,
3763
+ prior_high_water,
3764
+ protocol_evidence,
3765
+ )
3766
+ decoded.append(record)
3767
+ prior_high_water = (
3768
+ segment,
3769
+ offset + len(raw) + 1,
3770
+ )
3771
+ cutover_claude = resolve_cutover_claude_account()
3772
+ for record in decoded:
3773
+ _normalize_legacy_account_stamp(record, cutover_claude)
3774
+ selection = _lib_journal.resolve_effective_events(
3775
+ decoded,
3776
+ protocol_prefix_evidence=protocol_evidence,
3777
+ )
3778
+ expected = []
3779
+ for event_id, selected in selection.by_id.items():
3780
+ event_json = None
3781
+ if selected.record is not None:
3782
+ event_json = (
3783
+ _lib_journal.encode_line(selected.record)
3784
+ .decode("utf-8")
3785
+ .rstrip("\n")
3786
+ )
3787
+ expected.append(
3788
+ (
3789
+ event_id,
3790
+ selected.rev,
3791
+ selected.status,
3792
+ selected.content_hash,
3793
+ selected.batch_id,
3794
+ event_json,
3795
+ )
3796
+ )
3797
+ expected.sort(key=lambda row: row[0])
3798
+ actual = [
3799
+ tuple(row)
3800
+ for row in conn.execute(
3801
+ "SELECT event_id, rev, status, content_hash, batch_id, "
3802
+ "event_json FROM journal_effective_events ORDER BY event_id"
3803
+ )
3804
+ ]
3805
+ if actual != expected:
3806
+ return False
3807
+ expected_violation_rows = [
3808
+ *selection.protocol_violations,
3809
+ *selection.acknowledged_protocol_violations,
3810
+ ]
3811
+ expected_violation_rows.sort(
3812
+ key=lambda violation: (
3813
+ violation.batch_id,
3814
+ violation.kind,
3815
+ violation.fingerprint,
3816
+ )
3817
+ )
3818
+ expected_violations = [
3819
+ json.dumps(
3820
+ violation.to_dict(),
3821
+ sort_keys=True,
3822
+ separators=(",", ":"),
3823
+ )
3824
+ for violation in expected_violation_rows
3825
+ ]
3826
+ actual_violations = [
3827
+ str(row[0])
3828
+ for row in conn.execute(
3829
+ "SELECT violation_json FROM journal_protocol_violations "
3830
+ "ORDER BY batch_id, kind, fingerprint"
3831
+ )
3832
+ ]
3833
+ if actual_violations != expected_violations:
3834
+ return False
3835
+ for record in selection.active:
3836
+ if record.get("t") != "evt":
3837
+ continue
3838
+ spec = _EVT_SPECS.get((record.get("payload") or {}).get("kind"))
3839
+ if spec is None or spec.table is None:
3840
+ continue
3841
+ row = conn.execute(
3842
+ f"SELECT id FROM {spec.table} WHERE journal_id = ?",
3843
+ (record["id"],),
3844
+ ).fetchone()
3845
+ if row is None:
3846
+ return False
3847
+ if spec.applier is _apply_block_close:
3848
+ block_id = int(row[0])
3849
+ payload = record.get("payload") or {}
3850
+ for payload_key, child_table in _BLOCK_CHILDREN:
3851
+ child_count = conn.execute(
3852
+ f"SELECT COUNT(*) FROM {child_table} WHERE block_id = ?",
3853
+ (block_id,),
3854
+ ).fetchone()[0]
3855
+ if int(child_count) != len(payload.get(payload_key) or ()):
3856
+ return False
3857
+ return True
3858
+ finally:
3859
+ conn.close()
3860
+ except (
3861
+ OSError,
3862
+ sqlite3.DatabaseError,
3863
+ JournalError,
3864
+ _lib_journal.JournalProtocolError,
3865
+ ):
3866
+ return False
3867
+
3868
+
3869
+ def _prepare_existing_stats_for_cutover(path: pathlib.Path) -> None:
3870
+ """Checkpoint a readable old index so removing its sidecars is kill-safe."""
3871
+ import _cctally_db
3872
+
3873
+ try:
3874
+ conn = sqlite3.connect(str(path), timeout=15.0)
3875
+ try:
3876
+ conn.execute("PRAGMA schema_version").fetchone()
3877
+ checkpoint = conn.execute("PRAGMA wal_checkpoint(TRUNCATE)").fetchone()
3878
+ if checkpoint is not None and int(checkpoint[0]) != 0:
3879
+ raise JournalError(
3880
+ "old stats index WAL could not be drained before cutover"
3881
+ )
3882
+ finally:
3883
+ conn.close()
3884
+ except sqlite3.DatabaseError as exc:
3885
+ # Auto-heal necessarily starts from an unreadable family. Preserve its
3886
+ # exact bytes below, then publish the already-validated replacement.
3887
+ if _cctally_db._is_sqlite_corruption_error(exc):
3888
+ return
3889
+ raise
3890
+
3891
+
3892
+ def _preserve_stats_family_for_cutover(path: pathlib.Path) -> pathlib.Path:
3893
+ """Durably copy the old family into quarantine without removing the main."""
3894
+ import _cctally_db
3895
+
3896
+ root = _cctally_core.APP_DIR / "quarantine"
3897
+ root.mkdir(parents=True, exist_ok=True)
3898
+ # The quarantine entry itself must survive power loss before any old
3899
+ # sidecar can be removed. fsyncing only the new root/incident cannot make
3900
+ # the root's directory entry durable in APP_DIR.
3901
+ _fsync_dir(_cctally_core.APP_DIR)
3902
+ try:
3903
+ os.chmod(root, 0o700)
3904
+ except OSError:
3905
+ pass
3906
+ stamp = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%S_%f")
3907
+ incident = root / f"{path.name}-{stamp}"
3908
+ incident.mkdir(mode=0o700)
3909
+ destination = incident / path.name
3910
+ members = [
3911
+ pathlib.Path(str(path) + suffix).name
3912
+ for suffix in ("", "-wal", "-shm")
3913
+ if pathlib.Path(str(path) + suffix).exists()
3914
+ ]
3915
+ if not members:
3916
+ raise OSError(f"no database family exists to preserve at {path}")
3917
+ _cctally_db._copy_db_family(path, destination)
3918
+ manifest = {
3919
+ "schemaVersion": 1,
3920
+ "quarantinedAtUtc": dt.datetime.now(dt.timezone.utc).isoformat(
3921
+ timespec="seconds"
3922
+ ).replace("+00:00", "Z"),
3923
+ "originalPath": str(path),
3924
+ "movedFiles": members,
3925
+ "complete": True,
3926
+ "cutoverProtocol": "preserve-then-atomic-replace-v1",
3927
+ }
3928
+ _cctally_db._atomic_write_private_json(incident / "manifest.json", manifest)
3929
+ _fsync_dir(incident)
3930
+ _fsync_dir(root)
3931
+ return incident
3932
+
3933
+
3934
+ def _publish_rebuilt_stats_index(
3935
+ *,
3936
+ scratch: pathlib.Path,
3937
+ destination: pathlib.Path,
3938
+ preserve_existing: bool,
3939
+ before_swap=None,
3940
+ ) -> "pathlib.Path | None":
3941
+ """Publish one validated, closed, sidecar-free scratch index atomically."""
3942
+ import _cctally_store
3943
+
3944
+ family_exists = any(
3945
+ pathlib.Path(str(destination) + suffix).exists()
3946
+ for suffix in ("", "-wal", "-shm")
3947
+ )
3948
+ incident = None
3949
+ if family_exists:
3950
+ blocked = _cctally_store._stats_family_drained(destination)
3951
+ if blocked is not None:
3952
+ raise JournalError(f"stats.db cutover declined: {blocked}")
3953
+ _cctally_store._stats_storm_test_pause("stats_replace_drained")
3954
+ if preserve_existing:
3955
+ # Preserve the exact pre-cutover family, including a committed WAL
3956
+ # and SHM, before checkpointing mutates or removes those sidecars.
3957
+ incident = _preserve_stats_family_for_cutover(destination)
3958
+ if destination.exists():
3959
+ _prepare_existing_stats_for_cutover(destination)
3960
+ # The old main stays present and, when it was readable, fully
3961
+ # checkpointed. A kill from here until os.replace therefore still
3962
+ # leaves a usable old destination while preventing stale sidecars from
3963
+ # being paired with the replacement main.
3964
+ _remove_db_sidecars_strict(destination)
3965
+
3966
+ if before_swap is not None:
3967
+ before_swap()
3968
+ _stats_rebuild_test_pause("rebuild_before_cutover")
3969
+ os.replace(str(scratch), str(destination))
3970
+ _fsync_dir(destination.parent)
3971
+ return incident
3972
+
3973
+
2154
3974
  def _rebuild_quota_cache_leg(records) -> None:
2155
3975
  """Re-materialize cache.db `quota_window_snapshots` from the journal's Codex
2156
3976
  quota obs (spec §5.4). The journal obs are the DURABLE source (§1 latent
@@ -2209,7 +4029,13 @@ def _rebuild_quota_cache_leg(records) -> None:
2209
4029
  release_cache_writer_flocks(held)
2210
4030
 
2211
4031
 
2212
- def rebuild_stats_index(*, target_path=None) -> RebuildResult:
4032
+ def rebuild_stats_index(
4033
+ *,
4034
+ target_path=None,
4035
+ high_water: "tuple[str, int] | None" = None,
4036
+ update_quota_cache: bool = True,
4037
+ before_swap=None,
4038
+ ) -> RebuildResult:
2213
4039
  """Build a FRESH stats index from the journal alone (spec §5.4).
2214
4040
 
2215
4041
  Replays every segment in canonical `(segment, offset)` order into a fresh
@@ -2219,10 +4045,14 @@ def rebuild_stats_index(*, target_path=None) -> RebuildResult:
2219
4045
  no alerts, no `reconcile_config` (see the module note above). Post-rebuild the
2220
4046
  cursor equals the journal high-water.
2221
4047
 
2222
- `target_path` selects the destination (default `DB_PATH`). The caller
2223
- (auto-heal HEAL_HOOK / `db rebuild`) forensics-quarantines the damaged/old DB
2224
- FIRST, so the destination is absent at swap time; a `target_path` build (used
2225
- by determinism tests) writes an independent index without touching `DB_PATH`.
4048
+ `target_path` selects the destination (default `DB_PATH`). `high_water`
4049
+ optionally pins the exact inclusive journal prefix; later bytes stay beyond
4050
+ the rebuilt cursor. `update_quota_cache=False` is the Task-C Claude-only
4051
+ path whose caller already holds a stable cache exclusion. The common
4052
+ cutover keeps a live destination in place until its validated replacement
4053
+ is ready, preserves the old family, detaches old sidecars, and atomically
4054
+ replaces the main file. A `target_path` build uses the same atomic
4055
+ publication but does not create a live-family quarantine incident.
2226
4056
  """
2227
4057
  start = time.monotonic()
2228
4058
  dest = (pathlib.Path(target_path) if target_path is not None
@@ -2231,8 +4061,19 @@ def rebuild_stats_index(*, target_path=None) -> RebuildResult:
2231
4061
  # HW snapshot at the START — lines appended during the rebuild are past HW
2232
4062
  # and belong to the next ingest cycle (they replay idempotently); mirrors the
2233
4063
  # live cycle's §5.2.1 HW-prefix rule.
2234
- hw = journal_high_water()
4064
+ hw = high_water if high_water is not None else journal_high_water()
2235
4065
  segments = list_segments()
4066
+ if hw is not None:
4067
+ if hw[0] not in segments:
4068
+ raise JournalError(
4069
+ f"rebuild high-water segment is missing: {hw[0]}"
4070
+ )
4071
+ current_size = os.path.getsize(_cctally_core.JOURNAL_DIR / hw[0])
4072
+ if hw[1] < 0 or hw[1] > current_size:
4073
+ raise JournalError(
4074
+ f"rebuild high-water offset is invalid for {hw[0]}: {hw[1]}"
4075
+ )
4076
+ segments = segments[:segments.index(hw[0]) + 1]
2236
4077
 
2237
4078
  stamp = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%S_%f")
2238
4079
  scratch = dest.with_name(dest.name + f".rebuilding-{stamp}")
@@ -2246,13 +4087,28 @@ def rebuild_stats_index(*, target_path=None) -> RebuildResult:
2246
4087
  lines_folded = 0
2247
4088
  try:
2248
4089
  decoded: list = []
4090
+ protocol_evidence = []
4091
+ prior_high_water = None
2249
4092
  if hw is not None:
2250
- for _seg, _off, raw in _read_range(None, hw):
4093
+ for segment, offset, raw in _read_range(None, hw):
2251
4094
  rec = _lib_journal.decode_line(raw)
2252
4095
  if rec is None:
2253
4096
  malformed += 1
4097
+ prior_high_water = (
4098
+ segment,
4099
+ offset + len(raw) + 1,
4100
+ )
2254
4101
  continue
4102
+ _capture_protocol_prefix_evidence(
4103
+ rec,
4104
+ prior_high_water,
4105
+ protocol_evidence,
4106
+ )
2255
4107
  decoded.append(rec)
4108
+ prior_high_water = (
4109
+ segment,
4110
+ offset + len(raw) + 1,
4111
+ )
2256
4112
 
2257
4113
  # Legacy account normalisation (#341, spec §2 / handoff item 2): a
2258
4114
  # pre-#341 real-account line lacks an account stamp — inject the cutover
@@ -2265,9 +4121,18 @@ def rebuild_stats_index(*, target_path=None) -> RebuildResult:
2265
4121
  for rec in decoded:
2266
4122
  _normalize_legacy_account_stamp(rec, cutover_claude)
2267
4123
 
4124
+ # Resolve corrections BEFORE either disposable index is mutated. A
4125
+ # malformed revision, divergent same-revision candidate, or invalid
4126
+ # committed manifest leaves the existing destination untouched.
4127
+ effective = _lib_journal.resolve_effective_events(
4128
+ decoded,
4129
+ protocol_prefix_evidence=protocol_evidence,
4130
+ )
4131
+
2268
4132
  # Cache leg BEFORE any stats txn (provider-flock lock-order): journal
2269
4133
  # Codex quota obs -> cache.db quota_window_snapshots.
2270
- _rebuild_quota_cache_leg(decoded)
4134
+ if update_quota_cache:
4135
+ _rebuild_quota_cache_leg(decoded)
2271
4136
 
2272
4137
  # One ordered fold stream: op-folds (order 5) + evts, keyed by
2273
4138
  # (fold_order, canonical seq) so referenced families resolve before
@@ -2278,16 +4143,19 @@ def rebuild_stats_index(*, target_path=None) -> RebuildResult:
2278
4143
  kind = (rec.get("payload") or {}).get("kind")
2279
4144
  if t == "op" and kind in FOLD_APPLIERS:
2280
4145
  stream.append((_OP_FOLD_ORDER, seq, "op", rec))
2281
- elif t == "evt":
2282
- stream.append((_fold_order(rec), seq, "evt", rec))
4146
+ for seq, rec in enumerate(effective.active):
4147
+ stream.append((_fold_order(rec), seq, "evt", rec))
2283
4148
  stream.sort(key=lambda x: (x[0], x[1]))
2284
4149
  structural = [s for s in stream if s[0] < _REBUILD_MILESTONE_ORDER]
2285
4150
  tail = [s for s in stream if s[0] >= _REBUILD_MILESTONE_ORDER]
2286
4151
 
4152
+ _stats_rebuild_test_pause("rebuild_fold_started")
4153
+
2287
4154
  # Phase 1 (txn A) — structural folds: op floors, snapshot_accept, cost
2288
4155
  # snapshots, resets+suppression, block_close, arming, credit effects.
2289
4156
  conn.execute("BEGIN IMMEDIATE")
2290
4157
  try:
4158
+ _write_effective_metadata(conn, effective)
2291
4159
  for _order, _seq, kind, rec in structural:
2292
4160
  if kind == "op":
2293
4161
  FOLD_APPLIERS[(rec.get("payload") or {}).get("kind")](conn, rec)
@@ -2350,21 +4218,35 @@ def rebuild_stats_index(*, target_path=None) -> RebuildResult:
2350
4218
  except sqlite3.Error:
2351
4219
  rows_by_table[tbl] = 0
2352
4220
  # Drain the WAL into the main file so the atomic rename carries all data.
2353
- conn.execute("PRAGMA wal_checkpoint(TRUNCATE)")
4221
+ checkpoint = conn.execute("PRAGMA wal_checkpoint(TRUNCATE)").fetchone()
4222
+ if checkpoint is not None and int(checkpoint[0]) != 0:
4223
+ raise JournalError("rebuilt stats index WAL could not be drained")
4224
+ _validate_rebuilt_stats_index(conn, hw)
4225
+ _stats_rebuild_test_pause("rebuild_scratch_complete")
2354
4226
  finally:
2355
4227
  conn.close()
2356
4228
 
2357
- # Atomic swap: the freshly-built scratch becomes the destination. Its WAL was
2358
- # drained above; drop the empty sidecars, rename, and clear any stale
2359
- # destination sidecars (a fresh open recreates its own).
2360
- _remove_db_sidecars(scratch)
2361
- os.replace(str(scratch), str(dest))
2362
- _remove_db_sidecars(dest)
4229
+ # Closed, drained, validated, and durable before the old family is touched.
4230
+ _remove_db_sidecars_strict(scratch)
4231
+ with scratch.open("rb") as handle:
4232
+ os.fsync(handle.fileno())
4233
+ _fsync_dir(scratch.parent)
4234
+ incident = _publish_rebuilt_stats_index(
4235
+ scratch=scratch,
4236
+ destination=dest,
4237
+ preserve_existing=target_path is None,
4238
+ before_swap=before_swap,
4239
+ )
2363
4240
 
2364
4241
  return RebuildResult(
2365
4242
  rows_by_table=rows_by_table, malformed=malformed,
2366
4243
  duration_s=time.monotonic() - start, segments_read=len(segments),
2367
- lines_folded=lines_folded,
4244
+ lines_folded=lines_folded, conflicts=effective.conflicts,
4245
+ protocol_violations=effective.protocol_violations,
4246
+ acknowledged_protocol_violations=(
4247
+ effective.acknowledged_protocol_violations
4248
+ ),
4249
+ quarantine_dir=incident,
2368
4250
  )
2369
4251
 
2370
4252
 
@@ -2396,10 +4278,26 @@ def rebuild_stats_index(*, target_path=None) -> RebuildResult:
2396
4278
  # the stamping runs; a re-run after any crash re-exports byte-identical lines
2397
4279
  # (ids are `b:<table>:<rowid>`, independent of the retry's timestamp), so a
2398
4280
  # duplicate/leftover bootstrap folds idempotently (`INSERT OR IGNORE`). The
2399
- # cutover does NOT take the ingest lock (open_db reaches it from INSIDE
2400
- # run_stats_ingest's own ingest-lock hold — re-acquiring would self-deadlock):
2401
- # single-flight of the STAMP is provided by `BEGIN IMMEDIATE`, and concurrent
2402
- # cutovers converge by id.
4281
+ # cutover does NOT take the ingest lock. **The conclusion is right; the reason
4282
+ # once written here was false and is corrected (#386).** It was: "open_db
4283
+ # reaches it from INSIDE run_stats_ingest's own ingest-lock hold — re-acquiring
4284
+ # would self-deadlock." It does not: `run_stats_ingest` takes maintenance
4285
+ # EXCLUSIVE *before* `open_db()` on the legacy/fresh branch and only acquires
4286
+ # `journal.ingest.lock` after `open_db` has returned, so cutover never runs
4287
+ # under an ingest hold from that path.
4288
+ #
4289
+ # The real reason is that cutover runs under maintenance-EXCLUSIVE — since #386,
4290
+ # unconditionally, via `_cctally_store.stats_open_time_guard()` around
4291
+ # `open_db`'s whole open-time mutation region, whichever command reached it.
4292
+ # Maintenance-exclusive already serializes it against ingest, so the ingest lock
4293
+ # would add nothing; single-flight of the STAMP is provided by `BEGIN
4294
+ # IMMEDIATE`, and concurrent cutovers converge by id.
4295
+ #
4296
+ # DO NOT "fix" this by making cutover take the ingest lock. `holds_ingest_lock()`
4297
+ # now exists, and the deleted sentence reads like an invitation to add the
4298
+ # acquire with a re-entrancy check. Taking maintenance then ingest here would be
4299
+ # in lock order and would not deadlock — it would simply be a second, redundant
4300
+ # lock on a path that already holds the stronger one.
2403
4301
 
2404
4302
 
2405
4303
  def _cutover_iso(dt_utc: dt.datetime) -> str:
@@ -2434,9 +4332,8 @@ class _CutoverSpec:
2434
4332
  # (quota_alert_arming): its fold applier converges by NATURAL-KEY upsert, so
2435
4333
  # there is nothing to stamp back and it is excluded from the no-NULL-survivors
2436
4334
  # invariant (§8). When `natural_key_id` is set, the exported evt id is the
2437
- # natural-key form (`<natural_key_prefix>:<col>:<col>…`) matching the LIVE
2438
- # emission (so a cutover-exported record and a later live re-emission share
2439
- # one id) instead of the `b:<table>:<rowid>` bootstrap id.
4335
+ # state-instance form (`<natural_key_prefix>:<col>:<col>…`) matching the LIVE
4336
+ # emission, instead of the `b:<table>:<rowid>` bootstrap id.
2440
4337
  stamp: bool = True
2441
4338
  natural_key_prefix: str = "" # evt_id kind prefix (e.g. "qaa")
2442
4339
  natural_key_id: tuple = () # columns forming the natural-key evt id
@@ -2480,23 +4377,18 @@ _CUTOVER_SPECS = (
2480
4377
  # forward-only alert clock (`activated_at_utc`) that must survive rebuild so
2481
4378
  # the reconcile honors it (no historical re-fires). No journal_id column →
2482
4379
  # NOT stamped; the fold applier upserts by natural key. The evt id is the
2483
- # `qaa:` natural-key form (matching the live emission in
2484
- # `_cctally_quota._codex_leg._emit_arming`), so a cutover-exported arming
2485
- # record and a later live re-emission for the same identity are ONE record.
4380
+ # `qaa:` state-instance form (matching the live emission in
4381
+ # `_cctally_quota._codex_leg._emit_arming`): the natural row key is followed
4382
+ # by fingerprint + activation boundary so distinct state transitions never
4383
+ # collide at rev 0, while exact re-emission of one state converges.
2486
4384
  _CutoverSpec("quota_alert_arming", "quota_alert_arming", "evt",
2487
4385
  "activated_at_utc", stamp=False, natural_key_prefix="qaa",
2488
4386
  natural_key_id=("source", "source_root_key", "account_key",
2489
4387
  "logical_limit_key", "observed_slot",
2490
- "window_minutes")),
4388
+ "window_minutes", "rule_fingerprint",
4389
+ "activated_at_utc")),
2491
4390
  )
2492
4391
 
2493
- # Journal-covered stats tables whose rows get a `journal_id` stamp at cutover.
2494
- # five_hour_blocks stamps only its CLOSED rows (open blocks stay NULL — they are
2495
- # re-materialized projections). `stamp=False` families (quota_alert_arming: no
2496
- # journal_id column) are excluded — they converge by natural-key upsert.
2497
- _CUTOVER_STAMP_TABLES = tuple(s.table for s in _CUTOVER_SPECS if s.stamp)
2498
-
2499
-
2500
4392
  def _export_stats_table(conn, spec) -> list:
2501
4393
  """Return `[(line_record, rowid), ...]` for every row of `spec.table`
2502
4394
  (closed rows only when `spec.closed_only`). Bootstrap id = b:<table>:<rowid>;
@@ -2528,13 +4420,13 @@ def _export_stats_table(conn, spec) -> list:
2528
4420
  payload[payload_key] = [
2529
4421
  {k: cr[k] for k in cr.keys() if k not in ("id", "block_id")}
2530
4422
  for cr in child_rows]
4423
+ if spec.kind == "quota_alert_arming":
4424
+ payload["journal_identity_version"] = 2
2531
4425
  if spec.natural_key_id:
2532
- # §5.3 "state" family: the evt id is the natural-key form (matching
2533
- # the live emission), NOT the b:<table>:<rowid> bootstrap id. A legacy
2534
- # (pre-#341) stats.db has no `account_key` column, so a natural-key
2535
- # component absent from the row is the sentinel (#341): the exported
2536
- # qaa id becomes `qaa:...:unattributed:...`, matching what a live
2537
- # re-emission for the unattributed identity would build.
4426
+ # §5.3 "state" family: the evt id is the state-instance form (matching
4427
+ # the live emission), NOT the b:<table>:<rowid> bootstrap id. A
4428
+ # legacy (pre-#341) stats.db has no `account_key` column, so that
4429
+ # missing component is the sentinel (#341).
2538
4430
  row_cols = set(row.keys())
2539
4431
  bid = _lib_journal.evt_id(
2540
4432
  spec.natural_key_prefix,
@@ -2704,9 +4596,10 @@ def run_cutover(conn, *, now_utc: dt.datetime | None = None) -> "str | None":
2704
4596
  # Account epoch-transition coordinator (#341, spec §2)
2705
4597
  # ==========================================================================
2706
4598
  #
2707
- # Epoch 1000 -> 1001 adds the account dimension. An existing epoch-1000 stats.db
2708
- # reaches `resolve_stats_epoch_mismatch`, which runs this coordinator BEFORE the
2709
- # rebuild, in exact order (spec §2, review finding 1):
4599
+ # The epoch-transition coordinator was introduced when 1000 -> 1001 added the
4600
+ # account dimension. Later disposable-index epoch bumps reuse it idempotently:
4601
+ # `resolve_stats_epoch_mismatch` runs this coordinator BEFORE the rebuild, in
4602
+ # exact order (spec §2, review finding 1):
2710
4603
  # (1) resolve the cutover identity WITHOUT opening stats.db — a stable-read of
2711
4604
  # ~/.claude.json; stably-absent / torn -> `unattributed` (never a guess);
2712
4605
  # (2) atomically check/append the canonical cutover op (stable semantic id
@@ -2796,7 +4689,8 @@ def run_epoch_transition(*, claude_json_path=None) -> str:
2796
4689
  identity, check/append the canonical cutover op, THEN rebuild — in that exact
2797
4690
  order, so the op is inside the rebuild's input. Returns the resolved
2798
4691
  ``claude_legacy_account``. Exposed for tests; the epoch-mismatch path calls
2799
- it (under the maintenance + ingest locks) after quarantining the old index."""
4692
+ it under the maintenance + ingest locks, and the rebuild's common cutover
4693
+ preserves the old index only after the replacement is validated."""
2800
4694
  claude_key = _resolve_claude_cutover_identity(claude_json_path)
2801
4695
  recorded = append_accounts_cutover_op(claude_key)
2802
4696
  rebuild_stats_index()