cctally 1.92.2 → 1.93.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/CHANGELOG.md +43 -0
  2. package/bin/_cctally_cache.py +354 -0
  3. package/bin/_cctally_core.py +180 -3
  4. package/bin/_cctally_dashboard.py +71 -1
  5. package/bin/_cctally_dashboard_envelope.py +28 -2
  6. package/bin/_cctally_dashboard_share.py +75 -19
  7. package/bin/_cctally_dashboard_sources.py +12 -0
  8. package/bin/_cctally_db.py +89 -1
  9. package/bin/_cctally_doctor.py +31 -0
  10. package/bin/_cctally_forecast.py +4 -2
  11. package/bin/_cctally_journal.py +3482 -258
  12. package/bin/_cctally_journal_repair.py +123 -32
  13. package/bin/_cctally_milestone_history.py +4 -1
  14. package/bin/_cctally_project.py +8 -6
  15. package/bin/_cctally_quota.py +420 -20
  16. package/bin/_cctally_rederive.py +57 -23
  17. package/bin/_cctally_reporting.py +8 -6
  18. package/bin/_cctally_share.py +74 -37
  19. package/bin/_cctally_source_analytics.py +6 -8
  20. package/bin/_cctally_store.py +13 -2
  21. package/bin/_cctally_tui.py +53 -0
  22. package/bin/_lib_cache_coverage.py +547 -0
  23. package/bin/_lib_doctor.py +54 -2
  24. package/bin/_lib_journal.py +235 -95
  25. package/bin/_lib_journal_router.py +21 -0
  26. package/bin/_lib_segment_summary.py +374 -0
  27. package/bin/_lib_selector_state.py +959 -0
  28. package/bin/_lib_share.py +1073 -165
  29. package/bin/_lib_share_templates.py +35 -11
  30. package/bin/_lib_stats_wal.py +327 -0
  31. package/bin/_lib_view_models.py +2 -1
  32. package/dashboard/static/assets/index-DwWJOYxd.css +1 -0
  33. package/dashboard/static/assets/{index-Dat-mza6.js → index-HlIK7k8Q.js} +47 -47
  34. package/dashboard/static/dashboard.html +2 -2
  35. package/package.json +5 -1
  36. package/dashboard/static/assets/index-DnWdv8um.css +0 -1
@@ -25,6 +25,8 @@ segment files.
25
25
  """
26
26
  from __future__ import annotations
27
27
 
28
+ import collections
29
+ import contextlib
28
30
  import datetime as dt
29
31
  import fcntl
30
32
  import hashlib
@@ -39,9 +41,12 @@ from dataclasses import dataclass, field, replace as _dc_replace
39
41
 
40
42
  import _cctally_core
41
43
  import _lib_accounts
44
+ import _lib_cache_coverage
42
45
  import _lib_journal
43
46
  import _lib_journal_router
44
47
  import _lib_record
48
+ import _lib_segment_summary
49
+ import _lib_selector_state
45
50
 
46
51
 
47
52
  # Torn-tail scan window (spec §4.3). A single line must fit inside it — a
@@ -65,12 +70,32 @@ class JournalError(Exception):
65
70
  """A structural journal-append failure (line too long, unrepairable tail)."""
66
71
 
67
72
 
73
+ #: A newly completed correction batch whose effect the live index lacks. The
74
+ #: commit marker IS the narrowest complete prefix, and convergence is decided by
75
+ #: the exact `(rev, status, content_hash, batch_id)` the signal carries.
76
+ CORRECTION_KIND_NEWLY_COMPLETED = "newly_completed"
77
+
78
+ #: A batch whose durable status was `completed` and which a later record
79
+ #: tainted. Its recovery coordinate is the END OFFSET OF THAT RECORD, never the
80
+ #: batch's earliest commit: a rebuild bounded at the commit excludes the
81
+ #: tainting record, faithfully reproduces the completed correction, and meets
82
+ #: the same taint on the next tick — signalling forever instead of converging
83
+ #: (#496 S5b §3.7).
84
+ CORRECTION_KIND_COMPLETED_TO_TAINTED = "completed_to_tainted"
85
+
86
+
68
87
  class CorrectionRebuildRequired(JournalError):
69
88
  """A completed correction cannot be applied incrementally to a live index.
70
89
 
71
90
  The recovery boundary needs more than a message: it must rebuild through
72
- the exact completed-batch commit that triggered the mismatch, then
73
- revalidate the exact effective metadata under exclusive ownership.
91
+ the exact prefix that triggered the mismatch, then revalidate under
92
+ exclusive ownership.
93
+
94
+ `kind` selects WHICH prefix and WHICH revalidation. The two kinds do not
95
+ share a convergence predicate: after a taint withdraws a completed batch the
96
+ post-rebuild winner may be an older candidate that durable selector state
97
+ deliberately does not store, so an exact expected metadata tuple is not
98
+ computable for it.
74
99
  """
75
100
 
76
101
  def __init__(
@@ -82,6 +107,7 @@ class CorrectionRebuildRequired(JournalError):
82
107
  high_water=None,
83
108
  expected_metadata=None,
84
109
  recovery_eligible=False,
110
+ kind=CORRECTION_KIND_NEWLY_COMPLETED,
85
111
  ):
86
112
  super().__init__(message)
87
113
  self.batch_id = batch_id
@@ -89,12 +115,26 @@ class CorrectionRebuildRequired(JournalError):
89
115
  self.high_water = high_water
90
116
  self.expected_metadata = expected_metadata
91
117
  self.recovery_eligible = recovery_eligible
118
+ self.kind = kind
92
119
 
93
120
 
94
121
  class CorrectionRecoveryError(JournalError):
95
122
  """Bounded correction recovery could not safely replace the live index."""
96
123
 
97
124
 
125
+ class JournalAppendTargetStale(JournalError):
126
+ """The resolved append target is no longer the canonically-last segment.
127
+
128
+ Raised when a writer resolved its month segment before taking the leaf lock
129
+ and the journal moved on underneath it (#511, #496 S5b §2.4). Retryable:
130
+ the caller re-resolves and appends again. It is a DISTINCT class precisely
131
+ so `_cctally_cache._append_codex_quota_obs` can re-raise it while still
132
+ swallowing genuine errors — swallowing this one would advance a file offset
133
+ past bytes whose observation was never journaled, and the rollout JSONL
134
+ those bytes came from evaporates.
135
+ """
136
+
137
+
98
138
  # --------------------------------------------------------------------------
99
139
  # leaf lock
100
140
  # --------------------------------------------------------------------------
@@ -243,7 +283,7 @@ def _load_quota_dedup_keys() -> None:
243
283
  _QUOTA_DEDUP_KEYS.clear()
244
284
 
245
285
  for name in list_segments():
246
- with (journal_dir / name).open("rb") as fh:
286
+ with _open_segment_for_read(journal_dir / name) as fh:
247
287
  for raw in fh:
248
288
  if not raw.endswith(b"\n"):
249
289
  break
@@ -276,6 +316,50 @@ def _load_quota_dedup_keys() -> None:
276
316
  _QUOTA_DEDUP_LOADED = True
277
317
 
278
318
 
319
+ def _utc_now() -> dt.datetime:
320
+ """The append path's clock, named so both reads are the same call.
321
+
322
+ An appender resolves its month segment once before the leaf lock and once
323
+ again after taking it (#511). Reading the clock through one helper keeps the
324
+ two reads identical in everything but time.
325
+ """
326
+ return dt.datetime.now(dt.timezone.utc)
327
+
328
+
329
+ def _validate_append_target(journal_dir, seg_name: str) -> None:
330
+ """Refuse an append whose target is not the canonically-last segment.
331
+
332
+ MUST be called with the leaf lock held: the whole point is that the
333
+ canonical order cannot move between this check and the write. Both
334
+ appenders route through it rather than duplicating the comparison, because
335
+ fixing only one leaves the claimed immutability false for exactly the
336
+ correction and audit group appends the durable selector depends on.
337
+
338
+ An EXISTING target must equal the canonically-last segment. An ABSENT
339
+ target must sort last when provisionally added to the segment list, which
340
+ is the ordinary month rollover: the new month's file does not exist yet.
341
+
342
+ It refuses rather than redirects. Refusal preserves physical order and lets
343
+ a writer retry against a freshly resolved target, where silently
344
+ redirecting a planned correction group would move it out from under a
345
+ caller that had already reasoned about its placement (#496 S5b §2.4).
346
+ """
347
+ segments = list_segments()
348
+ if not segments:
349
+ return
350
+ if seg_name == segments[-1]:
351
+ return
352
+ if seg_name not in segments:
353
+ provisional = sorted(
354
+ [*segments, seg_name], key=_lib_journal.segment_sort_key)
355
+ if provisional[-1] == seg_name:
356
+ return
357
+ raise JournalAppendTargetStale(
358
+ f"append target {seg_name} is not the canonically-last segment "
359
+ f"({segments[-1]}) in {journal_dir}; re-resolve and retry"
360
+ )
361
+
362
+
279
363
  def _append_quota_dedup_key(natural_key: str) -> None:
280
364
  """Journal-first second leg: append+fsync one key to the compact index."""
281
365
  index_path = _cctally_core.JOURNAL_DIR / _QUOTA_DEDUP_INDEX_NAME
@@ -306,9 +390,15 @@ def append_record(
306
390
  and returns ``None`` on a skip. Otherwise returns ``(segment_basename,
307
391
  end_offset)`` where ``end_offset`` is the file size just past the appended
308
392
  line — the byte position the ingest cursor advances to when it consumes
309
- this line."""
393
+ this line.
394
+
395
+ An explicitly supplied ``now_utc`` is honoured verbatim and NOT validated
396
+ against canonical order: no production caller supplies one, and roughly a
397
+ dozen tests pass a fixed timestamp precisely to pin segment placement. A
398
+ deliberate placement choice is not a stall (#496 S5b §2.4)."""
399
+ explicit_now = now_utc is not None
310
400
  if now_utc is None:
311
- now_utc = dt.datetime.now(dt.timezone.utc)
401
+ now_utc = _utc_now()
312
402
  data = _lib_journal.encode_line(record)
313
403
  if len(data) > _MAX_LINE_BYTES:
314
404
  raise JournalError(
@@ -343,6 +433,15 @@ def append_record(
343
433
  if natural_key in _QUOTA_DEDUP_KEYS:
344
434
  return None
345
435
 
436
+ # ORDER MATTERS (#496 S5b §2.4). The dedupe no-write return above runs
437
+ # FIRST, because a skipped append writes nothing and so has no target to
438
+ # validate. Re-resolution and validation then run BEFORE the file is
439
+ # opened or torn-tail-repaired, so the repair sequence is untouched.
440
+ if not explicit_now:
441
+ seg_name = _lib_journal.segment_name(_utc_now())
442
+ seg_path = journal_dir / seg_name
443
+ _validate_append_target(journal_dir, seg_name)
444
+
346
445
  seg_created = not seg_path.exists()
347
446
  fd = os.open(str(seg_path), os.O_RDWR | os.O_APPEND | os.O_CREAT, 0o600)
348
447
  try:
@@ -385,11 +484,17 @@ def append_records(
385
484
  correction batch remains physically ordered. ``expected_high_water`` is
386
485
  checked while holding the same leaf lock that performs the append, closing
387
486
  the plan/revalidate/append race.
487
+
488
+ That check is NOT a substitute for target validation: it proves nothing
489
+ about canonical order when ``expected_high_water`` is unset, which is the
490
+ default. A defaulted ``now_utc`` is therefore re-resolved and validated
491
+ under the same lock, exactly as in ``append_record`` (#511, #496 S5b §2.4).
388
492
  """
389
493
  if not isinstance(records, list) or not records:
390
494
  raise ValueError("journal record group must be a non-empty list")
495
+ explicit_now = now_utc is not None
391
496
  if now_utc is None:
392
- now_utc = dt.datetime.now(dt.timezone.utc)
497
+ now_utc = _utc_now()
393
498
  encoded = []
394
499
  for record in records:
395
500
  data = _lib_journal.encode_line(record)
@@ -430,6 +535,13 @@ def append_records(
430
535
  f"(expected {expected_high_water!r}, found {actual_high_water!r})"
431
536
  )
432
537
 
538
+ # After the caller's own precondition, and before the file is opened or
539
+ # torn-tail-repaired (#496 S5b §2.4).
540
+ if not explicit_now:
541
+ seg_name = _lib_journal.segment_name(_utc_now())
542
+ seg_path = journal_dir / seg_name
543
+ _validate_append_target(journal_dir, seg_name)
544
+
433
545
  seg_created = not seg_path.exists()
434
546
  fd = os.open(str(seg_path), os.O_RDWR | os.O_APPEND | os.O_CREAT, 0o600)
435
547
  try:
@@ -506,6 +618,434 @@ def _journal_rebuild_snapshot() -> tuple[tuple[str, int] | None, bool]:
506
618
  _release_leaf_lock(lock_fd)
507
619
 
508
620
 
621
+ #: How far back `_complete_line_offset` will look for the last newline when a
622
+ #: segment does not end on one. A torn tail is one interrupted `write`, so it is
623
+ #: bounded by one record; a megabyte is orders of magnitude beyond any record
624
+ #: this journal produces, and reading the whole segment to answer a question
625
+ #: about its last byte is exactly the cost the coverage certificate exists to
626
+ #: remove.
627
+ _COMPLETE_LINE_TAIL_WINDOW = 1 << 20
628
+
629
+
630
+ def _complete_line_offset(path, size: int) -> int:
631
+ """The offset after the last COMPLETE line in ``path``, given its ``size``.
632
+
633
+ The coverage certificate promises decoded records, not bytes, so its covered
634
+ boundary must be a verified newline boundary — covering a raw torn-tail
635
+ extent would let `_repair_torn_tail` truncate that suffix and append a
636
+ complete record ending at the same size, leaving `(segment, size)` identical
637
+ while the covered contribution changed (spec §4.2).
638
+
639
+ Answering costs one `read` of the segment's last byte in the healthy case,
640
+ because every appender writes a trailing newline. Only a torn tail pays for
641
+ the bounded backward scan, and a tail longer than
642
+ `_COMPLETE_LINE_TAIL_WINDOW` answers 0 rather than reading the whole
643
+ segment: 0 is a valid boundary, it is the conservative direction, and it is
644
+ computed identically by every caller, so two callers cannot disagree.
645
+ """
646
+ if size <= 0:
647
+ return 0
648
+ try:
649
+ with open(path, "rb") as handle:
650
+ handle.seek(size - 1)
651
+ if handle.read(1) == b"\n":
652
+ return size
653
+ window = min(size, _COMPLETE_LINE_TAIL_WINDOW)
654
+ handle.seek(size - window)
655
+ tail = handle.read(window)
656
+ except OSError:
657
+ return 0
658
+ index = tail.rfind(b"\n")
659
+ if index < 0:
660
+ return 0
661
+ return size - window + index + 1
662
+
663
+
664
+ def segment_summary_sidecar_path():
665
+ """The journal-side segment-summary sidecar (#496 S5b section 5.3).
666
+
667
+ Beside the segments it describes rather than in `stats.db`, because a
668
+ rebuild frequently runs precisely when `stats.db` is absent or unreadable —
669
+ the one case a summary stored there could never serve.
670
+ """
671
+ return _cctally_core.JOURNAL_DIR / _lib_segment_summary.SIDECAR_NAME
672
+
673
+
674
+ def read_segment_summaries():
675
+ """``{segment_name: SegmentSummary}`` from the sidecar, or ``{}``.
676
+
677
+ A missing, torn, version-mismatched or checksum-mismatched sidecar answers
678
+ an EMPTY map rather than raising, which is exactly a pass that elides
679
+ nothing (spec section 6.3's silent full-replay fallback).
680
+ """
681
+ return _lib_segment_summary.read_sidecar(segment_summary_sidecar_path()) or {}
682
+
683
+
684
+ class SegmentElisionPlan:
685
+ """The one elision planner, shared by both whole-prefix readers.
686
+
687
+ `rebuild_stats_index` and `stats_index_matches_journal_prefix` construct it
688
+ the same way from the same inputs, so an eliding rebuild and a prefix
689
+ validation cannot reach different answers about the same journal (spec
690
+ §5.6).
691
+
692
+ **Constructed BEFORE the pass, from a short-lived read of the coverage
693
+ certificate.** The certificate has to be consulted before the first segment
694
+ is reached, and holding a `cache.db` read transaction open across a whole
695
+ journal traversal would block WAL checkpointing for the length of the
696
+ rebuild — the bloat issue #297 is about. The leg re-resolves coverage under
697
+ its own snapshot afterwards and may reach a different verdict, because an
698
+ ordinary Codex batch landing mid-pass advances the certificate over a
699
+ journal this pass did not pin. `_refill_elided_quota_raw` is what makes that
700
+ race cost a read rather than a wrong answer.
701
+ """
702
+
703
+ __slots__ = ("summaries", "covered", "verdict", "resolution_seen",
704
+ "elided", "elided_lines", "elided_bytes", "scanned",
705
+ "reasons", "quota_gaps")
706
+
707
+ def __init__(self, *, summaries, covered, verdict) -> None:
708
+ self.summaries = summaries
709
+ #: The certificate's `coveredHighWater`, or None when it is unusable.
710
+ self.covered = covered
711
+ self.verdict = verdict
712
+ #: Set once a `journal_protocol_resolution` op is decoded. From then on
713
+ #: nothing further is elided and the prefix digest is recomputed from
714
+ #: disk, because `PrefixHashAccumulator` cannot compose over a gap.
715
+ self.resolution_seen = False
716
+ self.elided = 0
717
+ self.elided_lines = 0
718
+ self.elided_bytes = 0
719
+ self.scanned = 0
720
+ self.reasons: dict = {}
721
+ #: `(segment, index into quota_raw, summarized_size, summarized line
722
+ #: count)` per elided segment, so the leg can re-read exactly what
723
+ #: elision skipped if its own coverage verdict turns out not to be `ok`.
724
+ #: The line count is carried because `_iter_segment_lines` stops
725
+ #: SILENTLY at EOF, so a short read is invisible to the re-read's
726
+ #: `except` clause and can only be caught by counting what came back.
727
+ self.quota_gaps: list = []
728
+
729
+ def decide(self, name, hi, stat_result):
730
+ """The summary to elide ``name`` with, or None to read it.
731
+
732
+ ``hi`` is this pass's pinned raw extent for the segment, which is its
733
+ `st_size` for every segment the planner is allowed to consider — the
734
+ canonically-last one is refused by the caller before this runs.
735
+ """
736
+ summary = self.summaries.get(name)
737
+ if summary is None:
738
+ self.scanned += 1
739
+ self.reasons.setdefault(name, "summaryAbsent")
740
+ return None
741
+ ok, reason = _lib_segment_summary.summary_is_elidable(
742
+ summary,
743
+ pinned_raw_extent=hi,
744
+ is_last=False,
745
+ certificate_covers=self._covers(name, summary),
746
+ resolution_seen=self.resolution_seen,
747
+ stat_identity=(stat_result.st_dev, stat_result.st_ino),
748
+ )
749
+ if not ok:
750
+ self.scanned += 1
751
+ self.reasons.setdefault(name, reason)
752
+ return None
753
+ self.elided += 1
754
+ self.elided_lines += summary.lines
755
+ self.elided_bytes += summary.bytes
756
+ self.reasons.setdefault(name, _lib_segment_summary.REASON_OK)
757
+ return summary
758
+
759
+ def _covers(self, name, summary) -> bool:
760
+ if self.covered is None:
761
+ return False
762
+ return _coordinate_covers(
763
+ self.covered, (name, int(summary.summarized_size)))
764
+
765
+ def counters(self) -> dict:
766
+ """The additive rebuild-record block (§6.3, "recorded, not silent")."""
767
+ return {
768
+ "elidedSegments": self.elided,
769
+ "elidedLines": self.elided_lines,
770
+ "elidedBytes": self.elided_bytes,
771
+ "scannedSegments": self.scanned,
772
+ "coverage": self.verdict,
773
+ "resolutionSeen": self.resolution_seen,
774
+ "refusals": dict(self.reasons),
775
+ }
776
+
777
+
778
+ def plan_segment_elision(segments, high_water):
779
+ """One `SegmentElisionPlan` for the prefix ``segments`` up to ``high_water``.
780
+
781
+ Every failure — no journal, no sidecar, no readable cache, an invalid
782
+ certificate — answers a plan that elides nothing, which is exactly today's
783
+ behaviour. Spec §6.3 makes every one of those a SILENT full replay.
784
+
785
+ The prefix's own pinned vector is deliberately NOT computed here. Only the
786
+ WHOLE journal's vector validates a certificate (see below), and the per-
787
+ segment extents this plan checks arrive from `_iter_range_with_segments`'s
788
+ own `stat`. Computing a second vector restricted to ``segments`` would cost
789
+ one `os.stat` plus one `open` per segment — through `_complete_line_offset`'s
790
+ tail probe — on every plan construction, including inside
791
+ `stats_index_matches_journal_prefix`, and nothing would read it.
792
+ """
793
+ summaries = read_segment_summaries()
794
+ if not summaries or high_water is None:
795
+ return SegmentElisionPlan(
796
+ summaries={}, covered=None,
797
+ verdict=_lib_cache_coverage.REASON_ABSENT)
798
+ snapshot = _read_coverage_snapshot(_cctally_core.CACHE_DB_PATH)
799
+ if snapshot is None:
800
+ return SegmentElisionPlan(
801
+ summaries=summaries, covered=None,
802
+ verdict=_lib_cache_coverage.REASON_ABSENT)
803
+ try:
804
+ # The vector the certificate is validated against is the WHOLE journal's,
805
+ # exactly as `_resolve_quota_cache_coverage` computes it: a root over the
806
+ # pinned prefix alone would never match one a writer stored over every
807
+ # segment.
808
+ full_vector = coverage_pinned_vector()
809
+ ok, reason = _lib_cache_coverage.certificate_is_valid(
810
+ snapshot.certificate,
811
+ pinned_vector=full_vector,
812
+ physical_seq=snapshot.physical_seq,
813
+ )
814
+ covered = (
815
+ snapshot.certificate.get("coveredHighWater") if ok else None)
816
+ finally:
817
+ _close_coverage_snapshot(snapshot)
818
+ return SegmentElisionPlan(
819
+ summaries=summaries, covered=covered, verdict=reason)
820
+
821
+
822
+ def _refill_elided_quota_raw(quota_raw, gaps):
823
+ """``(stream, complete)`` — ``quota_raw`` with the elided observations back.
824
+
825
+ Reached only when the pass elided and the cache leg's own coverage verdict
826
+ then came back something other than `ok` — an ordinary Codex batch landing
827
+ mid-pass is enough. The leg is about to REPLAY, and replaying a stream that
828
+ is missing the elided segments would leave those observations unmaterialized
829
+ while the mint asserted coverage over them.
830
+
831
+ **Rebuilt in one ascending sweep rather than spliced in place.** An elided
832
+ segment appends nothing to ``quota_raw``, so every segment of a contiguous
833
+ elided run records the SAME insertion index, and inserting each one at that
834
+ shared index pushes its predecessor later — which replays the run backwards.
835
+ Six of the maintainer's seven bootstrap segments are quota-only and
836
+ adjacent, so on that journal the whole run reverses. Order is not cosmetic:
837
+ `_QUOTA_SNAPSHOT_INSERT` is `INSERT OR IGNORE` and resolves first-wins on
838
+ the natural key, and `CodexResetAnchorResolver` decides per record in stream
839
+ order.
840
+
841
+ ``complete`` is False when any elided segment could not be re-read IN FULL.
842
+ The caller must then drop its covered boundary: the observations this stream
843
+ is missing are exactly the ones a minted certificate would claim.
844
+
845
+ **A short read is detected by counting, not by catching.**
846
+ `_iter_segment_lines` reads until its own `read()` returns nothing, so a
847
+ segment that yields fewer lines than it held ends the loop with no exception
848
+ at all. Each gap therefore carries the line count its summary recorded, and
849
+ a re-read that returns a different number reports the shortfall rather than
850
+ leaving the leg to certify a stream it never saw.
851
+ """
852
+ by_index: dict = {}
853
+ complete = True
854
+ for name, index, extent, expected_lines in gaps:
855
+ recovered = []
856
+ seen = 0
857
+ try:
858
+ for _seg, _off, raw in _iter_segment_lines(
859
+ _cctally_core.JOURNAL_DIR / name, 0, int(extent)):
860
+ seen += 1
861
+ record = _lib_journal.decode_line(raw)
862
+ if record is not None and _is_codex_quota_obs(record):
863
+ recovered.append(raw)
864
+ except Exception:
865
+ # A vanished segment self-heals, because the pinned vector no longer
866
+ # matches the journal and the certificate is invalid the moment it
867
+ # is read. A transient read error on an UNCHANGED file does not: the
868
+ # vector still matches, so a certificate minted over this shortened
869
+ # stream would be valid and would claim observations nobody applied.
870
+ # Reporting the shortfall is what stops the mint.
871
+ #
872
+ # EVERY exception, not just `OSError`: spec §6.3 makes every
873
+ # degraded state here a silent full-replay fallback, and neither the
874
+ # call site nor its caller catches anything, so one that escaped
875
+ # would abort the whole rebuild over a re-derivable optimization.
876
+ complete = False
877
+ continue
878
+ if seen != int(expected_lines):
879
+ complete = False
880
+ by_index.setdefault(int(index), []).extend(recovered)
881
+ out: list = []
882
+ cursor = 0
883
+ for index in sorted(by_index):
884
+ bounded = min(int(index), len(quota_raw))
885
+ out.extend(quota_raw[cursor:bounded])
886
+ out.extend(by_index[index])
887
+ cursor = max(cursor, bounded)
888
+ out.extend(quota_raw[cursor:])
889
+ return out, complete
890
+
891
+
892
+ class _SegmentSummaryCollector:
893
+ """Per-segment traversal facts, accumulated as the pass streams.
894
+
895
+ The counters are per segment rather than per pass because a later pass has
896
+ to contribute an ELIDED segment's share of them without reading it, and a
897
+ whole-pass total cannot be decomposed after the fact.
898
+
899
+ The last-seen fold is deliberately accumulated per segment and merged into
900
+ the caller's running accumulator at each boundary, rather than folded into
901
+ both. `LastSeenAccumulator.observe` runs once per record on a 1.95-million
902
+ line journal, and folding twice would double that; merging a small map once
903
+ per segment does not.
904
+ """
905
+
906
+ __slots__ = ("summaries", "last_seen", "_name", "_lo", "_stat",
907
+ "_lines", "_bytes", "_decodes", "_malformed", "_retained",
908
+ "_line_end")
909
+
910
+ def __init__(self) -> None:
911
+ self.summaries: dict = {}
912
+ #: The open segment's own accumulator, or None between segments.
913
+ self.last_seen = None
914
+ self._name = None
915
+ self._reset()
916
+
917
+ def _reset(self) -> None:
918
+ self._lo = 0
919
+ self._stat = None
920
+ self._lines = 0
921
+ self._bytes = 0
922
+ self._decodes = 0
923
+ self._malformed = 0
924
+ self._retained = 0
925
+ self._line_end = 0
926
+
927
+ def begin(self, name, lo, hi, stat_result, running) -> None:
928
+ """Open ``name``, closing whatever segment was open before it.
929
+
930
+ ``hi`` is this pass's pinned READ extent, which for the high-water
931
+ segment is the pinned high-water rather than the file's size. The
932
+ summary records `st_size` instead (spec §5.3), so ``hi`` is accepted for
933
+ the shared `on_extent` signature and deliberately not stored.
934
+ """
935
+ self.close(running)
936
+ self._name = name
937
+ self._reset()
938
+ self._lo = int(lo)
939
+ self._stat = stat_result
940
+ self._line_end = int(lo)
941
+ self.last_seen = _lib_journal_router.LastSeenAccumulator()
942
+
943
+ def line(self, raw, end_offset) -> None:
944
+ self._lines += 1
945
+ self._bytes += len(raw) + 1
946
+ self._line_end = int(end_offset)
947
+
948
+ def decoded(self, retained: bool) -> None:
949
+ """One decoded record: one traversal decode AND one `decoded` element.
950
+
951
+ ONE counter, because the two are the same number by construction — the
952
+ rebuild appends exactly one element to `decoded` per successfully
953
+ decoded record, the record itself when retained and a `None` placeholder
954
+ otherwise. The summary stores it under two names because they answer
955
+ different questions: `decodes` is the traversal counter the rebuild
956
+ record reports, and `decoded_entry_count` is the elision contract, which
957
+ carries a `None` sentinel a counter cannot.
958
+ """
959
+ self._decodes += 1
960
+ if retained:
961
+ self._retained += 1
962
+
963
+ def malformed_line(self) -> None:
964
+ self._malformed += 1
965
+
966
+ def close(self, running) -> None:
967
+ """Finalize the open segment and merge its last-seen into ``running``."""
968
+ if self.last_seen is not None and running is not None:
969
+ running.merge(
970
+ self.last_seen.stamped,
971
+ self.last_seen.legacy_claude_at,
972
+ self.last_seen.legacy_codex_at,
973
+ )
974
+ if self._name is not None and self._lo == 0 and self._stat is not None:
975
+ partial = self.last_seen
976
+ self.summaries[self._name] = _lib_segment_summary.SegmentSummary(
977
+ segment_name=self._name,
978
+ st_dev=int(self._stat.st_dev),
979
+ st_ino=int(self._stat.st_ino),
980
+ # The raw `st_size` observed when this summary was written (spec
981
+ # §5.3), NOT this pass's read extent. They differ only for the
982
+ # high-water segment, whose extent is the pinned high-water; a
983
+ # summary claiming that extent as the file's size would let a
984
+ # later pass elide a segment this one did not read to the end.
985
+ summarized_size=int(self._stat.st_size),
986
+ complete_line_covered_offset=self._line_end,
987
+ lines=self._lines,
988
+ bytes=self._bytes,
989
+ decodes=self._decodes,
990
+ malformed=self._malformed,
991
+ quota_only=self._retained == 0,
992
+ decoded_entry_count=self._decodes,
993
+ last_seen_stamped=(
994
+ {} if partial is None else dict(partial.stamped)),
995
+ last_seen_legacy_claude_at=(
996
+ None if partial is None else partial.legacy_claude_at),
997
+ last_seen_legacy_codex_at=(
998
+ None if partial is None else partial.legacy_codex_at),
999
+ )
1000
+ self._name = None
1001
+ self.last_seen = None
1002
+
1003
+ def adopt(self, summary) -> None:
1004
+ """Carry an ELIDED segment's stored summary forward unchanged.
1005
+
1006
+ The pass did not re-derive it, so re-deriving it here would be inventing
1007
+ it. Carrying it forward is what keeps the sidecar complete across a run
1008
+ of consecutive eliding passes.
1009
+ """
1010
+ self.summaries[summary.segment_name] = summary
1011
+
1012
+
1013
+ def coverage_pinned_vector():
1014
+ """The ordered ``(name, raw st_size, complete-line offset)`` triple per segment.
1015
+
1016
+ This is the vector the #496 S5b coverage certificate's identity root binds
1017
+ (spec §4.2), and it is deliberately ONE function rather than two: the writer
1018
+ that advances a certificate and the rebuild that validates it must compute
1019
+ the same triples from the same rules, or the root would differ for reasons
1020
+ that have nothing to do with coverage.
1021
+
1022
+ Both operands are carried because they are independent. `_repair_torn_tail`
1023
+ truncates a partial trailing line before appending, so a segment's size can
1024
+ decrease or return to a previous value — "an append strictly increases
1025
+ `st_size`" is false. A segment whose last newline sits at 100 and whose size
1026
+ is 120 is a different physical state from one where both are 100, and
1027
+ collapsing them would let a permanently torn segment certify as covered.
1028
+
1029
+ It takes no segment list. Every caller wants the WHOLE journal's vector,
1030
+ because that is the only one a stored certificate's identity root can be
1031
+ validated against: a root computed over a prefix would never match one a
1032
+ writer stored over every segment.
1033
+ """
1034
+ vector = []
1035
+ for name in list_segments():
1036
+ path = _cctally_core.JOURNAL_DIR / name
1037
+ try:
1038
+ size = os.stat(path).st_size
1039
+ except OSError:
1040
+ # A segment that vanished between the listing and the stat leaves a
1041
+ # vector nothing can match, which falls back to a full replay. That
1042
+ # is the right answer: the journal shape moved under this pass.
1043
+ size = -1
1044
+ vector.append(
1045
+ (name, size, 0 if size < 0 else _complete_line_offset(path, size)))
1046
+ return tuple(vector)
1047
+
1048
+
509
1049
  def journal_high_water() -> tuple[str, int] | None:
510
1050
  """Snapshot ``(latest segment basename, size)`` under a µs leaf-lock hold.
511
1051
 
@@ -951,6 +1491,24 @@ def _write_cursor(conn: sqlite3.Connection, segment: str, offset: int) -> None:
951
1491
  _SEGMENT_READ_CHUNK = 256 * 1024
952
1492
 
953
1493
 
1494
+ def _open_segment_for_read(seg_path):
1495
+ """The single physical read boundary for a journal segment.
1496
+
1497
+ Both read routes go through here — the streaming line reader and the
1498
+ prefix hasher — so a test can observe the exact sequence of segment opens
1499
+ a pass performs. A per-pass line or byte counter cannot: it stays correct
1500
+ while an implementation reopens a segment behind it, which is precisely the
1501
+ hidden re-read this session removes (#496 S5 §4.2).
1502
+
1503
+ The cutover's bootstrap-reuse digest and the quota dedupe-index rebuild are
1504
+ the module's other two read-only segment scans, and they come through here
1505
+ as well, so "every physical read of a segment" is a property a test can
1506
+ check rather than a claim in prose. The append path is deliberately NOT
1507
+ routed here: it holds a read-write handle of its own for the torn-tail scan.
1508
+ """
1509
+ return open(seg_path, "rb")
1510
+
1511
+
954
1512
  def _iter_segment_lines(seg_path, lo: int, hi: int, *, on_bytes=None):
955
1513
  """Stream `(basename, absolute-offset, raw-line-without-newline)` for every
956
1514
  complete line in `[lo, hi)`, holding at most one chunk plus one partial line
@@ -964,7 +1522,7 @@ def _iter_segment_lines(seg_path, lo: int, hi: int, *, on_bytes=None):
964
1522
  including a torn trailing partial line that is never yielded.
965
1523
  """
966
1524
  name = seg_path.name
967
- with open(seg_path, "rb") as fh:
1525
+ with _open_segment_for_read(seg_path) as fh:
968
1526
  fh.seek(lo)
969
1527
  pos = lo
970
1528
  buf = b""
@@ -1006,7 +1564,7 @@ def iter_range(cursor, hw):
1006
1564
 
1007
1565
 
1008
1566
  def _iter_range_with_segments(cursor, hw, segments, *, on_segment=None,
1009
- on_bytes=None):
1567
+ on_bytes=None, on_extent=None, elide=None):
1010
1568
  """`iter_range` over a segment list the CALLER snapshotted (#496 S4 §4).
1011
1569
 
1012
1570
  `list_segments()` enumerates the journal directory at call time and orders
@@ -1020,6 +1578,18 @@ def _iter_range_with_segments(cursor, hw, segments, *, on_segment=None,
1020
1578
  function then skips because it holds no bytes in range: `journal_prefix_hash`
1021
1579
  frames a zero-byte segment, so a hash accumulator has to be told it exists.
1022
1580
  `on_bytes` is forwarded to `_iter_segment_lines`.
1581
+
1582
+ `on_extent(seg, lo, hi, stat_result)` is called after this pass has pinned
1583
+ the range it will read from that segment, so a caller building a durable
1584
+ summary records the extent the pass actually covered rather than one it
1585
+ re-stats afterwards (#496 S5b section 5.3).
1586
+
1587
+ `elide(seg, lo, hi, stat_result)` decides whether to SKIP the segment
1588
+ entirely — no `_open_segment_for_read`, no bytes, no lines. It runs after
1589
+ `on_extent` and instead of `on_segment`, because a skipped segment
1590
+ contributes nothing to a hash accumulator that could not compose over it
1591
+ anyway (spec section 5.1). Only #496 S5b Stage 4's elision planner supplies
1592
+ it; every other caller reads every segment exactly as before.
1023
1593
  """
1024
1594
  hw_seg, hw_size = hw
1025
1595
  if hw_seg not in segments:
@@ -1037,7 +1607,20 @@ def _iter_range_with_segments(cursor, hw, segments, *, on_segment=None,
1037
1607
  seg = segments[idx]
1038
1608
  seg_path = _cctally_core.JOURNAL_DIR / seg
1039
1609
  lo = start_off if idx == start_idx else 0
1040
- hi = hw_size if idx == hw_idx else os.path.getsize(seg_path)
1610
+ stat_result = None
1611
+ if idx == hw_idx:
1612
+ hi = hw_size
1613
+ else:
1614
+ stat_result = os.stat(seg_path)
1615
+ hi = stat_result.st_size
1616
+ if on_extent is not None or elide is not None:
1617
+ if stat_result is None:
1618
+ stat_result = os.stat(seg_path)
1619
+ if elide is not None and lo == 0 and idx != hw_idx and elide(
1620
+ seg, lo, hi, stat_result):
1621
+ continue
1622
+ if on_extent is not None:
1623
+ on_extent(seg, lo, hi, stat_result)
1041
1624
  if on_segment is not None:
1042
1625
  on_segment(seg)
1043
1626
  if lo >= hi:
@@ -1051,7 +1634,13 @@ def _read_range(cursor, hw) -> list[tuple[str, int, bytes]]:
1051
1634
 
1052
1635
 
1053
1636
  def journal_prefix_hash(high_water) -> "str | None":
1054
- """Hash exact raw segment bytes through one canonical high-water."""
1637
+ """Hash exact raw segment bytes through one canonical high-water.
1638
+
1639
+ The framing is durable: per segment, the 4-byte big-endian name length, the
1640
+ name, the 8-byte big-endian data length, then the data. These digests are
1641
+ recorded inside `journal_protocol_resolution` payloads, so a change to the
1642
+ framing invalidates every acknowledgement already written.
1643
+ """
1055
1644
  if high_water is None:
1056
1645
  return None
1057
1646
  digest = hashlib.sha256()
@@ -1059,7 +1648,8 @@ def journal_prefix_hash(high_water) -> "str | None":
1059
1648
  for segment in list_segments():
1060
1649
  path = _cctally_core.JOURNAL_DIR / segment
1061
1650
  size = high_water[1] if segment == high_water[0] else path.stat().st_size
1062
- data = path.read_bytes()[:size]
1651
+ with _open_segment_for_read(path) as handle:
1652
+ data = handle.read(size)
1063
1653
  if len(data) != size:
1064
1654
  raise OSError(f"journal segment changed while reading: {segment}")
1065
1655
  name = segment.encode("utf-8")
@@ -1318,7 +1908,7 @@ def _apply_file_account_records(cache, records) -> "tuple[int, int]":
1318
1908
  return restored, conflicts
1319
1909
 
1320
1910
 
1321
- def _report_file_account_conflicts(conflicts: int) -> None:
1911
+ def _report_file_account_conflicts(conflicts: int, *, quiet: bool = False) -> None:
1322
1912
  """One stderr line for a run of replayed decisions that contradicted a
1323
1913
  different account already recorded at the same
1324
1914
  ``(file_identity, incarnation, from_offset)`` and were therefore DECLINED
@@ -1328,8 +1918,13 @@ def _report_file_account_conflicts(conflicts: int) -> None:
1328
1918
 
1329
1919
  Every call site must invoke this AFTER its commit (closeout review C5): a
1330
1920
  rolled-back transaction applied nothing, so reporting from inside it would
1331
- tell the operator about a decline that did not happen."""
1332
- if conflicts > 0:
1921
+ tell the operator about a decline that did not happen.
1922
+
1923
+ ``quiet`` is the reconciliation's caller. That path is reachable from an
1924
+ ordinary command, where acceptance criterion 10 requires no new stderr
1925
+ line; the same decline is still reported by the ingest and rebuild paths
1926
+ that own the remedy this line names."""
1927
+ if conflicts > 0 and not quiet:
1333
1928
  print(
1334
1929
  f"[ingest] codex attribution replay declined {conflicts} "
1335
1930
  "contradicting decision(s); the first journalled decision for each "
@@ -1525,11 +2120,27 @@ def _resolve_obs_anchor(resolver, rec: dict) -> "str | None":
1525
2120
  return None
1526
2121
 
1527
2122
 
1528
- def _apply_quota_records(cache, records) -> None:
2123
+ def _apply_quota_records(
2124
+ cache, records, *, reported_conflicts=None, quiet=False,
2125
+ ) -> None:
1529
2126
  """Materialize Codex quota obs into an OPEN cache.db transaction, applying
1530
2127
  the §3.5 precedence rule. Callers must apply the batch's file-account
1531
2128
  decisions FIRST, so a decision arriving in the same batch already governs
1532
- the observations it covers."""
2129
+ the observations it covers.
2130
+
2131
+ ``reported_conflicts`` lets ONE logical pass thread its conflict-report set
2132
+ across several calls (spec §4.6). The rebuild's recovery pass is chunked into
2133
+ many transactions, and without threading it would emit one line per chunk for
2134
+ a condition a single unchunked call reports once. Every other caller passes
2135
+ None and gets today's per-call set.
2136
+
2137
+ ``quiet`` suppresses the conflict line, for the same reason
2138
+ `_report_file_account_conflicts` takes it: routing
2139
+ `recover_quota_cache_from_journal` through the shared leg made this line
2140
+ reachable from the open-time reconciliation, so `cache-sync` and the
2141
+ dashboard would emit a line they never emitted before, and acceptance
2142
+ criterion 10 requires no new stderr on an ordinary command. Its sibling was
2143
+ guarded and this one was not."""
1533
2144
  oracle = _CodexAttributionOracle(cache)
1534
2145
  # One line per FILE PER BATCH, not per record. A mid-file account switch
1535
2146
  # legitimately produces a run of observations whose first-stamp-wins account
@@ -1541,7 +2152,8 @@ def _apply_quota_records(cache, records) -> None:
1541
2152
  # standing condition a handful of times is the cheaper error. The condition
1542
2153
  # is worth reporting at all because a genuine correction is expressed as an
1543
2154
  # explicit new range decision, never by mutating history.
1544
- reported_conflicts: set = set()
2155
+ if reported_conflicts is None:
2156
+ reported_conflicts = set()
1545
2157
  # #416 spec §4.2: this leg is a genuine INGEST into cache.db (it materializes
1546
2158
  # observations whose source rollout may have evaporated), so it must resolve
1547
2159
  # the canonical anchor too. Without it, a journal-replayed row lands with a
@@ -1568,20 +2180,74 @@ def _apply_quota_records(cache, records) -> None:
1568
2180
  conflict_key = (payload.get("source_root_key"), payload.get("source_path"))
1569
2181
  if (observed is not None and observed != decided
1570
2182
  and conflict_key not in reported_conflicts):
2183
+ # The key is recorded even when quiet, so a later non-quiet call
2184
+ # threading the same set still reports each file at most once.
1571
2185
  reported_conflicts.add(conflict_key)
1572
- print(
1573
- "[ingest] codex attribution conflict: "
1574
- f"{payload.get('source_path')}@{payload.get('line_offset')} "
1575
- f"observation stamped {observed} but the durable decision says "
1576
- f"{decided if decided is not None else 'unattributed'}; "
1577
- "keeping the decision",
1578
- file=sys.stderr,
1579
- )
2186
+ if not quiet:
2187
+ print(
2188
+ "[ingest] codex attribution conflict: "
2189
+ f"{payload.get('source_path')}"
2190
+ f"@{payload.get('line_offset')} "
2191
+ "observation stamped "
2192
+ f"{observed} but the durable decision says "
2193
+ f"{decided if decided is not None else 'unattributed'}; "
2194
+ "keeping the decision",
2195
+ file=sys.stderr,
2196
+ )
1580
2197
  values = list(row_values)
1581
2198
  values[16] = decided
1582
2199
  cache.execute(upsert_sql, tuple(values))
1583
2200
 
1584
2201
 
2202
+ class CoverageInvariantViolation(RuntimeError):
2203
+ """A caller reached a covered-family delete without invalidating coverage.
2204
+
2205
+ Raised rather than asserted, because `python -O` strips `assert` and this is
2206
+ the only guard standing between a new `authoritative=True` caller and a
2207
+ certificate left standing over a table that was just emptied.
2208
+ """
2209
+
2210
+
2211
+ def _assert_coverage_already_invalidated(cache_conn) -> None:
2212
+ """Refuse an authoritative replay while a coverage certificate stands.
2213
+
2214
+ `rehydrate_codex_file_accounts(authoritative=True)` empties
2215
+ `codex_file_accounts`, a member of `COVERAGE_CACHE_FAMILIES`, before
2216
+ replaying, yet its inventory entry is `preserve`. That holds only because
2217
+ its one caller — `sync_codex_cache` under `--rebuild` — already ran
2218
+ `_clear_codex_derived_rows` and committed, which deleted both the
2219
+ certificate and any recovery progress. A second caller, or a reordering
2220
+ inside `sync_codex_cache`, would break it silently and the writer-surface
2221
+ scanner could not see it, because the key is already in the inventory with a
2222
+ green label. This turns that ordering into a checked precondition.
2223
+
2224
+ **Where to look when it fires.** No legitimate ordering reaches it today,
2225
+ and six were checked. If a future caller does, `sync_codex_cache`'s
2226
+ `except Exception` catches `CoverageInvariantViolation`, rolls back, sets
2227
+ `deferred_reason = "attribution_rehydration"`, prints one line and skips the
2228
+ whole Codex walk — so the symptom is a permanently deferred Codex sync
2229
+ rather than a loud failure. That containment is the right production trade
2230
+ and is not changed here; the note exists so the next debugger looks for the
2231
+ new caller rather than for a broken walk.
2232
+ """
2233
+ try:
2234
+ row = cache_conn.execute(
2235
+ "SELECT key FROM cache_meta WHERE key IN (?, ?) LIMIT 1",
2236
+ (_lib_cache_coverage.CERTIFICATE_KEY,
2237
+ _lib_cache_coverage.PROGRESS_KEY),
2238
+ ).fetchone()
2239
+ except sqlite3.Error:
2240
+ # An unreadable `cache_meta` cannot witness the invariant either way,
2241
+ # and raising here would turn a degraded cache into a failed sync.
2242
+ return
2243
+ if row is not None:
2244
+ raise CoverageInvariantViolation(
2245
+ "rehydrate_codex_file_accounts(authoritative=True) clears a covered "
2246
+ f"family while {row[0]!r} is still stored; the caller must "
2247
+ "invalidate coverage in the transaction that clears"
2248
+ )
2249
+
2250
+
1585
2251
  def rehydrate_codex_file_accounts(
1586
2252
  cache_conn, *, authoritative: bool = False, since=None,
1587
2253
  ) -> "tuple[int, tuple[str, int] | None, int]":
@@ -1649,6 +2315,8 @@ def rehydrate_codex_file_accounts(
1649
2315
  upsert already converges the counter, a clear has no upside and that
1650
2316
  downside.
1651
2317
  """
2318
+ if authoritative:
2319
+ _assert_coverage_already_invalidated(cache_conn)
1652
2320
  hw = journal_high_water()
1653
2321
  if hw is None:
1654
2322
  if authoritative:
@@ -1681,7 +2349,78 @@ def rehydrate_codex_file_accounts(
1681
2349
  return applied, hw, conflicts
1682
2350
 
1683
2351
 
1684
- def _cache_applier(decoded) -> int | None:
2352
+ def _bounded_covered_offset(segment, raw_offset, covered_offset, decoded_end):
2353
+ """The covered boundary for ``segment``, bounded by what a pass DECODED.
2354
+
2355
+ Three independent upper bounds, and the smallest wins:
2356
+
2357
+ - ``raw_offset`` — the pass's own high water; nothing past it was read.
2358
+ - ``covered_offset`` — the segment's complete-line offset, so a torn
2359
+ trailing line is never inside the claim.
2360
+ - ``decoded_end`` — the end of the last line the pass CONSUMED, never past
2361
+ it. The two callers supply that operand differently and both are safe.
2362
+ `_coverage_advance_plan` (the ingest cycle) passes the end of the last
2363
+ record it DECODED, so a malformed line the batch skipped is excluded.
2364
+ `rebuild_stats_index` passes `prior_high_water`, which the streaming pass
2365
+ advances for a malformed line too — consumed but not decoded — and that is
2366
+ still an upper bound on what was read, so the claim stays true. Reading
2367
+ the operand as "the last DECODED record" is literally true only of the
2368
+ first caller.
2369
+
2370
+ The third is not redundant with the second, and leaving it out was a real
2371
+ defect. Both other operands are read from the file, and the file is re-stat'd
2372
+ AFTER the pass finished reading: a segment torn at the pinned high water and
2373
+ repaired by a later append has a current complete-line offset at or past that
2374
+ high water, so `min(raw, covered)` returns the raw high water — a boundary
2375
+ whose trailing bytes the traversal never decoded, because the partial line
2376
+ was skipped. A newline-terminated MALFORMED trailing line produces the same
2377
+ divergence without any repair at all. ``decoded_end`` is the only operand
2378
+ the pass observed rather than inferred.
2379
+
2380
+ A ``decoded_end`` in an EARLIER segment answers 0: the pass covered
2381
+ everything before this segment and none of it, which is exactly expressible
2382
+ and stays true.
2383
+ """
2384
+ bound = min(int(raw_offset), int(covered_offset))
2385
+ if decoded_end is None:
2386
+ return bound
2387
+ if str(decoded_end[0]) != str(segment):
2388
+ return 0
2389
+ return min(bound, int(decoded_end[1]))
2390
+
2391
+
2392
+ def _coverage_advance_plan(cursor, covered_to, decoded_end=None):
2393
+ """``(pinned_vector, covered, applied_through)`` for an advance, or None.
2394
+
2395
+ Captured BEFORE the cache flocks are acquired, and that ordering is the
2396
+ safety property. A vector captured before a concurrent append describes a
2397
+ SMALLER journal than the one that exists when the certificate is stored, so
2398
+ the stored root stops matching and the next rebuild replays. A vector
2399
+ captured after such an append would match the current journal while a record
2400
+ nobody applied sat inside it — coverage asserted over an unapplied record,
2401
+ which is the one failure the certificate must not produce.
2402
+
2403
+ ``covered`` is bounded by `_bounded_covered_offset`; ``applied_through`` is
2404
+ the raw coordinate the cycle advances its cursor to. The two are returned
2405
+ separately because the certificate stores both: the next writer's contiguity
2406
+ check compares its starting cursor against `appliedThrough`, and comparing
2407
+ it against the clamped boundary instead made a single torn or malformed
2408
+ trailing line freeze the certificate permanently.
2409
+ """
2410
+ if cursor is None or covered_to is None:
2411
+ return None
2412
+ vector = coverage_pinned_vector()
2413
+ segment, offset = str(covered_to[0]), int(covered_to[1])
2414
+ for name, _raw_extent, covered_offset in vector:
2415
+ if name == segment:
2416
+ bounded = _bounded_covered_offset(
2417
+ segment, offset, covered_offset, decoded_end)
2418
+ return vector, (segment, bounded), (segment, offset)
2419
+ return None
2420
+
2421
+
2422
+ def _cache_applier(decoded, *, cursor=None, covered_to=None,
2423
+ decoded_end=None) -> int | None:
1685
2424
  """Composite cache leg (spec §5.2 step 3 + #416 spec §3.4): materialize this
1686
2425
  batch's Codex quota obs into `quota_window_snapshots` AND its
1687
2426
  `codex_file_account` ops into the attribution map, under the NON-BLOCKING
@@ -1698,6 +2437,17 @@ def _cache_applier(decoded) -> int | None:
1698
2437
  - Flock acquired + everything upserted → return None (full consumption).
1699
2438
  A quota-row change advances ``codex_physical_mutation_seq`` in the same
1700
2439
  transaction; an idempotent replay leaves the sequence unchanged.
2440
+
2441
+ ``cursor``, ``covered_to`` and ``decoded_end`` carry the cycle's journal
2442
+ range so this leg can ADVANCE the #496 S5b coverage certificate (spec §4.3).
2443
+ It is the writer that can do so soundly, because it owns a CONTIGUOUS batch:
2444
+ it consumed every record in `[cursor, covered_to]` and applied every
2445
+ cache-relevant one, so a predecessor applied through exactly `cursor`
2446
+ extends to `covered_to`. ``decoded_end`` is the end coordinate of the last
2447
+ record the cycle decoded, and it bounds the covered CLAIM below the raw
2448
+ cursor target whenever the two differ. All three default to None, which
2449
+ advances nothing — the ingest cycle is the only caller that knows the range,
2450
+ and a test calling this directly must not mint coverage it cannot justify.
1701
2451
  """
1702
2452
  quota_idx = [i for i, (rec, _s, _o) in enumerate(decoded)
1703
2453
  if _is_codex_quota_obs(rec)]
@@ -1705,6 +2455,10 @@ def _cache_applier(decoded) -> int | None:
1705
2455
  if _is_codex_file_account_op(rec)]
1706
2456
  if not quota_idx and not file_idx:
1707
2457
  return None
2458
+ # BEFORE the flocks, for the reason `_coverage_advance_plan` states, and
2459
+ # before them for a second reason too: it is journal file I/O, and the leg's
2460
+ # whole purpose is to hold the global cache writer lock as briefly as it can.
2461
+ plan = _coverage_advance_plan(cursor, covered_to, decoded_end)
1708
2462
  # All-or-nothing across the two families: one stop, the earliest of either.
1709
2463
  stop_idx = min(quota_idx[0] if quota_idx else file_idx[0],
1710
2464
  file_idx[0] if file_idx else quota_idx[0])
@@ -1730,7 +2484,25 @@ def _cache_applier(decoded) -> int | None:
1730
2484
  print(f"[ingest] cache leg connect failed: {exc}", file=sys.stderr)
1731
2485
  return stop_idx
1732
2486
  try:
2487
+ import _cctally_cache
1733
2488
  cache.execute("PRAGMA busy_timeout=15000")
2489
+ # Read and check the predecessor BEFORE the transaction opens
2490
+ # (spec §4.3, as corrected). `prior_is_extendable` checks contiguity
2491
+ # against this cycle's STARTING cursor — a predecessor applied
2492
+ # through less leaves a gap nobody applied, and one applied through
2493
+ # more was written by a pass that saw records this batch does not
2494
+ # carry — AND the two version fields, because `advance` re-stamps
2495
+ # the current module constants and discards `prior`, so extending a
2496
+ # certificate written under an older `interpretationVersion` would
2497
+ # launder it into a current-version one. It deliberately does not
2498
+ # run the full `certificate_is_valid`: an advance's predecessor
2499
+ # necessarily describes an older, smaller journal.
2500
+ prior = _cctally_cache.load_codex_journal_coverage_certificate(cache)
2501
+ if plan is not None:
2502
+ extendable, _why = _lib_cache_coverage.prior_is_extendable(
2503
+ prior, applied_through=(str(cursor[0]), int(cursor[1])))
2504
+ if not extendable:
2505
+ prior = None
1734
2506
  cache.execute("BEGIN IMMEDIATE")
1735
2507
  # Decisions FIRST: §3.5 makes the file/range decision authoritative
1736
2508
  # over the observation stamp, so a decision arriving in this batch
@@ -1743,8 +2515,15 @@ def _cache_applier(decoded) -> int | None:
1743
2515
  # #457: this path is independent of the fused rollout writer,
1744
2516
  # but its quota rows feed the same certificate and dashboard
1745
2517
  # signatures. Keep the token atomic with the materialization.
1746
- import _cctally_cache
1747
2518
  _cctally_cache._bump_codex_physical_mutation_seq(cache)
2519
+ if plan is not None:
2520
+ # AFTER the bump, so the certificate carries the post-bump
2521
+ # sequence, and inside this transaction so a rollback leaves the
2522
+ # predecessor standing even when the journal appends survived.
2523
+ vector, covered, applied_through = plan
2524
+ _cctally_cache._advance_codex_journal_coverage(
2525
+ cache, prior=prior, covered=covered,
2526
+ applied_through=applied_through, pinned_vector=vector)
1748
2527
  cache.commit()
1749
2528
  _report_file_account_conflicts(_file_conflicts)
1750
2529
  except sqlite3.Error as exc:
@@ -3059,64 +3838,479 @@ def _fold_order(evt) -> int:
3059
3838
  return (_EVT_SPECS.get(kind) or _UNKNOWN_EVT_SPEC).order
3060
3839
 
3061
3840
 
3062
- def _write_effective_metadata(conn, selection) -> None:
3063
- """Replace the disposable effective-event summary from a pure selection."""
3841
+ def _replace_protocol_violations(conn, rows) -> None:
3842
+ """Replace the disposable structural-violation summary from kernel rows.
3843
+
3844
+ ONE writer, shared by the rebuild's whole-generation write and the live
3845
+ path's full-prefix fallback, because the two must produce IDENTICAL rows.
3846
+ They did not: the fallback wrote four columns and left `available_after`
3847
+ NULL, and it serialized with `ensure_ascii=True` while the kernel uses
3848
+ `ensure_ascii=False`. A fallback tick therefore left rows a fresh derivation
3849
+ would not match — always on `available_after`, and on the JSON bytes for any
3850
+ non-ASCII character in a violation payload — so
3851
+ `stats_index_matches_journal_prefix` answered False afterwards and the
3852
+ incremental path carried the wrong evidence forward verbatim.
3853
+ """
3854
+ conn.execute("DELETE FROM journal_protocol_violations")
3855
+ for row in rows:
3856
+ conn.execute(
3857
+ "INSERT INTO journal_protocol_violations "
3858
+ "(fingerprint, batch_id, kind, violation_json, available_after) "
3859
+ "VALUES (?, ?, ?, ?, ?)",
3860
+ (
3861
+ row.fingerprint,
3862
+ row.batch_id,
3863
+ row.kind,
3864
+ row.violation_json,
3865
+ row.available_after,
3866
+ ),
3867
+ )
3868
+
3869
+
3870
+ def _write_selector_state(conn, rows) -> None:
3871
+ """Replace the whole durable selector generation from pure-kernel rows.
3872
+
3873
+ This took over from `_write_effective_metadata`, which it superseded and
3874
+ which is now deleted: it writes the same `journal_effective_events` and
3875
+ `journal_protocol_violations` content plus the two added columns, the
3876
+ retained violation evidence, and the three selector tables. Everything lands
3877
+ in ONE transaction with the index content, so a generation never publishes
3878
+ selector state describing a different fold.
3879
+ """
3064
3880
  conn.execute("DELETE FROM journal_effective_events")
3065
- for event_id, selected in selection.by_id.items():
3066
- event_json = None
3067
- if selected.record is not None:
3068
- event_json = (
3069
- _lib_journal.encode_line(selected.record)
3070
- .decode("utf-8")
3071
- .rstrip("\n")
3072
- )
3881
+ for row in rows.effective:
3073
3882
  conn.execute(
3074
3883
  "INSERT INTO journal_effective_events "
3075
- "(event_id, rev, status, content_hash, batch_id, event_json) "
3076
- "VALUES (?, ?, ?, ?, ?, ?)",
3884
+ "(event_id, rev, status, content_hash, batch_id, event_json, "
3885
+ " winning_sequence, conflict_hashes_json) "
3886
+ "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
3077
3887
  (
3078
- event_id,
3079
- selected.rev,
3080
- selected.status,
3081
- selected.content_hash,
3082
- selected.batch_id,
3083
- event_json,
3888
+ row.event_id,
3889
+ row.rev,
3890
+ row.status,
3891
+ row.content_hash,
3892
+ row.batch_id,
3893
+ row.event_json,
3894
+ row.winning_sequence,
3895
+ row.conflict_hashes_json,
3084
3896
  ),
3085
3897
  )
3086
- _write_protocol_violations(
3087
- conn,
3088
- selection.protocol_violations,
3089
- selection.acknowledged_protocol_violations,
3898
+ _replace_protocol_violations(conn, rows.violations)
3899
+ conn.execute("DELETE FROM journal_selector_batch_records")
3900
+ for row in rows.batch_records:
3901
+ conn.execute(
3902
+ "INSERT INTO journal_selector_batch_records "
3903
+ "(batch_id, kind, key, record_digest, identity_digest, sequence, "
3904
+ " action_core_json) VALUES (?, ?, ?, ?, ?, ?, ?)",
3905
+ (
3906
+ row.batch_id,
3907
+ row.kind,
3908
+ row.key,
3909
+ row.record_digest,
3910
+ row.identity_digest,
3911
+ row.sequence,
3912
+ row.action_core_json,
3913
+ ),
3914
+ )
3915
+ conn.execute("DELETE FROM journal_selector_batches")
3916
+ for row in rows.batches:
3917
+ conn.execute(
3918
+ "INSERT INTO journal_selector_batches "
3919
+ "(batch_id, status, action_count, action_set_hash, begin_segment, "
3920
+ " begin_offset, earliest_commit_segment, earliest_commit_offset) "
3921
+ "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
3922
+ (
3923
+ row.batch_id,
3924
+ row.status,
3925
+ row.action_count,
3926
+ row.action_set_hash,
3927
+ row.begin_segment,
3928
+ row.begin_offset,
3929
+ row.earliest_commit_segment,
3930
+ row.earliest_commit_offset,
3931
+ ),
3932
+ )
3933
+ conn.execute("DELETE FROM journal_selector_state")
3934
+ state = rows.state
3935
+ conn.execute(
3936
+ "INSERT INTO journal_selector_state "
3937
+ "(id, generation_record_path, generation_stamped_at_utc, "
3938
+ " covered_segment, covered_offset, next_sequence, selector_version, "
3939
+ " cutover_seen, cutover_account_key) "
3940
+ "VALUES (1, ?, ?, ?, ?, ?, ?, ?, ?)",
3941
+ (
3942
+ state.generation_record_path,
3943
+ state.generation_stamped_at_utc,
3944
+ state.covered_segment,
3945
+ state.covered_offset,
3946
+ state.next_sequence,
3947
+ state.selector_version,
3948
+ 1 if state.cutover_seen else 0,
3949
+ state.cutover_account_key,
3950
+ ),
3090
3951
  )
3091
3952
 
3092
3953
 
3093
- def _write_protocol_violations(conn, violations, acknowledged=()) -> None:
3094
- """Replace the disposable structural-violation summary."""
3095
- conn.execute("DELETE FROM journal_protocol_violations")
3096
- rows = [*violations, *acknowledged]
3097
- rows.sort(
3098
- key=lambda violation: (
3099
- violation.batch_id,
3100
- violation.kind,
3101
- violation.fingerprint,
3102
- )
3954
+ def _write_selector_delta(conn, before, after) -> None:
3955
+ """Advance durable selector state from ``before`` to ``after`` in place.
3956
+
3957
+ A whole-generation replace is what the rebuild does, and it is the wrong
3958
+ shape for an ingest tick: `journal_effective_events` holds one row per event
3959
+ id 34,644 on the maintainer's production journal — and rewriting all of
3960
+ every status-line tick would trade the read this session removes for a write
3961
+ of the same size. So only the rows that actually changed are written.
3962
+
3963
+ Three of the four groups express upserts only, and the reason is that **the
3964
+ fold emits no removals** for them: `after ⊇ before` in every one. It is NOT
3965
+ that a key missing from ``after`` lies outside the delta's scope — ``after``
3966
+ is computed FROM ``before``, so every key in ``before`` is by construction
3967
+ inside the read scope, and a key the kernel dropped would be a genuine
3968
+ removal. A merged generation keeps every winner and every batch, and a
3969
+ completed batch's action rows keep their keys with a NULL core rather than
3970
+ disappearing.
3971
+
3972
+ Violations are the exception, so that group DOES express removals. A phase-2
3973
+ verdict is re-derived on every pass and a later record can withdraw one — an
3974
+ incomplete action set completed by a late action stops producing
3975
+ `manifest_action_sequence_mismatch` — and `_check_journal_protocol` reads
3976
+ that table, so a stale row makes `doctor` exit 2 and names a fingerprint no
3977
+ fresh derivation reproduces. The removal is safe precisely because it is
3978
+ scoped: ``before.violations`` is the delta's own scoped read, so the
3979
+ difference can only name rows this delta looked at.
3980
+ """
3981
+ state = after.state
3982
+ conn.execute(
3983
+ "INSERT INTO journal_selector_state "
3984
+ "(id, generation_record_path, generation_stamped_at_utc, "
3985
+ " covered_segment, covered_offset, next_sequence, selector_version, "
3986
+ " cutover_seen, cutover_account_key) "
3987
+ "VALUES (1, ?, ?, ?, ?, ?, ?, ?, ?) "
3988
+ "ON CONFLICT(id) DO UPDATE SET "
3989
+ "generation_record_path = excluded.generation_record_path, "
3990
+ "generation_stamped_at_utc = excluded.generation_stamped_at_utc, "
3991
+ "covered_segment = excluded.covered_segment, "
3992
+ "covered_offset = excluded.covered_offset, "
3993
+ "next_sequence = excluded.next_sequence, "
3994
+ "selector_version = excluded.selector_version, "
3995
+ "cutover_seen = excluded.cutover_seen, "
3996
+ "cutover_account_key = excluded.cutover_account_key",
3997
+ (
3998
+ state.generation_record_path,
3999
+ state.generation_stamped_at_utc,
4000
+ state.covered_segment,
4001
+ state.covered_offset,
4002
+ state.next_sequence,
4003
+ state.selector_version,
4004
+ 1 if state.cutover_seen else 0,
4005
+ state.cutover_account_key,
4006
+ ),
3103
4007
  )
3104
- for violation in rows:
3105
- conn.execute(
3106
- "INSERT INTO journal_protocol_violations "
3107
- "(fingerprint, batch_id, kind, violation_json) "
3108
- "VALUES (?, ?, ?, ?)",
3109
- (
3110
- violation.fingerprint,
3111
- violation.batch_id,
3112
- violation.kind,
3113
- json.dumps(
3114
- violation.to_dict(),
3115
- sort_keys=True,
3116
- separators=(",", ":"),
4008
+
4009
+ # Identity, not equality: `advance_counter` returns the row tuples BY
4010
+ # REFERENCE for a delta the fold does not consume, so a tick that only
4011
+ # moved the counters skips these diffs entirely rather than walking every
4012
+ # durable row to conclude that nothing changed.
4013
+ if before.batches is not after.batches:
4014
+ prior_batches = {row.batch_id: row for row in before.batches}
4015
+ for row in after.batches:
4016
+ if prior_batches.get(row.batch_id) == row:
4017
+ continue
4018
+ conn.execute(
4019
+ "INSERT OR REPLACE INTO journal_selector_batches "
4020
+ "(batch_id, status, action_count, action_set_hash, begin_segment, "
4021
+ " begin_offset, earliest_commit_segment, earliest_commit_offset) "
4022
+ "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
4023
+ (
4024
+ row.batch_id,
4025
+ row.status,
4026
+ row.action_count,
4027
+ row.action_set_hash,
4028
+ row.begin_segment,
4029
+ row.begin_offset,
4030
+ row.earliest_commit_segment,
4031
+ row.earliest_commit_offset,
3117
4032
  ),
3118
- ),
4033
+ )
4034
+
4035
+ if before.batch_records is not after.batch_records:
4036
+ prior_records = {
4037
+ (row.batch_id, row.kind, row.key): row for row in before.batch_records
4038
+ }
4039
+ for row in after.batch_records:
4040
+ key = (row.batch_id, row.kind, row.key)
4041
+ if prior_records.get(key) == row:
4042
+ continue
4043
+ conn.execute(
4044
+ "INSERT OR REPLACE INTO journal_selector_batch_records "
4045
+ "(batch_id, kind, key, record_digest, identity_digest, sequence, "
4046
+ " action_core_json) VALUES (?, ?, ?, ?, ?, ?, ?)",
4047
+ (
4048
+ row.batch_id,
4049
+ row.kind,
4050
+ row.key,
4051
+ row.record_digest,
4052
+ row.identity_digest,
4053
+ row.sequence,
4054
+ row.action_core_json,
4055
+ ),
4056
+ )
4057
+
4058
+ if before.effective is not after.effective:
4059
+ prior_effective = {row.event_id: row for row in before.effective}
4060
+ for row in after.effective:
4061
+ if prior_effective.get(row.event_id) == row:
4062
+ continue
4063
+ conn.execute(
4064
+ "INSERT OR REPLACE INTO journal_effective_events "
4065
+ "(event_id, rev, status, content_hash, batch_id, event_json, "
4066
+ " winning_sequence, conflict_hashes_json) "
4067
+ "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
4068
+ (
4069
+ row.event_id,
4070
+ row.rev,
4071
+ row.status,
4072
+ row.content_hash,
4073
+ row.batch_id,
4074
+ row.event_json,
4075
+ row.winning_sequence,
4076
+ row.conflict_hashes_json,
4077
+ ),
4078
+ )
4079
+
4080
+ if before.violations is not after.violations:
4081
+ prior_violations = {row.fingerprint: row for row in before.violations}
4082
+ withdrawn = set(prior_violations) - {
4083
+ row.fingerprint for row in after.violations
4084
+ }
4085
+ for fingerprint in sorted(withdrawn):
4086
+ conn.execute(
4087
+ "DELETE FROM journal_protocol_violations WHERE fingerprint = ?",
4088
+ (fingerprint,),
4089
+ )
4090
+ for row in after.violations:
4091
+ if prior_violations.get(row.fingerprint) == row:
4092
+ continue
4093
+ conn.execute(
4094
+ "INSERT OR REPLACE INTO journal_protocol_violations "
4095
+ "(fingerprint, batch_id, kind, violation_json, available_after) "
4096
+ "VALUES (?, ?, ?, ?, ?)",
4097
+ (
4098
+ row.fingerprint,
4099
+ row.batch_id,
4100
+ row.kind,
4101
+ row.violation_json,
4102
+ row.available_after,
4103
+ ),
4104
+ )
4105
+
4106
+
4107
+ def _read_selector_state(conn):
4108
+ """`journal_selector_state`'s single row as a kernel row, or None.
4109
+
4110
+ Split out of `_read_selector_rows` because the three cheap validity checks —
4111
+ `selector_version`, cursor agreement and generation identity — are decided
4112
+ from this row alone. Materializing the four row groups ahead of them made a
4113
+ degraded tick and a counters-only tick read roughly 100,000 rows to conclude
4114
+ that nothing was needed, on the status-line path.
4115
+
4116
+ Returns None when the table does not hold exactly one row, or is unreadable
4117
+ at all: a stats index predating epoch 1009 has no selector tables, and
4118
+ `open_db` returns a legacy one unchanged until its migration or epoch path
4119
+ runs. Both are degraded states that fall back to full selection, and neither
4120
+ may be mistaken for an empty-but-valid generation.
4121
+ """
4122
+ try:
4123
+ state_rows = conn.execute(
4124
+ "SELECT generation_record_path, generation_stamped_at_utc, "
4125
+ "covered_segment, covered_offset, next_sequence, selector_version, "
4126
+ "cutover_seen, cutover_account_key FROM journal_selector_state"
4127
+ ).fetchall()
4128
+ except sqlite3.Error:
4129
+ return None
4130
+ if len(state_rows) != 1:
4131
+ return None
4132
+ row = state_rows[0]
4133
+ return _lib_selector_state.SelectorStateRow(
4134
+ next_sequence=int(row[4]),
4135
+ selector_version=int(row[5]),
4136
+ covered_segment=row[2],
4137
+ covered_offset=None if row[3] is None else int(row[3]),
4138
+ generation_record_path=row[0],
4139
+ generation_stamped_at_utc=row[1],
4140
+ cutover_seen=bool(row[6]),
4141
+ cutover_account_key=row[7],
4142
+ )
4143
+
4144
+
4145
+ def _read_selector_rows(conn, *, batch_ids=None, event_ids=None):
4146
+ """Read one generation's durable selector state back as kernel rows.
4147
+
4148
+ Returns `None` when `journal_selector_state` does not hold exactly one row.
4149
+ A zero-row or unreadable state is one of the degraded cases that falls back
4150
+ to full selection, and it must never be mistaken for an empty-but-valid
4151
+ generation.
4152
+
4153
+ ``batch_ids`` and ``event_ids`` SCOPE the four row groups to what one delta
4154
+ can reach. Both default to None, which reads the whole generation — the
4155
+ shape validation and the rebuild need. The live path always scopes, because
4156
+ the unscoped read materializes 34,644 `journal_effective_events` rows (mean
4157
+ 825 bytes of retained JSON) and 64,248 `journal_selector_batch_records` rows
4158
+ on a production journal, on every status-line tick.
4159
+
4160
+ Under a scope the returned rows are a SUBSET, so the caller must not treat a
4161
+ key absent from them as a key that was removed — `_write_selector_delta`
4162
+ documents that consequence at the write end.
4163
+
4164
+ The five reads run inside ONE deferred read transaction, so they share a
4165
+ snapshot. Without it each takes its own, and a publication landing between
4166
+ two of them would return rows from two generations under one state row's
4167
+ identity. Every current caller holds the ingest or maintenance lock, so that
4168
+ interleaving is not reachable today; the transaction is what keeps it
4169
+ unreachable if a future caller does not.
4170
+ """
4171
+ _ks = _lib_selector_state
4172
+
4173
+ with _deferred_read(conn):
4174
+ state = _read_selector_state(conn)
4175
+ if state is None:
4176
+ return None
4177
+ return _ks.SelectorRows(
4178
+ state=state,
4179
+ batches=_read_selector_batch_rows(conn, batch_ids),
4180
+ batch_records=_read_selector_batch_record_rows(conn, batch_ids),
4181
+ effective=_read_selector_effective_rows(conn, event_ids),
4182
+ violations=_read_selector_violation_rows(conn, batch_ids),
4183
+ )
4184
+
4185
+
4186
+ @contextlib.contextmanager
4187
+ def _deferred_read(conn):
4188
+ """One snapshot across several SELECTs, without disturbing an open txn.
4189
+
4190
+ A caller already inside a transaction keeps it — beginning a second one
4191
+ raises — and the snapshot it holds is the one this block wanted anyway.
4192
+ """
4193
+ if conn.in_transaction:
4194
+ yield
4195
+ return
4196
+ try:
4197
+ conn.execute("BEGIN")
4198
+ except sqlite3.Error:
4199
+ yield
4200
+ return
4201
+ try:
4202
+ yield
4203
+ finally:
4204
+ try:
4205
+ conn.rollback()
4206
+ except sqlite3.Error:
4207
+ pass
4208
+
4209
+
4210
+ #: SQLite's default parameter ceiling is 999, so a scope wider than this is read
4211
+ #: in chunks rather than in one statement.
4212
+ _SCOPE_CHUNK = 500
4213
+
4214
+
4215
+ def _scoped_query(conn, sql, order_by, column, scope):
4216
+ """Run ``sql`` unscoped, or once per chunk of ``scope`` on ``column``."""
4217
+ if scope is None:
4218
+ yield from conn.execute(f"{sql} ORDER BY {order_by}")
4219
+ return
4220
+ keys = sorted(scope)
4221
+ for start in range(0, len(keys), _SCOPE_CHUNK):
4222
+ chunk = keys[start:start + _SCOPE_CHUNK]
4223
+ placeholders = ",".join("?" * len(chunk))
4224
+ yield from conn.execute(
4225
+ f"{sql} WHERE {column} IN ({placeholders}) ORDER BY {order_by}",
4226
+ chunk,
4227
+ )
4228
+
4229
+
4230
+ def _read_selector_batch_rows(conn, scope):
4231
+ _ks = _lib_selector_state
4232
+ return tuple(
4233
+ _ks.SelectorBatchRow(
4234
+ batch_id=item[0],
4235
+ status=item[1],
4236
+ action_count=None if item[2] is None else int(item[2]),
4237
+ action_set_hash=item[3],
4238
+ begin_segment=item[4],
4239
+ begin_offset=None if item[5] is None else int(item[5]),
4240
+ earliest_commit_segment=item[6],
4241
+ earliest_commit_offset=None if item[7] is None else int(item[7]),
4242
+ )
4243
+ for item in _scoped_query(
4244
+ conn,
4245
+ "SELECT batch_id, status, action_count, action_set_hash, "
4246
+ "begin_segment, begin_offset, earliest_commit_segment, "
4247
+ "earliest_commit_offset FROM journal_selector_batches",
4248
+ "batch_id", "batch_id", scope,
4249
+ )
4250
+ )
4251
+
4252
+
4253
+ def _read_selector_batch_record_rows(conn, scope):
4254
+ _ks = _lib_selector_state
4255
+ return tuple(
4256
+ _ks.SelectorBatchRecordRow(
4257
+ batch_id=item[0],
4258
+ kind=item[1],
4259
+ key=item[2],
4260
+ record_digest=item[3],
4261
+ sequence=int(item[5]),
4262
+ identity_digest=item[4],
4263
+ action_core_json=item[6],
4264
+ )
4265
+ for item in _scoped_query(
4266
+ conn,
4267
+ "SELECT batch_id, kind, key, record_digest, identity_digest, "
4268
+ "sequence, action_core_json FROM journal_selector_batch_records",
4269
+ "batch_id, kind, key", "batch_id", scope,
4270
+ )
4271
+ )
4272
+
4273
+
4274
+ def _read_selector_effective_rows(conn, scope):
4275
+ _ks = _lib_selector_state
4276
+ return tuple(
4277
+ _ks.SelectorEffectiveRow(
4278
+ event_id=item[0],
4279
+ rev=int(item[1]),
4280
+ status=item[2],
4281
+ content_hash=item[3],
4282
+ batch_id=item[4],
4283
+ event_json=item[5],
4284
+ winning_sequence=None if item[6] is None else int(item[6]),
4285
+ conflict_hashes_json=item[7],
4286
+ )
4287
+ for item in _scoped_query(
4288
+ conn,
4289
+ "SELECT event_id, rev, status, content_hash, batch_id, event_json, "
4290
+ "winning_sequence, conflict_hashes_json "
4291
+ "FROM journal_effective_events",
4292
+ "event_id", "event_id", scope,
4293
+ )
4294
+ )
4295
+
4296
+
4297
+ def _read_selector_violation_rows(conn, scope):
4298
+ _ks = _lib_selector_state
4299
+ return tuple(
4300
+ _ks.SelectorViolationRow(
4301
+ fingerprint=item[0],
4302
+ batch_id=item[1],
4303
+ kind=item[2],
4304
+ violation_json=item[3],
4305
+ available_after=None if item[4] is None else int(item[4]),
4306
+ )
4307
+ for item in _scoped_query(
4308
+ conn,
4309
+ "SELECT fingerprint, batch_id, kind, violation_json, "
4310
+ "available_after FROM journal_protocol_violations",
4311
+ "batch_id, kind, fingerprint", "batch_id", scope,
3119
4312
  )
4313
+ )
3120
4314
 
3121
4315
 
3122
4316
  def _metadata_row(conn, event_id):
@@ -3438,7 +4632,7 @@ def _validate_excluded_derived_fks(conn, spec, row) -> None:
3438
4632
  f"(re-derived {expected})")
3439
4633
 
3440
4634
 
3441
- def _full_effective_selection(hw):
4635
+ def _full_effective_selection(hw, accumulators=None):
3442
4636
  records = []
3443
4637
  evidence = []
3444
4638
  prior_high_water = None
@@ -3459,25 +4653,427 @@ def _full_effective_selection(hw):
3459
4653
  return _lib_journal.resolve_effective_events(
3460
4654
  records,
3461
4655
  protocol_prefix_evidence=evidence,
4656
+ accumulators=accumulators,
3462
4657
  )
3463
4658
 
3464
4659
 
3465
- def _correction_commit_high_water(batch_id, hw=None):
3466
- """Return the exact end offset of one completed-batch commit marker.
4660
+ def _selector_generation_matches(conn, state) -> bool:
4661
+ """Whether ``state`` names the generation this connection is reading.
3467
4662
 
3468
- The batch was already structurally validated either by the full effective
3469
- selector or by the live metadata row that names it. The earliest matching
3470
- commit is the narrowest complete prefix and remains stable even when later
3471
- journal bytes or crash-replayed duplicate markers exist.
4663
+ `stats_publication_stamp` carries no counter, so this is an IDENTITY
4664
+ comparison rather than an ordering one, and the stamp is re-read on a FRESH
4665
+ read-only connection: `conn` may be sitting on a superseded generation
4666
+ precisely because in-place publication keeps an open reader alive on the one
4667
+ it started with, which is the case this check exists to catch.
3472
4668
 
3473
- Streams rather than materializing (#496 S4): the previous form built the
3474
- whole prefix through `_read_range` before its first-match return, so a
3475
- marker in the first segment still paid for every later one.
4669
+ Absent, duplicated, unreadable or mismatched all answer False, which is the
4670
+ fall-back-to-full-selection direction (spec §3.4). A stamp read from a
4671
+ destination at any other epoch also answers False, for the same reason
4672
+ `read_publication_stamp` returns None there.
3476
4673
  """
3477
- if not batch_id:
3478
- return None
3479
- if hw is None:
3480
- hw = journal_high_water()
4674
+ if state.generation_record_path is None or (
4675
+ state.generation_stamped_at_utc is None
4676
+ ):
4677
+ return False
4678
+ try:
4679
+ row = conn.execute(
4680
+ "SELECT file FROM pragma_database_list WHERE name = 'main'"
4681
+ ).fetchone()
4682
+ except sqlite3.Error:
4683
+ return False
4684
+ path = None if row is None else row[0]
4685
+ if not path:
4686
+ return False
4687
+ try:
4688
+ probe = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
4689
+ try:
4690
+ epoch = int(probe.execute("PRAGMA user_version").fetchone()[0])
4691
+ if epoch != _cctally_core.STATS_INDEX_EPOCH:
4692
+ return False
4693
+ rows = probe.execute(
4694
+ "SELECT record_path, stamped_at_utc FROM stats_publication_stamp"
4695
+ ).fetchall()
4696
+ finally:
4697
+ probe.close()
4698
+ except BaseException:
4699
+ return False
4700
+ if len(rows) != 1:
4701
+ return False
4702
+ return (str(rows[0][0]), str(rows[0][1])) == (
4703
+ state.generation_record_path,
4704
+ state.generation_stamped_at_utc,
4705
+ )
4706
+
4707
+
4708
+ def _cutover_for_selection(seen: bool, account_key, hw):
4709
+ """The cutover account a selection over this prefix must normalize with.
4710
+
4711
+ When durable state SAW the op, the recorded value is exact and no scan
4712
+ happens — which is what removes `resolve_cutover_claude_account()`'s
4713
+ whole-journal traversal from the live path (spec §3.6). When it did not, the
4714
+ answer is `_resolve_cutover_for_rebuild`'s, so the live path and a rebuild
4715
+ pinned at the same high-water cannot disagree; on the live path the pinned
4716
+ high-water IS the current one, so that call returns immediately without
4717
+ opening a segment.
4718
+ """
4719
+ if seen:
4720
+ return (
4721
+ account_key if account_key is not None
4722
+ else _lib_accounts.UNATTRIBUTED
4723
+ )
4724
+ return _resolve_cutover_for_rebuild(_CUTOVER_UNSEEN, hw, list_segments())
4725
+
4726
+
4727
+ def _causal_offset_of(sequence, coordinates):
4728
+ """The journal coordinate of the record at ``sequence``, or None.
4729
+
4730
+ A separate function so the fail-closed path has something to exercise. The
4731
+ coordinate is MANDATORY: substituting the pinned high-water is unsafe,
4732
+ because it is `st_size`, `_iter_segment_lines` omits an incomplete trailing
4733
+ line, and `_repair_torn_tail` can truncate below it — so a cursor written
4734
+ there can sit beyond unread data (spec §3.7).
4735
+ """
4736
+ return coordinates.get(sequence)
4737
+
4738
+
4739
+ def _prefix_position(coordinate):
4740
+ """A comparable position for a journal prefix; None is the very start.
4741
+
4742
+ Ordered through `segment_sort_key`, because bootstrap segments sort before
4743
+ observation segments and a plain string comparison would get the order
4744
+ wrong across that boundary.
4745
+ """
4746
+ if not coordinate or coordinate[0] is None or coordinate[1] is None:
4747
+ return None
4748
+ return (_lib_journal.segment_sort_key(coordinate[0]), int(coordinate[1]))
4749
+
4750
+
4751
+ def _coordinate_covers(covered, target) -> bool:
4752
+ """Whether ``covered`` reaches at least as far as ``target``."""
4753
+ left = _prefix_position(covered)
4754
+ right = _prefix_position(target)
4755
+ if left is None or right is None:
4756
+ return False
4757
+ return left >= right
4758
+
4759
+
4760
+ #: The most journal bytes one live tick may re-fold to realign a durable
4761
+ #: selector prefix that fell behind the caller's cursor.
4762
+ #:
4763
+ #: The gap is normally one tick wide and closes on that tick, because the common
4764
+ #: transient — a `database is locked` swallowed by `_selector_generation_matches`
4765
+ #: — is decided BEFORE the gap read and the next successful tick writes the
4766
+ #: realigned prefix. One refusal is not the shape this cap exists for.
4767
+ #:
4768
+ #: The shape it exists for is a refusal that repeats. `merge_delta`'s
4769
+ #: durably-completed-batch shape refusal falls back to
4770
+ #: `_full_effective_selection`, whose loop acts only on entries whose `batch_id`
4771
+ #: is not `None` — and after a taint the winner reverts to the base journal
4772
+ #: event, whose `batch_id` IS `None`. That is issue #510, so no
4773
+ #: `CorrectionRebuildRequired` is raised and no rebuild follows to realign the
4774
+ #: durable prefix. Every later tick would then re-read and re-decode a
4775
+ #: monotonically growing range, which is the whole-prefix read spec §3.3 exists
4776
+ #: to prevent under another name.
4777
+ #:
4778
+ #: Past the cap the tick degrades to full selection like every other degraded
4779
+ #: case, and `_observe_selector_desynchronization` reports the gap and this cap
4780
+ #: in the structured rebuild record, so the state is observable rather than
4781
+ #: silent. Four mebibytes is roughly 4,600 records at the maintainer's measured
4782
+ #: mean journal line length; a genuine one-tick gap is a handful.
4783
+ _GAP_REFOLD_BYTE_CAP = 4 * 1024 * 1024
4784
+
4785
+
4786
+ def _selector_gap_bytes(durable, cursor) -> "int | None":
4787
+ """Journal bytes in ``[durable, cursor)``, from `stat` alone.
4788
+
4789
+ Mirrors `_iter_range_with_segments`' bounds arithmetic without opening a
4790
+ single segment, which is what lets the cap refuse BEFORE the read it caps.
4791
+
4792
+ Returns `None` when the distance cannot be determined — an unreadable
4793
+ segment, or a cursor naming a segment the journal no longer lists. The
4794
+ caller treats that as over the cap, which is the fall-back direction.
4795
+ """
4796
+ if cursor is None or cursor[0] is None or cursor[1] is None:
4797
+ return None
4798
+ segments = list_segments()
4799
+ if cursor[0] not in segments:
4800
+ return None
4801
+ end_idx = segments.index(cursor[0])
4802
+ if (
4803
+ durable is None
4804
+ or durable[0] is None
4805
+ or durable[1] is None
4806
+ or durable[0] not in segments
4807
+ ):
4808
+ start_idx, start_off = 0, 0
4809
+ else:
4810
+ start_idx, start_off = segments.index(durable[0]), int(durable[1])
4811
+ total = 0
4812
+ for idx in range(start_idx, end_idx + 1):
4813
+ lo = start_off if idx == start_idx else 0
4814
+ if idx == end_idx:
4815
+ hi = int(cursor[1])
4816
+ else:
4817
+ try:
4818
+ hi = os.path.getsize(_cctally_core.JOURNAL_DIR / segments[idx])
4819
+ except OSError:
4820
+ return None
4821
+ if hi > lo:
4822
+ total += hi - lo
4823
+ return total
4824
+
4825
+
4826
+ def _selector_gap_entries(durable, cursor):
4827
+ """Decoded records in ``[durable, cursor)`` with each one's end offset.
4828
+
4829
+ This is non-empty only after a tick that could not validate the durable
4830
+ generation. `_write_cursor` runs whether or not a selector delta was
4831
+ written, so such a tick advances `journal_cursor` and leaves
4832
+ `journal_selector_state` where the last rebuild put it — and nothing on the
4833
+ live path ever rewrites that table, so an EQUALITY comparison between the
4834
+ two could never hold again and F20 stayed off until a full rebuild.
4835
+
4836
+ Re-folding the gap is what brings them back together. It is correct because
4837
+ the selector fold is independent of application, so folding records the
4838
+ cycle already applied changes only the selector's own accumulators, and the
4839
+ read is bounded by the gap rather than by the whole prefix.
4840
+ """
4841
+ entries = []
4842
+ start = None if _prefix_position(durable) is None else (
4843
+ durable[0], int(durable[1]))
4844
+ for segment, offset, raw in iter_range(start, cursor):
4845
+ record = _lib_journal.decode_line(raw)
4846
+ if record is None:
4847
+ # A malformed line consumes no sequence number, exactly as the
4848
+ # cycle's own read loop and `decoded_entry_count` treat it.
4849
+ continue
4850
+ entries.append((record, segment, offset + len(raw) + 1))
4851
+ return entries
4852
+
4853
+
4854
+ def _normalized_for_selection(records, cutover_claude):
4855
+ """Copies of ``records`` with the legacy account stamp applied.
4856
+
4857
+ COPIES, because `_normalize_legacy_account_stamp` mutates in place and these
4858
+ same dicts are what the cycle's pipeline folds; injecting a stamp into them
4859
+ would change what step 4b sees. The payload is copied too, since the evt/op
4860
+ branch writes into it.
4861
+ """
4862
+ copies = []
4863
+ for record in records:
4864
+ if not isinstance(record, dict):
4865
+ copies.append(record)
4866
+ continue
4867
+ copy = dict(record)
4868
+ payload = record.get("payload")
4869
+ if isinstance(payload, dict):
4870
+ copy["payload"] = dict(payload)
4871
+ _normalize_legacy_account_stamp(copy, cutover_claude)
4872
+ copies.append(copy)
4873
+ return copies
4874
+
4875
+
4876
+ def _incremental_selection(conn, records, entries, cursor, covered):
4877
+ """Continue the durable fold over one cycle's delta, or return None.
4878
+
4879
+ ``entries`` is positionally parallel to ``records``: each element is the
4880
+ ``(segment, end offset)`` of the record that consumed one sequence number.
4881
+ The map from ABSOLUTE sequence to coordinate is built HERE rather than by the
4882
+ caller, because a durable prefix behind the cursor prepends its unfolded gap
4883
+ and shifts every delta record's number.
4884
+
4885
+ None means every degraded case: absent or unreadable selector state, a
4886
+ `selector_version` mismatch, a durable prefix AHEAD of the caller's cursor, a
4887
+ gap wider than `_GAP_REFOLD_BYTE_CAP`, a generation identity that does not
4888
+ match a freshly read publication stamp, a generation that moved while the
4889
+ gap was being read, a cutover operation the durable prefix has not folded,
4890
+ and anything the pure kernel refuses. Every one of them takes the existing
4891
+ full-prefix path unchanged.
4892
+
4893
+ Read order is load-bearing. The three cheap checks are decided from ONE row
4894
+ and run BEFORE any row group is materialized, a delta the fold does not
4895
+ consume reads no group at all, and a delta that does is scoped to the
4896
+ batches and event ids it names. This function is on `cmd_record_usage`'s
4897
+ path, which runs on every Claude Code status-line tick, and an unscoped read
4898
+ materializes 34,644 effective rows and 64,248 batch-record rows on a
4899
+ production journal.
4900
+
4901
+ There are TWO read snapshots rather than one, and the gap re-fold sits
4902
+ between them deliberately: it is arbitrary journal file I/O, and holding a
4903
+ stats.db read snapshot open across it pins the WAL against checkpointing for
4904
+ the whole read, which is the bloat #297 documents. The second snapshot
4905
+ re-reads the one state row and refuses on any difference, which is what
4906
+ keeps the row groups and the state row from coming out of two generations.
4907
+
4908
+ That re-read covers in-place publication and any live delta writer, because
4909
+ both mutate the state row this connection can see. It does NOT cover
4910
+ PHYSICAL replacement: an `os.replace` between the two blocks leaves this
4911
+ connection on the unlinked old inode, whose state row and row groups are
4912
+ mutually consistent but stale. Physical replacement is excluded by the ingest
4913
+ lock the caller already holds, not by this check.
4914
+ """
4915
+ _ks = _lib_selector_state
4916
+ with _deferred_read(conn):
4917
+ state = _read_selector_state(conn)
4918
+ if state is None:
4919
+ return None
4920
+ if state.selector_version != _ks.SELECTOR_VERSION:
4921
+ return None
4922
+ durable = (state.covered_segment, state.covered_offset)
4923
+ durable_at = _prefix_position(durable)
4924
+ cursor_at = _prefix_position(cursor)
4925
+ if durable_at is not None and (
4926
+ cursor_at is None or durable_at > cursor_at
4927
+ ):
4928
+ # The durable prefix has folded records the caller has not applied,
4929
+ # so numbering the delta from `next_sequence` would leave a hole the
4930
+ # size of the overlap. The other direction is recoverable and is
4931
+ # recovered below.
4932
+ return None
4933
+ if not _selector_generation_matches(conn, state):
4934
+ return None
4935
+
4936
+ if durable_at == cursor_at:
4937
+ gap = ()
4938
+ else:
4939
+ gap_bytes = _selector_gap_bytes(durable, cursor)
4940
+ if gap_bytes is None or gap_bytes > _GAP_REFOLD_BYTE_CAP:
4941
+ # Bounded rather than unbounded: a refusal that repeats leaves the
4942
+ # durable prefix behind forever, and an uncapped re-fold would grow
4943
+ # the per-tick read without limit. See `_GAP_REFOLD_BYTE_CAP`.
4944
+ return None
4945
+ gap = _selector_gap_entries(durable, cursor)
4946
+
4947
+ stream = [item[0] for item in gap] + list(records)
4948
+ coordinates = {}
4949
+ for index, item in enumerate(gap):
4950
+ coordinates[state.next_sequence + index] = (item[1], item[2])
4951
+ for index, coordinate in enumerate(entries or ()):
4952
+ coordinates[state.next_sequence + len(gap) + index] = coordinate
4953
+
4954
+ for record in stream:
4955
+ if isinstance(record, dict) and record.get("id") == CUTOVER_OP_ID:
4956
+ if not state.cutover_seen:
4957
+ # A cutover the durable prefix has not folded re-normalizes
4958
+ # every legacy Claude line in that prefix, which changes those
4959
+ # events' `content_hash` and `event_json`. The durable winners
4960
+ # were folded WITHOUT it, and this path normalizes only the
4961
+ # delta, so carrying them forward would diverge from a full
4962
+ # pass. Falling back is provably equivalent to one.
4963
+ return None
4964
+ break
4965
+
4966
+ if not any(
4967
+ isinstance(record, dict)
4968
+ and (
4969
+ record.get("t") in _ks.FOLD_RECORD_TYPES
4970
+ or (
4971
+ record.get("t") == "op"
4972
+ and isinstance(record.get("payload"), dict)
4973
+ and record["payload"].get("kind")
4974
+ == _lib_journal._PROTOCOL_RESOLUTION_KIND
4975
+ )
4976
+ )
4977
+ for record in stream
4978
+ ):
4979
+ # Nothing here changes the fold, so only the counters move — and the
4980
+ # four row groups are not read at all. `advance_counter` returns them
4981
+ # by reference, so an empty placeholder makes the delta writer skip
4982
+ # every group by identity, exactly as a full read would have.
4983
+ placeholder = _ks.SelectorRows(state=state)
4984
+ return {
4985
+ "before": placeholder,
4986
+ "after": _ks.advance_counter(
4987
+ placeholder,
4988
+ consumed=state.next_sequence + len(stream),
4989
+ covered=covered,
4990
+ ),
4991
+ "transitions": [],
4992
+ "coordinates": coordinates,
4993
+ }
4994
+
4995
+ batch_ids = _ks.delta_batch_scope(stream)
4996
+ with _deferred_read(conn):
4997
+ if _read_selector_state(conn) != state:
4998
+ return None
4999
+ batches = _read_selector_batch_rows(conn, batch_ids)
5000
+ batch_records = _read_selector_batch_record_rows(conn, batch_ids)
5001
+ effective = _read_selector_effective_rows(
5002
+ conn, _ks.delta_event_scope(stream, batch_records))
5003
+ violations = _read_selector_violation_rows(conn, batch_ids)
5004
+ # No coordinate widening beyond `batch_ids`, and the reason is a
5005
+ # reachability argument rather than an economy. `_preflight_live_events`
5006
+ # consults a batch row only for a winner whose four-tuple DIFFERS from
5007
+ # `journal_effective_events`. `_read_selector_effective_rows` reads that
5008
+ # same table, so a winner the merge passed through compares equal and is
5009
+ # skipped; and a winner the merge re-decided took a correction candidate
5010
+ # from a batch the delta named, so its batch row is already in `batches`.
5011
+ # A read over `{winner batches} - batch_ids` therefore returned rows the
5012
+ # caller could not reach, on every tick that named any correction batch.
5013
+ rows = _ks.SelectorRows(
5014
+ state=state,
5015
+ batches=batches,
5016
+ batch_records=batch_records,
5017
+ effective=effective,
5018
+ violations=violations,
5019
+ )
5020
+
5021
+ cutover_claude = _cutover_for_selection(
5022
+ state.cutover_seen, state.cutover_account_key, covered)
5023
+
5024
+ try:
5025
+ merged, transitions = _ks.merge_delta(
5026
+ rows,
5027
+ _normalized_for_selection(stream, cutover_claude),
5028
+ next_sequence=state.next_sequence,
5029
+ coordinates=coordinates,
5030
+ covered=covered,
5031
+ )
5032
+ except _ks.IncrementalSelectionUnavailable:
5033
+ return None
5034
+ return {
5035
+ "before": rows,
5036
+ "after": merged,
5037
+ "transitions": transitions,
5038
+ "coordinates": coordinates,
5039
+ }
5040
+
5041
+
5042
+ def _raise_taint_transition(transition, coordinates) -> None:
5043
+ """Turn one completed-to-tainted transition into its recovery signal."""
5044
+ coordinate = _causal_offset_of(transition.causal_sequence, coordinates)
5045
+ if coordinate is None:
5046
+ raise JournalError(
5047
+ f"correction batch {transition.batch_id} moved completed -> "
5048
+ "tainted but its causal offset could not be resolved; refusing to "
5049
+ "substitute the pinned high-water"
5050
+ )
5051
+ raise CorrectionRebuildRequired(
5052
+ f"completed correction batch {transition.batch_id} was tainted by a "
5053
+ "later record and requires a stats index rebuild through it",
5054
+ batch_id=transition.batch_id,
5055
+ high_water=coordinate,
5056
+ recovery_eligible=True,
5057
+ kind=CORRECTION_KIND_COMPLETED_TO_TAINTED,
5058
+ )
5059
+
5060
+
5061
+ def _correction_commit_high_water(batch_id, hw=None):
5062
+ """Return the exact end offset of one completed-batch commit marker.
5063
+
5064
+ The batch was already structurally validated either by the full effective
5065
+ selector or by the live metadata row that names it. The earliest matching
5066
+ commit is the narrowest complete prefix and remains stable even when later
5067
+ journal bytes or crash-replayed duplicate markers exist.
5068
+
5069
+ Streams rather than materializing (#496 S4): the previous form built the
5070
+ whole prefix through `_read_range` before its first-match return, so a
5071
+ marker in the first segment still paid for every later one.
5072
+ """
5073
+ if not batch_id:
5074
+ return None
5075
+ if hw is None:
5076
+ hw = journal_high_water()
3481
5077
  if hw is None:
3482
5078
  return None
3483
5079
  for segment, offset, raw in iter_range(None, hw):
@@ -3492,8 +5088,22 @@ def _correction_commit_high_water(batch_id, hw=None):
3492
5088
  return None
3493
5089
 
3494
5090
 
5091
+ def _has_correction_records(records) -> bool:
5092
+ return any(
5093
+ record.get("t") in {"correction", "correction_batch"}
5094
+ or (
5095
+ record.get("t") == "op"
5096
+ and isinstance(record.get("payload"), dict)
5097
+ and record["payload"].get("kind")
5098
+ == _lib_journal._PROTOCOL_RESOLUTION_KIND
5099
+ )
5100
+ for record in records
5101
+ )
5102
+
5103
+
3495
5104
  def _preflight_live_events(
3496
- conn, records, hw, conflicts=None, protocol_scan=None
5105
+ conn, records, hw, conflicts=None, protocol_scan=None,
5106
+ selector=None, cursor=None, entries=None,
3497
5107
  ):
3498
5108
  """Validate unread evt/correction records before the stats transaction.
3499
5109
 
@@ -3502,7 +5112,22 @@ def _preflight_live_events(
3502
5112
  every cycle over an already-poisoned journal, exactly like the rebuild. The
3503
5113
  divergent evt is DROPPED from the apply set, the prior effective event
3504
5114
  stands, and the group is appended to `conflicts` when a sink is supplied.
3505
- `CorrectionRebuildRequired` stays fatal."""
5115
+ `CorrectionRebuildRequired` stays fatal.
5116
+
5117
+ `selector` is an out-dict (#496 S5b §3.3). When supplied and the durable
5118
+ generation validates, it receives the merged selector rows so the caller can
5119
+ advance them inside its own transaction — the delta and the cursor it
5120
+ describes then commit together. When the generation does not validate the
5121
+ dict is left empty and durable state stays where the last rebuild put it,
5122
+ which is the conservative direction: `stats_index_matches_journal_prefix`
5123
+ then answers False and its one caller rebuilds.
5124
+
5125
+ `cursor` is the caller's applied journal cursor. The durable covered prefix
5126
+ must not be ahead of it; a prefix BEHIND it is re-folded from the journal
5127
+ over exactly that gap. `entries` is positionally parallel to `records` and
5128
+ carries each record's `(segment, end offset)`, from which the incremental
5129
+ step derives the absolute-sequence coordinate map.
5130
+ """
3506
5131
  event_records = [record for record in records if record.get("t") == "evt"]
3507
5132
  selected_new = _lib_journal.resolve_effective_events(event_records)
3508
5133
  if conflicts is not None:
@@ -3552,51 +5177,117 @@ def _preflight_live_events(
3552
5177
  ),
3553
5178
  )
3554
5179
 
3555
- if any(
3556
- record.get("t") in {"correction", "correction_batch"}
3557
- or (
3558
- record.get("t") == "op"
3559
- and isinstance(record.get("payload"), dict)
3560
- and record["payload"].get("kind")
3561
- == _lib_journal._PROTOCOL_RESOLUTION_KIND
3562
- )
3563
- for record in records
3564
- ):
3565
- full = _full_effective_selection(hw)
3566
- if protocol_scan is not None:
3567
- protocol_scan["scanned"] = True
3568
- protocol_scan["violations"] = full.protocol_violations
3569
- protocol_scan["acknowledged"] = (
3570
- full.acknowledged_protocol_violations
3571
- )
3572
- for selected in full.by_id.values():
3573
- if selected.batch_id is None:
5180
+ has_corrections = _has_correction_records(records)
5181
+ incremental = None
5182
+ if selector is not None or has_corrections:
5183
+ incremental = _incremental_selection(conn, records, entries, cursor, hw)
5184
+ if selector is not None and incremental is not None:
5185
+ selector["before"] = incremental["before"]
5186
+ selector["after"] = incremental["after"]
5187
+
5188
+ if incremental is not None:
5189
+ # The incremental path already knows every batch's verdict, so the
5190
+ # whole-prefix read is not performed at all. `protocol_scan` stays unset
5191
+ # deliberately: the merged violation rows are advanced by the selector
5192
+ # delta the caller writes, and running `_write_protocol_violations` on
5193
+ # top of that would replace them with a set derived from a scan that did
5194
+ # not happen.
5195
+ #
5196
+ # This runs even when the DELTA carries no correction record, because a
5197
+ # re-folded gap can: the merged rows are about to be written into
5198
+ # `journal_effective_events`, and installing a corrected winner without
5199
+ # the rebuild the correction demands is exactly what the loop refuses.
5200
+ # Both loops are bounded by the delta's own scope, so the added work on
5201
+ # an ordinary tick is zero rows.
5202
+ for transition in incremental["transitions"]:
5203
+ _raise_taint_transition(transition, incremental["coordinates"])
5204
+ batches = {row.batch_id: row for row in incremental["after"].batches}
5205
+ for row in incremental["after"].effective:
5206
+ if row.batch_id is None:
5207
+ continue
5208
+ prior = _metadata_row(conn, row.event_id)
5209
+ if prior is not None and (
5210
+ int(prior[0]), prior[1], prior[2], prior[3]
5211
+ ) == (row.rev, row.status, row.content_hash, row.batch_id):
3574
5212
  continue
3575
- prior = _metadata_row(conn, selected.event_id)
3576
- if prior is not None:
3577
- prior_tuple = (int(prior[0]), prior[1], prior[2], prior[3])
3578
- selected_tuple = (
3579
- selected.rev,
3580
- selected.status,
3581
- selected.content_hash,
3582
- selected.batch_id,
5213
+ batch = batches.get(row.batch_id)
5214
+ commit = None
5215
+ if batch is not None and batch.earliest_commit_segment is not None:
5216
+ # The durable coordinate, which is what removes
5217
+ # `_correction_commit_high_water`'s traversal from this path.
5218
+ commit = (
5219
+ batch.earliest_commit_segment,
5220
+ int(batch.earliest_commit_offset),
3583
5221
  )
3584
- if prior_tuple == selected_tuple:
3585
- continue
5222
+ if commit is None:
5223
+ # A whole-prefix stream, and a correctness fallback rather than
5224
+ # a fast path: acceptance criteria 6 and 17 ("no whole-journal
5225
+ # read on the interactive ingest path") do NOT hold here.
5226
+ #
5227
+ # It is reached only when no durable batch row carries an
5228
+ # `earliest_commit_*` for this winner, which `marker_coordinates`
5229
+ # and `_carry_coordinates` normally prevent. It is NOT reached
5230
+ # for a winner the delta merely passed through: `_metadata_row`
5231
+ # and `_read_selector_effective_rows` read the same table, so a
5232
+ # passed-through winner compares equal to its stored metadata
5233
+ # above and is skipped before ever arriving here. A winner the
5234
+ # delta RE-DECIDED names a batch the delta itself carried, whose
5235
+ # row this pass is about to write.
5236
+ commit = _correction_commit_high_water(row.batch_id, hw)
3586
5237
  raise CorrectionRebuildRequired(
3587
- f"completed correction batch {selected.batch_id} requires "
5238
+ f"completed correction batch {row.batch_id} requires "
3588
5239
  "stats index rebuild",
3589
- batch_id=selected.batch_id,
3590
- event_id=selected.event_id,
3591
- high_water=_correction_commit_high_water(selected.batch_id, hw),
5240
+ batch_id=row.batch_id,
5241
+ event_id=row.event_id,
5242
+ high_water=commit,
3592
5243
  expected_metadata=(
3593
- selected.rev,
3594
- selected.status,
3595
- selected.content_hash,
3596
- selected.batch_id,
3597
- ),
5244
+ row.rev, row.status, row.content_hash, row.batch_id),
3598
5245
  recovery_eligible=True,
5246
+ kind=CORRECTION_KIND_NEWLY_COMPLETED,
5247
+ )
5248
+ return to_apply
5249
+
5250
+ if not has_corrections:
5251
+ return to_apply
5252
+
5253
+ accumulators: dict = {}
5254
+ full = _full_effective_selection(hw, accumulators)
5255
+ if protocol_scan is not None:
5256
+ protocol_scan["scanned"] = True
5257
+ # The KERNEL's rows, not a second derivation: the rebuild writes these
5258
+ # exact rows, and a fallback that wrote a different shape left the index
5259
+ # unable to match its own journal prefix afterwards.
5260
+ protocol_scan["rows"] = _lib_selector_state.violation_rows(
5261
+ full, accumulators["fold"])
5262
+ for selected in full.by_id.values():
5263
+ if selected.batch_id is None:
5264
+ continue
5265
+ prior = _metadata_row(conn, selected.event_id)
5266
+ if prior is not None:
5267
+ prior_tuple = (int(prior[0]), prior[1], prior[2], prior[3])
5268
+ selected_tuple = (
5269
+ selected.rev,
5270
+ selected.status,
5271
+ selected.content_hash,
5272
+ selected.batch_id,
3599
5273
  )
5274
+ if prior_tuple == selected_tuple:
5275
+ continue
5276
+ raise CorrectionRebuildRequired(
5277
+ f"completed correction batch {selected.batch_id} requires "
5278
+ "stats index rebuild",
5279
+ batch_id=selected.batch_id,
5280
+ event_id=selected.event_id,
5281
+ high_water=_correction_commit_high_water(selected.batch_id, hw),
5282
+ expected_metadata=(
5283
+ selected.rev,
5284
+ selected.status,
5285
+ selected.content_hash,
5286
+ selected.batch_id,
5287
+ ),
5288
+ recovery_eligible=True,
5289
+ kind=CORRECTION_KIND_NEWLY_COMPLETED,
5290
+ )
3600
5291
  return to_apply
3601
5292
 
3602
5293
 
@@ -3620,7 +5311,14 @@ def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
3620
5311
  # advance — any harvested budget evt lands in the freshly-created first
3621
5312
  # segment past the (absent) HW and replays idempotently on the next cycle.
3622
5313
  decoded: list = [] # (record, segment, offset)
5314
+ # End offset per decoded entry, positionally parallel to `decoded`. Kept
5315
+ # separate rather than widened into those tuples because `QUOTA_APPLIER`
5316
+ # consumes them by shape (#496 S5b): the selector needs each record's end
5317
+ # coordinate, and re-encoding a record to recover its length would only be
5318
+ # right for lines this binary wrote.
5319
+ end_offsets: list = []
3623
5320
  malformed = 0
5321
+ cursor = None
3624
5322
  cursor_target = None
3625
5323
  if hw is None:
3626
5324
  if reconcile_config is None and codex_apply is None:
@@ -3639,6 +5337,7 @@ def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
3639
5337
  malformed += 1
3640
5338
  continue
3641
5339
  decoded.append((rec, seg, off))
5340
+ end_offsets.append(off + len(raw) + 1)
3642
5341
 
3643
5342
  # Step 3: cache leg (Codex quota) BEFORE the stats txn (lock-order law).
3644
5343
  # QUOTA_APPLIER attempts the global-then-Codex cache flock NB upsert; on
@@ -3648,11 +5347,23 @@ def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
3648
5347
  # scalar cursor sound).
3649
5348
  cursor_target = (hw_seg, hw_size)
3650
5349
  if QUOTA_APPLIER is not None:
3651
- stop = QUOTA_APPLIER(decoded)
5350
+ # The cycle's own range, so the leg can advance the #496 S5b
5351
+ # coverage certificate over a batch it can prove contiguous. A
5352
+ # prefix-stop advances nothing, because the leg then committed
5353
+ # neither family and `cursor_target` moves to the stop coordinate.
5354
+ # `decoded_end` is the last DECODED record's end coordinate, which
5355
+ # bounds the covered claim below the raw cursor target whenever the
5356
+ # traversal stopped short of it — a torn or malformed trailing line.
5357
+ stop = QUOTA_APPLIER(
5358
+ decoded, cursor=cursor, covered_to=cursor_target,
5359
+ decoded_end=(
5360
+ None if not decoded
5361
+ else (decoded[-1][1], end_offsets[-1])))
3652
5362
  if stop is not None:
3653
5363
  _rec, stop_seg, stop_off = decoded[stop]
3654
5364
  cursor_target = (stop_seg, stop_off)
3655
5365
  decoded = decoded[:stop]
5366
+ end_offsets = end_offsets[:stop]
3656
5367
 
3657
5368
  records = [r for (r, _s, _o) in decoded]
3658
5369
  batch = [r for r in records if r.get("t") in ("obs", "op")]
@@ -3660,12 +5371,27 @@ def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
3660
5371
  # raising; the groups it drops are counted on the cycle summary.
3661
5372
  preflight_conflicts: list = []
3662
5373
  protocol_scan: dict = {}
5374
+ # #496 S5b §3.3: the delta's own sequence numbering starts where the durable
5375
+ # prefix stopped, and every decoded entry consumes one number — the same
5376
+ # numbering the rebuild produces by appending a placeholder for each valid
5377
+ # non-retained record. `_incremental_selection` owns that arithmetic: it
5378
+ # rejects a durable prefix AHEAD of `cursor`, and re-folds the gap when the
5379
+ # prefix is behind it, so an absent or stale state cannot make these numbers
5380
+ # mean something else.
5381
+ selector_state: dict = {}
5382
+ selector_entries = [
5383
+ (segment, end_offsets[index])
5384
+ for index, (_rec, segment, _off) in enumerate(decoded)
5385
+ ]
3663
5386
  journal_evts = _preflight_live_events(
3664
5387
  conn,
3665
5388
  records,
3666
5389
  cursor_target,
3667
5390
  conflicts=preflight_conflicts,
3668
5391
  protocol_scan=protocol_scan,
5392
+ selector=selector_state,
5393
+ cursor=cursor,
5394
+ entries=selector_entries,
3669
5395
  )
3670
5396
 
3671
5397
  # Step 4: ONE BEGIN IMMEDIATE — replay + pipeline + derived-fact journaling +
@@ -3732,11 +5458,18 @@ def _run_cycle(conn: sqlite3.Connection, *, reconcile_config=None,
3732
5458
  # the cursor so shallow Doctor paths observe either the old complete
3733
5459
  # result or the new complete result, never an in-between state.
3734
5460
  if protocol_scan.get("scanned"):
3735
- _write_protocol_violations(
3736
- conn,
3737
- protocol_scan.get("violations", ()),
3738
- protocol_scan.get("acknowledged", ()),
3739
- )
5461
+ _replace_protocol_violations(conn, protocol_scan.get("rows", ()))
5462
+ # 4c''. Advance durable selector state (#496 S5b §3.3), in the SAME
5463
+ # transaction as the cursor it describes, so the durable prefix equals
5464
+ # the applied journal cursor at every commit and a crash rolls both back
5465
+ # together. It runs AFTER step 4a: that step inserts the plain
5466
+ # six-column effective row for a newly applied evt, and the delta
5467
+ # replaces it with the eight-column row carrying the winning sequence.
5468
+ # Absent when the generation did not validate, which leaves durable
5469
+ # state where the last rebuild put it.
5470
+ if selector_state.get("after") is not None:
5471
+ _write_selector_delta(
5472
+ conn, selector_state["before"], selector_state["after"])
3740
5473
  # 4d. Advance the cursor (to HW, or to the cache-leg prefix boundary).
3741
5474
  # `cursor_target is None` ONLY on a reconcile-only cycle over a still-
3742
5475
  # empty journal (§5.2 above): there are no consumed lines to advance
@@ -3974,8 +5707,48 @@ def _correction_error_result(error) -> IngestResult:
3974
5707
  )
3975
5708
 
3976
5709
 
5710
+ def _tainted_batch_converged(signal: CorrectionRebuildRequired) -> bool:
5711
+ """Convergence for a completed-to-tainted signal (#496 S5b §3.7).
5712
+
5713
+ `_recover_completed_correction`'s existing predicate needs an exact
5714
+ `(rev, status, content_hash, batch_id)`, and after a taint withdraws a
5715
+ completed batch the post-rebuild winner may be an OLDER candidate that
5716
+ durable selector state deliberately does not store — §3.2 keeps no losing
5717
+ candidates. So this kind converges on state that is available: the selector
5718
+ batch is tainted, selector coverage includes the causal offset, and no
5719
+ effective winner names that batch.
5720
+ """
5721
+ if signal.batch_id is None or signal.high_water is None:
5722
+ return False
5723
+ path = pathlib.Path(_cctally_core.DB_PATH)
5724
+ if not path.exists():
5725
+ return False
5726
+ conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True, timeout=5.0)
5727
+ try:
5728
+ rows = _read_selector_rows(conn)
5729
+ except sqlite3.Error:
5730
+ return False
5731
+ finally:
5732
+ conn.close()
5733
+ if rows is None:
5734
+ return False
5735
+ batch = {row.batch_id: row for row in rows.batches}.get(signal.batch_id)
5736
+ if batch is None or batch.status != "tainted":
5737
+ return False
5738
+ if not _coordinate_covers(
5739
+ (rows.state.covered_segment, rows.state.covered_offset),
5740
+ signal.high_water,
5741
+ ):
5742
+ return False
5743
+ return not any(
5744
+ row.batch_id == signal.batch_id for row in rows.effective
5745
+ )
5746
+
5747
+
3977
5748
  def _correction_index_converged(signal: CorrectionRebuildRequired) -> bool:
3978
5749
  """Revalidate the triggering effective row without open-time mutation."""
5750
+ if signal.kind == CORRECTION_KIND_COMPLETED_TO_TAINTED:
5751
+ return _tainted_batch_converged(signal)
3979
5752
  if signal.event_id is None or signal.expected_metadata is None:
3980
5753
  return False
3981
5754
  path = pathlib.Path(_cctally_core.DB_PATH)
@@ -4025,7 +5798,20 @@ def _recover_completed_correction(
4025
5798
  timeout_s: float,
4026
5799
  ) -> None:
4027
5800
  """Revalidate and, when still needed, replace through the trigger prefix."""
4028
- if (
5801
+ if signal.kind == CORRECTION_KIND_COMPLETED_TO_TAINTED:
5802
+ # A different contract, deliberately: this kind carries no `event_id`
5803
+ # and no expected metadata tuple, because the post-taint winner may be a
5804
+ # candidate the durable state does not store. The causal offset is
5805
+ # mandatory and there is nothing to substitute for it, so an absent one
5806
+ # is refused rather than widened to the pinned high-water.
5807
+ if signal.batch_id is None or signal.high_water is None:
5808
+ raise CorrectionRecoveryError(
5809
+ _correction_recovery_guidance(
5810
+ "completed-to-tainted correction lacks the causal record "
5811
+ "offset its rebuild must include"
5812
+ )
5813
+ )
5814
+ elif (
4029
5815
  signal.batch_id is None
4030
5816
  or signal.event_id is None
4031
5817
  or signal.high_water is None
@@ -4142,6 +5928,7 @@ def run_stats_ingest(
4142
5928
  high_water=signal.high_water,
4143
5929
  expected_metadata=signal.expected_metadata,
4144
5930
  recovery_eligible=True,
5931
+ kind=signal.kind,
4145
5932
  ) from signal
4146
5933
 
4147
5934
  try:
@@ -4290,7 +6077,12 @@ class RebuildResult:
4290
6077
  rows_by_table: dict # journal-covered table -> row count in the rebuild
4291
6078
  malformed: int # journal lines that failed to decode (spec §4.4)
4292
6079
  duration_s: float # wall time of the whole rebuild
4293
- segments_read: int # journal segments folded
6080
+ # Segments in the pinned prefix, which is NOT the same as segments this
6081
+ # pass opened: #496 S5b's elision skips some of them and contributes their
6082
+ # stored summary instead. `traversal["elision"]["scannedSegments"]` is the
6083
+ # opened count. The value here is unchanged from before elision existed, so
6084
+ # the public `segmentsRead` key keeps meaning what it always meant.
6085
+ segments_read: int # journal segments in the pinned prefix
4294
6086
  lines_folded: int # op + evt lines applied (obs are rederive input)
4295
6087
  # #374: divergent same-revision groups quarantined behind a lowest-sequence
4296
6088
  # provisional winner. The rebuild COMPLETES and exits 0 — reporting them is
@@ -4323,6 +6115,32 @@ class RebuildResult:
4323
6115
  peak_heap_bytes: int = 0
4324
6116
  #: cache writer flock acquisition to release, in seconds.
4325
6117
  quota_lock_hold_seconds: float = 0.0
6118
+ #: #496 S5b — the durable selector prefix of the index this rebuild is about
6119
+ #: to REPLACE, when that prefix was behind the index's own applied journal
6120
+ #: cursor. `None` when the two agreed, when the destination did not exist, or
6121
+ #: when either could not be read. A tick that cannot validate the durable
6122
+ #: generation advances the cursor without advancing the selector, and the
6123
+ #: next tick re-folds the gap silently (§6.3); this is the only surface on
6124
+ #: which a persistent desynchronization is reported.
6125
+ selector_desynchronized: "dict | None" = None
6126
+ #: #496 S5b F11 — what the quota cache leg did: `status` in
6127
+ #: `{skipped, covered, recovered, failed}`, `reason` naming the coverage
6128
+ #: verdict, `coveredHighWater` and `replayedObservations`. `covered` is the
6129
+ #: intact path, where the leg takes no cache writer flock and replays
6130
+ #: nothing. Additive to the `schemaVersion: 1` rebuild record.
6131
+ quota_cache_coverage: dict = field(default_factory=dict)
6132
+ #: #496 S5b §4.7 — TRUE when this generation's stats quota projection was
6133
+ #: materialized from a cache whose recovery left an uncovered remainder.
6134
+ #:
6135
+ #: It is a separate field from `quota_cache_coverage` because the two answer
6136
+ #: different questions and a consumer must not read one as the other:
6137
+ #: coverage describes the CACHE, this describes the PUBLISHED INDEX. The
6138
+ #: quota projection is materialized FROM `cache.db`, so a partial cache
6139
+ #: produces a semantically partial projection inside the generation being
6140
+ #: published, and completing cache recovery later does not by itself
6141
+ #: reconcile that projection. `RebuildResult` still has no success or
6142
+ #: failure boolean, and this is not one — publication proceeds either way.
6143
+ stats_quota_projection_incomplete: bool = False
4326
6144
 
4327
6145
 
4328
6146
  def _remove_db_sidecars_strict(path) -> None:
@@ -4367,6 +6185,9 @@ _REBUILD_REQUIRED_TABLES = frozenset(
4367
6185
  "journal_cursor",
4368
6186
  "journal_effective_events",
4369
6187
  "journal_protocol_violations",
6188
+ "journal_selector_batch_records",
6189
+ "journal_selector_batches",
6190
+ "journal_selector_state",
4370
6191
  "percent_milestones",
4371
6192
  "project_budget_milestones",
4372
6193
  "projected_milestones",
@@ -4380,6 +6201,7 @@ _REBUILD_REQUIRED_TABLES = frozenset(
4380
6201
  "schema_migrations_skipped",
4381
6202
  "stats_open_fixups",
4382
6203
  "stats_publication_stamp",
6204
+ "stats_quota_projection_state",
4383
6205
  "week_reset_events",
4384
6206
  "weekly_cost_snapshots",
4385
6207
  "weekly_credit_floors",
@@ -4404,6 +6226,7 @@ _REBUILD_REQUIRED_INDEXES = frozenset(
4404
6226
  "idx_five_hour_milestones_journal_id_null",
4405
6227
  "idx_five_hour_reset_events_journal_id",
4406
6228
  "idx_five_hour_reset_events_journal_id_null",
6229
+ "idx_journal_protocol_violations_batch",
4407
6230
  "idx_percent_milestones_journal_id",
4408
6231
  "idx_percent_milestones_journal_id_null",
4409
6232
  "idx_project_budget_milestones_journal_id",
@@ -4428,7 +6251,7 @@ _REBUILD_REQUIRED_INDEXES = frozenset(
4428
6251
  # omitted column, constraint, partial predicate, or index definition. An epoch
4429
6252
  # schema change must update this contract alongside STATS_INDEX_EPOCH.
4430
6253
  _REBUILD_SCHEMA_FINGERPRINT = (
4431
- "7dde5a7995f441558d08b0204136824d6ff7208b221e576c79a76854b76aa178"
6254
+ "2b378fc3be1c7bb249bb0c3ddd2111a802f689cf30b3fd42806a116611c799e6"
4432
6255
  )
4433
6256
 
4434
6257
 
@@ -4529,6 +6352,31 @@ def _validate_rebuilt_stats_index(
4529
6352
  )
4530
6353
 
4531
6354
 
6355
+ def _validate_selector_state(conn, expected) -> None:
6356
+ """Refuse an index whose durable selector state is not what the journal says.
6357
+
6358
+ ``expected`` is the kernel rows a full selection over the same pinned
6359
+ traversal produced. Raises `JournalError`; `stats_index_matches_journal_
6360
+ prefix` catches that and answers False, while the rebuild lets it abort the
6361
+ publication.
6362
+
6363
+ Single-row cardinality is part of the contract, so a zero-row state fails
6364
+ here rather than reading as an empty-but-valid generation.
6365
+ """
6366
+ stored = _read_selector_rows(conn)
6367
+ if stored is None:
6368
+ raise JournalError(
6369
+ "durable selector state is absent or not a single row"
6370
+ )
6371
+ if _lib_selector_state.comparable(stored) != _lib_selector_state.comparable(
6372
+ expected
6373
+ ):
6374
+ raise JournalError(
6375
+ "durable selector state does not match the journal selection "
6376
+ "derived from the same pinned prefix"
6377
+ )
6378
+
6379
+
4532
6380
  def stats_index_matches_journal_prefix(
4533
6381
  path: pathlib.Path, high_water: "tuple[str, int] | None"
4534
6382
  ) -> bool:
@@ -4543,6 +6391,26 @@ def stats_index_matches_journal_prefix(
4543
6391
  if not pathlib.Path(path).exists():
4544
6392
  return False
4545
6393
  try:
6394
+ all_segments = list_segments()
6395
+ segments = all_segments
6396
+ elision = None
6397
+ if high_water is not None:
6398
+ if high_water[0] in segments:
6399
+ segments = segments[:segments.index(high_water[0]) + 1]
6400
+ # THE SAME planner the rebuild uses, constructed the same way from
6401
+ # the same inputs (#496 S5b §5.6). An eliding rebuild and a prefix
6402
+ # validation that disagreed about the same journal would make the
6403
+ # validation meaningless.
6404
+ #
6405
+ # Constructed BEFORE the `stats.db` connection, deliberately: the
6406
+ # planner reads `cache.db`, and the lock-order law runs cache before
6407
+ # stats. Building it inside the stats connection's lifetime would
6408
+ # open the cache underneath an already-open stats reader. Neither
6409
+ # read takes a flock and Python's sqlite3 holds no implicit read
6410
+ # transaction across `fetchall()`, so nothing deadlocks today — but
6411
+ # the law is a TOTAL order, and the cost of keeping it is one short
6412
+ # cache read on a path that was about to fail structurally.
6413
+ elision = plan_segment_elision(segments, high_water)
4546
6414
  conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
4547
6415
  try:
4548
6416
  _validate_rebuilt_stats_index(conn, high_water)
@@ -4557,17 +6425,28 @@ def stats_index_matches_journal_prefix(
4557
6425
  protocol_evidence = []
4558
6426
  prior_high_water = None
4559
6427
  cutover_captured = _CUTOVER_UNSEEN
4560
- all_segments = list_segments()
4561
- segments = all_segments
6428
+ marker_coordinates: dict = {}
4562
6429
  hasher = _lib_journal_router.PrefixHashAccumulator()
4563
6430
  if high_water is not None:
4564
- if high_water[0] in segments:
4565
- segments = segments[:segments.index(high_water[0]) + 1]
6431
+ def _elide_segment(name, lo, hi, stat_result) -> bool:
6432
+ nonlocal prior_high_water
6433
+ summary = elision.decide(name, hi, stat_result)
6434
+ if summary is None:
6435
+ return False
6436
+ # Only the placeholders and the boundary: this pass builds
6437
+ # no counters, retains no observation bytes and folds no
6438
+ # last-seen map, so the `decoded`-entry count is the whole
6439
+ # contribution.
6440
+ decoded.extend([None] * int(summary.decoded_entry_count))
6441
+ prior_high_water = (name, int(summary.summarized_size))
6442
+ return True
6443
+
4566
6444
  for segment, offset, raw in _iter_range_with_segments(
4567
6445
  None, high_water, segments,
4568
6446
  on_segment=lambda name: hasher.begin_segment(
4569
6447
  name, prior_high_water),
4570
6448
  on_bytes=hasher.extend,
6449
+ elide=_elide_segment,
4571
6450
  ):
4572
6451
  record = _lib_journal.decode_line(raw)
4573
6452
  if record is not None:
@@ -4575,11 +6454,23 @@ def stats_index_matches_journal_prefix(
4575
6454
  record,
4576
6455
  prior_high_water,
4577
6456
  protocol_evidence,
4578
- hasher=hasher,
6457
+ # See the rebuild's twin: a digest cannot be composed
6458
+ # over an elided prefix, so it is recomputed from
6459
+ # disk instead (§5.1).
6460
+ hasher=None if elision.elided else hasher,
4579
6461
  )
6462
+ if (isinstance(record.get("payload"), dict)
6463
+ and record["payload"].get("kind")
6464
+ == _lib_journal._PROTOCOL_RESOLUTION_KIND):
6465
+ elision.resolution_seen = True
4580
6466
  if (cutover_captured is _CUTOVER_UNSEEN
4581
6467
  and record.get("id") == CUTOVER_OP_ID):
4582
6468
  cutover_captured = _cutover_value_of(record)
6469
+ if record.get("t") == "correction_batch":
6470
+ marker_coordinates[len(decoded)] = (
6471
+ segment,
6472
+ offset + len(raw) + 1,
6473
+ )
4583
6474
  decoded.append(
4584
6475
  record
4585
6476
  if record.get("t")
@@ -4596,9 +6487,30 @@ def stats_index_matches_journal_prefix(
4596
6487
  for record in decoded:
4597
6488
  if record is not None:
4598
6489
  _normalize_legacy_account_stamp(record, cutover_claude)
6490
+ selector_accumulators: dict = {}
4599
6491
  selection = _lib_journal.resolve_effective_events(
4600
6492
  decoded,
4601
6493
  protocol_prefix_evidence=protocol_evidence,
6494
+ accumulators=selector_accumulators,
6495
+ )
6496
+ # The same semantic half the rebuild's oracle runs (#496 S5b §6.2).
6497
+ # Both planners derive it from THEIR OWN pinned traversal, so an
6498
+ # eliding rebuild and a prefix validation cannot reach different
6499
+ # answers about the same journal.
6500
+ _validate_selector_state(
6501
+ conn,
6502
+ _lib_selector_state.rows_from_selection(
6503
+ selection,
6504
+ accumulators=selector_accumulators,
6505
+ next_sequence=len(decoded),
6506
+ coordinates=marker_coordinates,
6507
+ covered=high_water,
6508
+ cutover_seen=cutover_captured is not _CUTOVER_UNSEEN,
6509
+ cutover_account_key=(
6510
+ None if cutover_captured is _CUTOVER_UNSEEN
6511
+ else cutover_captured
6512
+ ),
6513
+ ),
4602
6514
  )
4603
6515
  expected = []
4604
6516
  for event_id, selected in selection.by_id.items():
@@ -4701,6 +6613,27 @@ def _prepare_existing_stats_for_cutover(path: pathlib.Path) -> str:
4701
6613
  replacement main file.
4702
6614
  """
4703
6615
  import _cctally_db
6616
+ import _lib_stats_wal
6617
+
6618
+ wal_index = _lib_stats_wal.inspect_wal_index_family(path)
6619
+ wal_verdict = wal_index.get("verdict")
6620
+ if _lib_stats_wal.is_incoherent_wal_index(wal_index):
6621
+ # The caller has already proved whole-family drain and, for a heal,
6622
+ # preserved the complete pre-checkpoint family. Opening SQLite here
6623
+ # would let a stale aPgno[] map direct valid WAL frames to wrong main
6624
+ # pages before quarantine records the original bytes.
6625
+ return "skipped_incoherent_wal_index"
6626
+ if wal_verdict in {"capture_raced", "analysis_truncated"}:
6627
+ raise JournalError(
6628
+ "old stats index WAL/SHM coherence could not be established "
6629
+ f"before cutover ({wal_verdict})"
6630
+ )
6631
+ if wal_verdict not in {"coherent", "wal_absent", "wal_empty"}:
6632
+ # A malformed/non-empty WAL, missing SHM, or another unrecognized raw
6633
+ # shape is not permission to let SQLite reconstruct or checkpoint it.
6634
+ # The caller has already preserved the complete family and can publish
6635
+ # the independently rebuilt index without mutating these old bytes.
6636
+ return "skipped_unproven_wal_index"
4704
6637
 
4705
6638
  try:
4706
6639
  conn = sqlite3.connect(str(path), timeout=15.0)
@@ -5189,6 +7122,7 @@ def _preserve_stats_family_for_cutover(
5189
7122
  "rebuildRecordPath": context.record_path,
5190
7123
  "binaryVersion": _binary_version(),
5191
7124
  "binaryEpoch": _cctally_core.STATS_INDEX_EPOCH,
7125
+ "sqliteRuntimeVersion": sqlite3.sqlite_version,
5192
7126
  "preservedUserVersion": preserved_user_version,
5193
7127
  "familySizes": family_sizes,
5194
7128
  # The retained COPY is described, not the live file, because the copy
@@ -5377,12 +7311,40 @@ def _publish_generation_in_place(
5377
7311
  conn.execute(f"PRAGMA main.user_version={epoch:d}")
5378
7312
  # The publication's own identity, committed atomically with the
5379
7313
  # content and the epoch it describes (#496 S3 §5).
7314
+ stamped_at = _utc_iso_now()
5380
7315
  conn.execute("DELETE FROM main.stats_publication_stamp")
5381
7316
  conn.execute(
5382
7317
  "INSERT INTO main.stats_publication_stamp "
5383
7318
  "(record_path, started_at_utc, stamped_at_utc) VALUES (?, ?, ?)",
5384
- (str(record_path), started_at, _utc_iso_now()),
7319
+ (str(record_path), started_at, stamped_at),
5385
7320
  )
7321
+ # Spec §7 case 5 asks for a failure BETWEEN the publication-stamp
7322
+ # write and the selector-identity write, so the seam sits here
7323
+ # rather than after both. An injection after both still established
7324
+ # the property, but it exercised a different interleaving than the
7325
+ # one the spec names.
7326
+ _stats_rebuild_test_pause("publication_before_commit")
7327
+ # The SAME identity onto the durable selector row, in the SAME
7328
+ # transaction (#496 S5b §3.4). The scratch was built before this
7329
+ # stamp existed, so a row populated there cannot carry the identity
7330
+ # it will publish under; writing it here is what lets a live tick
7331
+ # prove its durable state belongs to the generation it is reading.
7332
+ #
7333
+ # The existence probe is not tolerance for a missing table on a real
7334
+ # generation — `_REBUILD_REQUIRED_TABLES` makes validation refuse a
7335
+ # scratch that lacks it. This function publishes whatever schema its
7336
+ # scratch carries, and the publication-protocol tests build minimal
7337
+ # synthetic generations that legitimately have no selector state.
7338
+ if conn.execute(
7339
+ "SELECT 1 FROM main.sqlite_schema WHERE type='table' "
7340
+ "AND name='journal_selector_state'"
7341
+ ).fetchone() is not None:
7342
+ conn.execute(
7343
+ "UPDATE main.journal_selector_state "
7344
+ "SET generation_record_path = ?, "
7345
+ "generation_stamped_at_utc = ?",
7346
+ (str(record_path), stamped_at),
7347
+ )
5386
7348
  phase = sp.COMMIT_UNKNOWN
5387
7349
  conn.commit()
5388
7350
  phase = sp.COMMITTED
@@ -5426,30 +7388,33 @@ def _checkpoint_after_publication(conn: sqlite3.Connection) -> str:
5426
7388
  return "checkpointed" if int(row[0]) == 0 else "busy"
5427
7389
 
5428
7390
 
5429
- def _remove_empty_db_sidecars(path) -> None:
5430
- """Remove the sidecars a read-only validation open leaves behind.
5431
-
5432
- Unlike the physical path, an in-place publish leaves a LEGITIMATE WAL: it
5433
- belongs to the live database, and deleting a non-empty one would discard
5434
- committed frames. Only a zero-length WAL is removed the state a
5435
- successful TRUNCATE checkpoint and a clean last close leave so the
5436
- documented sidecar-free end state still holds without ever risking data.
5437
- """
5438
- path = pathlib.Path(path)
5439
- wal = pathlib.Path(str(path) + "-wal")
5440
- try:
5441
- if wal.stat().st_size:
5442
- return
5443
- except OSError:
5444
- pass
5445
- for suffix in ("-wal", "-shm"):
5446
- try:
5447
- pathlib.Path(str(path) + suffix).unlink()
5448
- except FileNotFoundError:
5449
- pass
5450
- except OSError:
5451
- return
5452
- _fsync_dir(path.parent)
7391
+ # An in-place publication used to unlink the live `-wal` and `-shm` here once
7392
+ # the TRUNCATE checkpoint had emptied the WAL, to reach a sidecar-free end
7393
+ # state. That is issue #516 and the call is gone: unlinking the sidecars of a
7394
+ # database other connections still hold open is outside SQLite's contract, and
7395
+ # what it cost is a data-correctness fault and a crash, NOT a cosmetic end
7396
+ # state. Two conditions have to combine, and an earlier thirteen-arrangement
7397
+ # run never combined them: something must WRITE after the unlink, and the
7398
+ # reader must already have READ before it. Re-measured on both LAN runners
7399
+ # (macOS, Python 3.13.14, SQLite 3.53.4, byte-identical on each):
7400
+ #
7401
+ # - later writer in the SAME process — a reader that had read before the
7402
+ # unlink, read-write or `mode=ro`, raises `OperationalError: disk I/O
7403
+ # error`; and a connection that keeps writing through the unlinked inodes
7404
+ # breaks the NEXT connection opened in that process the same way;
7405
+ # - later writer in a CHILD process — that reader silently reads a STALE
7406
+ # generation, and so does a freshly opened connection in the parent;
7407
+ # - a reader that had NOT read before the unlink reads the current value,
7408
+ # which is why the earlier run observed nothing;
7409
+ # - a reader holding a pinned read transaction makes the checkpoint busy, so
7410
+ # the unlink is refused before it can happen.
7411
+ #
7412
+ # One mechanism seen from three ends: whichever connection holds an fd on the
7413
+ # stale `-wal` inode while the shared wal-index describes frames in the other
7414
+ # `-wal` inode takes the short read. A clean last close removes both sidecars
7415
+ # by itself, so nothing is leaked; only the case where a handle is open, which
7416
+ # is exactly the case that must not be unlinked, now leaves a zero-length WAL
7417
+ # behind.
5453
7418
 
5454
7419
 
5455
7420
  #: Probe 2 measured the live WAL peaking at 1.01x the main file across five
@@ -5651,7 +7616,6 @@ def _publish_stats_index_in_place(
5651
7616
  post_error = validate_published_stats_index(
5652
7617
  destination, high_water, expected_record_path=str(record_path),
5653
7618
  )
5654
- _remove_empty_db_sidecars(destination)
5655
7619
 
5656
7620
  live["publicationCheckpoint"] = checkpoint_outcome
5657
7621
  live["postPublicationValidation"] = {
@@ -5878,97 +7842,990 @@ def _decoded_quota_stream(quota_raw, cutover_claude, counters=None):
5878
7842
  yield record
5879
7843
 
5880
7844
 
5881
- def _rebuild_quota_cache_leg_raw(
5882
- quota_raw, decoded, cutover_claude, counters=None
5883
- ) -> float:
5884
- """Re-materialize cache.db `quota_window_snapshots` AND the #416 Codex
5885
- attribution map from the journal (spec §5.4 + #416 spec §3.4), fed RAW
5886
- ENCODED LINES for the observations instead of decoded dicts.
7845
+ #: Every field the rebuild's F11 fast path needs, captured from ONE read-only
7846
+ #: snapshot so the certificate and the sequence it is validated against cannot
7847
+ #: come from two different cache states.
7848
+ #:
7849
+ #: ``conn`` is that snapshot's connection, and it is returned still inside its
7850
+ #: read transaction. Under WAL a read transaction's snapshot is fixed at its
7851
+ #: first read, so a caller that keeps it open sees the SAME cache state for
7852
+ #: every later read — which is what lets §4.4's projection bundle be captured
7853
+ #: from the snapshot the coverage verdict was decided against, rather than from
7854
+ #: a second connection opened later.
7855
+ _CoverageSnapshot = collections.namedtuple(
7856
+ "_CoverageSnapshot", ("certificate", "physical_seq", "conn"))
5887
7857
 
5888
- The journal records are the DURABLE source (§1 latent data-loss hole — the
5889
- rollout JSONL evaporates); this INSERT-OR-IGNOREs the quota obs on their
5890
- natural key and the attribution decisions on theirs, mirroring
5891
- `_cache_applier` family for family. The map half is not optional: without it
5892
- a rebuild would leave the map empty and the following rollout walk would
5893
- have nothing to replay, sending every file back to the live `auth.json` —
5894
- the exact defect this mechanism exists to prevent.
5895
7858
 
5896
- Runs BEFORE any stats transaction, under the global cache writer lock
5897
- followed by the `cache.db.codex.lock` provider flock (lock-order law).
5898
- Best-effort: a missing/busy cache.db is a clean skip (the records stay
5899
- durable in the journal; the stats quota projection pass then degrades
5900
- cleanly).
7859
+ def _close_coverage_snapshot(snapshot) -> None:
7860
+ """End the snapshot's read transaction and close it. Never raises."""
7861
+ if snapshot is None or snapshot.conn is None:
7862
+ return
7863
+ try:
7864
+ snapshot.conn.rollback()
7865
+ except sqlite3.Error:
7866
+ pass
7867
+ try:
7868
+ snapshot.conn.close()
7869
+ except sqlite3.Error:
7870
+ pass
5901
7871
 
5902
- Taking raw bytes is what makes the rebuild affordable. 1.81M decoded
5903
- observation dictionaries are roughly six gigabytes against 1.64 GB of raw
5904
- bytes, so retaining bytes removes about four gigabytes of peak heap while
5905
- adding NO file input and NO second traversal — only the JSON decode of
5906
- records already in memory moves inside the flocks (#496 S4 §6.3). That
5907
- decode is why the measured hold is longer than it was before S4; see the
5908
- spec's §6.3 for the measured figures.
5909
7872
 
5910
- Ordering is preserved: file-account decisions are ops, so the router already
5911
- retains them decoded and they are available before the observation loop
5912
- begins, exactly as the §3.5 precedence rule requires.
7873
+ def _read_coverage_snapshot(cache_path) -> "_CoverageSnapshot | None":
7874
+ """The stored coverage certificate and physical sequence, in one `BEGIN`.
5913
7875
 
5914
- Returns the measured flock hold in seconds, which acceptance criterion 7
5915
- caps.
5916
- """
5917
- file_accounts = [
5918
- r for r in decoded if r is not None and _is_codex_file_account_op(r)
5919
- ]
5920
- if not quota_raw and not file_accounts:
5921
- return 0.0
5922
- cache_path = _cctally_core.CACHE_DB_PATH
5923
- if not cache_path.exists():
5924
- return 0.0
7876
+ Read-only and WAL, so it takes NO writer flock and blocks no writer which
7877
+ is the whole point of the fast path it feeds. Two separate reads could see
7878
+ a certificate from before a writer's transaction and a sequence from after
7879
+ it, which would validate a certificate the writer had just superseded.
7880
+
7881
+ **The transaction is left OPEN and the caller owns closing it.** Committing
7882
+ here would release the snapshot, and §4.4 requires the source roots, the
7883
+ observations and the ledger state to come from the same one. Every caller
7884
+ goes through `_close_coverage_snapshot`.
7885
+
7886
+ Any failure answers None, and None means replay. Every degraded state here
7887
+ falls back silently (spec §6.3).
7888
+ """
7889
+ import _cctally_cache
7890
+ try:
7891
+ conn = sqlite3.connect(
7892
+ f"file:{cache_path}?mode=ro", uri=True, timeout=5.0)
7893
+ except sqlite3.Error:
7894
+ return None
7895
+ try:
7896
+ conn.execute("PRAGMA busy_timeout=5000")
7897
+ conn.execute("BEGIN")
7898
+ certificate = _cctally_cache.load_codex_journal_coverage_certificate(conn)
7899
+ row = conn.execute(
7900
+ "SELECT value FROM cache_meta "
7901
+ "WHERE key='codex_physical_mutation_seq'"
7902
+ ).fetchone()
7903
+ except sqlite3.Error:
7904
+ _close_coverage_snapshot(_CoverageSnapshot(None, 0, conn))
7905
+ return None
7906
+ try:
7907
+ physical_seq = 0 if row is None or row[0] is None else int(row[0])
7908
+ except (TypeError, ValueError):
7909
+ _close_coverage_snapshot(_CoverageSnapshot(None, 0, conn))
7910
+ return None
7911
+ return _CoverageSnapshot(certificate, physical_seq, conn)
7912
+
7913
+
7914
+ def recover_quota_cache_from_journal(high_water=None, *, quiet=False) -> dict:
7915
+ """Replay the journal's Codex quota records into `cache.db`, no stats work.
7916
+
7917
+ This is the cache half of a rebuild, reachable on its own, which is what
7918
+ lets §4.7's open-time reconciliation resume a recovery a previous rebuild
7919
+ could not finish without re-publishing a whole generation. It reads the
7920
+ journal once, retains the observations as RAW BYTES for the same reason the
7921
+ rebuild does, and hands both populations to the same bounded, revalidating
7922
+ leg — so the caps, the per-chunk lock release, the restart rules and the
7923
+ mint are literally the same code, not a second implementation of them.
7924
+
7925
+ ``quiet`` suppresses the leg's `[rebuild]`-prefixed stderr. This function is
7926
+ reachable from an ordinary command, where acceptance criterion 10 requires
7927
+ no new stderr line and where a `[rebuild]` prefix would be a lie about which
7928
+ operation produced it. The rebuild itself passes the default.
7929
+
7930
+ Returns the leg's coverage record. `complete` is what the caller gates on.
7931
+ """
7932
+ hw = high_water if high_water is not None else journal_high_water()
7933
+ coverage: dict = {
7934
+ "status": "skipped", "reason": None, "replayedObservations": 0,
7935
+ "complete": True, "remainder": None,
7936
+ }
7937
+ if hw is None:
7938
+ return coverage
7939
+ segments = list_segments()
7940
+ if hw[0] not in segments:
7941
+ coverage.update({"status": "incomplete", "complete": False,
7942
+ "remainder": {"reason": "missingSegment"}})
7943
+ return coverage
7944
+ segments = segments[:segments.index(hw[0]) + 1]
7945
+
7946
+ quota_raw: list = []
7947
+ file_ops: list = []
7948
+ cutover_captured = _CUTOVER_UNSEEN
7949
+ # Advances for a malformed line too — consumed but not decoded — exactly as
7950
+ # the rebuild's `prior_high_water` does, because `_bounded_covered_offset`
7951
+ # is documented against the last line the pass CONSUMED.
7952
+ consumed_end = None
7953
+ for segment, offset, raw in _iter_range_with_segments(None, hw, segments):
7954
+ consumed_end = (segment, offset + len(raw) + 1)
7955
+ rec = _lib_journal.decode_line(raw)
7956
+ if rec is None:
7957
+ continue
7958
+ if (cutover_captured is _CUTOVER_UNSEEN
7959
+ and rec.get("id") == CUTOVER_OP_ID):
7960
+ cutover_captured = _cutover_value_of(rec)
7961
+ if _is_codex_file_account_op(rec):
7962
+ file_ops.append(rec)
7963
+ elif _is_codex_quota_obs(rec):
7964
+ quota_raw.append(raw)
7965
+ cutover_claude = _resolve_cutover_for_rebuild(
7966
+ cutover_captured, hw, segments)
7967
+ _rebuild_quota_cache_leg_raw(
7968
+ quota_raw, file_ops, cutover_claude, None,
7969
+ high_water=hw, coverage=coverage, decoded_end=consumed_end, quiet=quiet)
7970
+ # An absent cache.db is a clean skip for the REBUILD — the records stay
7971
+ # durable in the journal and a later `cache-sync` re-materializes them — but
7972
+ # it is not a completed recovery. The leg's `complete: True` describes "no
7973
+ # duty", and this caller's gate reads it as "the cache reached its target",
7974
+ # which for an absent cache with records to replay it certainly has not. The
7975
+ # distinction is made here rather than in the leg so the rebuild's
7976
+ # documented missing-cache behaviour is unchanged.
7977
+ if (coverage.get("complete") is True
7978
+ and (quota_raw or file_ops)
7979
+ and not _cctally_core.CACHE_DB_PATH.exists()):
7980
+ coverage.update({
7981
+ "status": "incomplete", "complete": False,
7982
+ "remainder": {
7983
+ "observations": len(quota_raw),
7984
+ "chunksRemaining": None,
7985
+ "reason": "cacheAbsent",
7986
+ },
7987
+ })
7988
+ return coverage
7989
+
7990
+
7991
+ #: How long one process waits before re-attempting a reconciliation that did not
7992
+ #: clear the flag.
7993
+ #:
7994
+ #: Nothing bounded the repetition before this. The states that leave the flag
7995
+ #: set are ordinary — `locksBusy` under a multi-agent hook storm, `restartLimit`
7996
+ #: under a competing `cache-sync --rebuild`, `noCoverageEstablished` and
7997
+ #: `mintRefused` — and in every one of them the next attempt re-read the whole
7998
+ #: journal and failed the same way. The marker is stamped on ATTEMPT rather than
7999
+ #: on outcome, because the cost this bounds is the read, which a failing attempt
8000
+ #: pays in full.
8001
+ _PROJECTION_RECONCILE_RETRY_SECONDS = 300.0
8002
+
8003
+
8004
+ def _projection_reconcile_marker_path():
8005
+ """Where the last reconciliation attempt is recorded.
8006
+
8007
+ Resolved at call time from `APP_DIR`, not captured at import, because the
8008
+ path constants are re-pointed by `_init_paths_from_env` and by every test
8009
+ fixture. It is a marker file rather than a column so the throttle costs no
8010
+ schema change: `stats_quota_projection_state` is part of the epoch-1009
8011
+ fingerprint, and widening it would move the hardcoded literal and every
8012
+ doctor golden with it.
8013
+ """
8014
+ return _cctally_core.APP_DIR / "stats.quota-reconcile.attempt"
8015
+
8016
+
8017
+ def _projection_reconcile_throttled() -> bool:
8018
+ """True when the previous attempt is too recent to repeat."""
8019
+ if _PROJECTION_RECONCILE_RETRY_SECONDS <= 0:
8020
+ return False
8021
+ try:
8022
+ age = time.time() - _projection_reconcile_marker_path().stat().st_mtime
8023
+ except OSError:
8024
+ return False
8025
+ return 0 <= age < _PROJECTION_RECONCILE_RETRY_SECONDS
8026
+
8027
+
8028
+ def _stamp_projection_reconcile_attempt() -> None:
8029
+ """Record an attempt. Never raises — a marker that cannot be written costs
8030
+ only a repeat, and failing an unrelated command over it would be worse."""
8031
+ try:
8032
+ _cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
8033
+ _projection_reconcile_marker_path().touch()
8034
+ os.utime(_projection_reconcile_marker_path(), None)
8035
+ except OSError:
8036
+ pass
8037
+
8038
+
8039
+ def _clear_projection_reconcile_attempt() -> None:
8040
+ """Drop the marker after a reconciliation that actually cleared the flag.
8041
+
8042
+ Same never-raises posture as the stamp: a marker that cannot be removed
8043
+ costs only a delayed retry, and failing the caller over it would be worse.
8044
+ """
8045
+ try:
8046
+ _projection_reconcile_marker_path().unlink()
8047
+ except OSError:
8048
+ pass
8049
+
8050
+
8051
+ def _cache_writer_flocks_available() -> bool:
8052
+ """One non-blocking probe of the two cache writer flocks.
8053
+
8054
+ The journal read is the expensive half of a recovery and was paid FIRST: a
8055
+ pass that cannot take these flocks applies no row, so reading 1.64 GB to
8056
+ discover that is pure waste. Taken here, after the maintenance and ingest
8057
+ locks, so the probe respects §4.7's lock order rather than inverting it.
8058
+ """
5925
8059
  from _lib_cache_writer_lock import (
5926
- acquire_cache_writer_flocks,
5927
- release_cache_writer_flocks,
8060
+ acquire_cache_writer_flocks, release_cache_writer_flocks,
5928
8061
  )
5929
-
5930
- _cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
5931
8062
  try:
5932
8063
  held = acquire_cache_writer_flocks(
5933
8064
  _cctally_core.CACHE_LOCK_PATH,
5934
8065
  _cctally_core.CACHE_LOCK_CODEX_PATH,
5935
- timeout=15.0,
8066
+ timeout=None,
5936
8067
  )
5937
- except OSError as exc:
5938
- print(f"[rebuild] quota cache leg lock failed: {exc}", file=sys.stderr)
5939
- return 0.0
8068
+ except OSError:
8069
+ return False
5940
8070
  if held is None:
5941
- print("[rebuild] quota cache leg locks busy; skipping", file=sys.stderr)
5942
- return 0.0
5943
- held_from = time.monotonic()
8071
+ return False
8072
+ release_cache_writer_flocks(held)
8073
+ return True
8074
+
8075
+
8076
+ def reconcile_incomplete_quota_projection(conn) -> bool:
8077
+ """Resume cache recovery and re-materialize a gated quota projection.
8078
+
8079
+ Returns whether the flag was cleared. Called from `open_db` ahead of the
8080
+ current-epoch fast return, because §4.7's "the next open reconciles it" is
8081
+ only enforceable if some open actually does it — `RebuildResult` is
8082
+ process-local and a current-epoch open otherwise returns without any
8083
+ reconciliation at all.
8084
+
8085
+ `open_db` only reaches this for an ARMED process
8086
+ (`_cctally_core.enable_quota_projection_reconciliation`), because the work
8087
+ below reads the journal from zero to the current high water and `open_db`
8088
+ is on the status-line path. Three things bound what an armed process pays:
8089
+ the maintenance acquire is non-blocking rather than a thirty-second wait,
8090
+ the cache flocks are probed BEFORE the journal read, and a durable marker
8091
+ throttles the repeat.
8092
+
8093
+ Lock order is the repository's, stated in §4.7 and taken in this order:
8094
+ maintenance-exclusive, then the ingest lock, then (inside the leg) the
8095
+ global and Codex cache flocks, then the cache snapshot, then the stats
8096
+ transaction. The ingest acquire is OPPORTUNISTIC: an open that loses it
8097
+ leaves the flag set and the projection gated, which is the fail-closed
8098
+ direction, rather than blocking an interactive command behind an ingest
8099
+ cycle.
8100
+
8101
+ A caller that already holds the ingest lock is skipped outright. That
8102
+ context is the serialized writer — a rebuild or an ingest cycle — and it
8103
+ sets or clears this flag itself; reconciling underneath it would run a
8104
+ second recovery inside its transaction.
8105
+ """
8106
+ import _cctally_db
8107
+ import _cctally_store
8108
+ if conn is None or _cctally_store.holds_ingest_lock():
8109
+ return False
8110
+ # #146's rule is literally about advancing `user_version`, and this path
8111
+ # runs after the `_uv == STATS_INDEX_EPOCH` fast return, so it changes no
8112
+ # schema and no version. But it WRITES data rows to the real prod stats.db
8113
+ # and cache.db, and the cutover branch beside it in `open_db` already
8114
+ # refuses that from a dev checkout. `db rebuild` is the only setter of the
8115
+ # flag and already carries the guard, so the exposure is small — small is
8116
+ # not a reason for the two neighbouring write paths to disagree.
8117
+ if _cctally_db._would_block_prod_stats(_cctally_core.DB_PATH):
8118
+ return False
8119
+ try:
8120
+ row = conn.execute(
8121
+ "SELECT incomplete FROM stats_quota_projection_state WHERE id = 1"
8122
+ ).fetchone()
8123
+ except sqlite3.Error:
8124
+ # NOT "the flag must be clear" — that justification is false and
8125
+ # `assert_projection_readable` no longer uses it. This is the opposite
8126
+ # decision on the opposite question: a connection whose probe fails
8127
+ # cannot safely START a reconciliation, because the reconciliation
8128
+ # writes through this very connection and would take the maintenance,
8129
+ # ingest and cache locks to do it. Declining leaves the flag set, and
8130
+ # the gate — which fails CLOSED on the same error — still refuses every
8131
+ # projection read. The two directions agree on the outcome the user
8132
+ # sees: no incomplete projection is served.
8133
+ return False
8134
+ if row is None or not int(row[0] or 0):
8135
+ return False
8136
+ if _projection_reconcile_throttled():
8137
+ return False
5944
8138
  try:
8139
+ with _cctally_store.stats_open_time_guard(
8140
+ live=True, wait_seconds=0.0):
8141
+ # Re-read under the exclusive: another process may have reconciled
8142
+ # it while this one waited, and re-running the whole recovery to
8143
+ # discover that would cost a journal read for nothing.
8144
+ try:
8145
+ row = conn.execute(
8146
+ "SELECT incomplete FROM stats_quota_projection_state "
8147
+ "WHERE id = 1").fetchone()
8148
+ except sqlite3.Error:
8149
+ return False
8150
+ if row is None or not int(row[0] or 0):
8151
+ return False
8152
+ fd = _acquire_ingest_lock("opportunistic", 0.0)
8153
+ if fd is None:
8154
+ return False
8155
+ try:
8156
+ if not _cache_writer_flocks_available():
8157
+ return False
8158
+ _stamp_projection_reconcile_attempt()
8159
+ coverage = recover_quota_cache_from_journal(quiet=True)
8160
+ if coverage.get("complete") is not True:
8161
+ return False
8162
+ cleared = _rematerialize_and_clear_projection_gate(
8163
+ conn, quiet=True)
8164
+ if cleared:
8165
+ # The marker bounds the cost of a FAILING attempt. A success
8166
+ # leaves nothing to bound, and keeping it makes the throttle
8167
+ # punish the next genuine incompleteness: a flag set again
8168
+ # within the interval — a second interrupted rebuild, which
8169
+ # is exactly the sequence an upgrade under lock contention
8170
+ # produces — would wait the interval out for no reason.
8171
+ _clear_projection_reconcile_attempt()
8172
+ return cleared
8173
+ finally:
8174
+ _release_ingest_lock(fd)
8175
+ except _cctally_db.StatsDbMaintenanceError:
8176
+ # Another process owns stats maintenance. Leaving the flag set is the
8177
+ # fail-closed direction and costs nothing but a later attempt.
8178
+ return False
8179
+
8180
+
8181
+ def _rematerialize_and_clear_projection_gate(conn, *, quiet=False) -> bool:
8182
+ """Re-materialize the projection from a complete cache and clear the flag.
8183
+
8184
+ Both happen in ONE stats transaction, so a crash between them cannot leave a
8185
+ cleared flag over a projection that was never rewritten.
8186
+
8187
+ The bundle is read from a read-only cache snapshot taken AFTER recovery
8188
+ completed, which is the same ordering the rebuild's recovery path uses and
8189
+ for the same reason: a snapshot from before those writes would miss exactly
8190
+ the rows recovery restored.
8191
+
8192
+ **No bundle, no clear.** `rematerialize_quota_projection_for_rebuild`
8193
+ treats an absent or unreadable cache as a clean no-op, so calling it with
8194
+ `bundle=None` returns without touching a single projection row — and
8195
+ clearing the flag afterwards would serve the partial projection the flag
8196
+ exists to refuse over a projection this function never rewrote. Every other
8197
+ degraded path here fails closed and this one must too.
8198
+
8199
+ This is the SECOND of the two guards that close that path, and the two are
8200
+ NESTED rather than disjoint: the `cacheAbsent` remainder in
8201
+ `recover_quota_cache_from_journal` fires on a subset of the states this one
8202
+ covers, because an absent cache also yields no bundle. Both still earn their
8203
+ place. The first names the state in the coverage record, which is what the
8204
+ `db rebuild --json` `cacheRecovery.remainder.reason` reports and what an
8205
+ operator reads; this one additionally covers a cache that EXISTS but cannot
8206
+ be read, which the first sees as present and would let through.
8207
+ """
8208
+ import _cctally_quota as _q
8209
+ cache_path = _cctally_core.CACHE_DB_PATH
8210
+ bundle = None
8211
+ if cache_path.exists():
5945
8212
  try:
5946
- cache = sqlite3.connect(str(cache_path), timeout=15.0)
5947
- except sqlite3.Error as exc: # pragma: no cover — cache.db unopenable
5948
- print(f"[rebuild] quota cache leg connect failed: {exc}", file=sys.stderr)
5949
- return time.monotonic() - held_from
8213
+ cache = sqlite3.connect(
8214
+ f"file:{cache_path}?mode=ro", uri=True, timeout=5.0)
8215
+ except sqlite3.Error:
8216
+ return False
5950
8217
  try:
5951
- cache.execute("PRAGMA busy_timeout=15000")
5952
- cache.execute("BEGIN IMMEDIATE")
5953
- # Decisions FIRST — same §3.5 precedence ordering as `_cache_applier`.
5954
- _, _file_conflicts = _apply_file_account_records(cache, file_accounts)
5955
- _apply_quota_records(
5956
- cache,
5957
- _decoded_quota_stream(quota_raw, cutover_claude, counters),
5958
- )
5959
- cache.commit()
5960
- _report_file_account_conflicts(_file_conflicts)
5961
- except sqlite3.Error as exc:
8218
+ cache.execute("PRAGMA busy_timeout=5000")
8219
+ cache.execute("BEGIN")
8220
+ bundle = _q.load_quota_projection_bundle(cache)
8221
+ except sqlite3.Error:
8222
+ return False
8223
+ finally:
5962
8224
  try:
5963
8225
  cache.rollback()
5964
8226
  except sqlite3.Error:
5965
8227
  pass
5966
- print(f"[rebuild] quota cache leg write failed: {exc}", file=sys.stderr)
5967
- finally:
5968
8228
  cache.close()
8229
+ if bundle is None:
8230
+ return False
8231
+ try:
8232
+ conn.execute("BEGIN IMMEDIATE")
8233
+ _q.rematerialize_quota_projection_for_rebuild(conn, bundle=bundle)
8234
+ conn.execute(
8235
+ "UPDATE stats_quota_projection_state SET incomplete = 0, "
8236
+ "target_version = 0, recovery_target_json = NULL WHERE id = 1")
8237
+ conn.commit()
8238
+ except sqlite3.Error as exc:
8239
+ try:
8240
+ conn.rollback()
8241
+ except sqlite3.Error:
8242
+ pass
8243
+ if not quiet:
8244
+ print(f"[stats] quota projection reconciliation failed: {exc}",
8245
+ file=sys.stderr)
8246
+ return False
8247
+ return True
8248
+
8249
+
8250
+ #: The version stamped into `stats_quota_projection_state.target_version`.
8251
+ #:
8252
+ #: The target is VERSIONED rather than a bare coordinate so a target written by
8253
+ #: one binary is never misread by another: a later binary that changes what the
8254
+ #: target names reads a version it does not recognize and treats the projection
8255
+ #: as reconcilable-by-full-recovery instead of interpreting fields it would
8256
+ #: misunderstand.
8257
+ PROJECTION_RECOVERY_TARGET_VERSION = 1
8258
+
8259
+
8260
+ def _write_quota_projection_state(conn, *, coverage, high_water):
8261
+ """Record whether this generation's quota projection is complete.
8262
+
8263
+ Returns the flag it wrote, or **None** when it could not write one. The
8264
+ third state is the point: returning `False` for both "wrote clear" and
8265
+ "could not write" is what made a shipped test vacuous, because the caller
8266
+ then reported a complete projection for a generation whose flag says
8267
+ nothing. `None` is fail-closed at the caller — an unwritten flag is
8268
+ reported as incomplete, which costs a reconciliation and never serves a
8269
+ partial projection.
8270
+
8271
+ Runs inside the caller's transaction, which is the one that materialized
8272
+ the projection — the flag and the rows it describes must commit or roll
8273
+ back together.
8274
+
8275
+ A coverage record with no `complete` key is a leg that never ran (a
8276
+ `update_quota_cache=False` rebuild, or one with nothing to do), and that is
8277
+ complete by absence rather than incomplete by ignorance.
8278
+ """
8279
+ incomplete = bool(coverage) and coverage.get("complete") is False
8280
+ target = None
8281
+ if incomplete:
8282
+ target = json.dumps(
8283
+ {
8284
+ "highWater": (
8285
+ None if high_water is None
8286
+ else [str(high_water[0]), int(high_water[1])]
8287
+ ),
8288
+ "coveredHighWater": coverage.get("coveredHighWater"),
8289
+ "remainder": coverage.get("remainder"),
8290
+ },
8291
+ separators=(",", ":"), sort_keys=True,
8292
+ )
8293
+ try:
8294
+ conn.execute(
8295
+ "INSERT INTO stats_quota_projection_state"
8296
+ "(id, incomplete, target_version, recovery_target_json) "
8297
+ "VALUES (1, ?, ?, ?) "
8298
+ "ON CONFLICT(id) DO UPDATE SET incomplete = excluded.incomplete, "
8299
+ "target_version = excluded.target_version, "
8300
+ "recovery_target_json = excluded.recovery_target_json",
8301
+ (1 if incomplete else 0,
8302
+ PROJECTION_RECOVERY_TARGET_VERSION if incomplete else 0,
8303
+ target),
8304
+ )
8305
+ except sqlite3.Error as exc: # pragma: no cover — pre-1009 index
8306
+ print(f"[rebuild] quota projection state not recorded: {exc}",
8307
+ file=sys.stderr)
8308
+ return None
8309
+ return incomplete
8310
+
8311
+
8312
+ def _read_quota_projection_bundle(snapshot_out):
8313
+ """§4.4's projection bundle from a retained coverage snapshot, or None.
8314
+
8315
+ ``snapshot_out`` is the single-element list the leg appends its still-open
8316
+ read transaction to on the intact path. The bundle is read from that same
8317
+ transaction — under WAL its snapshot is fixed at the certificate read, so
8318
+ the certificate, the sequence, the source roots, the observations and the
8319
+ ledger state all describe ONE cache state.
8320
+
8321
+ None means "no snapshot to consume", and the projection then opens its own
8322
+ connection. That is the recovery path, where a snapshot from before the
8323
+ leg's writes would miss the rows recovery restored, and every degraded case,
8324
+ where falling back is the same silent full behaviour §6.3 asks for.
8325
+ """
8326
+ if not snapshot_out:
8327
+ return None
8328
+ snapshot = snapshot_out[0]
8329
+ if snapshot is None or snapshot.conn is None:
8330
+ return None
8331
+ try:
8332
+ import _cctally_quota as _q
8333
+ return _q.load_quota_projection_bundle(snapshot.conn)
8334
+ except (sqlite3.Error, ValueError):
8335
+ return None
5969
8336
  finally:
5970
- release_cache_writer_flocks(held)
5971
- return time.monotonic() - held_from
8337
+ _close_coverage_snapshot(snapshot)
8338
+
8339
+
8340
+ def _resolve_quota_cache_coverage(cache_path, high_water, decoded_end=None):
8341
+ """``(vector, covered, verdict, snapshot)`` for this coverage decision.
8342
+
8343
+ ``snapshot`` is returned with its read transaction still open, or None when
8344
+ none could be taken. The caller closes it — on the intact path after it has
8345
+ read §4.4's projection bundle from it, and immediately otherwise.
8346
+
8347
+ ``verdict`` is one of `_lib_cache_coverage`'s reason strings, and `REASON_OK`
8348
+ means the intact path: every cache-relevant journal record in the pinned
8349
+ prefix is already materialized, so the leg takes no writer flock at all.
8350
+
8351
+ Two independent checks, and BOTH are required. `certificate_is_valid` is an
8352
+ identity check — it asks whether the certificate describes the journal and
8353
+ cache physically in front of it — and a certificate covering only the first
8354
+ of three segments passes it. Whether coverage REACHES this rebuild's pinned
8355
+ high-water is a separate comparison, made here.
8356
+
8357
+ ``decoded_end`` is the traversal's own last complete-line boundary, and it
8358
+ bounds the covered claim for the reason `_bounded_covered_offset` gives.
8359
+ """
8360
+ vector = coverage_pinned_vector()
8361
+ covered = None
8362
+ for name, _raw_extent, covered_offset in vector:
8363
+ if high_water is not None and name == str(high_water[0]):
8364
+ covered = (name, _bounded_covered_offset(
8365
+ name, int(high_water[1]), covered_offset, decoded_end))
8366
+ if covered is None:
8367
+ # No boundary could be resolved at all. Reporting `identityRoot` here
8368
+ # would tell an operator the journal identity moved when in fact the
8369
+ # certificate was never consulted.
8370
+ return vector, None, _lib_cache_coverage.REASON_NO_BOUNDARY, None
8371
+ snapshot = _read_coverage_snapshot(cache_path)
8372
+ if snapshot is None:
8373
+ return vector, covered, _lib_cache_coverage.REASON_ABSENT, None
8374
+ ok, reason = _lib_cache_coverage.certificate_is_valid(
8375
+ snapshot.certificate,
8376
+ pinned_vector=vector,
8377
+ physical_seq=snapshot.physical_seq,
8378
+ )
8379
+ if not ok:
8380
+ return vector, covered, reason, snapshot
8381
+ if not _coordinate_covers(snapshot.certificate["coveredHighWater"], covered):
8382
+ return (vector, covered,
8383
+ _lib_cache_coverage.REASON_COVERED_HIGH_WATER, snapshot)
8384
+ return vector, covered, _lib_cache_coverage.REASON_OK, snapshot
8385
+
8386
+
8387
+ def _rebuild_quota_cache_leg_raw(
8388
+ quota_raw, decoded, cutover_claude, counters=None, *,
8389
+ high_water=None, coverage=None, decoded_end=None, snapshot_out=None,
8390
+ quiet=False, elision_gaps=None,
8391
+ ) -> float:
8392
+ """Re-materialize cache.db `quota_window_snapshots` AND the #416 Codex
8393
+ attribution map from the journal (spec §5.4 + #416 spec §3.4), fed RAW
8394
+ ENCODED LINES for the observations instead of decoded dicts.
8395
+
8396
+ The journal records are the DURABLE source (§1 latent data-loss hole — the
8397
+ rollout JSONL evaporates); this INSERT-OR-IGNOREs the quota obs on their
8398
+ natural key and the attribution decisions on theirs, mirroring
8399
+ `_cache_applier` family for family. The map half is not optional: without it
8400
+ a rebuild would leave the map empty and the following rollout walk would
8401
+ have nothing to replay, sending every file back to the live `auth.json` —
8402
+ the exact defect this mechanism exists to prevent.
8403
+
8404
+ Runs BEFORE any stats transaction, under the global cache writer lock
8405
+ followed by the `cache.db.codex.lock` provider flock (lock-order law).
8406
+ Best-effort: a missing/busy cache.db is a clean skip (the records stay
8407
+ durable in the journal; the stats quota projection pass then degrades
8408
+ cleanly).
8409
+
8410
+ Taking raw bytes is what makes the rebuild affordable. 1.81M decoded
8411
+ observation dictionaries are roughly six gigabytes against 1.64 GB of raw
8412
+ bytes, so retaining bytes removes about four gigabytes of peak heap while
8413
+ adding NO file input and NO second traversal — only the JSON decode of
8414
+ records already in memory moves inside the flocks (#496 S4 §6.3). That
8415
+ decode is why the measured hold is longer than it was before S4; see the
8416
+ spec's §6.3 for the measured figures.
8417
+
8418
+ Ordering is preserved: file-account decisions are ops, so the router already
8419
+ retains them decoded and they are available before the observation loop
8420
+ begins, exactly as the §3.5 precedence rule requires.
8421
+
8422
+ Returns the measured flock hold in seconds, which acceptance criterion 7
8423
+ caps.
8424
+ """
8425
+ file_accounts = [
8426
+ r for r in decoded if r is not None and _is_codex_file_account_op(r)
8427
+ ]
8428
+ if coverage is not None:
8429
+ # `complete` is TRUE for a skip, and that is not a slip. Spec §4.7
8430
+ # distinguishes stats publication success from cache-recovery
8431
+ # completeness, and a leg with nothing to recover has no shortfall to
8432
+ # report — the incompleteness this flag exists for is an uncovered
8433
+ # REMAINDER, not an absent duty.
8434
+ coverage.update({
8435
+ "status": "skipped", "reason": None, "replayedObservations": 0,
8436
+ "chunks": 0, "plannedChunks": 0, "restarts": 0,
8437
+ "concurrentWriter": False, "complete": True, "remainder": None,
8438
+ })
8439
+ # `elision_gaps` keeps this from returning early on an empty stream: a pass
8440
+ # that elided every quota-bearing segment has an empty `quota_raw` and a
8441
+ # cache that may still need those observations if the certificate stopped
8442
+ # being valid while the pass ran.
8443
+ if not quota_raw and not file_accounts and not elision_gaps:
8444
+ return 0.0
8445
+ cache_path = _cctally_core.CACHE_DB_PATH
8446
+ if not cache_path.exists():
8447
+ return 0.0
8448
+
8449
+ # F11's intact path (spec §4.4). Decided from a read-only WAL snapshot
8450
+ # BEFORE any flock is requested, so a covered cache costs this rebuild zero
8451
+ # writer-lock hold rather than the measured 23.0 s it costs today. The
8452
+ # vector is pinned here for the same reason `_coverage_advance_plan` pins
8453
+ # its own before the flocks: one captured after a concurrent append would
8454
+ # match the current journal while a record nobody applied sat inside it.
8455
+ vector, covered, verdict, snapshot = _resolve_quota_cache_coverage(
8456
+ cache_path, high_water, decoded_end)
8457
+ if coverage is not None:
8458
+ coverage["reason"] = verdict
8459
+ if verdict == _lib_cache_coverage.REASON_OK:
8460
+ if coverage is not None:
8461
+ coverage.update({
8462
+ "status": "covered",
8463
+ "coveredHighWater": [covered[0], covered[1]],
8464
+ "replayedObservations": 0,
8465
+ "complete": True, "remainder": None,
8466
+ })
8467
+ # The intact path wrote nothing, so this snapshot still describes the
8468
+ # cache the verdict was decided against and §4.4's projection bundle may
8469
+ # be read from it. Hand it to the caller with its read transaction open.
8470
+ if snapshot_out is not None:
8471
+ snapshot_out.append(snapshot)
8472
+ else:
8473
+ _close_coverage_snapshot(snapshot)
8474
+ return 0.0
8475
+
8476
+ # The recovery path is about to WRITE to this cache, and a snapshot taken
8477
+ # before those writes would miss exactly the rows recovery restores. It is
8478
+ # closed here and the projection reads its own afterwards, which is what the
8479
+ # pre-change leg did at the same point.
8480
+ _close_coverage_snapshot(snapshot)
8481
+ if elision_gaps:
8482
+ # The read pass elided on a certificate that was valid when it planned,
8483
+ # and this leg's own verdict disagrees — an ordinary Codex batch landing
8484
+ # mid-pass is enough, because it advances the certificate over a journal
8485
+ # this pass did not pin. Recovery is about to replay, so the elided
8486
+ # segments' observations are re-read now, OUTSIDE both flocks, and the
8487
+ # replay proceeds over the same stream a non-eliding pass would have
8488
+ # built. Elision then costs one read in the racy case rather than an
8489
+ # unmaterialized observation under a certificate claiming coverage.
8490
+ before = len(quota_raw)
8491
+ quota_raw, refilled = _refill_elided_quota_raw(quota_raw, elision_gaps)
8492
+ if not refilled:
8493
+ # A segment this pass elided could not be re-read, so its
8494
+ # observations are absent from the stream about to be replayed.
8495
+ # `_run_bounded_recovery` mints whenever `covered` is not None, and
8496
+ # `covered` is the whole pinned prefix — so leaving it set would
8497
+ # certify coverage over rows nobody applied. Dropping it routes the
8498
+ # pass through `noCoverageEstablished`: it applies what it holds and
8499
+ # certifies nothing.
8500
+ covered = None
8501
+ if coverage is not None:
8502
+ coverage["elisionRefill"] = {
8503
+ "segments": len(elision_gaps),
8504
+ "observations": len(quota_raw) - before,
8505
+ "complete": refilled,
8506
+ }
8507
+ return _run_bounded_recovery(
8508
+ quota_raw, file_accounts, cutover_claude, counters,
8509
+ cache_path=cache_path, vector=vector, covered=covered,
8510
+ high_water=high_water, coverage=coverage, quiet=quiet,
8511
+ )
8512
+
8513
+
8514
+ #: Per-chunk caps for the recovery pass. BOTH are enforced (spec §4.5): capping
8515
+ #: by records alone lets one chunk of large observations blow the memory bound,
8516
+ #: and capping by bytes alone lets a chunk of tiny ones carry far more rows than
8517
+ #: one transaction should.
8518
+ #:
8519
+ #: 8 MiB against the maintainer's 905-byte mean observation is roughly 9,000
8520
+ #: records, so the byte cap binds first on real data and the record cap is the
8521
+ #: backstop for a journal of unusually small lines. The pair keeps peak decoded
8522
+ #: memory at one chunk, which is what makes per-chunk decode satisfy F11 and S4's
8523
+ #: measured 2.09 GB together.
8524
+ _RECOVERY_CHUNK_BYTES = 8 * 1024 * 1024
8525
+ _RECOVERY_CHUNK_RECORDS = 20_000
8526
+
8527
+ #: How many times one pass may restart from zero before giving up and reporting
8528
+ #: an uncovered remainder. A restart is triggered by a destructive writer, and a
8529
+ #: writer that keeps clearing the cache would otherwise make this pass loop for
8530
+ #: as long as it kept doing so. Three is enough to ride out a single competing
8531
+ #: `cache-sync --rebuild`; past that the honest answer is an incomplete pass,
8532
+ #: which §4.7 already has a contract for.
8533
+ _RECOVERY_MAX_RESTARTS = 3
8534
+
8535
+ #: Test-only seam, called with the number of chunks committed so far AFTER both
8536
+ #: flocks are released and BEFORE the next chunk requests them. That is exactly
8537
+ #: the window a competing writer can use, so a test proving a hold was RELEASED
8538
+ #: rather than merely shortened writes here rather than racing a thread.
8539
+ _RECOVERY_BETWEEN_CHUNKS = None
8540
+
8541
+
8542
+ def _recovery_pass_identity():
8543
+ """``(pass_id, started_at)`` for one recovery pass.
8544
+
8545
+ ``started_at`` is a coarse wall clock in microseconds, and it exists only to
8546
+ ORDER two passes for the monotonic compare-and-swap. It is never compared for
8547
+ equality and never used as a deadline.
8548
+ """
8549
+ return (
8550
+ hashlib.sha256(
8551
+ f"{os.getpid()}:{time.time_ns()}:{id(object())}".encode("utf-8")
8552
+ ).hexdigest()[:16],
8553
+ int(time.time() * 1_000_000),
8554
+ )
8555
+
8556
+
8557
+ def _recovery_state(cache):
8558
+ """``(physical_seq, source_roots_digest)`` read inside the caller's txn."""
8559
+ import _cctally_quota
8560
+ seq = _cctally_quota.codex_physical_mutation_seq(cache)
8561
+ digest = _lib_cache_coverage.source_roots_digest(
8562
+ _cctally_quota._cache_root_keys(cache))
8563
+ return seq, digest
8564
+
8565
+
8566
+ def _run_bounded_recovery(
8567
+ quota_raw, file_accounts, cutover_claude, counters, *,
8568
+ cache_path, vector, covered, high_water, coverage, quiet=False,
8569
+ ) -> float:
8570
+ """Recovery as resumable chunks, each capped by bytes AND record count.
8571
+
8572
+ Per chunk: decode that chunk's raw lines, acquire the global then the Codex
8573
+ flock, `BEGIN IMMEDIATE`, revalidate, apply, persist progress, commit,
8574
+ release both locks, discard the decoded chunk. The decode is DELIBERATELY
8575
+ per chunk and not once up front: decoding everything first would
8576
+ re-materialize the roughly six gigabytes of observation dictionaries S4
8577
+ removed to bring peak heap from 6.50 GB to 2.09 GB.
8578
+
8579
+ **Every chunk revalidates after reacquiring the locks, because releasing
8580
+ them admits a destructive writer.** A concurrent `cache-sync --rebuild` can
8581
+ take the locks between two chunks, run `_clear_codex_derived_rows`, and
8582
+ delete both the materialized quota state and the certificate. A pass that
8583
+ resumed from an in-memory cursor would continue over cleared state and mint
8584
+ a certificate claiming coverage the cache does not have. Progress is
8585
+ therefore persisted separately from the certificate, every destructive clear
8586
+ deletes it in the same transaction (`_invalidate_codex_journal_coverage_
8587
+ certificate` does both), and a pass that finds it gone restarts from zero.
8588
+ Restarting is always sound, because every apply is idempotent on its natural
8589
+ key.
8590
+
8591
+ Two pieces of per-record state survive the chunking (spec §4.6). The anchor
8592
+ resolver is RECONSTRUCTED per chunk — `_apply_quota_records` builds one per
8593
+ call, and it seeds lazily from the rows already stored, so each chunk's
8594
+ anchor decisions are made against the state committed at that moment rather
8595
+ than against a cache another writer may have mutated underneath a
8596
+ long-lived resolver. The conflict-report set is THREADED across the chunks
8597
+ of one pass, so a single recovery reports at most what a single unchunked
8598
+ call reports today.
8599
+
8600
+ Returns the total measured flock hold across every chunk.
8601
+ """
8602
+ import _cctally_cache
8603
+ from _lib_cache_writer_lock import (
8604
+ acquire_cache_writer_flocks,
8605
+ release_cache_writer_flocks,
8606
+ )
8607
+
8608
+ pass_id, started_at = _recovery_pass_identity()
8609
+ identity_root = _lib_cache_coverage.identity_root(vector)
8610
+ spans = _lib_cache_coverage.chunk_spans(
8611
+ [len(raw) + 1 for raw in quota_raw],
8612
+ byte_cap=_RECOVERY_CHUNK_BYTES, record_cap=_RECOVERY_CHUNK_RECORDS,
8613
+ )
8614
+ # NOT cleared on a restart, deliberately. §4.6 requires that one recovery
8615
+ # report at most what a single unchunked call reports today, and a restart
8616
+ # replays records this pass has already reported a conflict for.
8617
+ reported_conflicts: set = set()
8618
+ # The replay counters are cumulative across every decode this pass makes, so
8619
+ # a restart would count the re-decoded prefix twice and inflate
8620
+ # `traversal["quota_replay"]` against what one unchunked call reports. They
8621
+ # are snapshotted here and restored on each restart.
8622
+ counter_baseline = dict(counters) if counters is not None else None
8623
+ hold_seconds = 0.0
8624
+ restarts = 0
8625
+ chunk_index = 0
8626
+ applied = 0
8627
+ committed_chunks = 0
8628
+ outcome = "recovered"
8629
+ stop_reason = None
8630
+ concurrent_writer = False
8631
+
8632
+ _cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
8633
+
8634
+ def _record(status):
8635
+ if coverage is None:
8636
+ return
8637
+ coverage.update({
8638
+ "status": status,
8639
+ "coveredHighWater": (
8640
+ None if covered is None else [covered[0], covered[1]]),
8641
+ "replayedObservations": applied,
8642
+ "chunks": committed_chunks,
8643
+ "plannedChunks": len(plan),
8644
+ "restarts": restarts,
8645
+ "concurrentWriter": concurrent_writer,
8646
+ "complete": status == "recovered",
8647
+ "remainder": (
8648
+ None if status == "recovered"
8649
+ else {
8650
+ "observations": max(0, len(quota_raw) - applied),
8651
+ "chunksRemaining": max(
8652
+ 0, len(plan) - committed_chunks),
8653
+ "reason": stop_reason,
8654
+ }
8655
+ ),
8656
+ })
8657
+
8658
+ # Chunk 0 carries EVERY file-account decision, because §3.5 makes the
8659
+ # file/range decision authoritative over the observation stamp and a
8660
+ # decision must already govern the observations it covers. That population
8661
+ # is the retained decision records — 5.08% of a production journal — and it
8662
+ # is bounded by the decisions rather than by the observations, so it does
8663
+ # not reintroduce the unbounded hold the chunking exists to remove.
8664
+ #
8665
+ # It shares chunk 0's transaction with the first observation span rather
8666
+ # than taking a lock cycle of its own, so a recovery small enough to fit one
8667
+ # chunk takes exactly ONE hold — byte-identical in shape to the unchunked
8668
+ # form this replaces.
8669
+ if spans:
8670
+ plan = [(True, spans[0])] + [(False, span) for span in spans[1:]]
8671
+ else:
8672
+ plan = [(True, None)]
8673
+
8674
+ while chunk_index < len(plan):
8675
+ with_decisions, span = plan[chunk_index]
8676
+ decoded_chunk = None
8677
+ if span is not None:
8678
+ start, stop, _bytes = span
8679
+ decoded_chunk = list(_decoded_quota_stream(
8680
+ quota_raw[start:stop], cutover_claude, counters))
8681
+ try:
8682
+ held = acquire_cache_writer_flocks(
8683
+ _cctally_core.CACHE_LOCK_PATH,
8684
+ _cctally_core.CACHE_LOCK_CODEX_PATH,
8685
+ timeout=15.0,
8686
+ )
8687
+ except OSError as exc:
8688
+ if not quiet:
8689
+ print(f"[rebuild] quota cache leg lock failed: {exc}",
8690
+ file=sys.stderr)
8691
+ outcome, stop_reason = "incomplete", "lockFailed"
8692
+ break
8693
+ if held is None:
8694
+ if not quiet:
8695
+ print("[rebuild] quota cache leg locks busy; skipping",
8696
+ file=sys.stderr)
8697
+ outcome, stop_reason = "incomplete", "locksBusy"
8698
+ break
8699
+ held_from = time.monotonic()
8700
+ cache = None
8701
+ try:
8702
+ try:
8703
+ cache = sqlite3.connect(str(cache_path), timeout=15.0)
8704
+ cache.execute("PRAGMA busy_timeout=15000")
8705
+ cache.execute("BEGIN IMMEDIATE")
8706
+ if chunk_index > 0:
8707
+ seq, digest = _recovery_state(cache)
8708
+ verdict, why, saw_writer = (
8709
+ _lib_cache_coverage.resume_verdict(
8710
+ _cctally_cache.load_codex_recovery_progress(cache),
8711
+ pass_id=pass_id, started_at=started_at,
8712
+ identity_root=identity_root, physical_seq=seq,
8713
+ source_roots_digest=digest,
8714
+ )
8715
+ )
8716
+ concurrent_writer = concurrent_writer or saw_writer
8717
+ if verdict == _lib_cache_coverage.YIELD:
8718
+ cache.rollback()
8719
+ outcome, stop_reason = "incomplete", why
8720
+ break
8721
+ if verdict == _lib_cache_coverage.RESTART:
8722
+ cache.rollback()
8723
+ restarts += 1
8724
+ if restarts > _RECOVERY_MAX_RESTARTS:
8725
+ outcome, stop_reason = "incomplete", "restartLimit"
8726
+ break
8727
+ chunk_index = 0
8728
+ applied = 0
8729
+ committed_chunks = 0
8730
+ if counter_baseline is not None:
8731
+ counters.clear()
8732
+ counters.update(counter_baseline)
8733
+ continue
8734
+ file_conflicts = 0
8735
+ if with_decisions:
8736
+ # Decisions FIRST inside this transaction — the same §3.5
8737
+ # precedence ordering `_cache_applier` keeps.
8738
+ _restored, file_conflicts = _apply_file_account_records(
8739
+ cache, file_accounts)
8740
+ if decoded_chunk is not None:
8741
+ _apply_quota_records(
8742
+ cache, decoded_chunk,
8743
+ reported_conflicts=reported_conflicts, quiet=quiet)
8744
+ applied += len(decoded_chunk)
8745
+ last = chunk_index == len(plan) - 1
8746
+ if last and covered is None:
8747
+ # No boundary was resolvable at all, so this pass covered
8748
+ # bytes it cannot name and established nothing. Reporting
8749
+ # `complete` here would say "cache recovery complete" for a
8750
+ # pass that certified no prefix. The progress record goes
8751
+ # too: it describes a run that will never be finished.
8752
+ cache.execute(
8753
+ "DELETE FROM cache_meta WHERE key = ?",
8754
+ (_lib_cache_coverage.PROGRESS_KEY,))
8755
+ cache.commit()
8756
+ committed_chunks += 1
8757
+ chunk_index += 1
8758
+ _report_file_account_conflicts(file_conflicts, quiet=quiet)
8759
+ outcome, stop_reason = "incomplete", "noCoverageEstablished"
8760
+ break
8761
+ if last and covered is not None:
8762
+ # The pass has now replayed the whole pinned prefix, so it
8763
+ # is the one caller allowed to ESTABLISH coverage rather
8764
+ # than only extend it — it is the only writer that reads the
8765
+ # journal. Inside the same transaction as the last rows it
8766
+ # certifies, so a rollback leaves no certificate over rows
8767
+ # that never landed. The progress record goes away with it:
8768
+ # a certificate supersedes progress, and leaving both would
8769
+ # let a later pass resume a run that already finished.
8770
+ minted = _cctally_cache._advance_codex_journal_coverage(
8771
+ cache, prior=None, covered=covered,
8772
+ applied_through=(
8773
+ str(high_water[0]), int(high_water[1])),
8774
+ pinned_vector=vector, allow_mint=True)
8775
+ cache.execute(
8776
+ "DELETE FROM cache_meta WHERE key = ?",
8777
+ (_lib_cache_coverage.PROGRESS_KEY,))
8778
+ if not minted:
8779
+ # The mint refused — a stored certificate already reaches
8780
+ # further than this pass consumed to. Nothing false was
8781
+ # certified, but no coverage was established either, so
8782
+ # the honest report is an incomplete pass rather than
8783
+ # `complete: True` over an absent certificate.
8784
+ cache.commit()
8785
+ committed_chunks += 1
8786
+ chunk_index += 1
8787
+ _report_file_account_conflicts(file_conflicts, quiet=quiet)
8788
+ outcome, stop_reason = "incomplete", "mintRefused"
8789
+ break
8790
+ else:
8791
+ seq, digest = _recovery_state(cache)
8792
+ _cctally_cache._store_codex_recovery_progress(
8793
+ cache,
8794
+ _lib_cache_coverage.make_progress(
8795
+ pass_id=pass_id, started_at=started_at,
8796
+ chunks=chunk_index + 1,
8797
+ identity_root=identity_root, physical_seq=seq,
8798
+ source_roots_digest=digest,
8799
+ covered=(covered if covered is not None
8800
+ else ("", 0))),
8801
+ )
8802
+ cache.commit()
8803
+ committed_chunks += 1
8804
+ chunk_index += 1
8805
+ _report_file_account_conflicts(file_conflicts, quiet=quiet)
8806
+ except sqlite3.Error as exc:
8807
+ if cache is not None:
8808
+ try:
8809
+ cache.rollback()
8810
+ except sqlite3.Error:
8811
+ pass
8812
+ if not quiet:
8813
+ print(f"[rebuild] quota cache leg write failed: {exc}",
8814
+ file=sys.stderr)
8815
+ outcome, stop_reason = "failed", "writeFailed"
8816
+ break
8817
+ finally:
8818
+ if cache is not None:
8819
+ cache.close()
8820
+ finally:
8821
+ hold_seconds += time.monotonic() - held_from
8822
+ release_cache_writer_flocks(held)
8823
+ decoded_chunk = None
8824
+ if _RECOVERY_BETWEEN_CHUNKS is not None:
8825
+ _RECOVERY_BETWEEN_CHUNKS(committed_chunks)
8826
+
8827
+ _record(outcome)
8828
+ return hold_seconds
5972
8829
 
5973
8830
 
5974
8831
  #: `_resolve_cutover_for_rebuild` distinguishes "the streaming pass never saw the
@@ -6030,6 +8887,61 @@ def _resolve_cutover_for_rebuild(captured, hw, segments, counters=None) -> str:
6030
8887
  return _lib_accounts.UNATTRIBUTED
6031
8888
 
6032
8889
 
8890
+ def _observe_selector_desynchronization(path) -> "dict | None":
8891
+ """The selector prefix of the index at ``path``, when it is behind its cursor.
8892
+
8893
+ Returns `None` for the healthy case and for every unreadable one — an absent
8894
+ destination, a pre-1009 index without the selector tables, a missing cursor
8895
+ row. This is diagnostic reporting over a re-derivable artifact, so it must
8896
+ never be the reason a rebuild fails.
8897
+ """
8898
+ import _lib_stats_wal
8899
+
8900
+ path = pathlib.Path(path)
8901
+ if not path.exists():
8902
+ return None
8903
+ wal_index = _lib_stats_wal.inspect_wal_index_family(path)
8904
+ if wal_index.get("verdict") not in {"coherent", "wal_absent", "wal_empty"}:
8905
+ # This observation is optional. Opening an unproven WAL/SHM family can
8906
+ # rewrite its headers before the cutover path preserves the incident
8907
+ # bytes, so only let SQLite see a raw-classified safe family (#514).
8908
+ return None
8909
+ try:
8910
+ conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True, timeout=5.0)
8911
+ except sqlite3.Error:
8912
+ return None
8913
+ try:
8914
+ state = _read_selector_state(conn)
8915
+ if state is None:
8916
+ return None
8917
+ cursor = _read_cursor(conn)
8918
+ except sqlite3.Error:
8919
+ return None
8920
+ finally:
8921
+ conn.close()
8922
+ covered = (state.covered_segment, state.covered_offset)
8923
+ covered_at = _prefix_position(covered)
8924
+ cursor_at = _prefix_position(cursor)
8925
+ if cursor_at is None or (covered_at is not None and covered_at >= cursor_at):
8926
+ return None
8927
+ # A gap the live path REFUSES to re-fold is a different state from one it
8928
+ # will close on the next tick, and only the rebuild record makes either
8929
+ # visible. `gapBytes` is `None` when the distance could not be determined,
8930
+ # which the live path also treats as over the cap.
8931
+ gap_bytes = _selector_gap_bytes(covered, cursor)
8932
+ return {
8933
+ "coveredSegment": covered[0],
8934
+ "coveredOffset": covered[1],
8935
+ "cursorSegment": cursor[0],
8936
+ "cursorOffset": int(cursor[1]),
8937
+ "gapBytes": gap_bytes,
8938
+ "gapByteCap": _GAP_REFOLD_BYTE_CAP,
8939
+ "gapExceedsCap": (
8940
+ gap_bytes is None or gap_bytes > _GAP_REFOLD_BYTE_CAP
8941
+ ),
8942
+ }
8943
+
8944
+
6033
8945
  def rebuild_stats_index(
6034
8946
  *,
6035
8947
  context: RebuildContext,
@@ -6103,6 +9015,13 @@ def rebuild_stats_index(
6103
9015
  )
6104
9016
  segments = segments[:segments.index(hw[0]) + 1]
6105
9017
 
9018
+ # Observed BEFORE the scratch is built, because it describes the index this
9019
+ # rebuild is about to replace (#496 S5b). It is the only report a persistent
9020
+ # selector desynchronization gets: the live path's own recovery is silent by
9021
+ # §6.3's uniform policy, so without this a repeatedly degrading generation
9022
+ # check would leave no trace anywhere.
9023
+ selector_desync = _observe_selector_desynchronization(dest)
9024
+
6106
9025
  stamp = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%S_%f")
6107
9026
  scratch = dest.with_name(dest.name + f".rebuilding-{stamp}")
6108
9027
  _remove_db_family(scratch)
@@ -6120,6 +9039,25 @@ def rebuild_stats_index(
6120
9039
  "quota_replay")
6121
9040
  }
6122
9041
  quota_lock_hold = 0.0
9042
+ #: Whether this generation's quota projection was materialized from a cache
9043
+ #: whose recovery left an uncovered remainder. Declared out here because the
9044
+ #: record and the result are assembled outside the try that sets it.
9045
+ stats_projection_incomplete = False
9046
+ #: §4.4's single read-only WAL snapshot, carried from the quota cache leg to
9047
+ #: the projection pass. The leg appends it ONLY on the intact path, where it
9048
+ #: wrote nothing and its snapshot therefore still describes the cache the
9049
+ #: coverage verdict was decided against. A destructive clear landing between
9050
+ #: the verdict and the projection would otherwise publish a generation whose
9051
+ #: quota projection was materialized from a cleared cache while the verdict
9052
+ #: already read `covered`. Declared out here so the `finally` that closes it
9053
+ #: cannot meet an unbound name.
9054
+ quota_snapshot: list = []
9055
+ #: #496 S5b §4.7 "recorded, not silent": the coverage verdict, the boundary
9056
+ #: it reached and how many observations it had to replay. Additive to a
9057
+ #: `schemaVersion: 1` record under S4's rule, and the ONLY surface reporting
9058
+ #: it — §6.3 makes every degraded coverage state a silent full replay.
9059
+ quota_coverage: dict = {
9060
+ "status": "skipped", "reason": None, "replayedObservations": 0}
6123
9061
  tracing = tracemalloc.is_tracing()
6124
9062
  if tracing:
6125
9063
  tracemalloc.reset_peak()
@@ -6135,23 +9073,70 @@ def rebuild_stats_index(
6135
9073
  protocol_evidence = []
6136
9074
  prior_high_water = None
6137
9075
  cutover_captured = _CUTOVER_UNSEEN
9076
+ # Sequence -> journal coordinate, for correction-batch MARKERS only
9077
+ # (#496 S5b §3.5). That is what removes `_correction_commit_high_water`'s
9078
+ # separate traversal from the live fast path, and it is bounded to the
9079
+ # markers because nothing else is ever looked up: a production journal's
9080
+ # 64,248 correction records carry far fewer markers than lines.
9081
+ marker_coordinates: dict = {}
6138
9082
  last_seen = _lib_journal_router.LastSeenAccumulator()
9083
+ summaries = _SegmentSummaryCollector()
9084
+ # F12 (#496 S5b §5). Planned before the first segment is reached, from
9085
+ # the sidecar this pass's predecessor wrote and the coverage certificate
9086
+ # Stage 3 mints. A plan that elides nothing is exactly today's pass.
9087
+ elision = plan_segment_elision(segments, hw)
6139
9088
  hasher = _lib_journal_router.PrefixHashAccumulator()
6140
9089
  evidence_seconds = 0.0
6141
9090
  prefix = traversal["stats_prefix"]
6142
9091
  read_started = time.monotonic()
9092
+
9093
+ def _elide_segment(name, lo, hi, stat_result) -> bool:
9094
+ """Contribute an elided segment's share instead of reading it."""
9095
+ nonlocal prior_high_water
9096
+ nonlocal malformed
9097
+ summary = elision.decide(name, hi, stat_result)
9098
+ if summary is None:
9099
+ return False
9100
+ prefix["lines"] += summary.lines
9101
+ prefix["bytes"] += summary.bytes
9102
+ prefix["decodes"] += summary.decodes
9103
+ malformed += summary.malformed
9104
+ # EXACTLY the segment's `decoded`-entry count. A quota-only segment
9105
+ # contributes only placeholders, and `resolve_effective_events`
9106
+ # numbers candidates with `enumerate(records)` — three of the seven
9107
+ # structural violation kinds hash that number into a DURABLE
9108
+ # fingerprint that `journal_protocol_resolution` ops reference by
9109
+ # name, so contributing the wrong count makes an acknowledged
9110
+ # violation unresolvable (#496 S5b §5.4).
9111
+ decoded.extend([None] * int(summary.decoded_entry_count))
9112
+ last_seen.merge(
9113
+ summary.last_seen_stamped,
9114
+ summary.last_seen_legacy_claude_at,
9115
+ summary.last_seen_legacy_codex_at,
9116
+ )
9117
+ summaries.adopt(summary)
9118
+ elision.quota_gaps.append(
9119
+ (name, len(quota_raw), summary.summarized_size, summary.lines))
9120
+ prior_high_water = (name, int(summary.summarized_size))
9121
+ return True
9122
+
6143
9123
  if hw is not None:
6144
9124
  for segment, offset, raw in _iter_range_with_segments(
6145
9125
  None, hw, segments,
6146
9126
  on_segment=lambda name: hasher.begin_segment(
6147
9127
  name, prior_high_water),
6148
9128
  on_bytes=hasher.extend,
9129
+ on_extent=lambda name, lo, hi, st: summaries.begin(
9130
+ name, lo, hi, st, last_seen),
9131
+ elide=_elide_segment,
6149
9132
  ):
6150
9133
  prefix["lines"] += 1
6151
9134
  prefix["bytes"] += len(raw) + 1
9135
+ summaries.line(raw, offset + len(raw) + 1)
6152
9136
  rec = _lib_journal.decode_line(raw)
6153
9137
  if rec is None:
6154
9138
  malformed += 1
9139
+ summaries.malformed_line()
6155
9140
  prior_high_water = (
6156
9141
  segment,
6157
9142
  offset + len(raw) + 1,
@@ -6170,16 +9155,49 @@ def rebuild_stats_index(
6170
9155
  rec,
6171
9156
  prior_high_water,
6172
9157
  protocol_evidence,
6173
- hasher=hasher,
9158
+ # OPTIMISTIC ELISION, RE-READ FALLBACK (#496 S5b §5.1).
9159
+ # `PrefixHashAccumulator` absorbs completed segments
9160
+ # into ONE sequential sha256 and `hashlib` can neither
9161
+ # export nor restore midstate, so a digest over a prefix
9162
+ # containing an elided segment is not computable from
9163
+ # the bytes this pass read. Handing `None` routes the
9164
+ # digest to `journal_prefix_hash`, which re-reads the
9165
+ # prefix from disk — including the elided segments — and
9166
+ # produces the byte-identical durable digest. The op's
9167
+ # own claimed hash is never accepted without this
9168
+ # recomputation.
9169
+ hasher=None if elision.elided else hasher,
6174
9170
  )
6175
9171
  evidence_seconds += time.monotonic() - evidence_started
9172
+ if (isinstance(rec.get("payload"), dict)
9173
+ and rec["payload"].get("kind")
9174
+ == _lib_journal._PROTOCOL_RESOLUTION_KIND):
9175
+ # Nothing further is elided in this pass. The
9176
+ # accumulator stays non-composable for the rest of it —
9177
+ # a gap already read is still a gap — so every later
9178
+ # evidence point also recomputes from disk; what this
9179
+ # stops is a NEW gap opening after an op that will
9180
+ # certainly be followed by more evidence points.
9181
+ elision.resolution_seen = True
6176
9182
  # First cutover op wins, exactly as `find_accounts_cutover_op`
6177
9183
  # scans — captured here so the rebuild reads the journal once.
6178
9184
  if (cutover_captured is _CUTOVER_UNSEEN
6179
9185
  and rec.get("id") == CUTOVER_OP_ID):
6180
9186
  cutover_captured = _cutover_value_of(rec)
6181
- last_seen.observe(rec, classify_legacy_provider)
6182
- if rec.get("t") in _lib_journal_router.RETAINED_RECORD_TYPES:
9187
+ # Folded into the OPEN SEGMENT's accumulator, which
9188
+ # `_SegmentSummaryCollector.close` merges into `last_seen` at the
9189
+ # boundary. The merge is a per-key maximum, so the resolved map
9190
+ # is identical to a single whole-pass fold (#496 S5b §5.4).
9191
+ summaries.last_seen.observe(rec, classify_legacy_provider)
9192
+ if rec.get("t") == "correction_batch":
9193
+ marker_coordinates[len(decoded)] = (
9194
+ segment,
9195
+ offset + len(raw) + 1,
9196
+ )
9197
+ retained = (
9198
+ rec.get("t") in _lib_journal_router.RETAINED_RECORD_TYPES)
9199
+ summaries.decoded(retained)
9200
+ if retained:
6183
9201
  decoded.append(rec)
6184
9202
  else:
6185
9203
  if update_quota_cache and _is_codex_quota_obs(rec):
@@ -6201,12 +9219,26 @@ def rebuild_stats_index(
6201
9219
  segment,
6202
9220
  offset + len(raw) + 1,
6203
9221
  )
9222
+ summaries.close(last_seen)
9223
+ # §6.3 "recorded, not silent": every elision refusal is a SILENT
9224
+ # fallback, and this block is the only surface that says which one.
9225
+ traversal["elision"] = elision.counters()
6204
9226
  phase_seconds["journal_read_decode"] = round(
6205
9227
  max(0.0, time.monotonic() - read_started - evidence_seconds), 6)
6206
9228
  phase_seconds["protocol_evidence"] = round(evidence_seconds, 6)
6207
9229
  traversal["protocol_evidence"]["bytes"] = hasher.bytes_hashed
6208
9230
  traversal["protocol_evidence"]["lines"] = hasher.digests_computed
6209
9231
  hasher = None
9232
+ # The sidecar is a pure re-derivable cache, so it is refreshed on the
9233
+ # way past rather than guarded by anything: this pass just pinned every
9234
+ # extent it describes, and `write_sidecar` swallows its own I/O failures
9235
+ # because a pass that could not write it is still a correct pass.
9236
+ if summaries.summaries:
9237
+ _lib_segment_summary.write_sidecar(
9238
+ segment_summary_sidecar_path(),
9239
+ [summaries.summaries[name] for name in segments
9240
+ if name in summaries.summaries],
9241
+ )
6210
9242
 
6211
9243
  # Legacy account normalisation (#341, spec §2 / handoff item 2): a
6212
9244
  # pre-#341 real-account line lacks an account stamp — inject the cutover
@@ -6229,9 +9261,29 @@ def rebuild_stats_index(
6229
9261
  # malformed revision, divergent same-revision candidate, or invalid
6230
9262
  # committed manifest leaves the existing destination untouched.
6231
9263
  selection_started = time.monotonic()
9264
+ selector_accumulators: dict = {}
6232
9265
  effective = _lib_journal.resolve_effective_events(
6233
9266
  decoded,
6234
9267
  protocol_prefix_evidence=protocol_evidence,
9268
+ accumulators=selector_accumulators,
9269
+ )
9270
+ # Durable selector state comes from THIS pinned traversal (#496 S5b
9271
+ # §3.3), so publication carries the fold and the index content together
9272
+ # and nothing has to read the journal a second time to derive it. The
9273
+ # generation identity is deliberately absent here: `stats_publication_
9274
+ # stamp` is written after the scratch is built, so a row populated now
9275
+ # cannot carry the identity it will publish under.
9276
+ selector_rows = _lib_selector_state.rows_from_selection(
9277
+ effective,
9278
+ accumulators=selector_accumulators,
9279
+ next_sequence=len(decoded),
9280
+ coordinates=marker_coordinates,
9281
+ covered=hw,
9282
+ cutover_seen=cutover_captured is not _CUTOVER_UNSEEN,
9283
+ cutover_account_key=(
9284
+ None if cutover_captured is _CUTOVER_UNSEEN
9285
+ else cutover_captured
9286
+ ),
6235
9287
  )
6236
9288
  phase_seconds["effective_selection"] = round(
6237
9289
  time.monotonic() - selection_started, 6)
@@ -6243,7 +9295,10 @@ def rebuild_stats_index(
6243
9295
  leg_started = time.monotonic()
6244
9296
  if update_quota_cache:
6245
9297
  quota_lock_hold = _rebuild_quota_cache_leg_raw(
6246
- quota_raw, decoded, cutover_claude, traversal["quota_replay"])
9298
+ quota_raw, decoded, cutover_claude, traversal["quota_replay"],
9299
+ high_water=hw, coverage=quota_coverage,
9300
+ decoded_end=prior_high_water, snapshot_out=quota_snapshot,
9301
+ elision_gaps=elision.quota_gaps)
6247
9302
  quota_raw = []
6248
9303
  phase_seconds["quota_cache_leg"] = round(
6249
9304
  time.monotonic() - leg_started, 6)
@@ -6272,7 +9327,7 @@ def rebuild_stats_index(
6272
9327
  # snapshots, resets+suppression, block_close, arming, credit effects.
6273
9328
  conn.execute("BEGIN IMMEDIATE")
6274
9329
  try:
6275
- _write_effective_metadata(conn, effective)
9330
+ _write_selector_state(conn, selector_rows)
6276
9331
  for _order, _seq, kind, rec in structural:
6277
9332
  if kind == "op":
6278
9333
  FOLD_APPLIERS[(rec.get("payload") or {}).get("kind")](conn, rec)
@@ -6286,11 +9341,44 @@ def rebuild_stats_index(
6286
9341
  except Exception:
6287
9342
  pass
6288
9343
  raise
9344
+ # Reported separately from `stats_fold` because this span is PART of how
9345
+ # long the retained cache read snapshot stays open past the point it
9346
+ # could have been consumed — see the WAL-pinning note in
9347
+ # docs/journal-gotchas.md. Measured at 0.56 s over a 1.6 GB journal, and
9348
+ # that figure is a LOWER BOUND on the pin, not the pin itself: this
9349
+ # timer starts at `fold_started`, which is set after the fold stream is
9350
+ # built and sorted above, and the snapshot opened earlier still inside
9351
+ # `_rebuild_quota_cache_leg_raw`. The full pin is the leg's post-snapshot
9352
+ # span plus the stream build and sort plus this fold. `quota_cache_leg`
9353
+ # is reported separately, so an operator can bound the first term.
9354
+ phase_seconds["structural_fold"] = round(
9355
+ time.monotonic() - fold_started, 6)
9356
+
9357
+ # §4.4's bundle is read from the retained snapshot here, BEFORE phase 2a
9358
+ # and before the stats transaction, so the cache read transaction closes
9359
+ # as early as it can rather than being held across a stats write. An
9360
+ # absent or unreadable snapshot leaves the bundle None, and the
9361
+ # projection reads its own connection exactly as it did before.
9362
+ #
9363
+ # An open WAL read transaction holds a read mark, so
9364
+ # `wal_checkpoint(TRUNCATE)` from any other process returns busy for as
9365
+ # long as this snapshot lives — which disables both of issue #297's
9366
+ # persistent defences. Reading it here rather than after phase 2a is
9367
+ # what bounds that window: measured on a 1.6 GB journal, the structural
9368
+ # fold above is 0.56 s and the open-block projection below is 27.7 s, so
9369
+ # this placement removes about 98% of the pin. The RELATIVE claim is
9370
+ # what those two figures support; neither is the absolute pin, because
9371
+ # the snapshot opens inside the quota cache leg and both timers start
9372
+ # later (see the note on `structural_fold` above). It cannot move any
9373
+ # earlier without holding the bundle's 254 MiB of observations (measured
9374
+ # on the same store, 232,466 rows) across the structural fold as well.
9375
+ quota_bundle = _read_quota_projection_bundle(quota_snapshot)
6289
9376
 
6290
9377
  # Phase 2a — OPEN 5h block projection (own txn; block-only). Closed blocks
6291
9378
  # came from block_close evts; this materializes the never-closed window(s)
6292
9379
  # so the five_hour_milestone block_id derived_fk resolves. Best-effort
6293
9380
  # (the open block is a projection, §5.3).
9381
+ projection_started = time.monotonic()
6294
9382
  try:
6295
9383
  cctally = sys.modules.get("cctally")
6296
9384
  bf = getattr(cctally, "_backfill_five_hour_blocks", None)
@@ -6299,6 +9387,8 @@ def rebuild_stats_index(
6299
9387
  except Exception as exc: # pragma: no cover — projection is best-effort
6300
9388
  print(f"[rebuild] open 5h block re-materialization failed: {exc}",
6301
9389
  file=sys.stderr)
9390
+ phase_seconds["open_block_projection"] = round(
9391
+ time.monotonic() - projection_started, 6)
6302
9392
 
6303
9393
  # Phase 2b + 3 (txn B) — quota projection re-materialization (after the
6304
9394
  # order-45 arming folds) + milestone/budget folds + cursor advance.
@@ -6306,10 +9396,26 @@ def rebuild_stats_index(
6306
9396
  try:
6307
9397
  try:
6308
9398
  import _cctally_quota as _q
6309
- _q.rematerialize_quota_projection_for_rebuild(conn)
9399
+ _q.rematerialize_quota_projection_for_rebuild(
9400
+ conn, bundle=quota_bundle)
6310
9401
  except Exception as exc: # pragma: no cover — projection best-effort
6311
9402
  print(f"[rebuild] quota projection re-materialization failed: {exc}",
6312
9403
  file=sys.stderr)
9404
+ # §4.7's durable, per-transaction gate. It is set in the SAME
9405
+ # transaction that materializes the projection it describes, so a
9406
+ # rollback leaves neither, and it rides into the live index with the
9407
+ # generation's own content — a process-local `RebuildResult` field
9408
+ # could not survive publication, and in-place publication
9409
+ # deliberately keeps already-open readers alive, so a connection can
9410
+ # observe the incomplete generation without ever calling `open_db`
9411
+ # again.
9412
+ _written = _write_quota_projection_state(
9413
+ conn, coverage=quota_coverage, high_water=hw)
9414
+ # `None` is "could not write the flag at all". Fail CLOSED: report
9415
+ # incomplete, so the generation is reconciled rather than served on
9416
+ # the strength of a flag nobody could store.
9417
+ stats_projection_incomplete = (
9418
+ True if _written is None else _written)
6313
9419
  for _order, _seq, _kind, rec in tail:
6314
9420
  _apply_evt(conn, rec)
6315
9421
  lines_folded += 1
@@ -6346,11 +9452,29 @@ def rebuild_stats_index(
6346
9452
  if checkpoint is not None and int(checkpoint[0]) != 0:
6347
9453
  raise JournalError("rebuilt stats index WAL could not be drained")
6348
9454
  _validate_rebuilt_stats_index(conn, hw)
9455
+ # The SEMANTIC half (#496 S5b §6.2). The structural checks above cover
9456
+ # the new tables' existence and definition; without this, a scratch
9457
+ # carrying correct legacy rows and WRONG selector state would validate,
9458
+ # receive a matching generation stamp, and then be trusted by the fast
9459
+ # path — which is the precise failure shape epic #496 exists to
9460
+ # eliminate, reintroduced one layer up. `selector_rows` is the full
9461
+ # selection this pass already derived from the pinned traversal, so the
9462
+ # comparison costs no second journal read.
9463
+ _validate_selector_state(conn, selector_rows)
6349
9464
  _stats_rebuild_test_pause("rebuild_scratch_complete")
6350
9465
  phase_seconds["scratch_validate"] = round(
6351
9466
  time.monotonic() - validate_started, 6)
6352
9467
  finally:
6353
9468
  conn.close()
9469
+ # A retained coverage snapshot holds an open read transaction on
9470
+ # `cache.db`, which pins the WAL against checkpointing for as long as it
9471
+ # lives. `_read_quota_projection_bundle` closes it on every ordinary
9472
+ # path; this covers the paths that raise before reaching it, which
9473
+ # matters most in a long-lived process such as the dashboard's auto-heal
9474
+ # thread. Closing twice is harmless.
9475
+ for _retained in quota_snapshot:
9476
+ _close_coverage_snapshot(_retained)
9477
+ quota_snapshot = []
6354
9478
 
6355
9479
  # Closed, drained, validated, and durable before the old family is touched.
6356
9480
  _remove_db_sidecars_strict(scratch)
@@ -6402,6 +9526,7 @@ def rebuild_stats_index(
6402
9526
  "forensicsPath": context.forensics_path,
6403
9527
  "binaryVersion": _binary_version(),
6404
9528
  "binaryEpoch": _cctally_core.STATS_INDEX_EPOCH,
9529
+ "sqliteRuntimeVersion": sqlite3.sqlite_version,
6405
9530
  "highWater": [hw[0], hw[1]] if hw is not None else None,
6406
9531
  "destination": str(dest),
6407
9532
  "targetPath": str(target_path) if target_path is not None else None,
@@ -6422,6 +9547,14 @@ def rebuild_stats_index(
6422
9547
  },
6423
9548
  "peakHeapBytes": peak_heap_bytes,
6424
9549
  "quotaLockHoldSeconds": round(quota_lock_hold, 6),
9550
+ "selectorDesynchronized": selector_desync,
9551
+ "quotaCacheCoverage": quota_coverage,
9552
+ # §4.7: distinct from `quotaCacheCoverage`, which describes the
9553
+ # CACHE. This describes the PUBLISHED INDEX, and a consumer must not
9554
+ # read one as the other. The record is written before publication
9555
+ # returns, so a crash afterwards still leaves the remainder
9556
+ # discoverable.
9557
+ "statsQuotaProjectionIncomplete": stats_projection_incomplete,
6425
9558
  },
6426
9559
  )
6427
9560
  phase_seconds["publication"] = round(
@@ -6438,6 +9571,9 @@ def rebuild_stats_index(
6438
9571
  traversal=traversal,
6439
9572
  peak_heap_bytes=peak_heap_bytes,
6440
9573
  quota_lock_hold_seconds=round(quota_lock_hold, 6),
9574
+ selector_desynchronized=selector_desync,
9575
+ quota_cache_coverage=quota_coverage,
9576
+ stats_quota_projection_incomplete=stats_projection_incomplete,
6441
9577
  )
6442
9578
 
6443
9579
 
@@ -6706,19 +9842,14 @@ def _cutover_segment_name(now_utc: dt.datetime) -> str:
6706
9842
  return f"{_lib_journal.BOOTSTRAP_PREFIX}{ts}.jsonl"
6707
9843
 
6708
9844
 
6709
- def _write_bootstrap_segment(seg_name: str, lines: list) -> int:
6710
- """Materialize the bootstrap segment atomically (spec §8 rename-then-stamp):
6711
- encode all lines, write to a `.partial` sibling, fsync file + dir, verify the
6712
- line count, then `os.replace` into `seg_name`. Returns the final byte size.
6713
- Every line must fit the torn-tail window (append discipline)."""
6714
- journal_dir = _cctally_core.JOURNAL_DIR
6715
- dir_created = not journal_dir.exists()
6716
- journal_dir.mkdir(parents=True, exist_ok=True)
6717
- if dir_created:
6718
- try:
6719
- os.chmod(journal_dir, 0o700)
6720
- except OSError:
6721
- pass
9845
+ def _encode_bootstrap_lines(lines: list) -> bytes:
9846
+ """The cutover export as one verified blob (spec §8 verify step).
9847
+
9848
+ Encoding is separated from writing because `run_cutover` digests the blob
9849
+ before it decides whether a byte-identical segment already exists (#496 S5
9850
+ §3). Encoding twice would compute the reuse digest over a different object
9851
+ than the one written, so this is the single encode both uses.
9852
+ """
6722
9853
  encoded = []
6723
9854
  for rec in lines:
6724
9855
  data = _lib_journal.encode_line(rec)
@@ -6731,6 +9862,81 @@ def _write_bootstrap_segment(seg_name: str, lines: list) -> int:
6731
9862
  if blob.count(b"\n") != len(lines):
6732
9863
  raise JournalError(
6733
9864
  "cutover export line count mismatch (spec §8 verify step)")
9865
+ return blob
9866
+
9867
+
9868
+ def _reusable_bootstrap(candidate_digest: str, candidate_size: int):
9869
+ """`(name, size)` of a published bootstrap identical to the candidate blob.
9870
+
9871
+ Every published bootstrap is REPORTED, because `reusable_bootstrap_name`
9872
+ refuses a match that is not the canonically newest one; only segments whose
9873
+ byte length already equals the candidate's are READ, so the comparison costs
9874
+ one pass over the same-size candidates rather than one over the journal.
9875
+ `list_segments` excludes `.partial` files, so a cutover that is still writing
9876
+ can never be reused (#496 S5 §3). A segment whose length cannot be read is
9877
+ reported with a `None` length rather than dropped, which refuses reuse
9878
+ instead of promoting an older segment to canonically newest.
9879
+ """
9880
+ journal_dir = _cctally_core.JOURNAL_DIR
9881
+ if not journal_dir.exists():
9882
+ return None
9883
+ existing = []
9884
+ for name in list_segments():
9885
+ if not name.startswith(_lib_journal.BOOTSTRAP_PREFIX):
9886
+ continue
9887
+ path = journal_dir / name
9888
+ try:
9889
+ size = os.path.getsize(path)
9890
+ except OSError:
9891
+ # Report it anyway. Dropping the entry would let an OLDER match
9892
+ # look canonically newest, which is the reuse this scan refuses.
9893
+ existing.append((name, None, None))
9894
+ continue
9895
+ if size != candidate_size:
9896
+ existing.append((name, size, None))
9897
+ continue
9898
+ digest = hashlib.sha256()
9899
+ with _open_segment_for_read(path) as handle:
9900
+ while True:
9901
+ chunk = handle.read(_SEGMENT_READ_CHUNK)
9902
+ if not chunk:
9903
+ break
9904
+ digest.update(chunk)
9905
+ existing.append((name, size, digest.hexdigest()))
9906
+ name = _lib_journal.reusable_bootstrap_name(
9907
+ candidate_digest, candidate_size, existing)
9908
+ return None if name is None else (name, candidate_size)
9909
+
9910
+
9911
+ def _fsync_published_segment(seg_name: str) -> None:
9912
+ """Make an already-renamed segment and its directory entry durable.
9913
+
9914
+ `_write_bootstrap_segment` establishes this for a segment it writes itself.
9915
+ A reused segment was published by a different attempt, which may have
9916
+ crashed anywhere in that sequence, so the reuse path repeats the file and
9917
+ directory fsyncs before anything durable is allowed to name the file.
9918
+ """
9919
+ journal_dir = _cctally_core.JOURNAL_DIR
9920
+ fd = os.open(str(journal_dir / seg_name), os.O_RDONLY)
9921
+ try:
9922
+ os.fsync(fd)
9923
+ finally:
9924
+ os.close(fd)
9925
+ _fsync_dir(journal_dir)
9926
+
9927
+
9928
+ def _write_bootstrap_segment(seg_name: str, blob: bytes) -> int:
9929
+ """Materialize the bootstrap segment atomically (spec §8 rename-then-stamp):
9930
+ write the verified blob to a `.partial` sibling, fsync file + dir, then
9931
+ `os.replace` into `seg_name`. Returns the final byte size."""
9932
+ journal_dir = _cctally_core.JOURNAL_DIR
9933
+ dir_created = not journal_dir.exists()
9934
+ journal_dir.mkdir(parents=True, exist_ok=True)
9935
+ if dir_created:
9936
+ try:
9937
+ os.chmod(journal_dir, 0o700)
9938
+ except OSError:
9939
+ pass
6734
9940
  partial = journal_dir / (seg_name + ".partial")
6735
9941
  fd = os.open(str(partial), os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
6736
9942
  try:
@@ -6759,7 +9965,14 @@ def run_cutover(conn, *, now_utc: dt.datetime | None = None) -> "str | None":
6759
9965
  commit rolls the whole thing back (the legacy DB stays fully usable); the
6760
9966
  next open retries idempotently (stable bootstrap ids). A truly empty install
6761
9967
  (nothing to export) just stamps the epoch — no bootstrap file. Returns the
6762
- bootstrap segment basename, or None when nothing was exported."""
9968
+ bootstrap segment basename, or None when nothing was exported.
9969
+
9970
+ A crash between the `os.replace` and the commit leaves a published segment
9971
+ the rolled-back transaction never referenced. The retry re-exports the same
9972
+ rows byte for byte, so it reuses that orphan rather than publishing a twin
9973
+ (#496 S5 §3) — the retry is now idempotent on disk, not only on fold. When
9974
+ the retry's export genuinely differs, no digest matches and a new segment is
9975
+ written exactly as before."""
6763
9976
  if now_utc is None:
6764
9977
  now_utc = dt.datetime.now(dt.timezone.utc)
6765
9978
  epoch = _cctally_core.STATS_INDEX_EPOCH
@@ -6781,8 +9994,19 @@ def run_cutover(conn, *, now_utc: dt.datetime | None = None) -> "str | None":
6781
9994
  conn.commit()
6782
9995
  return None
6783
9996
 
6784
- seg_name = _cutover_segment_name(now_utc)
6785
- seg_size = _write_bootstrap_segment(seg_name, lines)
9997
+ blob = _encode_bootstrap_lines(lines)
9998
+ reuse = _reusable_bootstrap(
9999
+ hashlib.sha256(blob).hexdigest(), len(blob))
10000
+ if reuse is None:
10001
+ seg_name = _cutover_segment_name(now_utc)
10002
+ seg_size = _write_bootstrap_segment(seg_name, blob)
10003
+ else:
10004
+ seg_name, seg_size = reuse
10005
+ # The adopted segment was renamed by ANOTHER attempt, whose rename
10006
+ # may still be only in the page cache. The cursor stamped below is
10007
+ # made durable by SQLite's own commit fsync, so without this the
10008
+ # index could name a bootstrap that a power loss then leaves absent.
10009
+ _fsync_published_segment(seg_name)
6786
10010
 
6787
10011
  for table, rowid in stamp:
6788
10012
  conn.execute(