cctally 1.82.0 → 1.83.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/CHANGELOG.md +70 -0
  2. package/README.md +52 -74
  3. package/bin/_cctally_alerts.py +8 -1
  4. package/bin/_cctally_cache.py +963 -149
  5. package/bin/_cctally_config.py +43 -4
  6. package/bin/_cctally_core.py +933 -759
  7. package/bin/_cctally_dashboard.py +157 -47
  8. package/bin/_cctally_dashboard_cache_report.py +13 -6
  9. package/bin/_cctally_dashboard_conversation.py +1 -0
  10. package/bin/_cctally_dashboard_envelope.py +186 -8
  11. package/bin/_cctally_dashboard_share.py +60 -20
  12. package/bin/_cctally_dashboard_sources.py +427 -128
  13. package/bin/_cctally_db.py +605 -128
  14. package/bin/_cctally_doctor.py +413 -28
  15. package/bin/_cctally_five_hour.py +12 -5
  16. package/bin/_cctally_journal.py +2050 -156
  17. package/bin/_cctally_journal_repair.py +519 -0
  18. package/bin/_cctally_milestone_history.py +142 -56
  19. package/bin/_cctally_milestones.py +179 -111
  20. package/bin/_cctally_parser.py +42 -0
  21. package/bin/_cctally_project.py +24 -18
  22. package/bin/_cctally_quota.py +139 -25
  23. package/bin/_cctally_record.py +279 -108
  24. package/bin/_cctally_rederive.py +1052 -0
  25. package/bin/_cctally_reporting.py +58 -53
  26. package/bin/_cctally_setup.py +1 -0
  27. package/bin/_cctally_source_analytics.py +4 -1
  28. package/bin/_cctally_statusline.py +11 -11
  29. package/bin/_cctally_store.py +1039 -31
  30. package/bin/_cctally_sync_week.py +17 -8
  31. package/bin/_cctally_tui.py +421 -54
  32. package/bin/_cctally_update.py +133 -8
  33. package/bin/_cctally_weekrefs.py +14 -0
  34. package/bin/_lib_aggregators.py +10 -6
  35. package/bin/_lib_cache_report.py +101 -9
  36. package/bin/_lib_codex_pools.py +82 -0
  37. package/bin/_lib_conversation_query.py +126 -33
  38. package/bin/_lib_dashboard_sources.py +126 -1
  39. package/bin/_lib_diff_kernel.py +28 -15
  40. package/bin/_lib_doctor.py +342 -4
  41. package/bin/_lib_journal.py +924 -2
  42. package/bin/_lib_jsonl.py +43 -14
  43. package/bin/_lib_pricing.py +140 -21
  44. package/bin/_lib_readme_refresh.py +401 -0
  45. package/bin/_lib_rederive.py +395 -0
  46. package/bin/_lib_share.py +58 -2
  47. package/bin/cctally +56 -8
  48. package/dashboard/static/assets/{index-DJP4gEB7.js → index-3bgCMVHb.js} +52 -52
  49. package/dashboard/static/assets/index-D27EIHEI.css +1 -0
  50. package/dashboard/static/dashboard.html +2 -2
  51. package/package.json +6 -1
  52. package/dashboard/static/assets/index-Dk1nplOz.css +0 -1
@@ -40,14 +40,31 @@ provider flocks (Claude → Codex) → SQLite transactions → ``journal.lock``
40
40
  flock; no SQLite write transaction ever spans a flock acquisition.
41
41
 
42
42
  **Raw-connect escape hatches stay OUT of this module by design** (spec §6.1):
43
- ``db checkpoint``'s ``mode=rw`` connect, ``db vacuum``'s exclusive connect, and
44
- doctor's read-only gather deliberately bypass the opener so they carry no
45
- schema-apply / migration side effects on maintenance/diagnostic paths.
43
+ ``db checkpoint``'s ``mode=rw`` connect and ``db vacuum``'s exclusive connect
44
+ deliberately bypass ``open_index``/``open_db`` so they carry no schema-apply /
45
+ migration side effects on maintenance paths.
46
+
47
+ **#386 narrowed that carve-out for stats.** Skipping the *schema apply* is not
48
+ the same as skipping the *opener protocol*: spec §3.1's third clause requires
49
+ every opener of the live stats family to observe the repair marker and the
50
+ quarantine-pending record under maintenance-SHARED across connect. Doctor's
51
+ read-write probes (``bin/_cctally_doctor.py``), ``db backup --db stats``'s
52
+ ``mode=ro`` source, and ``_db_status_for``'s status connect therefore all route
53
+ through ``stats_open_guarded`` with their OWN ``connect`` callable — they keep
54
+ their open mode and their freedom from schema side effects while still
55
+ participating. The claim that doctor's gather "bypasses the opener" is no longer
56
+ true and must not be restored.
46
57
  """
47
58
  from __future__ import annotations
48
59
 
60
+ import contextlib
61
+ import datetime as dt
49
62
  import fcntl
63
+ import json
50
64
  import os
65
+ import pathlib
66
+ import re
67
+ import signal
51
68
  import sqlite3
52
69
  import sys
53
70
  import time
@@ -323,7 +340,858 @@ def mark_stats_open_fixups_done(conn: sqlite3.Connection) -> None:
323
340
  _HEAL_ACTIVE = False
324
341
 
325
342
 
343
+ # --------------------------------------------------------------------------
344
+ # #386 sanctioned-write context
345
+ # --------------------------------------------------------------------------
346
+ #
347
+ # A stats.db mutation is legal only inside this scope, entered by the ingester
348
+ # while it holds ``journal.ingest.lock`` and by maintenance paths while they hold
349
+ # ``stats.db.maintenance.lock`` (spec section 3.1's three regimes).
350
+ #
351
+ # A ``ContextVar``, NOT a module global. The dashboard is threaded, so a
352
+ # process-global boolean would let one sanctioned thread authorize an unrelated
353
+ # one — precisely the false-positive the review rejected the trace-callback
354
+ # design over. ``ContextVar`` values do not propagate into a ``threading.Thread``
355
+ # started from inside the scope, which is the property under test in
356
+ # ``tests/test_stats_writer_guard_386.py::test_scope_does_not_leak_across_threads``.
357
+ #
358
+ # ``holds_ingest_lock()`` is deliberately NARROWER than ``in_stats_write_scope()``
359
+ # and is not implied by it: maintenance paths are sanctioned writers that do NOT
360
+ # hold the ingest lock. The heal path keys its self-deadlock avoidance on the
361
+ # narrow fact (see ``_heal_flock_bounded``), so conflating the two would let a
362
+ # maintenance caller skip an ingest acquire it never made.
363
+
364
+ # The two ContextVars live in `_cctally_core` (see the block beside
365
+ # `holds_stats_maintenance`): `tests/conftest.py`'s `load_script()` reloads
366
+ # every `_cctally_*` sibling but never the kernel, so state kept here would be
367
+ # silently reset mid-test and a sanctioned write would then be denied.
368
+ _STATS_WRITE_SCOPE = _cctally_core._STATS_WRITE_SCOPE
369
+ _INGEST_LOCK_HELD = _cctally_core._STATS_INGEST_LOCK_HELD
370
+ _INTERRUPTED_RECOVERY_SUPPRESSED = (
371
+ _cctally_core._STATS_INTERRUPTED_RECOVERY_SUPPRESSED
372
+ )
373
+
374
+
375
+ def in_stats_write_scope() -> bool:
376
+ """True when THIS execution context is inside a sanctioned stats-write scope."""
377
+ return _STATS_WRITE_SCOPE.get() > 0
378
+
379
+
380
+ def holds_ingest_lock() -> bool:
381
+ """True when THIS execution context already holds ``journal.ingest.lock``.
382
+
383
+ Used by the heal path to distinguish "I am the serialized writer" from
384
+ "someone else holds it": a corruption surfacing from inside a
385
+ ``run_stats_ingest`` cycle already owns the lock, so waiting for it would
386
+ self-deadlock, while any other caller must genuinely wait or decline.
387
+ """
388
+ return _INGEST_LOCK_HELD.get() > 0
389
+
390
+
391
+ @contextlib.contextmanager
392
+ def suppress_interrupted_stats_recovery():
393
+ """Keep every nested Doctor stats opener read-only in this context."""
394
+ token = _INTERRUPTED_RECOVERY_SUPPRESSED.set(
395
+ _INTERRUPTED_RECOVERY_SUPPRESSED.get() + 1
396
+ )
397
+ try:
398
+ yield
399
+ finally:
400
+ _INTERRUPTED_RECOVERY_SUPPRESSED.reset(token)
401
+
402
+
403
+ @contextlib.contextmanager
404
+ def stats_write_scope(reason: str, *, ingest_lock: bool = False):
405
+ """Mark the enclosed block as a sanctioned stats.db writer.
406
+
407
+ ``reason`` is diagnostic only (it names the regime for the Stage 3 guard log).
408
+ ``ingest_lock=True`` additionally asserts that the caller holds
409
+ ``journal.ingest.lock`` for the duration. Nests; restored on exception via
410
+ the ``ContextVar`` tokens, so an unwinding error never leaves the process
411
+ permanently sanctioned.
412
+ """
413
+ depth = _STATS_WRITE_SCOPE.set(_STATS_WRITE_SCOPE.get() + 1)
414
+ held = _INGEST_LOCK_HELD.set(
415
+ _INGEST_LOCK_HELD.get() + (1 if ingest_lock else 0)
416
+ )
417
+ try:
418
+ yield reason
419
+ finally:
420
+ _INGEST_LOCK_HELD.reset(held)
421
+ _STATS_WRITE_SCOPE.reset(depth)
422
+
423
+
424
+ # --------------------------------------------------------------------------
425
+ # #386 enforcement — the stats sole-writer authorizer (spec §6.1)
426
+ # --------------------------------------------------------------------------
427
+ #
428
+ # The mechanism is ``Connection.set_authorizer``, NOT ``set_trace_callback``.
429
+ # Two independent reasons, both verified rather than assumed:
430
+ #
431
+ # 1. Python SUPPRESSES exceptions raised inside a trace callback. A trace hook
432
+ # that raises on INSERT does NOT prevent the write — the row commits and
433
+ # `SELECT count(*)` returns 1. An authorizer returning SQLITE_DENY blocks it
434
+ # (count 0). A mechanism that cannot stop the write is diagnostics, not
435
+ # enforcement.
436
+ # 2. `_TRACE_HOOK` is only installed by `open_index`, which the stats openers
437
+ # do not go through.
438
+ #
439
+ # Action CODES, never SQL text. A text classifier mishandles DDL, dynamic table
440
+ # names (`UPDATE {table}` in cutover and in eight journal folds), CTEs, and
441
+ # comments — and the mutation inventory found 15 dynamic-SQL sites a lexical
442
+ # scan cannot resolve at all.
443
+ #
444
+ # Scoped to the ``main`` schema. Stats connections legitimately create TEMP
445
+ # VIEWs outside any write scope (`bin/_cctally_tui.py`), and the dashboard/TUI
446
+ # build TEMP views over an ATTACHed cache.db. Those report `temp`/the attach
447
+ # alias as `db_name` and are none of this guard's business.
448
+ #
449
+ # NOT covered, and deliberately so: `PRAGMA user_version = …`, `VACUUM`,
450
+ # `os.replace`/`rename`/`unlink`. The first two would require guarding
451
+ # SQLITE_PRAGMA (which every `apply_policy` call trips) and the last three are
452
+ # invisible to every SQLite hook. 12 of the 14 physical mutation sites are in
453
+ # that last class — they are covered by the opener protocol and the lock
454
+ # corrections instead. Enforcement and serialization do different jobs here and
455
+ # neither substitutes for the other.
456
+
457
+ _GUARD_MUTATIONS = frozenset({
458
+ sqlite3.SQLITE_INSERT,
459
+ sqlite3.SQLITE_UPDATE,
460
+ sqlite3.SQLITE_DELETE,
461
+ sqlite3.SQLITE_CREATE_TABLE,
462
+ sqlite3.SQLITE_CREATE_INDEX,
463
+ sqlite3.SQLITE_DROP_TABLE,
464
+ sqlite3.SQLITE_DROP_INDEX,
465
+ sqlite3.SQLITE_ALTER_TABLE,
466
+ })
467
+
468
+ #: One line per throttle window across all processes. Rotation is still the
469
+ #: hard disk-growth bound if the marker is removed or the throttle is disabled.
470
+ _GUARD_THROTTLE_S = 60.0
471
+ _GUARD_LOG_ROTATE_BYTES = 1024 * 1024
472
+ _guard_last_logged = 0.0
473
+
474
+
475
+ def _guard_log_path() -> pathlib.Path:
476
+ """``logs/stats-writer-guard.log`` — the doctor leg's input (spec §6.4)."""
477
+ return pathlib.Path(_cctally_core.HOOK_TICK_LOG_DIR) / "stats-writer-guard.log"
478
+
479
+
480
+ def _guard_rotated_log_path() -> pathlib.Path:
481
+ path = _guard_log_path()
482
+ return path.with_name(path.name + ".1")
483
+
484
+
485
+ def _guard_log_lock_path() -> pathlib.Path:
486
+ path = _guard_log_path()
487
+ return path.with_name(path.name + ".lock")
488
+
489
+
490
+ def _guard_throttle_path() -> pathlib.Path:
491
+ path = _guard_log_path()
492
+ return path.with_name(path.name + ".last")
493
+
494
+
495
+ def _guard_rotate_if_needed(path: pathlib.Path) -> bool:
496
+ """Keep one rotated generation; return False when rotation cannot proceed."""
497
+ try:
498
+ size = path.stat().st_size
499
+ except FileNotFoundError:
500
+ return True
501
+ except OSError:
502
+ return False
503
+ if size < _GUARD_LOG_ROTATE_BYTES:
504
+ return True
505
+ try:
506
+ os.replace(path, _guard_rotated_log_path())
507
+ except OSError:
508
+ return False
509
+ return True
510
+
511
+
512
+ def _log_unsanctioned_write(action, table) -> None:
513
+ global _guard_last_logged
514
+ now_monotonic = time.monotonic()
515
+ if (
516
+ _GUARD_THROTTLE_S > 0
517
+ and _guard_last_logged
518
+ and now_monotonic - _guard_last_logged < _GUARD_THROTTLE_S
519
+ ):
520
+ return
521
+ lock_fd = None
522
+ try:
523
+ path = _guard_log_path()
524
+ path.parent.mkdir(parents=True, exist_ok=True)
525
+ lock_fd = os.open(
526
+ _guard_log_lock_path(), os.O_RDWR | os.O_CREAT, 0o600)
527
+ # This diagnostic must never park the command whose write triggered it.
528
+ # One storm participant records the incident; contenders fail soft.
529
+ fcntl.flock(lock_fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
530
+
531
+ marker = _guard_throttle_path()
532
+ if _GUARD_THROTTLE_S > 0:
533
+ try:
534
+ marker_age = time.time() - marker.stat().st_mtime
535
+ except FileNotFoundError:
536
+ marker_age = _GUARD_THROTTLE_S
537
+ if 0 <= marker_age < _GUARD_THROTTLE_S:
538
+ _guard_last_logged = now_monotonic
539
+ return
540
+
541
+ if not _guard_rotate_if_needed(path):
542
+ return
543
+ line = (
544
+ f"{_cctally_core.now_utc_iso()}\tunsanctioned stats write\t"
545
+ f"action={action}\ttable={table}\n"
546
+ ).encode("utf-8", errors="replace")
547
+ log_fd = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_APPEND, 0o600)
548
+ try:
549
+ os.write(log_fd, line)
550
+ finally:
551
+ os.close(log_fd)
552
+ marker.touch(mode=0o600, exist_ok=True)
553
+ _guard_last_logged = now_monotonic
554
+ except OSError:
555
+ pass
556
+ finally:
557
+ if lock_fd is not None:
558
+ try:
559
+ fcntl.flock(lock_fd, fcntl.LOCK_UN)
560
+ except OSError:
561
+ pass
562
+ os.close(lock_fd)
563
+
564
+
565
+ def _guard_should_raise() -> bool:
566
+ """Raise on a dev checkout or under pytest; log-only on installed builds.
567
+
568
+ ``_is_dev_checkout()``, deliberately NOT ``DEV_MODE`` — CLAUDE.md forbids
569
+ collapsing those two predicates, and this is the checkout question, not the
570
+ verbosity one.
571
+ """
572
+ return bool(
573
+ _cctally_core._is_dev_checkout()
574
+ or os.environ.get("PYTEST_CURRENT_TEST")
575
+ )
576
+
577
+
578
+ def arm_stats_authorizer(conn: sqlite3.Connection) -> None:
579
+ """Deny mutations of the stats ``main`` schema outside a sanctioned scope."""
580
+
581
+ def _auth(action, arg1, arg2, db_name, trigger):
582
+ if action not in _GUARD_MUTATIONS:
583
+ return sqlite3.SQLITE_OK
584
+ if db_name not in (None, "main"):
585
+ return sqlite3.SQLITE_OK
586
+ if in_stats_write_scope():
587
+ return sqlite3.SQLITE_OK
588
+ if _guard_should_raise():
589
+ return sqlite3.SQLITE_DENY
590
+ _log_unsanctioned_write(action, arg1)
591
+ return sqlite3.SQLITE_OK
592
+
593
+ conn.set_authorizer(_auth)
594
+
595
+
596
+ # --------------------------------------------------------------------------
597
+ # #386 the open-time mutation regime (spec §3.1's second clause, Gaps C/GAP-1..4)
598
+ # --------------------------------------------------------------------------
599
+ #
600
+ # `open_db`'s post-gate body runs the full schema DDL, the quota-projection
601
+ # schema, the migration dispatcher, two backfills, the fixups marker and the
602
+ # in-place cutover. Every one of those is a mutation, and before #386 they ran
603
+ # under NO lock whatever command reached them — 57 production `open_db` call
604
+ # sites, any of which could be racing another. Spec §3.1: "First-open, legacy,
605
+ # epoch and administrative mutation … must acquire stats.db.maintenance.lock
606
+ # (exclusive) BEFORE mutating, regardless of the command that reached them."
607
+ #
608
+ # Re-entrant on the same signal the opener uses, because the common case is
609
+ # reaching here from a caller that ALREADY holds the lock (`run_stats_ingest`'s
610
+ # legacy branch takes it exclusive before `open_db()`; `rebuild_stats_index`
611
+ # opens its scratch under the rebuild's hold).
612
+
613
+ #: Open-time mutation is a one-shot path (first open / legacy / post-rebuild),
614
+ #: not the steady-state hot path — the epoch gate returns before it. So the wait
615
+ #: is generous where the opener's is short. On expiry we DECLINE; proceeding
616
+ #: unlocked would be the "described itself as serialized while running
617
+ #: unserialized" defect this session removed from the heal path.
618
+ _STATS_OPEN_TIME_MAINTENANCE_WAIT_S = 30.0
619
+
620
+
621
+ @contextlib.contextmanager
622
+ def stats_open_time_guard(*, live: bool = True):
623
+ """Hold maintenance-exclusive + the sanctioned scope across open-time DDL.
624
+
625
+ ``live=False`` marks a ``_target_path`` (scratch) build: it enters the
626
+ sanctioned scope but takes NO flock, mirroring the divergence
627
+ ``stats_open_guarded`` already documents. Two reasons, both concrete:
628
+
629
+ 1. A scratch index is not the live family, so serializing it against live
630
+ maintenance buys nothing — and every scratch build already runs under a
631
+ HELD maintenance exclusive (rebuild, rederive) or against a private
632
+ temp copy (`db rederive`'s preview snapshot). Taking the LIVE lock on a
633
+ second fd from inside a held exclusive is the self-deadlock
634
+ `holds_stats_maintenance` exists to prevent.
635
+ 2. Creating `stats.db.maintenance.lock` is itself a persistent side
636
+ effect. `db rederive`'s preview has a literal zero-persistent-write
637
+ contract ("without creating any coordination files"), pinned by
638
+ `tests/test_rederive_command.py::test_preview_is_write_free_…`, and its
639
+ snapshot open would otherwise mint that file in the real APP_DIR.
640
+ """
641
+ if not live or _cctally_core.holds_stats_maintenance():
642
+ with stats_write_scope("open-time"):
643
+ yield
644
+ return
645
+ _cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
646
+ fd = os.open(
647
+ str(_cctally_core.STATS_LOCK_MAINTENANCE_PATH),
648
+ os.O_RDWR | os.O_CREAT, 0o600,
649
+ )
650
+ acquired = False
651
+ try:
652
+ deadline = time.monotonic() + _STATS_OPEN_TIME_MAINTENANCE_WAIT_S
653
+ while True:
654
+ try:
655
+ fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
656
+ acquired = True
657
+ break
658
+ except (BlockingIOError, OSError):
659
+ if time.monotonic() >= deadline:
660
+ raise _cctally_db.StatsDbMaintenanceError(
661
+ _STATS_OPEN_MAINTENANCE_TIMEOUT_MSG
662
+ )
663
+ time.sleep(0.02)
664
+ _cctally_core.note_stats_maintenance_acquired()
665
+ try:
666
+ with stats_write_scope("open-time"):
667
+ yield
668
+ finally:
669
+ _cctally_core.note_stats_maintenance_released()
670
+ finally:
671
+ if acquired:
672
+ try:
673
+ fcntl.flock(fd, fcntl.LOCK_UN)
674
+ except OSError:
675
+ pass
676
+ os.close(fd)
677
+
678
+
679
+ # --------------------------------------------------------------------------
680
+ # #386 opener half of the physical-replacement protocol
681
+ # --------------------------------------------------------------------------
682
+ #
683
+ # Spec section 3.1, third clause: EVERY opener of the live stats family --
684
+ # read-only consumers included -- observes the repair marker and the
685
+ # quarantine-pending record under maintenance-SHARED, held across the marker
686
+ # checks AND the connect. That is what makes the pending record's claim to
687
+ # "block every opener" true, and it is the half stats never had: `open_db`
688
+ # checked `stats.db.repairing` three times around a bare `sqlite3.connect` and
689
+ # consulted no pending record at all, so a destructive maintenance path could
690
+ # publish its record, scan for handles, and still have a brand-new opener arrive
691
+ # in the window before the first rename (spec section 1.1 Gap A's TOCTOU).
692
+ #
693
+ # Modelled directly on `bin/_cctally_cache.py::_cache_open_guarded`. Two
694
+ # deliberate divergences from the cache version, both narrowing:
695
+ #
696
+ # 1. Repair-marker STALENESS reclaim is NOT ported. The cache upgrades to
697
+ # exclusive, asks `_repair_marker_is_live`, and reclaims a dead owner's
698
+ # marker. Stats has always simply raised, and every existing stats
699
+ # regression asserts that. Changing it is a behaviour change with its own
700
+ # test surface and belongs to whoever owns `db repair`, not to a
701
+ # corruption-prevention pass. Consequence, recorded rather than fixed: a
702
+ # SIGKILLed `db repair --db stats` still strands its marker.
703
+ # 2. `_target_path` (scratch) opens skip the flock entirely and keep the
704
+ # pre-#386 marker-only behaviour. A scratch index is not the live family,
705
+ # and every scratch open happens under a HELD maintenance exclusive
706
+ # (rebuild, rederive) -- taking shared on a second fd there is the
707
+ # self-deadlock described in `_cctally_core.holds_stats_maintenance`.
708
+
709
+
710
+ #: Bounded wait for ``stats.db.maintenance.lock`` on the OPENER path (#386).
711
+ #
712
+ # Stage 2 took this flock UNTIMED, on the path every command uses — every
713
+ # `statusline` render and every detached `hook-tick`. Stage 2 also made the
714
+ # exclusive holders long: `db rebuild` replays the whole journal, `db vacuum`
715
+ # rewrites the file, `db rederive --yes` runs a full scratch replay, `db repair`
716
+ # shells out to `sqlite3 .recover`. During any of them EVERY stats open parked
717
+ # forever — the #297-class `database is locked` stall that spec §5.2 ground 4
718
+ # rejected the checkpoint policy over, reintroduced by the corruption fix.
719
+ #
720
+ # Bounding it is strictly safe: the TOCTOU guarantee is "you cannot open WHILE
721
+ # exclusive is held", and failing after a timeout preserves it — we simply
722
+ # decline instead of waiting. Every caller already handles the exception
723
+ # (`main()` maps it to exit 3, `doctor` degrades, `cmd_db_status` reports
724
+ # `_open_error`).
725
+ #
726
+ # 5 s is chosen against the two failure directions: it is well under the 15 s
727
+ # `busy_timeout` the DB already tolerates (so the opener is never the slowest
728
+ # thing on the hot path), and it is orders of magnitude above the handshake it
729
+ # must NOT trip — marker publish + drain scan + rename is milliseconds.
730
+ _STATS_OPEN_MAINTENANCE_WAIT_S = 5.0
731
+
732
+ #: How long the opener will wait to UPGRADE to exclusive to resume a pending
733
+ #: quarantine. Same reasoning; kept separate so the two can diverge.
734
+ _STATS_OPEN_RESUME_WAIT_S = 5.0
735
+
736
+ _STATS_OPEN_MAINTENANCE_TIMEOUT_MSG = (
737
+ "stats.db maintenance is in progress; retry after the maintenance command "
738
+ "exits"
739
+ )
740
+
741
+
742
+ def _flock_bounded(lock_fh, operation: int, timeout_s: float) -> bool:
743
+ """Poll for ``operation`` on ``lock_fh`` until ``timeout_s`` expires.
744
+
745
+ ``True`` when the lock is held, ``False`` on expiry (nothing acquired, so
746
+ the caller must not release). Same shape as ``_heal_flock_bounded``, but it
747
+ operates on an already-open file object rather than minting an fd, because
748
+ the opener holds one file object across its whole marker/pending/connect
749
+ sequence.
750
+ """
751
+ deadline = time.monotonic() + timeout_s
752
+ while True:
753
+ try:
754
+ fcntl.flock(lock_fh, operation | fcntl.LOCK_NB)
755
+ return True
756
+ except (BlockingIOError, OSError):
757
+ if time.monotonic() >= deadline:
758
+ return False
759
+ time.sleep(0.02)
760
+
761
+
762
+ def _stats_repair_marker(db_path) -> pathlib.Path:
763
+ """The live stats repair marker, whatever `db_path` is.
764
+
765
+ Deliberately `with_name`, not `<db_path>.repairing`: a scratch/rebuild
766
+ target sitting beside the live DB must still observe the LIVE marker, which
767
+ is the pre-#386 behaviour this preserves byte for byte.
768
+ """
769
+ return pathlib.Path(db_path).with_name("stats.db.repairing")
770
+
771
+
772
+ def _resume_pending_quarantine(db_path: pathlib.Path) -> None:
773
+ """Finish a strict quarantine that a previous owner did not complete.
774
+
775
+ Caller holds maintenance EXCLUSIVE. Fails closed: if any handle on the
776
+ family is still open -- or the platform cannot tell us -- we refuse rather
777
+ than rename files out from under a live mapping, which is precisely the
778
+ "cctally performs file-family surgery with live mappings" class that spec
779
+ section 1.2 identifies as where SQLite's crash guarantees stop applying.
780
+ """
781
+ try:
782
+ open_pids = _cctally_db._db_family_open_pids(db_path)
783
+ if open_pids is None:
784
+ raise OSError(
785
+ "could not verify that the database family has no open handles"
786
+ )
787
+ if open_pids:
788
+ raise OSError(
789
+ "database family is still open in process(es) "
790
+ + ", ".join(str(pid) for pid in sorted(open_pids))
791
+ )
792
+ # Resumes the SAME incident from the pending record -- never a second
793
+ # incident dir, never a recreation.
794
+ _cctally_db.quarantine_db_family(db_path, strict=True)
795
+ except OSError as exc:
796
+ raise sqlite3.OperationalError(
797
+ f"stats.db pending quarantine could not resume: {exc}"
798
+ ) from exc
799
+
800
+
801
+ _STATS_REBUILD_ARTIFACT_RE = re.compile(
802
+ r"^(?P<base>stats\.db\.rebuilding-\d{8}T\d{6}_\d{6})"
803
+ r"(?P<sidecar>-wal|-shm)?$"
804
+ )
805
+ _STATS_QUARANTINE_INCIDENT_RE = re.compile(
806
+ r"^stats\.db-(?:\d{8}T\d{6}Z|\d{8}T\d{6}_\d{6})$"
807
+ )
808
+
809
+
810
+ def _stats_rebuild_artifact_bases(db_path: pathlib.Path) -> tuple[pathlib.Path, ...]:
811
+ """Return only Task A's exact scratch-family bases beside ``db_path``."""
812
+ bases: set[pathlib.Path] = set()
813
+ for candidate in db_path.parent.glob(f"{db_path.name}.rebuilding-*"):
814
+ match = _STATS_REBUILD_ARTIFACT_RE.fullmatch(candidate.name)
815
+ if match is not None:
816
+ bases.add(db_path.parent / match.group("base"))
817
+ return tuple(sorted(bases))
818
+
819
+
820
+ def _rebuild_artifact_time(path: pathlib.Path) -> "dt.datetime | None":
821
+ match = _STATS_REBUILD_ARTIFACT_RE.fullmatch(path.name)
822
+ if match is None:
823
+ return None
824
+ try:
825
+ return dt.datetime.strptime(
826
+ match.group("base").removeprefix("stats.db.rebuilding-"),
827
+ "%Y%m%dT%H%M%S_%f",
828
+ ).replace(tzinfo=dt.timezone.utc)
829
+ except ValueError:
830
+ return None
831
+
832
+
833
+ def _quarantine_incident_time(path: pathlib.Path) -> "dt.datetime | None":
834
+ timestamp = path.name.removeprefix("stats.db-")
835
+ try:
836
+ if timestamp.endswith("Z"):
837
+ return dt.datetime.strptime(timestamp, "%Y%m%dT%H%M%SZ").replace(
838
+ tzinfo=dt.timezone.utc
839
+ )
840
+ return dt.datetime.strptime(timestamp, "%Y%m%dT%H%M%S_%f").replace(
841
+ tzinfo=dt.timezone.utc
842
+ )
843
+ except ValueError:
844
+ return None
845
+
846
+
847
+ def _has_completed_stats_quarantine_incident(
848
+ db_path: pathlib.Path, artifacts: tuple[pathlib.Path, ...]
849
+ ) -> bool:
850
+ """Positive evidence that the same legacy rebuild removed the live family."""
851
+ root = _cctally_core.APP_DIR / "quarantine"
852
+ if not root.is_dir():
853
+ return False
854
+ artifact_times = tuple(
855
+ timestamp
856
+ for path in artifacts
857
+ if (timestamp := _rebuild_artifact_time(path)) is not None
858
+ )
859
+ if not artifact_times:
860
+ return False
861
+ for manifest_path in sorted(root.glob(f"{db_path.name}-*/manifest.json")):
862
+ if _STATS_QUARANTINE_INCIDENT_RE.fullmatch(
863
+ manifest_path.parent.name
864
+ ) is None:
865
+ continue
866
+ incident_time = _quarantine_incident_time(manifest_path.parent)
867
+ if incident_time is None or not any(
868
+ dt.timedelta(0) <= artifact_time - incident_time <= dt.timedelta(minutes=5)
869
+ for artifact_time in artifact_times
870
+ ):
871
+ continue
872
+ try:
873
+ manifest = json.loads(manifest_path.read_text())
874
+ except (OSError, json.JSONDecodeError):
875
+ continue
876
+ if (
877
+ manifest.get("complete") is True
878
+ and manifest.get("originalPath") == str(db_path)
879
+ and db_path.name in (manifest.get("movedFiles") or ())
880
+ ):
881
+ return True
882
+ return False
883
+
884
+
885
+ def stats_interrupted_rebuild_evidence(
886
+ db_path: pathlib.Path,
887
+ ) -> "dict | None":
888
+ """Read-only classification for Doctor; never creates or reclaims files."""
889
+ db_path = pathlib.Path(db_path)
890
+ artifacts = _stats_rebuild_artifact_bases(db_path)
891
+ lock_path = pathlib.Path(_cctally_core.STATS_LOCK_MAINTENANCE_PATH)
892
+ if (
893
+ not artifacts
894
+ or not lock_path.exists()
895
+ or not _has_completed_stats_quarantine_incident(db_path, artifacts)
896
+ ):
897
+ return None
898
+ import _cctally_journal
899
+
900
+ high_water = _cctally_journal.journal_high_water()
901
+ if high_water is None or high_water[1] == 0:
902
+ return None
903
+ lock_fh = open(lock_path, "r+")
904
+ acquired = False
905
+ try:
906
+ try:
907
+ fcntl.flock(lock_fh, fcntl.LOCK_EX | fcntl.LOCK_NB)
908
+ acquired = True
909
+ except (BlockingIOError, OSError):
910
+ return {
911
+ "live": True,
912
+ "artifacts": [path.name for path in artifacts],
913
+ "journalHighWater": [high_water[0], high_water[1]],
914
+ }
915
+ return {
916
+ "live": False,
917
+ "artifacts": [path.name for path in artifacts],
918
+ "journalHighWater": [high_water[0], high_water[1]],
919
+ "destinationExists": db_path.exists(),
920
+ }
921
+ finally:
922
+ if acquired:
923
+ fcntl.flock(lock_fh, fcntl.LOCK_UN)
924
+ lock_fh.close()
925
+
926
+
927
+ def _remove_stale_stats_rebuild_artifacts(
928
+ artifacts: tuple[pathlib.Path, ...],
929
+ ) -> None:
930
+ """Remove only exact scratch families, failing loudly on incomplete cleanup."""
931
+ for artifact in artifacts:
932
+ for suffix in ("", "-wal", "-shm"):
933
+ candidate = pathlib.Path(f"{artifact}{suffix}")
934
+ try:
935
+ candidate.unlink()
936
+ except FileNotFoundError:
937
+ pass
938
+ _cctally_journal = __import__("_cctally_journal")
939
+ _cctally_journal._fsync_dir(artifacts[0].parent)
940
+ leftovers = [
941
+ str(pathlib.Path(f"{artifact}{suffix}"))
942
+ for artifact in artifacts
943
+ for suffix in ("", "-wal", "-shm")
944
+ if pathlib.Path(f"{artifact}{suffix}").exists()
945
+ ]
946
+ if leftovers:
947
+ raise OSError(
948
+ "stale stats rebuild artifacts remain after cleanup: "
949
+ + ", ".join(leftovers)
950
+ )
951
+
952
+
953
+ def _recover_or_reclaim_interrupted_stats_rebuild(
954
+ db_path: pathlib.Path, artifacts: tuple[pathlib.Path, ...]
955
+ ) -> bool:
956
+ """Recover the legacy crash shape or reclaim a proven-stale scratch.
957
+
958
+ Caller holds maintenance EXCLUSIVE. A fully journal-consistent destination
959
+ proves every exact scratch family stale and needs cleanup only. Rebuilding
960
+ an absent or inconsistent destination additionally requires the completed
961
+ prebuild-quarantine incident that distinguishes the legacy interruption
962
+ from an unrelated file.
963
+ """
964
+ if not artifacts:
965
+ return False
966
+ import _cctally_journal
967
+
968
+ if _cctally_db._would_block_prod_stats(db_path):
969
+ raise _cctally_db.ProdMigrationRefused(
970
+ "stats.db", "interrupted-rebuild-recovery"
971
+ )
972
+ high_water = _cctally_journal.journal_high_water()
973
+ matching_incident = _has_completed_stats_quarantine_incident(
974
+ db_path, artifacts
975
+ )
976
+ if not matching_incident:
977
+ # Task A never removes the old destination before publication. With no
978
+ # matching legacy prebuild-quarantine incident, exact scratch names are
979
+ # unpublished Task A artifacts and are safe to reclaim under the
980
+ # caller's maintenance EXCLUSIVE hold.
981
+ _remove_stale_stats_rebuild_artifacts(artifacts)
982
+ return True
983
+ if db_path.exists() and _cctally_journal.stats_index_matches_journal_prefix(
984
+ db_path, high_water
985
+ ):
986
+ _remove_stale_stats_rebuild_artifacts(artifacts)
987
+ return True
988
+ if high_water is None or high_water[1] == 0:
989
+ return False
990
+ ingest_fd = _cctally_journal._acquire_ingest_lock("authoritative", 10.0)
991
+ if ingest_fd is None:
992
+ raise _cctally_db.StatsDbMaintenanceError(
993
+ "stats.db interrupted-rebuild recovery timed out waiting for "
994
+ "journal ingest serialization; retry after active cctally commands exit"
995
+ )
996
+ try:
997
+ with stats_write_scope("maintenance-interrupted-rebuild"):
998
+ _cctally_journal.rebuild_stats_index(high_water=high_water)
999
+ _remove_stale_stats_rebuild_artifacts(artifacts)
1000
+ return True
1001
+ finally:
1002
+ _cctally_journal._release_ingest_lock(ingest_fd)
1003
+
1004
+
1005
+ def stats_open_guarded(
1006
+ db_path=None, *, connect=None, recover_interruptions: bool = True
1007
+ ) -> sqlite3.Connection:
1008
+ """Open stats.db while excluding a destructive maintenance handshake (#386).
1009
+
1010
+ The shared maintenance flock covers the marker/pending checks AND the
1011
+ connect. A destructive maintenance path owns the exclusive side; after it
1012
+ publishes its marker/pending record, no new opener can escape into the live
1013
+ family while it verifies that pre-marker handles have drained.
1014
+
1015
+ ``connect`` lets a caller keep its own open mode (``mode=ro`` for
1016
+ ``db backup``, ``mode=rw`` for ``db status``) while still participating; it
1017
+ receives the path and returns the connection. Defaults to
1018
+ ``sqlite3.connect``.
1019
+ """
1020
+ db_path = pathlib.Path(
1021
+ db_path if db_path is not None else _cctally_core.DB_PATH
1022
+ )
1023
+ marker = _stats_repair_marker(db_path)
1024
+ _connect = connect if connect is not None else sqlite3.connect
1025
+
1026
+ live = db_path == pathlib.Path(_cctally_core.DB_PATH)
1027
+ if not live or _cctally_core.holds_stats_maintenance():
1028
+ # Scratch target, or this context already owns the exclusive side.
1029
+ # Pre-#386 behaviour, unchanged.
1030
+ if marker.exists():
1031
+ raise _cctally_db.StatsDbMaintenanceError()
1032
+ conn = _connect(db_path)
1033
+ arm_stats_authorizer(conn)
1034
+ return conn
1035
+
1036
+ pending = _cctally_db._quarantine_pending_path(db_path)
1037
+ lock_path = pathlib.Path(_cctally_core.STATS_LOCK_MAINTENANCE_PATH)
1038
+ lock_path.parent.mkdir(parents=True, exist_ok=True)
1039
+ lock_fh = open(lock_path, "a+")
1040
+ try:
1041
+ for _attempt in range(2):
1042
+ conn = None
1043
+ # BOUNDED, never blocking (#386 Stage 2 review P1-1) — see
1044
+ # _STATS_OPEN_MAINTENANCE_WAIT_S.
1045
+ if not _flock_bounded(
1046
+ lock_fh, fcntl.LOCK_SH, _STATS_OPEN_MAINTENANCE_WAIT_S
1047
+ ):
1048
+ raise _cctally_db.StatsDbMaintenanceError(
1049
+ _STATS_OPEN_MAINTENANCE_TIMEOUT_MSG
1050
+ )
1051
+ if marker.exists():
1052
+ fcntl.flock(lock_fh, fcntl.LOCK_UN)
1053
+ raise _cctally_db.StatsDbMaintenanceError()
1054
+ if pending.exists():
1055
+ # Drop shared BEFORE taking exclusive so two resumers cannot
1056
+ # deadlock while upgrading; recheck under exclusive because a
1057
+ # live owner may have completed it in the gap.
1058
+ fcntl.flock(lock_fh, fcntl.LOCK_UN)
1059
+ if not _flock_bounded(
1060
+ lock_fh, fcntl.LOCK_EX, _STATS_OPEN_RESUME_WAIT_S
1061
+ ):
1062
+ raise _cctally_db.StatsDbMaintenanceError(
1063
+ _STATS_OPEN_MAINTENANCE_TIMEOUT_MSG
1064
+ )
1065
+ try:
1066
+ if marker.exists():
1067
+ raise _cctally_db.StatsDbMaintenanceError()
1068
+ if pending.exists():
1069
+ _resume_pending_quarantine(db_path)
1070
+ finally:
1071
+ fcntl.flock(lock_fh, fcntl.LOCK_UN)
1072
+ continue
1073
+ artifacts = _stats_rebuild_artifact_bases(db_path)
1074
+ if (
1075
+ artifacts
1076
+ and recover_interruptions
1077
+ and _INTERRUPTED_RECOVERY_SUPPRESSED.get() == 0
1078
+ ):
1079
+ # A live rebuild owns maintenance EXCLUSIVE, so reaching this
1080
+ # shared hold proves the owner is gone. Upgrade without holding
1081
+ # shared, then re-check every fact under exclusive before any
1082
+ # file-family mutation or live connect.
1083
+ fcntl.flock(lock_fh, fcntl.LOCK_UN)
1084
+ if not _flock_bounded(
1085
+ lock_fh, fcntl.LOCK_EX, _STATS_OPEN_RESUME_WAIT_S
1086
+ ):
1087
+ raise _cctally_db.StatsDbMaintenanceError(
1088
+ _STATS_OPEN_MAINTENANCE_TIMEOUT_MSG
1089
+ )
1090
+ recovered = False
1091
+ try:
1092
+ current_artifacts = _stats_rebuild_artifact_bases(db_path)
1093
+ try:
1094
+ recovered = (
1095
+ _recover_or_reclaim_interrupted_stats_rebuild(
1096
+ db_path, current_artifacts
1097
+ )
1098
+ )
1099
+ except (
1100
+ _cctally_db.ProdMigrationRefused,
1101
+ _cctally_db.StatsDbMaintenanceError,
1102
+ ):
1103
+ raise
1104
+ except Exception as exc:
1105
+ raise _cctally_db.StatsDbMaintenanceError(
1106
+ "stats.db interrupted-rebuild recovery failed: "
1107
+ f"{exc}. The best usable index was preserved; stale "
1108
+ "artifact cleanup may be incomplete. Run "
1109
+ "`cctally doctor`, resolve "
1110
+ "the reported journal problem, then run "
1111
+ "`cctally db rebuild --db stats`."
1112
+ ) from exc
1113
+ finally:
1114
+ fcntl.flock(lock_fh, fcntl.LOCK_UN)
1115
+ if recovered:
1116
+ continue
1117
+ if not _flock_bounded(
1118
+ lock_fh, fcntl.LOCK_SH, _STATS_OPEN_MAINTENANCE_WAIT_S
1119
+ ):
1120
+ raise _cctally_db.StatsDbMaintenanceError(
1121
+ _STATS_OPEN_MAINTENANCE_TIMEOUT_MSG
1122
+ )
1123
+ if marker.exists() or pending.exists():
1124
+ fcntl.flock(lock_fh, fcntl.LOCK_UN)
1125
+ continue
1126
+ try:
1127
+ conn = _connect(db_path)
1128
+ # Re-check inside the same shared hold: cheap, and it closes the
1129
+ # window between the checks above and a slow connect.
1130
+ if marker.exists() or pending.exists():
1131
+ conn.close()
1132
+ conn = None
1133
+ raise _cctally_db.StatsDbMaintenanceError()
1134
+ # #386 enforcement: EVERY stats connection this module hands out
1135
+ # carries the authorizer. Arming HERE and nowhere else is what
1136
+ # keeps raw `sqlite3.connect` escape hatches (the storm suite's
1137
+ # `_grow_wal`, `db checkpoint`'s `mode=rw`) unaffected — a
1138
+ # broader arming point would make their writes unsanctioned and
1139
+ # the correct fix would then be to narrow the arming, never to
1140
+ # weaken the guard.
1141
+ arm_stats_authorizer(conn)
1142
+ return conn
1143
+ except BaseException:
1144
+ if conn is not None:
1145
+ try:
1146
+ conn.close()
1147
+ except Exception:
1148
+ pass
1149
+ raise
1150
+ finally:
1151
+ fcntl.flock(lock_fh, fcntl.LOCK_UN)
1152
+ raise _cctally_db.StatsDbMaintenanceError()
1153
+ finally:
1154
+ lock_fh.close()
1155
+
1156
+
1157
+ def _acquire_stats_maintenance_reentrant(path) -> "int | None":
1158
+ """Take ``stats.db.maintenance.lock`` EXCLUSIVE unless we already hold it.
1159
+
1160
+ Returns the held fd, or ``None`` when THIS execution context already owns the
1161
+ lock (in which case the caller must not release anything).
1162
+
1163
+ #386 Stage 2 review P1-2. ``flock`` conflicts are per open-file-DESCRIPTION
1164
+ and apply WITHIN a process: holding SHARED on one fd and then requesting
1165
+ EXCLUSIVE on a second fd of the same file blocks the process against itself,
1166
+ indefinitely. ``run_stats_ingest`` holds maintenance SHARED across its entire
1167
+ cycle, and both callers of this helper — the heal hook and the epoch resolver
1168
+ — are reachable from a nested ``open_db()`` inside that cycle. Without this
1169
+ check that nested open is an unconditional self-deadlock.
1170
+
1171
+ Proceeding on a shared hold is a deliberate, narrow weakening: the caller
1172
+ still runs ``_stats_family_drained`` before any physical replacement, which
1173
+ is a WHOLE-SYSTEM handle scan and therefore catches any sibling that could
1174
+ be harmed. The alternative — hanging forever — is strictly worse.
1175
+ """
1176
+ if _cctally_core.holds_stats_maintenance():
1177
+ return None
1178
+ return _heal_flock_blocking(path)
1179
+
1180
+
1181
+ def _release_stats_maintenance_reentrant(fd: "int | None") -> None:
1182
+ """Release what ``_acquire_stats_maintenance_reentrant`` took, if anything."""
1183
+ if fd is not None:
1184
+ _heal_release_maintenance_flock(fd)
1185
+
1186
+
326
1187
  def _heal_flock_blocking(path) -> int:
1188
+ """Blocking EX flock. Both call sites target STATS_LOCK_MAINTENANCE_PATH, so
1189
+ a successful acquire also records the #386 maintenance hold — pair it with
1190
+ ``_heal_release_maintenance_flock``, never the plain release.
1191
+
1192
+ Callers must reach this through ``_acquire_stats_maintenance_reentrant``, so
1193
+ a context that already owns the lock never requests it a second time.
1194
+ """
327
1195
  _cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
328
1196
  fd = os.open(str(path), os.O_RDWR | os.O_CREAT, 0o600)
329
1197
  try:
@@ -331,12 +1199,45 @@ def _heal_flock_blocking(path) -> int:
331
1199
  except BaseException:
332
1200
  os.close(fd)
333
1201
  raise
1202
+ if str(path) == str(_cctally_core.STATS_LOCK_MAINTENANCE_PATH):
1203
+ _cctally_core.note_stats_maintenance_acquired()
334
1204
  return fd
335
1205
 
336
1206
 
337
- def _heal_flock_bounded(path, timeout_s: float) -> int:
338
- """Bounded EX flock: return the held fd, or (on timeout) the OPEN fd WITHOUT
339
- the lock held — the caller proceeds best-effort (see the module note)."""
1207
+ def _heal_release_maintenance_flock(fd: int) -> None:
1208
+ """Release a stats maintenance flock taken by ``_heal_flock_blocking``.
1209
+
1210
+ Distinct from ``_heal_release_flock`` because that helper is also used for
1211
+ the INGEST fd, which carries no maintenance hold to unwind.
1212
+ """
1213
+ try:
1214
+ _cctally_core.note_stats_maintenance_released()
1215
+ finally:
1216
+ _heal_release_flock(fd)
1217
+
1218
+
1219
+ def _heal_flock_bounded(path, timeout_s: float) -> "int | None":
1220
+ """Bounded EX flock. Returns the HELD fd, or ``None`` on timeout.
1221
+
1222
+ #386: this previously returned the OPEN fd *without* the lock held and let
1223
+ the caller proceed "best-effort", which meant the heal path described itself
1224
+ as serialized while running unserialized — and no caller could tell the two
1225
+ outcomes apart, because both were an ``int``.
1226
+
1227
+ The re-entrancy case that motivated the old behaviour is real and is NOT
1228
+ solved by simply aborting on timeout: a corruption surfacing from INSIDE a
1229
+ ``run_stats_ingest`` cycle already holds ``journal.ingest.lock``, so an
1230
+ indefinite (or fail-closed) wait would deadlock the process against itself.
1231
+ That case is now detected EXPLICITLY at the call sites via
1232
+ ``holds_ingest_lock()`` — the ingester enters ``stats_write_scope(...,
1233
+ ingest_lock=True)`` around its cycle — so a timeout here means some OTHER
1234
+ holder has it, and failing soft is correct: decline the heal and let a later
1235
+ open retry.
1236
+
1237
+ Spec §5.1 says "a timeout aborts the operation rather than continuing
1238
+ unlocked". Read literally that would reintroduce the self-deadlock; the
1239
+ plan's Stage 2 correction (and this docstring) is the operative version.
1240
+ """
340
1241
  _cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
341
1242
  fd = os.open(str(path), os.O_RDWR | os.O_CREAT, 0o600)
342
1243
  deadline = time.monotonic() + timeout_s
@@ -347,13 +1248,54 @@ def _heal_flock_bounded(path, timeout_s: float) -> int:
347
1248
  return fd
348
1249
  except (BlockingIOError, OSError):
349
1250
  if time.monotonic() >= deadline:
350
- return fd
1251
+ os.close(fd)
1252
+ return None
351
1253
  time.sleep(0.02)
352
1254
  except BaseException:
353
1255
  os.close(fd)
354
1256
  raise
355
1257
 
356
1258
 
1259
+ def _stats_storm_test_pause(point: str) -> None:
1260
+ """Private process-control seam for the #386 stats writer-storm harness.
1261
+
1262
+ Production is a zero-cost string comparison. A test arms one exact point plus
1263
+ a marker path, waits for the marker, and then drives the SIGSTOPped child
1264
+ from the parent. Mirrors `_cctally_cache._cache_storm_test_pause`, which has
1265
+ carried the cache half of this since #344.
1266
+
1267
+ This is the ONLY way to hit spec section 1.1 Gap A's window deterministically:
1268
+ the instant after the handle scan says "drained" and before the first rename,
1269
+ which is exactly where a new opener must not be able to arrive.
1270
+ """
1271
+ if os.environ.get("CCTALLY_TEST_STATS_STORM_PAUSE_AT") != point:
1272
+ return
1273
+ marker = os.environ.get("CCTALLY_TEST_STATS_STORM_MARKER")
1274
+ if not marker:
1275
+ return
1276
+ pathlib.Path(marker).write_text(f"{os.getpid()}\n")
1277
+ os.kill(os.getpid(), signal.SIGSTOP)
1278
+
1279
+
1280
+ def _stats_family_drained(path) -> "str | None":
1281
+ """``None`` when no handle is open on the stats family; else why not.
1282
+
1283
+ Physical replacement renames files out from under whatever has them mapped.
1284
+ SQLite's crash guarantees stop applying at that point (spec §1.2), so the
1285
+ drain check is a precondition, not a nicety — and "the platform could not
1286
+ tell us" is a refusal, not a pass.
1287
+ """
1288
+ open_pids = _cctally_db._db_family_open_pids(path)
1289
+ if open_pids is None:
1290
+ return "could not verify that the database family has no open handles"
1291
+ if open_pids:
1292
+ return (
1293
+ "family is still open in process(es) "
1294
+ + ", ".join(str(pid) for pid in sorted(open_pids))
1295
+ )
1296
+ return None
1297
+
1298
+
357
1299
  def _heal_release_flock(fd: int) -> None:
358
1300
  try:
359
1301
  fcntl.flock(fd, fcntl.LOCK_UN)
@@ -380,7 +1322,34 @@ def _probe_stats_ok(path) -> bool:
380
1322
  return False
381
1323
 
382
1324
 
383
- def _stats_heal_hook(store: str, exc: Exception) -> bool:
1325
+ def _probe_stats_integrity_ok(path) -> bool:
1326
+ """Positive whole-index re-check for a post-query corruption report.
1327
+
1328
+ The ordinary locked probe intentionally stays cheap because it serves the
1329
+ open-time boundary. A dashboard leg has already observed a corruption
1330
+ error after that boundary, so its sibling-healed re-check must exercise the
1331
+ index B-trees rather than repeat ``PRAGMA schema_version``.
1332
+ """
1333
+
1334
+ if not path.exists():
1335
+ return False
1336
+ try:
1337
+ c = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
1338
+ try:
1339
+ rows = c.execute("PRAGMA quick_check").fetchall()
1340
+ finally:
1341
+ c.close()
1342
+ return rows == [("ok",)]
1343
+ except sqlite3.DatabaseError:
1344
+ return False
1345
+
1346
+
1347
+ def _stats_heal_hook(
1348
+ store: str,
1349
+ exc: Exception,
1350
+ *,
1351
+ post_query: bool = False,
1352
+ ) -> bool:
384
1353
  """Classifier-gated corruption auto-heal for stats.db (spec §6.3). Returns
385
1354
  True when it healed (quarantined + rebuilt) OR a sibling already healed under
386
1355
  the maintenance lock; False when it DECLINES — a non-corruption
@@ -412,19 +1381,38 @@ def _stats_heal_hook(store: str, exc: Exception) -> bool:
412
1381
  return False
413
1382
  _HEAL_ACTIVE = True
414
1383
  try:
415
- maint_fd = _heal_flock_blocking(_cctally_core.STATS_LOCK_MAINTENANCE_PATH)
1384
+ maint_fd = _acquire_stats_maintenance_reentrant(
1385
+ _cctally_core.STATS_LOCK_MAINTENANCE_PATH)
416
1386
  try:
417
- if _probe_stats_ok(path):
1387
+ probe = _probe_stats_integrity_ok if post_query else _probe_stats_ok
1388
+ if probe(path):
418
1389
  return True # a sibling process already healed it — retry the open
1390
+ # Forensics FIRST — before anything disturbs the evidence.
419
1391
  _cctally_db.write_corruption_forensics(path, db_label="stats")
420
- ingest_fd = _heal_flock_bounded(
421
- _cctally_core.JOURNAL_INGEST_LOCK_PATH, 5.0)
1392
+ if holds_ingest_lock():
1393
+ ingest_fd = None # this context IS the serialized writer
1394
+ else:
1395
+ ingest_fd = _heal_flock_bounded(
1396
+ _cctally_core.JOURNAL_INGEST_LOCK_PATH, 5.0)
1397
+ if ingest_fd is None:
1398
+ print(
1399
+ "[heal] stats.db auto-heal declined: another ingest "
1400
+ "holds journal.ingest.lock; a later open will retry.",
1401
+ file=sys.stderr,
1402
+ )
1403
+ return False
422
1404
  try:
423
- _cctally_db.quarantine_db_family(path)
424
- import _cctally_journal
425
- _cctally_journal.rebuild_stats_index()
1405
+ # #386: the rebuild writes the fresh scratch index through
1406
+ # `open_db(_target_path=...)`, whose connection carries the
1407
+ # authorizer. Declare the sanctioned maintenance regime for the
1408
+ # whole replacement — we hold (or already held) maintenance
1409
+ # exclusive, which is exactly what spec §3.1 sanctions.
1410
+ with stats_write_scope("maintenance-heal"):
1411
+ import _cctally_journal
1412
+ _cctally_journal.rebuild_stats_index()
426
1413
  finally:
427
- _heal_release_flock(ingest_fd)
1414
+ if ingest_fd is not None:
1415
+ _heal_release_flock(ingest_fd)
428
1416
  print(
429
1417
  f"[heal] stats.db was corrupt ({exc}); quarantined its file family "
430
1418
  "under quarantine/ (forensics in logs/) and rebuilt a fresh index "
@@ -433,7 +1421,7 @@ def _stats_heal_hook(store: str, exc: Exception) -> bool:
433
1421
  )
434
1422
  return True
435
1423
  finally:
436
- _heal_release_flock(maint_fd)
1424
+ _release_stats_maintenance_reentrant(maint_fd)
437
1425
  except Exception as heal_exc:
438
1426
  print(f"[heal] stats.db auto-heal failed: {heal_exc}", file=sys.stderr)
439
1427
  return False
@@ -490,12 +1478,15 @@ def resolve_stats_epoch_mismatch():
490
1478
  "journal and run `cctally db rebuild --db stats`.")
491
1479
  _EPOCH_MISMATCH_ACTIVE = True
492
1480
  try:
493
- maint_fd = _heal_flock_blocking(_cctally_core.STATS_LOCK_MAINTENANCE_PATH)
1481
+ maint_fd = _acquire_stats_maintenance_reentrant(
1482
+ _cctally_core.STATS_LOCK_MAINTENANCE_PATH)
494
1483
  try:
495
1484
  # Locked re-check: a sibling process may have already rebuilt it.
496
1485
  if _raw_user_version(path) != _cctally_core.STATS_INDEX_EPOCH:
497
- hw = _cctally_journal.journal_high_water()
498
- if hw is None or hw[1] == 0:
1486
+ hw, journal_has_bytes = (
1487
+ _cctally_journal._journal_rebuild_snapshot()
1488
+ )
1489
+ if hw is None or not journal_has_bytes:
499
1490
  raise _cctally_db.StatsEpochMismatchError(
500
1491
  f"stats.db is at index epoch {_raw_user_version(path)}, "
501
1492
  f"but this cctally builds epoch "
@@ -503,18 +1494,35 @@ def resolve_stats_epoch_mismatch():
503
1494
  "present to rebuild from. The journal/ directory is the "
504
1495
  "durable source — restore it from backup, then run "
505
1496
  "`cctally db rebuild --db stats`.")
506
- # Preserve the version-ahead DB (nothing deleted), then run the
507
- # #341 account epoch-transition coordinator into the now-absent
508
- # destination: resolve the cutover identity (no stats.db open),
509
- # append the canonical cutover op, then rebuild account-scoped
510
- # (spec §2 — op strictly before the rebuild HW). A same-epoch or
511
- # legacy DB never reaches here; a fresh cutover install stamps the
512
- # epoch directly (run_cutover), so this path is the 1000->1001
513
- # upgrade of an already-journaled install.
514
- _cctally_db.quarantine_db_family(path)
515
- _cctally_journal.run_epoch_transition()
1497
+ # Total lock order is maintenance -> ingest. The epoch bump can
1498
+ # be discovered by an ordinary open or an ingest caller, so it
1499
+ # must take the ingest lock here before building and publishing
1500
+ # the replacement; callers must never enter this resolver while
1501
+ # already holding that later lock. #386: if this context DOES
1502
+ # already hold it, say so rather than deadlocking against
1503
+ # ourselves.
1504
+ if holds_ingest_lock():
1505
+ ingest_fd = None
1506
+ else:
1507
+ ingest_fd = _cctally_journal._acquire_ingest_lock(
1508
+ "authoritative", 10.0
1509
+ )
1510
+ if ingest_fd is None:
1511
+ raise _cctally_db.StatsEpochMismatchError(
1512
+ "timed out waiting for journal ingest serialization "
1513
+ "during stats.db epoch rebuild"
1514
+ )
1515
+ try:
1516
+ # Append the idempotent account coordinator input, then let
1517
+ # the common rebuild cutover preserve the version-ahead
1518
+ # family and atomically publish the current epoch.
1519
+ with stats_write_scope("maintenance-epoch"):
1520
+ _cctally_journal.run_epoch_transition()
1521
+ finally:
1522
+ if ingest_fd is not None:
1523
+ _cctally_journal._release_ingest_lock(ingest_fd)
516
1524
  finally:
517
- _heal_release_flock(maint_fd)
1525
+ _release_stats_maintenance_reentrant(maint_fd)
518
1526
  return _cctally_core.open_db()
519
1527
  finally:
520
1528
  _EPOCH_MISMATCH_ACTIVE = False