cctally 1.82.0 → 1.83.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +70 -0
- package/README.md +52 -74
- package/bin/_cctally_alerts.py +8 -1
- package/bin/_cctally_cache.py +963 -149
- package/bin/_cctally_config.py +43 -4
- package/bin/_cctally_core.py +933 -759
- package/bin/_cctally_dashboard.py +157 -47
- package/bin/_cctally_dashboard_cache_report.py +13 -6
- package/bin/_cctally_dashboard_conversation.py +1 -0
- package/bin/_cctally_dashboard_envelope.py +186 -8
- package/bin/_cctally_dashboard_share.py +60 -20
- package/bin/_cctally_dashboard_sources.py +427 -128
- package/bin/_cctally_db.py +605 -128
- package/bin/_cctally_doctor.py +413 -28
- package/bin/_cctally_five_hour.py +12 -5
- package/bin/_cctally_journal.py +2050 -156
- package/bin/_cctally_journal_repair.py +519 -0
- package/bin/_cctally_milestone_history.py +142 -56
- package/bin/_cctally_milestones.py +179 -111
- package/bin/_cctally_parser.py +42 -0
- package/bin/_cctally_project.py +24 -18
- package/bin/_cctally_quota.py +139 -25
- package/bin/_cctally_record.py +279 -108
- package/bin/_cctally_rederive.py +1052 -0
- package/bin/_cctally_reporting.py +58 -53
- package/bin/_cctally_setup.py +1 -0
- package/bin/_cctally_source_analytics.py +4 -1
- package/bin/_cctally_statusline.py +11 -11
- package/bin/_cctally_store.py +1039 -31
- package/bin/_cctally_sync_week.py +17 -8
- package/bin/_cctally_tui.py +421 -54
- package/bin/_cctally_update.py +133 -8
- package/bin/_cctally_weekrefs.py +14 -0
- package/bin/_lib_aggregators.py +10 -6
- package/bin/_lib_cache_report.py +101 -9
- package/bin/_lib_codex_pools.py +82 -0
- package/bin/_lib_conversation_query.py +126 -33
- package/bin/_lib_dashboard_sources.py +126 -1
- package/bin/_lib_diff_kernel.py +28 -15
- package/bin/_lib_doctor.py +342 -4
- package/bin/_lib_journal.py +924 -2
- package/bin/_lib_jsonl.py +43 -14
- package/bin/_lib_pricing.py +140 -21
- package/bin/_lib_readme_refresh.py +401 -0
- package/bin/_lib_rederive.py +395 -0
- package/bin/_lib_share.py +58 -2
- package/bin/cctally +56 -8
- package/dashboard/static/assets/{index-DJP4gEB7.js → index-3bgCMVHb.js} +52 -52
- package/dashboard/static/assets/index-D27EIHEI.css +1 -0
- package/dashboard/static/dashboard.html +2 -2
- package/package.json +6 -1
- package/dashboard/static/assets/index-Dk1nplOz.css +0 -1
package/bin/_cctally_store.py
CHANGED
|
@@ -40,14 +40,31 @@ provider flocks (Claude → Codex) → SQLite transactions → ``journal.lock``
|
|
|
40
40
|
flock; no SQLite write transaction ever spans a flock acquisition.
|
|
41
41
|
|
|
42
42
|
**Raw-connect escape hatches stay OUT of this module by design** (spec §6.1):
|
|
43
|
-
``db checkpoint``'s ``mode=rw`` connect
|
|
44
|
-
|
|
45
|
-
|
|
43
|
+
``db checkpoint``'s ``mode=rw`` connect and ``db vacuum``'s exclusive connect
|
|
44
|
+
deliberately bypass ``open_index``/``open_db`` so they carry no schema-apply /
|
|
45
|
+
migration side effects on maintenance paths.
|
|
46
|
+
|
|
47
|
+
**#386 narrowed that carve-out for stats.** Skipping the *schema apply* is not
|
|
48
|
+
the same as skipping the *opener protocol*: spec §3.1's third clause requires
|
|
49
|
+
every opener of the live stats family to observe the repair marker and the
|
|
50
|
+
quarantine-pending record under maintenance-SHARED across connect. Doctor's
|
|
51
|
+
read-write probes (``bin/_cctally_doctor.py``), ``db backup --db stats``'s
|
|
52
|
+
``mode=ro`` source, and ``_db_status_for``'s status connect therefore all route
|
|
53
|
+
through ``stats_open_guarded`` with their OWN ``connect`` callable — they keep
|
|
54
|
+
their open mode and their freedom from schema side effects while still
|
|
55
|
+
participating. The claim that doctor's gather "bypasses the opener" is no longer
|
|
56
|
+
true and must not be restored.
|
|
46
57
|
"""
|
|
47
58
|
from __future__ import annotations
|
|
48
59
|
|
|
60
|
+
import contextlib
|
|
61
|
+
import datetime as dt
|
|
49
62
|
import fcntl
|
|
63
|
+
import json
|
|
50
64
|
import os
|
|
65
|
+
import pathlib
|
|
66
|
+
import re
|
|
67
|
+
import signal
|
|
51
68
|
import sqlite3
|
|
52
69
|
import sys
|
|
53
70
|
import time
|
|
@@ -323,7 +340,858 @@ def mark_stats_open_fixups_done(conn: sqlite3.Connection) -> None:
|
|
|
323
340
|
_HEAL_ACTIVE = False
|
|
324
341
|
|
|
325
342
|
|
|
343
|
+
# --------------------------------------------------------------------------
|
|
344
|
+
# #386 sanctioned-write context
|
|
345
|
+
# --------------------------------------------------------------------------
|
|
346
|
+
#
|
|
347
|
+
# A stats.db mutation is legal only inside this scope, entered by the ingester
|
|
348
|
+
# while it holds ``journal.ingest.lock`` and by maintenance paths while they hold
|
|
349
|
+
# ``stats.db.maintenance.lock`` (spec section 3.1's three regimes).
|
|
350
|
+
#
|
|
351
|
+
# A ``ContextVar``, NOT a module global. The dashboard is threaded, so a
|
|
352
|
+
# process-global boolean would let one sanctioned thread authorize an unrelated
|
|
353
|
+
# one — precisely the false-positive the review rejected the trace-callback
|
|
354
|
+
# design over. ``ContextVar`` values do not propagate into a ``threading.Thread``
|
|
355
|
+
# started from inside the scope, which is the property under test in
|
|
356
|
+
# ``tests/test_stats_writer_guard_386.py::test_scope_does_not_leak_across_threads``.
|
|
357
|
+
#
|
|
358
|
+
# ``holds_ingest_lock()`` is deliberately NARROWER than ``in_stats_write_scope()``
|
|
359
|
+
# and is not implied by it: maintenance paths are sanctioned writers that do NOT
|
|
360
|
+
# hold the ingest lock. The heal path keys its self-deadlock avoidance on the
|
|
361
|
+
# narrow fact (see ``_heal_flock_bounded``), so conflating the two would let a
|
|
362
|
+
# maintenance caller skip an ingest acquire it never made.
|
|
363
|
+
|
|
364
|
+
# The two ContextVars live in `_cctally_core` (see the block beside
|
|
365
|
+
# `holds_stats_maintenance`): `tests/conftest.py`'s `load_script()` reloads
|
|
366
|
+
# every `_cctally_*` sibling but never the kernel, so state kept here would be
|
|
367
|
+
# silently reset mid-test and a sanctioned write would then be denied.
|
|
368
|
+
_STATS_WRITE_SCOPE = _cctally_core._STATS_WRITE_SCOPE
|
|
369
|
+
_INGEST_LOCK_HELD = _cctally_core._STATS_INGEST_LOCK_HELD
|
|
370
|
+
_INTERRUPTED_RECOVERY_SUPPRESSED = (
|
|
371
|
+
_cctally_core._STATS_INTERRUPTED_RECOVERY_SUPPRESSED
|
|
372
|
+
)
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def in_stats_write_scope() -> bool:
|
|
376
|
+
"""True when THIS execution context is inside a sanctioned stats-write scope."""
|
|
377
|
+
return _STATS_WRITE_SCOPE.get() > 0
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def holds_ingest_lock() -> bool:
|
|
381
|
+
"""True when THIS execution context already holds ``journal.ingest.lock``.
|
|
382
|
+
|
|
383
|
+
Used by the heal path to distinguish "I am the serialized writer" from
|
|
384
|
+
"someone else holds it": a corruption surfacing from inside a
|
|
385
|
+
``run_stats_ingest`` cycle already owns the lock, so waiting for it would
|
|
386
|
+
self-deadlock, while any other caller must genuinely wait or decline.
|
|
387
|
+
"""
|
|
388
|
+
return _INGEST_LOCK_HELD.get() > 0
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
@contextlib.contextmanager
|
|
392
|
+
def suppress_interrupted_stats_recovery():
|
|
393
|
+
"""Keep every nested Doctor stats opener read-only in this context."""
|
|
394
|
+
token = _INTERRUPTED_RECOVERY_SUPPRESSED.set(
|
|
395
|
+
_INTERRUPTED_RECOVERY_SUPPRESSED.get() + 1
|
|
396
|
+
)
|
|
397
|
+
try:
|
|
398
|
+
yield
|
|
399
|
+
finally:
|
|
400
|
+
_INTERRUPTED_RECOVERY_SUPPRESSED.reset(token)
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
@contextlib.contextmanager
|
|
404
|
+
def stats_write_scope(reason: str, *, ingest_lock: bool = False):
|
|
405
|
+
"""Mark the enclosed block as a sanctioned stats.db writer.
|
|
406
|
+
|
|
407
|
+
``reason`` is diagnostic only (it names the regime for the Stage 3 guard log).
|
|
408
|
+
``ingest_lock=True`` additionally asserts that the caller holds
|
|
409
|
+
``journal.ingest.lock`` for the duration. Nests; restored on exception via
|
|
410
|
+
the ``ContextVar`` tokens, so an unwinding error never leaves the process
|
|
411
|
+
permanently sanctioned.
|
|
412
|
+
"""
|
|
413
|
+
depth = _STATS_WRITE_SCOPE.set(_STATS_WRITE_SCOPE.get() + 1)
|
|
414
|
+
held = _INGEST_LOCK_HELD.set(
|
|
415
|
+
_INGEST_LOCK_HELD.get() + (1 if ingest_lock else 0)
|
|
416
|
+
)
|
|
417
|
+
try:
|
|
418
|
+
yield reason
|
|
419
|
+
finally:
|
|
420
|
+
_INGEST_LOCK_HELD.reset(held)
|
|
421
|
+
_STATS_WRITE_SCOPE.reset(depth)
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
# --------------------------------------------------------------------------
|
|
425
|
+
# #386 enforcement — the stats sole-writer authorizer (spec §6.1)
|
|
426
|
+
# --------------------------------------------------------------------------
|
|
427
|
+
#
|
|
428
|
+
# The mechanism is ``Connection.set_authorizer``, NOT ``set_trace_callback``.
|
|
429
|
+
# Two independent reasons, both verified rather than assumed:
|
|
430
|
+
#
|
|
431
|
+
# 1. Python SUPPRESSES exceptions raised inside a trace callback. A trace hook
|
|
432
|
+
# that raises on INSERT does NOT prevent the write — the row commits and
|
|
433
|
+
# `SELECT count(*)` returns 1. An authorizer returning SQLITE_DENY blocks it
|
|
434
|
+
# (count 0). A mechanism that cannot stop the write is diagnostics, not
|
|
435
|
+
# enforcement.
|
|
436
|
+
# 2. `_TRACE_HOOK` is only installed by `open_index`, which the stats openers
|
|
437
|
+
# do not go through.
|
|
438
|
+
#
|
|
439
|
+
# Action CODES, never SQL text. A text classifier mishandles DDL, dynamic table
|
|
440
|
+
# names (`UPDATE {table}` in cutover and in eight journal folds), CTEs, and
|
|
441
|
+
# comments — and the mutation inventory found 15 dynamic-SQL sites a lexical
|
|
442
|
+
# scan cannot resolve at all.
|
|
443
|
+
#
|
|
444
|
+
# Scoped to the ``main`` schema. Stats connections legitimately create TEMP
|
|
445
|
+
# VIEWs outside any write scope (`bin/_cctally_tui.py`), and the dashboard/TUI
|
|
446
|
+
# build TEMP views over an ATTACHed cache.db. Those report `temp`/the attach
|
|
447
|
+
# alias as `db_name` and are none of this guard's business.
|
|
448
|
+
#
|
|
449
|
+
# NOT covered, and deliberately so: `PRAGMA user_version = …`, `VACUUM`,
|
|
450
|
+
# `os.replace`/`rename`/`unlink`. The first two would require guarding
|
|
451
|
+
# SQLITE_PRAGMA (which every `apply_policy` call trips) and the last three are
|
|
452
|
+
# invisible to every SQLite hook. 12 of the 14 physical mutation sites are in
|
|
453
|
+
# that last class — they are covered by the opener protocol and the lock
|
|
454
|
+
# corrections instead. Enforcement and serialization do different jobs here and
|
|
455
|
+
# neither substitutes for the other.
|
|
456
|
+
|
|
457
|
+
_GUARD_MUTATIONS = frozenset({
|
|
458
|
+
sqlite3.SQLITE_INSERT,
|
|
459
|
+
sqlite3.SQLITE_UPDATE,
|
|
460
|
+
sqlite3.SQLITE_DELETE,
|
|
461
|
+
sqlite3.SQLITE_CREATE_TABLE,
|
|
462
|
+
sqlite3.SQLITE_CREATE_INDEX,
|
|
463
|
+
sqlite3.SQLITE_DROP_TABLE,
|
|
464
|
+
sqlite3.SQLITE_DROP_INDEX,
|
|
465
|
+
sqlite3.SQLITE_ALTER_TABLE,
|
|
466
|
+
})
|
|
467
|
+
|
|
468
|
+
#: One line per throttle window across all processes. Rotation is still the
|
|
469
|
+
#: hard disk-growth bound if the marker is removed or the throttle is disabled.
|
|
470
|
+
_GUARD_THROTTLE_S = 60.0
|
|
471
|
+
_GUARD_LOG_ROTATE_BYTES = 1024 * 1024
|
|
472
|
+
_guard_last_logged = 0.0
|
|
473
|
+
|
|
474
|
+
|
|
475
|
+
def _guard_log_path() -> pathlib.Path:
|
|
476
|
+
"""``logs/stats-writer-guard.log`` — the doctor leg's input (spec §6.4)."""
|
|
477
|
+
return pathlib.Path(_cctally_core.HOOK_TICK_LOG_DIR) / "stats-writer-guard.log"
|
|
478
|
+
|
|
479
|
+
|
|
480
|
+
def _guard_rotated_log_path() -> pathlib.Path:
|
|
481
|
+
path = _guard_log_path()
|
|
482
|
+
return path.with_name(path.name + ".1")
|
|
483
|
+
|
|
484
|
+
|
|
485
|
+
def _guard_log_lock_path() -> pathlib.Path:
|
|
486
|
+
path = _guard_log_path()
|
|
487
|
+
return path.with_name(path.name + ".lock")
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
def _guard_throttle_path() -> pathlib.Path:
|
|
491
|
+
path = _guard_log_path()
|
|
492
|
+
return path.with_name(path.name + ".last")
|
|
493
|
+
|
|
494
|
+
|
|
495
|
+
def _guard_rotate_if_needed(path: pathlib.Path) -> bool:
|
|
496
|
+
"""Keep one rotated generation; return False when rotation cannot proceed."""
|
|
497
|
+
try:
|
|
498
|
+
size = path.stat().st_size
|
|
499
|
+
except FileNotFoundError:
|
|
500
|
+
return True
|
|
501
|
+
except OSError:
|
|
502
|
+
return False
|
|
503
|
+
if size < _GUARD_LOG_ROTATE_BYTES:
|
|
504
|
+
return True
|
|
505
|
+
try:
|
|
506
|
+
os.replace(path, _guard_rotated_log_path())
|
|
507
|
+
except OSError:
|
|
508
|
+
return False
|
|
509
|
+
return True
|
|
510
|
+
|
|
511
|
+
|
|
512
|
+
def _log_unsanctioned_write(action, table) -> None:
|
|
513
|
+
global _guard_last_logged
|
|
514
|
+
now_monotonic = time.monotonic()
|
|
515
|
+
if (
|
|
516
|
+
_GUARD_THROTTLE_S > 0
|
|
517
|
+
and _guard_last_logged
|
|
518
|
+
and now_monotonic - _guard_last_logged < _GUARD_THROTTLE_S
|
|
519
|
+
):
|
|
520
|
+
return
|
|
521
|
+
lock_fd = None
|
|
522
|
+
try:
|
|
523
|
+
path = _guard_log_path()
|
|
524
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
525
|
+
lock_fd = os.open(
|
|
526
|
+
_guard_log_lock_path(), os.O_RDWR | os.O_CREAT, 0o600)
|
|
527
|
+
# This diagnostic must never park the command whose write triggered it.
|
|
528
|
+
# One storm participant records the incident; contenders fail soft.
|
|
529
|
+
fcntl.flock(lock_fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
530
|
+
|
|
531
|
+
marker = _guard_throttle_path()
|
|
532
|
+
if _GUARD_THROTTLE_S > 0:
|
|
533
|
+
try:
|
|
534
|
+
marker_age = time.time() - marker.stat().st_mtime
|
|
535
|
+
except FileNotFoundError:
|
|
536
|
+
marker_age = _GUARD_THROTTLE_S
|
|
537
|
+
if 0 <= marker_age < _GUARD_THROTTLE_S:
|
|
538
|
+
_guard_last_logged = now_monotonic
|
|
539
|
+
return
|
|
540
|
+
|
|
541
|
+
if not _guard_rotate_if_needed(path):
|
|
542
|
+
return
|
|
543
|
+
line = (
|
|
544
|
+
f"{_cctally_core.now_utc_iso()}\tunsanctioned stats write\t"
|
|
545
|
+
f"action={action}\ttable={table}\n"
|
|
546
|
+
).encode("utf-8", errors="replace")
|
|
547
|
+
log_fd = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_APPEND, 0o600)
|
|
548
|
+
try:
|
|
549
|
+
os.write(log_fd, line)
|
|
550
|
+
finally:
|
|
551
|
+
os.close(log_fd)
|
|
552
|
+
marker.touch(mode=0o600, exist_ok=True)
|
|
553
|
+
_guard_last_logged = now_monotonic
|
|
554
|
+
except OSError:
|
|
555
|
+
pass
|
|
556
|
+
finally:
|
|
557
|
+
if lock_fd is not None:
|
|
558
|
+
try:
|
|
559
|
+
fcntl.flock(lock_fd, fcntl.LOCK_UN)
|
|
560
|
+
except OSError:
|
|
561
|
+
pass
|
|
562
|
+
os.close(lock_fd)
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
def _guard_should_raise() -> bool:
|
|
566
|
+
"""Raise on a dev checkout or under pytest; log-only on installed builds.
|
|
567
|
+
|
|
568
|
+
``_is_dev_checkout()``, deliberately NOT ``DEV_MODE`` — CLAUDE.md forbids
|
|
569
|
+
collapsing those two predicates, and this is the checkout question, not the
|
|
570
|
+
verbosity one.
|
|
571
|
+
"""
|
|
572
|
+
return bool(
|
|
573
|
+
_cctally_core._is_dev_checkout()
|
|
574
|
+
or os.environ.get("PYTEST_CURRENT_TEST")
|
|
575
|
+
)
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
def arm_stats_authorizer(conn: sqlite3.Connection) -> None:
|
|
579
|
+
"""Deny mutations of the stats ``main`` schema outside a sanctioned scope."""
|
|
580
|
+
|
|
581
|
+
def _auth(action, arg1, arg2, db_name, trigger):
|
|
582
|
+
if action not in _GUARD_MUTATIONS:
|
|
583
|
+
return sqlite3.SQLITE_OK
|
|
584
|
+
if db_name not in (None, "main"):
|
|
585
|
+
return sqlite3.SQLITE_OK
|
|
586
|
+
if in_stats_write_scope():
|
|
587
|
+
return sqlite3.SQLITE_OK
|
|
588
|
+
if _guard_should_raise():
|
|
589
|
+
return sqlite3.SQLITE_DENY
|
|
590
|
+
_log_unsanctioned_write(action, arg1)
|
|
591
|
+
return sqlite3.SQLITE_OK
|
|
592
|
+
|
|
593
|
+
conn.set_authorizer(_auth)
|
|
594
|
+
|
|
595
|
+
|
|
596
|
+
# --------------------------------------------------------------------------
|
|
597
|
+
# #386 the open-time mutation regime (spec §3.1's second clause, Gaps C/GAP-1..4)
|
|
598
|
+
# --------------------------------------------------------------------------
|
|
599
|
+
#
|
|
600
|
+
# `open_db`'s post-gate body runs the full schema DDL, the quota-projection
|
|
601
|
+
# schema, the migration dispatcher, two backfills, the fixups marker and the
|
|
602
|
+
# in-place cutover. Every one of those is a mutation, and before #386 they ran
|
|
603
|
+
# under NO lock whatever command reached them — 57 production `open_db` call
|
|
604
|
+
# sites, any of which could be racing another. Spec §3.1: "First-open, legacy,
|
|
605
|
+
# epoch and administrative mutation … must acquire stats.db.maintenance.lock
|
|
606
|
+
# (exclusive) BEFORE mutating, regardless of the command that reached them."
|
|
607
|
+
#
|
|
608
|
+
# Re-entrant on the same signal the opener uses, because the common case is
|
|
609
|
+
# reaching here from a caller that ALREADY holds the lock (`run_stats_ingest`'s
|
|
610
|
+
# legacy branch takes it exclusive before `open_db()`; `rebuild_stats_index`
|
|
611
|
+
# opens its scratch under the rebuild's hold).
|
|
612
|
+
|
|
613
|
+
#: Open-time mutation is a one-shot path (first open / legacy / post-rebuild),
|
|
614
|
+
#: not the steady-state hot path — the epoch gate returns before it. So the wait
|
|
615
|
+
#: is generous where the opener's is short. On expiry we DECLINE; proceeding
|
|
616
|
+
#: unlocked would be the "described itself as serialized while running
|
|
617
|
+
#: unserialized" defect this session removed from the heal path.
|
|
618
|
+
_STATS_OPEN_TIME_MAINTENANCE_WAIT_S = 30.0
|
|
619
|
+
|
|
620
|
+
|
|
621
|
+
@contextlib.contextmanager
|
|
622
|
+
def stats_open_time_guard(*, live: bool = True):
|
|
623
|
+
"""Hold maintenance-exclusive + the sanctioned scope across open-time DDL.
|
|
624
|
+
|
|
625
|
+
``live=False`` marks a ``_target_path`` (scratch) build: it enters the
|
|
626
|
+
sanctioned scope but takes NO flock, mirroring the divergence
|
|
627
|
+
``stats_open_guarded`` already documents. Two reasons, both concrete:
|
|
628
|
+
|
|
629
|
+
1. A scratch index is not the live family, so serializing it against live
|
|
630
|
+
maintenance buys nothing — and every scratch build already runs under a
|
|
631
|
+
HELD maintenance exclusive (rebuild, rederive) or against a private
|
|
632
|
+
temp copy (`db rederive`'s preview snapshot). Taking the LIVE lock on a
|
|
633
|
+
second fd from inside a held exclusive is the self-deadlock
|
|
634
|
+
`holds_stats_maintenance` exists to prevent.
|
|
635
|
+
2. Creating `stats.db.maintenance.lock` is itself a persistent side
|
|
636
|
+
effect. `db rederive`'s preview has a literal zero-persistent-write
|
|
637
|
+
contract ("without creating any coordination files"), pinned by
|
|
638
|
+
`tests/test_rederive_command.py::test_preview_is_write_free_…`, and its
|
|
639
|
+
snapshot open would otherwise mint that file in the real APP_DIR.
|
|
640
|
+
"""
|
|
641
|
+
if not live or _cctally_core.holds_stats_maintenance():
|
|
642
|
+
with stats_write_scope("open-time"):
|
|
643
|
+
yield
|
|
644
|
+
return
|
|
645
|
+
_cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
|
|
646
|
+
fd = os.open(
|
|
647
|
+
str(_cctally_core.STATS_LOCK_MAINTENANCE_PATH),
|
|
648
|
+
os.O_RDWR | os.O_CREAT, 0o600,
|
|
649
|
+
)
|
|
650
|
+
acquired = False
|
|
651
|
+
try:
|
|
652
|
+
deadline = time.monotonic() + _STATS_OPEN_TIME_MAINTENANCE_WAIT_S
|
|
653
|
+
while True:
|
|
654
|
+
try:
|
|
655
|
+
fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
656
|
+
acquired = True
|
|
657
|
+
break
|
|
658
|
+
except (BlockingIOError, OSError):
|
|
659
|
+
if time.monotonic() >= deadline:
|
|
660
|
+
raise _cctally_db.StatsDbMaintenanceError(
|
|
661
|
+
_STATS_OPEN_MAINTENANCE_TIMEOUT_MSG
|
|
662
|
+
)
|
|
663
|
+
time.sleep(0.02)
|
|
664
|
+
_cctally_core.note_stats_maintenance_acquired()
|
|
665
|
+
try:
|
|
666
|
+
with stats_write_scope("open-time"):
|
|
667
|
+
yield
|
|
668
|
+
finally:
|
|
669
|
+
_cctally_core.note_stats_maintenance_released()
|
|
670
|
+
finally:
|
|
671
|
+
if acquired:
|
|
672
|
+
try:
|
|
673
|
+
fcntl.flock(fd, fcntl.LOCK_UN)
|
|
674
|
+
except OSError:
|
|
675
|
+
pass
|
|
676
|
+
os.close(fd)
|
|
677
|
+
|
|
678
|
+
|
|
679
|
+
# --------------------------------------------------------------------------
|
|
680
|
+
# #386 opener half of the physical-replacement protocol
|
|
681
|
+
# --------------------------------------------------------------------------
|
|
682
|
+
#
|
|
683
|
+
# Spec section 3.1, third clause: EVERY opener of the live stats family --
|
|
684
|
+
# read-only consumers included -- observes the repair marker and the
|
|
685
|
+
# quarantine-pending record under maintenance-SHARED, held across the marker
|
|
686
|
+
# checks AND the connect. That is what makes the pending record's claim to
|
|
687
|
+
# "block every opener" true, and it is the half stats never had: `open_db`
|
|
688
|
+
# checked `stats.db.repairing` three times around a bare `sqlite3.connect` and
|
|
689
|
+
# consulted no pending record at all, so a destructive maintenance path could
|
|
690
|
+
# publish its record, scan for handles, and still have a brand-new opener arrive
|
|
691
|
+
# in the window before the first rename (spec section 1.1 Gap A's TOCTOU).
|
|
692
|
+
#
|
|
693
|
+
# Modelled directly on `bin/_cctally_cache.py::_cache_open_guarded`. Two
|
|
694
|
+
# deliberate divergences from the cache version, both narrowing:
|
|
695
|
+
#
|
|
696
|
+
# 1. Repair-marker STALENESS reclaim is NOT ported. The cache upgrades to
|
|
697
|
+
# exclusive, asks `_repair_marker_is_live`, and reclaims a dead owner's
|
|
698
|
+
# marker. Stats has always simply raised, and every existing stats
|
|
699
|
+
# regression asserts that. Changing it is a behaviour change with its own
|
|
700
|
+
# test surface and belongs to whoever owns `db repair`, not to a
|
|
701
|
+
# corruption-prevention pass. Consequence, recorded rather than fixed: a
|
|
702
|
+
# SIGKILLed `db repair --db stats` still strands its marker.
|
|
703
|
+
# 2. `_target_path` (scratch) opens skip the flock entirely and keep the
|
|
704
|
+
# pre-#386 marker-only behaviour. A scratch index is not the live family,
|
|
705
|
+
# and every scratch open happens under a HELD maintenance exclusive
|
|
706
|
+
# (rebuild, rederive) -- taking shared on a second fd there is the
|
|
707
|
+
# self-deadlock described in `_cctally_core.holds_stats_maintenance`.
|
|
708
|
+
|
|
709
|
+
|
|
710
|
+
#: Bounded wait for ``stats.db.maintenance.lock`` on the OPENER path (#386).
|
|
711
|
+
#
|
|
712
|
+
# Stage 2 took this flock UNTIMED, on the path every command uses — every
|
|
713
|
+
# `statusline` render and every detached `hook-tick`. Stage 2 also made the
|
|
714
|
+
# exclusive holders long: `db rebuild` replays the whole journal, `db vacuum`
|
|
715
|
+
# rewrites the file, `db rederive --yes` runs a full scratch replay, `db repair`
|
|
716
|
+
# shells out to `sqlite3 .recover`. During any of them EVERY stats open parked
|
|
717
|
+
# forever — the #297-class `database is locked` stall that spec §5.2 ground 4
|
|
718
|
+
# rejected the checkpoint policy over, reintroduced by the corruption fix.
|
|
719
|
+
#
|
|
720
|
+
# Bounding it is strictly safe: the TOCTOU guarantee is "you cannot open WHILE
|
|
721
|
+
# exclusive is held", and failing after a timeout preserves it — we simply
|
|
722
|
+
# decline instead of waiting. Every caller already handles the exception
|
|
723
|
+
# (`main()` maps it to exit 3, `doctor` degrades, `cmd_db_status` reports
|
|
724
|
+
# `_open_error`).
|
|
725
|
+
#
|
|
726
|
+
# 5 s is chosen against the two failure directions: it is well under the 15 s
|
|
727
|
+
# `busy_timeout` the DB already tolerates (so the opener is never the slowest
|
|
728
|
+
# thing on the hot path), and it is orders of magnitude above the handshake it
|
|
729
|
+
# must NOT trip — marker publish + drain scan + rename is milliseconds.
|
|
730
|
+
_STATS_OPEN_MAINTENANCE_WAIT_S = 5.0
|
|
731
|
+
|
|
732
|
+
#: How long the opener will wait to UPGRADE to exclusive to resume a pending
|
|
733
|
+
#: quarantine. Same reasoning; kept separate so the two can diverge.
|
|
734
|
+
_STATS_OPEN_RESUME_WAIT_S = 5.0
|
|
735
|
+
|
|
736
|
+
_STATS_OPEN_MAINTENANCE_TIMEOUT_MSG = (
|
|
737
|
+
"stats.db maintenance is in progress; retry after the maintenance command "
|
|
738
|
+
"exits"
|
|
739
|
+
)
|
|
740
|
+
|
|
741
|
+
|
|
742
|
+
def _flock_bounded(lock_fh, operation: int, timeout_s: float) -> bool:
|
|
743
|
+
"""Poll for ``operation`` on ``lock_fh`` until ``timeout_s`` expires.
|
|
744
|
+
|
|
745
|
+
``True`` when the lock is held, ``False`` on expiry (nothing acquired, so
|
|
746
|
+
the caller must not release). Same shape as ``_heal_flock_bounded``, but it
|
|
747
|
+
operates on an already-open file object rather than minting an fd, because
|
|
748
|
+
the opener holds one file object across its whole marker/pending/connect
|
|
749
|
+
sequence.
|
|
750
|
+
"""
|
|
751
|
+
deadline = time.monotonic() + timeout_s
|
|
752
|
+
while True:
|
|
753
|
+
try:
|
|
754
|
+
fcntl.flock(lock_fh, operation | fcntl.LOCK_NB)
|
|
755
|
+
return True
|
|
756
|
+
except (BlockingIOError, OSError):
|
|
757
|
+
if time.monotonic() >= deadline:
|
|
758
|
+
return False
|
|
759
|
+
time.sleep(0.02)
|
|
760
|
+
|
|
761
|
+
|
|
762
|
+
def _stats_repair_marker(db_path) -> pathlib.Path:
|
|
763
|
+
"""The live stats repair marker, whatever `db_path` is.
|
|
764
|
+
|
|
765
|
+
Deliberately `with_name`, not `<db_path>.repairing`: a scratch/rebuild
|
|
766
|
+
target sitting beside the live DB must still observe the LIVE marker, which
|
|
767
|
+
is the pre-#386 behaviour this preserves byte for byte.
|
|
768
|
+
"""
|
|
769
|
+
return pathlib.Path(db_path).with_name("stats.db.repairing")
|
|
770
|
+
|
|
771
|
+
|
|
772
|
+
def _resume_pending_quarantine(db_path: pathlib.Path) -> None:
|
|
773
|
+
"""Finish a strict quarantine that a previous owner did not complete.
|
|
774
|
+
|
|
775
|
+
Caller holds maintenance EXCLUSIVE. Fails closed: if any handle on the
|
|
776
|
+
family is still open -- or the platform cannot tell us -- we refuse rather
|
|
777
|
+
than rename files out from under a live mapping, which is precisely the
|
|
778
|
+
"cctally performs file-family surgery with live mappings" class that spec
|
|
779
|
+
section 1.2 identifies as where SQLite's crash guarantees stop applying.
|
|
780
|
+
"""
|
|
781
|
+
try:
|
|
782
|
+
open_pids = _cctally_db._db_family_open_pids(db_path)
|
|
783
|
+
if open_pids is None:
|
|
784
|
+
raise OSError(
|
|
785
|
+
"could not verify that the database family has no open handles"
|
|
786
|
+
)
|
|
787
|
+
if open_pids:
|
|
788
|
+
raise OSError(
|
|
789
|
+
"database family is still open in process(es) "
|
|
790
|
+
+ ", ".join(str(pid) for pid in sorted(open_pids))
|
|
791
|
+
)
|
|
792
|
+
# Resumes the SAME incident from the pending record -- never a second
|
|
793
|
+
# incident dir, never a recreation.
|
|
794
|
+
_cctally_db.quarantine_db_family(db_path, strict=True)
|
|
795
|
+
except OSError as exc:
|
|
796
|
+
raise sqlite3.OperationalError(
|
|
797
|
+
f"stats.db pending quarantine could not resume: {exc}"
|
|
798
|
+
) from exc
|
|
799
|
+
|
|
800
|
+
|
|
801
|
+
_STATS_REBUILD_ARTIFACT_RE = re.compile(
|
|
802
|
+
r"^(?P<base>stats\.db\.rebuilding-\d{8}T\d{6}_\d{6})"
|
|
803
|
+
r"(?P<sidecar>-wal|-shm)?$"
|
|
804
|
+
)
|
|
805
|
+
_STATS_QUARANTINE_INCIDENT_RE = re.compile(
|
|
806
|
+
r"^stats\.db-(?:\d{8}T\d{6}Z|\d{8}T\d{6}_\d{6})$"
|
|
807
|
+
)
|
|
808
|
+
|
|
809
|
+
|
|
810
|
+
def _stats_rebuild_artifact_bases(db_path: pathlib.Path) -> tuple[pathlib.Path, ...]:
|
|
811
|
+
"""Return only Task A's exact scratch-family bases beside ``db_path``."""
|
|
812
|
+
bases: set[pathlib.Path] = set()
|
|
813
|
+
for candidate in db_path.parent.glob(f"{db_path.name}.rebuilding-*"):
|
|
814
|
+
match = _STATS_REBUILD_ARTIFACT_RE.fullmatch(candidate.name)
|
|
815
|
+
if match is not None:
|
|
816
|
+
bases.add(db_path.parent / match.group("base"))
|
|
817
|
+
return tuple(sorted(bases))
|
|
818
|
+
|
|
819
|
+
|
|
820
|
+
def _rebuild_artifact_time(path: pathlib.Path) -> "dt.datetime | None":
|
|
821
|
+
match = _STATS_REBUILD_ARTIFACT_RE.fullmatch(path.name)
|
|
822
|
+
if match is None:
|
|
823
|
+
return None
|
|
824
|
+
try:
|
|
825
|
+
return dt.datetime.strptime(
|
|
826
|
+
match.group("base").removeprefix("stats.db.rebuilding-"),
|
|
827
|
+
"%Y%m%dT%H%M%S_%f",
|
|
828
|
+
).replace(tzinfo=dt.timezone.utc)
|
|
829
|
+
except ValueError:
|
|
830
|
+
return None
|
|
831
|
+
|
|
832
|
+
|
|
833
|
+
def _quarantine_incident_time(path: pathlib.Path) -> "dt.datetime | None":
|
|
834
|
+
timestamp = path.name.removeprefix("stats.db-")
|
|
835
|
+
try:
|
|
836
|
+
if timestamp.endswith("Z"):
|
|
837
|
+
return dt.datetime.strptime(timestamp, "%Y%m%dT%H%M%SZ").replace(
|
|
838
|
+
tzinfo=dt.timezone.utc
|
|
839
|
+
)
|
|
840
|
+
return dt.datetime.strptime(timestamp, "%Y%m%dT%H%M%S_%f").replace(
|
|
841
|
+
tzinfo=dt.timezone.utc
|
|
842
|
+
)
|
|
843
|
+
except ValueError:
|
|
844
|
+
return None
|
|
845
|
+
|
|
846
|
+
|
|
847
|
+
def _has_completed_stats_quarantine_incident(
|
|
848
|
+
db_path: pathlib.Path, artifacts: tuple[pathlib.Path, ...]
|
|
849
|
+
) -> bool:
|
|
850
|
+
"""Positive evidence that the same legacy rebuild removed the live family."""
|
|
851
|
+
root = _cctally_core.APP_DIR / "quarantine"
|
|
852
|
+
if not root.is_dir():
|
|
853
|
+
return False
|
|
854
|
+
artifact_times = tuple(
|
|
855
|
+
timestamp
|
|
856
|
+
for path in artifacts
|
|
857
|
+
if (timestamp := _rebuild_artifact_time(path)) is not None
|
|
858
|
+
)
|
|
859
|
+
if not artifact_times:
|
|
860
|
+
return False
|
|
861
|
+
for manifest_path in sorted(root.glob(f"{db_path.name}-*/manifest.json")):
|
|
862
|
+
if _STATS_QUARANTINE_INCIDENT_RE.fullmatch(
|
|
863
|
+
manifest_path.parent.name
|
|
864
|
+
) is None:
|
|
865
|
+
continue
|
|
866
|
+
incident_time = _quarantine_incident_time(manifest_path.parent)
|
|
867
|
+
if incident_time is None or not any(
|
|
868
|
+
dt.timedelta(0) <= artifact_time - incident_time <= dt.timedelta(minutes=5)
|
|
869
|
+
for artifact_time in artifact_times
|
|
870
|
+
):
|
|
871
|
+
continue
|
|
872
|
+
try:
|
|
873
|
+
manifest = json.loads(manifest_path.read_text())
|
|
874
|
+
except (OSError, json.JSONDecodeError):
|
|
875
|
+
continue
|
|
876
|
+
if (
|
|
877
|
+
manifest.get("complete") is True
|
|
878
|
+
and manifest.get("originalPath") == str(db_path)
|
|
879
|
+
and db_path.name in (manifest.get("movedFiles") or ())
|
|
880
|
+
):
|
|
881
|
+
return True
|
|
882
|
+
return False
|
|
883
|
+
|
|
884
|
+
|
|
885
|
+
def stats_interrupted_rebuild_evidence(
|
|
886
|
+
db_path: pathlib.Path,
|
|
887
|
+
) -> "dict | None":
|
|
888
|
+
"""Read-only classification for Doctor; never creates or reclaims files."""
|
|
889
|
+
db_path = pathlib.Path(db_path)
|
|
890
|
+
artifacts = _stats_rebuild_artifact_bases(db_path)
|
|
891
|
+
lock_path = pathlib.Path(_cctally_core.STATS_LOCK_MAINTENANCE_PATH)
|
|
892
|
+
if (
|
|
893
|
+
not artifacts
|
|
894
|
+
or not lock_path.exists()
|
|
895
|
+
or not _has_completed_stats_quarantine_incident(db_path, artifacts)
|
|
896
|
+
):
|
|
897
|
+
return None
|
|
898
|
+
import _cctally_journal
|
|
899
|
+
|
|
900
|
+
high_water = _cctally_journal.journal_high_water()
|
|
901
|
+
if high_water is None or high_water[1] == 0:
|
|
902
|
+
return None
|
|
903
|
+
lock_fh = open(lock_path, "r+")
|
|
904
|
+
acquired = False
|
|
905
|
+
try:
|
|
906
|
+
try:
|
|
907
|
+
fcntl.flock(lock_fh, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
908
|
+
acquired = True
|
|
909
|
+
except (BlockingIOError, OSError):
|
|
910
|
+
return {
|
|
911
|
+
"live": True,
|
|
912
|
+
"artifacts": [path.name for path in artifacts],
|
|
913
|
+
"journalHighWater": [high_water[0], high_water[1]],
|
|
914
|
+
}
|
|
915
|
+
return {
|
|
916
|
+
"live": False,
|
|
917
|
+
"artifacts": [path.name for path in artifacts],
|
|
918
|
+
"journalHighWater": [high_water[0], high_water[1]],
|
|
919
|
+
"destinationExists": db_path.exists(),
|
|
920
|
+
}
|
|
921
|
+
finally:
|
|
922
|
+
if acquired:
|
|
923
|
+
fcntl.flock(lock_fh, fcntl.LOCK_UN)
|
|
924
|
+
lock_fh.close()
|
|
925
|
+
|
|
926
|
+
|
|
927
|
+
def _remove_stale_stats_rebuild_artifacts(
|
|
928
|
+
artifacts: tuple[pathlib.Path, ...],
|
|
929
|
+
) -> None:
|
|
930
|
+
"""Remove only exact scratch families, failing loudly on incomplete cleanup."""
|
|
931
|
+
for artifact in artifacts:
|
|
932
|
+
for suffix in ("", "-wal", "-shm"):
|
|
933
|
+
candidate = pathlib.Path(f"{artifact}{suffix}")
|
|
934
|
+
try:
|
|
935
|
+
candidate.unlink()
|
|
936
|
+
except FileNotFoundError:
|
|
937
|
+
pass
|
|
938
|
+
_cctally_journal = __import__("_cctally_journal")
|
|
939
|
+
_cctally_journal._fsync_dir(artifacts[0].parent)
|
|
940
|
+
leftovers = [
|
|
941
|
+
str(pathlib.Path(f"{artifact}{suffix}"))
|
|
942
|
+
for artifact in artifacts
|
|
943
|
+
for suffix in ("", "-wal", "-shm")
|
|
944
|
+
if pathlib.Path(f"{artifact}{suffix}").exists()
|
|
945
|
+
]
|
|
946
|
+
if leftovers:
|
|
947
|
+
raise OSError(
|
|
948
|
+
"stale stats rebuild artifacts remain after cleanup: "
|
|
949
|
+
+ ", ".join(leftovers)
|
|
950
|
+
)
|
|
951
|
+
|
|
952
|
+
|
|
953
|
+
def _recover_or_reclaim_interrupted_stats_rebuild(
|
|
954
|
+
db_path: pathlib.Path, artifacts: tuple[pathlib.Path, ...]
|
|
955
|
+
) -> bool:
|
|
956
|
+
"""Recover the legacy crash shape or reclaim a proven-stale scratch.
|
|
957
|
+
|
|
958
|
+
Caller holds maintenance EXCLUSIVE. A fully journal-consistent destination
|
|
959
|
+
proves every exact scratch family stale and needs cleanup only. Rebuilding
|
|
960
|
+
an absent or inconsistent destination additionally requires the completed
|
|
961
|
+
prebuild-quarantine incident that distinguishes the legacy interruption
|
|
962
|
+
from an unrelated file.
|
|
963
|
+
"""
|
|
964
|
+
if not artifacts:
|
|
965
|
+
return False
|
|
966
|
+
import _cctally_journal
|
|
967
|
+
|
|
968
|
+
if _cctally_db._would_block_prod_stats(db_path):
|
|
969
|
+
raise _cctally_db.ProdMigrationRefused(
|
|
970
|
+
"stats.db", "interrupted-rebuild-recovery"
|
|
971
|
+
)
|
|
972
|
+
high_water = _cctally_journal.journal_high_water()
|
|
973
|
+
matching_incident = _has_completed_stats_quarantine_incident(
|
|
974
|
+
db_path, artifacts
|
|
975
|
+
)
|
|
976
|
+
if not matching_incident:
|
|
977
|
+
# Task A never removes the old destination before publication. With no
|
|
978
|
+
# matching legacy prebuild-quarantine incident, exact scratch names are
|
|
979
|
+
# unpublished Task A artifacts and are safe to reclaim under the
|
|
980
|
+
# caller's maintenance EXCLUSIVE hold.
|
|
981
|
+
_remove_stale_stats_rebuild_artifacts(artifacts)
|
|
982
|
+
return True
|
|
983
|
+
if db_path.exists() and _cctally_journal.stats_index_matches_journal_prefix(
|
|
984
|
+
db_path, high_water
|
|
985
|
+
):
|
|
986
|
+
_remove_stale_stats_rebuild_artifacts(artifacts)
|
|
987
|
+
return True
|
|
988
|
+
if high_water is None or high_water[1] == 0:
|
|
989
|
+
return False
|
|
990
|
+
ingest_fd = _cctally_journal._acquire_ingest_lock("authoritative", 10.0)
|
|
991
|
+
if ingest_fd is None:
|
|
992
|
+
raise _cctally_db.StatsDbMaintenanceError(
|
|
993
|
+
"stats.db interrupted-rebuild recovery timed out waiting for "
|
|
994
|
+
"journal ingest serialization; retry after active cctally commands exit"
|
|
995
|
+
)
|
|
996
|
+
try:
|
|
997
|
+
with stats_write_scope("maintenance-interrupted-rebuild"):
|
|
998
|
+
_cctally_journal.rebuild_stats_index(high_water=high_water)
|
|
999
|
+
_remove_stale_stats_rebuild_artifacts(artifacts)
|
|
1000
|
+
return True
|
|
1001
|
+
finally:
|
|
1002
|
+
_cctally_journal._release_ingest_lock(ingest_fd)
|
|
1003
|
+
|
|
1004
|
+
|
|
1005
|
+
def stats_open_guarded(
|
|
1006
|
+
db_path=None, *, connect=None, recover_interruptions: bool = True
|
|
1007
|
+
) -> sqlite3.Connection:
|
|
1008
|
+
"""Open stats.db while excluding a destructive maintenance handshake (#386).
|
|
1009
|
+
|
|
1010
|
+
The shared maintenance flock covers the marker/pending checks AND the
|
|
1011
|
+
connect. A destructive maintenance path owns the exclusive side; after it
|
|
1012
|
+
publishes its marker/pending record, no new opener can escape into the live
|
|
1013
|
+
family while it verifies that pre-marker handles have drained.
|
|
1014
|
+
|
|
1015
|
+
``connect`` lets a caller keep its own open mode (``mode=ro`` for
|
|
1016
|
+
``db backup``, ``mode=rw`` for ``db status``) while still participating; it
|
|
1017
|
+
receives the path and returns the connection. Defaults to
|
|
1018
|
+
``sqlite3.connect``.
|
|
1019
|
+
"""
|
|
1020
|
+
db_path = pathlib.Path(
|
|
1021
|
+
db_path if db_path is not None else _cctally_core.DB_PATH
|
|
1022
|
+
)
|
|
1023
|
+
marker = _stats_repair_marker(db_path)
|
|
1024
|
+
_connect = connect if connect is not None else sqlite3.connect
|
|
1025
|
+
|
|
1026
|
+
live = db_path == pathlib.Path(_cctally_core.DB_PATH)
|
|
1027
|
+
if not live or _cctally_core.holds_stats_maintenance():
|
|
1028
|
+
# Scratch target, or this context already owns the exclusive side.
|
|
1029
|
+
# Pre-#386 behaviour, unchanged.
|
|
1030
|
+
if marker.exists():
|
|
1031
|
+
raise _cctally_db.StatsDbMaintenanceError()
|
|
1032
|
+
conn = _connect(db_path)
|
|
1033
|
+
arm_stats_authorizer(conn)
|
|
1034
|
+
return conn
|
|
1035
|
+
|
|
1036
|
+
pending = _cctally_db._quarantine_pending_path(db_path)
|
|
1037
|
+
lock_path = pathlib.Path(_cctally_core.STATS_LOCK_MAINTENANCE_PATH)
|
|
1038
|
+
lock_path.parent.mkdir(parents=True, exist_ok=True)
|
|
1039
|
+
lock_fh = open(lock_path, "a+")
|
|
1040
|
+
try:
|
|
1041
|
+
for _attempt in range(2):
|
|
1042
|
+
conn = None
|
|
1043
|
+
# BOUNDED, never blocking (#386 Stage 2 review P1-1) — see
|
|
1044
|
+
# _STATS_OPEN_MAINTENANCE_WAIT_S.
|
|
1045
|
+
if not _flock_bounded(
|
|
1046
|
+
lock_fh, fcntl.LOCK_SH, _STATS_OPEN_MAINTENANCE_WAIT_S
|
|
1047
|
+
):
|
|
1048
|
+
raise _cctally_db.StatsDbMaintenanceError(
|
|
1049
|
+
_STATS_OPEN_MAINTENANCE_TIMEOUT_MSG
|
|
1050
|
+
)
|
|
1051
|
+
if marker.exists():
|
|
1052
|
+
fcntl.flock(lock_fh, fcntl.LOCK_UN)
|
|
1053
|
+
raise _cctally_db.StatsDbMaintenanceError()
|
|
1054
|
+
if pending.exists():
|
|
1055
|
+
# Drop shared BEFORE taking exclusive so two resumers cannot
|
|
1056
|
+
# deadlock while upgrading; recheck under exclusive because a
|
|
1057
|
+
# live owner may have completed it in the gap.
|
|
1058
|
+
fcntl.flock(lock_fh, fcntl.LOCK_UN)
|
|
1059
|
+
if not _flock_bounded(
|
|
1060
|
+
lock_fh, fcntl.LOCK_EX, _STATS_OPEN_RESUME_WAIT_S
|
|
1061
|
+
):
|
|
1062
|
+
raise _cctally_db.StatsDbMaintenanceError(
|
|
1063
|
+
_STATS_OPEN_MAINTENANCE_TIMEOUT_MSG
|
|
1064
|
+
)
|
|
1065
|
+
try:
|
|
1066
|
+
if marker.exists():
|
|
1067
|
+
raise _cctally_db.StatsDbMaintenanceError()
|
|
1068
|
+
if pending.exists():
|
|
1069
|
+
_resume_pending_quarantine(db_path)
|
|
1070
|
+
finally:
|
|
1071
|
+
fcntl.flock(lock_fh, fcntl.LOCK_UN)
|
|
1072
|
+
continue
|
|
1073
|
+
artifacts = _stats_rebuild_artifact_bases(db_path)
|
|
1074
|
+
if (
|
|
1075
|
+
artifacts
|
|
1076
|
+
and recover_interruptions
|
|
1077
|
+
and _INTERRUPTED_RECOVERY_SUPPRESSED.get() == 0
|
|
1078
|
+
):
|
|
1079
|
+
# A live rebuild owns maintenance EXCLUSIVE, so reaching this
|
|
1080
|
+
# shared hold proves the owner is gone. Upgrade without holding
|
|
1081
|
+
# shared, then re-check every fact under exclusive before any
|
|
1082
|
+
# file-family mutation or live connect.
|
|
1083
|
+
fcntl.flock(lock_fh, fcntl.LOCK_UN)
|
|
1084
|
+
if not _flock_bounded(
|
|
1085
|
+
lock_fh, fcntl.LOCK_EX, _STATS_OPEN_RESUME_WAIT_S
|
|
1086
|
+
):
|
|
1087
|
+
raise _cctally_db.StatsDbMaintenanceError(
|
|
1088
|
+
_STATS_OPEN_MAINTENANCE_TIMEOUT_MSG
|
|
1089
|
+
)
|
|
1090
|
+
recovered = False
|
|
1091
|
+
try:
|
|
1092
|
+
current_artifacts = _stats_rebuild_artifact_bases(db_path)
|
|
1093
|
+
try:
|
|
1094
|
+
recovered = (
|
|
1095
|
+
_recover_or_reclaim_interrupted_stats_rebuild(
|
|
1096
|
+
db_path, current_artifacts
|
|
1097
|
+
)
|
|
1098
|
+
)
|
|
1099
|
+
except (
|
|
1100
|
+
_cctally_db.ProdMigrationRefused,
|
|
1101
|
+
_cctally_db.StatsDbMaintenanceError,
|
|
1102
|
+
):
|
|
1103
|
+
raise
|
|
1104
|
+
except Exception as exc:
|
|
1105
|
+
raise _cctally_db.StatsDbMaintenanceError(
|
|
1106
|
+
"stats.db interrupted-rebuild recovery failed: "
|
|
1107
|
+
f"{exc}. The best usable index was preserved; stale "
|
|
1108
|
+
"artifact cleanup may be incomplete. Run "
|
|
1109
|
+
"`cctally doctor`, resolve "
|
|
1110
|
+
"the reported journal problem, then run "
|
|
1111
|
+
"`cctally db rebuild --db stats`."
|
|
1112
|
+
) from exc
|
|
1113
|
+
finally:
|
|
1114
|
+
fcntl.flock(lock_fh, fcntl.LOCK_UN)
|
|
1115
|
+
if recovered:
|
|
1116
|
+
continue
|
|
1117
|
+
if not _flock_bounded(
|
|
1118
|
+
lock_fh, fcntl.LOCK_SH, _STATS_OPEN_MAINTENANCE_WAIT_S
|
|
1119
|
+
):
|
|
1120
|
+
raise _cctally_db.StatsDbMaintenanceError(
|
|
1121
|
+
_STATS_OPEN_MAINTENANCE_TIMEOUT_MSG
|
|
1122
|
+
)
|
|
1123
|
+
if marker.exists() or pending.exists():
|
|
1124
|
+
fcntl.flock(lock_fh, fcntl.LOCK_UN)
|
|
1125
|
+
continue
|
|
1126
|
+
try:
|
|
1127
|
+
conn = _connect(db_path)
|
|
1128
|
+
# Re-check inside the same shared hold: cheap, and it closes the
|
|
1129
|
+
# window between the checks above and a slow connect.
|
|
1130
|
+
if marker.exists() or pending.exists():
|
|
1131
|
+
conn.close()
|
|
1132
|
+
conn = None
|
|
1133
|
+
raise _cctally_db.StatsDbMaintenanceError()
|
|
1134
|
+
# #386 enforcement: EVERY stats connection this module hands out
|
|
1135
|
+
# carries the authorizer. Arming HERE and nowhere else is what
|
|
1136
|
+
# keeps raw `sqlite3.connect` escape hatches (the storm suite's
|
|
1137
|
+
# `_grow_wal`, `db checkpoint`'s `mode=rw`) unaffected — a
|
|
1138
|
+
# broader arming point would make their writes unsanctioned and
|
|
1139
|
+
# the correct fix would then be to narrow the arming, never to
|
|
1140
|
+
# weaken the guard.
|
|
1141
|
+
arm_stats_authorizer(conn)
|
|
1142
|
+
return conn
|
|
1143
|
+
except BaseException:
|
|
1144
|
+
if conn is not None:
|
|
1145
|
+
try:
|
|
1146
|
+
conn.close()
|
|
1147
|
+
except Exception:
|
|
1148
|
+
pass
|
|
1149
|
+
raise
|
|
1150
|
+
finally:
|
|
1151
|
+
fcntl.flock(lock_fh, fcntl.LOCK_UN)
|
|
1152
|
+
raise _cctally_db.StatsDbMaintenanceError()
|
|
1153
|
+
finally:
|
|
1154
|
+
lock_fh.close()
|
|
1155
|
+
|
|
1156
|
+
|
|
1157
|
+
def _acquire_stats_maintenance_reentrant(path) -> "int | None":
|
|
1158
|
+
"""Take ``stats.db.maintenance.lock`` EXCLUSIVE unless we already hold it.
|
|
1159
|
+
|
|
1160
|
+
Returns the held fd, or ``None`` when THIS execution context already owns the
|
|
1161
|
+
lock (in which case the caller must not release anything).
|
|
1162
|
+
|
|
1163
|
+
#386 Stage 2 review P1-2. ``flock`` conflicts are per open-file-DESCRIPTION
|
|
1164
|
+
and apply WITHIN a process: holding SHARED on one fd and then requesting
|
|
1165
|
+
EXCLUSIVE on a second fd of the same file blocks the process against itself,
|
|
1166
|
+
indefinitely. ``run_stats_ingest`` holds maintenance SHARED across its entire
|
|
1167
|
+
cycle, and both callers of this helper — the heal hook and the epoch resolver
|
|
1168
|
+
— are reachable from a nested ``open_db()`` inside that cycle. Without this
|
|
1169
|
+
check that nested open is an unconditional self-deadlock.
|
|
1170
|
+
|
|
1171
|
+
Proceeding on a shared hold is a deliberate, narrow weakening: the caller
|
|
1172
|
+
still runs ``_stats_family_drained`` before any physical replacement, which
|
|
1173
|
+
is a WHOLE-SYSTEM handle scan and therefore catches any sibling that could
|
|
1174
|
+
be harmed. The alternative — hanging forever — is strictly worse.
|
|
1175
|
+
"""
|
|
1176
|
+
if _cctally_core.holds_stats_maintenance():
|
|
1177
|
+
return None
|
|
1178
|
+
return _heal_flock_blocking(path)
|
|
1179
|
+
|
|
1180
|
+
|
|
1181
|
+
def _release_stats_maintenance_reentrant(fd: "int | None") -> None:
|
|
1182
|
+
"""Release what ``_acquire_stats_maintenance_reentrant`` took, if anything."""
|
|
1183
|
+
if fd is not None:
|
|
1184
|
+
_heal_release_maintenance_flock(fd)
|
|
1185
|
+
|
|
1186
|
+
|
|
326
1187
|
def _heal_flock_blocking(path) -> int:
|
|
1188
|
+
"""Blocking EX flock. Both call sites target STATS_LOCK_MAINTENANCE_PATH, so
|
|
1189
|
+
a successful acquire also records the #386 maintenance hold — pair it with
|
|
1190
|
+
``_heal_release_maintenance_flock``, never the plain release.
|
|
1191
|
+
|
|
1192
|
+
Callers must reach this through ``_acquire_stats_maintenance_reentrant``, so
|
|
1193
|
+
a context that already owns the lock never requests it a second time.
|
|
1194
|
+
"""
|
|
327
1195
|
_cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
|
|
328
1196
|
fd = os.open(str(path), os.O_RDWR | os.O_CREAT, 0o600)
|
|
329
1197
|
try:
|
|
@@ -331,12 +1199,45 @@ def _heal_flock_blocking(path) -> int:
|
|
|
331
1199
|
except BaseException:
|
|
332
1200
|
os.close(fd)
|
|
333
1201
|
raise
|
|
1202
|
+
if str(path) == str(_cctally_core.STATS_LOCK_MAINTENANCE_PATH):
|
|
1203
|
+
_cctally_core.note_stats_maintenance_acquired()
|
|
334
1204
|
return fd
|
|
335
1205
|
|
|
336
1206
|
|
|
337
|
-
def
|
|
338
|
-
"""
|
|
339
|
-
|
|
1207
|
+
def _heal_release_maintenance_flock(fd: int) -> None:
|
|
1208
|
+
"""Release a stats maintenance flock taken by ``_heal_flock_blocking``.
|
|
1209
|
+
|
|
1210
|
+
Distinct from ``_heal_release_flock`` because that helper is also used for
|
|
1211
|
+
the INGEST fd, which carries no maintenance hold to unwind.
|
|
1212
|
+
"""
|
|
1213
|
+
try:
|
|
1214
|
+
_cctally_core.note_stats_maintenance_released()
|
|
1215
|
+
finally:
|
|
1216
|
+
_heal_release_flock(fd)
|
|
1217
|
+
|
|
1218
|
+
|
|
1219
|
+
def _heal_flock_bounded(path, timeout_s: float) -> "int | None":
|
|
1220
|
+
"""Bounded EX flock. Returns the HELD fd, or ``None`` on timeout.
|
|
1221
|
+
|
|
1222
|
+
#386: this previously returned the OPEN fd *without* the lock held and let
|
|
1223
|
+
the caller proceed "best-effort", which meant the heal path described itself
|
|
1224
|
+
as serialized while running unserialized — and no caller could tell the two
|
|
1225
|
+
outcomes apart, because both were an ``int``.
|
|
1226
|
+
|
|
1227
|
+
The re-entrancy case that motivated the old behaviour is real and is NOT
|
|
1228
|
+
solved by simply aborting on timeout: a corruption surfacing from INSIDE a
|
|
1229
|
+
``run_stats_ingest`` cycle already holds ``journal.ingest.lock``, so an
|
|
1230
|
+
indefinite (or fail-closed) wait would deadlock the process against itself.
|
|
1231
|
+
That case is now detected EXPLICITLY at the call sites via
|
|
1232
|
+
``holds_ingest_lock()`` — the ingester enters ``stats_write_scope(...,
|
|
1233
|
+
ingest_lock=True)`` around its cycle — so a timeout here means some OTHER
|
|
1234
|
+
holder has it, and failing soft is correct: decline the heal and let a later
|
|
1235
|
+
open retry.
|
|
1236
|
+
|
|
1237
|
+
Spec §5.1 says "a timeout aborts the operation rather than continuing
|
|
1238
|
+
unlocked". Read literally that would reintroduce the self-deadlock; the
|
|
1239
|
+
plan's Stage 2 correction (and this docstring) is the operative version.
|
|
1240
|
+
"""
|
|
340
1241
|
_cctally_core.APP_DIR.mkdir(parents=True, exist_ok=True)
|
|
341
1242
|
fd = os.open(str(path), os.O_RDWR | os.O_CREAT, 0o600)
|
|
342
1243
|
deadline = time.monotonic() + timeout_s
|
|
@@ -347,13 +1248,54 @@ def _heal_flock_bounded(path, timeout_s: float) -> int:
|
|
|
347
1248
|
return fd
|
|
348
1249
|
except (BlockingIOError, OSError):
|
|
349
1250
|
if time.monotonic() >= deadline:
|
|
350
|
-
|
|
1251
|
+
os.close(fd)
|
|
1252
|
+
return None
|
|
351
1253
|
time.sleep(0.02)
|
|
352
1254
|
except BaseException:
|
|
353
1255
|
os.close(fd)
|
|
354
1256
|
raise
|
|
355
1257
|
|
|
356
1258
|
|
|
1259
|
+
def _stats_storm_test_pause(point: str) -> None:
|
|
1260
|
+
"""Private process-control seam for the #386 stats writer-storm harness.
|
|
1261
|
+
|
|
1262
|
+
Production is a zero-cost string comparison. A test arms one exact point plus
|
|
1263
|
+
a marker path, waits for the marker, and then drives the SIGSTOPped child
|
|
1264
|
+
from the parent. Mirrors `_cctally_cache._cache_storm_test_pause`, which has
|
|
1265
|
+
carried the cache half of this since #344.
|
|
1266
|
+
|
|
1267
|
+
This is the ONLY way to hit spec section 1.1 Gap A's window deterministically:
|
|
1268
|
+
the instant after the handle scan says "drained" and before the first rename,
|
|
1269
|
+
which is exactly where a new opener must not be able to arrive.
|
|
1270
|
+
"""
|
|
1271
|
+
if os.environ.get("CCTALLY_TEST_STATS_STORM_PAUSE_AT") != point:
|
|
1272
|
+
return
|
|
1273
|
+
marker = os.environ.get("CCTALLY_TEST_STATS_STORM_MARKER")
|
|
1274
|
+
if not marker:
|
|
1275
|
+
return
|
|
1276
|
+
pathlib.Path(marker).write_text(f"{os.getpid()}\n")
|
|
1277
|
+
os.kill(os.getpid(), signal.SIGSTOP)
|
|
1278
|
+
|
|
1279
|
+
|
|
1280
|
+
def _stats_family_drained(path) -> "str | None":
|
|
1281
|
+
"""``None`` when no handle is open on the stats family; else why not.
|
|
1282
|
+
|
|
1283
|
+
Physical replacement renames files out from under whatever has them mapped.
|
|
1284
|
+
SQLite's crash guarantees stop applying at that point (spec §1.2), so the
|
|
1285
|
+
drain check is a precondition, not a nicety — and "the platform could not
|
|
1286
|
+
tell us" is a refusal, not a pass.
|
|
1287
|
+
"""
|
|
1288
|
+
open_pids = _cctally_db._db_family_open_pids(path)
|
|
1289
|
+
if open_pids is None:
|
|
1290
|
+
return "could not verify that the database family has no open handles"
|
|
1291
|
+
if open_pids:
|
|
1292
|
+
return (
|
|
1293
|
+
"family is still open in process(es) "
|
|
1294
|
+
+ ", ".join(str(pid) for pid in sorted(open_pids))
|
|
1295
|
+
)
|
|
1296
|
+
return None
|
|
1297
|
+
|
|
1298
|
+
|
|
357
1299
|
def _heal_release_flock(fd: int) -> None:
|
|
358
1300
|
try:
|
|
359
1301
|
fcntl.flock(fd, fcntl.LOCK_UN)
|
|
@@ -380,7 +1322,34 @@ def _probe_stats_ok(path) -> bool:
|
|
|
380
1322
|
return False
|
|
381
1323
|
|
|
382
1324
|
|
|
383
|
-
def
|
|
1325
|
+
def _probe_stats_integrity_ok(path) -> bool:
|
|
1326
|
+
"""Positive whole-index re-check for a post-query corruption report.
|
|
1327
|
+
|
|
1328
|
+
The ordinary locked probe intentionally stays cheap because it serves the
|
|
1329
|
+
open-time boundary. A dashboard leg has already observed a corruption
|
|
1330
|
+
error after that boundary, so its sibling-healed re-check must exercise the
|
|
1331
|
+
index B-trees rather than repeat ``PRAGMA schema_version``.
|
|
1332
|
+
"""
|
|
1333
|
+
|
|
1334
|
+
if not path.exists():
|
|
1335
|
+
return False
|
|
1336
|
+
try:
|
|
1337
|
+
c = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
|
|
1338
|
+
try:
|
|
1339
|
+
rows = c.execute("PRAGMA quick_check").fetchall()
|
|
1340
|
+
finally:
|
|
1341
|
+
c.close()
|
|
1342
|
+
return rows == [("ok",)]
|
|
1343
|
+
except sqlite3.DatabaseError:
|
|
1344
|
+
return False
|
|
1345
|
+
|
|
1346
|
+
|
|
1347
|
+
def _stats_heal_hook(
|
|
1348
|
+
store: str,
|
|
1349
|
+
exc: Exception,
|
|
1350
|
+
*,
|
|
1351
|
+
post_query: bool = False,
|
|
1352
|
+
) -> bool:
|
|
384
1353
|
"""Classifier-gated corruption auto-heal for stats.db (spec §6.3). Returns
|
|
385
1354
|
True when it healed (quarantined + rebuilt) OR a sibling already healed under
|
|
386
1355
|
the maintenance lock; False when it DECLINES — a non-corruption
|
|
@@ -412,19 +1381,38 @@ def _stats_heal_hook(store: str, exc: Exception) -> bool:
|
|
|
412
1381
|
return False
|
|
413
1382
|
_HEAL_ACTIVE = True
|
|
414
1383
|
try:
|
|
415
|
-
maint_fd =
|
|
1384
|
+
maint_fd = _acquire_stats_maintenance_reentrant(
|
|
1385
|
+
_cctally_core.STATS_LOCK_MAINTENANCE_PATH)
|
|
416
1386
|
try:
|
|
417
|
-
if _probe_stats_ok
|
|
1387
|
+
probe = _probe_stats_integrity_ok if post_query else _probe_stats_ok
|
|
1388
|
+
if probe(path):
|
|
418
1389
|
return True # a sibling process already healed it — retry the open
|
|
1390
|
+
# Forensics FIRST — before anything disturbs the evidence.
|
|
419
1391
|
_cctally_db.write_corruption_forensics(path, db_label="stats")
|
|
420
|
-
|
|
421
|
-
|
|
1392
|
+
if holds_ingest_lock():
|
|
1393
|
+
ingest_fd = None # this context IS the serialized writer
|
|
1394
|
+
else:
|
|
1395
|
+
ingest_fd = _heal_flock_bounded(
|
|
1396
|
+
_cctally_core.JOURNAL_INGEST_LOCK_PATH, 5.0)
|
|
1397
|
+
if ingest_fd is None:
|
|
1398
|
+
print(
|
|
1399
|
+
"[heal] stats.db auto-heal declined: another ingest "
|
|
1400
|
+
"holds journal.ingest.lock; a later open will retry.",
|
|
1401
|
+
file=sys.stderr,
|
|
1402
|
+
)
|
|
1403
|
+
return False
|
|
422
1404
|
try:
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
1405
|
+
# #386: the rebuild writes the fresh scratch index through
|
|
1406
|
+
# `open_db(_target_path=...)`, whose connection carries the
|
|
1407
|
+
# authorizer. Declare the sanctioned maintenance regime for the
|
|
1408
|
+
# whole replacement — we hold (or already held) maintenance
|
|
1409
|
+
# exclusive, which is exactly what spec §3.1 sanctions.
|
|
1410
|
+
with stats_write_scope("maintenance-heal"):
|
|
1411
|
+
import _cctally_journal
|
|
1412
|
+
_cctally_journal.rebuild_stats_index()
|
|
426
1413
|
finally:
|
|
427
|
-
|
|
1414
|
+
if ingest_fd is not None:
|
|
1415
|
+
_heal_release_flock(ingest_fd)
|
|
428
1416
|
print(
|
|
429
1417
|
f"[heal] stats.db was corrupt ({exc}); quarantined its file family "
|
|
430
1418
|
"under quarantine/ (forensics in logs/) and rebuilt a fresh index "
|
|
@@ -433,7 +1421,7 @@ def _stats_heal_hook(store: str, exc: Exception) -> bool:
|
|
|
433
1421
|
)
|
|
434
1422
|
return True
|
|
435
1423
|
finally:
|
|
436
|
-
|
|
1424
|
+
_release_stats_maintenance_reentrant(maint_fd)
|
|
437
1425
|
except Exception as heal_exc:
|
|
438
1426
|
print(f"[heal] stats.db auto-heal failed: {heal_exc}", file=sys.stderr)
|
|
439
1427
|
return False
|
|
@@ -490,12 +1478,15 @@ def resolve_stats_epoch_mismatch():
|
|
|
490
1478
|
"journal and run `cctally db rebuild --db stats`.")
|
|
491
1479
|
_EPOCH_MISMATCH_ACTIVE = True
|
|
492
1480
|
try:
|
|
493
|
-
maint_fd =
|
|
1481
|
+
maint_fd = _acquire_stats_maintenance_reentrant(
|
|
1482
|
+
_cctally_core.STATS_LOCK_MAINTENANCE_PATH)
|
|
494
1483
|
try:
|
|
495
1484
|
# Locked re-check: a sibling process may have already rebuilt it.
|
|
496
1485
|
if _raw_user_version(path) != _cctally_core.STATS_INDEX_EPOCH:
|
|
497
|
-
hw =
|
|
498
|
-
|
|
1486
|
+
hw, journal_has_bytes = (
|
|
1487
|
+
_cctally_journal._journal_rebuild_snapshot()
|
|
1488
|
+
)
|
|
1489
|
+
if hw is None or not journal_has_bytes:
|
|
499
1490
|
raise _cctally_db.StatsEpochMismatchError(
|
|
500
1491
|
f"stats.db is at index epoch {_raw_user_version(path)}, "
|
|
501
1492
|
f"but this cctally builds epoch "
|
|
@@ -503,18 +1494,35 @@ def resolve_stats_epoch_mismatch():
|
|
|
503
1494
|
"present to rebuild from. The journal/ directory is the "
|
|
504
1495
|
"durable source — restore it from backup, then run "
|
|
505
1496
|
"`cctally db rebuild --db stats`.")
|
|
506
|
-
#
|
|
507
|
-
#
|
|
508
|
-
#
|
|
509
|
-
#
|
|
510
|
-
#
|
|
511
|
-
#
|
|
512
|
-
#
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
1497
|
+
# Total lock order is maintenance -> ingest. The epoch bump can
|
|
1498
|
+
# be discovered by an ordinary open or an ingest caller, so it
|
|
1499
|
+
# must take the ingest lock here before building and publishing
|
|
1500
|
+
# the replacement; callers must never enter this resolver while
|
|
1501
|
+
# already holding that later lock. #386: if this context DOES
|
|
1502
|
+
# already hold it, say so rather than deadlocking against
|
|
1503
|
+
# ourselves.
|
|
1504
|
+
if holds_ingest_lock():
|
|
1505
|
+
ingest_fd = None
|
|
1506
|
+
else:
|
|
1507
|
+
ingest_fd = _cctally_journal._acquire_ingest_lock(
|
|
1508
|
+
"authoritative", 10.0
|
|
1509
|
+
)
|
|
1510
|
+
if ingest_fd is None:
|
|
1511
|
+
raise _cctally_db.StatsEpochMismatchError(
|
|
1512
|
+
"timed out waiting for journal ingest serialization "
|
|
1513
|
+
"during stats.db epoch rebuild"
|
|
1514
|
+
)
|
|
1515
|
+
try:
|
|
1516
|
+
# Append the idempotent account coordinator input, then let
|
|
1517
|
+
# the common rebuild cutover preserve the version-ahead
|
|
1518
|
+
# family and atomically publish the current epoch.
|
|
1519
|
+
with stats_write_scope("maintenance-epoch"):
|
|
1520
|
+
_cctally_journal.run_epoch_transition()
|
|
1521
|
+
finally:
|
|
1522
|
+
if ingest_fd is not None:
|
|
1523
|
+
_cctally_journal._release_ingest_lock(ingest_fd)
|
|
516
1524
|
finally:
|
|
517
|
-
|
|
1525
|
+
_release_stats_maintenance_reentrant(maint_fd)
|
|
518
1526
|
return _cctally_core.open_db()
|
|
519
1527
|
finally:
|
|
520
1528
|
_EPOCH_MISMATCH_ACTIVE = False
|