cctally 1.93.0 → 1.94.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -66,6 +66,7 @@ import pathlib
66
66
  import re
67
67
  import signal
68
68
  import sqlite3
69
+ import stat
69
70
  import sys
70
71
  import time
71
72
  from dataclasses import dataclass
@@ -186,6 +187,63 @@ def apply_connection_policy(conn: sqlite3.Connection, store: str) -> None:
186
187
  conn.execute(f"PRAGMA journal_size_limit={policy.journal_size_limit}")
187
188
 
188
189
 
190
+ # --------------------------------------------------------------------------
191
+ # §9.2 — the stats family is published privately (#496 S6 F23)
192
+ # --------------------------------------------------------------------------
193
+
194
+ #: The at-rest mode of `stats.db` and its sidecars. `cache.db`,
195
+ #: `conversations.db` and their sidecars are already published this way; the
196
+ #: stats family was created at the process umask, which is `022` on the
197
+ #: maintainer's machine, so a bare create landed at `0644`.
198
+ STATS_FAMILY_MODE = 0o600
199
+
200
+
201
+ def _harden_stats_family(path) -> None:
202
+ """Best-effort `0600` on a stats database and its `-wal`/`-shm` sidecars.
203
+
204
+ **This closes a defense-in-depth inconsistency, not a live exposure**
205
+ (§9.1). `ensure_dirs()` chmods the data directory to `0700`, so no other
206
+ local user can traverse to the file. It is corrected because a copied file
207
+ carries its own mode, a data directory created outside `ensure_dirs()` need
208
+ not be `0700`, and an inconsistent layer becomes a real hole when the layer
209
+ above changes.
210
+
211
+ **Every invocation checks the mode; there is no path memo.** A memo that
212
+ skipped the check after one success would leave a long-lived dashboard
213
+ process unable to repair an external `chmod`, and a same-path `os.replace`
214
+ installs a new inode behind a memoized path. The comparison is what avoids
215
+ the repeated `chmod`: an `lstat` is cheap, and a `chmod` on an
216
+ already-correct file is the cost being avoided. The statusline opens the
217
+ index at three sites, so the steady state is three `lstat` triples per
218
+ render and zero `chmod` calls.
219
+
220
+ The `lstat` does not follow symlinks, and a member that IS a symlink is
221
+ left alone rather than chmod'd through. This function never unlinks and
222
+ never renames (§9.3): `chmod` changes metadata on an existing inode and
223
+ does not invalidate an open descriptor, which is categorically different
224
+ from removing a live sidecar — the defect that closed as #516. A sidecar
225
+ SQLite removes between the `lstat` and the `chmod` surfaces as a swallowed
226
+ `ENOENT`.
227
+ """
228
+ base = str(path)
229
+ for member in (base, base + "-wal", base + "-shm"):
230
+ try:
231
+ info = os.lstat(member)
232
+ except OSError:
233
+ continue
234
+ if stat.S_ISLNK(info.st_mode):
235
+ continue
236
+ if stat.S_IMODE(info.st_mode) == STATS_FAMILY_MODE:
237
+ continue
238
+ try:
239
+ os.chmod(member, STATS_FAMILY_MODE)
240
+ except OSError as exc:
241
+ print(
242
+ f"[store] could not chmod {member} 0600 ({exc}); continuing",
243
+ file=sys.stderr,
244
+ )
245
+
246
+
189
247
  # --------------------------------------------------------------------------
190
248
  # §6.2 version gate
191
249
  # --------------------------------------------------------------------------
@@ -1068,8 +1126,14 @@ def _resume_pending_quarantine(db_path: pathlib.Path) -> None:
1068
1126
  + ", ".join(str(pid) for pid in sorted(open_pids))
1069
1127
  )
1070
1128
  # Resumes the SAME incident from the pending record -- never a second
1071
- # incident dir, never a recreation.
1072
- _cctally_db.quarantine_db_family(db_path, strict=True)
1129
+ # incident dir, never a recreation. #496 S6 §5.3: held SHARED across
1130
+ # the resume, because the final manifest is published here.
1131
+ import _cctally_retention
1132
+
1133
+ with _cctally_retention.retention_shared(
1134
+ label="stats quarantine resume"
1135
+ ):
1136
+ _cctally_db.quarantine_db_family(db_path, strict=True)
1073
1137
  except OSError as exc:
1074
1138
  raise sqlite3.OperationalError(
1075
1139
  f"stats.db pending quarantine could not resume: {exc}"
@@ -1754,8 +1818,8 @@ def _stats_heal_hook(
1754
1818
  catches it and degrades, while ``_cctally_core.open_db`` deliberately lets
1755
1819
  it propagate. The heal that replaced the index and then failed to validate
1756
1820
  it must report that itself, because ``open_db``'s decline branch would tell
1757
- the user the database was "Not auto-recreated" — false once replacement has
1758
- occurred."""
1821
+ the user the database "is never auto-recreated" — false once replacement
1822
+ has occurred."""
1759
1823
  global _HEAL_ACTIVE
1760
1824
  if store != "stats":
1761
1825
  return False
@@ -1799,10 +1863,20 @@ def _stats_heal_hook(
1799
1863
  file=sys.stderr,
1800
1864
  )
1801
1865
  return False
1866
+ import _cctally_retention
1867
+
1868
+ # #496 S6 §5.3, the parent's phase. SHARED from before the forensics
1869
+ # bundle until either the request marker naming it is fsynced or the
1870
+ # decision is terminal. A decline writes no marker and needs no bridge,
1871
+ # because nothing will act on the bundle later.
1872
+ retention = contextlib.ExitStack()
1802
1873
  try:
1803
1874
  probe = _probe_stats_integrity_ok if post_query else _probe_stats_ok
1804
1875
  if probe(path):
1805
1876
  return True # a sibling process already healed it — retry the open
1877
+ retention.enter_context(
1878
+ _cctally_retention.retention_shared(label="stats auto-heal")
1879
+ )
1806
1880
  # Forensics FIRST — before anything disturbs the evidence. The
1807
1881
  # trigger pair is what arms the #496 S1 forensics-time WAL capture
1808
1882
  # and what lets the quarantine incident name the bundle that
@@ -1852,13 +1926,43 @@ def _stats_heal_hook(
1852
1926
  )
1853
1927
  return False
1854
1928
  append_stats_heal_event(build_stats_heal_event(request, "confirmed"))
1929
+ # Phase 1: reserve and persist, UNDER maintenance and the shared
1930
+ # retention hold. The marker is durable when this returns
1931
+ # `reserved`, which is what closes the bundle-to-request gap.
1932
+ reservation = reserve_stats_corruption_heal(request)
1933
+ if reservation != "reserved":
1934
+ # #496 S6 §5.3, the SECOND terminal decision. The shared hold
1935
+ # is about to be released over a forensics bundle the durable
1936
+ # request marker does not name — it names the earlier
1937
+ # detection's bundle. That is legitimate for the same reason a
1938
+ # decline is: nothing will act on THIS bundle later. Rewriting
1939
+ # the marker to name it instead would be worse, because a
1940
+ # worker that has already read the marker would attribute its
1941
+ # incident to a bundle it never captured, and the record would
1942
+ # mix two detections. §3.3 then governs the stranded bundle as
1943
+ # an unreferenced one, which self-classifies from its own
1944
+ # `trigger.origin` and is reclaimable under the ordinary
1945
+ # bounds. Terminalizing it here, inside the hold, is what makes
1946
+ # the decision durable before the hold drops, and stops the
1947
+ # ring entry sitting at `detected` forever.
1948
+ update_stats_heal_event(
1949
+ request["healId"],
1950
+ outcome=(
1951
+ "coalesced" if reservation == "pending"
1952
+ else "admission-failed"
1953
+ ),
1954
+ )
1855
1955
  finally:
1856
- # Released BEFORE deferring: the worker takes maintenance
1956
+ # Retention is released first: it sits BELOW maintenance in the
1957
+ # lock order, so it must not outlive the hold it was taken under.
1958
+ retention.close()
1959
+ # Maintenance released BEFORE spawning: the worker takes it
1857
1960
  # EXCLUSIVE as a fresh process holding nothing, and a caller still
1858
1961
  # holding it here would make that acquire wait for a request it is
1859
1962
  # itself in the middle of filing.
1860
1963
  _release_stats_maintenance_for_heal(maint_fd)
1861
- outcome = defer_stats_corruption_heal(request)
1964
+ # Phase 3: spawn, under no lock at all.
1965
+ outcome = complete_stats_corruption_heal(reservation)
1862
1966
  # F15 (#496 S3 §7). Detachment supplies the timing for free: report at
1863
1967
  # DETECTION, naming the absolute forensics path and the heal id. It
1864
1968
  # cannot name an incident path, because the quarantine directory is
@@ -1898,7 +2002,7 @@ def _stats_heal_hook(
1898
2002
  print(f"[heal] stats.db auto-heal failed: {heal_exc}", file=sys.stderr)
1899
2003
  # A post-publication validation failure has ALREADY replaced the index.
1900
2004
  # Declining here sends `open_db` to its pre-existing branch, which tells
1901
- # the user the database was "Not auto-recreated" and to run
2005
+ # the user the database "is never auto-recreated" and to run
1902
2006
  # `db repair --db stats --yes` — both false once replacement occurred.
1903
2007
  # The durable marker makes the NEXT process say the right thing; the
1904
2008
  # process that caused the failure must say it too (#496 S1 F1).
@@ -2085,8 +2189,27 @@ def _log_stats_heal(
2085
2189
  pass
2086
2190
 
2087
2191
 
2088
- def defer_stats_corruption_heal(request: dict) -> str:
2089
- """Schedule one retryable detached corruption heal without blocking."""
2192
+ def reserve_stats_corruption_heal(request: dict) -> str:
2193
+ """Phase 1 of the deferral: decide admission and make the marker durable.
2194
+
2195
+ #496 S6 §5.3 splits `defer_stats_corruption_heal` in two so the documented
2196
+ lock order is satisfiable at all. Reservation — the admission flock, the
2197
+ marker-age check, the worker-active probe and the fsynced marker write —
2198
+ must happen while the caller still holds stats maintenance and the SHARED
2199
+ retention lock, because the marker is what bridges the parent-to-worker
2200
+ process boundary and tells reclamation that the forensics bundle it names
2201
+ is still live. The spawn must happen AFTER maintenance is released,
2202
+ because the worker takes maintenance exclusive as a fresh process.
2203
+
2204
+ The heal ring cannot serve as that bridge: `_mutate_stats_heal_ring`
2205
+ returns False on a mkdir failure, a bounded-lock timeout and an `OSError`
2206
+ during the write, and both hook call sites discard that result — so on a
2207
+ legitimate failure path the ring entry does not exist and the bundle would
2208
+ be unprotected.
2209
+
2210
+ Returns `reserved` (the marker is durable and the caller MUST spawn),
2211
+ `pending` (coalesced onto an existing request) or `failed`.
2212
+ """
2090
2213
  try:
2091
2214
  pathlib.Path(_cctally_core.APP_DIR).mkdir(parents=True, exist_ok=True)
2092
2215
  admission_fd = os.open(
@@ -2121,11 +2244,50 @@ def defer_stats_corruption_heal(request: dict) -> str:
2121
2244
  _cctally_db._atomic_write_private_json(marker, request)
2122
2245
  except OSError:
2123
2246
  return "failed"
2124
- from _cctally_update import _spawn_detached
2125
- if _spawn_detached(STATS_CORRUPTION_HEAL_COMMAND):
2126
- return "spawned"
2247
+ return "reserved"
2248
+ finally:
2249
+ try:
2250
+ fcntl.flock(admission_fd, fcntl.LOCK_UN)
2251
+ except OSError:
2252
+ pass
2253
+ os.close(admission_fd)
2254
+
2255
+
2256
+ def complete_stats_corruption_heal(reservation: str) -> str:
2257
+ """Phase 3 of the deferral: spawn the worker a reservation admitted.
2258
+
2259
+ Takes no lock: the spawn need not occur under any, and the maintenance hold
2260
+ must already have been released so the worker can take it exclusive. A
2261
+ non-`reserved` reservation is returned unchanged, so a coalesced or failed
2262
+ admission reports exactly what it reported before the split.
2263
+ """
2264
+ if reservation != "reserved":
2265
+ return reservation
2266
+ from _cctally_update import _spawn_detached
2267
+ if _spawn_detached(STATS_CORRUPTION_HEAL_COMMAND):
2268
+ return "spawned"
2269
+ # Drop our own marker so the next detection is admitted immediately rather
2270
+ # than waiting out the retry window. Done under the admission flock, as it
2271
+ # was before the split, so no concurrent caller can coalesce onto a marker
2272
+ # that is about to disappear. If the flock is unavailable the marker stays
2273
+ # and the retry window clears it — the retryable direction.
2274
+ _unlink_stats_heal_marker_under_admission()
2275
+ return "failed"
2276
+
2277
+
2278
+ def _unlink_stats_heal_marker_under_admission() -> None:
2279
+ try:
2280
+ admission_fd = os.open(
2281
+ _stats_heal_admission_path(), os.O_WRONLY | os.O_CREAT, 0o600
2282
+ )
2283
+ except OSError:
2284
+ return
2285
+ try:
2286
+ try:
2287
+ fcntl.flock(admission_fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
2288
+ except OSError:
2289
+ return
2127
2290
  _unlink_stats_heal_marker()
2128
- return "failed"
2129
2291
  finally:
2130
2292
  try:
2131
2293
  fcntl.flock(admission_fd, fcntl.LOCK_UN)
@@ -2134,6 +2296,18 @@ def defer_stats_corruption_heal(request: dict) -> str:
2134
2296
  os.close(admission_fd)
2135
2297
 
2136
2298
 
2299
+ def defer_stats_corruption_heal(request: dict) -> str:
2300
+ """Schedule one retryable detached corruption heal without blocking.
2301
+
2302
+ The composed form of the two phases above, kept for callers that hold no
2303
+ maintenance lock and therefore have no handoff to bridge. The auto-heal
2304
+ hook calls the two phases separately (§5.3).
2305
+ """
2306
+ return complete_stats_corruption_heal(
2307
+ reserve_stats_corruption_heal(request)
2308
+ )
2309
+
2310
+
2137
2311
  def _run_stats_corruption_heal(request: dict) -> str:
2138
2312
  """The worker's body, under its own maintenance-EXCLUSIVE hold.
2139
2313
 
@@ -2184,8 +2358,14 @@ def _run_stats_corruption_heal(request: dict) -> str:
2184
2358
  )
2185
2359
  if ingest_fd is None:
2186
2360
  return "ingest-busy"
2361
+ # #496 S6 §5.3, the worker's phase: SHARED after maintenance and
2362
+ # ingest, before publication, released after the final rebuild record.
2363
+ # The request marker is cleared last, by the caller, once this returns.
2364
+ import _cctally_retention
2365
+
2187
2366
  try:
2188
- with stats_write_scope("maintenance-heal"):
2367
+ with _cctally_retention.retention_shared(label="stats heal worker"), \
2368
+ stats_write_scope("maintenance-heal"):
2189
2369
  result = _cctally_journal.rebuild_stats_index(
2190
2370
  context=_cctally_journal.RebuildContext(
2191
2371
  trigger="corruption-heal",