cctally 1.93.1 → 1.94.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,2890 @@
1
+ """Glue for retained-artifact retention (#496 S6).
2
+
3
+ Everything in this module touches the filesystem. The decisions it feeds live
4
+ in the pure kernel `bin/_lib_artifact_retention.py`, which takes no filesystem,
5
+ no locks, no clock and no config.
6
+
7
+ This file currently carries the producer lock of §5.3, the family-parameterized
8
+ discovery and classification backfill of §4.3 and §4.5, and the two-phase
9
+ mark-then-delete engine of §5.4 and §5.5. The metadata walk, the detached
10
+ worker and `cmd_db_prune` join it later in the session.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import contextlib
15
+ import dataclasses
16
+ import datetime as dt
17
+ import fcntl
18
+ import json
19
+ import os
20
+ import pathlib
21
+ import re
22
+ import stat as _stat
23
+ import sys
24
+ import threading
25
+ import time
26
+
27
+ import _cctally_core
28
+ import _lib_artifact_retention as _kernel
29
+
30
+ # Public surface: shipped in the npm tarball + brew formula + public mirror.
31
+
32
+ # --------------------------------------------------------------------------
33
+ # §5.3 — the producer side of `artifact-retention.lock`
34
+ # --------------------------------------------------------------------------
35
+ #
36
+ # Placement in the lock-order law: after the conversation provider flocks and
37
+ # before SQLite transactions, so `journal.lock` stays the leaf. The worker takes
38
+ # it EXCLUSIVE holding nothing earlier, so a producer waiting for SHARED while
39
+ # holding an earlier lock cannot cycle against it.
40
+ #
41
+ # The hold is REFCOUNTED rather than re-acquired. `rebuild_stats_index` takes it
42
+ # for its own preservation span and is reached from producers that already hold
43
+ # it, so a nested acquire is ordinary rather than exceptional.
44
+ #
45
+ # Refcounting is NOT about `flock` fairness. Neither Linux nor macOS gives a
46
+ # queued exclusive waiter priority, so a second SHARED acquisition would be
47
+ # granted immediately even with the worker waiting. The reason is
48
+ # `_RETENTION_FD`: it is a single module slot, so a second acquisition would
49
+ # overwrite it and ORPHAN the outer descriptor, whose shared lock would then be
50
+ # held until the process exits. In a long-lived dashboard or TUI that
51
+ # permanently blocks the worker's exclusive request.
52
+ #
53
+ # The same single slot is why the depth check and the acquisition must be ONE
54
+ # atomic decision across threads. `_RETENTION_ACQUIRING` publishes the state
55
+ # between them, so a second thread waits for the first thread's result and then
56
+ # nests on it instead of opening a descriptor of its own.
57
+
58
+ #: A producer waits this long before giving up. The worker's exclusive hold
59
+ #: spans a re-stat and a set of renames, never a deletion, so a wait this long
60
+ #: means the lock is stuck rather than busy.
61
+ RETENTION_SHARED_WAIT_S = 30.0
62
+
63
+ _RETENTION_STATE = threading.Condition()
64
+ _RETENTION_FD: "int | None" = None
65
+ #: The mode `_RETENTION_FD` was taken in. Recorded because depth alone cannot
66
+ #: answer "is this hold exclusive": `retention_is_held()` is true under a shared
67
+ #: hold too, so a guard written against it admits marking from inside
68
+ #: `retention_shared()` — which is the concurrent-marking race §5.3 says the
69
+ #: exclusive hold exists to prevent.
70
+ _RETENTION_MODE: "int | None" = None
71
+ _RETENTION_DEPTH = 0
72
+ _RETENTION_ACQUIRING = False
73
+
74
+
75
+ def _retention_lock_path() -> pathlib.Path:
76
+ return pathlib.Path(_cctally_core.ARTIFACT_RETENTION_LOCK_PATH)
77
+
78
+
79
+ def _acquire_retention_flock(mode: int, timeout: float) -> bool:
80
+ """Take the flock in `mode`, bounded. Seam for tests; not for callers."""
81
+ global _RETENTION_FD, _RETENTION_MODE
82
+ path = _retention_lock_path()
83
+ try:
84
+ path.parent.mkdir(parents=True, exist_ok=True)
85
+ fd = os.open(str(path), os.O_RDWR | os.O_CREAT, 0o600)
86
+ except OSError:
87
+ return False
88
+ deadline = time.monotonic() + max(timeout, 0.0)
89
+ while True:
90
+ try:
91
+ fcntl.flock(fd, mode | fcntl.LOCK_NB)
92
+ except OSError:
93
+ if time.monotonic() >= deadline:
94
+ os.close(fd)
95
+ return False
96
+ time.sleep(0.05)
97
+ continue
98
+ _RETENTION_FD = fd
99
+ _RETENTION_MODE = mode & (fcntl.LOCK_SH | fcntl.LOCK_EX)
100
+ return True
101
+
102
+
103
+ def _release_retention_flock() -> None:
104
+ """Release the flock. Seam for tests; not for callers."""
105
+ global _RETENTION_FD, _RETENTION_MODE
106
+ fd, _RETENTION_FD = _RETENTION_FD, None
107
+ _RETENTION_MODE = None
108
+ if fd is None:
109
+ return
110
+ try:
111
+ fcntl.flock(fd, fcntl.LOCK_UN)
112
+ except OSError:
113
+ pass
114
+ finally:
115
+ os.close(fd)
116
+
117
+
118
+ def retention_depth() -> int:
119
+ """How many nested holds this process currently has."""
120
+ return _RETENTION_DEPTH
121
+
122
+
123
+ def retention_is_held() -> bool:
124
+ return _RETENTION_DEPTH > 0
125
+
126
+
127
+ def retention_is_held_exclusive() -> bool:
128
+ """Whether this process holds the lock EXCLUSIVE, not merely held.
129
+
130
+ `retention_is_held()` cannot answer this. A shared hold satisfies it, so a
131
+ guard written against it lets a marking pass run inside `retention_shared()`
132
+ concurrently with another process's marking pass.
133
+ """
134
+ return _RETENTION_DEPTH > 0 and _RETENTION_MODE == fcntl.LOCK_EX
135
+
136
+
137
+ def _retention_drop_one() -> None:
138
+ """Drop one hold and release the flock when the LAST one goes.
139
+
140
+ The release is keyed on the depth reaching zero, never on which context
141
+ manager is exiting. Once a nest can be entered by a thread other than the
142
+ one that acquired, the acquirer is no longer guaranteed to exit last, and
143
+ tying the release to it leaks the descriptor whenever it does not.
144
+ """
145
+ global _RETENTION_DEPTH
146
+ with _RETENTION_STATE:
147
+ _RETENTION_DEPTH = max(0, _RETENTION_DEPTH - 1)
148
+ if _RETENTION_DEPTH == 0:
149
+ _release_retention_flock()
150
+
151
+
152
+ @contextlib.contextmanager
153
+ def retention_shared(*, timeout: "float | None" = None, label: str = ""):
154
+ """Hold `artifact-retention.lock` SHARED across an evidence write (§5.3).
155
+
156
+ Yields True when the lock is held and False when it could not be taken
157
+ within the bound. A producer that could not take it PROCEEDS ANYWAY: this
158
+ lock exists to keep reclamation off evidence that is mid-publication, and
159
+ refusing to preserve corruption evidence because a lock file is stuck would
160
+ turn a safeguard into an outage. The failure is reported once on stderr so
161
+ it is visible rather than silent, and the worker's own protection gate
162
+ still covers the artifact — a bundle written seconds ago is younger than
163
+ any age bound and is unclassified until its manifest exists.
164
+ """
165
+ global _RETENTION_DEPTH, _RETENTION_ACQUIRING
166
+ wait = RETENTION_SHARED_WAIT_S if timeout is None else timeout
167
+ with _RETENTION_STATE:
168
+ # A thread already acquiring settles the question for everyone: wait
169
+ # for its result rather than opening a second descriptor. `depth > 0`
170
+ # and `acquiring` are mutually exclusive, so a nested acquire by a
171
+ # thread that already holds the lock never waits here.
172
+ while _RETENTION_ACQUIRING:
173
+ _RETENTION_STATE.wait()
174
+ if _RETENTION_DEPTH > 0:
175
+ _RETENTION_DEPTH += 1
176
+ nested = True
177
+ else:
178
+ _RETENTION_ACQUIRING = True
179
+ nested = False
180
+ if nested:
181
+ try:
182
+ yield True
183
+ finally:
184
+ _retention_drop_one()
185
+ return
186
+
187
+ held = False
188
+ try:
189
+ held = _acquire_retention_flock(fcntl.LOCK_SH, wait)
190
+ finally:
191
+ # Clearing the flag and publishing the result must be ONE critical
192
+ # section. A waiter that woke between them would read depth 0 over a
193
+ # lock this thread already holds and acquire a second descriptor,
194
+ # which is the orphan this protocol exists to prevent.
195
+ with _RETENTION_STATE:
196
+ _RETENTION_ACQUIRING = False
197
+ if held:
198
+ _RETENTION_DEPTH += 1
199
+ _RETENTION_STATE.notify_all()
200
+ if not held:
201
+ suffix = f" ({label})" if label else ""
202
+ print(
203
+ "[retention] could not take artifact-retention.lock within "
204
+ f"{wait:g}s{suffix}; continuing without it — evidence is still "
205
+ "written, and reclamation re-checks every artifact under its own "
206
+ "exclusive hold.",
207
+ file=sys.stderr,
208
+ )
209
+ try:
210
+ yield held
211
+ finally:
212
+ if held:
213
+ _retention_drop_one()
214
+
215
+
216
+ @contextlib.contextmanager
217
+ def retention_exclusive(*, timeout: "float | None" = None, label: str = ""):
218
+ """Hold `artifact-retention.lock` EXCLUSIVE, holding no earlier lock (§5.3).
219
+
220
+ The reclamation worker's acquisition primitive. Yields True when the lock
221
+ is held and False when it could not be taken within the bound; unlike a
222
+ producer, a worker that could not take it must mark NOTHING, because the
223
+ hold is the only thing that keeps it off evidence mid-publication.
224
+
225
+ THE BINDING CONSTRAINT, recorded in `docs/journal-gotchas.md`: nothing
226
+ inside this hold may acquire a lock that sits EARLIER in the total order —
227
+ not `cache.db.lock`, not the cache Codex provider flock, and not the
228
+ conversation provider flocks. A holder of this lock that waits on
229
+ `cache.db.lock` closes a real cycle against `db rederive --yes`, which
230
+ holds `cache.db.lock` and then requests this lock SHARED. Producers are
231
+ allowed the inverted acquisition precisely because a shared request is
232
+ never blocked by another shared holder; an exclusive holder removes that
233
+ property. `tests/test_artifact_retention_lock_order.py` enforces it.
234
+
235
+ An exclusive hold is never nested inside a shared one. `flock` cannot
236
+ upgrade in place, and `_RETENTION_FD` is a single module slot, so a caller
237
+ already holding the lock has no way to ask for a stronger mode. It takes
238
+ the same `_RETENTION_ACQUIRING` state a shared acquire takes, so it can
239
+ never race a concurrent shared acquire onto that one slot.
240
+ """
241
+ global _RETENTION_DEPTH, _RETENTION_ACQUIRING
242
+ wait = RETENTION_SHARED_WAIT_S if timeout is None else timeout
243
+ with _RETENTION_STATE:
244
+ while _RETENTION_ACQUIRING:
245
+ _RETENTION_STATE.wait()
246
+ if _RETENTION_DEPTH > 0:
247
+ raise RuntimeError(
248
+ "artifact-retention.lock cannot be upgraded from SHARED to "
249
+ "EXCLUSIVE; the worker takes it holding nothing earlier"
250
+ )
251
+ _RETENTION_ACQUIRING = True
252
+ held = False
253
+ try:
254
+ held = _acquire_retention_flock(fcntl.LOCK_EX, wait)
255
+ finally:
256
+ with _RETENTION_STATE:
257
+ _RETENTION_ACQUIRING = False
258
+ if held:
259
+ _RETENTION_DEPTH += 1
260
+ _RETENTION_STATE.notify_all()
261
+ if not held:
262
+ suffix = f" ({label})" if label else ""
263
+ print(
264
+ "[retention] could not take artifact-retention.lock exclusively "
265
+ f"within {wait:g}s{suffix}; reclaiming nothing this pass.",
266
+ file=sys.stderr,
267
+ )
268
+ try:
269
+ yield held
270
+ finally:
271
+ if held:
272
+ _retention_drop_one()
273
+
274
+
275
+ # --------------------------------------------------------------------------
276
+ # §5.4 / §5.5 — two-phase mark-then-delete
277
+ # --------------------------------------------------------------------------
278
+ #
279
+ # The worker re-stats every member under the EXCLUSIVE hold, writes and fsyncs
280
+ # a reclaim-pending record carrying the exact source-to-tombstone mapping and
281
+ # each member's identity, renames each member to a plan-qualified tombstone
282
+ # WITHIN ITS OWN PARENT, fsyncs that parent, flips the record to `marked`,
283
+ # fsyncs it, releases the lock, and only then unlinks.
284
+ #
285
+ # NOTHING BELOW MAY ACQUIRE A LOCK EARLIER IN THE TOTAL ORDER. Everything here
286
+ # is reachable from `reclaim_artifacts`, which holds `artifact-retention.lock`
287
+ # exclusively; a holder that waits on `cache.db.lock` closes a real cycle
288
+ # against `db rederive --yes`. `tests/test_artifact_retention_lock_order.py`
289
+ # scans this module for exactly that.
290
+
291
+ #: The reclaim record's schema. Bumped only on a breaking change; a resuming
292
+ #: worker refuses a version it does not know rather than guessing.
293
+ RECLAIM_RECORD_SCHEMA_VERSION = 1
294
+
295
+ #: Tombstones are named `.reclaiming-<plan>-<original>` inside the member's own
296
+ #: parent, so the rename is a same-directory operation and a resume can find
297
+ #: the tombstone from the record alone.
298
+ RECLAIM_TOMBSTONE_PREFIX = ".reclaiming-"
299
+
300
+ #: Pending records live directly in the data directory as dotfiles, beside the
301
+ #: `*.quarantine-pending.json` markers they resemble — never under `logs/` or
302
+ #: `quarantine/`, which the metadata walk enumerates as artifacts.
303
+ RECLAIM_RECORD_PREFIX = ".reclaim-pending-"
304
+
305
+
306
+ @dataclasses.dataclass(frozen=True)
307
+ class ReclaimTarget:
308
+ """One member to reclaim, with the identity observed when it was planned.
309
+
310
+ `root_id` names the root whose deletion closure this member belongs to.
311
+ Marking is decided per ROOT, not per member (§5.4): a member that is skipped
312
+ or fails abandons the rest of its own group, because the group is ordered
313
+ referrer-before-referent and continuing past a skipped referrer would delete
314
+ the referent out from under a manifest that survives and names it.
315
+ """
316
+
317
+ id: str
318
+ is_dir: bool
319
+ device: int
320
+ inode: int
321
+ size: int
322
+ mtime_ns: int
323
+ root_id: str = ""
324
+
325
+ @property
326
+ def group_id(self) -> str:
327
+ return self.root_id or self.id
328
+
329
+
330
+ @dataclasses.dataclass(frozen=True)
331
+ class MarkResult:
332
+ """What one marking pass achieved.
333
+
334
+ `skipped_ids` is a first-class outcome and not an error: a member that is
335
+ missing, symlinked or holds a different inode than the plan described was
336
+ deliberately left alone. It still has to reach the caller — an operator
337
+ running `db prune --yes` must be able to learn that a member was skipped and
338
+ why — which is what `reasons` carries.
339
+ """
340
+
341
+ plan_id: str
342
+ record_path: "pathlib.Path | None"
343
+ marked_ids: "tuple[str, ...]"
344
+ failed_roots: "tuple[str, ...]"
345
+ skipped_ids: "tuple[str, ...]"
346
+ reasons: "dict[str, str]"
347
+
348
+
349
+ @dataclasses.dataclass(frozen=True)
350
+ class ReclaimOutcome:
351
+ """What one whole reclamation achieved, marking and deletion together."""
352
+
353
+ held: bool
354
+ plan_ids: "tuple[str, ...]"
355
+ marked_ids: "tuple[str, ...]"
356
+ failed_roots: "tuple[str, ...]"
357
+ skipped_ids: "tuple[str, ...]"
358
+ deleted_ids: "tuple[str, ...]"
359
+ errors: "dict[str, str]"
360
+
361
+
362
+ def _reclaim_root(root) -> pathlib.Path:
363
+ return pathlib.Path(_cctally_core.APP_DIR if root is None else root)
364
+
365
+
366
+ def _member_path(root: pathlib.Path, member_id: str) -> pathlib.Path:
367
+ """Resolve a member id under `root`, refusing anything that escapes it.
368
+
369
+ The check is lexical and deliberately does NOT call `resolve()`: resolving
370
+ follows symlinks, and a symlinked member is something this engine refuses
371
+ rather than dereferences.
372
+ """
373
+ if not member_id or os.path.isabs(member_id):
374
+ raise ValueError(f"retained-artifact id must be relative: {member_id!r}")
375
+ normalized = os.path.normpath(member_id)
376
+ if normalized == os.pardir or normalized.startswith(os.pardir + os.sep):
377
+ raise ValueError(f"retained-artifact id escapes the data directory: {member_id!r}")
378
+ return root / normalized
379
+
380
+
381
+ def _tombstone_path(path: pathlib.Path, plan_id: str) -> pathlib.Path:
382
+ return path.with_name(f"{RECLAIM_TOMBSTONE_PREFIX}{plan_id}-{path.name}")
383
+
384
+
385
+ def _entry_tombstone_path(root: pathlib.Path, entry: dict) -> pathlib.Path:
386
+ """The tombstone an entry names, refusing anything the engine never wrote.
387
+
388
+ `root / entry["tombstone"]` on its own is not equivalent: `pathlib` discards
389
+ the left operand when the right is absolute, so a record naming an absolute
390
+ path outside the data directory would hand that path straight to the
391
+ unlink. The record is a `0600` file, so reaching this needs write access —
392
+ but the source-id sibling has had the escape guard since it was written, the
393
+ operation on this end is an unrecoverable delete, and a CORRUPTED record can
394
+ produce an out-of-range value with no adversary at all, which is the failure
395
+ mode this whole epic exists to survive.
396
+
397
+ Two further conditions, because the engine only ever writes one shape: the
398
+ tombstone lives in the same parent as its source (the rename is always
399
+ within the parent) and carries the reclaiming prefix.
400
+ """
401
+ recorded = entry.get("tombstone")
402
+ if not isinstance(recorded, str):
403
+ raise ValueError(f"reclaim entry has no tombstone path: {entry.get('id')!r}")
404
+ tombstone = _member_path(root, recorded)
405
+ source = _member_path(root, entry["id"])
406
+ if tombstone.parent != source.parent:
407
+ raise ValueError(
408
+ f"reclaim tombstone {recorded!r} does not sit beside its source "
409
+ f"{entry['id']!r}"
410
+ )
411
+ if not tombstone.name.startswith(RECLAIM_TOMBSTONE_PREFIX):
412
+ raise ValueError(f"reclaim tombstone {recorded!r} is not a tombstone name")
413
+ return tombstone
414
+
415
+
416
+ def _lstat_or_none(path):
417
+ try:
418
+ return _walk_lstat(path)
419
+ except OSError:
420
+ return None
421
+
422
+
423
+ def _fsync_directory(path) -> None:
424
+ """Make a rename or an unlink in `path` durable. Best effort by design."""
425
+ try:
426
+ fd = os.open(str(path), os.O_RDONLY)
427
+ except OSError:
428
+ return
429
+ try:
430
+ os.fsync(fd)
431
+ except OSError:
432
+ pass
433
+ finally:
434
+ os.close(fd)
435
+
436
+
437
+ def _rename_within_parent(src, dst) -> None:
438
+ """Rename a member to its tombstone. Same parent, never across a device."""
439
+ os.rename(str(src), str(dst))
440
+
441
+
442
+ def _unlink_children(path: pathlib.Path) -> None:
443
+ """Remove everything inside `path`, never following a symlink out of it."""
444
+ with os.scandir(path) as entries:
445
+ for entry in entries:
446
+ if entry.is_dir(follow_symlinks=False):
447
+ _unlink_children(pathlib.Path(entry.path))
448
+ os.rmdir(entry.path)
449
+ else:
450
+ os.unlink(entry.path)
451
+
452
+
453
+ def _unlink_tree(path) -> None:
454
+ """Delete one tombstone, file or directory, following no symlink.
455
+
456
+ A symlink INSIDE a tombstone is unlinked as a link; its target is never
457
+ reached. That is why this is hand-rolled rather than `shutil.rmtree`,
458
+ whose top-level symlink handling is a different contract.
459
+ """
460
+ path = pathlib.Path(path)
461
+ info = os.lstat(path)
462
+ if not _stat.S_ISLNK(info.st_mode) and _stat.S_ISDIR(info.st_mode):
463
+ _unlink_children(path)
464
+ os.rmdir(path)
465
+ else:
466
+ os.unlink(path)
467
+
468
+
469
+ def targets_for_plan(plan, *, root=None) -> "list[ReclaimTarget]":
470
+ """Stat every member a plan would delete, carrying its group (§5.4).
471
+
472
+ The sanctioned way to build a marking plan. Assembling the list by hand and
473
+ letting `root_id` default per member declares every member its own root,
474
+ which is precisely the flat-plan shape that let a skipped referrer's
475
+ referent be renamed — so the grouping comes from `RetentionPlan.delete_groups`
476
+ rather than from the caller's memory.
477
+ """
478
+ root = _reclaim_root(root)
479
+ targets: "list[ReclaimTarget]" = []
480
+ for root_id, member_ids in plan.delete_groups:
481
+ for member_id in member_ids:
482
+ target = stat_reclaim_target(member_id, root_id=root_id, root=root)
483
+ if target is not None:
484
+ targets.append(target)
485
+ return targets
486
+
487
+
488
+ def stat_reclaim_target(
489
+ member_id: str, *, root_id: str, root=None,
490
+ ) -> "ReclaimTarget | None":
491
+ """Observe a member's identity now, or None when it is not there.
492
+
493
+ A symlink is reported with `is_dir=False` and its own identity, so the
494
+ marking pass can refuse it explicitly rather than silently treating it as
495
+ the thing it points at.
496
+
497
+ `root_id` names the deletion-closure group this member belongs to. It is
498
+ REQUIRED rather than defaulted: a default of "the member itself" declares
499
+ every member its own root, which silently restores the flat plan whose
500
+ per-member decisions let a skipped referrer's referent be deleted. Prefer
501
+ `targets_for_plan`, which takes the grouping from the plan.
502
+ """
503
+ root = _reclaim_root(root)
504
+ path = _member_path(root, member_id)
505
+ info = _lstat_or_none(path)
506
+ if info is None:
507
+ return None
508
+ return ReclaimTarget(
509
+ id=member_id,
510
+ is_dir=bool(
511
+ not _stat.S_ISLNK(info.st_mode)
512
+ and _stat.S_ISDIR(info.st_mode)
513
+ ),
514
+ device=int(info.st_dev),
515
+ inode=int(info.st_ino),
516
+ size=int(info.st_size),
517
+ mtime_ns=int(info.st_mtime_ns),
518
+ root_id=root_id,
519
+ )
520
+
521
+
522
+ def _entry_for(target: ReclaimTarget, plan_id: str, root: pathlib.Path) -> dict:
523
+ path = _member_path(root, target.id)
524
+ tombstone = _tombstone_path(path, plan_id)
525
+ return {
526
+ "id": target.id,
527
+ "rootId": target.group_id,
528
+ "tombstone": os.path.relpath(str(tombstone), str(root)),
529
+ "phase": _kernel.RECLAIM_PHASE_MARKING,
530
+ "isDir": bool(target.is_dir),
531
+ "device": int(target.device),
532
+ "inode": int(target.inode),
533
+ "size": int(target.size),
534
+ "mtimeNs": int(target.mtime_ns),
535
+ "error": None,
536
+ }
537
+
538
+
539
+ def _reclaim_record_path(root: pathlib.Path, plan_id: str) -> pathlib.Path:
540
+ return root / f"{RECLAIM_RECORD_PREFIX}{plan_id}.json"
541
+
542
+
543
+ def _write_reclaim_record(record: dict, *, root=None) -> pathlib.Path:
544
+ """Persist the pending record durably and return where it landed."""
545
+ root = _reclaim_root(root)
546
+ path = _reclaim_record_path(root, record["planId"])
547
+ _atomic_write_private(path, record)
548
+ _fsync_directory(root)
549
+ return path
550
+
551
+
552
+ def _read_reclaim_record(path) -> "dict | None":
553
+ payload = _load_json(pathlib.Path(path))
554
+ if payload.get("schemaVersion") != RECLAIM_RECORD_SCHEMA_VERSION:
555
+ return None
556
+ if not isinstance(payload.get("entries"), list):
557
+ return None
558
+ if not isinstance(payload.get("planId"), str):
559
+ return None
560
+ return payload
561
+
562
+
563
+ def _identity_matches(entry: dict, info, *, allow_size_drift: bool) -> bool:
564
+ """Whether what is on disk is still the inode the plan described.
565
+
566
+ Device and inode always. Size and mtime too for a file, because a file is
567
+ never partially deleted — but never for a directory, where a partial
568
+ `rmtree` legitimately changes both (§5.5).
569
+ """
570
+ if int(info.st_dev) != entry.get("device"):
571
+ return False
572
+ if int(info.st_ino) != entry.get("inode"):
573
+ return False
574
+ if allow_size_drift:
575
+ return True
576
+ return (
577
+ int(info.st_size) == entry.get("size")
578
+ and int(info.st_mtime_ns) == entry.get("mtimeNs")
579
+ )
580
+
581
+
582
+ def mark_reclaim_plan(targets, *, plan_id=None, root=None) -> MarkResult:
583
+ """Rename every eligible member to its tombstone, durably (§5.4).
584
+
585
+ Must be called with `artifact-retention.lock` held EXCLUSIVE. The record is
586
+ written before the first rename and rewritten at `marked` after the last
587
+ one, so a resuming worker can always tell which side of the rename it
588
+ crashed on.
589
+
590
+ Marking is decided per ROOT. A member that is skipped or that fails
591
+ abandons every LATER member of its own group, because §5.4 orders a group
592
+ referrer-before-referent: continuing past a skipped referrer renames the
593
+ referent and leaves a surviving manifest naming a tombstone, which protects
594
+ that incident permanently and makes its corpus unreclaimable forever. The
595
+ members of the group already renamed are not unwound — they are referrers of
596
+ what is being abandoned, so a surviving referent with no referrer is the safe
597
+ direction.
598
+
599
+ A member is skipped, not failed, when it is gone or its identity moved; a
600
+ root whose tombstone path is already taken fails CLOSED and is left
601
+ untouched. Neither unwinds the members already renamed.
602
+ """
603
+ if not retention_is_held_exclusive():
604
+ raise RuntimeError(
605
+ "mark_reclaim_plan requires artifact-retention.lock held exclusive"
606
+ )
607
+ root = _reclaim_root(root)
608
+ plan_id = plan_id or f"{int(time.time())}-{os.getpid()}"
609
+ targets = [target for target in targets if target is not None]
610
+ if not targets:
611
+ # `skipped_ids` and `reasons` are both required, and an empty plan is
612
+ # the ORDINARY steady state — every sweep on a corpus already inside
613
+ # its bounds reaches here — so a five-argument construction raised
614
+ # `TypeError` on the common path rather than on a rare one.
615
+ return MarkResult(plan_id, None, (), (), (), {})
616
+
617
+ record = {
618
+ "schemaVersion": RECLAIM_RECORD_SCHEMA_VERSION,
619
+ "planId": plan_id,
620
+ "createdAtUtc": dt.datetime.now(dt.timezone.utc)
621
+ .isoformat(timespec="seconds")
622
+ .replace("+00:00", "Z"),
623
+ "entries": [_entry_for(target, plan_id, root) for target in targets],
624
+ }
625
+ record_path = _write_reclaim_record(record, root=root)
626
+
627
+ marked: "list[str]" = []
628
+ failed: "list[str]" = []
629
+ skipped: "list[str]" = []
630
+ reasons: "dict[str, str]" = {}
631
+
632
+ groups: "dict[str, list[dict]]" = {}
633
+ for entry in record["entries"]:
634
+ groups.setdefault(entry["rootId"], []).append(entry)
635
+
636
+ for group_id, entries in groups.items():
637
+ abandoned: "str | None" = None
638
+ for entry in entries:
639
+ if abandoned is not None:
640
+ reasons[entry["id"]] = (
641
+ f"group-abandoned: {abandoned} was not marked, and this "
642
+ "member is reachable from it"
643
+ )
644
+ entry["error"] = reasons[entry["id"]]
645
+ skipped.append(entry["id"])
646
+ continue
647
+ source = _member_path(root, entry["id"])
648
+ tombstone = _entry_tombstone_path(root, entry)
649
+ info = _lstat_or_none(source)
650
+ if info is None:
651
+ reasons[entry["id"]] = "missing: nothing at the planned path"
652
+ skipped.append(entry["id"])
653
+ elif _stat.S_ISLNK(info.st_mode):
654
+ reasons[entry["id"]] = (
655
+ "symlink: refusing to reclaim a symlinked member"
656
+ )
657
+ skipped.append(entry["id"])
658
+ elif not _identity_matches(entry, info, allow_size_drift=False):
659
+ reasons[entry["id"]] = (
660
+ "identity-mismatch: the path holds a different inode than "
661
+ "the plan described"
662
+ )
663
+ skipped.append(entry["id"])
664
+ elif _lstat_or_none(tombstone) is not None:
665
+ # Rename would overwrite. §5.4: an existing tombstone target
666
+ # fails that root closed rather than clobbering what is there.
667
+ reasons[entry["id"]] = (
668
+ "tombstone-exists: refusing to overwrite an existing "
669
+ "tombstone"
670
+ )
671
+ failed.append(group_id)
672
+ else:
673
+ try:
674
+ _rename_within_parent(source, tombstone)
675
+ except OSError as exc:
676
+ reasons[entry["id"]] = f"rename-failed: {exc}"
677
+ failed.append(group_id)
678
+ else:
679
+ _fsync_directory(source.parent)
680
+ entry["phase"] = _kernel.RECLAIM_PHASE_MARKED
681
+ marked.append(entry["id"])
682
+ continue
683
+ entry["error"] = reasons[entry["id"]]
684
+ abandoned = entry["id"]
685
+
686
+ record["entries"] = [
687
+ entry for entry in record["entries"]
688
+ if entry["phase"] == _kernel.RECLAIM_PHASE_MARKED
689
+ ]
690
+ if record["entries"]:
691
+ record_path = _write_reclaim_record(record, root=root)
692
+ else:
693
+ _discard_reclaim_record(record_path, root)
694
+ record_path = None
695
+ return MarkResult(
696
+ plan_id, record_path, tuple(marked), tuple(dict.fromkeys(failed)),
697
+ tuple(skipped), reasons,
698
+ )
699
+
700
+
701
+ def _discard_reclaim_record(path, root: pathlib.Path) -> None:
702
+ with contextlib.suppress(OSError):
703
+ os.unlink(path)
704
+ _fsync_directory(root)
705
+
706
+
707
+ def _utc_now_iso() -> str:
708
+ return (
709
+ dt.datetime.now(dt.timezone.utc)
710
+ .isoformat(timespec="seconds")
711
+ .replace("+00:00", "Z")
712
+ )
713
+
714
+
715
+ def _note_entry_failure(entry: dict, reason: str) -> None:
716
+ """Record an entry's error durably, with when it was FIRST seen.
717
+
718
+ The stamp is what bounds §5.5's fail-closed state. Most errors clear on the
719
+ next pass because the resume re-decides every entry; the `marking` row with
720
+ neither the source nor the tombstone present cannot, so its record would
721
+ otherwise accumulate in the data directory with nothing naming it. The
722
+ first-seen time and the count are what let the doctor leg report it rather
723
+ than the subsystem guessing at a member something outside it moved.
724
+ """
725
+ entry["error"] = reason
726
+ entry.setdefault("firstFailedAtUtc", _utc_now_iso())
727
+ entry["failureCount"] = int(entry.get("failureCount") or 0) + 1
728
+
729
+
730
+ def _entry_first_failed_epoch(entry: dict) -> "float | None":
731
+ stamp = entry.get("firstFailedAtUtc")
732
+ if not isinstance(stamp, str) or not stamp:
733
+ return None
734
+ try:
735
+ return dt.datetime.fromisoformat(stamp.replace("Z", "+00:00")).timestamp()
736
+ except ValueError:
737
+ return None
738
+
739
+
740
+ def list_stuck_reclaim_records(
741
+ *, root=None, now_epoch=None,
742
+ threshold_seconds: int = _kernel.RECLAIM_STUCK_AFTER_SECONDS,
743
+ ) -> "list[dict]":
744
+ """Every pending reclaim record carrying an error that has not cleared.
745
+
746
+ Read-only, taking no lock: the doctor leg (`db.retained_artifacts`) and
747
+ `db prune`'s report both need to name a stuck record without arming a
748
+ deletion. Each result is `{"path", "planId", "entries": {id: reason},
749
+ "stuck": bool}`; `stuck` is true once §5.5's fail-closed condition has
750
+ persisted past `threshold_seconds`, which is the state an operator has to
751
+ resolve by hand because no pass can decide it.
752
+ """
753
+ root = _reclaim_root(root)
754
+ now = time.time() if now_epoch is None else now_epoch
755
+ found: "list[dict]" = []
756
+ for path in sorted(root.glob(f"{RECLAIM_RECORD_PREFIX}*.json")):
757
+ record = _read_reclaim_record(path)
758
+ if record is None:
759
+ continue
760
+ failing = {
761
+ entry["id"]: entry["error"]
762
+ for entry in record["entries"]
763
+ if isinstance(entry, dict) and entry.get("error")
764
+ }
765
+ if not failing:
766
+ continue
767
+ first_failed = [
768
+ _entry_first_failed_epoch(entry)
769
+ for entry in record["entries"]
770
+ if isinstance(entry, dict) and entry.get("error")
771
+ ]
772
+ oldest = min(
773
+ (stamp for stamp in first_failed if stamp is not None), default=None,
774
+ )
775
+ found.append({
776
+ "path": path,
777
+ "planId": record["planId"],
778
+ "entries": failing,
779
+ #: How long the oldest failing entry has been failing. Additive:
780
+ #: the doctor leg has to say how long the condition has persisted,
781
+ #: and the first-failure stamp is kept across passes precisely so
782
+ #: the age cannot reset.
783
+ "ageSeconds": None if oldest is None else max(int(now - oldest), 0),
784
+ "stuck": any(
785
+ _kernel.reclaim_entry_is_stuck(
786
+ error=entry.get("error"),
787
+ first_failed_at_epoch=_entry_first_failed_epoch(entry),
788
+ now_epoch=now,
789
+ threshold_seconds=threshold_seconds,
790
+ )
791
+ for entry in record["entries"]
792
+ if isinstance(entry, dict)
793
+ ),
794
+ })
795
+ return found
796
+
797
+
798
+ def _resume_marking_pass(root: pathlib.Path) -> "tuple[list[pathlib.Path], dict]":
799
+ """Bring every pending record to `marked`, durably, before any unlink.
800
+
801
+ Runs under the exclusive hold. Every entry is decided by the pure phase
802
+ table in the kernel, so the decision is testable without a filesystem and
803
+ the filesystem work here is only the rename it asks for.
804
+ """
805
+ records: "list[pathlib.Path]" = []
806
+ errors: "dict[str, str]" = {}
807
+ for path in sorted(root.glob(f"{RECLAIM_RECORD_PREFIX}*.json")):
808
+ record = _read_reclaim_record(path)
809
+ if record is None:
810
+ continue
811
+ kept: "list[dict]" = []
812
+ abandoned: "dict[str, str]" = {}
813
+ for entry in record["entries"]:
814
+ group_id = entry.get("rootId") or entry["id"]
815
+ source = _member_path(root, entry["id"])
816
+ tombstone = _entry_tombstone_path(root, entry)
817
+ source_info = _lstat_or_none(source)
818
+ tombstone_info = _lstat_or_none(tombstone)
819
+ action = _kernel.resume_action(
820
+ entry.get("phase"), source_info is not None, tombstone_info is not None,
821
+ )
822
+ if action == "entry-complete":
823
+ continue
824
+ # A group whose earlier member could not be marked must not have its
825
+ # later members renamed here either, for the same reason marking
826
+ # stops: the later member is the referent of the one that stayed.
827
+ if group_id in abandoned and action != "continue-deletion":
828
+ errors[entry["id"]] = (
829
+ f"group-abandoned: {abandoned[group_id]} was not marked, and "
830
+ "this member is reachable from it"
831
+ )
832
+ _note_entry_failure(entry, errors[entry["id"]])
833
+ kept.append(entry)
834
+ continue
835
+ if action == "fail-closed":
836
+ errors[entry["id"]] = (
837
+ "fail-closed: source and tombstone are both present, or "
838
+ "neither is and the rename never completed"
839
+ )
840
+ _note_entry_failure(entry, errors[entry["id"]])
841
+ abandoned[group_id] = entry["id"]
842
+ kept.append(entry)
843
+ continue
844
+ if action == "continue-deletion":
845
+ # This pass did NO work on this entry: it was already `marked`
846
+ # and its tombstone is still on disk. Clearing `error` re-arms
847
+ # the deletion retry, which `_apply_reclaim_record` skips while
848
+ # an error stands — but the first-failure stamp is KEPT. An age
849
+ # that reset on every pass could never cross
850
+ # `RECLAIM_STUCK_AFTER_SECONDS`, so a member that will never
851
+ # delete (EPERM, an immutable flag, a vanished mount) stayed
852
+ # `stuck: False` forever and the §7.3 WARN that bounds the
853
+ # accumulation could not fire for it.
854
+ entry["phase"] = _kernel.RECLAIM_PHASE_MARKED
855
+ entry["error"] = None
856
+ kept.append(entry)
857
+ continue
858
+ if action == "resume-rename":
859
+ if _stat.S_ISLNK(source_info.st_mode):
860
+ # Marking refuses a symlinked member, so a resume must too:
861
+ # the recorded identity is the LINK's own, so an identity
862
+ # comparison alone would happily rename it.
863
+ errors[entry["id"]] = (
864
+ "symlink: refusing to reclaim a symlinked member"
865
+ )
866
+ _note_entry_failure(entry, errors[entry["id"]])
867
+ abandoned[group_id] = entry["id"]
868
+ kept.append(entry)
869
+ continue
870
+ if not _identity_matches(entry, source_info, allow_size_drift=False):
871
+ errors[entry["id"]] = (
872
+ "identity-mismatch: the source holds a different inode "
873
+ "than the plan described"
874
+ )
875
+ _note_entry_failure(entry, errors[entry["id"]])
876
+ abandoned[group_id] = entry["id"]
877
+ kept.append(entry)
878
+ continue
879
+ try:
880
+ _rename_within_parent(source, tombstone)
881
+ except OSError as exc:
882
+ errors[entry["id"]] = f"rename-failed: {exc}"
883
+ _note_entry_failure(entry, errors[entry["id"]])
884
+ abandoned[group_id] = entry["id"]
885
+ kept.append(entry)
886
+ continue
887
+ _fsync_directory(source.parent)
888
+ entry["phase"] = _kernel.RECLAIM_PHASE_MARKED
889
+ entry["error"] = None
890
+ entry.pop("firstFailedAtUtc", None)
891
+ entry.pop("failureCount", None)
892
+ kept.append(entry)
893
+ record["entries"] = kept
894
+ if kept:
895
+ records.append(_write_reclaim_record(record, root=root))
896
+ else:
897
+ _discard_reclaim_record(path, root)
898
+ return records, errors
899
+
900
+
901
+ def _apply_reclaim_record(path, *, root) -> "tuple[list[str], dict[str, str]]":
902
+ """Unlink every marked tombstone, outside the lock (§5.4).
903
+
904
+ One member that will not delete is recorded and skipped; successful work is
905
+ never unwound. Only an `OSError` is a deletion failure — anything else is a
906
+ defect and propagates.
907
+
908
+ Every failure here goes through `_note_entry_failure`, not through a bare
909
+ `entry["error"] = ...`. The stamp it writes is what §5.5 bounds the
910
+ fail-closed state with, and without it a permanently undeletable member
911
+ produced a record `list_stuck_reclaim_records` reported at
912
+ `stuck: False` forever — invisible to the §7.3 WARN that exists to bound
913
+ exactly this accumulation.
914
+ """
915
+ root = _reclaim_root(root)
916
+ record = _read_reclaim_record(path)
917
+ if record is None:
918
+ return [], {}
919
+ deleted: "list[str]" = []
920
+ errors: "dict[str, str]" = {}
921
+ entries = list(record["entries"])
922
+ remaining: "list[dict]" = []
923
+
924
+ def retire(index: int) -> None:
925
+ """Clear this entry from the durable record, keeping the untouched tail.
926
+
927
+ Rewriting after every unlink is what bounds the crash window to one
928
+ entry. When nothing is left the record is NOT rewritten empty: the
929
+ final discard does that, and an entry still listed after its tombstone
930
+ is gone is exactly the `marked`/absent/absent row §5.5 calls complete.
931
+ """
932
+ tail = remaining + entries[index + 1:]
933
+ if tail:
934
+ record["entries"] = tail
935
+ _write_reclaim_record(record, root=root)
936
+
937
+ for index, entry in enumerate(entries):
938
+ if entry.get("phase") != _kernel.RECLAIM_PHASE_MARKED or entry.get("error"):
939
+ remaining.append(entry)
940
+ continue
941
+ tombstone = _entry_tombstone_path(root, entry)
942
+ info = _lstat_or_none(tombstone)
943
+ if info is None:
944
+ # The success window: the unlink landed and the entry had not been
945
+ # cleared yet. Clearing it now is the completion, not an error — but
946
+ # THIS pass deleted nothing, so it does not claim the id. Counting
947
+ # it would make `deleted_ids` untruthful on every resume of a plan
948
+ # that had already finished.
949
+ retire(index)
950
+ continue
951
+ if bool(entry.get("isDir")) != bool(_stat.S_ISDIR(info.st_mode)):
952
+ errors[entry["id"]] = "identity-mismatch: the tombstone changed kind"
953
+ _note_entry_failure(entry, errors[entry["id"]])
954
+ remaining.append(entry)
955
+ continue
956
+ if not _identity_matches(
957
+ entry, info, allow_size_drift=bool(entry.get("isDir")),
958
+ ):
959
+ errors[entry["id"]] = (
960
+ "identity-mismatch: the tombstone holds a different inode than "
961
+ "the one that was marked"
962
+ )
963
+ _note_entry_failure(entry, errors[entry["id"]])
964
+ remaining.append(entry)
965
+ continue
966
+ try:
967
+ _unlink_tree(tombstone)
968
+ except OSError as exc:
969
+ errors[entry["id"]] = f"delete-failed: {exc}"
970
+ _note_entry_failure(entry, errors[entry["id"]])
971
+ remaining.append(entry)
972
+ continue
973
+ _fsync_directory(tombstone.parent)
974
+ deleted.append(entry["id"])
975
+ retire(index)
976
+
977
+ record["entries"] = remaining
978
+ if remaining:
979
+ _write_reclaim_record(record, root=root)
980
+ else:
981
+ _discard_reclaim_record(path, root)
982
+ return deleted, errors
983
+
984
+
985
+ def reclaim_artifacts(
986
+ targets=(), *, plan_id=None, root=None, timeout=None, resume=True,
987
+ ) -> ReclaimOutcome:
988
+ """Mark under the exclusive hold, then delete outside it (§5.4).
989
+
990
+ A worker that cannot take the lock marks NOTHING: the hold is the only
991
+ thing that keeps it off evidence mid-publication.
992
+ """
993
+ root = _reclaim_root(root)
994
+ records: "list[pathlib.Path]" = []
995
+ errors: "dict[str, str]" = {}
996
+ marked: "tuple[str, ...]" = ()
997
+ failed: "tuple[str, ...]" = ()
998
+ skipped: "tuple[str, ...]" = ()
999
+ plan_ids: "list[str]" = []
1000
+ with retention_exclusive(timeout=timeout, label="artifact reclamation") as held:
1001
+ if not held:
1002
+ return ReclaimOutcome(False, (), (), (), (), (), {})
1003
+ if resume:
1004
+ resumed, resume_errors = _resume_marking_pass(root)
1005
+ records.extend(resumed)
1006
+ errors.update(resume_errors)
1007
+ if targets:
1008
+ result = mark_reclaim_plan(targets, plan_id=plan_id, root=root)
1009
+ marked, failed = result.marked_ids, result.failed_roots
1010
+ skipped = result.skipped_ids
1011
+ # EVERY reason reaches the caller, not just the failures. §5.4 makes
1012
+ # "skipped" a first-class outcome, and a member that is silently
1013
+ # absent from `marked_ids` with nothing said about it is exactly how
1014
+ # an abandoned group would go unnoticed.
1015
+ errors.update(result.reasons)
1016
+ plan_ids.append(result.plan_id)
1017
+ if result.record_path is not None:
1018
+ records.append(result.record_path)
1019
+
1020
+ deleted: "list[str]" = []
1021
+ for path in records:
1022
+ applied, apply_errors = _apply_reclaim_record(path, root=root)
1023
+ deleted.extend(applied)
1024
+ errors.update(apply_errors)
1025
+ return ReclaimOutcome(
1026
+ True, tuple(plan_ids), marked, failed, skipped, tuple(deleted), errors,
1027
+ )
1028
+
1029
+
1030
+ def resume_reclaim(*, root=None, timeout=None) -> ReclaimOutcome:
1031
+ """Finish every pending reclaim plan left by an interrupted worker."""
1032
+ return reclaim_artifacts((), root=root, timeout=timeout, resume=True)
1033
+
1034
+
1035
+ #: The families that produce quarantine incidents. Keyed by the database file
1036
+ #: name, which is what both the incident directory and the forensics bundle are
1037
+ #: named from (`bin/_cctally_db.py:987` and `:1319`).
1038
+ KNOWN_FAMILIES = ("stats.db", "cache.db", "conversations.db")
1039
+
1040
+ #: Two incident name shapes exist in the retained corpus. The cutover protocol
1041
+ #: uses microsecond precision; the strict-quarantine path uses
1042
+ #: `_db_backup_timestamp()`'s trailing-Z second precision. Recognizing only one
1043
+ #: would silently drop the other, which is exactly the omission the shipped
1044
+ #: correlator's comment warns about.
1045
+ _INCIDENT_STAMP = r"(?P<stamp>\d{8}T\d{6}(?:_\d{6}|Z))"
1046
+ _BUNDLE_STAMP = r"(?P<stamp>\d{8}T\d{6})Z"
1047
+
1048
+
1049
+ def _incident_re(family: str) -> "re.Pattern[str]":
1050
+ return re.compile(rf"^{re.escape(family)}-{_INCIDENT_STAMP}$")
1051
+
1052
+
1053
+ def _bundle_re(family: str) -> "re.Pattern[str]":
1054
+ return re.compile(
1055
+ rf"^{re.escape(family)}-corruption-forensics-{_BUNDLE_STAMP}\.json$"
1056
+ )
1057
+
1058
+
1059
+ def family_of_incident(name: str) -> "str | None":
1060
+ """The family an incident directory name belongs to, or None."""
1061
+ for family in KNOWN_FAMILIES:
1062
+ if _incident_re(family).match(name) is not None:
1063
+ return family
1064
+ return None
1065
+
1066
+
1067
+ def incident_time(family: str, name: str) -> "dt.datetime | None":
1068
+ """Parse an incident directory name's UTC timestamp, or None."""
1069
+ match = _incident_re(family).match(name)
1070
+ if match is None:
1071
+ return None
1072
+ stamp = match["stamp"]
1073
+ fmt = "%Y%m%dT%H%M%SZ" if stamp.endswith("Z") else "%Y%m%dT%H%M%S_%f"
1074
+ try:
1075
+ return dt.datetime.strptime(stamp, fmt).replace(tzinfo=dt.timezone.utc)
1076
+ except ValueError:
1077
+ return None
1078
+
1079
+
1080
+ def bundle_time(family: str, name: str) -> "dt.datetime | None":
1081
+ """Parse a forensics bundle file name's UTC timestamp, or None."""
1082
+ match = _bundle_re(family).match(name)
1083
+ if match is None:
1084
+ return None
1085
+ try:
1086
+ return dt.datetime.strptime(match["stamp"], "%Y%m%dT%H%M%S").replace(
1087
+ tzinfo=dt.timezone.utc
1088
+ )
1089
+ except ValueError:
1090
+ return None
1091
+
1092
+
1093
+ def _load_json(path: pathlib.Path) -> dict:
1094
+ try:
1095
+ payload = json.loads(path.read_text(encoding="utf-8"))
1096
+ except (OSError, ValueError):
1097
+ return {}
1098
+ return payload if isinstance(payload, dict) else {}
1099
+
1100
+
1101
+ def list_family_bundles(
1102
+ logs_dir, family: str,
1103
+ ) -> "list[tuple[dt.datetime, str, dict]]":
1104
+ """Every forensics bundle of one family, ascending by timestamp.
1105
+
1106
+ Two independent exclusions, and they cover different entries. The
1107
+ WAL-evidence DIRECTORIES beside a bundle share its stem but not its
1108
+ `.json` suffix (`bin/_cctally_db.py:894` against `:992`), so `_bundle_re`
1109
+ rejects them on the NAME. The `is_file(follow_symlinks=False)` test covers
1110
+ what that cannot: a directory or a symlink whose name does match a
1111
+ bundle's. A symlink is refused rather than followed, in the same direction
1112
+ §3.2 protects a symlinked root.
1113
+
1114
+ The payload is loaded here because the verdict depends on whether the
1115
+ bundle names its own `trigger.origin` (§4.3).
1116
+ """
1117
+ logs_dir = pathlib.Path(logs_dir)
1118
+ if not logs_dir.is_dir():
1119
+ return []
1120
+ found: "list[tuple[dt.datetime, str, dict]]" = []
1121
+ with os.scandir(logs_dir) as entries:
1122
+ for entry in entries:
1123
+ if not entry.is_file(follow_symlinks=False):
1124
+ continue
1125
+ when = bundle_time(family, entry.name)
1126
+ if when is None:
1127
+ continue
1128
+ found.append((when, entry.path, _load_json(pathlib.Path(entry.path))))
1129
+ found.sort(key=lambda item: (item[0], item[1]))
1130
+ return found
1131
+
1132
+
1133
+ def load_incident_manifest(incident) -> dict:
1134
+ """The incident's `manifest.json`, or an empty dict when unreadable."""
1135
+ return _load_json(pathlib.Path(incident) / "manifest.json")
1136
+
1137
+
1138
+ def classify_incident_dir(incident, *, family=None, bundles=(), window_seconds=None):
1139
+ """Classify one incident directory on disk (§4.3)."""
1140
+ incident = pathlib.Path(incident)
1141
+ resolved = family or family_of_incident(incident.name)
1142
+ if resolved is None:
1143
+ raise ValueError(f"unrecognized incident directory name: {incident.name}")
1144
+ kwargs = {}
1145
+ if window_seconds is not None:
1146
+ kwargs["window_seconds"] = window_seconds
1147
+ return _kernel.classify_incident(
1148
+ family=resolved,
1149
+ incident_name=incident.name,
1150
+ manifest=load_incident_manifest(incident),
1151
+ bundles=bundles,
1152
+ incident_time=incident_time(resolved, incident.name),
1153
+ **kwargs,
1154
+ )
1155
+
1156
+
1157
+ def _atomic_write_private(path: pathlib.Path, payload: dict) -> None:
1158
+ token = f"{os.getpid()}-{path.name}"
1159
+ temp = path.with_name(f".{path.name}.{token}.tmp")
1160
+ fd = os.open(temp, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
1161
+ try:
1162
+ os.write(fd, (json.dumps(payload, indent=2, sort_keys=True) + "\n").encode())
1163
+ os.fsync(fd)
1164
+ finally:
1165
+ os.close(fd)
1166
+ try:
1167
+ os.replace(temp, path)
1168
+ except OSError:
1169
+ # Leaving the scratch file behind would put an unclassifiable dotfile
1170
+ # inside an incident directory, which the retention planner then has to
1171
+ # reason about. `_atomic_write_private_json` already cleans up here.
1172
+ with contextlib.suppress(OSError):
1173
+ os.unlink(temp)
1174
+ raise
1175
+ os.chmod(path, 0o600)
1176
+
1177
+
1178
+ def backfill_classification(
1179
+ incident, *, family=None, bundles=(), window_seconds=None,
1180
+ ) -> bool:
1181
+ """Write `classification.json` beside an incident's manifest (§4.5).
1182
+
1183
+ Returns True when the file changed. Idempotent by construction: a re-run
1184
+ over an unchanged corpus rewrites nothing, so repeated runs are
1185
+ byte-identical rather than merely equivalent.
1186
+
1187
+ A verdict already on disk is NEVER overridden, whatever this pass would
1188
+ decide, and that includes raising an `unknown` to something stronger. An
1189
+ `unknown` written by an earlier classifier is a CONSIDERED verdict — it
1190
+ records that no bundle correlated inside the window — so replacing it here
1191
+ would discard a decision made with evidence this pass does not have (§4.4).
1192
+ """
1193
+ incident = pathlib.Path(incident)
1194
+ path = incident / "classification.json"
1195
+ if path.exists():
1196
+ return False
1197
+ verdict = classify_incident_dir(
1198
+ incident, family=family, bundles=bundles, window_seconds=window_seconds,
1199
+ )
1200
+ payload = _kernel.verdict_to_record(verdict)
1201
+ payload["classifiedAtUtc"] = dt.datetime.now(dt.timezone.utc).isoformat(
1202
+ timespec="seconds"
1203
+ ).replace("+00:00", "Z")
1204
+ _atomic_write_private(path, payload)
1205
+ return True
1206
+
1207
+
1208
+ # --------------------------------------------------------------------------
1209
+ # §7.5 — the bounded metadata walk
1210
+ # --------------------------------------------------------------------------
1211
+ #
1212
+ # `os.scandir` plus a non-following `lstat` over the recognized shallow shapes,
1213
+ # summing `st_blocks * 512` for disk bytes with `st_size` as the logical figure
1214
+ # and the fallback. A v2 manifest's `familySizes` corroborates, never decides.
1215
+ #
1216
+ # The walk is NEW work on a periodic path: the previous gather only enumerated
1217
+ # top-level entries, and `doctor_gather_state` is reached from the TUI and the
1218
+ # dashboard snapshot precompute as well as from `GET /api/doctor`. It is
1219
+ # therefore bounded twice — a depth cap and an entry cap — and reports a
1220
+ # partial scan rather than silently under-reporting.
1221
+ #
1222
+ # Measured on the maintainer's install: 350 roots, ~1150 members, and a warm
1223
+ # depth-2 walk over 1036 entries in 3.5–4.8 ms. The regression gate that
1224
+ # actually catches a change is the operation count, not the wall clock.
1225
+
1226
+ #: One directory below each recognized root: `quarantine/<incident>/<member>`
1227
+ #: and `logs/<evidence>/<member>` are the deepest shapes that exist.
1228
+ WALK_MAX_DEPTH = 2
1229
+
1230
+ #: Beyond this the leg reports `partial` and degrades to WARN. Roughly five
1231
+ #: times the maintainer's whole corpus.
1232
+ WALK_MAX_ENTRIES = 5000
1233
+
1234
+ #: The databases whose control markers name a retained artifact.
1235
+ _MARKER_DB_NAMES = ("stats.db", "cache.db", "conversations.db")
1236
+
1237
+ #: A heal-ring entry at this outcome may still be acted on by a worker, so the
1238
+ #: evidence it names is `active` (§3.2). Every other outcome is terminal.
1239
+ _LIVE_HEAL_OUTCOMES = frozenset({"", "detected"})
1240
+
1241
+ _REBUILD_RECORD_RE = re.compile(
1242
+ r"^stats-rebuild-(?P<stamp>\d{8}T\d{6}_\d{6})\.json$"
1243
+ )
1244
+ _EVIDENCE_DIR_RE = re.compile(
1245
+ r"^(?P<family>.+)-corruption-forensics-(?P<stamp>\d{8}T\d{6})Z$"
1246
+ )
1247
+ _BACKUP_SIDECAR_SUFFIX = ".classification.json"
1248
+
1249
+
1250
+ def _walk_scandir(path):
1251
+ """Seam for the §7.5 operation-count test; not for callers."""
1252
+ return os.scandir(path)
1253
+
1254
+
1255
+ def _walk_lstat(path):
1256
+ """Seam for the §7.5 operation-count test; not for callers."""
1257
+ return os.lstat(path)
1258
+
1259
+
1260
+ @dataclasses.dataclass(frozen=True)
1261
+ class RetentionScan:
1262
+ """What one metadata walk observed.
1263
+
1264
+ `partial` is True when the entry cap stopped the walk, which the doctor leg
1265
+ reports as WARN rather than presenting an under-count as the truth.
1266
+ """
1267
+
1268
+ members: "tuple[_kernel.RetentionMember, ...]"
1269
+ partial: bool
1270
+ entries_seen: int
1271
+ free_disk_bytes: "int | None"
1272
+ incidents: "tuple[str, ...]"
1273
+ #: Why an incident carries no verdict — `unknown` when a classification
1274
+ #: file exists and reports an undecided confidence, `absent` when there is
1275
+ #: none at all. The kernel cannot tell these apart, because both resolve to
1276
+ #: `classification=None`, but `db prune` must state which one an operator
1277
+ #: is looking at: one has been considered and the other has not.
1278
+ classification_detail: "dict[str, str]" = dataclasses.field(
1279
+ default_factory=dict
1280
+ )
1281
+ #: `{family: [(when, path, payload), ...]}` ascending — the correlation
1282
+ #: input §4.3 needs, taken from the walk that already read every bundle
1283
+ #: rather than from a second directory scan that could disagree with it.
1284
+ bundles_by_family: "dict[str, list]" = dataclasses.field(
1285
+ default_factory=dict
1286
+ )
1287
+
1288
+
1289
+ class _EntryBudget:
1290
+ def __init__(self, limit: int):
1291
+ self.limit = int(limit)
1292
+ self.seen = 0
1293
+ self.exhausted = False
1294
+
1295
+ def take(self) -> bool:
1296
+ if self.seen >= self.limit:
1297
+ self.exhausted = True
1298
+ return False
1299
+ self.seen += 1
1300
+ return True
1301
+
1302
+
1303
+ def _disk_bytes(info) -> int:
1304
+ """Allocated bytes, falling back to the logical size when unavailable."""
1305
+ blocks = getattr(info, "st_blocks", None)
1306
+ if blocks is None:
1307
+ return int(info.st_size)
1308
+ return int(blocks) * 512
1309
+
1310
+
1311
+ def _scan_entries(path, budget: _EntryBudget) -> "list[tuple[str, object]]":
1312
+ """One `scandir` over `path`, one non-following `lstat` per entry.
1313
+
1314
+ Every entry costs one unit of the budget, including the ones the caller
1315
+ goes on to ignore, because the walk paid for it either way.
1316
+ """
1317
+ found: "list[tuple[str, object]]" = []
1318
+ try:
1319
+ with _walk_scandir(path) as entries:
1320
+ names = sorted(entry.name for entry in entries)
1321
+ except OSError:
1322
+ return found
1323
+ for name in names:
1324
+ if not budget.take():
1325
+ return found
1326
+ info = _lstat_or_none(pathlib.Path(path) / name)
1327
+ if info is not None:
1328
+ found.append((name, info))
1329
+ return found
1330
+
1331
+
1332
+ def _tree_bytes(path, budget: _EntryBudget, depth: int) -> "tuple[int, int, list[str]]":
1333
+ """`(disk, logical, entry names)` for one directory and its children."""
1334
+ disk = 0
1335
+ logical = 0
1336
+ names: "list[str]" = []
1337
+ if depth > WALK_MAX_DEPTH:
1338
+ return disk, logical, names
1339
+ for name, info in _scan_entries(path, budget):
1340
+ names.append(name)
1341
+ disk += _disk_bytes(info)
1342
+ logical += int(info.st_size)
1343
+ if _stat.S_ISDIR(info.st_mode) and not _stat.S_ISLNK(info.st_mode):
1344
+ child_disk, child_logical, _ = _tree_bytes(
1345
+ pathlib.Path(path) / name, budget, depth + 1,
1346
+ )
1347
+ disk += child_disk
1348
+ logical += child_logical
1349
+ return disk, logical, names
1350
+
1351
+
1352
+ def _epoch_of(when) -> "float | None":
1353
+ return None if when is None else when.timestamp()
1354
+
1355
+
1356
+ def _reference_id(root: pathlib.Path, raw, known) -> "str | None":
1357
+ """A recorded absolute path as a stable relative id, or a dangling token.
1358
+
1359
+ A reference that resolves outside the data directory, or names something
1360
+ the walk did not recognize, is returned as a token that is deliberately NOT
1361
+ a member id — `build_graph` then reports `dangling-reference` and §3.2
1362
+ protects every root that can reach it. Returning None here instead would
1363
+ silently drop the very condition the gate exists to catch.
1364
+ """
1365
+ if not isinstance(raw, str) or not raw:
1366
+ return None
1367
+ try:
1368
+ relative = os.path.relpath(raw, str(root))
1369
+ except ValueError:
1370
+ return f"!unresolved:{raw}"
1371
+ if relative.startswith(os.pardir) or os.path.isabs(relative):
1372
+ return f"!outside-root:{raw}"
1373
+ if relative not in known:
1374
+ return f"!missing:{relative}"
1375
+ return relative
1376
+
1377
+
1378
+ def _read_control_markers(root: pathlib.Path, top_names) -> "dict[str, object]":
1379
+ """Every durable marker that makes a retained artifact `active` (§3.2).
1380
+
1381
+ Read by name rather than by enumeration, and read WITHOUT any lock: these
1382
+ are `0600` JSON files, and the walk runs inside the exclusive retention
1383
+ hold, where acquiring a lock earlier in the total order is forbidden.
1384
+ """
1385
+ active_paths: "set[str]" = set()
1386
+ pending_incidents: "set[str]" = set()
1387
+
1388
+ heal_request = _load_json(root / "stats-corruption-heal.pending")
1389
+ if heal_request.get("forensicsPath"):
1390
+ active_paths.add(str(heal_request["forensicsPath"]))
1391
+
1392
+ for db_name in _MARKER_DB_NAMES:
1393
+ marker = f"{db_name}.publication"
1394
+ if marker in top_names:
1395
+ state = _load_json(root / marker)
1396
+ for key in ("recordPath", "scratchPath"):
1397
+ if state.get(key):
1398
+ active_paths.add(str(state[key]))
1399
+ pending = f"{db_name}.quarantine-pending.json"
1400
+ if pending in top_names:
1401
+ state = _load_json(root / pending)
1402
+ if state.get("incidentPath"):
1403
+ pending_incidents.add(str(state["incidentPath"]))
1404
+
1405
+ ring = _load_json(pathlib.Path(_cctally_core.LOG_DIR) / "stats-heal-events.json")
1406
+ events = ring.get("events")
1407
+ if isinstance(events, list):
1408
+ for event in events:
1409
+ if not isinstance(event, dict):
1410
+ continue
1411
+ if str(event.get("outcome") or "") not in _LIVE_HEAL_OUTCOMES:
1412
+ continue
1413
+ for key in ("forensicsPath", "incidentPath"):
1414
+ if event.get(key):
1415
+ active_paths.add(str(event[key]))
1416
+
1417
+ return {"active_paths": active_paths, "pending_incidents": pending_incidents}
1418
+
1419
+
1420
+ def _is_active(root: pathlib.Path, member_id: str, active_paths) -> bool:
1421
+ return str(root / member_id) in active_paths
1422
+
1423
+
1424
+ def _backup_family_name(stem_name: str) -> str:
1425
+ head = stem_name.split(".bak-", 1)[0]
1426
+ return head or stem_name
1427
+
1428
+
1429
+ def _backup_stamp_epoch(stem_name: str) -> "float | None":
1430
+ match = re.search(r"\.bak-(?:corrupt-malformed-)?(\d{8}T\d{6})Z$", stem_name)
1431
+ if match is None:
1432
+ return None
1433
+ try:
1434
+ return dt.datetime.strptime(match.group(1), "%Y%m%dT%H%M%S").replace(
1435
+ tzinfo=dt.timezone.utc
1436
+ ).timestamp()
1437
+ except ValueError:
1438
+ return None
1439
+
1440
+
1441
+ def gather_retained_artifacts(
1442
+ *,
1443
+ root=None,
1444
+ include_backups: bool = False,
1445
+ max_entries: int = WALK_MAX_ENTRIES,
1446
+ measure_free_disk: bool = True,
1447
+ ) -> RetentionScan:
1448
+ """Observe every recognized retained artifact, bounded (§7.5).
1449
+
1450
+ Produces the `RetentionMember`s `build_graph` needs and nothing else: the
1451
+ kernel re-derives no field from disk, so every protection condition is
1452
+ decided here and handed over as a value.
1453
+
1454
+ Takes NO lock. The worker calls it inside its exclusive retention hold,
1455
+ where acquiring anything earlier in the total order would close a real
1456
+ deadlock cycle; `db prune`'s preview calls it under no hold at all.
1457
+ """
1458
+ root = _reclaim_root(root)
1459
+ budget = _EntryBudget(max_entries)
1460
+ members: "list[_kernel.RetentionMember]" = []
1461
+
1462
+ top = _scan_entries(root, budget)
1463
+ top_names = {name for name, _info in top}
1464
+ markers = _read_control_markers(root, top_names)
1465
+ active_paths = markers["active_paths"]
1466
+ pending_incidents = markers["pending_incidents"]
1467
+
1468
+ logs_dir = pathlib.Path(_cctally_core.LOG_DIR)
1469
+ quarantine_dir = root / "quarantine"
1470
+
1471
+ # ---- incidents -------------------------------------------------------
1472
+ incident_records: "list[dict]" = []
1473
+ for name, info in _scan_entries(quarantine_dir, budget):
1474
+ member_id = f"quarantine/{name}"
1475
+ is_link = bool(_stat.S_ISLNK(info.st_mode))
1476
+ family = family_of_incident(name)
1477
+ if family is None or not _stat.S_ISDIR(info.st_mode) or is_link:
1478
+ # Not a shape this subsystem wrote. It still occupies the disk it
1479
+ # occupies, so it is reported as an unrecognized kind — which
1480
+ # `build_graph` roots and protects rather than sweeping.
1481
+ members.append(_unknown_member(
1482
+ member_id, info, family or "quarantine", is_link,
1483
+ ))
1484
+ continue
1485
+ disk, logical, observed = _tree_bytes(
1486
+ quarantine_dir / name, budget, WALK_MAX_DEPTH,
1487
+ )
1488
+ manifest = load_incident_manifest(quarantine_dir / name)
1489
+ verdict = _load_json(quarantine_dir / name / "classification.json")
1490
+ incident_records.append({
1491
+ "id": member_id,
1492
+ "name": name,
1493
+ "family": family,
1494
+ "info": info,
1495
+ "disk": disk + _disk_bytes(info),
1496
+ "logical": logical + int(info.st_size),
1497
+ "observed": observed,
1498
+ "manifest": manifest,
1499
+ "verdict": verdict,
1500
+ })
1501
+
1502
+ # ---- logs: bundles, WAL evidence, rebuild records ---------------------
1503
+ bundle_records: "list[dict]" = []
1504
+ evidence_records: "list[dict]" = []
1505
+ record_records: "list[dict]" = []
1506
+ for name, info in _scan_entries(logs_dir, budget):
1507
+ member_id = f"logs/{name}"
1508
+ is_link = bool(_stat.S_ISLNK(info.st_mode))
1509
+ is_dir = bool(_stat.S_ISDIR(info.st_mode)) and not is_link
1510
+ rebuild = _REBUILD_RECORD_RE.match(name)
1511
+ if rebuild is not None and not is_dir:
1512
+ payload = _load_json(logs_dir / name)
1513
+ record_records.append({
1514
+ "id": member_id, "name": name, "info": info,
1515
+ "payload": payload, "is_link": is_link,
1516
+ "stamp": rebuild["stamp"],
1517
+ })
1518
+ continue
1519
+ evidence = _EVIDENCE_DIR_RE.match(name)
1520
+ if evidence is not None and is_dir:
1521
+ disk, logical, _children = _tree_bytes(
1522
+ logs_dir / name, budget, WALK_MAX_DEPTH,
1523
+ )
1524
+ evidence_records.append({
1525
+ "id": member_id, "name": name, "info": info,
1526
+ "family": evidence["family"], "stamp": evidence["stamp"],
1527
+ "disk": disk + _disk_bytes(info),
1528
+ "logical": logical + int(info.st_size),
1529
+ })
1530
+ continue
1531
+ family = next(
1532
+ (fam for fam in KNOWN_FAMILIES if bundle_time(fam, name) is not None),
1533
+ None,
1534
+ )
1535
+ if family is not None and not is_dir:
1536
+ bundle_records.append({
1537
+ "id": member_id, "name": name, "info": info, "family": family,
1538
+ "payload": _load_json(logs_dir / name), "is_link": is_link,
1539
+ })
1540
+
1541
+ # ---- backup families -------------------------------------------------
1542
+ backup_records = _collect_backup_families(root, top, include_backups)
1543
+
1544
+ known = (
1545
+ {record["id"] for record in incident_records}
1546
+ | {record["id"] for record in bundle_records}
1547
+ | {record["id"] for record in evidence_records}
1548
+ | {record["id"] for record in record_records}
1549
+ )
1550
+
1551
+ evidence_by_id = {record["id"]: record for record in evidence_records}
1552
+ referenced_evidence: "set[str]" = set()
1553
+
1554
+ bundles_by_family: "dict[str, list]" = {}
1555
+ for record in bundle_records:
1556
+ when = bundle_time(record["family"], record["name"])
1557
+ if when is not None:
1558
+ bundles_by_family.setdefault(record["family"], []).append(
1559
+ (when, str(logs_dir / record["name"]), record["payload"])
1560
+ )
1561
+ for entries in bundles_by_family.values():
1562
+ entries.sort(key=lambda item: (item[0], item[1]))
1563
+
1564
+ for record in incident_records:
1565
+ manifest = record["manifest"]
1566
+ references = tuple(
1567
+ ref for ref in (
1568
+ _reference_id(root, manifest.get("forensicsPath"), known),
1569
+ _reference_id(root, manifest.get("rebuildRecordPath"), known),
1570
+ ) if ref is not None
1571
+ )
1572
+ damage = manifest.get("damage")
1573
+ preserved = damage.get("preserved") if isinstance(damage, dict) else None
1574
+ shape = (
1575
+ preserved.get("shapeToken") if isinstance(preserved, dict) else None
1576
+ )
1577
+ members.append(_kernel.RetentionMember(
1578
+ id=record["id"],
1579
+ kind="incident",
1580
+ family=record["family"],
1581
+ created_at_epoch=(
1582
+ _epoch_of(incident_time(record["family"], record["name"]))
1583
+ or float(record["info"].st_mtime)
1584
+ ),
1585
+ disk_bytes=record["disk"],
1586
+ logical_bytes=record["logical"],
1587
+ references=references,
1588
+ is_symlink=False,
1589
+ in_root=True,
1590
+ exists=True,
1591
+ valid=_kernel.validate_incident(
1592
+ manifest=manifest, observed=record["observed"],
1593
+ ),
1594
+ classification=_incident_confidence(
1595
+ record, bundles_by_family.get(record["family"], ()),
1596
+ ),
1597
+ shape_token=shape if isinstance(shape, str) else None,
1598
+ finalized=_kernel.incident_is_finalized(
1599
+ manifest=manifest,
1600
+ pending_marker_present=(
1601
+ str(root / record["id"]) in pending_incidents
1602
+ ),
1603
+ ),
1604
+ active=_is_active(root, record["id"], active_paths),
1605
+ ))
1606
+
1607
+ for record in bundle_records:
1608
+ payload = record["payload"]
1609
+ evidence_id = f"logs/{record['name'][:-len('.json')]}"
1610
+ references: "tuple[str, ...]" = ()
1611
+ if evidence_id in evidence_by_id:
1612
+ references = (evidence_id,)
1613
+ referenced_evidence.add(evidence_id)
1614
+ trigger = payload.get("trigger")
1615
+ origin = trigger.get("origin") if isinstance(trigger, dict) else None
1616
+ members.append(_kernel.RetentionMember(
1617
+ id=record["id"],
1618
+ kind="bundle",
1619
+ family=record["family"],
1620
+ created_at_epoch=(
1621
+ _epoch_of(bundle_time(record["family"], record["name"]))
1622
+ or float(record["info"].st_mtime)
1623
+ ),
1624
+ disk_bytes=_disk_bytes(record["info"]),
1625
+ logical_bytes=int(record["info"].st_size),
1626
+ references=references,
1627
+ is_symlink=record["is_link"],
1628
+ in_root=True,
1629
+ exists=True,
1630
+ valid=_kernel.validate_bundle(payload),
1631
+ # §3.3: a referenced bundle INHERITS its referrer's verdict, which
1632
+ # the kernel gives for free by reading classification only on the
1633
+ # root. An unreferenced one classifies by its own `trigger.origin`.
1634
+ classification="exact" if isinstance(origin, str) and origin else None,
1635
+ shape_token=_bundle_shape_token(payload),
1636
+ finalized=True,
1637
+ active=_is_active(root, record["id"], active_paths),
1638
+ ))
1639
+
1640
+ for record in record_records:
1641
+ payload = record["payload"]
1642
+ references = tuple(
1643
+ ref for ref in (
1644
+ _reference_id(root, payload.get("forensicsPath"), known),
1645
+ _reference_id(root, payload.get("incidentPath"), known),
1646
+ ) if ref is not None
1647
+ )
1648
+ trigger = payload.get("trigger")
1649
+ members.append(_kernel.RetentionMember(
1650
+ id=record["id"],
1651
+ kind="rebuild_record",
1652
+ family="stats.db",
1653
+ created_at_epoch=(
1654
+ _record_stamp_epoch(record["stamp"])
1655
+ or float(record["info"].st_mtime)
1656
+ ),
1657
+ disk_bytes=_disk_bytes(record["info"]),
1658
+ logical_bytes=int(record["info"].st_size),
1659
+ references=references,
1660
+ is_symlink=record["is_link"],
1661
+ in_root=True,
1662
+ exists=True,
1663
+ valid=_kernel.validate_rebuild_record(payload),
1664
+ classification=(
1665
+ "exact" if isinstance(trigger, str) and trigger else None
1666
+ ),
1667
+ shape_token=None,
1668
+ finalized=True,
1669
+ active=(
1670
+ payload.get("status") == "pending"
1671
+ or _is_active(root, record["id"], active_paths)
1672
+ ),
1673
+ ))
1674
+
1675
+ for record in evidence_records:
1676
+ members.append(_kernel.RetentionMember(
1677
+ id=record["id"],
1678
+ kind="wal_evidence",
1679
+ family=record["family"],
1680
+ created_at_epoch=(
1681
+ _record_stamp_epoch(record["stamp"], fmt="%Y%m%dT%H%M%S")
1682
+ or float(record["info"].st_mtime)
1683
+ ),
1684
+ disk_bytes=record["disk"],
1685
+ logical_bytes=record["logical"],
1686
+ references=(),
1687
+ is_symlink=False,
1688
+ in_root=True,
1689
+ exists=True,
1690
+ # §3.3: valid when a bundle or incident references it. Nothing
1691
+ # does, so it cannot be validated and must not be swept alone —
1692
+ # `build_graph` adds `unreferenced-evidence` on top for a root.
1693
+ valid=record["id"] in referenced_evidence,
1694
+ classification=None,
1695
+ shape_token=None,
1696
+ finalized=True,
1697
+ active=_is_active(root, record["id"], active_paths),
1698
+ ))
1699
+
1700
+ members.extend(backup_records)
1701
+
1702
+ free_disk = None
1703
+ if measure_free_disk:
1704
+ try:
1705
+ free_disk = int(_disk_usage(str(root)).free)
1706
+ except OSError:
1707
+ free_disk = None
1708
+
1709
+ return RetentionScan(
1710
+ members=tuple(members),
1711
+ partial=budget.exhausted,
1712
+ entries_seen=budget.seen,
1713
+ free_disk_bytes=free_disk,
1714
+ incidents=tuple(record["id"] for record in incident_records),
1715
+ classification_detail={
1716
+ record["id"]: (
1717
+ "unknown" if record["verdict"] else "absent"
1718
+ )
1719
+ for record in incident_records
1720
+ },
1721
+ bundles_by_family=bundles_by_family,
1722
+ )
1723
+
1724
+
1725
+ def _incident_confidence(record, bundles):
1726
+ """The verdict this incident carries, or the one a backfill would write.
1727
+
1728
+ A PREVIEW must plan the same deletion `--yes` plans (§5.7), and the apply
1729
+ classifies before it plans (§4.5). Reading only what is already on disk
1730
+ would therefore under-report by exactly the incidents the apply reclaims.
1731
+
1732
+ A verdict already recorded is never second-guessed, INCLUDING an `unknown`
1733
+ one: §4.4 makes that a considered decision, and `backfill_classification`
1734
+ refuses to overwrite it, so folding a fresh correlation in here would make
1735
+ the preview promise a deletion the apply will not perform.
1736
+ """
1737
+ recorded = _kernel.incident_classification(
1738
+ manifest=record["manifest"], verdict=record["verdict"],
1739
+ incident_name=record["name"],
1740
+ )
1741
+ if recorded is not None or record["verdict"]:
1742
+ return recorded
1743
+ verdict = _kernel.classify_incident(
1744
+ family=record["family"],
1745
+ incident_name=record["name"],
1746
+ manifest=record["manifest"],
1747
+ bundles=bundles,
1748
+ incident_time=incident_time(record["family"], record["name"]),
1749
+ )
1750
+ return verdict.confidence if _kernel.is_classified(verdict.confidence) else None
1751
+
1752
+
1753
+ def _disk_usage(path):
1754
+ import shutil
1755
+
1756
+ return shutil.disk_usage(path)
1757
+
1758
+
1759
+ def _bundle_shape_token(payload) -> "str | None":
1760
+ damage = payload.get("damage")
1761
+ token = damage.get("shapeToken") if isinstance(damage, dict) else None
1762
+ return token if isinstance(token, str) else None
1763
+
1764
+
1765
+ def _record_stamp_epoch(stamp: str, fmt: str = "%Y%m%dT%H%M%S_%f") -> "float | None":
1766
+ try:
1767
+ return dt.datetime.strptime(stamp, fmt).replace(
1768
+ tzinfo=dt.timezone.utc
1769
+ ).timestamp()
1770
+ except ValueError:
1771
+ return None
1772
+
1773
+
1774
+ def _unknown_member(member_id, info, family, is_link):
1775
+ """Something inside a recognized directory that this subsystem did not write.
1776
+
1777
+ Reported with a kind no validator claims, which `build_graph` roots and
1778
+ protects. Its bytes still count toward the budget, so the operator sees the
1779
+ disk it occupies rather than a total that quietly omits it.
1780
+ """
1781
+ return _kernel.RetentionMember(
1782
+ id=member_id,
1783
+ kind="unknown",
1784
+ family=family,
1785
+ created_at_epoch=float(info.st_mtime),
1786
+ disk_bytes=_disk_bytes(info),
1787
+ logical_bytes=int(info.st_size),
1788
+ references=(),
1789
+ is_symlink=bool(is_link),
1790
+ in_root=True,
1791
+ exists=True,
1792
+ valid=False,
1793
+ classification=None,
1794
+ shape_token=None,
1795
+ finalized=True,
1796
+ active=False,
1797
+ )
1798
+
1799
+
1800
+ def _collect_backup_families(
1801
+ root: pathlib.Path, top, include_backups: bool,
1802
+ ) -> "list[_kernel.RetentionMember]":
1803
+ """Group `<db>.bak-*` entries by STEM, including `-wal` and `-shm` (§3.7).
1804
+
1805
+ `_copy_db_family` copies all three, and `tests/test_db_repair_314.py`
1806
+ pins that the backup WAL survives, so the family is one root and its
1807
+ sidecars are members of it rather than roots of their own.
1808
+ """
1809
+ by_name = {name: info for name, info in top if ".bak-" in name}
1810
+ satellites: "dict[str, list[str]]" = {}
1811
+ stems: "list[str]" = []
1812
+ for name in sorted(by_name):
1813
+ owner = None
1814
+ for suffix in ("-wal", "-shm", _BACKUP_SIDECAR_SUFFIX):
1815
+ if name.endswith(suffix) and name[: -len(suffix)] in by_name:
1816
+ owner = name[: -len(suffix)]
1817
+ break
1818
+ if owner is None:
1819
+ stems.append(name)
1820
+ else:
1821
+ satellites.setdefault(owner, []).append(name)
1822
+
1823
+ members: "list[_kernel.RetentionMember]" = []
1824
+ for stem in stems:
1825
+ info = by_name[stem]
1826
+ origin = _kernel.backup_origin(stem)
1827
+ family_names = [stem] + sorted(satellites.get(stem, []))
1828
+ observed = [
1829
+ {
1830
+ "name": name,
1831
+ "size": int(by_name[name].st_size),
1832
+ "mtime": float(by_name[name].st_mtime),
1833
+ "device": int(by_name[name].st_dev),
1834
+ "inode": int(by_name[name].st_ino),
1835
+ }
1836
+ for name in family_names
1837
+ if not name.endswith(_BACKUP_SIDECAR_SUFFIX)
1838
+ ]
1839
+ sidecar = _load_json(root / f"{stem}{_BACKUP_SIDECAR_SUFFIX}")
1840
+ if origin == "machine":
1841
+ classification = _kernel.backup_classification(
1842
+ sidecar=sidecar, observed=observed,
1843
+ )
1844
+ elif origin == "user" and include_backups:
1845
+ # §6.1: `--include-backups` reaches exactly the backups Q4
1846
+ # excludes. An UNRECOGNIZED name stays out even then — §3.7's
1847
+ # third row is the fail-safe, and this install carries several.
1848
+ classification = "user-requested"
1849
+ else:
1850
+ classification = None
1851
+ references = tuple(
1852
+ f"{name}" for name in family_names if name != stem
1853
+ )
1854
+ members.append(_kernel.RetentionMember(
1855
+ id=stem,
1856
+ kind="backup",
1857
+ family=_backup_family_name(stem),
1858
+ created_at_epoch=(
1859
+ _backup_stamp_epoch(stem) or float(info.st_mtime)
1860
+ ),
1861
+ disk_bytes=_disk_bytes(info),
1862
+ logical_bytes=int(info.st_size),
1863
+ references=references,
1864
+ is_symlink=bool(_stat.S_ISLNK(info.st_mode)),
1865
+ in_root=True,
1866
+ exists=True,
1867
+ valid=True,
1868
+ classification=classification,
1869
+ shape_token=None,
1870
+ finalized=True,
1871
+ active=False,
1872
+ ))
1873
+ for name in references:
1874
+ sat = by_name[name]
1875
+ members.append(_kernel.RetentionMember(
1876
+ id=name,
1877
+ kind="backup_member",
1878
+ family=_backup_family_name(stem),
1879
+ created_at_epoch=float(sat.st_mtime),
1880
+ disk_bytes=_disk_bytes(sat),
1881
+ logical_bytes=int(sat.st_size),
1882
+ references=(),
1883
+ is_symlink=bool(_stat.S_ISLNK(sat.st_mode)),
1884
+ in_root=True,
1885
+ exists=True,
1886
+ valid=True,
1887
+ classification=None,
1888
+ shape_token=None,
1889
+ finalized=True,
1890
+ active=False,
1891
+ ))
1892
+ return members
1893
+
1894
+
1895
+ # --------------------------------------------------------------------------
1896
+ # §6.5 — the strict policy read, and §5.6's production guard
1897
+ # --------------------------------------------------------------------------
1898
+
1899
+ #: The one nested config key this subsystem reads.
1900
+ RETENTION_CONFIG_KEY = "storage.artifact_retention"
1901
+
1902
+
1903
+ def read_retention_policy() -> "_kernel.PolicyResolution":
1904
+ """Resolve the persisted policy from a RAW read of `config.json` (§6.5).
1905
+
1906
+ `load_config()` must not be used: it turns corrupt JSON into defaults,
1907
+ which would silently arm deletion with a policy the user never wrote. A
1908
+ file that cannot be parsed is `malformed`, exactly like a block that fails
1909
+ validation, because in both cases the policy on disk is not the policy the
1910
+ operator would read back.
1911
+ """
1912
+ path = pathlib.Path(_cctally_core.CONFIG_PATH)
1913
+ if not path.exists():
1914
+ return _kernel.resolve_retention_policy(None)
1915
+ try:
1916
+ loaded = json.loads(path.read_text(encoding="utf-8"))
1917
+ except (OSError, ValueError) as exc:
1918
+ return _kernel.PolicyResolution(
1919
+ "malformed", None, f"config.json could not be read: {exc}",
1920
+ )
1921
+ if not isinstance(loaded, dict):
1922
+ return _kernel.PolicyResolution(
1923
+ "malformed", None, "config.json is not a JSON object",
1924
+ )
1925
+ return _kernel.resolve_retention_policy(loaded.get(RETENTION_CONFIG_KEY))
1926
+
1927
+
1928
+ def would_block_prod_retention(root=None) -> bool:
1929
+ """§5.6: a development binary never prunes the production data directory.
1930
+
1931
+ Mirrors `_would_block_prod_stats`: a suppressor-independent raw `.git`
1932
+ check against a password-DB-resolved prod directory, so a fake-`HOME` test
1933
+ is exempt and `CCTALLY_ALLOW_PROD_MIGRATION` is the escape.
1934
+ """
1935
+ if _cctally_core._truthy_env("CCTALLY_ALLOW_PROD_MIGRATION"):
1936
+ return False
1937
+ if not (_cctally_core._repo_root() / ".git").exists():
1938
+ return False
1939
+ try:
1940
+ return (
1941
+ pathlib.Path(_reclaim_root(root)).resolve()
1942
+ == _cctally_core._real_prod_data_dir().resolve()
1943
+ )
1944
+ except OSError:
1945
+ return False
1946
+
1947
+
1948
+ # --------------------------------------------------------------------------
1949
+ # §5.1 / §5.7 — one sweep implementation, shared by the worker and `db prune`
1950
+ # --------------------------------------------------------------------------
1951
+
1952
+ #: Every status a sweep can report. `blocked` is not an error: it means the
1953
+ #: worker could not take the exclusive hold, so it marked nothing.
1954
+ SWEEP_OK = "ok"
1955
+ SWEEP_BLOCKED = "blocked"
1956
+ SWEEP_POLICY_MALFORMED = "policy-malformed"
1957
+ SWEEP_PROD_REFUSED = "prod-refused"
1958
+ SWEEP_PARTIAL = "partial"
1959
+
1960
+
1961
+ @dataclasses.dataclass(frozen=True)
1962
+ class SweepResult:
1963
+ """What one sweep decided and, when it applied, what it achieved."""
1964
+
1965
+ status: str
1966
+ policy: "_kernel.RetentionPolicy | None"
1967
+ reason: "str | None"
1968
+ scan: "RetentionScan | None"
1969
+ plan: "object | None"
1970
+ outcome: "ReclaimOutcome | None"
1971
+ stuck: "tuple[dict, ...]"
1972
+ applied: bool
1973
+
1974
+
1975
+ def plan_retention(
1976
+ *, policy, root=None, include_backups: bool = False, now_epoch=None,
1977
+ max_entries: int = WALK_MAX_ENTRIES, with_graph: bool = False,
1978
+ ):
1979
+ """Walk, build the graph and plan. No lock, no mutation, no clock but this.
1980
+
1981
+ Shared by `db prune`'s preview and the worker, so the preview describes the
1982
+ same decision `--yes` would make.
1983
+
1984
+ `with_graph=True` returns the graph alongside, so a caller that also needs
1985
+ `summarize_prune` does not rebuild it. The default stays the two-tuple
1986
+ every existing caller destructures.
1987
+ """
1988
+ scan = gather_retained_artifacts(
1989
+ root=root, include_backups=include_backups, max_entries=max_entries,
1990
+ )
1991
+ graph = _kernel.build_graph(scan.members)
1992
+ state = _kernel.RetentionState(
1993
+ graph=graph,
1994
+ now_epoch=time.time() if now_epoch is None else float(now_epoch),
1995
+ free_disk_bytes=scan.free_disk_bytes,
1996
+ )
1997
+ plan = _kernel.plan_artifact_retention(state, policy)
1998
+ return (scan, plan, graph) if with_graph else (scan, plan)
1999
+
2000
+
2001
+ def _backfill_scan_classifications(root: pathlib.Path, scan: RetentionScan) -> int:
2002
+ """Persist §4.3's verdict for every incident that has none (§4.5).
2003
+
2004
+ Runs inside the worker's EXISTING exclusive hold, taking no second flock.
2005
+ Idempotent: `backfill_classification` refuses to overwrite a verdict
2006
+ already on disk, including an `unknown` one, because that is a CONSIDERED
2007
+ decision made with evidence this pass does not have.
2008
+ """
2009
+ written = 0
2010
+ for member_id in scan.incidents:
2011
+ incident = root / member_id
2012
+ family = family_of_incident(incident.name)
2013
+ if family is None:
2014
+ continue
2015
+ try:
2016
+ if backfill_classification(
2017
+ incident, family=family,
2018
+ bundles=scan.bundles_by_family.get(family, ()),
2019
+ ):
2020
+ written += 1
2021
+ except (OSError, ValueError):
2022
+ continue
2023
+ return written
2024
+
2025
+
2026
+ def run_retention_sweep(
2027
+ *, root=None, policy=None, include_backups: bool = False,
2028
+ apply: bool = True, timeout=None, now_epoch=None, backfill: bool = True,
2029
+ ) -> SweepResult:
2030
+ """The one sweep: classify, plan and mark under the hold; delete outside it.
2031
+
2032
+ `cctally db prune --yes` and the detached `_artifact-retention` worker both
2033
+ run exactly this, which is what makes the preview honest about what the
2034
+ apply will do (§5.7).
2035
+
2036
+ NOTHING INSIDE THE EXCLUSIVE HOLD MAY ACQUIRE AN EARLIER LOCK. The walk,
2037
+ the classification backfill and the planner are all filesystem-and-memory
2038
+ only; a `cache.db.lock` acquisition anywhere below closes a real cycle
2039
+ against `db rederive --yes`, and
2040
+ `tests/test_artifact_retention_lock_order.py` scans this module for
2041
+ exactly that.
2042
+ """
2043
+ root = _reclaim_root(root)
2044
+ resolution = (
2045
+ _kernel.PolicyResolution("valid", policy, None) if policy is not None
2046
+ else read_retention_policy()
2047
+ )
2048
+ if resolution.status == "malformed":
2049
+ # §6.5: automatic admission skips deletion ENTIRELY. Nothing is marked,
2050
+ # nothing is deleted, and the reason reaches the caller.
2051
+ return SweepResult(
2052
+ SWEEP_POLICY_MALFORMED, None, resolution.reason, None, None, None,
2053
+ tuple(list_stuck_reclaim_records(root=root)), False,
2054
+ )
2055
+ resolved = resolution.policy
2056
+ if apply and would_block_prod_retention(root):
2057
+ return SweepResult(
2058
+ SWEEP_PROD_REFUSED, resolved,
2059
+ "a development checkout will not prune the production data "
2060
+ "directory; set CCTALLY_ALLOW_PROD_MIGRATION=1 to override",
2061
+ None, None, None, tuple(list_stuck_reclaim_records(root=root)),
2062
+ False,
2063
+ )
2064
+
2065
+ if not apply:
2066
+ scan, plan = plan_retention(
2067
+ policy=resolved, root=root, include_backups=include_backups,
2068
+ now_epoch=now_epoch,
2069
+ )
2070
+ return SweepResult(
2071
+ SWEEP_PARTIAL if scan.partial else SWEEP_OK, resolved, None,
2072
+ scan, plan, None,
2073
+ tuple(list_stuck_reclaim_records(root=root)), False,
2074
+ )
2075
+
2076
+ records: "list[pathlib.Path]" = []
2077
+ errors: "dict[str, str]" = {}
2078
+ scan = None
2079
+ plan = None
2080
+ marked: "tuple[str, ...]" = ()
2081
+ failed: "tuple[str, ...]" = ()
2082
+ skipped: "tuple[str, ...]" = ()
2083
+ plan_ids: "list[str]" = []
2084
+ with retention_exclusive(timeout=timeout, label="artifact retention") as held:
2085
+ if not held:
2086
+ return SweepResult(
2087
+ SWEEP_BLOCKED, resolved,
2088
+ "artifact-retention.lock is held; nothing was marked",
2089
+ None, None, ReclaimOutcome(False, (), (), (), (), (), {}),
2090
+ tuple(list_stuck_reclaim_records(root=root)), False,
2091
+ )
2092
+ resumed, resume_errors = _resume_marking_pass(root)
2093
+ records.extend(resumed)
2094
+ errors.update(resume_errors)
2095
+ scan, plan = plan_retention(
2096
+ policy=resolved, root=root, include_backups=include_backups,
2097
+ now_epoch=now_epoch,
2098
+ )
2099
+ if backfill:
2100
+ # AFTER planning, not before: the walk already folds in the verdict
2101
+ # this would write, so a second walk would only re-derive the same
2102
+ # plan. `targets_for_plan` re-stats every member below, which is
2103
+ # what absorbs the incident directory's moved inode.
2104
+ _backfill_scan_classifications(root, scan)
2105
+ result = mark_reclaim_plan(
2106
+ targets_for_plan(plan, root=root), root=root,
2107
+ )
2108
+ marked, failed, skipped = (
2109
+ result.marked_ids, result.failed_roots, result.skipped_ids,
2110
+ )
2111
+ errors.update(result.reasons)
2112
+ plan_ids.append(result.plan_id)
2113
+ if result.record_path is not None:
2114
+ records.append(result.record_path)
2115
+
2116
+ deleted: "list[str]" = []
2117
+ for path in records:
2118
+ applied, apply_errors = _apply_reclaim_record(path, root=root)
2119
+ deleted.extend(applied)
2120
+ errors.update(apply_errors)
2121
+ outcome = ReclaimOutcome(
2122
+ True, tuple(plan_ids), marked, failed, skipped, tuple(deleted), errors,
2123
+ )
2124
+ return SweepResult(
2125
+ SWEEP_PARTIAL if scan.partial else SWEEP_OK, resolved, None,
2126
+ scan, plan, outcome,
2127
+ tuple(list_stuck_reclaim_records(root=root)), True,
2128
+ )
2129
+
2130
+
2131
+ # --------------------------------------------------------------------------
2132
+ # §5.1 / §5.2 — the hidden detached worker and its admission
2133
+ # --------------------------------------------------------------------------
2134
+ #
2135
+ # The shipped `_stats-corruption-heal` shape, for the same reason it was
2136
+ # adopted there: a non-blocking admission flock decides exactly one request, the
2137
+ # marker is made durable BEFORE the spawn so a crash between the two leaves a
2138
+ # retryable record rather than a lost one, and a worker-active probe stops a
2139
+ # second process being launched only to lose the worker flock and exit.
2140
+
2141
+ #: The hidden subcommand `_spawn_detached` launches.
2142
+ ARTIFACT_RETENTION_COMMAND = "_artifact-retention"
2143
+
2144
+ #: One automatic sweep a day. Recovery of a pending plan is NOT subject to it:
2145
+ #: a crashed deletion must not wait 24 hours to finish.
2146
+ RETENTION_SWEEP_INTERVAL_S = 86400.0
2147
+
2148
+ #: A marker younger than this coalesces rather than admitting a second worker,
2149
+ #: exactly as the heal admission does.
2150
+ RETENTION_REQUEST_RETRY_S = 60.0
2151
+
2152
+ RETENTION_MODE_NEW_PLAN = "new-plan"
2153
+ RETENTION_MODE_RECOVERY = "recovery"
2154
+
2155
+
2156
+ def _retention_path(name: str) -> pathlib.Path:
2157
+ return pathlib.Path(_cctally_core.APP_DIR) / name
2158
+
2159
+
2160
+ def _retention_request_path() -> pathlib.Path:
2161
+ return _retention_path("artifact-retention.pending")
2162
+
2163
+
2164
+ def _retention_admission_path() -> pathlib.Path:
2165
+ return _retention_path("artifact-retention.admission.lock")
2166
+
2167
+
2168
+ def _retention_worker_path() -> pathlib.Path:
2169
+ return _retention_path("artifact-retention.worker.lock")
2170
+
2171
+
2172
+ def _retention_stamp_path() -> pathlib.Path:
2173
+ return _retention_path("artifact-retention.last-sweep")
2174
+
2175
+
2176
+ def _retention_log_path() -> pathlib.Path:
2177
+ return pathlib.Path(_cctally_core.LOG_DIR) / "artifact-retention.log"
2178
+
2179
+
2180
+ def retention_rate_limited(*, now=None) -> bool:
2181
+ """Whether an automatic sweep already ran inside the daily window.
2182
+
2183
+ The stamp is written when a sweep is ADMITTED, not when one succeeds. A
2184
+ throttle stamped only on success re-spawns a failing worker on every
2185
+ command, which is the failure mode the telemetry beat already learned.
2186
+ """
2187
+ try:
2188
+ age = (time.time() if now is None else float(now)) - (
2189
+ _retention_stamp_path().stat().st_mtime
2190
+ )
2191
+ except OSError:
2192
+ return False
2193
+ return age < RETENTION_SWEEP_INTERVAL_S
2194
+
2195
+
2196
+ def pending_reclaim_plan_present(*, root=None) -> bool:
2197
+ """Whether an interrupted worker left a plan to finish."""
2198
+ root = _reclaim_root(root)
2199
+ try:
2200
+ return any(root.glob(f"{RECLAIM_RECORD_PREFIX}*.json"))
2201
+ except OSError:
2202
+ return False
2203
+
2204
+
2205
+ def _stamp_retention_sweep() -> None:
2206
+ path = _retention_stamp_path()
2207
+ try:
2208
+ path.parent.mkdir(parents=True, exist_ok=True)
2209
+ fd = os.open(path, os.O_WRONLY | os.O_CREAT, 0o600)
2210
+ except OSError:
2211
+ return
2212
+ try:
2213
+ os.utime(path, None)
2214
+ except OSError:
2215
+ pass
2216
+ finally:
2217
+ os.close(fd)
2218
+
2219
+
2220
+ def _retention_worker_active() -> bool:
2221
+ """Probe the worker flock without waiting or disturbing its owner."""
2222
+ try:
2223
+ fd = os.open(_retention_worker_path(), os.O_WRONLY | os.O_CREAT, 0o600)
2224
+ except OSError:
2225
+ return False
2226
+ try:
2227
+ try:
2228
+ fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
2229
+ except BlockingIOError:
2230
+ return True
2231
+ except OSError:
2232
+ return False
2233
+ try:
2234
+ fcntl.flock(fd, fcntl.LOCK_UN)
2235
+ except OSError:
2236
+ pass
2237
+ return False
2238
+ finally:
2239
+ os.close(fd)
2240
+
2241
+
2242
+ def _read_retention_request() -> "dict | None":
2243
+ try:
2244
+ payload = json.loads(_retention_request_path().read_text(encoding="utf-8"))
2245
+ except FileNotFoundError:
2246
+ return None
2247
+ except (OSError, ValueError):
2248
+ return {}
2249
+ return payload if isinstance(payload, dict) else {}
2250
+
2251
+
2252
+ def _unlink_retention_request() -> None:
2253
+ try:
2254
+ _retention_request_path().unlink()
2255
+ except FileNotFoundError:
2256
+ pass
2257
+
2258
+
2259
+ def reserve_artifact_retention(mode: str) -> str:
2260
+ """Phase 1: decide admission and make the request marker durable.
2261
+
2262
+ Returns `reserved` (the caller MUST spawn), `pending` (coalesced onto a
2263
+ request already filed) or `failed`. Split from the spawn for the same
2264
+ reason `defer_stats_corruption_heal` is: the marker has to be durable
2265
+ before a worker can be launched to read it.
2266
+ """
2267
+ try:
2268
+ pathlib.Path(_cctally_core.APP_DIR).mkdir(parents=True, exist_ok=True)
2269
+ admission_fd = os.open(
2270
+ _retention_admission_path(), os.O_WRONLY | os.O_CREAT, 0o600
2271
+ )
2272
+ except OSError:
2273
+ return "failed"
2274
+ try:
2275
+ try:
2276
+ fcntl.flock(admission_fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
2277
+ except OSError:
2278
+ return "pending"
2279
+ marker = _retention_request_path()
2280
+ try:
2281
+ age = time.time() - marker.stat().st_mtime
2282
+ except FileNotFoundError:
2283
+ age = None
2284
+ except OSError:
2285
+ return "failed"
2286
+ if age is not None and age < RETENTION_REQUEST_RETRY_S:
2287
+ return "pending"
2288
+ if _retention_worker_active():
2289
+ try:
2290
+ os.utime(marker, None)
2291
+ except OSError:
2292
+ pass
2293
+ return "pending"
2294
+ request = {
2295
+ "schemaVersion": 1,
2296
+ "mode": mode,
2297
+ "requestedAtUtc": _utc_now_iso(),
2298
+ }
2299
+ try:
2300
+ _atomic_write_private(marker, request)
2301
+ except OSError:
2302
+ return "failed"
2303
+ if mode == RETENTION_MODE_NEW_PLAN:
2304
+ # Stamped on the ADMISSION, so a worker that keeps failing does not
2305
+ # get re-admitted by every subsequent command.
2306
+ _stamp_retention_sweep()
2307
+ return "reserved"
2308
+ finally:
2309
+ try:
2310
+ fcntl.flock(admission_fd, fcntl.LOCK_UN)
2311
+ except OSError:
2312
+ pass
2313
+ os.close(admission_fd)
2314
+
2315
+
2316
+ def _unlink_retention_request_under_admission() -> None:
2317
+ try:
2318
+ admission_fd = os.open(
2319
+ _retention_admission_path(), os.O_WRONLY | os.O_CREAT, 0o600
2320
+ )
2321
+ except OSError:
2322
+ return
2323
+ try:
2324
+ try:
2325
+ fcntl.flock(admission_fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
2326
+ except OSError:
2327
+ return
2328
+ _unlink_retention_request()
2329
+ finally:
2330
+ try:
2331
+ fcntl.flock(admission_fd, fcntl.LOCK_UN)
2332
+ except OSError:
2333
+ pass
2334
+ os.close(admission_fd)
2335
+
2336
+
2337
+ def complete_artifact_retention(reservation: str) -> str:
2338
+ """Phase 2: spawn the worker a reservation admitted. Takes no lock."""
2339
+ if reservation != "reserved":
2340
+ return reservation
2341
+ from _cctally_update import _spawn_detached
2342
+
2343
+ if _spawn_detached(ARTIFACT_RETENTION_COMMAND):
2344
+ return "spawned"
2345
+ # Drop our own marker so the next eligible command is admitted immediately
2346
+ # rather than waiting out the retry window for a worker never launched.
2347
+ _unlink_retention_request_under_admission()
2348
+ return "failed"
2349
+
2350
+
2351
+ def defer_artifact_retention(mode: str = RETENTION_MODE_NEW_PLAN) -> str:
2352
+ """Schedule one detached retention sweep without blocking the caller."""
2353
+ return complete_artifact_retention(reserve_artifact_retention(mode))
2354
+
2355
+
2356
+ def _log_retention(outcome: str, detail: str = "") -> None:
2357
+ """Append one path-safe worker result line.
2358
+
2359
+ Counts and status only. The worker's streams are `/dev/null`, so this is
2360
+ the only channel a background sweep has, and it must not carry a private
2361
+ path into a file a user may paste into a bug report.
2362
+ """
2363
+ try:
2364
+ path = _retention_log_path()
2365
+ path.parent.mkdir(parents=True, exist_ok=True)
2366
+ line = f"{_utc_now_iso()} artifact-retention {outcome}{detail}\n"
2367
+ fd = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_APPEND, 0o600)
2368
+ try:
2369
+ os.write(fd, line.encode("utf-8"))
2370
+ finally:
2371
+ os.close(fd)
2372
+ except Exception:
2373
+ pass
2374
+
2375
+
2376
+ def cmd_artifact_retention_internal(args) -> int:
2377
+ """Hidden detached worker: reclaim retained artifacts exactly once.
2378
+
2379
+ Always returns 0. Failures are logged and stay retryable, exactly like the
2380
+ corruption-heal worker: a background sweep that made a command fail would
2381
+ be worse than one that quietly did nothing this pass.
2382
+ """
2383
+ del args
2384
+ try:
2385
+ pathlib.Path(_cctally_core.APP_DIR).mkdir(parents=True, exist_ok=True)
2386
+ worker_fd = os.open(
2387
+ _retention_worker_path(), os.O_WRONLY | os.O_CREAT, 0o600
2388
+ )
2389
+ except OSError as exc:
2390
+ _log_retention("error", f" error={type(exc).__name__}")
2391
+ return 0
2392
+ try:
2393
+ try:
2394
+ fcntl.flock(worker_fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
2395
+ except OSError:
2396
+ return 0
2397
+ request = _read_retention_request()
2398
+ if not request:
2399
+ _unlink_retention_request()
2400
+ _log_retention("no-request")
2401
+ return 0
2402
+ mode = str(request.get("mode") or RETENTION_MODE_NEW_PLAN)
2403
+ try:
2404
+ if mode == RETENTION_MODE_RECOVERY:
2405
+ outcome = resume_reclaim()
2406
+ _log_retention(
2407
+ "recovered",
2408
+ f" deleted={len(outcome.deleted_ids)} "
2409
+ f"errors={len(outcome.errors)}",
2410
+ )
2411
+ else:
2412
+ result = run_retention_sweep()
2413
+ deleted = (
2414
+ len(result.outcome.deleted_ids)
2415
+ if result.outcome is not None else 0
2416
+ )
2417
+ _log_retention(
2418
+ result.status,
2419
+ f" deleted={deleted} "
2420
+ f"unsatisfied={len(result.plan.unsatisfied_rules) if result.plan else 0}",
2421
+ )
2422
+ except Exception as exc: # noqa: BLE001 — retryable, never fatal
2423
+ _log_retention("error", f" error={type(exc).__name__}")
2424
+ return 0
2425
+ _unlink_retention_request()
2426
+ return 0
2427
+ finally:
2428
+ try:
2429
+ fcntl.flock(worker_fd, fcntl.LOCK_UN)
2430
+ except OSError:
2431
+ pass
2432
+ os.close(worker_fd)
2433
+
2434
+
2435
+ def maybe_defer_artifact_retention(
2436
+ *, command, action=None, exit_code=0, hook_forked=None,
2437
+ hook_explain=False, hook_foreground=False, applied=None, root=None,
2438
+ ) -> str:
2439
+ """Apply §5.2's predicate to one finished invocation and act on it.
2440
+
2441
+ The single call site shape for both the post-command hook in `bin/cctally`
2442
+ and `hook-tick`'s forked child. Best-effort throughout: an admission
2443
+ failure must never perturb the command the user actually ran.
2444
+ """
2445
+ if _cctally_core._truthy_env("CCTALLY_DISABLE_RETENTION_SWEEP"):
2446
+ return ""
2447
+ try:
2448
+ invocation = dict(
2449
+ command=command,
2450
+ action=action,
2451
+ exit_code=exit_code,
2452
+ hook_forked=hook_forked,
2453
+ hook_explain=hook_explain,
2454
+ hook_foreground=hook_foreground,
2455
+ applied=applied,
2456
+ )
2457
+ # The pure rejection FIRST. `retention_rate_limited()` stats a marker
2458
+ # and `pending_reclaim_plan_present()` globs the data directory, and
2459
+ # passing them as arguments made `cctally statusline` — which can never
2460
+ # admit — pay a readdir on every render.
2461
+ if not _kernel.retention_admission_possible(**invocation):
2462
+ return ""
2463
+ decision = _kernel.retention_admission(
2464
+ **invocation,
2465
+ rate_limited=retention_rate_limited(),
2466
+ pending_plan_present=pending_reclaim_plan_present(root=root),
2467
+ )
2468
+ if not decision:
2469
+ return ""
2470
+ return defer_artifact_retention(decision)
2471
+ except Exception: # noqa: BLE001 — never break the parent command
2472
+ return ""
2473
+
2474
+
2475
+ # --------------------------------------------------------------------------
2476
+ # §6 — `cctally db prune`
2477
+ # --------------------------------------------------------------------------
2478
+ #
2479
+ # Preview by default; `--yes` applies. A separate `db` child rather than an
2480
+ # extension of `db vacuum`, whose free-space prerequisite can disable it
2481
+ # precisely when this operation is the thing that would restore the space.
2482
+
2483
+ PRUNE_SCHEMA_VERSION = 1
2484
+
2485
+ #: A protection reason rendered for a person. `unclassified` splits in two,
2486
+ #: because "considered and undecided" and "never looked at" are different
2487
+ #: things for an operator even though the kernel resolves both to no verdict.
2488
+ _PROTECTION_PHRASES = {
2489
+ "unclassified": "no classification recorded",
2490
+ "unclassified:unknown": "classification is unknown",
2491
+ "invalid": "the artifact does not match its own manifest",
2492
+ "unfinished": "the quarantine that wrote it did not finish",
2493
+ "active": "a durable marker still names it",
2494
+ "symlink": "it is a symlink",
2495
+ "outside-root": "it resolves outside the data directory",
2496
+ "missing": "it is no longer on disk",
2497
+ "dangling-reference": "it references something that is missing",
2498
+ "unrecognized-kind": "cctally did not write it",
2499
+ "unreferenced-evidence": "no bundle or incident references this evidence",
2500
+ }
2501
+
2502
+ _KIND_LABELS = {
2503
+ "incident": "{family} incidents",
2504
+ "bundle": "{family} forensics bundles",
2505
+ "wal_evidence": "{family} WAL evidence",
2506
+ "rebuild_record": "rebuild records",
2507
+ "unknown": "unrecognized artifacts",
2508
+ }
2509
+
2510
+ _BACKUP_LABELS = {
2511
+ "machine": "machine backups",
2512
+ "user": "user backups",
2513
+ "unknown": "unrecognized backups",
2514
+ }
2515
+
2516
+
2517
+ def _gib(value: int) -> str:
2518
+ """§6.4's two-decimal figure, adaptive below a GiB.
2519
+
2520
+ `db prune` is the user-facing surface of this whole feature, and a fixed
2521
+ GiB rendering printed `0.00 GiB` in every column on any corpus below about
2522
+ 50 MiB — the exact defect the shared formatter exists for.
2523
+ """
2524
+ return _kernel.format_disk_bytes(value, digits=2)
2525
+
2526
+
2527
+ def _row_label(root) -> str:
2528
+ if root.kind == "backup":
2529
+ return _BACKUP_LABELS[_kernel.backup_origin(root.id)]
2530
+ template = _KIND_LABELS.get(root.kind, "{family} artifacts")
2531
+ return template.format(family=root.family)
2532
+
2533
+
2534
+ def _protection_phrase(root, detail) -> str:
2535
+ phrases = []
2536
+ for reason in root.protected_reasons:
2537
+ key = reason
2538
+ if reason == "unclassified" and detail.get(root.id) == "unknown":
2539
+ key = "unclassified:unknown"
2540
+ phrases.append(_PROTECTION_PHRASES.get(key, reason))
2541
+ return ", ".join(phrases) or "protected"
2542
+
2543
+
2544
+ def summarize_prune(scan: RetentionScan, plan, *, graph=None) -> dict:
2545
+ """Fold the plan into the rows and reason counts §6.4 renders.
2546
+
2547
+ Every member is attributed to exactly ONE row — the row of the lowest root
2548
+ that reaches it — so the columns partition the corpus rather than double
2549
+ counting a member two roots share.
2550
+
2551
+ `graph` is accepted so a caller that already built it from the same scan
2552
+ does not pay for a second identical construction; it is rebuilt only when
2553
+ none is supplied.
2554
+ """
2555
+ graph = _kernel.build_graph(scan.members) if graph is None else graph
2556
+ deleted = set(plan.delete_ids)
2557
+ protected_ids = set(plan.protected_ids)
2558
+ detail = scan.classification_detail
2559
+
2560
+ owner: "dict[str, str]" = {}
2561
+ for member_id in graph.members:
2562
+ inbound = graph.inbound_roots.get(member_id) or frozenset({member_id})
2563
+ owner[member_id] = min(inbound)
2564
+
2565
+ rows: "dict[str, dict]" = {}
2566
+ for root in graph.roots:
2567
+ row = rows.setdefault(_row_label(root), {
2568
+ "label": _row_label(root), "roots": 0, "disk": 0, "delete": 0,
2569
+ "keep": 0, "protected": 0,
2570
+ })
2571
+ row["roots"] += 1
2572
+ for member_id, member in graph.members.items():
2573
+ root_id = owner[member_id]
2574
+ root = graph.roots_by_id.get(root_id)
2575
+ if root is None:
2576
+ continue
2577
+ row = rows[_row_label(root)]
2578
+ row["disk"] += member.disk_bytes
2579
+ if member_id in deleted:
2580
+ row["delete"] += member.disk_bytes
2581
+ elif root_id in protected_ids:
2582
+ row["protected"] += member.disk_bytes
2583
+ else:
2584
+ row["keep"] += member.disk_bytes
2585
+
2586
+ # Bytes per owning root, folded ONCE. Re-scanning `owner` per protected
2587
+ # root made this loop cost roots x members, which on a corpus with many
2588
+ # protected incidents is the same quadratic the planner had.
2589
+ owned_bytes: "dict[str, int]" = {}
2590
+ for member_id, owner_id in owner.items():
2591
+ owned_bytes[owner_id] = (
2592
+ owned_bytes.get(owner_id, 0) + graph.members[member_id].disk_bytes
2593
+ )
2594
+
2595
+ reasons: "dict[str, int]" = {}
2596
+ reason_bytes: "dict[str, int]" = {}
2597
+ for root_id in plan.protected_ids:
2598
+ root = graph.roots_by_id[root_id]
2599
+ phrase = _protection_phrase(root, detail)
2600
+ reasons[phrase] = reasons.get(phrase, 0) + 1
2601
+ reason_bytes[phrase] = (
2602
+ reason_bytes.get(phrase, 0) + owned_bytes.get(root_id, 0)
2603
+ )
2604
+
2605
+ protected_bytes = sum(row["protected"] for row in rows.values())
2606
+ return {
2607
+ "rows": sorted(rows.values(), key=lambda row: row["label"]),
2608
+ "reasons": sorted(
2609
+ (
2610
+ {"reason": phrase, "roots": count,
2611
+ "diskBytes": reason_bytes[phrase]}
2612
+ for phrase, count in reasons.items()
2613
+ ),
2614
+ key=lambda item: (-item["roots"], item["reason"]),
2615
+ ),
2616
+ "protectedRoots": len(plan.protected_ids),
2617
+ "protectedBytes": protected_bytes,
2618
+ "roots": len(graph.roots),
2619
+ "members": len(graph.members),
2620
+ }
2621
+
2622
+
2623
+ def _policy_sentence(policy) -> str:
2624
+ clauses = []
2625
+ if policy.max_age_seconds is not None:
2626
+ clauses.append(f"keep {policy.max_age_seconds // 86400} days")
2627
+ if policy.max_count_per_family is not None:
2628
+ clauses.append(f"{policy.max_count_per_family} per family")
2629
+ if policy.max_total_bytes is not None:
2630
+ clauses.append(f"{policy.max_total_bytes // (1024 ** 2)} MiB total")
2631
+ head = ", ".join(clauses) or "no size rule enabled"
2632
+ tail = ""
2633
+ if policy.min_free_bytes is not None:
2634
+ tail = (
2635
+ f"; reclaim below {policy.min_free_bytes // (1024 ** 2)} MiB free"
2636
+ )
2637
+ return (
2638
+ f"Policy: {head}{tail};\n"
2639
+ f" keep {policy.max_shape_examples} damage-shape examples."
2640
+ )
2641
+
2642
+
2643
+ def render_prune_text(result: SweepResult) -> str:
2644
+ """§6.4's report. Shape normative, figures per run."""
2645
+ plan = result.plan
2646
+ summary = summarize_prune(result.scan, plan)
2647
+ lines = [_policy_sentence(result.policy), ""]
2648
+ header = (
2649
+ f"{'':<24}{'groups':>8}{'on disk':>13}{'delete':>13}"
2650
+ f"{'keep':>13}{'protected':>13}"
2651
+ )
2652
+ lines.append(header)
2653
+ for row in summary["rows"]:
2654
+ lines.append(
2655
+ f"{row['label']:<24}{row['roots']:>8}{_gib(row['disk']):>13}"
2656
+ f"{_gib(row['delete']):>13}{_gib(row['keep']):>13}"
2657
+ + (
2658
+ f"{_gib(row['protected']):>13}" if row["protected"]
2659
+ else f"{'--':>13}"
2660
+ )
2661
+ )
2662
+ if not summary["rows"]:
2663
+ lines.append("(no retained artifacts)")
2664
+ if summary["reasons"]:
2665
+ lines.append("")
2666
+ lines.append(
2667
+ f"Protected and never deleted ({summary['protectedRoots']} groups, "
2668
+ f"{_gib(summary['protectedBytes'])}):"
2669
+ )
2670
+ for item in summary["reasons"]:
2671
+ lines.append(f" {item['roots']} {item['reason']}")
2672
+ if plan.floor_retained_ids:
2673
+ lines.append("")
2674
+ lines.append(
2675
+ f"Keeping {len(plan.floor_retained_ids)} damage-shape example"
2676
+ f"{'' if len(plan.floor_retained_ids) == 1 else 's'} that age and "
2677
+ "count would otherwise have removed."
2678
+ )
2679
+ if result.scan.partial:
2680
+ lines.append("")
2681
+ lines.append(
2682
+ "The scan stopped at its entry cap, so these figures cover only "
2683
+ "part of the retained corpus."
2684
+ )
2685
+ for record in result.stuck:
2686
+ if record.get("stuck"):
2687
+ lines.append("")
2688
+ lines.append(
2689
+ f"Reclaim plan {record['planId']} has been stuck for over a "
2690
+ f"day on: {', '.join(sorted(record['entries']))}. No pass can "
2691
+ "decide it — inspect the named member, then remove "
2692
+ f"{pathlib.Path(record['path']).name}."
2693
+ )
2694
+ lines.append("")
2695
+ free_after = (
2696
+ None if result.scan.free_disk_bytes is None
2697
+ else result.scan.free_disk_bytes + plan.reclaimable_bytes
2698
+ )
2699
+ remaining = f", leaving {_gib(plan.projected_bytes)} retained"
2700
+ if result.applied:
2701
+ deleted_ids = set(result.outcome.deleted_ids)
2702
+ freed = sum(
2703
+ member.disk_bytes for member in result.scan.members
2704
+ if member.id in deleted_ids
2705
+ )
2706
+ lines.append(
2707
+ f"Freed {_gib(freed)}{remaining}."
2708
+ + (f" {_gib(free_after)} free on disk." if free_after else "")
2709
+ )
2710
+ if result.outcome.failed_roots or result.outcome.errors:
2711
+ lines.append(
2712
+ f"{len(result.outcome.failed_roots)} group(s) could not be "
2713
+ f"reclaimed; {len(result.outcome.errors)} member(s) reported a "
2714
+ "reason."
2715
+ )
2716
+ else:
2717
+ lines.append(
2718
+ f"Would free {_gib(plan.reclaimable_bytes)}{remaining}. "
2719
+ "Nothing was deleted — re-run with --yes."
2720
+ )
2721
+ if plan.unsatisfied_rules:
2722
+ lines.append(
2723
+ "Protected evidence holds the corpus over: "
2724
+ + ", ".join(plan.unsatisfied_rules)
2725
+ + "."
2726
+ )
2727
+ return "\n".join(lines)
2728
+
2729
+
2730
+ def _prune_status(result: SweepResult) -> str:
2731
+ if result.plan is not None and result.plan.unsatisfied_rules:
2732
+ return "blocked"
2733
+ if result.applied:
2734
+ outcome = result.outcome
2735
+ if outcome.failed_roots or outcome.errors:
2736
+ return "partial"
2737
+ return "applied" if outcome.deleted_ids else "no-op"
2738
+ return "preview"
2739
+
2740
+
2741
+ def _prune_exit_code(result: SweepResult) -> int:
2742
+ """§6.2, and the asymmetry in it is deliberate.
2743
+
2744
+ A PREVIEW that reports a blocked bound still exits 0: it deleted nothing
2745
+ and nothing failed, so there is no staged failure to report. The same
2746
+ condition on an APPLY exits 3, because the apply is the operation that was
2747
+ supposed to resolve it.
2748
+ """
2749
+ if result.status == SWEEP_POLICY_MALFORMED:
2750
+ return 2
2751
+ if result.status == SWEEP_PROD_REFUSED:
2752
+ return 2
2753
+ if not result.applied and result.status != SWEEP_BLOCKED:
2754
+ return 0
2755
+ if result.status == SWEEP_BLOCKED:
2756
+ return 3
2757
+ outcome = result.outcome
2758
+ if outcome.failed_roots or outcome.errors:
2759
+ return 3
2760
+ if result.plan is not None and result.plan.unsatisfied_rules:
2761
+ return 3
2762
+ return 0
2763
+
2764
+
2765
+ def prune_payload(result: SweepResult) -> dict:
2766
+ """§6.3's envelope body, stamped by the caller. camelCase throughout, and
2767
+ every artifact named by its stable relative id — never an absolute path."""
2768
+ payload: "dict[str, object]" = {"status": _prune_status(result)}
2769
+ if result.status == SWEEP_POLICY_MALFORMED:
2770
+ payload["status"] = "malformedPolicy"
2771
+ elif result.status == SWEEP_PROD_REFUSED:
2772
+ payload["status"] = "prodRefused"
2773
+ elif result.status == SWEEP_BLOCKED:
2774
+ payload["status"] = "blocked"
2775
+ payload["policy"] = (
2776
+ None if result.policy is None
2777
+ else {
2778
+ "maxAgeDays": (
2779
+ None if result.policy.max_age_seconds is None
2780
+ else result.policy.max_age_seconds // 86400
2781
+ ),
2782
+ "maxCountPerFamily": result.policy.max_count_per_family,
2783
+ "maxTotalMib": (
2784
+ None if result.policy.max_total_bytes is None
2785
+ else result.policy.max_total_bytes // (1024 ** 2)
2786
+ ),
2787
+ "minFreeMib": (
2788
+ None if result.policy.min_free_bytes is None
2789
+ else result.policy.min_free_bytes // (1024 ** 2)
2790
+ ),
2791
+ "maxShapeExamples": result.policy.max_shape_examples,
2792
+ }
2793
+ )
2794
+ summary = (
2795
+ summarize_prune(result.scan, result.plan)
2796
+ if result.scan is not None and result.plan is not None else None
2797
+ )
2798
+ payload["before"] = None if summary is None else {
2799
+ "roots": summary["roots"],
2800
+ "members": summary["members"],
2801
+ "diskBytes": result.plan.before_bytes,
2802
+ "freeDiskBytes": result.scan.free_disk_bytes,
2803
+ "entriesScanned": result.scan.entries_seen,
2804
+ "partialScan": result.scan.partial,
2805
+ }
2806
+ payload["plan"] = None if result.plan is None else {
2807
+ "deleteIds": list(result.plan.delete_ids),
2808
+ "keepIds": list(result.plan.keep_ids),
2809
+ "reclaimableBytes": result.plan.reclaimable_bytes,
2810
+ "projectedBytes": result.plan.projected_bytes,
2811
+ "referencePinnedBytes": result.plan.reference_pinned_bytes,
2812
+ "floorRetainedIds": list(result.plan.floor_retained_ids),
2813
+ "floorRetainedBytes": result.plan.floor_retained_bytes,
2814
+ "groups": [] if summary is None else [
2815
+ {
2816
+ "label": row["label"], "roots": row["roots"],
2817
+ "diskBytes": row["disk"], "deleteBytes": row["delete"],
2818
+ "keepBytes": row["keep"], "protectedBytes": row["protected"],
2819
+ }
2820
+ for row in summary["rows"]
2821
+ ],
2822
+ }
2823
+ payload["protected"] = None if summary is None else {
2824
+ "roots": summary["protectedRoots"],
2825
+ "diskBytes": summary["protectedBytes"],
2826
+ "ids": list(result.plan.protected_ids),
2827
+ "reasons": summary["reasons"],
2828
+ }
2829
+ payload["result"] = {
2830
+ "applied": bool(result.applied),
2831
+ "deletedIds": list(result.outcome.deleted_ids) if result.outcome else [],
2832
+ "markedIds": list(result.outcome.marked_ids) if result.outcome else [],
2833
+ "skippedIds": list(result.outcome.skipped_ids) if result.outcome else [],
2834
+ "failedRoots": list(result.outcome.failed_roots) if result.outcome else [],
2835
+ }
2836
+ payload["unsatisfiedRules"] = (
2837
+ [] if result.plan is None else list(result.plan.unsatisfied_rules)
2838
+ )
2839
+ errors = []
2840
+ if result.reason:
2841
+ errors.append({"id": None, "reason": result.reason})
2842
+ if result.outcome is not None:
2843
+ errors.extend(
2844
+ {"id": member_id, "reason": reason}
2845
+ for member_id, reason in sorted(result.outcome.errors.items())
2846
+ )
2847
+ payload["errors"] = errors
2848
+ payload["stuckRecords"] = [
2849
+ {
2850
+ "planId": record["planId"],
2851
+ "memberIds": sorted(record["entries"]),
2852
+ "stuck": bool(record.get("stuck")),
2853
+ }
2854
+ for record in result.stuck
2855
+ ]
2856
+ return payload
2857
+
2858
+
2859
+ def cmd_db_prune(args) -> int:
2860
+ """`cctally db prune` — preview by default, `--yes` applies (§6)."""
2861
+ from _lib_json_envelope import stamp_schema_version
2862
+
2863
+ as_json = bool(getattr(args, "json", False))
2864
+ apply = bool(getattr(args, "yes", False))
2865
+ result = run_retention_sweep(
2866
+ include_backups=bool(getattr(args, "include_backups", False)),
2867
+ apply=apply,
2868
+ )
2869
+ code = _prune_exit_code(result)
2870
+ if as_json:
2871
+ print(json.dumps(stamp_schema_version(
2872
+ prune_payload(result), version=PRUNE_SCHEMA_VERSION,
2873
+ )))
2874
+ return code
2875
+ if result.status == SWEEP_POLICY_MALFORMED:
2876
+ print(
2877
+ f"cctally: the retention policy in config.json is malformed "
2878
+ f"({result.reason}). Automatic reclamation is off and nothing was "
2879
+ "deleted; fix or remove the storage.artifact_retention block.",
2880
+ file=sys.stderr,
2881
+ )
2882
+ return code
2883
+ if result.status == SWEEP_PROD_REFUSED:
2884
+ print(f"cctally: {result.reason}.", file=sys.stderr)
2885
+ return code
2886
+ if result.status == SWEEP_BLOCKED:
2887
+ print(f"cctally: {result.reason}.", file=sys.stderr)
2888
+ return code
2889
+ print(render_prune_text(result))
2890
+ return code