cctally 1.93.1 → 1.94.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1645 @@
1
+ """Pure kernel for retained-artifact retention (#496 S6, spec §3 / §4 / §6.5).
2
+
3
+ This module takes **no filesystem, no locks, no clock and no config**. Every
4
+ decision it makes is a function of the values it is handed. All I/O — the
5
+ metadata walk, JSON parsing, `disk_usage`, config reading, locking, tombstone
6
+ renames, deletion — lives in `bin/_cctally_retention.py` and the producer glue.
7
+
8
+ `resolve_retention_policy` (§6.5) is the **strict** resolver over the raw
9
+ `storage.artifact_retention` block. It never falls back to defaults on
10
+ malformed input, because a policy the user never wrote must not arm deletion;
11
+ `load_config()` cannot be used for this reason (spec C14).
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import dataclasses
16
+ import re
17
+
18
+ # Public surface: shipped in the npm tarball + brew formula + public mirror.
19
+
20
+ _SECONDS_PER_DAY = 86400
21
+ _BYTES_PER_MIB = 1024 * 1024
22
+
23
+
24
+ @dataclasses.dataclass(frozen=True)
25
+ class RetentionPolicy:
26
+ """A resolved, validated retention policy in canonical units.
27
+
28
+ `None` on any of the first four fields means that rule is disabled.
29
+ `max_shape_examples` is never `None` — the shape floor (§3.4) is binding
30
+ by maintainer decision Q5.
31
+ """
32
+
33
+ max_age_seconds: "int | None"
34
+ max_count_per_family: "int | None"
35
+ max_total_bytes: "int | None"
36
+ min_free_bytes: "int | None"
37
+ max_shape_examples: int
38
+
39
+
40
+ @dataclasses.dataclass(frozen=True)
41
+ class PolicyResolution:
42
+ """The outcome of reading the persisted policy.
43
+
44
+ `status` is one of ``missing`` (no block written — use `DEFAULT_POLICY`),
45
+ ``valid`` (a policy the user wrote) or ``malformed`` (refuse to act; §6.5
46
+ requires doctor to FAIL and both `db prune` modes to exit 2).
47
+ """
48
+
49
+ status: str
50
+ policy: "RetentionPolicy | None"
51
+ reason: "str | None"
52
+
53
+
54
+ #: Maintainer decision Q8: 30 days / 20 per family / 4096 MiB total /
55
+ #: 10240 MiB free floor, keeping 8 damage-shape examples. The free floor is
56
+ #: measured rather than chosen — `db vacuum` needs roughly
57
+ #: ``2 * db_bytes + wal_bytes`` free and `conversations.db` is 5.15 GB.
58
+ DEFAULT_POLICY = RetentionPolicy(
59
+ max_age_seconds=30 * _SECONDS_PER_DAY,
60
+ max_count_per_family=20,
61
+ max_total_bytes=4096 * _BYTES_PER_MIB,
62
+ min_free_bytes=10240 * _BYTES_PER_MIB,
63
+ max_shape_examples=8,
64
+ )
65
+
66
+ #: field name -> (floor, nullable, unit multiplier, policy attribute)
67
+ _POLICY_FIELDS: "dict[str, tuple[int, bool, int, str]]" = {
68
+ "max_age_days": (1, True, _SECONDS_PER_DAY, "max_age_seconds"),
69
+ "max_count_per_family": (1, True, 1, "max_count_per_family"),
70
+ "max_total_mib": (1, True, _BYTES_PER_MIB, "max_total_bytes"),
71
+ "min_free_mib": (0, True, _BYTES_PER_MIB, "min_free_bytes"),
72
+ "max_shape_examples": (1, False, 1, "max_shape_examples"),
73
+ }
74
+
75
+ #: At least one of these must stay enabled, or nothing bounds growth.
76
+ #: `min_free_mib` is deliberately not one of them: it is a floor that reacts
77
+ #: to disk pressure, not a bound on what this subsystem retains.
78
+ _SIZE_RULE_FIELDS = ("max_age_days", "max_count_per_family", "max_total_mib")
79
+
80
+
81
+ def default_policy_block() -> "dict[str, int]":
82
+ """The default policy rendered in the config file's own units.
83
+
84
+ Used by `config get` so an operator sees the effective policy in the same
85
+ shape `config set` accepts.
86
+ """
87
+ return {
88
+ "max_age_days": DEFAULT_POLICY.max_age_seconds // _SECONDS_PER_DAY,
89
+ "max_count_per_family": DEFAULT_POLICY.max_count_per_family,
90
+ "max_total_mib": DEFAULT_POLICY.max_total_bytes // _BYTES_PER_MIB,
91
+ "min_free_mib": DEFAULT_POLICY.min_free_bytes // _BYTES_PER_MIB,
92
+ "max_shape_examples": DEFAULT_POLICY.max_shape_examples,
93
+ }
94
+
95
+
96
+ def policy_to_block(policy: RetentionPolicy) -> "dict[str, int | None]":
97
+ """Render a resolved policy back into the config file's units."""
98
+ return {
99
+ "max_age_days": (
100
+ None if policy.max_age_seconds is None
101
+ else policy.max_age_seconds // _SECONDS_PER_DAY
102
+ ),
103
+ "max_count_per_family": policy.max_count_per_family,
104
+ "max_total_mib": (
105
+ None if policy.max_total_bytes is None
106
+ else policy.max_total_bytes // _BYTES_PER_MIB
107
+ ),
108
+ "min_free_mib": (
109
+ None if policy.min_free_bytes is None
110
+ else policy.min_free_bytes // _BYTES_PER_MIB
111
+ ),
112
+ "max_shape_examples": policy.max_shape_examples,
113
+ }
114
+
115
+
116
+ def resolve_retention_policy(raw: object) -> PolicyResolution:
117
+ """Resolve the raw `storage.artifact_retention` block, strictly (§6.5).
118
+
119
+ `raw` is whatever a raw read of `config.json` produced for that key —
120
+ `None` when the block is absent. This function never repairs input: an
121
+ unknown field, a boolean, a non-integer, a value below its floor, a null
122
+ on the non-nullable `max_shape_examples`, or disabling every size rule all
123
+ return ``malformed`` with a reason. Only ``missing`` and ``valid`` carry a
124
+ policy a caller may act on.
125
+ """
126
+ if raw is None:
127
+ return PolicyResolution("missing", DEFAULT_POLICY, None)
128
+ if isinstance(raw, dict) and not raw:
129
+ return PolicyResolution("missing", DEFAULT_POLICY, None)
130
+ if not isinstance(raw, dict):
131
+ return PolicyResolution(
132
+ "malformed",
133
+ None,
134
+ "storage.artifact_retention must be a JSON object, got "
135
+ f"{type(raw).__name__}",
136
+ )
137
+
138
+ unknown = sorted(set(raw) - set(_POLICY_FIELDS))
139
+ if unknown:
140
+ return PolicyResolution(
141
+ "malformed",
142
+ None,
143
+ "storage.artifact_retention has unknown field(s): "
144
+ + ", ".join(unknown),
145
+ )
146
+
147
+ resolved: "dict[str, int | None]" = {}
148
+ for field, (floor, nullable, unit, attribute) in _POLICY_FIELDS.items():
149
+ if field not in raw:
150
+ resolved[attribute] = getattr(DEFAULT_POLICY, attribute)
151
+ continue
152
+ value = raw[field]
153
+ if value is None:
154
+ if not nullable:
155
+ return PolicyResolution(
156
+ "malformed",
157
+ None,
158
+ f"storage.artifact_retention.{field} may not be null "
159
+ "(the damage-shape floor is always in force)",
160
+ )
161
+ resolved[attribute] = None
162
+ continue
163
+ # isinstance(True, int) is True in Python: check bool FIRST, or
164
+ # `max_count_per_family: true` would silently resolve to 1.
165
+ if isinstance(value, bool) or not isinstance(value, int):
166
+ return PolicyResolution(
167
+ "malformed",
168
+ None,
169
+ f"storage.artifact_retention.{field} must be an integer "
170
+ f"or null, got {value!r}",
171
+ )
172
+ if value < floor:
173
+ return PolicyResolution(
174
+ "malformed",
175
+ None,
176
+ f"storage.artifact_retention.{field} must be >= {floor}, "
177
+ f"got {value}",
178
+ )
179
+ resolved[attribute] = value * unit
180
+
181
+ if all(
182
+ resolved[_POLICY_FIELDS[field][3]] is None for field in _SIZE_RULE_FIELDS
183
+ ):
184
+ return PolicyResolution(
185
+ "malformed",
186
+ None,
187
+ "storage.artifact_retention must keep at least one of "
188
+ + ", ".join(_SIZE_RULE_FIELDS)
189
+ + " enabled",
190
+ )
191
+
192
+ return PolicyResolution("valid", RetentionPolicy(**resolved), None)
193
+
194
+
195
+ # --------------------------------------------------------------------------
196
+ # §3.1 / §3.2 — the reference graph and the absolute protection gate
197
+ # --------------------------------------------------------------------------
198
+
199
+ #: The artifact kinds this subsystem recognizes. Anything else is protected
200
+ #: rather than swept: §3.2 names "outside the recognized artifact set" as a
201
+ #: protection condition, and a kind nobody wrote a validator for is exactly
202
+ #: that.
203
+ RETENTION_KINDS = (
204
+ "incident", "bundle", "wal_evidence", "rebuild_record", "backup",
205
+ "backup_member",
206
+ )
207
+
208
+ #: Kinds that are a root whatever references them. An incident is the unit a
209
+ #: policy counts and ages, and a backup stem has no referrer at all. The
210
+ #: remaining kinds are a root only when nothing references them (§3.1: "a
211
+ #: bundle owned by a rebuild record is a member of that record").
212
+ #:
213
+ #: This is a KIND rule and not an in-degree rule, and the difference is load
214
+ #: bearing: a stats incident manifest names its rebuild record and that record
215
+ #: names the incident back (`bin/_cctally_journal.py:7122` and `:7770`), so a
216
+ #: pure "referenced by nobody" test would leave both nodes of that real cycle
217
+ #: unrooted and the whole component permanently invisible to the planner.
218
+ _ROOT_WHATEVER_REFERENCES_IT = frozenset({"incident", "backup"})
219
+
220
+ #: The confidence values a decided verdict never carries. §3.3 states the rule
221
+ #: as "other than `unknown`" rather than as a list of accepted values, so a
222
+ #: confidence this subsystem has not seen before classifies rather than
223
+ #: silently protecting — the classifier and the gate must not disagree about
224
+ #: what "classified" means.
225
+ UNDECIDED_CONFIDENCES = frozenset({"unknown", ""})
226
+
227
+
228
+ def is_classified(confidence) -> bool:
229
+ """Whether a recorded confidence counts as a decision (§3.3).
230
+
231
+ The field is `confidence`. Reading a `verdict` key instead reports every
232
+ incident as unclassified, which happened once during design.
233
+ """
234
+ return isinstance(confidence, str) and confidence not in UNDECIDED_CONFIDENCES
235
+
236
+ #: Every recognized kind must be classified before it can be deleted. The map
237
+ #: is explicit rather than a constant `True` so that adding a kind forces a
238
+ #: decision about it instead of inheriting deletability by default.
239
+ _KIND_REQUIRES_CLASSIFICATION = {
240
+ "incident": True,
241
+ "bundle": True,
242
+ "wal_evidence": True,
243
+ "rebuild_record": True,
244
+ "backup": True,
245
+ # A `-wal`, `-shm` or `.classification.json` sidecar of a backup stem.
246
+ # §3.7 keys a backup family by its STEM, so the sidecars must be MEMBERS
247
+ # of that stem rather than roots of their own — which is why they cannot
248
+ # carry kind `backup`, whose entry in `_ROOT_WHATEVER_REFERENCES_IT` would
249
+ # root each of them separately and let the count bound see one family as
250
+ # three. Their own classification is never read, because §3.2 takes the
251
+ # classification condition from the root.
252
+ "backup_member": True,
253
+ }
254
+
255
+
256
+ @dataclasses.dataclass(frozen=True)
257
+ class RetentionMember:
258
+ """One retained artifact, as the metadata walk observed it.
259
+
260
+ Every field is supplied by the glue. The kernel never re-derives one from
261
+ the filesystem, so a test states a condition by setting a field rather than
262
+ by building a directory.
263
+ """
264
+
265
+ id: str
266
+ kind: str
267
+ family: str
268
+ created_at_epoch: float
269
+ disk_bytes: int
270
+ logical_bytes: int
271
+ references: "tuple[str, ...]"
272
+ is_symlink: bool
273
+ in_root: bool
274
+ exists: bool
275
+ valid: bool
276
+ classification: "str | None"
277
+ shape_token: "str | None"
278
+ finalized: bool
279
+ active: bool
280
+
281
+
282
+ @dataclasses.dataclass(frozen=True)
283
+ class RetentionRoot:
284
+ """A thing a policy counts and ages (§3.1).
285
+
286
+ `reachable_ids` includes the root itself, because the deletion closure
287
+ deletes the root along with the members exclusive to it. There is
288
+ deliberately **no `exclusive_closure_ids` field**: for roots A and B that
289
+ both reference target T, T is exclusive to neither at construction time, so
290
+ exclusivity is a property of a SELECTION and is computed by
291
+ `deletion_closure` against the currently selected set.
292
+ """
293
+
294
+ id: str
295
+ kind: str
296
+ family: str
297
+ created_at_epoch: float
298
+ reachable_ids: "frozenset[str]"
299
+ own_member_ids: "frozenset[str]"
300
+ requires_classification: bool
301
+ classification: "str | None"
302
+ shape_token: "str | None"
303
+ protected_reasons: "tuple[str, ...]"
304
+
305
+
306
+ @dataclasses.dataclass(frozen=True)
307
+ class RetentionGraph:
308
+ """The whole retained corpus: roots, members, and the inverse index."""
309
+
310
+ roots: "tuple[RetentionRoot, ...]"
311
+ members: "dict[str, RetentionMember]"
312
+ inbound_roots: "dict[str, frozenset[str]]"
313
+ roots_by_id: "dict[str, RetentionRoot]"
314
+
315
+
316
+ def _member_protection_reasons(member, known_ids) -> "list[str]":
317
+ """Why this member protects every root that can reach it (§3.2).
318
+
319
+ Protection propagates BACKWARD through the whole reference graph, not
320
+ through a per-root exclusive closure. An invalid SHARED member belongs to
321
+ no root's exclusive closure, so a closure-only gate would let the planner
322
+ delete it once every inbound root happened to be selected.
323
+ """
324
+ reasons: "list[str]" = []
325
+ if member.kind not in _KIND_REQUIRES_CLASSIFICATION:
326
+ reasons.append("unrecognized-kind")
327
+ if not member.exists:
328
+ reasons.append("missing")
329
+ if not member.valid:
330
+ reasons.append("invalid")
331
+ if not member.finalized:
332
+ reasons.append("unfinished")
333
+ if member.active:
334
+ reasons.append("active")
335
+ if member.is_symlink:
336
+ reasons.append("symlink")
337
+ if not member.in_root:
338
+ reasons.append("outside-root")
339
+ if any(ref not in known_ids for ref in member.references):
340
+ reasons.append("dangling-reference")
341
+ return reasons
342
+
343
+
344
+ def _reachable_from(by_id, known_ids, root_ids, member_id) -> "set[str]":
345
+ """Everything `member_id` reaches, **stopping at every other root**.
346
+
347
+ Reachability carries a visited set: the corpus contains a real cycle — an
348
+ incident manifest naming its rebuild record and that record naming the
349
+ incident back — so a walk without one does not terminate on the
350
+ maintainer's own store.
351
+
352
+ Stopping at another root is what makes the root set an ANTICHAIN, and that
353
+ is a correctness requirement rather than an optimization. `deletion_closure`
354
+ admits a member only when every inbound root is selected, so a root reached
355
+ by a second root is excluded from its own singleton closure: it would be
356
+ credited against the count bound, absent from `delete_ids`, `keep_ids` and
357
+ `protected_ids` alike, and the bound reported satisfied over a corpus still
358
+ over budget. With this walk `inbound_roots[root] == {root}` for every root,
359
+ so a selected root is always inside its own closure.
360
+ """
361
+ reachable = {member_id}
362
+ queue = [ref for ref in by_id[member_id].references if ref in known_ids]
363
+ while queue:
364
+ current = queue.pop()
365
+ if current in reachable or current in root_ids:
366
+ continue
367
+ reachable.add(current)
368
+ queue.extend(ref for ref in by_id[current].references if ref in known_ids)
369
+ return reachable
370
+
371
+
372
+ def _resolve_root_ids(by_id, known_ids) -> "set[str]":
373
+ """The root set: kind-and-orphanhood, then completed so nothing is stranded.
374
+
375
+ The base rule is §3.1's: incidents and backup stems are roots whatever
376
+ references them, an unrecognized kind is a root because nobody wrote a
377
+ validator for it, and a bundle, WAL-evidence directory or rebuild record is
378
+ a root only when nothing references it.
379
+
380
+ That rule alone can strand a member. Two bundles that reference each other
381
+ and nothing else are both "referenced", so neither is a root, and no root
382
+ reaches either — they are invisible to the planner and can never be
383
+ reclaimed. The completion loop below promotes such a member to a root, so
384
+ the graph carries the second invariant this subsystem needs: **every member
385
+ is reachable from at least one root**. Preferring an orphan that no other
386
+ orphan references keeps a satellite a satellite; the lexicographic fallback
387
+ is what breaks a pure cycle deterministically.
388
+ """
389
+ referenced: "set[str]" = set()
390
+ for member in by_id.values():
391
+ referenced.update(ref for ref in member.references if ref in known_ids)
392
+
393
+ root_ids = {
394
+ member_id
395
+ for member_id, member in by_id.items()
396
+ if member.kind in _ROOT_WHATEVER_REFERENCES_IT
397
+ or member.kind not in _KIND_REQUIRES_CLASSIFICATION
398
+ or member_id not in referenced
399
+ }
400
+
401
+ while True:
402
+ covered: "set[str]" = set()
403
+ for root_id in root_ids:
404
+ covered |= _reachable_from(by_id, known_ids, root_ids, root_id)
405
+ stranded = known_ids - covered
406
+ if not stranded:
407
+ return root_ids
408
+ inner = {
409
+ ref
410
+ for member_id in stranded
411
+ for ref in by_id[member_id].references
412
+ if ref in stranded
413
+ }
414
+ promote = sorted(stranded - inner) or [min(stranded)]
415
+ root_ids.update(promote)
416
+
417
+
418
+ def build_graph(members) -> RetentionGraph:
419
+ """Build the reference graph, identify roots and index inbound roots (§3.1).
420
+
421
+ Two invariants hold over the result, and the planner and the doctor leg
422
+ both rest on them:
423
+
424
+ - **no root is reachable from another root**, so every root appears in the
425
+ deletion closure of any selection containing it;
426
+ - **every member is reachable from some root**, so nothing in the corpus is
427
+ invisible to the planner.
428
+ """
429
+ by_id: "dict[str, RetentionMember]" = {}
430
+ for member in members:
431
+ if member.id in by_id:
432
+ raise ValueError(f"duplicate retained-artifact id: {member.id}")
433
+ by_id[member.id] = member
434
+ known_ids = set(by_id)
435
+
436
+ root_ids = _resolve_root_ids(by_id, known_ids)
437
+
438
+ own_reasons = {
439
+ member_id: _member_protection_reasons(member, known_ids)
440
+ for member_id, member in by_id.items()
441
+ }
442
+ for member_id in root_ids:
443
+ member = by_id[member_id]
444
+ if member.kind == "wal_evidence":
445
+ # §3.3: WAL evidence is valid when a valid bundle or incident
446
+ # references it. Nothing does, so it cannot be validated and must
447
+ # not be swept on its own.
448
+ own_reasons[member_id].append("unreferenced-evidence")
449
+
450
+ inbound: "dict[str, set[str]]" = {member_id: set() for member_id in by_id}
451
+ roots: "list[RetentionRoot]" = []
452
+ for member_id in sorted(root_ids):
453
+ member = by_id[member_id]
454
+ reachable = _reachable_from(by_id, known_ids, root_ids, member_id)
455
+ for reached in reachable:
456
+ inbound[reached].add(member_id)
457
+
458
+ reasons: "set[str]" = set()
459
+ for reached in reachable:
460
+ reasons.update(own_reasons[reached])
461
+ requires_classification = _KIND_REQUIRES_CLASSIFICATION.get(
462
+ member.kind, True
463
+ )
464
+ if requires_classification and not is_classified(member.classification):
465
+ reasons.add("unclassified")
466
+
467
+ roots.append(RetentionRoot(
468
+ id=member_id,
469
+ kind=member.kind,
470
+ family=member.family,
471
+ created_at_epoch=member.created_at_epoch,
472
+ reachable_ids=frozenset(reachable),
473
+ own_member_ids=frozenset(reachable - root_ids),
474
+ requires_classification=requires_classification,
475
+ classification=member.classification,
476
+ shape_token=member.shape_token,
477
+ protected_reasons=tuple(sorted(reasons)),
478
+ ))
479
+
480
+ roots.sort(key=lambda root: (root.created_at_epoch, root.id))
481
+ return RetentionGraph(
482
+ roots=tuple(roots),
483
+ members=by_id,
484
+ inbound_roots={
485
+ member_id: frozenset(found) for member_id, found in inbound.items()
486
+ },
487
+ roots_by_id={root.id: root for root in roots},
488
+ )
489
+
490
+
491
+ # --------------------------------------------------------------------------
492
+ # §5.2 — one central admission predicate
493
+ # --------------------------------------------------------------------------
494
+ #
495
+ # new_plan = eligible_hook_tick_branch OR eligible_mutating_command_branch
496
+ # recovery = eligible_invocation AND a pending plan exists
497
+ #
498
+ # Revision 3 stated `hook-tick` as a mandatory condition and then said ordinary
499
+ # commands enter through the same predicate, which read literally means no
500
+ # ordinary command can ever qualify. It is a DISJUNCTION.
501
+
502
+ #: The commands whose success may schedule a sweep. A central ANNOTATION, not
503
+ #: a computed property: reports legitimately write to the entry cache while
504
+ #: remaining user-facing reads, so "does it mutate the filesystem" would
505
+ #: misclassify `daily` and `report` as mutating. `db` children are named by
506
+ #: their qualified `db <action>` form.
507
+ RETENTION_MUTATING_COMMANDS = frozenset({
508
+ "sync-week",
509
+ "record-usage",
510
+ "record-credit",
511
+ "cache-sync",
512
+ "db rebuild",
513
+ "db rederive",
514
+ "db journal-repair",
515
+ "db vacuum",
516
+ "db checkpoint",
517
+ "db backup",
518
+ })
519
+
520
+ #: `cctally db prune` never admits, in EITHER path and in EVERY mode. A
521
+ #: successful `--yes` has already applied the plan, so a redundant automatic
522
+ #: one immediately afterwards is pure waste; and a preview that triggered a
523
+ #: real deletion would contradict its own output.
524
+ RETENTION_NEVER_ADMITS = frozenset({"db prune"})
525
+
526
+ #: On the mutating allowlist because their APPLY mutates, but preview-by-
527
+ #: default — so their PREVIEW must not admit. `db journal-repair`'s preview
528
+ #: carries a literal no-mutation contract that `bin/cctally` already honours by
529
+ #: skipping the update hooks for it, and a preview that filed an admission
530
+ #: marker broke exactly that contract. It is also correct on the merits: a
531
+ #: preview created no new evidence, so there is nothing new to reclaim.
532
+ #: `record-credit` is here for the same reason and is NOT a `db` child: it is
533
+ #: preview-and-confirm by default, so its preview created no new evidence and
534
+ #: must not schedule a sweep.
535
+ RETENTION_PREVIEW_BY_DEFAULT = frozenset({
536
+ "db rederive", "db journal-repair", "db prune", "record-credit",
537
+ })
538
+
539
+
540
+ def qualified_command(command, action=None) -> str:
541
+ """`db rebuild` for a db child, the bare command otherwise."""
542
+ name = str(command or "")
543
+ child = str(action or "")
544
+ return f"{name} {child}" if name == "db" and child else name
545
+
546
+
547
+ def retention_admission_possible(
548
+ *,
549
+ command,
550
+ action=None,
551
+ exit_code: int = 0,
552
+ hook_forked: "bool | None" = None,
553
+ hook_explain: bool = False,
554
+ hook_foreground: bool = False,
555
+ prune_mode: "str | None" = None,
556
+ applied: "bool | None" = None,
557
+ ) -> bool:
558
+ """Whether this invocation could admit at all, from the invocation alone.
559
+
560
+ Split out of `retention_admission` so the glue can reject before it
561
+ measures anything. The daily rate limit costs a `stat` and the pending-plan
562
+ probe costs a `glob` of the data directory, and both were evaluated as
563
+ ARGUMENTS — so `cctally statusline`, which can never admit, paid a readdir
564
+ on every render to reach a decision this function makes for free.
565
+ """
566
+ name = qualified_command(command, action)
567
+ if prune_mode is not None or name in RETENTION_NEVER_ADMITS:
568
+ return False
569
+ if name in RETENTION_PREVIEW_BY_DEFAULT and applied is not True:
570
+ return False
571
+ if int(exit_code) != 0:
572
+ # A failed command never admits, in either path: whatever went wrong
573
+ # may be the very corruption whose evidence a sweep would reclaim.
574
+ return False
575
+ hook_branch = (
576
+ name == "hook-tick"
577
+ and not hook_explain
578
+ and not hook_foreground
579
+ and hook_forked is True
580
+ )
581
+ # Every hidden worker, `statusline`, `doctor`, share previews and every
582
+ # read-only command fail both branches, because the allowlist is a
583
+ # whitelist.
584
+ return bool(hook_branch or name in RETENTION_MUTATING_COMMANDS)
585
+
586
+
587
+ def retention_admission(
588
+ *,
589
+ command,
590
+ action=None,
591
+ exit_code: int = 0,
592
+ hook_forked: "bool | None" = None,
593
+ hook_explain: bool = False,
594
+ hook_foreground: bool = False,
595
+ prune_mode: "str | None" = None,
596
+ applied: "bool | None" = None,
597
+ rate_limited: bool = False,
598
+ pending_plan_present: bool = False,
599
+ ) -> str:
600
+ """What this invocation may schedule: `new-plan`, `recovery`, or nothing.
601
+
602
+ Every input is explicit. The daily rate limit and the presence of a pending
603
+ plan are booleans the glue measures, because this function reads no clock
604
+ and no filesystem.
605
+
606
+ `applied` is the `--yes` state of a preview-by-default command, and `None`
607
+ on a command that has no preview mode at all. Only a command in
608
+ `RETENTION_PREVIEW_BY_DEFAULT` is gated on it.
609
+ """
610
+ if not retention_admission_possible(
611
+ command=command, action=action, exit_code=exit_code,
612
+ hook_forked=hook_forked, hook_explain=hook_explain,
613
+ hook_foreground=hook_foreground, prune_mode=prune_mode,
614
+ applied=applied,
615
+ ):
616
+ return ""
617
+ if not rate_limited:
618
+ return "new-plan"
619
+ if pending_plan_present:
620
+ # Unconditional and NOT rate-limited: a crashed deletion must not wait
621
+ # 24 hours to finish. A new plan already resumes first (§5.4), so this
622
+ # branch only matters once the daily limit has closed the other one.
623
+ return "recovery"
624
+ return ""
625
+
626
+
627
+ # --------------------------------------------------------------------------
628
+ # §5.5 — the resume decision table
629
+ # --------------------------------------------------------------------------
630
+
631
+ #: The two durable phases a reclaim entry passes through. No unlink happens
632
+ #: until `marked` is fsynced, which is what makes the table below decidable.
633
+ RECLAIM_PHASE_MARKING = "marking"
634
+ RECLAIM_PHASE_MARKED = "marked"
635
+
636
+
637
+ #: How long a reclaim entry may carry an error before the condition counts as
638
+ #: stuck rather than transient. Most errors are retried and clear themselves:
639
+ #: `_resume_marking_pass` re-decides every entry on every pass, so a rename that
640
+ #: failed on a busy file succeeds later. One case cannot clear itself — phase
641
+ #: `marking` with neither the source nor the tombstone present, which means
642
+ #: something outside this subsystem moved the member — and the record holding it
643
+ #: would otherwise sit in the data directory forever with nothing reporting it.
644
+ RECLAIM_STUCK_AFTER_SECONDS = 86400
645
+
646
+
647
+ def format_disk_bytes(value, *, digits: int = 1) -> str:
648
+ """Bytes in the largest unit that still carries a significant figure.
649
+
650
+ ONE home for the rule, because a fixed GiB rendering prints `0.00 GiB` in
651
+ every column of a corpus below about 50 MiB — which is most of them on a
652
+ healthy install and all of them in a fixture, and is exactly the figure
653
+ that tells an operator nothing. `digits` differs by surface: `db prune`'s
654
+ table renders two decimals per §6.4, doctor's one-line summaries one.
655
+ """
656
+ value = int(value or 0)
657
+ for threshold, unit in (
658
+ (1024 ** 3, "GiB"), (1024 ** 2, "MiB"), (1024, "KiB"),
659
+ ):
660
+ if value >= threshold:
661
+ return f"{value / threshold:.{digits}f} {unit}"
662
+ return f"{value} B"
663
+
664
+
665
+ def reclaim_entry_is_stuck(
666
+ *, error, first_failed_at_epoch, now_epoch,
667
+ threshold_seconds: int = RECLAIM_STUCK_AFTER_SECONDS,
668
+ ) -> bool:
669
+ """Whether one reclaim entry's error has persisted long enough to report.
670
+
671
+ An entry with no error is never stuck, and an entry whose first failure has
672
+ no recorded time is treated as fresh rather than as stuck — a record written
673
+ by an earlier binary carries no stamp, and reporting it immediately would
674
+ raise an alarm about a condition nobody has observed to persist.
675
+ """
676
+ if not error:
677
+ return False
678
+ if first_failed_at_epoch is None:
679
+ return False
680
+ return (now_epoch - first_failed_at_epoch) >= threshold_seconds
681
+
682
+
683
+ def resume_action(phase: str, source_present: bool, tombstone_present: bool) -> str:
684
+ """What a resuming worker must do with one reclaim entry (§5.5).
685
+
686
+ The durable phase is what makes "neither exists" decidable. There is an
687
+ unavoidable crash window after a tombstone is unlinked and before its entry
688
+ is cleared from the pending record, so at `marked` that state is a
689
+ completed entry rather than an error — reporting it as a permanent failure
690
+ would make a SUCCESSFUL deletion non-resumable.
691
+
692
+ Two rows §5.5's table does not enumerate:
693
+
694
+ - `marking` with neither present. The rename never completed and no unlink
695
+ can have happened yet, so the member was moved by something outside this
696
+ subsystem. Fail closed.
697
+ - `marked` with the source present and the tombstone gone. The deletion
698
+ completed and something re-created the original path; the new inode is
699
+ not ours to remove, and there is nothing left to delete.
700
+ """
701
+ if source_present and tombstone_present:
702
+ return "fail-closed"
703
+ if phase == RECLAIM_PHASE_MARKING:
704
+ if source_present:
705
+ return "resume-rename"
706
+ if tombstone_present:
707
+ return "advance-to-marked"
708
+ return "fail-closed"
709
+ if tombstone_present:
710
+ return "continue-deletion"
711
+ return "entry-complete"
712
+
713
+
714
+ # --------------------------------------------------------------------------
715
+ # §3.1 / §3.4 / §3.5 / §3.6 — the planner
716
+ # --------------------------------------------------------------------------
717
+
718
+ #: The damage shape that is not a shape (§3.4). An incident whose preserved
719
+ #: damage token is the literal `none` earns no floor; 11 of the 29 tokens on
720
+ #: the maintainer's store are this value.
721
+ NON_SHAPE_TOKEN = "none"
722
+
723
+ #: The bounds, in the order §3.5 applies them. The names are the policy's own
724
+ #: attribute names, so `unsatisfied_rules` says which knob to change.
725
+ BOUND_ORDER = (
726
+ "max_age_seconds", "max_count_per_family", "max_total_bytes", "min_free_bytes",
727
+ )
728
+
729
+
730
+ @dataclasses.dataclass(frozen=True)
731
+ class RetentionState:
732
+ """The graph plus the two values the kernel refuses to read for itself.
733
+
734
+ `now_epoch` and `free_disk_bytes` are handed in because the kernel takes
735
+ no clock and no filesystem. `free_disk_bytes` is None when the glue could
736
+ not measure it, and the free-disk floor is then skipped rather than
737
+ guessed at.
738
+ """
739
+
740
+ graph: RetentionGraph
741
+ now_epoch: float
742
+ free_disk_bytes: "int | None"
743
+
744
+
745
+ @dataclasses.dataclass(frozen=True)
746
+ class RetentionPlan:
747
+ """What a sweep would delete, keep and refuse to touch."""
748
+
749
+ delete_ids: "tuple[str, ...]"
750
+ #: `(root_id, ordered member ids)` per selected root, in deletion order.
751
+ #: The marking engine decides per ROOT (§5.4), so it needs the grouping
752
+ #: explicitly rather than having to re-derive it from `reasons`. Within a
753
+ #: group every referrer precedes its referent.
754
+ delete_groups: "tuple[tuple[str, tuple[str, ...]], ...]"
755
+ keep_ids: "tuple[str, ...]"
756
+ protected_ids: "tuple[str, ...]"
757
+ reasons: "dict[str, str]"
758
+ before_bytes: int
759
+ projected_bytes: int
760
+ reclaimable_bytes: int
761
+ reference_pinned_bytes: int
762
+ unsatisfied_rules: "tuple[str, ...]"
763
+ #: Roots a bound would have taken and the shape floor kept (§3.6), with
764
+ #: what deleting them on top of the plan would have freed. Reported
765
+ #: SEPARATELY from `unsatisfied_rules` because the operator asked for this
766
+ #: retention through `max_shape_examples` and has nothing to act on.
767
+ floor_retained_ids: "tuple[str, ...]" = ()
768
+ floor_retained_bytes: int = 0
769
+
770
+
771
+ def deletion_closure(graph: RetentionGraph, selected) -> "frozenset[str]":
772
+ """The members deletable given exactly this selected set of roots (§3.1).
773
+
774
+ deletion_closure(S) = { m : m reachable from some root in S
775
+ and m reachable from no root outside S }
776
+
777
+ Exclusivity is a property of the SELECTION, never of a root. For roots A
778
+ and B that both reference target T, T is in neither singleton closure and
779
+ enters only once both are selected — which is why a statically computed
780
+ `exclusive_closure_ids` field either never reclaims T or subtracts its
781
+ bytes too early.
782
+ """
783
+ selected = frozenset(selected)
784
+ if not selected:
785
+ return frozenset()
786
+ reached: "set[str]" = set()
787
+ for root_id in selected:
788
+ root = graph.roots_by_id.get(root_id)
789
+ if root is not None:
790
+ reached.update(root.reachable_ids)
791
+ return frozenset(
792
+ member_id
793
+ for member_id in reached
794
+ if graph.inbound_roots.get(member_id, frozenset()) <= selected
795
+ )
796
+
797
+
798
+ def _closure_order(graph: RetentionGraph, root_id: str, closure) -> "list[str]":
799
+ """`root_id` first, then the members it reaches, **every** referrer first.
800
+
801
+ §5.4 renames a reference-bearing root before the bundle it references, so a
802
+ crash cannot leave a surviving manifest pointing at a tombstone. A pre-order
803
+ walk alone does NOT give that: it emits a member the first time it is
804
+ reached, which can precede a referrer reached only by a longer path. The
805
+ production shape is exactly that diamond — 28 of the maintainer's 142
806
+ incidents name both a forensics bundle and a rebuild record, and that record
807
+ names the same bundle — so under one reference ordering the bundle would be
808
+ renamed before the record that points at it.
809
+
810
+ The pre-order below therefore only fixes a deterministic candidate order.
811
+ The emission loop is a topological pass over it: a member is emitted once
812
+ every in-closure referrer has been emitted, and when a cycle leaves nothing
813
+ ready the earliest pre-order candidate wins — which is the entry root on the
814
+ production incident-to-rebuild-record cycle.
815
+ """
816
+ candidates: "list[str]" = []
817
+ seen: "set[str]" = set()
818
+ stack = [root_id]
819
+ while stack:
820
+ current = stack.pop()
821
+ if current in seen or current not in closure:
822
+ continue
823
+ seen.add(current)
824
+ candidates.append(current)
825
+ member = graph.members.get(current)
826
+ if member is None:
827
+ continue
828
+ stack.extend(reversed([
829
+ ref for ref in member.references if ref in graph.members
830
+ ]))
831
+
832
+ referrers: "dict[str, set[str]]" = {member_id: set() for member_id in candidates}
833
+ for member_id in candidates:
834
+ member = graph.members.get(member_id)
835
+ if member is None:
836
+ continue
837
+ for ref in member.references:
838
+ if ref in referrers and ref != member_id:
839
+ referrers[ref].add(member_id)
840
+
841
+ ordered: "list[str]" = []
842
+ emitted: "set[str]" = set()
843
+ pending = list(candidates)
844
+ while pending:
845
+ ready = next(
846
+ (member_id for member_id in pending if referrers[member_id] <= emitted),
847
+ None,
848
+ )
849
+ chosen = pending[0] if ready is None else ready
850
+ ordered.append(chosen)
851
+ emitted.add(chosen)
852
+ pending.remove(chosen)
853
+ return ordered
854
+
855
+
856
+ def _shape_floor_ids(graph: RetentionGraph, eligible, max_shape_examples: int):
857
+ """The roots §3.4 keeps because they are the last example of a shape.
858
+
859
+ One example per distinct shape, capped at `max_shape_examples` shapes. The
860
+ kept example is the NEWEST, which under oldest-first selection is exactly
861
+ "never delete the last one". Shapes compete for the cap by the recency of
862
+ their newest example, so a cap smaller than the number of shapes is still
863
+ deterministic.
864
+ """
865
+ if max_shape_examples <= 0:
866
+ return frozenset()
867
+ newest: "dict[str, RetentionRoot]" = {}
868
+ for root in graph.roots:
869
+ token = root.shape_token
870
+ if not isinstance(token, str) or not token or token == NON_SHAPE_TOKEN:
871
+ continue
872
+ if root.id not in eligible:
873
+ continue
874
+ current = newest.get(token)
875
+ if current is None or (
876
+ root.created_at_epoch, root.id
877
+ ) > (current.created_at_epoch, current.id):
878
+ newest[token] = root
879
+ ranked = sorted(
880
+ newest.values(),
881
+ key=lambda root: (-root.created_at_epoch, root.id),
882
+ )
883
+ return frozenset(root.id for root in ranked[:max_shape_examples])
884
+
885
+
886
+ class _Selection:
887
+ """Roots chosen so far, with the closure updated after every addition.
888
+
889
+ Recomputing once per phase is not enough: the byte phase alone selects
890
+ several roots, and each one changes what the next one would reclaim
891
+ (§3.1).
892
+
893
+ **The update is incremental, and that is a cost decision, not a semantic
894
+ one.** `deletion_closure` remains the authoritative statement of §3.1. The
895
+ equality is enforced by
896
+ `tests/test_artifact_retention_kernel.py::test_the_incremental_selection_equals_the_authoritative_closure`,
897
+ which compares this class against `deletion_closure` after every addition
898
+ over 400 generated corpora, and NOT by an assertion in `add`: re-deriving
899
+ the whole closure once per candidate is precisely the cost the incremental
900
+ form removes.
901
+
902
+ **That differential covers exactly three of this class's outputs** —
903
+ `closure`, `bytes_reclaimed` and `surviving_root_counts`. It does not
904
+ compare `order` or `groups`, and cannot: §5.4's topological deletion order
905
+ is produced by `_closure_order`, so a differential would have to
906
+ re-implement it. That ordering is covered instead by
907
+ `test_delete_ids_puts_a_referrer_before_its_referent`,
908
+ `test_a_diamond_emits_every_referrer_before_the_shared_referent`,
909
+ `test_the_production_cycle_still_puts_the_incident_first`,
910
+ `test_delete_groups_lead_with_their_root_and_carry_its_closure`,
911
+ `test_delete_groups_flatten_back_to_delete_ids` and
912
+ `test_a_shared_member_joins_the_group_of_the_root_that_completed_it`.
913
+
914
+ What changed is how the same answer is reached: recomputing the whole
915
+ closure and the whole surviving-root tally per candidate made
916
+ `plan_artifact_retention` quadratic
917
+ in the root count — measured at 8.0 ms over 142 roots, 32.1 over 284, 125.8
918
+ over 568 and 535.7 over 1136, a clean 4x per doubling. The walk's
919
+ 5000-entry cap admits roughly 1600 roots, and this runs on the periodic
920
+ doctor gather, which is the shape that has pegged this repository's CPU
921
+ before.
922
+
923
+ The rewrite rests on the definition itself. A member joins the closure
924
+ exactly when the last of its inbound roots is selected, so one countdown
925
+ per member — decremented when a root that reaches it is added — decides
926
+ membership in O(1) per inbound edge over the whole selection. A member with
927
+ NO inbound root can never join, because it is reachable from no selected
928
+ root; its countdown starts at zero and is excluded explicitly rather than
929
+ by arithmetic.
930
+ """
931
+
932
+ def __init__(self, graph: RetentionGraph):
933
+ self._graph = graph
934
+ self.root_ids: "list[str]" = []
935
+ self.order: "list[str]" = []
936
+ self.groups: "list[tuple[str, tuple[str, ...]]]" = []
937
+ self.reasons: "dict[str, str]" = {}
938
+
939
+ #: member -> how many of its inbound roots are still unselected.
940
+ self._pending: "dict[str, int]" = {}
941
+ #: root -> the members that root reaches, i.e. the inverse of
942
+ #: `inbound_roots`, so one addition touches only its own edges.
943
+ self._reached_by: "dict[str, list[str]]" = {}
944
+ for member_id, inbound in graph.inbound_roots.items():
945
+ if not inbound:
946
+ continue
947
+ self._pending[member_id] = len(inbound)
948
+ for root_id in inbound:
949
+ self._reached_by.setdefault(root_id, []).append(member_id)
950
+
951
+ self._closed: "set[str]" = set()
952
+ self._frozen: "frozenset[str] | None" = frozenset()
953
+ self._bytes = 0
954
+ self._counts: "dict[str, int]" = {}
955
+ for root in graph.roots:
956
+ self._counts[root.family] = self._counts.get(root.family, 0) + 1
957
+
958
+ @property
959
+ def closure(self) -> "frozenset[str]":
960
+ """The deletion closure of the roots selected so far.
961
+
962
+ Materialized on demand and cached until the next addition. Freezing the
963
+ set inside `add` instead copied the whole closure once per selected
964
+ root, which is a second roots-times-members term and was most of what
965
+ remained after the countdown replaced the recomputation.
966
+ """
967
+ if self._frozen is None:
968
+ self._frozen = frozenset(self._closed)
969
+ return self._frozen
970
+
971
+ @property
972
+ def bytes_reclaimed(self) -> int:
973
+ return self._bytes
974
+
975
+ def add(self, root_id: str, reason: str) -> None:
976
+ self.root_ids.append(root_id)
977
+ self.reasons[root_id] = reason
978
+ newly: "set[str]" = set()
979
+ for member_id in self._reached_by.get(root_id, ()):
980
+ remaining = self._pending[member_id] - 1
981
+ self._pending[member_id] = remaining
982
+ if remaining:
983
+ continue
984
+ newly.add(member_id)
985
+ self._closed.add(member_id)
986
+ member = self._graph.members.get(member_id)
987
+ if member is not None:
988
+ self._bytes += member.disk_bytes
989
+ root = self._graph.roots_by_id.get(member_id)
990
+ if root is not None and self._counts.get(root.family):
991
+ self._counts[root.family] -= 1
992
+ self._frozen = None
993
+ added: "list[str]" = []
994
+ # Every member that newly closed is reachable from the root just added
995
+ # — that is what made its countdown reach zero — so ordering the walk
996
+ # from this root reaches all of them, in §5.4's topological order.
997
+ for member_id in _closure_order(self._graph, root_id, self._closed):
998
+ if member_id in newly:
999
+ self.order.append(member_id)
1000
+ added.append(member_id)
1001
+ self.reasons.setdefault(member_id, "closure")
1002
+ if added:
1003
+ self.groups.append((root_id, tuple(added)))
1004
+
1005
+ def surviving_root_counts(self) -> "dict[str, int]":
1006
+ """Roots per family that this selection would leave on disk.
1007
+
1008
+ A root survives when it is not in the deletion closure — NOT merely
1009
+ when it has not been selected. The distinction is what §3.5 means by
1010
+ "projected state is recomputed after every root added": the age phase
1011
+ removes its selections from the candidate list without touching any
1012
+ per-family tally, so a count phase that measured the original
1013
+ population would keep taking until the candidates ran out. Measured on
1014
+ the maintainer's corpus shape with classification complete, a 20-per-
1015
+ family bound left 40 survivors on its own and 8 once a firing age bound
1016
+ preceded it.
1017
+
1018
+ The tally is maintained by `add` rather than rebuilt here, which is
1019
+ what makes the count phase linear; a family whose survivors reach zero
1020
+ is dropped so the mapping keeps agreeing with a rebuilt one key for
1021
+ key.
1022
+ """
1023
+ return {
1024
+ family: count for family, count in self._counts.items() if count
1025
+ }
1026
+
1027
+
1028
+ def _run_selection(
1029
+ state: RetentionState, policy: RetentionPolicy, floored,
1030
+ ) -> "tuple[_Selection, int]":
1031
+ """Select roots oldest-first against each bound in turn (§3.5).
1032
+
1033
+ Factored out so §3.6's "is this bound blocked by PROTECTION?" question can
1034
+ be answered by running the very same selection with the shape floor
1035
+ disabled, rather than by four per-bound subtractions that the byte and
1036
+ free-disk bounds cannot express (a floored root's share of a shared member
1037
+ is not a per-root quantity).
1038
+ """
1039
+ graph = state.graph
1040
+ # Already oldest-first from `build_graph`. Each phase consumes this list in
1041
+ # order and hands its leftovers to the next, rather than removing from one
1042
+ # shared list: `list.remove` compares whole frozen dataclasses, so the
1043
+ # removals alone cost roots-squared field comparisons at the walk's cap.
1044
+ remaining = [
1045
+ root for root in graph.roots
1046
+ if not root.protected_reasons and root.id not in floored
1047
+ ]
1048
+
1049
+ selection = _Selection(graph)
1050
+
1051
+ if policy.max_age_seconds is not None:
1052
+ kept = []
1053
+ for root in remaining:
1054
+ if state.now_epoch - root.created_at_epoch > policy.max_age_seconds:
1055
+ selection.add(root.id, "max_age_seconds")
1056
+ else:
1057
+ kept.append(root)
1058
+ remaining = kept
1059
+
1060
+ if policy.max_count_per_family is not None:
1061
+ # Re-read the surviving population before EVERY candidate. Maintaining
1062
+ # a private tally seeded from `graph.roots` is what let the age phase's
1063
+ # selections be counted twice; a protected or shape-floored root does
1064
+ # survive and must keep counting, which is why the tally is derived
1065
+ # from the closure rather than from `remaining`.
1066
+ kept = []
1067
+ for root in remaining:
1068
+ counts = selection.surviving_root_counts()
1069
+ if counts.get(root.family, 0) > policy.max_count_per_family:
1070
+ selection.add(root.id, "max_count_per_family")
1071
+ else:
1072
+ kept.append(root)
1073
+ remaining = kept
1074
+
1075
+ before_bytes = sum(member.disk_bytes for member in graph.members.values())
1076
+ cursor = 0
1077
+
1078
+ if policy.max_total_bytes is not None:
1079
+ while (
1080
+ cursor < len(remaining)
1081
+ and before_bytes - selection.bytes_reclaimed > policy.max_total_bytes
1082
+ ):
1083
+ selection.add(remaining[cursor].id, "max_total_bytes")
1084
+ cursor += 1
1085
+
1086
+ if state.free_disk_bytes is not None and policy.min_free_bytes is not None:
1087
+ while (
1088
+ cursor < len(remaining)
1089
+ and state.free_disk_bytes + selection.bytes_reclaimed
1090
+ < policy.min_free_bytes
1091
+ ):
1092
+ selection.add(remaining[cursor].id, "min_free_bytes")
1093
+ cursor += 1
1094
+
1095
+ return selection, before_bytes
1096
+
1097
+
1098
+ def _unsatisfied_rules(
1099
+ state: RetentionState, policy: RetentionPolicy, selection: "_Selection",
1100
+ before_bytes: int,
1101
+ ) -> "list[str]":
1102
+ """The bounds this selection leaves unmet."""
1103
+ graph = state.graph
1104
+ deleted_ids = selection.closure
1105
+ reclaimable = selection.bytes_reclaimed
1106
+ unsatisfied: "list[str]" = []
1107
+ if policy.max_age_seconds is not None and any(
1108
+ state.now_epoch - root.created_at_epoch > policy.max_age_seconds
1109
+ for root in graph.roots
1110
+ if root.id not in deleted_ids
1111
+ ):
1112
+ unsatisfied.append("max_age_seconds")
1113
+ if policy.max_count_per_family is not None:
1114
+ surviving = selection.surviving_root_counts()
1115
+ if any(count > policy.max_count_per_family for count in surviving.values()):
1116
+ unsatisfied.append("max_count_per_family")
1117
+ if (
1118
+ policy.max_total_bytes is not None
1119
+ and before_bytes - reclaimable > policy.max_total_bytes
1120
+ ):
1121
+ unsatisfied.append("max_total_bytes")
1122
+ if (
1123
+ state.free_disk_bytes is not None
1124
+ and policy.min_free_bytes is not None
1125
+ and state.free_disk_bytes + reclaimable < policy.min_free_bytes
1126
+ ):
1127
+ unsatisfied.append("min_free_bytes")
1128
+ return unsatisfied
1129
+
1130
+
1131
+ def plan_artifact_retention(
1132
+ state: RetentionState, policy: RetentionPolicy,
1133
+ ) -> RetentionPlan:
1134
+ """Select roots oldest-first against each bound in turn (§3.5).
1135
+
1136
+ The protection gate runs before every bound and is absolute (§3.2). When
1137
+ protection alone exceeds a bound, everything eligible is still deleted and
1138
+ the bound is reported `unsatisfied` rather than silently unmet (§3.6).
1139
+
1140
+ **The shape floor is excused from `unsatisfied_rules`; protection is not**
1141
+ (§3.6). Both leave a root on disk that a bound would otherwise have taken,
1142
+ but only one of them is something the operator can act on. A floored root
1143
+ is retained by the policy the operator themselves set through
1144
+ `max_shape_examples`, and reporting that as an unsatisfied bound states
1145
+ that they must act while the system does exactly what they asked. Two of
1146
+ the four damage shapes on the maintainer's corpus appear exactly once, so
1147
+ without this the age bound becomes permanently unsatisfiable on a healthy
1148
+ install and `db.retained_artifacts` FAILs forever — a FAIL no action
1149
+ clears, which trains the operator to ignore doctor.
1150
+
1151
+ The question is therefore answered by re-running the identical selection
1152
+ with the floor disabled. Whatever that run still leaves unmet is what
1153
+ protection blocks.
1154
+ """
1155
+ graph = state.graph
1156
+ protected = [root for root in graph.roots if root.protected_reasons]
1157
+ eligible_ids = {
1158
+ root.id for root in graph.roots if not root.protected_reasons
1159
+ }
1160
+ floored = _shape_floor_ids(graph, eligible_ids, policy.max_shape_examples)
1161
+
1162
+ selection, before_bytes = _run_selection(state, policy, floored)
1163
+
1164
+ reclaimable = selection.bytes_reclaimed
1165
+ projected = before_bytes - reclaimable
1166
+ selected_ids = set(selection.root_ids)
1167
+ # What survives is what the closure does NOT delete. `build_graph`
1168
+ # guarantees every root is inside its own closure, so for a graph it built
1169
+ # these two sets agree on roots; keeping the distinction here means a graph
1170
+ # assembled by any other route still reports every root in exactly one of
1171
+ # `delete_ids`, `keep_ids` and `protected_ids` instead of dropping it.
1172
+ deleted_ids = selection.closure
1173
+
1174
+ pinned = 0
1175
+ for member_id, inbound in graph.inbound_roots.items():
1176
+ if member_id in deleted_ids or not inbound:
1177
+ continue
1178
+ if inbound & selected_ids:
1179
+ pinned += graph.members[member_id].disk_bytes
1180
+
1181
+ if floored:
1182
+ floor_free, _ = _run_selection(state, policy, frozenset())
1183
+ unsatisfied = _unsatisfied_rules(state, policy, floor_free, before_bytes)
1184
+ # What the floor kept is what a bound WOULD have taken (§6.4's "that
1185
+ # age and count would otherwise have removed"), never every root that
1186
+ # merely holds a floor.
1187
+ floor_retained = tuple(
1188
+ root.id for root in graph.roots
1189
+ if root.id in floored and root.id in floor_free.closure
1190
+ )
1191
+ floor_retained_bytes = sum(
1192
+ graph.members[member_id].disk_bytes
1193
+ for member_id in deletion_closure(
1194
+ graph, selected_ids | set(floor_retained)
1195
+ )
1196
+ - deleted_ids
1197
+ )
1198
+ else:
1199
+ unsatisfied = _unsatisfied_rules(state, policy, selection, before_bytes)
1200
+ floor_retained = ()
1201
+ floor_retained_bytes = 0
1202
+
1203
+ reasons = dict(selection.reasons)
1204
+ for root in protected:
1205
+ reasons[root.id] = ",".join(root.protected_reasons)
1206
+ keep_ids = []
1207
+ for root in graph.roots:
1208
+ if root.id in deleted_ids or root.protected_reasons:
1209
+ continue
1210
+ keep_ids.append(root.id)
1211
+ if root.id in selected_ids:
1212
+ # Selected, and still on disk: another root that survives can reach
1213
+ # it. Overwriting the selection reason is deliberate — reporting it
1214
+ # as deleted under `max_count_per_family` would describe work the
1215
+ # plan is not going to do.
1216
+ reasons[root.id] = "reference-pinned"
1217
+ continue
1218
+ reasons.setdefault(
1219
+ root.id, "shape-floor" if root.id in floored else "retained",
1220
+ )
1221
+
1222
+ return RetentionPlan(
1223
+ delete_ids=tuple(selection.order),
1224
+ delete_groups=tuple(selection.groups),
1225
+ keep_ids=tuple(keep_ids),
1226
+ protected_ids=tuple(root.id for root in protected),
1227
+ reasons=reasons,
1228
+ before_bytes=before_bytes,
1229
+ projected_bytes=projected,
1230
+ reclaimable_bytes=reclaimable,
1231
+ reference_pinned_bytes=pinned,
1232
+ unsatisfied_rules=tuple(
1233
+ rule for rule in BOUND_ORDER if rule in unsatisfied
1234
+ ),
1235
+ floor_retained_ids=floor_retained,
1236
+ floor_retained_bytes=floor_retained_bytes,
1237
+ )
1238
+
1239
+
1240
+ # --------------------------------------------------------------------------
1241
+ # §3.3 — validation and classification, per member kind
1242
+ # --------------------------------------------------------------------------
1243
+ #
1244
+ # A universal "no manifest means protected" rule would permanently protect
1245
+ # every standalone bundle, rebuild record and ordinary backup family, none of
1246
+ # which has an incident manifest. Each predicate below is therefore about one
1247
+ # kind, and each returns a plain bool the metadata walk stores on the member.
1248
+
1249
+ #: Files an incident directory acquires AFTER its members were moved, so they
1250
+ #: can never appear in `movedFiles`. Excluding them is what keeps a classified
1251
+ #: incident valid (§3.3).
1252
+ INCIDENT_CONTROL_METADATA = frozenset({"manifest.json", "classification.json"})
1253
+
1254
+ #: The statuses a rebuild record may terminally carry. A missing or
1255
+ #: unrecognized status is INVALID rather than terminal, because a record whose
1256
+ #: rebuild never reported an outcome may still describe live work.
1257
+ TERMINAL_REBUILD_STATUSES = frozenset({"ok", "failed"})
1258
+
1259
+ _MACHINE_BACKUP_RE = re.compile(r"\.bak-corrupt-malformed-\d{8}T\d{6}Z$")
1260
+ _USER_BACKUP_RE = re.compile(r"\.bak-\d{8}T\d{6}Z$")
1261
+
1262
+
1263
+ def validate_incident(*, manifest, observed) -> bool:
1264
+ """An incident is valid when its manifest agrees with what is on disk.
1265
+
1266
+ `observed` is every entry name directly inside the incident directory.
1267
+ Control metadata is excluded from the comparison because it is written
1268
+ after the move — an incident that has since been classified would
1269
+ otherwise become invalid, and therefore protected, precisely because it
1270
+ was classified.
1271
+ """
1272
+ if not isinstance(manifest, dict):
1273
+ return False
1274
+ moved = manifest.get("movedFiles")
1275
+ if not isinstance(moved, list) or any(
1276
+ not isinstance(name, str) for name in moved
1277
+ ):
1278
+ return False
1279
+ return set(moved) == set(observed) - INCIDENT_CONTROL_METADATA
1280
+
1281
+
1282
+ def incident_is_finalized(*, manifest, pending_marker_present) -> bool:
1283
+ """Whether the quarantine that produced this incident ran to completion.
1284
+
1285
+ §3.2 words the condition as "a `.quarantine-pending.json` marker is
1286
+ present, or `complete` is not `true`". The second half of that reading is
1287
+ wrong on this tree, and the first half is what actually carries the
1288
+ invariant.
1289
+
1290
+ Both quarantine paths write `manifest.json` only AFTER every member has
1291
+ been renamed into the incident directory (`bin/_cctally_db.py:1409` and
1292
+ `:1487`), and the strict path unlinks its pending record only after that.
1293
+ So a readable manifest already proves the move finished, and a crash
1294
+ mid-move leaves an incident with no manifest at all — which §3.3 rejects
1295
+ as invalid, and which is therefore protected anyway.
1296
+
1297
+ Reading an ABSENT `complete` key as unfinished would protect 22 of the 142
1298
+ incidents on the maintainer's store, holding 8.83 of its 18.2 GiB,
1299
+ including every cache incident §4.3's correlation was measured to unlock —
1300
+ which would make acceptance criterion 2 unreachable. Only an explicit
1301
+ `complete: false`, or a live pending marker naming this incident, means
1302
+ unfinished.
1303
+ """
1304
+ if pending_marker_present:
1305
+ return False
1306
+ if isinstance(manifest, dict) and manifest.get("complete") is False:
1307
+ return False
1308
+ return True
1309
+
1310
+
1311
+ def validate_bundle(payload) -> bool:
1312
+ """A forensics bundle is valid when it parses and carries a schema version.
1313
+
1314
+ Deliberately weaker than the incident rule: a successful synchronous
1315
+ `db rebuild` writes a bundle with no `trigger` at all
1316
+ (`bin/_cctally_db.py:1006` against the `:1426` call site), so demanding
1317
+ one would invalidate every bundle that command has ever written.
1318
+ """
1319
+ return isinstance(payload, dict) and payload.get("schemaVersion") is not None
1320
+
1321
+
1322
+ def validate_rebuild_record(payload) -> bool:
1323
+ """A rebuild record is valid only at an ENUMERATED terminal status."""
1324
+ if not isinstance(payload, dict):
1325
+ return False
1326
+ return payload.get("status") in TERMINAL_REBUILD_STATUSES
1327
+
1328
+
1329
+ def classification_applies(*, verdict, incident_name) -> bool:
1330
+ """Whether a `classification.json` classifies THIS incident (§3.3).
1331
+
1332
+ Binding the verdict to the directory it names prevents a classification
1333
+ file from authorizing deletion of a directory it was not written for.
1334
+ """
1335
+ if not isinstance(verdict, dict):
1336
+ return False
1337
+ if verdict.get("incident") != incident_name:
1338
+ return False
1339
+ return is_classified(verdict.get("confidence"))
1340
+
1341
+
1342
+ def incident_classification(*, manifest, verdict, incident_name):
1343
+ """The confidence that classifies an incident, or None (§3.3).
1344
+
1345
+ A `schemaVersion: 2` manifest with a truthy `trigger` classifies itself
1346
+ `exact`. A manifest claiming version 2 with no trigger is semantically
1347
+ invalid: it does NOT fall through to the sidecar, because a producer that
1348
+ stamped the version and omitted the trigger left a defect rather than a
1349
+ weaker verdict.
1350
+ """
1351
+ if isinstance(manifest, dict) and manifest.get("schemaVersion") == 2:
1352
+ trigger = manifest.get("trigger")
1353
+ if isinstance(trigger, str) and trigger:
1354
+ return "exact"
1355
+ return None
1356
+ if classification_applies(verdict=verdict, incident_name=incident_name):
1357
+ return verdict.get("confidence")
1358
+ return None
1359
+
1360
+
1361
+ def backup_origin(name: str) -> str:
1362
+ """Which of §3.7's three naming shapes a backup stem has.
1363
+
1364
+ ``machine`` is `db repair`'s own copy and is the only sweepable origin.
1365
+ ``user`` is `cctally db backup`, excluded by maintainer decision Q4.
1366
+ ``unknown`` is a hand-made copy, and this install carries several — which
1367
+ is why an unrecognized name is never deleted automatically.
1368
+ """
1369
+ if _MACHINE_BACKUP_RE.search(name):
1370
+ return "machine"
1371
+ if _USER_BACKUP_RE.search(name):
1372
+ return "user"
1373
+ return "unknown"
1374
+
1375
+
1376
+ def _backup_identities(entries):
1377
+ """`{name: (device, inode, size)}` for a backup family, or None if unusable."""
1378
+ if not isinstance(entries, list):
1379
+ return None
1380
+ identities = {}
1381
+ for entry in entries:
1382
+ if not isinstance(entry, dict):
1383
+ return None
1384
+ name = entry.get("name")
1385
+ if not isinstance(name, str) or name in identities:
1386
+ return None
1387
+ identities[name] = (
1388
+ entry.get("device"), entry.get("inode"), entry.get("size"),
1389
+ )
1390
+ return identities
1391
+
1392
+
1393
+ def backup_sidecar_applies(*, sidecar, observed) -> bool:
1394
+ """Whether a backup classification sidecar still describes this family.
1395
+
1396
+ Device and inode are what §3.3 names, and they are what stops a
1397
+ REPLACEMENT family written at the same machine-shaped stem from inheriting
1398
+ an earlier `exact` verdict and becoming deletable without ever having been
1399
+ classified. Size is compared as well, so an in-place rewrite of the same
1400
+ inode is caught too.
1401
+
1402
+ `mtime` is recorded by the writer but deliberately NOT compared, and the
1403
+ reason is availability rather than safety. Comparing it WOULD add safety: a
1404
+ filesystem may reuse an inode number once the recorded file is gone, and a
1405
+ device, inode and size match on a reused inode authorizes deleting a family
1406
+ this sidecar never described. And omitting it is the safe direction on the
1407
+ other side too — a copy or restore tool that perturbs mtime would make the
1408
+ family unclassified, which PROTECTS it.
1409
+
1410
+ The decision stands because that protection has no way back: nothing rewrites
1411
+ a backup sidecar, so a family protected by a perturbed mtime stays protected
1412
+ forever with no operator-facing remedy. The residual inode-reuse risk is
1413
+ bounded by the device and size comparison and by §3.7's rule that only a
1414
+ machine-shaped stem is ever swept.
1415
+ """
1416
+ if not isinstance(sidecar, dict):
1417
+ return False
1418
+ if not is_classified(sidecar.get("confidence")):
1419
+ return False
1420
+ recorded = _backup_identities(sidecar.get("members"))
1421
+ present = _backup_identities(observed)
1422
+ if recorded is None or present is None or not recorded:
1423
+ return False
1424
+ return recorded == present
1425
+
1426
+
1427
+ def backup_classification(*, sidecar, observed):
1428
+ """The confidence a backup family carries, or None (§3.3)."""
1429
+ if backup_sidecar_applies(sidecar=sidecar, observed=observed):
1430
+ return sidecar.get("confidence")
1431
+ return None
1432
+
1433
+
1434
+ # --------------------------------------------------------------------------
1435
+ # §4.3 — family-parameterized incident classification
1436
+ # --------------------------------------------------------------------------
1437
+
1438
+
1439
+ @dataclasses.dataclass(frozen=True)
1440
+ class Verdict:
1441
+ """One classification decision about one incident directory.
1442
+
1443
+ `incident` names the directory this verdict describes, so a
1444
+ `classification.json` can never authorize deletion of a directory it was
1445
+ not written for (§3.3).
1446
+ """
1447
+
1448
+ schema_version: int
1449
+ method: str
1450
+ confidence: str
1451
+ incident: str
1452
+ trigger: "str | None"
1453
+ candidates: "tuple[str, ...]"
1454
+ forensics_path: "str | None"
1455
+ evidence: dict
1456
+
1457
+
1458
+ #: The trigger origins a *correlated but trigger-less* bundle leaves open, per
1459
+ #: family. A v1 bundle carries no `trigger` object at all, so correlation can
1460
+ #: only narrow the cause to the producers that write one for that family.
1461
+ FAMILY_CANDIDATE_TRIGGERS: "dict[str, tuple[str, ...]]" = {
1462
+ "stats.db": ("corruption-heal", "db-rebuild"),
1463
+ "cache.db": ("cache.open", "cache.sync"),
1464
+ "conversations.db": ("conversations.open", "conversations.recovery"),
1465
+ }
1466
+
1467
+ #: A rebuild measured at 71 to 77 seconds, so the window is bounded well above
1468
+ #: that while staying far too small to reach an unrelated incident. The value
1469
+ #: `bin/cctally-classify-incidents` has always used.
1470
+ DEFAULT_CORRELATION_WINDOW_SECONDS = 600
1471
+
1472
+ #: The classification record's own schema version, matching the one the
1473
+ #: shipped correlator writes.
1474
+ CLASSIFICATION_SCHEMA_VERSION = 1
1475
+
1476
+
1477
+ def nearest_preceding_bundle(bundles, when):
1478
+ """The latest bundle at or before `when`, or None.
1479
+
1480
+ Ported unchanged from `bin/cctally-classify-incidents:130`; only the tuple
1481
+ shape widens, from `(stamp, path)` to `(stamp, path, payload)`. `bundles`
1482
+ is ascending by time and already filtered to one family. The window is NOT
1483
+ applied here, exactly as in the original — the caller applies it.
1484
+ """
1485
+ best = None
1486
+ for entry in bundles:
1487
+ if entry[0] <= when:
1488
+ best = entry
1489
+ else:
1490
+ break
1491
+ return best
1492
+
1493
+
1494
+ def classify_incident(
1495
+ *,
1496
+ family: str,
1497
+ incident_name: str,
1498
+ manifest: "dict | None",
1499
+ bundles,
1500
+ incident_time,
1501
+ window_seconds: int = DEFAULT_CORRELATION_WINDOW_SECONDS,
1502
+ ) -> Verdict:
1503
+ """Render a verdict for one incident of one family (§4.3).
1504
+
1505
+ Precedence: a self-classifying v2 manifest wins outright; otherwise the
1506
+ nearest preceding bundle inside the window decides — `exact` when it names
1507
+ its own `trigger.origin`, `candidate` when it does not — and otherwise the
1508
+ incident stays `unknown` and therefore protected.
1509
+
1510
+ A `schemaVersion: 2` manifest with no truthy trigger is semantically
1511
+ invalid: it claims to classify itself and does not. It is reported
1512
+ `unknown` rather than falling through to correlation, because correlating
1513
+ an incident whose own manifest contradicts itself would dress a defect up
1514
+ as evidence.
1515
+ """
1516
+ manifest = manifest if isinstance(manifest, dict) else {}
1517
+ candidates = FAMILY_CANDIDATE_TRIGGERS.get(family, ())
1518
+ manifest_version = manifest.get("schemaVersion")
1519
+
1520
+ if manifest_version == 2:
1521
+ trigger = manifest.get("trigger")
1522
+ if isinstance(trigger, str) and trigger:
1523
+ forensics = manifest.get("forensicsPath")
1524
+ return Verdict(
1525
+ schema_version=CLASSIFICATION_SCHEMA_VERSION,
1526
+ method="manifest-v2",
1527
+ confidence="exact",
1528
+ incident=incident_name,
1529
+ trigger=trigger,
1530
+ candidates=(),
1531
+ forensics_path=forensics if isinstance(forensics, str) else None,
1532
+ evidence={
1533
+ "manifestSchemaVersion": 2,
1534
+ "triggerError": manifest.get("triggerError"),
1535
+ "binaryVersion": manifest.get("binaryVersion"),
1536
+ },
1537
+ )
1538
+ return Verdict(
1539
+ schema_version=CLASSIFICATION_SCHEMA_VERSION,
1540
+ method="header-only",
1541
+ confidence="unknown",
1542
+ incident=incident_name,
1543
+ trigger=None,
1544
+ candidates=candidates,
1545
+ forensics_path=None,
1546
+ evidence={
1547
+ "manifestSchemaVersion": manifest_version,
1548
+ "windowSeconds": window_seconds,
1549
+ "reason": (
1550
+ "a schemaVersion 2 manifest without a trigger is "
1551
+ "semantically invalid"
1552
+ ),
1553
+ },
1554
+ )
1555
+
1556
+ match = (
1557
+ None
1558
+ if incident_time is None
1559
+ else nearest_preceding_bundle(bundles, incident_time)
1560
+ )
1561
+ if match is not None:
1562
+ when, path, bundle = match
1563
+ gap = int((incident_time - when).total_seconds())
1564
+ if gap <= window_seconds:
1565
+ trigger_block = bundle.get("trigger") if isinstance(bundle, dict) else None
1566
+ origin = (
1567
+ trigger_block.get("origin")
1568
+ if isinstance(trigger_block, dict)
1569
+ else None
1570
+ )
1571
+ evidence = {
1572
+ "manifestSchemaVersion": manifest_version,
1573
+ "gapSeconds": gap,
1574
+ "windowSeconds": window_seconds,
1575
+ }
1576
+ if isinstance(origin, str) and origin:
1577
+ # The bundle names its own trigger. Calling that `candidate`
1578
+ # would understate the evidence: cache and conversations
1579
+ # producers pass a trigger into `write_corruption_forensics`,
1580
+ # and `tests/test_cache_corruption_recovery.py:994` pins it.
1581
+ return Verdict(
1582
+ schema_version=CLASSIFICATION_SCHEMA_VERSION,
1583
+ method="forensics-trigger",
1584
+ confidence="exact",
1585
+ incident=incident_name,
1586
+ trigger=origin,
1587
+ candidates=(),
1588
+ forensics_path=path,
1589
+ evidence=evidence,
1590
+ )
1591
+ return Verdict(
1592
+ schema_version=CLASSIFICATION_SCHEMA_VERSION,
1593
+ method="forensics-correlation",
1594
+ confidence="candidate",
1595
+ incident=incident_name,
1596
+ trigger=None,
1597
+ candidates=candidates,
1598
+ forensics_path=path,
1599
+ evidence={
1600
+ **evidence,
1601
+ "reason": (
1602
+ "a bundle carrying no trigger cannot distinguish the "
1603
+ "producers of this family"
1604
+ ),
1605
+ },
1606
+ )
1607
+
1608
+ return Verdict(
1609
+ schema_version=CLASSIFICATION_SCHEMA_VERSION,
1610
+ method="header-only",
1611
+ confidence="unknown",
1612
+ incident=incident_name,
1613
+ trigger=None,
1614
+ candidates=candidates,
1615
+ forensics_path=None,
1616
+ evidence={
1617
+ "manifestSchemaVersion": manifest_version,
1618
+ "windowSeconds": window_seconds,
1619
+ "reason": (
1620
+ "no forensics bundle within the window preceding this incident"
1621
+ ),
1622
+ },
1623
+ )
1624
+
1625
+
1626
+ def verdict_to_record(verdict: Verdict) -> "dict[str, object]":
1627
+ """The persisted `classification.json` body, byte-comparable across runs.
1628
+
1629
+ Keeps the key names `bin/cctally-classify-incidents` already writes so a
1630
+ reader does not have to know which producer wrote the file.
1631
+ """
1632
+ record: "dict[str, object]" = {
1633
+ "schemaVersion": verdict.schema_version,
1634
+ "incident": verdict.incident,
1635
+ "method": verdict.method,
1636
+ "confidence": verdict.confidence,
1637
+ "evidence": verdict.evidence,
1638
+ }
1639
+ if verdict.trigger is not None:
1640
+ record["trigger"] = verdict.trigger
1641
+ if verdict.candidates:
1642
+ record["candidates"] = list(verdict.candidates)
1643
+ if verdict.forensics_path is not None:
1644
+ record["forensicsPath"] = verdict.forensics_path
1645
+ return record