sidegraph 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sidegraph/store.py ADDED
@@ -0,0 +1,3363 @@
1
+ """Owned, append-only decision store — git-native canonical files + a derived local index.
2
+
3
+ The store is a **repo-committed sidecar** — never the engine's ``graph.json`` (which is
4
+ read-only input and regenerated on every commit). It enforces the write-path invariants
5
+ (see CLAUDE.md and ``docs/concepts/data-model.md``):
6
+
7
+ - append-only: no hard deletes; reversal = ``valid_to`` + a new ``supersedes`` record;
8
+ - ``valid_to >= valid_from`` (also checked in the Pydantic model);
9
+ - a ``superseded`` decision must have a successor;
10
+ - every :class:`AnchorBinding` references an existing :class:`Entity`;
11
+ - provenance is always present;
12
+ - the committed store is **sync-clean**: a graph rebuild must never produce a git diff.
13
+
14
+ See ``docs/reference/store-format.md`` for the full layout reference.
15
+ A single repo-committed SQLite file cannot be merged by git (two branches ratifying
16
+ different decisions collide on one binary blob with no meaningful resolution), so the
17
+ canonical, git-committed record is now **file-per-record JSON** under ``self.path``::
18
+
19
+ decisions/<ulid>.json full Decision row (status/valid_to included — human facts)
20
+ facts/<ulid>.json full Fact row (status/valid_to included — non-derivable facts)
21
+ domains/<ulid>.json Domain row MINUS the volatile `communities` field
22
+ entities/<ulid>.json identity ONLY: entity_id, canonical_name, kind, descriptor
23
+ bindings/<decision_ulid>.json that decision's anchors: [{entity_id, tier, relation,
24
+ weight}] — NO status
25
+ initiatives/<ulid>.json full Initiative row (no volatile fields on this model)
26
+ index.db DERIVED, gitignored: every table below plus the volatile
27
+ fields (entity last_seen_*, binding status, domain
28
+ communities, capture ledger, toc_cache, schema/digest meta)
29
+
30
+ ``index.db`` keeps the exact same tables/queries this store has always used — reads are
31
+ unchanged. Writes to a canonical record update the matching file (atomic: tmp + ``os.replace``)
32
+ **and** the index, under the same lock. A write that touches ONLY a sanctioned volatile
33
+ field (an entity's engine mapping, a binding's status, a domain's ``communities`` — see
34
+ ``sync.py``) touches ONLY the index; the committed files are untouched, so a
35
+ ``sidegraph-sync`` pass never dirties git (Invariant #1, extended).
36
+
37
+ Thread-safety: fastmcp 3 dispatches sync ``@mcp.tool`` calls onto worker threads, so a
38
+ single process-wide :class:`Store` (see ``server.py``) is used from multiple threads. The
39
+ index connection is opened with ``check_same_thread=False`` and every use of it (reads and
40
+ writes alike) is serialized through an instance-level :class:`threading.RLock`. Iterator
41
+ methods fetch eagerly under the lock and yield afterwards, so the lock is never held across
42
+ a paused generator.
43
+ """
44
+
45
+ from __future__ import annotations
46
+
47
+ import hashlib
48
+ import json
49
+ import os
50
+ import shutil
51
+ import sqlite3
52
+ import sys
53
+ import tempfile
54
+ import threading
55
+ import time
56
+ import uuid
57
+ from collections.abc import Callable, Iterable, Iterator, Sequence
58
+ from contextlib import contextmanager, suppress
59
+ from datetime import UTC, datetime, timedelta
60
+ from pathlib import Path
61
+ from typing import Literal
62
+
63
+ from pydantic import BaseModel, Field
64
+
65
+ from .schema import (
66
+ SCHEMA_VERSION,
67
+ AnchorBinding,
68
+ Decision,
69
+ DecisionStatus,
70
+ Descriptor,
71
+ Domain,
72
+ DomainStatus,
73
+ Entity,
74
+ EntityKind,
75
+ Fact,
76
+ Initiative,
77
+ Scope,
78
+ canonicalize,
79
+ )
80
+
81
+ # One definition, two consumers: __init__ bootstraps with executescript (not inside a
82
+ # transaction, so its implicit COMMIT is harmless), while _reload_index_from_canonical must
83
+ # issue these individually -- executescript would commit the rebuild's transaction out from
84
+ # under it and re-publish the DROPs (design D2, measured).
85
+ _SCHEMA_STATEMENTS: tuple[str, ...] = (
86
+ """CREATE TABLE IF NOT EXISTS meta (
87
+ key TEXT PRIMARY KEY,
88
+ value TEXT NOT NULL
89
+ )""",
90
+ """CREATE TABLE IF NOT EXISTS entities (
91
+ entity_id TEXT PRIMARY KEY,
92
+ canonical_name TEXT NOT NULL,
93
+ data TEXT NOT NULL
94
+ )""",
95
+ """CREATE TABLE IF NOT EXISTS decisions (
96
+ id TEXT PRIMARY KEY,
97
+ status TEXT NOT NULL,
98
+ supersedes TEXT,
99
+ data TEXT NOT NULL
100
+ )""",
101
+ """CREATE TABLE IF NOT EXISTS facts (
102
+ id TEXT PRIMARY KEY,
103
+ status TEXT NOT NULL,
104
+ supersedes TEXT,
105
+ data TEXT NOT NULL
106
+ )""",
107
+ """CREATE TABLE IF NOT EXISTS anchor_bindings (
108
+ record_id TEXT NOT NULL,
109
+ entity_id TEXT NOT NULL,
110
+ data TEXT NOT NULL,
111
+ PRIMARY KEY (record_id, entity_id)
112
+ )""",
113
+ """CREATE TABLE IF NOT EXISTS initiatives (
114
+ id TEXT PRIMARY KEY,
115
+ name TEXT NOT NULL,
116
+ data TEXT NOT NULL
117
+ )""",
118
+ """CREATE TABLE IF NOT EXISTS capture_sessions (
119
+ session_id TEXT PRIMARY KEY,
120
+ captured_at TEXT NOT NULL
121
+ )""",
122
+ """CREATE TABLE IF NOT EXISTS domains (
123
+ domain_id TEXT PRIMARY KEY,
124
+ slug TEXT NOT NULL,
125
+ status TEXT NOT NULL,
126
+ supersedes TEXT,
127
+ data TEXT NOT NULL
128
+ )""",
129
+ """CREATE TABLE IF NOT EXISTS retrieval_shows (
130
+ record_id TEXT PRIMARY KEY,
131
+ shows INTEGER NOT NULL,
132
+ last_shown_at TEXT NOT NULL
133
+ )""",
134
+ """CREATE TABLE IF NOT EXISTS retrieval_seeds (
135
+ seed TEXT PRIMARY KEY,
136
+ queries INTEGER NOT NULL,
137
+ last_seen_at TEXT NOT NULL
138
+ )""",
139
+ """CREATE TABLE IF NOT EXISTS retrieval_events (
140
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
141
+ session_id TEXT NOT NULL,
142
+ at TEXT NOT NULL,
143
+ kind TEXT NOT NULL,
144
+ key TEXT NOT NULL,
145
+ detail TEXT
146
+ )""",
147
+ """CREATE INDEX IF NOT EXISTS idx_retrieval_events_session
148
+ ON retrieval_events (session_id)""",
149
+ """CREATE TABLE IF NOT EXISTS canonical_stat (
150
+ subdir TEXT NOT NULL,
151
+ stem TEXT NOT NULL,
152
+ size INTEGER NOT NULL,
153
+ mtime_ns INTEGER NOT NULL,
154
+ PRIMARY KEY (subdir, stem)
155
+ )""",
156
+ )
157
+
158
+ _SCHEMA_SQL = ";\n".join(_SCHEMA_STATEMENTS) + ";\n"
159
+
160
+ # Legacy (pre-0.4.0) single-SQLite-file stores that are safe to migrate forward on open (see
161
+ # ``Store._migrate_legacy`` and docs/reference/store-format.md#migration-to-040-from-02x-and-03x).
162
+ # Anything else (unknown/garbled/future versions) is a hard rejection — migration
163
+ # tooling beyond this one forward step remains deferred (CLAUDE.md invariant #3).
164
+ _MIGRATABLE_SCHEMA_VERSIONS = frozenset({"0.2.0", "0.3.0"})
165
+
166
+ # Stores stamped with these versions have fully forward-compatible canonical files;
167
+ # on open we rebuild the derived index (which re-stamps SCHEMA_VERSION) instead of
168
+ # raising. see design/superpowers/specs/2026-07-10-facts-layer-design.md
169
+ # "0.5.0" added for derived community bindings (see design/superpowers/specs/
170
+ # 2026-07-10-derived-community-bindings-design.md): a 0.5.0 store's canonical files are
171
+ # forward-compatible as-is (a reload just tolerates any pre-existing community entity/
172
+ # binding entries — see _is_derived_entity and _write_bindings_canonical_for_record).
173
+ _RELOADABLE_SCHEMA_VERSIONS = frozenset({"0.4.0", "0.5.0"})
174
+
175
+ # The canonical, git-committed record directories (relative to Store.path). Order matters
176
+ # only for readability; digest/reload iterate them in this order.
177
+ _CANONICAL_SUBDIRS = ("decisions", "facts", "domains", "entities", "bindings", "initiatives")
178
+
179
+ # Compaction (design §7): immutable archive segments. NOT one of _CANONICAL_SUBDIRS above —
180
+ # segments are packed multi-record files, not one-file-per-record, and the directory only
181
+ # comes into existence the first time ``Store.compact`` actually writes a segment (a store
182
+ # that never compacts has no ``archive/`` at all). Still fully canonical/git-committed: the
183
+ # freshness digest and the cold-load reload path both cover it (see
184
+ # ``_compute_canonical_digest`` / ``_reload_index_from_canonical`` / ``_archived_records``).
185
+ #
186
+ # Segment filenames: ``<date>-<seq>-<hash12>.jsonl`` (amended in N5 review, Minor-4;
187
+ # ``_parse_segment_seq`` still accepts the original bare ``<date>-<seq>.jsonl`` shape for
188
+ # any pre-amendment segment). ``<hash12>`` is the first 12 hex chars of a sha256 over the
189
+ # segment's own content (see ``Store._write_archive_segment``): without it, two branches
190
+ # each running ``sidegraph-compact`` on the SAME day could both pick the same
191
+ # ``<date>-<seq>`` name with DIFFERENT content — a real git-add/merge conflict on a file
192
+ # design §7 promised could "never merge-conflict". With the content baked into the name,
193
+ # different content always gets a different filename (both segments survive a merge, no
194
+ # conflict — the loader's ULID dedup absorbs any record overlap), and identical content
195
+ # always gets the identical name AND bytes (no conflict either — trivially the same file).
196
+ _ARCHIVE_SUBDIR = "archive"
197
+
198
+ # Terminal = a status after which append-only rules guarantee the record can never change
199
+ # again (see CLAUDE.md invariant #2 and design §7). PROPOSED/ACCEPTED decisions can still be
200
+ # ratified/dropped/superseded; PROPOSED/ACCEPTED domains can still be ratified/superseded —
201
+ # neither is ever a compaction candidate. DEPRECATED has no write path today (dead enum
202
+ # value — nothing in this codebase ever sets it) but is terminal by the same definition, so
203
+ # it is included for forward-compat rather than silently left hot forever the day something
204
+ # does start setting it.
205
+ _TERMINAL_DECISION_STATUSES = frozenset(
206
+ {DecisionStatus.SUPERSEDED, DecisionStatus.REJECTED, DecisionStatus.DEPRECATED}
207
+ )
208
+ _TERMINAL_DOMAIN_STATUSES = frozenset({DomainStatus.SUPERSEDED, DomainStatus.DROPPED})
209
+
210
+ # Committed format marker: a one-line ``<path>/format`` file (``sidegraph-store <version>``)
211
+ # written at layout creation and migration, checked on every open — see
212
+ # ``Store._ensure_format_marker``.
213
+ _FORMAT_MARKER_NAME = "format"
214
+ _FORMAT_MARKER_PREFIX = "sidegraph-store "
215
+
216
+ # Committed layout convenience (design §1/§5): ``<path>/.gitignore`` ignoring the derived
217
+ # index (and its sqlite WAL/SHM sidecars via the ``index.db*`` glob) and any crash-debris
218
+ # ``*.tmp`` — see ``Store._ensure_gitignore``.
219
+ _GITIGNORE_NAME = ".gitignore"
220
+ _GITIGNORE_CONTENT = "index.db*\n*.tmp\n"
221
+
222
+ # Committed creation marker: a one-line ``<path>/stamping_live_since`` file (an aware
223
+ # ISO-8601 UTC timestamp) written ONLY on the open that finds the store genuinely new — see
224
+ # ``Store._ensure_stamping_marker``. Unlike the format marker above, this one is never
225
+ # backfilled onto a pre-existing store: its whole job is to tell doctor.py's
226
+ # ``unratified-accept`` check the moment stamping capability became live for THIS store,
227
+ # so that check has a trustworthy scope-start even when the store has never ratified
228
+ # anything (see that check's docstring in doctor.py).
229
+ _STAMPING_MARKER_NAME = "stamping_live_since"
230
+
231
+ # Glob patterns matching the store's OWN root-level tmp artifacts (the format marker, the
232
+ # stamping marker, and .gitignore -- everything else lives under the canonical subdirs,
233
+ # _CANONICAL_SUBDIRS). Matches ``_atomic_write_text_race_tolerant``'s per-attempt UNIQUE tmp
234
+ # naming (``<name>.<random>.tmp``, never a fixed ``<name>.tmp`` -- see that function's
235
+ # docstring). Scoped by a real prefix/suffix pattern, not a blanket ``*.tmp`` glob
236
+ # (review Minor-3): the store root is also a directory a user may keep other files in (a
237
+ # build artifact, a scratch file, ...) -- sweeping every ``*.tmp`` there on open would
238
+ # delete something that was never ours.
239
+ _ROOT_TMP_GLOB_PATTERNS = (
240
+ f"{_FORMAT_MARKER_NAME}.*.tmp",
241
+ f"{_GITIGNORE_NAME}.*.tmp",
242
+ f"{_STAMPING_MARKER_NAME}.*.tmp",
243
+ )
244
+
245
+ # Attempts _atomic_write_text_race_tolerant makes before giving up and re-raising. Each
246
+ # concurrent opener sweeps stale tmp debris exactly once, at its own open start (see
247
+ # Store._sweep_stale_tmp_files) -- so a bounded handful of retries is enough for the set
248
+ # of "still mid-sweep" openers to converge to empty and a write to finally stick.
249
+ _RACE_TOLERANT_ATTEMPTS = 3
250
+
251
+ # How old a *.tmp file must be before the open-time sweep will remove it. The sweep cannot
252
+ # tell crash debris from another process's in-flight buffer, and it used to delete live ones
253
+ # (design D4): a buffer exists for microseconds, debris is old by definition, so age is the
254
+ # one signal that separates them without a lock or a liveness check. The trade: debris from a
255
+ # crash less than a minute ago survives until the next open after that. Harmless -- the sweep
256
+ # exists to stop unbounded accumulation, not to be prompt.
257
+ _TMP_DEBRIS_MIN_AGE_SECONDS = 60.0
258
+
259
+ # Attempts Store._write_archive_segment makes to publish a new segment under an unused
260
+ # name before giving up (review Important-1). Higher than _RACE_TOLERANT_ATTEMPTS above:
261
+ # that helper's racers are all writing IDENTICAL content and converge once the single
262
+ # winner lands, but two concurrent compacts can pick genuinely DIFFERENT terminal-record
263
+ # sets (different content, different hash suffix -- see _ARCHIVE_SUBDIR's docstring
264
+ # amendment) that only collide on the human-readable <date>-<seq> prefix; each retry here
265
+ # can therefore collide with a DIFFERENT racer, not just re-attempt against one winner.
266
+ _ARCHIVE_SEGMENT_PUBLISH_ATTEMPTS = 10
267
+
268
+ # Meta key: set to "1" whenever __init__ reloads the index from canonical files (missing
269
+ # index, digest mismatch, or a fresh migration) and cleared by a completed `sidegraph-sync`
270
+ # pass (see sync.py) — see design §3.
271
+ VOLATILE_STALE_KEY = "volatile_stale"
272
+
273
+ _CANONICAL_DIGEST_KEY = "canonical_digest"
274
+
275
+
276
+ def _atomic_write_text(path: Path, text: str) -> os.stat_result:
277
+ """Write ``text`` to ``path`` atomically (tmp + ``os.replace``, same directory so the
278
+ replace is same-filesystem). A crash strictly between the tmp write and the replace
279
+ leaves the ORIGINAL file (or its absence) intact — see design §2/§Testing. Writes the
280
+ canonical records (via ``_atomic_write_json`` below); the format marker and
281
+ ``.gitignore`` do NOT come through here — they use
282
+ ``_atomic_write_text_race_tolerant``, which has its own tmp+replace implementation.
283
+
284
+ The tmp name is per-call UNIQUE (``<name>.<pid>.<random>.tmp``), matching
285
+ ``_atomic_write_text_race_tolerant`` (design D5): a FIXED ``<name>.tmp`` means two
286
+ writers of the SAME target share one inode, so one's ``open(..., "w")`` truncation can
287
+ land inside the other's ``os.replace`` window and atomically install an EMPTY file as
288
+ the "committed" canonical record — silent content corruption, not a crash. That
289
+ corruption class is already known-real on this project (it is why the race-tolerant
290
+ variant has unique names); this closes the same hole for every canonical record write.
291
+ The name still ends in ``.tmp``, so the sweep's globs (``_sweep_stale_tmp_files``) and
292
+ the store's own ``.gitignore`` (``*.tmp``) still cover it.
293
+
294
+ Returns the TMP file's stat, taken BEFORE the replace (digest-integrity design §3, the
295
+ rule the whole ``canonical_stat`` mechanism rests on): the stat a caller records must
296
+ describe the content it is about to publish, captured at or before the write becomes
297
+ visible, never after — a stat taken after the replace could describe some OTHER
298
+ writer's file that replaced this one in the interim (two sessions rewriting the same
299
+ record, replace order inverted against commit order). ``os.replace`` preserves the
300
+ inode's ``(size, mtime_ns)`` (POSIX; measured), so the tmp's pre-replace stat is exactly
301
+ the published file's stat too — no second ``stat()`` call needed, and no window in
302
+ which to observe someone else's replace instead of this one."""
303
+ path.parent.mkdir(parents=True, exist_ok=True)
304
+ tmp = path.with_name(f"{path.name}.{os.getpid()}.{uuid.uuid4().hex[:12]}.tmp")
305
+ tmp.write_text(text, encoding="utf-8")
306
+ st = tmp.stat()
307
+ os.replace(tmp, path)
308
+ return st
309
+
310
+
311
+ def _atomic_write_json(path: Path, obj: object) -> os.stat_result:
312
+ """Write ``obj`` as pretty, stably-sorted JSON to ``path`` atomically (see
313
+ ``_atomic_write_text``, including the returned pre-replace stat).
314
+
315
+ ``ensure_ascii=False``: non-ASCII decision prose (Cyrillic, em-dashes, ...) must appear
316
+ verbatim in the committed file, not as ``\\uXXXX`` escapes — these files are meant to be
317
+ read and diffed by humans in a normal git workflow."""
318
+ text = json.dumps(obj, sort_keys=True, indent=2, ensure_ascii=False) + "\n"
319
+ return _atomic_write_text(path, text)
320
+
321
+
322
+ def _atomic_write_text_race_tolerant(path: Path, text: str) -> None:
323
+ """Like ``_atomic_write_text``, but safe for the specific way ``Store``'s FIRST-OPEN
324
+ writes (the format marker, ``.gitignore``) can race across THREADS (fastmcp's
325
+ worker-thread dispatch) or genuinely separate PROCESSES (a hook racing a long-lived
326
+ MCP server, or two CLI invocations opening the same fresh directory at once) — review
327
+ Important-2a plus a residual finding a cross-process stress probe surfaced on top of
328
+ it. Both writers always produce IDENTICAL content (the same running
329
+ ``SCHEMA_VERSION`` / the same fixed ``.gitignore`` body), so there is never a real
330
+ disagreement to resolve — only races over WHO gets to write it.
331
+
332
+ Unlike ``_atomic_write_text``, each attempt here writes to a per-attempt UNIQUE tmp
333
+ filename (``<name>.<random>.tmp``, via ``uuid4``), never a fixed ``<name>.tmp``. This
334
+ matters beyond just avoiding a name clash: a FIXED shared tmp name lets one writer's
335
+ ``open(..., "w")`` (which truncates in place) land on the SAME inode ANOTHER writer's
336
+ ``os.replace`` is about to consume — the second writer's truncate can zero out the
337
+ first writer's already-written bytes a moment before that first writer's replace call
338
+ fires, atomically installing an EMPTY file as the "committed" marker (silent content
339
+ corruption, not a crash — this is what a cross-process stress probe caught: an
340
+ ``unrecognized store format marker ''`` on a subsequent open). A unique tmp name per
341
+ attempt makes that impossible: no two attempts, in this process or any other, ever
342
+ write through the same path, so nothing can be truncated out from under a writer that
343
+ still owns its own tmp file.
344
+
345
+ What a unique name does NOT prevent: another process's open-time debris SWEEP
346
+ (``Store._sweep_stale_tmp_files``) deleting THIS attempt's tmp file before its
347
+ ``os.replace`` runs — the sweep has no way to tell a truly stale leftover from a live
348
+ in-flight buffer, unique name or not. That still surfaces as ``FileNotFoundError`` on
349
+ ``os.replace``. If the target already exists by then, some attempt (ours or a
350
+ concurrent one) already landed the identical content, so it's a success, not a
351
+ failure. If not, the WHOLE cycle (a fresh unique tmp, a fresh write, a fresh replace)
352
+ is retried, up to ``_RACE_TOLERANT_ATTEMPTS`` times: every opener sweeps only once, at
353
+ its own open start, so the set of "still mid-sweep" openers shrinks to empty within a
354
+ bounded number of rounds and a later attempt sticks. Exhausting every attempt
355
+ re-raises the last ``FileNotFoundError`` — a genuine, non-race failure (e.g. the
356
+ store's parent directory itself disappeared) must not be swallowed forever.
357
+ """
358
+ path.parent.mkdir(parents=True, exist_ok=True)
359
+ last_error: FileNotFoundError | None = None
360
+ for _ in range(_RACE_TOLERANT_ATTEMPTS):
361
+ tmp = path.with_name(f"{path.name}.{os.getpid()}.{uuid.uuid4().hex[:12]}.tmp")
362
+ tmp.write_text(text, encoding="utf-8")
363
+ try:
364
+ os.replace(tmp, path)
365
+ return
366
+ except FileNotFoundError as e:
367
+ if path.is_file():
368
+ return
369
+ last_error = e
370
+ assert last_error is not None # the loop always sets it before falling through
371
+ raise last_error
372
+
373
+
374
+ def _store_has_any_records(path: Path) -> bool:
375
+ """True iff the store rooted at ``path`` holds a record of ANY kind — the complete
376
+ inventory of places one can live, per this module's own docstring layout: a hot file
377
+ under one of the six ``_CANONICAL_SUBDIRS`` (``*.json``), or an archive segment under
378
+ ``_ARCHIVE_SUBDIR`` (``*.jsonl``, ``Store.compact``'s output). Both are checked because
379
+ ``Store.compact`` can move EVERY decision/domain out of its hot directory into an
380
+ archive segment — a store in that state has zero hot files but is emphatically not
381
+ new; only ``archive/`` still proves it. Used solely to decide whether
382
+ ``Store._ensure_stamping_marker`` may write its creation marker: that marker must
383
+ never appear on a store that already had a record, by any route, before this open."""
384
+ for sub in _CANONICAL_SUBDIRS:
385
+ d = path / sub
386
+ if d.is_dir() and any(d.glob("*.json")):
387
+ return True
388
+ archive_dir = path / _ARCHIVE_SUBDIR
389
+ return archive_dir.is_dir() and any(archive_dir.glob("*.jsonl"))
390
+
391
+
392
+ def _entity_identity_payload(entity: Entity) -> dict:
393
+ """The canonical (git-committed) subset of an Entity: identity only, never the
394
+ engine-mapping fields (``last_seen_*``) sync refreshes — see design §1."""
395
+ return {
396
+ "entity_id": entity.entity_id,
397
+ "canonical_name": entity.canonical_name,
398
+ "kind": entity.kind.value,
399
+ "descriptor": entity.descriptor.model_dump() if entity.descriptor else None,
400
+ }
401
+
402
+
403
+ def _domain_canonical_payload(domain: Domain) -> dict:
404
+ """The canonical (git-committed) subset of a Domain: everything except the volatile
405
+ ``communities`` mapping, which ``sync.py``'s ``refresh_domain_communities`` refreshes
406
+ index-only — see design §1."""
407
+ data = domain.model_dump(mode="json")
408
+ data.pop("communities", None)
409
+ return data
410
+
411
+
412
+ def _is_derived_entity(entity: Entity) -> bool:
413
+ """True iff ``entity`` is engine-volatile DERIVED state that must never reach a
414
+ canonical (git-committed) file — currently: Tier-1 community abstract entities
415
+ (``canonical_name`` starting with ``"community:"``). Communities are snapshot labels
416
+ Leiden renumbers on every rebuild, not stable identities — see design/superpowers/specs/
417
+ 2026-07-10-derived-community-bindings-design.md. The scope condition is the ENTITY, not
418
+ the tier number: ``domain:*``/``tag:*``/initiative abstract entities stay canonical."""
419
+ return entity.kind == EntityKind.ABSTRACT and entity.canonical_name.startswith("community:")
420
+
421
+
422
+ def _binding_identity_payload(binding: AnchorBinding) -> dict:
423
+ """The canonical (git-committed) subset of an AnchorBinding: the anchor itself, never
424
+ ``status`` (live/degraded/orphaned), which sync flips index-only — see design §1.
425
+ ``record_id`` is implied by the enclosing ``bindings/<record_id>.json`` file, so it
426
+ is not part of this payload either."""
427
+ return {
428
+ "entity_id": binding.entity_id,
429
+ "tier": binding.tier,
430
+ "relation": binding.relation,
431
+ "weight": binding.weight,
432
+ }
433
+
434
+
435
+ def _archive_record_line(record_type: Literal["decision", "domain"], payload: dict) -> str:
436
+ """One ``archive/<date>-<seq>.jsonl`` line: the record's canonical payload (exactly
437
+ what its hot file contained — see ``_write_decision_canonical`` /
438
+ ``_domain_canonical_payload``) plus a ``record_type`` discriminator so the loader knows
439
+ which model to reconstruct (design §7). Same ``json.dumps`` settings as
440
+ ``_atomic_write_json`` (``sort_keys=True``, ``ensure_ascii=False``) minus ``indent`` —
441
+ flattened onto one line, since a segment is one record per line, not one record per
442
+ file."""
443
+ return json.dumps({"record_type": record_type, **payload}, sort_keys=True, ensure_ascii=False)
444
+
445
+
446
+ def _warn_hot_archive_mismatch(kind: str, record_id: str) -> None:
447
+ """A ULID present in BOTH a hot canonical file and an archive segment with DIFFERENT
448
+ content — terminal records are supposed to be immutable once archived, so this is
449
+ corruption-shaped (e.g. a hand-edited archive segment, or a bug). Never destroy data:
450
+ the hot file always wins (see ``_reload_index_from_canonical`` / ``Store.compact``);
451
+ this only surfaces the disagreement for a human to investigate."""
452
+ print(
453
+ f"sidegraph: WARNING {kind} {record_id!r} exists in BOTH a hot canonical file and "
454
+ "an archive segment with DIFFERENT content — archive segments are meant to be "
455
+ "immutable; keeping the hot file and ignoring the stale archive copy. This is "
456
+ "corruption-shaped (e.g. a hand-edited archive segment); investigate manually.",
457
+ file=sys.stderr,
458
+ )
459
+
460
+
461
+ def _warn_domain_slug_conflict(slug: str, domain_ids: list[str]) -> None:
462
+ """A slug held by more than one LIVE (proposed|accepted) domain at once — design §6's
463
+ cross-branch race: two branches independently proposed/accepted a domain with the same
464
+ slug; both files merge in cleanly (different ``domain_id``s, no file conflict), so
465
+ nothing at write time ever catches this after the merge. ``find_domain_by_slug``
466
+ still resolves deterministically in the meantime (accepted > proposed, newest first),
467
+ so retrieval never breaks — but the duplicate itself is never auto-resolved (never
468
+ guess which one a human meant to keep); this only surfaces it, same convention as
469
+ ``_warn_hot_archive_mismatch``."""
470
+ print(
471
+ f"sidegraph: WARNING domain slug {slug!r} is held by {len(domain_ids)} live "
472
+ f"domains ({', '.join(domain_ids)}) — likely a cross-branch merge race (design "
473
+ "§6); drop one (sidegraph-ratify --drop <loser-id>), or supersede it under a "
474
+ "different slug.",
475
+ file=sys.stderr,
476
+ )
477
+
478
+
479
+ def _parse_segment_seq(stem: str, date_str: str) -> int | None:
480
+ """The ``<seq>`` component of an archive segment's filename stem, given its date.
481
+
482
+ Tolerates both the current ``<date>-<seq>-<hash12>.jsonl`` shape and the original
483
+ bare ``<date>-<seq>.jsonl`` shape (review Minor-4: the hash suffix was added after
484
+ N5 shipped, to fix same-day cross-branch compacts colliding on filename — see
485
+ ``_ARCHIVE_SUBDIR``'s docstring and design §7's amendment note. No released store
486
+ ever wrote the bare shape, but nothing stops a hand-authored/older fixture from
487
+ having one, and the loader must never choke on it). Returns ``None`` for a stem that
488
+ doesn't start with ``<date_str>-`` or whose seq component isn't a plain integer."""
489
+ prefix = f"{date_str}-"
490
+ if not stem.startswith(prefix):
491
+ return None
492
+ rest = stem[len(prefix) :]
493
+ seq_part = rest.split("-", 1)[0]
494
+ return int(seq_part) if seq_part.isdigit() else None
495
+
496
+
497
+ def _hot_file_matches(path: Path, expected_payload: dict) -> bool | None:
498
+ """Compare a hot canonical file's ON-DISK content against ``expected_payload`` (never
499
+ trust an in-memory snapshot for a decision as destructive as unlinking a file — review
500
+ Minor-6). Returns ``True``/``False`` if the file exists and does/doesn't match, or
501
+ ``None`` if there is no hot file there at all (nothing to compare, nothing to
502
+ remove)."""
503
+ if not path.is_file():
504
+ return None
505
+ return json.loads(path.read_text(encoding="utf-8")) == expected_payload
506
+
507
+
508
+ class CompactedRecord(BaseModel):
509
+ """One line of ``sidegraph-compact``'s report / ``--dry-run`` listing — see
510
+ :meth:`Store.compact`."""
511
+
512
+ ulid: str
513
+ kind: Literal["decision", "domain"]
514
+ status: str
515
+ title: str
516
+
517
+
518
+ class CompactReport(BaseModel):
519
+ """Result of :meth:`Store.compact` (design §7) — mirrors the report-object convention
520
+ other feature modules use (e.g. ``sync.py``'s ``RebindOutcome``)."""
521
+
522
+ #: The new segment's path relative to the store root (e.g.
523
+ #: ``"archive/2026-07-08-1-a1b2c3d4e5f6.jsonl"`` — ``<date>-<seq>-<hash12>``), or
524
+ #: ``None`` when no NEW segment was written (nothing selected, or ``dry_run=True``).
525
+ segment_path: str | None = None
526
+ decisions_compacted: int = 0
527
+ domains_compacted: int = 0
528
+ #: TERMINAL decisions excluded because ``--older-than`` was given and either the
529
+ #: decision hasn't been terminal long enough, or (defensive) its ``valid_to`` is
530
+ #: unexpectedly unset. Decisions only — see ``domains_excluded_age_unknown`` below for
531
+ #: the separate domain count (review Minor-7: the two have different root causes and
532
+ #: read confusingly merged into one number — "too recent" vs. "no timestamp exists at
533
+ #: all to check").
534
+ skipped_age_filtered: int = 0
535
+ #: TERMINAL domains excluded because ``--older-than`` was given. Always every eligible
536
+ #: domain, unconditionally: unlike a decision's ``valid_to``, nothing on ``Domain``
537
+ #: records when it entered a terminal status, so there is no age to even evaluate.
538
+ domains_excluded_age_unknown: int = 0
539
+ #: Hot files removed because they duplicated a record ALREADY in an existing archive
540
+ #: segment (crash-window debris from a prior compact that wrote its segment but was
541
+ #: interrupted before removing the hot file) — no new segment was written for these.
542
+ cleaned_up_hot_files: int = 0
543
+ items: list[CompactedRecord] = Field(default_factory=list)
544
+ dry_run: bool = False
545
+
546
+ @property
547
+ def total_compacted(self) -> int:
548
+ return self.decisions_compacted + self.domains_compacted
549
+
550
+
551
+ def _ratifier_identity(actor: str | None = None) -> str | None:
552
+ """Best-effort `git config user.name` for the ratifier stamp (design D4), extended
553
+ for auto-ratification (design D2/T17) with an explicit override.
554
+
555
+ An explicit, non-blank ``actor`` — the auto-ratify stamp, ``"auto:<policy>"``
556
+ (design D2) — wins outright: no git lookup, returned as-is. ``actor`` that is blank
557
+ or whitespace-only is treated exactly like an absent one and falls through to the
558
+ git lookup below, never stored verbatim: the caller can never chain an empty or
559
+ whitespace string into a trust field, extending this function's existing principle
560
+ (a wrong identity is worse than an absent one) to a caller-supplied value as well as
561
+ a failed git lookup.
562
+
563
+ None on ANY failure — no git, no configured name, sandboxed env, or a blank actor
564
+ with no git identity behind it either. The stamp never guesses (the anchors'
565
+ never-guess stance, applied to identity): a wrong or blank name in a trust field is
566
+ worse than an absent one. Module-level so tests monkeypatch it.
567
+ """
568
+ if actor is not None and actor.strip():
569
+ return actor
570
+
571
+ import subprocess
572
+
573
+ try:
574
+ out = subprocess.run(
575
+ ["git", "config", "user.name"], capture_output=True, text=True, timeout=5
576
+ )
577
+ name = out.stdout.strip()
578
+ return name or None
579
+ except Exception:
580
+ return None
581
+
582
+
583
+ class Store:
584
+ """Thin persistence layer that owns the invariants. Use as a context manager.
585
+
586
+ ``Store(path)`` accepts a directory (canonical layout; created if missing — the
587
+ convention, default ``.sidegraph``) or a legacy single-file store (a bare ``*.db``
588
+ path, or a directory containing an un-migrated ``decisions.db``): opening either
589
+ triggers a one-time export to the canonical layout (see ``_migrate_legacy`` and design
590
+ §4). This is a persistence swap, not an API change — every public method, invariant,
591
+ and result shape is unchanged from the pre-0.4.0 single-file store.
592
+ """
593
+
594
+ def __init__(self, path: str | Path = ".sidegraph") -> None:
595
+ raw = Path(path)
596
+ self._lock = threading.RLock()
597
+ # Depth counter for `_mutation` (design D1/D2): only the OUTERMOST `_mutation` scope
598
+ # commits or rolls back -- see that method's docstring. Lives beside `_lock` since
599
+ # the two are set up together and only ever mutate together.
600
+ self._mutation_depth = 0
601
+ self.path = raw
602
+
603
+ if raw.is_file():
604
+ self._reject_file_inside_a_store(raw)
605
+ self._migrate_legacy(raw)
606
+ else:
607
+ legacy = raw / "decisions.db"
608
+ if legacy.is_file():
609
+ self._migrate_legacy(legacy)
610
+
611
+ self._ensure_canonical_subdirs()
612
+ self._sweep_stale_tmp_files()
613
+ self._ensure_format_marker()
614
+ self._ensure_stamping_marker()
615
+ self._ensure_gitignore()
616
+ self._conn = sqlite3.connect(str(self.path / "index.db"), check_same_thread=False)
617
+ try:
618
+ self._conn.row_factory = sqlite3.Row
619
+ self._conn.executescript(_SCHEMA_SQL)
620
+ self._refresh_freshness()
621
+ except BaseException:
622
+ self._conn.close()
623
+ raise
624
+
625
+ # -- canonical layout / migration ----------------------------------------
626
+
627
+ def _reject_file_inside_a_store(self, target: Path) -> None:
628
+ """A file sitting inside a canonical store is an artifact OF that store — never a
629
+ legacy single-file store, and never something to migrate.
630
+
631
+ ``index.db`` is the obvious-looking "the database" file, so opening it instead of the
632
+ store directory is an easy mistake. It used to fall through to ``_migrate_legacy``,
633
+ which read the derived index's own ``schema_version`` (the CURRENT one), found it
634
+ outside the migratable set, and reported a mismatch between two identical strings
635
+ while advising the operator to discard a perfectly healthy store. Fail here instead,
636
+ naming what to open.
637
+
638
+ Detection is the format marker, not the filename: any artifact next to a valid marker
639
+ is inside a store, whatever it is called.
640
+ """
641
+ marker = target.parent / _FORMAT_MARKER_NAME
642
+ if not marker.is_file():
643
+ return
644
+ try:
645
+ text = marker.read_text(encoding="utf-8")
646
+ except (OSError, UnicodeDecodeError):
647
+ return # unreadable marker: fall through, the ordinary paths will complain
648
+ if not text.startswith(_FORMAT_MARKER_PREFIX):
649
+ return
650
+ raise ValueError(
651
+ f"{target} is a file inside the sidegraph store at {target.parent} — open the "
652
+ f"store DIRECTORY, not a file within it (e.g. Store({str(target.parent)!r})). "
653
+ "Nothing is wrong with the store itself."
654
+ )
655
+
656
+ def _ensure_canonical_subdirs(self) -> None:
657
+ for sub in _CANONICAL_SUBDIRS:
658
+ (self.path / sub).mkdir(parents=True, exist_ok=True)
659
+
660
+ def _unlink_if_stale(self, path: Path) -> None:
661
+ """Remove ``path`` only if it is older than ``_TMP_DEBRIS_MIN_AGE_SECONDS`` (design
662
+ D4) — the age gate that keeps the sweep below from ever seeing a live in-flight tmp
663
+ buffer as a candidate for removal in the first place; see
664
+ ``_sweep_stale_tmp_files``. Tolerates ``FileNotFoundError`` from ``stat`` the same
665
+ TOCTOU way ``missing_ok=True`` already tolerates it on the unlink below: the file
666
+ can vanish between the caller's glob and this stat (another opener's sweep, or its
667
+ own writer's ``os.replace`` finally landing) — already gone is just as harmless as
668
+ never having found it there."""
669
+ try:
670
+ age = time.time() - path.stat().st_mtime
671
+ except FileNotFoundError:
672
+ return # already gone -- another opener's sweep, or its os.replace consumed it
673
+ if age > _TMP_DEBRIS_MIN_AGE_SECONDS:
674
+ path.unlink(missing_ok=True)
675
+
676
+ def _sweep_stale_tmp_files(self) -> None:
677
+ """Remove stray ``*.tmp`` files left by a crash strictly between a writer's tmp
678
+ write and its ``os.replace`` swap (see ``_atomic_write_text``/
679
+ ``_atomic_write_text_race_tolerant``): the original file (or its absence) is
680
+ already the durable truth, so a leftover tmp is pure crash debris — harmless to
681
+ the invariants but left unswept it would accumulate forever and could confuse a
682
+ naive directory listing. Covers both the canonical record dirs (a real ``*.tmp``
683
+ glob -- every file there is ours) and the store root (only the glob patterns
684
+ matching the store's OWN root-level tmp artifacts, ``_ROOT_TMP_GLOB_PATTERNS`` --
685
+ never a blanket root-level ``*.tmp`` glob, which would delete a file a user
686
+ happens to keep in the store root that was never ours; see review Minor-3).
687
+
688
+ Age-gated (design D4): a bare ``*.tmp`` glob can't distinguish crash debris from
689
+ another process's tmp file that is mid-write RIGHT NOW — this used to delete a live
690
+ buffer out from under its writer, whose only defence was
691
+ ``_atomic_write_text_race_tolerant``'s bounded retry, a constant (3) smaller than
692
+ the number of concurrent openers it has to survive above that concurrency
693
+ (``FileNotFoundError`` on ``os.replace``, 1/1000 opens measured). Every removal below
694
+ goes through ``_unlink_if_stale``, which only unlinks a ``*.tmp`` file older than
695
+ ``_TMP_DEBRIS_MIN_AGE_SECONDS`` — a live buffer exists for microseconds, so it is
696
+ never old enough to match; only genuine crash debris is. This replaces the retry
697
+ bound as the mechanism that keeps the sweep from destroying a live write; the retry
698
+ loop itself stays (a writer that stalls past the age gate — e.g. the machine
699
+ sleeps — still needs it, see design §5).
700
+
701
+ ``_unlink_if_stale`` internally uses ``missing_ok=True`` on the actual unlink
702
+ (review residual on Important-2a/Minor-3): a bare ``if f.is_file(): f.unlink()`` is
703
+ a check-then-act TOCTOU against another concurrent opener's ``os.replace``
704
+ consuming that exact tmp file between the check and the unlink -- a cross-process
705
+ stress probe hit this reliably. ``unlink``ing a path that's already gone by the
706
+ time the syscall runs is just as harmless as never having found it there at all, so
707
+ tolerating that (rather than crashing this open entirely) is correct, not merely
708
+ convenient.
709
+ """
710
+ for pattern in _ROOT_TMP_GLOB_PATTERNS:
711
+ for f in self.path.glob(pattern):
712
+ self._unlink_if_stale(f)
713
+ for sub in _CANONICAL_SUBDIRS:
714
+ d = self.path / sub
715
+ if not d.is_dir():
716
+ continue
717
+ for f in d.glob("*.tmp"):
718
+ self._unlink_if_stale(f)
719
+ # archive/ is NOT one of _CANONICAL_SUBDIRS (see its definition) — it only exists
720
+ # once a compact actually ran, and only its own *.tmp glob is swept, same as any
721
+ # other canonical dir; segment .jsonl files themselves are never touched here.
722
+ archive_dir = self.path / _ARCHIVE_SUBDIR
723
+ if archive_dir.is_dir():
724
+ for f in archive_dir.glob("*.tmp"):
725
+ self._unlink_if_stale(f)
726
+
727
+ def _ensure_format_marker(self) -> None:
728
+ """Committed, git-visible format marker (``<path>/format``, one line:
729
+ ``sidegraph-store <version>``) — a correctness gate that lives OUTSIDE the
730
+ gitignored ``index.db``, so a teammate on an incompatible store format is rejected
731
+ the moment they open it, not just when their local index happens to agree with
732
+ their code. Missing marker on an existing layout (N1-era stores predate this file)
733
+ is backfilled, never rejected. An unknown MAJOR component IS a hard rejection, same
734
+ spirit as the ``schema_version`` gate in ``_refresh_freshness``.
735
+
736
+ Written atomically (tmp + ``os.replace``, via ``_atomic_write_text_race_tolerant``):
737
+ a crash mid-write must never leave a truncated marker behind — that would
738
+ hard-reject every subsequent open of an otherwise-healthy store. Race-tolerant
739
+ because two openers can reach this on a store's first-ever open — see
740
+ ``_atomic_write_text_race_tolerant``'s docstring."""
741
+ marker = self.path / _FORMAT_MARKER_NAME
742
+ if not marker.is_file():
743
+ _atomic_write_text_race_tolerant(marker, f"{_FORMAT_MARKER_PREFIX}{SCHEMA_VERSION}\n")
744
+ return
745
+ text = marker.read_text(encoding="utf-8").strip()
746
+ if not text.startswith(_FORMAT_MARKER_PREFIX):
747
+ raise ValueError(
748
+ f"unrecognized store format marker {text!r} at {marker}; "
749
+ "migration tooling is deferred — use a fresh store"
750
+ )
751
+ stamped_version = text[len(_FORMAT_MARKER_PREFIX) :]
752
+ stamped_major = stamped_version.split(".", 1)[0]
753
+ current_major = SCHEMA_VERSION.split(".", 1)[0]
754
+ if stamped_major != current_major:
755
+ raise ValueError(
756
+ f"store format marker {text!r} (major {stamped_major!r}) is incompatible "
757
+ f"with code's {SCHEMA_VERSION!r} (major {current_major!r}); "
758
+ "migration tooling is deferred — use a fresh store"
759
+ )
760
+
761
+ def _ensure_stamping_marker(self) -> None:
762
+ """Committed, git-visible creation marker (``<path>/stamping_live_since``, one
763
+ line: an aware UTC ISO-8601 timestamp) recording the moment THIS store was
764
+ genuinely new — written ONLY on the open that finds it new, never backfilled onto
765
+ a pre-existing store (contrast ``_ensure_format_marker``, which explicitly DOES
766
+ backfill).
767
+
768
+ Why this exists: ``doctor.py``'s ``unratified-accept`` check scopes its scan to
769
+ records created at/after the earliest ``ratified_at`` stamp anywhere in the store,
770
+ because the signature it looks for (accepted, agent-sourced, no ratifier stamp) is
771
+ indistinguishable from an ordinary pre-ratification-stamp record before that point
772
+ (see that check's docstring). A store that has never ratified anything has no
773
+ stamp to scope from, so it produces zero findings — forever, even after a later
774
+ ratification arms the check, because every record already in the store predates
775
+ that first stamp. A store adopting an auto-ratification policy from scratch starts
776
+ exactly there: inside the blind window, at the moment the gate matters most. This
777
+ marker gives that check an alternative scope-start that does not depend on any
778
+ ratification ever having happened: the instant stamping became live for this
779
+ store, recorded by a version that already writes this file.
780
+
781
+ "Genuinely new" is derived, not assumed: a store is new here iff, at the moment
782
+ this runs (after ``_ensure_canonical_subdirs`` has created the six canonical
783
+ subdirectories, so they exist to check), it holds NO record of any kind —
784
+ ``_store_has_any_records`` checks every hot canonical subdir AND ``archive/``,
785
+ which together are the complete inventory of where a record can live (module
786
+ docstring). That second half is deliberate: a store whose records were ALL moved
787
+ into archive segments by ``Store.compact`` has empty hot directories but is an
788
+ EXISTING store with real history, not a new one — treating it as new would stamp
789
+ ``stamping_live_since`` at the moment of this open, years after the store (and its
790
+ never-ratified records, if any) actually came into being, which is not "no
791
+ backfill", it is exactly the backfill this marker must never do. A store that is
792
+ merely emptied, never archived (nothing has ever been written to it), correctly
793
+ has no records anywhere and correctly counts as new.
794
+
795
+ Idempotent by construction, the same way ``_ensure_format_marker`` is: once
796
+ written, the marker file itself is present, so every later open of this same
797
+ store — however many records it accumulates in between — takes the `is_file()`
798
+ branch below and never re-derives or rewrites it. A legacy store that already had
799
+ records the very first time a stamping version opened it never has this file at
800
+ all, on any subsequent open, ever — there is no path back into the "new" branch
801
+ once records exist.
802
+
803
+ Published KEEP-FIRST, not via ``_atomic_write_text_race_tolerant`` (review round 2,
804
+ Minor 1 — a real bug, not a nit): two openers (worker threads, or two genuinely
805
+ separate processes) can reach a store's first-ever open at once, and unlike the
806
+ format marker/``.gitignore`` (where every racer writes IDENTICAL content, so
807
+ whoever's ``os.replace`` lands last is harmless), each racer here stamps its OWN
808
+ ``datetime.now(UTC)``. ``_atomic_write_text_race_tolerant``'s ``os.replace`` is
809
+ LAST-WRITER-WINS: forced interleaving (measured) shows opener A can pass the
810
+ "no records yet" check, opener B then open, publish an EARLIER marker, and write a
811
+ real bypass record — and A, resuming, unconditionally overwrites B's marker with
812
+ A's LATER timestamp, leaving a record that predates the file that is supposed to
813
+ bound "before this store existed". A marker whose entire job is "the earliest
814
+ instant this store could have held a record" must never be replaced by a LATER
815
+ one. So this publishes via ``os.link`` (an exclusive create — raises
816
+ ``FileExistsError`` if ``marker`` already exists, never silently clobbers it,
817
+ same idiom ``_write_archive_segment`` uses and for the same reason) onto a
818
+ per-attempt UNIQUE tmp name (never a fixed one — a fixed shared tmp name would
819
+ reopen the truncation-race hole ``_atomic_write_text_race_tolerant`` closes for
820
+ its own callers). Whichever racer's ``os.link`` reaches the filesystem FIRST wins
821
+ and stays forever: every later attempt — a losing racer finishing its own
822
+ ``__init__`` right after, or any later ``Store`` open entirely — sees
823
+ ``marker.is_file()`` true at THIS method's very first line and never attempts a
824
+ write at all, so a published marker can never be replaced, only ever raced to be
825
+ the first one written. (A losing racer's OWN computed timestamp can occasionally
826
+ be a few microseconds smaller than the winner's — real-time submission order, not
827
+ computation order, decides the race — but that residual imprecision is bounded to
828
+ the width of a single first-open race, not "years later", and is exactly the
829
+ "equally valid near-simultaneous now" case this marker's precision was always only
830
+ good for.)
831
+
832
+ Advisory write, never a load-bearing one (review round 2, Minor 2 — a regression
833
+ this fix must not introduce): a read-only store directory or a full disk must
834
+ raise on nothing here. ``format`` failing to write IS supposed to fail an open —
835
+ it is a correctness gate ``_ensure_format_marker``'s own docstring describes as
836
+ such. This marker only feeds one advisory doctor check; failing an open over it
837
+ would make Store() strictly LESS robust than before this feature existed, on a
838
+ plausible shape (a sandboxed/read-only-mounted project that has run sidegraph but
839
+ recorded nothing yet — e.g. a SessionStart hook in a read-only sandbox). Any
840
+ ``OSError`` anywhere in the attempt — the tmp write, or ``os.link`` raising
841
+ anything other than ``FileExistsError`` — is swallowed; the store then behaves
842
+ exactly like a legacy store (no marker), the safe direction, and a later open on
843
+ writable storage still gets a real chance (the "no records yet" gate is re-checked
844
+ fresh every open, not remembered from a failed attempt).
845
+ """
846
+ marker = self.path / _STAMPING_MARKER_NAME
847
+ if marker.is_file():
848
+ return
849
+ if _store_has_any_records(self.path):
850
+ return
851
+ stamp = datetime.now(UTC).isoformat()
852
+ tmp = marker.with_name(f"{marker.name}.{os.getpid()}.{uuid.uuid4().hex[:12]}.tmp")
853
+ with suppress(OSError): # advisory write only -- never fail Store.open (Minor 2)
854
+ tmp.write_text(f"{stamp}\n", encoding="utf-8")
855
+ with suppress(FileExistsError): # keep-first: another opener already won
856
+ os.link(tmp, marker)
857
+ # The cleanup needs its own suppression, not the block above (fix round 2
858
+ # re-review, Minor A): `missing_ok=True` only tolerates a MISSING file, so an
859
+ # unlink that raises for any other reason -- a read-only directory being the
860
+ # realistic one -- escaped and failed `Store.open` from the one write this method
861
+ # promises can never do that. The debris it leaves behind is swept by
862
+ # `_sweep_stale_tmp_files` on a later open, so skipping it costs nothing.
863
+ with suppress(OSError):
864
+ tmp.unlink(missing_ok=True)
865
+
866
+ def _ensure_gitignore(self) -> None:
867
+ """Committed layout convenience (design §1/§5): write ``<path>/.gitignore``
868
+ (``index.db*`` + ``*.tmp``) when missing, so the derived index and any crash
869
+ debris never show up in ``git status``. Written by ANY ``Store`` open that touches
870
+ the canonical layout — not just ``sidegraph-init`` — so a bare
871
+ ``sidegraph-mcp``/hook invocation against a pre-existing (pre-N2) store also
872
+ leaves git clean. NEVER overwrites an existing ``.gitignore``: a user may have
873
+ hand-edited it."""
874
+ gitignore = self.path / _GITIGNORE_NAME
875
+ if gitignore.is_file():
876
+ return
877
+ _atomic_write_text_race_tolerant(gitignore, _GITIGNORE_CONTENT)
878
+
879
+ def _migrate_legacy(self, legacy_file: Path) -> None:
880
+ """One-time export of a legacy single-file SQLite store (schema_version in
881
+ ``_MIGRATABLE_SCHEMA_VERSIONS``) into the canonical file-per-record layout at
882
+ ``self.path`` (design §4). Raises (without touching the legacy file) if its stamp
883
+ isn't forward-migratable. On success, renames the legacy file to
884
+ ``<name>.migrated-backup`` (never deleted — the store never destroys data) and
885
+ prints one stderr notice.
886
+
887
+ Fail-closed: every legacy row is read AND pydantic-validated in FULL before a
888
+ single byte is written anywhere or the legacy file is touched. A garbled row
889
+ raises a clear error naming the offending table and id, leaving the legacy db
890
+ exactly as it was — retryable, never partially migrated. Only once every row is
891
+ known-good is the export written, and only into a private staging directory (see
892
+ below); the legacy file's RENAME to ``.migrated-backup`` is the point of no return,
893
+ immediately followed by moving the staged export into ``self.path``.
894
+
895
+ Staging, not writing ``self.path`` directly, is required because ``self.path`` can
896
+ BE ``legacy_file`` itself (the bare-file dispatch, e.g. ``Store("old.db")``) — a
897
+ plain file, not yet a directory a canonical subdir could be created under. Writing
898
+ to an isolated temp directory first (same filesystem as ``legacy_file`` so the
899
+ final move is a cheap, atomic-per-entry rename) sidesteps that ordering constraint
900
+ entirely while still keeping every write fully isolated from the real store until
901
+ the data is proven good.
902
+
903
+ Volatile fields (entity ``last_seen_*``, binding ``status``, domain
904
+ ``communities``) are NOT carried over: only canonical (identity/content) files are
905
+ written here, and ``__init__``'s freshness check (index missing -> reload) rebuilds
906
+ the index from exactly those files afterwards, exactly like any other
907
+ canonical-only rebuild — the next sync re-derives volatile state from the live
908
+ graph, same as design §3's git-pull-absorption path.
909
+ """
910
+ conn = sqlite3.connect(str(legacy_file))
911
+ conn.row_factory = sqlite3.Row
912
+ try:
913
+ row = conn.execute("SELECT value FROM meta WHERE key = 'schema_version'").fetchone()
914
+ stamped = row["value"] if row else None
915
+ if stamped not in _MIGRATABLE_SCHEMA_VERSIONS:
916
+ # State the condition actually being tested. The old wording compared the
917
+ # stamp against the RUNNING CODE's version, which is not what this branch
918
+ # checks — so a file stamped with the current version produced the nonsense
919
+ # "'0.6.0' != '0.6.0'", and its "use a fresh store" advice would have
920
+ # discarded real data. Say what this migrator handles and stop.
921
+ handled = ", ".join(repr(v) for v in sorted(_MIGRATABLE_SCHEMA_VERSIONS))
922
+ raise ValueError(
923
+ f"{legacy_file} is stamped schema_version {stamped!r}, which this "
924
+ f"migration path does not handle — it migrates {handled} single-file "
925
+ "stores only (migration tooling beyond that is deferred)."
926
+ )
927
+ legacy_rows = self._read_legacy_rows(conn)
928
+ finally:
929
+ conn.close()
930
+
931
+ # Validate EVERYTHING before touching the filesystem at all — a garbled row raises
932
+ # here, before the staging directory even exists.
933
+ validated = self._validate_legacy_rows(legacy_file, legacy_rows)
934
+
935
+ staging = Path(tempfile.mkdtemp(prefix=".sidegraph-migrating-", dir=legacy_file.parent))
936
+ try:
937
+ for sub in _CANONICAL_SUBDIRS:
938
+ (staging / sub).mkdir(parents=True, exist_ok=True)
939
+
940
+ for entity in validated["entities"]:
941
+ _atomic_write_json(
942
+ staging / "entities" / f"{entity.entity_id}.json",
943
+ _entity_identity_payload(entity),
944
+ )
945
+ for decision in validated["decisions"]:
946
+ _atomic_write_json(
947
+ staging / "decisions" / f"{decision.id}.json",
948
+ decision.model_dump(mode="json"),
949
+ )
950
+ for domain in validated["domains"]:
951
+ _atomic_write_json(
952
+ staging / "domains" / f"{domain.domain_id}.json",
953
+ _domain_canonical_payload(domain),
954
+ )
955
+ for initiative in validated["initiatives"]:
956
+ _atomic_write_json(
957
+ staging / "initiatives" / f"{initiative.id}.json",
958
+ initiative.model_dump(mode="json"),
959
+ )
960
+
961
+ by_decision: dict[str, list[dict]] = {}
962
+ for binding in validated["anchor_bindings"]:
963
+ by_decision.setdefault(binding.record_id, []).append(
964
+ _binding_identity_payload(binding)
965
+ )
966
+ for record_id, items in by_decision.items():
967
+ ordered = sorted(items, key=lambda x: (x["tier"], x["entity_id"]))
968
+ _atomic_write_json(staging / "bindings" / f"{record_id}.json", ordered)
969
+
970
+ # -- commit: the legacy file's rename is the point of no return (everything
971
+ # above is fully staged and known-good); moving the staged subdirs into place
972
+ # is then just directory-level renames, not per-record writes.
973
+ backup = legacy_file.with_name(legacy_file.name + ".migrated-backup")
974
+ legacy_file.rename(backup)
975
+
976
+ self.path.mkdir(parents=True, exist_ok=True)
977
+ for sub in _CANONICAL_SUBDIRS:
978
+ target = self.path / sub
979
+ if target.exists():
980
+ # pre-existing (empty, by construction of the dispatch above) subdir —
981
+ # move each staged file into it individually.
982
+ for f in (staging / sub).iterdir():
983
+ os.replace(f, target / f.name)
984
+ else:
985
+ os.replace(staging / sub, target)
986
+ finally:
987
+ shutil.rmtree(staging, ignore_errors=True)
988
+
989
+ print(
990
+ f"sidegraph: migrated legacy store {legacy_file} (schema {stamped}) -> "
991
+ f"canonical layout at {self.path} (backup: {backup})",
992
+ file=sys.stderr,
993
+ )
994
+
995
+ @staticmethod
996
+ def _read_legacy_rows(conn: sqlite3.Connection) -> dict[str, list[tuple[str, str]]]:
997
+ """Every row of every legacy table as ``(primary_key, data_json)`` pairs — the
998
+ primary key travels alongside the payload purely so a validation failure (see
999
+ ``_validate_legacy_rows``) can name the offending row even when its JSON itself is
1000
+ the thing that's garbled."""
1001
+
1002
+ def _table_exists(name: str) -> bool:
1003
+ return (
1004
+ conn.execute(
1005
+ "SELECT 1 FROM sqlite_master WHERE type='table' AND name=?", (name,)
1006
+ ).fetchone()
1007
+ is not None
1008
+ )
1009
+
1010
+ return {
1011
+ "entities": [
1012
+ (r["entity_id"], r["data"])
1013
+ for r in conn.execute("SELECT entity_id, data FROM entities")
1014
+ ],
1015
+ "decisions": [
1016
+ (r["id"], r["data"]) for r in conn.execute("SELECT id, data FROM decisions")
1017
+ ],
1018
+ "anchor_bindings": [
1019
+ # legacy (0.2.0/0.3.0) tables literally have a `decision_id` column -- this
1020
+ # reads the real on-disk legacy schema, not the current AnchorBinding model.
1021
+ (f"{r['decision_id']}:{r['entity_id']}", r["data"])
1022
+ for r in conn.execute("SELECT decision_id, entity_id, data FROM anchor_bindings")
1023
+ ],
1024
+ "initiatives": (
1025
+ [(r["id"], r["data"]) for r in conn.execute("SELECT id, data FROM initiatives")]
1026
+ if _table_exists("initiatives")
1027
+ else []
1028
+ ),
1029
+ "domains": (
1030
+ [
1031
+ (r["domain_id"], r["data"])
1032
+ for r in conn.execute("SELECT domain_id, data FROM domains")
1033
+ ]
1034
+ if _table_exists("domains")
1035
+ else []
1036
+ ),
1037
+ }
1038
+
1039
+ _LEGACY_ROW_MODELS: dict[str, type[BaseModel]] = {
1040
+ "entities": Entity,
1041
+ "decisions": Decision,
1042
+ "domains": Domain,
1043
+ "initiatives": Initiative,
1044
+ "anchor_bindings": AnchorBinding,
1045
+ }
1046
+
1047
+ @classmethod
1048
+ def _validate_legacy_rows(
1049
+ cls, legacy_file: Path, rows: dict[str, list[tuple[str, str]]]
1050
+ ) -> dict[str, list]:
1051
+ """Pydantic-validate every legacy row BEFORE any canonical file is written or the
1052
+ legacy file is renamed (fail-closed migration, design §4): a single garbled row
1053
+ raises a clear ``ValueError`` naming the offending table and id, and nothing about
1054
+ the legacy db has been touched yet — the caller can simply retry later."""
1055
+ out: dict[str, list] = {}
1056
+ for table, model in cls._LEGACY_ROW_MODELS.items():
1057
+ validated = []
1058
+ for row_id, data in rows.get(table, []):
1059
+ try:
1060
+ if table == "anchor_bindings":
1061
+ # A genuine 0.2.0/0.3.0-era row was serialized by code that named
1062
+ # this field `decision_id` (the AnchorBinding rename post-dates
1063
+ # every _MIGRATABLE_SCHEMA_VERSIONS store) -- normalize the legacy
1064
+ # key before validating against the current model.
1065
+ payload = json.loads(data)
1066
+ if "decision_id" in payload and "record_id" not in payload:
1067
+ payload["record_id"] = payload.pop("decision_id")
1068
+ validated.append(model.model_validate(payload))
1069
+ else:
1070
+ validated.append(model.model_validate_json(data))
1071
+ except Exception as e:
1072
+ raise ValueError(
1073
+ f"legacy store {legacy_file} has an invalid row in table "
1074
+ f"{table!r} (id={row_id!r}): {e}"
1075
+ ) from e
1076
+ out[table] = validated
1077
+ return out
1078
+
1079
+ # -- canonical file writers (git-committed) ------------------------------
1080
+
1081
+ def _record_canonical_stat(self, subdir: str, stem: str, st: os.stat_result) -> None:
1082
+ """Insert/update ``canonical_stat``'s row for ``(subdir, stem)`` — called from
1083
+ every one of the seven canonical writers, in the SAME transaction as the canonical
1084
+ file write and (for six of the seven) the caller's own index write (digest-integrity
1085
+ design, Task 1). Recording it HERE, beside the file write, rather than at the
1086
+ caller's index-write site, is what makes the ordering between the two stop
1087
+ mattering (design D4a): ``upsert_entity``/``_write_domain`` touch the digest BEFORE
1088
+ their index write while ``_write_decision``/``_write_fact`` do the reverse, and an
1089
+ id-based (or call-site-based) check would have refused the stamp on every new
1090
+ entity/domain. The stat row and the index row now always commit or roll back
1091
+ together regardless of which one the caller writes first."""
1092
+ self._conn.execute(
1093
+ "INSERT INTO canonical_stat (subdir, stem, size, mtime_ns) VALUES (?, ?, ?, ?) "
1094
+ "ON CONFLICT(subdir, stem) DO UPDATE SET "
1095
+ "size=excluded.size, mtime_ns=excluded.mtime_ns",
1096
+ (subdir, stem, st.st_size, st.st_mtime_ns),
1097
+ )
1098
+
1099
+ def _delete_canonical_stat(self, subdir: str, stem: str) -> None:
1100
+ """Remove ``canonical_stat``'s row for ``(subdir, stem)`` — called wherever a
1101
+ canonical file is actually removed (``compact``, Task 2): the row must not outlive
1102
+ the file it describes being GONE-by-design, mirroring ``_hot_file_matches``'s own
1103
+ skip (a file kept because it diverged from the archive keeps its row too — this is
1104
+ simply never called for that case)."""
1105
+ self._conn.execute(
1106
+ "DELETE FROM canonical_stat WHERE subdir = ? AND stem = ?", (subdir, stem)
1107
+ )
1108
+
1109
+ def _write_entity_canonical(self, entity: Entity) -> None:
1110
+ st = _atomic_write_json(
1111
+ self.path / "entities" / f"{entity.entity_id}.json",
1112
+ _entity_identity_payload(entity),
1113
+ )
1114
+ self._record_canonical_stat("entities", entity.entity_id, st)
1115
+
1116
+ def _write_decision_canonical(self, decision: Decision) -> None:
1117
+ st = _atomic_write_json(
1118
+ self.path / "decisions" / f"{decision.id}.json", decision.model_dump(mode="json")
1119
+ )
1120
+ self._record_canonical_stat("decisions", decision.id, st)
1121
+
1122
+ def _write_fact_canonical(self, fact: Fact) -> None:
1123
+ st = _atomic_write_json(
1124
+ self.path / "facts" / f"{fact.id}.json", fact.model_dump(mode="json")
1125
+ )
1126
+ self._record_canonical_stat("facts", fact.id, st)
1127
+
1128
+ def _write_domain_canonical(self, domain: Domain) -> None:
1129
+ st = _atomic_write_json(
1130
+ self.path / "domains" / f"{domain.domain_id}.json",
1131
+ _domain_canonical_payload(domain),
1132
+ )
1133
+ self._record_canonical_stat("domains", domain.domain_id, st)
1134
+
1135
+ def _write_initiative_canonical(self, initiative: Initiative) -> None:
1136
+ st = _atomic_write_json(
1137
+ self.path / "initiatives" / f"{initiative.id}.json",
1138
+ initiative.model_dump(mode="json"),
1139
+ )
1140
+ self._record_canonical_stat("initiatives", initiative.id, st)
1141
+
1142
+ def _write_bindings_file(self, record_id: str, items: list[dict]) -> None:
1143
+ ordered = sorted(items, key=lambda x: (x["tier"], x["entity_id"]))
1144
+ st = _atomic_write_json(self.path / "bindings" / f"{record_id}.json", ordered)
1145
+ self._record_canonical_stat("bindings", record_id, st)
1146
+
1147
+ def _write_bindings_canonical_for_record(self, record_id: str) -> None:
1148
+ """Recompute the FULL canonical anchor set for ``record_id`` from the index
1149
+ (source of truth for "current") and rewrite its bindings file. Called only when
1150
+ some binding's identity (entity_id/tier/relation/weight) actually changed — a pure
1151
+ status flip never reaches here (see ``add_binding``).
1152
+
1153
+ Filters out any binding whose entity is DERIVED (``community:*`` abstract
1154
+ entities — see ``_is_derived_entity``): those are index-only snapshot labels and
1155
+ must never land in a committed file, even when a genuine (non-derived) identity
1156
+ change on the SAME record is what triggered this rewrite in the first place —
1157
+ this is also how a stale community entry left by pre-0.6.0 code (tolerated on
1158
+ reload) decays away, lazily, the next time this record's canonical file is
1159
+ legitimately rewritten. One ``get_entity`` lookup per binding here is fine —
1160
+ rewrites are rare (see design doc, Mechanics §1)."""
1161
+ items = []
1162
+ for b in self.bindings_for_record(record_id):
1163
+ entity = self.get_entity(b.entity_id)
1164
+ if entity is not None and _is_derived_entity(entity):
1165
+ continue
1166
+ items.append(_binding_identity_payload(b))
1167
+ self._write_bindings_file(record_id, items)
1168
+
1169
+ # -- digest / freshness (design §3) --------------------------------------
1170
+
1171
+ def _compute_canonical_digest(self) -> tuple[str, dict[tuple[str, str], tuple[int, int]]]:
1172
+ """sha256 over sorted (relpath, size, mtime_ns) of every canonical record file, plus
1173
+ the format marker — cheap enough to run on every open (hundreds/thousands of small
1174
+ JSONs). The marker is included like any other committed file, but since it never
1175
+ changes within a version (``_ensure_format_marker`` writes it once and never again),
1176
+ its presence here never causes spurious digest churn.
1177
+
1178
+ Also covers ``archive/*.jsonl`` segments (design §7): a segment is write-once and
1179
+ never rewritten in place, but it is still new committed content the first time it
1180
+ shows up (freshly compacted locally, or absorbed from a teammate's branch via git
1181
+ pull) — without it here, a reopen after either would keep serving the stale index
1182
+ (missing the archived records / still showing their now-removed hot files) until
1183
+ something else happened to bust the digest.
1184
+
1185
+ Returns ``(digest, stats)`` where ``stats`` maps every walked file's
1186
+ ``(subdir, stem)`` — every ``_CANONICAL_SUBDIRS`` entry plus ``archive/*.jsonl``
1187
+ segments, NOT the format marker (write-once, never rewritten, so it never needs a
1188
+ ``canonical_stat`` row to prove freshness against) — to its ``(size, mtime_ns)``,
1189
+ accumulated in this SAME walk (digest-integrity design, Task 3): the walk already
1190
+ stats every file for the digest itself, so this is free. ``_touch_digest`` compares
1191
+ ``stats`` against ``canonical_stat`` before stamping."""
1192
+ entries: list[str] = []
1193
+ stats: dict[tuple[str, str], tuple[int, int]] = {}
1194
+ marker = self.path / _FORMAT_MARKER_NAME
1195
+ if marker.is_file():
1196
+ st = marker.stat()
1197
+ entries.append(f"{_FORMAT_MARKER_NAME}\0{st.st_size}\0{st.st_mtime_ns}")
1198
+ for sub in _CANONICAL_SUBDIRS:
1199
+ d = self.path / sub
1200
+ if not d.is_dir():
1201
+ continue
1202
+ for f in d.iterdir():
1203
+ if f.suffix != ".json":
1204
+ continue
1205
+ st = f.stat()
1206
+ entries.append(f"{sub}/{f.name}\0{st.st_size}\0{st.st_mtime_ns}")
1207
+ stats[(sub, f.stem)] = (st.st_size, st.st_mtime_ns)
1208
+ archive_dir = self.path / _ARCHIVE_SUBDIR
1209
+ if archive_dir.is_dir():
1210
+ for f in archive_dir.iterdir():
1211
+ if f.suffix != ".jsonl":
1212
+ continue
1213
+ st = f.stat()
1214
+ entries.append(f"{_ARCHIVE_SUBDIR}/{f.name}\0{st.st_size}\0{st.st_mtime_ns}")
1215
+ stats[(_ARCHIVE_SUBDIR, f.stem)] = (st.st_size, st.st_mtime_ns)
1216
+ entries.sort()
1217
+ h = hashlib.sha256()
1218
+ for e in entries:
1219
+ h.update(e.encode())
1220
+ h.update(b"\n")
1221
+ return h.hexdigest(), stats
1222
+
1223
+ def _canonical_stat_mismatch(self, stats: dict[tuple[str, str], tuple[int, int]]) -> bool:
1224
+ """True iff any ``(subdir, stem)`` the digest walk just saw has NO ``canonical_stat``
1225
+ row, or a row with a DIFFERENT ``(size, mtime_ns)`` — i.e. this index did not load
1226
+ that file in the state it is currently in on disk (design D1/D4). One-directional
1227
+ (design D2): a row whose file is now GONE is normal and never flags a mismatch —
1228
+ ``compact`` deletes those rows itself (Task 2), and a derived ``community:*`` entity
1229
+ or an archive-loaded row never had a canonical file to begin with."""
1230
+ if not stats:
1231
+ return False
1232
+ rows = self._conn.execute("SELECT subdir, stem, size, mtime_ns FROM canonical_stat")
1233
+ table = {(r["subdir"], r["stem"]): (r["size"], r["mtime_ns"]) for r in rows}
1234
+ return any(table.get(key) != st for key, st in stats.items())
1235
+
1236
+ def _touch_digest(self) -> None:
1237
+ """Recompute + store the canonical digest right after a canonical-file write, so
1238
+ THIS process's own legitimate writes never look like an external change (git pull,
1239
+ manual edit) to the next Store that opens this path.
1240
+
1241
+ Refuses to stamp — and CLEARS the existing stamp — when the digest walk saw a
1242
+ canonical file this index never loaded in its current state (digest-integrity
1243
+ design D1/D3, rev 4). An earlier draft merely withheld the new stamp, on the
1244
+ assumption that doing so always leaves ``stored`` behind ``current`` for the next
1245
+ open. That assumption is false, and needs no crash at all: S1 replaces a record's
1246
+ file; S2 replaces over it and FULLY completes, stamping a digest that validly
1247
+ matches current disk; S1 then lands its own (now stale) index write and refuses to
1248
+ stamp. Merely withholding changes nothing — S2's certificate still matches disk, so
1249
+ ``stored == current`` while the index has drifted under it, permanently, since
1250
+ nothing ever busts that digest again. Declining to issue a new certificate does
1251
+ nothing about the one already on the wall — a refusal means "this index may not
1252
+ match disk", and that has to invalidate the existing claim, not just withhold the
1253
+ next one. Deleting the ``canonical_digest`` meta row does that: ``stored`` becomes
1254
+ ``None``, and ``_refresh_freshness`` already treats a missing row as "reload"
1255
+ unconditionally, regardless of what ``current`` happens to compute to.
1256
+
1257
+ Still not reloading mid-write (D3): that would nest a second transaction owner
1258
+ inside ``_mutation`` and drop the caller's own uncommitted work, since the reload
1259
+ drops and repopulates every record table. Clearing costs one reload on the next
1260
+ open — the existing, tested heal path, just entered unconditionally instead of by
1261
+ a value comparison that this scenario can defeat. ``BaseException`` is not needed
1262
+ here: this never raises on the refusal path, it silently clears and returns, by
1263
+ design (D7 — a refusal means another process wrote concurrently, which is normal in
1264
+ this project's shape and not worth a routine warning)."""
1265
+ digest, stats = self._compute_canonical_digest()
1266
+ if self._canonical_stat_mismatch(stats):
1267
+ self._conn.execute("DELETE FROM meta WHERE key = ?", (_CANONICAL_DIGEST_KEY,))
1268
+ return # invalidate, don't merely withhold (D1/D3)
1269
+ self._conn.execute(
1270
+ "INSERT INTO meta (key, value) VALUES (?, ?) "
1271
+ "ON CONFLICT(key) DO UPDATE SET value=excluded.value",
1272
+ (_CANONICAL_DIGEST_KEY, digest),
1273
+ )
1274
+
1275
+ def _refresh_freshness(self) -> None:
1276
+ """On open: missing/stale digest -> full reload of canonical files into the index,
1277
+ volatile fields reset to cold defaults (design §3). Digest match -> fast path, but
1278
+ still guard against a hand-corrupted ``schema_version`` stamp (the index's OWN
1279
+ claim of freshness is only trustworthy if its declared version agrees with the
1280
+ running code) — UNLESS the stamped version is a known-reloadable one
1281
+ (``_RELOADABLE_SCHEMA_VERSIONS``), in which case a full reload re-stamps it instead
1282
+ of hard-failing (see ``_RELOADABLE_SCHEMA_VERSIONS``)."""
1283
+ with self._lock:
1284
+ stored = self._conn.execute(
1285
+ "SELECT value FROM meta WHERE key = ?", (_CANONICAL_DIGEST_KEY,)
1286
+ ).fetchone()
1287
+ current, _current_stats = self._compute_canonical_digest()
1288
+ if stored is None or stored["value"] != current:
1289
+ self._reload_index_from_canonical(current)
1290
+ else:
1291
+ stamped = self._conn.execute(
1292
+ "SELECT value FROM meta WHERE key = 'schema_version'"
1293
+ ).fetchone()
1294
+ stamped_value = stamped["value"] if stamped else None
1295
+ if stamped_value != SCHEMA_VERSION:
1296
+ if stamped_value in _RELOADABLE_SCHEMA_VERSIONS:
1297
+ self._reload_index_from_canonical(current)
1298
+ else:
1299
+ raise ValueError(
1300
+ f"store schema_version {stamped_value!r} != code "
1301
+ f"{SCHEMA_VERSION!r}; migration tooling is deferred — use a "
1302
+ "fresh store"
1303
+ )
1304
+ # No commit here: the rebuild owns its transaction (design D1/D9) and the fast
1305
+ # path is SELECT-only, which opens none. This commit used to be the only thing
1306
+ # that closed the rebuild's implicit transaction; leaving it now would invite
1307
+ # removing the rebuild's own commit, which would silently restore mechanism 1.
1308
+
1309
+ def _reload_index_from_canonical(self, digest: str) -> None:
1310
+ """Rebuild the ENTIRE index from the canonical files on disk. Every volatile field
1311
+ resets to its cold default (design §3): entity ``last_seen_*`` -> None, binding
1312
+ ``status`` -> "live" ("bindings live-as-written"), domain ``communities`` -> [].
1313
+ The capture ledger and non-schema meta (e.g. ``last_synced_graph_version``,
1314
+ the TOC cache) are untouched — they are pure local bookkeeping with no canonical
1315
+ file to rebuild from, not tied to the digest at all.
1316
+
1317
+ The RECORD tables (entities/decisions/facts/domains/anchor_bindings/initiatives)
1318
+ are dropped and recreated here, not just cleared with ``DELETE FROM`` — a real
1319
+ pre-0.5.0 store's ``index.db`` was built by OLDER code against an OLDER
1320
+ ``_SCHEMA_SQL`` (e.g. ``anchor_bindings.decision_id`` before this branch renamed
1321
+ it to ``record_id``), and ``__init__``'s ``executescript(_SCHEMA_SQL)`` is
1322
+ ``CREATE TABLE IF NOT EXISTS`` — a no-op against an existing table, so it never
1323
+ heals column drift. These tables are fully DERIVED (every row is rebuilt below
1324
+ from the canonical files this loop reads), so dropping and recreating them is
1325
+ safe and heals any DDL drift; re-running the schema statements immediately after
1326
+ recreates exactly the dropped tables (``IF NOT EXISTS``) without touching
1327
+ ``meta`` or ``capture_sessions``, which are NOT record tables and must survive
1328
+ this reload untouched: ``meta`` is re-stamped by this same method below, and
1329
+ ``capture_sessions`` is the local once-per-session capture ledger — dropping it
1330
+ would replay capture nudges the store already saw.
1331
+
1332
+ The whole body — the six DROPs, the CREATEs, every reload loop, the archive
1333
+ merge, the slug-conflict warning, and the three meta stamps — runs inside one
1334
+ explicit ``BEGIN IMMEDIATE`` … ``COMMIT`` (design D1). SQLite DDL is fully
1335
+ transactional; the previous code just never opened a transaction around it, so
1336
+ the bare ``DROP TABLE`` statements autocommitted one at a time and a concurrent
1337
+ reader could observe the record tables gone (``OperationalError: no such table:
1338
+ domains``, 2/1600 opens measured) — including a long-lived MCP server that never
1339
+ re-opens and so has no way to retry. Under one transaction, a concurrent reader
1340
+ instead sees the OLD committed rows for the entire rebuild and the NEW ones only
1341
+ after this method's ``commit()`` — never a missing table, on every platform, with
1342
+ no new file, lock, or timeout. An exception anywhere in the body rolls the whole
1343
+ rebuild back, so a failed reload leaves the previous index intact instead of
1344
+ half-dropped (see ``test_store_rebuild_atomicity.py``).
1345
+
1346
+ The CREATEs are issued as individual ``execute()`` calls against
1347
+ ``_SCHEMA_STATEMENTS``, never ``self._conn.executescript(_SCHEMA_SQL)`` (design
1348
+ D2 — measured, not stylistic): ``executescript`` issues an implicit ``COMMIT``
1349
+ before running its script, which would publish the just-executed DROPs and
1350
+ restore exactly this bug. Do not bring ``executescript`` back into this method.
1351
+ """
1352
+ with self._lock:
1353
+ # Two concurrent rebuilds now serialize on SQLite's RESERVED lock instead of
1354
+ # interleaving, so the default 5s busy timeout becomes a new failure mode under
1355
+ # load. Raise it for the rebuild ONLY -- raising it on connect would apply to
1356
+ # every statement for the connection's lifetime, stalling an ordinary tool call
1357
+ # behind a wedged writer for 30s where it stalls 5s today (design D3). A
1358
+ # measured rebuild of this repo's own store (683 canonical files) is 40ms.
1359
+ previous_timeout = self._conn.execute("PRAGMA busy_timeout").fetchone()[0]
1360
+ self._conn.execute("PRAGMA busy_timeout = 30000")
1361
+ try:
1362
+ self._conn.execute("BEGIN IMMEDIATE")
1363
+ try:
1364
+ self._conn.execute("DROP TABLE IF EXISTS entities")
1365
+ self._conn.execute("DROP TABLE IF EXISTS decisions")
1366
+ self._conn.execute("DROP TABLE IF EXISTS facts")
1367
+ self._conn.execute("DROP TABLE IF EXISTS anchor_bindings")
1368
+ self._conn.execute("DROP TABLE IF EXISTS domains")
1369
+ self._conn.execute("DROP TABLE IF EXISTS initiatives")
1370
+ # canonical_stat is DROPped with the record tables, not upserted over:
1371
+ # a row for a file that is ABSENT during this reload must not survive it.
1372
+ # Left in place, a file returning later with an identical (size, mtime_ns)
1373
+ # -- mv out and back, rsync -a, a backup restore -- would match the stale
1374
+ # row and get certified as loaded, recreating the exact poisoned state
1375
+ # this design exists to kill, through this design's own mechanism
1376
+ # (branch review). Recreated by the _SCHEMA_STATEMENTS loop just below,
1377
+ # inside this same transaction, so a failed reload still rolls back whole.
1378
+ self._conn.execute("DROP TABLE IF EXISTS canonical_stat")
1379
+ for statement in _SCHEMA_STATEMENTS:
1380
+ self._conn.execute(statement)
1381
+
1382
+ # Every loop below STATS the file BEFORE reading its content
1383
+ # (digest-integrity design §3's rule, applied to the reload): a
1384
+ # concurrent writer could replace the file between the stat and the
1385
+ # read, and stat-before-read means that race can only make the
1386
+ # recorded stat OLDER than the content this reload actually indexed --
1387
+ # never newer. An older stat is safe (it just fails to match on a
1388
+ # later digest walk and forces one more reload -- D3's tolerated heal
1389
+ # path); a stat taken AFTER the read could describe content NEWER than
1390
+ # what got parsed, which is the same lie in the other direction (§3).
1391
+ for f in sorted((self.path / "entities").glob("*.json")):
1392
+ st = f.stat()
1393
+ self._index_write_entity(Entity.model_validate(json.loads(f.read_text())))
1394
+ self._record_canonical_stat("entities", f.stem, st)
1395
+ for f in sorted((self.path / "decisions").glob("*.json")):
1396
+ st = f.stat()
1397
+ self._index_write_decision(
1398
+ Decision.model_validate(json.loads(f.read_text()))
1399
+ )
1400
+ self._record_canonical_stat("decisions", f.stem, st)
1401
+ for f in sorted((self.path / "facts").glob("*.json")):
1402
+ st = f.stat()
1403
+ self._index_write_fact(Fact.model_validate(json.loads(f.read_text())))
1404
+ self._record_canonical_stat("facts", f.stem, st)
1405
+ for f in sorted((self.path / "domains").glob("*.json")):
1406
+ st = f.stat()
1407
+ self._index_write_domain(Domain.model_validate(json.loads(f.read_text())))
1408
+ self._record_canonical_stat("domains", f.stem, st)
1409
+ for f in sorted((self.path / "bindings").glob("*.json")):
1410
+ record_id = f.stem
1411
+ st = f.stat()
1412
+ for item in json.loads(f.read_text()):
1413
+ binding = AnchorBinding.model_validate({**item, "record_id": record_id})
1414
+ self._index_write_binding(binding)
1415
+ # An empty `[]` bindings file (hand edit / merge artifact — §3's
1416
+ # debris shape) produces no index rows above, but still gets its
1417
+ # row here: the check reads THIS table, not the record tables, so
1418
+ # there is no branch left to wedge on it (spec item 7).
1419
+ self._record_canonical_stat("bindings", record_id, st)
1420
+ for f in sorted((self.path / "initiatives").glob("*.json")):
1421
+ st = f.stat()
1422
+ self._index_write_initiative(
1423
+ Initiative.model_validate(json.loads(f.read_text()))
1424
+ )
1425
+ self._record_canonical_stat("initiatives", f.stem, st)
1426
+
1427
+ # Archive segments (design Task 2 item 3): stat every segment BEFORE
1428
+ # `_archived_records()` reads its content below, same stat-before-read
1429
+ # rule as the loops above -- and record a row for EVERY segment file,
1430
+ # regardless of whether any of its individual records end up
1431
+ # hot-shadowed a few lines down: the row is at segment-FILE
1432
+ # granularity (this is what makes a `git pull`ed segment absorbed by
1433
+ # this reload get a row too, whether or not its records are shadowed).
1434
+ archive_dir = self.path / _ARCHIVE_SUBDIR
1435
+ archive_stats: dict[str, os.stat_result] = {}
1436
+ if archive_dir.is_dir():
1437
+ for f in sorted(archive_dir.glob("*.jsonl")):
1438
+ archive_stats[f.stem] = f.stat()
1439
+
1440
+ # Archived decisions/domains (design §7): a hot file for the same id
1441
+ # ALWAYS wins (it was just indexed above) — either it's the
1442
+ # crash-window duplicate a compact leaves behind (byte-identical,
1443
+ # silently a no-op here; ``Store.compact`` is what actually cleans
1444
+ # those up) or, if it genuinely differs, that's corruption-shaped and
1445
+ # never something a mere reload should resolve by preferring the
1446
+ # archive over live disk state. Only ids with NO hot file are indexed
1447
+ # from the archive.
1448
+ archived_decisions, archived_domains = self._archived_records()
1449
+ hot_decision_ids = {f.stem for f in (self.path / "decisions").glob("*.json")}
1450
+ hot_domain_ids = {f.stem for f in (self.path / "domains").glob("*.json")}
1451
+ for did, payload in archived_decisions.items():
1452
+ hot_path = self.path / "decisions" / f"{did}.json"
1453
+ if did in hot_decision_ids:
1454
+ if json.loads(hot_path.read_text(encoding="utf-8")) != payload:
1455
+ _warn_hot_archive_mismatch("decision", did)
1456
+ continue
1457
+ self._index_write_decision(Decision.model_validate(payload))
1458
+ for dmid, payload in archived_domains.items():
1459
+ hot_path = self.path / "domains" / f"{dmid}.json"
1460
+ if dmid in hot_domain_ids:
1461
+ if json.loads(hot_path.read_text(encoding="utf-8")) != payload:
1462
+ _warn_hot_archive_mismatch("domain", dmid)
1463
+ continue
1464
+ self._index_write_domain(Domain.model_validate(payload))
1465
+
1466
+ for stem, st in archive_stats.items():
1467
+ self._record_canonical_stat(_ARCHIVE_SUBDIR, stem, st)
1468
+
1469
+ # design §6: warn about a cross-branch domain-slug race NOW, with the
1470
+ # full, just-merged domain set in hand -- see
1471
+ # _compute_domain_slug_conflicts. This is a one-shot stderr notice for
1472
+ # whoever's process just did the reload; it is NOT persisted anywhere
1473
+ # (see Store.domain_slug_conflicts -- fix, review round 3: a
1474
+ # single-clone workflow with nothing else ever busting the digest would
1475
+ # otherwise never re-run this, so a resolved conflict could report as
1476
+ # unresolved forever. domain_slug_conflicts() recomputes live instead).
1477
+ # This also reads the tables THIS transaction just (re)wrote, so it
1478
+ # must run before the commit below — outside the transaction it would
1479
+ # be reading a rebuild that had already been published, the very race
1480
+ # D1 closes.
1481
+ for conflict in self._compute_domain_slug_conflicts():
1482
+ _warn_domain_slug_conflict(conflict["slug"], conflict["domain_ids"])
1483
+
1484
+ for key, value in (
1485
+ ("schema_version", SCHEMA_VERSION),
1486
+ (_CANONICAL_DIGEST_KEY, digest),
1487
+ (VOLATILE_STALE_KEY, "1"),
1488
+ ):
1489
+ self._conn.execute(
1490
+ "INSERT INTO meta (key, value) VALUES (?, ?) "
1491
+ "ON CONFLICT(key) DO UPDATE SET value=excluded.value",
1492
+ (key, value),
1493
+ )
1494
+ self._conn.commit()
1495
+ except BaseException:
1496
+ self._conn.rollback()
1497
+ raise
1498
+ finally:
1499
+ self._conn.execute(f"PRAGMA busy_timeout = {previous_timeout}")
1500
+
1501
+ def _compute_domain_slug_conflicts(self) -> list[dict]:
1502
+ """Slugs currently held by MORE than one LIVE (``proposed``/``accepted`` — the same
1503
+ "live" set ``_validate_domain_write``'s all-holders uniqueness query uses) domain at
1504
+ once. Superseded/dropped domains intentionally free their slug back up and are
1505
+ never part of a conflict — mirrors that method's own status filter exactly, so
1506
+ "would this write have been rejected had it happened in one process" and "is this
1507
+ flagged as a merge race" never disagree.
1508
+
1509
+ Reads directly off the INDEX (not the canonical files), so it reflects whatever is
1510
+ CURRENTLY loaded — every hot file plus every merged-in archive entry, always
1511
+ up to date with the running process's own writes too (see ``domain_slug_conflicts``,
1512
+ the public, live-computing wrapper around this).
1513
+
1514
+ Returns a list of ``{"slug": str, "domain_ids": list[str]}``, sorted by slug then
1515
+ by ``domain_id`` (ULID, so also creation order) for a deterministic report."""
1516
+ with self._lock:
1517
+ rows = self._conn.execute(
1518
+ "SELECT domain_id, slug FROM domains WHERE status IN (?, ?)",
1519
+ (DomainStatus.PROPOSED.value, DomainStatus.ACCEPTED.value),
1520
+ ).fetchall()
1521
+ by_slug: dict[str, list[str]] = {}
1522
+ for row in rows:
1523
+ by_slug.setdefault(row["slug"], []).append(row["domain_id"])
1524
+ return [
1525
+ {"slug": slug, "domain_ids": sorted(ids)}
1526
+ for slug, ids in sorted(by_slug.items())
1527
+ if len(ids) > 1
1528
+ ]
1529
+
1530
+ def domain_slug_conflicts(self) -> list[dict]:
1531
+ """Slugs currently held by more than one live domain — see
1532
+ ``_compute_domain_slug_conflicts`` for the exact definition. Computed LIVE, on
1533
+ every call (review round 3 fix: an earlier version cached this at index-reload time
1534
+ only, in index meta — but nothing else in a single-clone workflow ever busts the
1535
+ digest to force a fresh reload, so a conflict resolved in-process, or even in a
1536
+ brand-new process against the same already-fresh store, would have kept reporting
1537
+ as unresolved forever). A single indexed ``SELECT`` is cheap enough to run on every
1538
+ ``sidegraph-sync`` pass without caching it at all.
1539
+
1540
+ ``sidegraph-sync`` surfaces this list in its report (see
1541
+ ``sync.SyncReport.slug_conflicts``); ``Store.__init__`` also warns to stderr the
1542
+ moment a cold load/digest-mismatch reload detects one, for a bare CLI invocation
1543
+ that never looks at a report at all (see ``_warn_domain_slug_conflict``, called
1544
+ from ``_reload_index_from_canonical`` — that warning is a one-shot notice, never
1545
+ persisted; this method is the source of truth)."""
1546
+ return self._compute_domain_slug_conflicts()
1547
+
1548
+ def _commit(self) -> None:
1549
+ self._conn.commit()
1550
+
1551
+ @contextmanager
1552
+ def _mutation(self, *, immediate: bool = False) -> Iterator[None]:
1553
+ """Every public write runs inside this: commit on success, roll back on any failure
1554
+ (design D1/D2 — the fix for Defect A, external review).
1555
+
1556
+ Canonical files are already on disk by the time anything here fails and cannot be
1557
+ un-replaced (see ``supersede_domain``'s ordering comment). What must not survive is a
1558
+ half-staged SQLite transaction: the connection is shared, so a long-lived process
1559
+ that returns one error would keep the write lock and every hook and CLI after it
1560
+ would get ``database is locked`` (measured: 2.1s wait, then failure).
1561
+
1562
+ Reentrant (design D2): ``ratify_domains`` -> ``get_or_create_abstract_entity`` ->
1563
+ ``upsert_entity`` is a real nesting chain today, and only the OUTERMOST scope may
1564
+ commit or roll back — an earlier draft of the design spec claimed nesting was
1565
+ hypothetical; it is not, and ``ratify_domains`` wraps per item instead of once
1566
+ around the whole method for exactly this reason (D2a).
1567
+
1568
+ The commit is INSIDE the try on purpose (design H1): COMMIT is exactly where
1569
+ SQLITE_BUSY lands, so a commit raising must roll back too, not propagate untouched.
1570
+ An earlier draft of the design put the commit after the whole try/except/finally and
1571
+ review caught that it silently reintroduces Defect A on its single most likely
1572
+ trigger.
1573
+
1574
+ ``BaseException``, not ``Exception`` — matches ``_reload_index_from_canonical``'s
1575
+ existing guard (design D3, that method's own transaction, NOT this helper): a
1576
+ ``KeyboardInterrupt`` mid-write must roll back before propagating.
1577
+
1578
+ ``immediate=True`` (design D6, entity-identity-uniqueness spec) issues ``BEGIN
1579
+ IMMEDIATE`` before yielding, so a get-or-create's check-then-create runs as one
1580
+ atomic unit and a second connection cannot slip its own lookup in between (external
1581
+ review, finding 2). Gated on the CONNECTION (``self._conn.in_transaction``), NOT on
1582
+ the depth. Python's legacy transaction control opens nothing until the first DML, so
1583
+ an outer ``_mutation()`` that has only READ holds no lock at all — measured,
1584
+ ``in_transaction`` is ``False`` after a read inside a mutation. An earlier draft
1585
+ gated this on depth instead (a nested immediate at depth 2 is a no-op because "the
1586
+ outer transaction already holds RESERVED"); that reasoning is false whenever the
1587
+ outer scope hasn't written yet, and review reproduced the consequence directly:
1588
+ outer plain mutation -> nested depth-2 immediate treated as a no-op -> two ids
1589
+ minted again, straight through the hole this flag exists to close. Checking
1590
+ ``in_transaction`` instead is correct at any depth: if the outer scope already
1591
+ performed DML (so a transaction is already open), the nested ``BEGIN IMMEDIATE`` is
1592
+ skipped — issuing a second ``BEGIN`` on an open transaction would raise
1593
+ ``OperationalError: cannot start a transaction within a transaction`` — and if it
1594
+ hasn't, the nested call is the one that actually takes the write lock, for real,
1595
+ while the outermost scope still owns commit and rollback either way.
1596
+ """
1597
+ with self._lock:
1598
+ self._mutation_depth += 1
1599
+ try:
1600
+ if immediate and not self._conn.in_transaction:
1601
+ self._conn.execute("BEGIN IMMEDIATE")
1602
+ yield
1603
+ if self._mutation_depth == 1:
1604
+ self._commit()
1605
+ except BaseException:
1606
+ if self._mutation_depth == 1:
1607
+ self._conn.rollback()
1608
+ raise
1609
+ finally:
1610
+ self._mutation_depth -= 1
1611
+
1612
+ # -- index writers (derived; index.db only) ------------------------------
1613
+
1614
+ def _index_write_entity(self, entity: Entity) -> None:
1615
+ self._conn.execute(
1616
+ "INSERT INTO entities (entity_id, canonical_name, data) VALUES (?, ?, ?) "
1617
+ "ON CONFLICT(entity_id) DO UPDATE SET canonical_name=excluded.canonical_name, "
1618
+ "data=excluded.data",
1619
+ (entity.entity_id, entity.canonical_name, entity.model_dump_json()),
1620
+ )
1621
+
1622
+ def _index_write_decision(self, decision: Decision) -> None:
1623
+ self._conn.execute(
1624
+ "INSERT INTO decisions (id, status, supersedes, data) VALUES (?, ?, ?, ?) "
1625
+ "ON CONFLICT(id) DO UPDATE SET status=excluded.status, data=excluded.data",
1626
+ (
1627
+ decision.id,
1628
+ decision.status.value,
1629
+ decision.supersedes,
1630
+ decision.model_dump_json(),
1631
+ ),
1632
+ )
1633
+
1634
+ def _index_write_fact(self, fact: Fact) -> None:
1635
+ self._conn.execute(
1636
+ "INSERT INTO facts (id, status, supersedes, data) VALUES (?, ?, ?, ?) "
1637
+ "ON CONFLICT(id) DO UPDATE SET status=excluded.status, data=excluded.data",
1638
+ (
1639
+ fact.id,
1640
+ fact.status.value,
1641
+ fact.supersedes,
1642
+ fact.model_dump_json(),
1643
+ ),
1644
+ )
1645
+
1646
+ def _index_write_domain(self, domain: Domain) -> None:
1647
+ self._conn.execute(
1648
+ "INSERT INTO domains (domain_id, slug, status, supersedes, data) "
1649
+ "VALUES (?, ?, ?, ?, ?) "
1650
+ "ON CONFLICT(domain_id) DO UPDATE SET slug=excluded.slug, "
1651
+ "status=excluded.status, supersedes=excluded.supersedes, data=excluded.data",
1652
+ (
1653
+ domain.domain_id,
1654
+ domain.slug,
1655
+ domain.status.value,
1656
+ domain.supersedes,
1657
+ domain.model_dump_json(),
1658
+ ),
1659
+ )
1660
+
1661
+ def _index_write_binding(self, binding: AnchorBinding) -> None:
1662
+ self._conn.execute(
1663
+ "INSERT INTO anchor_bindings (record_id, entity_id, data) VALUES (?, ?, ?) "
1664
+ "ON CONFLICT(record_id, entity_id) DO UPDATE SET data=excluded.data",
1665
+ (binding.record_id, binding.entity_id, binding.model_dump_json()),
1666
+ )
1667
+
1668
+ def _index_write_initiative(self, initiative: Initiative) -> None:
1669
+ self._conn.execute(
1670
+ "INSERT INTO initiatives (id, name, data) VALUES (?, ?, ?) "
1671
+ "ON CONFLICT(id) DO UPDATE SET name=excluded.name, data=excluded.data",
1672
+ (initiative.id, initiative.name, initiative.model_dump_json()),
1673
+ )
1674
+
1675
+ def _index_get_binding(self, record_id: str, entity_id: str) -> AnchorBinding | None:
1676
+ row = self._conn.execute(
1677
+ "SELECT data FROM anchor_bindings WHERE record_id = ? AND entity_id = ?",
1678
+ (record_id, entity_id),
1679
+ ).fetchone()
1680
+ return AnchorBinding.model_validate_json(row["data"]) if row else None
1681
+
1682
+ # -- entities -------------------------------------------------------------
1683
+
1684
+ def upsert_entity(self, entity: Entity) -> Entity:
1685
+ """Persist an entity. Entities are created lazily by the capture path.
1686
+
1687
+ The canonical file (``entities/<id>.json`` — identity only) is written ONLY when
1688
+ the entity is new, or one of its identity fields actually changed (e.g. a rename
1689
+ that moves its ``descriptor`` — see ``sync.rebind_entity``'s "moved" outcome). A
1690
+ pure engine-mapping refresh (``last_seen_node_id``/``last_seen_graph_version``/
1691
+ ``last_seen_community`` — see ``sync.py``) touches ONLY the index; the committed
1692
+ file is untouched (design §1: "sync never writes a committed file again").
1693
+
1694
+ DERIVED entities (``community:*`` abstract entities — see ``_is_derived_entity``)
1695
+ never get a canonical file either, even on this direct path: the guard lives HERE,
1696
+ not just in ``get_or_create_abstract_entity``'s caller-side special case, so any
1697
+ current or future caller that constructs a brand-new ``community:*`` ``Entity`` and
1698
+ upserts it directly gets the same index-only treatment — the invariant is airtight
1699
+ by construction, not by caller discipline (see design/superpowers/specs/
1700
+ 2026-07-10-derived-community-bindings-design.md)."""
1701
+ with self._mutation():
1702
+ existing = self.get_entity(entity.entity_id)
1703
+ identity_changed = existing is None or _entity_identity_payload(
1704
+ existing
1705
+ ) != _entity_identity_payload(entity)
1706
+ if identity_changed and not _is_derived_entity(entity):
1707
+ self._write_entity_canonical(entity)
1708
+ self._touch_digest()
1709
+ self._index_write_entity(entity)
1710
+ return entity
1711
+
1712
+ def get_entity(self, entity_id: str) -> Entity | None:
1713
+ with self._lock:
1714
+ row = self._conn.execute(
1715
+ "SELECT data FROM entities WHERE entity_id = ?", (entity_id,)
1716
+ ).fetchone()
1717
+ return Entity.model_validate_json(row["data"]) if row else None
1718
+
1719
+ def find_entity(self, name: str, file_path: str | None) -> Entity | None:
1720
+ """Dedup lookup by canonical name + file_path (see schema.canonicalize).
1721
+
1722
+ Deterministic under a duplicate (design D4): a duplicate logical identity is LEGAL
1723
+ (two branches each minting the same name produce two ULIDs -> two files -> a clean
1724
+ git merge, see D2), so more than one row can match here. The LOWEST ``entity_id``
1725
+ wins — ULIDs sort lexicographically by mint time, so "lowest" means "the one that
1726
+ existed first", and it is stable across processes and index rebuilds. Without this,
1727
+ a bare table scan returns whichever row SQLite happens to hand back first, which is
1728
+ an accident of insertion order, not a rule — see
1729
+ ``tests/test_entity_duplicate_resolution.py``'s inversion construction, which pins
1730
+ exactly that. Still a full Python scan (real, and a separate change — see spec §5);
1731
+ not this wave's subject."""
1732
+ target = canonicalize(name)
1733
+ with self._lock:
1734
+ rows = self._conn.execute("SELECT data FROM entities").fetchall()
1735
+ matches: list[Entity] = []
1736
+ for row in rows:
1737
+ e = Entity.model_validate_json(row["data"])
1738
+ desc_file = e.descriptor.file_path if e.descriptor else None
1739
+ e_name = e.descriptor.name if e.descriptor else e.canonical_name
1740
+ if canonicalize(e_name) == target and desc_file == file_path:
1741
+ matches.append(e)
1742
+ return min(matches, key=lambda e: e.entity_id) if matches else None
1743
+
1744
+ def get_or_create_entity(self, descriptor: Descriptor) -> Entity:
1745
+ """Atomic get-or-create for a CONCRETE entity (design D3).
1746
+
1747
+ The lookup and the mint run in one ``BEGIN IMMEDIATE`` transaction, so a second
1748
+ process cannot execute its own lookup in between and mint a rival id for the same
1749
+ logical entity — ``self._lock`` is per-instance and never protected against that
1750
+ (external review, finding 2; see ``get_or_create_abstract_entity``'s docstring for
1751
+ the full mechanism). Collapses the longhand ``find_entity`` + ``upsert_entity``
1752
+ sequence that used to be written out at two call sites (``anchoring.py``,
1753
+ ``capture.py``) onto one place, so a third caller cannot repeat the same omission.
1754
+
1755
+ The nested ``upsert_entity`` runs at depth 2 — it neither begins nor commits.
1756
+
1757
+ A descriptor with NO ``file_path`` falls back to a name-only lookup before minting.
1758
+ ``find_entity`` matches on name AND path, so a path-less descriptor could never find
1759
+ the entity that already existed for that symbol with a real path — it minted a
1760
+ path-less twin that resolves to nothing in the graph, hiding every decision anchored
1761
+ through it (38 twin names across the four committed stores). ``doc_import`` produces
1762
+ exactly such descriptors for backticked mentions: it has no path it could know.
1763
+ Adoption requires the candidates to agree on ONE path — a name owned by two files is
1764
+ genuinely ambiguous and keeps the old behaviour rather than guessing (measured: all
1765
+ 38 were unambiguous). Abstract entities carry no descriptor and are never adopted."""
1766
+ with self._mutation(immediate=True):
1767
+ existing = self.resolve_descriptor(descriptor.name, descriptor.file_path)
1768
+ if existing is not None:
1769
+ return existing
1770
+ return self.upsert_entity(Entity(canonical_name=descriptor.name, descriptor=descriptor))
1771
+
1772
+ def resolve_descriptor(self, name: str, file_path: str | None) -> Entity | None:
1773
+ """One descriptor -> one existing entity: the lookup EVERY caller must use, read or
1774
+ write, or the two disagree.
1775
+
1776
+ ``get_or_create_entity``'s find-half, exposed so the read path resolves a descriptor
1777
+ to the same entity the write path bound it to. Splitting them is how the twin bug
1778
+ worked in the first place, and re-splitting them broke capture's dedup: once the
1779
+ write adopts the carrier, a reader still keying on ``(name, None)`` finds nothing,
1780
+ so an identical re-proposal sails through ``_is_duplicate``/``_is_duplicate_fact``
1781
+ (external review, finding 1 — demonstrated, not theorised). Returns None when the
1782
+ descriptor names nothing yet; only ``get_or_create_entity`` mints.
1783
+
1784
+ ONE scan for the path-less case, not two. Every entity lookup here is a full Python
1785
+ scan that re-parses each row, and ``resolve_seeds`` runs this per graph node — a
1786
+ path-less node is not exotic (45 in this repo's own graph, 6646 in airflow's). Asking
1787
+ ``_adopt_path_carrying_entity`` and then falling back to ``find_entity`` scanned
1788
+ twice and measured 2.00 ms/call against ``find_entity``'s 0.99 ms at 219 entities:
1789
+ 13 s of pure scanning on an airflow-sized seed. Both answers come out of the same
1790
+ candidate list instead.
1791
+
1792
+ ``""`` counts as no path. Graphify emits ``source_file: ""`` — an empty STRING, not
1793
+ a missing key — for every path-less node (measured: 6646/6646 in airflow's graph,
1794
+ 45/45 in this repo's), so a reader-fed lookup arrives as ``(name, "")``. Keying
1795
+ adoption on ``is None`` alone left the read path inert while the write path adopted,
1796
+ which is the very split this method exists to close; ``retrieval.py`` already
1797
+ documents the same trap for its own truthiness check. No entity can carry ``""`` as
1798
+ its ``file_path`` (descriptors hold a real path or None), so folding the two costs
1799
+ nothing elsewhere."""
1800
+ if file_path:
1801
+ return self.find_entity(name, file_path)
1802
+ candidates = self.find_entities_by_name(name)
1803
+ adopted = self._adopt_path_carrying_entity(candidates)
1804
+ if adopted is not None:
1805
+ return adopted
1806
+ # No carrier: reproduce find_entity(name, None) — same predicate (``desc_file is
1807
+ # None``, which an abstract entity satisfies by having no descriptor at all), same
1808
+ # D4 tie-break — off the list already in hand.
1809
+ exact = [
1810
+ e for e in candidates if (e.descriptor.file_path if e.descriptor else None) is None
1811
+ ]
1812
+ return min(exact, key=lambda e: e.entity_id) if exact else None
1813
+
1814
+ def _adopt_path_carrying_entity(self, candidates: list[Entity]) -> Entity | None:
1815
+ """The CONCRETE entity a path-less descriptor should adopt, out of ``candidates``
1816
+ (every entity sharing its canonical name), when they agree on exactly one path.
1817
+ Returns None when nothing carries a path, or when two files do — the caller then
1818
+ falls back to the plain path-less lookup. Ties on one path resolve by lowest
1819
+ ``entity_id``, the store's one duplicate rule (design D4)."""
1820
+ concrete = [
1821
+ e for e in candidates if e.descriptor is not None and e.descriptor.file_path is not None
1822
+ ]
1823
+ if len({e.descriptor.file_path for e in concrete if e.descriptor}) != 1:
1824
+ return None
1825
+ return min(concrete, key=lambda e: e.entity_id)
1826
+
1827
+ def find_entities_by_name(self, name: str) -> list[Entity]:
1828
+ """Name-only scan across all entities, ignoring ``file_path`` — the fallback
1829
+ ``find_entity`` (the MCP tool) uses when no exact descriptor match exists. May
1830
+ return more than one entity when a name is reused across files; the caller decides
1831
+ whether that's ambiguous (never guess which one the query meant)."""
1832
+ target = canonicalize(name)
1833
+ with self._lock:
1834
+ rows = self._conn.execute("SELECT data FROM entities").fetchall()
1835
+ out: list[Entity] = []
1836
+ for row in rows:
1837
+ e = Entity.model_validate_json(row["data"])
1838
+ e_name = e.descriptor.name if e.descriptor else e.canonical_name
1839
+ if canonicalize(e_name) == target:
1840
+ out.append(e)
1841
+ return out
1842
+
1843
+ def get_or_create_abstract_entity(self, canonical_name: str) -> Entity:
1844
+ """Get-or-create an abstract Entity (community / domain / tag / initiative anchor).
1845
+ Reused.
1846
+
1847
+ The whole check-then-create sequence runs inside ONE ``BEGIN IMMEDIATE`` transaction
1848
+ (design D1/D3, entity-identity-uniqueness spec), not merely under ``self._lock``:
1849
+ ``self._lock`` is per-``Store``-instance, so it serializes two THREADS sharing one
1850
+ instance but does nothing at all across two separate ``Store`` instances (an MCP
1851
+ server, a hook, and a CLI each hold their own) — those raced their lookups straight
1852
+ through it and each minted a different id for the same logical entity (external
1853
+ review, finding 2). ``BEGIN IMMEDIATE`` takes SQLite's write lock before the lookup
1854
+ even runs, so a second connection's own ``BEGIN IMMEDIATE`` blocks until this one
1855
+ commits, and its lookup then sees this one's row instead of racing it.
1856
+
1857
+ ``community:*`` names are DERIVED (see ``_is_derived_entity``): the new entity gets
1858
+ an index row only — never an ``entities/<id>.json`` canonical file — since Leiden
1859
+ renumbers communities on every rebuild and a committed file would just accumulate
1860
+ one dead entity per renumbering forever. Every other abstract name (``domain:*``,
1861
+ ``tag:*``, initiative anchors, ...) is unchanged: a brand-new one still writes its
1862
+ canonical file. The derived check itself now lives entirely in ``upsert_entity``
1863
+ (see its docstring) — this just calls it unconditionally. The nested
1864
+ ``upsert_entity`` runs at depth 2 — it neither begins nor commits (see
1865
+ ``_mutation``).
1866
+
1867
+ Deterministic under a duplicate (design D4): sorts matches and returns the LOWEST
1868
+ ``entity_id`` — see ``find_entity``'s docstring for why. Left unsorted, this inline
1869
+ lookup could disagree with ``find_abstract_entity``'s own read path about which
1870
+ duplicate wins; sorting both the same way means the create path and the read path
1871
+ always agree."""
1872
+ with self._mutation(immediate=True):
1873
+ rows = self._conn.execute(
1874
+ "SELECT data FROM entities WHERE canonical_name = ?", (canonical_name,)
1875
+ ).fetchall()
1876
+ matches = [
1877
+ e
1878
+ for e in (Entity.model_validate_json(row["data"]) for row in rows)
1879
+ if e.kind == EntityKind.ABSTRACT
1880
+ ]
1881
+ if matches:
1882
+ return min(matches, key=lambda e: e.entity_id)
1883
+ entity = Entity(canonical_name=canonical_name, kind=EntityKind.ABSTRACT)
1884
+ return self.upsert_entity(entity)
1885
+
1886
+ # -- decisions ----------------------------------------------------------
1887
+
1888
+ def add_decision(self, decision: Decision, *, close_predecessor: bool = True) -> Decision:
1889
+ """Append a decision, enforcing the write-path invariants.
1890
+
1891
+ If ``decision`` supersedes another, the predecessor is closed (its ``valid_to`` is
1892
+ set and its status flipped to ``superseded``) in the same transaction — keeping the
1893
+ "a superseded decision must have a successor" invariant true at all times. This is
1894
+ the default (``close_predecessor=True``) and unchanged for every existing caller
1895
+ (``capture.propose``, ``server._supersede_decision_impl``, doc-import's non-
1896
+ ``--propose`` path).
1897
+
1898
+ The close only ever applies to a predecessor that is STILL OPEN (status
1899
+ ``accepted`` or ``proposed`` — mirrors :meth:`ratify`'s own guard). A predecessor
1900
+ that is ALREADY terminal (``superseded``/``rejected``/``deprecated`` — see design
1901
+ §7) is left completely untouched: its hot file (which may already have been
1902
+ compacted into an archive segment — see :meth:`compact`) is never rewritten, and
1903
+ its status never flips again. Without this guard, superseding an archived
1904
+ ``rejected`` decision would resurrect a mutated copy of it as a fresh hot file —
1905
+ permanently corruption-shaped from compaction's point of view (every teammate's
1906
+ next cold reload would warn about a hot/archive mismatch, and the record could
1907
+ never compact again). The successor's ``supersedes`` field still records the
1908
+ intended relationship either way — append-only semantics: the link is a fact that
1909
+ happened, even when the target itself can no longer be edited.
1910
+
1911
+ The SUCCESSOR is written FIRST, and the predecessor is only flipped afterwards
1912
+ (mirrors :meth:`ratify`'s order): if the successor's write fails partway, nothing
1913
+ has flipped the predecessor's canonical file yet, so the store is left in the
1914
+ tolerated "deferred supersession" shape (both records still live, successor simply
1915
+ absent) rather than a durable "superseded with zero successors" record that a later
1916
+ digest reload would silently adopt as-is.
1917
+
1918
+ ``close_predecessor=False`` (doc-import's ``--propose`` + edited-doc path, design
1919
+ §3 option (b)) writes the successor with ``supersedes`` set but leaves the
1920
+ predecessor exactly as it is — an accepted decision must never be silently closed
1921
+ by a mere proposal nobody has reviewed yet. The existence check below (``supersedes``
1922
+ must reference a real decision) still runs either way; only the CLOSE side effect is
1923
+ skipped. The predecessor closes later, when a human actually ratifies the successor
1924
+ — see :meth:`ratify`'s deferred-supersession logic.
1925
+
1926
+ Raises if ``decision.id`` already exists: this method is append-only, and without
1927
+ the guard an existing id would silently rewrite the row via the ``ON CONFLICT DO
1928
+ UPDATE`` in ``_write_decision``, erasing history. Nothing legitimate re-adds the
1929
+ same id — ``ratify``/``drop`` and the supersede path above call ``_write_decision``
1930
+ directly.
1931
+ """
1932
+ with self._mutation():
1933
+ if self.get_decision(decision.id) is not None:
1934
+ raise ValueError(
1935
+ f"decision {decision.id} already exists — add_decision is append-only; "
1936
+ "use its supersedes field or ratify/drop instead"
1937
+ )
1938
+ predecessor = None
1939
+ if decision.supersedes:
1940
+ predecessor = self.get_decision(decision.supersedes)
1941
+ if predecessor is None:
1942
+ raise ValueError(
1943
+ f"supersedes references unknown decision {decision.supersedes!r}"
1944
+ )
1945
+
1946
+ self._write_decision(decision)
1947
+
1948
+ if (
1949
+ close_predecessor
1950
+ and predecessor is not None
1951
+ and predecessor.status in (DecisionStatus.ACCEPTED, DecisionStatus.PROPOSED)
1952
+ ):
1953
+ predecessor.status = DecisionStatus.SUPERSEDED
1954
+ predecessor.valid_to = predecessor.valid_to or max(
1955
+ decision.valid_from, predecessor.valid_from
1956
+ )
1957
+ self._write_decision(predecessor)
1958
+ return decision
1959
+
1960
+ def _write_decision(self, decision: Decision) -> None:
1961
+ with self._lock:
1962
+ self._write_decision_canonical(decision)
1963
+ self._index_write_decision(decision)
1964
+ self._touch_digest()
1965
+
1966
+ def _write_fact(self, fact: Fact) -> None:
1967
+ with self._lock:
1968
+ self._write_fact_canonical(fact)
1969
+ self._index_write_fact(fact)
1970
+ self._touch_digest()
1971
+
1972
+ def get_decision(self, decision_id: str) -> Decision | None:
1973
+ with self._lock:
1974
+ row = self._conn.execute(
1975
+ "SELECT data FROM decisions WHERE id = ?", (decision_id,)
1976
+ ).fetchone()
1977
+ return Decision.model_validate_json(row["data"]) if row else None
1978
+
1979
+ def iter_decisions(self) -> Iterator[Decision]:
1980
+ with self._lock:
1981
+ rows = self._conn.execute("SELECT data FROM decisions").fetchall()
1982
+ for row in rows:
1983
+ yield Decision.model_validate_json(row["data"])
1984
+
1985
+ def find_decision_by_title(
1986
+ self, canonical_title: str, source: str, ref: str | None
1987
+ ) -> Decision | None:
1988
+ """Idempotency lookup for bulk importers (see importer.py): the first non-superseded
1989
+ decision whose canonicalized title matches ``canonical_title``, whose
1990
+ ``provenance.source`` equals ``source``, AND whose ``provenance.ref`` equals ``ref``.
1991
+ ``ref`` keys on the rationale's origin (its ``file_path``, falling back to
1992
+ ``node_id`` when the rationale has none) — title alone over-dedups: identical first
1993
+ lines in different files are distinct memories and must both import (S2 review).
1994
+ Full-table scan — acceptable at import volume (hundreds, not a hot retrieval path)."""
1995
+ with self._lock:
1996
+ rows = self._conn.execute(
1997
+ "SELECT data FROM decisions WHERE status != ?",
1998
+ (DecisionStatus.SUPERSEDED.value,),
1999
+ ).fetchall()
2000
+ for row in rows:
2001
+ d = Decision.model_validate_json(row["data"])
2002
+ if (
2003
+ d.provenance.source == source
2004
+ and d.provenance.ref == ref
2005
+ and canonicalize(d.title) == canonical_title
2006
+ ):
2007
+ return d
2008
+ return None
2009
+
2010
+ def find_decisions_by_ref(
2011
+ self, source: str, ref: str, *, statuses: tuple[DecisionStatus, ...] | None = None
2012
+ ) -> list[Decision]:
2013
+ """Idempotency lookup for doc-import (see doc_import.py): EVERY OPEN (status
2014
+ ``accepted`` or ``proposed``) decision whose ``provenance.source`` equals
2015
+ ``source`` and whose ``provenance.ref`` equals ``ref`` — title-AGNOSTIC, unlike
2016
+ ``find_decision_by_title``. Importer #1 never supersedes, so it needs an exact
2017
+ title match too (a changed title there just means "a different rationale");
2018
+ doc-import must find every CURRENTLY-OPEN record for a given doc path, even when
2019
+ titles differ (the doc was edited) or a proposal is still awaiting ratification.
2020
+
2021
+ Replaces the former singular ``find_decision_by_ref`` (returned only the FIRST
2022
+ non-superseded match in whatever order sqlite's full-table scan happened to
2023
+ produce). That was the root cause of a real regression: with ``--propose``, an
2024
+ edited ref can have TWO open records at once — a still-``accepted`` predecessor
2025
+ and a newer ``proposed`` draft — and the single-result lookup would return
2026
+ whichever came first in scan order (typically the older accepted row, since it was
2027
+ inserted first), silently hiding the pending proposal from the idempotency check.
2028
+ Re-running the importer then compared the fresh parse against the WRONG record,
2029
+ never matched, and proposed a fresh duplicate every single run. Returning the full
2030
+ set lets the caller compare against every open record and pick the right
2031
+ predecessor deliberately instead of trusting scan order.
2032
+
2033
+ ``rejected`` (a human explicitly declined it via ``ratify``/``drop``) and
2034
+ ``superseded`` records are excluded — a dead-end draft or a closed predecessor
2035
+ must never block or feed a fresh import; only genuinely open state counts.
2036
+
2037
+ ``statuses`` overrides that default set. Doc-import passes ``REJECTED`` alongside
2038
+ the open ones: a rejected record is invisible to the default filter, so an
2039
+ unchanged ``status: rejected`` document was re-imported as a brand-new record on
2040
+ every single run (review finding 1 — three runs, three identical records, and the
2041
+ report line said "0 existing").
2042
+
2043
+ Deterministic order: sorted by ``id`` ascending (a ULID, so this is also creation
2044
+ order — oldest first). Full-table scan, like ``find_decision_by_title``; same
2045
+ volume rationale.
2046
+ """
2047
+ wanted = statuses or (DecisionStatus.ACCEPTED, DecisionStatus.PROPOSED)
2048
+ placeholders = ", ".join("?" for _ in wanted)
2049
+ with self._lock:
2050
+ rows = self._conn.execute(
2051
+ f"SELECT data FROM decisions WHERE status IN ({placeholders})",
2052
+ tuple(st.value for st in wanted),
2053
+ ).fetchall()
2054
+ out: list[Decision] = []
2055
+ for row in rows:
2056
+ d = Decision.model_validate_json(row["data"])
2057
+ if d.provenance.source == source and d.provenance.ref == ref:
2058
+ out.append(d)
2059
+ out.sort(key=lambda d: d.id)
2060
+ return out
2061
+
2062
+ # -- facts ----------------------------------------------------------------
2063
+ #
2064
+ # Append-only, mirroring decisions: add_fact writes a brand-new row; supersession
2065
+ # closes the predecessor (status -> superseded) and writes the successor first, in the
2066
+ # same transaction (see add_decision for the parallel). A fact's `supports` links it to
2067
+ # the decision(s) it informed — every id must reference an existing Decision.
2068
+
2069
+ def add_fact(self, fact: Fact, *, close_predecessor: bool = True) -> Fact:
2070
+ """Append a fact, enforcing the write-path invariants — mirrors :meth:`add_decision`.
2071
+
2072
+ Every id in ``fact.supports`` must reference an existing :class:`Decision` (facts
2073
+ inform decisions; the reverse link isn't a thing). If ``fact`` supersedes another
2074
+ fact, the predecessor is closed (``valid_to`` set, status flipped to
2075
+ ``superseded``) in the same transaction — same "a superseded record must have a
2076
+ successor" guarantee ``add_decision`` gives, and the same STILL-OPEN guard (only a
2077
+ predecessor with status ``accepted``/``proposed`` is closed). The SUCCESSOR is
2078
+ written FIRST, mirroring ``add_decision``'s crash-ordering rationale.
2079
+
2080
+ Raises if ``fact.id`` already exists: append-only, mirroring ``add_decision`` — an
2081
+ existing id must never be silently rewritten via ``_index_write_fact``'s
2082
+ ``ON CONFLICT DO UPDATE``.
2083
+ """
2084
+ with self._mutation():
2085
+ if self.get_fact(fact.id) is not None:
2086
+ raise ValueError(
2087
+ f"fact {fact.id} already exists — add_fact is append-only; "
2088
+ "use its supersedes field instead"
2089
+ )
2090
+ for sid in fact.supports:
2091
+ if self.get_decision(sid) is None:
2092
+ raise ValueError(f"supports references unknown decision: {sid}")
2093
+
2094
+ predecessor = None
2095
+ if fact.supersedes:
2096
+ predecessor = self.get_fact(fact.supersedes)
2097
+ if predecessor is None:
2098
+ raise ValueError(f"supersedes references unknown fact {fact.supersedes!r}")
2099
+
2100
+ self._write_fact(fact)
2101
+
2102
+ if (
2103
+ close_predecessor
2104
+ and predecessor is not None
2105
+ and predecessor.status in (DecisionStatus.ACCEPTED, DecisionStatus.PROPOSED)
2106
+ ):
2107
+ predecessor.status = DecisionStatus.SUPERSEDED
2108
+ predecessor.valid_to = predecessor.valid_to or max(
2109
+ fact.valid_from, predecessor.valid_from
2110
+ )
2111
+ self._write_fact(predecessor)
2112
+ return fact
2113
+
2114
+ def get_fact(self, fact_id: str) -> Fact | None:
2115
+ with self._lock:
2116
+ row = self._conn.execute("SELECT data FROM facts WHERE id = ?", (fact_id,)).fetchone()
2117
+ return Fact.model_validate_json(row["data"]) if row else None
2118
+
2119
+ def iter_facts(self) -> Iterator[Fact]:
2120
+ with self._lock:
2121
+ rows = self._conn.execute("SELECT data FROM facts").fetchall()
2122
+ for row in rows:
2123
+ yield Fact.model_validate_json(row["data"])
2124
+
2125
+ def iter_proposed_facts(self) -> Iterator[Fact]:
2126
+ """Facts awaiting ratification (status == proposed) — mirrors :meth:`iter_proposed`."""
2127
+ with self._lock:
2128
+ rows = self._conn.execute("SELECT data FROM facts WHERE status = 'proposed'").fetchall()
2129
+ for row in rows:
2130
+ yield Fact.model_validate_json(row["data"])
2131
+
2132
+ def facts_for_decision(self, decision_id: str) -> list[Fact]:
2133
+ """Live + proposed facts (status ACCEPTED/PROPOSED, any validity) whose
2134
+ ``supports`` contains ``decision_id`` — filtered in Python; volumes are small (same
2135
+ rationale as ``find_decision_by_title``'s full-table scan)."""
2136
+ with self._lock:
2137
+ rows = self._conn.execute(
2138
+ "SELECT data FROM facts WHERE status IN (?, ?)",
2139
+ (DecisionStatus.ACCEPTED.value, DecisionStatus.PROPOSED.value),
2140
+ ).fetchall()
2141
+ out: list[Fact] = []
2142
+ for row in rows:
2143
+ f = Fact.model_validate_json(row["data"])
2144
+ if decision_id in f.supports:
2145
+ out.append(f)
2146
+ return out
2147
+
2148
+ def valid_facts_for_entity(self, entity_id: str, as_of: datetime | None = None) -> list[Fact]:
2149
+ """Currently-valid, non-superseded facts bound to entity_id via a live|degraded
2150
+ binding (orphaned bindings are skipped) — mirrors :meth:`valid_decisions_for_entity`.
2151
+ ``as_of`` defaults to now (UTC)."""
2152
+ as_of = as_of or datetime.now(UTC)
2153
+ out: list[Fact] = []
2154
+ for b in self.bindings_for_entity(entity_id):
2155
+ if b.status not in ("live", "degraded"):
2156
+ continue
2157
+ f = self.get_fact(b.record_id)
2158
+ if f is None or f.status in (DecisionStatus.SUPERSEDED, DecisionStatus.REJECTED):
2159
+ continue
2160
+ if f.valid_to is None or f.valid_to > as_of:
2161
+ out.append(f)
2162
+ return out
2163
+
2164
+ # -- anchor bindings ----------------------------------------------------
2165
+
2166
+ def add_binding(self, binding: AnchorBinding) -> AnchorBinding:
2167
+ """Link a record (a decision OR a fact) to an entity. The entity must already
2168
+ exist, and ``binding.record_id`` must resolve to a real decision or fact —
2169
+ neither existing is a hard error.
2170
+
2171
+ The canonical file (``bindings/<record_id>.json`` — the anchor set, no
2172
+ ``status``) is rewritten ONLY when this binding is new, or its identity
2173
+ (entity_id/tier/relation/weight) actually changed. A pure status flip (sync's
2174
+ live/degraded/orphaned transitions — see ``sync._set_leaf_status``,
2175
+ ``_repoint_communities``) touches ONLY the index.
2176
+
2177
+ An identity change on a binding whose entity is DERIVED (a ``community:*``
2178
+ abstract entity — see ``_is_derived_entity``) never triggers a canonical rewrite
2179
+ either: community rebinding is index-only, at capture time as much as at sync
2180
+ time (see design/superpowers/specs/2026-07-10-derived-community-bindings-design.md).
2181
+ Such a binding would be filtered back out of the payload anyway (see
2182
+ ``_write_bindings_canonical_for_record``) — skipping the rewrite here just avoids
2183
+ a pointless write (and ``_touch_digest`` bump) whose result is byte-identical to
2184
+ what's already on disk.
2185
+ """
2186
+ with self._mutation():
2187
+ entity = self.get_entity(binding.entity_id)
2188
+ if entity is None:
2189
+ raise ValueError(f"binding references unknown entity {binding.entity_id!r}")
2190
+ if (
2191
+ self.get_decision(binding.record_id) is None
2192
+ and self.get_fact(binding.record_id) is None
2193
+ ):
2194
+ raise ValueError(f"binding references unknown record {binding.record_id!r}")
2195
+ existing = self._index_get_binding(binding.record_id, binding.entity_id)
2196
+ identity_changed = existing is None or _binding_identity_payload(
2197
+ existing
2198
+ ) != _binding_identity_payload(binding)
2199
+ self._index_write_binding(binding)
2200
+ if identity_changed and not _is_derived_entity(entity):
2201
+ self._write_bindings_canonical_for_record(binding.record_id)
2202
+ self._touch_digest()
2203
+ return binding
2204
+
2205
+ def bindings_for_entity(self, entity_id: str) -> list[AnchorBinding]:
2206
+ with self._lock:
2207
+ rows = self._conn.execute(
2208
+ "SELECT data FROM anchor_bindings WHERE entity_id = ?", (entity_id,)
2209
+ ).fetchall()
2210
+ return [AnchorBinding.model_validate_json(r["data"]) for r in rows]
2211
+
2212
+ def bindings_for_record(self, record_id: str) -> list[AnchorBinding]:
2213
+ with self._lock:
2214
+ rows = self._conn.execute(
2215
+ "SELECT data FROM anchor_bindings WHERE record_id = ?", (record_id,)
2216
+ ).fetchall()
2217
+ return [AnchorBinding.model_validate_json(r["data"]) for r in rows]
2218
+
2219
+ def valid_decisions_for_entity(
2220
+ self, entity_id: str, as_of: datetime | None = None
2221
+ ) -> list[Decision]:
2222
+ """Currently-valid, non-superseded decisions bound to entity_id via a live|degraded
2223
+ binding (orphaned bindings are skipped). ``as_of`` defaults to now (UTC)."""
2224
+ as_of = as_of or datetime.now(UTC)
2225
+ out: list[Decision] = []
2226
+ for b in self.bindings_for_entity(entity_id):
2227
+ if b.status not in ("live", "degraded"):
2228
+ continue
2229
+ d = self.get_decision(b.record_id)
2230
+ if d is None or d.status in (DecisionStatus.SUPERSEDED, DecisionStatus.REJECTED):
2231
+ continue
2232
+ if d.valid_to is None or d.valid_to > as_of:
2233
+ out.append(d)
2234
+ return out
2235
+
2236
+ def superseded_for_entity(self, entity_id: str) -> list[Decision]:
2237
+ """Superseded decisions bound to entity_id (for the 'tried, reverted' one-liner)."""
2238
+ out: list[Decision] = []
2239
+ for b in self.bindings_for_entity(entity_id):
2240
+ d = self.get_decision(b.record_id)
2241
+ if d is not None and d.status == DecisionStatus.SUPERSEDED:
2242
+ out.append(d)
2243
+ return out
2244
+
2245
+ def decisions_by_scope(self, scope: Scope, as_of: datetime | None = None) -> list[Decision]:
2246
+ """Currently-valid, non-superseded decisions with the given scope."""
2247
+ as_of = as_of or datetime.now(UTC)
2248
+ out: list[Decision] = []
2249
+ for d in self.iter_decisions():
2250
+ if (
2251
+ d.scope == scope
2252
+ and d.status not in (DecisionStatus.SUPERSEDED, DecisionStatus.REJECTED)
2253
+ and (d.valid_to is None or d.valid_to > as_of)
2254
+ ):
2255
+ out.append(d)
2256
+ return out
2257
+
2258
+ def find_abstract_entity(self, canonical_name: str) -> Entity | None:
2259
+ """Read-only lookup of an abstract entity by canonical_name (never creates).
2260
+
2261
+ Deterministic under a duplicate (design D4): returns the LOWEST ``entity_id`` when
2262
+ more than one row matches — see ``find_entity``'s docstring for the full reasoning
2263
+ (a duplicate is legal, D2; "lowest" is stable, not merely "first-returned")."""
2264
+ with self._lock:
2265
+ rows = self._conn.execute(
2266
+ "SELECT data FROM entities WHERE canonical_name = ?", (canonical_name,)
2267
+ ).fetchall()
2268
+ matches = [
2269
+ e
2270
+ for e in (Entity.model_validate_json(row["data"]) for row in rows)
2271
+ if e.kind == EntityKind.ABSTRACT
2272
+ ]
2273
+ return min(matches, key=lambda e: e.entity_id) if matches else None
2274
+
2275
+ def iter_initiatives(self) -> Iterator[Initiative]:
2276
+ with self._lock:
2277
+ rows = self._conn.execute("SELECT data FROM initiatives").fetchall()
2278
+ for row in rows:
2279
+ yield Initiative.model_validate_json(row["data"])
2280
+
2281
+ def iter_concrete_entities(self) -> Iterator[Entity]:
2282
+ """Concrete entities with a descriptor — the sync job's rebind population."""
2283
+ with self._lock:
2284
+ rows = self._conn.execute("SELECT data FROM entities").fetchall()
2285
+ for row in rows:
2286
+ e = Entity.model_validate_json(row["data"])
2287
+ if e.kind == EntityKind.CONCRETE and e.descriptor is not None:
2288
+ yield e
2289
+
2290
+ # -- domains --------------------------------------------------------------
2291
+ #
2292
+ # Append-only, mirroring decisions: add_domain writes a brand-new row; supersede_domain
2293
+ # closes the predecessor (status -> superseded) and writes a new row with `supersedes`
2294
+ # set in the same transaction (see add_decision for the parallel). Slug uniqueness is
2295
+ # enforced only against *live* domains (proposed | accepted) so a superseded or dropped
2296
+ # domain frees its slug back up — see docs/concepts/mind-model.md#domain-lifecycle.
2297
+
2298
+ def add_domain(self, domain: Domain) -> Domain:
2299
+ """Append a new Domain, enforcing slug-uniqueness (among live domains) and
2300
+ parent existence + acyclicity.
2301
+
2302
+ Raises if ``domain.domain_id`` already exists: append-only, mirroring
2303
+ ``add_decision`` — without the guard, an existing id would silently rewrite the
2304
+ row via the ``ON CONFLICT DO UPDATE`` in ``_write_domain``, erasing history.
2305
+ Reversal goes through ``supersede_domain`` instead.
2306
+ """
2307
+ with self._mutation():
2308
+ if self.get_domain(domain.domain_id) is not None:
2309
+ raise ValueError(f"domain {domain.domain_id} already exists — use supersede_domain")
2310
+ self._validate_domain_write(domain)
2311
+ self._write_domain(domain)
2312
+ return domain
2313
+
2314
+ def supersede_domain(self, old_id: str, new_domain: Domain) -> Domain:
2315
+ """Append-only reversal: close ``old_id`` (status -> superseded) and write
2316
+ ``new_domain`` (whose ``supersedes`` must equal ``old_id``) in the same
2317
+ transaction. The same slug is allowed because the successor's slug-uniqueness
2318
+ check excludes ``old_id`` — see ``_validate_domain_write``'s ``exclude_id``.
2319
+
2320
+ The successor is validated FIRST, before the predecessor's row is touched: if
2321
+ validation raises (e.g. a slug collision against some *other* live domain, or a
2322
+ bad parent), nothing has been written yet — neither the index nor a canonical
2323
+ file — so there is nothing to leak. Once validation passes, the SUCCESSOR is
2324
+ written first and the predecessor is flipped only afterwards (mirrors
2325
+ ``add_decision``/``ratify``'s order): a canonical-file write is a durable
2326
+ filesystem replace that ``self._conn.rollback()`` cannot undo, so flipping the
2327
+ predecessor FIRST (the old order) could leave a durable "superseded with zero
2328
+ successors" domain on disk if the successor's write then failed. Writing the
2329
+ successor first means a failure there leaves the predecessor untouched — the
2330
+ tolerated "both still live" shape — and the trailing rollback below still guards
2331
+ the INDEX transaction for the (now much narrower) window between the two writes.
2332
+
2333
+ Also part of that up-front validation: ``new_domain.domain_id`` must not already
2334
+ exist. Without this guard a successor reusing ``old_id`` (or any THIRD domain's
2335
+ id) would silently rewrite that row in place via the ``ON CONFLICT DO UPDATE`` in
2336
+ ``_write_domain`` — erasing history for an id that was supposed to stay append-
2337
+ only, exactly like ``add_domain``'s own existing-id guard.
2338
+ """
2339
+ with self._mutation():
2340
+ old = self.get_domain(old_id)
2341
+ if old is None:
2342
+ raise ValueError(f"supersede references unknown domain {old_id!r}")
2343
+ if new_domain.supersedes != old_id:
2344
+ raise ValueError("new_domain.supersedes must equal old_id")
2345
+ if self.get_domain(new_domain.domain_id) is not None:
2346
+ raise ValueError(
2347
+ f"domain {new_domain.domain_id} already exists — successor must be a new domain"
2348
+ )
2349
+ self._validate_domain_write(new_domain, exclude_id=old_id)
2350
+ self._write_domain(new_domain)
2351
+ old.status = DomainStatus.SUPERSEDED
2352
+ self._write_domain(old)
2353
+ return new_domain
2354
+
2355
+ def _validate_domain_write(self, domain: Domain, *, exclude_id: str | None = None) -> None:
2356
+ """Slug-uniqueness (among live: proposed | accepted) + parent existence and
2357
+ acyclicity. Shared by add_domain and supersede_domain.
2358
+
2359
+ ``exclude_id`` lets supersede_domain validate the successor while the
2360
+ predecessor is still live (not yet flipped to superseded): without it, a
2361
+ same-slug supersede would spuriously collide with its own still-live
2362
+ predecessor.
2363
+
2364
+ Checks EVERY live holder of the slug, not just one (review round 3 fix: the prior
2365
+ version picked a single, arbitrary row via a bare ``SELECT`` with no ``ORDER BY``
2366
+ — harmless with at most one live holder, which write-time uniqueness normally
2367
+ guarantees, but a cross-branch merge can legitimately leave TWO live domains
2368
+ sharing a slug — see design §6 and ``domain_slug_conflicts``. In that shape, the
2369
+ old single-row check could non-deterministically raise or pass depending on which
2370
+ of the two rows sqlite happened to return, even when ``exclude_id`` correctly named
2371
+ the OTHER one: resolving the conflict via ``supersede_domain`` on either duplicate
2372
+ could spuriously fail. Naming every blocking id in the error is also strictly more
2373
+ useful than naming just whichever one won the race.
2374
+ """
2375
+ with self._lock:
2376
+ rows = self._conn.execute(
2377
+ "SELECT domain_id FROM domains WHERE slug = ? AND status IN (?, ?)",
2378
+ (domain.slug, DomainStatus.PROPOSED.value, DomainStatus.ACCEPTED.value),
2379
+ ).fetchall()
2380
+ blocking = sorted(
2381
+ r["domain_id"] for r in rows if r["domain_id"] not in (domain.domain_id, exclude_id)
2382
+ )
2383
+ if blocking:
2384
+ raise ValueError(
2385
+ f"slug {domain.slug!r} already used by a live domain ({', '.join(blocking)})"
2386
+ )
2387
+ if domain.parent_id is not None:
2388
+ self._check_domain_parent_acyclic(domain.domain_id, domain.parent_id)
2389
+
2390
+ def _check_domain_parent_acyclic(self, domain_id: str, parent_id: str) -> None:
2391
+ """``parent_id`` must reference an existing domain, and walking the parent chain
2392
+ from it must never loop back to ``domain_id`` (self-parent included)."""
2393
+ seen: set[str] = set()
2394
+ current: str | None = parent_id
2395
+ while current is not None:
2396
+ if current == domain_id or current in seen:
2397
+ raise ValueError(f"parent_id {parent_id!r} would create a cycle")
2398
+ seen.add(current)
2399
+ parent = self.get_domain(current)
2400
+ if parent is None:
2401
+ raise ValueError(f"parent_id references unknown domain {current!r}")
2402
+ current = parent.parent_id
2403
+
2404
+ def _write_domain(self, domain: Domain, *, write_canonical: bool = True) -> None:
2405
+ """``write_canonical=False`` is the volatile-only path: used exclusively by
2406
+ ``refresh_domain_communities`` to update ONLY the index (the canonical
2407
+ ``domains/<id>.json`` file never contains ``communities`` in the first place, so
2408
+ skipping it isn't just an optimization — it's what keeps a routine sync pass from
2409
+ touching a committed file at all)."""
2410
+ with self._lock:
2411
+ if write_canonical:
2412
+ self._write_domain_canonical(domain)
2413
+ self._touch_digest()
2414
+ self._index_write_domain(domain)
2415
+
2416
+ def get_domain(self, domain_id: str) -> Domain | None:
2417
+ with self._lock:
2418
+ row = self._conn.execute(
2419
+ "SELECT data FROM domains WHERE domain_id = ?", (domain_id,)
2420
+ ).fetchone()
2421
+ return Domain.model_validate_json(row["data"]) if row else None
2422
+
2423
+ def find_domain_by_slug(self, slug: str) -> Domain | None:
2424
+ """Best non-superseded domain with this slug (mirrors the "non-superseded"
2425
+ convention used for decisions, e.g. find_decision_by_title: only SUPERSEDED is
2426
+ excluded, so a dropped domain is still findable by slug).
2427
+
2428
+ Deterministic preference among candidates: accepted > proposed > dropped, and
2429
+ newest first within a status (domain_id is a ULID, so it sorts by creation time).
2430
+ Without an explicit order, sqlite returns rows in rowid/insertion order, which
2431
+ previously meant a drop-then-recreate at the same slug could resolve back to the
2432
+ older, dropped row instead of the live recreation.
2433
+ """
2434
+ with self._lock:
2435
+ rows = self._conn.execute(
2436
+ "SELECT data FROM domains WHERE slug = ? AND status != ? "
2437
+ "ORDER BY CASE status WHEN ? THEN 0 WHEN ? THEN 1 WHEN ? THEN 2 ELSE 3 END, "
2438
+ "domain_id DESC",
2439
+ (
2440
+ slug,
2441
+ DomainStatus.SUPERSEDED.value,
2442
+ DomainStatus.ACCEPTED.value,
2443
+ DomainStatus.PROPOSED.value,
2444
+ DomainStatus.DROPPED.value,
2445
+ ),
2446
+ ).fetchall()
2447
+ for row in rows:
2448
+ return Domain.model_validate_json(row["data"])
2449
+ return None
2450
+
2451
+ def find_domains_by_community(self, community_id: str) -> list[Domain]:
2452
+ """ALL ACCEPTED domains whose ``communities`` contains ``community_id``, newest
2453
+ first — the retrieval read path's bucket-C union (``retrieval.rank_decisions``,
2454
+ Gate-5 finding 2). Two accepted domains can legitimately cover the same community
2455
+ at once (not prevented at write time — an "orphan window" between one domain's
2456
+ acceptance and a later re-scope), and retrieval must surface every one of them: a
2457
+ decision tier-1-bound to an OLDER covering domain must still surface even after a
2458
+ NEWER domain also claims the community. ``domain_id`` is a ULID, so ``ORDER BY
2459
+ domain_id DESC`` is newest-first (matters only to callers that want just the
2460
+ newest — see ``find_domain_by_community`` below).
2461
+ """
2462
+ with self._lock:
2463
+ rows = self._conn.execute(
2464
+ "SELECT data FROM domains WHERE status = ? ORDER BY domain_id DESC",
2465
+ (DomainStatus.ACCEPTED.value,),
2466
+ ).fetchall()
2467
+ out: list[Domain] = []
2468
+ for row in rows:
2469
+ d = Domain.model_validate_json(row["data"])
2470
+ if community_id in d.communities:
2471
+ out.append(d)
2472
+ return out
2473
+
2474
+ def find_domain_by_community(self, community_id: str) -> Domain | None:
2475
+ """The single newest ACCEPTED domain covering ``community_id`` — what
2476
+ ``anchoring.resolve_and_bind`` uses to pick ONE Tier-1 binding target at capture
2477
+ time. This is deliberately narrower than ``find_domains_by_community`` above: a
2478
+ binding write is a one-time pick, and "newest accepted domain" is a fine,
2479
+ deterministic tie-break for it — the asymmetry with retrieval (which needs the
2480
+ FULL set, not just this newest one) is why the two methods exist side by side
2481
+ rather than retrieval calling this one and taking ``[0]`` itself.
2482
+ """
2483
+ domains = self.find_domains_by_community(community_id)
2484
+ return domains[0] if domains else None
2485
+
2486
+ def refresh_domain_communities(self, domain_id: str, communities: list[str]) -> Domain:
2487
+ """Sanctioned mutable-field update for ``Domain.communities`` — the domain analog
2488
+ of ``Entity.last_seen_*`` (see CLAUDE.md): a durable->engine mapping refreshed by
2489
+ ``sync``, NOT content. Updates ONLY the ``communities`` field in place; title,
2490
+ summary, parent_id, and status are untouched and stay governed exclusively by
2491
+ ``supersede_domain``/``ratify_domains``. No-ops (no write) when the value is
2492
+ unchanged, so repeated sync passes over a stable graph never touch the row —
2493
+ writes ONLY the index (the canonical ``domains/<id>.json`` file never contains
2494
+ ``communities`` — see ``_write_domain``), so this never dirties git.
2495
+ """
2496
+ with self._mutation():
2497
+ d = self.get_domain(domain_id)
2498
+ if d is None:
2499
+ raise ValueError(f"domain {domain_id!r} not found")
2500
+ if d.communities == communities:
2501
+ return d
2502
+ updated = d.model_copy(update={"communities": communities})
2503
+ self._write_domain(updated, write_canonical=False)
2504
+ return updated
2505
+
2506
+ def iter_domains(self, status: DomainStatus | None = None) -> Iterator[Domain]:
2507
+ with self._lock:
2508
+ if status is None:
2509
+ rows = self._conn.execute("SELECT data FROM domains").fetchall()
2510
+ else:
2511
+ rows = self._conn.execute(
2512
+ "SELECT data FROM domains WHERE status = ?", (status.value,)
2513
+ ).fetchall()
2514
+ for row in rows:
2515
+ yield Domain.model_validate_json(row["data"])
2516
+
2517
+ def ratify_domains(
2518
+ self,
2519
+ accept: list[str] | None = None,
2520
+ drop: list[str] | None = None,
2521
+ *,
2522
+ actor: str | None = None,
2523
+ ) -> dict[str, str]:
2524
+ """Bulk ratification of proposed domains — mirrors the decisions ratify result
2525
+ shape (server.py's ``_ratify_decisions_impl``): a dict of domain_id -> outcome
2526
+ string ("accepted" | "dropped" | "error: ..."), so a batch partially succeeds
2527
+ instead of one bad id aborting the rest.
2528
+
2529
+ ``actor`` (keyword-only, design D2/T17) stamps every domain accepted in this
2530
+ call with that identity instead of the git one — the auto-ratify caller's
2531
+ ``"auto:<policy>"`` stamp. Forwarded to ``_ratifier_identity`` unconditionally
2532
+ (even at its default ``None``, which that function treats exactly like never
2533
+ having called it with an argument at all): every existing caller (MCP,
2534
+ ``sidegraph-ratify``) passes nothing, so behavior is byte-identical to before —
2535
+ the git identity lookup runs unchanged.
2536
+
2537
+ Accepting flips proposed -> accepted (PROPOSED only) AND mints (get-or-create,
2538
+ idempotent) the paired abstract entity ``domain:<slug>`` so AnchorBinding machinery
2539
+ works unchanged.
2540
+
2541
+ Dropping flips proposed OR accepted -> dropped; no entity is minted (an already-
2542
+ minted ``domain:<slug>`` abstract entity from a prior accept is left as-is — see
2543
+ CLAUDE.md invariant #2, entities are never deleted either). Extended beyond
2544
+ proposed-only in review round 3 (design §6, resolving a cross-branch slug
2545
+ conflict — see ``domain_slug_conflicts``): two branches can each independently
2546
+ ACCEPT a domain with the same slug, and ``supersede_domain`` cannot resolve that
2547
+ shape at all (a successor keeping the same slug would just collide with the OTHER
2548
+ still-live duplicate — see ``_validate_domain_write``). Domains are the owned
2549
+ abstraction layer, not a memory record — unlike ``Decision``, whose drop is
2550
+ deliberately proposal-only (see :meth:`drop`, UNCHANGED by this), retiring an
2551
+ accepted ``Domain`` is a legitimate, append-only-safe operation: the file stays,
2552
+ only its status flips, and it remains fully retrievable (and, once terminal,
2553
+ compactable — see design §7) via ``DomainStatus.DROPPED``, which already existed
2554
+ for exactly this "no longer wanted" case.
2555
+
2556
+ Each accept/drop item is wrapped in its OWN ``_mutation()`` (design D2a) — not one
2557
+ shared transaction for the whole batch (that's the point: one bad id must not abort
2558
+ the rest), and deliberately not the whole METHOD in a single ``_mutation()`` either:
2559
+ that would roll back an EARLIER item that already committed the moment a LATER one
2560
+ failed, which is exactly the behavior this method's own contract rules out. Within a
2561
+ single accept, the entity mint's nested call (``get_or_create_abstract_entity`` ->
2562
+ ``upsert_entity`` -> its own ``_mutation()``) runs at depth 2 and so does not commit
2563
+ or roll back on its own (design D2/D2a) — only the item's own outermost
2564
+ ``_mutation()`` (depth 1) does, committing the domain's status flip and the entity
2565
+ mint together, once, at the end of that item. What *is* shared across the whole call
2566
+ is lock scope, not transaction scope: the entire loop still runs under
2567
+ ``self._lock`` (an outer acquisition around both loops, not replaced by each item's
2568
+ own nested one — ``RLock`` makes nesting the two free), so no other thread's write
2569
+ interleaves mid-batch.
2570
+ """
2571
+ out: dict[str, str] = {}
2572
+ with self._lock:
2573
+ for domain_id in accept or []:
2574
+ try:
2575
+ with self._mutation():
2576
+ d = self.get_domain(domain_id)
2577
+ if d is None or d.status != DomainStatus.PROPOSED:
2578
+ raise ValueError(f"domain {domain_id!r} is not proposed")
2579
+ d.status = DomainStatus.ACCEPTED
2580
+ d.ratified_at = datetime.now(UTC)
2581
+ d.ratified_by = _ratifier_identity(actor)
2582
+ self._write_domain(d)
2583
+ self.get_or_create_abstract_entity(f"domain:{d.slug}")
2584
+ out[domain_id] = "accepted"
2585
+ except ValueError as e:
2586
+ out[domain_id] = f"error: {e}"
2587
+ for domain_id in drop or []:
2588
+ if domain_id in out:
2589
+ out[domain_id] = f"{out[domain_id]} (drop ignored)"
2590
+ continue
2591
+ try:
2592
+ with self._mutation():
2593
+ d = self.get_domain(domain_id)
2594
+ if d is None or d.status not in (
2595
+ DomainStatus.PROPOSED,
2596
+ DomainStatus.ACCEPTED,
2597
+ ):
2598
+ raise ValueError(f"domain {domain_id!r} is not proposed or accepted")
2599
+ d.status = DomainStatus.DROPPED
2600
+ self._write_domain(d)
2601
+ out[domain_id] = "dropped"
2602
+ except ValueError as e:
2603
+ out[domain_id] = f"error: {e}"
2604
+ return out
2605
+
2606
+ # -- ratification (Stage 5) ----------------------------------------------
2607
+
2608
+ def iter_proposed(self) -> Iterator[Decision]:
2609
+ """Decisions awaiting ratification (status == proposed)."""
2610
+ with self._lock:
2611
+ rows = self._conn.execute(
2612
+ "SELECT data FROM decisions WHERE status = 'proposed'"
2613
+ ).fetchall()
2614
+ for row in rows:
2615
+ yield Decision.model_validate_json(row["data"])
2616
+
2617
+ def pending_ratification_counts(self) -> tuple[int, int, int]:
2618
+ """``(decisions, standalone_facts, domains)`` awaiting ratification — the queue
2619
+ SessionStart's pending-ratification line summarizes. See design/superpowers/specs/
2620
+ 2026-07-10-ratification-ux-and-mcp-gaps-design.md.
2621
+
2622
+ Mirrors ``server._list_proposed_impl``'s decomposition (same "items to review"
2623
+ mental model): proposed decisions; proposed facts NOT supporting any proposed
2624
+ decision (facts riding a proposed decision's cascade are covered by it and must
2625
+ not be double-counted — see ``Store.ratify``); proposed domains."""
2626
+ proposals = list(self.iter_proposed())
2627
+ nested = {
2628
+ f.id
2629
+ for d in proposals
2630
+ for f in self.facts_for_decision(d.id)
2631
+ if f.status == DecisionStatus.PROPOSED
2632
+ }
2633
+ standalone = [f for f in self.iter_proposed_facts() if f.id not in nested]
2634
+ domains = list(self.iter_domains(status=DomainStatus.PROPOSED))
2635
+ return (len(proposals), len(standalone), len(domains))
2636
+
2637
+ def ratify(
2638
+ self,
2639
+ decision_id: str,
2640
+ *,
2641
+ actor: str | None = None,
2642
+ cascade_guard: Callable[[Sequence[Fact]], bool] | None = None,
2643
+ ) -> tuple[Decision, list[Fact]]:
2644
+ """Flip a proposed decision to accepted (the human's one-tap gesture).
2645
+
2646
+ Deferred supersession (design §3 option (b), shared by every ratify path — MCP
2647
+ ``ratify``/``ratify_decisions`` and ``sidegraph-ratify`` alike, since both route
2648
+ through this method): when the now-accepted decision ``supersedes`` a predecessor
2649
+ that is STILL OPEN (status ``accepted`` or ``proposed`` — i.e. never closed at
2650
+ write time, see ``add_decision(close_predecessor=False)``), that predecessor is
2651
+ closed (``valid_to`` + status ``superseded``) in this SAME operation — this is what
2652
+ actually performs the supersession a doc-import ``--propose`` deferred. A
2653
+ predecessor that's already ``superseded`` (the ordinary path, where
2654
+ ``add_decision`` closed it immediately at write time) is left alone — a no-op here,
2655
+ not a double-close. Dropping the proposal instead (:meth:`drop`) never touches the
2656
+ predecessor at all.
2657
+
2658
+ ``actor`` (keyword-only, design D2/T17): an explicit stamp for this ratification,
2659
+ used by auto-ratification (the ``"auto:<policy>"`` stamp, never called by a human
2660
+ path) instead of the git identity. Forwarded to ``_ratifier_identity``
2661
+ unconditionally, including at its default ``None`` — every existing caller
2662
+ passes nothing, so behavior is byte-identical to before it existed: the git
2663
+ identity lookup runs unchanged. A blank/whitespace ``actor`` is handled by
2664
+ ``_ratifier_identity`` itself (falls back to the git identity, never stores the
2665
+ blank string).
2666
+
2667
+ ``cascade_guard`` (keyword-only, design D2 checkpoint-2 fix, Ruling Q): closes the
2668
+ race window between an outer eligibility check and this method's own write (external
2669
+ review finding A1, reproduced by ``scratchpad/probe_race.py`` — a fact landing
2670
+ between the two could self-certify with no anchor). When supplied, the whole
2671
+ transition runs under ``_mutation(immediate=True)`` so the write lock is taken
2672
+ BEFORE the cascade set is even read — see ``_mutation``'s own docstring for why
2673
+ ``immediate`` is required here: an outer mutation that has only READ so far holds no
2674
+ lock, so without it a second connection could still slip a write in between. The
2675
+ cascade set (still-``proposed`` facts whose ``supports`` names this decision, read
2676
+ fresh under that lock) is handed to the guard; a ``False`` return raises
2677
+ ``ValueError`` BEFORE any write happens — no status flip, no stamp, no predecessor
2678
+ close — leaving the decision proposed. Omitted (every existing human path — MCP,
2679
+ CLI, doc-import): behavior is byte-identical to before this parameter existed, plain
2680
+ ``_mutation()``, no guard, no re-check. Only ``capture._auto_ratify``'s decision route
2681
+ ever supplies one.
2682
+
2683
+ Cascade (facts layer, design §2026-07-10): every still-``proposed`` :class:`Fact`
2684
+ whose ``supports`` names this decision rides its verdict — flipped to ``accepted``
2685
+ here, in the same locked body, and returned alongside the decision. The cascaded
2686
+ facts inherit ``ratified_by`` directly from ``d`` below, so an explicit ``actor``
2687
+ reaches them the same way the git identity always did — no separate threading.
2688
+ """
2689
+ with self._mutation(immediate=cascade_guard is not None):
2690
+ d = self.get_decision(decision_id)
2691
+ if d is None or d.status != DecisionStatus.PROPOSED:
2692
+ raise ValueError(f"decision {decision_id!r} is not proposed")
2693
+
2694
+ if cascade_guard is not None:
2695
+ cascade_candidates = [
2696
+ fact for fact in self.iter_proposed_facts() if decision_id in fact.supports
2697
+ ]
2698
+ if not cascade_guard(cascade_candidates):
2699
+ raise ValueError(
2700
+ f"cascade guard refused auto-ratification of decision {decision_id!r}"
2701
+ )
2702
+
2703
+ d.status = DecisionStatus.ACCEPTED
2704
+ d.ratified_at = datetime.now(UTC)
2705
+ d.ratified_by = _ratifier_identity(actor)
2706
+ self._write_decision(d)
2707
+ if d.supersedes:
2708
+ predecessor = self.get_decision(d.supersedes)
2709
+ if predecessor is not None and predecessor.status in (
2710
+ DecisionStatus.ACCEPTED,
2711
+ DecisionStatus.PROPOSED,
2712
+ ):
2713
+ predecessor.status = DecisionStatus.SUPERSEDED
2714
+ predecessor.valid_to = predecessor.valid_to or max(
2715
+ datetime.now(UTC), predecessor.valid_from
2716
+ )
2717
+ self._write_decision(predecessor)
2718
+
2719
+ cascaded: list[Fact] = []
2720
+ for fact in self.iter_proposed_facts():
2721
+ if decision_id in fact.supports:
2722
+ fact.status = DecisionStatus.ACCEPTED
2723
+ fact.ratified_at = d.ratified_at
2724
+ fact.ratified_by = d.ratified_by
2725
+ self._write_fact(fact)
2726
+ cascaded.append(fact)
2727
+ return d, cascaded
2728
+
2729
+ def drop(self, decision_id: str) -> tuple[Decision, list[Fact]]:
2730
+ """Reject a proposed decision. Append-only: sets valid_to + rejected, never deletes.
2731
+
2732
+ Cascade (facts layer, design §2026-07-10): every still-``proposed`` :class:`Fact`
2733
+ for which ALL ``supports`` targets are now ``rejected`` is dropped too (``valid_to``
2734
+ + status ``rejected``) — a fact with any surviving supporter is kept; a fact with no
2735
+ supporters at all (``supports == []``) never cascades. Runs in the same locked body,
2736
+ after the decision flip, and the dropped facts are returned alongside it.
2737
+ """
2738
+ with self._mutation():
2739
+ d = self.get_decision(decision_id)
2740
+ if d is None or d.status != DecisionStatus.PROPOSED:
2741
+ raise ValueError(f"decision {decision_id!r} is not proposed")
2742
+ d.status = DecisionStatus.REJECTED
2743
+ d.valid_to = d.valid_to or max(datetime.now(UTC), d.valid_from)
2744
+ self._write_decision(d)
2745
+
2746
+ cascaded: list[Fact] = []
2747
+ for fact in self.iter_proposed_facts():
2748
+ if self._all_supporters_rejected(fact):
2749
+ fact.status = DecisionStatus.REJECTED
2750
+ fact.valid_to = fact.valid_to or max(datetime.now(UTC), fact.valid_from)
2751
+ self._write_fact(fact)
2752
+ cascaded.append(fact)
2753
+ return d, cascaded
2754
+
2755
+ def _all_supporters_rejected(self, fact: Fact) -> bool:
2756
+ """Drop-cascade predicate: true iff ``fact.supports`` is non-empty and every
2757
+ decision it names is now ``rejected`` — an empty ``supports`` never cascades."""
2758
+ if not fact.supports:
2759
+ return False
2760
+ return all(
2761
+ (dec := self.get_decision(sid)) is not None and dec.status == DecisionStatus.REJECTED
2762
+ for sid in fact.supports
2763
+ )
2764
+
2765
+ def ratify_fact(self, fact_id: str, *, actor: str | None = None) -> Fact:
2766
+ """Flip a proposed fact to accepted directly (no supporting decision involved).
2767
+
2768
+ ``actor`` (keyword-only, design D2/T17): same contract as :meth:`ratify` — an
2769
+ explicit stamp for auto-ratification's standalone-fact path, defaulting to
2770
+ ``None`` so every existing caller (which passes nothing) is unaffected and still
2771
+ gets the plain git-identity lookup.
2772
+ """
2773
+ with self._mutation():
2774
+ f = self.get_fact(fact_id)
2775
+ if f is None or f.status != DecisionStatus.PROPOSED:
2776
+ raise ValueError(f"fact {fact_id!r} is not proposed")
2777
+ f.status = DecisionStatus.ACCEPTED
2778
+ f.ratified_at = datetime.now(UTC)
2779
+ f.ratified_by = _ratifier_identity(actor)
2780
+ self._write_fact(f)
2781
+ return f
2782
+
2783
+ def drop_fact(self, fact_id: str) -> Fact:
2784
+ """Reject a proposed fact directly. Append-only: sets valid_to + rejected."""
2785
+ with self._mutation():
2786
+ f = self.get_fact(fact_id)
2787
+ if f is None or f.status != DecisionStatus.PROPOSED:
2788
+ raise ValueError(f"fact {fact_id!r} is not proposed")
2789
+ f.status = DecisionStatus.REJECTED
2790
+ f.valid_to = f.valid_to or max(datetime.now(UTC), f.valid_from)
2791
+ self._write_fact(f)
2792
+ return f
2793
+
2794
+ # -- capture ledger (Stage 5) --------------------------------------------
2795
+ #
2796
+ # Local-only bookkeeping (design §1: "capture ledger" is explicitly volatile) — no
2797
+ # canonical file, index.db only.
2798
+
2799
+ def was_captured(self, session_id: str) -> bool:
2800
+ with self._lock:
2801
+ row = self._conn.execute(
2802
+ "SELECT 1 FROM capture_sessions WHERE session_id = ?", (session_id,)
2803
+ ).fetchone()
2804
+ return row is not None
2805
+
2806
+ def mark_captured(self, session_id: str) -> None:
2807
+ with self._mutation():
2808
+ self._conn.execute(
2809
+ "INSERT OR IGNORE INTO capture_sessions (session_id, captured_at) VALUES (?, ?)",
2810
+ (session_id, datetime.now(UTC).isoformat()),
2811
+ )
2812
+
2813
+ # -- retrieval telemetry (design/superpowers/specs/
2814
+ # 2026-07-25-retrieval-telemetry-design.md) -----------------------------
2815
+ #
2816
+ # Local-only bookkeeping, index.db only — no canonical file. Same precedent as the
2817
+ # capture ledger above: derived state that must survive `_reload_index_from_canonical`
2818
+ # (its DROP list names the six record tables explicitly), because losing the history on
2819
+ # every `git pull` would make the counts meaningless for the long-lived question they
2820
+ # answer.
2821
+
2822
+ def record_retrieval(self, record_ids: Iterable[str], seeds: Iterable[str]) -> None:
2823
+ """Count records that reached a render, and areas that were asked about.
2824
+
2825
+ Derived state, index-only: this is a *read* path, and writing anything canonical
2826
+ here would make retrieval dirty git — the sync-clean invariant's most direct
2827
+ possible violation. Deduped per call: one render showing a record twice is one
2828
+ showing, and a seed counts once per query however many records came back, or the
2829
+ ratio the doctor check rests on stops meaning anything.
2830
+ """
2831
+ now = datetime.now(UTC).isoformat()
2832
+ with self._mutation():
2833
+ for rid in dict.fromkeys(record_ids):
2834
+ self._conn.execute(
2835
+ "INSERT INTO retrieval_shows (record_id, shows, last_shown_at) "
2836
+ "VALUES (?, 1, ?) ON CONFLICT(record_id) DO UPDATE SET "
2837
+ "shows = shows + 1, last_shown_at = excluded.last_shown_at",
2838
+ (rid, now),
2839
+ )
2840
+ for seed in dict.fromkeys(seeds):
2841
+ self._conn.execute(
2842
+ "INSERT INTO retrieval_seeds (seed, queries, last_seen_at) "
2843
+ "VALUES (?, 1, ?) ON CONFLICT(seed) DO UPDATE SET "
2844
+ "queries = queries + 1, last_seen_at = excluded.last_seen_at",
2845
+ (seed, now),
2846
+ )
2847
+
2848
+ def retrieval_shows(self) -> dict[str, int]:
2849
+ with self._lock:
2850
+ return {
2851
+ row["record_id"]: row["shows"]
2852
+ for row in self._conn.execute("SELECT record_id, shows FROM retrieval_shows")
2853
+ }
2854
+
2855
+ def retrieval_seed_queries(self) -> dict[str, int]:
2856
+ with self._lock:
2857
+ return {
2858
+ row["seed"]: row["queries"]
2859
+ for row in self._conn.execute("SELECT seed, queries FROM retrieval_seeds")
2860
+ }
2861
+
2862
+ # -- coverage telemetry: the ordered journal (design/superpowers/specs/
2863
+ # 2026-07-26-retrieval-coverage-telemetry-design.md, D1) ------------------
2864
+ #
2865
+ # Same index-only contract as the two counter tables above, for the same reason: this
2866
+ # is a read path, and writing anything canonical here would make retrieval dirty git.
2867
+ # What the counters cannot express is *when* and *in which session*, which is the whole
2868
+ # question this journal exists to answer.
2869
+
2870
+ def _append_event(self, session_id: str, kind: str, key: str, detail: str | None) -> None:
2871
+ with self._mutation():
2872
+ self._conn.execute(
2873
+ "INSERT INTO retrieval_events (session_id, at, kind, key, detail) "
2874
+ "VALUES (?, ?, ?, ?, ?)",
2875
+ (session_id, datetime.now(UTC).isoformat(), kind, key, detail),
2876
+ )
2877
+
2878
+ def record_touch(self, session_id: str, path: str, tool: str) -> None:
2879
+ """Record that a file was touched during a session.
2880
+
2881
+ All three values are **opaque** to the core: ``session_id`` is whatever the host
2882
+ calls a session and ``tool`` whatever it calls the tool. The core never interprets
2883
+ either — this keeps the host seam one-directional (see CLAUDE.md; ``capture_sessions``
2884
+ is the precedent). ``path`` must already be repo-relative; normalizing it is the
2885
+ host's job, because only the host knows the root it was given (D3).
2886
+ """
2887
+ self._append_event(session_id, "touch", path, tool)
2888
+
2889
+ def record_retrieval_events(
2890
+ self,
2891
+ session_id: str,
2892
+ seeds: Iterable[str],
2893
+ shows: Iterable[tuple[str, str]],
2894
+ ) -> None:
2895
+ """Record what a retrieval asked about and what it surfaced.
2896
+
2897
+ ``shows`` are ``(record_id, anchored_file_path)`` pairs — one event per pair, since
2898
+ a record anchored to three files can be reached by a touch of any of them. Both
2899
+ sequences are deduped per call: one render showing a record twice is one showing.
2900
+ """
2901
+ now = datetime.now(UTC).isoformat()
2902
+ with self._mutation():
2903
+ for seed in dict.fromkeys(seeds):
2904
+ self._conn.execute(
2905
+ "INSERT INTO retrieval_events (session_id, at, kind, key, detail) "
2906
+ "VALUES (?, ?, 'seed', ?, NULL)",
2907
+ (session_id, now, seed),
2908
+ )
2909
+ for record_id, path in dict.fromkeys(shows):
2910
+ self._conn.execute(
2911
+ "INSERT INTO retrieval_events (session_id, at, kind, key, detail) "
2912
+ "VALUES (?, ?, 'show_anchor', ?, ?)",
2913
+ (session_id, now, path, record_id),
2914
+ )
2915
+
2916
+ def retrieval_events(self, session_id: str | None = None) -> list[dict[str, str | None]]:
2917
+ """Journal rows in insertion order, optionally scoped to one session."""
2918
+ sql = "SELECT session_id, at, kind, key, detail FROM retrieval_events"
2919
+ params: tuple[str, ...] = ()
2920
+ if session_id is not None:
2921
+ sql += " WHERE session_id = ?"
2922
+ params = (session_id,)
2923
+ sql += " ORDER BY id"
2924
+ with self._lock:
2925
+ return [dict(row) for row in self._conn.execute(sql, params)]
2926
+
2927
+ def prune_retrieval_events(self, older_than_days: int = 30) -> int:
2928
+ """Delete events older than the retention window; returns the row count.
2929
+
2930
+ Called once per session from ``SessionStart`` (D7) — an append-only journal with no
2931
+ pruning is a predictable disk-growth bug, and doing it off every hot path keeps the
2932
+ cost invisible.
2933
+ """
2934
+ cutoff = (datetime.now(UTC) - timedelta(days=older_than_days)).isoformat()
2935
+ with self._mutation():
2936
+ return self._conn.execute(
2937
+ "DELETE FROM retrieval_events WHERE at < ?", (cutoff,)
2938
+ ).rowcount
2939
+
2940
+ # -- initiatives --------------------------------------------------------
2941
+
2942
+ def upsert_initiative(self, initiative: Initiative) -> Initiative:
2943
+ with self._mutation():
2944
+ self._write_initiative_canonical(initiative)
2945
+ self._index_write_initiative(initiative)
2946
+ self._touch_digest()
2947
+ return initiative
2948
+
2949
+ # -- compaction (design §7) ------------------------------------------------
2950
+ #
2951
+ # Terminal-status decisions/domains never change again under append-only rules, so they
2952
+ # need neither mergeability nor PR review. ``compact`` packs them into an immutable
2953
+ # ``archive/<date>-<seq>.jsonl`` segment and removes their individual hot files in the
2954
+ # same operation — CLAUDE.md invariant #2 (append-only) still holds: the records MOVE,
2955
+ # they are never lost or mutated (``_reload_index_from_canonical`` above and the loader
2956
+ # helpers below still surface them exactly as before). Entities, bindings, and
2957
+ # initiatives stay hot in v1 — a compacted decision's bindings remain live, ordinary hot
2958
+ # files (retrieval of superseded history still needs them). Explicit, human-run
2959
+ # maintenance only: nothing in sync/retrieval/ratify calls this.
2960
+
2961
+ def _list_archive_segments(self) -> list[Path]:
2962
+ archive_dir = self.path / _ARCHIVE_SUBDIR
2963
+ if not archive_dir.is_dir():
2964
+ return []
2965
+ return sorted(archive_dir.glob("*.jsonl"))
2966
+
2967
+ @staticmethod
2968
+ def _read_archive_segment(path: Path) -> list[dict]:
2969
+ """Parse every line of one archive segment. A malformed line (review Minor-5: a
2970
+ merge-mangled segment, truncated write, or hand edit) raises a ``ValueError``
2971
+ naming the offending segment PATH and LINE NUMBER — a bare ``JSONDecodeError``
2972
+ gives no clue which of potentially many segments is at fault (mirrors
2973
+ ``_validate_legacy_rows``'s "name the offending row" convention)."""
2974
+ out: list[dict] = []
2975
+ for lineno, raw_line in enumerate(path.read_text(encoding="utf-8").splitlines(), start=1):
2976
+ line = raw_line.strip()
2977
+ if not line:
2978
+ continue
2979
+ try:
2980
+ out.append(json.loads(line))
2981
+ except json.JSONDecodeError as e:
2982
+ raise ValueError(f"corrupt archive segment {path} at line {lineno}: {e}") from e
2983
+ return out
2984
+
2985
+ def _archived_records(self) -> tuple[dict[str, dict], dict[str, dict]]:
2986
+ """Every archived decision/domain payload (``record_type`` stripped), keyed by id,
2987
+ across ALL segments — oldest segment first, first occurrence wins. Two segments
2988
+ legitimately containing the same id (e.g. compaction run independently on two
2989
+ branches, later merged) are guaranteed byte-identical by design (terminal records
2990
+ never change once archived), so which one wins is never a correctness question —
2991
+ see design §7."""
2992
+ decisions: dict[str, dict] = {}
2993
+ domains: dict[str, dict] = {}
2994
+ for segment in self._list_archive_segments():
2995
+ for raw in self._read_archive_segment(segment):
2996
+ record_type = raw.get("record_type")
2997
+ payload = {k: v for k, v in raw.items() if k != "record_type"}
2998
+ if record_type == "decision" and "id" in payload:
2999
+ decisions.setdefault(payload["id"], payload)
3000
+ elif record_type == "domain" and "domain_id" in payload:
3001
+ domains.setdefault(payload["domain_id"], payload)
3002
+ # an unrecognized record_type is silently skipped -- forward-compat with a
3003
+ # future record kind this version of the code doesn't know how to load yet,
3004
+ # rather than a hard failure that blocks opening the whole store.
3005
+ return decisions, domains
3006
+
3007
+ def _archived_record_ids(self) -> tuple[set[str], set[str]]:
3008
+ decisions, domains = self._archived_records()
3009
+ return set(decisions), set(domains)
3010
+
3011
+ def _leftover_archived_hot(
3012
+ self, archived_decisions: dict[str, dict], archived_domains: dict[str, dict]
3013
+ ) -> tuple[set[str], set[str], int]:
3014
+ """Ids whose hot canonical file duplicates an ALREADY-archived record exactly —
3015
+ crash-window debris from a compact that durably wrote its segment but was
3016
+ interrupted before removing the hot file (see :meth:`compact`). These are safe to
3017
+ remove without writing a new segment: the archive already has them.
3018
+
3019
+ A hot file that instead DIFFERS from its archived counterpart is corruption-shaped,
3020
+ not crash debris — it is left exactly as it is (never destroy data) and surfaced via
3021
+ a stderr warning. Returns ``(decision_ids_safe_to_remove, domain_ids_safe_to_remove,
3022
+ mismatch_count)``.
3023
+ """
3024
+ safe_decisions: set[str] = set()
3025
+ safe_domains: set[str] = set()
3026
+ mismatches = 0
3027
+ for did, archived_payload in archived_decisions.items():
3028
+ match = _hot_file_matches(self.path / "decisions" / f"{did}.json", archived_payload)
3029
+ if match is True:
3030
+ safe_decisions.add(did)
3031
+ elif match is False:
3032
+ mismatches += 1
3033
+ _warn_hot_archive_mismatch("decision", did)
3034
+ for dmid, archived_payload in archived_domains.items():
3035
+ match = _hot_file_matches(self.path / "domains" / f"{dmid}.json", archived_payload)
3036
+ if match is True:
3037
+ safe_domains.add(dmid)
3038
+ elif match is False:
3039
+ mismatches += 1
3040
+ _warn_hot_archive_mismatch("domain", dmid)
3041
+ return safe_decisions, safe_domains, mismatches
3042
+
3043
+ def _next_archive_seq(self, archive_dir: Path, date_str: str) -> int:
3044
+ """The next unused ``<seq>`` for ``date_str``, given segments already on disk —
3045
+ re-globbed fresh on every call (see ``_write_archive_segment``'s retry loop: a
3046
+ concurrent writer may have just published between one call and the next, and
3047
+ "next available" must reflect that, not a stale snapshot)."""
3048
+ existing = [
3049
+ seq
3050
+ for f in archive_dir.glob(f"{date_str}-*.jsonl")
3051
+ if (seq := _parse_segment_seq(f.stem, date_str)) is not None
3052
+ ]
3053
+ return max(existing, default=0) + 1
3054
+
3055
+ def _write_archive_segment(self, decisions: list[Decision], domains: list[Domain]) -> str:
3056
+ """Write ONE new segment containing every one of ``decisions``/``domains``,
3057
+ ULID-sorted together, one record per line (see ``_archive_record_line``). Segments
3058
+ are write-once (design §7) and this always creates a brand-new file — but unlike
3059
+ every other committed file, publishing one is NOT a plain tmp+``os.replace``:
3060
+
3061
+ Review Important-1 (fault-injection finding): the original implementation used a
3062
+ FIXED tmp filename and unconditional ``os.replace``. Two compacts racing (two
3063
+ processes, or two threads via fastmcp's worker dispatch) could both open that same
3064
+ fixed tmp name for writing at once — one writer's truncating ``open(mode="w")``
3065
+ can zero out bytes the other had already written, and ``os.replace`` would then
3066
+ atomically install that TRUNCATED (even 0-byte) content as "the segment" while the
3067
+ caller had already unlinked the hot files it was supposed to be a durable copy of
3068
+ — silent, permanent data loss on the next cold reload. ``os.replace`` unconditionally
3069
+ overwriting an existing target is also a second problem on its own: it can clobber
3070
+ an already-published segment, violating "write once, never rewritten".
3071
+
3072
+ Fixed by: (1) writing to a tmp file with a per-attempt UNIQUE name (pid+uuid, same
3073
+ convention as ``_atomic_write_text_race_tolerant`` — no two writers, in this
3074
+ process or any other, ever share a tmp path, so no truncation race is possible);
3075
+ (2) fsync-ing that tmp file's content before it is ever reachable under a real
3076
+ name (review Minor-3 — compaction is the one operation where a power-loss gap
3077
+ between "hot files gone" and "segment durable" would delete the only surviving
3078
+ copy of a record); (3) publishing via ``os.link`` (an exclusive create — raises
3079
+ ``FileExistsError`` if the target name is already taken, never silently clobbers
3080
+ it) instead of ``os.replace``; (4) on that ``FileExistsError``, recomputing the
3081
+ next available ``<seq>`` and retrying under a NEW name, up to
3082
+ ``_ARCHIVE_SEGMENT_PUBLISH_ATTEMPTS`` times; (5) fsync-ing the archive DIRECTORY
3083
+ after a successful link, so the new directory entry itself survives a crash, not
3084
+ just the file's bytes. The tmp file is always unlinked afterward (its content
3085
+ lives on under the published name via the hard link) — success, failure, or
3086
+ exhausted retries alike.
3087
+
3088
+ The published name is ``<date>-<seq>-<hash12>.jsonl`` (review Minor-4 amendment —
3089
+ see ``_ARCHIVE_SUBDIR``'s docstring for why the content hash is part of the name).
3090
+ Any OTHER exception from the publish attempt (anything but ``FileExistsError`` —
3091
+ e.g. a genuine disk error) propagates immediately: the caller (``compact``) must
3092
+ never proceed to remove a single hot file once this raises, since there is then no
3093
+ durable segment guaranteed to contain their content.
3094
+
3095
+ Returns the new segment's path relative to ``self.path``.
3096
+ """
3097
+ records: list[tuple[str, str]] = [
3098
+ (d.id, _archive_record_line("decision", d.model_dump(mode="json"))) for d in decisions
3099
+ ] + [
3100
+ (dom.domain_id, _archive_record_line("domain", _domain_canonical_payload(dom)))
3101
+ for dom in domains
3102
+ ]
3103
+ records.sort(key=lambda item: item[0])
3104
+ text = "\n".join(line for _, line in records) + "\n"
3105
+ content_bytes = text.encode("utf-8")
3106
+ content_hash12 = hashlib.sha256(content_bytes).hexdigest()[:12]
3107
+
3108
+ archive_dir = self.path / _ARCHIVE_SUBDIR
3109
+ archive_dir.mkdir(parents=True, exist_ok=True)
3110
+ date_str = datetime.now(UTC).date().isoformat()
3111
+
3112
+ tmp = archive_dir / f"archive-segment.{os.getpid()}.{uuid.uuid4().hex[:12]}.tmp"
3113
+ target: Path | None = None
3114
+ try:
3115
+ with open(tmp, "wb") as fh:
3116
+ fh.write(content_bytes)
3117
+ fh.flush()
3118
+ os.fsync(fh.fileno())
3119
+
3120
+ for _ in range(_ARCHIVE_SEGMENT_PUBLISH_ATTEMPTS):
3121
+ seq = self._next_archive_seq(archive_dir, date_str)
3122
+ candidate = archive_dir / f"{date_str}-{seq}-{content_hash12}.jsonl"
3123
+ try:
3124
+ os.link(tmp, candidate)
3125
+ target = candidate
3126
+ break
3127
+ except FileExistsError:
3128
+ continue
3129
+ else:
3130
+ raise OSError(
3131
+ f"could not publish archive segment for {date_str} after "
3132
+ f"{_ARCHIVE_SEGMENT_PUBLISH_ATTEMPTS} attempts (persistent seq "
3133
+ "collision with concurrent writers)"
3134
+ )
3135
+ finally:
3136
+ tmp.unlink(missing_ok=True)
3137
+
3138
+ try:
3139
+ dir_fd = os.open(archive_dir, os.O_RDONLY)
3140
+ try:
3141
+ os.fsync(dir_fd)
3142
+ finally:
3143
+ os.close(dir_fd)
3144
+ except OSError:
3145
+ pass # best-effort directory-entry durability -- unsupported on some platforms
3146
+
3147
+ assert target is not None # the loop always sets it before falling through to here
3148
+ # Record the stat row only AFTER the link has actually won (digest-integrity design
3149
+ # §3): the retry loop above can change the final name, so recording any earlier
3150
+ # candidate name would describe a name that was never published. `target` and `tmp`
3151
+ # are hard links to the SAME inode, so `target.stat()` here is identical to a stat
3152
+ # taken on `tmp` right after its fsync -- either way it is THIS process's own
3153
+ # published bytes, never a later replace (archive segments are write-once and never
3154
+ # replaced again, so there is no "later writer" window to lose a race in at all).
3155
+ self._record_canonical_stat(_ARCHIVE_SUBDIR, target.stem, target.stat())
3156
+ return f"{_ARCHIVE_SUBDIR}/{target.name}"
3157
+
3158
+ def compact(
3159
+ self, *, older_than_days: int | None = None, dry_run: bool = False
3160
+ ) -> CompactReport:
3161
+ """Pack every TERMINAL-status decision/domain into a new archive segment and remove
3162
+ its hot canonical file, in one operation (design §7; see the class-level comment
3163
+ above this section for the append-only guarantee this preserves).
3164
+
3165
+ Facts are NOT compacted in this wave (deferred) — ``facts/`` stays entirely hot
3166
+ regardless of status; a compacted fact is future scope, not this task's.
3167
+
3168
+ Terminal = decisions ``superseded``/``rejected``/``deprecated``, domains
3169
+ ``superseded``/``dropped`` (see ``_TERMINAL_DECISION_STATUSES`` /
3170
+ ``_TERMINAL_DOMAIN_STATUSES``) — statuses append-only rules guarantee can never
3171
+ change again. ``proposed``/``accepted`` records of either kind are never touched.
3172
+
3173
+ ``older_than_days``, when given, additionally requires the record's best available
3174
+ TERMINAL timestamp to be at least that many days in the past:
3175
+
3176
+ - a decision's ``valid_to`` — always stamped the moment ``add_decision``/``ratify``/
3177
+ ``drop`` actually closes it, so this is a reliable signal for every real record.
3178
+ Excluded decisions (too young, or defensively when ``valid_to`` is unexpectedly
3179
+ ``None`` despite a terminal status) are counted in
3180
+ ``CompactReport.skipped_age_filtered``.
3181
+ - a domain has NO field anywhere that records when it entered a terminal status (no
3182
+ ``valid_to``, nothing in ``Provenance`` either) — rather than guess from e.g. its
3183
+ ULID's creation timestamp (which is when it was PROPOSED, not when it became
3184
+ superseded/dropped — those can be arbitrarily far apart), every domain is
3185
+ conservatively EXCLUDED (kept hot) whenever ``older_than_days`` is set. Counted
3186
+ separately, in ``CompactReport.domains_excluded_age_unknown`` (review Minor-7:
3187
+ "too recent" and "no timestamp exists to check at all" are different situations
3188
+ and read confusingly merged into one number).
3189
+
3190
+ Excluded-by-age records of either kind are left completely untouched — not
3191
+ archived, not removed.
3192
+
3193
+ ``dry_run=True`` computes and returns the full report (including which records WOULD
3194
+ be archived/cleaned up) without writing or removing anything — not the new segment,
3195
+ not a hot file, not even the digest.
3196
+
3197
+ Idempotent / crash-safe: a prior run that durably wrote its segment but crashed
3198
+ before removing the now-redundant hot files leaves those files as harmless
3199
+ byte-identical duplicates of their archived copy. This run detects them (see
3200
+ ``_leftover_archived_hot``) and removes them WITHOUT writing a second, duplicate
3201
+ segment for records that are already durably archived — ``CompactReport.
3202
+ cleaned_up_hot_files`` counts these separately from freshly-archived records.
3203
+
3204
+ Every hot-file removal (fresh candidates AND crash-window leftovers alike) is
3205
+ gated on the file's ON-DISK content still matching exactly what got archived
3206
+ (review Minor-6): a candidate is read from the INDEX, which should always mirror
3207
+ its hot file, but a removal is destructive enough that this never just trusts that
3208
+ invariant — a mismatch (something changed the file out from under this pass, e.g.
3209
+ an external hand edit) leaves the file in place and surfaces a warning instead of
3210
+ silently discarding newer state nothing else preserved a copy of.
3211
+ """
3212
+ with self._mutation():
3213
+ archived_decisions, archived_domains = self._archived_records()
3214
+
3215
+ cutoff = (
3216
+ datetime.now(UTC) - timedelta(days=older_than_days)
3217
+ if older_than_days is not None
3218
+ else None
3219
+ )
3220
+
3221
+ new_decisions: list[Decision] = []
3222
+ new_domains: list[Domain] = []
3223
+ skipped_age_filtered = 0
3224
+ domains_excluded_age_unknown = 0
3225
+
3226
+ for d in self.iter_decisions():
3227
+ if d.status not in _TERMINAL_DECISION_STATUSES or d.id in archived_decisions:
3228
+ continue
3229
+ if cutoff is not None and (d.valid_to is None or d.valid_to > cutoff):
3230
+ skipped_age_filtered += 1
3231
+ continue
3232
+ new_decisions.append(d)
3233
+
3234
+ for dom in self.iter_domains():
3235
+ if dom.status not in _TERMINAL_DOMAIN_STATUSES or dom.domain_id in archived_domains:
3236
+ continue
3237
+ if cutoff is not None:
3238
+ # No terminal-timestamp field exists on Domain at all -- always
3239
+ # conservative when age-filtering is active.
3240
+ domains_excluded_age_unknown += 1
3241
+ continue
3242
+ new_domains.append(dom)
3243
+
3244
+ new_decisions.sort(key=lambda d: d.id)
3245
+ new_domains.sort(key=lambda dom: dom.domain_id)
3246
+
3247
+ items = sorted(
3248
+ [
3249
+ CompactedRecord(
3250
+ ulid=d.id, kind="decision", status=d.status.value, title=d.title[:60]
3251
+ )
3252
+ for d in new_decisions
3253
+ ]
3254
+ + [
3255
+ CompactedRecord(
3256
+ ulid=dom.domain_id,
3257
+ kind="domain",
3258
+ status=dom.status.value,
3259
+ title=dom.title[:60],
3260
+ )
3261
+ for dom in new_domains
3262
+ ],
3263
+ key=lambda item: item.ulid,
3264
+ )
3265
+
3266
+ safe_decisions, safe_domains, _mismatches = self._leftover_archived_hot(
3267
+ archived_decisions, archived_domains
3268
+ )
3269
+
3270
+ if dry_run:
3271
+ return CompactReport(
3272
+ segment_path=None,
3273
+ decisions_compacted=len(new_decisions),
3274
+ domains_compacted=len(new_domains),
3275
+ skipped_age_filtered=skipped_age_filtered,
3276
+ domains_excluded_age_unknown=domains_excluded_age_unknown,
3277
+ cleaned_up_hot_files=len(safe_decisions) + len(safe_domains),
3278
+ items=items,
3279
+ dry_run=True,
3280
+ )
3281
+
3282
+ segment_relpath = None
3283
+ if new_decisions or new_domains:
3284
+ segment_relpath = self._write_archive_segment(new_decisions, new_domains)
3285
+
3286
+ # Fresh candidates: only remove a hot file if its ON-DISK content still
3287
+ # matches exactly what was just archived (Minor-6) -- never trust the INDEX
3288
+ # snapshot alone for a destructive removal. Its canonical_stat row is removed
3289
+ # in the SAME step (digest-integrity design Task 2): a file kept because it
3290
+ # MISMATCHED (the `continue` below) keeps its row too -- mirroring
3291
+ # `_hot_file_matches`'s own skip exactly, so a diverged-but-kept file is never
3292
+ # treated as though this index stopped having loaded it.
3293
+ for d in new_decisions:
3294
+ p = self.path / "decisions" / f"{d.id}.json"
3295
+ if _hot_file_matches(p, d.model_dump(mode="json")) is False:
3296
+ _warn_hot_archive_mismatch("decision", d.id)
3297
+ continue
3298
+ p.unlink(missing_ok=True)
3299
+ self._delete_canonical_stat("decisions", d.id)
3300
+ for dom in new_domains:
3301
+ p = self.path / "domains" / f"{dom.domain_id}.json"
3302
+ if _hot_file_matches(p, _domain_canonical_payload(dom)) is False:
3303
+ _warn_hot_archive_mismatch("domain", dom.domain_id)
3304
+ continue
3305
+ p.unlink(missing_ok=True)
3306
+ self._delete_canonical_stat("domains", dom.domain_id)
3307
+ # Crash-window leftovers: _leftover_archived_hot already did the compare above.
3308
+ for did in safe_decisions:
3309
+ (self.path / "decisions" / f"{did}.json").unlink(missing_ok=True)
3310
+ self._delete_canonical_stat("decisions", did)
3311
+ for dmid in safe_domains:
3312
+ (self.path / "domains" / f"{dmid}.json").unlink(missing_ok=True)
3313
+ self._delete_canonical_stat("domains", dmid)
3314
+
3315
+ cleaned_up = len(safe_decisions) + len(safe_domains)
3316
+ if new_decisions or new_domains or cleaned_up:
3317
+ self._touch_digest()
3318
+
3319
+ return CompactReport(
3320
+ segment_path=segment_relpath,
3321
+ decisions_compacted=len(new_decisions),
3322
+ domains_compacted=len(new_domains),
3323
+ skipped_age_filtered=skipped_age_filtered,
3324
+ domains_excluded_age_unknown=domains_excluded_age_unknown,
3325
+ cleaned_up_hot_files=cleaned_up,
3326
+ items=items,
3327
+ dry_run=False,
3328
+ )
3329
+
3330
+ @property
3331
+ def schema_version(self) -> str:
3332
+ with self._lock:
3333
+ row = self._conn.execute(
3334
+ "SELECT value FROM meta WHERE key = 'schema_version'"
3335
+ ).fetchone()
3336
+ return row["value"]
3337
+
3338
+ def get_meta(self, key: str) -> str | None:
3339
+ with self._lock:
3340
+ row = self._conn.execute("SELECT value FROM meta WHERE key = ?", (key,)).fetchone()
3341
+ return row["value"] if row else None
3342
+
3343
+ def set_meta(self, key: str, value: str) -> None:
3344
+ if key == "schema_version":
3345
+ raise ValueError(
3346
+ "schema_version is stamped at store creation and must not be overwritten"
3347
+ )
3348
+ with self._mutation():
3349
+ self._conn.execute(
3350
+ "INSERT INTO meta (key, value) VALUES (?, ?) "
3351
+ "ON CONFLICT(key) DO UPDATE SET value=excluded.value",
3352
+ (key, value),
3353
+ )
3354
+
3355
+ def close(self) -> None:
3356
+ with self._lock:
3357
+ self._conn.close()
3358
+
3359
+ def __enter__(self) -> Store:
3360
+ return self
3361
+
3362
+ def __exit__(self, *exc: object) -> None:
3363
+ self.close()