markdown-memory 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
markdown_memory/db.py ADDED
@@ -0,0 +1,1546 @@
1
+ """SQLite storage: connection factory, migrations, and the section repository.
2
+
3
+ The store combines three indexes over the same ``sections`` rows:
4
+
5
+ * ``sections`` - canonical relational data (cascade-deleted with its document)
6
+ * ``sections_fts`` - FTS5 external-content index (BM25 keyword search)
7
+ * ``sections_vec`` - sqlite-vec ``vec0`` index, one vector per section with a body
8
+ * ``units`` / ``units_vec`` - the section's passages (paragraph, list item, table row,
9
+ code block) and one vector for each; a section is ranked by its best passage
10
+
11
+ Triggers keep both virtual tables in lock-step with ``sections``, including rows
12
+ removed by ``ON DELETE CASCADE``, so callers only ever write to ``sections``.
13
+
14
+ Connections are per-thread (MCP tool handlers run in worker threads and hybrid
15
+ search queries both indexes concurrently); WAL mode lets readers proceed while
16
+ an indexing transaction is open.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import logging
22
+ import math
23
+ import os
24
+ import sqlite3
25
+ import struct
26
+ import threading
27
+ import time
28
+ from collections.abc import Callable, Iterable, Iterator, Mapping, Sequence
29
+ from contextlib import contextmanager
30
+ from pathlib import Path
31
+ from types import TracebackType
32
+ from typing import Self
33
+
34
+ import sqlite_vec
35
+
36
+ from markdown_memory.exceptions import DatabaseError
37
+ from markdown_memory.models import (
38
+ Document,
39
+ DocumentSummary,
40
+ FileFailure,
41
+ IndexStatus,
42
+ Section,
43
+ SectionDraft,
44
+ SectionVectors,
45
+ )
46
+
47
+ logger = logging.getLogger(__name__)
48
+
49
+ DEFAULT_EMBEDDING_DIM = 384
50
+ SCHEMA_VERSION = 6
51
+
52
+ #: How a section's vector is built. 1 embedded the whole section text, truncated at the
53
+ #: model's token limit; 2 is the mean of the section's passage vectors. Stored per document
54
+ #: so that a document written under the old scheme re-indexes itself and one written under
55
+ #: the new one is left alone - a format change repairs a tree file by file, and resumes
56
+ #: where it stopped if it is interrupted.
57
+ VECTOR_FORMAT = 2
58
+ #: Meta key holding the weights revision the stored vectors were built from. It lives here
59
+ #: because `clear()` has to forget it in the same transaction that deletes them.
60
+ WEIGHTS_META_KEY = "embedding_weights_revision"
61
+ #: Set when the weights behind an unchanged model name changed under an existing index.
62
+ #: It holds the sentence an agent is shown, because the index is then answering from
63
+ #: vectors one model built while the next query would be embedded by another.
64
+ WEIGHTS_MISMATCH_KEY = "embedding_weights_mismatch"
65
+ #: What `WEIGHTS_META_KEY` holds while an index is being re-embedded by other weights. It
66
+ #: equals no revision, so every search - under the old weights or the new - finds it
67
+ #: differs from its own and ranks by keyword alone until a run re-certifies the index. A
68
+ #: revision is a hash, sometimes with a graph path after it; this can be neither.
69
+ WEIGHTS_REVOKED = "(revoked: being re-embedded)"
70
+ _LEGACY_VECTORS = 1
71
+
72
+ _SECTION_ID_META_KEY = "next_section_id"
73
+ _GENERATION_META_KEY = "index_generation"
74
+ _BUSY_TIMEOUT_MS = 10_000
75
+ _WAL_ATTEMPTS = 40
76
+ _WAL_RETRY_SECONDS = 0.05
77
+ _SQL_VARIABLE_BATCH = 500
78
+
79
+ _SECTION_COLUMNS = (
80
+ "id, doc_id, heading_title, heading_level, heading_path, content, "
81
+ "start_line, end_line, part_index"
82
+ )
83
+
84
+
85
+ def _schema_v1(embedding_dim: int) -> tuple[str, ...]:
86
+ return (
87
+ """
88
+ CREATE TABLE meta (
89
+ key TEXT PRIMARY KEY,
90
+ value TEXT NOT NULL
91
+ )
92
+ """,
93
+ """
94
+ CREATE TABLE documents (
95
+ id INTEGER PRIMARY KEY,
96
+ file_path TEXT NOT NULL UNIQUE,
97
+ title TEXT NOT NULL,
98
+ content_hash TEXT NOT NULL,
99
+ last_modified INTEGER NOT NULL
100
+ )
101
+ """,
102
+ """
103
+ CREATE TABLE sections (
104
+ id INTEGER PRIMARY KEY,
105
+ doc_id INTEGER NOT NULL REFERENCES documents(id) ON DELETE CASCADE,
106
+ heading_title TEXT NOT NULL,
107
+ heading_level INTEGER NOT NULL,
108
+ heading_path TEXT NOT NULL,
109
+ content TEXT NOT NULL,
110
+ start_line INTEGER NOT NULL,
111
+ end_line INTEGER NOT NULL,
112
+ part_index INTEGER NOT NULL DEFAULT 0
113
+ )
114
+ """,
115
+ "CREATE INDEX idx_sections_doc ON sections(doc_id, id)",
116
+ """
117
+ CREATE VIRTUAL TABLE sections_fts USING fts5(
118
+ heading_title,
119
+ heading_path,
120
+ content,
121
+ content='sections',
122
+ content_rowid='id',
123
+ tokenize='porter unicode61 remove_diacritics 2'
124
+ )
125
+ """,
126
+ # Headings are short and highly descriptive: weight them above body text.
127
+ "INSERT INTO sections_fts(sections_fts, rank) VALUES ('rank', 'bm25(5.0, 3.0, 1.0)')",
128
+ _vector_tables(embedding_dim)[0],
129
+ """
130
+ CREATE TRIGGER sections_after_insert AFTER INSERT ON sections BEGIN
131
+ INSERT INTO sections_fts(rowid, heading_title, heading_path, content)
132
+ VALUES (new.id, new.heading_title, new.heading_path, new.content);
133
+ END
134
+ """,
135
+ """
136
+ CREATE TRIGGER sections_after_delete AFTER DELETE ON sections BEGIN
137
+ INSERT INTO sections_fts(sections_fts, rowid, heading_title, heading_path, content)
138
+ VALUES ('delete', old.id, old.heading_title, old.heading_path, old.content);
139
+ DELETE FROM sections_vec WHERE section_id = old.id;
140
+ END
141
+ """,
142
+ """
143
+ CREATE TRIGGER sections_after_update AFTER UPDATE ON sections BEGIN
144
+ INSERT INTO sections_fts(sections_fts, rowid, heading_title, heading_path, content)
145
+ VALUES ('delete', old.id, old.heading_title, old.heading_path, old.content);
146
+ INSERT INTO sections_fts(rowid, heading_title, heading_path, content)
147
+ VALUES (new.id, new.heading_title, new.heading_path, new.content);
148
+ END
149
+ """,
150
+ )
151
+
152
+
153
+ def _vector_tables(embedding_dim: int) -> tuple[str, ...]:
154
+ return (
155
+ f"""
156
+ CREATE VIRTUAL TABLE sections_vec USING vec0(
157
+ section_id INTEGER PRIMARY KEY,
158
+ embedding FLOAT[{embedding_dim}] distance_metric=cosine
159
+ )
160
+ """,
161
+ f"""
162
+ CREATE VIRTUAL TABLE units_vec USING vec0(
163
+ unit_id INTEGER PRIMARY KEY,
164
+ embedding FLOAT[{embedding_dim}] distance_metric=cosine
165
+ )
166
+ """,
167
+ )
168
+
169
+
170
+ def _schema_v2(embedding_dim: int) -> tuple[str, ...]:
171
+ """Passage-level vectors: ``units`` rows cascade with their section."""
172
+ return (
173
+ """
174
+ CREATE TABLE units (
175
+ id INTEGER PRIMARY KEY,
176
+ section_id INTEGER NOT NULL REFERENCES sections(id) ON DELETE CASCADE,
177
+ ordinal INTEGER NOT NULL,
178
+ content TEXT NOT NULL
179
+ )
180
+ """,
181
+ "CREATE INDEX idx_units_section ON units(section_id, ordinal)",
182
+ _vector_tables(embedding_dim)[1],
183
+ """
184
+ CREATE TRIGGER units_after_delete AFTER DELETE ON units BEGIN
185
+ DELETE FROM units_vec WHERE unit_id = old.id;
186
+ END
187
+ """,
188
+ )
189
+
190
+
191
+ def _schema_v4() -> tuple[str, ...]:
192
+ """What is known to be wrong with the index, and whether anything vouches for it.
193
+
194
+ Two questions, kept apart because they have different answers. `index_failures` says
195
+ which paths could not be read, one row per path, written by whichever scan last looked
196
+ at that path. `index_coverage` says whether a full walk of a root ever finished without
197
+ failures - the only thing that can distinguish "every file was seen" from "only these
198
+ files were seen", which no amount of per-file state can tell you: a run killed on its
199
+ tenth file leaves the other 990 with no rows at all.
200
+
201
+ Earlier attempts hashed the root (so containment could not be asked), then kept a
202
+ wall-clock guard (wrong in both orderings), then a crash marker that masked the file
203
+ rows. Those are gone. What makes this sound instead is the scan lock: one scan at a
204
+ time, so the run that just walked a tree is entitled to speak for it.
205
+ """
206
+ return (
207
+ """
208
+ CREATE TABLE index_failures (
209
+ file_path TEXT PRIMARY KEY,
210
+ message TEXT NOT NULL
211
+ )
212
+ """,
213
+ """
214
+ CREATE TABLE index_coverage (
215
+ root TEXT PRIMARY KEY,
216
+ verified INTEGER NOT NULL
217
+ )
218
+ """,
219
+ )
220
+
221
+
222
+ def serialize_embedding(embedding: Sequence[float]) -> bytes:
223
+ """Pack a vector into the little-endian float32 blob format sqlite-vec expects."""
224
+ return struct.pack(f"<{len(embedding)}f", *embedding)
225
+
226
+
227
+ def _is_usable_vector(embedding: Sequence[float]) -> bool:
228
+ """Cosine distance is undefined (NaN) for non-finite or zero-length vectors."""
229
+ return all(math.isfinite(value) for value in embedding) and any(embedding)
230
+
231
+
232
+ def _bump_generation(conn: sqlite3.Connection) -> None:
233
+ """Mark that everything indexed before this moment is gone."""
234
+ conn.execute(
235
+ "INSERT INTO meta(key, value) VALUES (?, '1') "
236
+ "ON CONFLICT(key) DO UPDATE SET value = CAST(CAST(value AS INTEGER) + 1 AS TEXT)",
237
+ (_GENERATION_META_KEY,),
238
+ )
239
+
240
+
241
+ def _forget_weights_without_vectors(conn: sqlite3.Connection) -> None:
242
+ """Drop the weights metadata once the vectors it describes are gone.
243
+
244
+ Which weights produced the vectors is a fact about the vectors, so it belongs to the
245
+ same transaction that removes them - a purge of the last document, a rebuild for a new
246
+ vector size, a discard for a changed model. Kept behind, it would have the next run
247
+ compare a new model against the revision of a model whose output no longer exists, and
248
+ have search rank on keywords alone over an index with nothing wrong with it.
249
+
250
+ Heading-only documents hold no vectors, so the test is the vectors themselves rather
251
+ than the document rows above them.
252
+ """
253
+ if conn.execute("SELECT 1 FROM units_vec LIMIT 1").fetchone() is not None:
254
+ return
255
+ conn.execute("DELETE FROM meta WHERE key = ?", (WEIGHTS_META_KEY,))
256
+ conn.execute("DELETE FROM meta WHERE key = ?", (WEIGHTS_MISMATCH_KEY,))
257
+
258
+
259
+ def _revoke_weights(conn: sqlite3.Connection, message: str) -> None:
260
+ """Mark the index as being re-embedded, and say why, in the caller's transaction.
261
+
262
+ One transaction for both: a revocation with no message would leave `index_status`
263
+ calling the index healthy while search refuses its vectors, and a message with no
264
+ revocation would let search rank the mixture the message warns about.
265
+ """
266
+ conn.executemany(
267
+ "INSERT INTO meta (key, value) VALUES (?, ?) "
268
+ "ON CONFLICT(key) DO UPDATE SET value = excluded.value",
269
+ ((WEIGHTS_META_KEY, WEIGHTS_REVOKED), (WEIGHTS_MISMATCH_KEY, message)),
270
+ )
271
+
272
+
273
+ def _directory_prefix(directory: str) -> str:
274
+ return directory if directory.endswith(os.sep) else directory + os.sep
275
+
276
+
277
+ class Database:
278
+ """Thread-safe SQLite client and repository for documents and sections.
279
+
280
+ Use as a context manager to guarantee every per-thread connection is closed::
281
+
282
+ with Database(path) as db:
283
+ with db.transaction() as conn:
284
+ ...
285
+ """
286
+
287
+ def __init__(self, path: str | Path, *, embedding_dim: int = DEFAULT_EMBEDDING_DIM) -> None:
288
+ if str(path) == ":memory:" or str(path).startswith("file::memory:"):
289
+ raise DatabaseError(
290
+ "In-memory databases are not supported: connections are per-thread "
291
+ "and would each see an empty database. Use a file path."
292
+ )
293
+ if embedding_dim <= 0:
294
+ raise DatabaseError(f"embedding_dim must be positive, got {embedding_dim}")
295
+ self._path = Path(path).expanduser()
296
+ self._embedding_dim = embedding_dim
297
+ self._local = threading.local()
298
+ self._connections: list[tuple[threading.Thread, sqlite3.Connection]] = []
299
+ self._lock = threading.Lock()
300
+ self._closed = False
301
+ try:
302
+ self._path.parent.mkdir(parents=True, exist_ok=True)
303
+ except OSError as exc:
304
+ raise DatabaseError(f"Cannot create database directory {self._path.parent}") from exc
305
+ self._migrate()
306
+
307
+ # ------------------------------------------------------------------ lifecycle
308
+
309
+ def __enter__(self) -> Self:
310
+ return self
311
+
312
+ def __exit__(
313
+ self,
314
+ exc_type: type[BaseException] | None,
315
+ exc: BaseException | None,
316
+ traceback: TracebackType | None,
317
+ ) -> None:
318
+ self.close()
319
+
320
+ @property
321
+ def path(self) -> Path:
322
+ return self._path
323
+
324
+ @property
325
+ def embedding_dim(self) -> int:
326
+ return self._embedding_dim
327
+
328
+ def close(self) -> None:
329
+ """Close every connection opened by any thread. Idempotent."""
330
+ with self._lock:
331
+ self._closed = True
332
+ connections, self._connections = self._connections, []
333
+ for _, conn in connections:
334
+ _close_quietly(conn)
335
+
336
+ @property
337
+ def open_connection_count(self) -> int:
338
+ """Connections currently held open across all threads (diagnostics)."""
339
+ with self._lock:
340
+ return len(self._connections)
341
+
342
+ def connection(self) -> sqlite3.Connection:
343
+ """Return this thread's connection, opening and configuring it on first use.
344
+
345
+ Tool handlers run on pooled worker threads that the runtime retires when idle,
346
+ so opening a connection is also the moment the connections of threads that have
347
+ since exited are closed; otherwise a long-lived server leaks file descriptors.
348
+ """
349
+ if self._closed:
350
+ raise DatabaseError("Database is closed")
351
+ existing: sqlite3.Connection | None = getattr(self._local, "conn", None)
352
+ if existing is not None:
353
+ return existing
354
+ conn = self._open()
355
+ abandoned: list[sqlite3.Connection] = []
356
+ with self._lock:
357
+ if self._closed:
358
+ conn.close()
359
+ raise DatabaseError("Database is closed")
360
+ alive: list[tuple[threading.Thread, sqlite3.Connection]] = []
361
+ for thread, other in self._connections:
362
+ if thread.is_alive():
363
+ alive.append((thread, other))
364
+ else:
365
+ abandoned.append(other)
366
+ alive.append((threading.current_thread(), conn))
367
+ self._connections = alive
368
+ self._local.conn = conn
369
+ for other in abandoned:
370
+ _close_quietly(other)
371
+ return conn
372
+
373
+ def _open(self) -> sqlite3.Connection:
374
+ try:
375
+ # isolation_level=None: autocommit; transactions are explicit (see transaction()).
376
+ # check_same_thread=False only so close() may run from another thread;
377
+ # each connection is otherwise used exclusively by the thread that opened it.
378
+ conn = sqlite3.connect(
379
+ self._path,
380
+ isolation_level=None,
381
+ check_same_thread=False,
382
+ timeout=_BUSY_TIMEOUT_MS / 1000,
383
+ )
384
+ except sqlite3.Error as exc:
385
+ raise DatabaseError(f"Cannot open database at {self._path}: {exc}") from exc
386
+ try:
387
+ conn.enable_load_extension(True)
388
+ sqlite_vec.load(conn)
389
+ conn.enable_load_extension(False)
390
+ except (sqlite3.Error, AttributeError) as exc:
391
+ conn.close()
392
+ raise DatabaseError(
393
+ "Cannot load the sqlite-vec extension. This Python build's sqlite3 module "
394
+ f"must support loadable extensions: {exc}"
395
+ ) from exc
396
+ try:
397
+ conn.execute(f"PRAGMA busy_timeout = {_BUSY_TIMEOUT_MS}")
398
+ _enable_wal(conn)
399
+ conn.execute("PRAGMA synchronous = NORMAL")
400
+ conn.execute("PRAGMA foreign_keys = ON")
401
+ except sqlite3.Error as exc:
402
+ conn.close()
403
+ raise DatabaseError(f"Cannot configure database connection: {exc}") from exc
404
+ return conn
405
+
406
+ @contextmanager
407
+ def transaction(self) -> Iterator[sqlite3.Connection]:
408
+ """Run a write transaction: commit on success, roll back on any exception."""
409
+ conn = self.connection()
410
+ try:
411
+ conn.execute("BEGIN IMMEDIATE")
412
+ except sqlite3.Error as exc:
413
+ raise DatabaseError(f"Cannot begin transaction: {exc}") from exc
414
+ try:
415
+ yield conn
416
+ conn.execute("COMMIT")
417
+ except BaseException as exc:
418
+ if conn.in_transaction:
419
+ try:
420
+ conn.execute("ROLLBACK")
421
+ except sqlite3.Error: # pragma: no cover - rollback is best effort
422
+ logger.exception("Rollback failed")
423
+ if isinstance(exc, sqlite3.Error):
424
+ raise DatabaseError(f"Transaction failed: {exc}") from exc
425
+ raise
426
+
427
+ @contextmanager
428
+ def _reading(self) -> Iterator[sqlite3.Connection]:
429
+ """Yield the connection for reads, translating driver errors."""
430
+ conn = self.connection()
431
+ try:
432
+ yield conn
433
+ except sqlite3.Error as exc:
434
+ raise DatabaseError(f"Query failed: {exc}") from exc
435
+
436
+ # ------------------------------------------------------------------ migrations
437
+
438
+ def _migrate(self) -> None:
439
+ conn = self.connection()
440
+ try:
441
+ version = int(conn.execute("PRAGMA user_version").fetchone()[0])
442
+ except sqlite3.Error as exc:
443
+ raise DatabaseError(f"Cannot read schema version: {exc}") from exc
444
+ if version > SCHEMA_VERSION:
445
+ raise DatabaseError(
446
+ f"Database schema v{version} is newer than this build supports "
447
+ f"(v{SCHEMA_VERSION}); upgrade markdown-memory."
448
+ )
449
+ if version < SCHEMA_VERSION:
450
+ applied: list[int] = []
451
+ with self.transaction() as tx:
452
+ # Re-check under the write lock: another process starting at the same
453
+ # moment may have migrated the schema while this one waited for it.
454
+ current = int(tx.execute("PRAGMA user_version").fetchone()[0])
455
+ if current < 1:
456
+ for statement in _schema_v1(self._embedding_dim):
457
+ tx.execute(statement)
458
+ tx.execute(
459
+ "INSERT INTO meta(key, value) VALUES ('embedding_dim', ?)",
460
+ (str(self._embedding_dim),),
461
+ )
462
+ applied.append(1)
463
+ if current < 2:
464
+ stored = tx.execute(
465
+ "SELECT value FROM meta WHERE key = 'embedding_dim'"
466
+ ).fetchone()
467
+ for statement in _schema_v2(int(stored[0])):
468
+ tx.execute(statement)
469
+ # Sections indexed before v2 have no passages, and their unchanged
470
+ # SHA-256 would make every later run skip them. The index is a cache
471
+ # of the files: drop it so the next index_directory rebuilds it whole.
472
+ dropped = tx.execute("DELETE FROM documents").rowcount
473
+ if dropped:
474
+ _add_notice(
475
+ tx,
476
+ f"The index format changed (passage-level vectors): discarded all "
477
+ f"{dropped} previously indexed documents from every directory. "
478
+ "Re-run index_directory for each documentation root.",
479
+ )
480
+ applied.append(2)
481
+ if current < 4:
482
+ for statement in _schema_v4():
483
+ tx.execute(statement)
484
+ tx.execute(
485
+ "ALTER TABLE documents "
486
+ f"ADD COLUMN vector_format INTEGER NOT NULL DEFAULT {_LEGACY_VECTORS}"
487
+ )
488
+ # v3 was never released; it exists only in working copies of the
489
+ # abandoned design. Its table is dropped rather than migrated.
490
+ tx.execute("DROP TABLE IF EXISTS index_problems")
491
+ tx.execute("DELETE FROM meta WHERE key LIKE 'incomplete:%'")
492
+ applied.append(4)
493
+ if current < 5:
494
+ # Rows written before v5 recorded whole seconds, which cannot see an
495
+ # edit made in the same second as the scan that indexed it. They get
496
+ # NULL, which means "no modification time was recorded" - not a
497
+ # timestamp of any kind, so no real one can collide with it, the epoch
498
+ # included. The freshness check reads their bytes instead of trusting a
499
+ # time, which is the slow answer but never the wrong one, and the next
500
+ # index_directory writes a real value and the file stops paying it.
501
+ # Nothing is discarded for this: the vectors are still good, and a
502
+ # rebuild would cost half an hour of embedding to learn nothing.
503
+ tx.execute("ALTER TABLE documents ADD COLUMN mtime_ns INTEGER")
504
+ applied.append(5)
505
+ if current < 6:
506
+ # Which weights embedded each document, so a change of model repairs the
507
+ # index file by file instead of discarding it. Copied from the index-wide
508
+ # revision where one is recorded, and that copy is exact: the revision is
509
+ # written only while no vector exists, every vector after it was checked
510
+ # against it, and it is forgotten only once no vector is left. Where none
511
+ # is recorded but vectors are, nobody can say what built them - they are
512
+ # quarantined now, in this transaction, rather than ranked against a query
513
+ # until some later run happens to notice.
514
+ tx.execute("ALTER TABLE documents ADD COLUMN weights_revision TEXT")
515
+ recorded = tx.execute(
516
+ "SELECT value FROM meta WHERE key = ?", (WEIGHTS_META_KEY,)
517
+ ).fetchone()
518
+ if recorded is not None:
519
+ tx.execute("UPDATE documents SET weights_revision = ?", (recorded[0],))
520
+ elif tx.execute("SELECT 1 FROM units_vec LIMIT 1").fetchone() is not None:
521
+ _revoke_weights(
522
+ tx,
523
+ "No record says which weights built this index's vectors, so "
524
+ "they are not compared with a query: only keyword ranking is used "
525
+ "until index_directory re-embeds them.",
526
+ )
527
+ applied.append(6)
528
+ tx.execute(f"PRAGMA user_version = {SCHEMA_VERSION}")
529
+ if applied:
530
+ logger.info("Applied schema migration(s) %s at %s", applied, self._path)
531
+ self._seed_section_ids()
532
+ stored_dim = self.get_meta("embedding_dim")
533
+ if stored_dim is not None and int(stored_dim) != self._embedding_dim:
534
+ self._rebuild_for_dimension(int(stored_dim))
535
+
536
+ def _seed_section_ids(self) -> None:
537
+ """Give a database from an earlier release its section-id high-water mark.
538
+
539
+ Without it the mark would be derived from ``MAX(id)`` after rows had already been
540
+ deleted, so a re-index - or a purge followed by one - would hand the freed ids
541
+ straight back out, which is exactly what ``_next_section_id`` exists to prevent.
542
+ """
543
+ if self.get_meta(_SECTION_ID_META_KEY) is not None:
544
+ return
545
+ with self.transaction() as tx:
546
+ if tx.execute("SELECT 1 FROM meta WHERE key = ?", (_SECTION_ID_META_KEY,)).fetchone():
547
+ return # another process seeded it while this one waited for the lock
548
+ tx.execute(
549
+ "INSERT INTO meta(key, value) "
550
+ "SELECT ?, CAST(COALESCE(MAX(id), 0) + 1 AS TEXT) FROM sections",
551
+ (_SECTION_ID_META_KEY,),
552
+ )
553
+
554
+ def _rebuild_for_dimension(self, stored_dim: int) -> None:
555
+ """Re-create the vector tables for a model with a different output size.
556
+
557
+ The index is a cache of the Markdown files: vectors of another dimensionality
558
+ are useless, so everything is dropped and the next ``index_directory`` rebuilds it.
559
+ """
560
+ logger.warning(
561
+ "Embedding size changed (%d -> %d): discarding the index at %s; re-run "
562
+ "index_directory to rebuild it",
563
+ stored_dim,
564
+ self._embedding_dim,
565
+ self._path,
566
+ )
567
+ with self.transaction() as tx:
568
+ dropped = tx.execute("DELETE FROM documents").rowcount
569
+ _forget_weights_without_vectors(tx)
570
+ # Every root's documents are gone, including roots this process never looked
571
+ # at; a certificate that survived would vouch for an empty tree.
572
+ self.revoke_coverage(tx)
573
+ if dropped:
574
+ _add_notice(
575
+ tx,
576
+ f"The embedding size changed ({stored_dim} -> {self._embedding_dim} "
577
+ f"dimensions): discarded all {dropped} previously indexed documents from "
578
+ "every directory. Re-run index_directory for each documentation root.",
579
+ )
580
+ tx.execute("DROP TABLE sections_vec")
581
+ tx.execute("DROP TABLE units_vec")
582
+ for statement in _vector_tables(self._embedding_dim):
583
+ tx.execute(statement)
584
+ tx.execute(
585
+ "UPDATE meta SET value = ? WHERE key = 'embedding_dim'",
586
+ (str(self._embedding_dim),),
587
+ )
588
+
589
+ # ------------------------------------------------------------------ meta / pragmas
590
+
591
+ def get_meta(self, key: str) -> str | None:
592
+ with self._reading() as conn:
593
+ row = conn.execute("SELECT value FROM meta WHERE key = ?", (key,)).fetchone()
594
+ return None if row is None else str(row[0])
595
+
596
+ def set_meta(self, key: str, value: str) -> None:
597
+ with self.transaction() as conn:
598
+ conn.execute(
599
+ "INSERT INTO meta(key, value) VALUES (?, ?) "
600
+ "ON CONFLICT(key) DO UPDATE SET value = excluded.value",
601
+ (key, value),
602
+ )
603
+
604
+ def pending_notices(self) -> dict[str, str]:
605
+ """Messages left by whatever discarded the index, oldest first, keyed for dismissal.
606
+
607
+ They are persisted, not logged only, so that whoever next runs ``index_directory``
608
+ - possibly another process, much later - is told why the index was empty. Reading
609
+ does not clear them: a run that fails before it can report them must not eat them.
610
+ """
611
+ with self._reading() as conn:
612
+ rows = conn.execute(
613
+ "SELECT key, value FROM meta WHERE key LIKE 'notice:%' "
614
+ "ORDER BY CAST(substr(key, 8) AS INTEGER)"
615
+ ).fetchall()
616
+ return {str(key): str(value) for key, value in rows}
617
+
618
+ def failure_paths(self, root: str) -> list[str]:
619
+ """Every path recorded as unreadable under ``root``."""
620
+ prefix = _directory_prefix(root)
621
+ with self._reading() as conn:
622
+ rows = conn.execute(
623
+ "SELECT file_path FROM index_failures "
624
+ "WHERE file_path = ? OR substr(file_path, 1, length(?)) = ?",
625
+ (root, prefix, prefix),
626
+ ).fetchall()
627
+ return [str(row[0]) for row in rows]
628
+
629
+ def record_failures(self, clear: Sequence[str], failures: Mapping[str, str]) -> None:
630
+ """Forget the failures in ``clear``, then record ``failures``.
631
+
632
+ The caller names what to forget rather than passing a root, because a walk does
633
+ not reach everything beneath its root: `.venv` and `node_modules` are pruned, and
634
+ a directory that cannot be listed is skipped. Clearing by prefix would erase what
635
+ a scan never looked at - a file recorded as broken inside a pruned directory would
636
+ be quietly declared fine by a run of its parent.
637
+
638
+ What is cleared is replaced wholesale, because a failure can outlive every chance
639
+ to clear it one at a time: a file that fails on its *first* index never reaches
640
+ `replace_document`, so it never enters `documents` and a later purge cannot find
641
+ it either. Sound because one scan runs at a time - the run that just walked these
642
+ paths is the freshest word on them.
643
+ """
644
+ with self.transaction() as conn:
645
+ conn.executemany(
646
+ "DELETE FROM index_failures WHERE file_path = ?", [(path,) for path in clear]
647
+ )
648
+ if failures:
649
+ conn.executemany(
650
+ "INSERT INTO index_failures(file_path, message) VALUES (?, ?)",
651
+ sorted(failures.items()),
652
+ )
653
+
654
+ def mark_scan_started(self, root: str) -> None:
655
+ """This root, and every root containing it, is no longer vouched for.
656
+
657
+ Called at a scan's first write, not at its start: a run that changes nothing -
658
+ the model will not load, every file is unchanged - has no business retracting a
659
+ certificate. A scan of `docs/api` retracts `docs` too, because a half-written
660
+ subtree is a half-written tree.
661
+ """
662
+ prefix = _directory_prefix(root)
663
+ with self.transaction() as conn:
664
+ conn.execute(
665
+ "UPDATE index_coverage SET verified = 0 "
666
+ # itself, anything containing it, and anything inside it: this scan may
667
+ # rewrite any of them, and none may go on vouching for itself while it does
668
+ "WHERE root = ? "
669
+ "OR substr(?, 1, length(root) + 1) = root || ? "
670
+ "OR substr(root, 1, length(?)) = ?",
671
+ (root, root, os.sep, prefix, prefix),
672
+ )
673
+ conn.execute(
674
+ "INSERT INTO index_coverage(root, verified) VALUES (?, 0) "
675
+ "ON CONFLICT(root) DO UPDATE SET verified = 0",
676
+ (root,),
677
+ )
678
+
679
+ def mark_scan_complete(self, root: str, generation: int) -> None:
680
+ """A full walk of ``root`` ran to the end.
681
+
682
+ Not "and everything was readable" - that is what `index_failures` is for, and
683
+ `index_status` will not call a tree whole while anything under it is listed there.
684
+ Keeping the two apart means a run does not have to decide what a later question
685
+ will mean.
686
+
687
+ Only this root's row is written. An earlier draft also deleted the rows of roots
688
+ inside it, on the grounds that this walk covered them - but it does not cover a
689
+ pruned subtree, and a walk that hit failures covered even less. Deleting them
690
+ turned a nested root that was perfectly fine into one that reported unknown,
691
+ which is a worse answer than the one it replaced.
692
+ """
693
+ with self.transaction() as conn:
694
+ current = int(
695
+ (
696
+ conn.execute(
697
+ "SELECT value FROM meta WHERE key = ?", (_GENERATION_META_KEY,)
698
+ ).fetchone()
699
+ or ("0",)
700
+ )[0]
701
+ )
702
+ if current != generation:
703
+ # The index was discarded while this scan was walking; what it just
704
+ # measured describes a database that no longer exists.
705
+ return
706
+ conn.execute(
707
+ "INSERT INTO index_coverage(root, verified) VALUES (?, 1) "
708
+ "ON CONFLICT(root) DO UPDATE SET verified = 1",
709
+ (root,),
710
+ )
711
+
712
+ def revoke_coverage(self, conn: sqlite3.Connection | None = None) -> None:
713
+ """Nothing is vouched for any more - the index itself was discarded.
714
+
715
+ A model or dimension change empties every document in the database, including
716
+ roots this process never looked at. A certificate that outlives its subject is
717
+ worse than none: it says a tree is whole when nothing of it is left.
718
+
719
+ The generation is bumped in the same breath. Revoking only settles the
720
+ certificates that exist *now*; a scan already running has read its file hashes,
721
+ will skip every file as unchanged, and would write a fresh certificate over an
722
+ empty database. It compares the generation instead and stands down.
723
+ """
724
+ if conn is not None:
725
+ _bump_generation(conn)
726
+ conn.execute("DELETE FROM index_coverage")
727
+ return
728
+ with self.transaction() as owned:
729
+ _bump_generation(owned)
730
+ owned.execute("DELETE FROM index_coverage")
731
+
732
+ def generation(self) -> int:
733
+ """How many times this database has been emptied wholesale."""
734
+ return int(self.get_meta(_GENERATION_META_KEY) or "0")
735
+
736
+ def index_status(self, root: str, scope: str | None = None) -> IndexStatus:
737
+ """What can honestly be said about answers drawn from ``root``.
738
+
739
+ ``scope`` narrows *what is named* - the failures and stale documents worth
740
+ mentioning - without changing whose coverage is being reported: a caller asking
741
+ about one directory is still served from the whole root, and the certificate
742
+ belongs to the root.
743
+
744
+ Every read is one snapshot. Taken separately, a scan committing between them hands
745
+ back a verdict that was never true at any instant: the failures read as empty, the
746
+ certificate still reads valid, and the answer claims a whole tree while the row
747
+ proving otherwise is already committed. Composing two snapshots in the caller has
748
+ exactly the same hole, which is why the narrowing happens here.
749
+ """
750
+ named = scope if scope is not None else root
751
+ prefix = _directory_prefix(named)
752
+ with self._reading() as conn:
753
+ conn.execute("BEGIN")
754
+ try:
755
+ rows = conn.execute(
756
+ "SELECT file_path, message FROM index_failures "
757
+ "WHERE file_path = ? OR substr(file_path, 1, length(?)) = ? "
758
+ "ORDER BY file_path",
759
+ (named, prefix, prefix),
760
+ ).fetchall()
761
+ certificate = conn.execute(
762
+ "SELECT verified FROM index_coverage WHERE root = ?", (root,)
763
+ ).fetchone()
764
+ # In the same snapshot as the rest: a verdict that mixes one moment's
765
+ # certificate with another's provenance describes no moment at all.
766
+ mismatch = conn.execute(
767
+ "SELECT value FROM meta WHERE key = ?", (WEIGHTS_MISMATCH_KEY,)
768
+ ).fetchone()
769
+ stale_vectors = int(
770
+ conn.execute(
771
+ "SELECT COUNT(*) FROM documents "
772
+ "WHERE (file_path = ? OR substr(file_path, 1, length(?)) = ?) "
773
+ "AND vector_format != ?",
774
+ (named, prefix, prefix, VECTOR_FORMAT),
775
+ ).fetchone()[0]
776
+ )
777
+ # Whether the ROOT is whole, in the same snapshot. A caller asking about
778
+ # one directory is still answered from the whole root, so a clean
779
+ # subdirectory of a root that lost files is not itself a safe answer -
780
+ # only what is *named* narrows.
781
+ whole = rows == [] and stale_vectors == 0
782
+ if named != root:
783
+ root_prefix = _directory_prefix(root)
784
+ whole = not conn.execute(
785
+ "SELECT 1 FROM index_failures "
786
+ "WHERE file_path = ? OR substr(file_path, 1, length(?)) = ? "
787
+ "UNION ALL SELECT 1 FROM documents "
788
+ "WHERE (file_path = ? OR substr(file_path, 1, length(?)) = ?) "
789
+ "AND vector_format != ? LIMIT 1",
790
+ (root, root_prefix, root_prefix,
791
+ root, root_prefix, root_prefix, VECTOR_FORMAT),
792
+ ).fetchone() # fmt: skip
793
+ finally:
794
+ conn.execute("COMMIT")
795
+ failures = tuple(
796
+ FileFailure(file_path=str(path), message=str(message)) for path, message in rows
797
+ )
798
+ verified = certificate is not None and bool(certificate[0]) and whole
799
+ weights_mismatch = str(mismatch[0]) if mismatch else None
800
+ return IndexStatus(
801
+ # A walk that read every file still cannot vouch for vectors built by a
802
+ # model that is no longer the one answering.
803
+ verified=verified and weights_mismatch is None,
804
+ failures=failures,
805
+ stale_vectors=stale_vectors,
806
+ weights_mismatch=weights_mismatch,
807
+ )
808
+
809
+ def dismiss_notices(self, keys: Iterable[str]) -> None:
810
+ """Forget the notices that have been delivered; any added since are kept."""
811
+ with self.transaction() as conn:
812
+ conn.executemany("DELETE FROM meta WHERE key = ?", [(key,) for key in keys])
813
+
814
+ def pragma(self, name: str) -> str:
815
+ """Return a PRAGMA's current value on this thread's connection."""
816
+ if not name.isidentifier():
817
+ raise DatabaseError(f"Invalid pragma name: {name!r}")
818
+ with self._reading() as conn:
819
+ row = conn.execute(f"PRAGMA {name}").fetchone()
820
+ return "" if row is None else str(row[0])
821
+
822
+ # ------------------------------------------------------------------ documents
823
+
824
+ def get_document(self, file_path: str) -> Document | None:
825
+ with self._reading() as conn:
826
+ row = conn.execute(
827
+ "SELECT id, file_path, title, content_hash, last_modified "
828
+ "FROM documents WHERE file_path = ?",
829
+ (file_path,),
830
+ ).fetchone()
831
+ return None if row is None else _document_from_row(row)
832
+
833
+ def find_documents_by_suffix(self, relative_path: str) -> list[Document]:
834
+ """Documents whose stored path ends with ``/<relative_path>``."""
835
+ suffix = os.sep + relative_path.lstrip("/\\")
836
+ with self._reading() as conn:
837
+ rows = conn.execute(
838
+ "SELECT id, file_path, title, content_hash, last_modified FROM documents "
839
+ "WHERE substr(file_path, -length(?)) = ? ORDER BY file_path",
840
+ (suffix, suffix),
841
+ ).fetchall()
842
+ return [_document_from_row(row) for row in rows]
843
+
844
+ def document_fingerprints(self, directory: str) -> dict[str, tuple[str, int | None]]:
845
+ """Map ``file_path -> (content_hash, mtime_ns)`` under ``directory``.
846
+
847
+ What a caller needs to ask the filesystem whether the index is still current:
848
+ the cheap question first (has the modification time moved?) and the expensive
849
+ one - re-hashing the bytes - only for the files where it has.
850
+ """
851
+ prefix = _directory_prefix(directory)
852
+ with self._reading() as conn:
853
+ rows = conn.execute(
854
+ "SELECT file_path, content_hash, mtime_ns FROM documents "
855
+ "WHERE substr(file_path, 1, length(?)) = ?",
856
+ (prefix, prefix),
857
+ ).fetchall()
858
+ return {
859
+ str(path): (str(content_hash), None if mtime_ns is None else int(mtime_ns))
860
+ for path, content_hash, mtime_ns in rows
861
+ }
862
+
863
+ def document_hashes(self, directory: str) -> dict[str, tuple[str, int, int | None, str | None]]:
864
+ """Map ``file_path -> (hash, vector_format, mtime_ns, weights_revision)`` under it.
865
+
866
+ The format and the weights travel with the hash because all three answer the same
867
+ question - may this file be skipped? - and a file whose vectors predate the current
868
+ pooling, or came from other weights, must be rebuilt however unchanged its bytes
869
+ are. The recorded modification time travels with them because a file that may be
870
+ skipped still has to have that time brought up to date, or the freshness check reads
871
+ the bytes of an unchanged file for ever.
872
+ """
873
+ prefix = _directory_prefix(directory)
874
+ with self._reading() as conn:
875
+ rows = conn.execute(
876
+ "SELECT file_path, content_hash, vector_format, mtime_ns, weights_revision "
877
+ "FROM documents WHERE substr(file_path, 1, length(?)) = ?",
878
+ (prefix, prefix),
879
+ ).fetchall()
880
+ return {
881
+ str(path): (
882
+ str(content_hash),
883
+ int(vector_format),
884
+ None if mtime_ns is None else int(mtime_ns),
885
+ None if weights is None else str(weights),
886
+ )
887
+ for path, content_hash, vector_format, mtime_ns, weights in rows
888
+ }
889
+
890
+ def record_modification_time(
891
+ self, file_path: str, content_hash: str, previous_ns: int | None, mtime_ns: int
892
+ ) -> None:
893
+ """Note when an unchanged file was last written, without touching its content.
894
+
895
+ A file whose bytes are what was indexed is skipped, and used to keep whatever time
896
+ it was stored with - a `touch`, a checkout, or a row migrated from a schema that
897
+ had no nanoseconds at all. Every freshness sweep then found a time that did not
898
+ match and hashed the file again, forever, to conclude what the hash it already
899
+ held could have said. One narrow UPDATE ends that: no sections, no vectors, no
900
+ reindex.
901
+
902
+ A compare-and-swap on both the hash and the time the caller started from. Whoever
903
+ checked those bytes did so outside this transaction: an indexing run may have
904
+ replaced the document in between - writing a time against somebody else's content
905
+ is how a stale row comes to look current - or may have recorded a *newer* time for
906
+ the same content, which this must not roll back, or the file it just verified gets
907
+ hashed all over again. `IS` rather than `=` so a row that had no time recorded at
908
+ all is matched rather than skipped.
909
+ """
910
+ with self.transaction() as conn:
911
+ conn.execute(
912
+ "UPDATE documents SET last_modified = ?, mtime_ns = ? "
913
+ "WHERE file_path = ? AND content_hash = ? AND mtime_ns IS ?",
914
+ (mtime_ns // 1_000_000_000, mtime_ns, file_path, content_hash, previous_ns),
915
+ )
916
+
917
+ def list_documents(self, directory: str = "") -> list[DocumentSummary]:
918
+ """All documents (optionally restricted to ``directory``) with section counts."""
919
+ sql = (
920
+ "SELECT d.file_path, d.title, COUNT(s.id), d.last_modified "
921
+ "FROM documents d LEFT JOIN sections s ON s.doc_id = d.id "
922
+ )
923
+ params: tuple[str, ...] = ()
924
+ if directory:
925
+ prefix = _directory_prefix(directory)
926
+ sql += "WHERE substr(d.file_path, 1, length(?)) = ? "
927
+ params = (prefix, prefix)
928
+ sql += "GROUP BY d.id ORDER BY d.file_path"
929
+ with self._reading() as conn:
930
+ rows = conn.execute(sql, params).fetchall()
931
+ return [
932
+ DocumentSummary(
933
+ file_path=str(path),
934
+ title=str(title),
935
+ section_count=int(count),
936
+ last_modified=int(modified),
937
+ )
938
+ for path, title, count, modified in rows
939
+ ]
940
+
941
+ def replace_document(
942
+ self,
943
+ *,
944
+ file_path: str,
945
+ title: str,
946
+ content_hash: str,
947
+ last_modified: int,
948
+ mtime_ns: int,
949
+ sections: Sequence[SectionDraft],
950
+ vectors: Sequence[SectionVectors],
951
+ weights_revision: str | None = None,
952
+ ) -> Document:
953
+ """Atomically insert or fully replace one document, its sections and vectors.
954
+
955
+ ``weights_revision`` names the weights that produced ``vectors``, and is written in
956
+ the same transaction as them: it is the one moment the two are known to belong
957
+ together. None means unknown, which no known revision will match.
958
+ """
959
+ if len(sections) != len(vectors):
960
+ raise DatabaseError(
961
+ f"Got {len(sections)} sections but {len(vectors)} vector sets for {file_path}"
962
+ )
963
+ for section, vector in zip(sections, vectors, strict=True):
964
+ if len(vector.units) != len(section.units):
965
+ raise DatabaseError(
966
+ f"Section '{section.heading_path}' of {file_path} has {len(section.units)} "
967
+ f"passages but {len(vector.units)} passage vectors"
968
+ )
969
+ if (vector.section is None) != (not section.units):
970
+ raise DatabaseError(
971
+ f"Section '{section.heading_path}' of {file_path}: a section vector is "
972
+ "required exactly when the section has passages"
973
+ )
974
+ present = [] if vector.section is None else [vector.section]
975
+ for embedding in (*present, *vector.units):
976
+ self._check_vector(embedding, f"Embedding for {file_path}")
977
+ with self.transaction() as conn:
978
+ # Sampled before the delete below: on a database from an earlier release the
979
+ # high-water mark is missing, and MAX(id) taken afterwards would hand the ids
980
+ # of the rows just deleted straight back out.
981
+ section_id = _next_section_id(conn) - 1
982
+ row = conn.execute(
983
+ "SELECT id FROM documents WHERE file_path = ?", (file_path,)
984
+ ).fetchone()
985
+ if row is None:
986
+ cursor = conn.execute(
987
+ "INSERT INTO documents(file_path, title, content_hash, last_modified, "
988
+ "mtime_ns, vector_format, weights_revision) VALUES (?, ?, ?, ?, ?, ?, ?)",
989
+ (
990
+ file_path,
991
+ title,
992
+ content_hash,
993
+ last_modified,
994
+ mtime_ns,
995
+ VECTOR_FORMAT,
996
+ weights_revision,
997
+ ),
998
+ )
999
+ if cursor.lastrowid is None: # pragma: no cover - sqlite always sets it
1000
+ raise DatabaseError("INSERT INTO documents returned no rowid")
1001
+ doc_id = cursor.lastrowid
1002
+ else:
1003
+ doc_id = int(row[0])
1004
+ conn.execute(
1005
+ "UPDATE documents SET title = ?, content_hash = ?, last_modified = ?, "
1006
+ "mtime_ns = ?, vector_format = ?, weights_revision = ? WHERE id = ?",
1007
+ (
1008
+ title,
1009
+ content_hash,
1010
+ last_modified,
1011
+ mtime_ns,
1012
+ VECTOR_FORMAT,
1013
+ weights_revision,
1014
+ doc_id,
1015
+ ),
1016
+ )
1017
+ conn.execute("DELETE FROM sections WHERE doc_id = ?", (doc_id,))
1018
+ for section, vector in zip(sections, vectors, strict=True):
1019
+ section_id += 1
1020
+ conn.execute(
1021
+ "INSERT INTO sections(id, doc_id, heading_title, heading_level, "
1022
+ "heading_path, content, start_line, end_line, part_index) "
1023
+ "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
1024
+ (
1025
+ section_id,
1026
+ doc_id,
1027
+ section.heading_title,
1028
+ section.heading_level,
1029
+ section.heading_path,
1030
+ section.content,
1031
+ section.start_line,
1032
+ section.end_line,
1033
+ section.part_index,
1034
+ ),
1035
+ )
1036
+ if vector.section is not None:
1037
+ conn.execute(
1038
+ "INSERT INTO sections_vec(section_id, embedding) VALUES (?, ?)",
1039
+ (section_id, serialize_embedding(vector.section)),
1040
+ )
1041
+ for ordinal, (text, embedding) in enumerate(
1042
+ zip(section.units, vector.units, strict=True)
1043
+ ):
1044
+ unit_id = conn.execute(
1045
+ "INSERT INTO units(section_id, ordinal, content) VALUES (?, ?, ?)",
1046
+ (section_id, ordinal, text),
1047
+ ).lastrowid
1048
+ conn.execute(
1049
+ "INSERT INTO units_vec(unit_id, embedding) VALUES (?, ?)",
1050
+ (unit_id, serialize_embedding(embedding)),
1051
+ )
1052
+ conn.execute(
1053
+ "INSERT INTO meta(key, value) VALUES (?, ?) "
1054
+ "ON CONFLICT(key) DO UPDATE SET value = excluded.value",
1055
+ (_SECTION_ID_META_KEY, str(section_id + 1)),
1056
+ )
1057
+ # This write can remove the last vector in the index as well as add one: a
1058
+ # document whose prose became headings alone embeds nothing, and its old
1059
+ # vectors went with its old sections.
1060
+ _forget_weights_without_vectors(conn)
1061
+ return Document(
1062
+ id=doc_id,
1063
+ file_path=file_path,
1064
+ title=title,
1065
+ content_hash=content_hash,
1066
+ last_modified=last_modified,
1067
+ )
1068
+
1069
+ def _check_vector(self, embedding: Sequence[float], what: str) -> None:
1070
+ if len(embedding) != self._embedding_dim:
1071
+ raise DatabaseError(
1072
+ f"{what} has {len(embedding)} dimensions, expected {self._embedding_dim}"
1073
+ )
1074
+ if not _is_usable_vector(embedding):
1075
+ raise DatabaseError(f"{what} is all zeros or contains NaN/inf values")
1076
+
1077
+ def delete_documents(self, file_paths: Iterable[str]) -> int:
1078
+ """Delete documents by path; sections, FTS rows and vectors cascade. Returns count."""
1079
+ paths = list(file_paths)
1080
+ if not paths:
1081
+ return 0
1082
+ deleted = 0
1083
+ with self.transaction() as conn:
1084
+ for path in paths:
1085
+ deleted += conn.execute(
1086
+ "DELETE FROM documents WHERE file_path = ?", (path,)
1087
+ ).rowcount
1088
+ _forget_weights_without_vectors(conn)
1089
+ return deleted
1090
+
1091
+ def clear(self, notice: Callable[[int], str] | None = None) -> int:
1092
+ """Delete every document (and, by cascade, every section and vector).
1093
+
1094
+ ``notice`` words the message for the number of documents discarded; it is persisted
1095
+ in the same transaction, so the explanation cannot be lost while the data is.
1096
+ """
1097
+ with self.transaction() as conn:
1098
+ discarded = conn.execute("DELETE FROM documents").rowcount
1099
+ _forget_weights_without_vectors(conn)
1100
+ self.revoke_coverage(conn)
1101
+ if discarded and notice is not None:
1102
+ _add_notice(conn, notice(discarded))
1103
+ return discarded
1104
+
1105
+ # ------------------------------------------------------------------ sections
1106
+
1107
+ def record_weights_mismatch(self, message: str | None, *, replace: bool = True) -> None:
1108
+ """Remember (or clear) that the index and the loaded model disagree.
1109
+
1110
+ Persisted rather than held in memory: every `search_docs` and `list_documents`
1111
+ answer carries an `index_status`, and a fact this serious may not depend on
1112
+ which process, or which run, happens to have noticed it. ``replace=False`` keeps a
1113
+ message already recorded, in the same statement that would have written this one.
1114
+ """
1115
+ with self.transaction() as conn:
1116
+ if message is None:
1117
+ conn.execute("DELETE FROM meta WHERE key = ?", (WEIGHTS_MISMATCH_KEY,))
1118
+ else:
1119
+ conn.execute(
1120
+ "INSERT INTO meta (key, value) VALUES (?, ?) ON CONFLICT(key) "
1121
+ + ("DO UPDATE SET value = excluded.value" if replace else "DO NOTHING"),
1122
+ (WEIGHTS_MISMATCH_KEY, message),
1123
+ )
1124
+
1125
+ def revoke_weights(self, message: str) -> None:
1126
+ """Withdraw the index's revision before other weights write into it.
1127
+
1128
+ From here until `settle_weights` re-certifies it, the index holds - or may hold -
1129
+ vectors from two models, and no search may rank them against one query.
1130
+ """
1131
+ with self.transaction() as conn:
1132
+ _revoke_weights(conn, message)
1133
+
1134
+ def settle_weights(self, weights: str | None) -> None:
1135
+ """Close a run: vouch for the vectors again, or say what still stands in the way.
1136
+
1137
+ The revision may be written back only once every vector in the database - every
1138
+ root's, and rows no walk reaches, such as a document indexed inside `.venv` - came
1139
+ from ``weights``. Checking only what this run visited would re-certify an index
1140
+ still holding another model's vectors, which is the failure a certificate exists
1141
+ to rule out. An index holding no vector is never certified: there is nothing to
1142
+ vouch for, and a revision over nothing would turn the next model away. None means
1143
+ this run could not tell which weights it ran, and so vouches for nothing.
1144
+ """
1145
+ recorded = self.get_meta(WEIGHTS_META_KEY)
1146
+ mismatch = self.get_meta(WEIGHTS_MISMATCH_KEY)
1147
+ if (recorded, mismatch) != (None, None) and self.count_rows("units_vec") == 0:
1148
+ # A revision claimed for vectors that never arrived - the run died between
1149
+ # the claim and the write - describes nothing, whoever is asking.
1150
+ with self.transaction() as conn:
1151
+ _forget_weights_without_vectors(conn)
1152
+ return
1153
+ # Unknown weights vouch for nothing. Known and already certified, with nothing
1154
+ # said against it: the invariant holds by construction, so the whole-database
1155
+ # check would find nothing, and a no-op run need not take the write lock to learn
1156
+ # that.
1157
+ if weights is None or (recorded == weights and mismatch is None):
1158
+ return
1159
+ with self.transaction() as conn:
1160
+ if conn.execute("SELECT 1 FROM units_vec LIMIT 1").fetchone() is None:
1161
+ _forget_weights_without_vectors(conn)
1162
+ return
1163
+ stale = [
1164
+ str(row[0])
1165
+ for row in conn.execute(
1166
+ "SELECT d.file_path FROM documents AS d "
1167
+ "WHERE (d.weights_revision IS NULL OR d.weights_revision != ?) "
1168
+ "AND EXISTS (SELECT 1 FROM sections AS s JOIN units AS u "
1169
+ "ON u.section_id = s.id WHERE s.doc_id = d.id) "
1170
+ "ORDER BY d.file_path",
1171
+ (weights,),
1172
+ )
1173
+ ]
1174
+ if not stale:
1175
+ conn.execute(
1176
+ "INSERT INTO meta (key, value) VALUES (?, ?) "
1177
+ "ON CONFLICT(key) DO UPDATE SET value = excluded.value",
1178
+ (WEIGHTS_META_KEY, weights),
1179
+ )
1180
+ conn.execute("DELETE FROM meta WHERE key = ?", (WEIGHTS_MISMATCH_KEY,))
1181
+ return
1182
+ # Named by directory, because a row this run could not reach - one indexed
1183
+ # deliberately inside a pruned directory - is repaired only by indexing that
1184
+ # directory itself, and a count alone would not say where to point it.
1185
+ folders = sorted({os.path.dirname(path) for path in stale})
1186
+ shown = ", ".join(folders[:3]) + (
1187
+ f" and {len(folders) - 3} more" if len(folders) > 3 else ""
1188
+ )
1189
+ conn.execute(
1190
+ "INSERT INTO meta (key, value) VALUES (?, ?) "
1191
+ "ON CONFLICT(key) DO UPDATE SET value = excluded.value",
1192
+ (
1193
+ WEIGHTS_MISMATCH_KEY,
1194
+ f"{len(stale)} document(s) still hold vectors from other weights than "
1195
+ f"the ones answering now, under {shown}. Only keyword ranking is used "
1196
+ "until index_directory re-embeds them - run it on those directories.",
1197
+ ),
1198
+ )
1199
+
1200
+ def forget_weights_revision(self) -> None:
1201
+ """Drop the recorded weights revision: no documents, so nothing it can describe.
1202
+
1203
+ `clear()` does this in the same transaction as the delete. This exists for every
1204
+ other way the index empties - a purge of the last document, a rebuild for a new
1205
+ vector size, the v1 format discard - where the rows go without going through it.
1206
+ """
1207
+ with self.transaction() as conn:
1208
+ conn.execute("DELETE FROM meta WHERE key = ?", (WEIGHTS_META_KEY,))
1209
+
1210
+ def get_sections(self, doc_id: int) -> list[Section]:
1211
+ """Every section of a document in source order."""
1212
+ with self._reading() as conn:
1213
+ rows = conn.execute(
1214
+ f"SELECT {_SECTION_COLUMNS} FROM sections WHERE doc_id = ? ORDER BY id",
1215
+ (doc_id,),
1216
+ ).fetchall()
1217
+ return [_section_from_row(row) for row in rows]
1218
+
1219
+ def get_sections_with_documents(
1220
+ self, section_ids: Sequence[int]
1221
+ ) -> dict[int, tuple[Section, Document]]:
1222
+ """Hydrate section ids into ``(Section, Document)`` pairs."""
1223
+ hydrated: dict[int, tuple[Section, Document]] = {}
1224
+ columns = ", ".join(f"s.{name.strip()}" for name in _SECTION_COLUMNS.split(","))
1225
+ with self._reading() as conn:
1226
+ for start in range(0, len(section_ids), _SQL_VARIABLE_BATCH):
1227
+ batch = section_ids[start : start + _SQL_VARIABLE_BATCH]
1228
+ placeholders = ", ".join("?" for _ in batch)
1229
+ rows = conn.execute(
1230
+ f"SELECT {columns}, d.id, d.file_path, d.title, d.content_hash, "
1231
+ "d.last_modified FROM sections s JOIN documents d ON d.id = s.doc_id "
1232
+ f"WHERE s.id IN ({placeholders})",
1233
+ tuple(batch),
1234
+ ).fetchall()
1235
+ for row in rows:
1236
+ section = _section_from_row(row[:9])
1237
+ hydrated[section.id] = (section, _document_from_row(row[9:]))
1238
+ return hydrated
1239
+
1240
+ # ------------------------------------------------------------------ search primitives
1241
+
1242
+ def fts_search(self, match_query: str, limit: int, scope: str | None = None) -> list[int]:
1243
+ """Section ids matching an FTS5 query, best BM25 rank first.
1244
+
1245
+ ``scope`` restricts the search to documents under one directory. The predicate
1246
+ joins inside the query so the limit applies to what survives it: filtering a
1247
+ page afterwards would return fewer rows than asked for whenever a neighbouring
1248
+ documentation root in the same database ranks higher.
1249
+ """
1250
+ with self._reading() as conn:
1251
+ if scope is None:
1252
+ rows = conn.execute(
1253
+ "SELECT rowid FROM sections_fts WHERE sections_fts MATCH ? "
1254
+ "ORDER BY rank LIMIT ?",
1255
+ (match_query, limit),
1256
+ ).fetchall()
1257
+ else:
1258
+ prefix = _directory_prefix(scope)
1259
+ rows = conn.execute(
1260
+ "SELECT f.rowid FROM sections_fts f "
1261
+ "JOIN sections s ON s.id = f.rowid JOIN documents d ON d.id = s.doc_id "
1262
+ "WHERE sections_fts MATCH ? AND substr(d.file_path, 1, length(?)) = ? "
1263
+ "ORDER BY rank LIMIT ?",
1264
+ (match_query, prefix, prefix, limit),
1265
+ ).fetchall()
1266
+ return [int(row[0]) for row in rows]
1267
+
1268
+ def vec_search(self, embedding: Sequence[float], limit: int) -> list[tuple[int, float]]:
1269
+ """``(section_id, cosine_distance)`` for the nearest section vectors, closest first."""
1270
+ self._check_vector(embedding, "Query embedding")
1271
+ with self._reading() as conn:
1272
+ rows = conn.execute(
1273
+ "SELECT section_id, distance FROM sections_vec "
1274
+ "WHERE embedding MATCH ? AND k = ? ORDER BY distance",
1275
+ (serialize_embedding(embedding), limit),
1276
+ ).fetchall()
1277
+ return [
1278
+ (int(section_id), float(distance))
1279
+ for section_id, distance in rows
1280
+ if distance is not None # defensive: NaN distances surface as NULL
1281
+ ]
1282
+
1283
+ def unit_search(self, embedding: Sequence[float], limit: int) -> list[tuple[int, float, str]]:
1284
+ """``(section_id, cosine_distance, passage)`` for the nearest passages, closest first.
1285
+
1286
+ A section appears once per matching passage; callers keep its best one.
1287
+ """
1288
+ self._check_vector(embedding, "Query embedding")
1289
+ with self._reading() as conn:
1290
+ rows = conn.execute(
1291
+ "WITH nearest AS (SELECT unit_id, distance FROM units_vec "
1292
+ "WHERE embedding MATCH ? AND k = ?) "
1293
+ "SELECT u.section_id, nearest.distance, u.content FROM nearest "
1294
+ "JOIN units u ON u.id = nearest.unit_id ORDER BY nearest.distance",
1295
+ (serialize_embedding(embedding), limit),
1296
+ ).fetchall()
1297
+ return [
1298
+ (int(section_id), float(distance), str(content))
1299
+ for section_id, distance, content in rows
1300
+ if distance is not None
1301
+ ]
1302
+
1303
+ def fts_matching(self, term: str, within: Sequence[int]) -> set[int]:
1304
+ """Which of the sections ``within`` match the single FTS5 ``term``."""
1305
+ if not within:
1306
+ return set()
1307
+ placeholders = ", ".join("?" for _ in within)
1308
+ with self._reading() as conn:
1309
+ rows = conn.execute(
1310
+ "SELECT rowid FROM sections_fts WHERE sections_fts MATCH ? "
1311
+ f"AND rowid IN ({placeholders})",
1312
+ (term, *within),
1313
+ ).fetchall()
1314
+ return {int(row[0]) for row in rows}
1315
+
1316
+ def fts_document_frequency(self, term: str) -> int:
1317
+ """Number of sections matching the single FTS5 ``term``."""
1318
+ with self._reading() as conn:
1319
+ row = conn.execute(
1320
+ "SELECT COUNT(*) FROM sections_fts WHERE sections_fts MATCH ?", (term,)
1321
+ ).fetchone()
1322
+ return int(row[0])
1323
+
1324
+ def sections_with_passages(self, section_ids: Sequence[int]) -> set[int]:
1325
+ """The subset of ``section_ids`` that has a body (heading-only sections have none)."""
1326
+ if not section_ids:
1327
+ return set()
1328
+ placeholders = ", ".join("?" for _ in section_ids)
1329
+ with self._reading() as conn:
1330
+ rows = conn.execute(
1331
+ f"SELECT DISTINCT section_id FROM units WHERE section_id IN ({placeholders})",
1332
+ tuple(section_ids),
1333
+ ).fetchall()
1334
+ return {int(row[0]) for row in rows}
1335
+
1336
+ def sections_under(self, section_ids: Sequence[int], directory: str) -> set[int]:
1337
+ """The subset of ``section_ids`` whose document lives under ``directory``.
1338
+
1339
+ Search is scoped with this rather than with a predicate inside the FTS5 and
1340
+ vec0 queries: both apply their own limit before any join would filter, so a
1341
+ scoped predicate there silently returns fewer results than asked for.
1342
+ """
1343
+ if not section_ids:
1344
+ return set()
1345
+ prefix = _directory_prefix(directory)
1346
+ placeholders = ", ".join("?" for _ in section_ids)
1347
+ with self._reading() as conn:
1348
+ rows = conn.execute(
1349
+ f"SELECT s.id FROM sections s JOIN documents d ON d.id = s.doc_id "
1350
+ f"WHERE s.id IN ({placeholders}) "
1351
+ "AND substr(d.file_path, 1, length(?)) = ?",
1352
+ (*section_ids, prefix, prefix),
1353
+ ).fetchall()
1354
+ return {int(row[0]) for row in rows}
1355
+
1356
+ def integrity_problems(self) -> list[str]:
1357
+ """Everything that is wrong with the store; an empty list means it is sound.
1358
+
1359
+ Covers SQLite's own page and foreign-key checks, the FTS5 index against the
1360
+ ``sections`` table, and the invariants the triggers exist to uphold: one FTS row
1361
+ per section, one vector per passage, a section vector exactly for the sections
1362
+ that have passages, and every stored vector of the configured dimension.
1363
+ """
1364
+ problems: list[str] = []
1365
+ blob_bytes = self._embedding_dim * 4
1366
+ conn = self.connection()
1367
+ try:
1368
+ # rank = 1 makes FTS5 compare the index with the external content table; the
1369
+ # plain form only checks the index's internal consistency. The command is an
1370
+ # INSERT, so it needs the write lock - failing to get it proves nothing.
1371
+ conn.execute(
1372
+ "INSERT INTO sections_fts(sections_fts, rank) VALUES ('integrity-check', 1)"
1373
+ )
1374
+ except sqlite3.Error as exc:
1375
+ if _is_lock_error(exc):
1376
+ problems.append(
1377
+ "could not verify the FTS5 index: the database is locked by another "
1378
+ "writer (not a sign of damage - retry when indexing has finished)"
1379
+ )
1380
+ else:
1381
+ problems.append(f"FTS5 index does not match the sections table: {exc}")
1382
+ try:
1383
+ conn.execute("BEGIN") # one read snapshot: counts taken mid-write would disagree
1384
+ try:
1385
+ version = int(conn.execute("PRAGMA user_version").fetchone()[0])
1386
+ pages = [str(row[0]) for row in conn.execute("PRAGMA integrity_check")]
1387
+ orphans = conn.execute("PRAGMA foreign_key_check").fetchall()
1388
+ counts = {
1389
+ table: int(conn.execute(f"SELECT COUNT(*) FROM {source}").fetchone()[0])
1390
+ for table, source in (
1391
+ ("sections", "sections"),
1392
+ ("sections_fts", "sections_fts_docsize"),
1393
+ ("sections_vec", "sections_vec"),
1394
+ ("units", "units"),
1395
+ ("units_vec", "units_vec"),
1396
+ )
1397
+ }
1398
+ with_passages = int(
1399
+ conn.execute("SELECT COUNT(DISTINCT section_id) FROM units").fetchone()[0]
1400
+ )
1401
+ wrong_size = sum(
1402
+ int(
1403
+ conn.execute(
1404
+ f"SELECT COUNT(*) FROM {table} WHERE length(embedding) != ?",
1405
+ (blob_bytes,),
1406
+ ).fetchone()[0]
1407
+ )
1408
+ for table in ("sections_vec", "units_vec")
1409
+ )
1410
+ row = conn.execute("SELECT value FROM meta WHERE key = 'embedding_dim'").fetchone()
1411
+ stored_dim = None if row is None else str(row[0])
1412
+ finally:
1413
+ conn.execute("ROLLBACK")
1414
+ except sqlite3.Error as exc:
1415
+ raise DatabaseError(f"Integrity check could not read the database: {exc}") from exc
1416
+
1417
+ if version != SCHEMA_VERSION:
1418
+ problems.append(f"schema version is {version}, expected {SCHEMA_VERSION}")
1419
+ if pages != ["ok"]:
1420
+ problems.append("PRAGMA integrity_check: " + "; ".join(pages[:5]))
1421
+ if orphans:
1422
+ problems.append(f"PRAGMA foreign_key_check: {len(orphans)} orphaned row(s)")
1423
+ if counts["sections"] != counts["sections_fts"]:
1424
+ problems.append(f"{counts['sections']} sections but {counts['sections_fts']} FTS rows")
1425
+ if counts["units"] != counts["units_vec"]:
1426
+ problems.append(f"{counts['units']} passages but {counts['units_vec']} passage vectors")
1427
+ if counts["sections_vec"] != with_passages:
1428
+ problems.append(
1429
+ f"{counts['sections_vec']} section vectors but "
1430
+ f"{with_passages} sections with passages"
1431
+ )
1432
+ if wrong_size:
1433
+ problems.append(
1434
+ f"{wrong_size} stored vector(s) are not {self._embedding_dim}-dimensional"
1435
+ )
1436
+ if stored_dim != str(self._embedding_dim):
1437
+ problems.append(f"meta embedding_dim is {stored_dim}, expected {self._embedding_dim}")
1438
+ return problems
1439
+
1440
+ def count_rows(self, table: str) -> int:
1441
+ """Row count of one of the known tables (diagnostics and integrity tests)."""
1442
+ if table not in {
1443
+ "documents", "sections", "sections_fts", "sections_vec", "units", "units_vec",
1444
+ "index_failures", "index_coverage",
1445
+ }: # fmt: skip
1446
+ raise DatabaseError(f"Unknown table: {table!r}")
1447
+ # COUNT(*) on an external-content FTS5 table is answered from `sections`; the
1448
+ # docsize shadow table has one row per entry actually present in the index.
1449
+ source = "sections_fts_docsize" if table == "sections_fts" else table
1450
+ with self._reading() as conn:
1451
+ row = conn.execute(f"SELECT COUNT(*) FROM {source}").fetchone()
1452
+ return int(row[0])
1453
+
1454
+
1455
+ def _next_section_id(conn: sqlite3.Connection) -> int:
1456
+ """A section id that was never used before, not even by a row deleted since.
1457
+
1458
+ SQLite hands the rowid of a deleted row out again. A search ranks ids on one connection
1459
+ and fetches them on another, so a re-index in between must make the old ids *vanish*
1460
+ (the search then ranks again) rather than point at whatever section was stored next.
1461
+ """
1462
+ stored = conn.execute("SELECT value FROM meta WHERE key = ?", (_SECTION_ID_META_KEY,))
1463
+ row = stored.fetchone()
1464
+ highest = conn.execute("SELECT COALESCE(MAX(id), 0) FROM sections").fetchone()[0]
1465
+ return max(int(row[0]) if row is not None else 1, int(highest) + 1)
1466
+
1467
+
1468
+ def _add_notice(conn: sqlite3.Connection, message: str) -> None:
1469
+ logger.warning(message)
1470
+ # Numbered after the newest, not by count: notices are dismissed one by one. The key is
1471
+ # TEXT, so the successor is taken numerically - 'notice:10000' sorts below 'notice:9999'.
1472
+ newest = conn.execute(
1473
+ "SELECT MAX(CAST(substr(key, 8) AS INTEGER)) FROM meta WHERE key LIKE 'notice:%'"
1474
+ ).fetchone()[0]
1475
+ number = 0 if newest is None else int(newest) + 1
1476
+ conn.execute("INSERT INTO meta(key, value) VALUES (?, ?)", (f"notice:{number:04d}", message))
1477
+
1478
+
1479
+ def _close_quietly(conn: sqlite3.Connection) -> None:
1480
+ try:
1481
+ conn.close()
1482
+ except sqlite3.Error: # pragma: no cover - best effort
1483
+ logger.warning("Failed to close a SQLite connection", exc_info=True)
1484
+
1485
+
1486
+ def _is_lock_error(exc: sqlite3.Error) -> bool:
1487
+ message = str(exc).lower()
1488
+ return "locked" in message or "busy" in message
1489
+
1490
+
1491
+ def _enable_wal(conn: sqlite3.Connection) -> None:
1492
+ """Switch to WAL, retrying while another process holds the lock.
1493
+
1494
+ Changing the journal mode needs an exclusive lock and SQLite reports contention
1495
+ on it immediately instead of honouring the busy timeout.
1496
+ """
1497
+ for attempt in range(_WAL_ATTEMPTS):
1498
+ try:
1499
+ conn.execute("PRAGMA journal_mode = WAL")
1500
+ return
1501
+ except sqlite3.OperationalError as exc:
1502
+ if attempt == _WAL_ATTEMPTS - 1 or not _is_lock_error(exc):
1503
+ raise
1504
+ time.sleep(_WAL_RETRY_SECONDS * (attempt + 1))
1505
+
1506
+
1507
+ def _document_from_row(row: Sequence[object]) -> Document:
1508
+ doc_id, file_path, title, content_hash, last_modified = row
1509
+ return Document(
1510
+ id=_as_int(doc_id),
1511
+ file_path=str(file_path),
1512
+ title=str(title),
1513
+ content_hash=str(content_hash),
1514
+ last_modified=_as_int(last_modified),
1515
+ )
1516
+
1517
+
1518
+ def _section_from_row(row: Sequence[object]) -> Section:
1519
+ (
1520
+ section_id,
1521
+ doc_id,
1522
+ heading_title,
1523
+ heading_level,
1524
+ heading_path,
1525
+ content,
1526
+ start_line,
1527
+ end_line,
1528
+ part_index,
1529
+ ) = row
1530
+ return Section(
1531
+ id=_as_int(section_id),
1532
+ doc_id=_as_int(doc_id),
1533
+ heading_title=str(heading_title),
1534
+ heading_level=_as_int(heading_level),
1535
+ heading_path=str(heading_path),
1536
+ content=str(content),
1537
+ start_line=_as_int(start_line),
1538
+ end_line=_as_int(end_line),
1539
+ part_index=_as_int(part_index),
1540
+ )
1541
+
1542
+
1543
+ def _as_int(value: object) -> int:
1544
+ if isinstance(value, int):
1545
+ return value
1546
+ raise DatabaseError(f"Expected an integer column value, got {type(value).__name__}")