markdown-memory 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- markdown_memory/__init__.py +47 -0
- markdown_memory/autoindex.py +170 -0
- markdown_memory/config.py +235 -0
- markdown_memory/db.py +1546 -0
- markdown_memory/discovery.py +195 -0
- markdown_memory/embedders.py +513 -0
- markdown_memory/exceptions.py +71 -0
- markdown_memory/freshness.py +138 -0
- markdown_memory/headings.py +185 -0
- markdown_memory/indexer.py +725 -0
- markdown_memory/model_cache.py +272 -0
- markdown_memory/models.py +339 -0
- markdown_memory/parser.py +869 -0
- markdown_memory/py.typed +0 -0
- markdown_memory/search.py +518 -0
- markdown_memory/server.py +520 -0
- markdown_memory-0.1.0.dist-info/METADATA +579 -0
- markdown_memory-0.1.0.dist-info/RECORD +21 -0
- markdown_memory-0.1.0.dist-info/WHEEL +4 -0
- markdown_memory-0.1.0.dist-info/entry_points.txt +3 -0
- markdown_memory-0.1.0.dist-info/licenses/LICENSE +21 -0
markdown_memory/db.py
ADDED
|
@@ -0,0 +1,1546 @@
|
|
|
1
|
+
"""SQLite storage: connection factory, migrations, and the section repository.
|
|
2
|
+
|
|
3
|
+
The store combines three indexes over the same ``sections`` rows:
|
|
4
|
+
|
|
5
|
+
* ``sections`` - canonical relational data (cascade-deleted with its document)
|
|
6
|
+
* ``sections_fts`` - FTS5 external-content index (BM25 keyword search)
|
|
7
|
+
* ``sections_vec`` - sqlite-vec ``vec0`` index, one vector per section with a body
|
|
8
|
+
* ``units`` / ``units_vec`` - the section's passages (paragraph, list item, table row,
|
|
9
|
+
code block) and one vector for each; a section is ranked by its best passage
|
|
10
|
+
|
|
11
|
+
Triggers keep both virtual tables in lock-step with ``sections``, including rows
|
|
12
|
+
removed by ``ON DELETE CASCADE``, so callers only ever write to ``sections``.
|
|
13
|
+
|
|
14
|
+
Connections are per-thread (MCP tool handlers run in worker threads and hybrid
|
|
15
|
+
search queries both indexes concurrently); WAL mode lets readers proceed while
|
|
16
|
+
an indexing transaction is open.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import logging
|
|
22
|
+
import math
|
|
23
|
+
import os
|
|
24
|
+
import sqlite3
|
|
25
|
+
import struct
|
|
26
|
+
import threading
|
|
27
|
+
import time
|
|
28
|
+
from collections.abc import Callable, Iterable, Iterator, Mapping, Sequence
|
|
29
|
+
from contextlib import contextmanager
|
|
30
|
+
from pathlib import Path
|
|
31
|
+
from types import TracebackType
|
|
32
|
+
from typing import Self
|
|
33
|
+
|
|
34
|
+
import sqlite_vec
|
|
35
|
+
|
|
36
|
+
from markdown_memory.exceptions import DatabaseError
|
|
37
|
+
from markdown_memory.models import (
|
|
38
|
+
Document,
|
|
39
|
+
DocumentSummary,
|
|
40
|
+
FileFailure,
|
|
41
|
+
IndexStatus,
|
|
42
|
+
Section,
|
|
43
|
+
SectionDraft,
|
|
44
|
+
SectionVectors,
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
logger = logging.getLogger(__name__)
|
|
48
|
+
|
|
49
|
+
DEFAULT_EMBEDDING_DIM = 384
|
|
50
|
+
SCHEMA_VERSION = 6
|
|
51
|
+
|
|
52
|
+
#: How a section's vector is built. 1 embedded the whole section text, truncated at the
|
|
53
|
+
#: model's token limit; 2 is the mean of the section's passage vectors. Stored per document
|
|
54
|
+
#: so that a document written under the old scheme re-indexes itself and one written under
|
|
55
|
+
#: the new one is left alone - a format change repairs a tree file by file, and resumes
|
|
56
|
+
#: where it stopped if it is interrupted.
|
|
57
|
+
VECTOR_FORMAT = 2
|
|
58
|
+
#: Meta key holding the weights revision the stored vectors were built from. It lives here
|
|
59
|
+
#: because `clear()` has to forget it in the same transaction that deletes them.
|
|
60
|
+
WEIGHTS_META_KEY = "embedding_weights_revision"
|
|
61
|
+
#: Set when the weights behind an unchanged model name changed under an existing index.
|
|
62
|
+
#: It holds the sentence an agent is shown, because the index is then answering from
|
|
63
|
+
#: vectors one model built while the next query would be embedded by another.
|
|
64
|
+
WEIGHTS_MISMATCH_KEY = "embedding_weights_mismatch"
|
|
65
|
+
#: What `WEIGHTS_META_KEY` holds while an index is being re-embedded by other weights. It
|
|
66
|
+
#: equals no revision, so every search - under the old weights or the new - finds it
|
|
67
|
+
#: differs from its own and ranks by keyword alone until a run re-certifies the index. A
|
|
68
|
+
#: revision is a hash, sometimes with a graph path after it; this can be neither.
|
|
69
|
+
WEIGHTS_REVOKED = "(revoked: being re-embedded)"
|
|
70
|
+
_LEGACY_VECTORS = 1
|
|
71
|
+
|
|
72
|
+
_SECTION_ID_META_KEY = "next_section_id"
|
|
73
|
+
_GENERATION_META_KEY = "index_generation"
|
|
74
|
+
_BUSY_TIMEOUT_MS = 10_000
|
|
75
|
+
_WAL_ATTEMPTS = 40
|
|
76
|
+
_WAL_RETRY_SECONDS = 0.05
|
|
77
|
+
_SQL_VARIABLE_BATCH = 500
|
|
78
|
+
|
|
79
|
+
_SECTION_COLUMNS = (
|
|
80
|
+
"id, doc_id, heading_title, heading_level, heading_path, content, "
|
|
81
|
+
"start_line, end_line, part_index"
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _schema_v1(embedding_dim: int) -> tuple[str, ...]:
|
|
86
|
+
return (
|
|
87
|
+
"""
|
|
88
|
+
CREATE TABLE meta (
|
|
89
|
+
key TEXT PRIMARY KEY,
|
|
90
|
+
value TEXT NOT NULL
|
|
91
|
+
)
|
|
92
|
+
""",
|
|
93
|
+
"""
|
|
94
|
+
CREATE TABLE documents (
|
|
95
|
+
id INTEGER PRIMARY KEY,
|
|
96
|
+
file_path TEXT NOT NULL UNIQUE,
|
|
97
|
+
title TEXT NOT NULL,
|
|
98
|
+
content_hash TEXT NOT NULL,
|
|
99
|
+
last_modified INTEGER NOT NULL
|
|
100
|
+
)
|
|
101
|
+
""",
|
|
102
|
+
"""
|
|
103
|
+
CREATE TABLE sections (
|
|
104
|
+
id INTEGER PRIMARY KEY,
|
|
105
|
+
doc_id INTEGER NOT NULL REFERENCES documents(id) ON DELETE CASCADE,
|
|
106
|
+
heading_title TEXT NOT NULL,
|
|
107
|
+
heading_level INTEGER NOT NULL,
|
|
108
|
+
heading_path TEXT NOT NULL,
|
|
109
|
+
content TEXT NOT NULL,
|
|
110
|
+
start_line INTEGER NOT NULL,
|
|
111
|
+
end_line INTEGER NOT NULL,
|
|
112
|
+
part_index INTEGER NOT NULL DEFAULT 0
|
|
113
|
+
)
|
|
114
|
+
""",
|
|
115
|
+
"CREATE INDEX idx_sections_doc ON sections(doc_id, id)",
|
|
116
|
+
"""
|
|
117
|
+
CREATE VIRTUAL TABLE sections_fts USING fts5(
|
|
118
|
+
heading_title,
|
|
119
|
+
heading_path,
|
|
120
|
+
content,
|
|
121
|
+
content='sections',
|
|
122
|
+
content_rowid='id',
|
|
123
|
+
tokenize='porter unicode61 remove_diacritics 2'
|
|
124
|
+
)
|
|
125
|
+
""",
|
|
126
|
+
# Headings are short and highly descriptive: weight them above body text.
|
|
127
|
+
"INSERT INTO sections_fts(sections_fts, rank) VALUES ('rank', 'bm25(5.0, 3.0, 1.0)')",
|
|
128
|
+
_vector_tables(embedding_dim)[0],
|
|
129
|
+
"""
|
|
130
|
+
CREATE TRIGGER sections_after_insert AFTER INSERT ON sections BEGIN
|
|
131
|
+
INSERT INTO sections_fts(rowid, heading_title, heading_path, content)
|
|
132
|
+
VALUES (new.id, new.heading_title, new.heading_path, new.content);
|
|
133
|
+
END
|
|
134
|
+
""",
|
|
135
|
+
"""
|
|
136
|
+
CREATE TRIGGER sections_after_delete AFTER DELETE ON sections BEGIN
|
|
137
|
+
INSERT INTO sections_fts(sections_fts, rowid, heading_title, heading_path, content)
|
|
138
|
+
VALUES ('delete', old.id, old.heading_title, old.heading_path, old.content);
|
|
139
|
+
DELETE FROM sections_vec WHERE section_id = old.id;
|
|
140
|
+
END
|
|
141
|
+
""",
|
|
142
|
+
"""
|
|
143
|
+
CREATE TRIGGER sections_after_update AFTER UPDATE ON sections BEGIN
|
|
144
|
+
INSERT INTO sections_fts(sections_fts, rowid, heading_title, heading_path, content)
|
|
145
|
+
VALUES ('delete', old.id, old.heading_title, old.heading_path, old.content);
|
|
146
|
+
INSERT INTO sections_fts(rowid, heading_title, heading_path, content)
|
|
147
|
+
VALUES (new.id, new.heading_title, new.heading_path, new.content);
|
|
148
|
+
END
|
|
149
|
+
""",
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _vector_tables(embedding_dim: int) -> tuple[str, ...]:
|
|
154
|
+
return (
|
|
155
|
+
f"""
|
|
156
|
+
CREATE VIRTUAL TABLE sections_vec USING vec0(
|
|
157
|
+
section_id INTEGER PRIMARY KEY,
|
|
158
|
+
embedding FLOAT[{embedding_dim}] distance_metric=cosine
|
|
159
|
+
)
|
|
160
|
+
""",
|
|
161
|
+
f"""
|
|
162
|
+
CREATE VIRTUAL TABLE units_vec USING vec0(
|
|
163
|
+
unit_id INTEGER PRIMARY KEY,
|
|
164
|
+
embedding FLOAT[{embedding_dim}] distance_metric=cosine
|
|
165
|
+
)
|
|
166
|
+
""",
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _schema_v2(embedding_dim: int) -> tuple[str, ...]:
|
|
171
|
+
"""Passage-level vectors: ``units`` rows cascade with their section."""
|
|
172
|
+
return (
|
|
173
|
+
"""
|
|
174
|
+
CREATE TABLE units (
|
|
175
|
+
id INTEGER PRIMARY KEY,
|
|
176
|
+
section_id INTEGER NOT NULL REFERENCES sections(id) ON DELETE CASCADE,
|
|
177
|
+
ordinal INTEGER NOT NULL,
|
|
178
|
+
content TEXT NOT NULL
|
|
179
|
+
)
|
|
180
|
+
""",
|
|
181
|
+
"CREATE INDEX idx_units_section ON units(section_id, ordinal)",
|
|
182
|
+
_vector_tables(embedding_dim)[1],
|
|
183
|
+
"""
|
|
184
|
+
CREATE TRIGGER units_after_delete AFTER DELETE ON units BEGIN
|
|
185
|
+
DELETE FROM units_vec WHERE unit_id = old.id;
|
|
186
|
+
END
|
|
187
|
+
""",
|
|
188
|
+
)
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _schema_v4() -> tuple[str, ...]:
|
|
192
|
+
"""What is known to be wrong with the index, and whether anything vouches for it.
|
|
193
|
+
|
|
194
|
+
Two questions, kept apart because they have different answers. `index_failures` says
|
|
195
|
+
which paths could not be read, one row per path, written by whichever scan last looked
|
|
196
|
+
at that path. `index_coverage` says whether a full walk of a root ever finished without
|
|
197
|
+
failures - the only thing that can distinguish "every file was seen" from "only these
|
|
198
|
+
files were seen", which no amount of per-file state can tell you: a run killed on its
|
|
199
|
+
tenth file leaves the other 990 with no rows at all.
|
|
200
|
+
|
|
201
|
+
Earlier attempts hashed the root (so containment could not be asked), then kept a
|
|
202
|
+
wall-clock guard (wrong in both orderings), then a crash marker that masked the file
|
|
203
|
+
rows. Those are gone. What makes this sound instead is the scan lock: one scan at a
|
|
204
|
+
time, so the run that just walked a tree is entitled to speak for it.
|
|
205
|
+
"""
|
|
206
|
+
return (
|
|
207
|
+
"""
|
|
208
|
+
CREATE TABLE index_failures (
|
|
209
|
+
file_path TEXT PRIMARY KEY,
|
|
210
|
+
message TEXT NOT NULL
|
|
211
|
+
)
|
|
212
|
+
""",
|
|
213
|
+
"""
|
|
214
|
+
CREATE TABLE index_coverage (
|
|
215
|
+
root TEXT PRIMARY KEY,
|
|
216
|
+
verified INTEGER NOT NULL
|
|
217
|
+
)
|
|
218
|
+
""",
|
|
219
|
+
)
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def serialize_embedding(embedding: Sequence[float]) -> bytes:
|
|
223
|
+
"""Pack a vector into the little-endian float32 blob format sqlite-vec expects."""
|
|
224
|
+
return struct.pack(f"<{len(embedding)}f", *embedding)
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _is_usable_vector(embedding: Sequence[float]) -> bool:
|
|
228
|
+
"""Cosine distance is undefined (NaN) for non-finite or zero-length vectors."""
|
|
229
|
+
return all(math.isfinite(value) for value in embedding) and any(embedding)
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _bump_generation(conn: sqlite3.Connection) -> None:
|
|
233
|
+
"""Mark that everything indexed before this moment is gone."""
|
|
234
|
+
conn.execute(
|
|
235
|
+
"INSERT INTO meta(key, value) VALUES (?, '1') "
|
|
236
|
+
"ON CONFLICT(key) DO UPDATE SET value = CAST(CAST(value AS INTEGER) + 1 AS TEXT)",
|
|
237
|
+
(_GENERATION_META_KEY,),
|
|
238
|
+
)
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _forget_weights_without_vectors(conn: sqlite3.Connection) -> None:
|
|
242
|
+
"""Drop the weights metadata once the vectors it describes are gone.
|
|
243
|
+
|
|
244
|
+
Which weights produced the vectors is a fact about the vectors, so it belongs to the
|
|
245
|
+
same transaction that removes them - a purge of the last document, a rebuild for a new
|
|
246
|
+
vector size, a discard for a changed model. Kept behind, it would have the next run
|
|
247
|
+
compare a new model against the revision of a model whose output no longer exists, and
|
|
248
|
+
have search rank on keywords alone over an index with nothing wrong with it.
|
|
249
|
+
|
|
250
|
+
Heading-only documents hold no vectors, so the test is the vectors themselves rather
|
|
251
|
+
than the document rows above them.
|
|
252
|
+
"""
|
|
253
|
+
if conn.execute("SELECT 1 FROM units_vec LIMIT 1").fetchone() is not None:
|
|
254
|
+
return
|
|
255
|
+
conn.execute("DELETE FROM meta WHERE key = ?", (WEIGHTS_META_KEY,))
|
|
256
|
+
conn.execute("DELETE FROM meta WHERE key = ?", (WEIGHTS_MISMATCH_KEY,))
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _revoke_weights(conn: sqlite3.Connection, message: str) -> None:
|
|
260
|
+
"""Mark the index as being re-embedded, and say why, in the caller's transaction.
|
|
261
|
+
|
|
262
|
+
One transaction for both: a revocation with no message would leave `index_status`
|
|
263
|
+
calling the index healthy while search refuses its vectors, and a message with no
|
|
264
|
+
revocation would let search rank the mixture the message warns about.
|
|
265
|
+
"""
|
|
266
|
+
conn.executemany(
|
|
267
|
+
"INSERT INTO meta (key, value) VALUES (?, ?) "
|
|
268
|
+
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
|
|
269
|
+
((WEIGHTS_META_KEY, WEIGHTS_REVOKED), (WEIGHTS_MISMATCH_KEY, message)),
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _directory_prefix(directory: str) -> str:
|
|
274
|
+
return directory if directory.endswith(os.sep) else directory + os.sep
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
class Database:
|
|
278
|
+
"""Thread-safe SQLite client and repository for documents and sections.
|
|
279
|
+
|
|
280
|
+
Use as a context manager to guarantee every per-thread connection is closed::
|
|
281
|
+
|
|
282
|
+
with Database(path) as db:
|
|
283
|
+
with db.transaction() as conn:
|
|
284
|
+
...
|
|
285
|
+
"""
|
|
286
|
+
|
|
287
|
+
def __init__(self, path: str | Path, *, embedding_dim: int = DEFAULT_EMBEDDING_DIM) -> None:
|
|
288
|
+
if str(path) == ":memory:" or str(path).startswith("file::memory:"):
|
|
289
|
+
raise DatabaseError(
|
|
290
|
+
"In-memory databases are not supported: connections are per-thread "
|
|
291
|
+
"and would each see an empty database. Use a file path."
|
|
292
|
+
)
|
|
293
|
+
if embedding_dim <= 0:
|
|
294
|
+
raise DatabaseError(f"embedding_dim must be positive, got {embedding_dim}")
|
|
295
|
+
self._path = Path(path).expanduser()
|
|
296
|
+
self._embedding_dim = embedding_dim
|
|
297
|
+
self._local = threading.local()
|
|
298
|
+
self._connections: list[tuple[threading.Thread, sqlite3.Connection]] = []
|
|
299
|
+
self._lock = threading.Lock()
|
|
300
|
+
self._closed = False
|
|
301
|
+
try:
|
|
302
|
+
self._path.parent.mkdir(parents=True, exist_ok=True)
|
|
303
|
+
except OSError as exc:
|
|
304
|
+
raise DatabaseError(f"Cannot create database directory {self._path.parent}") from exc
|
|
305
|
+
self._migrate()
|
|
306
|
+
|
|
307
|
+
# ------------------------------------------------------------------ lifecycle
|
|
308
|
+
|
|
309
|
+
def __enter__(self) -> Self:
|
|
310
|
+
return self
|
|
311
|
+
|
|
312
|
+
def __exit__(
|
|
313
|
+
self,
|
|
314
|
+
exc_type: type[BaseException] | None,
|
|
315
|
+
exc: BaseException | None,
|
|
316
|
+
traceback: TracebackType | None,
|
|
317
|
+
) -> None:
|
|
318
|
+
self.close()
|
|
319
|
+
|
|
320
|
+
@property
|
|
321
|
+
def path(self) -> Path:
|
|
322
|
+
return self._path
|
|
323
|
+
|
|
324
|
+
@property
|
|
325
|
+
def embedding_dim(self) -> int:
|
|
326
|
+
return self._embedding_dim
|
|
327
|
+
|
|
328
|
+
def close(self) -> None:
|
|
329
|
+
"""Close every connection opened by any thread. Idempotent."""
|
|
330
|
+
with self._lock:
|
|
331
|
+
self._closed = True
|
|
332
|
+
connections, self._connections = self._connections, []
|
|
333
|
+
for _, conn in connections:
|
|
334
|
+
_close_quietly(conn)
|
|
335
|
+
|
|
336
|
+
@property
|
|
337
|
+
def open_connection_count(self) -> int:
|
|
338
|
+
"""Connections currently held open across all threads (diagnostics)."""
|
|
339
|
+
with self._lock:
|
|
340
|
+
return len(self._connections)
|
|
341
|
+
|
|
342
|
+
def connection(self) -> sqlite3.Connection:
|
|
343
|
+
"""Return this thread's connection, opening and configuring it on first use.
|
|
344
|
+
|
|
345
|
+
Tool handlers run on pooled worker threads that the runtime retires when idle,
|
|
346
|
+
so opening a connection is also the moment the connections of threads that have
|
|
347
|
+
since exited are closed; otherwise a long-lived server leaks file descriptors.
|
|
348
|
+
"""
|
|
349
|
+
if self._closed:
|
|
350
|
+
raise DatabaseError("Database is closed")
|
|
351
|
+
existing: sqlite3.Connection | None = getattr(self._local, "conn", None)
|
|
352
|
+
if existing is not None:
|
|
353
|
+
return existing
|
|
354
|
+
conn = self._open()
|
|
355
|
+
abandoned: list[sqlite3.Connection] = []
|
|
356
|
+
with self._lock:
|
|
357
|
+
if self._closed:
|
|
358
|
+
conn.close()
|
|
359
|
+
raise DatabaseError("Database is closed")
|
|
360
|
+
alive: list[tuple[threading.Thread, sqlite3.Connection]] = []
|
|
361
|
+
for thread, other in self._connections:
|
|
362
|
+
if thread.is_alive():
|
|
363
|
+
alive.append((thread, other))
|
|
364
|
+
else:
|
|
365
|
+
abandoned.append(other)
|
|
366
|
+
alive.append((threading.current_thread(), conn))
|
|
367
|
+
self._connections = alive
|
|
368
|
+
self._local.conn = conn
|
|
369
|
+
for other in abandoned:
|
|
370
|
+
_close_quietly(other)
|
|
371
|
+
return conn
|
|
372
|
+
|
|
373
|
+
def _open(self) -> sqlite3.Connection:
|
|
374
|
+
try:
|
|
375
|
+
# isolation_level=None: autocommit; transactions are explicit (see transaction()).
|
|
376
|
+
# check_same_thread=False only so close() may run from another thread;
|
|
377
|
+
# each connection is otherwise used exclusively by the thread that opened it.
|
|
378
|
+
conn = sqlite3.connect(
|
|
379
|
+
self._path,
|
|
380
|
+
isolation_level=None,
|
|
381
|
+
check_same_thread=False,
|
|
382
|
+
timeout=_BUSY_TIMEOUT_MS / 1000,
|
|
383
|
+
)
|
|
384
|
+
except sqlite3.Error as exc:
|
|
385
|
+
raise DatabaseError(f"Cannot open database at {self._path}: {exc}") from exc
|
|
386
|
+
try:
|
|
387
|
+
conn.enable_load_extension(True)
|
|
388
|
+
sqlite_vec.load(conn)
|
|
389
|
+
conn.enable_load_extension(False)
|
|
390
|
+
except (sqlite3.Error, AttributeError) as exc:
|
|
391
|
+
conn.close()
|
|
392
|
+
raise DatabaseError(
|
|
393
|
+
"Cannot load the sqlite-vec extension. This Python build's sqlite3 module "
|
|
394
|
+
f"must support loadable extensions: {exc}"
|
|
395
|
+
) from exc
|
|
396
|
+
try:
|
|
397
|
+
conn.execute(f"PRAGMA busy_timeout = {_BUSY_TIMEOUT_MS}")
|
|
398
|
+
_enable_wal(conn)
|
|
399
|
+
conn.execute("PRAGMA synchronous = NORMAL")
|
|
400
|
+
conn.execute("PRAGMA foreign_keys = ON")
|
|
401
|
+
except sqlite3.Error as exc:
|
|
402
|
+
conn.close()
|
|
403
|
+
raise DatabaseError(f"Cannot configure database connection: {exc}") from exc
|
|
404
|
+
return conn
|
|
405
|
+
|
|
406
|
+
@contextmanager
|
|
407
|
+
def transaction(self) -> Iterator[sqlite3.Connection]:
|
|
408
|
+
"""Run a write transaction: commit on success, roll back on any exception."""
|
|
409
|
+
conn = self.connection()
|
|
410
|
+
try:
|
|
411
|
+
conn.execute("BEGIN IMMEDIATE")
|
|
412
|
+
except sqlite3.Error as exc:
|
|
413
|
+
raise DatabaseError(f"Cannot begin transaction: {exc}") from exc
|
|
414
|
+
try:
|
|
415
|
+
yield conn
|
|
416
|
+
conn.execute("COMMIT")
|
|
417
|
+
except BaseException as exc:
|
|
418
|
+
if conn.in_transaction:
|
|
419
|
+
try:
|
|
420
|
+
conn.execute("ROLLBACK")
|
|
421
|
+
except sqlite3.Error: # pragma: no cover - rollback is best effort
|
|
422
|
+
logger.exception("Rollback failed")
|
|
423
|
+
if isinstance(exc, sqlite3.Error):
|
|
424
|
+
raise DatabaseError(f"Transaction failed: {exc}") from exc
|
|
425
|
+
raise
|
|
426
|
+
|
|
427
|
+
@contextmanager
|
|
428
|
+
def _reading(self) -> Iterator[sqlite3.Connection]:
|
|
429
|
+
"""Yield the connection for reads, translating driver errors."""
|
|
430
|
+
conn = self.connection()
|
|
431
|
+
try:
|
|
432
|
+
yield conn
|
|
433
|
+
except sqlite3.Error as exc:
|
|
434
|
+
raise DatabaseError(f"Query failed: {exc}") from exc
|
|
435
|
+
|
|
436
|
+
# ------------------------------------------------------------------ migrations
|
|
437
|
+
|
|
438
|
+
def _migrate(self) -> None:
|
|
439
|
+
conn = self.connection()
|
|
440
|
+
try:
|
|
441
|
+
version = int(conn.execute("PRAGMA user_version").fetchone()[0])
|
|
442
|
+
except sqlite3.Error as exc:
|
|
443
|
+
raise DatabaseError(f"Cannot read schema version: {exc}") from exc
|
|
444
|
+
if version > SCHEMA_VERSION:
|
|
445
|
+
raise DatabaseError(
|
|
446
|
+
f"Database schema v{version} is newer than this build supports "
|
|
447
|
+
f"(v{SCHEMA_VERSION}); upgrade markdown-memory."
|
|
448
|
+
)
|
|
449
|
+
if version < SCHEMA_VERSION:
|
|
450
|
+
applied: list[int] = []
|
|
451
|
+
with self.transaction() as tx:
|
|
452
|
+
# Re-check under the write lock: another process starting at the same
|
|
453
|
+
# moment may have migrated the schema while this one waited for it.
|
|
454
|
+
current = int(tx.execute("PRAGMA user_version").fetchone()[0])
|
|
455
|
+
if current < 1:
|
|
456
|
+
for statement in _schema_v1(self._embedding_dim):
|
|
457
|
+
tx.execute(statement)
|
|
458
|
+
tx.execute(
|
|
459
|
+
"INSERT INTO meta(key, value) VALUES ('embedding_dim', ?)",
|
|
460
|
+
(str(self._embedding_dim),),
|
|
461
|
+
)
|
|
462
|
+
applied.append(1)
|
|
463
|
+
if current < 2:
|
|
464
|
+
stored = tx.execute(
|
|
465
|
+
"SELECT value FROM meta WHERE key = 'embedding_dim'"
|
|
466
|
+
).fetchone()
|
|
467
|
+
for statement in _schema_v2(int(stored[0])):
|
|
468
|
+
tx.execute(statement)
|
|
469
|
+
# Sections indexed before v2 have no passages, and their unchanged
|
|
470
|
+
# SHA-256 would make every later run skip them. The index is a cache
|
|
471
|
+
# of the files: drop it so the next index_directory rebuilds it whole.
|
|
472
|
+
dropped = tx.execute("DELETE FROM documents").rowcount
|
|
473
|
+
if dropped:
|
|
474
|
+
_add_notice(
|
|
475
|
+
tx,
|
|
476
|
+
f"The index format changed (passage-level vectors): discarded all "
|
|
477
|
+
f"{dropped} previously indexed documents from every directory. "
|
|
478
|
+
"Re-run index_directory for each documentation root.",
|
|
479
|
+
)
|
|
480
|
+
applied.append(2)
|
|
481
|
+
if current < 4:
|
|
482
|
+
for statement in _schema_v4():
|
|
483
|
+
tx.execute(statement)
|
|
484
|
+
tx.execute(
|
|
485
|
+
"ALTER TABLE documents "
|
|
486
|
+
f"ADD COLUMN vector_format INTEGER NOT NULL DEFAULT {_LEGACY_VECTORS}"
|
|
487
|
+
)
|
|
488
|
+
# v3 was never released; it exists only in working copies of the
|
|
489
|
+
# abandoned design. Its table is dropped rather than migrated.
|
|
490
|
+
tx.execute("DROP TABLE IF EXISTS index_problems")
|
|
491
|
+
tx.execute("DELETE FROM meta WHERE key LIKE 'incomplete:%'")
|
|
492
|
+
applied.append(4)
|
|
493
|
+
if current < 5:
|
|
494
|
+
# Rows written before v5 recorded whole seconds, which cannot see an
|
|
495
|
+
# edit made in the same second as the scan that indexed it. They get
|
|
496
|
+
# NULL, which means "no modification time was recorded" - not a
|
|
497
|
+
# timestamp of any kind, so no real one can collide with it, the epoch
|
|
498
|
+
# included. The freshness check reads their bytes instead of trusting a
|
|
499
|
+
# time, which is the slow answer but never the wrong one, and the next
|
|
500
|
+
# index_directory writes a real value and the file stops paying it.
|
|
501
|
+
# Nothing is discarded for this: the vectors are still good, and a
|
|
502
|
+
# rebuild would cost half an hour of embedding to learn nothing.
|
|
503
|
+
tx.execute("ALTER TABLE documents ADD COLUMN mtime_ns INTEGER")
|
|
504
|
+
applied.append(5)
|
|
505
|
+
if current < 6:
|
|
506
|
+
# Which weights embedded each document, so a change of model repairs the
|
|
507
|
+
# index file by file instead of discarding it. Copied from the index-wide
|
|
508
|
+
# revision where one is recorded, and that copy is exact: the revision is
|
|
509
|
+
# written only while no vector exists, every vector after it was checked
|
|
510
|
+
# against it, and it is forgotten only once no vector is left. Where none
|
|
511
|
+
# is recorded but vectors are, nobody can say what built them - they are
|
|
512
|
+
# quarantined now, in this transaction, rather than ranked against a query
|
|
513
|
+
# until some later run happens to notice.
|
|
514
|
+
tx.execute("ALTER TABLE documents ADD COLUMN weights_revision TEXT")
|
|
515
|
+
recorded = tx.execute(
|
|
516
|
+
"SELECT value FROM meta WHERE key = ?", (WEIGHTS_META_KEY,)
|
|
517
|
+
).fetchone()
|
|
518
|
+
if recorded is not None:
|
|
519
|
+
tx.execute("UPDATE documents SET weights_revision = ?", (recorded[0],))
|
|
520
|
+
elif tx.execute("SELECT 1 FROM units_vec LIMIT 1").fetchone() is not None:
|
|
521
|
+
_revoke_weights(
|
|
522
|
+
tx,
|
|
523
|
+
"No record says which weights built this index's vectors, so "
|
|
524
|
+
"they are not compared with a query: only keyword ranking is used "
|
|
525
|
+
"until index_directory re-embeds them.",
|
|
526
|
+
)
|
|
527
|
+
applied.append(6)
|
|
528
|
+
tx.execute(f"PRAGMA user_version = {SCHEMA_VERSION}")
|
|
529
|
+
if applied:
|
|
530
|
+
logger.info("Applied schema migration(s) %s at %s", applied, self._path)
|
|
531
|
+
self._seed_section_ids()
|
|
532
|
+
stored_dim = self.get_meta("embedding_dim")
|
|
533
|
+
if stored_dim is not None and int(stored_dim) != self._embedding_dim:
|
|
534
|
+
self._rebuild_for_dimension(int(stored_dim))
|
|
535
|
+
|
|
536
|
+
def _seed_section_ids(self) -> None:
|
|
537
|
+
"""Give a database from an earlier release its section-id high-water mark.
|
|
538
|
+
|
|
539
|
+
Without it the mark would be derived from ``MAX(id)`` after rows had already been
|
|
540
|
+
deleted, so a re-index - or a purge followed by one - would hand the freed ids
|
|
541
|
+
straight back out, which is exactly what ``_next_section_id`` exists to prevent.
|
|
542
|
+
"""
|
|
543
|
+
if self.get_meta(_SECTION_ID_META_KEY) is not None:
|
|
544
|
+
return
|
|
545
|
+
with self.transaction() as tx:
|
|
546
|
+
if tx.execute("SELECT 1 FROM meta WHERE key = ?", (_SECTION_ID_META_KEY,)).fetchone():
|
|
547
|
+
return # another process seeded it while this one waited for the lock
|
|
548
|
+
tx.execute(
|
|
549
|
+
"INSERT INTO meta(key, value) "
|
|
550
|
+
"SELECT ?, CAST(COALESCE(MAX(id), 0) + 1 AS TEXT) FROM sections",
|
|
551
|
+
(_SECTION_ID_META_KEY,),
|
|
552
|
+
)
|
|
553
|
+
|
|
554
|
+
def _rebuild_for_dimension(self, stored_dim: int) -> None:
|
|
555
|
+
"""Re-create the vector tables for a model with a different output size.
|
|
556
|
+
|
|
557
|
+
The index is a cache of the Markdown files: vectors of another dimensionality
|
|
558
|
+
are useless, so everything is dropped and the next ``index_directory`` rebuilds it.
|
|
559
|
+
"""
|
|
560
|
+
logger.warning(
|
|
561
|
+
"Embedding size changed (%d -> %d): discarding the index at %s; re-run "
|
|
562
|
+
"index_directory to rebuild it",
|
|
563
|
+
stored_dim,
|
|
564
|
+
self._embedding_dim,
|
|
565
|
+
self._path,
|
|
566
|
+
)
|
|
567
|
+
with self.transaction() as tx:
|
|
568
|
+
dropped = tx.execute("DELETE FROM documents").rowcount
|
|
569
|
+
_forget_weights_without_vectors(tx)
|
|
570
|
+
# Every root's documents are gone, including roots this process never looked
|
|
571
|
+
# at; a certificate that survived would vouch for an empty tree.
|
|
572
|
+
self.revoke_coverage(tx)
|
|
573
|
+
if dropped:
|
|
574
|
+
_add_notice(
|
|
575
|
+
tx,
|
|
576
|
+
f"The embedding size changed ({stored_dim} -> {self._embedding_dim} "
|
|
577
|
+
f"dimensions): discarded all {dropped} previously indexed documents from "
|
|
578
|
+
"every directory. Re-run index_directory for each documentation root.",
|
|
579
|
+
)
|
|
580
|
+
tx.execute("DROP TABLE sections_vec")
|
|
581
|
+
tx.execute("DROP TABLE units_vec")
|
|
582
|
+
for statement in _vector_tables(self._embedding_dim):
|
|
583
|
+
tx.execute(statement)
|
|
584
|
+
tx.execute(
|
|
585
|
+
"UPDATE meta SET value = ? WHERE key = 'embedding_dim'",
|
|
586
|
+
(str(self._embedding_dim),),
|
|
587
|
+
)
|
|
588
|
+
|
|
589
|
+
# ------------------------------------------------------------------ meta / pragmas
|
|
590
|
+
|
|
591
|
+
def get_meta(self, key: str) -> str | None:
|
|
592
|
+
with self._reading() as conn:
|
|
593
|
+
row = conn.execute("SELECT value FROM meta WHERE key = ?", (key,)).fetchone()
|
|
594
|
+
return None if row is None else str(row[0])
|
|
595
|
+
|
|
596
|
+
def set_meta(self, key: str, value: str) -> None:
|
|
597
|
+
with self.transaction() as conn:
|
|
598
|
+
conn.execute(
|
|
599
|
+
"INSERT INTO meta(key, value) VALUES (?, ?) "
|
|
600
|
+
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
|
|
601
|
+
(key, value),
|
|
602
|
+
)
|
|
603
|
+
|
|
604
|
+
def pending_notices(self) -> dict[str, str]:
|
|
605
|
+
"""Messages left by whatever discarded the index, oldest first, keyed for dismissal.
|
|
606
|
+
|
|
607
|
+
They are persisted, not logged only, so that whoever next runs ``index_directory``
|
|
608
|
+
- possibly another process, much later - is told why the index was empty. Reading
|
|
609
|
+
does not clear them: a run that fails before it can report them must not eat them.
|
|
610
|
+
"""
|
|
611
|
+
with self._reading() as conn:
|
|
612
|
+
rows = conn.execute(
|
|
613
|
+
"SELECT key, value FROM meta WHERE key LIKE 'notice:%' "
|
|
614
|
+
"ORDER BY CAST(substr(key, 8) AS INTEGER)"
|
|
615
|
+
).fetchall()
|
|
616
|
+
return {str(key): str(value) for key, value in rows}
|
|
617
|
+
|
|
618
|
+
def failure_paths(self, root: str) -> list[str]:
|
|
619
|
+
"""Every path recorded as unreadable under ``root``."""
|
|
620
|
+
prefix = _directory_prefix(root)
|
|
621
|
+
with self._reading() as conn:
|
|
622
|
+
rows = conn.execute(
|
|
623
|
+
"SELECT file_path FROM index_failures "
|
|
624
|
+
"WHERE file_path = ? OR substr(file_path, 1, length(?)) = ?",
|
|
625
|
+
(root, prefix, prefix),
|
|
626
|
+
).fetchall()
|
|
627
|
+
return [str(row[0]) for row in rows]
|
|
628
|
+
|
|
629
|
+
def record_failures(self, clear: Sequence[str], failures: Mapping[str, str]) -> None:
|
|
630
|
+
"""Forget the failures in ``clear``, then record ``failures``.
|
|
631
|
+
|
|
632
|
+
The caller names what to forget rather than passing a root, because a walk does
|
|
633
|
+
not reach everything beneath its root: `.venv` and `node_modules` are pruned, and
|
|
634
|
+
a directory that cannot be listed is skipped. Clearing by prefix would erase what
|
|
635
|
+
a scan never looked at - a file recorded as broken inside a pruned directory would
|
|
636
|
+
be quietly declared fine by a run of its parent.
|
|
637
|
+
|
|
638
|
+
What is cleared is replaced wholesale, because a failure can outlive every chance
|
|
639
|
+
to clear it one at a time: a file that fails on its *first* index never reaches
|
|
640
|
+
`replace_document`, so it never enters `documents` and a later purge cannot find
|
|
641
|
+
it either. Sound because one scan runs at a time - the run that just walked these
|
|
642
|
+
paths is the freshest word on them.
|
|
643
|
+
"""
|
|
644
|
+
with self.transaction() as conn:
|
|
645
|
+
conn.executemany(
|
|
646
|
+
"DELETE FROM index_failures WHERE file_path = ?", [(path,) for path in clear]
|
|
647
|
+
)
|
|
648
|
+
if failures:
|
|
649
|
+
conn.executemany(
|
|
650
|
+
"INSERT INTO index_failures(file_path, message) VALUES (?, ?)",
|
|
651
|
+
sorted(failures.items()),
|
|
652
|
+
)
|
|
653
|
+
|
|
654
|
+
def mark_scan_started(self, root: str) -> None:
|
|
655
|
+
"""This root, and every root containing it, is no longer vouched for.
|
|
656
|
+
|
|
657
|
+
Called at a scan's first write, not at its start: a run that changes nothing -
|
|
658
|
+
the model will not load, every file is unchanged - has no business retracting a
|
|
659
|
+
certificate. A scan of `docs/api` retracts `docs` too, because a half-written
|
|
660
|
+
subtree is a half-written tree.
|
|
661
|
+
"""
|
|
662
|
+
prefix = _directory_prefix(root)
|
|
663
|
+
with self.transaction() as conn:
|
|
664
|
+
conn.execute(
|
|
665
|
+
"UPDATE index_coverage SET verified = 0 "
|
|
666
|
+
# itself, anything containing it, and anything inside it: this scan may
|
|
667
|
+
# rewrite any of them, and none may go on vouching for itself while it does
|
|
668
|
+
"WHERE root = ? "
|
|
669
|
+
"OR substr(?, 1, length(root) + 1) = root || ? "
|
|
670
|
+
"OR substr(root, 1, length(?)) = ?",
|
|
671
|
+
(root, root, os.sep, prefix, prefix),
|
|
672
|
+
)
|
|
673
|
+
conn.execute(
|
|
674
|
+
"INSERT INTO index_coverage(root, verified) VALUES (?, 0) "
|
|
675
|
+
"ON CONFLICT(root) DO UPDATE SET verified = 0",
|
|
676
|
+
(root,),
|
|
677
|
+
)
|
|
678
|
+
|
|
679
|
+
def mark_scan_complete(self, root: str, generation: int) -> None:
|
|
680
|
+
"""A full walk of ``root`` ran to the end.
|
|
681
|
+
|
|
682
|
+
Not "and everything was readable" - that is what `index_failures` is for, and
|
|
683
|
+
`index_status` will not call a tree whole while anything under it is listed there.
|
|
684
|
+
Keeping the two apart means a run does not have to decide what a later question
|
|
685
|
+
will mean.
|
|
686
|
+
|
|
687
|
+
Only this root's row is written. An earlier draft also deleted the rows of roots
|
|
688
|
+
inside it, on the grounds that this walk covered them - but it does not cover a
|
|
689
|
+
pruned subtree, and a walk that hit failures covered even less. Deleting them
|
|
690
|
+
turned a nested root that was perfectly fine into one that reported unknown,
|
|
691
|
+
which is a worse answer than the one it replaced.
|
|
692
|
+
"""
|
|
693
|
+
with self.transaction() as conn:
|
|
694
|
+
current = int(
|
|
695
|
+
(
|
|
696
|
+
conn.execute(
|
|
697
|
+
"SELECT value FROM meta WHERE key = ?", (_GENERATION_META_KEY,)
|
|
698
|
+
).fetchone()
|
|
699
|
+
or ("0",)
|
|
700
|
+
)[0]
|
|
701
|
+
)
|
|
702
|
+
if current != generation:
|
|
703
|
+
# The index was discarded while this scan was walking; what it just
|
|
704
|
+
# measured describes a database that no longer exists.
|
|
705
|
+
return
|
|
706
|
+
conn.execute(
|
|
707
|
+
"INSERT INTO index_coverage(root, verified) VALUES (?, 1) "
|
|
708
|
+
"ON CONFLICT(root) DO UPDATE SET verified = 1",
|
|
709
|
+
(root,),
|
|
710
|
+
)
|
|
711
|
+
|
|
712
|
+
def revoke_coverage(self, conn: sqlite3.Connection | None = None) -> None:
|
|
713
|
+
"""Nothing is vouched for any more - the index itself was discarded.
|
|
714
|
+
|
|
715
|
+
A model or dimension change empties every document in the database, including
|
|
716
|
+
roots this process never looked at. A certificate that outlives its subject is
|
|
717
|
+
worse than none: it says a tree is whole when nothing of it is left.
|
|
718
|
+
|
|
719
|
+
The generation is bumped in the same breath. Revoking only settles the
|
|
720
|
+
certificates that exist *now*; a scan already running has read its file hashes,
|
|
721
|
+
will skip every file as unchanged, and would write a fresh certificate over an
|
|
722
|
+
empty database. It compares the generation instead and stands down.
|
|
723
|
+
"""
|
|
724
|
+
if conn is not None:
|
|
725
|
+
_bump_generation(conn)
|
|
726
|
+
conn.execute("DELETE FROM index_coverage")
|
|
727
|
+
return
|
|
728
|
+
with self.transaction() as owned:
|
|
729
|
+
_bump_generation(owned)
|
|
730
|
+
owned.execute("DELETE FROM index_coverage")
|
|
731
|
+
|
|
732
|
+
def generation(self) -> int:
|
|
733
|
+
"""How many times this database has been emptied wholesale."""
|
|
734
|
+
return int(self.get_meta(_GENERATION_META_KEY) or "0")
|
|
735
|
+
|
|
736
|
+
def index_status(self, root: str, scope: str | None = None) -> IndexStatus:
|
|
737
|
+
"""What can honestly be said about answers drawn from ``root``.
|
|
738
|
+
|
|
739
|
+
``scope`` narrows *what is named* - the failures and stale documents worth
|
|
740
|
+
mentioning - without changing whose coverage is being reported: a caller asking
|
|
741
|
+
about one directory is still served from the whole root, and the certificate
|
|
742
|
+
belongs to the root.
|
|
743
|
+
|
|
744
|
+
Every read is one snapshot. Taken separately, a scan committing between them hands
|
|
745
|
+
back a verdict that was never true at any instant: the failures read as empty, the
|
|
746
|
+
certificate still reads valid, and the answer claims a whole tree while the row
|
|
747
|
+
proving otherwise is already committed. Composing two snapshots in the caller has
|
|
748
|
+
exactly the same hole, which is why the narrowing happens here.
|
|
749
|
+
"""
|
|
750
|
+
named = scope if scope is not None else root
|
|
751
|
+
prefix = _directory_prefix(named)
|
|
752
|
+
with self._reading() as conn:
|
|
753
|
+
conn.execute("BEGIN")
|
|
754
|
+
try:
|
|
755
|
+
rows = conn.execute(
|
|
756
|
+
"SELECT file_path, message FROM index_failures "
|
|
757
|
+
"WHERE file_path = ? OR substr(file_path, 1, length(?)) = ? "
|
|
758
|
+
"ORDER BY file_path",
|
|
759
|
+
(named, prefix, prefix),
|
|
760
|
+
).fetchall()
|
|
761
|
+
certificate = conn.execute(
|
|
762
|
+
"SELECT verified FROM index_coverage WHERE root = ?", (root,)
|
|
763
|
+
).fetchone()
|
|
764
|
+
# In the same snapshot as the rest: a verdict that mixes one moment's
|
|
765
|
+
# certificate with another's provenance describes no moment at all.
|
|
766
|
+
mismatch = conn.execute(
|
|
767
|
+
"SELECT value FROM meta WHERE key = ?", (WEIGHTS_MISMATCH_KEY,)
|
|
768
|
+
).fetchone()
|
|
769
|
+
stale_vectors = int(
|
|
770
|
+
conn.execute(
|
|
771
|
+
"SELECT COUNT(*) FROM documents "
|
|
772
|
+
"WHERE (file_path = ? OR substr(file_path, 1, length(?)) = ?) "
|
|
773
|
+
"AND vector_format != ?",
|
|
774
|
+
(named, prefix, prefix, VECTOR_FORMAT),
|
|
775
|
+
).fetchone()[0]
|
|
776
|
+
)
|
|
777
|
+
# Whether the ROOT is whole, in the same snapshot. A caller asking about
|
|
778
|
+
# one directory is still answered from the whole root, so a clean
|
|
779
|
+
# subdirectory of a root that lost files is not itself a safe answer -
|
|
780
|
+
# only what is *named* narrows.
|
|
781
|
+
whole = rows == [] and stale_vectors == 0
|
|
782
|
+
if named != root:
|
|
783
|
+
root_prefix = _directory_prefix(root)
|
|
784
|
+
whole = not conn.execute(
|
|
785
|
+
"SELECT 1 FROM index_failures "
|
|
786
|
+
"WHERE file_path = ? OR substr(file_path, 1, length(?)) = ? "
|
|
787
|
+
"UNION ALL SELECT 1 FROM documents "
|
|
788
|
+
"WHERE (file_path = ? OR substr(file_path, 1, length(?)) = ?) "
|
|
789
|
+
"AND vector_format != ? LIMIT 1",
|
|
790
|
+
(root, root_prefix, root_prefix,
|
|
791
|
+
root, root_prefix, root_prefix, VECTOR_FORMAT),
|
|
792
|
+
).fetchone() # fmt: skip
|
|
793
|
+
finally:
|
|
794
|
+
conn.execute("COMMIT")
|
|
795
|
+
failures = tuple(
|
|
796
|
+
FileFailure(file_path=str(path), message=str(message)) for path, message in rows
|
|
797
|
+
)
|
|
798
|
+
verified = certificate is not None and bool(certificate[0]) and whole
|
|
799
|
+
weights_mismatch = str(mismatch[0]) if mismatch else None
|
|
800
|
+
return IndexStatus(
|
|
801
|
+
# A walk that read every file still cannot vouch for vectors built by a
|
|
802
|
+
# model that is no longer the one answering.
|
|
803
|
+
verified=verified and weights_mismatch is None,
|
|
804
|
+
failures=failures,
|
|
805
|
+
stale_vectors=stale_vectors,
|
|
806
|
+
weights_mismatch=weights_mismatch,
|
|
807
|
+
)
|
|
808
|
+
|
|
809
|
+
def dismiss_notices(self, keys: Iterable[str]) -> None:
|
|
810
|
+
"""Forget the notices that have been delivered; any added since are kept."""
|
|
811
|
+
with self.transaction() as conn:
|
|
812
|
+
conn.executemany("DELETE FROM meta WHERE key = ?", [(key,) for key in keys])
|
|
813
|
+
|
|
814
|
+
def pragma(self, name: str) -> str:
|
|
815
|
+
"""Return a PRAGMA's current value on this thread's connection."""
|
|
816
|
+
if not name.isidentifier():
|
|
817
|
+
raise DatabaseError(f"Invalid pragma name: {name!r}")
|
|
818
|
+
with self._reading() as conn:
|
|
819
|
+
row = conn.execute(f"PRAGMA {name}").fetchone()
|
|
820
|
+
return "" if row is None else str(row[0])
|
|
821
|
+
|
|
822
|
+
# ------------------------------------------------------------------ documents
|
|
823
|
+
|
|
824
|
+
def get_document(self, file_path: str) -> Document | None:
|
|
825
|
+
with self._reading() as conn:
|
|
826
|
+
row = conn.execute(
|
|
827
|
+
"SELECT id, file_path, title, content_hash, last_modified "
|
|
828
|
+
"FROM documents WHERE file_path = ?",
|
|
829
|
+
(file_path,),
|
|
830
|
+
).fetchone()
|
|
831
|
+
return None if row is None else _document_from_row(row)
|
|
832
|
+
|
|
833
|
+
def find_documents_by_suffix(self, relative_path: str) -> list[Document]:
|
|
834
|
+
"""Documents whose stored path ends with ``/<relative_path>``."""
|
|
835
|
+
suffix = os.sep + relative_path.lstrip("/\\")
|
|
836
|
+
with self._reading() as conn:
|
|
837
|
+
rows = conn.execute(
|
|
838
|
+
"SELECT id, file_path, title, content_hash, last_modified FROM documents "
|
|
839
|
+
"WHERE substr(file_path, -length(?)) = ? ORDER BY file_path",
|
|
840
|
+
(suffix, suffix),
|
|
841
|
+
).fetchall()
|
|
842
|
+
return [_document_from_row(row) for row in rows]
|
|
843
|
+
|
|
844
|
+
def document_fingerprints(self, directory: str) -> dict[str, tuple[str, int | None]]:
|
|
845
|
+
"""Map ``file_path -> (content_hash, mtime_ns)`` under ``directory``.
|
|
846
|
+
|
|
847
|
+
What a caller needs to ask the filesystem whether the index is still current:
|
|
848
|
+
the cheap question first (has the modification time moved?) and the expensive
|
|
849
|
+
one - re-hashing the bytes - only for the files where it has.
|
|
850
|
+
"""
|
|
851
|
+
prefix = _directory_prefix(directory)
|
|
852
|
+
with self._reading() as conn:
|
|
853
|
+
rows = conn.execute(
|
|
854
|
+
"SELECT file_path, content_hash, mtime_ns FROM documents "
|
|
855
|
+
"WHERE substr(file_path, 1, length(?)) = ?",
|
|
856
|
+
(prefix, prefix),
|
|
857
|
+
).fetchall()
|
|
858
|
+
return {
|
|
859
|
+
str(path): (str(content_hash), None if mtime_ns is None else int(mtime_ns))
|
|
860
|
+
for path, content_hash, mtime_ns in rows
|
|
861
|
+
}
|
|
862
|
+
|
|
863
|
+
def document_hashes(self, directory: str) -> dict[str, tuple[str, int, int | None, str | None]]:
|
|
864
|
+
"""Map ``file_path -> (hash, vector_format, mtime_ns, weights_revision)`` under it.
|
|
865
|
+
|
|
866
|
+
The format and the weights travel with the hash because all three answer the same
|
|
867
|
+
question - may this file be skipped? - and a file whose vectors predate the current
|
|
868
|
+
pooling, or came from other weights, must be rebuilt however unchanged its bytes
|
|
869
|
+
are. The recorded modification time travels with them because a file that may be
|
|
870
|
+
skipped still has to have that time brought up to date, or the freshness check reads
|
|
871
|
+
the bytes of an unchanged file for ever.
|
|
872
|
+
"""
|
|
873
|
+
prefix = _directory_prefix(directory)
|
|
874
|
+
with self._reading() as conn:
|
|
875
|
+
rows = conn.execute(
|
|
876
|
+
"SELECT file_path, content_hash, vector_format, mtime_ns, weights_revision "
|
|
877
|
+
"FROM documents WHERE substr(file_path, 1, length(?)) = ?",
|
|
878
|
+
(prefix, prefix),
|
|
879
|
+
).fetchall()
|
|
880
|
+
return {
|
|
881
|
+
str(path): (
|
|
882
|
+
str(content_hash),
|
|
883
|
+
int(vector_format),
|
|
884
|
+
None if mtime_ns is None else int(mtime_ns),
|
|
885
|
+
None if weights is None else str(weights),
|
|
886
|
+
)
|
|
887
|
+
for path, content_hash, vector_format, mtime_ns, weights in rows
|
|
888
|
+
}
|
|
889
|
+
|
|
890
|
+
def record_modification_time(
|
|
891
|
+
self, file_path: str, content_hash: str, previous_ns: int | None, mtime_ns: int
|
|
892
|
+
) -> None:
|
|
893
|
+
"""Note when an unchanged file was last written, without touching its content.
|
|
894
|
+
|
|
895
|
+
A file whose bytes are what was indexed is skipped, and used to keep whatever time
|
|
896
|
+
it was stored with - a `touch`, a checkout, or a row migrated from a schema that
|
|
897
|
+
had no nanoseconds at all. Every freshness sweep then found a time that did not
|
|
898
|
+
match and hashed the file again, forever, to conclude what the hash it already
|
|
899
|
+
held could have said. One narrow UPDATE ends that: no sections, no vectors, no
|
|
900
|
+
reindex.
|
|
901
|
+
|
|
902
|
+
A compare-and-swap on both the hash and the time the caller started from. Whoever
|
|
903
|
+
checked those bytes did so outside this transaction: an indexing run may have
|
|
904
|
+
replaced the document in between - writing a time against somebody else's content
|
|
905
|
+
is how a stale row comes to look current - or may have recorded a *newer* time for
|
|
906
|
+
the same content, which this must not roll back, or the file it just verified gets
|
|
907
|
+
hashed all over again. `IS` rather than `=` so a row that had no time recorded at
|
|
908
|
+
all is matched rather than skipped.
|
|
909
|
+
"""
|
|
910
|
+
with self.transaction() as conn:
|
|
911
|
+
conn.execute(
|
|
912
|
+
"UPDATE documents SET last_modified = ?, mtime_ns = ? "
|
|
913
|
+
"WHERE file_path = ? AND content_hash = ? AND mtime_ns IS ?",
|
|
914
|
+
(mtime_ns // 1_000_000_000, mtime_ns, file_path, content_hash, previous_ns),
|
|
915
|
+
)
|
|
916
|
+
|
|
917
|
+
def list_documents(self, directory: str = "") -> list[DocumentSummary]:
|
|
918
|
+
"""All documents (optionally restricted to ``directory``) with section counts."""
|
|
919
|
+
sql = (
|
|
920
|
+
"SELECT d.file_path, d.title, COUNT(s.id), d.last_modified "
|
|
921
|
+
"FROM documents d LEFT JOIN sections s ON s.doc_id = d.id "
|
|
922
|
+
)
|
|
923
|
+
params: tuple[str, ...] = ()
|
|
924
|
+
if directory:
|
|
925
|
+
prefix = _directory_prefix(directory)
|
|
926
|
+
sql += "WHERE substr(d.file_path, 1, length(?)) = ? "
|
|
927
|
+
params = (prefix, prefix)
|
|
928
|
+
sql += "GROUP BY d.id ORDER BY d.file_path"
|
|
929
|
+
with self._reading() as conn:
|
|
930
|
+
rows = conn.execute(sql, params).fetchall()
|
|
931
|
+
return [
|
|
932
|
+
DocumentSummary(
|
|
933
|
+
file_path=str(path),
|
|
934
|
+
title=str(title),
|
|
935
|
+
section_count=int(count),
|
|
936
|
+
last_modified=int(modified),
|
|
937
|
+
)
|
|
938
|
+
for path, title, count, modified in rows
|
|
939
|
+
]
|
|
940
|
+
|
|
941
|
+
def replace_document(
|
|
942
|
+
self,
|
|
943
|
+
*,
|
|
944
|
+
file_path: str,
|
|
945
|
+
title: str,
|
|
946
|
+
content_hash: str,
|
|
947
|
+
last_modified: int,
|
|
948
|
+
mtime_ns: int,
|
|
949
|
+
sections: Sequence[SectionDraft],
|
|
950
|
+
vectors: Sequence[SectionVectors],
|
|
951
|
+
weights_revision: str | None = None,
|
|
952
|
+
) -> Document:
|
|
953
|
+
"""Atomically insert or fully replace one document, its sections and vectors.
|
|
954
|
+
|
|
955
|
+
``weights_revision`` names the weights that produced ``vectors``, and is written in
|
|
956
|
+
the same transaction as them: it is the one moment the two are known to belong
|
|
957
|
+
together. None means unknown, which no known revision will match.
|
|
958
|
+
"""
|
|
959
|
+
if len(sections) != len(vectors):
|
|
960
|
+
raise DatabaseError(
|
|
961
|
+
f"Got {len(sections)} sections but {len(vectors)} vector sets for {file_path}"
|
|
962
|
+
)
|
|
963
|
+
for section, vector in zip(sections, vectors, strict=True):
|
|
964
|
+
if len(vector.units) != len(section.units):
|
|
965
|
+
raise DatabaseError(
|
|
966
|
+
f"Section '{section.heading_path}' of {file_path} has {len(section.units)} "
|
|
967
|
+
f"passages but {len(vector.units)} passage vectors"
|
|
968
|
+
)
|
|
969
|
+
if (vector.section is None) != (not section.units):
|
|
970
|
+
raise DatabaseError(
|
|
971
|
+
f"Section '{section.heading_path}' of {file_path}: a section vector is "
|
|
972
|
+
"required exactly when the section has passages"
|
|
973
|
+
)
|
|
974
|
+
present = [] if vector.section is None else [vector.section]
|
|
975
|
+
for embedding in (*present, *vector.units):
|
|
976
|
+
self._check_vector(embedding, f"Embedding for {file_path}")
|
|
977
|
+
with self.transaction() as conn:
|
|
978
|
+
# Sampled before the delete below: on a database from an earlier release the
|
|
979
|
+
# high-water mark is missing, and MAX(id) taken afterwards would hand the ids
|
|
980
|
+
# of the rows just deleted straight back out.
|
|
981
|
+
section_id = _next_section_id(conn) - 1
|
|
982
|
+
row = conn.execute(
|
|
983
|
+
"SELECT id FROM documents WHERE file_path = ?", (file_path,)
|
|
984
|
+
).fetchone()
|
|
985
|
+
if row is None:
|
|
986
|
+
cursor = conn.execute(
|
|
987
|
+
"INSERT INTO documents(file_path, title, content_hash, last_modified, "
|
|
988
|
+
"mtime_ns, vector_format, weights_revision) VALUES (?, ?, ?, ?, ?, ?, ?)",
|
|
989
|
+
(
|
|
990
|
+
file_path,
|
|
991
|
+
title,
|
|
992
|
+
content_hash,
|
|
993
|
+
last_modified,
|
|
994
|
+
mtime_ns,
|
|
995
|
+
VECTOR_FORMAT,
|
|
996
|
+
weights_revision,
|
|
997
|
+
),
|
|
998
|
+
)
|
|
999
|
+
if cursor.lastrowid is None: # pragma: no cover - sqlite always sets it
|
|
1000
|
+
raise DatabaseError("INSERT INTO documents returned no rowid")
|
|
1001
|
+
doc_id = cursor.lastrowid
|
|
1002
|
+
else:
|
|
1003
|
+
doc_id = int(row[0])
|
|
1004
|
+
conn.execute(
|
|
1005
|
+
"UPDATE documents SET title = ?, content_hash = ?, last_modified = ?, "
|
|
1006
|
+
"mtime_ns = ?, vector_format = ?, weights_revision = ? WHERE id = ?",
|
|
1007
|
+
(
|
|
1008
|
+
title,
|
|
1009
|
+
content_hash,
|
|
1010
|
+
last_modified,
|
|
1011
|
+
mtime_ns,
|
|
1012
|
+
VECTOR_FORMAT,
|
|
1013
|
+
weights_revision,
|
|
1014
|
+
doc_id,
|
|
1015
|
+
),
|
|
1016
|
+
)
|
|
1017
|
+
conn.execute("DELETE FROM sections WHERE doc_id = ?", (doc_id,))
|
|
1018
|
+
for section, vector in zip(sections, vectors, strict=True):
|
|
1019
|
+
section_id += 1
|
|
1020
|
+
conn.execute(
|
|
1021
|
+
"INSERT INTO sections(id, doc_id, heading_title, heading_level, "
|
|
1022
|
+
"heading_path, content, start_line, end_line, part_index) "
|
|
1023
|
+
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
|
1024
|
+
(
|
|
1025
|
+
section_id,
|
|
1026
|
+
doc_id,
|
|
1027
|
+
section.heading_title,
|
|
1028
|
+
section.heading_level,
|
|
1029
|
+
section.heading_path,
|
|
1030
|
+
section.content,
|
|
1031
|
+
section.start_line,
|
|
1032
|
+
section.end_line,
|
|
1033
|
+
section.part_index,
|
|
1034
|
+
),
|
|
1035
|
+
)
|
|
1036
|
+
if vector.section is not None:
|
|
1037
|
+
conn.execute(
|
|
1038
|
+
"INSERT INTO sections_vec(section_id, embedding) VALUES (?, ?)",
|
|
1039
|
+
(section_id, serialize_embedding(vector.section)),
|
|
1040
|
+
)
|
|
1041
|
+
for ordinal, (text, embedding) in enumerate(
|
|
1042
|
+
zip(section.units, vector.units, strict=True)
|
|
1043
|
+
):
|
|
1044
|
+
unit_id = conn.execute(
|
|
1045
|
+
"INSERT INTO units(section_id, ordinal, content) VALUES (?, ?, ?)",
|
|
1046
|
+
(section_id, ordinal, text),
|
|
1047
|
+
).lastrowid
|
|
1048
|
+
conn.execute(
|
|
1049
|
+
"INSERT INTO units_vec(unit_id, embedding) VALUES (?, ?)",
|
|
1050
|
+
(unit_id, serialize_embedding(embedding)),
|
|
1051
|
+
)
|
|
1052
|
+
conn.execute(
|
|
1053
|
+
"INSERT INTO meta(key, value) VALUES (?, ?) "
|
|
1054
|
+
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
|
|
1055
|
+
(_SECTION_ID_META_KEY, str(section_id + 1)),
|
|
1056
|
+
)
|
|
1057
|
+
# This write can remove the last vector in the index as well as add one: a
|
|
1058
|
+
# document whose prose became headings alone embeds nothing, and its old
|
|
1059
|
+
# vectors went with its old sections.
|
|
1060
|
+
_forget_weights_without_vectors(conn)
|
|
1061
|
+
return Document(
|
|
1062
|
+
id=doc_id,
|
|
1063
|
+
file_path=file_path,
|
|
1064
|
+
title=title,
|
|
1065
|
+
content_hash=content_hash,
|
|
1066
|
+
last_modified=last_modified,
|
|
1067
|
+
)
|
|
1068
|
+
|
|
1069
|
+
def _check_vector(self, embedding: Sequence[float], what: str) -> None:
|
|
1070
|
+
if len(embedding) != self._embedding_dim:
|
|
1071
|
+
raise DatabaseError(
|
|
1072
|
+
f"{what} has {len(embedding)} dimensions, expected {self._embedding_dim}"
|
|
1073
|
+
)
|
|
1074
|
+
if not _is_usable_vector(embedding):
|
|
1075
|
+
raise DatabaseError(f"{what} is all zeros or contains NaN/inf values")
|
|
1076
|
+
|
|
1077
|
+
def delete_documents(self, file_paths: Iterable[str]) -> int:
|
|
1078
|
+
"""Delete documents by path; sections, FTS rows and vectors cascade. Returns count."""
|
|
1079
|
+
paths = list(file_paths)
|
|
1080
|
+
if not paths:
|
|
1081
|
+
return 0
|
|
1082
|
+
deleted = 0
|
|
1083
|
+
with self.transaction() as conn:
|
|
1084
|
+
for path in paths:
|
|
1085
|
+
deleted += conn.execute(
|
|
1086
|
+
"DELETE FROM documents WHERE file_path = ?", (path,)
|
|
1087
|
+
).rowcount
|
|
1088
|
+
_forget_weights_without_vectors(conn)
|
|
1089
|
+
return deleted
|
|
1090
|
+
|
|
1091
|
+
def clear(self, notice: Callable[[int], str] | None = None) -> int:
|
|
1092
|
+
"""Delete every document (and, by cascade, every section and vector).
|
|
1093
|
+
|
|
1094
|
+
``notice`` words the message for the number of documents discarded; it is persisted
|
|
1095
|
+
in the same transaction, so the explanation cannot be lost while the data is.
|
|
1096
|
+
"""
|
|
1097
|
+
with self.transaction() as conn:
|
|
1098
|
+
discarded = conn.execute("DELETE FROM documents").rowcount
|
|
1099
|
+
_forget_weights_without_vectors(conn)
|
|
1100
|
+
self.revoke_coverage(conn)
|
|
1101
|
+
if discarded and notice is not None:
|
|
1102
|
+
_add_notice(conn, notice(discarded))
|
|
1103
|
+
return discarded
|
|
1104
|
+
|
|
1105
|
+
# ------------------------------------------------------------------ sections
|
|
1106
|
+
|
|
1107
|
+
def record_weights_mismatch(self, message: str | None, *, replace: bool = True) -> None:
|
|
1108
|
+
"""Remember (or clear) that the index and the loaded model disagree.
|
|
1109
|
+
|
|
1110
|
+
Persisted rather than held in memory: every `search_docs` and `list_documents`
|
|
1111
|
+
answer carries an `index_status`, and a fact this serious may not depend on
|
|
1112
|
+
which process, or which run, happens to have noticed it. ``replace=False`` keeps a
|
|
1113
|
+
message already recorded, in the same statement that would have written this one.
|
|
1114
|
+
"""
|
|
1115
|
+
with self.transaction() as conn:
|
|
1116
|
+
if message is None:
|
|
1117
|
+
conn.execute("DELETE FROM meta WHERE key = ?", (WEIGHTS_MISMATCH_KEY,))
|
|
1118
|
+
else:
|
|
1119
|
+
conn.execute(
|
|
1120
|
+
"INSERT INTO meta (key, value) VALUES (?, ?) ON CONFLICT(key) "
|
|
1121
|
+
+ ("DO UPDATE SET value = excluded.value" if replace else "DO NOTHING"),
|
|
1122
|
+
(WEIGHTS_MISMATCH_KEY, message),
|
|
1123
|
+
)
|
|
1124
|
+
|
|
1125
|
+
def revoke_weights(self, message: str) -> None:
|
|
1126
|
+
"""Withdraw the index's revision before other weights write into it.
|
|
1127
|
+
|
|
1128
|
+
From here until `settle_weights` re-certifies it, the index holds - or may hold -
|
|
1129
|
+
vectors from two models, and no search may rank them against one query.
|
|
1130
|
+
"""
|
|
1131
|
+
with self.transaction() as conn:
|
|
1132
|
+
_revoke_weights(conn, message)
|
|
1133
|
+
|
|
1134
|
+
def settle_weights(self, weights: str | None) -> None:
|
|
1135
|
+
"""Close a run: vouch for the vectors again, or say what still stands in the way.
|
|
1136
|
+
|
|
1137
|
+
The revision may be written back only once every vector in the database - every
|
|
1138
|
+
root's, and rows no walk reaches, such as a document indexed inside `.venv` - came
|
|
1139
|
+
from ``weights``. Checking only what this run visited would re-certify an index
|
|
1140
|
+
still holding another model's vectors, which is the failure a certificate exists
|
|
1141
|
+
to rule out. An index holding no vector is never certified: there is nothing to
|
|
1142
|
+
vouch for, and a revision over nothing would turn the next model away. None means
|
|
1143
|
+
this run could not tell which weights it ran, and so vouches for nothing.
|
|
1144
|
+
"""
|
|
1145
|
+
recorded = self.get_meta(WEIGHTS_META_KEY)
|
|
1146
|
+
mismatch = self.get_meta(WEIGHTS_MISMATCH_KEY)
|
|
1147
|
+
if (recorded, mismatch) != (None, None) and self.count_rows("units_vec") == 0:
|
|
1148
|
+
# A revision claimed for vectors that never arrived - the run died between
|
|
1149
|
+
# the claim and the write - describes nothing, whoever is asking.
|
|
1150
|
+
with self.transaction() as conn:
|
|
1151
|
+
_forget_weights_without_vectors(conn)
|
|
1152
|
+
return
|
|
1153
|
+
# Unknown weights vouch for nothing. Known and already certified, with nothing
|
|
1154
|
+
# said against it: the invariant holds by construction, so the whole-database
|
|
1155
|
+
# check would find nothing, and a no-op run need not take the write lock to learn
|
|
1156
|
+
# that.
|
|
1157
|
+
if weights is None or (recorded == weights and mismatch is None):
|
|
1158
|
+
return
|
|
1159
|
+
with self.transaction() as conn:
|
|
1160
|
+
if conn.execute("SELECT 1 FROM units_vec LIMIT 1").fetchone() is None:
|
|
1161
|
+
_forget_weights_without_vectors(conn)
|
|
1162
|
+
return
|
|
1163
|
+
stale = [
|
|
1164
|
+
str(row[0])
|
|
1165
|
+
for row in conn.execute(
|
|
1166
|
+
"SELECT d.file_path FROM documents AS d "
|
|
1167
|
+
"WHERE (d.weights_revision IS NULL OR d.weights_revision != ?) "
|
|
1168
|
+
"AND EXISTS (SELECT 1 FROM sections AS s JOIN units AS u "
|
|
1169
|
+
"ON u.section_id = s.id WHERE s.doc_id = d.id) "
|
|
1170
|
+
"ORDER BY d.file_path",
|
|
1171
|
+
(weights,),
|
|
1172
|
+
)
|
|
1173
|
+
]
|
|
1174
|
+
if not stale:
|
|
1175
|
+
conn.execute(
|
|
1176
|
+
"INSERT INTO meta (key, value) VALUES (?, ?) "
|
|
1177
|
+
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
|
|
1178
|
+
(WEIGHTS_META_KEY, weights),
|
|
1179
|
+
)
|
|
1180
|
+
conn.execute("DELETE FROM meta WHERE key = ?", (WEIGHTS_MISMATCH_KEY,))
|
|
1181
|
+
return
|
|
1182
|
+
# Named by directory, because a row this run could not reach - one indexed
|
|
1183
|
+
# deliberately inside a pruned directory - is repaired only by indexing that
|
|
1184
|
+
# directory itself, and a count alone would not say where to point it.
|
|
1185
|
+
folders = sorted({os.path.dirname(path) for path in stale})
|
|
1186
|
+
shown = ", ".join(folders[:3]) + (
|
|
1187
|
+
f" and {len(folders) - 3} more" if len(folders) > 3 else ""
|
|
1188
|
+
)
|
|
1189
|
+
conn.execute(
|
|
1190
|
+
"INSERT INTO meta (key, value) VALUES (?, ?) "
|
|
1191
|
+
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
|
|
1192
|
+
(
|
|
1193
|
+
WEIGHTS_MISMATCH_KEY,
|
|
1194
|
+
f"{len(stale)} document(s) still hold vectors from other weights than "
|
|
1195
|
+
f"the ones answering now, under {shown}. Only keyword ranking is used "
|
|
1196
|
+
"until index_directory re-embeds them - run it on those directories.",
|
|
1197
|
+
),
|
|
1198
|
+
)
|
|
1199
|
+
|
|
1200
|
+
def forget_weights_revision(self) -> None:
|
|
1201
|
+
"""Drop the recorded weights revision: no documents, so nothing it can describe.
|
|
1202
|
+
|
|
1203
|
+
`clear()` does this in the same transaction as the delete. This exists for every
|
|
1204
|
+
other way the index empties - a purge of the last document, a rebuild for a new
|
|
1205
|
+
vector size, the v1 format discard - where the rows go without going through it.
|
|
1206
|
+
"""
|
|
1207
|
+
with self.transaction() as conn:
|
|
1208
|
+
conn.execute("DELETE FROM meta WHERE key = ?", (WEIGHTS_META_KEY,))
|
|
1209
|
+
|
|
1210
|
+
def get_sections(self, doc_id: int) -> list[Section]:
|
|
1211
|
+
"""Every section of a document in source order."""
|
|
1212
|
+
with self._reading() as conn:
|
|
1213
|
+
rows = conn.execute(
|
|
1214
|
+
f"SELECT {_SECTION_COLUMNS} FROM sections WHERE doc_id = ? ORDER BY id",
|
|
1215
|
+
(doc_id,),
|
|
1216
|
+
).fetchall()
|
|
1217
|
+
return [_section_from_row(row) for row in rows]
|
|
1218
|
+
|
|
1219
|
+
def get_sections_with_documents(
|
|
1220
|
+
self, section_ids: Sequence[int]
|
|
1221
|
+
) -> dict[int, tuple[Section, Document]]:
|
|
1222
|
+
"""Hydrate section ids into ``(Section, Document)`` pairs."""
|
|
1223
|
+
hydrated: dict[int, tuple[Section, Document]] = {}
|
|
1224
|
+
columns = ", ".join(f"s.{name.strip()}" for name in _SECTION_COLUMNS.split(","))
|
|
1225
|
+
with self._reading() as conn:
|
|
1226
|
+
for start in range(0, len(section_ids), _SQL_VARIABLE_BATCH):
|
|
1227
|
+
batch = section_ids[start : start + _SQL_VARIABLE_BATCH]
|
|
1228
|
+
placeholders = ", ".join("?" for _ in batch)
|
|
1229
|
+
rows = conn.execute(
|
|
1230
|
+
f"SELECT {columns}, d.id, d.file_path, d.title, d.content_hash, "
|
|
1231
|
+
"d.last_modified FROM sections s JOIN documents d ON d.id = s.doc_id "
|
|
1232
|
+
f"WHERE s.id IN ({placeholders})",
|
|
1233
|
+
tuple(batch),
|
|
1234
|
+
).fetchall()
|
|
1235
|
+
for row in rows:
|
|
1236
|
+
section = _section_from_row(row[:9])
|
|
1237
|
+
hydrated[section.id] = (section, _document_from_row(row[9:]))
|
|
1238
|
+
return hydrated
|
|
1239
|
+
|
|
1240
|
+
# ------------------------------------------------------------------ search primitives
|
|
1241
|
+
|
|
1242
|
+
def fts_search(self, match_query: str, limit: int, scope: str | None = None) -> list[int]:
|
|
1243
|
+
"""Section ids matching an FTS5 query, best BM25 rank first.
|
|
1244
|
+
|
|
1245
|
+
``scope`` restricts the search to documents under one directory. The predicate
|
|
1246
|
+
joins inside the query so the limit applies to what survives it: filtering a
|
|
1247
|
+
page afterwards would return fewer rows than asked for whenever a neighbouring
|
|
1248
|
+
documentation root in the same database ranks higher.
|
|
1249
|
+
"""
|
|
1250
|
+
with self._reading() as conn:
|
|
1251
|
+
if scope is None:
|
|
1252
|
+
rows = conn.execute(
|
|
1253
|
+
"SELECT rowid FROM sections_fts WHERE sections_fts MATCH ? "
|
|
1254
|
+
"ORDER BY rank LIMIT ?",
|
|
1255
|
+
(match_query, limit),
|
|
1256
|
+
).fetchall()
|
|
1257
|
+
else:
|
|
1258
|
+
prefix = _directory_prefix(scope)
|
|
1259
|
+
rows = conn.execute(
|
|
1260
|
+
"SELECT f.rowid FROM sections_fts f "
|
|
1261
|
+
"JOIN sections s ON s.id = f.rowid JOIN documents d ON d.id = s.doc_id "
|
|
1262
|
+
"WHERE sections_fts MATCH ? AND substr(d.file_path, 1, length(?)) = ? "
|
|
1263
|
+
"ORDER BY rank LIMIT ?",
|
|
1264
|
+
(match_query, prefix, prefix, limit),
|
|
1265
|
+
).fetchall()
|
|
1266
|
+
return [int(row[0]) for row in rows]
|
|
1267
|
+
|
|
1268
|
+
def vec_search(self, embedding: Sequence[float], limit: int) -> list[tuple[int, float]]:
|
|
1269
|
+
"""``(section_id, cosine_distance)`` for the nearest section vectors, closest first."""
|
|
1270
|
+
self._check_vector(embedding, "Query embedding")
|
|
1271
|
+
with self._reading() as conn:
|
|
1272
|
+
rows = conn.execute(
|
|
1273
|
+
"SELECT section_id, distance FROM sections_vec "
|
|
1274
|
+
"WHERE embedding MATCH ? AND k = ? ORDER BY distance",
|
|
1275
|
+
(serialize_embedding(embedding), limit),
|
|
1276
|
+
).fetchall()
|
|
1277
|
+
return [
|
|
1278
|
+
(int(section_id), float(distance))
|
|
1279
|
+
for section_id, distance in rows
|
|
1280
|
+
if distance is not None # defensive: NaN distances surface as NULL
|
|
1281
|
+
]
|
|
1282
|
+
|
|
1283
|
+
def unit_search(self, embedding: Sequence[float], limit: int) -> list[tuple[int, float, str]]:
|
|
1284
|
+
"""``(section_id, cosine_distance, passage)`` for the nearest passages, closest first.
|
|
1285
|
+
|
|
1286
|
+
A section appears once per matching passage; callers keep its best one.
|
|
1287
|
+
"""
|
|
1288
|
+
self._check_vector(embedding, "Query embedding")
|
|
1289
|
+
with self._reading() as conn:
|
|
1290
|
+
rows = conn.execute(
|
|
1291
|
+
"WITH nearest AS (SELECT unit_id, distance FROM units_vec "
|
|
1292
|
+
"WHERE embedding MATCH ? AND k = ?) "
|
|
1293
|
+
"SELECT u.section_id, nearest.distance, u.content FROM nearest "
|
|
1294
|
+
"JOIN units u ON u.id = nearest.unit_id ORDER BY nearest.distance",
|
|
1295
|
+
(serialize_embedding(embedding), limit),
|
|
1296
|
+
).fetchall()
|
|
1297
|
+
return [
|
|
1298
|
+
(int(section_id), float(distance), str(content))
|
|
1299
|
+
for section_id, distance, content in rows
|
|
1300
|
+
if distance is not None
|
|
1301
|
+
]
|
|
1302
|
+
|
|
1303
|
+
def fts_matching(self, term: str, within: Sequence[int]) -> set[int]:
|
|
1304
|
+
"""Which of the sections ``within`` match the single FTS5 ``term``."""
|
|
1305
|
+
if not within:
|
|
1306
|
+
return set()
|
|
1307
|
+
placeholders = ", ".join("?" for _ in within)
|
|
1308
|
+
with self._reading() as conn:
|
|
1309
|
+
rows = conn.execute(
|
|
1310
|
+
"SELECT rowid FROM sections_fts WHERE sections_fts MATCH ? "
|
|
1311
|
+
f"AND rowid IN ({placeholders})",
|
|
1312
|
+
(term, *within),
|
|
1313
|
+
).fetchall()
|
|
1314
|
+
return {int(row[0]) for row in rows}
|
|
1315
|
+
|
|
1316
|
+
def fts_document_frequency(self, term: str) -> int:
|
|
1317
|
+
"""Number of sections matching the single FTS5 ``term``."""
|
|
1318
|
+
with self._reading() as conn:
|
|
1319
|
+
row = conn.execute(
|
|
1320
|
+
"SELECT COUNT(*) FROM sections_fts WHERE sections_fts MATCH ?", (term,)
|
|
1321
|
+
).fetchone()
|
|
1322
|
+
return int(row[0])
|
|
1323
|
+
|
|
1324
|
+
def sections_with_passages(self, section_ids: Sequence[int]) -> set[int]:
|
|
1325
|
+
"""The subset of ``section_ids`` that has a body (heading-only sections have none)."""
|
|
1326
|
+
if not section_ids:
|
|
1327
|
+
return set()
|
|
1328
|
+
placeholders = ", ".join("?" for _ in section_ids)
|
|
1329
|
+
with self._reading() as conn:
|
|
1330
|
+
rows = conn.execute(
|
|
1331
|
+
f"SELECT DISTINCT section_id FROM units WHERE section_id IN ({placeholders})",
|
|
1332
|
+
tuple(section_ids),
|
|
1333
|
+
).fetchall()
|
|
1334
|
+
return {int(row[0]) for row in rows}
|
|
1335
|
+
|
|
1336
|
+
def sections_under(self, section_ids: Sequence[int], directory: str) -> set[int]:
|
|
1337
|
+
"""The subset of ``section_ids`` whose document lives under ``directory``.
|
|
1338
|
+
|
|
1339
|
+
Search is scoped with this rather than with a predicate inside the FTS5 and
|
|
1340
|
+
vec0 queries: both apply their own limit before any join would filter, so a
|
|
1341
|
+
scoped predicate there silently returns fewer results than asked for.
|
|
1342
|
+
"""
|
|
1343
|
+
if not section_ids:
|
|
1344
|
+
return set()
|
|
1345
|
+
prefix = _directory_prefix(directory)
|
|
1346
|
+
placeholders = ", ".join("?" for _ in section_ids)
|
|
1347
|
+
with self._reading() as conn:
|
|
1348
|
+
rows = conn.execute(
|
|
1349
|
+
f"SELECT s.id FROM sections s JOIN documents d ON d.id = s.doc_id "
|
|
1350
|
+
f"WHERE s.id IN ({placeholders}) "
|
|
1351
|
+
"AND substr(d.file_path, 1, length(?)) = ?",
|
|
1352
|
+
(*section_ids, prefix, prefix),
|
|
1353
|
+
).fetchall()
|
|
1354
|
+
return {int(row[0]) for row in rows}
|
|
1355
|
+
|
|
1356
|
+
def integrity_problems(self) -> list[str]:
|
|
1357
|
+
"""Everything that is wrong with the store; an empty list means it is sound.
|
|
1358
|
+
|
|
1359
|
+
Covers SQLite's own page and foreign-key checks, the FTS5 index against the
|
|
1360
|
+
``sections`` table, and the invariants the triggers exist to uphold: one FTS row
|
|
1361
|
+
per section, one vector per passage, a section vector exactly for the sections
|
|
1362
|
+
that have passages, and every stored vector of the configured dimension.
|
|
1363
|
+
"""
|
|
1364
|
+
problems: list[str] = []
|
|
1365
|
+
blob_bytes = self._embedding_dim * 4
|
|
1366
|
+
conn = self.connection()
|
|
1367
|
+
try:
|
|
1368
|
+
# rank = 1 makes FTS5 compare the index with the external content table; the
|
|
1369
|
+
# plain form only checks the index's internal consistency. The command is an
|
|
1370
|
+
# INSERT, so it needs the write lock - failing to get it proves nothing.
|
|
1371
|
+
conn.execute(
|
|
1372
|
+
"INSERT INTO sections_fts(sections_fts, rank) VALUES ('integrity-check', 1)"
|
|
1373
|
+
)
|
|
1374
|
+
except sqlite3.Error as exc:
|
|
1375
|
+
if _is_lock_error(exc):
|
|
1376
|
+
problems.append(
|
|
1377
|
+
"could not verify the FTS5 index: the database is locked by another "
|
|
1378
|
+
"writer (not a sign of damage - retry when indexing has finished)"
|
|
1379
|
+
)
|
|
1380
|
+
else:
|
|
1381
|
+
problems.append(f"FTS5 index does not match the sections table: {exc}")
|
|
1382
|
+
try:
|
|
1383
|
+
conn.execute("BEGIN") # one read snapshot: counts taken mid-write would disagree
|
|
1384
|
+
try:
|
|
1385
|
+
version = int(conn.execute("PRAGMA user_version").fetchone()[0])
|
|
1386
|
+
pages = [str(row[0]) for row in conn.execute("PRAGMA integrity_check")]
|
|
1387
|
+
orphans = conn.execute("PRAGMA foreign_key_check").fetchall()
|
|
1388
|
+
counts = {
|
|
1389
|
+
table: int(conn.execute(f"SELECT COUNT(*) FROM {source}").fetchone()[0])
|
|
1390
|
+
for table, source in (
|
|
1391
|
+
("sections", "sections"),
|
|
1392
|
+
("sections_fts", "sections_fts_docsize"),
|
|
1393
|
+
("sections_vec", "sections_vec"),
|
|
1394
|
+
("units", "units"),
|
|
1395
|
+
("units_vec", "units_vec"),
|
|
1396
|
+
)
|
|
1397
|
+
}
|
|
1398
|
+
with_passages = int(
|
|
1399
|
+
conn.execute("SELECT COUNT(DISTINCT section_id) FROM units").fetchone()[0]
|
|
1400
|
+
)
|
|
1401
|
+
wrong_size = sum(
|
|
1402
|
+
int(
|
|
1403
|
+
conn.execute(
|
|
1404
|
+
f"SELECT COUNT(*) FROM {table} WHERE length(embedding) != ?",
|
|
1405
|
+
(blob_bytes,),
|
|
1406
|
+
).fetchone()[0]
|
|
1407
|
+
)
|
|
1408
|
+
for table in ("sections_vec", "units_vec")
|
|
1409
|
+
)
|
|
1410
|
+
row = conn.execute("SELECT value FROM meta WHERE key = 'embedding_dim'").fetchone()
|
|
1411
|
+
stored_dim = None if row is None else str(row[0])
|
|
1412
|
+
finally:
|
|
1413
|
+
conn.execute("ROLLBACK")
|
|
1414
|
+
except sqlite3.Error as exc:
|
|
1415
|
+
raise DatabaseError(f"Integrity check could not read the database: {exc}") from exc
|
|
1416
|
+
|
|
1417
|
+
if version != SCHEMA_VERSION:
|
|
1418
|
+
problems.append(f"schema version is {version}, expected {SCHEMA_VERSION}")
|
|
1419
|
+
if pages != ["ok"]:
|
|
1420
|
+
problems.append("PRAGMA integrity_check: " + "; ".join(pages[:5]))
|
|
1421
|
+
if orphans:
|
|
1422
|
+
problems.append(f"PRAGMA foreign_key_check: {len(orphans)} orphaned row(s)")
|
|
1423
|
+
if counts["sections"] != counts["sections_fts"]:
|
|
1424
|
+
problems.append(f"{counts['sections']} sections but {counts['sections_fts']} FTS rows")
|
|
1425
|
+
if counts["units"] != counts["units_vec"]:
|
|
1426
|
+
problems.append(f"{counts['units']} passages but {counts['units_vec']} passage vectors")
|
|
1427
|
+
if counts["sections_vec"] != with_passages:
|
|
1428
|
+
problems.append(
|
|
1429
|
+
f"{counts['sections_vec']} section vectors but "
|
|
1430
|
+
f"{with_passages} sections with passages"
|
|
1431
|
+
)
|
|
1432
|
+
if wrong_size:
|
|
1433
|
+
problems.append(
|
|
1434
|
+
f"{wrong_size} stored vector(s) are not {self._embedding_dim}-dimensional"
|
|
1435
|
+
)
|
|
1436
|
+
if stored_dim != str(self._embedding_dim):
|
|
1437
|
+
problems.append(f"meta embedding_dim is {stored_dim}, expected {self._embedding_dim}")
|
|
1438
|
+
return problems
|
|
1439
|
+
|
|
1440
|
+
def count_rows(self, table: str) -> int:
|
|
1441
|
+
"""Row count of one of the known tables (diagnostics and integrity tests)."""
|
|
1442
|
+
if table not in {
|
|
1443
|
+
"documents", "sections", "sections_fts", "sections_vec", "units", "units_vec",
|
|
1444
|
+
"index_failures", "index_coverage",
|
|
1445
|
+
}: # fmt: skip
|
|
1446
|
+
raise DatabaseError(f"Unknown table: {table!r}")
|
|
1447
|
+
# COUNT(*) on an external-content FTS5 table is answered from `sections`; the
|
|
1448
|
+
# docsize shadow table has one row per entry actually present in the index.
|
|
1449
|
+
source = "sections_fts_docsize" if table == "sections_fts" else table
|
|
1450
|
+
with self._reading() as conn:
|
|
1451
|
+
row = conn.execute(f"SELECT COUNT(*) FROM {source}").fetchone()
|
|
1452
|
+
return int(row[0])
|
|
1453
|
+
|
|
1454
|
+
|
|
1455
|
+
def _next_section_id(conn: sqlite3.Connection) -> int:
|
|
1456
|
+
"""A section id that was never used before, not even by a row deleted since.
|
|
1457
|
+
|
|
1458
|
+
SQLite hands the rowid of a deleted row out again. A search ranks ids on one connection
|
|
1459
|
+
and fetches them on another, so a re-index in between must make the old ids *vanish*
|
|
1460
|
+
(the search then ranks again) rather than point at whatever section was stored next.
|
|
1461
|
+
"""
|
|
1462
|
+
stored = conn.execute("SELECT value FROM meta WHERE key = ?", (_SECTION_ID_META_KEY,))
|
|
1463
|
+
row = stored.fetchone()
|
|
1464
|
+
highest = conn.execute("SELECT COALESCE(MAX(id), 0) FROM sections").fetchone()[0]
|
|
1465
|
+
return max(int(row[0]) if row is not None else 1, int(highest) + 1)
|
|
1466
|
+
|
|
1467
|
+
|
|
1468
|
+
def _add_notice(conn: sqlite3.Connection, message: str) -> None:
|
|
1469
|
+
logger.warning(message)
|
|
1470
|
+
# Numbered after the newest, not by count: notices are dismissed one by one. The key is
|
|
1471
|
+
# TEXT, so the successor is taken numerically - 'notice:10000' sorts below 'notice:9999'.
|
|
1472
|
+
newest = conn.execute(
|
|
1473
|
+
"SELECT MAX(CAST(substr(key, 8) AS INTEGER)) FROM meta WHERE key LIKE 'notice:%'"
|
|
1474
|
+
).fetchone()[0]
|
|
1475
|
+
number = 0 if newest is None else int(newest) + 1
|
|
1476
|
+
conn.execute("INSERT INTO meta(key, value) VALUES (?, ?)", (f"notice:{number:04d}", message))
|
|
1477
|
+
|
|
1478
|
+
|
|
1479
|
+
def _close_quietly(conn: sqlite3.Connection) -> None:
|
|
1480
|
+
try:
|
|
1481
|
+
conn.close()
|
|
1482
|
+
except sqlite3.Error: # pragma: no cover - best effort
|
|
1483
|
+
logger.warning("Failed to close a SQLite connection", exc_info=True)
|
|
1484
|
+
|
|
1485
|
+
|
|
1486
|
+
def _is_lock_error(exc: sqlite3.Error) -> bool:
|
|
1487
|
+
message = str(exc).lower()
|
|
1488
|
+
return "locked" in message or "busy" in message
|
|
1489
|
+
|
|
1490
|
+
|
|
1491
|
+
def _enable_wal(conn: sqlite3.Connection) -> None:
|
|
1492
|
+
"""Switch to WAL, retrying while another process holds the lock.
|
|
1493
|
+
|
|
1494
|
+
Changing the journal mode needs an exclusive lock and SQLite reports contention
|
|
1495
|
+
on it immediately instead of honouring the busy timeout.
|
|
1496
|
+
"""
|
|
1497
|
+
for attempt in range(_WAL_ATTEMPTS):
|
|
1498
|
+
try:
|
|
1499
|
+
conn.execute("PRAGMA journal_mode = WAL")
|
|
1500
|
+
return
|
|
1501
|
+
except sqlite3.OperationalError as exc:
|
|
1502
|
+
if attempt == _WAL_ATTEMPTS - 1 or not _is_lock_error(exc):
|
|
1503
|
+
raise
|
|
1504
|
+
time.sleep(_WAL_RETRY_SECONDS * (attempt + 1))
|
|
1505
|
+
|
|
1506
|
+
|
|
1507
|
+
def _document_from_row(row: Sequence[object]) -> Document:
|
|
1508
|
+
doc_id, file_path, title, content_hash, last_modified = row
|
|
1509
|
+
return Document(
|
|
1510
|
+
id=_as_int(doc_id),
|
|
1511
|
+
file_path=str(file_path),
|
|
1512
|
+
title=str(title),
|
|
1513
|
+
content_hash=str(content_hash),
|
|
1514
|
+
last_modified=_as_int(last_modified),
|
|
1515
|
+
)
|
|
1516
|
+
|
|
1517
|
+
|
|
1518
|
+
def _section_from_row(row: Sequence[object]) -> Section:
|
|
1519
|
+
(
|
|
1520
|
+
section_id,
|
|
1521
|
+
doc_id,
|
|
1522
|
+
heading_title,
|
|
1523
|
+
heading_level,
|
|
1524
|
+
heading_path,
|
|
1525
|
+
content,
|
|
1526
|
+
start_line,
|
|
1527
|
+
end_line,
|
|
1528
|
+
part_index,
|
|
1529
|
+
) = row
|
|
1530
|
+
return Section(
|
|
1531
|
+
id=_as_int(section_id),
|
|
1532
|
+
doc_id=_as_int(doc_id),
|
|
1533
|
+
heading_title=str(heading_title),
|
|
1534
|
+
heading_level=_as_int(heading_level),
|
|
1535
|
+
heading_path=str(heading_path),
|
|
1536
|
+
content=str(content),
|
|
1537
|
+
start_line=_as_int(start_line),
|
|
1538
|
+
end_line=_as_int(end_line),
|
|
1539
|
+
part_index=_as_int(part_index),
|
|
1540
|
+
)
|
|
1541
|
+
|
|
1542
|
+
|
|
1543
|
+
def _as_int(value: object) -> int:
|
|
1544
|
+
if isinstance(value, int):
|
|
1545
|
+
return value
|
|
1546
|
+
raise DatabaseError(f"Expected an integer column value, got {type(value).__name__}")
|