markdown-memory 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- markdown_memory/__init__.py +47 -0
- markdown_memory/autoindex.py +170 -0
- markdown_memory/config.py +235 -0
- markdown_memory/db.py +1546 -0
- markdown_memory/discovery.py +195 -0
- markdown_memory/embedders.py +513 -0
- markdown_memory/exceptions.py +71 -0
- markdown_memory/freshness.py +138 -0
- markdown_memory/headings.py +185 -0
- markdown_memory/indexer.py +725 -0
- markdown_memory/model_cache.py +272 -0
- markdown_memory/models.py +339 -0
- markdown_memory/parser.py +869 -0
- markdown_memory/py.typed +0 -0
- markdown_memory/search.py +518 -0
- markdown_memory/server.py +520 -0
- markdown_memory-0.1.0.dist-info/METADATA +579 -0
- markdown_memory-0.1.0.dist-info/RECORD +21 -0
- markdown_memory-0.1.0.dist-info/WHEEL +4 -0
- markdown_memory-0.1.0.dist-info/entry_points.txt +3 -0
- markdown_memory-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,725 @@
|
|
|
1
|
+
"""Incremental indexing: directory scan, SHA-256 change detection, embedding sync.
|
|
2
|
+
|
|
3
|
+
Workers read, parse and embed; one driver thread writes. The embedders themselves live in
|
|
4
|
+
``embedders.py``, the model cache in ``model_cache.py``, and the directory walk in
|
|
5
|
+
``discovery.py``.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import contextlib
|
|
11
|
+
import dataclasses
|
|
12
|
+
import fcntl
|
|
13
|
+
import logging
|
|
14
|
+
import math
|
|
15
|
+
import os
|
|
16
|
+
import stat
|
|
17
|
+
import threading
|
|
18
|
+
import time
|
|
19
|
+
from collections import deque
|
|
20
|
+
from collections.abc import Callable, Iterator, Mapping, Sequence
|
|
21
|
+
from concurrent.futures import Future, ThreadPoolExecutor
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
|
|
24
|
+
from markdown_memory import discovery
|
|
25
|
+
from markdown_memory.db import (
|
|
26
|
+
VECTOR_FORMAT,
|
|
27
|
+
WEIGHTS_META_KEY,
|
|
28
|
+
WEIGHTS_MISMATCH_KEY,
|
|
29
|
+
WEIGHTS_REVOKED,
|
|
30
|
+
Database,
|
|
31
|
+
)
|
|
32
|
+
from markdown_memory.embedders import Embedder, short_weights
|
|
33
|
+
from markdown_memory.exceptions import (
|
|
34
|
+
EmbeddingError,
|
|
35
|
+
ForeignWeightsError,
|
|
36
|
+
IndexBusyError,
|
|
37
|
+
IndexCancelled,
|
|
38
|
+
IndexingError,
|
|
39
|
+
MarkdownMemoryError,
|
|
40
|
+
ModelLoadError,
|
|
41
|
+
)
|
|
42
|
+
from markdown_memory.models import FileFailure, IndexReport, SectionDraft, SectionVectors
|
|
43
|
+
from markdown_memory.parser import MarkdownParser
|
|
44
|
+
|
|
45
|
+
logger = logging.getLogger(__name__)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
#: Files embedded at the same time. One ONNX session is shared by all of them: the weights
|
|
49
|
+
#: are mmapped and counted once however many threads run against them, so a second worker
|
|
50
|
+
#: costs the ~150 MB of one in-flight forward pass and nothing more. Embedding alone, on 16
|
|
51
|
+
#: cores over 32 passages of 512 tokens: 1 worker 0.72 vectors/s at 725 MB peak, 2 workers
|
|
52
|
+
#: 1.63 at 883 MB, 4 workers 2.17 at 1,168 MB, 8 workers 2.85 at 1,784 MB. End to end over
|
|
53
|
+
#: 24 files of the eval corpus (879 passages) the gain is smaller, because parsing and the
|
|
54
|
+
#: writes are serial and a long file holds the head of the queue: 196.3 s at 1 worker,
|
|
55
|
+
#: 145.9 s at 2 (1.35x), 96.3 s at 4 (2.04x). Two is the default because it is the last
|
|
56
|
+
#: setting whose peak - 757-814 MB across live-test runs - is nowhere near the 1.2 GB this
|
|
57
|
+
#: tool budgets for itself while running beside an editor. A machine with cores to spare
|
|
58
|
+
#: sets the variable higher and is paid ~2x for four.
|
|
59
|
+
_INDEX_WORKERS_ENV = "MARKDOWN_MEMORY_INDEX_WORKERS"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
DEFAULT_INDEX_WORKERS = 2
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
# Below this the pooled direction is rounding noise rather than a direction. Unit vectors
|
|
66
|
+
# that genuinely cancel land near 1e-16; a real centroid of normalised passages is >= 1/n
|
|
67
|
+
# of one passage, which for the 64-passage ceiling is ~0.015.
|
|
68
|
+
_MIN_POOLED_NORM = 1e-6
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
_MODEL_META_KEY = "embedding_model"
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _index_workers() -> int:
|
|
75
|
+
"""How many files are read, parsed and embedded at once. Never below one."""
|
|
76
|
+
override = os.environ.get(_INDEX_WORKERS_ENV, "").strip()
|
|
77
|
+
if override.isdigit() and int(override) > 0:
|
|
78
|
+
return int(override)
|
|
79
|
+
return DEFAULT_INDEX_WORKERS
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _section_vector(units: Sequence[Sequence[float]]) -> list[float] | None:
|
|
83
|
+
"""The vector stored for a section: pooled, or one of its passages if pooling fails.
|
|
84
|
+
|
|
85
|
+
A section with passages must have a vector - the storage layer rejects the whole file
|
|
86
|
+
otherwise - so passages that cancel each other out cannot be allowed to cost the file
|
|
87
|
+
its place in the index. Falling back to the first passage keeps a direction that is
|
|
88
|
+
at least the section's own text.
|
|
89
|
+
"""
|
|
90
|
+
if not units:
|
|
91
|
+
return None
|
|
92
|
+
pooled = _mean_vector(units)
|
|
93
|
+
return pooled if pooled is not None else list(units[0])
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _mean_vector(vectors: Sequence[Sequence[float]]) -> list[float] | None:
|
|
97
|
+
"""The centroid of ``vectors``, renormalised, or ``None`` for a section with no body.
|
|
98
|
+
|
|
99
|
+
A section's own vector used to be a separate embedding of its whole text, which the
|
|
100
|
+
model truncates at 512 tokens: 126 of 1,589 sections in the vendored corpus were
|
|
101
|
+
longer than that, the largest half again over, and their tails were simply absent
|
|
102
|
+
from the section-level signal. Averaging the passages covers the section entirely,
|
|
103
|
+
and costs one embedding call fewer per section rather than one more.
|
|
104
|
+
"""
|
|
105
|
+
if not vectors:
|
|
106
|
+
return None
|
|
107
|
+
totals = [math.fsum(values) for values in zip(*vectors, strict=True)]
|
|
108
|
+
norm = math.sqrt(math.fsum(value * value for value in totals))
|
|
109
|
+
# Not `== 0.0`: passages that point opposite ways cancel to float residue near 1e-16,
|
|
110
|
+
# and dividing that by its own magnitude turns rounding noise into a full-length
|
|
111
|
+
# vector aimed in an arbitrary direction, which then matches arbitrary queries.
|
|
112
|
+
if norm < _MIN_POOLED_NORM:
|
|
113
|
+
return None
|
|
114
|
+
return [value / norm for value in totals]
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
@dataclasses.dataclass(slots=True, frozen=True)
|
|
118
|
+
class _Prepared:
|
|
119
|
+
"""One file, read and embedded, waiting to be written.
|
|
120
|
+
|
|
121
|
+
Everything a worker produces and nothing it may do: the write is the driver's, so
|
|
122
|
+
that one thread owns section ids, the certificate and the provenance metadata.
|
|
123
|
+
"""
|
|
124
|
+
|
|
125
|
+
file_path: str
|
|
126
|
+
title: str
|
|
127
|
+
content_hash: str
|
|
128
|
+
last_modified: int
|
|
129
|
+
mtime_ns: int
|
|
130
|
+
sections: tuple[SectionDraft, ...]
|
|
131
|
+
vectors: tuple[SectionVectors, ...]
|
|
132
|
+
#: The file is what the index already holds; only when it was last written has moved.
|
|
133
|
+
unchanged: bool = False
|
|
134
|
+
#: The time the row held when this was prepared, for the write-back to compare against.
|
|
135
|
+
previous_mtime_ns: int | None = None
|
|
136
|
+
|
|
137
|
+
@property
|
|
138
|
+
def has_vectors(self) -> bool:
|
|
139
|
+
return any(vector.units for vector in self.vectors)
|
|
140
|
+
|
|
141
|
+
@property
|
|
142
|
+
def counts(self) -> tuple[int, int]:
|
|
143
|
+
return len(self.sections), sum(len(section.units) for section in self.sections)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
class Indexer:
|
|
147
|
+
"""Keeps the database in sync with the Markdown files of a directory tree."""
|
|
148
|
+
|
|
149
|
+
def __init__(
|
|
150
|
+
self,
|
|
151
|
+
db: Database,
|
|
152
|
+
embedder: Embedder,
|
|
153
|
+
workers: int | None = None,
|
|
154
|
+
exclude: Sequence[str] = (),
|
|
155
|
+
) -> None:
|
|
156
|
+
if embedder.dimension != db.embedding_dim:
|
|
157
|
+
raise IndexingError(
|
|
158
|
+
f"Embedder produces {embedder.dimension}-dimensional vectors but the "
|
|
159
|
+
f"database stores {db.embedding_dim}"
|
|
160
|
+
)
|
|
161
|
+
self._db = db
|
|
162
|
+
self._embedder = embedder
|
|
163
|
+
# One parser per thread. `MarkdownParser` holds a `MarkdownIt` with mutable
|
|
164
|
+
# ruler and env state, so two files parsed through one instance at the same time
|
|
165
|
+
# would read each other's tokens.
|
|
166
|
+
self._parsers = threading.local()
|
|
167
|
+
self._workers = max(1, workers if workers is not None else _index_workers())
|
|
168
|
+
self._exclude = tuple(exclude)
|
|
169
|
+
self._run_lock = threading.Lock()
|
|
170
|
+
|
|
171
|
+
@property
|
|
172
|
+
def _parser(self) -> MarkdownParser:
|
|
173
|
+
parser: MarkdownParser | None = getattr(self._parsers, "parser", None)
|
|
174
|
+
if parser is None:
|
|
175
|
+
parser = MarkdownParser()
|
|
176
|
+
self._parsers.parser = parser
|
|
177
|
+
return parser
|
|
178
|
+
|
|
179
|
+
@contextlib.contextmanager
|
|
180
|
+
def _scan_lock(self) -> Iterator[None]:
|
|
181
|
+
"""One scan at a time over this database, across threads and across processes.
|
|
182
|
+
|
|
183
|
+
A thread lock cannot see another process, and every ordering rule this feature
|
|
184
|
+
tried instead of a lock was wrong in one direction or the other. Both halves
|
|
185
|
+
refuse rather than wait: a scan can run for 25 minutes, and a tool call that
|
|
186
|
+
blocks that long is a client timeout, which reads to the agent as a broken
|
|
187
|
+
server rather than a busy one.
|
|
188
|
+
|
|
189
|
+
`flock` is released by the kernel when the process dies, so a killed run cannot
|
|
190
|
+
strand it - the one guarantee a row in the database could not give.
|
|
191
|
+
"""
|
|
192
|
+
if not self._run_lock.acquire(blocking=False):
|
|
193
|
+
raise IndexBusyError(
|
|
194
|
+
"Another index run is in progress in this process; try again shortly."
|
|
195
|
+
)
|
|
196
|
+
try:
|
|
197
|
+
# Inside the try: opening the lock file can fail on its own (a read-only
|
|
198
|
+
# directory, no file descriptors left), and a thread lock taken above and
|
|
199
|
+
# never released would refuse every later run in this process for good.
|
|
200
|
+
# Resolved: two spellings of one database - a symlink, a relative path -
|
|
201
|
+
# would otherwise take two different locks and both scans would proceed.
|
|
202
|
+
lock_path = str(Path(self._db.path).resolve()) + ".lock"
|
|
203
|
+
handle = os.open(lock_path, os.O_CREAT | os.O_RDWR, 0o644)
|
|
204
|
+
except OSError:
|
|
205
|
+
self._run_lock.release()
|
|
206
|
+
raise
|
|
207
|
+
try:
|
|
208
|
+
try:
|
|
209
|
+
fcntl.flock(handle, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
210
|
+
except OSError as error:
|
|
211
|
+
raise IndexBusyError(
|
|
212
|
+
f"Another process is indexing {self._db.path}; try again shortly."
|
|
213
|
+
) from error
|
|
214
|
+
os.truncate(handle, 0)
|
|
215
|
+
os.write(handle, f"{os.getpid()}\n".encode("ascii"))
|
|
216
|
+
yield
|
|
217
|
+
finally:
|
|
218
|
+
os.close(handle)
|
|
219
|
+
self._run_lock.release()
|
|
220
|
+
|
|
221
|
+
def index_directory(
|
|
222
|
+
self, directory: Path, should_stop: Callable[[], bool] | None = None
|
|
223
|
+
) -> IndexReport:
|
|
224
|
+
"""Index new/changed files, skip unchanged ones, purge files that disappeared.
|
|
225
|
+
|
|
226
|
+
A failure in one file is recorded in the report and does not abort the run.
|
|
227
|
+
``should_stop`` is asked between documents; once it answers true the run raises
|
|
228
|
+
`IndexCancelled` and leaves what a killed run leaves.
|
|
229
|
+
"""
|
|
230
|
+
try:
|
|
231
|
+
root = directory.expanduser().resolve(strict=True)
|
|
232
|
+
except (OSError, RuntimeError, ValueError) as exc: # missing, ~unknown, NUL, loop
|
|
233
|
+
raise IndexingError(f"Directory does not exist: {directory}") from exc
|
|
234
|
+
if not root.is_dir():
|
|
235
|
+
raise IndexingError(f"Not a directory: {root}")
|
|
236
|
+
try:
|
|
237
|
+
str(root).encode("utf-8")
|
|
238
|
+
except UnicodeEncodeError:
|
|
239
|
+
raise IndexingError(
|
|
240
|
+
"Directory name is not valid UTF-8 and cannot be indexed: "
|
|
241
|
+
f"{discovery._printable(str(root))}"
|
|
242
|
+
) from None
|
|
243
|
+
|
|
244
|
+
with self._scan_lock():
|
|
245
|
+
started = time.perf_counter()
|
|
246
|
+
# Set at the first write of the run, not here: a run that changes nothing has
|
|
247
|
+
# no business retracting a certificate that is still true. Once it does write,
|
|
248
|
+
# the certificate stays retracted until a full pass finishes - so a run killed
|
|
249
|
+
# partway leaves the tree honestly described as unvouched-for, with no marker
|
|
250
|
+
# to clean up and nothing to go stale.
|
|
251
|
+
retracted = False
|
|
252
|
+
|
|
253
|
+
def about_to_write() -> None:
|
|
254
|
+
nonlocal retracted
|
|
255
|
+
if not retracted:
|
|
256
|
+
self._db.mark_scan_started(str(root))
|
|
257
|
+
retracted = True
|
|
258
|
+
|
|
259
|
+
previous_model = self._db.get_meta(_MODEL_META_KEY)
|
|
260
|
+
renamed = previous_model not in {None, self._embedder.model_name}
|
|
261
|
+
if renamed and self._embedder.weights_revision is None:
|
|
262
|
+
# A model that names its weights only once loaded is loaded now: this run
|
|
263
|
+
# re-embeds everything anyway, and knowing the weights is what spares the
|
|
264
|
+
# index the discard below. One that will not load is discarded as before.
|
|
265
|
+
with contextlib.suppress(ModelLoadError):
|
|
266
|
+
self._embedder.warm_up()
|
|
267
|
+
if renamed and self._embedder.weights_revision is None:
|
|
268
|
+
# Vectors from different models are not comparable, and they share one
|
|
269
|
+
# vector table. A model that names its weights up front leaves this to the
|
|
270
|
+
# per-document stamps, which re-embed each document in place while keyword
|
|
271
|
+
# search keeps answering; one that cannot would leave nothing to tell old
|
|
272
|
+
# vectors from new, so every document has to go, not only those under
|
|
273
|
+
# `root`. Nothing is announced when a size change already emptied it.
|
|
274
|
+
self._db.clear(
|
|
275
|
+
notice=lambda discarded: (
|
|
276
|
+
f"Embedding model changed ({previous_model} -> "
|
|
277
|
+
f"{self._embedder.model_name}): discarded all {discarded} previously "
|
|
278
|
+
"indexed documents from every directory. Re-run index_directory for "
|
|
279
|
+
"any other documentation root."
|
|
280
|
+
)
|
|
281
|
+
)
|
|
282
|
+
self._db.set_meta(_MODEL_META_KEY, self._embedder.model_name)
|
|
283
|
+
# Whatever emptied the index (new format, new vector size, new model) left a
|
|
284
|
+
# notice. They are dismissed only once the report carrying them exists: a run
|
|
285
|
+
# that aborts - the model cannot be loaded - leaves them for the next one.
|
|
286
|
+
notices = self._db.pending_notices()
|
|
287
|
+
# Captured here, after this run has done its own discarding and just before it
|
|
288
|
+
# reads the hashes it will trust: a discard *after* this point means the walk
|
|
289
|
+
# measured a database that no longer exists. Captured any earlier and the run
|
|
290
|
+
# counts its own model-change wipe as somebody else's, then refuses to certify
|
|
291
|
+
# the index it just rebuilt from scratch.
|
|
292
|
+
generation = self._db.generation()
|
|
293
|
+
identity = self._run_identity()
|
|
294
|
+
known_hashes = self._db.document_hashes(str(root))
|
|
295
|
+
weights_settled = False
|
|
296
|
+
|
|
297
|
+
seen: set[str] = set()
|
|
298
|
+
indexed = unchanged = sections_indexed = passages_indexed = 0
|
|
299
|
+
failures: list[FileFailure] = []
|
|
300
|
+
unreadable: list[str] = []
|
|
301
|
+
|
|
302
|
+
def record_unreadable(error: OSError) -> None:
|
|
303
|
+
location = str(error.filename or root)
|
|
304
|
+
unreadable.append(location)
|
|
305
|
+
failures.append(
|
|
306
|
+
FileFailure(
|
|
307
|
+
file_path=discovery._printable(location),
|
|
308
|
+
message=f"Cannot list directory: {error.strerror or error}",
|
|
309
|
+
)
|
|
310
|
+
)
|
|
311
|
+
|
|
312
|
+
def store(prepared: _Prepared) -> None:
|
|
313
|
+
nonlocal weights_settled, indexed, unchanged, sections_indexed
|
|
314
|
+
nonlocal passages_indexed
|
|
315
|
+
if prepared.unchanged:
|
|
316
|
+
# Not an index: the document stands, and only the time it was last
|
|
317
|
+
# written is brought up to date. The certificate is not retracted for
|
|
318
|
+
# it either - nothing an answer is drawn from has changed.
|
|
319
|
+
self._db.record_modification_time(
|
|
320
|
+
prepared.file_path,
|
|
321
|
+
prepared.content_hash,
|
|
322
|
+
prepared.previous_mtime_ns,
|
|
323
|
+
prepared.mtime_ns,
|
|
324
|
+
)
|
|
325
|
+
unchanged += 1
|
|
326
|
+
return
|
|
327
|
+
if prepared.has_vectors and not weights_settled:
|
|
328
|
+
# Once per run, on the one thread that writes, and only once a vector
|
|
329
|
+
# really exists to be written. A document of headings alone produces
|
|
330
|
+
# none, and asking would make a file that needs no model fail when no
|
|
331
|
+
# model can be loaded. The embedding is already spent by the time a
|
|
332
|
+
# refusal lands, but nothing is stored, which is what the guard is for.
|
|
333
|
+
self._settle_weights()
|
|
334
|
+
weights_settled = True
|
|
335
|
+
# Here, and not in the worker: parsing and embedding can fail without
|
|
336
|
+
# touching the index, and a run that changed nothing must leave a standing
|
|
337
|
+
# certificate alone.
|
|
338
|
+
about_to_write()
|
|
339
|
+
self._db.replace_document(
|
|
340
|
+
file_path=prepared.file_path,
|
|
341
|
+
title=prepared.title,
|
|
342
|
+
content_hash=prepared.content_hash,
|
|
343
|
+
last_modified=prepared.last_modified,
|
|
344
|
+
mtime_ns=prepared.mtime_ns,
|
|
345
|
+
sections=prepared.sections,
|
|
346
|
+
vectors=prepared.vectors,
|
|
347
|
+
weights_revision=self._embedder.weights_revision,
|
|
348
|
+
)
|
|
349
|
+
indexed += 1
|
|
350
|
+
sections_indexed += prepared.counts[0]
|
|
351
|
+
passages_indexed += prepared.counts[1]
|
|
352
|
+
|
|
353
|
+
# Workers embed; this thread writes. Embedding is ~17 s per file and a write
|
|
354
|
+
# is under 5 ms, so nothing is gained by letting workers write and a great
|
|
355
|
+
# deal is given up: SQLite takes one writer at a time anyway, and section ids
|
|
356
|
+
# are allocated as documents are stored. Results are therefore drained in
|
|
357
|
+
# submission order - `search.py` breaks a scoring tie by section id, so ids
|
|
358
|
+
# handed out in some completion order would quietly reorder equal hits, and
|
|
359
|
+
# no later sort can give them back. The window bounds what is held in memory:
|
|
360
|
+
# a prepared file carries every vector of every passage it has.
|
|
361
|
+
files = discovery.iter_markdown_files(root, record_unreadable, self._exclude)
|
|
362
|
+
window = max(2 * self._workers, 2)
|
|
363
|
+
pending: deque[tuple[str, Future[_Prepared | None]]] = deque()
|
|
364
|
+
with ThreadPoolExecutor(
|
|
365
|
+
max_workers=self._workers, thread_name_prefix="markdown-memory-index"
|
|
366
|
+
) as pool:
|
|
367
|
+
try:
|
|
368
|
+
exhausted = False
|
|
369
|
+
while True:
|
|
370
|
+
if should_stop is not None and should_stop():
|
|
371
|
+
raise IndexCancelled("index run stopped by its owner")
|
|
372
|
+
while not exhausted and len(pending) < window:
|
|
373
|
+
path = next(files, None)
|
|
374
|
+
if path is None:
|
|
375
|
+
exhausted = True
|
|
376
|
+
break
|
|
377
|
+
file_path = str(path)
|
|
378
|
+
seen.add(file_path)
|
|
379
|
+
pending.append(
|
|
380
|
+
(
|
|
381
|
+
file_path,
|
|
382
|
+
pool.submit(
|
|
383
|
+
self._prepare_file,
|
|
384
|
+
path,
|
|
385
|
+
known_hashes.get(file_path),
|
|
386
|
+
identity,
|
|
387
|
+
),
|
|
388
|
+
)
|
|
389
|
+
)
|
|
390
|
+
if not pending:
|
|
391
|
+
break
|
|
392
|
+
file_path, future = pending.popleft()
|
|
393
|
+
try:
|
|
394
|
+
prepared = future.result()
|
|
395
|
+
# The write is inside the same guard as the read: storing one
|
|
396
|
+
# document can fail on its own - a vector the storage layer
|
|
397
|
+
# rejects, a row that will not go in - and that is this file's
|
|
398
|
+
# failure to carry, not the run's to die of.
|
|
399
|
+
if prepared is None:
|
|
400
|
+
unchanged += 1
|
|
401
|
+
elif should_stop is not None and should_stop():
|
|
402
|
+
# Checked again after the wait: embedding one file can take
|
|
403
|
+
# seconds, and a stop asked meanwhile writes nothing more.
|
|
404
|
+
raise IndexCancelled("index run stopped by its owner")
|
|
405
|
+
else:
|
|
406
|
+
store(prepared)
|
|
407
|
+
except (ModelLoadError, ForeignWeightsError):
|
|
408
|
+
raise # not this file's fault: every other file fails the same
|
|
409
|
+
except (MarkdownMemoryError, OSError) as exc:
|
|
410
|
+
logger.warning(
|
|
411
|
+
"Failed to index %s: %s", discovery._printable(file_path), exc
|
|
412
|
+
)
|
|
413
|
+
failures.append(
|
|
414
|
+
FileFailure(
|
|
415
|
+
file_path=discovery._printable(file_path), message=str(exc)
|
|
416
|
+
)
|
|
417
|
+
)
|
|
418
|
+
continue
|
|
419
|
+
except BaseException:
|
|
420
|
+
# Whatever has not started will not start. What is already running is
|
|
421
|
+
# joined by the pool on the way out; there is nowhere to put its result.
|
|
422
|
+
for _, queued in pending:
|
|
423
|
+
queued.cancel()
|
|
424
|
+
raise
|
|
425
|
+
|
|
426
|
+
vanished = self._vanished(root, known_hashes, seen, unreadable)
|
|
427
|
+
if vanished:
|
|
428
|
+
about_to_write() # deleting is changing it, even if no file was read
|
|
429
|
+
purged = self._db.delete_documents(vanished)
|
|
430
|
+
# This run's own failures are already in `errors`, and the search tools read
|
|
431
|
+
# the recorded ones straight from the database, so nothing is added to
|
|
432
|
+
# `notes`: a warning repeated in three places is how a warning becomes noise.
|
|
433
|
+
# Only what this walk could have reached: a failure inside a pruned directory
|
|
434
|
+
# or one that could not be listed is not this run's to forget, however far
|
|
435
|
+
# under its root it sits.
|
|
436
|
+
reachable = self._reachable(root, self._db.failure_paths(str(root)), unreadable)
|
|
437
|
+
self._db.record_failures(
|
|
438
|
+
reachable,
|
|
439
|
+
{
|
|
440
|
+
failure.file_path: (
|
|
441
|
+
f"{failure.message} (indexing {discovery._printable(str(root))})"
|
|
442
|
+
)
|
|
443
|
+
for failure in failures
|
|
444
|
+
},
|
|
445
|
+
)
|
|
446
|
+
# Whole-database, so it cannot vouch for rows this walk never reached.
|
|
447
|
+
self._db.settle_weights(self._embedder.weights_revision)
|
|
448
|
+
# The walk finished, which is all this records; what it could not read is
|
|
449
|
+
# recorded separately, and `index_status` refuses to call a tree whole while
|
|
450
|
+
# anything under it is still listed there. Two facts, two places, one answer.
|
|
451
|
+
self._db.mark_scan_complete(str(root), generation)
|
|
452
|
+
report = IndexReport(
|
|
453
|
+
directory=discovery._printable(str(root)),
|
|
454
|
+
files_scanned=len(seen),
|
|
455
|
+
files_indexed=indexed,
|
|
456
|
+
files_unchanged=unchanged,
|
|
457
|
+
files_purged=purged,
|
|
458
|
+
sections_indexed=sections_indexed,
|
|
459
|
+
passages_indexed=passages_indexed,
|
|
460
|
+
elapsed_seconds=time.perf_counter() - started,
|
|
461
|
+
errors=tuple(failures),
|
|
462
|
+
notes=tuple(notices.values()),
|
|
463
|
+
)
|
|
464
|
+
self._db.dismiss_notices(notices)
|
|
465
|
+
logger.info(report.summary())
|
|
466
|
+
return report
|
|
467
|
+
|
|
468
|
+
def _run_identity(self) -> str | None:
|
|
469
|
+
"""The weights this run embeds with, when they can be known before it embeds.
|
|
470
|
+
|
|
471
|
+
A model that names its weights without loading (EmbeddingGemma) always answers.
|
|
472
|
+
One that learns them by loading (bge-small) is loaded here only when a repair is
|
|
473
|
+
pending - the index is being re-embedded, disagrees with some model, or holds
|
|
474
|
+
vectors no revision vouches for - because such a run has embedding to do anyway;
|
|
475
|
+
otherwise a run that changes nothing would pay for a model load. None means
|
|
476
|
+
documents are compared on content and format alone, as before stamps existed; a
|
|
477
|
+
stale one left behind that way keeps the certificate withheld at the end of the
|
|
478
|
+
run, and the next run repairs it.
|
|
479
|
+
"""
|
|
480
|
+
recorded = self._db.get_meta(WEIGHTS_META_KEY)
|
|
481
|
+
if self._embedder.weights_revision is None and (
|
|
482
|
+
recorded == WEIGHTS_REVOKED
|
|
483
|
+
or self._db.get_meta(WEIGHTS_MISMATCH_KEY) is not None
|
|
484
|
+
or (recorded is None and self._db.count_rows("units_vec") > 0)
|
|
485
|
+
):
|
|
486
|
+
try:
|
|
487
|
+
self._embedder.warm_up()
|
|
488
|
+
except ModelLoadError as exc:
|
|
489
|
+
# Degrade rather than fail: a run whose files need no embedding can still
|
|
490
|
+
# finish, and the recorded message keeps saying what is wrong.
|
|
491
|
+
logger.warning("Model not loaded, stored weights not compared: %s", exc)
|
|
492
|
+
return self._embedder.weights_revision
|
|
493
|
+
|
|
494
|
+
def _settle_weights(self) -> None:
|
|
495
|
+
"""Make sure the index says it is being re-embedded before other weights write in.
|
|
496
|
+
|
|
497
|
+
Called once per run, from the driver, at the first document that really has
|
|
498
|
+
vectors: the earliest moment a lazily-loaded embedder can be asked what it is
|
|
499
|
+
without making a run that needs no model load one, and on the only thread allowed
|
|
500
|
+
to write what the answer implies. Weights that differ from the recorded ones - or
|
|
501
|
+
any known weights joining vectors nobody vouched for - revoke the certificate and
|
|
502
|
+
carry on: every search, whatever model it runs, then ranks on keywords alone until
|
|
503
|
+
`settle_weights` finds every vector-bearing document stamped with one revision.
|
|
504
|
+
Nothing is discarded first, so keyword search answers throughout, and a run killed
|
|
505
|
+
partway resumes where it stopped, because each document's stamp is written with
|
|
506
|
+
its vectors.
|
|
507
|
+
|
|
508
|
+
Weights that cannot be named still refuse: vectors written now would be
|
|
509
|
+
indistinguishable from the ones already stored, and no stamp could repair that.
|
|
510
|
+
"""
|
|
511
|
+
if self._db.count_rows("units_vec") == 0:
|
|
512
|
+
# No vector here for any of this to be about. Documents are the wrong
|
|
513
|
+
# question: a file of nothing but headings is stored and embeds nothing, so an
|
|
514
|
+
# index can hold documents and no vectors at all. Whatever this run is about
|
|
515
|
+
# to write is therefore the whole of it, and it may say so - before the write
|
|
516
|
+
# rather than after, so that another process reading these vectors a moment
|
|
517
|
+
# from now finds them labelled. A run that dies in between leaves a revision
|
|
518
|
+
# over no vectors, which the next one clears exactly here.
|
|
519
|
+
self._claim_empty_index()
|
|
520
|
+
return
|
|
521
|
+
recorded = self._db.get_meta(WEIGHTS_META_KEY)
|
|
522
|
+
# Already loaded: a worker embedded the document this is about to store. Kept
|
|
523
|
+
# anyway, because `warm_up` is what makes `weights_revision` answerable and this
|
|
524
|
+
# is called from tests and from runs whose first document came from a cache.
|
|
525
|
+
self._embedder.warm_up()
|
|
526
|
+
weights = self._embedder.weights_revision
|
|
527
|
+
if weights is None:
|
|
528
|
+
if recorded in {None, WEIGHTS_REVOKED}:
|
|
529
|
+
# No provenance to contradict, or none left to protect: a revoked index is
|
|
530
|
+
# already ranked by keyword alone, and only a run whose weights have a name
|
|
531
|
+
# can stamp its way back out of that.
|
|
532
|
+
return
|
|
533
|
+
# The model loaded, so something answered - it just cannot say which weights
|
|
534
|
+
# it is. That is not "nothing to compare": the vectors written now would be
|
|
535
|
+
# unlabelled and indistinguishable from the ones already stored, which is the
|
|
536
|
+
# state this guard exists to prevent.
|
|
537
|
+
message = (
|
|
538
|
+
f"Which weights {self._embedder.model_name} is running could not be read, so "
|
|
539
|
+
"there is no way to tell whether they are the ones that built this index "
|
|
540
|
+
f"({short_weights(recorded)}). Nothing has been discarded and no vector has been "
|
|
541
|
+
"stored - a document of headings alone, which embeds nothing, may have "
|
|
542
|
+
"been updated before this was reached; repair the model cache and run "
|
|
543
|
+
"index_directory again."
|
|
544
|
+
)
|
|
545
|
+
self._db.record_weights_mismatch(message)
|
|
546
|
+
self._db.revoke_coverage()
|
|
547
|
+
raise ForeignWeightsError(message)
|
|
548
|
+
if weights == recorded:
|
|
549
|
+
self._db.record_weights_mismatch(None)
|
|
550
|
+
return
|
|
551
|
+
self._db.revoke_weights(
|
|
552
|
+
f"{self._embedder.model_name} is re-embedding this index with other weights "
|
|
553
|
+
f"({short_weights(weights)}). Only keyword ranking is used until every "
|
|
554
|
+
"document has been re-embedded; semantic ranking resumes when index_directory "
|
|
555
|
+
"finishes."
|
|
556
|
+
)
|
|
557
|
+
|
|
558
|
+
def _claim_empty_index(self) -> None:
|
|
559
|
+
"""Take ownership of an index that holds no vectors, and drop what described none.
|
|
560
|
+
|
|
561
|
+
A model *name* is not enough for bge-small: fastembed pins no revision, so a
|
|
562
|
+
re-download can bring different weights under the same name and nothing in the
|
|
563
|
+
index would notice. This is the one moment a revision can honestly be written -
|
|
564
|
+
the vectors that follow are all there will be, and there are none yet to
|
|
565
|
+
contradict. A model that cannot say which weights it is records nothing, and the
|
|
566
|
+
index carries no provenance rather than a provenance that might be wrong.
|
|
567
|
+
"""
|
|
568
|
+
self._db.forget_weights_revision()
|
|
569
|
+
self._db.record_weights_mismatch(None)
|
|
570
|
+
self._embedder.warm_up()
|
|
571
|
+
weights = self._embedder.weights_revision
|
|
572
|
+
if weights is not None:
|
|
573
|
+
self._db.set_meta(WEIGHTS_META_KEY, weights)
|
|
574
|
+
|
|
575
|
+
def _reachable(self, root: Path, paths: Sequence[str], unreadable: Sequence[str]) -> list[str]:
|
|
576
|
+
"""The subset of ``paths`` a walk of ``root`` would have visited.
|
|
577
|
+
|
|
578
|
+
Pruned directories (`.venv`, `node_modules`), directories excluded by
|
|
579
|
+
configuration, and directories that could not be listed are never entered, so this
|
|
580
|
+
run saw nothing inside them and may not speak for what it did not see.
|
|
581
|
+
|
|
582
|
+
Every component is tested, including the last. A recorded failure is usually a
|
|
583
|
+
file, but an unreadable *directory* is recorded under its own path - and dropping
|
|
584
|
+
the final component would ask whether `.venv`'s parent is walkable rather than
|
|
585
|
+
whether `.venv` is, and then clear it.
|
|
586
|
+
|
|
587
|
+
A path out of the walk's sight is still retired once it is observably gone
|
|
588
|
+
(`discovery._certainly_gone`), or its row would outlive the file and no run could ever
|
|
589
|
+
retire it.
|
|
590
|
+
"""
|
|
591
|
+
blocked = tuple(location.rstrip(os.sep) + os.sep for location in unreadable)
|
|
592
|
+
visitable = []
|
|
593
|
+
for path in paths:
|
|
594
|
+
if self._walk_would_visit(root, path, blocked) or discovery._certainly_gone(root, path):
|
|
595
|
+
visitable.append(path)
|
|
596
|
+
return visitable
|
|
597
|
+
|
|
598
|
+
def _walk_would_visit(self, root: Path, path: str, blocked: tuple[str, ...]) -> bool:
|
|
599
|
+
"""Whether a walk of ``root`` reaches ``path``, given the directories it could not list."""
|
|
600
|
+
if blocked and path.startswith(blocked):
|
|
601
|
+
return False
|
|
602
|
+
if not discovery._is_walkable(os.path.relpath(path, root).split(os.sep)):
|
|
603
|
+
return False
|
|
604
|
+
if discovery._behind_symlink(root, path) or discovery._is_shadowing_symlink(path):
|
|
605
|
+
return False
|
|
606
|
+
return not (self._exclude and discovery._is_excluded(Path(path), root, self._exclude))
|
|
607
|
+
|
|
608
|
+
@staticmethod
|
|
609
|
+
def _vanished(
|
|
610
|
+
root: Path,
|
|
611
|
+
known: Mapping[str, object],
|
|
612
|
+
seen: set[str],
|
|
613
|
+
unreadable: Sequence[str],
|
|
614
|
+
) -> list[str]:
|
|
615
|
+
"""Known documents that this walk *would* have found had they still existed.
|
|
616
|
+
|
|
617
|
+
A document is only purged when its absence is evidence of deletion. It is kept
|
|
618
|
+
when the walk could not have reached it: it lives under a directory that could
|
|
619
|
+
not be listed, or under a pruned tree (``node_modules`` ...) that was indexed
|
|
620
|
+
explicitly by pointing ``index_directory`` inside it.
|
|
621
|
+
|
|
622
|
+
Unless the file is observably gone (`discovery._certainly_gone`). Not being visited is not
|
|
623
|
+
evidence of deletion; `ENOENT` on that one name is exactly that evidence, and
|
|
624
|
+
without it a deleted document under a pruned tree keeps answering searches with
|
|
625
|
+
text that is not on disk any more, until someone re-indexes that tree by hand.
|
|
626
|
+
"""
|
|
627
|
+
blocked = tuple(location.rstrip(os.sep) + os.sep for location in unreadable)
|
|
628
|
+
vanished: list[str] = []
|
|
629
|
+
for file_path in sorted(set(known) - seen):
|
|
630
|
+
if discovery._certainly_gone(root, file_path):
|
|
631
|
+
vanished.append(file_path)
|
|
632
|
+
continue
|
|
633
|
+
if blocked and file_path.startswith(blocked):
|
|
634
|
+
continue
|
|
635
|
+
relative = os.path.relpath(file_path, root)
|
|
636
|
+
if not discovery._is_walkable(relative.split(os.sep)[:-1]):
|
|
637
|
+
continue
|
|
638
|
+
if discovery._behind_symlink(root, file_path):
|
|
639
|
+
continue
|
|
640
|
+
vanished.append(file_path)
|
|
641
|
+
return vanished
|
|
642
|
+
|
|
643
|
+
def _prepare_file(
|
|
644
|
+
self,
|
|
645
|
+
path: Path,
|
|
646
|
+
known: tuple[str, int, int | None, str | None] | None,
|
|
647
|
+
identity: str | None,
|
|
648
|
+
) -> _Prepared | None:
|
|
649
|
+
"""Read, parse and embed one file. ``None`` when it is unchanged.
|
|
650
|
+
|
|
651
|
+
Runs on a worker thread and writes nothing: every database write of a run belongs
|
|
652
|
+
to the driver, so that section ids are handed out in walk order and the index's
|
|
653
|
+
account of itself - certificate, provenance, failures - has a single author.
|
|
654
|
+
"""
|
|
655
|
+
file_path = str(path)
|
|
656
|
+
try:
|
|
657
|
+
file_path.encode("utf-8")
|
|
658
|
+
except UnicodeEncodeError:
|
|
659
|
+
raise IndexingError("File name is not valid UTF-8; skipped") from None
|
|
660
|
+
info = path.stat()
|
|
661
|
+
if not stat.S_ISREG(info.st_mode):
|
|
662
|
+
# A FIFO or device named *.md would block or stream forever when read.
|
|
663
|
+
raise IndexingError("Not a regular file; skipped")
|
|
664
|
+
# Checked again on the descriptor: between that stat and this open the path can be
|
|
665
|
+
# replaced by a FIFO, and a blocking open would then wait for a writer that may
|
|
666
|
+
# never come - with a worker of the pool in its hand.
|
|
667
|
+
data = discovery.read_regular_file(path)
|
|
668
|
+
if data is None:
|
|
669
|
+
raise IndexingError("Not a regular file; skipped")
|
|
670
|
+
if len(data) > discovery.MAX_FILE_BYTES:
|
|
671
|
+
raise IndexingError(f"File is larger than {discovery.MAX_FILE_BYTES} bytes; skipped")
|
|
672
|
+
content_hash = discovery.hash_bytes(data)
|
|
673
|
+
# The format counts as much as the content: a file whose bytes never changed still
|
|
674
|
+
# has to be rebuilt if its vectors were pooled by an older scheme, or it would keep
|
|
675
|
+
# them forever and the table would answer one query two different ways. So do the
|
|
676
|
+
# weights, once this run knows its own: a document stamped by other ones is
|
|
677
|
+
# re-embedded in place, which is how a changed model repairs the index file by file.
|
|
678
|
+
if (
|
|
679
|
+
known is not None
|
|
680
|
+
and known[:2] == (content_hash, VECTOR_FORMAT)
|
|
681
|
+
and (identity is None or known[3] == identity)
|
|
682
|
+
):
|
|
683
|
+
if known[2] == info.st_mtime_ns:
|
|
684
|
+
return None
|
|
685
|
+
# Same bytes, a different timestamp: nothing to parse, embed or store, but the
|
|
686
|
+
# time has to be written down or the freshness check hashes this file again on
|
|
687
|
+
# every sweep from here on.
|
|
688
|
+
return _Prepared(
|
|
689
|
+
file_path=file_path,
|
|
690
|
+
title="",
|
|
691
|
+
content_hash=content_hash,
|
|
692
|
+
last_modified=int(info.st_mtime),
|
|
693
|
+
mtime_ns=info.st_mtime_ns,
|
|
694
|
+
sections=(),
|
|
695
|
+
vectors=(),
|
|
696
|
+
unchanged=True,
|
|
697
|
+
previous_mtime_ns=known[2],
|
|
698
|
+
)
|
|
699
|
+
parsed = self._parser.parse(
|
|
700
|
+
data.decode("utf-8", errors="replace"), fallback_title=path.stem
|
|
701
|
+
)
|
|
702
|
+
# One embedding call per file, over the passages alone. The section vector is the
|
|
703
|
+
# mean of its passages rather than a separate embedding of the whole section:
|
|
704
|
+
# that text ran past the model's 512-token limit for 7.9% of the vendored corpus
|
|
705
|
+
# and lost its tail, and embedding it cost one extra call per section.
|
|
706
|
+
texts: list[str] = []
|
|
707
|
+
for section in parsed.sections:
|
|
708
|
+
texts.extend(section.unit_texts)
|
|
709
|
+
embeddings = self._embedder.embed_documents(texts)
|
|
710
|
+
if len(embeddings) != len(texts):
|
|
711
|
+
raise EmbeddingError(f"Got {len(embeddings)} vectors for {len(texts)} texts")
|
|
712
|
+
embedded = iter(embeddings)
|
|
713
|
+
vectors = []
|
|
714
|
+
for section in parsed.sections:
|
|
715
|
+
units = tuple(next(embedded) for _ in section.units)
|
|
716
|
+
vectors.append(SectionVectors(section=_section_vector(units), units=units))
|
|
717
|
+
return _Prepared(
|
|
718
|
+
file_path=file_path,
|
|
719
|
+
title=parsed.title,
|
|
720
|
+
content_hash=content_hash,
|
|
721
|
+
last_modified=int(info.st_mtime),
|
|
722
|
+
mtime_ns=info.st_mtime_ns,
|
|
723
|
+
sections=tuple(parsed.sections),
|
|
724
|
+
vectors=tuple(vectors),
|
|
725
|
+
)
|