markdown-memory 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,725 @@
1
+ """Incremental indexing: directory scan, SHA-256 change detection, embedding sync.
2
+
3
+ Workers read, parse and embed; one driver thread writes. The embedders themselves live in
4
+ ``embedders.py``, the model cache in ``model_cache.py``, and the directory walk in
5
+ ``discovery.py``.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import contextlib
11
+ import dataclasses
12
+ import fcntl
13
+ import logging
14
+ import math
15
+ import os
16
+ import stat
17
+ import threading
18
+ import time
19
+ from collections import deque
20
+ from collections.abc import Callable, Iterator, Mapping, Sequence
21
+ from concurrent.futures import Future, ThreadPoolExecutor
22
+ from pathlib import Path
23
+
24
+ from markdown_memory import discovery
25
+ from markdown_memory.db import (
26
+ VECTOR_FORMAT,
27
+ WEIGHTS_META_KEY,
28
+ WEIGHTS_MISMATCH_KEY,
29
+ WEIGHTS_REVOKED,
30
+ Database,
31
+ )
32
+ from markdown_memory.embedders import Embedder, short_weights
33
+ from markdown_memory.exceptions import (
34
+ EmbeddingError,
35
+ ForeignWeightsError,
36
+ IndexBusyError,
37
+ IndexCancelled,
38
+ IndexingError,
39
+ MarkdownMemoryError,
40
+ ModelLoadError,
41
+ )
42
+ from markdown_memory.models import FileFailure, IndexReport, SectionDraft, SectionVectors
43
+ from markdown_memory.parser import MarkdownParser
44
+
45
+ logger = logging.getLogger(__name__)
46
+
47
+
48
+ #: Files embedded at the same time. One ONNX session is shared by all of them: the weights
49
+ #: are mmapped and counted once however many threads run against them, so a second worker
50
+ #: costs the ~150 MB of one in-flight forward pass and nothing more. Embedding alone, on 16
51
+ #: cores over 32 passages of 512 tokens: 1 worker 0.72 vectors/s at 725 MB peak, 2 workers
52
+ #: 1.63 at 883 MB, 4 workers 2.17 at 1,168 MB, 8 workers 2.85 at 1,784 MB. End to end over
53
+ #: 24 files of the eval corpus (879 passages) the gain is smaller, because parsing and the
54
+ #: writes are serial and a long file holds the head of the queue: 196.3 s at 1 worker,
55
+ #: 145.9 s at 2 (1.35x), 96.3 s at 4 (2.04x). Two is the default because it is the last
56
+ #: setting whose peak - 757-814 MB across live-test runs - is nowhere near the 1.2 GB this
57
+ #: tool budgets for itself while running beside an editor. A machine with cores to spare
58
+ #: sets the variable higher and is paid ~2x for four.
59
+ _INDEX_WORKERS_ENV = "MARKDOWN_MEMORY_INDEX_WORKERS"
60
+
61
+
62
+ DEFAULT_INDEX_WORKERS = 2
63
+
64
+
65
+ # Below this the pooled direction is rounding noise rather than a direction. Unit vectors
66
+ # that genuinely cancel land near 1e-16; a real centroid of normalised passages is >= 1/n
67
+ # of one passage, which for the 64-passage ceiling is ~0.015.
68
+ _MIN_POOLED_NORM = 1e-6
69
+
70
+
71
+ _MODEL_META_KEY = "embedding_model"
72
+
73
+
74
+ def _index_workers() -> int:
75
+ """How many files are read, parsed and embedded at once. Never below one."""
76
+ override = os.environ.get(_INDEX_WORKERS_ENV, "").strip()
77
+ if override.isdigit() and int(override) > 0:
78
+ return int(override)
79
+ return DEFAULT_INDEX_WORKERS
80
+
81
+
82
+ def _section_vector(units: Sequence[Sequence[float]]) -> list[float] | None:
83
+ """The vector stored for a section: pooled, or one of its passages if pooling fails.
84
+
85
+ A section with passages must have a vector - the storage layer rejects the whole file
86
+ otherwise - so passages that cancel each other out cannot be allowed to cost the file
87
+ its place in the index. Falling back to the first passage keeps a direction that is
88
+ at least the section's own text.
89
+ """
90
+ if not units:
91
+ return None
92
+ pooled = _mean_vector(units)
93
+ return pooled if pooled is not None else list(units[0])
94
+
95
+
96
+ def _mean_vector(vectors: Sequence[Sequence[float]]) -> list[float] | None:
97
+ """The centroid of ``vectors``, renormalised, or ``None`` for a section with no body.
98
+
99
+ A section's own vector used to be a separate embedding of its whole text, which the
100
+ model truncates at 512 tokens: 126 of 1,589 sections in the vendored corpus were
101
+ longer than that, the largest half again over, and their tails were simply absent
102
+ from the section-level signal. Averaging the passages covers the section entirely,
103
+ and costs one embedding call fewer per section rather than one more.
104
+ """
105
+ if not vectors:
106
+ return None
107
+ totals = [math.fsum(values) for values in zip(*vectors, strict=True)]
108
+ norm = math.sqrt(math.fsum(value * value for value in totals))
109
+ # Not `== 0.0`: passages that point opposite ways cancel to float residue near 1e-16,
110
+ # and dividing that by its own magnitude turns rounding noise into a full-length
111
+ # vector aimed in an arbitrary direction, which then matches arbitrary queries.
112
+ if norm < _MIN_POOLED_NORM:
113
+ return None
114
+ return [value / norm for value in totals]
115
+
116
+
117
+ @dataclasses.dataclass(slots=True, frozen=True)
118
+ class _Prepared:
119
+ """One file, read and embedded, waiting to be written.
120
+
121
+ Everything a worker produces and nothing it may do: the write is the driver's, so
122
+ that one thread owns section ids, the certificate and the provenance metadata.
123
+ """
124
+
125
+ file_path: str
126
+ title: str
127
+ content_hash: str
128
+ last_modified: int
129
+ mtime_ns: int
130
+ sections: tuple[SectionDraft, ...]
131
+ vectors: tuple[SectionVectors, ...]
132
+ #: The file is what the index already holds; only when it was last written has moved.
133
+ unchanged: bool = False
134
+ #: The time the row held when this was prepared, for the write-back to compare against.
135
+ previous_mtime_ns: int | None = None
136
+
137
+ @property
138
+ def has_vectors(self) -> bool:
139
+ return any(vector.units for vector in self.vectors)
140
+
141
+ @property
142
+ def counts(self) -> tuple[int, int]:
143
+ return len(self.sections), sum(len(section.units) for section in self.sections)
144
+
145
+
146
+ class Indexer:
147
+ """Keeps the database in sync with the Markdown files of a directory tree."""
148
+
149
+ def __init__(
150
+ self,
151
+ db: Database,
152
+ embedder: Embedder,
153
+ workers: int | None = None,
154
+ exclude: Sequence[str] = (),
155
+ ) -> None:
156
+ if embedder.dimension != db.embedding_dim:
157
+ raise IndexingError(
158
+ f"Embedder produces {embedder.dimension}-dimensional vectors but the "
159
+ f"database stores {db.embedding_dim}"
160
+ )
161
+ self._db = db
162
+ self._embedder = embedder
163
+ # One parser per thread. `MarkdownParser` holds a `MarkdownIt` with mutable
164
+ # ruler and env state, so two files parsed through one instance at the same time
165
+ # would read each other's tokens.
166
+ self._parsers = threading.local()
167
+ self._workers = max(1, workers if workers is not None else _index_workers())
168
+ self._exclude = tuple(exclude)
169
+ self._run_lock = threading.Lock()
170
+
171
+ @property
172
+ def _parser(self) -> MarkdownParser:
173
+ parser: MarkdownParser | None = getattr(self._parsers, "parser", None)
174
+ if parser is None:
175
+ parser = MarkdownParser()
176
+ self._parsers.parser = parser
177
+ return parser
178
+
179
+ @contextlib.contextmanager
180
+ def _scan_lock(self) -> Iterator[None]:
181
+ """One scan at a time over this database, across threads and across processes.
182
+
183
+ A thread lock cannot see another process, and every ordering rule this feature
184
+ tried instead of a lock was wrong in one direction or the other. Both halves
185
+ refuse rather than wait: a scan can run for 25 minutes, and a tool call that
186
+ blocks that long is a client timeout, which reads to the agent as a broken
187
+ server rather than a busy one.
188
+
189
+ `flock` is released by the kernel when the process dies, so a killed run cannot
190
+ strand it - the one guarantee a row in the database could not give.
191
+ """
192
+ if not self._run_lock.acquire(blocking=False):
193
+ raise IndexBusyError(
194
+ "Another index run is in progress in this process; try again shortly."
195
+ )
196
+ try:
197
+ # Inside the try: opening the lock file can fail on its own (a read-only
198
+ # directory, no file descriptors left), and a thread lock taken above and
199
+ # never released would refuse every later run in this process for good.
200
+ # Resolved: two spellings of one database - a symlink, a relative path -
201
+ # would otherwise take two different locks and both scans would proceed.
202
+ lock_path = str(Path(self._db.path).resolve()) + ".lock"
203
+ handle = os.open(lock_path, os.O_CREAT | os.O_RDWR, 0o644)
204
+ except OSError:
205
+ self._run_lock.release()
206
+ raise
207
+ try:
208
+ try:
209
+ fcntl.flock(handle, fcntl.LOCK_EX | fcntl.LOCK_NB)
210
+ except OSError as error:
211
+ raise IndexBusyError(
212
+ f"Another process is indexing {self._db.path}; try again shortly."
213
+ ) from error
214
+ os.truncate(handle, 0)
215
+ os.write(handle, f"{os.getpid()}\n".encode("ascii"))
216
+ yield
217
+ finally:
218
+ os.close(handle)
219
+ self._run_lock.release()
220
+
221
+ def index_directory(
222
+ self, directory: Path, should_stop: Callable[[], bool] | None = None
223
+ ) -> IndexReport:
224
+ """Index new/changed files, skip unchanged ones, purge files that disappeared.
225
+
226
+ A failure in one file is recorded in the report and does not abort the run.
227
+ ``should_stop`` is asked between documents; once it answers true the run raises
228
+ `IndexCancelled` and leaves what a killed run leaves.
229
+ """
230
+ try:
231
+ root = directory.expanduser().resolve(strict=True)
232
+ except (OSError, RuntimeError, ValueError) as exc: # missing, ~unknown, NUL, loop
233
+ raise IndexingError(f"Directory does not exist: {directory}") from exc
234
+ if not root.is_dir():
235
+ raise IndexingError(f"Not a directory: {root}")
236
+ try:
237
+ str(root).encode("utf-8")
238
+ except UnicodeEncodeError:
239
+ raise IndexingError(
240
+ "Directory name is not valid UTF-8 and cannot be indexed: "
241
+ f"{discovery._printable(str(root))}"
242
+ ) from None
243
+
244
+ with self._scan_lock():
245
+ started = time.perf_counter()
246
+ # Set at the first write of the run, not here: a run that changes nothing has
247
+ # no business retracting a certificate that is still true. Once it does write,
248
+ # the certificate stays retracted until a full pass finishes - so a run killed
249
+ # partway leaves the tree honestly described as unvouched-for, with no marker
250
+ # to clean up and nothing to go stale.
251
+ retracted = False
252
+
253
+ def about_to_write() -> None:
254
+ nonlocal retracted
255
+ if not retracted:
256
+ self._db.mark_scan_started(str(root))
257
+ retracted = True
258
+
259
+ previous_model = self._db.get_meta(_MODEL_META_KEY)
260
+ renamed = previous_model not in {None, self._embedder.model_name}
261
+ if renamed and self._embedder.weights_revision is None:
262
+ # A model that names its weights only once loaded is loaded now: this run
263
+ # re-embeds everything anyway, and knowing the weights is what spares the
264
+ # index the discard below. One that will not load is discarded as before.
265
+ with contextlib.suppress(ModelLoadError):
266
+ self._embedder.warm_up()
267
+ if renamed and self._embedder.weights_revision is None:
268
+ # Vectors from different models are not comparable, and they share one
269
+ # vector table. A model that names its weights up front leaves this to the
270
+ # per-document stamps, which re-embed each document in place while keyword
271
+ # search keeps answering; one that cannot would leave nothing to tell old
272
+ # vectors from new, so every document has to go, not only those under
273
+ # `root`. Nothing is announced when a size change already emptied it.
274
+ self._db.clear(
275
+ notice=lambda discarded: (
276
+ f"Embedding model changed ({previous_model} -> "
277
+ f"{self._embedder.model_name}): discarded all {discarded} previously "
278
+ "indexed documents from every directory. Re-run index_directory for "
279
+ "any other documentation root."
280
+ )
281
+ )
282
+ self._db.set_meta(_MODEL_META_KEY, self._embedder.model_name)
283
+ # Whatever emptied the index (new format, new vector size, new model) left a
284
+ # notice. They are dismissed only once the report carrying them exists: a run
285
+ # that aborts - the model cannot be loaded - leaves them for the next one.
286
+ notices = self._db.pending_notices()
287
+ # Captured here, after this run has done its own discarding and just before it
288
+ # reads the hashes it will trust: a discard *after* this point means the walk
289
+ # measured a database that no longer exists. Captured any earlier and the run
290
+ # counts its own model-change wipe as somebody else's, then refuses to certify
291
+ # the index it just rebuilt from scratch.
292
+ generation = self._db.generation()
293
+ identity = self._run_identity()
294
+ known_hashes = self._db.document_hashes(str(root))
295
+ weights_settled = False
296
+
297
+ seen: set[str] = set()
298
+ indexed = unchanged = sections_indexed = passages_indexed = 0
299
+ failures: list[FileFailure] = []
300
+ unreadable: list[str] = []
301
+
302
+ def record_unreadable(error: OSError) -> None:
303
+ location = str(error.filename or root)
304
+ unreadable.append(location)
305
+ failures.append(
306
+ FileFailure(
307
+ file_path=discovery._printable(location),
308
+ message=f"Cannot list directory: {error.strerror or error}",
309
+ )
310
+ )
311
+
312
+ def store(prepared: _Prepared) -> None:
313
+ nonlocal weights_settled, indexed, unchanged, sections_indexed
314
+ nonlocal passages_indexed
315
+ if prepared.unchanged:
316
+ # Not an index: the document stands, and only the time it was last
317
+ # written is brought up to date. The certificate is not retracted for
318
+ # it either - nothing an answer is drawn from has changed.
319
+ self._db.record_modification_time(
320
+ prepared.file_path,
321
+ prepared.content_hash,
322
+ prepared.previous_mtime_ns,
323
+ prepared.mtime_ns,
324
+ )
325
+ unchanged += 1
326
+ return
327
+ if prepared.has_vectors and not weights_settled:
328
+ # Once per run, on the one thread that writes, and only once a vector
329
+ # really exists to be written. A document of headings alone produces
330
+ # none, and asking would make a file that needs no model fail when no
331
+ # model can be loaded. The embedding is already spent by the time a
332
+ # refusal lands, but nothing is stored, which is what the guard is for.
333
+ self._settle_weights()
334
+ weights_settled = True
335
+ # Here, and not in the worker: parsing and embedding can fail without
336
+ # touching the index, and a run that changed nothing must leave a standing
337
+ # certificate alone.
338
+ about_to_write()
339
+ self._db.replace_document(
340
+ file_path=prepared.file_path,
341
+ title=prepared.title,
342
+ content_hash=prepared.content_hash,
343
+ last_modified=prepared.last_modified,
344
+ mtime_ns=prepared.mtime_ns,
345
+ sections=prepared.sections,
346
+ vectors=prepared.vectors,
347
+ weights_revision=self._embedder.weights_revision,
348
+ )
349
+ indexed += 1
350
+ sections_indexed += prepared.counts[0]
351
+ passages_indexed += prepared.counts[1]
352
+
353
+ # Workers embed; this thread writes. Embedding is ~17 s per file and a write
354
+ # is under 5 ms, so nothing is gained by letting workers write and a great
355
+ # deal is given up: SQLite takes one writer at a time anyway, and section ids
356
+ # are allocated as documents are stored. Results are therefore drained in
357
+ # submission order - `search.py` breaks a scoring tie by section id, so ids
358
+ # handed out in some completion order would quietly reorder equal hits, and
359
+ # no later sort can give them back. The window bounds what is held in memory:
360
+ # a prepared file carries every vector of every passage it has.
361
+ files = discovery.iter_markdown_files(root, record_unreadable, self._exclude)
362
+ window = max(2 * self._workers, 2)
363
+ pending: deque[tuple[str, Future[_Prepared | None]]] = deque()
364
+ with ThreadPoolExecutor(
365
+ max_workers=self._workers, thread_name_prefix="markdown-memory-index"
366
+ ) as pool:
367
+ try:
368
+ exhausted = False
369
+ while True:
370
+ if should_stop is not None and should_stop():
371
+ raise IndexCancelled("index run stopped by its owner")
372
+ while not exhausted and len(pending) < window:
373
+ path = next(files, None)
374
+ if path is None:
375
+ exhausted = True
376
+ break
377
+ file_path = str(path)
378
+ seen.add(file_path)
379
+ pending.append(
380
+ (
381
+ file_path,
382
+ pool.submit(
383
+ self._prepare_file,
384
+ path,
385
+ known_hashes.get(file_path),
386
+ identity,
387
+ ),
388
+ )
389
+ )
390
+ if not pending:
391
+ break
392
+ file_path, future = pending.popleft()
393
+ try:
394
+ prepared = future.result()
395
+ # The write is inside the same guard as the read: storing one
396
+ # document can fail on its own - a vector the storage layer
397
+ # rejects, a row that will not go in - and that is this file's
398
+ # failure to carry, not the run's to die of.
399
+ if prepared is None:
400
+ unchanged += 1
401
+ elif should_stop is not None and should_stop():
402
+ # Checked again after the wait: embedding one file can take
403
+ # seconds, and a stop asked meanwhile writes nothing more.
404
+ raise IndexCancelled("index run stopped by its owner")
405
+ else:
406
+ store(prepared)
407
+ except (ModelLoadError, ForeignWeightsError):
408
+ raise # not this file's fault: every other file fails the same
409
+ except (MarkdownMemoryError, OSError) as exc:
410
+ logger.warning(
411
+ "Failed to index %s: %s", discovery._printable(file_path), exc
412
+ )
413
+ failures.append(
414
+ FileFailure(
415
+ file_path=discovery._printable(file_path), message=str(exc)
416
+ )
417
+ )
418
+ continue
419
+ except BaseException:
420
+ # Whatever has not started will not start. What is already running is
421
+ # joined by the pool on the way out; there is nowhere to put its result.
422
+ for _, queued in pending:
423
+ queued.cancel()
424
+ raise
425
+
426
+ vanished = self._vanished(root, known_hashes, seen, unreadable)
427
+ if vanished:
428
+ about_to_write() # deleting is changing it, even if no file was read
429
+ purged = self._db.delete_documents(vanished)
430
+ # This run's own failures are already in `errors`, and the search tools read
431
+ # the recorded ones straight from the database, so nothing is added to
432
+ # `notes`: a warning repeated in three places is how a warning becomes noise.
433
+ # Only what this walk could have reached: a failure inside a pruned directory
434
+ # or one that could not be listed is not this run's to forget, however far
435
+ # under its root it sits.
436
+ reachable = self._reachable(root, self._db.failure_paths(str(root)), unreadable)
437
+ self._db.record_failures(
438
+ reachable,
439
+ {
440
+ failure.file_path: (
441
+ f"{failure.message} (indexing {discovery._printable(str(root))})"
442
+ )
443
+ for failure in failures
444
+ },
445
+ )
446
+ # Whole-database, so it cannot vouch for rows this walk never reached.
447
+ self._db.settle_weights(self._embedder.weights_revision)
448
+ # The walk finished, which is all this records; what it could not read is
449
+ # recorded separately, and `index_status` refuses to call a tree whole while
450
+ # anything under it is still listed there. Two facts, two places, one answer.
451
+ self._db.mark_scan_complete(str(root), generation)
452
+ report = IndexReport(
453
+ directory=discovery._printable(str(root)),
454
+ files_scanned=len(seen),
455
+ files_indexed=indexed,
456
+ files_unchanged=unchanged,
457
+ files_purged=purged,
458
+ sections_indexed=sections_indexed,
459
+ passages_indexed=passages_indexed,
460
+ elapsed_seconds=time.perf_counter() - started,
461
+ errors=tuple(failures),
462
+ notes=tuple(notices.values()),
463
+ )
464
+ self._db.dismiss_notices(notices)
465
+ logger.info(report.summary())
466
+ return report
467
+
468
+ def _run_identity(self) -> str | None:
469
+ """The weights this run embeds with, when they can be known before it embeds.
470
+
471
+ A model that names its weights without loading (EmbeddingGemma) always answers.
472
+ One that learns them by loading (bge-small) is loaded here only when a repair is
473
+ pending - the index is being re-embedded, disagrees with some model, or holds
474
+ vectors no revision vouches for - because such a run has embedding to do anyway;
475
+ otherwise a run that changes nothing would pay for a model load. None means
476
+ documents are compared on content and format alone, as before stamps existed; a
477
+ stale one left behind that way keeps the certificate withheld at the end of the
478
+ run, and the next run repairs it.
479
+ """
480
+ recorded = self._db.get_meta(WEIGHTS_META_KEY)
481
+ if self._embedder.weights_revision is None and (
482
+ recorded == WEIGHTS_REVOKED
483
+ or self._db.get_meta(WEIGHTS_MISMATCH_KEY) is not None
484
+ or (recorded is None and self._db.count_rows("units_vec") > 0)
485
+ ):
486
+ try:
487
+ self._embedder.warm_up()
488
+ except ModelLoadError as exc:
489
+ # Degrade rather than fail: a run whose files need no embedding can still
490
+ # finish, and the recorded message keeps saying what is wrong.
491
+ logger.warning("Model not loaded, stored weights not compared: %s", exc)
492
+ return self._embedder.weights_revision
493
+
494
+ def _settle_weights(self) -> None:
495
+ """Make sure the index says it is being re-embedded before other weights write in.
496
+
497
+ Called once per run, from the driver, at the first document that really has
498
+ vectors: the earliest moment a lazily-loaded embedder can be asked what it is
499
+ without making a run that needs no model load one, and on the only thread allowed
500
+ to write what the answer implies. Weights that differ from the recorded ones - or
501
+ any known weights joining vectors nobody vouched for - revoke the certificate and
502
+ carry on: every search, whatever model it runs, then ranks on keywords alone until
503
+ `settle_weights` finds every vector-bearing document stamped with one revision.
504
+ Nothing is discarded first, so keyword search answers throughout, and a run killed
505
+ partway resumes where it stopped, because each document's stamp is written with
506
+ its vectors.
507
+
508
+ Weights that cannot be named still refuse: vectors written now would be
509
+ indistinguishable from the ones already stored, and no stamp could repair that.
510
+ """
511
+ if self._db.count_rows("units_vec") == 0:
512
+ # No vector here for any of this to be about. Documents are the wrong
513
+ # question: a file of nothing but headings is stored and embeds nothing, so an
514
+ # index can hold documents and no vectors at all. Whatever this run is about
515
+ # to write is therefore the whole of it, and it may say so - before the write
516
+ # rather than after, so that another process reading these vectors a moment
517
+ # from now finds them labelled. A run that dies in between leaves a revision
518
+ # over no vectors, which the next one clears exactly here.
519
+ self._claim_empty_index()
520
+ return
521
+ recorded = self._db.get_meta(WEIGHTS_META_KEY)
522
+ # Already loaded: a worker embedded the document this is about to store. Kept
523
+ # anyway, because `warm_up` is what makes `weights_revision` answerable and this
524
+ # is called from tests and from runs whose first document came from a cache.
525
+ self._embedder.warm_up()
526
+ weights = self._embedder.weights_revision
527
+ if weights is None:
528
+ if recorded in {None, WEIGHTS_REVOKED}:
529
+ # No provenance to contradict, or none left to protect: a revoked index is
530
+ # already ranked by keyword alone, and only a run whose weights have a name
531
+ # can stamp its way back out of that.
532
+ return
533
+ # The model loaded, so something answered - it just cannot say which weights
534
+ # it is. That is not "nothing to compare": the vectors written now would be
535
+ # unlabelled and indistinguishable from the ones already stored, which is the
536
+ # state this guard exists to prevent.
537
+ message = (
538
+ f"Which weights {self._embedder.model_name} is running could not be read, so "
539
+ "there is no way to tell whether they are the ones that built this index "
540
+ f"({short_weights(recorded)}). Nothing has been discarded and no vector has been "
541
+ "stored - a document of headings alone, which embeds nothing, may have "
542
+ "been updated before this was reached; repair the model cache and run "
543
+ "index_directory again."
544
+ )
545
+ self._db.record_weights_mismatch(message)
546
+ self._db.revoke_coverage()
547
+ raise ForeignWeightsError(message)
548
+ if weights == recorded:
549
+ self._db.record_weights_mismatch(None)
550
+ return
551
+ self._db.revoke_weights(
552
+ f"{self._embedder.model_name} is re-embedding this index with other weights "
553
+ f"({short_weights(weights)}). Only keyword ranking is used until every "
554
+ "document has been re-embedded; semantic ranking resumes when index_directory "
555
+ "finishes."
556
+ )
557
+
558
+ def _claim_empty_index(self) -> None:
559
+ """Take ownership of an index that holds no vectors, and drop what described none.
560
+
561
+ A model *name* is not enough for bge-small: fastembed pins no revision, so a
562
+ re-download can bring different weights under the same name and nothing in the
563
+ index would notice. This is the one moment a revision can honestly be written -
564
+ the vectors that follow are all there will be, and there are none yet to
565
+ contradict. A model that cannot say which weights it is records nothing, and the
566
+ index carries no provenance rather than a provenance that might be wrong.
567
+ """
568
+ self._db.forget_weights_revision()
569
+ self._db.record_weights_mismatch(None)
570
+ self._embedder.warm_up()
571
+ weights = self._embedder.weights_revision
572
+ if weights is not None:
573
+ self._db.set_meta(WEIGHTS_META_KEY, weights)
574
+
575
+ def _reachable(self, root: Path, paths: Sequence[str], unreadable: Sequence[str]) -> list[str]:
576
+ """The subset of ``paths`` a walk of ``root`` would have visited.
577
+
578
+ Pruned directories (`.venv`, `node_modules`), directories excluded by
579
+ configuration, and directories that could not be listed are never entered, so this
580
+ run saw nothing inside them and may not speak for what it did not see.
581
+
582
+ Every component is tested, including the last. A recorded failure is usually a
583
+ file, but an unreadable *directory* is recorded under its own path - and dropping
584
+ the final component would ask whether `.venv`'s parent is walkable rather than
585
+ whether `.venv` is, and then clear it.
586
+
587
+ A path out of the walk's sight is still retired once it is observably gone
588
+ (`discovery._certainly_gone`), or its row would outlive the file and no run could ever
589
+ retire it.
590
+ """
591
+ blocked = tuple(location.rstrip(os.sep) + os.sep for location in unreadable)
592
+ visitable = []
593
+ for path in paths:
594
+ if self._walk_would_visit(root, path, blocked) or discovery._certainly_gone(root, path):
595
+ visitable.append(path)
596
+ return visitable
597
+
598
+ def _walk_would_visit(self, root: Path, path: str, blocked: tuple[str, ...]) -> bool:
599
+ """Whether a walk of ``root`` reaches ``path``, given the directories it could not list."""
600
+ if blocked and path.startswith(blocked):
601
+ return False
602
+ if not discovery._is_walkable(os.path.relpath(path, root).split(os.sep)):
603
+ return False
604
+ if discovery._behind_symlink(root, path) or discovery._is_shadowing_symlink(path):
605
+ return False
606
+ return not (self._exclude and discovery._is_excluded(Path(path), root, self._exclude))
607
+
608
+ @staticmethod
609
+ def _vanished(
610
+ root: Path,
611
+ known: Mapping[str, object],
612
+ seen: set[str],
613
+ unreadable: Sequence[str],
614
+ ) -> list[str]:
615
+ """Known documents that this walk *would* have found had they still existed.
616
+
617
+ A document is only purged when its absence is evidence of deletion. It is kept
618
+ when the walk could not have reached it: it lives under a directory that could
619
+ not be listed, or under a pruned tree (``node_modules`` ...) that was indexed
620
+ explicitly by pointing ``index_directory`` inside it.
621
+
622
+ Unless the file is observably gone (`discovery._certainly_gone`). Not being visited is not
623
+ evidence of deletion; `ENOENT` on that one name is exactly that evidence, and
624
+ without it a deleted document under a pruned tree keeps answering searches with
625
+ text that is not on disk any more, until someone re-indexes that tree by hand.
626
+ """
627
+ blocked = tuple(location.rstrip(os.sep) + os.sep for location in unreadable)
628
+ vanished: list[str] = []
629
+ for file_path in sorted(set(known) - seen):
630
+ if discovery._certainly_gone(root, file_path):
631
+ vanished.append(file_path)
632
+ continue
633
+ if blocked and file_path.startswith(blocked):
634
+ continue
635
+ relative = os.path.relpath(file_path, root)
636
+ if not discovery._is_walkable(relative.split(os.sep)[:-1]):
637
+ continue
638
+ if discovery._behind_symlink(root, file_path):
639
+ continue
640
+ vanished.append(file_path)
641
+ return vanished
642
+
643
+ def _prepare_file(
644
+ self,
645
+ path: Path,
646
+ known: tuple[str, int, int | None, str | None] | None,
647
+ identity: str | None,
648
+ ) -> _Prepared | None:
649
+ """Read, parse and embed one file. ``None`` when it is unchanged.
650
+
651
+ Runs on a worker thread and writes nothing: every database write of a run belongs
652
+ to the driver, so that section ids are handed out in walk order and the index's
653
+ account of itself - certificate, provenance, failures - has a single author.
654
+ """
655
+ file_path = str(path)
656
+ try:
657
+ file_path.encode("utf-8")
658
+ except UnicodeEncodeError:
659
+ raise IndexingError("File name is not valid UTF-8; skipped") from None
660
+ info = path.stat()
661
+ if not stat.S_ISREG(info.st_mode):
662
+ # A FIFO or device named *.md would block or stream forever when read.
663
+ raise IndexingError("Not a regular file; skipped")
664
+ # Checked again on the descriptor: between that stat and this open the path can be
665
+ # replaced by a FIFO, and a blocking open would then wait for a writer that may
666
+ # never come - with a worker of the pool in its hand.
667
+ data = discovery.read_regular_file(path)
668
+ if data is None:
669
+ raise IndexingError("Not a regular file; skipped")
670
+ if len(data) > discovery.MAX_FILE_BYTES:
671
+ raise IndexingError(f"File is larger than {discovery.MAX_FILE_BYTES} bytes; skipped")
672
+ content_hash = discovery.hash_bytes(data)
673
+ # The format counts as much as the content: a file whose bytes never changed still
674
+ # has to be rebuilt if its vectors were pooled by an older scheme, or it would keep
675
+ # them forever and the table would answer one query two different ways. So do the
676
+ # weights, once this run knows its own: a document stamped by other ones is
677
+ # re-embedded in place, which is how a changed model repairs the index file by file.
678
+ if (
679
+ known is not None
680
+ and known[:2] == (content_hash, VECTOR_FORMAT)
681
+ and (identity is None or known[3] == identity)
682
+ ):
683
+ if known[2] == info.st_mtime_ns:
684
+ return None
685
+ # Same bytes, a different timestamp: nothing to parse, embed or store, but the
686
+ # time has to be written down or the freshness check hashes this file again on
687
+ # every sweep from here on.
688
+ return _Prepared(
689
+ file_path=file_path,
690
+ title="",
691
+ content_hash=content_hash,
692
+ last_modified=int(info.st_mtime),
693
+ mtime_ns=info.st_mtime_ns,
694
+ sections=(),
695
+ vectors=(),
696
+ unchanged=True,
697
+ previous_mtime_ns=known[2],
698
+ )
699
+ parsed = self._parser.parse(
700
+ data.decode("utf-8", errors="replace"), fallback_title=path.stem
701
+ )
702
+ # One embedding call per file, over the passages alone. The section vector is the
703
+ # mean of its passages rather than a separate embedding of the whole section:
704
+ # that text ran past the model's 512-token limit for 7.9% of the vendored corpus
705
+ # and lost its tail, and embedding it cost one extra call per section.
706
+ texts: list[str] = []
707
+ for section in parsed.sections:
708
+ texts.extend(section.unit_texts)
709
+ embeddings = self._embedder.embed_documents(texts)
710
+ if len(embeddings) != len(texts):
711
+ raise EmbeddingError(f"Got {len(embeddings)} vectors for {len(texts)} texts")
712
+ embedded = iter(embeddings)
713
+ vectors = []
714
+ for section in parsed.sections:
715
+ units = tuple(next(embedded) for _ in section.units)
716
+ vectors.append(SectionVectors(section=_section_vector(units), units=units))
717
+ return _Prepared(
718
+ file_path=file_path,
719
+ title=parsed.title,
720
+ content_hash=content_hash,
721
+ last_modified=int(info.st_mtime),
722
+ mtime_ns=info.st_mtime_ns,
723
+ sections=tuple(parsed.sections),
724
+ vectors=tuple(vectors),
725
+ )