markdown-memory 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- markdown_memory/__init__.py +47 -0
- markdown_memory/autoindex.py +170 -0
- markdown_memory/config.py +235 -0
- markdown_memory/db.py +1546 -0
- markdown_memory/discovery.py +195 -0
- markdown_memory/embedders.py +513 -0
- markdown_memory/exceptions.py +71 -0
- markdown_memory/freshness.py +138 -0
- markdown_memory/headings.py +185 -0
- markdown_memory/indexer.py +725 -0
- markdown_memory/model_cache.py +272 -0
- markdown_memory/models.py +339 -0
- markdown_memory/parser.py +869 -0
- markdown_memory/py.typed +0 -0
- markdown_memory/search.py +518 -0
- markdown_memory/server.py +520 -0
- markdown_memory-0.1.0.dist-info/METADATA +579 -0
- markdown_memory-0.1.0.dist-info/RECORD +21 -0
- markdown_memory-0.1.0.dist-info/WHEEL +4 -0
- markdown_memory-0.1.0.dist-info/entry_points.txt +3 -0
- markdown_memory-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,513 @@
|
|
|
1
|
+
"""The embedders: two local ONNX models behind one protocol.
|
|
2
|
+
|
|
3
|
+
``EmbeddingGemmaEmbedder`` (the default) is what retrieval quality was tuned on;
|
|
4
|
+
``FastEmbedEmbedder`` (bge-small) is the light option: a tenth of the size and ~25x
|
|
5
|
+
faster, at a clear cost in recall on paraphrased queries. ``numpy`` and ``onnxruntime``
|
|
6
|
+
live here and nowhere else in the package.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import contextlib
|
|
12
|
+
import logging
|
|
13
|
+
import os
|
|
14
|
+
import threading
|
|
15
|
+
import time
|
|
16
|
+
from collections.abc import Sequence
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import TYPE_CHECKING, Protocol
|
|
19
|
+
|
|
20
|
+
from markdown_memory import model_cache
|
|
21
|
+
from markdown_memory.db import DEFAULT_EMBEDDING_DIM
|
|
22
|
+
from markdown_memory.exceptions import (
|
|
23
|
+
EmbeddingError,
|
|
24
|
+
IndexingError,
|
|
25
|
+
MarkdownMemoryError,
|
|
26
|
+
ModelLoadError,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
if TYPE_CHECKING:
|
|
30
|
+
from fastembed import TextEmbedding
|
|
31
|
+
from tokenizers import Tokenizer
|
|
32
|
+
|
|
33
|
+
logger = logging.getLogger(__name__)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
DEFAULT_EMBEDDER = "embeddinggemma"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
# bge v1.5 is asymmetric: queries (never passages) need this instruction, which
|
|
40
|
+
# fastembed's query_embed() does not add.
|
|
41
|
+
_BGE_QUERY_INSTRUCTION = "Represent this sentence for searching relevant passages: "
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
GEMMA_DIMENSION = 768
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
GEMMA_MAX_TOKENS = 512
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
# Prompts from the EmbeddingGemma model card; the model is trained to expect them.
|
|
51
|
+
GEMMA_QUERY_PROMPT = "task: search result | query: "
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
GEMMA_DOCUMENT_PROMPT = "title: none | text: "
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
# Measured, and kept at 4 deliberately. Embedding passages in isolation, a larger batch
|
|
58
|
+
# looks much faster; indexing a real directory it is not, because each file is embedded on
|
|
59
|
+
# its own and its sections and passages differ enough in length that the padding eats the
|
|
60
|
+
# gain. End to end over the same corpus: batch 4 gave 3.76 vectors/s at 1,639 MB peak RSS,
|
|
61
|
+
# batch 16 gave 4.41 vectors/s at 2,780 MB. +17% throughput does not buy +1.1 GB on a tool
|
|
62
|
+
# that runs beside an editor. Sorting by token count instead of characters was measured
|
|
63
|
+
# too: worth ~30% at batch 4 only, which a second tokenisation pass cancels out.
|
|
64
|
+
_GEMMA_BATCH_SIZE = 4
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
# Thread count is left to onnxruntime; what is *not* left to it is spinning (see
|
|
68
|
+
# _SPIN_CONFIG). Pinning the count was measured from 4 to 16 threads and every value sat
|
|
69
|
+
# inside the run-to-run noise on wall time. That measurement missed the cost that matters
|
|
70
|
+
# for a tool running beside an editor: with spinning off, 16 threads and 4 threads differ
|
|
71
|
+
# by ~30% of CPU and nothing in wall time, so the count stays onnxruntime's business and
|
|
72
|
+
# only a machine that disagrees with it needs MARKDOWN_MEMORY_THREADS.
|
|
73
|
+
_THREADS_ENV = "MARKDOWN_MEMORY_THREADS"
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
# onnxruntime's intra-op threads spin-wait between operators by default. That is a good
|
|
77
|
+
# trade for a server answering back-to-back requests and a bad one here: measured on a
|
|
78
|
+
# 16-core machine, one warm query cost 7.2 s of CPU across 16 spinning threads, and the
|
|
79
|
+
# pool kept burning ~0.5 core-seconds per second *after* the query returned. Turning
|
|
80
|
+
# spinning off made the same query 0.6 s of CPU and ~40% faster in wall time, because the
|
|
81
|
+
# spinners were competing with the thread doing the work. Queries here arrive seconds
|
|
82
|
+
# apart, so the wake-up cost spinning buys is never recovered.
|
|
83
|
+
_SPIN_CONFIG = ("session.intra_op.allow_spinning", "0")
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
_EMBED_BATCH_SIZE = 32
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def short_weights(identity: str | None) -> str:
|
|
90
|
+
"""A weights identity short enough for a message, keeping what distinguishes it.
|
|
91
|
+
|
|
92
|
+
Twelve characters of the revision used to be enough. It is not any more: one revision
|
|
93
|
+
publishes several graphs, so both sides of "the weights changed from X to Y" would
|
|
94
|
+
print the same string and the message would read as nonsense while being true.
|
|
95
|
+
"""
|
|
96
|
+
if not identity:
|
|
97
|
+
return "no readable revision"
|
|
98
|
+
revision, separator, graph = identity.partition("/")
|
|
99
|
+
return f"{revision[:12]}/{graph}" if separator else revision[:12]
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
class Embedder(Protocol):
|
|
103
|
+
"""Turns text into fixed-size vectors. Implementations must be thread-safe."""
|
|
104
|
+
|
|
105
|
+
@property
|
|
106
|
+
def model_name(self) -> str: ...
|
|
107
|
+
|
|
108
|
+
@property
|
|
109
|
+
def dimension(self) -> int: ...
|
|
110
|
+
|
|
111
|
+
@property
|
|
112
|
+
def weights_revision(self) -> str | None: ...
|
|
113
|
+
|
|
114
|
+
def warm_up(self) -> None: ...
|
|
115
|
+
|
|
116
|
+
def embed_documents(self, texts: Sequence[str]) -> list[list[float]]: ...
|
|
117
|
+
|
|
118
|
+
def embed_query(self, text: str) -> list[float]: ...
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
class FastEmbedEmbedder:
|
|
122
|
+
"""Local ONNX embeddings via ``fastembed``; the model loads lazily on first use."""
|
|
123
|
+
|
|
124
|
+
def __init__(
|
|
125
|
+
self,
|
|
126
|
+
model_name: str = model_cache.BGE_SMALL_MODEL_NAME,
|
|
127
|
+
*,
|
|
128
|
+
cache_dir: Path | None = None,
|
|
129
|
+
dimension: int = DEFAULT_EMBEDDING_DIM,
|
|
130
|
+
) -> None:
|
|
131
|
+
self._model_name = model_name
|
|
132
|
+
self._cache_dir = cache_dir
|
|
133
|
+
self._dimension = dimension
|
|
134
|
+
self._lock = threading.Lock()
|
|
135
|
+
self._model: TextEmbedding | None = None
|
|
136
|
+
self._weights_revision: str | None = None
|
|
137
|
+
|
|
138
|
+
@property
|
|
139
|
+
def model_name(self) -> str:
|
|
140
|
+
return self._model_name
|
|
141
|
+
|
|
142
|
+
@property
|
|
143
|
+
def dimension(self) -> int:
|
|
144
|
+
return self._dimension
|
|
145
|
+
|
|
146
|
+
@property
|
|
147
|
+
def weights_revision(self) -> str | None:
|
|
148
|
+
"""Which snapshot the loaded weights came from, or None before they are loaded.
|
|
149
|
+
|
|
150
|
+
fastembed pins no revision, so a deleted cache can come back with different
|
|
151
|
+
weights under an unchanged model name - and stored passage vectors would then be
|
|
152
|
+
compared against query vectors from a different model, with nothing to notice it.
|
|
153
|
+
huggingface_hub records the snapshot it fetched in `refs/main`.
|
|
154
|
+
|
|
155
|
+
Read once, when the model loads, and not on every call: the file can change under
|
|
156
|
+
a running process, and what matters is the weights that produced the vectors, not
|
|
157
|
+
whatever is on disk by the time somebody asks.
|
|
158
|
+
"""
|
|
159
|
+
return self._weights_revision
|
|
160
|
+
|
|
161
|
+
def _read_weights_revision(self) -> str | None:
|
|
162
|
+
if self._model_name != model_cache.BGE_SMALL_MODEL_NAME:
|
|
163
|
+
# `model_cache.fastembed_model_dir` resolves one model's folder. Reading it for a
|
|
164
|
+
# different model would report a revision belonging to weights that are not
|
|
165
|
+
# the ones answering - worse than reporting none, which is merely unknown.
|
|
166
|
+
return None
|
|
167
|
+
directory = model_cache.fastembed_model_dir(self._cache_dir)
|
|
168
|
+
if directory is None:
|
|
169
|
+
return None
|
|
170
|
+
try:
|
|
171
|
+
return (directory / "refs" / "main").read_text(encoding="utf-8").strip() or None
|
|
172
|
+
except OSError:
|
|
173
|
+
return None
|
|
174
|
+
|
|
175
|
+
def warm_up(self) -> None:
|
|
176
|
+
"""Load (and if necessary download) the model now instead of on first query."""
|
|
177
|
+
self._load()
|
|
178
|
+
|
|
179
|
+
def embed_documents(self, texts: Sequence[str]) -> list[list[float]]:
|
|
180
|
+
if not texts:
|
|
181
|
+
return []
|
|
182
|
+
model = self._load()
|
|
183
|
+
try:
|
|
184
|
+
vectors = [
|
|
185
|
+
_to_floats(vector.tolist())
|
|
186
|
+
for vector in model.embed(list(texts), batch_size=_EMBED_BATCH_SIZE)
|
|
187
|
+
]
|
|
188
|
+
except Exception as exc: # onnxruntime raises a variety of unrelated types
|
|
189
|
+
raise EmbeddingError(f"Embedding {len(texts)} passages failed: {exc}") from exc
|
|
190
|
+
return self._validated(vectors, expected=len(texts))
|
|
191
|
+
|
|
192
|
+
def embed_query(self, text: str) -> list[float]:
|
|
193
|
+
model = self._load()
|
|
194
|
+
try:
|
|
195
|
+
prompt = _BGE_QUERY_INSTRUCTION if "bge-" in self._model_name else ""
|
|
196
|
+
vectors = [_to_floats(vector.tolist()) for vector in model.query_embed(prompt + text)]
|
|
197
|
+
except Exception as exc:
|
|
198
|
+
raise EmbeddingError(f"Embedding the query failed: {exc}") from exc
|
|
199
|
+
return self._validated(vectors, expected=1)[0]
|
|
200
|
+
|
|
201
|
+
def _validated(self, vectors: list[list[float]], *, expected: int) -> list[list[float]]:
|
|
202
|
+
if len(vectors) != expected:
|
|
203
|
+
raise EmbeddingError(f"Model returned {len(vectors)} vectors for {expected} texts")
|
|
204
|
+
for vector in vectors:
|
|
205
|
+
if len(vector) != self._dimension:
|
|
206
|
+
raise EmbeddingError(
|
|
207
|
+
f"Model {self._model_name} produced {len(vector)}-dimensional vectors, "
|
|
208
|
+
f"expected {self._dimension}"
|
|
209
|
+
)
|
|
210
|
+
return vectors
|
|
211
|
+
|
|
212
|
+
def _load(self) -> TextEmbedding:
|
|
213
|
+
with self._lock:
|
|
214
|
+
if self._model is None:
|
|
215
|
+
started = time.perf_counter()
|
|
216
|
+
try:
|
|
217
|
+
from fastembed import TextEmbedding # heavy import: defer until needed
|
|
218
|
+
|
|
219
|
+
if self._cache_dir is not None:
|
|
220
|
+
self._cache_dir.mkdir(parents=True, exist_ok=True)
|
|
221
|
+
self._model = TextEmbedding(
|
|
222
|
+
model_name=self._model_name,
|
|
223
|
+
cache_dir=None if self._cache_dir is None else str(self._cache_dir),
|
|
224
|
+
# fastembed builds its own session, so the override reaches it only
|
|
225
|
+
# through this argument; without it MARKDOWN_MEMORY_THREADS was
|
|
226
|
+
# documented but ignored for this preset. fastembed exposes no
|
|
227
|
+
# spinning switch (its add_extra_session_options knows only
|
|
228
|
+
# enable_cpu_mem_arena), so capping the threads is the whole lever
|
|
229
|
+
# here: at 4, a query cost 95 ms of CPU instead of 718 ms.
|
|
230
|
+
threads=_inference_threads() or None,
|
|
231
|
+
)
|
|
232
|
+
self._weights_revision = self._read_weights_revision()
|
|
233
|
+
except Exception as exc:
|
|
234
|
+
raise ModelLoadError(
|
|
235
|
+
f"Cannot load embedding model {self._model_name}: {exc}"
|
|
236
|
+
) from exc
|
|
237
|
+
logger.info(
|
|
238
|
+
"Loaded embedding model %s in %.2fs",
|
|
239
|
+
self._model_name,
|
|
240
|
+
time.perf_counter() - started,
|
|
241
|
+
)
|
|
242
|
+
return self._model
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
class EmbeddingGemmaEmbedder:
|
|
246
|
+
"""Google's EmbeddingGemma-300m (quantized ONNX, 768 dimensions) run with onnxruntime.
|
|
247
|
+
|
|
248
|
+
The weights live in an external data file next to the graph. onnxruntime refuses
|
|
249
|
+
such a file when it is a symlink out of the model directory - which is how the
|
|
250
|
+
Hugging Face cache stores it - so the files are downloaded as real files into
|
|
251
|
+
``cache_dir``. Nothing is fetched when they are already there.
|
|
252
|
+
"""
|
|
253
|
+
|
|
254
|
+
def __init__(self, *, cache_dir: Path | None = None) -> None:
|
|
255
|
+
self._cache_dir = cache_dir
|
|
256
|
+
self._model_dir = model_cache.gemma_model_dir(cache_dir)
|
|
257
|
+
self._lock = threading.Lock()
|
|
258
|
+
self._session: _OrtSession | None = None
|
|
259
|
+
self._tokenizer: Tokenizer | None = None
|
|
260
|
+
|
|
261
|
+
@property
|
|
262
|
+
def model_name(self) -> str:
|
|
263
|
+
return (
|
|
264
|
+
f"{model_cache.GEMMA_REPOSITORY}@{model_cache.GEMMA_REVISION[:12]}"
|
|
265
|
+
f"/{model_cache.GEMMA_MODEL_FILE}"
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
@property
|
|
269
|
+
def dimension(self) -> int:
|
|
270
|
+
return GEMMA_DIMENSION
|
|
271
|
+
|
|
272
|
+
@property
|
|
273
|
+
def weights_revision(self) -> str | None:
|
|
274
|
+
"""Which weights these vectors came from - the revision *and* the graph.
|
|
275
|
+
|
|
276
|
+
The revision alone is not enough. This repository publishes several graphs at one
|
|
277
|
+
revision, and they do not agree: swapping the int8 graph for the 4-bit one moves a
|
|
278
|
+
query's vector by about 0.03 cosine, which is far more than the distance search
|
|
279
|
+
ranks on. With the bare revision, an index built by one graph and searched by the
|
|
280
|
+
other passes this check, so the query is embedded by one model and compared
|
|
281
|
+
against another's vectors - no error, just quietly worse answers.
|
|
282
|
+
"""
|
|
283
|
+
return f"{model_cache.GEMMA_REVISION}/{model_cache.GEMMA_MODEL_FILE}"
|
|
284
|
+
|
|
285
|
+
def warm_up(self) -> None:
|
|
286
|
+
"""Download (first run only) and load the model now instead of on first use."""
|
|
287
|
+
self._load()
|
|
288
|
+
self._report_other_versions()
|
|
289
|
+
|
|
290
|
+
def embed_documents(self, texts: Sequence[str]) -> list[list[float]]:
|
|
291
|
+
return self._embed([GEMMA_DOCUMENT_PROMPT + text for text in texts])
|
|
292
|
+
|
|
293
|
+
def embed_query(self, text: str) -> list[float]:
|
|
294
|
+
return self._embed([GEMMA_QUERY_PROMPT + text])[0]
|
|
295
|
+
|
|
296
|
+
def _embed(self, texts: Sequence[str]) -> list[list[float]]:
|
|
297
|
+
if not texts:
|
|
298
|
+
return []
|
|
299
|
+
session, tokenizer = self._load()
|
|
300
|
+
import numpy as np
|
|
301
|
+
|
|
302
|
+
vectors: list[list[float]] = [[] for _ in texts]
|
|
303
|
+
try:
|
|
304
|
+
# Similar lengths share a batch: padding is wasted compute.
|
|
305
|
+
order = sorted(range(len(texts)), key=lambda index: len(texts[index]))
|
|
306
|
+
for start in range(0, len(order), _GEMMA_BATCH_SIZE):
|
|
307
|
+
batch = order[start : start + _GEMMA_BATCH_SIZE]
|
|
308
|
+
encodings = tokenizer.encode_batch([texts[index] for index in batch])
|
|
309
|
+
outputs = session.run(
|
|
310
|
+
["sentence_embedding"],
|
|
311
|
+
{
|
|
312
|
+
"input_ids": np.array([e.ids for e in encodings], dtype=np.int64),
|
|
313
|
+
"attention_mask": np.array(
|
|
314
|
+
[e.attention_mask for e in encodings], dtype=np.int64
|
|
315
|
+
),
|
|
316
|
+
},
|
|
317
|
+
)[0]
|
|
318
|
+
norms = np.maximum(np.linalg.norm(outputs, axis=-1, keepdims=True), 1e-12)
|
|
319
|
+
for index, vector in zip(batch, (outputs / norms).tolist(), strict=True):
|
|
320
|
+
vectors[index] = _to_floats(vector)
|
|
321
|
+
except Exception as exc: # onnxruntime raises a variety of unrelated types
|
|
322
|
+
raise EmbeddingError(f"Embedding {len(texts)} texts failed: {exc}") from exc
|
|
323
|
+
if any(len(vector) != GEMMA_DIMENSION for vector in vectors):
|
|
324
|
+
raise EmbeddingError(f"Model did not return {GEMMA_DIMENSION}-dimensional vectors")
|
|
325
|
+
return vectors
|
|
326
|
+
|
|
327
|
+
def _load(self) -> tuple[_OrtSession, Tokenizer]:
|
|
328
|
+
with self._lock:
|
|
329
|
+
if self._session is not None and self._tokenizer is not None:
|
|
330
|
+
return self._session, self._tokenizer
|
|
331
|
+
started = time.perf_counter()
|
|
332
|
+
# Two passes at most: the first can find nothing worth trusting, and the
|
|
333
|
+
# second runs on a cache that was repaired under the exclusive lock in
|
|
334
|
+
# between - by this process or by whichever one held the lock first.
|
|
335
|
+
for attempt in range(2):
|
|
336
|
+
with model_cache._model_cache_lock(self._cache_dir, exclusive=False):
|
|
337
|
+
if model_cache._stamp_is_current(self._model_dir):
|
|
338
|
+
self._session, self._tokenizer = self._open()
|
|
339
|
+
logger.info(
|
|
340
|
+
"Loaded embedding model %s in %.2fs",
|
|
341
|
+
model_cache.GEMMA_REPOSITORY,
|
|
342
|
+
time.perf_counter() - started,
|
|
343
|
+
)
|
|
344
|
+
return self._session, self._tokenizer
|
|
345
|
+
if attempt == 0:
|
|
346
|
+
with model_cache._model_cache_lock(self._cache_dir, exclusive=True):
|
|
347
|
+
try:
|
|
348
|
+
self._repair()
|
|
349
|
+
except MarkdownMemoryError:
|
|
350
|
+
raise
|
|
351
|
+
except Exception as exc:
|
|
352
|
+
# Downloading, hashing and writing the stamp all raise things
|
|
353
|
+
# the SDK would hide behind "Error executing tool": a full
|
|
354
|
+
# disk, a revoked token, a read-only cache.
|
|
355
|
+
raise ModelLoadError(
|
|
356
|
+
f"Cannot prepare the model cache at {self._model_dir}: {exc}"
|
|
357
|
+
) from exc
|
|
358
|
+
raise ModelLoadError(
|
|
359
|
+
f"The files under {self._model_dir} still do not match "
|
|
360
|
+
f"{model_cache.GEMMA_REPOSITORY} "
|
|
361
|
+
f"at {model_cache.GEMMA_REVISION[:12]} after being replaced"
|
|
362
|
+
)
|
|
363
|
+
|
|
364
|
+
def _open(self) -> tuple[_OrtSession, Tokenizer]:
|
|
365
|
+
"""Build the tokenizer and session from files verification has just trusted.
|
|
366
|
+
|
|
367
|
+
A failure here is not a corruption signal, because these bytes were checked
|
|
368
|
+
against the manifest moments ago: nothing is deleted and nothing is downloaded.
|
|
369
|
+
What is left is an onnxruntime that cannot load this graph, a permission problem,
|
|
370
|
+
or a machine out of memory, and the original exception says which. (The server
|
|
371
|
+
still retries on the next request; what it will not do is fetch 218 MB again.)
|
|
372
|
+
"""
|
|
373
|
+
try:
|
|
374
|
+
import onnxruntime
|
|
375
|
+
from tokenizers import Tokenizer
|
|
376
|
+
|
|
377
|
+
tokenizer = Tokenizer.from_file(str(self._model_dir / "tokenizer.json"))
|
|
378
|
+
tokenizer.enable_truncation(max_length=GEMMA_MAX_TOKENS)
|
|
379
|
+
tokenizer.enable_padding()
|
|
380
|
+
options = onnxruntime.SessionOptions()
|
|
381
|
+
options.add_session_config_entry(*_SPIN_CONFIG)
|
|
382
|
+
if threads := _inference_threads():
|
|
383
|
+
options.intra_op_num_threads = threads
|
|
384
|
+
session: _OrtSession = onnxruntime.InferenceSession(
|
|
385
|
+
str(self._model_dir / model_cache.GEMMA_MODEL_FILE),
|
|
386
|
+
options,
|
|
387
|
+
providers=["CPUExecutionProvider"],
|
|
388
|
+
)
|
|
389
|
+
except Exception as exc:
|
|
390
|
+
raise ModelLoadError(
|
|
391
|
+
f"Cannot load embedding model {model_cache.GEMMA_REPOSITORY}: {exc}"
|
|
392
|
+
) from exc
|
|
393
|
+
return session, tokenizer
|
|
394
|
+
|
|
395
|
+
def _repair(self) -> None:
|
|
396
|
+
"""Bring the cache up to the manifest. Runs under the exclusive lock."""
|
|
397
|
+
if model_cache._stamp_is_current(self._model_dir):
|
|
398
|
+
return # another process did the work while this one waited for the lock
|
|
399
|
+
self._model_dir.mkdir(parents=True, exist_ok=True)
|
|
400
|
+
wrong = model_cache._unverified(self._model_dir)
|
|
401
|
+
for name in wrong:
|
|
402
|
+
path = model_cache._cache_path(self._model_dir, name)
|
|
403
|
+
if path is None: # a symlinked directory on the way: refuse to write through it
|
|
404
|
+
raise ModelLoadError(
|
|
405
|
+
f"{self._model_dir / name} leaves the model cache through a symlinked "
|
|
406
|
+
"directory; move it aside by hand"
|
|
407
|
+
)
|
|
408
|
+
model_cache._remove(path) # only what is proven wrong
|
|
409
|
+
if wrong:
|
|
410
|
+
self._fetch()
|
|
411
|
+
still_wrong = model_cache._unverified(self._model_dir)
|
|
412
|
+
if still_wrong:
|
|
413
|
+
raise ModelLoadError(
|
|
414
|
+
f"Downloaded {model_cache.GEMMA_REPOSITORY} at "
|
|
415
|
+
f"{model_cache.GEMMA_REVISION[:12]}, but "
|
|
416
|
+
f"{', '.join(sorted(still_wrong))} does not match the expected size "
|
|
417
|
+
"and checksum"
|
|
418
|
+
)
|
|
419
|
+
model_cache._write_stamp(self._model_dir)
|
|
420
|
+
|
|
421
|
+
def _fetch(self) -> None:
|
|
422
|
+
from huggingface_hub import snapshot_download
|
|
423
|
+
|
|
424
|
+
logger.info("Downloading %s (~218 MB, first run only)", model_cache.GEMMA_REPOSITORY)
|
|
425
|
+
snapshot_download(
|
|
426
|
+
model_cache.GEMMA_REPOSITORY,
|
|
427
|
+
revision=model_cache.GEMMA_REVISION,
|
|
428
|
+
allow_patterns=list(model_cache.GEMMA_FILES),
|
|
429
|
+
local_dir=self._model_dir, # real files, not symlinks into a blob store
|
|
430
|
+
)
|
|
431
|
+
|
|
432
|
+
def _report_other_versions(self) -> None:
|
|
433
|
+
"""Say what weights this version does not use cost, once, and delete none of them.
|
|
434
|
+
|
|
435
|
+
Weights another checkout is using, or one pinned deliberately, are not this
|
|
436
|
+
process's to remove; saying how much room they take is. Two kinds qualify: a folder
|
|
437
|
+
for another revision, and - because one revision publishes several graphs - a file
|
|
438
|
+
sitting in *this* folder that the manifest does not name. The second is what an
|
|
439
|
+
upgrade from the int8 graph leaves behind, and looking only at other folders would
|
|
440
|
+
miss all 310 MB of it.
|
|
441
|
+
"""
|
|
442
|
+
others: dict[Path, int] = {}
|
|
443
|
+
for path in sorted(
|
|
444
|
+
model_cache.model_cache_root(self._cache_dir).glob(f"{model_cache._GEMMA_DIR_PREFIX}*")
|
|
445
|
+
):
|
|
446
|
+
if path == self._model_dir or not path.is_dir():
|
|
447
|
+
continue
|
|
448
|
+
with contextlib.suppress(OSError):
|
|
449
|
+
others[path] = sum(
|
|
450
|
+
entry.stat().st_size for entry in path.rglob("*") if entry.is_file()
|
|
451
|
+
)
|
|
452
|
+
with contextlib.suppress(OSError):
|
|
453
|
+
wanted = {self._model_dir / name for name in model_cache.GEMMA_FILES}
|
|
454
|
+
# `snapshot_download(local_dir=...)` keeps its own locks and metadata under
|
|
455
|
+
# `.cache/`: every download leaves them, and none of them is weights.
|
|
456
|
+
bookkeeping = self._model_dir / ".cache"
|
|
457
|
+
for entry in sorted(self._model_dir.rglob("*")):
|
|
458
|
+
stamp = entry.name.startswith(model_cache._VERIFIED_STAMP)
|
|
459
|
+
ours = entry.is_relative_to(bookkeeping)
|
|
460
|
+
if entry.is_file() and entry not in wanted and not stamp and not ours:
|
|
461
|
+
others[entry] = entry.stat().st_size
|
|
462
|
+
if others:
|
|
463
|
+
logger.info(
|
|
464
|
+
"The model cache also holds %d copy/copies of %s this version does not use "
|
|
465
|
+
"(%.0f MB in total): %s. Nothing is deleted automatically; remove them to "
|
|
466
|
+
"reclaim the space.",
|
|
467
|
+
len(others),
|
|
468
|
+
model_cache.GEMMA_REPOSITORY,
|
|
469
|
+
sum(others.values()) / 1e6,
|
|
470
|
+
", ".join(str(path) for path in others),
|
|
471
|
+
)
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
class _OrtSession(Protocol):
|
|
475
|
+
"""The one onnxruntime call this module makes (the package ships no type stubs)."""
|
|
476
|
+
|
|
477
|
+
def run(
|
|
478
|
+
self, output_names: Sequence[str], input_feed: dict[str, object]
|
|
479
|
+
) -> Sequence[_Array]: ...
|
|
480
|
+
|
|
481
|
+
|
|
482
|
+
class _Array(Protocol):
|
|
483
|
+
def tolist(self) -> list[list[float]]: ...
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
def create_embedder(preset: str = DEFAULT_EMBEDDER, *, cache_dir: Path | None = None) -> Embedder:
|
|
487
|
+
"""Build one of the supported local embedders by preset name."""
|
|
488
|
+
if preset == "embeddinggemma":
|
|
489
|
+
return EmbeddingGemmaEmbedder(cache_dir=cache_dir)
|
|
490
|
+
if preset == "bge-small":
|
|
491
|
+
return FastEmbedEmbedder(model_cache.BGE_SMALL_MODEL_NAME, cache_dir=cache_dir)
|
|
492
|
+
raise IndexingError(f"Unknown embedder {preset!r}; choose 'embeddinggemma' or 'bge-small'")
|
|
493
|
+
|
|
494
|
+
|
|
495
|
+
def _inference_threads() -> int:
|
|
496
|
+
"""Thread count for one embedding pass: onnxruntime's own choice unless overridden.
|
|
497
|
+
|
|
498
|
+
Deriving it from the machine was tried and rejected: on a 16-core VM the topology
|
|
499
|
+
says 16, which measured worse than the default, while every count from 4 to 12 sat
|
|
500
|
+
inside the run-to-run noise. A wrong number is slower than no number.
|
|
501
|
+
|
|
502
|
+
That comparison was wall time only, which is the smaller half of the story. Once
|
|
503
|
+
spinning is off (``_SPIN_CONFIG``), the count barely moves wall time but does move
|
|
504
|
+
CPU: indexing the same passages took ~52 s of CPU at onnxruntime's count and ~37 s
|
|
505
|
+
capped at 4. The default stays onnxruntime's, because the right cap depends on what
|
|
506
|
+
else the machine is doing; this is the knob for saying so.
|
|
507
|
+
"""
|
|
508
|
+
override = os.environ.get(_THREADS_ENV, "").strip()
|
|
509
|
+
return int(override) if override.isdigit() and int(override) > 0 else 0
|
|
510
|
+
|
|
511
|
+
|
|
512
|
+
def _to_floats(values: Sequence[float]) -> list[float]:
|
|
513
|
+
return [float(value) for value in values]
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""Domain-specific exception hierarchy for markdown-memory."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class MarkdownMemoryError(Exception):
|
|
7
|
+
"""Base class for every error raised deliberately by this package."""
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ConfigurationError(MarkdownMemoryError):
|
|
11
|
+
"""The server was started with configuration it cannot act on."""
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class DatabaseError(MarkdownMemoryError):
|
|
15
|
+
"""The SQLite store could not be opened, migrated, read, or written."""
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class ASTParseError(MarkdownMemoryError):
|
|
19
|
+
"""A Markdown source could not be tokenised into an AST."""
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class IndexingError(MarkdownMemoryError):
|
|
23
|
+
"""The indexing pipeline failed (bad directory, unreadable file, ...)."""
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class ForeignWeightsError(IndexingError):
|
|
27
|
+
"""The model about to embed is not the one whose vectors the index already holds.
|
|
28
|
+
|
|
29
|
+
Raised from the one place that can tell - just before a vector is produced, where a
|
|
30
|
+
lazily-loaded embedder has had to load and can finally say what it is. It aborts the
|
|
31
|
+
whole run rather than failing one file, because every other file would fail the same
|
|
32
|
+
way and for the same reason.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class IndexBusyError(IndexingError):
|
|
37
|
+
"""Another indexing run holds the lock on this database.
|
|
38
|
+
|
|
39
|
+
Its own class because the answer is "try again shortly", not "this failed": the
|
|
40
|
+
caller did nothing wrong and nothing is broken.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class IndexCancelled(Exception): # noqa: N818 - a request honoured, not an error
|
|
45
|
+
"""An index run stopped because its owner asked it to, between two documents.
|
|
46
|
+
|
|
47
|
+
Deliberately outside `MarkdownMemoryError`: the indexer's per-file handler catches
|
|
48
|
+
that hierarchy and carries on with the next file, and a stop that was swallowed as
|
|
49
|
+
one file's failure would not be a stop. What it leaves is what a killed run leaves -
|
|
50
|
+
the documents written so far, and coverage withdrawn until a run finishes.
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class EmbeddingError(IndexingError):
|
|
55
|
+
"""The embedding model failed to load or to produce usable vectors."""
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class ModelLoadError(EmbeddingError):
|
|
59
|
+
"""The embedding model itself cannot be loaded: no file can be embedded at all."""
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class SearchError(MarkdownMemoryError):
|
|
63
|
+
"""A search index failed in a way the storage layer did not anticipate."""
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class DocumentNotFoundError(MarkdownMemoryError):
|
|
67
|
+
"""The requested file is not present in the index."""
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class SectionNotFoundError(MarkdownMemoryError):
|
|
71
|
+
"""The requested heading path does not exist in the indexed document."""
|