markdown-memory 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- markdown_memory/__init__.py +47 -0
- markdown_memory/autoindex.py +170 -0
- markdown_memory/config.py +235 -0
- markdown_memory/db.py +1546 -0
- markdown_memory/discovery.py +195 -0
- markdown_memory/embedders.py +513 -0
- markdown_memory/exceptions.py +71 -0
- markdown_memory/freshness.py +138 -0
- markdown_memory/headings.py +185 -0
- markdown_memory/indexer.py +725 -0
- markdown_memory/model_cache.py +272 -0
- markdown_memory/models.py +339 -0
- markdown_memory/parser.py +869 -0
- markdown_memory/py.typed +0 -0
- markdown_memory/search.py +518 -0
- markdown_memory/server.py +520 -0
- markdown_memory-0.1.0.dist-info/METADATA +579 -0
- markdown_memory-0.1.0.dist-info/RECORD +21 -0
- markdown_memory-0.1.0.dist-info/WHEEL +4 -0
- markdown_memory-0.1.0.dist-info/entry_points.txt +3 -0
- markdown_memory-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,272 @@
|
|
|
1
|
+
"""The on-disk model cache: where weights live, and how we know they are ours.
|
|
2
|
+
|
|
3
|
+
The embedder that uses this is in ``embedders.py``; the cache is its own module because
|
|
4
|
+
the pin, the manifest and the verification stamp are consulted from outside it too -
|
|
5
|
+
``scripts/eval_cache.py`` keys the retrieval gate on them.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import contextlib
|
|
11
|
+
import fcntl
|
|
12
|
+
import hashlib
|
|
13
|
+
import json
|
|
14
|
+
import os
|
|
15
|
+
import shutil
|
|
16
|
+
import stat
|
|
17
|
+
from collections.abc import Iterator, Mapping
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
from markdown_memory.exceptions import ModelLoadError
|
|
21
|
+
|
|
22
|
+
BGE_SMALL_MODEL_NAME = "BAAI/bge-small-en-v1.5"
|
|
23
|
+
GEMMA_REPOSITORY = "onnx-community/embeddinggemma-300m-ONNX"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
# Pinned so that an upstream re-export can never silently change stored vectors.
|
|
27
|
+
GEMMA_REVISION = "5090578d9565bb06545b4552f76e6bc2c93e4a66"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# The 4-bit graph, which quantizes the vocabulary table with `GatherBlockQuantized` and the
|
|
31
|
+
# projections with `MatMulNBits`. Earlier versions ran `onnx/model_quantized.onnx` (int8)
|
|
32
|
+
# and rewrote it on each machine to gather the vocabulary before dequantizing it; this
|
|
33
|
+
# graph does that natively, in half the download and half the CPU per query. The file name
|
|
34
|
+
# is part of the embedder's `model_name`, so changing it discards every stored vector -
|
|
35
|
+
# which is correct, because the two graphs' vectors are not comparable.
|
|
36
|
+
GEMMA_MODEL_FILE = "onnx/model_q4.onnx"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
# Size and sha256 of every file at GEMMA_REVISION, from the Hub's paths-info API. This is
|
|
40
|
+
# what stands between a damaged cache and onnxruntime: huggingface_hub checks only the
|
|
41
|
+
# size of what it downloads, and hands back a file that is already on disk without reading
|
|
42
|
+
# it at all.
|
|
43
|
+
GEMMA_MANIFEST: Mapping[str, tuple[int, str]] = {
|
|
44
|
+
GEMMA_MODEL_FILE: (
|
|
45
|
+
519_322,
|
|
46
|
+
"ad1dfee81a70f7944b9b9d1cc6e48075b832881cf33fab2f2b248be78f3f0043",
|
|
47
|
+
),
|
|
48
|
+
GEMMA_MODEL_FILE + "_data": (
|
|
49
|
+
196_725_760,
|
|
50
|
+
"599962c3143b040de2dd05e5975be3e9091dd067cacc6a8f7186e3203bab9e02",
|
|
51
|
+
),
|
|
52
|
+
"tokenizer.json": (
|
|
53
|
+
20_323_312,
|
|
54
|
+
"4dda02faaf32bc91031dc8c88457ac272b00c1016cc679757d1c441b248b9c47",
|
|
55
|
+
),
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
GEMMA_FILES: tuple[str, ...] = tuple(GEMMA_MANIFEST)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
# The directory carries the revision, so moving the pin fetches the new weights instead of
|
|
63
|
+
# serving the old ones under a name that claims to be the new ones. The graph file is not
|
|
64
|
+
# in the path: a stamp recording one graph's files cannot match another's manifest, so a
|
|
65
|
+
# folder holding the wrong graph is rejected and refetched rather than trusted.
|
|
66
|
+
_GEMMA_DIR_PREFIX = "embeddinggemma-300m-onnx"
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
# One lock for every revision, so two versions starting at once still exclude each other.
|
|
70
|
+
_GEMMA_LOCK_NAME = f"{_GEMMA_DIR_PREFIX}.lock"
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
_VERIFIED_STAMP = ".verified"
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def model_cache_root(cache_dir: Path | None = None) -> Path:
|
|
77
|
+
"""Where every model this package downloads is kept."""
|
|
78
|
+
return cache_dir or Path.home() / ".cache" / "markdown-memory" / "models"
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def gemma_model_dir(cache_dir: Path | None = None) -> Path:
|
|
82
|
+
"""The folder holding the pinned EmbeddingGemma revision."""
|
|
83
|
+
return model_cache_root(cache_dir) / f"{_GEMMA_DIR_PREFIX}-{GEMMA_REVISION[:12]}"
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def fastembed_model_dir(cache_dir: Path | None) -> Path | None:
|
|
87
|
+
"""The folder fastembed keeps bge-small in, or None when it cannot be derived.
|
|
88
|
+
|
|
89
|
+
Not guessable from the model name: fastembed downloads its own re-export of the
|
|
90
|
+
model (`qdrant/bge-small-en-v1.5-onnx-q`), so the repository is read out of its
|
|
91
|
+
registry rather than assumed.
|
|
92
|
+
"""
|
|
93
|
+
if cache_dir is None:
|
|
94
|
+
return None
|
|
95
|
+
from fastembed import TextEmbedding
|
|
96
|
+
|
|
97
|
+
for entry in TextEmbedding.list_supported_models():
|
|
98
|
+
if not isinstance(entry, dict) or entry.get("model") != BGE_SMALL_MODEL_NAME:
|
|
99
|
+
continue
|
|
100
|
+
sources = entry.get("sources")
|
|
101
|
+
repository = sources.get("hf") if isinstance(sources, dict) else None
|
|
102
|
+
if isinstance(repository, str) and repository:
|
|
103
|
+
return cache_dir / ("models--" + repository.replace("/", "--"))
|
|
104
|
+
return None
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _cache_path(model_dir: Path, name: str) -> Path | None:
|
|
108
|
+
"""``model_dir/name``, or None when any directory on the way there is a symlink.
|
|
109
|
+
|
|
110
|
+
`_file_identity` only ever looked at the last component, so an `onnx` that pointed
|
|
111
|
+
somewhere else was trusted - and then repaired, which meant deleting and overwriting
|
|
112
|
+
files outside the cache entirely.
|
|
113
|
+
"""
|
|
114
|
+
current = model_dir
|
|
115
|
+
for part in Path(name).parts:
|
|
116
|
+
if current.is_symlink():
|
|
117
|
+
return None
|
|
118
|
+
current = current / part
|
|
119
|
+
return current
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _file_identity(path: Path) -> dict[str, int] | None:
|
|
123
|
+
"""What a stamp remembers about one model file; None when it is not a plain file.
|
|
124
|
+
|
|
125
|
+
`ctime_ns` earns its place: `cp -p`, `tar x` and `rsync --inplace` all rewrite a
|
|
126
|
+
file's contents and then restore its old mtime, so size, mtime and inode can agree
|
|
127
|
+
across different bytes. Nothing in user space can set ctime back.
|
|
128
|
+
"""
|
|
129
|
+
try:
|
|
130
|
+
info = path.lstat()
|
|
131
|
+
except OSError:
|
|
132
|
+
return None
|
|
133
|
+
if not stat.S_ISREG(info.st_mode):
|
|
134
|
+
return None # a symlink into a blob store is not a file this cache vouches for
|
|
135
|
+
return {
|
|
136
|
+
"size": info.st_size,
|
|
137
|
+
"mtime_ns": info.st_mtime_ns,
|
|
138
|
+
"ctime_ns": info.st_ctime_ns,
|
|
139
|
+
"inode": info.st_ino,
|
|
140
|
+
"device": info.st_dev,
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _identity_in(model_dir: Path, name: str) -> dict[str, int] | None:
|
|
145
|
+
"""The identity of one cached file, refusing a path that leaves the cache."""
|
|
146
|
+
path = _cache_path(model_dir, name)
|
|
147
|
+
return None if path is None else _file_identity(path)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _stamp_is_current(model_dir: Path) -> bool:
|
|
151
|
+
"""Whether every file still looks exactly as it did when it was last verified.
|
|
152
|
+
|
|
153
|
+
Hashing 218 MB costs most of a second of one core, which is too much for every
|
|
154
|
+
server start when an editor starts one per session. This is a handful of `stat`
|
|
155
|
+
calls; anything that disagrees sends the files back to be hashed.
|
|
156
|
+
"""
|
|
157
|
+
try:
|
|
158
|
+
stamp = json.loads((model_dir / _VERIFIED_STAMP).read_text(encoding="utf-8"))
|
|
159
|
+
except (OSError, ValueError):
|
|
160
|
+
return False
|
|
161
|
+
if not isinstance(stamp, dict) or stamp.get("revision") != GEMMA_REVISION:
|
|
162
|
+
return False
|
|
163
|
+
recorded = stamp.get("files")
|
|
164
|
+
if not isinstance(recorded, dict) or set(recorded) != set(GEMMA_FILES):
|
|
165
|
+
# Also how a folder left behind by a different graph is rejected: its stamp names
|
|
166
|
+
# files this manifest does not, so it is refetched rather than half-trusted.
|
|
167
|
+
return False
|
|
168
|
+
if model_dir.is_symlink():
|
|
169
|
+
return False
|
|
170
|
+
for name in recorded:
|
|
171
|
+
identity = _identity_in(model_dir, name)
|
|
172
|
+
# `None` means the path is not a regular file - a symlink, most likely. Nothing
|
|
173
|
+
# should be able to stamp one (`_unverified` calls it wrong), and this is the
|
|
174
|
+
# line that makes a stamp that somehow recorded `None` stop matching `None`.
|
|
175
|
+
if identity is None or recorded[name] != identity:
|
|
176
|
+
return False
|
|
177
|
+
return True
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _write_stamp(model_dir: Path) -> None:
|
|
181
|
+
"""Record what was just verified. Atomically: a half-written stamp is a false claim."""
|
|
182
|
+
stamp = {
|
|
183
|
+
"revision": GEMMA_REVISION,
|
|
184
|
+
"files": {name: _file_identity(model_dir / name) for name in GEMMA_FILES},
|
|
185
|
+
}
|
|
186
|
+
temporary = model_dir / f"{_VERIFIED_STAMP}.{os.getpid()}"
|
|
187
|
+
_remove(temporary) # a stamp left behind by a crash under this same pid
|
|
188
|
+
_write_new_file(temporary, json.dumps(stamp).encode("utf-8"))
|
|
189
|
+
stamped = model_dir / _VERIFIED_STAMP
|
|
190
|
+
if stamped.is_dir() and not stamped.is_symlink():
|
|
191
|
+
# `os.replace` will not put a file where a directory is, and the stamp is read
|
|
192
|
+
# back as simply invalid, so the repair it ends would run again on every start
|
|
193
|
+
# and end the same way. Nothing of ours is ever a directory here.
|
|
194
|
+
_remove(stamped)
|
|
195
|
+
os.replace(temporary, stamped)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _write_new_file(path: Path, payload: bytes) -> None:
|
|
199
|
+
"""Create ``path`` with its contents, refusing to follow a symlink or reuse a file.
|
|
200
|
+
|
|
201
|
+
The temporary names these writes use are predictable (the pid), and a plain write
|
|
202
|
+
follows a symlink planted at one of them: the caller would then overwrite whatever it
|
|
203
|
+
points at, anywhere the user can write. `O_EXCL | O_NOFOLLOW` makes both refusals the
|
|
204
|
+
kernel's.
|
|
205
|
+
"""
|
|
206
|
+
descriptor = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, 0o600)
|
|
207
|
+
try:
|
|
208
|
+
with os.fdopen(descriptor, "wb") as handle:
|
|
209
|
+
handle.write(payload)
|
|
210
|
+
except BaseException:
|
|
211
|
+
with contextlib.suppress(OSError):
|
|
212
|
+
path.unlink()
|
|
213
|
+
raise
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def _remove(path: Path) -> None:
|
|
217
|
+
"""Delete whatever sits at ``path``, file or directory.
|
|
218
|
+
|
|
219
|
+
A directory where a model file belongs is not something `unlink` can clear, and a
|
|
220
|
+
cache that cannot be repaired is a server that never starts again.
|
|
221
|
+
"""
|
|
222
|
+
if path.is_dir() and not path.is_symlink():
|
|
223
|
+
shutil.rmtree(path, ignore_errors=True)
|
|
224
|
+
else:
|
|
225
|
+
path.unlink(missing_ok=True)
|
|
226
|
+
if path.exists() or path.is_symlink():
|
|
227
|
+
# Say so here, where the path is known, rather than failing three lines later on
|
|
228
|
+
# a rename that cannot explain itself.
|
|
229
|
+
raise ModelLoadError(f"Cannot clear {path} to repair the model cache")
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _hash_file(path: Path) -> str:
|
|
233
|
+
digest = hashlib.sha256()
|
|
234
|
+
with path.open("rb") as handle:
|
|
235
|
+
for block in iter(lambda: handle.read(1 << 20), b""):
|
|
236
|
+
digest.update(block)
|
|
237
|
+
return digest.hexdigest()
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _unverified(model_dir: Path) -> list[str]:
|
|
241
|
+
"""The model files that are missing, the wrong size, or the wrong bytes."""
|
|
242
|
+
problems: list[str] = []
|
|
243
|
+
for name, (size, checksum) in GEMMA_MANIFEST.items():
|
|
244
|
+
path = _cache_path(model_dir, name)
|
|
245
|
+
before = None if path is None else _file_identity(path)
|
|
246
|
+
if path is None or before is None or before["size"] != size:
|
|
247
|
+
problems.append(name)
|
|
248
|
+
elif _hash_file(path) != checksum or _file_identity(path) != before:
|
|
249
|
+
# Second identity: a file rewritten while it was being read was never hashed
|
|
250
|
+
# as it now stands, so the answer that came back means nothing.
|
|
251
|
+
problems.append(name)
|
|
252
|
+
return problems
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
@contextlib.contextmanager
|
|
256
|
+
def _model_cache_lock(cache_dir: Path | None, *, exclusive: bool) -> Iterator[None]:
|
|
257
|
+
"""Serialise verification, repair and session construction across processes.
|
|
258
|
+
|
|
259
|
+
Shared while a verified cache is being opened, exclusive while it is being changed,
|
|
260
|
+
so nothing can repair files another process has verified but not yet handed to
|
|
261
|
+
onnxruntime. The kernel drops a `flock` when the process dies, so a crash leaves
|
|
262
|
+
nothing held. A model cache on NFS or SMB shared between machines is out of scope:
|
|
263
|
+
`flock` can be local-only there - the boundary SQLite's WAL already has.
|
|
264
|
+
"""
|
|
265
|
+
root = model_cache_root(cache_dir)
|
|
266
|
+
root.mkdir(parents=True, exist_ok=True)
|
|
267
|
+
with (root / _GEMMA_LOCK_NAME).open("a") as handle:
|
|
268
|
+
fcntl.flock(handle, fcntl.LOCK_EX if exclusive else fcntl.LOCK_SH)
|
|
269
|
+
try:
|
|
270
|
+
yield
|
|
271
|
+
finally:
|
|
272
|
+
fcntl.flock(handle, fcntl.LOCK_UN)
|
|
@@ -0,0 +1,339 @@
|
|
|
1
|
+
"""Immutable, fully typed domain models.
|
|
2
|
+
|
|
3
|
+
Every entity is a frozen, slotted dataclass. ``to_dict`` methods produce the
|
|
4
|
+
JSON-serialisable payloads returned by the MCP tools.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from collections.abc import Sequence
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from typing import TypeAlias
|
|
12
|
+
|
|
13
|
+
from pydantic import JsonValue
|
|
14
|
+
|
|
15
|
+
JsonDict: TypeAlias = dict[str, JsonValue]
|
|
16
|
+
|
|
17
|
+
PREAMBLE_TITLE = "[Overview / Preamble]"
|
|
18
|
+
PATH_SEPARATOR = " > "
|
|
19
|
+
CHARS_PER_TOKEN = 4
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def estimate_tokens(text: str) -> int:
|
|
23
|
+
"""Cheap token estimate (~4 characters per token), never below 1 for non-empty text."""
|
|
24
|
+
if not text:
|
|
25
|
+
return 0
|
|
26
|
+
return max(1, -(-len(text) // CHARS_PER_TOKEN))
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def part_path(base_path: str, part_index: int) -> str:
|
|
30
|
+
"""Breadcrumb for one part of an oversized section, e.g. ``A > B (Part 2)``."""
|
|
31
|
+
return f"{base_path} (Part {part_index})"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass(slots=True, frozen=True)
|
|
35
|
+
class SectionDraft:
|
|
36
|
+
"""A section produced by the parser, not yet persisted.
|
|
37
|
+
|
|
38
|
+
``part_index`` is ``0`` for a section stored whole and ``1..n`` for the
|
|
39
|
+
sequential parts of an oversized section. ``base_path`` is the breadcrumb
|
|
40
|
+
without the ``(Part n)`` suffix. Line numbers are 1-based and inclusive.
|
|
41
|
+
|
|
42
|
+
``units`` are the section's passages as plain text - one per paragraph, list item,
|
|
43
|
+
table row or code block. Each is embedded separately, because one vector for a
|
|
44
|
+
whole section dilutes a single relevant table row beyond recognition. A section
|
|
45
|
+
without units is heading-only and gets no vectors at all.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
heading_title: str
|
|
49
|
+
heading_level: int
|
|
50
|
+
heading_path: str
|
|
51
|
+
base_path: str
|
|
52
|
+
content: str
|
|
53
|
+
start_line: int
|
|
54
|
+
end_line: int
|
|
55
|
+
part_index: int = 0
|
|
56
|
+
units: tuple[str, ...] = ()
|
|
57
|
+
|
|
58
|
+
@property
|
|
59
|
+
def embedding_text(self) -> str:
|
|
60
|
+
"""Text for the section-level vector: breadcrumb, then the body without markup."""
|
|
61
|
+
return f"{self.heading_path}\n\n{' '.join(self.units)}"
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def unit_texts(self) -> tuple[str, ...]:
|
|
65
|
+
"""Texts for the passage-level vectors, each carrying the breadcrumb for context."""
|
|
66
|
+
return tuple(f"{self.heading_path}: {unit}" for unit in self.units)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@dataclass(slots=True, frozen=True)
|
|
70
|
+
class SectionVectors:
|
|
71
|
+
"""Embeddings of one section: ``section`` is ``None`` for a heading-only section."""
|
|
72
|
+
|
|
73
|
+
section: Sequence[float] | None
|
|
74
|
+
units: tuple[Sequence[float], ...] = ()
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@dataclass(slots=True, frozen=True)
|
|
78
|
+
class ParsedDocument:
|
|
79
|
+
"""Result of parsing one Markdown source."""
|
|
80
|
+
|
|
81
|
+
title: str
|
|
82
|
+
sections: tuple[SectionDraft, ...]
|
|
83
|
+
line_count: int
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@dataclass(slots=True, frozen=True)
|
|
87
|
+
class Document:
|
|
88
|
+
"""A persisted Markdown file."""
|
|
89
|
+
|
|
90
|
+
id: int
|
|
91
|
+
file_path: str
|
|
92
|
+
title: str
|
|
93
|
+
content_hash: str
|
|
94
|
+
last_modified: int
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
@dataclass(slots=True, frozen=True)
|
|
98
|
+
class DocumentSummary:
|
|
99
|
+
"""A document together with the number of sections indexed for it."""
|
|
100
|
+
|
|
101
|
+
file_path: str
|
|
102
|
+
title: str
|
|
103
|
+
section_count: int
|
|
104
|
+
last_modified: int
|
|
105
|
+
|
|
106
|
+
def to_dict(self) -> JsonDict:
|
|
107
|
+
return {
|
|
108
|
+
"file_path": self.file_path,
|
|
109
|
+
"title": self.title,
|
|
110
|
+
"section_count": self.section_count,
|
|
111
|
+
"last_modified": self.last_modified,
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
@dataclass(slots=True, frozen=True)
|
|
116
|
+
class Section:
|
|
117
|
+
"""A persisted section (or one part of an oversized section)."""
|
|
118
|
+
|
|
119
|
+
id: int
|
|
120
|
+
doc_id: int
|
|
121
|
+
heading_title: str
|
|
122
|
+
heading_level: int
|
|
123
|
+
heading_path: str
|
|
124
|
+
content: str
|
|
125
|
+
start_line: int
|
|
126
|
+
end_line: int
|
|
127
|
+
part_index: int
|
|
128
|
+
|
|
129
|
+
@property
|
|
130
|
+
def base_path(self) -> str:
|
|
131
|
+
"""Breadcrumb with any ``(Part n)`` suffix removed."""
|
|
132
|
+
if self.part_index <= 0:
|
|
133
|
+
return self.heading_path
|
|
134
|
+
suffix = part_path("", self.part_index)
|
|
135
|
+
return self.heading_path.removesuffix(suffix)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
@dataclass(slots=True, frozen=True)
|
|
139
|
+
class OutlineNode:
|
|
140
|
+
"""One heading in a document's hierarchical table of contents."""
|
|
141
|
+
|
|
142
|
+
heading_title: str
|
|
143
|
+
heading_level: int
|
|
144
|
+
heading_path: str
|
|
145
|
+
start_line: int
|
|
146
|
+
end_line: int
|
|
147
|
+
token_estimate: int
|
|
148
|
+
part_count: int
|
|
149
|
+
children: tuple[OutlineNode, ...] = ()
|
|
150
|
+
|
|
151
|
+
def to_dict(self) -> JsonDict:
|
|
152
|
+
payload: JsonDict = {
|
|
153
|
+
"title": self.heading_title,
|
|
154
|
+
"level": self.heading_level,
|
|
155
|
+
"heading_path": self.heading_path,
|
|
156
|
+
"lines": f"{self.start_line}-{self.end_line}",
|
|
157
|
+
"tokens": self.token_estimate,
|
|
158
|
+
}
|
|
159
|
+
if self.part_count > 1:
|
|
160
|
+
payload["parts"] = self.part_count
|
|
161
|
+
if self.children:
|
|
162
|
+
payload["children"] = [child.to_dict() for child in self.children]
|
|
163
|
+
return payload
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
@dataclass(slots=True, frozen=True)
|
|
167
|
+
class SearchResult:
|
|
168
|
+
"""A section matched by hybrid search, with its fused and per-index ranks."""
|
|
169
|
+
|
|
170
|
+
section_id: int
|
|
171
|
+
file_path: str
|
|
172
|
+
document_title: str
|
|
173
|
+
heading_title: str
|
|
174
|
+
heading_path: str
|
|
175
|
+
content: str
|
|
176
|
+
start_line: int
|
|
177
|
+
end_line: int
|
|
178
|
+
score: float
|
|
179
|
+
fts_rank: int | None
|
|
180
|
+
vec_rank: int | None
|
|
181
|
+
matched_passage: str | None = None
|
|
182
|
+
|
|
183
|
+
def to_dict(self) -> JsonDict:
|
|
184
|
+
payload: JsonDict = {
|
|
185
|
+
"file_path": self.file_path,
|
|
186
|
+
"document_title": self.document_title,
|
|
187
|
+
"heading_path": self.heading_path,
|
|
188
|
+
"heading_title": self.heading_title,
|
|
189
|
+
"lines": f"{self.start_line}-{self.end_line}",
|
|
190
|
+
"score": round(self.score, 6),
|
|
191
|
+
"fts_rank": self.fts_rank,
|
|
192
|
+
"vec_rank": self.vec_rank,
|
|
193
|
+
"tokens": estimate_tokens(self.content),
|
|
194
|
+
"content": self.content,
|
|
195
|
+
}
|
|
196
|
+
if self.matched_passage is not None:
|
|
197
|
+
payload["matched_passage"] = self.matched_passage
|
|
198
|
+
return payload
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
@dataclass(slots=True, frozen=True)
|
|
202
|
+
class FileFailure:
|
|
203
|
+
"""A per-file failure recorded while indexing (the run itself continues)."""
|
|
204
|
+
|
|
205
|
+
file_path: str
|
|
206
|
+
message: str
|
|
207
|
+
|
|
208
|
+
def to_dict(self) -> JsonDict:
|
|
209
|
+
return {"file_path": self.file_path, "message": self.message}
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
#: Failures carried on a search answer. Enough to act on, few enough not to bury the
|
|
213
|
+
#: answer itself - the count in `message` says how many were left out.
|
|
214
|
+
MAX_REPORTED_FAILURES = 20
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
@dataclass(slots=True, frozen=True)
|
|
218
|
+
class IndexStatus:
|
|
219
|
+
"""Whether answers drawn from one tree can be trusted to be drawn from all of it.
|
|
220
|
+
|
|
221
|
+
`verified` means a full walk of this scope finished and read every file it found. It
|
|
222
|
+
is deliberately not a claim that the filesystem has stopped changing: a file created
|
|
223
|
+
after its directory was walked is not in the index and not in `failures`, and the next
|
|
224
|
+
run picks it up. A killed run and a tree nobody ever indexed both read unverified,
|
|
225
|
+
which is the same answer because it is the same situation - nothing walked it whole.
|
|
226
|
+
"""
|
|
227
|
+
|
|
228
|
+
verified: bool
|
|
229
|
+
failures: tuple[FileFailure, ...] = ()
|
|
230
|
+
#: Set when the weights behind the model name changed under an existing index: the
|
|
231
|
+
#: stored vectors and the vectors a query would produce now come from different
|
|
232
|
+
#: models. Nothing is discarded, and nothing new is written, until it is resolved.
|
|
233
|
+
weights_mismatch: str | None = None
|
|
234
|
+
#: Documents under this scope whose vectors were built by an older pooling scheme.
|
|
235
|
+
#: They still answer, less well, and only a run over the directory holding them
|
|
236
|
+
#: rebuilds - a parent run prunes `.venv`, `node_modules` and the like, so one indexed
|
|
237
|
+
#: deliberately inside such a directory is never reached again.
|
|
238
|
+
stale_vectors: int = 0
|
|
239
|
+
#: Indexed documents a cheap probe could not confirm are still what was indexed: their
|
|
240
|
+
#: bytes differ, or they are gone, unreadable, or no longer a regular file. It is
|
|
241
|
+
#: best-effort in both directions. Counted over the rows the index holds, so a file
|
|
242
|
+
#: nobody has indexed yet is not in it - finding those needs the directory walk, which
|
|
243
|
+
#: is the expensive half. And the bytes are only read where the modification time moved,
|
|
244
|
+
#: so an edit that restores a file's own timestamp is not seen. Zero means nothing was
|
|
245
|
+
#: detected, not that every indexed file was hashed.
|
|
246
|
+
changed_files: int = 0
|
|
247
|
+
#: This server's own background run is indexing the tree right now. A hint about one
|
|
248
|
+
#: process only: another process's run shows as coverage withdrawn, as it always did.
|
|
249
|
+
indexing: bool = False
|
|
250
|
+
|
|
251
|
+
def to_dict(self) -> JsonDict:
|
|
252
|
+
shown = self.failures[:MAX_REPORTED_FAILURES]
|
|
253
|
+
return {
|
|
254
|
+
"coverage": "verified" if self.verified else "unknown",
|
|
255
|
+
"failures": [failure.to_dict() for failure in shown],
|
|
256
|
+
"changed_files": self.changed_files,
|
|
257
|
+
"indexing": self.indexing,
|
|
258
|
+
"message": self.message(),
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
def message(self) -> str | None:
|
|
262
|
+
"""One sentence, or nothing at all when there is nothing to act on."""
|
|
263
|
+
# First, because it is the only one that says the answers themselves may be
|
|
264
|
+
# wrong rather than incomplete. Hoisted above `verified` rather than left below
|
|
265
|
+
# it: the two cannot both hold today, and a reader should not have to know that.
|
|
266
|
+
if self.weights_mismatch:
|
|
267
|
+
return self.weights_mismatch
|
|
268
|
+
if self.indexing:
|
|
269
|
+
# Before everything that ends in "run index_directory": that run would be
|
|
270
|
+
# refused as busy, and it would be refused for doing what is already being done.
|
|
271
|
+
return (
|
|
272
|
+
"An automatic index run is in progress, so an answer may be missing a file "
|
|
273
|
+
"changed or added since the last one finished; there is no need to run "
|
|
274
|
+
"index_directory."
|
|
275
|
+
)
|
|
276
|
+
if self.verified:
|
|
277
|
+
# A walk that finished still describes the moment it finished. Files edited
|
|
278
|
+
# since are the one thing a verified tree has left to say.
|
|
279
|
+
if self.changed_files:
|
|
280
|
+
return (
|
|
281
|
+
f"The last full index completed, but {self.changed_files} indexed "
|
|
282
|
+
"document(s) can no longer be confirmed to be what was indexed - "
|
|
283
|
+
"changed, unreadable or gone - so an answer may quote text that is no "
|
|
284
|
+
"longer there; run index_directory to refresh. The check is cheap and "
|
|
285
|
+
"best-effort: files created since that scan are not counted, and an "
|
|
286
|
+
"edit that puts a file's modification time back is not seen."
|
|
287
|
+
)
|
|
288
|
+
return None
|
|
289
|
+
if not self.failures:
|
|
290
|
+
if self.stale_vectors:
|
|
291
|
+
return (
|
|
292
|
+
f"{self.stale_vectors} document(s) here were indexed by an older "
|
|
293
|
+
"vector format and rank less well until the directory holding them is "
|
|
294
|
+
"indexed again."
|
|
295
|
+
)
|
|
296
|
+
return (
|
|
297
|
+
"This documentation root has not been indexed end to end since it last "
|
|
298
|
+
"changed, so an answer may be missing part of it. Run index_directory."
|
|
299
|
+
)
|
|
300
|
+
hidden = len(self.failures) - MAX_REPORTED_FAILURES
|
|
301
|
+
more = f" (showing the first {MAX_REPORTED_FAILURES})" if hidden > 0 else ""
|
|
302
|
+
return (
|
|
303
|
+
f"{len(self.failures)} path(s) could not be indexed{more}; answers here are "
|
|
304
|
+
"drawn from a tree that is missing them."
|
|
305
|
+
)
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
@dataclass(slots=True, frozen=True)
|
|
309
|
+
class IndexReport:
|
|
310
|
+
"""Outcome of one ``index_directory`` run."""
|
|
311
|
+
|
|
312
|
+
directory: str
|
|
313
|
+
files_scanned: int
|
|
314
|
+
files_indexed: int
|
|
315
|
+
files_unchanged: int
|
|
316
|
+
files_purged: int
|
|
317
|
+
sections_indexed: int
|
|
318
|
+
elapsed_seconds: float
|
|
319
|
+
passages_indexed: int = 0
|
|
320
|
+
errors: tuple[FileFailure, ...] = ()
|
|
321
|
+
notes: tuple[str, ...] = ()
|
|
322
|
+
|
|
323
|
+
def summary(self) -> str:
|
|
324
|
+
lines = [
|
|
325
|
+
f"Indexed {self.directory} in {self.elapsed_seconds:.2f}s: "
|
|
326
|
+
f"{self.files_scanned} scanned, {self.files_indexed} (re)indexed, "
|
|
327
|
+
f"{self.files_unchanged} unchanged, {self.files_purged} purged, "
|
|
328
|
+
f"{self.sections_indexed} sections embedded ({self.passages_indexed} passages)."
|
|
329
|
+
]
|
|
330
|
+
lines.extend(f"NOTE {note}" for note in self.notes)
|
|
331
|
+
lines.extend(f"ERROR {error.file_path}: {error.message}" for error in self.errors)
|
|
332
|
+
if self.errors:
|
|
333
|
+
# Without this the run reads as a success with some noise attached, and an
|
|
334
|
+
# index missing part of its tree answers questions as if it were whole.
|
|
335
|
+
lines.append(
|
|
336
|
+
f"INCOMPLETE: {len(self.errors)} file(s) could not be indexed; "
|
|
337
|
+
"this documentation root is only partly searchable."
|
|
338
|
+
)
|
|
339
|
+
return "\n".join(lines)
|