markdown-memory 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,272 @@
1
+ """The on-disk model cache: where weights live, and how we know they are ours.
2
+
3
+ The embedder that uses this is in ``embedders.py``; the cache is its own module because
4
+ the pin, the manifest and the verification stamp are consulted from outside it too -
5
+ ``scripts/eval_cache.py`` keys the retrieval gate on them.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import contextlib
11
+ import fcntl
12
+ import hashlib
13
+ import json
14
+ import os
15
+ import shutil
16
+ import stat
17
+ from collections.abc import Iterator, Mapping
18
+ from pathlib import Path
19
+
20
+ from markdown_memory.exceptions import ModelLoadError
21
+
22
+ BGE_SMALL_MODEL_NAME = "BAAI/bge-small-en-v1.5"
23
+ GEMMA_REPOSITORY = "onnx-community/embeddinggemma-300m-ONNX"
24
+
25
+
26
+ # Pinned so that an upstream re-export can never silently change stored vectors.
27
+ GEMMA_REVISION = "5090578d9565bb06545b4552f76e6bc2c93e4a66"
28
+
29
+
30
+ # The 4-bit graph, which quantizes the vocabulary table with `GatherBlockQuantized` and the
31
+ # projections with `MatMulNBits`. Earlier versions ran `onnx/model_quantized.onnx` (int8)
32
+ # and rewrote it on each machine to gather the vocabulary before dequantizing it; this
33
+ # graph does that natively, in half the download and half the CPU per query. The file name
34
+ # is part of the embedder's `model_name`, so changing it discards every stored vector -
35
+ # which is correct, because the two graphs' vectors are not comparable.
36
+ GEMMA_MODEL_FILE = "onnx/model_q4.onnx"
37
+
38
+
39
+ # Size and sha256 of every file at GEMMA_REVISION, from the Hub's paths-info API. This is
40
+ # what stands between a damaged cache and onnxruntime: huggingface_hub checks only the
41
+ # size of what it downloads, and hands back a file that is already on disk without reading
42
+ # it at all.
43
+ GEMMA_MANIFEST: Mapping[str, tuple[int, str]] = {
44
+ GEMMA_MODEL_FILE: (
45
+ 519_322,
46
+ "ad1dfee81a70f7944b9b9d1cc6e48075b832881cf33fab2f2b248be78f3f0043",
47
+ ),
48
+ GEMMA_MODEL_FILE + "_data": (
49
+ 196_725_760,
50
+ "599962c3143b040de2dd05e5975be3e9091dd067cacc6a8f7186e3203bab9e02",
51
+ ),
52
+ "tokenizer.json": (
53
+ 20_323_312,
54
+ "4dda02faaf32bc91031dc8c88457ac272b00c1016cc679757d1c441b248b9c47",
55
+ ),
56
+ }
57
+
58
+
59
+ GEMMA_FILES: tuple[str, ...] = tuple(GEMMA_MANIFEST)
60
+
61
+
62
+ # The directory carries the revision, so moving the pin fetches the new weights instead of
63
+ # serving the old ones under a name that claims to be the new ones. The graph file is not
64
+ # in the path: a stamp recording one graph's files cannot match another's manifest, so a
65
+ # folder holding the wrong graph is rejected and refetched rather than trusted.
66
+ _GEMMA_DIR_PREFIX = "embeddinggemma-300m-onnx"
67
+
68
+
69
+ # One lock for every revision, so two versions starting at once still exclude each other.
70
+ _GEMMA_LOCK_NAME = f"{_GEMMA_DIR_PREFIX}.lock"
71
+
72
+
73
+ _VERIFIED_STAMP = ".verified"
74
+
75
+
76
+ def model_cache_root(cache_dir: Path | None = None) -> Path:
77
+ """Where every model this package downloads is kept."""
78
+ return cache_dir or Path.home() / ".cache" / "markdown-memory" / "models"
79
+
80
+
81
+ def gemma_model_dir(cache_dir: Path | None = None) -> Path:
82
+ """The folder holding the pinned EmbeddingGemma revision."""
83
+ return model_cache_root(cache_dir) / f"{_GEMMA_DIR_PREFIX}-{GEMMA_REVISION[:12]}"
84
+
85
+
86
+ def fastembed_model_dir(cache_dir: Path | None) -> Path | None:
87
+ """The folder fastembed keeps bge-small in, or None when it cannot be derived.
88
+
89
+ Not guessable from the model name: fastembed downloads its own re-export of the
90
+ model (`qdrant/bge-small-en-v1.5-onnx-q`), so the repository is read out of its
91
+ registry rather than assumed.
92
+ """
93
+ if cache_dir is None:
94
+ return None
95
+ from fastembed import TextEmbedding
96
+
97
+ for entry in TextEmbedding.list_supported_models():
98
+ if not isinstance(entry, dict) or entry.get("model") != BGE_SMALL_MODEL_NAME:
99
+ continue
100
+ sources = entry.get("sources")
101
+ repository = sources.get("hf") if isinstance(sources, dict) else None
102
+ if isinstance(repository, str) and repository:
103
+ return cache_dir / ("models--" + repository.replace("/", "--"))
104
+ return None
105
+
106
+
107
+ def _cache_path(model_dir: Path, name: str) -> Path | None:
108
+ """``model_dir/name``, or None when any directory on the way there is a symlink.
109
+
110
+ `_file_identity` only ever looked at the last component, so an `onnx` that pointed
111
+ somewhere else was trusted - and then repaired, which meant deleting and overwriting
112
+ files outside the cache entirely.
113
+ """
114
+ current = model_dir
115
+ for part in Path(name).parts:
116
+ if current.is_symlink():
117
+ return None
118
+ current = current / part
119
+ return current
120
+
121
+
122
+ def _file_identity(path: Path) -> dict[str, int] | None:
123
+ """What a stamp remembers about one model file; None when it is not a plain file.
124
+
125
+ `ctime_ns` earns its place: `cp -p`, `tar x` and `rsync --inplace` all rewrite a
126
+ file's contents and then restore its old mtime, so size, mtime and inode can agree
127
+ across different bytes. Nothing in user space can set ctime back.
128
+ """
129
+ try:
130
+ info = path.lstat()
131
+ except OSError:
132
+ return None
133
+ if not stat.S_ISREG(info.st_mode):
134
+ return None # a symlink into a blob store is not a file this cache vouches for
135
+ return {
136
+ "size": info.st_size,
137
+ "mtime_ns": info.st_mtime_ns,
138
+ "ctime_ns": info.st_ctime_ns,
139
+ "inode": info.st_ino,
140
+ "device": info.st_dev,
141
+ }
142
+
143
+
144
+ def _identity_in(model_dir: Path, name: str) -> dict[str, int] | None:
145
+ """The identity of one cached file, refusing a path that leaves the cache."""
146
+ path = _cache_path(model_dir, name)
147
+ return None if path is None else _file_identity(path)
148
+
149
+
150
+ def _stamp_is_current(model_dir: Path) -> bool:
151
+ """Whether every file still looks exactly as it did when it was last verified.
152
+
153
+ Hashing 218 MB costs most of a second of one core, which is too much for every
154
+ server start when an editor starts one per session. This is a handful of `stat`
155
+ calls; anything that disagrees sends the files back to be hashed.
156
+ """
157
+ try:
158
+ stamp = json.loads((model_dir / _VERIFIED_STAMP).read_text(encoding="utf-8"))
159
+ except (OSError, ValueError):
160
+ return False
161
+ if not isinstance(stamp, dict) or stamp.get("revision") != GEMMA_REVISION:
162
+ return False
163
+ recorded = stamp.get("files")
164
+ if not isinstance(recorded, dict) or set(recorded) != set(GEMMA_FILES):
165
+ # Also how a folder left behind by a different graph is rejected: its stamp names
166
+ # files this manifest does not, so it is refetched rather than half-trusted.
167
+ return False
168
+ if model_dir.is_symlink():
169
+ return False
170
+ for name in recorded:
171
+ identity = _identity_in(model_dir, name)
172
+ # `None` means the path is not a regular file - a symlink, most likely. Nothing
173
+ # should be able to stamp one (`_unverified` calls it wrong), and this is the
174
+ # line that makes a stamp that somehow recorded `None` stop matching `None`.
175
+ if identity is None or recorded[name] != identity:
176
+ return False
177
+ return True
178
+
179
+
180
+ def _write_stamp(model_dir: Path) -> None:
181
+ """Record what was just verified. Atomically: a half-written stamp is a false claim."""
182
+ stamp = {
183
+ "revision": GEMMA_REVISION,
184
+ "files": {name: _file_identity(model_dir / name) for name in GEMMA_FILES},
185
+ }
186
+ temporary = model_dir / f"{_VERIFIED_STAMP}.{os.getpid()}"
187
+ _remove(temporary) # a stamp left behind by a crash under this same pid
188
+ _write_new_file(temporary, json.dumps(stamp).encode("utf-8"))
189
+ stamped = model_dir / _VERIFIED_STAMP
190
+ if stamped.is_dir() and not stamped.is_symlink():
191
+ # `os.replace` will not put a file where a directory is, and the stamp is read
192
+ # back as simply invalid, so the repair it ends would run again on every start
193
+ # and end the same way. Nothing of ours is ever a directory here.
194
+ _remove(stamped)
195
+ os.replace(temporary, stamped)
196
+
197
+
198
+ def _write_new_file(path: Path, payload: bytes) -> None:
199
+ """Create ``path`` with its contents, refusing to follow a symlink or reuse a file.
200
+
201
+ The temporary names these writes use are predictable (the pid), and a plain write
202
+ follows a symlink planted at one of them: the caller would then overwrite whatever it
203
+ points at, anywhere the user can write. `O_EXCL | O_NOFOLLOW` makes both refusals the
204
+ kernel's.
205
+ """
206
+ descriptor = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, 0o600)
207
+ try:
208
+ with os.fdopen(descriptor, "wb") as handle:
209
+ handle.write(payload)
210
+ except BaseException:
211
+ with contextlib.suppress(OSError):
212
+ path.unlink()
213
+ raise
214
+
215
+
216
+ def _remove(path: Path) -> None:
217
+ """Delete whatever sits at ``path``, file or directory.
218
+
219
+ A directory where a model file belongs is not something `unlink` can clear, and a
220
+ cache that cannot be repaired is a server that never starts again.
221
+ """
222
+ if path.is_dir() and not path.is_symlink():
223
+ shutil.rmtree(path, ignore_errors=True)
224
+ else:
225
+ path.unlink(missing_ok=True)
226
+ if path.exists() or path.is_symlink():
227
+ # Say so here, where the path is known, rather than failing three lines later on
228
+ # a rename that cannot explain itself.
229
+ raise ModelLoadError(f"Cannot clear {path} to repair the model cache")
230
+
231
+
232
+ def _hash_file(path: Path) -> str:
233
+ digest = hashlib.sha256()
234
+ with path.open("rb") as handle:
235
+ for block in iter(lambda: handle.read(1 << 20), b""):
236
+ digest.update(block)
237
+ return digest.hexdigest()
238
+
239
+
240
+ def _unverified(model_dir: Path) -> list[str]:
241
+ """The model files that are missing, the wrong size, or the wrong bytes."""
242
+ problems: list[str] = []
243
+ for name, (size, checksum) in GEMMA_MANIFEST.items():
244
+ path = _cache_path(model_dir, name)
245
+ before = None if path is None else _file_identity(path)
246
+ if path is None or before is None or before["size"] != size:
247
+ problems.append(name)
248
+ elif _hash_file(path) != checksum or _file_identity(path) != before:
249
+ # Second identity: a file rewritten while it was being read was never hashed
250
+ # as it now stands, so the answer that came back means nothing.
251
+ problems.append(name)
252
+ return problems
253
+
254
+
255
+ @contextlib.contextmanager
256
+ def _model_cache_lock(cache_dir: Path | None, *, exclusive: bool) -> Iterator[None]:
257
+ """Serialise verification, repair and session construction across processes.
258
+
259
+ Shared while a verified cache is being opened, exclusive while it is being changed,
260
+ so nothing can repair files another process has verified but not yet handed to
261
+ onnxruntime. The kernel drops a `flock` when the process dies, so a crash leaves
262
+ nothing held. A model cache on NFS or SMB shared between machines is out of scope:
263
+ `flock` can be local-only there - the boundary SQLite's WAL already has.
264
+ """
265
+ root = model_cache_root(cache_dir)
266
+ root.mkdir(parents=True, exist_ok=True)
267
+ with (root / _GEMMA_LOCK_NAME).open("a") as handle:
268
+ fcntl.flock(handle, fcntl.LOCK_EX if exclusive else fcntl.LOCK_SH)
269
+ try:
270
+ yield
271
+ finally:
272
+ fcntl.flock(handle, fcntl.LOCK_UN)
@@ -0,0 +1,339 @@
1
+ """Immutable, fully typed domain models.
2
+
3
+ Every entity is a frozen, slotted dataclass. ``to_dict`` methods produce the
4
+ JSON-serialisable payloads returned by the MCP tools.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from collections.abc import Sequence
10
+ from dataclasses import dataclass
11
+ from typing import TypeAlias
12
+
13
+ from pydantic import JsonValue
14
+
15
+ JsonDict: TypeAlias = dict[str, JsonValue]
16
+
17
+ PREAMBLE_TITLE = "[Overview / Preamble]"
18
+ PATH_SEPARATOR = " > "
19
+ CHARS_PER_TOKEN = 4
20
+
21
+
22
+ def estimate_tokens(text: str) -> int:
23
+ """Cheap token estimate (~4 characters per token), never below 1 for non-empty text."""
24
+ if not text:
25
+ return 0
26
+ return max(1, -(-len(text) // CHARS_PER_TOKEN))
27
+
28
+
29
+ def part_path(base_path: str, part_index: int) -> str:
30
+ """Breadcrumb for one part of an oversized section, e.g. ``A > B (Part 2)``."""
31
+ return f"{base_path} (Part {part_index})"
32
+
33
+
34
+ @dataclass(slots=True, frozen=True)
35
+ class SectionDraft:
36
+ """A section produced by the parser, not yet persisted.
37
+
38
+ ``part_index`` is ``0`` for a section stored whole and ``1..n`` for the
39
+ sequential parts of an oversized section. ``base_path`` is the breadcrumb
40
+ without the ``(Part n)`` suffix. Line numbers are 1-based and inclusive.
41
+
42
+ ``units`` are the section's passages as plain text - one per paragraph, list item,
43
+ table row or code block. Each is embedded separately, because one vector for a
44
+ whole section dilutes a single relevant table row beyond recognition. A section
45
+ without units is heading-only and gets no vectors at all.
46
+ """
47
+
48
+ heading_title: str
49
+ heading_level: int
50
+ heading_path: str
51
+ base_path: str
52
+ content: str
53
+ start_line: int
54
+ end_line: int
55
+ part_index: int = 0
56
+ units: tuple[str, ...] = ()
57
+
58
+ @property
59
+ def embedding_text(self) -> str:
60
+ """Text for the section-level vector: breadcrumb, then the body without markup."""
61
+ return f"{self.heading_path}\n\n{' '.join(self.units)}"
62
+
63
+ @property
64
+ def unit_texts(self) -> tuple[str, ...]:
65
+ """Texts for the passage-level vectors, each carrying the breadcrumb for context."""
66
+ return tuple(f"{self.heading_path}: {unit}" for unit in self.units)
67
+
68
+
69
+ @dataclass(slots=True, frozen=True)
70
+ class SectionVectors:
71
+ """Embeddings of one section: ``section`` is ``None`` for a heading-only section."""
72
+
73
+ section: Sequence[float] | None
74
+ units: tuple[Sequence[float], ...] = ()
75
+
76
+
77
+ @dataclass(slots=True, frozen=True)
78
+ class ParsedDocument:
79
+ """Result of parsing one Markdown source."""
80
+
81
+ title: str
82
+ sections: tuple[SectionDraft, ...]
83
+ line_count: int
84
+
85
+
86
+ @dataclass(slots=True, frozen=True)
87
+ class Document:
88
+ """A persisted Markdown file."""
89
+
90
+ id: int
91
+ file_path: str
92
+ title: str
93
+ content_hash: str
94
+ last_modified: int
95
+
96
+
97
+ @dataclass(slots=True, frozen=True)
98
+ class DocumentSummary:
99
+ """A document together with the number of sections indexed for it."""
100
+
101
+ file_path: str
102
+ title: str
103
+ section_count: int
104
+ last_modified: int
105
+
106
+ def to_dict(self) -> JsonDict:
107
+ return {
108
+ "file_path": self.file_path,
109
+ "title": self.title,
110
+ "section_count": self.section_count,
111
+ "last_modified": self.last_modified,
112
+ }
113
+
114
+
115
+ @dataclass(slots=True, frozen=True)
116
+ class Section:
117
+ """A persisted section (or one part of an oversized section)."""
118
+
119
+ id: int
120
+ doc_id: int
121
+ heading_title: str
122
+ heading_level: int
123
+ heading_path: str
124
+ content: str
125
+ start_line: int
126
+ end_line: int
127
+ part_index: int
128
+
129
+ @property
130
+ def base_path(self) -> str:
131
+ """Breadcrumb with any ``(Part n)`` suffix removed."""
132
+ if self.part_index <= 0:
133
+ return self.heading_path
134
+ suffix = part_path("", self.part_index)
135
+ return self.heading_path.removesuffix(suffix)
136
+
137
+
138
+ @dataclass(slots=True, frozen=True)
139
+ class OutlineNode:
140
+ """One heading in a document's hierarchical table of contents."""
141
+
142
+ heading_title: str
143
+ heading_level: int
144
+ heading_path: str
145
+ start_line: int
146
+ end_line: int
147
+ token_estimate: int
148
+ part_count: int
149
+ children: tuple[OutlineNode, ...] = ()
150
+
151
+ def to_dict(self) -> JsonDict:
152
+ payload: JsonDict = {
153
+ "title": self.heading_title,
154
+ "level": self.heading_level,
155
+ "heading_path": self.heading_path,
156
+ "lines": f"{self.start_line}-{self.end_line}",
157
+ "tokens": self.token_estimate,
158
+ }
159
+ if self.part_count > 1:
160
+ payload["parts"] = self.part_count
161
+ if self.children:
162
+ payload["children"] = [child.to_dict() for child in self.children]
163
+ return payload
164
+
165
+
166
+ @dataclass(slots=True, frozen=True)
167
+ class SearchResult:
168
+ """A section matched by hybrid search, with its fused and per-index ranks."""
169
+
170
+ section_id: int
171
+ file_path: str
172
+ document_title: str
173
+ heading_title: str
174
+ heading_path: str
175
+ content: str
176
+ start_line: int
177
+ end_line: int
178
+ score: float
179
+ fts_rank: int | None
180
+ vec_rank: int | None
181
+ matched_passage: str | None = None
182
+
183
+ def to_dict(self) -> JsonDict:
184
+ payload: JsonDict = {
185
+ "file_path": self.file_path,
186
+ "document_title": self.document_title,
187
+ "heading_path": self.heading_path,
188
+ "heading_title": self.heading_title,
189
+ "lines": f"{self.start_line}-{self.end_line}",
190
+ "score": round(self.score, 6),
191
+ "fts_rank": self.fts_rank,
192
+ "vec_rank": self.vec_rank,
193
+ "tokens": estimate_tokens(self.content),
194
+ "content": self.content,
195
+ }
196
+ if self.matched_passage is not None:
197
+ payload["matched_passage"] = self.matched_passage
198
+ return payload
199
+
200
+
201
+ @dataclass(slots=True, frozen=True)
202
+ class FileFailure:
203
+ """A per-file failure recorded while indexing (the run itself continues)."""
204
+
205
+ file_path: str
206
+ message: str
207
+
208
+ def to_dict(self) -> JsonDict:
209
+ return {"file_path": self.file_path, "message": self.message}
210
+
211
+
212
+ #: Failures carried on a search answer. Enough to act on, few enough not to bury the
213
+ #: answer itself - the count in `message` says how many were left out.
214
+ MAX_REPORTED_FAILURES = 20
215
+
216
+
217
+ @dataclass(slots=True, frozen=True)
218
+ class IndexStatus:
219
+ """Whether answers drawn from one tree can be trusted to be drawn from all of it.
220
+
221
+ `verified` means a full walk of this scope finished and read every file it found. It
222
+ is deliberately not a claim that the filesystem has stopped changing: a file created
223
+ after its directory was walked is not in the index and not in `failures`, and the next
224
+ run picks it up. A killed run and a tree nobody ever indexed both read unverified,
225
+ which is the same answer because it is the same situation - nothing walked it whole.
226
+ """
227
+
228
+ verified: bool
229
+ failures: tuple[FileFailure, ...] = ()
230
+ #: Set when the weights behind the model name changed under an existing index: the
231
+ #: stored vectors and the vectors a query would produce now come from different
232
+ #: models. Nothing is discarded, and nothing new is written, until it is resolved.
233
+ weights_mismatch: str | None = None
234
+ #: Documents under this scope whose vectors were built by an older pooling scheme.
235
+ #: They still answer, less well, and only a run over the directory holding them
236
+ #: rebuilds - a parent run prunes `.venv`, `node_modules` and the like, so one indexed
237
+ #: deliberately inside such a directory is never reached again.
238
+ stale_vectors: int = 0
239
+ #: Indexed documents a cheap probe could not confirm are still what was indexed: their
240
+ #: bytes differ, or they are gone, unreadable, or no longer a regular file. It is
241
+ #: best-effort in both directions. Counted over the rows the index holds, so a file
242
+ #: nobody has indexed yet is not in it - finding those needs the directory walk, which
243
+ #: is the expensive half. And the bytes are only read where the modification time moved,
244
+ #: so an edit that restores a file's own timestamp is not seen. Zero means nothing was
245
+ #: detected, not that every indexed file was hashed.
246
+ changed_files: int = 0
247
+ #: This server's own background run is indexing the tree right now. A hint about one
248
+ #: process only: another process's run shows as coverage withdrawn, as it always did.
249
+ indexing: bool = False
250
+
251
+ def to_dict(self) -> JsonDict:
252
+ shown = self.failures[:MAX_REPORTED_FAILURES]
253
+ return {
254
+ "coverage": "verified" if self.verified else "unknown",
255
+ "failures": [failure.to_dict() for failure in shown],
256
+ "changed_files": self.changed_files,
257
+ "indexing": self.indexing,
258
+ "message": self.message(),
259
+ }
260
+
261
+ def message(self) -> str | None:
262
+ """One sentence, or nothing at all when there is nothing to act on."""
263
+ # First, because it is the only one that says the answers themselves may be
264
+ # wrong rather than incomplete. Hoisted above `verified` rather than left below
265
+ # it: the two cannot both hold today, and a reader should not have to know that.
266
+ if self.weights_mismatch:
267
+ return self.weights_mismatch
268
+ if self.indexing:
269
+ # Before everything that ends in "run index_directory": that run would be
270
+ # refused as busy, and it would be refused for doing what is already being done.
271
+ return (
272
+ "An automatic index run is in progress, so an answer may be missing a file "
273
+ "changed or added since the last one finished; there is no need to run "
274
+ "index_directory."
275
+ )
276
+ if self.verified:
277
+ # A walk that finished still describes the moment it finished. Files edited
278
+ # since are the one thing a verified tree has left to say.
279
+ if self.changed_files:
280
+ return (
281
+ f"The last full index completed, but {self.changed_files} indexed "
282
+ "document(s) can no longer be confirmed to be what was indexed - "
283
+ "changed, unreadable or gone - so an answer may quote text that is no "
284
+ "longer there; run index_directory to refresh. The check is cheap and "
285
+ "best-effort: files created since that scan are not counted, and an "
286
+ "edit that puts a file's modification time back is not seen."
287
+ )
288
+ return None
289
+ if not self.failures:
290
+ if self.stale_vectors:
291
+ return (
292
+ f"{self.stale_vectors} document(s) here were indexed by an older "
293
+ "vector format and rank less well until the directory holding them is "
294
+ "indexed again."
295
+ )
296
+ return (
297
+ "This documentation root has not been indexed end to end since it last "
298
+ "changed, so an answer may be missing part of it. Run index_directory."
299
+ )
300
+ hidden = len(self.failures) - MAX_REPORTED_FAILURES
301
+ more = f" (showing the first {MAX_REPORTED_FAILURES})" if hidden > 0 else ""
302
+ return (
303
+ f"{len(self.failures)} path(s) could not be indexed{more}; answers here are "
304
+ "drawn from a tree that is missing them."
305
+ )
306
+
307
+
308
+ @dataclass(slots=True, frozen=True)
309
+ class IndexReport:
310
+ """Outcome of one ``index_directory`` run."""
311
+
312
+ directory: str
313
+ files_scanned: int
314
+ files_indexed: int
315
+ files_unchanged: int
316
+ files_purged: int
317
+ sections_indexed: int
318
+ elapsed_seconds: float
319
+ passages_indexed: int = 0
320
+ errors: tuple[FileFailure, ...] = ()
321
+ notes: tuple[str, ...] = ()
322
+
323
+ def summary(self) -> str:
324
+ lines = [
325
+ f"Indexed {self.directory} in {self.elapsed_seconds:.2f}s: "
326
+ f"{self.files_scanned} scanned, {self.files_indexed} (re)indexed, "
327
+ f"{self.files_unchanged} unchanged, {self.files_purged} purged, "
328
+ f"{self.sections_indexed} sections embedded ({self.passages_indexed} passages)."
329
+ ]
330
+ lines.extend(f"NOTE {note}" for note in self.notes)
331
+ lines.extend(f"ERROR {error.file_path}: {error.message}" for error in self.errors)
332
+ if self.errors:
333
+ # Without this the run reads as a success with some noise attached, and an
334
+ # index missing part of its tree answers questions as if it were whole.
335
+ lines.append(
336
+ f"INCOMPLETE: {len(self.errors)} file(s) could not be indexed; "
337
+ "this documentation root is only partly searchable."
338
+ )
339
+ return "\n".join(lines)