markdown-memory 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- markdown_memory/__init__.py +47 -0
- markdown_memory/autoindex.py +170 -0
- markdown_memory/config.py +235 -0
- markdown_memory/db.py +1546 -0
- markdown_memory/discovery.py +195 -0
- markdown_memory/embedders.py +513 -0
- markdown_memory/exceptions.py +71 -0
- markdown_memory/freshness.py +138 -0
- markdown_memory/headings.py +185 -0
- markdown_memory/indexer.py +725 -0
- markdown_memory/model_cache.py +272 -0
- markdown_memory/models.py +339 -0
- markdown_memory/parser.py +869 -0
- markdown_memory/py.typed +0 -0
- markdown_memory/search.py +518 -0
- markdown_memory/server.py +520 -0
- markdown_memory-0.1.0.dist-info/METADATA +579 -0
- markdown_memory-0.1.0.dist-info/RECORD +21 -0
- markdown_memory-0.1.0.dist-info/WHEEL +4 -0
- markdown_memory-0.1.0.dist-info/entry_points.txt +3 -0
- markdown_memory-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,520 @@
|
|
|
1
|
+
"""MCP server: the application service, tool handlers and the entrypoint.
|
|
2
|
+
|
|
3
|
+
Configuration lives in ``config.py``, the freshness sweep in ``freshness.py`` and
|
|
4
|
+
heading resolution in ``headings.py``.
|
|
5
|
+
|
|
6
|
+
stdout belongs to the JSON-RPC transport. Every log record goes to ``sys.stderr``;
|
|
7
|
+
nothing in this package calls ``print``.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import argparse
|
|
13
|
+
import dataclasses
|
|
14
|
+
import functools
|
|
15
|
+
import logging
|
|
16
|
+
import os
|
|
17
|
+
import sys
|
|
18
|
+
import threading
|
|
19
|
+
from collections.abc import AsyncIterator, Callable, Sequence
|
|
20
|
+
from contextlib import asynccontextmanager
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import ParamSpec, TypeVar
|
|
23
|
+
|
|
24
|
+
from mcp.server.mcpserver import MCPServer
|
|
25
|
+
from mcp.server.mcpserver.exceptions import ToolError
|
|
26
|
+
|
|
27
|
+
from markdown_memory import __version__, headings
|
|
28
|
+
from markdown_memory.autoindex import AutoIndexer
|
|
29
|
+
from markdown_memory.config import (
|
|
30
|
+
ENV_AUTO_INDEX,
|
|
31
|
+
ENV_DB_PATH,
|
|
32
|
+
ENV_DOCS_DIR,
|
|
33
|
+
ENV_EMBEDDER,
|
|
34
|
+
ENV_EXCLUDE,
|
|
35
|
+
ENV_LOG_LEVEL,
|
|
36
|
+
ServerConfig,
|
|
37
|
+
_config_from_cli,
|
|
38
|
+
)
|
|
39
|
+
from markdown_memory.db import Database
|
|
40
|
+
from markdown_memory.embedders import (
|
|
41
|
+
DEFAULT_EMBEDDER,
|
|
42
|
+
Embedder,
|
|
43
|
+
create_embedder,
|
|
44
|
+
)
|
|
45
|
+
from markdown_memory.exceptions import (
|
|
46
|
+
DocumentNotFoundError,
|
|
47
|
+
IndexingError,
|
|
48
|
+
MarkdownMemoryError,
|
|
49
|
+
SearchError,
|
|
50
|
+
)
|
|
51
|
+
from markdown_memory.freshness import FreshnessSweep
|
|
52
|
+
from markdown_memory.indexer import Indexer
|
|
53
|
+
from markdown_memory.models import (
|
|
54
|
+
Document,
|
|
55
|
+
DocumentSummary,
|
|
56
|
+
IndexReport,
|
|
57
|
+
IndexStatus,
|
|
58
|
+
JsonDict,
|
|
59
|
+
OutlineNode,
|
|
60
|
+
SearchResult,
|
|
61
|
+
)
|
|
62
|
+
from markdown_memory.parser import join_parts
|
|
63
|
+
from markdown_memory.search import HybridSearcher
|
|
64
|
+
|
|
65
|
+
logger = logging.getLogger(__name__)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
_P = ParamSpec("_P")
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
_R = TypeVar("_R")
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
SERVER_INSTRUCTIONS = (
|
|
75
|
+
"Markdown documentation memory. The server keeps its documentation root indexed by "
|
|
76
|
+
"itself (unless started with --no-auto-index); call index_directory only when "
|
|
77
|
+
"index_status says so. Use search_docs to locate relevant sections, or "
|
|
78
|
+
"get_document_outline followed by read_section to fetch one heading's text. Prefer "
|
|
79
|
+
"these tools over reading whole Markdown files."
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class MarkdownMemoryService:
|
|
84
|
+
"""Application service behind the MCP tools; returns typed domain models."""
|
|
85
|
+
|
|
86
|
+
def __init__(self, config: ServerConfig, embedder: Embedder | None = None) -> None:
|
|
87
|
+
self._config = config
|
|
88
|
+
self._embedder: Embedder = embedder or create_embedder(
|
|
89
|
+
config.embedder, cache_dir=config.model_cache_dir
|
|
90
|
+
)
|
|
91
|
+
self._db = Database(config.db_path, embedding_dim=self._embedder.dimension)
|
|
92
|
+
self._indexer = Indexer(
|
|
93
|
+
self._db, self._embedder, workers=config.index_workers, exclude=config.exclude
|
|
94
|
+
)
|
|
95
|
+
# Resolved, because indexing resolves: a document under a symlinked or relative
|
|
96
|
+
# docs root is stored by its real path, and a scope spelled any other way filters
|
|
97
|
+
# every one of them out and returns nothing. Resolved ONCE, and reused: resolving
|
|
98
|
+
# again per call lets a retargeted symlink answer from one tree while reporting on
|
|
99
|
+
# another, which is a lie told with two correct halves.
|
|
100
|
+
self._root = str(headings._absolute(self._config.docs_dir, SearchError))
|
|
101
|
+
self._freshness = FreshnessSweep(self._db)
|
|
102
|
+
self._searcher = HybridSearcher(self._db, self._embedder, scope=self._root)
|
|
103
|
+
#: Only the stdio server starts one (`main`): a service built by a test or a
|
|
104
|
+
#: script does exactly what it is asked and nothing in the background.
|
|
105
|
+
self._auto: AutoIndexer | None = None
|
|
106
|
+
|
|
107
|
+
@property
|
|
108
|
+
def db(self) -> Database:
|
|
109
|
+
return self._db
|
|
110
|
+
|
|
111
|
+
@property
|
|
112
|
+
def embedder(self) -> Embedder:
|
|
113
|
+
return self._embedder
|
|
114
|
+
|
|
115
|
+
def close(self) -> None:
|
|
116
|
+
# The background run first, and waited for: it writes through the database that
|
|
117
|
+
# is about to close, and a stop lands between two documents.
|
|
118
|
+
if self._auto is not None:
|
|
119
|
+
self._auto.stop()
|
|
120
|
+
self._searcher.close()
|
|
121
|
+
self._db.close()
|
|
122
|
+
|
|
123
|
+
def start_auto_index(self, *, request: bool = True) -> None:
|
|
124
|
+
"""Keep the docs root indexed from now on, as searches see change.
|
|
125
|
+
|
|
126
|
+
`request=False` arms the runner without starting a run: the first search then finds
|
|
127
|
+
it has never run and starts the catch-up itself. That is what the stdio server does,
|
|
128
|
+
so nothing loads the model before the client's handshake has been answered.
|
|
129
|
+
"""
|
|
130
|
+
if self._auto is None:
|
|
131
|
+
self._auto = AutoIndexer(
|
|
132
|
+
lambda should_stop: self.index_directory(None, should_stop=should_stop),
|
|
133
|
+
self._root_status,
|
|
134
|
+
)
|
|
135
|
+
if request:
|
|
136
|
+
self._auto.request()
|
|
137
|
+
|
|
138
|
+
# ------------------------------------------------------------------ operations
|
|
139
|
+
|
|
140
|
+
def index_directory(
|
|
141
|
+
self, directory: str | None = None, should_stop: Callable[[], bool] | None = None
|
|
142
|
+
) -> IndexReport:
|
|
143
|
+
try:
|
|
144
|
+
return self._indexer.index_directory(
|
|
145
|
+
self._resolve_directory(directory), should_stop=should_stop
|
|
146
|
+
)
|
|
147
|
+
finally:
|
|
148
|
+
# Whatever just happened, the sweep's answer is about the tree as it was
|
|
149
|
+
# before it: a run that refreshed the files it named would otherwise keep
|
|
150
|
+
# being reported as stale for the rest of the window, which is exactly the
|
|
151
|
+
# moment an agent looks. A run that failed partway invalidates it too - some
|
|
152
|
+
# of it may have been written.
|
|
153
|
+
self._freshness.invalidate()
|
|
154
|
+
|
|
155
|
+
def list_documents(self, directory: str = "") -> list[DocumentSummary]:
|
|
156
|
+
# An empty argument means "this project", not "everything this database holds".
|
|
157
|
+
# The default database is per-root now, but a configured MARKDOWN_MEMORY_DB can
|
|
158
|
+
# still be shared, and a database outlives the root it was first keyed to.
|
|
159
|
+
scope = self._resolve_directory(directory or None)
|
|
160
|
+
return self._db.list_documents(str(self._within_root(scope, IndexingError)))
|
|
161
|
+
|
|
162
|
+
def get_document_outline(self, file_path: str) -> list[OutlineNode]:
|
|
163
|
+
document = self._resolve_document(file_path)
|
|
164
|
+
return headings.build_outline(self._db.get_sections(document.id))
|
|
165
|
+
|
|
166
|
+
def read_section(
|
|
167
|
+
self, file_path: str, heading_path: str, *, include_subsections: bool = False
|
|
168
|
+
) -> str:
|
|
169
|
+
document = self._resolve_document(file_path)
|
|
170
|
+
sections = self._db.get_sections(document.id)
|
|
171
|
+
matched = headings.select_sections(
|
|
172
|
+
sections, heading_path, include_subsections=include_subsections
|
|
173
|
+
)
|
|
174
|
+
return join_parts(matched)
|
|
175
|
+
|
|
176
|
+
def search_docs(self, query: str, limit: int = 5) -> list[SearchResult]:
|
|
177
|
+
return self._searcher.search(query, limit)
|
|
178
|
+
|
|
179
|
+
def index_status(self, directory: str | None = None) -> IndexStatus:
|
|
180
|
+
"""What can honestly be said about answers drawn from this server's documents.
|
|
181
|
+
|
|
182
|
+
Indexing reports its own failures, but almost nothing calls indexing: an agent
|
|
183
|
+
opens a session, searches, and is served from whatever the index happens to hold.
|
|
184
|
+
Until this is asked at the point of use, a root that lost files to a permissions
|
|
185
|
+
error - or was never indexed at all - answers with confidence and no caveat.
|
|
186
|
+
|
|
187
|
+
Coverage is always the configured docs root's, because that is the tree every
|
|
188
|
+
answer is drawn from; `directory` only narrows which failures are worth naming.
|
|
189
|
+
"""
|
|
190
|
+
if directory is None:
|
|
191
|
+
status = self._root_status()
|
|
192
|
+
# The root's own status is what a search reads, so it is also what decides
|
|
193
|
+
# whether the background run is due - there is no other look at the disk.
|
|
194
|
+
if self._auto is not None:
|
|
195
|
+
self._auto.consider(status)
|
|
196
|
+
return self._with_indexing(status)
|
|
197
|
+
# Narrowed in one read, not composed from two: coverage stays the root's - that is
|
|
198
|
+
# the tree every answer is drawn from - while the failures and stale documents
|
|
199
|
+
# named are the ones that live here. Resolved against the root this service was
|
|
200
|
+
# built for, never against the configured path again, or a retargeted symlink
|
|
201
|
+
# pairs this root's certificate with another tree's failures.
|
|
202
|
+
scope = self._within_root(
|
|
203
|
+
headings._absolute(Path(self._root) / directory.strip(), SearchError), SearchError
|
|
204
|
+
)
|
|
205
|
+
return self._with_indexing(
|
|
206
|
+
self._with_freshness(self._db.index_status(self._root, str(scope)), str(scope))
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
def _root_status(self) -> IndexStatus:
|
|
210
|
+
return self._with_freshness(self._db.index_status(self._root), self._root)
|
|
211
|
+
|
|
212
|
+
def _with_indexing(self, status: IndexStatus) -> IndexStatus:
|
|
213
|
+
active = self._auto is not None and self._auto.active
|
|
214
|
+
return dataclasses.replace(status, indexing=active) if active else status
|
|
215
|
+
|
|
216
|
+
def _with_freshness(self, status: IndexStatus, scope: str) -> IndexStatus:
|
|
217
|
+
"""Add what only the filesystem knows: which indexed files have moved on.
|
|
218
|
+
|
|
219
|
+
The database's own status is one SQLite snapshot and deliberately says nothing
|
|
220
|
+
about the disk, so this is composed here rather than there. It asks only about
|
|
221
|
+
rows the index holds - a file nobody has indexed yet is found by walking the tree,
|
|
222
|
+
which is the expensive half of indexing and not something a search should pay for.
|
|
223
|
+
"""
|
|
224
|
+
return dataclasses.replace(status, changed_files=self._freshness.changed_files(scope))
|
|
225
|
+
|
|
226
|
+
# ------------------------------------------------------------------ resolution
|
|
227
|
+
|
|
228
|
+
def _resolve_directory(self, directory: str | None) -> Path:
|
|
229
|
+
"""Resolve against the root this server settled on, never the configured spelling.
|
|
230
|
+
|
|
231
|
+
The docs root is resolved once at construction precisely so a retargeted symlink
|
|
232
|
+
cannot make the server answer from one tree while reporting on another. Resolving
|
|
233
|
+
the configured path again here reopened that door from the other side: indexing
|
|
234
|
+
followed the link to its new target and wrote rows the frozen root can never see,
|
|
235
|
+
`list_documents()` with no argument then resolved outside its own root and raised,
|
|
236
|
+
and a restart keyed a different database and read as never indexed.
|
|
237
|
+
"""
|
|
238
|
+
if directory is None or not directory.strip():
|
|
239
|
+
return Path(self._root)
|
|
240
|
+
path = headings._user_path(directory.strip(), IndexingError)
|
|
241
|
+
if not path.is_absolute():
|
|
242
|
+
path = Path(self._root) / path
|
|
243
|
+
return headings._absolute(path, IndexingError)
|
|
244
|
+
|
|
245
|
+
def _within_root(self, resolved: Path, error: type[MarkdownMemoryError]) -> Path:
|
|
246
|
+
"""Refuse to *answer about* a directory outside the tree this server serves.
|
|
247
|
+
|
|
248
|
+
`..`, an absolute path and a symlink each reach out of the root, and each was
|
|
249
|
+
obeyed: `list_documents` handed back another project's file paths, and a status
|
|
250
|
+
lookup paired this root's certificate with that tree's failures - `coverage:
|
|
251
|
+
verified` beside a non-empty failure list, which the envelope promises cannot
|
|
252
|
+
happen. Search has been scoped to the root ever since it answered one project's
|
|
253
|
+
question from another's documentation; these are the two other ways in.
|
|
254
|
+
|
|
255
|
+
`index_directory` is deliberately not scoped this way. It is an instruction rather
|
|
256
|
+
than a question - go and index that tree - and it keys the tree it walked under its
|
|
257
|
+
own root, so nothing it writes is attributed here.
|
|
258
|
+
"""
|
|
259
|
+
root = Path(self._root)
|
|
260
|
+
if resolved != root and root not in resolved.parents:
|
|
261
|
+
raise error(
|
|
262
|
+
f"{str(resolved)!r} is outside this server's documentation root "
|
|
263
|
+
f"({self._root}); name a directory inside it."
|
|
264
|
+
)
|
|
265
|
+
return resolved
|
|
266
|
+
|
|
267
|
+
def _resolve_document(self, file_path: str) -> Document:
|
|
268
|
+
"""Find an indexed document by absolute path, relative path, or unique path suffix."""
|
|
269
|
+
requested = file_path.strip()
|
|
270
|
+
if not requested:
|
|
271
|
+
raise DocumentNotFoundError("file_path must not be empty")
|
|
272
|
+
path = headings._user_path(requested, DocumentNotFoundError)
|
|
273
|
+
candidates = [path]
|
|
274
|
+
if not path.is_absolute():
|
|
275
|
+
candidates = [self._config.docs_dir / path]
|
|
276
|
+
try:
|
|
277
|
+
candidates.append(Path.cwd() / path)
|
|
278
|
+
except OSError: # the working directory was deleted under the server
|
|
279
|
+
logger.debug("Working directory is gone; not resolving %s against it", requested)
|
|
280
|
+
for candidate in candidates:
|
|
281
|
+
resolved = headings._absolute(candidate, DocumentNotFoundError)
|
|
282
|
+
document = self._db.get_document(str(resolved))
|
|
283
|
+
if document is not None:
|
|
284
|
+
return document
|
|
285
|
+
matches = self._db.find_documents_by_suffix(requested)
|
|
286
|
+
if len(matches) == 1:
|
|
287
|
+
return matches[0]
|
|
288
|
+
if matches:
|
|
289
|
+
listing = ", ".join(match.file_path for match in matches[: headings._MAX_LISTED_PATHS])
|
|
290
|
+
raise DocumentNotFoundError(f"'{requested}' is ambiguous; it matches: {listing}")
|
|
291
|
+
raise DocumentNotFoundError(
|
|
292
|
+
f"'{requested}' is not indexed. Run index_directory, then list_documents "
|
|
293
|
+
"to see the available paths."
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
# ---------------------------------------------------------------------- MCP wiring
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def anticipated_errors(function: Callable[_P, _R]) -> Callable[_P, _R]:
|
|
301
|
+
"""Report domain failures to the model verbatim.
|
|
302
|
+
|
|
303
|
+
The SDK hides the message of any exception other than ``ToolError`` behind a generic
|
|
304
|
+
"Error executing tool" (treating it as a crash). Domain errors carry actionable text
|
|
305
|
+
- available heading paths, "run index_directory first" - that the agent needs to see.
|
|
306
|
+
"""
|
|
307
|
+
|
|
308
|
+
@functools.wraps(function)
|
|
309
|
+
def wrapper(*args: _P.args, **kwargs: _P.kwargs) -> _R:
|
|
310
|
+
try:
|
|
311
|
+
return function(*args, **kwargs)
|
|
312
|
+
except MarkdownMemoryError as exc:
|
|
313
|
+
raise ToolError(str(exc)) from exc
|
|
314
|
+
|
|
315
|
+
return wrapper
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
class _ServiceProvider:
|
|
319
|
+
"""Hands the tools their service and owns its lifetime when nobody else does.
|
|
320
|
+
|
|
321
|
+
A service passed in by the caller is never closed here. One the server creates itself
|
|
322
|
+
lives as long as any session is open, and is created again on demand afterwards, so
|
|
323
|
+
the same server object can serve several sessions (or concurrent ones) in a row.
|
|
324
|
+
"""
|
|
325
|
+
|
|
326
|
+
def __init__(self, config: ServerConfig | None, service: MarkdownMemoryService | None) -> None:
|
|
327
|
+
self._config = config
|
|
328
|
+
self._external = service
|
|
329
|
+
self._owned: MarkdownMemoryService | None = None
|
|
330
|
+
self._sessions = 0
|
|
331
|
+
self._lock = threading.Lock()
|
|
332
|
+
|
|
333
|
+
def get(self) -> MarkdownMemoryService:
|
|
334
|
+
if self._external is not None:
|
|
335
|
+
return self._external
|
|
336
|
+
with self._lock:
|
|
337
|
+
if self._owned is None:
|
|
338
|
+
self._owned = MarkdownMemoryService(self._config or ServerConfig.from_env())
|
|
339
|
+
return self._owned
|
|
340
|
+
|
|
341
|
+
def session_started(self) -> None:
|
|
342
|
+
with self._lock:
|
|
343
|
+
self._sessions += 1
|
|
344
|
+
|
|
345
|
+
def session_ended(self) -> None:
|
|
346
|
+
with self._lock:
|
|
347
|
+
self._sessions -= 1
|
|
348
|
+
if self._sessions > 0 or self._owned is None:
|
|
349
|
+
return
|
|
350
|
+
owned, self._owned = self._owned, None
|
|
351
|
+
owned.close()
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def create_server(
|
|
355
|
+
config: ServerConfig | None = None, *, service: MarkdownMemoryService | None = None
|
|
356
|
+
) -> MCPServer[None]:
|
|
357
|
+
"""Build the MCP server and register its tools."""
|
|
358
|
+
services = _ServiceProvider(config, service)
|
|
359
|
+
|
|
360
|
+
@asynccontextmanager
|
|
361
|
+
async def lifespan(_server: MCPServer[None]) -> AsyncIterator[None]:
|
|
362
|
+
services.session_started()
|
|
363
|
+
try:
|
|
364
|
+
yield
|
|
365
|
+
finally:
|
|
366
|
+
services.session_ended()
|
|
367
|
+
|
|
368
|
+
server: MCPServer[None] = MCPServer(
|
|
369
|
+
"markdown-memory", version=__version__, instructions=SERVER_INSTRUCTIONS, lifespan=lifespan
|
|
370
|
+
)
|
|
371
|
+
|
|
372
|
+
@server.tool()
|
|
373
|
+
@anticipated_errors
|
|
374
|
+
def index_directory(directory: str | None = None) -> str:
|
|
375
|
+
"""Scan a directory tree for Markdown files and (re)index new or changed ones.
|
|
376
|
+
|
|
377
|
+
Unchanged files are skipped via SHA-256 hashes; files deleted from disk are purged.
|
|
378
|
+
Omit `directory` to index the server's configured documentation root.
|
|
379
|
+
"""
|
|
380
|
+
return services.get().index_directory(directory).summary()
|
|
381
|
+
|
|
382
|
+
@server.tool()
|
|
383
|
+
@anticipated_errors
|
|
384
|
+
def list_documents(directory: str = "") -> JsonDict:
|
|
385
|
+
"""List indexed documents (path, title, section count), optionally under `directory`.
|
|
386
|
+
|
|
387
|
+
Returns `{"documents": [...], "index_status": {...}}`. `index_status.changed_files`
|
|
388
|
+
counts indexed documents that no longer match the index; it is independent of
|
|
389
|
+
coverage, so read `index_status.message` whenever either is set.
|
|
390
|
+
`index_status.coverage` is
|
|
391
|
+
"verified" only when a full index run of this documentation root finished and read
|
|
392
|
+
every file it found; otherwise it is "unknown" and `index_status.message` says why.
|
|
393
|
+
"""
|
|
394
|
+
service = services.get()
|
|
395
|
+
scope = directory if directory.strip() else None
|
|
396
|
+
return {
|
|
397
|
+
"documents": [summary.to_dict() for summary in service.list_documents(directory)],
|
|
398
|
+
"index_status": service.index_status(scope).to_dict(),
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
@server.tool()
|
|
402
|
+
@anticipated_errors
|
|
403
|
+
def get_document_outline(file_path: str) -> list[JsonDict]:
|
|
404
|
+
"""Hierarchical table of contents of one document: heading paths, line ranges and
|
|
405
|
+
token estimates. Costs a few hundred tokens; use it to pick a section to read."""
|
|
406
|
+
return [node.to_dict() for node in services.get().get_document_outline(file_path)]
|
|
407
|
+
|
|
408
|
+
@server.tool()
|
|
409
|
+
@anticipated_errors
|
|
410
|
+
def read_section(file_path: str, heading_path: str, include_subsections: bool = False) -> str:
|
|
411
|
+
"""Return the verbatim Markdown of one section, addressed by its breadcrumb
|
|
412
|
+
(`Root > Child > Subchild`, as shown by get_document_outline). By default only the
|
|
413
|
+
section's own text is returned; set `include_subsections` to append its children."""
|
|
414
|
+
return services.get().read_section(
|
|
415
|
+
file_path, heading_path, include_subsections=include_subsections
|
|
416
|
+
)
|
|
417
|
+
|
|
418
|
+
@server.tool()
|
|
419
|
+
@anticipated_errors
|
|
420
|
+
def search_docs(query: str, limit: int = 5) -> JsonDict:
|
|
421
|
+
"""Hybrid search (BM25 keywords + semantic vectors, fused with RRF) over all indexed
|
|
422
|
+
sections. Works for exact identifiers (flags, env vars) and for conceptual questions.
|
|
423
|
+
|
|
424
|
+
Returns `{"results": [...], "index_status": {...}}`, `results` holding at most
|
|
425
|
+
`limit` sections. `index_status.changed_files` counts indexed documents that no
|
|
426
|
+
longer match the index - a hit may quote text that is no longer there - and is
|
|
427
|
+
independent of coverage: it can be non-zero while coverage reads "verified", so
|
|
428
|
+
read `index_status.message` whenever either is set.
|
|
429
|
+
When `index_status.coverage` is "unknown", what you searched is
|
|
430
|
+
missing part of its documentation, or was never indexed end to end: an answer drawn
|
|
431
|
+
from it may be confidently incomplete, and `index_status.message` says what to run.
|
|
432
|
+
"""
|
|
433
|
+
service = services.get()
|
|
434
|
+
return {
|
|
435
|
+
"results": [result.to_dict() for result in service.search_docs(query, limit)],
|
|
436
|
+
"index_status": service.index_status().to_dict(),
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
return server
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def configure_logging(level: str | None = None) -> None:
|
|
443
|
+
"""Route all logging to stderr; stdout is reserved for JSON-RPC frames."""
|
|
444
|
+
name = (level or os.environ.get(ENV_LOG_LEVEL, "") or "INFO").upper()
|
|
445
|
+
logging.basicConfig(
|
|
446
|
+
level=getattr(logging, name, logging.INFO),
|
|
447
|
+
stream=sys.stderr,
|
|
448
|
+
format="%(asctime)s %(levelname)s %(name)s: %(message)s",
|
|
449
|
+
force=True,
|
|
450
|
+
)
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
def _download_model(config: ServerConfig) -> None:
|
|
454
|
+
"""Fetch and load the configured embedder, so no client waits on the first download."""
|
|
455
|
+
try:
|
|
456
|
+
create_embedder(config.embedder, cache_dir=config.model_cache_dir).warm_up()
|
|
457
|
+
except MarkdownMemoryError:
|
|
458
|
+
logger.exception("Cannot download the embedding model")
|
|
459
|
+
raise SystemExit(1) from None
|
|
460
|
+
logger.info("Embedding model %s is ready in %s", config.embedder, config.model_cache_dir)
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
def main(argv: Sequence[str] | None = None) -> None:
|
|
464
|
+
"""Console entrypoint: serve MCP over stdio."""
|
|
465
|
+
parser = argparse.ArgumentParser(
|
|
466
|
+
prog="markdown-memory", description="Markdown documentation memory MCP server (stdio)."
|
|
467
|
+
)
|
|
468
|
+
parser.add_argument("--db", type=Path, help=f"SQLite database path (env {ENV_DB_PATH})")
|
|
469
|
+
parser.add_argument("--docs-dir", type=Path, help=f"Default docs root (env {ENV_DOCS_DIR})")
|
|
470
|
+
parser.add_argument("--log-level", help=f"Logging level (env {ENV_LOG_LEVEL})")
|
|
471
|
+
parser.add_argument(
|
|
472
|
+
"--exclude",
|
|
473
|
+
action="append",
|
|
474
|
+
default=[],
|
|
475
|
+
metavar="GLOB",
|
|
476
|
+
help=f"Skip paths matching this glob, relative to the docs root; repeatable "
|
|
477
|
+
f"(env {ENV_EXCLUDE}, comma separated)",
|
|
478
|
+
)
|
|
479
|
+
parser.add_argument(
|
|
480
|
+
"--no-auto-index",
|
|
481
|
+
action="store_true",
|
|
482
|
+
help=f"Do not keep the docs root indexed in the background (env {ENV_AUTO_INDEX}=0)",
|
|
483
|
+
)
|
|
484
|
+
parser.add_argument(
|
|
485
|
+
"--embedder",
|
|
486
|
+
choices=("embeddinggemma", "bge-small"),
|
|
487
|
+
help=f"Embedding model preset (env {ENV_EMBEDDER}; default {DEFAULT_EMBEDDER})",
|
|
488
|
+
)
|
|
489
|
+
parser.add_argument(
|
|
490
|
+
"--download-model",
|
|
491
|
+
action="store_true",
|
|
492
|
+
help="Download and load the configured embedding model, then exit",
|
|
493
|
+
)
|
|
494
|
+
parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
|
|
495
|
+
arguments = parser.parse_args(argv)
|
|
496
|
+
|
|
497
|
+
configure_logging(arguments.log_level)
|
|
498
|
+
config = _config_from_cli(arguments)
|
|
499
|
+
if arguments.download_model:
|
|
500
|
+
_download_model(config)
|
|
501
|
+
return
|
|
502
|
+
try:
|
|
503
|
+
service = MarkdownMemoryService(config)
|
|
504
|
+
except MarkdownMemoryError:
|
|
505
|
+
logger.exception("Cannot start markdown-memory")
|
|
506
|
+
raise SystemExit(1) from None
|
|
507
|
+
logger.info("markdown-memory serving; db=%s docs_dir=%s", config.db_path, config.docs_dir)
|
|
508
|
+
# Nothing heavy starts here. Loading the model holds the GIL for seconds, and a client
|
|
509
|
+
# waits on the handshake with a short timeout - Codex's is 10 s. The first search loads
|
|
510
|
+
# it instead, and starts the catch-up index run on its way out.
|
|
511
|
+
if config.auto_index:
|
|
512
|
+
service.start_auto_index(request=False)
|
|
513
|
+
try:
|
|
514
|
+
create_server(config, service=service).run("stdio")
|
|
515
|
+
finally:
|
|
516
|
+
service.close()
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
if __name__ == "__main__":
|
|
520
|
+
main()
|