markdown-memory 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,520 @@
1
+ """MCP server: the application service, tool handlers and the entrypoint.
2
+
3
+ Configuration lives in ``config.py``, the freshness sweep in ``freshness.py`` and
4
+ heading resolution in ``headings.py``.
5
+
6
+ stdout belongs to the JSON-RPC transport. Every log record goes to ``sys.stderr``;
7
+ nothing in this package calls ``print``.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import argparse
13
+ import dataclasses
14
+ import functools
15
+ import logging
16
+ import os
17
+ import sys
18
+ import threading
19
+ from collections.abc import AsyncIterator, Callable, Sequence
20
+ from contextlib import asynccontextmanager
21
+ from pathlib import Path
22
+ from typing import ParamSpec, TypeVar
23
+
24
+ from mcp.server.mcpserver import MCPServer
25
+ from mcp.server.mcpserver.exceptions import ToolError
26
+
27
+ from markdown_memory import __version__, headings
28
+ from markdown_memory.autoindex import AutoIndexer
29
+ from markdown_memory.config import (
30
+ ENV_AUTO_INDEX,
31
+ ENV_DB_PATH,
32
+ ENV_DOCS_DIR,
33
+ ENV_EMBEDDER,
34
+ ENV_EXCLUDE,
35
+ ENV_LOG_LEVEL,
36
+ ServerConfig,
37
+ _config_from_cli,
38
+ )
39
+ from markdown_memory.db import Database
40
+ from markdown_memory.embedders import (
41
+ DEFAULT_EMBEDDER,
42
+ Embedder,
43
+ create_embedder,
44
+ )
45
+ from markdown_memory.exceptions import (
46
+ DocumentNotFoundError,
47
+ IndexingError,
48
+ MarkdownMemoryError,
49
+ SearchError,
50
+ )
51
+ from markdown_memory.freshness import FreshnessSweep
52
+ from markdown_memory.indexer import Indexer
53
+ from markdown_memory.models import (
54
+ Document,
55
+ DocumentSummary,
56
+ IndexReport,
57
+ IndexStatus,
58
+ JsonDict,
59
+ OutlineNode,
60
+ SearchResult,
61
+ )
62
+ from markdown_memory.parser import join_parts
63
+ from markdown_memory.search import HybridSearcher
64
+
65
+ logger = logging.getLogger(__name__)
66
+
67
+
68
+ _P = ParamSpec("_P")
69
+
70
+
71
+ _R = TypeVar("_R")
72
+
73
+
74
+ SERVER_INSTRUCTIONS = (
75
+ "Markdown documentation memory. The server keeps its documentation root indexed by "
76
+ "itself (unless started with --no-auto-index); call index_directory only when "
77
+ "index_status says so. Use search_docs to locate relevant sections, or "
78
+ "get_document_outline followed by read_section to fetch one heading's text. Prefer "
79
+ "these tools over reading whole Markdown files."
80
+ )
81
+
82
+
83
+ class MarkdownMemoryService:
84
+ """Application service behind the MCP tools; returns typed domain models."""
85
+
86
+ def __init__(self, config: ServerConfig, embedder: Embedder | None = None) -> None:
87
+ self._config = config
88
+ self._embedder: Embedder = embedder or create_embedder(
89
+ config.embedder, cache_dir=config.model_cache_dir
90
+ )
91
+ self._db = Database(config.db_path, embedding_dim=self._embedder.dimension)
92
+ self._indexer = Indexer(
93
+ self._db, self._embedder, workers=config.index_workers, exclude=config.exclude
94
+ )
95
+ # Resolved, because indexing resolves: a document under a symlinked or relative
96
+ # docs root is stored by its real path, and a scope spelled any other way filters
97
+ # every one of them out and returns nothing. Resolved ONCE, and reused: resolving
98
+ # again per call lets a retargeted symlink answer from one tree while reporting on
99
+ # another, which is a lie told with two correct halves.
100
+ self._root = str(headings._absolute(self._config.docs_dir, SearchError))
101
+ self._freshness = FreshnessSweep(self._db)
102
+ self._searcher = HybridSearcher(self._db, self._embedder, scope=self._root)
103
+ #: Only the stdio server starts one (`main`): a service built by a test or a
104
+ #: script does exactly what it is asked and nothing in the background.
105
+ self._auto: AutoIndexer | None = None
106
+
107
+ @property
108
+ def db(self) -> Database:
109
+ return self._db
110
+
111
+ @property
112
+ def embedder(self) -> Embedder:
113
+ return self._embedder
114
+
115
+ def close(self) -> None:
116
+ # The background run first, and waited for: it writes through the database that
117
+ # is about to close, and a stop lands between two documents.
118
+ if self._auto is not None:
119
+ self._auto.stop()
120
+ self._searcher.close()
121
+ self._db.close()
122
+
123
+ def start_auto_index(self, *, request: bool = True) -> None:
124
+ """Keep the docs root indexed from now on, as searches see change.
125
+
126
+ `request=False` arms the runner without starting a run: the first search then finds
127
+ it has never run and starts the catch-up itself. That is what the stdio server does,
128
+ so nothing loads the model before the client's handshake has been answered.
129
+ """
130
+ if self._auto is None:
131
+ self._auto = AutoIndexer(
132
+ lambda should_stop: self.index_directory(None, should_stop=should_stop),
133
+ self._root_status,
134
+ )
135
+ if request:
136
+ self._auto.request()
137
+
138
+ # ------------------------------------------------------------------ operations
139
+
140
+ def index_directory(
141
+ self, directory: str | None = None, should_stop: Callable[[], bool] | None = None
142
+ ) -> IndexReport:
143
+ try:
144
+ return self._indexer.index_directory(
145
+ self._resolve_directory(directory), should_stop=should_stop
146
+ )
147
+ finally:
148
+ # Whatever just happened, the sweep's answer is about the tree as it was
149
+ # before it: a run that refreshed the files it named would otherwise keep
150
+ # being reported as stale for the rest of the window, which is exactly the
151
+ # moment an agent looks. A run that failed partway invalidates it too - some
152
+ # of it may have been written.
153
+ self._freshness.invalidate()
154
+
155
+ def list_documents(self, directory: str = "") -> list[DocumentSummary]:
156
+ # An empty argument means "this project", not "everything this database holds".
157
+ # The default database is per-root now, but a configured MARKDOWN_MEMORY_DB can
158
+ # still be shared, and a database outlives the root it was first keyed to.
159
+ scope = self._resolve_directory(directory or None)
160
+ return self._db.list_documents(str(self._within_root(scope, IndexingError)))
161
+
162
+ def get_document_outline(self, file_path: str) -> list[OutlineNode]:
163
+ document = self._resolve_document(file_path)
164
+ return headings.build_outline(self._db.get_sections(document.id))
165
+
166
+ def read_section(
167
+ self, file_path: str, heading_path: str, *, include_subsections: bool = False
168
+ ) -> str:
169
+ document = self._resolve_document(file_path)
170
+ sections = self._db.get_sections(document.id)
171
+ matched = headings.select_sections(
172
+ sections, heading_path, include_subsections=include_subsections
173
+ )
174
+ return join_parts(matched)
175
+
176
+ def search_docs(self, query: str, limit: int = 5) -> list[SearchResult]:
177
+ return self._searcher.search(query, limit)
178
+
179
+ def index_status(self, directory: str | None = None) -> IndexStatus:
180
+ """What can honestly be said about answers drawn from this server's documents.
181
+
182
+ Indexing reports its own failures, but almost nothing calls indexing: an agent
183
+ opens a session, searches, and is served from whatever the index happens to hold.
184
+ Until this is asked at the point of use, a root that lost files to a permissions
185
+ error - or was never indexed at all - answers with confidence and no caveat.
186
+
187
+ Coverage is always the configured docs root's, because that is the tree every
188
+ answer is drawn from; `directory` only narrows which failures are worth naming.
189
+ """
190
+ if directory is None:
191
+ status = self._root_status()
192
+ # The root's own status is what a search reads, so it is also what decides
193
+ # whether the background run is due - there is no other look at the disk.
194
+ if self._auto is not None:
195
+ self._auto.consider(status)
196
+ return self._with_indexing(status)
197
+ # Narrowed in one read, not composed from two: coverage stays the root's - that is
198
+ # the tree every answer is drawn from - while the failures and stale documents
199
+ # named are the ones that live here. Resolved against the root this service was
200
+ # built for, never against the configured path again, or a retargeted symlink
201
+ # pairs this root's certificate with another tree's failures.
202
+ scope = self._within_root(
203
+ headings._absolute(Path(self._root) / directory.strip(), SearchError), SearchError
204
+ )
205
+ return self._with_indexing(
206
+ self._with_freshness(self._db.index_status(self._root, str(scope)), str(scope))
207
+ )
208
+
209
+ def _root_status(self) -> IndexStatus:
210
+ return self._with_freshness(self._db.index_status(self._root), self._root)
211
+
212
+ def _with_indexing(self, status: IndexStatus) -> IndexStatus:
213
+ active = self._auto is not None and self._auto.active
214
+ return dataclasses.replace(status, indexing=active) if active else status
215
+
216
+ def _with_freshness(self, status: IndexStatus, scope: str) -> IndexStatus:
217
+ """Add what only the filesystem knows: which indexed files have moved on.
218
+
219
+ The database's own status is one SQLite snapshot and deliberately says nothing
220
+ about the disk, so this is composed here rather than there. It asks only about
221
+ rows the index holds - a file nobody has indexed yet is found by walking the tree,
222
+ which is the expensive half of indexing and not something a search should pay for.
223
+ """
224
+ return dataclasses.replace(status, changed_files=self._freshness.changed_files(scope))
225
+
226
+ # ------------------------------------------------------------------ resolution
227
+
228
+ def _resolve_directory(self, directory: str | None) -> Path:
229
+ """Resolve against the root this server settled on, never the configured spelling.
230
+
231
+ The docs root is resolved once at construction precisely so a retargeted symlink
232
+ cannot make the server answer from one tree while reporting on another. Resolving
233
+ the configured path again here reopened that door from the other side: indexing
234
+ followed the link to its new target and wrote rows the frozen root can never see,
235
+ `list_documents()` with no argument then resolved outside its own root and raised,
236
+ and a restart keyed a different database and read as never indexed.
237
+ """
238
+ if directory is None or not directory.strip():
239
+ return Path(self._root)
240
+ path = headings._user_path(directory.strip(), IndexingError)
241
+ if not path.is_absolute():
242
+ path = Path(self._root) / path
243
+ return headings._absolute(path, IndexingError)
244
+
245
+ def _within_root(self, resolved: Path, error: type[MarkdownMemoryError]) -> Path:
246
+ """Refuse to *answer about* a directory outside the tree this server serves.
247
+
248
+ `..`, an absolute path and a symlink each reach out of the root, and each was
249
+ obeyed: `list_documents` handed back another project's file paths, and a status
250
+ lookup paired this root's certificate with that tree's failures - `coverage:
251
+ verified` beside a non-empty failure list, which the envelope promises cannot
252
+ happen. Search has been scoped to the root ever since it answered one project's
253
+ question from another's documentation; these are the two other ways in.
254
+
255
+ `index_directory` is deliberately not scoped this way. It is an instruction rather
256
+ than a question - go and index that tree - and it keys the tree it walked under its
257
+ own root, so nothing it writes is attributed here.
258
+ """
259
+ root = Path(self._root)
260
+ if resolved != root and root not in resolved.parents:
261
+ raise error(
262
+ f"{str(resolved)!r} is outside this server's documentation root "
263
+ f"({self._root}); name a directory inside it."
264
+ )
265
+ return resolved
266
+
267
+ def _resolve_document(self, file_path: str) -> Document:
268
+ """Find an indexed document by absolute path, relative path, or unique path suffix."""
269
+ requested = file_path.strip()
270
+ if not requested:
271
+ raise DocumentNotFoundError("file_path must not be empty")
272
+ path = headings._user_path(requested, DocumentNotFoundError)
273
+ candidates = [path]
274
+ if not path.is_absolute():
275
+ candidates = [self._config.docs_dir / path]
276
+ try:
277
+ candidates.append(Path.cwd() / path)
278
+ except OSError: # the working directory was deleted under the server
279
+ logger.debug("Working directory is gone; not resolving %s against it", requested)
280
+ for candidate in candidates:
281
+ resolved = headings._absolute(candidate, DocumentNotFoundError)
282
+ document = self._db.get_document(str(resolved))
283
+ if document is not None:
284
+ return document
285
+ matches = self._db.find_documents_by_suffix(requested)
286
+ if len(matches) == 1:
287
+ return matches[0]
288
+ if matches:
289
+ listing = ", ".join(match.file_path for match in matches[: headings._MAX_LISTED_PATHS])
290
+ raise DocumentNotFoundError(f"'{requested}' is ambiguous; it matches: {listing}")
291
+ raise DocumentNotFoundError(
292
+ f"'{requested}' is not indexed. Run index_directory, then list_documents "
293
+ "to see the available paths."
294
+ )
295
+
296
+
297
+ # ---------------------------------------------------------------------- MCP wiring
298
+
299
+
300
+ def anticipated_errors(function: Callable[_P, _R]) -> Callable[_P, _R]:
301
+ """Report domain failures to the model verbatim.
302
+
303
+ The SDK hides the message of any exception other than ``ToolError`` behind a generic
304
+ "Error executing tool" (treating it as a crash). Domain errors carry actionable text
305
+ - available heading paths, "run index_directory first" - that the agent needs to see.
306
+ """
307
+
308
+ @functools.wraps(function)
309
+ def wrapper(*args: _P.args, **kwargs: _P.kwargs) -> _R:
310
+ try:
311
+ return function(*args, **kwargs)
312
+ except MarkdownMemoryError as exc:
313
+ raise ToolError(str(exc)) from exc
314
+
315
+ return wrapper
316
+
317
+
318
+ class _ServiceProvider:
319
+ """Hands the tools their service and owns its lifetime when nobody else does.
320
+
321
+ A service passed in by the caller is never closed here. One the server creates itself
322
+ lives as long as any session is open, and is created again on demand afterwards, so
323
+ the same server object can serve several sessions (or concurrent ones) in a row.
324
+ """
325
+
326
+ def __init__(self, config: ServerConfig | None, service: MarkdownMemoryService | None) -> None:
327
+ self._config = config
328
+ self._external = service
329
+ self._owned: MarkdownMemoryService | None = None
330
+ self._sessions = 0
331
+ self._lock = threading.Lock()
332
+
333
+ def get(self) -> MarkdownMemoryService:
334
+ if self._external is not None:
335
+ return self._external
336
+ with self._lock:
337
+ if self._owned is None:
338
+ self._owned = MarkdownMemoryService(self._config or ServerConfig.from_env())
339
+ return self._owned
340
+
341
+ def session_started(self) -> None:
342
+ with self._lock:
343
+ self._sessions += 1
344
+
345
+ def session_ended(self) -> None:
346
+ with self._lock:
347
+ self._sessions -= 1
348
+ if self._sessions > 0 or self._owned is None:
349
+ return
350
+ owned, self._owned = self._owned, None
351
+ owned.close()
352
+
353
+
354
+ def create_server(
355
+ config: ServerConfig | None = None, *, service: MarkdownMemoryService | None = None
356
+ ) -> MCPServer[None]:
357
+ """Build the MCP server and register its tools."""
358
+ services = _ServiceProvider(config, service)
359
+
360
+ @asynccontextmanager
361
+ async def lifespan(_server: MCPServer[None]) -> AsyncIterator[None]:
362
+ services.session_started()
363
+ try:
364
+ yield
365
+ finally:
366
+ services.session_ended()
367
+
368
+ server: MCPServer[None] = MCPServer(
369
+ "markdown-memory", version=__version__, instructions=SERVER_INSTRUCTIONS, lifespan=lifespan
370
+ )
371
+
372
+ @server.tool()
373
+ @anticipated_errors
374
+ def index_directory(directory: str | None = None) -> str:
375
+ """Scan a directory tree for Markdown files and (re)index new or changed ones.
376
+
377
+ Unchanged files are skipped via SHA-256 hashes; files deleted from disk are purged.
378
+ Omit `directory` to index the server's configured documentation root.
379
+ """
380
+ return services.get().index_directory(directory).summary()
381
+
382
+ @server.tool()
383
+ @anticipated_errors
384
+ def list_documents(directory: str = "") -> JsonDict:
385
+ """List indexed documents (path, title, section count), optionally under `directory`.
386
+
387
+ Returns `{"documents": [...], "index_status": {...}}`. `index_status.changed_files`
388
+ counts indexed documents that no longer match the index; it is independent of
389
+ coverage, so read `index_status.message` whenever either is set.
390
+ `index_status.coverage` is
391
+ "verified" only when a full index run of this documentation root finished and read
392
+ every file it found; otherwise it is "unknown" and `index_status.message` says why.
393
+ """
394
+ service = services.get()
395
+ scope = directory if directory.strip() else None
396
+ return {
397
+ "documents": [summary.to_dict() for summary in service.list_documents(directory)],
398
+ "index_status": service.index_status(scope).to_dict(),
399
+ }
400
+
401
+ @server.tool()
402
+ @anticipated_errors
403
+ def get_document_outline(file_path: str) -> list[JsonDict]:
404
+ """Hierarchical table of contents of one document: heading paths, line ranges and
405
+ token estimates. Costs a few hundred tokens; use it to pick a section to read."""
406
+ return [node.to_dict() for node in services.get().get_document_outline(file_path)]
407
+
408
+ @server.tool()
409
+ @anticipated_errors
410
+ def read_section(file_path: str, heading_path: str, include_subsections: bool = False) -> str:
411
+ """Return the verbatim Markdown of one section, addressed by its breadcrumb
412
+ (`Root > Child > Subchild`, as shown by get_document_outline). By default only the
413
+ section's own text is returned; set `include_subsections` to append its children."""
414
+ return services.get().read_section(
415
+ file_path, heading_path, include_subsections=include_subsections
416
+ )
417
+
418
+ @server.tool()
419
+ @anticipated_errors
420
+ def search_docs(query: str, limit: int = 5) -> JsonDict:
421
+ """Hybrid search (BM25 keywords + semantic vectors, fused with RRF) over all indexed
422
+ sections. Works for exact identifiers (flags, env vars) and for conceptual questions.
423
+
424
+ Returns `{"results": [...], "index_status": {...}}`, `results` holding at most
425
+ `limit` sections. `index_status.changed_files` counts indexed documents that no
426
+ longer match the index - a hit may quote text that is no longer there - and is
427
+ independent of coverage: it can be non-zero while coverage reads "verified", so
428
+ read `index_status.message` whenever either is set.
429
+ When `index_status.coverage` is "unknown", what you searched is
430
+ missing part of its documentation, or was never indexed end to end: an answer drawn
431
+ from it may be confidently incomplete, and `index_status.message` says what to run.
432
+ """
433
+ service = services.get()
434
+ return {
435
+ "results": [result.to_dict() for result in service.search_docs(query, limit)],
436
+ "index_status": service.index_status().to_dict(),
437
+ }
438
+
439
+ return server
440
+
441
+
442
+ def configure_logging(level: str | None = None) -> None:
443
+ """Route all logging to stderr; stdout is reserved for JSON-RPC frames."""
444
+ name = (level or os.environ.get(ENV_LOG_LEVEL, "") or "INFO").upper()
445
+ logging.basicConfig(
446
+ level=getattr(logging, name, logging.INFO),
447
+ stream=sys.stderr,
448
+ format="%(asctime)s %(levelname)s %(name)s: %(message)s",
449
+ force=True,
450
+ )
451
+
452
+
453
+ def _download_model(config: ServerConfig) -> None:
454
+ """Fetch and load the configured embedder, so no client waits on the first download."""
455
+ try:
456
+ create_embedder(config.embedder, cache_dir=config.model_cache_dir).warm_up()
457
+ except MarkdownMemoryError:
458
+ logger.exception("Cannot download the embedding model")
459
+ raise SystemExit(1) from None
460
+ logger.info("Embedding model %s is ready in %s", config.embedder, config.model_cache_dir)
461
+
462
+
463
+ def main(argv: Sequence[str] | None = None) -> None:
464
+ """Console entrypoint: serve MCP over stdio."""
465
+ parser = argparse.ArgumentParser(
466
+ prog="markdown-memory", description="Markdown documentation memory MCP server (stdio)."
467
+ )
468
+ parser.add_argument("--db", type=Path, help=f"SQLite database path (env {ENV_DB_PATH})")
469
+ parser.add_argument("--docs-dir", type=Path, help=f"Default docs root (env {ENV_DOCS_DIR})")
470
+ parser.add_argument("--log-level", help=f"Logging level (env {ENV_LOG_LEVEL})")
471
+ parser.add_argument(
472
+ "--exclude",
473
+ action="append",
474
+ default=[],
475
+ metavar="GLOB",
476
+ help=f"Skip paths matching this glob, relative to the docs root; repeatable "
477
+ f"(env {ENV_EXCLUDE}, comma separated)",
478
+ )
479
+ parser.add_argument(
480
+ "--no-auto-index",
481
+ action="store_true",
482
+ help=f"Do not keep the docs root indexed in the background (env {ENV_AUTO_INDEX}=0)",
483
+ )
484
+ parser.add_argument(
485
+ "--embedder",
486
+ choices=("embeddinggemma", "bge-small"),
487
+ help=f"Embedding model preset (env {ENV_EMBEDDER}; default {DEFAULT_EMBEDDER})",
488
+ )
489
+ parser.add_argument(
490
+ "--download-model",
491
+ action="store_true",
492
+ help="Download and load the configured embedding model, then exit",
493
+ )
494
+ parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
495
+ arguments = parser.parse_args(argv)
496
+
497
+ configure_logging(arguments.log_level)
498
+ config = _config_from_cli(arguments)
499
+ if arguments.download_model:
500
+ _download_model(config)
501
+ return
502
+ try:
503
+ service = MarkdownMemoryService(config)
504
+ except MarkdownMemoryError:
505
+ logger.exception("Cannot start markdown-memory")
506
+ raise SystemExit(1) from None
507
+ logger.info("markdown-memory serving; db=%s docs_dir=%s", config.db_path, config.docs_dir)
508
+ # Nothing heavy starts here. Loading the model holds the GIL for seconds, and a client
509
+ # waits on the handshake with a short timeout - Codex's is 10 s. The first search loads
510
+ # it instead, and starts the catch-up index run on its way out.
511
+ if config.auto_index:
512
+ service.start_auto_index(request=False)
513
+ try:
514
+ create_server(config, service=service).run("stdio")
515
+ finally:
516
+ service.close()
517
+
518
+
519
+ if __name__ == "__main__":
520
+ main()