markdown-memory 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,47 @@
1
+ """markdown-memory: AST-aware Markdown indexing and hybrid retrieval over MCP."""
2
+
3
+ from importlib.metadata import PackageNotFoundError, version
4
+
5
+ from markdown_memory.exceptions import (
6
+ ASTParseError,
7
+ DatabaseError,
8
+ DocumentNotFoundError,
9
+ EmbeddingError,
10
+ IndexingError,
11
+ MarkdownMemoryError,
12
+ SectionNotFoundError,
13
+ )
14
+ from markdown_memory.models import (
15
+ Document,
16
+ DocumentSummary,
17
+ IndexReport,
18
+ OutlineNode,
19
+ SearchResult,
20
+ Section,
21
+ SectionDraft,
22
+ )
23
+
24
+ try:
25
+ # pyproject.toml is the one place the version is written; this reads it back from the
26
+ # installed metadata rather than repeating it.
27
+ __version__ = version("markdown-memory")
28
+ except PackageNotFoundError: # a bare source tree nobody installed
29
+ __version__ = "0+unknown"
30
+
31
+ __all__ = [
32
+ "ASTParseError",
33
+ "DatabaseError",
34
+ "Document",
35
+ "DocumentNotFoundError",
36
+ "DocumentSummary",
37
+ "EmbeddingError",
38
+ "IndexReport",
39
+ "IndexingError",
40
+ "MarkdownMemoryError",
41
+ "OutlineNode",
42
+ "SearchResult",
43
+ "Section",
44
+ "SectionDraft",
45
+ "SectionNotFoundError",
46
+ "__version__",
47
+ ]
@@ -0,0 +1,170 @@
1
+ """Keeping the server's own documentation root indexed without being asked.
2
+
3
+ Nothing watches the filesystem. The server already looks at it on every search - the
4
+ freshness sweep counts indexed files that moved on - so that look is what decides when to
5
+ re-index. The first search after a start finds the runner has never run, and that
6
+ catch-up run picks up whatever changed while no server was running. A run is the ordinary
7
+ incremental `index_directory`, in one background thread at a time.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import logging
13
+ import threading
14
+ import time
15
+ from collections.abc import Callable
16
+
17
+ from markdown_memory.exceptions import IndexBusyError, IndexCancelled
18
+ from markdown_memory.models import IndexReport, IndexStatus
19
+
20
+ logger = logging.getLogger(__name__)
21
+
22
+ #: How soon after a run finishes an edit found by a search may start the next one. Short
23
+ #: enough that an agent's edit is searchable within a turn or two; long enough that a
24
+ #: burst of saves is one run rather than one each.
25
+ CHANGE_GAP_SECONDS = 10.0
26
+ #: How long a run speaks for when nothing edited was seen. Only a walk finds a file
27
+ #: nobody indexed yet, or an edit that put its modification time back.
28
+ WALK_GAP_SECONDS = 300.0
29
+
30
+
31
+ class AutoIndexer:
32
+ """One background index run at a time, started by what the server sees on use.
33
+
34
+ Every field below is read and written under one lock, so a request can never slip
35
+ between a run finishing and the runner forgetting it, and nothing starts once `stop`
36
+ has been called.
37
+ """
38
+
39
+ def __init__(
40
+ self,
41
+ run: Callable[[Callable[[], bool]], IndexReport],
42
+ measure: Callable[[], IndexStatus],
43
+ *,
44
+ clock: Callable[[], float] = time.monotonic,
45
+ change_gap: float = CHANGE_GAP_SECONDS,
46
+ walk_gap: float = WALK_GAP_SECONDS,
47
+ ) -> None:
48
+ self._run = run
49
+ self._measure = measure
50
+ self._clock = clock
51
+ self._change_gap = change_gap
52
+ self._walk_gap = walk_gap
53
+ self._lock = threading.Lock()
54
+ self._thread: threading.Thread | None = None
55
+ self._stopping = False
56
+ self._last_finished: float | None = None
57
+ #: Nothing starts before this: set when another process held the lock, so that
58
+ #: contention costs one attempt per change gap, not one per search.
59
+ self._retry_after = float("-inf")
60
+ #: Changed files the last finished run could not clear - files it failed on. A
61
+ #: search that still sees exactly those is not a reason to walk again; one more
62
+ #: is. Reset to zero by a run that failed nothing, so an edit made while it ran
63
+ #: is picked up by the next search.
64
+ self._baseline = 0
65
+ #: The weights mismatch as the last run left it: a new one is repaired at once,
66
+ #: one this runner could not repair does not start a run per search.
67
+ self._seen_mismatch: str | None = None
68
+
69
+ @property
70
+ def active(self) -> bool:
71
+ with self._lock:
72
+ return self._thread is not None
73
+
74
+ def request(self) -> bool:
75
+ """Start a run now unless one is running or the runner is stopping."""
76
+ with self._lock:
77
+ return self._start()
78
+
79
+ def consider(self, status: IndexStatus) -> bool:
80
+ """Start a run if what a search just measured says one is due."""
81
+ with self._lock:
82
+ if self._stopping or self._thread is not None or self._clock() < self._retry_after:
83
+ return False
84
+ since = (
85
+ float("inf") if self._last_finished is None else self._clock() - self._last_finished
86
+ )
87
+ due = (
88
+ (status.changed_files > self._baseline and since >= self._change_gap)
89
+ or (
90
+ status.weights_mismatch is not None
91
+ and status.weights_mismatch != self._seen_mismatch
92
+ )
93
+ or since >= self._walk_gap
94
+ )
95
+ return self._start() if due else False
96
+
97
+ def stop(self) -> None:
98
+ """Stop the running run between two documents, and wait for it to let go."""
99
+ with self._lock:
100
+ self._stopping = True
101
+ thread = self._thread
102
+ if thread is not None:
103
+ thread.join()
104
+
105
+ def _start(self) -> bool:
106
+ if self._stopping or self._thread is not None:
107
+ return False
108
+ # Not a daemon: the interpreter would kill it mid-write at exit. `stop` is what
109
+ # ends it, and the service calls that before it closes the database.
110
+ self._thread = threading.Thread(target=self._work, name="mdmem-autoindex")
111
+ self._thread.start()
112
+ return True
113
+
114
+ def _is_stopping(self) -> bool:
115
+ with self._lock:
116
+ return self._stopping
117
+
118
+ def _work(self) -> None:
119
+ baseline: int | None = None
120
+ busy = False
121
+ before = self._mismatch()
122
+ try:
123
+ report = self._run(self._is_stopping)
124
+ baseline = self._changed() if report.errors else 0
125
+ except IndexCancelled:
126
+ logger.info("Automatic index run stopped")
127
+ except IndexBusyError as exc:
128
+ # Someone else holds the lock - maybe on another root of a shared database -
129
+ # so nothing is learnt about this tree; the next search asks again.
130
+ logger.info("Automatic index run skipped: %s", exc)
131
+ busy = True
132
+ except Exception:
133
+ # The whole run failed - a model that will not load, say. Retrying it for the
134
+ # same edit every few seconds would load the model every few seconds: what
135
+ # it left behind becomes the baseline, and only a further change is news.
136
+ logger.exception("Automatic index run failed")
137
+ baseline = self._changed()
138
+ after = self._mismatch()
139
+ with self._lock:
140
+ self._last_finished = self._clock()
141
+ if busy:
142
+ # Nothing ran, so nothing is known: ask again once the change gap has
143
+ # passed rather than a whole walk interval later - a root nobody has
144
+ # indexed has no changed files to prompt the next attempt.
145
+ self._last_finished -= self._walk_gap - self._change_gap
146
+ self._retry_after = self._last_finished + self._walk_gap
147
+ self._thread = None
148
+ return
149
+ if baseline is not None:
150
+ self._baseline = baseline
151
+ # Seen only if this run met it and could not clear it. One a search recorded
152
+ # while the run was busy is news the run never acted on.
153
+ if after is None or after == before:
154
+ self._seen_mismatch = after
155
+ self._thread = None
156
+
157
+ def _changed(self) -> int | None:
158
+ try:
159
+ return self._measure().changed_files
160
+ except Exception:
161
+ logger.exception("Cannot read the index status around an automatic run")
162
+ return None
163
+
164
+ def _mismatch(self) -> str | None:
165
+ try:
166
+ return self._measure().weights_mismatch
167
+ except Exception:
168
+ logger.exception("Cannot read the index status around an automatic run")
169
+ with self._lock:
170
+ return self._seen_mismatch
@@ -0,0 +1,235 @@
1
+ """Where the server reads its settings from, and in what order.
2
+
3
+ One precedence for every entry point: an explicit argument beats the environment, which
4
+ beats the project default. ``resolve_config`` is the only place that order lives.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import argparse
10
+ import hashlib
11
+ import os
12
+ import re
13
+ from collections.abc import Sequence
14
+ from dataclasses import dataclass
15
+ from pathlib import Path
16
+
17
+ from markdown_memory.embedders import DEFAULT_EMBEDDER
18
+ from markdown_memory.exceptions import ConfigurationError
19
+ from markdown_memory.indexer import DEFAULT_INDEX_WORKERS
20
+
21
+ ENV_DB_PATH = "MARKDOWN_MEMORY_DB"
22
+
23
+
24
+ ENV_DOCS_DIR = "MARKDOWN_MEMORY_DOCS_DIR"
25
+
26
+
27
+ ENV_MODEL_CACHE = "MARKDOWN_MEMORY_MODEL_CACHE"
28
+
29
+
30
+ ENV_EMBEDDER = "MARKDOWN_MEMORY_EMBEDDER"
31
+
32
+
33
+ ENV_LOG_LEVEL = "MARKDOWN_MEMORY_LOG_LEVEL"
34
+
35
+
36
+ ENV_EXCLUDE = "MARKDOWN_MEMORY_EXCLUDE"
37
+
38
+
39
+ ENV_INDEX_WORKERS = "MARKDOWN_MEMORY_INDEX_WORKERS"
40
+
41
+
42
+ ENV_AUTO_INDEX = "MARKDOWN_MEMORY_AUTO_INDEX"
43
+
44
+
45
+ # Claude Code exports this to every stdio MCP server it spawns, set to the project root.
46
+ # `.mcp.json` cannot interpolate it - measured on Claude Code 2.1.278, `${CLAUDE_PROJECT_DIR}`
47
+ # and `${workspaceFolder}` are both reported as "Missing environment variables" and passed
48
+ # through literally - so a project-scoped config uses relative paths and the server resolves
49
+ # them here instead.
50
+ ENV_PROJECT_DIR = "CLAUDE_PROJECT_DIR"
51
+
52
+
53
+ def _xdg_dir(variable: str, fallback: str) -> Path:
54
+ configured = os.environ.get(variable, "").strip()
55
+ return Path(configured) if configured else Path.home() / fallback
56
+
57
+
58
+ def _project_database(docs_dir: Path) -> Path:
59
+ """Where one documentation root's index lives when nothing configured it.
60
+
61
+ Keyed on the documentation root, never on the working directory. The working directory
62
+ belongs to whoever launched the server, so two servers started from one directory for
63
+ two different projects would share a database - which is the cross-project leak this
64
+ default exists to close, arrived at from the other side. Search is scoped to the docs
65
+ root, but a document stays resolvable across a whole database by path or unique suffix,
66
+ so sharing the file is enough to leak one project's documentation into another's answers.
67
+
68
+ Kept out of the project too. A database inside the repository is committed by accident,
69
+ deleted by `git clean -xdf`, rebuilt per worktree, unwritable when the checkout is
70
+ read-only, and - on a network share - sits where SQLite's WAL cannot take the locks it
71
+ needs. The name carries the root's own basename so a person can tell the indexes apart,
72
+ and a digest of its resolved path so two projects called `docs` cannot collide.
73
+ """
74
+ try:
75
+ resolved = docs_dir.expanduser().resolve()
76
+ except (OSError, RuntimeError): # symlink loop, or a path the OS will not resolve
77
+ resolved = docs_dir.expanduser().absolute()
78
+ digest = hashlib.sha256(os.fsencode(str(resolved))).hexdigest()[:12]
79
+ label = re.sub(r"[^A-Za-z0-9_.-]", "-", resolved.name) or "root"
80
+ return (
81
+ _xdg_dir("XDG_DATA_HOME", ".local/share")
82
+ / "markdown-memory"
83
+ / "projects"
84
+ / f"{label}-{digest}"
85
+ / "index.db"
86
+ )
87
+
88
+
89
+ @dataclass(slots=True, frozen=True)
90
+ class ServerConfig:
91
+ """Runtime configuration, resolved from the environment (CLI flags override)."""
92
+
93
+ db_path: Path
94
+ docs_dir: Path
95
+ embedder: str = DEFAULT_EMBEDDER
96
+ model_cache_dir: Path | None = None
97
+ # Glob patterns, relative to the docs root, that indexing must not descend into: a
98
+ # repository's own fixtures, vendored documentation or test corpus are not its docs.
99
+ exclude: tuple[str, ...] = ()
100
+ #: Files embedded at the same time while indexing.
101
+ index_workers: int = DEFAULT_INDEX_WORKERS
102
+ #: Whether the stdio server keeps its own docs root indexed in the background.
103
+ auto_index: bool = True
104
+
105
+ @classmethod
106
+ def from_env(cls) -> ServerConfig:
107
+ root = _project_root()
108
+ db_path = _configured_path(ENV_DB_PATH, root)
109
+ docs_dir = _configured_path(ENV_DOCS_DIR, root)
110
+ model_cache = _configured_path(ENV_MODEL_CACHE, root)
111
+ return cls(
112
+ # One index per documentation root, rather than one for the whole machine.
113
+ # Isolation should not depend on the user having set an environment variable.
114
+ # The model cache below stays shared on purpose: 218 MB of read-only weights,
115
+ # identical everywhere, and copying it per project would be pure waste.
116
+ db_path=(db_path if db_path else _project_database(docs_dir if docs_dir else root)),
117
+ docs_dir=docs_dir if docs_dir else root,
118
+ embedder=os.environ.get(ENV_EMBEDDER, "").strip() or DEFAULT_EMBEDDER,
119
+ model_cache_dir=(
120
+ model_cache
121
+ if model_cache
122
+ else _xdg_dir("XDG_CACHE_HOME", ".cache") / "markdown-memory" / "models"
123
+ ),
124
+ exclude=parse_exclusions(os.environ.get(ENV_EXCLUDE, "")),
125
+ index_workers=_positive_int(ENV_INDEX_WORKERS, DEFAULT_INDEX_WORKERS),
126
+ auto_index=os.environ.get(ENV_AUTO_INDEX, "").strip().lower()
127
+ not in {"0", "false", "off", "no"},
128
+ )
129
+
130
+
131
+ def resolve_config(
132
+ *,
133
+ db: Path | None = None,
134
+ docs_dir: Path | None = None,
135
+ embedder: str | None = None,
136
+ exclude: Sequence[str] = (),
137
+ auto_index: bool | None = None,
138
+ ) -> ServerConfig:
139
+ """Environment configuration with explicit overrides laid over it.
140
+
141
+ Naming a different documentation root re-keys the database, because the default is
142
+ keyed on that root: taking `ServerConfig.from_env().db_path` as the fallback reads a
143
+ path derived from the *environment's* root, and two callers pointed at different roots
144
+ from one directory would land in the launcher's single database - exactly the
145
+ cross-project leak keying was added to close. An explicitly configured database still
146
+ wins, from the argument or the environment, in that order.
147
+
148
+ Every entry point that takes overrides resolves them here - the server's own flags
149
+ and the scripts alike - so that precedence is written once and cannot drift between
150
+ them. The paths that accept none (the in-process service, `eval_retrieval.py`) go
151
+ straight to `ServerConfig.from_env`, which is the same answer with nothing laid over
152
+ it.
153
+ """
154
+ base = ServerConfig.from_env()
155
+ root = docs_dir.expanduser() if docs_dir else base.docs_dir
156
+ configured_db = _configured_path(ENV_DB_PATH, _project_root())
157
+ return ServerConfig(
158
+ db_path=(
159
+ db.expanduser() if db else configured_db if configured_db else _project_database(root)
160
+ ),
161
+ docs_dir=root,
162
+ embedder=embedder or base.embedder,
163
+ model_cache_dir=base.model_cache_dir,
164
+ exclude=tuple(exclude) or base.exclude,
165
+ index_workers=base.index_workers,
166
+ auto_index=base.auto_index if auto_index is None else auto_index,
167
+ )
168
+
169
+
170
+ def _config_from_cli(arguments: argparse.Namespace) -> ServerConfig:
171
+ """The command line laid over the environment."""
172
+ return resolve_config(
173
+ db=arguments.db,
174
+ docs_dir=arguments.docs_dir,
175
+ embedder=arguments.embedder,
176
+ exclude=arguments.exclude,
177
+ auto_index=False if getattr(arguments, "no_auto_index", False) else None,
178
+ )
179
+
180
+
181
+ def _project_root() -> Path:
182
+ """The directory a relative configured path is relative to.
183
+
184
+ Claude Code sets the working directory of a project-scoped server to the project root
185
+ as well, so the fallback agrees with the export in that case; it differs only for a
186
+ server started by hand from somewhere else.
187
+ """
188
+ exported = os.environ.get(ENV_PROJECT_DIR, "").strip()
189
+ return Path(exported).expanduser() if exported else Path.cwd()
190
+
191
+
192
+ def _positive_int(variable: str, default: int) -> int:
193
+ """A count read from the environment; anything that is not one keeps the default."""
194
+ value = os.environ.get(variable, "").strip()
195
+ return int(value) if value.isdigit() and int(value) > 0 else default
196
+
197
+
198
+ def _configured_path(variable: str, root: Path) -> Path | None:
199
+ """One configured path, resolved against ``root`` when it is relative.
200
+
201
+ An unexpanded `${...}` is rejected rather than used as a directory name: Claude Code
202
+ loads a config whose variables it could not expand and passes the literal text through,
203
+ which would otherwise index a directory named `${workspaceFolder}` and report success
204
+ over zero files.
205
+ """
206
+ value = os.environ.get(variable, "").strip()
207
+ if not value:
208
+ return None
209
+ # A directory really named `docs/${version}` is allowed: if the literal path exists,
210
+ # it is a path, not a variable nobody expanded.
211
+ if "${" in value and not Path(value).expanduser().exists():
212
+ raise ConfigurationError(
213
+ f"{variable} is set to {value!r}, which still contains an unexpanded variable. "
214
+ "Claude Code expands only environment variables in .mcp.json - not "
215
+ "${workspaceFolder} or ${CLAUDE_PROJECT_DIR} - so write the path relative to the "
216
+ "project root instead (for example '.markdown-memory/index.db')."
217
+ )
218
+ path = Path(value).expanduser()
219
+ return path if path.is_absolute() else (root / path)
220
+
221
+
222
+ def parse_exclusions(value: str) -> tuple[str, ...]:
223
+ """Split a configured exclusion list on commas; blanks and stray ``./`` dropped.
224
+
225
+ Comma only: a colon separator would split a pattern that contains one, and silently
226
+ excluding the wrong thing is worse than not accepting the separator.
227
+ """
228
+ patterns = []
229
+ for part in value.split(","):
230
+ # One leading "./" only: `lstrip("./")` would eat the dot of `.hidden` and
231
+ # exclude a `hidden` directory instead of the one that was named.
232
+ cleaned = part.strip().removeprefix("./").rstrip("/")
233
+ if cleaned:
234
+ patterns.append(cleaned)
235
+ return tuple(patterns)