markdown-memory 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- markdown_memory/__init__.py +47 -0
- markdown_memory/autoindex.py +170 -0
- markdown_memory/config.py +235 -0
- markdown_memory/db.py +1546 -0
- markdown_memory/discovery.py +195 -0
- markdown_memory/embedders.py +513 -0
- markdown_memory/exceptions.py +71 -0
- markdown_memory/freshness.py +138 -0
- markdown_memory/headings.py +185 -0
- markdown_memory/indexer.py +725 -0
- markdown_memory/model_cache.py +272 -0
- markdown_memory/models.py +339 -0
- markdown_memory/parser.py +869 -0
- markdown_memory/py.typed +0 -0
- markdown_memory/search.py +518 -0
- markdown_memory/server.py +520 -0
- markdown_memory-0.1.0.dist-info/METADATA +579 -0
- markdown_memory-0.1.0.dist-info/RECORD +21 -0
- markdown_memory-0.1.0.dist-info/WHEEL +4 -0
- markdown_memory-0.1.0.dist-info/entry_points.txt +3 -0
- markdown_memory-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
"""Whether the indexed documents are still what is on disk.
|
|
2
|
+
|
|
3
|
+
The database's own status is one SQLite snapshot and says nothing about the filesystem,
|
|
4
|
+
so this is composed beside it rather than inside it. ``FreshnessSweep`` owns the whole
|
|
5
|
+
of that state: the single-entry cache, its TTL, and the lock that makes reading the
|
|
6
|
+
cache, walking the tree and publishing the answer one step.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import stat
|
|
12
|
+
import threading
|
|
13
|
+
import time
|
|
14
|
+
from enum import Enum, auto
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
from markdown_memory.db import Database
|
|
18
|
+
from markdown_memory.discovery import MAX_FILE_BYTES, hash_bytes, read_regular_file
|
|
19
|
+
|
|
20
|
+
# How long one filesystem sweep speaks for. An agent fires several searches in a single
|
|
21
|
+
# turn, and every one of them asks for the status: on a local ext4 tree 100 stats cost
|
|
22
|
+
# ~0.1 ms, but across a WSL2 or network boundary they cost 100-300 ms, which would double
|
|
23
|
+
# the latency of a query to re-answer a question whose answer cannot have changed much.
|
|
24
|
+
# Short enough that an edit is reported by the next search but one.
|
|
25
|
+
_FRESHNESS_TTL_SECONDS = 3.0
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class _Verdict(Enum):
|
|
29
|
+
"""What one file's freshness probe found."""
|
|
30
|
+
|
|
31
|
+
UNCHANGED = auto()
|
|
32
|
+
CHANGED = auto()
|
|
33
|
+
#: Same bytes, a time that has moved: nothing to report, but worth writing down.
|
|
34
|
+
SAME_BYTES_NEW_TIME = auto()
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _compare(path: Path, content_hash: str, mtime_ns: int | None) -> tuple[_Verdict, int]:
|
|
38
|
+
"""Whether the file behind an indexed document differs from what was indexed.
|
|
39
|
+
|
|
40
|
+
The modification time is the cheap question and the bytes are the expensive one, so
|
|
41
|
+
the hash is only computed where the time has moved: a `touch`, a checkout that
|
|
42
|
+
rewrites a file with its own contents, or a copy that preserves nothing but the text
|
|
43
|
+
must not be reported as a change an agent should act on. A file that has vanished, no
|
|
44
|
+
longer resolves to a regular file, or cannot be read counts as changed - not because
|
|
45
|
+
its bytes are known to differ, but because they cannot be checked at all. That applies
|
|
46
|
+
where the bytes had to be read: a file whose recorded time still matches is answered
|
|
47
|
+
from the time alone, so losing permission to read it - without touching it - is not
|
|
48
|
+
reported here. The next index run cannot read it either, and records a failure, which
|
|
49
|
+
is what takes `coverage` to `"unknown"`.
|
|
50
|
+
|
|
51
|
+
A stored `None` means no modification time was recorded - a row written before the
|
|
52
|
+
column existed - rather than a time of zero, so no real timestamp can be mistaken for
|
|
53
|
+
it, the epoch included. Those files are answered by their bytes until an index run
|
|
54
|
+
writes a time for them.
|
|
55
|
+
|
|
56
|
+
The one edit this cannot see is a file rewritten with its modification time put back
|
|
57
|
+
to what it was: no timestamp moved, so no hash is taken. Indexing itself is not fooled
|
|
58
|
+
- it hashes every file it walks - so `index_directory` still rebuilds that document;
|
|
59
|
+
what is missed is only the hint that it is worth running. Seeing it here would mean
|
|
60
|
+
hashing every indexed file on every query, or storing a second timestamp to compare
|
|
61
|
+
against, and this signal is not worth either.
|
|
62
|
+
"""
|
|
63
|
+
try:
|
|
64
|
+
info = path.stat()
|
|
65
|
+
except OSError:
|
|
66
|
+
return _Verdict.CHANGED, 0
|
|
67
|
+
if not stat.S_ISREG(info.st_mode):
|
|
68
|
+
return _Verdict.CHANGED, 0
|
|
69
|
+
if mtime_ns is not None and info.st_mtime_ns == mtime_ns:
|
|
70
|
+
return _Verdict.UNCHANGED, info.st_mtime_ns
|
|
71
|
+
try:
|
|
72
|
+
data = read_regular_file(path)
|
|
73
|
+
except OSError:
|
|
74
|
+
return _Verdict.CHANGED, 0
|
|
75
|
+
if data is None or len(data) > MAX_FILE_BYTES or hash_bytes(data) != content_hash:
|
|
76
|
+
return _Verdict.CHANGED, 0
|
|
77
|
+
# The time from the stat that came *before* the read, never a fresher one: a file
|
|
78
|
+
# rewritten after these bytes were hashed must not be recorded as verified at the
|
|
79
|
+
# moment of its rewrite, or the next sweep would trust a time that belongs to content
|
|
80
|
+
# nobody checked.
|
|
81
|
+
return _Verdict.SAME_BYTES_NEW_TIME, info.st_mtime_ns
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
class FreshnessSweep:
|
|
85
|
+
"""How many indexed documents are no longer what was indexed, cheaply and often.
|
|
86
|
+
|
|
87
|
+
One owner for three things that only make sense together: the count, the moment it
|
|
88
|
+
was taken, and the lock that keeps a sweep whole. Before this they were three
|
|
89
|
+
attributes on the service, which made it possible to reset one and not the others.
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
def __init__(self, db: Database, ttl: float = _FRESHNESS_TTL_SECONDS) -> None:
|
|
93
|
+
self._db = db
|
|
94
|
+
self._ttl = ttl
|
|
95
|
+
#: One entry, not a map keyed on the caller's path: an agent fires several
|
|
96
|
+
#: searches per turn against the same scope, and a map would grow for the life of
|
|
97
|
+
#: the server, one entry per spelling anybody ever asked about.
|
|
98
|
+
self._cache: tuple[str, float, int] | None = None
|
|
99
|
+
#: Held for the whole of a sweep, so that reading the cache, walking the
|
|
100
|
+
#: filesystem and storing the answer are one step. Without it a sweep that
|
|
101
|
+
#: indexing overtook would publish a count of a tree that no longer exists - the
|
|
102
|
+
#: one moment an agent is most likely to ask - and two sweeps racing could leave
|
|
103
|
+
#: the older one's answer behind. It also means a second caller arriving mid-sweep
|
|
104
|
+
#: waits and is served the result rather than walking the tree again.
|
|
105
|
+
self._lock = threading.Lock()
|
|
106
|
+
|
|
107
|
+
def invalidate(self) -> None:
|
|
108
|
+
"""Forget the cached count: indexing has changed what the answer would be."""
|
|
109
|
+
with self._lock:
|
|
110
|
+
self._cache = None
|
|
111
|
+
|
|
112
|
+
def changed_files(self, scope: str) -> int:
|
|
113
|
+
with self._lock:
|
|
114
|
+
cached = self._cache
|
|
115
|
+
if (
|
|
116
|
+
cached is not None
|
|
117
|
+
and cached[0] == scope
|
|
118
|
+
and time.monotonic() - cached[1] < self._ttl
|
|
119
|
+
):
|
|
120
|
+
return cached[2]
|
|
121
|
+
fingerprints = self._db.document_fingerprints(scope)
|
|
122
|
+
changed = 0
|
|
123
|
+
for file_path, (content_hash, mtime_ns) in fingerprints.items():
|
|
124
|
+
verdict, seen_ns = _compare(Path(file_path), content_hash, mtime_ns)
|
|
125
|
+
if verdict is _Verdict.CHANGED:
|
|
126
|
+
changed += 1
|
|
127
|
+
elif verdict is _Verdict.SAME_BYTES_NEW_TIME:
|
|
128
|
+
# The hash was computed to answer this, and the answer was "unchanged".
|
|
129
|
+
# Writing the time down means the next sweep reads it instead of the
|
|
130
|
+
# file - otherwise one `touch` costs a full hash every window until an
|
|
131
|
+
# index run happens to come past. A file that moved again in between
|
|
132
|
+
# is simply hashed again next time; nothing is lost by missing it.
|
|
133
|
+
self._db.record_modification_time(file_path, content_hash, mtime_ns, seen_ns)
|
|
134
|
+
# Stamped when the answer was produced, not when the sweep began: the walk
|
|
135
|
+
# itself takes time on a slow mount, and a window that starts before the
|
|
136
|
+
# measurement is a window the measurement was never true for.
|
|
137
|
+
self._cache = (scope, time.monotonic(), changed)
|
|
138
|
+
return changed
|
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
"""Turning what an agent typed into the section it meant.
|
|
2
|
+
|
|
3
|
+
Heading paths arrive as breadcrumbs (``Root > Child``) and have to survive casing,
|
|
4
|
+
spacing and ambiguity; the outline is the same tree seen from above.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
from collections.abc import Callable, Sequence
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from markdown_memory.exceptions import (
|
|
15
|
+
MarkdownMemoryError,
|
|
16
|
+
SectionNotFoundError,
|
|
17
|
+
)
|
|
18
|
+
from markdown_memory.models import (
|
|
19
|
+
PATH_SEPARATOR,
|
|
20
|
+
OutlineNode,
|
|
21
|
+
Section,
|
|
22
|
+
estimate_tokens,
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
_PATH_SEPARATOR_PATTERN = re.compile(r"\s*>\s*")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
_MAX_LISTED_PATHS = 40
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
# ---------------------------------------------------------------------- pure helpers
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _user_path(text: str, error: type[MarkdownMemoryError]) -> Path:
|
|
35
|
+
"""``Path(text).expanduser()``; pathlib's RuntimeError/ValueError become ``error``."""
|
|
36
|
+
try:
|
|
37
|
+
text.encode("utf-8") # a lone surrogate cannot be bound as a SQLite parameter
|
|
38
|
+
return Path(text).expanduser()
|
|
39
|
+
except (RuntimeError, ValueError) as exc: # unknown ~user, embedded NUL, bad encoding
|
|
40
|
+
raise error(f"Invalid path {text!r}: {exc}") from exc
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _absolute(path: Path, error: type[MarkdownMemoryError]) -> Path:
|
|
44
|
+
try:
|
|
45
|
+
resolved = path.expanduser().resolve()
|
|
46
|
+
str(resolved).encode("utf-8")
|
|
47
|
+
except (OSError, RuntimeError, ValueError) as exc: # symlink loop, NUL, bad encoding
|
|
48
|
+
raise error(f"Invalid path {str(path)!r}: {exc}") from exc
|
|
49
|
+
return resolved
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def normalize_heading_path(heading_path: str) -> str:
|
|
53
|
+
"""Canonicalise breadcrumb spacing: ``A>B`` and ``A > B`` both become ``A > B``."""
|
|
54
|
+
return PATH_SEPARATOR.join(
|
|
55
|
+
segment.strip() for segment in _PATH_SEPARATOR_PATTERN.split(heading_path.strip())
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _casefolded(heading_path: str) -> str:
|
|
60
|
+
return normalize_heading_path(heading_path).casefold()
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
# Progressively looser ways to compare a requested path with a stored one. The same
|
|
64
|
+
# transformation is applied to BOTH sides, so a title containing '>' ("Step 1 -> Step 2",
|
|
65
|
+
# "Result<T, E>") still matches itself, and an exact-case request beats a case-folded one
|
|
66
|
+
# (sibling headings "Setup" and "SETUP" stay individually addressable).
|
|
67
|
+
_MATCH_KEYS: tuple[Callable[[str], str], ...] = (str, normalize_heading_path, _casefolded)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def resolve_heading_path(candidates: Sequence[str], requested: str) -> str:
|
|
71
|
+
"""The one stored path that ``requested`` designates.
|
|
72
|
+
|
|
73
|
+
Whole-path matches are tried before trailing fragments (``Child > Subchild`` or the
|
|
74
|
+
bare heading title); within each, stricter comparisons come first. The first tier
|
|
75
|
+
with any match decides: one match wins, several are reported as ambiguous.
|
|
76
|
+
"""
|
|
77
|
+
for as_suffix in (False, True):
|
|
78
|
+
for key in _MATCH_KEYS:
|
|
79
|
+
wanted = key(requested)
|
|
80
|
+
if as_suffix:
|
|
81
|
+
ending = PATH_SEPARATOR + wanted
|
|
82
|
+
matches = [path for path in candidates if key(path).endswith(ending)]
|
|
83
|
+
else:
|
|
84
|
+
matches = [path for path in candidates if key(path) == wanted]
|
|
85
|
+
if len(matches) == 1:
|
|
86
|
+
return matches[0]
|
|
87
|
+
if matches:
|
|
88
|
+
raise SectionNotFoundError(
|
|
89
|
+
f"'{requested}' is ambiguous; use one of (exact spelling): "
|
|
90
|
+
+ " | ".join(matches[:_MAX_LISTED_PATHS])
|
|
91
|
+
)
|
|
92
|
+
available = " | ".join(candidates[:_MAX_LISTED_PATHS]) or "(document has no sections)"
|
|
93
|
+
raise SectionNotFoundError(f"No section '{requested}'. Available heading paths: {available}")
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def select_sections(
|
|
97
|
+
sections: Sequence[Section], heading_path: str, *, include_subsections: bool = False
|
|
98
|
+
) -> list[Section]:
|
|
99
|
+
"""Sections addressed by ``heading_path``, in source order.
|
|
100
|
+
|
|
101
|
+
``heading_path`` may name a whole section (every part of it is returned) or a single
|
|
102
|
+
``(Part n)`` of an oversized one. ``include_subsections`` adds the section's
|
|
103
|
+
descendants; it has no meaning for a single part and is ignored there.
|
|
104
|
+
"""
|
|
105
|
+
requested = heading_path.strip()
|
|
106
|
+
if not requested:
|
|
107
|
+
raise SectionNotFoundError("heading_path must not be empty")
|
|
108
|
+
base_paths = list(dict.fromkeys(section.base_path for section in sections))
|
|
109
|
+
parts = {s.heading_path: s for s in sections if s.part_index > 0}
|
|
110
|
+
part_paths = [path for path in parts if path not in base_paths]
|
|
111
|
+
chosen = resolve_heading_path([*base_paths, *part_paths], requested)
|
|
112
|
+
if chosen not in base_paths:
|
|
113
|
+
return [parts[chosen]]
|
|
114
|
+
|
|
115
|
+
selected: list[Section] = []
|
|
116
|
+
level: int | None = None
|
|
117
|
+
for section in sections:
|
|
118
|
+
if section.base_path == chosen:
|
|
119
|
+
level = section.heading_level
|
|
120
|
+
selected.append(section)
|
|
121
|
+
elif level is not None:
|
|
122
|
+
# Descendants are the sections that follow until a heading at the same or a
|
|
123
|
+
# shallower level; walking the order is immune to '>' inside titles.
|
|
124
|
+
if not include_subsections or level < 1 or section.heading_level <= level:
|
|
125
|
+
break
|
|
126
|
+
selected.append(section)
|
|
127
|
+
return selected
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def build_outline(sections: Sequence[Section]) -> list[OutlineNode]:
|
|
131
|
+
"""Nest a document's sections into a hierarchical table of contents."""
|
|
132
|
+
|
|
133
|
+
@dataclass(slots=True)
|
|
134
|
+
class _Pending:
|
|
135
|
+
title: str
|
|
136
|
+
level: int
|
|
137
|
+
path: str
|
|
138
|
+
start_line: int
|
|
139
|
+
end_line: int
|
|
140
|
+
tokens: int
|
|
141
|
+
parts: int
|
|
142
|
+
children: list[_Pending]
|
|
143
|
+
|
|
144
|
+
roots: list[_Pending] = []
|
|
145
|
+
stack: list[_Pending] = []
|
|
146
|
+
by_path: dict[str, _Pending] = {}
|
|
147
|
+
for section in sections:
|
|
148
|
+
existing = by_path.get(section.base_path)
|
|
149
|
+
if existing is not None: # a further part of an oversized section
|
|
150
|
+
existing.end_line = max(existing.end_line, section.end_line)
|
|
151
|
+
existing.tokens += estimate_tokens(section.content)
|
|
152
|
+
existing.parts += 1
|
|
153
|
+
continue
|
|
154
|
+
node = _Pending(
|
|
155
|
+
title=section.heading_title,
|
|
156
|
+
level=section.heading_level,
|
|
157
|
+
path=section.base_path,
|
|
158
|
+
start_line=section.start_line,
|
|
159
|
+
end_line=section.end_line,
|
|
160
|
+
tokens=estimate_tokens(section.content),
|
|
161
|
+
parts=1,
|
|
162
|
+
children=[],
|
|
163
|
+
)
|
|
164
|
+
by_path[section.base_path] = node
|
|
165
|
+
if node.level < 1: # the preamble is a sibling of the headings, never their parent
|
|
166
|
+
roots.append(node)
|
|
167
|
+
continue
|
|
168
|
+
while stack and stack[-1].level >= node.level:
|
|
169
|
+
stack.pop()
|
|
170
|
+
(stack[-1].children if stack else roots).append(node)
|
|
171
|
+
stack.append(node)
|
|
172
|
+
|
|
173
|
+
def freeze(node: _Pending) -> OutlineNode:
|
|
174
|
+
return OutlineNode(
|
|
175
|
+
heading_title=node.title,
|
|
176
|
+
heading_level=node.level,
|
|
177
|
+
heading_path=node.path,
|
|
178
|
+
start_line=node.start_line,
|
|
179
|
+
end_line=node.end_line,
|
|
180
|
+
token_estimate=node.tokens,
|
|
181
|
+
part_count=node.parts,
|
|
182
|
+
children=tuple(freeze(child) for child in node.children),
|
|
183
|
+
)
|
|
184
|
+
|
|
185
|
+
return [freeze(node) for node in roots]
|