markdown-memory 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,138 @@
1
+ """Whether the indexed documents are still what is on disk.
2
+
3
+ The database's own status is one SQLite snapshot and says nothing about the filesystem,
4
+ so this is composed beside it rather than inside it. ``FreshnessSweep`` owns the whole
5
+ of that state: the single-entry cache, its TTL, and the lock that makes reading the
6
+ cache, walking the tree and publishing the answer one step.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import stat
12
+ import threading
13
+ import time
14
+ from enum import Enum, auto
15
+ from pathlib import Path
16
+
17
+ from markdown_memory.db import Database
18
+ from markdown_memory.discovery import MAX_FILE_BYTES, hash_bytes, read_regular_file
19
+
20
+ # How long one filesystem sweep speaks for. An agent fires several searches in a single
21
+ # turn, and every one of them asks for the status: on a local ext4 tree 100 stats cost
22
+ # ~0.1 ms, but across a WSL2 or network boundary they cost 100-300 ms, which would double
23
+ # the latency of a query to re-answer a question whose answer cannot have changed much.
24
+ # Short enough that an edit is reported by the next search but one.
25
+ _FRESHNESS_TTL_SECONDS = 3.0
26
+
27
+
28
+ class _Verdict(Enum):
29
+ """What one file's freshness probe found."""
30
+
31
+ UNCHANGED = auto()
32
+ CHANGED = auto()
33
+ #: Same bytes, a time that has moved: nothing to report, but worth writing down.
34
+ SAME_BYTES_NEW_TIME = auto()
35
+
36
+
37
+ def _compare(path: Path, content_hash: str, mtime_ns: int | None) -> tuple[_Verdict, int]:
38
+ """Whether the file behind an indexed document differs from what was indexed.
39
+
40
+ The modification time is the cheap question and the bytes are the expensive one, so
41
+ the hash is only computed where the time has moved: a `touch`, a checkout that
42
+ rewrites a file with its own contents, or a copy that preserves nothing but the text
43
+ must not be reported as a change an agent should act on. A file that has vanished, no
44
+ longer resolves to a regular file, or cannot be read counts as changed - not because
45
+ its bytes are known to differ, but because they cannot be checked at all. That applies
46
+ where the bytes had to be read: a file whose recorded time still matches is answered
47
+ from the time alone, so losing permission to read it - without touching it - is not
48
+ reported here. The next index run cannot read it either, and records a failure, which
49
+ is what takes `coverage` to `"unknown"`.
50
+
51
+ A stored `None` means no modification time was recorded - a row written before the
52
+ column existed - rather than a time of zero, so no real timestamp can be mistaken for
53
+ it, the epoch included. Those files are answered by their bytes until an index run
54
+ writes a time for them.
55
+
56
+ The one edit this cannot see is a file rewritten with its modification time put back
57
+ to what it was: no timestamp moved, so no hash is taken. Indexing itself is not fooled
58
+ - it hashes every file it walks - so `index_directory` still rebuilds that document;
59
+ what is missed is only the hint that it is worth running. Seeing it here would mean
60
+ hashing every indexed file on every query, or storing a second timestamp to compare
61
+ against, and this signal is not worth either.
62
+ """
63
+ try:
64
+ info = path.stat()
65
+ except OSError:
66
+ return _Verdict.CHANGED, 0
67
+ if not stat.S_ISREG(info.st_mode):
68
+ return _Verdict.CHANGED, 0
69
+ if mtime_ns is not None and info.st_mtime_ns == mtime_ns:
70
+ return _Verdict.UNCHANGED, info.st_mtime_ns
71
+ try:
72
+ data = read_regular_file(path)
73
+ except OSError:
74
+ return _Verdict.CHANGED, 0
75
+ if data is None or len(data) > MAX_FILE_BYTES or hash_bytes(data) != content_hash:
76
+ return _Verdict.CHANGED, 0
77
+ # The time from the stat that came *before* the read, never a fresher one: a file
78
+ # rewritten after these bytes were hashed must not be recorded as verified at the
79
+ # moment of its rewrite, or the next sweep would trust a time that belongs to content
80
+ # nobody checked.
81
+ return _Verdict.SAME_BYTES_NEW_TIME, info.st_mtime_ns
82
+
83
+
84
+ class FreshnessSweep:
85
+ """How many indexed documents are no longer what was indexed, cheaply and often.
86
+
87
+ One owner for three things that only make sense together: the count, the moment it
88
+ was taken, and the lock that keeps a sweep whole. Before this they were three
89
+ attributes on the service, which made it possible to reset one and not the others.
90
+ """
91
+
92
+ def __init__(self, db: Database, ttl: float = _FRESHNESS_TTL_SECONDS) -> None:
93
+ self._db = db
94
+ self._ttl = ttl
95
+ #: One entry, not a map keyed on the caller's path: an agent fires several
96
+ #: searches per turn against the same scope, and a map would grow for the life of
97
+ #: the server, one entry per spelling anybody ever asked about.
98
+ self._cache: tuple[str, float, int] | None = None
99
+ #: Held for the whole of a sweep, so that reading the cache, walking the
100
+ #: filesystem and storing the answer are one step. Without it a sweep that
101
+ #: indexing overtook would publish a count of a tree that no longer exists - the
102
+ #: one moment an agent is most likely to ask - and two sweeps racing could leave
103
+ #: the older one's answer behind. It also means a second caller arriving mid-sweep
104
+ #: waits and is served the result rather than walking the tree again.
105
+ self._lock = threading.Lock()
106
+
107
+ def invalidate(self) -> None:
108
+ """Forget the cached count: indexing has changed what the answer would be."""
109
+ with self._lock:
110
+ self._cache = None
111
+
112
+ def changed_files(self, scope: str) -> int:
113
+ with self._lock:
114
+ cached = self._cache
115
+ if (
116
+ cached is not None
117
+ and cached[0] == scope
118
+ and time.monotonic() - cached[1] < self._ttl
119
+ ):
120
+ return cached[2]
121
+ fingerprints = self._db.document_fingerprints(scope)
122
+ changed = 0
123
+ for file_path, (content_hash, mtime_ns) in fingerprints.items():
124
+ verdict, seen_ns = _compare(Path(file_path), content_hash, mtime_ns)
125
+ if verdict is _Verdict.CHANGED:
126
+ changed += 1
127
+ elif verdict is _Verdict.SAME_BYTES_NEW_TIME:
128
+ # The hash was computed to answer this, and the answer was "unchanged".
129
+ # Writing the time down means the next sweep reads it instead of the
130
+ # file - otherwise one `touch` costs a full hash every window until an
131
+ # index run happens to come past. A file that moved again in between
132
+ # is simply hashed again next time; nothing is lost by missing it.
133
+ self._db.record_modification_time(file_path, content_hash, mtime_ns, seen_ns)
134
+ # Stamped when the answer was produced, not when the sweep began: the walk
135
+ # itself takes time on a slow mount, and a window that starts before the
136
+ # measurement is a window the measurement was never true for.
137
+ self._cache = (scope, time.monotonic(), changed)
138
+ return changed
@@ -0,0 +1,185 @@
1
+ """Turning what an agent typed into the section it meant.
2
+
3
+ Heading paths arrive as breadcrumbs (``Root > Child``) and have to survive casing,
4
+ spacing and ambiguity; the outline is the same tree seen from above.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import re
10
+ from collections.abc import Callable, Sequence
11
+ from dataclasses import dataclass
12
+ from pathlib import Path
13
+
14
+ from markdown_memory.exceptions import (
15
+ MarkdownMemoryError,
16
+ SectionNotFoundError,
17
+ )
18
+ from markdown_memory.models import (
19
+ PATH_SEPARATOR,
20
+ OutlineNode,
21
+ Section,
22
+ estimate_tokens,
23
+ )
24
+
25
+ _PATH_SEPARATOR_PATTERN = re.compile(r"\s*>\s*")
26
+
27
+
28
+ _MAX_LISTED_PATHS = 40
29
+
30
+
31
+ # ---------------------------------------------------------------------- pure helpers
32
+
33
+
34
+ def _user_path(text: str, error: type[MarkdownMemoryError]) -> Path:
35
+ """``Path(text).expanduser()``; pathlib's RuntimeError/ValueError become ``error``."""
36
+ try:
37
+ text.encode("utf-8") # a lone surrogate cannot be bound as a SQLite parameter
38
+ return Path(text).expanduser()
39
+ except (RuntimeError, ValueError) as exc: # unknown ~user, embedded NUL, bad encoding
40
+ raise error(f"Invalid path {text!r}: {exc}") from exc
41
+
42
+
43
+ def _absolute(path: Path, error: type[MarkdownMemoryError]) -> Path:
44
+ try:
45
+ resolved = path.expanduser().resolve()
46
+ str(resolved).encode("utf-8")
47
+ except (OSError, RuntimeError, ValueError) as exc: # symlink loop, NUL, bad encoding
48
+ raise error(f"Invalid path {str(path)!r}: {exc}") from exc
49
+ return resolved
50
+
51
+
52
+ def normalize_heading_path(heading_path: str) -> str:
53
+ """Canonicalise breadcrumb spacing: ``A>B`` and ``A > B`` both become ``A > B``."""
54
+ return PATH_SEPARATOR.join(
55
+ segment.strip() for segment in _PATH_SEPARATOR_PATTERN.split(heading_path.strip())
56
+ )
57
+
58
+
59
+ def _casefolded(heading_path: str) -> str:
60
+ return normalize_heading_path(heading_path).casefold()
61
+
62
+
63
+ # Progressively looser ways to compare a requested path with a stored one. The same
64
+ # transformation is applied to BOTH sides, so a title containing '>' ("Step 1 -> Step 2",
65
+ # "Result<T, E>") still matches itself, and an exact-case request beats a case-folded one
66
+ # (sibling headings "Setup" and "SETUP" stay individually addressable).
67
+ _MATCH_KEYS: tuple[Callable[[str], str], ...] = (str, normalize_heading_path, _casefolded)
68
+
69
+
70
+ def resolve_heading_path(candidates: Sequence[str], requested: str) -> str:
71
+ """The one stored path that ``requested`` designates.
72
+
73
+ Whole-path matches are tried before trailing fragments (``Child > Subchild`` or the
74
+ bare heading title); within each, stricter comparisons come first. The first tier
75
+ with any match decides: one match wins, several are reported as ambiguous.
76
+ """
77
+ for as_suffix in (False, True):
78
+ for key in _MATCH_KEYS:
79
+ wanted = key(requested)
80
+ if as_suffix:
81
+ ending = PATH_SEPARATOR + wanted
82
+ matches = [path for path in candidates if key(path).endswith(ending)]
83
+ else:
84
+ matches = [path for path in candidates if key(path) == wanted]
85
+ if len(matches) == 1:
86
+ return matches[0]
87
+ if matches:
88
+ raise SectionNotFoundError(
89
+ f"'{requested}' is ambiguous; use one of (exact spelling): "
90
+ + " | ".join(matches[:_MAX_LISTED_PATHS])
91
+ )
92
+ available = " | ".join(candidates[:_MAX_LISTED_PATHS]) or "(document has no sections)"
93
+ raise SectionNotFoundError(f"No section '{requested}'. Available heading paths: {available}")
94
+
95
+
96
+ def select_sections(
97
+ sections: Sequence[Section], heading_path: str, *, include_subsections: bool = False
98
+ ) -> list[Section]:
99
+ """Sections addressed by ``heading_path``, in source order.
100
+
101
+ ``heading_path`` may name a whole section (every part of it is returned) or a single
102
+ ``(Part n)`` of an oversized one. ``include_subsections`` adds the section's
103
+ descendants; it has no meaning for a single part and is ignored there.
104
+ """
105
+ requested = heading_path.strip()
106
+ if not requested:
107
+ raise SectionNotFoundError("heading_path must not be empty")
108
+ base_paths = list(dict.fromkeys(section.base_path for section in sections))
109
+ parts = {s.heading_path: s for s in sections if s.part_index > 0}
110
+ part_paths = [path for path in parts if path not in base_paths]
111
+ chosen = resolve_heading_path([*base_paths, *part_paths], requested)
112
+ if chosen not in base_paths:
113
+ return [parts[chosen]]
114
+
115
+ selected: list[Section] = []
116
+ level: int | None = None
117
+ for section in sections:
118
+ if section.base_path == chosen:
119
+ level = section.heading_level
120
+ selected.append(section)
121
+ elif level is not None:
122
+ # Descendants are the sections that follow until a heading at the same or a
123
+ # shallower level; walking the order is immune to '>' inside titles.
124
+ if not include_subsections or level < 1 or section.heading_level <= level:
125
+ break
126
+ selected.append(section)
127
+ return selected
128
+
129
+
130
+ def build_outline(sections: Sequence[Section]) -> list[OutlineNode]:
131
+ """Nest a document's sections into a hierarchical table of contents."""
132
+
133
+ @dataclass(slots=True)
134
+ class _Pending:
135
+ title: str
136
+ level: int
137
+ path: str
138
+ start_line: int
139
+ end_line: int
140
+ tokens: int
141
+ parts: int
142
+ children: list[_Pending]
143
+
144
+ roots: list[_Pending] = []
145
+ stack: list[_Pending] = []
146
+ by_path: dict[str, _Pending] = {}
147
+ for section in sections:
148
+ existing = by_path.get(section.base_path)
149
+ if existing is not None: # a further part of an oversized section
150
+ existing.end_line = max(existing.end_line, section.end_line)
151
+ existing.tokens += estimate_tokens(section.content)
152
+ existing.parts += 1
153
+ continue
154
+ node = _Pending(
155
+ title=section.heading_title,
156
+ level=section.heading_level,
157
+ path=section.base_path,
158
+ start_line=section.start_line,
159
+ end_line=section.end_line,
160
+ tokens=estimate_tokens(section.content),
161
+ parts=1,
162
+ children=[],
163
+ )
164
+ by_path[section.base_path] = node
165
+ if node.level < 1: # the preamble is a sibling of the headings, never their parent
166
+ roots.append(node)
167
+ continue
168
+ while stack and stack[-1].level >= node.level:
169
+ stack.pop()
170
+ (stack[-1].children if stack else roots).append(node)
171
+ stack.append(node)
172
+
173
+ def freeze(node: _Pending) -> OutlineNode:
174
+ return OutlineNode(
175
+ heading_title=node.title,
176
+ heading_level=node.level,
177
+ heading_path=node.path,
178
+ start_line=node.start_line,
179
+ end_line=node.end_line,
180
+ token_estimate=node.tokens,
181
+ part_count=node.parts,
182
+ children=tuple(freeze(child) for child in node.children),
183
+ )
184
+
185
+ return [freeze(node) for node in roots]