sessionmemory 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,181 @@
1
+ """A page in a field: a markdown file with optional YAML frontmatter in a flat directory.
2
+
3
+ This is the one place a page's shape is known. The filename rule, the debris rule, and
4
+ the 8KB soft limit come from the memoryfield spec, and a page without frontmatter is a
5
+ valid page there, so `read_page` never refuses one.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import re
11
+ import uuid
12
+ from dataclasses import dataclass
13
+ from itertools import islice
14
+ from typing import TYPE_CHECKING, Any
15
+
16
+ from sessionmemory.lib import atomic
17
+ from sessionmemory.lib.frontmatter import FrontmatterError, parse, serialize
18
+ from sessionmemory.lib.ids import id_candidates, slugify
19
+
20
+ if TYPE_CHECKING:
21
+ from collections.abc import Mapping
22
+ from pathlib import Path
23
+
24
+ PAGE_LIMIT = 8192
25
+
26
+ _PAGE_NAME = re.compile(r"^[a-z0-9](?:[a-z0-9-]*[a-z0-9])?\.md$")
27
+ _DEBRIS_NAMES = frozenset({".DS_Store", "desktop.ini", "Thumbs.db"})
28
+ _MAX_CLAIM_ATTEMPTS = 1000
29
+
30
+
31
+ class PageError(ValueError):
32
+ """Raised when a page cannot be created as asked."""
33
+
34
+
35
+ @dataclass(frozen=True)
36
+ class Page:
37
+ """One page as read from disk."""
38
+
39
+ path: Path
40
+ meta: dict[str, Any]
41
+ body: str
42
+
43
+ def _text(self, key: str) -> str:
44
+ value = self.meta.get(key)
45
+ return value if isinstance(value, str) else ""
46
+
47
+ @property
48
+ def title(self) -> str:
49
+ """The page's title, or empty."""
50
+ return self._text("title")
51
+
52
+ @property
53
+ def uuid(self) -> str:
54
+ """The page's uuid, or empty."""
55
+ return self._text("uuid")
56
+
57
+ @property
58
+ def summary(self) -> str:
59
+ """The page's one-line summary, or empty."""
60
+ return self._text("summary")
61
+
62
+ @property
63
+ def created(self) -> str:
64
+ """The page's creation datetime, or empty."""
65
+ return self._text("created")
66
+
67
+ @property
68
+ def updated(self) -> str:
69
+ """The page's last-updated datetime, or empty."""
70
+ return self._text("updated")
71
+
72
+ @property
73
+ def size(self) -> int:
74
+ """The file's size in bytes."""
75
+ return self.path.stat().st_size
76
+
77
+
78
+ def is_page_name(name: str) -> bool:
79
+ """Report whether a filename conforms to the spec's page filename rule."""
80
+ return _PAGE_NAME.match(name) is not None
81
+
82
+
83
+ def is_debris(name: str) -> bool:
84
+ """Report whether a filename is sync, editor, or OS debris the spec says to ignore."""
85
+ return name in _DEBRIS_NAMES or ".sync-conflict-" in name or name.endswith("~")
86
+
87
+
88
+ def iter_pages(directory: Path) -> list[Path]:
89
+ """List the pages in one field, sorted by name.
90
+
91
+ Only conformant names at the top level count; the spec forbids indexing pages in
92
+ sub-directories, and a non-conformant name is reported by `doctor` rather than read.
93
+ """
94
+ if not directory.is_dir():
95
+ return []
96
+ return sorted(
97
+ path
98
+ for path in directory.iterdir()
99
+ if path.is_file() and not is_debris(path.name) and is_page_name(path.name)
100
+ )
101
+
102
+
103
+ def read_page(path: Path) -> Page:
104
+ """Read a page, tolerating a missing frontmatter block or invalid UTF-8."""
105
+ text = path.read_bytes().decode("utf-8", errors="replace")
106
+ try:
107
+ meta, body = parse(text)
108
+ except FrontmatterError:
109
+ return Page(path=path, meta={}, body=text)
110
+ return Page(path=path, meta=meta, body=body)
111
+
112
+
113
+ def _already_says(path: Path, meta: Mapping[str, Any], body: str) -> bool:
114
+ try:
115
+ current_meta, current_body = parse(path.read_text(encoding="utf-8"))
116
+ except (OSError, FrontmatterError, UnicodeDecodeError):
117
+ return False
118
+ return current_meta == dict(meta) and current_body.rstrip() == body.rstrip()
119
+
120
+
121
+ def write_page(path: Path, meta: Mapping[str, Any], body: str) -> None:
122
+ """Write a page atomically, leaving a file that already says this untouched.
123
+
124
+ A formatter restyles frontmatter quoting and spacing; comparing parsed content rather
125
+ than bytes keeps this CLI and a formatter from trading edits forever.
126
+ """
127
+ if _already_says(path, meta, body):
128
+ return
129
+ atomic.write_text(path, serialize(meta, body))
130
+
131
+
132
+ def claim_filename(directory: Path, title: str, *, stem: str | None = None) -> Path:
133
+ """Create an empty file for a new page, taking the first free name.
134
+
135
+ Exclusive creation is the gate, so two writers racing for one title get two files.
136
+
137
+ Raises:
138
+ PageError: If the title yields no slug or no name is free.
139
+ """
140
+ if stem is None:
141
+ try:
142
+ stem = slugify(title)
143
+ except ValueError as error:
144
+ raise PageError(str(error)) from error
145
+ taken = {path.stem for path in iter_pages(directory)}
146
+ for candidate in islice(id_candidates(stem, taken), _MAX_CLAIM_ATTEMPTS):
147
+ path = directory / f"{candidate}.md"
148
+ if atomic.claim(path):
149
+ return path
150
+ msg = f"no free filename for {stem!r} in {directory}"
151
+ raise PageError(msg)
152
+
153
+
154
+ def _create(directory: Path, meta: dict[str, Any], body: str, *, stem: str | None) -> Path:
155
+ path = claim_filename(directory, str(meta["title"]), stem=stem)
156
+ try:
157
+ atomic.write_text(path, serialize(meta, body))
158
+ except BaseException:
159
+ path.unlink(missing_ok=True)
160
+ raise
161
+ return path
162
+
163
+
164
+ def new_page(directory: Path, *, title: str, summary: str, body: str, now: str) -> Path:
165
+ """Create a memory page with the five fields the spec defines."""
166
+ meta = {
167
+ "title": title,
168
+ "uuid": str(uuid.uuid4()),
169
+ "summary": summary,
170
+ "created": now,
171
+ "updated": now,
172
+ }
173
+ return _create(directory, meta, body, stem=None)
174
+
175
+
176
+ def new_document(
177
+ directory: Path, *, title: str, body: str, now: str, stem: str | None = None
178
+ ) -> Path:
179
+ """Create a spec, plan, or log: a titled, dated file that is not a memory page."""
180
+ meta = {"title": title, "created": now, "updated": now}
181
+ return _create(directory, meta, body, stem=stem)
@@ -0,0 +1,214 @@
1
+ """One field's vector index: the spec's `pages` table in one SQLite file.
2
+
3
+ The index is derived from the pages beside it and can be deleted at any time. Freshness
4
+ is the sha256 of each file, so a page edited in any editor is re-embedded on the next
5
+ read, and every read refreshes first so nobody has to remember `reindex`.
6
+
7
+ There is no application lock. A page write is an atomic rename, SQLite serializes its
8
+ own writers, and a corrupt file is discarded and rebuilt rather than reported.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import datetime
14
+ import hashlib
15
+ import json
16
+ import sqlite3
17
+ from dataclasses import dataclass
18
+ from typing import TYPE_CHECKING
19
+
20
+ import sqlite_vec
21
+
22
+ from sessionmemory.lib import field
23
+
24
+ if TYPE_CHECKING:
25
+ from pathlib import Path
26
+
27
+ from sessionmemory.lib.embed import Embedder
28
+
29
+ # Measured on a real vault with nomic-embed-text-v1.5: a page that answers the query sits
30
+ # under 0.25, a related neighbor under 0.40, and the nearest page to an unrelated query
31
+ # sits at 0.45 or beyond. It matches the reference implementation's default for the model.
32
+ DEFAULT_MAX_DISTANCE = 0.45
33
+
34
+ _SCHEMA = """
35
+ CREATE TABLE IF NOT EXISTS pages (
36
+ filename TEXT PRIMARY KEY,
37
+ frontmatter JSON NOT NULL,
38
+ last_modified DATETIME NOT NULL,
39
+ sha256_hash BLOB NOT NULL,
40
+ embedding BLOB NOT NULL
41
+ );
42
+ """
43
+
44
+
45
+ @dataclass(frozen=True)
46
+ class Hit:
47
+ """One search result, nearest first when ordered by `distance`."""
48
+
49
+ path: Path
50
+ title: str
51
+ summary: str
52
+ distance: float
53
+
54
+
55
+ @dataclass(frozen=True)
56
+ class Refresh:
57
+ """What one refresh changed."""
58
+
59
+ added: int
60
+ updated: int
61
+ removed: int
62
+ unchanged: int
63
+
64
+
65
+ def index_path(field_dir: Path, embedder: Embedder) -> Path:
66
+ """Return the index file for one field, named for the model that fills it."""
67
+ return field_dir / f"{embedder.name}.sqlite3"
68
+
69
+
70
+ def connect(path: Path) -> sqlite3.Connection:
71
+ """Open an index file, creating the table if the file is new."""
72
+ conn = sqlite3.connect(path)
73
+ try:
74
+ conn.row_factory = sqlite3.Row
75
+ conn.enable_load_extension(True) # noqa: FBT003
76
+ sqlite_vec.load(conn)
77
+ conn.enable_load_extension(False) # noqa: FBT003
78
+ conn.execute("PRAGMA busy_timeout = 5000")
79
+ conn.executescript(_SCHEMA)
80
+ except BaseException:
81
+ conn.close()
82
+ raise
83
+ return conn
84
+
85
+
86
+ def _open(path: Path) -> sqlite3.Connection:
87
+ """Open the index, discarding and recreating a file SQLite cannot read."""
88
+ try:
89
+ return connect(path)
90
+ except sqlite3.DatabaseError:
91
+ path.unlink(missing_ok=True)
92
+ path.with_name(f"{path.name}-journal").unlink(missing_ok=True)
93
+ return connect(path)
94
+
95
+
96
+ def embedding_input(text: str) -> str:
97
+ """Return the first `PAGE_LIMIT` bytes of `text`, never splitting a character."""
98
+ encoded = text.encode("utf-8")
99
+ if len(encoded) <= field.PAGE_LIMIT:
100
+ return text
101
+ return encoded[: field.PAGE_LIMIT].decode("utf-8", errors="ignore")
102
+
103
+
104
+ def _mtime_iso(path: Path) -> str:
105
+ stamp = datetime.datetime.fromtimestamp(path.stat().st_mtime, tz=datetime.UTC)
106
+ return stamp.replace(microsecond=0).isoformat().replace("+00:00", "Z")
107
+
108
+
109
+ def _refresh(conn: sqlite3.Connection, field_dir: Path, embedder: Embedder) -> Refresh:
110
+ stored = {
111
+ row["filename"]: row["sha256_hash"]
112
+ for row in conn.execute("SELECT filename, sha256_hash FROM pages")
113
+ }
114
+ pending: list[tuple[str, str, str, bytes, str]] = []
115
+ unchanged = 0
116
+ seen: set[str] = set()
117
+ for path in field.iter_pages(field_dir):
118
+ seen.add(path.name)
119
+ raw = path.read_bytes()
120
+ digest = hashlib.sha256(raw).digest()
121
+ # Hash the raw bytes so freshness tracks the file as written; a page saved
122
+ # with invalid UTF-8 is still indexed rather than crashing the refresh.
123
+ text = raw.decode("utf-8", errors="replace")
124
+ if stored.get(path.name) == digest:
125
+ unchanged += 1
126
+ continue
127
+ page = field.read_page(path)
128
+ pending.append((path.name, json.dumps(page.meta), _mtime_iso(path), digest, text))
129
+
130
+ gone = sorted(set(stored) - seen)
131
+ conn.executemany("DELETE FROM pages WHERE filename = ?", [(name,) for name in gone])
132
+
133
+ # The spec's embedding input is the whole file, frontmatter included.
134
+ vectors = embedder.encode_documents([embedding_input(text) for *_rest, text in pending])
135
+ conn.executemany(
136
+ "INSERT OR REPLACE INTO pages (filename, frontmatter, last_modified, sha256_hash, embedding)"
137
+ " VALUES (?, ?, ?, ?, ?)",
138
+ [
139
+ (name, meta, modified, digest, sqlite_vec.serialize_float32(vector))
140
+ for (name, meta, modified, digest, _text), vector in zip(pending, vectors, strict=True)
141
+ ],
142
+ )
143
+ conn.commit()
144
+ added = sum(1 for name, *_rest in pending if name not in stored)
145
+ return Refresh(
146
+ added=added, updated=len(pending) - added, removed=len(gone), unchanged=unchanged
147
+ )
148
+
149
+
150
+ def refresh(field_dir: Path, embedder: Embedder) -> Refresh:
151
+ """Bring one field's index up to date with the pages beside it."""
152
+ if not field_dir.is_dir():
153
+ return Refresh(added=0, updated=0, removed=0, unchanged=0)
154
+ conn = _open(index_path(field_dir, embedder))
155
+ try:
156
+ return _refresh(conn, field_dir, embedder)
157
+ finally:
158
+ conn.close()
159
+
160
+
161
+ def search(
162
+ field_dir: Path,
163
+ embedder: Embedder,
164
+ query: str,
165
+ *,
166
+ limit: int,
167
+ max_distance: float = DEFAULT_MAX_DISTANCE,
168
+ ) -> list[Hit]:
169
+ """Return the pages within `max_distance` of `query`, nearest first, refreshing the index first.
170
+
171
+ A cutoff rather than a bare top-k, so a query nothing answers returns nothing instead
172
+ of the nearest pages dressed up as hits.
173
+ """
174
+ if not field_dir.is_dir():
175
+ return []
176
+ conn = _open(index_path(field_dir, embedder))
177
+ try:
178
+ _refresh(conn, field_dir, embedder)
179
+ rows = conn.execute(
180
+ "SELECT * FROM ("
181
+ " SELECT filename, frontmatter, vec_distance_cosine(embedding, ?) AS distance"
182
+ " FROM pages)"
183
+ " WHERE distance <= ? ORDER BY distance LIMIT ?",
184
+ (sqlite_vec.serialize_float32(embedder.encode_query(query)), max_distance, limit),
185
+ ).fetchall()
186
+ finally:
187
+ conn.close()
188
+ hits = []
189
+ for row in rows:
190
+ meta = json.loads(row["frontmatter"])
191
+ title = meta.get("title")
192
+ summary = meta.get("summary")
193
+ hits.append(
194
+ Hit(
195
+ path=field_dir / row["filename"],
196
+ title=title if isinstance(title, str) else "",
197
+ summary=summary if isinstance(summary, str) else "",
198
+ distance=float(row["distance"]),
199
+ )
200
+ )
201
+ return hits
202
+
203
+
204
+ def forget(field_dir: Path, embedder: Embedder, filename: str) -> None:
205
+ """Drop one page's row, so a deleted page stops matching before the next refresh."""
206
+ path = index_path(field_dir, embedder)
207
+ if not path.is_file():
208
+ return
209
+ conn = _open(path)
210
+ try:
211
+ conn.execute("DELETE FROM pages WHERE filename = ?", (filename,))
212
+ conn.commit()
213
+ finally:
214
+ conn.close()
@@ -0,0 +1,163 @@
1
+ """Read and write the YAML frontmatter block that heads every note.
2
+
3
+ Dates are normalized to ISO strings on the way in and quoted on the way out. PyYAML
4
+ would otherwise load `created: 2026-08-01` as a `datetime.date` and dump it back
5
+ unquoted, so a value's type would depend on whether the note had been round tripped.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import datetime
11
+ from typing import TYPE_CHECKING, Any
12
+
13
+ import yaml
14
+
15
+ if TYPE_CHECKING:
16
+ from collections.abc import Mapping
17
+
18
+ DELIMITER = "---"
19
+ BOM = "\ufeff"
20
+
21
+
22
+ class FrontmatterError(ValueError):
23
+ """Raised when a note's frontmatter is absent or malformed."""
24
+
25
+
26
+ class MissingFrontmatterError(FrontmatterError):
27
+ """Raised when a note has no frontmatter block at all."""
28
+
29
+
30
+ def _stringify_dates(value: Any) -> Any: # noqa: ANN401
31
+ """Convert any date or datetime in a nested structure to an ISO string.
32
+
33
+ Args:
34
+ value (Any): A value that may contain dates at any depth.
35
+
36
+ Returns:
37
+ Any: The same structure with dates replaced by ISO strings.
38
+ """
39
+ if isinstance(value, datetime.date):
40
+ return value.isoformat()
41
+ if isinstance(value, dict):
42
+ return {k: _stringify_dates(v) for k, v in value.items()}
43
+ if isinstance(value, list):
44
+ return [_stringify_dates(v) for v in value]
45
+ return value
46
+
47
+
48
+ def _split(text: str) -> tuple[str, str]:
49
+ """Return the raw YAML block and the body, tolerating a BOM and CRLF endings.
50
+
51
+ Raises:
52
+ FrontmatterError: If the block is missing or unterminated.
53
+ """
54
+ text = text.removeprefix(BOM).replace("\r\n", "\n")
55
+
56
+ if not text.startswith(f"{DELIMITER}\n"):
57
+ msg = "note has no frontmatter block"
58
+ raise MissingFrontmatterError(msg)
59
+
60
+ rest = text[len(DELIMITER) + 1 :]
61
+ if rest.startswith(f"{DELIMITER}\n"):
62
+ # An empty block's closing delimiter sits immediately after the opening
63
+ # one, so the "\n---\n" search below never gets the leading newline it
64
+ # needs: the opening strip already consumed it.
65
+ raw_meta = ""
66
+ body = rest[len(DELIMITER) + 1 :]
67
+ elif rest == DELIMITER:
68
+ # Same empty-block case, but the closing delimiter is also the end of
69
+ # the file, so there is no body to slice off.
70
+ raw_meta = ""
71
+ body = ""
72
+ else:
73
+ end = rest.find(f"\n{DELIMITER}\n")
74
+ if end == -1:
75
+ if rest.endswith(f"\n{DELIMITER}"):
76
+ # Closing delimiter is the last thing in the file, so there is no
77
+ # trailing "\n" to anchor the usual end + len(DELIMITER) + 2 offset.
78
+ raw_meta = rest[: -len(DELIMITER) - 1]
79
+ body = ""
80
+ else:
81
+ msg = "frontmatter block was never closed"
82
+ raise FrontmatterError(msg)
83
+ else:
84
+ raw_meta = rest[:end]
85
+ body = rest[end + len(DELIMITER) + 2 :]
86
+ return raw_meta, body
87
+
88
+
89
+ def _load(raw_meta: str) -> dict[str, Any]:
90
+ """Load a raw YAML block as the mapping it must be, with values as YAML typed them.
91
+
92
+ Raises:
93
+ FrontmatterError: If the block is not valid YAML or not a mapping.
94
+ """
95
+ try:
96
+ loaded = yaml.safe_load(raw_meta)
97
+ except yaml.YAMLError as error:
98
+ msg = f"frontmatter is not valid YAML: {error}"
99
+ raise FrontmatterError(msg) from error
100
+ loaded = {} if loaded is None else loaded
101
+ if not isinstance(loaded, dict):
102
+ msg = f"frontmatter must be a mapping, got {type(loaded).__name__}"
103
+ raise FrontmatterError(msg)
104
+ return loaded
105
+
106
+
107
+ def parse(text: str) -> tuple[dict[str, Any], str]:
108
+ """Split a note into its frontmatter mapping and its markdown body.
109
+
110
+ A leading byte order mark and Windows line endings are tolerated. Notes are
111
+ hand-edited in whatever editor is at hand, and a note is still a note when it comes
112
+ back with a BOM or CRLF endings, so refusing it would refuse a legible file.
113
+
114
+ Args:
115
+ text (str): The full contents of a note file.
116
+
117
+ Returns:
118
+ tuple[dict[str, Any], str]: The metadata mapping and the body.
119
+
120
+ Raises:
121
+ FrontmatterError: If the block is missing, unterminated, or not a mapping.
122
+ """
123
+ raw_meta, body = _split(text)
124
+ return _stringify_dates(_load(raw_meta)), body
125
+
126
+
127
+ def unquoted_datetime_keys(text: str) -> list[str]:
128
+ """Return the top-level keys whose bare values YAML typed as a date or datetime.
129
+
130
+ The memoryfield spec requires quoting datetimes, since a YAML 1.1 parser types a bare
131
+ one and a YAML 1.2 parser does not, so what the page says would depend on the reader.
132
+
133
+ Args:
134
+ text (str): The full contents of a note file.
135
+
136
+ Returns:
137
+ list[str]: The offending keys in file order, empty when every value is quoted.
138
+
139
+ Raises:
140
+ FrontmatterError: If the block is missing, unterminated, or not a mapping.
141
+ """
142
+ raw_meta, _body = _split(text)
143
+ return [key for key, value in _load(raw_meta).items() if isinstance(value, datetime.date)]
144
+
145
+
146
+ def serialize(meta: Mapping[str, Any], body: str) -> str:
147
+ """Render a metadata mapping and a body back into note file contents.
148
+
149
+ Args:
150
+ meta (Mapping[str, Any]): The metadata to write.
151
+ body (str): The markdown body.
152
+
153
+ Returns:
154
+ str: The complete file contents, ending in a single newline.
155
+ """
156
+ dumped = yaml.safe_dump(
157
+ dict(meta),
158
+ sort_keys=False,
159
+ default_flow_style=False,
160
+ allow_unicode=True,
161
+ width=88,
162
+ )
163
+ return f"{DELIMITER}\n{dumped}{DELIMITER}\n{body.rstrip()}\n"