sessionmemory 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sessionmemory/__init__.py +1 -0
- sessionmemory/cli.py +62 -0
- sessionmemory/commands/__init__.py +1 -0
- sessionmemory/commands/_common.py +207 -0
- sessionmemory/commands/delete.py +71 -0
- sessionmemory/commands/doctor.py +30 -0
- sessionmemory/commands/export.py +61 -0
- sessionmemory/commands/init.py +88 -0
- sessionmemory/commands/inject.py +25 -0
- sessionmemory/commands/log.py +64 -0
- sessionmemory/commands/new.py +102 -0
- sessionmemory/commands/project.py +321 -0
- sessionmemory/commands/reindex.py +46 -0
- sessionmemory/commands/search.py +83 -0
- sessionmemory/lib/__init__.py +1 -0
- sessionmemory/lib/atomic.py +71 -0
- sessionmemory/lib/bootstrap.py +138 -0
- sessionmemory/lib/config.py +102 -0
- sessionmemory/lib/doctor.py +186 -0
- sessionmemory/lib/embed.py +123 -0
- sessionmemory/lib/export.py +25 -0
- sessionmemory/lib/field.py +181 -0
- sessionmemory/lib/fieldindex.py +214 -0
- sessionmemory/lib/frontmatter.py +163 -0
- sessionmemory/lib/gitinfo.py +177 -0
- sessionmemory/lib/ids.py +101 -0
- sessionmemory/lib/inject.py +97 -0
- sessionmemory/lib/log.py +73 -0
- sessionmemory/lib/paths.py +77 -0
- sessionmemory/lib/registry.py +269 -0
- sessionmemory/lib/resolve.py +75 -0
- sessionmemory-0.2.0.dist-info/METADATA +250 -0
- sessionmemory-0.2.0.dist-info/RECORD +35 -0
- sessionmemory-0.2.0.dist-info/WHEEL +4 -0
- sessionmemory-0.2.0.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
"""Create the parts of a vault that cannot be created on demand.
|
|
2
|
+
|
|
3
|
+
Note directories are not among them. Git does not track an empty directory, so a
|
|
4
|
+
pre-created `learnings/` would not survive the vault's first commit, and every note
|
|
5
|
+
write already creates its own parent. What has to exist up front is what nothing else
|
|
6
|
+
will ever write: the marker that identifies the directory as a vault, the gitignore that
|
|
7
|
+
keeps each field's index out of the backup, and a README for the human who opens
|
|
8
|
+
the vault in Obsidian.
|
|
9
|
+
|
|
10
|
+
Nothing here overwrites an existing file. Initialization has to be safe to re-run
|
|
11
|
+
against a vault that has been in use for a year, because the reason to re-run it is that
|
|
12
|
+
a later version of the CLI adds a file to this list.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from typing import TYPE_CHECKING
|
|
19
|
+
|
|
20
|
+
import tomli_w
|
|
21
|
+
|
|
22
|
+
from sessionmemory.lib import atomic
|
|
23
|
+
from sessionmemory.lib.config import VAULT_MARKER, is_initialized, today
|
|
24
|
+
from sessionmemory.lib.embed import MODEL_CODE
|
|
25
|
+
from sessionmemory.lib.paths import SYSTEM_DIR
|
|
26
|
+
|
|
27
|
+
if TYPE_CHECKING:
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
|
|
30
|
+
GITIGNORE = """\
|
|
31
|
+
# Each field's vector index is derived from the pages beside it and is
|
|
32
|
+
# rebuilt by `sessionmemory reindex`. Committing it would churn history on every write.
|
|
33
|
+
*.sqlite3
|
|
34
|
+
*.sqlite3-journal
|
|
35
|
+
|
|
36
|
+
# Obsidian's per-machine window state, which conflicts on every sync.
|
|
37
|
+
.obsidian/workspace*.json
|
|
38
|
+
|
|
39
|
+
.DS_Store
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
README = f"""\
|
|
43
|
+
# Session Memory Vault
|
|
44
|
+
|
|
45
|
+
One folder per project under `projects/`. Inside each:
|
|
46
|
+
|
|
47
|
+
- `learnings/` is a field: flat markdown pages, embedded and searchable.
|
|
48
|
+
- `logs/` is a second field, one page per session, searched on request.
|
|
49
|
+
- `specs/`, `plans/`, and `backlog.md` are plain files, never indexed.
|
|
50
|
+
- `backlog.md` is the list of open items for the project, one line each, sized by
|
|
51
|
+
effort and grouped by commit type.
|
|
52
|
+
|
|
53
|
+
The `{MODEL_CODE}.sqlite3` file inside a field is its vector index. It is
|
|
54
|
+
derived from the pages beside it, gitignored, and rebuilt by `sessionmemory reindex`.
|
|
55
|
+
|
|
56
|
+
`_system/registry.toml` maps each project's git remote and root to its folder here, and
|
|
57
|
+
`_system/vault.toml` is the marker that tells the CLI this directory is a vault.
|
|
58
|
+
|
|
59
|
+
Pages are created with `sessionmemory new learning` and searched with
|
|
60
|
+
`sessionmemory search`. Everything else is an ordinary file you read and edit directly.
|
|
61
|
+
The format follows the memoryfield spec: https://github.com/calpaterson/memoryfield-spec
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
@dataclass(frozen=True)
|
|
66
|
+
class InitResult:
|
|
67
|
+
"""What initialization changed, and what it found already in place."""
|
|
68
|
+
|
|
69
|
+
vault: Path
|
|
70
|
+
created: tuple[str, ...]
|
|
71
|
+
existed: tuple[str, ...]
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class NotAVaultError(RuntimeError):
|
|
75
|
+
"""Raised when a non-empty directory holding no marker is asked to become a vault."""
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def is_empty(vault: Path) -> bool:
|
|
79
|
+
"""Report whether `vault` holds nothing but dot entries.
|
|
80
|
+
|
|
81
|
+
Shared with `commands/_common.require_vault`, which needs the same check to decide
|
|
82
|
+
between naming `sessionmemory init` and `sessionmemory init --force` in its fix-it message.
|
|
83
|
+
|
|
84
|
+
Args:
|
|
85
|
+
vault (Path): The directory to check.
|
|
86
|
+
|
|
87
|
+
Returns:
|
|
88
|
+
bool: True when the directory has no non-dot entries.
|
|
89
|
+
"""
|
|
90
|
+
return all(entry.name.startswith(".") for entry in vault.iterdir())
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _marker_contents() -> str:
|
|
94
|
+
"""Build the marker file's contents.
|
|
95
|
+
|
|
96
|
+
Returns:
|
|
97
|
+
str: The TOML text for `_system/vault.toml`.
|
|
98
|
+
"""
|
|
99
|
+
return tomli_w.dumps({"created": today()})
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def initialize(vault: Path, *, force: bool = False) -> InitResult:
|
|
103
|
+
"""Create every file a vault needs and report what was and was not already there.
|
|
104
|
+
|
|
105
|
+
Args:
|
|
106
|
+
vault (Path): The directory to initialize. It is created if absent.
|
|
107
|
+
force (bool): Accept a non-empty directory that holds no vault marker.
|
|
108
|
+
|
|
109
|
+
Returns:
|
|
110
|
+
InitResult: The vault-relative names created and the ones already present.
|
|
111
|
+
|
|
112
|
+
Raises:
|
|
113
|
+
NotAVaultError: If the directory holds files but no marker and `force` is unset.
|
|
114
|
+
"""
|
|
115
|
+
vault.mkdir(parents=True, exist_ok=True)
|
|
116
|
+
|
|
117
|
+
marker = f"{SYSTEM_DIR}/{VAULT_MARKER}"
|
|
118
|
+
if not force and not is_empty(vault) and not is_initialized(vault):
|
|
119
|
+
msg = f"{vault} is not empty and has no vault marker; pass --force to initialize it anyway"
|
|
120
|
+
raise NotAVaultError(msg)
|
|
121
|
+
|
|
122
|
+
contents: dict[str, str] = {
|
|
123
|
+
marker: _marker_contents(),
|
|
124
|
+
".gitignore": GITIGNORE,
|
|
125
|
+
"README.md": README,
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
created: list[str] = []
|
|
129
|
+
existed: list[str] = []
|
|
130
|
+
for relative_path, text in contents.items():
|
|
131
|
+
destination = vault / relative_path
|
|
132
|
+
if destination.exists():
|
|
133
|
+
existed.append(relative_path)
|
|
134
|
+
continue
|
|
135
|
+
atomic.write_text(destination, text)
|
|
136
|
+
created.append(relative_path)
|
|
137
|
+
|
|
138
|
+
return InitResult(vault=vault, created=tuple(sorted(created)), existed=tuple(sorted(existed)))
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""Locate the vault."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import datetime
|
|
6
|
+
import os
|
|
7
|
+
import tomllib
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from sessionmemory.lib.paths import SYSTEM_DIR
|
|
11
|
+
|
|
12
|
+
VAULT_ENV_VAR = "SESSIONMEMORY_VAULT"
|
|
13
|
+
|
|
14
|
+
VAULT_MARKER = "vault.toml"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class VaultNotConfiguredError(RuntimeError):
|
|
18
|
+
"""Raised when the vault location is unset or does not exist."""
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def vault_root() -> Path:
|
|
22
|
+
"""Return the vault directory named by the environment.
|
|
23
|
+
|
|
24
|
+
The path is resolved, so every directory the CLI prints is absolute, such as the
|
|
25
|
+
`project_dir` in `sessionmemory project --json`. A relative value would otherwise be
|
|
26
|
+
printed as given, and usable only from the directory the variable was written for.
|
|
27
|
+
|
|
28
|
+
Returns:
|
|
29
|
+
Path: The resolved vault directory.
|
|
30
|
+
|
|
31
|
+
Raises:
|
|
32
|
+
VaultNotConfiguredError: If the variable is unset or names a missing directory.
|
|
33
|
+
"""
|
|
34
|
+
raw = os.environ.get(VAULT_ENV_VAR)
|
|
35
|
+
if not raw:
|
|
36
|
+
msg = f"{VAULT_ENV_VAR} is not set. Point it at your vault repository."
|
|
37
|
+
raise VaultNotConfiguredError(msg)
|
|
38
|
+
|
|
39
|
+
root = Path(raw).expanduser().resolve()
|
|
40
|
+
if not root.is_dir():
|
|
41
|
+
msg = f"{VAULT_ENV_VAR} is {root}, which does not exist or is not a directory."
|
|
42
|
+
raise VaultNotConfiguredError(msg)
|
|
43
|
+
|
|
44
|
+
return root
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def today() -> str:
|
|
48
|
+
"""Return today's date as an ISO string.
|
|
49
|
+
|
|
50
|
+
Returns:
|
|
51
|
+
str: Today's date.
|
|
52
|
+
"""
|
|
53
|
+
return datetime.datetime.now(tz=datetime.UTC).date().isoformat()
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def now() -> str:
|
|
57
|
+
"""Return the current UTC time as the quoted-string ISO form the memoryfield spec wants.
|
|
58
|
+
|
|
59
|
+
Second precision and a `Z` suffix: PyYAML quotes a value of this shape on dump, which
|
|
60
|
+
is what keeps a YAML 1.1 parser from coercing it to a datetime on the way back in.
|
|
61
|
+
"""
|
|
62
|
+
stamp = datetime.datetime.now(tz=datetime.UTC).replace(microsecond=0)
|
|
63
|
+
return stamp.isoformat().replace("+00:00", "Z")
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def marker_path(vault: Path) -> Path:
|
|
67
|
+
"""Return the path of the file that proves a directory is a vault.
|
|
68
|
+
|
|
69
|
+
Args:
|
|
70
|
+
vault (Path): The vault root.
|
|
71
|
+
|
|
72
|
+
Returns:
|
|
73
|
+
Path: The marker file, whether or not it exists.
|
|
74
|
+
"""
|
|
75
|
+
return vault / SYSTEM_DIR / VAULT_MARKER
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def is_initialized(vault: Path) -> bool:
|
|
79
|
+
"""Report whether `vault` has been initialized by `sessionmemory init`.
|
|
80
|
+
|
|
81
|
+
A directory that merely exists is not a vault. `SESSIONMEMORY_VAULT` pointing at a
|
|
82
|
+
home directory or a mistyped path would otherwise be accepted, and the first note
|
|
83
|
+
written would scatter a `learnings/` tree into it.
|
|
84
|
+
|
|
85
|
+
A marker that cannot be parsed is treated as absent rather than as proof, so a
|
|
86
|
+
truncated file produces "run sessionmemory init" instead of an error from deep inside a
|
|
87
|
+
later command.
|
|
88
|
+
|
|
89
|
+
Args:
|
|
90
|
+
vault (Path): The vault root.
|
|
91
|
+
|
|
92
|
+
Returns:
|
|
93
|
+
bool: True when a readable marker file is present.
|
|
94
|
+
"""
|
|
95
|
+
path = marker_path(vault)
|
|
96
|
+
if not path.is_file():
|
|
97
|
+
return False
|
|
98
|
+
try:
|
|
99
|
+
tomllib.loads(path.read_text(encoding="utf-8"))
|
|
100
|
+
except (tomllib.TOMLDecodeError, OSError, UnicodeDecodeError):
|
|
101
|
+
return False
|
|
102
|
+
return True
|
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
"""The six things that can be wrong with a vault, stated without severity.
|
|
2
|
+
|
|
3
|
+
Every check is a suggestion. The article's argument holds: an irrelevant or imperfect
|
|
4
|
+
page is never surfaced by semantic search, so nothing here fails a build. There is no
|
|
5
|
+
repair, because repair is what grew to three hundred lines last time.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import hashlib
|
|
11
|
+
import sqlite3
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import TYPE_CHECKING
|
|
15
|
+
|
|
16
|
+
from sessionmemory.lib import field, fieldindex, paths, registry
|
|
17
|
+
from sessionmemory.lib.frontmatter import (
|
|
18
|
+
FrontmatterError,
|
|
19
|
+
MissingFrontmatterError,
|
|
20
|
+
parse,
|
|
21
|
+
unquoted_datetime_keys,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
if TYPE_CHECKING:
|
|
25
|
+
from collections.abc import Callable
|
|
26
|
+
|
|
27
|
+
from sessionmemory.lib.embed import Embedder
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True)
|
|
31
|
+
class Finding:
|
|
32
|
+
"""One thing a check found."""
|
|
33
|
+
|
|
34
|
+
check: str
|
|
35
|
+
path: str
|
|
36
|
+
message: str
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _fields(vault: Path) -> list[Path]:
|
|
40
|
+
return [
|
|
41
|
+
paths.project_dir(vault, slug) / name
|
|
42
|
+
for slug in paths.iter_project_slugs(vault)
|
|
43
|
+
for name in paths.FIELD_DIRS
|
|
44
|
+
if (paths.project_dir(vault, slug) / name).is_dir()
|
|
45
|
+
]
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def nonconformant_names(vault: Path, _embedder: Embedder) -> list[Finding]:
|
|
49
|
+
"""Report a markdown file in a field whose name breaks the spec's filename rule."""
|
|
50
|
+
return [
|
|
51
|
+
Finding("filename", str(path), "not lowercase ascii letters, digits, and hyphens")
|
|
52
|
+
for directory in _fields(vault)
|
|
53
|
+
for path in sorted(directory.glob("*.md"))
|
|
54
|
+
if path.is_file() and not field.is_debris(path.name) and not field.is_page_name(path.name)
|
|
55
|
+
]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def oversized_pages(vault: Path, _embedder: Embedder) -> list[Finding]:
|
|
59
|
+
"""Report a page over the 8KB limit; only its first 8KB is embedded."""
|
|
60
|
+
return [
|
|
61
|
+
Finding(
|
|
62
|
+
"size",
|
|
63
|
+
str(path),
|
|
64
|
+
f"{path.stat().st_size} bytes; split it, the limit is {field.PAGE_LIMIT}",
|
|
65
|
+
)
|
|
66
|
+
for directory in _fields(vault)
|
|
67
|
+
for path in field.iter_pages(directory)
|
|
68
|
+
if path.stat().st_size > field.PAGE_LIMIT
|
|
69
|
+
]
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _blank(value: object) -> bool:
|
|
73
|
+
"""Report whether a frontmatter value is missing or an empty/whitespace string."""
|
|
74
|
+
return not (isinstance(value, str) and value.strip())
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def malformed_frontmatter(vault: Path, _embedder: Embedder) -> list[Finding]:
|
|
78
|
+
"""Report a page whose frontmatter cannot be parsed, or a learning missing title or summary.
|
|
79
|
+
|
|
80
|
+
A logs page with no frontmatter block is fine; a learnings page needs a title and a
|
|
81
|
+
summary to be worth anything a search result shows, so it has nothing to be missing
|
|
82
|
+
when it has no block at all.
|
|
83
|
+
"""
|
|
84
|
+
findings = []
|
|
85
|
+
for directory in _fields(vault):
|
|
86
|
+
for path in field.iter_pages(directory):
|
|
87
|
+
raw = path.read_bytes()
|
|
88
|
+
try:
|
|
89
|
+
text = raw.decode("utf-8")
|
|
90
|
+
except UnicodeDecodeError:
|
|
91
|
+
findings.append(Finding("frontmatter", str(path), "not valid UTF-8"))
|
|
92
|
+
continue
|
|
93
|
+
try:
|
|
94
|
+
meta, _body = parse(text)
|
|
95
|
+
except MissingFrontmatterError:
|
|
96
|
+
if path.parent.name == paths.LEARNINGS_DIR:
|
|
97
|
+
findings.append(Finding("frontmatter", str(path), "missing title or summary"))
|
|
98
|
+
continue
|
|
99
|
+
except FrontmatterError as error:
|
|
100
|
+
findings.append(Finding("frontmatter", str(path), str(error)))
|
|
101
|
+
continue
|
|
102
|
+
if path.parent.name == paths.LEARNINGS_DIR and (
|
|
103
|
+
_blank(meta.get("title")) or _blank(meta.get("summary"))
|
|
104
|
+
):
|
|
105
|
+
findings.append(Finding("frontmatter", str(path), "missing title or summary"))
|
|
106
|
+
return findings
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def unquoted_datetimes(vault: Path, _embedder: Embedder) -> list[Finding]:
|
|
110
|
+
"""Report a page whose frontmatter carries a bare date or datetime.
|
|
111
|
+
|
|
112
|
+
The spec requires quoting them: a YAML 1.1 parser types a bare value and a YAML 1.2
|
|
113
|
+
parser leaves it a string, so what the page says would depend on which reads it. A
|
|
114
|
+
block that cannot be parsed at all is left to the frontmatter check.
|
|
115
|
+
"""
|
|
116
|
+
findings = []
|
|
117
|
+
for directory in _fields(vault):
|
|
118
|
+
for path in field.iter_pages(directory):
|
|
119
|
+
try:
|
|
120
|
+
keys = unquoted_datetime_keys(path.read_text(encoding="utf-8"))
|
|
121
|
+
except (UnicodeDecodeError, FrontmatterError):
|
|
122
|
+
continue
|
|
123
|
+
if keys:
|
|
124
|
+
message = f"{', '.join(keys)}: unquoted; quote the value"
|
|
125
|
+
findings.append(Finding("datetime", str(path), message))
|
|
126
|
+
return findings
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def dead_projects(vault: Path, _embedder: Embedder) -> list[Finding]:
|
|
130
|
+
"""Report a registered project whose repository root no longer exists."""
|
|
131
|
+
try:
|
|
132
|
+
projects = registry.load(vault)
|
|
133
|
+
except registry.RegistryError as error:
|
|
134
|
+
return [Finding("registry", "", str(error))]
|
|
135
|
+
return [
|
|
136
|
+
Finding("project", slug, f"root {project.root} does not exist")
|
|
137
|
+
for slug, project in sorted(projects.items())
|
|
138
|
+
if project.root and not Path(project.root).is_dir()
|
|
139
|
+
]
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _is_stale(directory: Path, embedder: Embedder) -> str | None:
|
|
143
|
+
index = fieldindex.index_path(directory, embedder)
|
|
144
|
+
if not index.is_file():
|
|
145
|
+
return None
|
|
146
|
+
try:
|
|
147
|
+
conn = fieldindex.connect(index)
|
|
148
|
+
except sqlite3.DatabaseError:
|
|
149
|
+
return "index is not a readable database; run: sessionmemory reindex"
|
|
150
|
+
try:
|
|
151
|
+
stored = {
|
|
152
|
+
row["filename"]: row["sha256_hash"]
|
|
153
|
+
for row in conn.execute("SELECT filename, sha256_hash FROM pages")
|
|
154
|
+
}
|
|
155
|
+
finally:
|
|
156
|
+
conn.close()
|
|
157
|
+
current = {
|
|
158
|
+
path.name: hashlib.sha256(path.read_bytes()).digest()
|
|
159
|
+
for path in field.iter_pages(directory)
|
|
160
|
+
}
|
|
161
|
+
return None if stored == current else "index is behind its pages; run: sessionmemory reindex"
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def stale_indexes(vault: Path, embedder: Embedder) -> list[Finding]:
|
|
165
|
+
"""Report a field whose index file is unreadable or behind its pages."""
|
|
166
|
+
findings = []
|
|
167
|
+
for directory in _fields(vault):
|
|
168
|
+
message = _is_stale(directory, embedder)
|
|
169
|
+
if message:
|
|
170
|
+
findings.append(Finding("index", str(directory), message))
|
|
171
|
+
return findings
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
CHECKS: tuple[Callable[[Path, Embedder], list[Finding]], ...] = (
|
|
175
|
+
nonconformant_names,
|
|
176
|
+
oversized_pages,
|
|
177
|
+
malformed_frontmatter,
|
|
178
|
+
unquoted_datetimes,
|
|
179
|
+
dead_projects,
|
|
180
|
+
stale_indexes,
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def run(vault: Path, embedder: Embedder) -> list[Finding]:
|
|
185
|
+
"""Run every check and return what they found."""
|
|
186
|
+
return [finding for check in CHECKS for finding in check(vault, embedder)]
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""Turn note text into vectors, in-process and without a daemon.
|
|
2
|
+
|
|
3
|
+
`fastembed` runs `nomic-embed-text-v1.5` on ONNX Runtime inside this process. The
|
|
4
|
+
alternative considered was Ollama, and it was rejected for one reason: it is a service,
|
|
5
|
+
and a service is something every note write would depend on being up. A model file on
|
|
6
|
+
disk cannot be down.
|
|
7
|
+
|
|
8
|
+
The model cache is pinned rather than left at fastembed's default, which lives under the
|
|
9
|
+
system temp directory. macOS purges that periodically, and a silent re-download of 520MB
|
|
10
|
+
in the middle of `sessionmemory new` is indistinguishable from a hang.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import hashlib
|
|
16
|
+
import math
|
|
17
|
+
import os
|
|
18
|
+
import random
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import TYPE_CHECKING, Protocol
|
|
21
|
+
|
|
22
|
+
if TYPE_CHECKING:
|
|
23
|
+
from collections.abc import Sequence
|
|
24
|
+
|
|
25
|
+
from fastembed import TextEmbedding
|
|
26
|
+
|
|
27
|
+
MODEL_CODE = "nomic-embed-text-v1.5"
|
|
28
|
+
MODEL_NAME = "nomic-ai/nomic-embed-text-v1.5"
|
|
29
|
+
DIM = 768
|
|
30
|
+
|
|
31
|
+
# nomic is trained with task prefixes and fastembed 0.8 does not add them, so they are
|
|
32
|
+
# added here, which the memoryfield spec permits for a model that mandates them.
|
|
33
|
+
DOCUMENT_PREFIX = "search_document: "
|
|
34
|
+
QUERY_PREFIX = "search_query: "
|
|
35
|
+
|
|
36
|
+
CACHE_ENV_VAR = "SESSIONMEMORY_MODEL_CACHE"
|
|
37
|
+
DEFAULT_CACHE = Path.home() / ".cache" / "sessionmemory" / "models"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class Embedder(Protocol):
|
|
41
|
+
"""Anything that turns pages and queries into vectors of `dim` floats.
|
|
42
|
+
|
|
43
|
+
`name` is the model code the index file is named for, so two embedders with
|
|
44
|
+
different names never share an index.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
name: str
|
|
48
|
+
dim: int
|
|
49
|
+
|
|
50
|
+
def encode_documents(self, texts: Sequence[str]) -> list[list[float]]:
|
|
51
|
+
"""Return one vector per page text, in the order given."""
|
|
52
|
+
... # pragma: no cover
|
|
53
|
+
|
|
54
|
+
def encode_query(self, text: str) -> list[float]:
|
|
55
|
+
"""Return the vector for one search query."""
|
|
56
|
+
... # pragma: no cover
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class FastEmbedder:
|
|
60
|
+
"""Embeds with the real model, loaded on first use rather than at construction."""
|
|
61
|
+
|
|
62
|
+
def __init__(self, cache_dir: Path = DEFAULT_CACHE) -> None:
|
|
63
|
+
self.name = MODEL_CODE
|
|
64
|
+
self.dim = DIM
|
|
65
|
+
self.cache_dir = cache_dir
|
|
66
|
+
self._model: TextEmbedding | None = None
|
|
67
|
+
|
|
68
|
+
def _load(self) -> TextEmbedding:
|
|
69
|
+
if self._model is None:
|
|
70
|
+
from fastembed import TextEmbedding
|
|
71
|
+
|
|
72
|
+
self.cache_dir.mkdir(parents=True, exist_ok=True)
|
|
73
|
+
self._model = TextEmbedding(model_name=MODEL_NAME, cache_dir=str(self.cache_dir))
|
|
74
|
+
return self._model
|
|
75
|
+
|
|
76
|
+
def _encode(self, texts: list[str]) -> list[list[float]]:
|
|
77
|
+
if not texts:
|
|
78
|
+
return []
|
|
79
|
+
return [vector.tolist() for vector in self._load().embed(texts)]
|
|
80
|
+
|
|
81
|
+
def encode_documents(self, texts: Sequence[str]) -> list[list[float]]:
|
|
82
|
+
"""Return one vector per page text, in the order given."""
|
|
83
|
+
return self._encode([f"{DOCUMENT_PREFIX}{text}" for text in texts])
|
|
84
|
+
|
|
85
|
+
def encode_query(self, text: str) -> list[float]:
|
|
86
|
+
"""Return the vector for one search query."""
|
|
87
|
+
return self._encode([f"{QUERY_PREFIX}{text}"])[0]
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class StubEmbedder:
|
|
91
|
+
"""Embeds deterministically from a hash, with no model and no network."""
|
|
92
|
+
|
|
93
|
+
def __init__(self) -> None:
|
|
94
|
+
self.name = "stub"
|
|
95
|
+
self.dim = DIM
|
|
96
|
+
|
|
97
|
+
def _one(self, text: str) -> list[float]:
|
|
98
|
+
seed = int.from_bytes(hashlib.sha256(text.encode("utf-8")).digest()[:8], "big")
|
|
99
|
+
rng = random.Random(seed)
|
|
100
|
+
raw = [rng.gauss(0.0, 1.0) for _ in range(self.dim)]
|
|
101
|
+
norm = math.sqrt(sum(value * value for value in raw)) or 1.0
|
|
102
|
+
return [value / norm for value in raw]
|
|
103
|
+
|
|
104
|
+
def encode_documents(self, texts: Sequence[str]) -> list[list[float]]:
|
|
105
|
+
"""Return one vector per page text, in the order given."""
|
|
106
|
+
return [self._one(text) for text in texts]
|
|
107
|
+
|
|
108
|
+
def encode_query(self, text: str) -> list[float]:
|
|
109
|
+
"""Return the vector for one search query."""
|
|
110
|
+
return self._one(text)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def default_embedder() -> FastEmbedder:
|
|
114
|
+
"""Build the embedder commands use, reading the model cache location from the environment.
|
|
115
|
+
|
|
116
|
+
Returns:
|
|
117
|
+
FastEmbedder: An embedder caching to `SESSIONMEMORY_MODEL_CACHE`, or
|
|
118
|
+
`DEFAULT_CACHE` when that variable is unset or empty.
|
|
119
|
+
"""
|
|
120
|
+
raw = os.environ.get(CACHE_ENV_VAR)
|
|
121
|
+
if not raw:
|
|
122
|
+
return FastEmbedder(cache_dir=DEFAULT_CACHE)
|
|
123
|
+
return FastEmbedder(cache_dir=Path(raw).expanduser())
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""Write one field as a `.memoryfield.zip`, the spec's archival transport."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import zipfile
|
|
6
|
+
from typing import TYPE_CHECKING
|
|
7
|
+
|
|
8
|
+
from sessionmemory.lib import field, fieldindex
|
|
9
|
+
|
|
10
|
+
if TYPE_CHECKING:
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from sessionmemory.lib.embed import Embedder
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def export_field(directory: Path, embedder: Embedder, output: Path) -> Path:
|
|
17
|
+
"""Refresh the index, then archive every page and the index flat at the zip root."""
|
|
18
|
+
fieldindex.refresh(directory, embedder)
|
|
19
|
+
with zipfile.ZipFile(output, "w", compression=zipfile.ZIP_DEFLATED) as archive:
|
|
20
|
+
for path in field.iter_pages(directory):
|
|
21
|
+
archive.write(path, arcname=path.name)
|
|
22
|
+
index = fieldindex.index_path(directory, embedder)
|
|
23
|
+
if index.is_file():
|
|
24
|
+
archive.write(index, arcname=index.name)
|
|
25
|
+
return output
|