memory-boost 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- memory_boost/__init__.py +3 -0
- memory_boost/cli.py +389 -0
- memory_boost/core.py +681 -0
- memory_boost/drift.py +145 -0
- memory_boost/example_transcripts/-work-acme-api/0a1b2c3d-1111-4aaa-8bbb-000000000001.jsonl +136 -0
- memory_boost/example_transcripts/-work-acme-api/0a1b2c3d-3333-4aaa-8bbb-000000000003.jsonl +46 -0
- memory_boost/example_transcripts/-work-acme-api/0a1b2c3d-4444-4aaa-8bbb-000000000004.jsonl +10 -0
- memory_boost/example_transcripts/-work-pixel-notes/0a1b2c3d-2222-4aaa-8bbb-000000000002.jsonl +54 -0
- memory_boost/example_wiki/concepts/money-as-integer-cents.md +15 -0
- memory_boost/example_wiki/concepts/retry-jobs-idempotently.md +16 -0
- memory_boost/example_wiki/context/preferences.md +7 -0
- memory_boost/example_wiki/decisions/cors-allow-all.md +9 -0
- memory_boost/example_wiki/decisions/float-money.md +9 -0
- memory_boost/example_wiki/decisions/pin-postgres-15.md +14 -0
- memory_boost/example_wiki/decisions/queue-in-postgres.md +15 -0
- memory_boost/example_wiki/decisions/sessions-in-redis.md +9 -0
- memory_boost/example_wiki/log.md +37 -0
- memory_boost/example_wiki/projects/acme-api.md +24 -0
- memory_boost/example_wiki/projects/legacy-dashboard.md +13 -0
- memory_boost/example_wiki/projects/pixel-notes.md +9 -0
- memory_boost/lessons.py +77 -0
- memory_boost/mine.py +172 -0
- memory_boost/server.py +102 -0
- memory_boost-0.1.0.dist-info/METADATA +229 -0
- memory_boost-0.1.0.dist-info/RECORD +28 -0
- memory_boost-0.1.0.dist-info/WHEEL +4 -0
- memory_boost-0.1.0.dist-info/entry_points.txt +2 -0
- memory_boost-0.1.0.dist-info/licenses/LICENSE +21 -0
memory_boost/core.py
ADDED
|
@@ -0,0 +1,681 @@
|
|
|
1
|
+
"""Core logic: FTS5 index over a markdown wiki, freshness ("vigencia") rules,
|
|
2
|
+
the cheap session brief, event/log writes and session checkpoints.
|
|
3
|
+
|
|
4
|
+
Shared by the CLI (hooks) and the MCP server so that "cheap recall at session
|
|
5
|
+
start" and "recall on demand" never drift apart. Stdlib only.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import datetime as dt
|
|
10
|
+
import hashlib
|
|
11
|
+
import json
|
|
12
|
+
import os
|
|
13
|
+
import re
|
|
14
|
+
import sqlite3
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
# --- locations -------------------------------------------------------------
|
|
18
|
+
#
|
|
19
|
+
# Everything lives under MEMORY_BOOST_HOME (default $XDG_DATA_HOME/memory-boost):
|
|
20
|
+
# wiki/ compiled pages (projects/, decisions/, concepts/, context/, log.md)
|
|
21
|
+
# events/ raw append-only JSONL, one dir per agent
|
|
22
|
+
# checkpoints/ per-session working state (JSON), survives context compaction
|
|
23
|
+
# aliases.json optional {"dir-name": "project-slug"} map
|
|
24
|
+
# The FTS index is a disposable cache under $XDG_CACHE_HOME, keyed by wiki path.
|
|
25
|
+
# Paths are resolved on every call so env changes (tests, several homes) apply.
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def plural(n: int, word: str, many: str | None = None) -> str:
|
|
29
|
+
"""'1 finding', '2 findings' — reports are read by people too."""
|
|
30
|
+
return f"{n} {word}" if n == 1 else f"{n} {many or word + 's'}"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def home() -> Path:
|
|
34
|
+
if env := os.environ.get("MEMORY_BOOST_HOME"):
|
|
35
|
+
return Path(env).expanduser()
|
|
36
|
+
base = os.environ.get("XDG_DATA_HOME") or Path.home() / ".local" / "share"
|
|
37
|
+
return Path(base) / "memory-boost"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def wiki_dir() -> Path:
|
|
41
|
+
env = os.environ.get("MEMORY_BOOST_WIKI")
|
|
42
|
+
return Path(env).expanduser() if env else home() / "wiki"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def events_dir() -> Path:
|
|
46
|
+
return home() / "events"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def checkpoints_dir() -> Path:
|
|
50
|
+
env = os.environ.get("MEMORY_BOOST_CHECKPOINTS")
|
|
51
|
+
return Path(env).expanduser() if env else home() / "checkpoints"
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def index_path() -> Path:
|
|
55
|
+
if env := os.environ.get("MEMORY_BOOST_INDEX"):
|
|
56
|
+
return Path(env).expanduser()
|
|
57
|
+
base = os.environ.get("XDG_CACHE_HOME") or Path.home() / ".cache"
|
|
58
|
+
key = hashlib.sha256(str(wiki_dir().resolve()).encode()).hexdigest()[:12]
|
|
59
|
+
return Path(base) / "memory-boost" / f"index-{key}.db"
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def aliases_path() -> Path:
|
|
63
|
+
return home() / "aliases.json"
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
# --- parsing ---------------------------------------------------------------
|
|
67
|
+
|
|
68
|
+
LOG_ENTRY_RE = re.compile(r"^## \[(\d{4}-\d{2}-\d{2})([a-z])?\]\s*(\S+)\s*\|\s*([^|]+?)\s*\|\s*(.+)$")
|
|
69
|
+
LOG_HEADER_LOOSE_RE = re.compile(r"^## \[")
|
|
70
|
+
FRONTMATTER_RE = re.compile(r"^---\n(.*?)\n---\n", re.DOTALL)
|
|
71
|
+
H2_RE = re.compile(r"^##\s+(.+?)\s*$")
|
|
72
|
+
# Markers authors write into prose to flag a section as no longer current.
|
|
73
|
+
SUPERSEDED_RE = re.compile(
|
|
74
|
+
r"\bsuperseded\b|\bobsolete\b|\bdeprecated\b|\bno longer applies\b|\boutdated\b"
|
|
75
|
+
r"|desactualizad|obsolet|superad[oa]|ya no aplica|deprecad", re.I)
|
|
76
|
+
HISTORY_SUFFIXES = ("-history", "-historial")
|
|
77
|
+
|
|
78
|
+
FRESHNESS = ("current", "unclassified", "review", "historic", "superseded")
|
|
79
|
+
FRESHNESS_ORDER = {v: i for i, v in enumerate(FRESHNESS)}
|
|
80
|
+
# Frontmatter `status:` values accepted as aliases (wikis written in Spanish).
|
|
81
|
+
STATUS_ALIASES = {"vigente": "current", "revisar": "review", "historico": "historic",
|
|
82
|
+
"histórico": "historic"}
|
|
83
|
+
|
|
84
|
+
# How many of a project's most recent log entries count as "current": used
|
|
85
|
+
# both to classify log chunks and to cap the recent activity in brief().
|
|
86
|
+
CURRENT_LOG_ENTRIES_PER_PROJECT = 3
|
|
87
|
+
|
|
88
|
+
SLUG_RE = re.compile(r"^[a-z0-9][a-z0-9-]{0,79}$")
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def slugify(name: str) -> str:
|
|
92
|
+
return re.sub(r"[^a-z0-9]+", "-", name.lower()).strip("-")
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def norm_project(name: str) -> str:
|
|
96
|
+
"""my_api / my-api / MY_API -> my-api"""
|
|
97
|
+
return slugify(name.replace("_", "-"))
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def safe_slug(value: str, field: str) -> str:
|
|
101
|
+
"""Normalise a user/LLM-supplied name used in a path; reject anything that
|
|
102
|
+
does not survive as a plain slug (no separators, no traversal)."""
|
|
103
|
+
s = norm_project(value)
|
|
104
|
+
if not SLUG_RE.match(s) or any(sep in value for sep in ("/", "\\", "..")):
|
|
105
|
+
raise ValueError(f"invalid {field}: {value!r} (use letters, digits, '-' or '_')")
|
|
106
|
+
return s
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _one_line(text: str) -> str:
|
|
110
|
+
return " ".join(text.split())
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _is_history(stem: str) -> bool:
|
|
114
|
+
return stem.endswith(HISTORY_SUFFIXES)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _today() -> str:
|
|
118
|
+
"""UTC date; MEMORY_BOOST_TODAY=YYYY-MM-DD pins it for reproducible reports and tests."""
|
|
119
|
+
return os.environ.get("MEMORY_BOOST_TODAY") or dt.datetime.now(dt.UTC).strftime("%Y-%m-%d")
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _now_iso() -> str:
|
|
123
|
+
return dt.datetime.now(dt.UTC).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def load_aliases() -> dict:
|
|
127
|
+
p = aliases_path()
|
|
128
|
+
if not p.exists():
|
|
129
|
+
return {}
|
|
130
|
+
try:
|
|
131
|
+
data = json.loads(p.read_text(encoding="utf-8"))
|
|
132
|
+
except json.JSONDecodeError as e:
|
|
133
|
+
raise ValueError(f"{p}: invalid JSON ({e})") from e
|
|
134
|
+
if not isinstance(data, dict):
|
|
135
|
+
raise ValueError(f"{p}: expected an object {{\"dir-name\": \"project\"}}")
|
|
136
|
+
return {norm_project(k): norm_project(v) for k, v in data.items()}
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def resolve_project(cwd_or_name: str) -> str | None:
|
|
140
|
+
"""basename(cwd) or a project name -> slug that has a page in wiki/projects/."""
|
|
141
|
+
base = norm_project(Path(cwd_or_name).name)
|
|
142
|
+
base = load_aliases().get(base, base)
|
|
143
|
+
if base and not _is_history(base) and (wiki_dir() / "projects" / f"{base}.md").exists():
|
|
144
|
+
return base
|
|
145
|
+
return None
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def parse_frontmatter(text: str) -> tuple[dict, str]:
|
|
149
|
+
m = FRONTMATTER_RE.match(text)
|
|
150
|
+
if not m:
|
|
151
|
+
return {}, text
|
|
152
|
+
meta = {}
|
|
153
|
+
for line in m.group(1).splitlines():
|
|
154
|
+
if ":" in line:
|
|
155
|
+
k, v = line.split(":", 1)
|
|
156
|
+
meta[k.strip()] = v.strip().strip('"')
|
|
157
|
+
return meta, text[m.end():]
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _status(meta: dict) -> str | None:
|
|
161
|
+
s = (meta.get("status") or "").lower()
|
|
162
|
+
s = STATUS_ALIASES.get(s, s)
|
|
163
|
+
return s if s in FRESHNESS else None
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _freshness(kind: str, title: str | None, body: str, meta: dict, is_history: bool) -> str:
|
|
167
|
+
if is_history:
|
|
168
|
+
return "historic"
|
|
169
|
+
if SUPERSEDED_RE.search(body):
|
|
170
|
+
return "superseded"
|
|
171
|
+
t = (title or "").lower()
|
|
172
|
+
if "history" in t or "historial" in t:
|
|
173
|
+
return "historic"
|
|
174
|
+
if kind == "decision":
|
|
175
|
+
if status := _status(meta):
|
|
176
|
+
return status
|
|
177
|
+
review_by = meta.get("review_by")
|
|
178
|
+
if review_by and review_by < _today():
|
|
179
|
+
return "review"
|
|
180
|
+
return "current"
|
|
181
|
+
return "unclassified"
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def chunk_page(path: Path, kind: str) -> tuple[list[dict], dict]:
|
|
185
|
+
"""Split a page into ##-level chunks. The chunk before the first ## uses anchor 'top'."""
|
|
186
|
+
meta, body = parse_frontmatter(path.read_text(encoding="utf-8"))
|
|
187
|
+
is_history = _is_history(path.stem)
|
|
188
|
+
project = None
|
|
189
|
+
if kind == "project":
|
|
190
|
+
project = path.stem
|
|
191
|
+
for suf in HISTORY_SUFFIXES:
|
|
192
|
+
project = project.removesuffix(suf)
|
|
193
|
+
|
|
194
|
+
sections: list[tuple[str | None, list[str]]] = []
|
|
195
|
+
cur_title: str | None = None
|
|
196
|
+
cur_lines: list[str] = []
|
|
197
|
+
for line in body.splitlines():
|
|
198
|
+
if hm := H2_RE.match(line):
|
|
199
|
+
sections.append((cur_title, cur_lines))
|
|
200
|
+
cur_title, cur_lines = hm.group(1), []
|
|
201
|
+
else:
|
|
202
|
+
cur_lines.append(line)
|
|
203
|
+
sections.append((cur_title, cur_lines))
|
|
204
|
+
|
|
205
|
+
rel = path.relative_to(wiki_dir()).as_posix()
|
|
206
|
+
chunks = []
|
|
207
|
+
for title, lines in sections:
|
|
208
|
+
content = "\n".join(lines).strip()
|
|
209
|
+
if not content and title is None:
|
|
210
|
+
continue
|
|
211
|
+
chunks.append({
|
|
212
|
+
"ref": f"{rel}#{slugify(title) if title else 'top'}",
|
|
213
|
+
"path": rel,
|
|
214
|
+
"kind": kind,
|
|
215
|
+
"project": project,
|
|
216
|
+
"title": title or meta.get("title", path.stem),
|
|
217
|
+
"date": meta.get("updated") or meta.get("created"),
|
|
218
|
+
"freshness": _freshness(kind, title, content, meta, is_history),
|
|
219
|
+
"body": content,
|
|
220
|
+
})
|
|
221
|
+
return chunks, meta
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def parse_log(path: Path) -> list[dict]:
|
|
225
|
+
"""Parse log.md entries. Headers that look like entries but do not match the
|
|
226
|
+
format are kept and flagged (malformed) instead of being silently merged."""
|
|
227
|
+
if not path.exists():
|
|
228
|
+
return []
|
|
229
|
+
entries: list[dict] = []
|
|
230
|
+
cur = None
|
|
231
|
+
for n, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1):
|
|
232
|
+
if m := LOG_ENTRY_RE.match(line):
|
|
233
|
+
if cur:
|
|
234
|
+
entries.append(cur)
|
|
235
|
+
date, _suffix, agent, project, action = m.groups()
|
|
236
|
+
cur = {"date": date, "agent": agent, "project": norm_project(project),
|
|
237
|
+
"project_raw": project, "action": action.strip(), "bullets": [],
|
|
238
|
+
"malformed": False}
|
|
239
|
+
elif LOG_HEADER_LOOSE_RE.match(line):
|
|
240
|
+
if cur:
|
|
241
|
+
entries.append(cur)
|
|
242
|
+
cur = {"date": None, "agent": None, "project": "__malformed__", "project_raw": line,
|
|
243
|
+
"action": line.strip(), "bullets": [], "malformed": True, "line_no": n}
|
|
244
|
+
elif cur is not None and line.strip():
|
|
245
|
+
cur["bullets"].append(line.strip())
|
|
246
|
+
if cur:
|
|
247
|
+
entries.append(cur)
|
|
248
|
+
entries.sort(key=lambda e: e["date"] or "")
|
|
249
|
+
return entries
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def log_problems(entries: list[dict]) -> list[str]:
|
|
253
|
+
return [f"log.md:L{e['line_no']} unparseable header: {e['project_raw']}"
|
|
254
|
+
for e in entries if e.get("malformed")]
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def _log_chunks(path: Path) -> list[dict]:
|
|
258
|
+
entries = [e for e in parse_log(path) if not e.get("malformed")]
|
|
259
|
+
per_project: dict[str, list[int]] = {}
|
|
260
|
+
for i, e in enumerate(entries):
|
|
261
|
+
per_project.setdefault(e["project"], []).append(i)
|
|
262
|
+
current = {i for idxs in per_project.values() for i in idxs[-CURRENT_LOG_ENTRIES_PER_PROJECT:]}
|
|
263
|
+
|
|
264
|
+
rel = path.relative_to(wiki_dir()).as_posix()
|
|
265
|
+
return [{
|
|
266
|
+
"ref": f"{rel}#{e['date']}-{e['project']}-{i}",
|
|
267
|
+
"path": rel,
|
|
268
|
+
"kind": "log",
|
|
269
|
+
"project": e["project"],
|
|
270
|
+
"title": f"{e['date']} {e['project_raw']}: {e['action'][:60]}",
|
|
271
|
+
"date": e["date"],
|
|
272
|
+
"freshness": "current" if i in current else "historic",
|
|
273
|
+
"body": e["action"] + ("\n" + "\n".join(e["bullets"]) if e["bullets"] else ""),
|
|
274
|
+
} for i, e in enumerate(entries)]
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
# --- index -----------------------------------------------------------------
|
|
278
|
+
|
|
279
|
+
SCHEMA = """
|
|
280
|
+
CREATE TABLE IF NOT EXISTS files(path TEXT PRIMARY KEY, mtime REAL, content_hash TEXT);
|
|
281
|
+
CREATE TABLE IF NOT EXISTS docs(
|
|
282
|
+
id INTEGER PRIMARY KEY, ref TEXT UNIQUE, path TEXT, kind TEXT, project TEXT,
|
|
283
|
+
title TEXT, date TEXT, freshness TEXT, body TEXT
|
|
284
|
+
);
|
|
285
|
+
CREATE VIRTUAL TABLE IF NOT EXISTS docs_fts USING fts5(
|
|
286
|
+
title, body, content='docs', content_rowid='id',
|
|
287
|
+
tokenize="unicode61 remove_diacritics 2"
|
|
288
|
+
);
|
|
289
|
+
CREATE TRIGGER IF NOT EXISTS docs_ai AFTER INSERT ON docs BEGIN
|
|
290
|
+
INSERT INTO docs_fts(rowid, title, body) VALUES (new.id, new.title, new.body);
|
|
291
|
+
END;
|
|
292
|
+
CREATE TRIGGER IF NOT EXISTS docs_ad AFTER DELETE ON docs BEGIN
|
|
293
|
+
INSERT INTO docs_fts(docs_fts, rowid, title, body) VALUES('delete', old.id, old.title, old.body);
|
|
294
|
+
END;
|
|
295
|
+
CREATE TRIGGER IF NOT EXISTS docs_au AFTER UPDATE ON docs BEGIN
|
|
296
|
+
INSERT INTO docs_fts(docs_fts, rowid, title, body) VALUES('delete', old.id, old.title, old.body);
|
|
297
|
+
INSERT INTO docs_fts(rowid, title, body) VALUES (new.id, new.title, new.body);
|
|
298
|
+
END;
|
|
299
|
+
"""
|
|
300
|
+
|
|
301
|
+
PAGE_DIRS = (("project", "projects"), ("decision", "decisions"),
|
|
302
|
+
("concept", "concepts"), ("context", "context"))
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def _connect() -> sqlite3.Connection:
|
|
306
|
+
p = index_path()
|
|
307
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
308
|
+
conn = sqlite3.connect(p)
|
|
309
|
+
conn.executescript(SCHEMA)
|
|
310
|
+
return conn
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def _wiki_files() -> list[tuple[Path, str]]:
|
|
314
|
+
w = wiki_dir()
|
|
315
|
+
out = [(p, kind) for kind, sub in PAGE_DIRS for p in sorted((w / sub).glob("*.md"))]
|
|
316
|
+
out += [(w / n, "__log__") for n in ("log.md", "log-archive.md") if (w / n).exists()]
|
|
317
|
+
return out
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
def refresh_index(force: bool = False) -> dict:
|
|
321
|
+
"""Incremental: re-chunk only files whose mtime and content hash changed;
|
|
322
|
+
drop rows of deleted files. Cheap enough to run before every recall."""
|
|
323
|
+
conn = _connect()
|
|
324
|
+
stats = {"indexed": 0, "skipped": 0, "chunks": 0}
|
|
325
|
+
warnings: list[str] = []
|
|
326
|
+
seen: set[str] = set()
|
|
327
|
+
today = _today()
|
|
328
|
+
|
|
329
|
+
for path, kind in _wiki_files():
|
|
330
|
+
rel = path.relative_to(wiki_dir()).as_posix()
|
|
331
|
+
seen.add(rel)
|
|
332
|
+
mtime = path.stat().st_mtime
|
|
333
|
+
row = conn.execute("SELECT mtime, content_hash FROM files WHERE path=?", (rel,)).fetchone()
|
|
334
|
+
if not force and row and row[0] == mtime:
|
|
335
|
+
stats["skipped"] += 1
|
|
336
|
+
continue
|
|
337
|
+
h = hashlib.sha256(path.read_bytes()).hexdigest()
|
|
338
|
+
if not force and row and row[1] == h:
|
|
339
|
+
conn.execute("UPDATE files SET mtime=? WHERE path=?", (mtime, rel))
|
|
340
|
+
stats["skipped"] += 1
|
|
341
|
+
continue
|
|
342
|
+
conn.execute("DELETE FROM docs WHERE path=?", (rel,))
|
|
343
|
+
if kind == "__log__":
|
|
344
|
+
chunks = _log_chunks(path)
|
|
345
|
+
warnings += log_problems(parse_log(path))
|
|
346
|
+
else:
|
|
347
|
+
chunks, meta = chunk_page(path, kind)
|
|
348
|
+
if kind == "decision" and not _status(meta):
|
|
349
|
+
review_by = meta.get("review_by")
|
|
350
|
+
if not review_by:
|
|
351
|
+
warnings.append(f"decision without review_by: {path.name}")
|
|
352
|
+
elif review_by < today:
|
|
353
|
+
warnings.append(f"decision past review_by={review_by}: {path.name}")
|
|
354
|
+
for c in chunks:
|
|
355
|
+
conn.execute(
|
|
356
|
+
"INSERT OR REPLACE INTO docs(ref,path,kind,project,title,date,freshness,body) "
|
|
357
|
+
"VALUES (:ref,:path,:kind,:project,:title,:date,:freshness,:body)", c)
|
|
358
|
+
conn.execute("INSERT OR REPLACE INTO files(path,mtime,content_hash) VALUES (?,?,?)",
|
|
359
|
+
(rel, mtime, h))
|
|
360
|
+
stats["indexed"] += 1
|
|
361
|
+
stats["chunks"] += len(chunks)
|
|
362
|
+
|
|
363
|
+
for (p,) in conn.execute("SELECT path FROM files").fetchall():
|
|
364
|
+
if p not in seen:
|
|
365
|
+
conn.execute("DELETE FROM docs WHERE path=?", (p,))
|
|
366
|
+
conn.execute("DELETE FROM files WHERE path=?", (p,))
|
|
367
|
+
conn.commit()
|
|
368
|
+
conn.close()
|
|
369
|
+
stats["warnings"] = warnings
|
|
370
|
+
return stats
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def _fts_query(query: str) -> str:
|
|
374
|
+
terms = re.findall(r"[\w-]+", query.lower())
|
|
375
|
+
return " OR ".join(f'"{t}"' for t in terms) if terms else '""'
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
def recall(query: str, project: str | None = None, limit: int = 8,
|
|
379
|
+
include_historic: bool = False) -> list[dict]:
|
|
380
|
+
"""Full-text search over pages and log entries, ranked by bm25 and then
|
|
381
|
+
re-ordered so current knowledge comes before historic/superseded."""
|
|
382
|
+
refresh_index()
|
|
383
|
+
sql = ("SELECT d.ref, d.path, d.kind, d.project, d.title, d.date, d.freshness, "
|
|
384
|
+
"snippet(docs_fts, 1, '', '', '…', 12) "
|
|
385
|
+
"FROM docs_fts JOIN docs d ON d.id = docs_fts.rowid WHERE docs_fts MATCH ?")
|
|
386
|
+
params: list = [_fts_query(query)]
|
|
387
|
+
if project:
|
|
388
|
+
sql += " AND d.project = ?"
|
|
389
|
+
params.append(norm_project(project))
|
|
390
|
+
if not include_historic:
|
|
391
|
+
sql += " AND d.freshness != 'historic'"
|
|
392
|
+
sql += " ORDER BY bm25(docs_fts) LIMIT ?"
|
|
393
|
+
params.append(limit * 3)
|
|
394
|
+
conn = _connect()
|
|
395
|
+
try:
|
|
396
|
+
rows = conn.execute(sql, params).fetchall()
|
|
397
|
+
finally:
|
|
398
|
+
conn.close()
|
|
399
|
+
rows.sort(key=lambda r: FRESHNESS_ORDER.get(r[6], 1))
|
|
400
|
+
return [{"ref": ref, "path": path, "kind": kind, "project": proj, "title": title,
|
|
401
|
+
"date": date, "freshness": fr, "extract": snippet}
|
|
402
|
+
for ref, path, kind, proj, title, date, fr, snippet in rows[:limit]]
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
# --- brief (the SessionStart payload) ----------------------------------------
|
|
406
|
+
|
|
407
|
+
def _catalog() -> str:
|
|
408
|
+
lines = []
|
|
409
|
+
for sub, label in (("projects", "Projects"), ("decisions", "Decisions")):
|
|
410
|
+
names = sorted(p.stem for p in (wiki_dir() / sub).glob("*.md") if not _is_history(p.stem))
|
|
411
|
+
if names:
|
|
412
|
+
lines.append(f"{label}: " + ", ".join(names))
|
|
413
|
+
return "\n".join(lines)
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def truncate(text: str, cap: int, ref_hint: str) -> str:
|
|
417
|
+
if len(text) <= cap:
|
|
418
|
+
return text
|
|
419
|
+
out, total = [], 0
|
|
420
|
+
for line in text.splitlines():
|
|
421
|
+
if total + len(line) > cap:
|
|
422
|
+
break
|
|
423
|
+
out.append(line)
|
|
424
|
+
total += len(line)
|
|
425
|
+
out.append(f"…(truncated, see {ref_hint})")
|
|
426
|
+
return "\n".join(out)
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
def brief(cwd_or_project: str | None, budget: int = 1200) -> dict:
|
|
430
|
+
"""Cheap, cwd-aware orientation (~budget chars): catalog + the project's
|
|
431
|
+
stable context + its latest log entries. Used by the hook and the MCP tool."""
|
|
432
|
+
project = resolve_project(cwd_or_project) if cwd_or_project else None
|
|
433
|
+
w = wiki_dir()
|
|
434
|
+
blocks = [
|
|
435
|
+
"## Memory\n"
|
|
436
|
+
"Tools: memory_recall(query), memory_page(name, section), memory_save(...), "
|
|
437
|
+
"memory_checkpoint(...), memory_lesson(...).",
|
|
438
|
+
"### Index\n" + truncate(_catalog() or "(empty wiki)", budget * 7 // 12, "memory_page"),
|
|
439
|
+
]
|
|
440
|
+
entries = parse_log(w / "log.md")
|
|
441
|
+
probs = log_problems(entries)
|
|
442
|
+
valid = [e for e in entries if not e.get("malformed")]
|
|
443
|
+
|
|
444
|
+
if project:
|
|
445
|
+
chunks, _ = chunk_page(w / "projects" / f"{project}.md", "project")
|
|
446
|
+
top = next((c for c in chunks if c["ref"].endswith("#top")), None)
|
|
447
|
+
if top and top["body"]:
|
|
448
|
+
blocks.append(f"### {project} — context\n" + truncate(top["body"], budget * 3 // 12, top["ref"]))
|
|
449
|
+
recent = [e for e in valid if e["project"] == project][-CURRENT_LOG_ENTRIES_PER_PROJECT:]
|
|
450
|
+
if recent:
|
|
451
|
+
lines = []
|
|
452
|
+
for i, e in enumerate(reversed(recent)):
|
|
453
|
+
lines.append(f"- [{e['date']}] {e['action']}")
|
|
454
|
+
lines += [f" - {b.lstrip('- ')}" for b in e["bullets"]
|
|
455
|
+
if i == 0 or b.lstrip("- ").lower().startswith(("todo", "pending", "pendiente"))]
|
|
456
|
+
blocks.append(f"### {project} — latest sessions\n"
|
|
457
|
+
+ truncate("\n".join(lines), budget * 6 // 12, "wiki/log.md"))
|
|
458
|
+
else:
|
|
459
|
+
blocks.append(f"### {project} — latest sessions\n(no log entries yet)")
|
|
460
|
+
else:
|
|
461
|
+
recent = valid[-CURRENT_LOG_ENTRIES_PER_PROJECT:]
|
|
462
|
+
if recent:
|
|
463
|
+
blocks.append("### Recent activity (all projects)\n" + "\n".join(
|
|
464
|
+
f"- [{e['date']}] {e['project_raw']}: {e['action']}" for e in reversed(recent)))
|
|
465
|
+
if cwd_or_project:
|
|
466
|
+
blocks.append(f"No project page matches '{Path(cwd_or_project).name}'. "
|
|
467
|
+
"memory_save(project=...) creates one.")
|
|
468
|
+
if project:
|
|
469
|
+
# Imported here: drift/lessons depend on core, and brief() is the only caller.
|
|
470
|
+
from memory_boost import drift, lessons
|
|
471
|
+
if lines := drift.for_project(project):
|
|
472
|
+
blocks.append(f"### {project} — drift\n" + "\n".join(lines))
|
|
473
|
+
if found := lessons.lessons_for(project):
|
|
474
|
+
blocks.append("### Lessons from other projects\n" + "\n".join(lessons.format_lessons(found)))
|
|
475
|
+
if probs:
|
|
476
|
+
blocks.append(f"### log.md — {len(probs)} unparseable headers\n" + "\n".join(probs))
|
|
477
|
+
return {"text": "\n\n".join(blocks), "project": project}
|
|
478
|
+
|
|
479
|
+
|
|
480
|
+
# --- writes ------------------------------------------------------------------
|
|
481
|
+
|
|
482
|
+
def save_event(agent: str, project: str, action: str, result: str = "ok",
|
|
483
|
+
detail: str | None = None) -> Path:
|
|
484
|
+
"""Append one raw JSONL event under events/<agent>/<day>.jsonl. Any agent
|
|
485
|
+
that can append a line (a remote bot, a CI job) can feed the memory."""
|
|
486
|
+
d = events_dir() / safe_slug(agent, "agent")
|
|
487
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
488
|
+
event = {"ts": _now_iso(), "agent": agent, "project": project, "action": action,
|
|
489
|
+
"result": result}
|
|
490
|
+
if detail:
|
|
491
|
+
event["detail"] = detail
|
|
492
|
+
path = d / f"{_today()}.jsonl"
|
|
493
|
+
with open(path, "a", encoding="utf-8") as f:
|
|
494
|
+
f.write(json.dumps(event, ensure_ascii=False) + "\n")
|
|
495
|
+
return path
|
|
496
|
+
|
|
497
|
+
|
|
498
|
+
def ensure_project_page(project: str) -> Path:
|
|
499
|
+
slug = safe_slug(project, "project")
|
|
500
|
+
p = wiki_dir() / "projects" / f"{slug}.md"
|
|
501
|
+
if not p.exists():
|
|
502
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
503
|
+
p.write_text(f"---\ntitle: {slug}\ncreated: {_today()}\n---\n# {slug}\n\n"
|
|
504
|
+
"(Stable context: what this project is, where it lives, how to run it.)\n",
|
|
505
|
+
encoding="utf-8")
|
|
506
|
+
return p
|
|
507
|
+
|
|
508
|
+
|
|
509
|
+
def append_log(project: str, agent: str, action: str, bullets: list[str] | None = None,
|
|
510
|
+
pending: str | None = None) -> None:
|
|
511
|
+
"""Append a compiled entry to wiki/log.md. Every field is collapsed to one
|
|
512
|
+
line so user/LLM text can never forge an extra '## [date]' header."""
|
|
513
|
+
agent, project = safe_slug(agent, "agent"), safe_slug(project, "project")
|
|
514
|
+
header = f"## [{_today()}] {agent} | {project} | {_one_line(action)}"
|
|
515
|
+
lines = [header] + [f"- {_one_line(b)}" for b in (bullets or []) if b.strip()]
|
|
516
|
+
if pending:
|
|
517
|
+
lines.append(f"- Pending: {_one_line(pending)}")
|
|
518
|
+
log = wiki_dir() / "log.md"
|
|
519
|
+
log.parent.mkdir(parents=True, exist_ok=True)
|
|
520
|
+
with open(log, "a", encoding="utf-8") as f:
|
|
521
|
+
f.write("\n" + "\n".join(lines) + "\n")
|
|
522
|
+
|
|
523
|
+
|
|
524
|
+
def save(project: str, action: str, detail: str | None = None, result: str = "ok",
|
|
525
|
+
agent: str = "mcp", pending: str | None = None) -> str:
|
|
526
|
+
if not action.strip():
|
|
527
|
+
raise ValueError("action is empty")
|
|
528
|
+
slug = safe_slug(project, "project")
|
|
529
|
+
ensure_project_page(slug)
|
|
530
|
+
save_event(agent, slug, action, result, detail)
|
|
531
|
+
append_log(slug, agent, action, bullets=[detail] if detail else None, pending=pending)
|
|
532
|
+
return slug
|
|
533
|
+
|
|
534
|
+
|
|
535
|
+
def page_path(name: str) -> tuple[str, Path] | None:
|
|
536
|
+
"""(kind, path) of the wiki page called `name`, or None if no section has it."""
|
|
537
|
+
slug = safe_slug(name, "page name")
|
|
538
|
+
for kind, sub in PAGE_DIRS:
|
|
539
|
+
p = wiki_dir() / sub / f"{slug}.md"
|
|
540
|
+
if p.exists():
|
|
541
|
+
return kind, p
|
|
542
|
+
return None
|
|
543
|
+
|
|
544
|
+
|
|
545
|
+
def get_page(name: str, section: str | None = None) -> str:
|
|
546
|
+
found = page_path(name)
|
|
547
|
+
if not found:
|
|
548
|
+
return f"(page '{safe_slug(name, 'page name')}' not found)"
|
|
549
|
+
kind, p = found
|
|
550
|
+
chunks, _ = chunk_page(p, kind)
|
|
551
|
+
if not section:
|
|
552
|
+
return "\n\n".join(f"## {c['title']}\n{c['body']}" for c in chunks)
|
|
553
|
+
for c in chunks:
|
|
554
|
+
if slugify(c["title"]) == slugify(section) or c["ref"].endswith(f"#{section}"):
|
|
555
|
+
return c["body"]
|
|
556
|
+
return (f"(section '{section}' not found in {p.stem}; available: "
|
|
557
|
+
f"{', '.join(c['title'] for c in chunks)})")
|
|
558
|
+
|
|
559
|
+
|
|
560
|
+
# --- checkpoints ---------------------------------------------------------------
|
|
561
|
+
#
|
|
562
|
+
# Live state of ONE working session, meant to survive context compaction and to
|
|
563
|
+
# let parallel sessions on the same project see each other. Deliberately cheap:
|
|
564
|
+
# one small JSON per session, no index involved. Ephemeral: at the end of the
|
|
565
|
+
# session it is distilled into log.md with save() and marked status=done.
|
|
566
|
+
|
|
567
|
+
CHECKPOINT_STALE_HOURS = 72
|
|
568
|
+
CHECKPOINT_DONE_CAP = 40
|
|
569
|
+
CHECKPOINT_STATUSES = ("active", "blocked", "done")
|
|
570
|
+
|
|
571
|
+
|
|
572
|
+
def _ckpt_path(project: str, session_id: str) -> Path:
|
|
573
|
+
sid = slugify(session_id)
|
|
574
|
+
if not sid:
|
|
575
|
+
raise ValueError(f"invalid session_id: {session_id!r}")
|
|
576
|
+
return checkpoints_dir() / safe_slug(project, "project") / f"{sid}.json"
|
|
577
|
+
|
|
578
|
+
|
|
579
|
+
def save_checkpoint(project: str, session_id: str, task: str, *, agent: str = "agent",
|
|
580
|
+
done: list[str] | None = None, next_step: str | None = None,
|
|
581
|
+
files: list[str] | None = None, notes: list[str] | None = None,
|
|
582
|
+
status: str = "active", plan_ref: str | None = None,
|
|
583
|
+
replace_done: bool = False) -> dict:
|
|
584
|
+
"""Merge-write a session checkpoint. `done` accumulates unless replace_done;
|
|
585
|
+
other fields overwrite when given."""
|
|
586
|
+
if status not in CHECKPOINT_STATUSES:
|
|
587
|
+
raise ValueError(f"invalid status: {status!r} ({'|'.join(CHECKPOINT_STATUSES)})")
|
|
588
|
+
if not task.strip():
|
|
589
|
+
raise ValueError("task is empty")
|
|
590
|
+
path = _ckpt_path(project, session_id)
|
|
591
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
592
|
+
prev: dict = {}
|
|
593
|
+
if path.exists():
|
|
594
|
+
try:
|
|
595
|
+
prev = json.loads(path.read_text(encoding="utf-8"))
|
|
596
|
+
except json.JSONDecodeError as e:
|
|
597
|
+
raise ValueError(f"corrupt checkpoint {path}: {e}") from e
|
|
598
|
+
merged_done = [] if replace_done else list(prev.get("done", []))
|
|
599
|
+
merged_done += [d for d in (done or []) if d and d not in merged_done]
|
|
600
|
+
ckpt = {
|
|
601
|
+
"project": norm_project(project),
|
|
602
|
+
"session_id": session_id,
|
|
603
|
+
"agent": agent,
|
|
604
|
+
"created": prev.get("created", _now_iso()),
|
|
605
|
+
"updated": _now_iso(),
|
|
606
|
+
"status": status,
|
|
607
|
+
"task": task.strip(),
|
|
608
|
+
"done": merged_done[-CHECKPOINT_DONE_CAP:],
|
|
609
|
+
"next": (next_step if next_step is not None else prev.get("next")) or "",
|
|
610
|
+
"files": sorted(set(prev.get("files", [])) | set(files or [])),
|
|
611
|
+
"notes": (notes if notes is not None else prev.get("notes")) or [],
|
|
612
|
+
"plan_ref": plan_ref if plan_ref is not None else prev.get("plan_ref"),
|
|
613
|
+
}
|
|
614
|
+
tmp = path.with_suffix(".json.tmp")
|
|
615
|
+
tmp.write_text(json.dumps(ckpt, ensure_ascii=False, indent=1), encoding="utf-8")
|
|
616
|
+
tmp.replace(path) # atomic: a parallel session never reads half a file
|
|
617
|
+
return ckpt
|
|
618
|
+
|
|
619
|
+
|
|
620
|
+
def load_checkpoint(project: str, session_id: str) -> dict | None:
|
|
621
|
+
path = _ckpt_path(project, session_id)
|
|
622
|
+
return json.loads(path.read_text(encoding="utf-8")) if path.exists() else None
|
|
623
|
+
|
|
624
|
+
|
|
625
|
+
def _is_stale(ckpt: dict) -> bool:
|
|
626
|
+
try:
|
|
627
|
+
upd = dt.datetime.strptime(ckpt["updated"], "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=dt.UTC)
|
|
628
|
+
except (KeyError, ValueError):
|
|
629
|
+
return True
|
|
630
|
+
return (dt.datetime.now(dt.UTC) - upd).total_seconds() / 3600 > CHECKPOINT_STALE_HOURS
|
|
631
|
+
|
|
632
|
+
|
|
633
|
+
def list_checkpoints(project: str | None = None, include_done: bool = False) -> list[dict]:
|
|
634
|
+
"""Active (not done, not stale) checkpoints, newest first."""
|
|
635
|
+
root = checkpoints_dir()
|
|
636
|
+
dirs = [root / safe_slug(project, "project")] if project else (
|
|
637
|
+
[p for p in root.iterdir() if p.is_dir()] if root.exists() else [])
|
|
638
|
+
out = []
|
|
639
|
+
for d in dirs:
|
|
640
|
+
for f in d.glob("*.json") if d.exists() else ():
|
|
641
|
+
try:
|
|
642
|
+
c = json.loads(f.read_text(encoding="utf-8"))
|
|
643
|
+
except json.JSONDecodeError:
|
|
644
|
+
continue
|
|
645
|
+
if include_done or (c.get("status") != "done" and not _is_stale(c)):
|
|
646
|
+
out.append(c)
|
|
647
|
+
out.sort(key=lambda c: c.get("updated", ""), reverse=True)
|
|
648
|
+
return out
|
|
649
|
+
|
|
650
|
+
|
|
651
|
+
def format_other_sessions(others: list[dict]) -> list[str]:
|
|
652
|
+
return [f"- [{c['agent']} · {c['updated'][:16]}] {c['task']} → next: {c.get('next') or '?'}"
|
|
653
|
+
+ (f" · files: {', '.join(c['files'][:5])}" if c.get("files") else "")
|
|
654
|
+
for c in others[:6]]
|
|
655
|
+
|
|
656
|
+
|
|
657
|
+
def resume_text(project: str, session_id: str | None = None, budget: int = 1200) -> str:
|
|
658
|
+
"""Compact text to re-inject after compaction: this session's checkpoint in
|
|
659
|
+
full + one line per other active session on the project."""
|
|
660
|
+
lines = [f"## Session checkpoint — {norm_project(project)}"]
|
|
661
|
+
own = load_checkpoint(project, session_id) if session_id else None
|
|
662
|
+
if own:
|
|
663
|
+
lines.append(f"session_id: {session_id} · status: {own['status']} · updated: {own['updated']}")
|
|
664
|
+
lines.append(f"**Task:** {own['task']}")
|
|
665
|
+
if own.get("plan_ref"):
|
|
666
|
+
lines.append(f"Plan: {own['plan_ref']}")
|
|
667
|
+
if own["done"]:
|
|
668
|
+
lines.append("Done:")
|
|
669
|
+
lines += [f"- {d}" for d in own["done"][-12:]]
|
|
670
|
+
if own.get("next"):
|
|
671
|
+
lines.append(f"**Next:** {own['next']}")
|
|
672
|
+
if own.get("files"):
|
|
673
|
+
lines.append("Files: " + ", ".join(own["files"][:20]))
|
|
674
|
+
lines += [f"- note: {n}" for n in own.get("notes", [])[:6]]
|
|
675
|
+
elif session_id:
|
|
676
|
+
lines.append(f"(no checkpoint for session_id={session_id}; create one with memory_checkpoint)")
|
|
677
|
+
others = [c for c in list_checkpoints(project) if c["session_id"] != session_id]
|
|
678
|
+
if others:
|
|
679
|
+
lines.append("Other active sessions on this project (don't step on their scope):")
|
|
680
|
+
lines += format_other_sessions(others)
|
|
681
|
+
return truncate("\n".join(lines), budget, "memory_resume")
|