memstem 0.17.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. memstem/__init__.py +3 -0
  2. memstem/__main__.py +6 -0
  3. memstem/adapters/__init__.py +1 -0
  4. memstem/adapters/base.py +64 -0
  5. memstem/adapters/claude_code.py +351 -0
  6. memstem/adapters/codex.py +508 -0
  7. memstem/adapters/openclaw.py +664 -0
  8. memstem/auth.py +124 -0
  9. memstem/cli.py +2723 -0
  10. memstem/client.py +280 -0
  11. memstem/config.py +532 -0
  12. memstem/core/__init__.py +1 -0
  13. memstem/core/dedup.py +117 -0
  14. memstem/core/embed_worker.py +426 -0
  15. memstem/core/embeddings.py +712 -0
  16. memstem/core/extraction.py +395 -0
  17. memstem/core/frontmatter.py +224 -0
  18. memstem/core/hyde.py +504 -0
  19. memstem/core/importance_seed.py +183 -0
  20. memstem/core/index.py +1351 -0
  21. memstem/core/media.py +102 -0
  22. memstem/core/mmr.py +167 -0
  23. memstem/core/pipeline.py +309 -0
  24. memstem/core/rerank.py +672 -0
  25. memstem/core/retrieval_log.py +193 -0
  26. memstem/core/search.py +602 -0
  27. memstem/core/storage.py +190 -0
  28. memstem/core/summarizer.py +480 -0
  29. memstem/discovery.py +227 -0
  30. memstem/eval/__init__.py +35 -0
  31. memstem/eval/harness.py +314 -0
  32. memstem/hygiene/__init__.py +1 -0
  33. memstem/hygiene/cleanup_retro.py +633 -0
  34. memstem/hygiene/dedup_candidates.py +334 -0
  35. memstem/hygiene/dedup_judge.py +555 -0
  36. memstem/hygiene/distillation.py +241 -0
  37. memstem/hygiene/importance.py +357 -0
  38. memstem/hygiene/loop.py +398 -0
  39. memstem/hygiene/project_records.py +702 -0
  40. memstem/hygiene/session_distill.py +808 -0
  41. memstem/hygiene/state.py +232 -0
  42. memstem/hygiene/verify.py +293 -0
  43. memstem/integration.py +783 -0
  44. memstem/migrate.py +247 -0
  45. memstem/progress.py +164 -0
  46. memstem/prompts/__init__.py +6 -0
  47. memstem/prompts/dedup_judge.txt +41 -0
  48. memstem/prompts/distill_project.txt +89 -0
  49. memstem/prompts/distill_session.txt +79 -0
  50. memstem/prompts/hyde.txt +27 -0
  51. memstem/prompts/rerank.txt +38 -0
  52. memstem/servers/__init__.py +1 -0
  53. memstem/servers/http_server.py +436 -0
  54. memstem/servers/mcp_server.py +539 -0
  55. memstem/servers/request_limits.py +47 -0
  56. memstem/star_nudge.py +65 -0
  57. memstem-0.17.0.dist-info/METADATA +505 -0
  58. memstem-0.17.0.dist-info/RECORD +61 -0
  59. memstem-0.17.0.dist-info/WHEEL +4 -0
  60. memstem-0.17.0.dist-info/entry_points.txt +2 -0
  61. memstem-0.17.0.dist-info/licenses/LICENSE +21 -0
memstem/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """Memstem: unified memory and skill infrastructure for AI agents."""
2
+
3
+ __version__ = "0.17.0"
memstem/__main__.py ADDED
@@ -0,0 +1,6 @@
1
+ """Entry point for `python -m memstem`."""
2
+
3
+ from memstem.cli import app
4
+
5
+ if __name__ == "__main__":
6
+ app()
@@ -0,0 +1 @@
1
+ """Per-AI adapters: pull memory and skills from external filesystems."""
@@ -0,0 +1,64 @@
1
+ """Adapter base class and registry."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from abc import ABC, abstractmethod
6
+ from collections.abc import AsyncGenerator
7
+ from pathlib import Path
8
+ from typing import Any
9
+
10
+ from pydantic import BaseModel
11
+
12
+
13
+ class MemoryRecord(BaseModel):
14
+ """A normalized memory record produced by an adapter."""
15
+
16
+ source: str
17
+ """Adapter name, e.g. 'claude-code', 'openclaw'."""
18
+
19
+ ref: str
20
+ """Source-specific identifier (session id, file path, etc.)."""
21
+
22
+ title: str | None = None
23
+ body: str
24
+ tags: list[str] = []
25
+ metadata: dict[str, Any] = {}
26
+
27
+
28
+ class Adapter(ABC):
29
+ """Base class for all Memstem adapters.
30
+
31
+ Adapters are responsible for watching one external AI's filesystem
32
+ and producing normalized MemoryRecord objects. Storage and indexing
33
+ are downstream — adapters never touch the index directly.
34
+ """
35
+
36
+ name: str
37
+ """Unique identifier, e.g. 'claude-code', 'openclaw'."""
38
+
39
+ _observer: Any = None
40
+ """The running ``watch()``'s watchdog observer, registered after
41
+ ``observer.start()`` and cleared on shutdown. Read via
42
+ :meth:`watcher_alive`; never touched by callers directly."""
43
+
44
+ @abstractmethod
45
+ def watch(self, paths: list[Path]) -> AsyncGenerator[MemoryRecord, None]:
46
+ """Yield records as files change. Long-running async generator."""
47
+ ...
48
+
49
+ @abstractmethod
50
+ def reconcile(self, paths: list[Path]) -> AsyncGenerator[MemoryRecord, None]:
51
+ """Yield records by scanning paths from scratch. One-shot async generator."""
52
+ ...
53
+
54
+ def watcher_alive(self) -> bool | None:
55
+ """Liveness of this adapter's watchdog observer thread.
56
+
57
+ ``None`` means no watch is running (``watch()`` not started yet,
58
+ shut down cleanly, or it has nothing to observe). ``False`` means
59
+ a watch IS running but its observer thread died — file events are
60
+ silently being dropped; ``/health`` reports this as a
61
+ ``watcher_dead:<name>`` problem. ``True`` is the healthy state.
62
+ """
63
+ observer = self._observer
64
+ return None if observer is None else bool(observer.is_alive())
@@ -0,0 +1,351 @@
1
+ """Claude Code session + instructions adapter.
2
+
3
+ Two ingestion paths:
4
+
5
+ 1. **Session JSONLs** under `~/.claude/projects/<encoded-cwd>/<uuid>.jsonl` —
6
+ each file is one conversation, folded into one `MemoryRecord` per
7
+ session (type=session) with the chronological transcript.
8
+
9
+ 2. **Instructions files** (e.g. `~/.claude/CLAUDE.md`, project-level
10
+ CLAUDE.md, etc.) — passed in as `extra_files` to the constructor.
11
+ Each becomes a single record tagged `instructions` so a search for
12
+ "what does the global CLAUDE.md say about X" actually finds it.
13
+
14
+ For v0.1 the policy is "re-emit the full file on every change." The
15
+ consuming pipeline upserts by `ref`, so re-emits idempotently replace
16
+ the prior record. A future version can track per-line offsets to emit
17
+ incrementally for sessions — see PLAN step 5.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import asyncio
23
+ import json
24
+ import logging
25
+ import os
26
+ import re
27
+ from collections.abc import AsyncGenerator, Iterator
28
+ from datetime import UTC, datetime
29
+ from pathlib import Path
30
+ from typing import Any
31
+
32
+ import frontmatter as fm
33
+ from watchdog.events import FileSystemEvent, FileSystemEventHandler
34
+ from watchdog.observers import Observer
35
+
36
+ from memstem.adapters.base import Adapter, MemoryRecord
37
+
38
+ logger = logging.getLogger(__name__)
39
+
40
+ H1_RE = re.compile(r"^#\s+(.+?)\s*$", re.MULTILINE)
41
+
42
+
43
+ def _extract_text(content: Any) -> str:
44
+ """Pull plain text out of a Claude message content payload.
45
+
46
+ Content is either a bare string or a list of typed blocks (`text`,
47
+ `tool_use`, `tool_result`, etc.). Tool blocks are summarized so the
48
+ transcript stays readable but doesn't pull in raw tool I/O blobs.
49
+ """
50
+ if isinstance(content, str):
51
+ return content
52
+ if not isinstance(content, list):
53
+ return ""
54
+ parts: list[str] = []
55
+ for block in content:
56
+ if not isinstance(block, dict):
57
+ continue
58
+ block_type = block.get("type")
59
+ if block_type == "text":
60
+ text = block.get("text", "")
61
+ if isinstance(text, str) and text:
62
+ parts.append(text)
63
+ elif block_type == "tool_use":
64
+ name = block.get("name", "tool")
65
+ parts.append(f"[tool_use: {name}]")
66
+ elif block_type == "tool_result":
67
+ parts.append("[tool_result]")
68
+ return "\n".join(parts)
69
+
70
+
71
+ def _format_turn(role: str, text: str) -> str:
72
+ if not text.strip():
73
+ return ""
74
+ return f"**{role.title()}:** {text}"
75
+
76
+
77
+ def _file_mtime_iso(path: Path) -> str:
78
+ return datetime.fromtimestamp(path.stat().st_mtime, tz=UTC).isoformat()
79
+
80
+
81
+ def _extract_h1(body: str) -> str | None:
82
+ match = H1_RE.search(body)
83
+ return match.group(1).strip() if match else None
84
+
85
+
86
+ def _instructions_record(path: Path, source_name: str = "claude-code") -> MemoryRecord | None:
87
+ """Read a markdown instructions file as an `instructions`-tagged record."""
88
+ if not path.is_file():
89
+ return None
90
+ try:
91
+ text = path.read_text(encoding="utf-8")
92
+ except (OSError, UnicodeDecodeError) as exc:
93
+ logger.warning("could not read %s: %s", path, exc)
94
+ return None
95
+ try:
96
+ post = fm.loads(text)
97
+ except Exception as exc:
98
+ logger.warning("frontmatter parse failed for %s: %s", path, exc)
99
+ return None
100
+
101
+ body = post.content
102
+ meta = dict(post.metadata)
103
+ title_raw = meta.get("title")
104
+ if isinstance(title_raw, str) and title_raw.strip():
105
+ title = title_raw.strip()
106
+ else:
107
+ title = _extract_h1(body) or path.stem
108
+
109
+ mtime = _file_mtime_iso(path)
110
+ return MemoryRecord(
111
+ source=source_name,
112
+ ref=str(path),
113
+ title=title,
114
+ body=body,
115
+ tags=["instructions"],
116
+ metadata={
117
+ "type": "memory",
118
+ "created": str(meta.get("created") or mtime),
119
+ "updated": mtime,
120
+ },
121
+ )
122
+
123
+
124
+ def _parse_session_file(path: Path) -> dict[str, Any] | None:
125
+ """Parse a session JSONL into a summary dict, or None if unreadable."""
126
+ if not path.is_file():
127
+ return None
128
+ try:
129
+ text = path.read_text(encoding="utf-8", errors="replace")
130
+ except OSError as exc:
131
+ logger.warning("could not read %s: %s", path, exc)
132
+ return None
133
+
134
+ turns: list[str] = []
135
+ title: str | None = None
136
+ session_id: str | None = None
137
+ first_timestamp: str | None = None
138
+ last_timestamp: str | None = None
139
+
140
+ for line in text.splitlines():
141
+ line = line.strip()
142
+ if not line:
143
+ continue
144
+ try:
145
+ entry: dict[str, Any] = json.loads(line)
146
+ except json.JSONDecodeError:
147
+ continue
148
+ if not isinstance(entry, dict):
149
+ continue
150
+
151
+ ts = entry.get("timestamp")
152
+ if isinstance(ts, str):
153
+ if first_timestamp is None:
154
+ first_timestamp = ts
155
+ last_timestamp = ts
156
+ sid = entry.get("sessionId")
157
+ if isinstance(sid, str) and session_id is None:
158
+ session_id = sid
159
+
160
+ entry_type = entry.get("type")
161
+ if entry_type == "ai-title":
162
+ candidate = entry.get("title") or entry.get("text")
163
+ if isinstance(candidate, str) and candidate.strip():
164
+ title = candidate.strip()
165
+ elif entry_type in ("user", "assistant"):
166
+ msg = entry.get("message")
167
+ if not isinstance(msg, dict):
168
+ continue
169
+ text_payload = _extract_text(msg.get("content", ""))
170
+ turn = _format_turn(entry_type, text_payload)
171
+ if turn:
172
+ turns.append(turn)
173
+
174
+ if not session_id:
175
+ session_id = path.stem
176
+ if title is None and turns:
177
+ for turn in turns:
178
+ if turn.startswith("**User:**"):
179
+ title = turn[len("**User:** ") :].splitlines()[0][:80].strip()
180
+ break
181
+ if not title:
182
+ title = f"session {session_id[:8]}"
183
+
184
+ return {
185
+ "session_id": session_id,
186
+ "title": title,
187
+ "body": "\n\n".join(turns),
188
+ "first_timestamp": first_timestamp,
189
+ "last_timestamp": last_timestamp,
190
+ "turn_count": len(turns),
191
+ }
192
+
193
+
194
+ def _session_to_record(path: Path, source_name: str = "claude-code") -> MemoryRecord | None:
195
+ parsed = _parse_session_file(path)
196
+ if parsed is None:
197
+ return None
198
+ body = parsed["body"]
199
+ if not isinstance(body, str) or not body.strip():
200
+ return None
201
+
202
+ project_dir = path.parent.name
203
+ tags = [project_dir.lstrip("-")] if project_dir.startswith("-") else []
204
+
205
+ return MemoryRecord(
206
+ source=source_name,
207
+ ref=str(path),
208
+ title=str(parsed["title"]),
209
+ body=body,
210
+ tags=tags,
211
+ metadata={
212
+ "type": "session",
213
+ "session_id": parsed["session_id"],
214
+ "created": parsed["first_timestamp"] or _file_mtime_iso(path),
215
+ "updated": parsed["last_timestamp"] or _file_mtime_iso(path),
216
+ "turn_count": parsed["turn_count"],
217
+ "project": project_dir,
218
+ },
219
+ )
220
+
221
+
222
+ def _iter_jsonl_files(root: Path) -> Iterator[Path]:
223
+ if not root.exists():
224
+ return
225
+ if root.is_file():
226
+ if root.suffix == ".jsonl":
227
+ yield root
228
+ return
229
+ for path in sorted(root.rglob("*.jsonl")):
230
+ if path.is_file():
231
+ yield path
232
+
233
+
234
+ class _EventHandler(FileSystemEventHandler):
235
+ """Coalesces rapid-fire file events via per-path debounce timers."""
236
+
237
+ DEFAULT_DEBOUNCE_SECONDS = 30.0
238
+
239
+ def __init__(
240
+ self,
241
+ loop: asyncio.AbstractEventLoop,
242
+ queue: asyncio.Queue[Path],
243
+ suffixes: tuple[str, ...] = (".jsonl",),
244
+ ) -> None:
245
+ super().__init__()
246
+ self._loop = loop
247
+ self._queue = queue
248
+ self._suffixes = suffixes
249
+ self._pending: dict[Path, asyncio.TimerHandle] = {}
250
+ self._debounce_seconds = float(
251
+ os.environ.get(
252
+ "MEMSTEM_CLAUDE_CODE_WATCH_DEBOUNCE_SECONDS",
253
+ str(self.DEFAULT_DEBOUNCE_SECONDS),
254
+ )
255
+ )
256
+
257
+ def _enqueue(self, src: str) -> None:
258
+ path = Path(src)
259
+ if path.suffix not in self._suffixes:
260
+ return
261
+ if self._debounce_seconds <= 0:
262
+ self._loop.call_soon_threadsafe(self._queue.put_nowait, path)
263
+ return
264
+ self._loop.call_soon_threadsafe(self._schedule, path)
265
+
266
+ def _schedule(self, path: Path) -> None:
267
+ prior = self._pending.get(path)
268
+ if prior is not None:
269
+ prior.cancel()
270
+ self._pending[path] = self._loop.call_later(self._debounce_seconds, self._fire, path)
271
+
272
+ def _fire(self, path: Path) -> None:
273
+ self._pending.pop(path, None)
274
+ self._queue.put_nowait(path)
275
+
276
+ def on_created(self, event: FileSystemEvent) -> None:
277
+ if not event.is_directory:
278
+ self._enqueue(str(event.src_path))
279
+
280
+ def on_modified(self, event: FileSystemEvent) -> None:
281
+ if not event.is_directory:
282
+ self._enqueue(str(event.src_path))
283
+
284
+ def on_moved(self, event: FileSystemEvent) -> None:
285
+ dest = getattr(event, "dest_path", None)
286
+ if not event.is_directory and dest:
287
+ self._enqueue(str(dest))
288
+
289
+
290
+ class ClaudeCodeAdapter(Adapter):
291
+ """Reads Claude Code session JSONLs and instructions files."""
292
+
293
+ name = "claude-code"
294
+
295
+ def __init__(self, extra_files: list[Path] | None = None) -> None:
296
+ # Resolve once up front so equality checks during watch are reliable.
297
+ self.extra_files = [Path(p).expanduser().resolve() for p in (extra_files or [])]
298
+
299
+ async def reconcile(self, paths: list[Path]) -> AsyncGenerator[MemoryRecord, None]:
300
+ for root in paths:
301
+ for path in _iter_jsonl_files(root):
302
+ record = _session_to_record(path, self.name)
303
+ if record is not None:
304
+ yield record
305
+ for extra in self.extra_files:
306
+ instr = _instructions_record(extra, self.name)
307
+ if instr is not None:
308
+ yield instr
309
+
310
+ async def watch(self, paths: list[Path]) -> AsyncGenerator[MemoryRecord, None]:
311
+ queue: asyncio.Queue[Path] = asyncio.Queue()
312
+ loop = asyncio.get_running_loop()
313
+ observer = Observer()
314
+ handler = _EventHandler(loop=loop, queue=queue, suffixes=(".jsonl", ".md"))
315
+
316
+ for root in paths:
317
+ if root.exists():
318
+ observer.schedule(handler, str(root), recursive=True)
319
+ # Watch the parent dir of each extras file so we can pick up its changes.
320
+ watched_parents: set[Path] = set()
321
+ for extra in self.extra_files:
322
+ if extra.parent.exists() and extra.parent not in watched_parents:
323
+ observer.schedule(handler, str(extra.parent), recursive=False)
324
+ watched_parents.add(extra.parent)
325
+
326
+ observer.start()
327
+ self._observer = observer # registered for watcher_alive() / health
328
+ try:
329
+ while True:
330
+ changed = await queue.get()
331
+ if not changed.is_file():
332
+ continue
333
+ resolved = changed.resolve()
334
+
335
+ if resolved.suffix == ".jsonl":
336
+ record = _session_to_record(resolved, self.name)
337
+ if record is not None:
338
+ yield record
339
+ continue
340
+
341
+ if resolved in self.extra_files:
342
+ instr = _instructions_record(resolved, self.name)
343
+ if instr is not None:
344
+ yield instr
345
+ finally:
346
+ self._observer = None
347
+ observer.stop()
348
+ observer.join()
349
+
350
+
351
+ __all__ = ["ClaudeCodeAdapter"]