sup-mem 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. sup_mem-0.7.0/.env.example +29 -0
  2. sup_mem-0.7.0/.github/dependabot.yml +11 -0
  3. sup_mem-0.7.0/.github/workflows/ci.yml +70 -0
  4. sup_mem-0.7.0/.github/workflows/release.yml +48 -0
  5. sup_mem-0.7.0/.gitignore +46 -0
  6. sup_mem-0.7.0/HANDOVER.md +399 -0
  7. sup_mem-0.7.0/LICENSE +21 -0
  8. sup_mem-0.7.0/PKG-INFO +252 -0
  9. sup_mem-0.7.0/README.md +219 -0
  10. sup_mem-0.7.0/docker-compose.qdrant.yml +15 -0
  11. sup_mem-0.7.0/docs/PHASE10-CAPTURE.md +49 -0
  12. sup_mem-0.7.0/docs/PHASE6-LOOP.md +87 -0
  13. sup_mem-0.7.0/docs/PHASE8-TEMPORAL.md +73 -0
  14. sup_mem-0.7.0/docs/PHASE9-ARCHIVAL.md +82 -0
  15. sup_mem-0.7.0/docs/demo.gif +0 -0
  16. sup_mem-0.7.0/docs/demo.tape +47 -0
  17. sup_mem-0.7.0/docs/demo_seed.py +67 -0
  18. sup_mem-0.7.0/install.sh +34 -0
  19. sup_mem-0.7.0/pyproject.toml +95 -0
  20. sup_mem-0.7.0/src/sup_mem/__init__.py +22 -0
  21. sup_mem-0.7.0/src/sup_mem/archival.py +165 -0
  22. sup_mem-0.7.0/src/sup_mem/backends/__init__.py +37 -0
  23. sup_mem-0.7.0/src/sup_mem/backends/base.py +78 -0
  24. sup_mem-0.7.0/src/sup_mem/backends/qdrant.py +317 -0
  25. sup_mem-0.7.0/src/sup_mem/backends/sqlite_fts.py +706 -0
  26. sup_mem-0.7.0/src/sup_mem/capture.py +198 -0
  27. sup_mem-0.7.0/src/sup_mem/cli.py +179 -0
  28. sup_mem-0.7.0/src/sup_mem/commands.py +731 -0
  29. sup_mem-0.7.0/src/sup_mem/config.py +497 -0
  30. sup_mem-0.7.0/src/sup_mem/embedding/__init__.py +47 -0
  31. sup_mem-0.7.0/src/sup_mem/embedding/base.py +62 -0
  32. sup_mem-0.7.0/src/sup_mem/embedding/detect.py +122 -0
  33. sup_mem-0.7.0/src/sup_mem/embedding/providers.py +310 -0
  34. sup_mem-0.7.0/src/sup_mem/hook/__init__.py +1 -0
  35. sup_mem-0.7.0/src/sup_mem/hook/pre_compact.py +47 -0
  36. sup_mem-0.7.0/src/sup_mem/hook/session_start.py +39 -0
  37. sup_mem-0.7.0/src/sup_mem/hook/stop.py +65 -0
  38. sup_mem-0.7.0/src/sup_mem/hook/user_prompt_submit.py +205 -0
  39. sup_mem-0.7.0/src/sup_mem/ledger.py +448 -0
  40. sup_mem-0.7.0/src/sup_mem/maintenance.py +315 -0
  41. sup_mem-0.7.0/src/sup_mem/manifest.py +93 -0
  42. sup_mem-0.7.0/src/sup_mem/mcp/__init__.py +1 -0
  43. sup_mem-0.7.0/src/sup_mem/mcp/server.py +127 -0
  44. sup_mem-0.7.0/src/sup_mem/migrate.py +134 -0
  45. sup_mem-0.7.0/src/sup_mem/models.py +53 -0
  46. sup_mem-0.7.0/src/sup_mem/provenance.py +244 -0
  47. sup_mem-0.7.0/src/sup_mem/py.typed +0 -0
  48. sup_mem-0.7.0/src/sup_mem/ranking.py +50 -0
  49. sup_mem-0.7.0/src/sup_mem/registration.py +188 -0
  50. sup_mem-0.7.0/src/sup_mem/service.py +217 -0
  51. sup_mem-0.7.0/src/sup_mem/status.py +205 -0
  52. sup_mem-0.7.0/tests/conftest.py +45 -0
  53. sup_mem-0.7.0/tests/test_archival.py +242 -0
  54. sup_mem-0.7.0/tests/test_backends.py +203 -0
  55. sup_mem-0.7.0/tests/test_capture.py +246 -0
  56. sup_mem-0.7.0/tests/test_consistency.py +65 -0
  57. sup_mem-0.7.0/tests/test_e2e.py +72 -0
  58. sup_mem-0.7.0/tests/test_embedding_detect.py +103 -0
  59. sup_mem-0.7.0/tests/test_hook_tiers.py +207 -0
  60. sup_mem-0.7.0/tests/test_latency.py +55 -0
  61. sup_mem-0.7.0/tests/test_ledger.py +175 -0
  62. sup_mem-0.7.0/tests/test_loop.py +140 -0
  63. sup_mem-0.7.0/tests/test_maintenance.py +259 -0
  64. sup_mem-0.7.0/tests/test_manifest_scale.py +67 -0
  65. sup_mem-0.7.0/tests/test_mcp.py +57 -0
  66. sup_mem-0.7.0/tests/test_migrate.py +108 -0
  67. sup_mem-0.7.0/tests/test_provenance.py +103 -0
  68. sup_mem-0.7.0/tests/test_registration.py +84 -0
  69. sup_mem-0.7.0/tests/test_status.py +99 -0
  70. sup_mem-0.7.0/tests/test_temporal.py +199 -0
  71. sup_mem-0.7.0/uv.lock +3353 -0
@@ -0,0 +1,29 @@
1
+ # sup-mem — optional environment overrides. Copy to .env or export in your shell.
2
+ # Every config key is mirrored as SUP_MEM_<SECTION>_<KEY> (see ~/.sup-mem/config.toml).
3
+
4
+ # --- Core ---------------------------------------------------------------------------------
5
+ # SUP_MEM_DATA_DIR=~/.sup-mem # where the store + config + logs live
6
+ # SUP_MEM_BACKEND=sqlite_fts # sqlite_fts (default) | qdrant | pgvector
7
+
8
+ # --- Retrieval tuning (the dials you actually turn) ---------------------------------------
9
+ # SUP_MEM_RETRIEVAL_K=6
10
+ # SUP_MEM_RETRIEVAL_THRESHOLD=0.35 # raise to inject less, lower to inject more
11
+ # SUP_MEM_LOGGING_RETRIEVAL_LOG=true # (query, ids, scores, tier) -> retrieval.jsonl
12
+
13
+ # --- Vector path (backend = qdrant) -------------------------------------------------------
14
+ # QDRANT_URL=http://localhost:6333
15
+ # SUP_MEM_QDRANT_COLLECTION=sup_mem
16
+ # SUP_MEM_QDRANT_HNSW_M=16
17
+ # SUP_MEM_QDRANT_HNSW_EF=128
18
+ # SUP_MEM_QDRANT_QUANTIZATION=false
19
+
20
+ # --- Embedding provider (pin one, or leave unset for auto-detection on `setup`) -----------
21
+ # SUP_MEM_EMBEDDING_PROVIDER=fastembed # ollama | fastembed | tei | voyage | openai
22
+ # SUP_MEM_EMBEDDING_MODEL=BAAI/bge-small-en-v1.5
23
+ # OLLAMA_HOST=http://localhost:11434
24
+ # TEI_URL=http://localhost:8080
25
+ # VOYAGE_API_KEY=... # hosted: network + cost, leaves your box
26
+ # OPENAI_API_KEY=... # hosted: network + cost, leaves your box
27
+
28
+ # --- Claude Code integration --------------------------------------------------------------
29
+ # CLAUDE_CONFIG_DIR=~/.claude # where `init`/`setup` write hooks + MCP config
@@ -0,0 +1,11 @@
1
+ version: 2
2
+ updates:
3
+ - package-ecosystem: "github-actions"
4
+ directory: "/"
5
+ schedule:
6
+ interval: "weekly"
7
+ - package-ecosystem: "pip"
8
+ directory: "/"
9
+ schedule:
10
+ interval: "weekly"
11
+ open-pull-requests-limit: 5
@@ -0,0 +1,70 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+
8
+ concurrency:
9
+ group: ci-${{ github.ref }}
10
+ cancel-in-progress: true
11
+
12
+ jobs:
13
+ lint-type:
14
+ name: Lint + type
15
+ runs-on: ubuntu-latest
16
+ steps:
17
+ - uses: actions/checkout@v4
18
+ - uses: astral-sh/setup-uv@v5
19
+ with:
20
+ enable-cache: true
21
+ # Install all extras so mypy type-checks the vector/hosted-embedder code paths too.
22
+ - run: uv sync --extra qdrant --extra pgvector --extra voyage --extra openai
23
+ - name: ruff (lint)
24
+ run: uv run ruff check .
25
+ - name: ruff (format check)
26
+ run: uv run ruff format --check .
27
+ - name: mypy
28
+ run: uv run mypy
29
+
30
+ test:
31
+ name: Tests (SQLite path, no optional deps)
32
+ runs-on: ubuntu-latest
33
+ strategy:
34
+ fail-fast: false
35
+ matrix:
36
+ python-version: ["3.11", "3.12"]
37
+ steps:
38
+ - uses: actions/checkout@v4
39
+ - uses: astral-sh/setup-uv@v5
40
+ with:
41
+ enable-cache: true
42
+ python-version: ${{ matrix.python-version }}
43
+ # No extras — proves the default install works with zero optional deps (§11.3).
44
+ - run: uv sync
45
+ - run: uv run pytest -q -m "not slow and not qdrant"
46
+
47
+ test-qdrant:
48
+ name: Tests (Qdrant vector path)
49
+ runs-on: ubuntu-latest
50
+ services:
51
+ qdrant:
52
+ image: qdrant/qdrant:v1.12.4
53
+ ports:
54
+ - 6333:6333
55
+ env:
56
+ QDRANT_URL: http://localhost:6333
57
+ steps:
58
+ - uses: actions/checkout@v4
59
+ - uses: astral-sh/setup-uv@v5
60
+ with:
61
+ enable-cache: true
62
+ - run: uv sync --extra qdrant
63
+ - name: Wait for Qdrant
64
+ run: |
65
+ for _ in $(seq 1 30); do
66
+ if curl -sf "$QDRANT_URL/readyz" >/dev/null; then echo "ready"; exit 0; fi
67
+ sleep 1
68
+ done
69
+ echo "Qdrant did not become ready" >&2; exit 1
70
+ - run: uv run pytest -q -m "qdrant"
@@ -0,0 +1,48 @@
1
+ name: Release
2
+
3
+ on:
4
+ push:
5
+ tags: ["v*"]
6
+
7
+ permissions:
8
+ contents: write
9
+
10
+ jobs:
11
+ release:
12
+ name: Build + GitHub Release
13
+ runs-on: ubuntu-latest
14
+ steps:
15
+ - uses: actions/checkout@v4
16
+ - uses: astral-sh/setup-uv@v5
17
+ with:
18
+ enable-cache: true
19
+ - name: Test (fast suite)
20
+ run: |
21
+ uv sync
22
+ uv run pytest -q -m "not slow and not qdrant"
23
+ - name: Build sdist + wheel
24
+ run: uv build
25
+ - name: Create GitHub Release
26
+ env:
27
+ GH_TOKEN: ${{ github.token }}
28
+ run: |
29
+ gh release create "${GITHUB_REF_NAME}" dist/* \
30
+ --title "sup-mem ${GITHUB_REF_NAME}" \
31
+ --generate-notes
32
+
33
+ pypi:
34
+ name: Publish to PyPI (trusted publishing)
35
+ runs-on: ubuntu-latest
36
+ needs: release # only publish what passed the release job's test gate
37
+ environment: pypi
38
+ permissions:
39
+ id-token: write # OIDC — no API token anywhere (PyPI trusted publisher)
40
+ steps:
41
+ - uses: actions/checkout@v4
42
+ - uses: astral-sh/setup-uv@v5
43
+ with:
44
+ enable-cache: true
45
+ - name: Build sdist + wheel
46
+ run: uv build
47
+ - name: Publish to PyPI
48
+ uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,46 @@
1
+ # Python
2
+ __pycache__/
3
+ *.py[cod]
4
+ *$py.class
5
+ *.so
6
+ .Python
7
+ build/
8
+ dist/
9
+ *.egg-info/
10
+ .eggs/
11
+ *.egg
12
+
13
+ # Virtual environments (uv)
14
+ .venv/
15
+ venv/
16
+ env/
17
+
18
+ # uv: keep uv.lock committed; ignore the cache
19
+ .uv/
20
+
21
+ # Test / coverage artifacts
22
+ .pytest_cache/
23
+ .mypy_cache/
24
+ .ruff_cache/
25
+ .coverage
26
+ .coverage.*
27
+ htmlcov/
28
+ coverage.xml
29
+
30
+ # Local runtime data (sup-mem writes to ~/.sup-mem by default;
31
+ # ignore any accidental in-repo store or logs)
32
+ .sup-mem/
33
+ *.db
34
+ *.db-shm
35
+ *.db-wal
36
+ *.log
37
+
38
+ # Env / secrets
39
+ .env
40
+ .env.local
41
+
42
+ # Editor / OS
43
+ .DS_Store
44
+ .idea/
45
+ .vscode/
46
+ *.swp
@@ -0,0 +1,399 @@
1
+ # HANDOVER — `claude-memory`
2
+ > A self-hosted, pluggable **global memory layer for Claude** that persists context across
3
+ > sessions. Default install is model-free and Docker-free with a **one-line setup**; users
4
+ > with large stores opt into vector search. Built for **ultra-fast per-turn retrieval**.
5
+ >
6
+ > **This document is the contract.** Build to it. The numbered invariants in §2 are
7
+ > non-negotiable — if an implementation choice would violate one, stop and flag it rather
8
+ > than working around it. Everything else is an implementation detail you may improve.
9
+ ---
10
+ ## 0. How to use this handover (for Claude Code)
11
+ 1. Read the whole file before writing any code. The invariants (§2) and anti-patterns (§9)
12
+ encode decisions already litigated — do not re-derive them.
13
+ 2. Build in the phases in §10 using subagents where marked `[parallel]`.
14
+ 3. After each phase, run its acceptance checks (§11). Do not proceed on red.
15
+ 4. The definition of done is: `git clone` → one command → working memory in Claude Code,
16
+ **and** the full acceptance suite (§11) passes in CI.
17
+ ---
18
+ ## 1. Assumptions (change these first if wrong)
19
+ | Assumption | Value | Change it in |
20
+ |---|---|---|
21
+ | Language / runtime | Python ≥ 3.11 | `pyproject.toml` |
22
+ | Package/deps manager | `uv` (for fast, reproducible, one-line setup) | `install.sh`, `pyproject.toml` |
23
+ | Primary client | Claude Code (hooks + MCP). Claude Desktop supported for MCP only. | `cli.py` registration |
24
+ | License | MIT | `LICENSE` |
25
+ | Repo / package name | `claude-memory` | everywhere |
26
+ | Default backend | SQLite FTS5 (zero-dependency, no Docker, no model) | `config.py` |
27
+ | Default embedder (vector path) | `fastembed` + `BAAI/bge-small-en-v1.5` (CPU, ONNX, no server) | `embedding/detect.py` |
28
+ If any of these is wrong, fix the table and propagate before building.
29
+ ---
30
+ ## 2. Non-negotiable invariants
31
+ **I1 — Two front-doors, one backend.** Retrieval happens two ways over the *same* storage:
32
+ (a) an automatic **hook** that injects context every turn without Claude choosing to, and
33
+ (b) **MCP tools** (`remember`, `recall`) that Claude calls explicitly. They share one
34
+ `MemoryBackend`. Never build two separate stores.
35
+ **I2 — The hook is a short-lived process. It must never load a model or do heavy init.**
36
+ It is spawned fresh per prompt, lives milliseconds, and dies. It may only make thin calls to
37
+ already-warm services or open an embedded DB file. Any model, pool, or cache that costs more
38
+ than a few ms to warm belongs in a long-running process, not the hook. This is the single
39
+ most load-bearing rule for "ultra fast."
40
+ **I3 — Tiered retrieval; regex is a skip-gate, never a relevance-gate.**
41
+ - Tier 0: pinned facts, always injected (flat file, no lookup).
42
+ - Tier 1: a cheap lexical **skip** check — a *whitelist of obviously-trivial turns*
43
+ (greetings, "thanks", self-contained one-off asks). Its only job is to short-circuit turns
44
+ that need no episodic memory. It must **never** decide whether a relevant memory *exists*.
45
+ - Tier 2: real retrieval (FTS or vector), gated on a **relevance score threshold**, not on
46
+ keyword presence.
47
+ **I4 — The write path is off the hot path.** Never embed-and-store synchronously while Claude
48
+ is responding. Writes happen via the explicit `remember` tool or a session-end batch. Reads
49
+ and writes share storage but must not block each other.
50
+ **I5 — Claude decides tool calls from descriptions + conversation + injected context +
51
+ manifest — never by surveying the store.** The tool descriptions are the control surface;
52
+ write them carefully (§6.5). Claude can see what the hook injected this turn and the
53
+ session-start manifest — those are its only "map" of what exists.
54
+ **I6 — Pluggable backend behind one interface (§6.1).** Everything above the interface (hook,
55
+ MCP tools, manifest) is backend-agnostic. Adding a backend must not touch them.
56
+ **I7 — Embedding-model consistency is a hard contract (vector backends).** The model that
57
+ writes and the model that reads must be identical. Store `(provider, model, dim)` in backend
58
+ metadata; refuse to start (or warn loudly + exit non-zero in `doctor`) on mismatch; ship a
59
+ `reindex` command. Vectors from different models are not comparable.
60
+ **I8 — The default experience is one command, no Docker, no model.** A person evaluating the
61
+ repo must get working memory (SQLite FTS) from a single command with zero services to spin
62
+ up. Vector search is strictly opt-in.
63
+ **I9 — Everything is tunable.** Threshold, `k`, tier-1 skip patterns, manifest strategy,
64
+ chunking, HNSW/quantization params — all live in config with sane defaults, none hard-coded
65
+ in logic. See §8.
66
+ **I10 — Manifest degrades gracefully with scale.** Full topic list when small; clustered /
67
+ summarized when large; never dump tens of thousands of tags into context.
68
+ ---
69
+ ## 3. Architecture (recap)
70
+ ```
71
+ Per prompt (READ, synchronous, latency-critical):
72
+ UserPromptSubmit hook (short-lived) ──► Tier 0 pinned facts (always)
73
+ ──► Tier 1 skip check (trivial? exit)
74
+ ──► Tier 2 backend.search(query, k, threshold)
75
+ │ (thin call to WARM service or embedded DB)
76
+ ▼
77
+ stdout ─► injected into Claude's context
78
+ Session start:
79
+ SessionStart hook ──► backend.manifest() ─► compact topic index injected
80
+ Explicit (WRITE + fallback READ, Claude-initiated):
81
+ MCP server ──► remember(text, metadata) → backend.store(...)
82
+ ──► recall(query, k) → backend.search(...) [fallback only]
83
+ Shared backend (one of):
84
+ SqliteFtsBackend (default: embedded file, BM25, no model, no Docker)
85
+ QdrantBackend (opt-in: vector kNN, auto-detected embedder)
86
+ PgVectorBackend (optional/future, same interface)
87
+ ```
88
+ Warm services (vector path only) stay up via `restart: unless-stopped`. The hook borrows
89
+ their warmth; it never generates its own (I2).
90
+ ---
91
+ ## 4. Repository layout
92
+ ```
93
+ claude-memory/
94
+ ├── README.md # see §12 for required sections
95
+ ├── LICENSE # MIT
96
+ ├── pyproject.toml # uv-managed, extras: [qdrant], [pgvector], [voyage], [openai]
97
+ ├── install.sh # one-line installer (wraps uv)
98
+ ├── docker-compose.qdrant.yml # brought up only by `setup --backend qdrant`
99
+ ├── .github/workflows/ci.yml # lint + type + test matrix; see §11.6
100
+ ├── .env.example
101
+ ├── src/claude_memory/
102
+ │ ├── __init__.py
103
+ │ ├── config.py # load/merge config (defaults ← file ← env ← flags)
104
+ │ ├── models.py # Hit, MemoryRecord dataclasses
105
+ │ ├── backends/
106
+ │ │ ├── base.py # MemoryBackend ABC (§6.1)
107
+ │ │ ├── sqlite_fts.py # default (§6.2)
108
+ │ │ ├── qdrant.py # opt-in (§6.3)
109
+ │ │ └── pgvector.py # optional stub, same interface
110
+ │ ├── embedding/
111
+ │ │ ├── __init__.py
112
+ │ │ ├── detect.py # auto-detection + fallback (§6.4)
113
+ │ │ └── providers.py # fastembed / ollama / tei / voyage / openai
114
+ │ ├── hook/
115
+ │ │ ├── user_prompt_submit.py # tiered router (§6.6) — the hot path
116
+ │ │ └── session_start.py # manifest injection
117
+ │ ├── mcp/
118
+ │ │ └── server.py # remember / recall (§6.5)
119
+ │ ├── manifest.py # scale-aware topic index (§6.7)
120
+ │ └── cli.py # init / setup / doctor / reindex / serve / manifest (§7)
121
+ └── tests/
122
+ ├── conftest.py
123
+ ├── test_backends.py # interface conformance, run against every backend
124
+ ├── test_hook_tiers.py
125
+ ├── test_embedding_detect.py
126
+ ├── test_mcp.py
127
+ ├── test_manifest_scale.py
128
+ └── test_e2e.py # clone→setup→retrieve smoke
129
+ ```
130
+ ---
131
+ ## 5. Tech stack & rationale
132
+ - **`uv`** for install/deps — chosen specifically for the "ultra fast one-line setup"
133
+ requirement; `uv sync` is an order of magnitude faster than pip and gives a lockfile.
134
+ - **SQLite FTS5 + BM25** (stdlib `sqlite3`, no extra dep) — the zero-friction default (I8).
135
+ - **Qdrant** (`qdrant-client`) as the opt-in vector store — small footprint, HNSW +
136
+ quantization, clean Docker story.
137
+ - **`fastembed`** as the default local embedder — ONNX, CPU-friendly, **runs inside the
138
+ long-lived MCP server process** so no separate model server is needed. This directly
139
+ answers "I don't want to run a model": for the FTS default there is no model at all, and
140
+ for the vector path the model is embedded in a process that's already running.
141
+ - **MCP Python SDK** for the server.
142
+ - Optional extras (`[voyage]`, `[openai]`) for users who prefer a hosted embedder.
143
+ ---
144
+ ## 6. Component specifications
145
+ ### 6.1 `MemoryBackend` interface — `backends/base.py`
146
+ Everything above the line (hook, MCP, manifest) depends only on this. Adding a backend must
147
+ not require touching them (I6).
148
+ ```python
149
+ from abc import ABC, abstractmethod
150
+ from claude_memory.models import Hit, MemoryRecord
151
+ class MemoryBackend(ABC):
152
+ @abstractmethod
153
+ def store(self, text: str, metadata: dict) -> str:
154
+ """Persist a memory. Returns its id. Idempotent on (text, source) if possible."""
155
+ @abstractmethod
156
+ def search(self, query: str, k: int, threshold: float) -> list[Hit]:
157
+ """Return up to k hits with score >= threshold, best first.
158
+ Score MUST be normalized to 0..1 across backends so the threshold is portable."""
159
+ @abstractmethod
160
+ def manifest(self, max_topics: int) -> list[str]:
161
+ """Return a compact, scale-aware topic index (see §6.7)."""
162
+ @abstractmethod
163
+ def health(self) -> dict:
164
+ """Liveness + config summary: backend name, count, embed (provider, model, dim)|None."""
165
+ @abstractmethod
166
+ def reindex(self, progress=None) -> None:
167
+ """Re-embed/rebuild. No-op for lexical backends; required for vector (I7)."""
168
+ ```
169
+ `Hit` = `{id, text, score, metadata}`. **Score normalization to 0..1 is mandatory** so
170
+ `threshold` means the same thing regardless of backend (BM25 needs a squashing function; kNN
171
+ cosine maps naturally).
172
+ ### 6.2 `SqliteFtsBackend` (default, I8)
173
+ - Single file at `~/.claude-memory/memory.db`. No server, no model, no Docker.
174
+ - FTS5 virtual table; rank with `bm25()`. Normalize BM25 → 0..1 (document the squash).
175
+ - `store()` inserts + updates the FTS index. `manifest()` returns distinct tags/topics.
176
+ - `reindex()` is a no-op (rebuild FTS if schema changed).
177
+ - Must be importable and usable with **zero optional deps installed**.
178
+ ### 6.3 `QdrantBackend` (opt-in, scale)
179
+ - Collection with named vector; HNSW params and (optional) scalar quantization in config (§8).
180
+ - Embeds via the selected provider (§6.4). **Embedding happens inside the long-lived MCP
181
+ server process, not the hook** (I2). The hook calls the backend which calls a warm embed
182
+ endpoint / in-process fastembed.
183
+ - Persists `(provider, model, dim)` in a Qdrant payload/meta doc on first write. On startup
184
+ and in `doctor`, compare against configured model → enforce I7.
185
+ - `reindex()` re-embeds every record with the current model, with progress callback.
186
+ ### 6.4 Embedding auto-detection — `embedding/detect.py` (they asked for this explicitly)
187
+ `detect_embedding_provider(config)` resolves a provider by **priority**, prints what it found,
188
+ and (interactive) lets the user choose, or (— `--yes` / non-interactive) auto-picks the top
189
+ available. Record the choice into backend meta (I7).
190
+ ```
191
+ Priority order:
192
+ 0. If config pins provider+model → validate it's reachable, use it. Respect the user.
193
+ 1. Ollama reachable (OLLAMA_HOST, default http://localhost:11434)?
194
+ GET /api/tags, filter embedding-capable models
195
+ (nomic-embed-text, mxbai-embed-large, all-minilm, snowflake-arctic-embed).
196
+ If any → offer; default nomic-embed-text.
197
+ 2. fastembed importable? → offer; default BAAI/bge-small-en-v1.5 (384-dim, CPU, no server).
198
+ 3. TEI reachable (TEI_URL)? → offer.
199
+ 4. VOYAGE_API_KEY set? → offer voyage-3 (hosted; warn: network + cost + leaves your box).
200
+ 5. OPENAI_API_KEY set? → offer text-embedding-3-small (same warnings).
201
+ 6. Fallback: install fastembed + pull bge-small-en-v1.5.
202
+ If offline AND nothing usable → hard error with exact remediation commands.
203
+ ```
204
+ Requirements:
205
+ - Detection must be **fast and side-effect-free** until a choice is made (no downloads while
206
+ probing).
207
+ - Emit a clear table: provider | model | dim | where it runs | latency class | selected.
208
+ - Non-interactive mode is what makes the Qdrant path still "one line": it picks the best
209
+ available automatically and records it.
210
+ - Always surface **fallbacks** in the output so the user knows the alternatives.
211
+ ### 6.5 MCP server — `mcp/server.py`
212
+ Exposes exactly two tools. **The descriptions are the control surface (I5)** — Claude decides
213
+ purely from these + the conversation + injected context + manifest. Ship these strings
214
+ roughly as-is:
215
+ ```
216
+ remember:
217
+ "Store a durable fact, decision, preference, or correction that should persist across
218
+ future sessions. Call when the user says things like 'remember that…', 'we decided…',
219
+ 'going forward, always…', or states a stable fact about their systems/preferences.
220
+ Do NOT call for transient, turn-specific details or things already obviously stored."
221
+ recall:
222
+ "Fallback retrieval from long-term memory. Relevant context is normally injected
223
+ automatically each turn, so call this ONLY when: the user references prior work you lack
224
+ context for (e.g. 'the fix we did', 'that ticket', possessives about past projects) AND
225
+ the context already present this turn does not cover it — optionally guided by a topic
226
+ from the session manifest. Pass a focused query."
227
+ ```
228
+ - Server is long-lived (`serve`), holds the backend (and in-process fastembed if used) warm.
229
+ - Also usable from Claude Desktop via MCP config.
230
+ ### 6.6 The hook — `hook/user_prompt_submit.py` (the hot path)
231
+ Implements Tiers 0–2 (I3). Reads Claude Code's `UserPromptSubmit` JSON from stdin; anything
232
+ printed to stdout is injected into context.
233
+ - **Lazy-import everything heavy.** On the Tier-1 skip path, the process must not import the
234
+ backend, embedding libs, or an HTTP client. Import inside the Tier-2 branch only. Verify
235
+ with the import-time test (§11.2).
236
+ - Tier 0: `cat` the pinned-facts file (fast, unconditional).
237
+ - Tier 1: compiled skip regex (whitelist) + a "never-skip" cue regex (possessives, definite
238
+ articles, past-tense references, ticket keys). Skip only if skip-match AND no cue.
239
+ - Tier 2: `backend.search(prompt, k, threshold)`; print hits under a short header.
240
+ - Total added latency budget: §8.
241
+ - Fail open and silent: any error → print nothing, exit 0. A broken memory layer must never
242
+ block the user's prompt.
243
+ `hook/session_start.py`: print `backend.manifest(max_topics)` under a header. Cache it
244
+ (§8) so it isn't recomputed when the store is unchanged.
245
+ ### 6.7 Manifest — `manifest.py` (scale-aware, I10)
246
+ - Small store (< `manifest.full_below`, default 300): list distinct topics/tags verbatim.
247
+ - Large store: cluster/group (tag rollups, or embed-cluster centroids labeled) and emit a
248
+ summarized index within a token budget.
249
+ - Cache keyed on store revision (max updated_at or a counter). Regenerate only on change.
250
+ ---
251
+ ## 7. CLI & setup UX — `cli.py`
252
+ The setup experience is a graded requirement (I8). All commands must be idempotent and
253
+ re-runnable.
254
+ | Command | Does |
255
+ |---|---|
256
+ | `claude-memory init` | **Default one-liner.** Create SQLite FTS store, write pinned-facts file, register the hook + MCP server into Claude Code settings, print next steps. No Docker, no model, no prompts. |
257
+ | `claude-memory setup --backend qdrant [--yes]` | Bring up `docker-compose.qdrant.yml`, run embedding detection (§6.4), record model, migrate/create collection, register hook + MCP. `--yes` = non-interactive, auto-pick embedder → still one line. |
258
+ | `claude-memory doctor` | Health of backend + services; **enforce I7** (exit non-zero on model mismatch) with exact fix commands. |
259
+ | `claude-memory reindex` | Re-embed the store with the current model (vector backends). Progress bar. |
260
+ | `claude-memory serve` | Run the long-lived MCP server. |
261
+ | `claude-memory manifest` | Print/refresh the manifest (debug + cache warm). |
262
+ **One-line installs** (README must show both):
263
+ ```bash
264
+ # Default — FTS, no Docker, no model:
265
+ curl -LsSf https://raw.githubusercontent.com/<you>/claude-memory/main/install.sh | sh
266
+ # install.sh: ensures uv, `uv tool install claude-memory`, then `claude-memory init`
267
+ # Large-scale — vector search:
268
+ claude-memory setup --backend qdrant --yes
269
+ ```
270
+ Registration must **detect existing config and merge, not clobber** the user's Claude Code
271
+ `settings.json` hooks / MCP servers.
272
+ ---
273
+ ## 8. Performance budget & optimization surface (I9, "ultra fast", "as optimizable as possible")
274
+ **Latency budgets (assert in tests where feasible, §11.5):**
275
+ | Path | Budget |
276
+ |---|---|
277
+ | Tier-1 skip (trivial turn), total hook overhead | < 5 ms |
278
+ | Tier-2 FTS query (10k records) | < 10 ms |
279
+ | Tier-2 vector query, warm service | < 50 ms |
280
+ | Manifest injection (cached) | < 5 ms |
281
+ **Required optimizations:**
282
+ - Lazy imports on the hook skip path (I2) — enforced by test §11.2.
283
+ - Warm services only for embedding/vectors; never in the hook.
284
+ - Manifest caching keyed on store revision.
285
+ - Batch writes; never one-embed-per-insert in a loop on the write path.
286
+ - In-process fastembed reused across MCP calls (don't re-instantiate per request).
287
+ - Qdrant: expose HNSW `m` / `ef_construct` / search `ef`, and optional scalar quantization,
288
+ all in config.
289
+ - FTS: expose BM25 `k1`/`b` and the score-squash constants.
290
+ **Tuning knobs (config, all with defaults):**
291
+ `retrieval.k`, `retrieval.threshold`, `tier1.skip_patterns`, `tier1.cue_patterns`,
292
+ `manifest.full_below`, `manifest.token_budget`, `chunking.*`, `qdrant.hnsw.*`,
293
+ `qdrant.quantization`, `embedding.provider`, `embedding.model`.
294
+ **Retrieval logging for tuning (ship it on by default, off switch in config):** log
295
+ `(query, injected_ids, scores, tier_taken)` to a local file so users can eyeball
296
+ precision/recall and tune the threshold. This is the honest answer to "the threshold is a
297
+ dial you must tune on your own data" — give them the data to tune with.
298
+ ---
299
+ ## 9. Anti-patterns — do NOT do these (each maps to an invariant)
300
+ - ❌ Load an embedding model / open pools inside the hook. (I2)
301
+ - ❌ Embed synchronously on the write hot path. (I4)
302
+ - ❌ Use regex/keywords to decide whether a relevant memory *exists*. (I3)
303
+ - ❌ Make Claude survey the store to decide whether to call a tool, or expose a "list all
304
+ memories" tool for that purpose — use descriptions + manifest. (I5)
305
+ - ❌ Mix embedding models between write and read, or switch models without `reindex`. (I7)
306
+ - ❌ Require Docker or a model for the default install. (I8)
307
+ - ❌ Hard-code threshold/k/patterns in logic instead of config. (I9)
308
+ - ❌ Dump the entire tag list into context at scale. (I10)
309
+ - ❌ Let a memory-layer error block or delay the user's prompt — fail open, silent, exit 0.
310
+ ---
311
+ ## 10. Build plan for agents
312
+ Run phases in order. Items marked `[parallel]` can be separate subagents; join at the phase's
313
+ acceptance gate before moving on.
314
+ **Phase 0 — Scaffold.** Repo layout (§4), `pyproject.toml` with extras, `install.sh`,
315
+ `.github/workflows/ci.yml`, `LICENSE`, `models.py`, `config.py` (defaults←file←env←flags).
316
+ Gate: `uv sync` succeeds; `claude-memory --help` runs.
317
+ **Phase 1 — Interface + default backend.** `backends/base.py`, then `sqlite_fts.py` with score
318
+ normalization. Gate: §11.1 conformance suite green against SQLite; §11.3 default-deps import
319
+ test green (no optional deps installed).
320
+ **Phase 2 `[parallel]`:**
321
+ - 2a — Hook: `user_prompt_submit.py` (tiers), `session_start.py`. Gate §11.2, §11.4.
322
+ - 2b — MCP server: `mcp/server.py` with the two tools + descriptions. Gate §11.4.
323
+ - 2c — Manifest: `manifest.py` scale-aware. Gate §11.7.
324
+ **Phase 3 — Vector backend + embedding detection `[parallel]`:**
325
+ - 3a — `embedding/providers.py` + `detect.py`. Gate §11.8 (mock each provider; detection
326
+ priority + fallback + non-interactive).
327
+ - 3b — `qdrant.py` + `docker-compose.qdrant.yml` + I7 enforcement + `reindex`. Gate §11.1
328
+ conformance suite green against Qdrant too.
329
+ **Phase 4 — CLI & registration.** `init`, `setup`, `doctor`, `reindex`, `serve`, `manifest`;
330
+ Claude Code settings merge (non-clobbering). Gate §11.9.
331
+ **Phase 5 — E2E, docs, polish.** `test_e2e.py`, README (§12), `.env.example`, retrieval
332
+ logging. Gate: full suite green in CI; both one-line installs verified in a clean container.
333
+ **Verification agent:** after Phases 1–5, a dedicated agent runs the *entire* §11 suite from a
334
+ clean checkout and produces a pass/fail report per criterion. Nothing ships red.
335
+ ---
336
+ ## 11. Acceptance criteria (the verification contract)
337
+ ### 11.1 Backend conformance (run against EVERY backend)
338
+ - store→search round-trips; returned hits respect `k` and `threshold`.
339
+ - scores are within 0..1 and monotonic with relevance.
340
+ - `health()` reports count and embed meta correctly.
341
+ - `manifest(max_topics)` never exceeds `max_topics`.
342
+ ### 11.2 Hook lazy-import (I2)
343
+ - With backend/embedding modules instrumented, a **trivial prompt** ("thanks") triggers the
344
+ Tier-1 skip and imports **none** of them. Assert via `sys.modules` inspection in a
345
+ subprocess. This test failing means the hot path is slow — treat as build-breaking.
346
+ ### 11.3 Default-deps install
347
+ - In an env with only base deps (no `qdrant-client`, no `fastembed`), `import claude_memory`,
348
+ `init`, store, and search all work.
349
+ ### 11.4 Tier logic (I3)
350
+ - Greeting/thanks → skip, nothing beyond Tier 0 injected.
351
+ - Same-topic paraphrase with weak keywords but a cue ("the thing we fixed") → NOT skipped.
352
+ - Tier-2 respects the score threshold (below-threshold hits are dropped).
353
+ ### 11.5 Latency budgets (§8)
354
+ - Assert Tier-1 overhead and FTS query budgets on a seeded 10k-record store. Vector budget
355
+ asserted against a warm local Qdrant in CI (or marked `@slow` if Qdrant unavailable).
356
+ ### 11.6 Fail-open
357
+ - Backend raises on `search` → hook prints nothing, exits 0, does not raise.
358
+ ### 11.7 Manifest scale (I10)
359
+ - < `full_below` records → verbatim list. 50k records → summarized, within token budget,
360
+ cached (second call does no store scan).
361
+ ### 11.8 Embedding detection (§6.4)
362
+ - Priority order honored with mocked availability of each provider.
363
+ - Non-interactive `--yes` auto-picks top available.
364
+ - Nothing available + offline → hard error with remediation text.
365
+ - Chosen `(provider, model, dim)` is recorded and read back.
366
+ ### 11.9 Model-consistency contract (I7)
367
+ - Writing with model A then configuring model B → `doctor` exits non-zero with a clear message;
368
+ `reindex` fixes it; post-reindex `doctor` is green.
369
+ ### 11.10 CI
370
+ - `ruff` + `mypy` clean; test matrix on Python 3.11/3.12; SQLite path runs everywhere; Qdrant
371
+ path runs via a service container.
372
+ ### 11.11 One-line install E2E
373
+ - Clean container: default install → memory works end-to-end. Separately: `setup --backend
374
+ qdrant --yes` on a Docker-enabled runner → memory works end-to-end.
375
+ ---
376
+ ## 12. README requirements (public-repo readiness)
377
+ Must include, in this order: one-sentence pitch; a 20-second demo GIF/asciicast placeholder;
378
+ **the two one-line installs (default vs vector) side by side** with a plain-English "use FTS
379
+ until you have ~10k+ memories, then switch"; the architecture diagram (two front-doors, one
380
+ backend); how the hook + MCP register in Claude Code; the **model-consistency warning** (I7)
381
+ called out in its own box, not a footnote; the tuning section (threshold/k + retrieval logs);
382
+ a backend comparison table (FTS vs Qdrant vs pgvector: setup cost, scale, ranks-by-meaning);
383
+ and a "designed for optimization" section pointing at §8's knobs. Keep the default path
384
+ frictionless above the fold; put vector/scale content below.
385
+ ---
386
+ ## 13. Config surface (single `config.py`, precedence defaults ← file ← env ← flags)
387
+ Ship a documented `~/.claude-memory/config.toml` with every knob from §8 present and defaulted.
388
+ Env vars mirror keys (`CLAUDE_MEMORY_RETRIEVAL_THRESHOLD`, etc.). `--yes` and `--backend`
389
+ are the only flags needed for one-line setup.
390
+ ---
391
+ ## 14. Open decisions left to the implementer (pick sensibly, document the choice)
392
+ - BM25→0..1 squash function (suggest a logistic on rank-score; document constants).
393
+ - Session-end summarization writer: implement now vs. leave a documented hook point. (Explicit
394
+ `remember` is the MVP; summarization can be Phase 6.)
395
+ - pgvector backend: stub with interface + tests skipped, or full. Stub is acceptable for v1.
396
+ - Manifest clustering method at scale (tag rollup is fine for v1; embedding-cluster labels are
397
+ a nice-to-have).
398
+ ---
399
+ *End of handover. Build to the invariants; improve everything else.*
sup_mem-0.7.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Kiran Jose
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.