memgres 0.6.0__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. {memgres-0.6.0 → memgres-0.7.0}/PKG-INFO +11 -7
  2. {memgres-0.6.0 → memgres-0.7.0}/README.md +10 -6
  3. {memgres-0.6.0 → memgres-0.7.0}/memgres/__init__.py +3 -1
  4. {memgres-0.6.0 → memgres-0.7.0}/memgres/_version.py +1 -1
  5. {memgres-0.6.0 → memgres-0.7.0}/memgres/config.py +13 -0
  6. {memgres-0.6.0 → memgres-0.7.0}/memgres/embed_worker.py +31 -58
  7. memgres-0.7.0/memgres/links.py +188 -0
  8. {memgres-0.6.0 → memgres-0.7.0}/memgres/mcp_server.py +102 -29
  9. memgres-0.7.0/memgres/migrations/0015_normalize_tags.sql +44 -0
  10. memgres-0.7.0/memgres/migrations/0016_valid_at.sql +16 -0
  11. memgres-0.7.0/memgres/migrations/0017_memory_link.sql +40 -0
  12. memgres-0.7.0/memgres/migrations/0018_links_built.sql +12 -0
  13. memgres-0.7.0/memgres/migrations/0019_memory_usage.sql +36 -0
  14. memgres-0.7.0/memgres/migrations/0020_memory_usage_no_fk.sql +17 -0
  15. memgres-0.7.0/memgres/periodic.py +153 -0
  16. memgres-0.7.0/memgres/relink.py +157 -0
  17. {memgres-0.6.0 → memgres-0.7.0}/memgres/schema.py +45 -2
  18. {memgres-0.6.0 → memgres-0.7.0}/memgres/search.py +51 -34
  19. {memgres-0.6.0 → memgres-0.7.0}/memgres/server.py +38 -24
  20. {memgres-0.6.0 → memgres-0.7.0}/memgres/store.py +568 -66
  21. memgres-0.7.0/memgres/tags.py +90 -0
  22. {memgres-0.6.0 → memgres-0.7.0}/memgres/vector/base.py +35 -15
  23. {memgres-0.6.0 → memgres-0.7.0}/memgres/vector/pgvector.py +4 -2
  24. {memgres-0.6.0 → memgres-0.7.0}/memgres/vector/qdrant.py +4 -2
  25. {memgres-0.6.0 → memgres-0.7.0}/memgres.egg-info/PKG-INFO +11 -7
  26. {memgres-0.6.0 → memgres-0.7.0}/memgres.egg-info/SOURCES.txt +16 -0
  27. {memgres-0.6.0 → memgres-0.7.0}/memgres.egg-info/entry_points.txt +1 -0
  28. {memgres-0.6.0 → memgres-0.7.0}/pyproject.toml +1 -0
  29. {memgres-0.6.0 → memgres-0.7.0}/tests/test_blame_integration.py +3 -0
  30. {memgres-0.6.0 → memgres-0.7.0}/tests/test_chunk_index.py +29 -0
  31. {memgres-0.6.0 → memgres-0.7.0}/tests/test_claim_and_reembed.py +3 -0
  32. {memgres-0.6.0 → memgres-0.7.0}/tests/test_embed_worker.py +3 -0
  33. {memgres-0.6.0 → memgres-0.7.0}/tests/test_identity_integration.py +3 -0
  34. {memgres-0.6.0 → memgres-0.7.0}/tests/test_lexical_match.py +3 -0
  35. {memgres-0.6.0 → memgres-0.7.0}/tests/test_limits.py +9 -0
  36. memgres-0.7.0/tests/test_links.py +738 -0
  37. {memgres-0.6.0 → memgres-0.7.0}/tests/test_list.py +8 -1
  38. {memgres-0.6.0 → memgres-0.7.0}/tests/test_mcp_admin_tools.py +6 -0
  39. {memgres-0.6.0 → memgres-0.7.0}/tests/test_mcp_instructions.py +5 -2
  40. {memgres-0.6.0 → memgres-0.7.0}/tests/test_mcp_recall_schema.py +29 -1
  41. {memgres-0.6.0 → memgres-0.7.0}/tests/test_mcp_tool_visibility.py +3 -0
  42. {memgres-0.6.0 → memgres-0.7.0}/tests/test_migration_upgrade.py +14 -9
  43. {memgres-0.6.0 → memgres-0.7.0}/tests/test_multi_space_search.py +8 -4
  44. {memgres-0.6.0 → memgres-0.7.0}/tests/test_path_addressing.py +3 -0
  45. {memgres-0.6.0 → memgres-0.7.0}/tests/test_qdrant_integration.py +3 -0
  46. memgres-0.7.0/tests/test_require_title.py +183 -0
  47. memgres-0.7.0/tests/test_retention.py +270 -0
  48. {memgres-0.6.0 → memgres-0.7.0}/tests/test_roles_bootstrap.py +3 -0
  49. {memgres-0.6.0 → memgres-0.7.0}/tests/test_search_integration.py +63 -0
  50. {memgres-0.6.0 → memgres-0.7.0}/tests/test_security_integration.py +30 -2
  51. {memgres-0.6.0 → memgres-0.7.0}/tests/test_segments_store.py +3 -0
  52. {memgres-0.6.0 → memgres-0.7.0}/tests/test_server_info.py +27 -0
  53. {memgres-0.6.0 → memgres-0.7.0}/tests/test_server_integration.py +17 -1
  54. {memgres-0.6.0 → memgres-0.7.0}/tests/test_snippets.py +3 -0
  55. {memgres-0.6.0 → memgres-0.7.0}/tests/test_store_integration.py +47 -10
  56. memgres-0.7.0/tests/test_tags.py +268 -0
  57. memgres-0.7.0/tests/test_usage.py +328 -0
  58. memgres-0.7.0/tests/test_valid_at.py +201 -0
  59. {memgres-0.6.0 → memgres-0.7.0}/tests/test_write_ergonomics.py +3 -0
  60. {memgres-0.6.0 → memgres-0.7.0}/LICENSE +0 -0
  61. {memgres-0.6.0 → memgres-0.7.0}/memgres/admin.py +0 -0
  62. {memgres-0.6.0 → memgres-0.7.0}/memgres/admin_cli.py +0 -0
  63. {memgres-0.6.0 → memgres-0.7.0}/memgres/blame.py +0 -0
  64. {memgres-0.6.0 → memgres-0.7.0}/memgres/bootstrap.py +0 -0
  65. {memgres-0.6.0 → memgres-0.7.0}/memgres/delimiters.py +0 -0
  66. {memgres-0.6.0 → memgres-0.7.0}/memgres/diffing.py +0 -0
  67. {memgres-0.6.0 → memgres-0.7.0}/memgres/embeddings.py +0 -0
  68. {memgres-0.6.0 → memgres-0.7.0}/memgres/healthcheck.py +0 -0
  69. {memgres-0.6.0 → memgres-0.7.0}/memgres/identity.py +0 -0
  70. {memgres-0.6.0 → memgres-0.7.0}/memgres/indexing.py +0 -0
  71. {memgres-0.6.0 → memgres-0.7.0}/memgres/info.py +0 -0
  72. {memgres-0.6.0 → memgres-0.7.0}/memgres/lines.py +0 -0
  73. {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0001_core.sql +0 -0
  74. {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0002_identity.sql +0 -0
  75. {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0003_history_author.sql +0 -0
  76. {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0004_title.sql +0 -0
  77. {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0005_chunk_index.sql +0 -0
  78. {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0006_reader_floor.sql +0 -0
  79. {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0007_embed_retry.sql +0 -0
  80. {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0008_service_roles.sql +0 -0
  81. {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0009_create_namespace_right.sql +0 -0
  82. {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0010_namespace_alias.sql +0 -0
  83. {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0011_drop_default_namespace.sql +0 -0
  84. {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0012_user_profile.sql +0 -0
  85. {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0013_hash_version.sql +0 -0
  86. {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0014_access_request_no_fk.sql +0 -0
  87. {memgres-0.6.0 → memgres-0.7.0}/memgres/reembed.py +0 -0
  88. {memgres-0.6.0 → memgres-0.7.0}/memgres/segments.py +0 -0
  89. {memgres-0.6.0 → memgres-0.7.0}/memgres/vector/__init__.py +0 -0
  90. {memgres-0.6.0 → memgres-0.7.0}/memgres/worker.py +0 -0
  91. {memgres-0.6.0 → memgres-0.7.0}/memgres.egg-info/dependency_links.txt +0 -0
  92. {memgres-0.6.0 → memgres-0.7.0}/memgres.egg-info/requires.txt +0 -0
  93. {memgres-0.6.0 → memgres-0.7.0}/memgres.egg-info/top_level.txt +0 -0
  94. {memgres-0.6.0 → memgres-0.7.0}/setup.cfg +0 -0
  95. {memgres-0.6.0 → memgres-0.7.0}/tests/test_config.py +0 -0
  96. {memgres-0.6.0 → memgres-0.7.0}/tests/test_diffing.py +0 -0
  97. {memgres-0.6.0 → memgres-0.7.0}/tests/test_embeddings.py +0 -0
  98. {memgres-0.6.0 → memgres-0.7.0}/tests/test_healthcheck.py +0 -0
  99. {memgres-0.6.0 → memgres-0.7.0}/tests/test_mcp_http_transport.py +0 -0
  100. {memgres-0.6.0 → memgres-0.7.0}/tests/test_qdrant_ca.py +0 -0
  101. {memgres-0.6.0 → memgres-0.7.0}/tests/test_replace_build.py +0 -0
  102. {memgres-0.6.0 → memgres-0.7.0}/tests/test_segments.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: memgres
3
- Version: 0.6.0
3
+ Version: 0.7.0
4
4
  Summary: Drop-in memory for AI agents: one Postgres, lexical + semantic recall, diff-versioned history, GDPR-erasable.
5
5
  Author: mozgsml
6
6
  License-Expression: MIT
@@ -99,7 +99,7 @@ optional layers on top of the same core:
99
99
  └─ MCP server write/recall/get/blame/move/forget as MCP tools
100
100
  ```
101
101
 
102
- **Record model:** one memory = one mutable body (up to a configurable ceiling, default 256 KB) plus metadata — `tags` (cross-cutting labels, `text[]` + GIN), a `path` (its place in an `ltree` tree), timestamps, and per-diff provenance (`source`/`reason`, kept in history). A single write/diff is capped smaller (default 16 KB), so large bodies accrue over many authored diffs. **Organization is two orthogonal axes:** the tree is *where a memory lives* (one place, subtree-selectable); tags are *what it's about* (many, overlapping). Both filter either search — narrow a semantic query to a subtree, or list a tag across the tree.
102
+ **Record model:** one memory = one mutable body (up to a configurable ceiling, default 256 KB) plus metadata — `tags` (cross-cutting labels, `text[]` + GIN), a `path` (its place in an `ltree` tree), timestamps, and per-diff provenance (`source`/`reason`, kept in history). A single write/diff is capped smaller (default 16 KB), so large bodies accrue over many authored diffs. **Organization is two orthogonal axes:** the tree is *where a memory lives* (one place, subtree-selectable); tags are *what it's about* (many, overlapping). Both filter either search — narrow a semantic query to a subtree, or list a tag across the tree. **Links are the third axis:** `[[path]]` in a body becomes a real edge — walkable in BOTH directions, so "what relies on this?" has an answer before you change a fact — and a move rewrites the bodies that named the old address, so a link keeps working and an address copied out of a body is still one. Each memory also records how often it surfaces in search and how often it is opened, which is how the part of a corpus nobody reads becomes visible.
103
103
 
104
104
  **Isolation:** optional multi-tenant identity keeps tenants from seeing each other's memories — users own *namespaces*, and rotatable, permission-scoped *tokens* authenticate as a user (turned on with `MEMGRES_KEY_MODE=open|managed`; see [docs/TENANCY.md](docs/TENANCY.md)). Encryption at rest is left to the deployment — Postgres/managed-PG/disk TDE stays transparent to queries, so search keeps working; memgres deliberately does **not** encrypt bodies application-side (that would make them unsearchable, which is why no comparable tool does it either). GDPR erasure is real: `forget()` hard-deletes the row, its vectors, and crypto-shreds the history chain. All limits are env-configurable, so the same code serves a single-user embed and a capped multi-tenant service.
105
105
 
@@ -116,7 +116,7 @@ docker compose up
116
116
 
117
117
  Defaults suit a single-user setup with no auth. To change limits, the embedding provider, tokens, … drop a `.env` beside it — every `MEMGRES_*` is optional (see [Configuration](#configuration) or [.env.example](.env.example)).
118
118
 
119
- **Give it to an LLM / agent — no code (MCP).** Point any URL-capable MCP client (Cursor, Cline, Claude Desktop, …) at the running server; the model gets `memory_write`, `memory_recall`, `memory_get`, `memory_list`, `memory_blame`, `memory_history`, `memory_move`, `memory_forget`, `memory_server_info` as tools:
119
+ **Give it to an LLM / agent — no code (MCP).** Point any URL-capable MCP client (Cursor, Cline, Claude Desktop, …) at the running server; the model gets `memory_write`, `memory_recall`, `memory_get`, `memory_list`, `memory_tags`, `memory_links`, `memory_blame`, `memory_history`, `memory_move`, `memory_forget`, `memory_server_info` as tools:
120
120
 
121
121
  ```json
122
122
  {
@@ -191,7 +191,7 @@ Not sure which fits? Start with the decision guide: [docs/CHOOSING.md](docs/CHOO
191
191
  1. **`docker compose up`** — `pgvector` + service, nothing to configure. For a dedicated vector service instead, `docker compose --profile qdrant up` and set `MEMGRES_VECTOR_BACKEND=qdrant` (Qdrant ranks vectors; Postgres still holds bodies and does tag/subtree/TTL filtering).
192
192
  2. **Your own Postgres** — install the `[server]` extra (above), point `MEMGRES_DATABASE_URL` at it, run `memgres-server` (migrates on startup).
193
193
  3. **Embedded library** — install the core package, use `Store` directly, no HTTP at all.
194
- 4. **Split service (many clients)** — a stateless API tier that only flags writes plus a scalable `memgres-worker` tier that embeds; see [docs/DEPLOYMENT.md](docs/DEPLOYMENT.md) and `deploy/docker-compose.yml`. Switch the embedding model later with `memgres-reembed`.
194
+ 4. **Split service (many clients)** — a stateless API tier that only flags writes plus a scalable `memgres-worker` tier that embeds; see [docs/DEPLOYMENT.md](docs/DEPLOYMENT.md) and `deploy/docker-compose.yml`. Switch the embedding model later with `memgres-reembed`; rebuild the link graph with `memgres-relink`.
195
195
 
196
196
  Semantic recall is optional: the default `MEMGRES_EMBED_PROVIDER=none` gives you lexical FTS with zero models. Turn on `local` (sentence-transformers), a cloud API (`openai`/`jina`), or any OpenAI-compatible server (LM Studio, Ollama, …) when you want meaning-based search — see [docs/EMBEDDINGS.md](docs/EMBEDDINGS.md) for choosing local vs cloud and [docs/BACKENDS.md](docs/BACKENDS.md) for copy-paste setups. The model id + dimension get stamped into the schema and a later mismatch hard-fails instead of silently returning garbage.
197
197
 
@@ -206,12 +206,16 @@ Everything is env, all optional (defaults suit a single-user embed). Full list i
206
206
  | `MEMGRES_MAX_BODY_BYTES` | `262144` | ceiling for a whole record body (256 KB) |
207
207
  | `MEMGRES_MAX_WRITE_BYTES` | `16384` | ceiling for one write/diff payload (≤ body) |
208
208
  | `MEMGRES_MAX_SOURCE_BYTES` / `_MAX_REASON_BYTES` | `2048` / `1024` | ceilings for a write's `source` / `reason` provenance |
209
- | `MEMGRES_RETENTION_DAYS` | `0` | `0` = keep forever (TTL off); `>0` = expire N days after last touch |
209
+ | `MEMGRES_RETENTION_DAYS` | `0` | how long data is kept: `0` = forever (off); `>0` = expire N days after **last touch**. Operator-only — there is no per-write TTL. Note the clock restarts on a touch, and a read counts as one unless `MEMGRES_RENEW_ON_READ=false`: the window covers data nobody uses |
210
210
  | `MEMGRES_RENEW_ON_READ` | `true` | a read pushes the expiry clock forward |
211
+ | `MEMGRES_RETENTION_SWEEP` | `true` | this process runs the retention sweep. Every server process starts one; set `false` where a dedicated sweeper already runs |
212
+ | `MEMGRES_RETENTION_SWEEP_INTERVAL` | `3600` | seconds between sweeps that DELETE expired rows (and their vectors). Only runs when `RETENTION_DAYS > 0` |
213
+ | `MEMGRES_USAGE_COUNTERS` | `true` | count how often each memory surfaces in search and is read in full (`memory_usage`, a separate table — never the `memory` row, never the hash chain). Off makes reads pure again, for a read-only replica |
211
214
  | `MEMGRES_KEY_MODE` | `single` | `single` (no auth, one space) · `open` (bring-your-own token, self-registers) · `managed` (admin-provisioned). See [docs/TENANCY.md](docs/TENANCY.md) |
212
215
  | `MEMGRES_ADMIN_TOKEN` | — | global admin bearer for provisioning (managed mode) |
213
216
  | `MEMGRES_TOKEN` | — | default token used when a call passes none (single-tenant endpoints) |
214
217
  | `MEMGRES_TREE` | `true` | `ltree` path column + GiST index (fast subtree select) |
218
+ | `MEMGRES_REQUIRE_TITLE` | `true` | `true` = a write that stores content must supply `title`. Captions are what name a memory in results and what title-weighted ranking weighs; `move`/`retag` are exempt (they store no content) |
215
219
  | `MEMGRES_REQUIRE_PARENT` | `false` | `true` = a node's parent path must already exist |
216
220
  | `MEMGRES_HISTORY` | `true` | keep the hash-chained diff history (deleted with the record) |
217
221
  | `MEMGRES_FTS_LANGUAGE` | `simple` | Postgres FTS dictionary (`simple`/`english`/…) |
@@ -327,8 +331,8 @@ pip install "memgres[mcp]"
327
331
  ```
328
332
 
329
333
  Either way the model gets tools `memory_write`, `memory_recall`, `memory_get`,
330
- `memory_list`, `memory_blame`, `memory_history`, `memory_move`, `memory_forget`,
331
- `memory_server_info`. Tell it *"remember X"* / *"what do you know about Y?"* and it
334
+ `memory_list`, `memory_tags`, `memory_links`, `memory_blame`, `memory_history`,
335
+ `memory_move`, `memory_forget`, `memory_server_info`. Tell it *"remember X"* / *"what do you know about Y?"* and it
332
336
  calls them. (For semantic recall add the embedding env vars — see
333
337
  [docs/BACKENDS.md](docs/BACKENDS.md).)
334
338
 
@@ -63,7 +63,7 @@ optional layers on top of the same core:
63
63
  └─ MCP server write/recall/get/blame/move/forget as MCP tools
64
64
  ```
65
65
 
66
- **Record model:** one memory = one mutable body (up to a configurable ceiling, default 256 KB) plus metadata — `tags` (cross-cutting labels, `text[]` + GIN), a `path` (its place in an `ltree` tree), timestamps, and per-diff provenance (`source`/`reason`, kept in history). A single write/diff is capped smaller (default 16 KB), so large bodies accrue over many authored diffs. **Organization is two orthogonal axes:** the tree is *where a memory lives* (one place, subtree-selectable); tags are *what it's about* (many, overlapping). Both filter either search — narrow a semantic query to a subtree, or list a tag across the tree.
66
+ **Record model:** one memory = one mutable body (up to a configurable ceiling, default 256 KB) plus metadata — `tags` (cross-cutting labels, `text[]` + GIN), a `path` (its place in an `ltree` tree), timestamps, and per-diff provenance (`source`/`reason`, kept in history). A single write/diff is capped smaller (default 16 KB), so large bodies accrue over many authored diffs. **Organization is two orthogonal axes:** the tree is *where a memory lives* (one place, subtree-selectable); tags are *what it's about* (many, overlapping). Both filter either search — narrow a semantic query to a subtree, or list a tag across the tree. **Links are the third axis:** `[[path]]` in a body becomes a real edge — walkable in BOTH directions, so "what relies on this?" has an answer before you change a fact — and a move rewrites the bodies that named the old address, so a link keeps working and an address copied out of a body is still one. Each memory also records how often it surfaces in search and how often it is opened, which is how the part of a corpus nobody reads becomes visible.
67
67
 
68
68
  **Isolation:** optional multi-tenant identity keeps tenants from seeing each other's memories — users own *namespaces*, and rotatable, permission-scoped *tokens* authenticate as a user (turned on with `MEMGRES_KEY_MODE=open|managed`; see [docs/TENANCY.md](docs/TENANCY.md)). Encryption at rest is left to the deployment — Postgres/managed-PG/disk TDE stays transparent to queries, so search keeps working; memgres deliberately does **not** encrypt bodies application-side (that would make them unsearchable, which is why no comparable tool does it either). GDPR erasure is real: `forget()` hard-deletes the row, its vectors, and crypto-shreds the history chain. All limits are env-configurable, so the same code serves a single-user embed and a capped multi-tenant service.
69
69
 
@@ -80,7 +80,7 @@ docker compose up
80
80
 
81
81
  Defaults suit a single-user setup with no auth. To change limits, the embedding provider, tokens, … drop a `.env` beside it — every `MEMGRES_*` is optional (see [Configuration](#configuration) or [.env.example](.env.example)).
82
82
 
83
- **Give it to an LLM / agent — no code (MCP).** Point any URL-capable MCP client (Cursor, Cline, Claude Desktop, …) at the running server; the model gets `memory_write`, `memory_recall`, `memory_get`, `memory_list`, `memory_blame`, `memory_history`, `memory_move`, `memory_forget`, `memory_server_info` as tools:
83
+ **Give it to an LLM / agent — no code (MCP).** Point any URL-capable MCP client (Cursor, Cline, Claude Desktop, …) at the running server; the model gets `memory_write`, `memory_recall`, `memory_get`, `memory_list`, `memory_tags`, `memory_links`, `memory_blame`, `memory_history`, `memory_move`, `memory_forget`, `memory_server_info` as tools:
84
84
 
85
85
  ```json
86
86
  {
@@ -155,7 +155,7 @@ Not sure which fits? Start with the decision guide: [docs/CHOOSING.md](docs/CHOO
155
155
  1. **`docker compose up`** — `pgvector` + service, nothing to configure. For a dedicated vector service instead, `docker compose --profile qdrant up` and set `MEMGRES_VECTOR_BACKEND=qdrant` (Qdrant ranks vectors; Postgres still holds bodies and does tag/subtree/TTL filtering).
156
156
  2. **Your own Postgres** — install the `[server]` extra (above), point `MEMGRES_DATABASE_URL` at it, run `memgres-server` (migrates on startup).
157
157
  3. **Embedded library** — install the core package, use `Store` directly, no HTTP at all.
158
- 4. **Split service (many clients)** — a stateless API tier that only flags writes plus a scalable `memgres-worker` tier that embeds; see [docs/DEPLOYMENT.md](docs/DEPLOYMENT.md) and `deploy/docker-compose.yml`. Switch the embedding model later with `memgres-reembed`.
158
+ 4. **Split service (many clients)** — a stateless API tier that only flags writes plus a scalable `memgres-worker` tier that embeds; see [docs/DEPLOYMENT.md](docs/DEPLOYMENT.md) and `deploy/docker-compose.yml`. Switch the embedding model later with `memgres-reembed`; rebuild the link graph with `memgres-relink`.
159
159
 
160
160
  Semantic recall is optional: the default `MEMGRES_EMBED_PROVIDER=none` gives you lexical FTS with zero models. Turn on `local` (sentence-transformers), a cloud API (`openai`/`jina`), or any OpenAI-compatible server (LM Studio, Ollama, …) when you want meaning-based search — see [docs/EMBEDDINGS.md](docs/EMBEDDINGS.md) for choosing local vs cloud and [docs/BACKENDS.md](docs/BACKENDS.md) for copy-paste setups. The model id + dimension get stamped into the schema and a later mismatch hard-fails instead of silently returning garbage.
161
161
 
@@ -170,12 +170,16 @@ Everything is env, all optional (defaults suit a single-user embed). Full list i
170
170
  | `MEMGRES_MAX_BODY_BYTES` | `262144` | ceiling for a whole record body (256 KB) |
171
171
  | `MEMGRES_MAX_WRITE_BYTES` | `16384` | ceiling for one write/diff payload (≤ body) |
172
172
  | `MEMGRES_MAX_SOURCE_BYTES` / `_MAX_REASON_BYTES` | `2048` / `1024` | ceilings for a write's `source` / `reason` provenance |
173
- | `MEMGRES_RETENTION_DAYS` | `0` | `0` = keep forever (TTL off); `>0` = expire N days after last touch |
173
+ | `MEMGRES_RETENTION_DAYS` | `0` | how long data is kept: `0` = forever (off); `>0` = expire N days after **last touch**. Operator-only — there is no per-write TTL. Note the clock restarts on a touch, and a read counts as one unless `MEMGRES_RENEW_ON_READ=false`: the window covers data nobody uses |
174
174
  | `MEMGRES_RENEW_ON_READ` | `true` | a read pushes the expiry clock forward |
175
+ | `MEMGRES_RETENTION_SWEEP` | `true` | this process runs the retention sweep. Every server process starts one; set `false` where a dedicated sweeper already runs |
176
+ | `MEMGRES_RETENTION_SWEEP_INTERVAL` | `3600` | seconds between sweeps that DELETE expired rows (and their vectors). Only runs when `RETENTION_DAYS > 0` |
177
+ | `MEMGRES_USAGE_COUNTERS` | `true` | count how often each memory surfaces in search and is read in full (`memory_usage`, a separate table — never the `memory` row, never the hash chain). Off makes reads pure again, for a read-only replica |
175
178
  | `MEMGRES_KEY_MODE` | `single` | `single` (no auth, one space) · `open` (bring-your-own token, self-registers) · `managed` (admin-provisioned). See [docs/TENANCY.md](docs/TENANCY.md) |
176
179
  | `MEMGRES_ADMIN_TOKEN` | — | global admin bearer for provisioning (managed mode) |
177
180
  | `MEMGRES_TOKEN` | — | default token used when a call passes none (single-tenant endpoints) |
178
181
  | `MEMGRES_TREE` | `true` | `ltree` path column + GiST index (fast subtree select) |
182
+ | `MEMGRES_REQUIRE_TITLE` | `true` | `true` = a write that stores content must supply `title`. Captions are what name a memory in results and what title-weighted ranking weighs; `move`/`retag` are exempt (they store no content) |
179
183
  | `MEMGRES_REQUIRE_PARENT` | `false` | `true` = a node's parent path must already exist |
180
184
  | `MEMGRES_HISTORY` | `true` | keep the hash-chained diff history (deleted with the record) |
181
185
  | `MEMGRES_FTS_LANGUAGE` | `simple` | Postgres FTS dictionary (`simple`/`english`/…) |
@@ -291,8 +295,8 @@ pip install "memgres[mcp]"
291
295
  ```
292
296
 
293
297
  Either way the model gets tools `memory_write`, `memory_recall`, `memory_get`,
294
- `memory_list`, `memory_blame`, `memory_history`, `memory_move`, `memory_forget`,
295
- `memory_server_info`. Tell it *"remember X"* / *"what do you know about Y?"* and it
298
+ `memory_list`, `memory_tags`, `memory_links`, `memory_blame`, `memory_history`,
299
+ `memory_move`, `memory_forget`, `memory_server_info`. Tell it *"remember X"* / *"what do you know about Y?"* and it
296
300
  calls them. (For semantic recall add the embedding env vars — see
297
301
  [docs/BACKENDS.md](docs/BACKENDS.md).)
298
302
 
@@ -22,7 +22,8 @@ from .embeddings import Embedder, get_embedder
22
22
  from .schema import migrate, SchemaMismatch, SCHEMA_VERSION
23
23
  from .search import Hit, recall
24
24
  from .blame import annotate, annotate_grouped, reconstruct, replay
25
- from .store import Store, Memory, Conflict, NotFound, TooLarge, NoParent
25
+ from .store import (Store, Memory, Conflict, NotFound, TooLarge, NoParent,
26
+ MissingTitle)
26
27
  from .identity import (
27
28
  Principal, AuthError, SpaceNotFound,
28
29
  resolve, resolve_space, new_token, valid_format,
@@ -36,6 +37,7 @@ __all__ = [
36
37
  "__version__",
37
38
  "Config", "load_config",
38
39
  "Store", "Memory", "Conflict", "NotFound", "TooLarge", "NoParent",
40
+ "MissingTitle",
39
41
  "make_diff", "apply_diff", "content_hash", "DiffConflict",
40
42
  "Embedder", "get_embedder",
41
43
  "migrate", "SchemaMismatch", "SCHEMA_VERSION",
@@ -8,4 +8,4 @@ here at release; nowhere else carries the number.
8
8
  PEP 440: a ``.devN`` suffix marks an unreleased build ahead of the last tag.
9
9
  """
10
10
 
11
- __version__ = "0.6.0"
11
+ __version__ = "0.7.0"
@@ -72,6 +72,7 @@ class Config:
72
72
  # user_manager (default) | superadmin
73
73
  # organization
74
74
  tree_enabled: bool # ltree path column + GiST index for fast subtree selection
75
+ require_title: bool # True = a write that stores CONTENT must caption it
75
76
  require_parent: bool # False = sparse paths (create food.apple with no food row);
76
77
  # True = a node's parent path must already exist as a memory
77
78
  # history
@@ -106,6 +107,12 @@ class Config:
106
107
  # deployment where an external worker embeds.
107
108
  embed_worker: bool # a server process runs an in-process embed worker
108
109
  embed_worker_interval: float # seconds the idle worker sleeps between drains
110
+ usage_counters: bool # count how often each memory surfaces in search and
111
+ # is read in full. Off makes reads pure again — for
112
+ # a read-only replica, or a deployment unwilling to
113
+ # pay one small write per read.
114
+ retention_sweep: bool # this process runs the retention sweep
115
+ retention_sweep_interval: float # seconds between retention sweeps (see retention_days)
109
116
  embed_max_attempts: int # after this many failed embed attempts a row is a
110
117
  # dead letter — left flagged but out of the claim
111
118
  # rotation (logged), so one poison body can't wedge
@@ -151,6 +158,8 @@ class Config:
151
158
  raise ValueError("MEMGRES_FULL_BODY_MAX_CHARS must be >= 0")
152
159
  if self.embed_worker_interval <= 0:
153
160
  raise ValueError("MEMGRES_EMBED_WORKER_INTERVAL must be > 0")
161
+ if self.retention_sweep_interval <= 0:
162
+ raise ValueError("MEMGRES_RETENTION_SWEEP_INTERVAL must be > 0")
154
163
  if self.embed_max_attempts < 1:
155
164
  raise ValueError("MEMGRES_EMBED_MAX_ATTEMPTS must be >= 1")
156
165
  if self.embed_retry_backoff_s < 0:
@@ -200,6 +209,7 @@ def load() -> Config:
200
209
  admin_token_file=_str("MEMGRES_ADMIN_TOKEN_FILE", ""),
201
210
  admin_role=_str("MEMGRES_ADMIN_ROLE", "user_manager"),
202
211
  tree_enabled=_bool("MEMGRES_TREE", True),
212
+ require_title=_bool("MEMGRES_REQUIRE_TITLE", True),
203
213
  require_parent=_bool("MEMGRES_REQUIRE_PARENT", False),
204
214
  history_enabled=_bool("MEMGRES_HISTORY", True),
205
215
  fts_language=_str("MEMGRES_FTS_LANGUAGE", "simple"),
@@ -216,6 +226,9 @@ def load() -> Config:
216
226
  embed_dispatch=_str("MEMGRES_EMBED_DISPATCH", "inline"),
217
227
  embed_worker=_bool("MEMGRES_EMBED_WORKER", True),
218
228
  embed_worker_interval=_float("MEMGRES_EMBED_WORKER_INTERVAL", 1.0),
229
+ usage_counters=_bool("MEMGRES_USAGE_COUNTERS", True),
230
+ retention_sweep=_bool("MEMGRES_RETENTION_SWEEP", True),
231
+ retention_sweep_interval=_float("MEMGRES_RETENTION_SWEEP_INTERVAL", 3600.0),
219
232
  embed_max_attempts=_int("MEMGRES_EMBED_MAX_ATTEMPTS", 5),
220
233
  embed_retry_backoff_s=_float("MEMGRES_EMBED_RETRY_BACKOFF_S", 60.0),
221
234
  list_preview_chars=_int("MEMGRES_LIST_PREVIEW_CHARS", 120),
@@ -9,84 +9,46 @@ One daemon thread, one dedicated connection. It backfills on start (so a restart
9
9
  catches up any rows left pending, including the one-time re-chunk after the
10
10
  schema upgrade), then polls. ``drain_once`` is the same code the loop runs and is
11
11
  directly callable from a test or a CLI. The real work lives in
12
- :func:`memgres.indexing.drain`; this is just its lifecycle.
12
+ :func:`memgres.indexing.drain`; the lifecycle lives in
13
+ :class:`memgres.periodic.PeriodicWorker`, shared with the retention sweep.
13
14
  """
14
15
 
15
16
  from __future__ import annotations
16
17
 
17
18
  import logging
18
- import threading
19
19
  from typing import Callable, Optional
20
20
 
21
21
  from .indexing import drain
22
+ from .periodic import PeriodicWorker, maybe_start_sweeper
22
23
 
23
24
  _log = logging.getLogger("memgres.embed_worker")
24
25
 
25
26
 
26
- class EmbedWorker:
27
+ class EmbedWorker(PeriodicWorker):
28
+ """Drains ``embed_pending`` — one tick is one drain pass.
29
+
30
+ The first tick IS the backfill (rows left pending across a restart or the
31
+ schema upgrade). It happens in the thread rather than in ``start()``, so
32
+ building a server never blocks on embedding a backlog."""
33
+
34
+ name = "memgres-embed"
35
+
27
36
  def __init__(self, cfg, embedder, backend,
28
37
  connect: Callable[[], "object"]):
29
- self.cfg = cfg
38
+ super().__init__(cfg, connect, cfg.embed_worker_interval)
30
39
  self.embedder = embedder
31
40
  self.backend = backend
32
- self._connect = connect # () -> a fresh psycopg connection
33
- self._conn = None
34
- self._stop = threading.Event()
35
- self._thread: Optional[threading.Thread] = None
36
-
37
- def _conn_ok(self):
38
- if self._conn is None or getattr(self._conn, "closed", False):
39
- self._conn = self._connect()
40
- return self._conn
41
41
 
42
42
  def drain_once(self) -> int:
43
43
  """One synchronous drain pass over all currently-pending rows. Returns the
44
44
  count embedded. Used by the loop and directly by tests."""
45
45
  return drain(self._conn_ok(), self.cfg, self.embedder, self.backend)
46
46
 
47
- def _run(self) -> None:
48
- # The first iteration IS the backfill (catch up rows left pending across a
49
- # restart / the schema upgrade). Done in the thread, not in start(), so
50
- # building a server never blocks on embedding a backlog.
51
- while not self._stop.is_set():
52
- try:
53
- self.drain_once()
54
- except Exception:
55
- _log.exception("embed worker drain failed; dropping connection, retrying")
56
- self._reset_conn()
57
- self._stop.wait(self.cfg.embed_worker_interval)
58
-
59
- def _reset_conn(self) -> None:
60
- try:
61
- if self._conn is not None:
62
- self._conn.close()
63
- except Exception:
64
- pass
65
- self._conn = None
66
-
67
- def start(self) -> "EmbedWorker":
68
- if self._thread is not None:
69
- return self
70
- self._thread = threading.Thread(target=self._run, name="memgres-embed",
71
- daemon=True)
72
- self._thread.start()
73
- return self
74
-
75
- def serve(self) -> None:
76
- """Run the drain loop in the CURRENT thread, blocking until ``stop()``.
77
- Used by the standalone ``memgres-worker`` process (a signal handler calls
78
- ``stop()``); the in-process server path uses ``start()`` instead."""
79
- self._run()
80
- self._reset_conn()
47
+ def _tick(self) -> None:
48
+ self.drain_once()
81
49
 
82
50
  def stop(self) -> None:
83
51
  self._stop.set()
84
- if self._thread is not None:
85
- self._thread.join(timeout=5)
86
- self._thread = None
87
- self._reset_conn()
88
-
89
-
90
52
  def maybe_start_worker(cfg, embedder, backend,
91
53
  connect: Callable[[], "object"]) -> Optional[EmbedWorker]:
92
54
  """Start an :class:`EmbedWorker` when there's something to embed and the
@@ -99,12 +61,17 @@ def maybe_start_worker(cfg, embedder, backend,
99
61
 
100
62
 
101
63
  def wire_server(cfg, embedder):
102
- """Server-side setup shared by the HTTP and MCP entrypoints: build the vector
103
- backend ONCE, start the in-process embed worker if warranted, and return
64
+ """Server-side background setup shared by the HTTP and MCP entrypoints: build
65
+ the vector backend ONCE, start the in-process embed worker if warranted, start
66
+ the retention sweep if the deployment has a retention policy, and return
104
67
  ``(worker, cfg, backend)`` with ``cfg.embed_dispatch`` set to what actually
105
68
  holds. The caller injects ``backend`` into every per-request ``Store`` so a
106
69
  qdrant client isn't rebuilt each call.
107
70
 
71
+ The sweep is started here and not returned: like the embed worker at both
72
+ call sites, nothing holds it — it is a daemon thread that dies with the
73
+ process. Drive it directly (``RetentionSweeper.sweep_once``) in a test.
74
+
108
75
  Dispatch resolution:
109
76
  * an in-process worker started (``embed_worker`` on, embedder present) →
110
77
  writes defer to it → ``async``. The all-in-one server default: fast writes
@@ -120,9 +87,15 @@ def wire_server(cfg, embedder):
120
87
  from .vector.base import make_backend
121
88
 
122
89
  backend = make_backend(cfg, embedder)
123
- worker = maybe_start_worker(
124
- cfg, embedder, backend,
125
- connect=lambda: psycopg.connect(cfg.database_url or ""))
90
+ connect = lambda: psycopg.connect(cfg.database_url or "") # noqa: E731
91
+ worker = maybe_start_worker(cfg, embedder, backend, connect=connect)
92
+ # Retention answers to its own policy, not to whether embeddings are on: a
93
+ # lexical-only deployment must still stop holding expired data.
94
+ maybe_start_sweeper(cfg, connect, embedder, backend)
95
+ # One-time, synchronous: a server that starts serving `memory_links` before
96
+ # the graph exists would answer "nothing points here" for every memory.
97
+ from .relink import maybe_backfill
98
+ maybe_backfill(cfg, connect)
126
99
  dispatch = "async" if worker is not None else cfg.embed_dispatch
127
100
  if dispatch == "async" and worker is None and backend is not None:
128
101
  # async + no local worker: writes will flag embed_pending and this process
@@ -0,0 +1,188 @@
1
+ """`[[wiki links]]` in a body, parsed into edges.
2
+
3
+ The convention arrived on its own: this repo's reference corpus already carried
4
+ 238 `[[…]]` links across 97 memories with no tool support at all, 91% of them
5
+ resolving. What it did not carry was a way to ask "what points HERE" — 42 of
6
+ those 97 memories had no inbound link and were reachable only by search — or any
7
+ way to notice when a link stopped resolving.
8
+
9
+ ## The syntax
10
+
11
+ [[target]] a link
12
+ [[target#anchor]] …to a section of it
13
+ [[target|label]] …shown as something else
14
+ [[target#anchor|label]] both
15
+
16
+ `#` before `|`, as in every wiki dialect since MediaWiki and in Obsidian today,
17
+ so a link written from habit parses. The split is positional and not clever:
18
+ first `|` ends the target part, first `#` inside that part starts the anchor. A
19
+ `#` in the LABEL is therefore just text, which is the reading that surprises
20
+ nobody.
21
+
22
+ ## What is a link, and what is merely square brackets
23
+
24
+ Only two shapes are treated as links, and everything else is left alone:
25
+
26
+ * a **path** — an ltree path in this memory's own namespace (`ops.x402.deploy`);
27
+ * a **scheme** we know (`idea:some-slug`, `file:some-note`) — a pointer at
28
+ another store entirely, recorded as an edge but never resolved here.
29
+
30
+ A URL is not a link for these purposes and neither is prose. The rule is stated
31
+ POSITIVELY — recognise a path or a known scheme — rather than as a list of things
32
+ to exclude, because the exclusions are unbounded and the inclusions are two.
33
+
34
+ Two exclusions are still needed, and both come straight from the corpus:
35
+
36
+ * **code spans and fenced blocks are not scanned.** Documentation that explains
37
+ the syntax writes ``[[path]]`` in backticks, and a validator that flags its own
38
+ documentation trains everyone to ignore it;
39
+ * single-segment targets that name nothing here are still recorded as edges
40
+ rather than dropped. A link to something not yet written is a deliberate move
41
+ ("this deserves a memory") and the tooling must let it stand — dangling, and
42
+ visible as dangling.
43
+ """
44
+
45
+ from __future__ import annotations
46
+
47
+ import re
48
+ from dataclasses import dataclass
49
+ from typing import Dict, List, Optional
50
+
51
+ # Stores other than this one that a memory may legitimately point at. Recorded as
52
+ # edges so they are visible, never resolved — we do not own the address space.
53
+ KNOWN_SCHEMES = ("idea", "file")
54
+
55
+ # An ltree path: labels of [A-Za-z0-9_] joined by dots. Deliberately strict —
56
+ # this is what tells a path apart from a slug belonging to another store, and
57
+ # from prose that happens to sit in double brackets.
58
+ _PATH = re.compile(r"^[A-Za-z0-9_]+(\.[A-Za-z0-9_]+)*$")
59
+
60
+ _LINK = re.compile(r"\[\[([^\[\]\n]+)\]\]")
61
+ _FENCE = re.compile(r"```.*?```|~~~.*?~~~", re.S)
62
+ _CODE_SPAN = re.compile(r"`[^`\n]*`")
63
+
64
+
65
+ @dataclass
66
+ class Link:
67
+ """One parsed link. `target` is verbatim as written; `scheme` is None for an
68
+ ordinary path and the scheme name for a pointer at another store.
69
+
70
+ `start`/`end` are the link's half-open span in the ORIGINAL body — the whole
71
+ `[[…]]`, brackets included. They are what lets a rewriter replace one link
72
+ without touching the rest of the text, and in particular without touching a
73
+ `[[path]]` written inside backticks: the parser never saw it, so it has no
74
+ span, so a rewrite cannot reach it.
75
+ """
76
+ raw_target: str
77
+ label: Optional[str] = None
78
+ anchor: Optional[str] = None
79
+ scheme: Optional[str] = None
80
+ ord: int = 0
81
+ start: int = 0
82
+ end: int = 0
83
+
84
+
85
+ def _blank_code(body: str) -> str:
86
+ """Replace code spans and fenced blocks with spaces of the same length.
87
+
88
+ Same length, not removal, so every offset in the returned text still lines up
89
+ with the original — the parser does not need that today, but an anchor
90
+ resolver or a renderer will, and a scanner that quietly shifts positions is a
91
+ bug waiting for the feature that depends on them.
92
+ """
93
+ def blank(m: re.Match) -> str:
94
+ return "".join(" " if c != "\n" else "\n" for c in m.group(0))
95
+ return _CODE_SPAN.sub(blank, _FENCE.sub(blank, body))
96
+
97
+
98
+ def _classify(target: str) -> Optional[str]:
99
+ """None for a path in this namespace, the scheme name for a known foreign
100
+ store, and the string ``"ignore"`` for everything else."""
101
+ if ":" in target:
102
+ scheme = target.split(":", 1)[0].strip().lower()
103
+ return scheme if scheme in KNOWN_SCHEMES else "ignore"
104
+ return None if _PATH.match(target) else "ignore"
105
+
106
+
107
+ def parse_links(body: Optional[str]) -> List[Link]:
108
+ """Every link in `body`, in the order it appears.
109
+
110
+ Matched against the ORIGINAL text, with the blanked copy used only to decide
111
+ whether a match sits inside code. Reading the fields out of the blanked copy
112
+ instead was harmless while nothing wrote them back — and destructive the
113
+ moment something did: a label containing an inline `code` span came back as
114
+ that many spaces, and a rewrite then committed the loss as a well-formed diff.
115
+ A match may also STRADDLE a blanked region (backticks that swallow a `]]` and
116
+ the `[[` after it), which in the blanked copy looks like one enormous link;
117
+ requiring the span to be untouched by blanking rejects that as well.
118
+ """
119
+ body = body or ""
120
+ blanked = _blank_code(body)
121
+ out: List[Link] = []
122
+ for m in _LINK.finditer(body):
123
+ # Is THIS `[[` inside code? Its own opening brackets are blanked if so.
124
+ # Deliberately not "is any part of the span blanked": a label may legally
125
+ # contain an inline `code` span, and rejecting the whole link for that
126
+ # would silently drop a real edge.
127
+ if blanked[m.start():m.start() + 2] != "[[":
128
+ continue
129
+ inner = m.group(1).strip()
130
+ target_part, _, label = inner.partition("|")
131
+ target, _, anchor = target_part.partition("#")
132
+ target, label, anchor = (target.strip(), label.strip(), anchor.strip())
133
+ if not target:
134
+ continue
135
+ kind = _classify(target)
136
+ if kind == "ignore":
137
+ continue
138
+ out.append(Link(raw_target=target, label=label or None,
139
+ anchor=anchor or None,
140
+ scheme=None if kind is None else kind,
141
+ ord=len(out), start=m.start(), end=m.end()))
142
+ return out
143
+
144
+
145
+ def render(target: str, anchor: Optional[str] = None,
146
+ label: Optional[str] = None) -> str:
147
+ """The canonical text for a link. Inverse of the parse, in its normal form:
148
+ whitespace the author put inside the brackets is not preserved, because there
149
+ is nothing to preserve it for and a single spelling is easier to read."""
150
+ inner = target
151
+ if anchor:
152
+ inner += "#" + anchor
153
+ if label:
154
+ inner += "|" + label
155
+ return f"[[{inner}]]"
156
+
157
+
158
+ def rewrite_targets(body: str, new_targets: Dict[int, str]) -> str:
159
+ """Point selected links somewhere else, leaving the rest of the body alone.
160
+
161
+ `new_targets` maps a link's `ord` — its index among the links this parser
162
+ RECOGNISES, which is exactly what the edge table stores — to the target it
163
+ should now carry. Anchor and label ride along unchanged: a rename moves the
164
+ address, not what the author called it or which section they meant.
165
+
166
+ Addressed by ord rather than by text on purpose. A body that links to the same
167
+ path twice has two edges, and only one of them may be the one to change; and a
168
+ body that mentions the old path in prose or in a code span must not be touched
169
+ at all. Substituting text would get both of those wrong.
170
+
171
+ Assembled in one forward pass over the spans rather than by slicing the whole
172
+ body once per link: a memory at the default 256 KB ceiling can hold tens of
173
+ thousands of links, and rebuilding the string each time made a single rewrite
174
+ quadratic — seconds of CPU while holding row locks.
175
+ """
176
+ if not new_targets:
177
+ return body
178
+ links = sorted((l for l in parse_links(body) if l.ord in new_targets),
179
+ key=lambda l: l.start)
180
+ if not links:
181
+ return body
182
+ parts, at = [], 0
183
+ for l in links:
184
+ parts.append(body[at:l.start])
185
+ parts.append(render(new_targets[l.ord], l.anchor, l.label))
186
+ at = l.end
187
+ parts.append(body[at:])
188
+ return "".join(parts)