memgres 0.6.0__tar.gz → 0.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {memgres-0.6.0 → memgres-0.7.0}/PKG-INFO +11 -7
- {memgres-0.6.0 → memgres-0.7.0}/README.md +10 -6
- {memgres-0.6.0 → memgres-0.7.0}/memgres/__init__.py +3 -1
- {memgres-0.6.0 → memgres-0.7.0}/memgres/_version.py +1 -1
- {memgres-0.6.0 → memgres-0.7.0}/memgres/config.py +13 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/embed_worker.py +31 -58
- memgres-0.7.0/memgres/links.py +188 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/mcp_server.py +102 -29
- memgres-0.7.0/memgres/migrations/0015_normalize_tags.sql +44 -0
- memgres-0.7.0/memgres/migrations/0016_valid_at.sql +16 -0
- memgres-0.7.0/memgres/migrations/0017_memory_link.sql +40 -0
- memgres-0.7.0/memgres/migrations/0018_links_built.sql +12 -0
- memgres-0.7.0/memgres/migrations/0019_memory_usage.sql +36 -0
- memgres-0.7.0/memgres/migrations/0020_memory_usage_no_fk.sql +17 -0
- memgres-0.7.0/memgres/periodic.py +153 -0
- memgres-0.7.0/memgres/relink.py +157 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/schema.py +45 -2
- {memgres-0.6.0 → memgres-0.7.0}/memgres/search.py +51 -34
- {memgres-0.6.0 → memgres-0.7.0}/memgres/server.py +38 -24
- {memgres-0.6.0 → memgres-0.7.0}/memgres/store.py +568 -66
- memgres-0.7.0/memgres/tags.py +90 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/vector/base.py +35 -15
- {memgres-0.6.0 → memgres-0.7.0}/memgres/vector/pgvector.py +4 -2
- {memgres-0.6.0 → memgres-0.7.0}/memgres/vector/qdrant.py +4 -2
- {memgres-0.6.0 → memgres-0.7.0}/memgres.egg-info/PKG-INFO +11 -7
- {memgres-0.6.0 → memgres-0.7.0}/memgres.egg-info/SOURCES.txt +16 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres.egg-info/entry_points.txt +1 -0
- {memgres-0.6.0 → memgres-0.7.0}/pyproject.toml +1 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_blame_integration.py +3 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_chunk_index.py +29 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_claim_and_reembed.py +3 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_embed_worker.py +3 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_identity_integration.py +3 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_lexical_match.py +3 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_limits.py +9 -0
- memgres-0.7.0/tests/test_links.py +738 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_list.py +8 -1
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_mcp_admin_tools.py +6 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_mcp_instructions.py +5 -2
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_mcp_recall_schema.py +29 -1
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_mcp_tool_visibility.py +3 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_migration_upgrade.py +14 -9
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_multi_space_search.py +8 -4
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_path_addressing.py +3 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_qdrant_integration.py +3 -0
- memgres-0.7.0/tests/test_require_title.py +183 -0
- memgres-0.7.0/tests/test_retention.py +270 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_roles_bootstrap.py +3 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_search_integration.py +63 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_security_integration.py +30 -2
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_segments_store.py +3 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_server_info.py +27 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_server_integration.py +17 -1
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_snippets.py +3 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_store_integration.py +47 -10
- memgres-0.7.0/tests/test_tags.py +268 -0
- memgres-0.7.0/tests/test_usage.py +328 -0
- memgres-0.7.0/tests/test_valid_at.py +201 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_write_ergonomics.py +3 -0
- {memgres-0.6.0 → memgres-0.7.0}/LICENSE +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/admin.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/admin_cli.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/blame.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/bootstrap.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/delimiters.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/diffing.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/embeddings.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/healthcheck.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/identity.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/indexing.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/info.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/lines.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0001_core.sql +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0002_identity.sql +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0003_history_author.sql +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0004_title.sql +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0005_chunk_index.sql +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0006_reader_floor.sql +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0007_embed_retry.sql +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0008_service_roles.sql +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0009_create_namespace_right.sql +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0010_namespace_alias.sql +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0011_drop_default_namespace.sql +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0012_user_profile.sql +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0013_hash_version.sql +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/migrations/0014_access_request_no_fk.sql +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/reembed.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/segments.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/vector/__init__.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres/worker.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres.egg-info/dependency_links.txt +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres.egg-info/requires.txt +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/memgres.egg-info/top_level.txt +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/setup.cfg +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_config.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_diffing.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_embeddings.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_healthcheck.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_mcp_http_transport.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_qdrant_ca.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_replace_build.py +0 -0
- {memgres-0.6.0 → memgres-0.7.0}/tests/test_segments.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: memgres
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.7.0
|
|
4
4
|
Summary: Drop-in memory for AI agents: one Postgres, lexical + semantic recall, diff-versioned history, GDPR-erasable.
|
|
5
5
|
Author: mozgsml
|
|
6
6
|
License-Expression: MIT
|
|
@@ -99,7 +99,7 @@ optional layers on top of the same core:
|
|
|
99
99
|
└─ MCP server write/recall/get/blame/move/forget as MCP tools
|
|
100
100
|
```
|
|
101
101
|
|
|
102
|
-
**Record model:** one memory = one mutable body (up to a configurable ceiling, default 256 KB) plus metadata — `tags` (cross-cutting labels, `text[]` + GIN), a `path` (its place in an `ltree` tree), timestamps, and per-diff provenance (`source`/`reason`, kept in history). A single write/diff is capped smaller (default 16 KB), so large bodies accrue over many authored diffs. **Organization is two orthogonal axes:** the tree is *where a memory lives* (one place, subtree-selectable); tags are *what it's about* (many, overlapping). Both filter either search — narrow a semantic query to a subtree, or list a tag across the tree.
|
|
102
|
+
**Record model:** one memory = one mutable body (up to a configurable ceiling, default 256 KB) plus metadata — `tags` (cross-cutting labels, `text[]` + GIN), a `path` (its place in an `ltree` tree), timestamps, and per-diff provenance (`source`/`reason`, kept in history). A single write/diff is capped smaller (default 16 KB), so large bodies accrue over many authored diffs. **Organization is two orthogonal axes:** the tree is *where a memory lives* (one place, subtree-selectable); tags are *what it's about* (many, overlapping). Both filter either search — narrow a semantic query to a subtree, or list a tag across the tree. **Links are the third axis:** `[[path]]` in a body becomes a real edge — walkable in BOTH directions, so "what relies on this?" has an answer before you change a fact — and a move rewrites the bodies that named the old address, so a link keeps working and an address copied out of a body is still one. Each memory also records how often it surfaces in search and how often it is opened, which is how the part of a corpus nobody reads becomes visible.
|
|
103
103
|
|
|
104
104
|
**Isolation:** optional multi-tenant identity keeps tenants from seeing each other's memories — users own *namespaces*, and rotatable, permission-scoped *tokens* authenticate as a user (turned on with `MEMGRES_KEY_MODE=open|managed`; see [docs/TENANCY.md](docs/TENANCY.md)). Encryption at rest is left to the deployment — Postgres/managed-PG/disk TDE stays transparent to queries, so search keeps working; memgres deliberately does **not** encrypt bodies application-side (that would make them unsearchable, which is why no comparable tool does it either). GDPR erasure is real: `forget()` hard-deletes the row, its vectors, and crypto-shreds the history chain. All limits are env-configurable, so the same code serves a single-user embed and a capped multi-tenant service.
|
|
105
105
|
|
|
@@ -116,7 +116,7 @@ docker compose up
|
|
|
116
116
|
|
|
117
117
|
Defaults suit a single-user setup with no auth. To change limits, the embedding provider, tokens, … drop a `.env` beside it — every `MEMGRES_*` is optional (see [Configuration](#configuration) or [.env.example](.env.example)).
|
|
118
118
|
|
|
119
|
-
**Give it to an LLM / agent — no code (MCP).** Point any URL-capable MCP client (Cursor, Cline, Claude Desktop, …) at the running server; the model gets `memory_write`, `memory_recall`, `memory_get`, `memory_list`, `memory_blame`, `memory_history`, `memory_move`, `memory_forget`, `memory_server_info` as tools:
|
|
119
|
+
**Give it to an LLM / agent — no code (MCP).** Point any URL-capable MCP client (Cursor, Cline, Claude Desktop, …) at the running server; the model gets `memory_write`, `memory_recall`, `memory_get`, `memory_list`, `memory_tags`, `memory_links`, `memory_blame`, `memory_history`, `memory_move`, `memory_forget`, `memory_server_info` as tools:
|
|
120
120
|
|
|
121
121
|
```json
|
|
122
122
|
{
|
|
@@ -191,7 +191,7 @@ Not sure which fits? Start with the decision guide: [docs/CHOOSING.md](docs/CHOO
|
|
|
191
191
|
1. **`docker compose up`** — `pgvector` + service, nothing to configure. For a dedicated vector service instead, `docker compose --profile qdrant up` and set `MEMGRES_VECTOR_BACKEND=qdrant` (Qdrant ranks vectors; Postgres still holds bodies and does tag/subtree/TTL filtering).
|
|
192
192
|
2. **Your own Postgres** — install the `[server]` extra (above), point `MEMGRES_DATABASE_URL` at it, run `memgres-server` (migrates on startup).
|
|
193
193
|
3. **Embedded library** — install the core package, use `Store` directly, no HTTP at all.
|
|
194
|
-
4. **Split service (many clients)** — a stateless API tier that only flags writes plus a scalable `memgres-worker` tier that embeds; see [docs/DEPLOYMENT.md](docs/DEPLOYMENT.md) and `deploy/docker-compose.yml`. Switch the embedding model later with `memgres-reembed`.
|
|
194
|
+
4. **Split service (many clients)** — a stateless API tier that only flags writes plus a scalable `memgres-worker` tier that embeds; see [docs/DEPLOYMENT.md](docs/DEPLOYMENT.md) and `deploy/docker-compose.yml`. Switch the embedding model later with `memgres-reembed`; rebuild the link graph with `memgres-relink`.
|
|
195
195
|
|
|
196
196
|
Semantic recall is optional: the default `MEMGRES_EMBED_PROVIDER=none` gives you lexical FTS with zero models. Turn on `local` (sentence-transformers), a cloud API (`openai`/`jina`), or any OpenAI-compatible server (LM Studio, Ollama, …) when you want meaning-based search — see [docs/EMBEDDINGS.md](docs/EMBEDDINGS.md) for choosing local vs cloud and [docs/BACKENDS.md](docs/BACKENDS.md) for copy-paste setups. The model id + dimension get stamped into the schema and a later mismatch hard-fails instead of silently returning garbage.
|
|
197
197
|
|
|
@@ -206,12 +206,16 @@ Everything is env, all optional (defaults suit a single-user embed). Full list i
|
|
|
206
206
|
| `MEMGRES_MAX_BODY_BYTES` | `262144` | ceiling for a whole record body (256 KB) |
|
|
207
207
|
| `MEMGRES_MAX_WRITE_BYTES` | `16384` | ceiling for one write/diff payload (≤ body) |
|
|
208
208
|
| `MEMGRES_MAX_SOURCE_BYTES` / `_MAX_REASON_BYTES` | `2048` / `1024` | ceilings for a write's `source` / `reason` provenance |
|
|
209
|
-
| `MEMGRES_RETENTION_DAYS` | `0` | `0` =
|
|
209
|
+
| `MEMGRES_RETENTION_DAYS` | `0` | how long data is kept: `0` = forever (off); `>0` = expire N days after **last touch**. Operator-only — there is no per-write TTL. Note the clock restarts on a touch, and a read counts as one unless `MEMGRES_RENEW_ON_READ=false`: the window covers data nobody uses |
|
|
210
210
|
| `MEMGRES_RENEW_ON_READ` | `true` | a read pushes the expiry clock forward |
|
|
211
|
+
| `MEMGRES_RETENTION_SWEEP` | `true` | this process runs the retention sweep. Every server process starts one; set `false` where a dedicated sweeper already runs |
|
|
212
|
+
| `MEMGRES_RETENTION_SWEEP_INTERVAL` | `3600` | seconds between sweeps that DELETE expired rows (and their vectors). Only runs when `RETENTION_DAYS > 0` |
|
|
213
|
+
| `MEMGRES_USAGE_COUNTERS` | `true` | count how often each memory surfaces in search and is read in full (`memory_usage`, a separate table — never the `memory` row, never the hash chain). Off makes reads pure again, for a read-only replica |
|
|
211
214
|
| `MEMGRES_KEY_MODE` | `single` | `single` (no auth, one space) · `open` (bring-your-own token, self-registers) · `managed` (admin-provisioned). See [docs/TENANCY.md](docs/TENANCY.md) |
|
|
212
215
|
| `MEMGRES_ADMIN_TOKEN` | — | global admin bearer for provisioning (managed mode) |
|
|
213
216
|
| `MEMGRES_TOKEN` | — | default token used when a call passes none (single-tenant endpoints) |
|
|
214
217
|
| `MEMGRES_TREE` | `true` | `ltree` path column + GiST index (fast subtree select) |
|
|
218
|
+
| `MEMGRES_REQUIRE_TITLE` | `true` | `true` = a write that stores content must supply `title`. Captions are what name a memory in results and what title-weighted ranking weighs; `move`/`retag` are exempt (they store no content) |
|
|
215
219
|
| `MEMGRES_REQUIRE_PARENT` | `false` | `true` = a node's parent path must already exist |
|
|
216
220
|
| `MEMGRES_HISTORY` | `true` | keep the hash-chained diff history (deleted with the record) |
|
|
217
221
|
| `MEMGRES_FTS_LANGUAGE` | `simple` | Postgres FTS dictionary (`simple`/`english`/…) |
|
|
@@ -327,8 +331,8 @@ pip install "memgres[mcp]"
|
|
|
327
331
|
```
|
|
328
332
|
|
|
329
333
|
Either way the model gets tools `memory_write`, `memory_recall`, `memory_get`,
|
|
330
|
-
`memory_list`, `
|
|
331
|
-
`memory_server_info`. Tell it *"remember X"* / *"what do you know about Y?"* and it
|
|
334
|
+
`memory_list`, `memory_tags`, `memory_links`, `memory_blame`, `memory_history`,
|
|
335
|
+
`memory_move`, `memory_forget`, `memory_server_info`. Tell it *"remember X"* / *"what do you know about Y?"* and it
|
|
332
336
|
calls them. (For semantic recall add the embedding env vars — see
|
|
333
337
|
[docs/BACKENDS.md](docs/BACKENDS.md).)
|
|
334
338
|
|
|
@@ -63,7 +63,7 @@ optional layers on top of the same core:
|
|
|
63
63
|
└─ MCP server write/recall/get/blame/move/forget as MCP tools
|
|
64
64
|
```
|
|
65
65
|
|
|
66
|
-
**Record model:** one memory = one mutable body (up to a configurable ceiling, default 256 KB) plus metadata — `tags` (cross-cutting labels, `text[]` + GIN), a `path` (its place in an `ltree` tree), timestamps, and per-diff provenance (`source`/`reason`, kept in history). A single write/diff is capped smaller (default 16 KB), so large bodies accrue over many authored diffs. **Organization is two orthogonal axes:** the tree is *where a memory lives* (one place, subtree-selectable); tags are *what it's about* (many, overlapping). Both filter either search — narrow a semantic query to a subtree, or list a tag across the tree.
|
|
66
|
+
**Record model:** one memory = one mutable body (up to a configurable ceiling, default 256 KB) plus metadata — `tags` (cross-cutting labels, `text[]` + GIN), a `path` (its place in an `ltree` tree), timestamps, and per-diff provenance (`source`/`reason`, kept in history). A single write/diff is capped smaller (default 16 KB), so large bodies accrue over many authored diffs. **Organization is two orthogonal axes:** the tree is *where a memory lives* (one place, subtree-selectable); tags are *what it's about* (many, overlapping). Both filter either search — narrow a semantic query to a subtree, or list a tag across the tree. **Links are the third axis:** `[[path]]` in a body becomes a real edge — walkable in BOTH directions, so "what relies on this?" has an answer before you change a fact — and a move rewrites the bodies that named the old address, so a link keeps working and an address copied out of a body is still one. Each memory also records how often it surfaces in search and how often it is opened, which is how the part of a corpus nobody reads becomes visible.
|
|
67
67
|
|
|
68
68
|
**Isolation:** optional multi-tenant identity keeps tenants from seeing each other's memories — users own *namespaces*, and rotatable, permission-scoped *tokens* authenticate as a user (turned on with `MEMGRES_KEY_MODE=open|managed`; see [docs/TENANCY.md](docs/TENANCY.md)). Encryption at rest is left to the deployment — Postgres/managed-PG/disk TDE stays transparent to queries, so search keeps working; memgres deliberately does **not** encrypt bodies application-side (that would make them unsearchable, which is why no comparable tool does it either). GDPR erasure is real: `forget()` hard-deletes the row, its vectors, and crypto-shreds the history chain. All limits are env-configurable, so the same code serves a single-user embed and a capped multi-tenant service.
|
|
69
69
|
|
|
@@ -80,7 +80,7 @@ docker compose up
|
|
|
80
80
|
|
|
81
81
|
Defaults suit a single-user setup with no auth. To change limits, the embedding provider, tokens, … drop a `.env` beside it — every `MEMGRES_*` is optional (see [Configuration](#configuration) or [.env.example](.env.example)).
|
|
82
82
|
|
|
83
|
-
**Give it to an LLM / agent — no code (MCP).** Point any URL-capable MCP client (Cursor, Cline, Claude Desktop, …) at the running server; the model gets `memory_write`, `memory_recall`, `memory_get`, `memory_list`, `memory_blame`, `memory_history`, `memory_move`, `memory_forget`, `memory_server_info` as tools:
|
|
83
|
+
**Give it to an LLM / agent — no code (MCP).** Point any URL-capable MCP client (Cursor, Cline, Claude Desktop, …) at the running server; the model gets `memory_write`, `memory_recall`, `memory_get`, `memory_list`, `memory_tags`, `memory_links`, `memory_blame`, `memory_history`, `memory_move`, `memory_forget`, `memory_server_info` as tools:
|
|
84
84
|
|
|
85
85
|
```json
|
|
86
86
|
{
|
|
@@ -155,7 +155,7 @@ Not sure which fits? Start with the decision guide: [docs/CHOOSING.md](docs/CHOO
|
|
|
155
155
|
1. **`docker compose up`** — `pgvector` + service, nothing to configure. For a dedicated vector service instead, `docker compose --profile qdrant up` and set `MEMGRES_VECTOR_BACKEND=qdrant` (Qdrant ranks vectors; Postgres still holds bodies and does tag/subtree/TTL filtering).
|
|
156
156
|
2. **Your own Postgres** — install the `[server]` extra (above), point `MEMGRES_DATABASE_URL` at it, run `memgres-server` (migrates on startup).
|
|
157
157
|
3. **Embedded library** — install the core package, use `Store` directly, no HTTP at all.
|
|
158
|
-
4. **Split service (many clients)** — a stateless API tier that only flags writes plus a scalable `memgres-worker` tier that embeds; see [docs/DEPLOYMENT.md](docs/DEPLOYMENT.md) and `deploy/docker-compose.yml`. Switch the embedding model later with `memgres-reembed`.
|
|
158
|
+
4. **Split service (many clients)** — a stateless API tier that only flags writes plus a scalable `memgres-worker` tier that embeds; see [docs/DEPLOYMENT.md](docs/DEPLOYMENT.md) and `deploy/docker-compose.yml`. Switch the embedding model later with `memgres-reembed`; rebuild the link graph with `memgres-relink`.
|
|
159
159
|
|
|
160
160
|
Semantic recall is optional: the default `MEMGRES_EMBED_PROVIDER=none` gives you lexical FTS with zero models. Turn on `local` (sentence-transformers), a cloud API (`openai`/`jina`), or any OpenAI-compatible server (LM Studio, Ollama, …) when you want meaning-based search — see [docs/EMBEDDINGS.md](docs/EMBEDDINGS.md) for choosing local vs cloud and [docs/BACKENDS.md](docs/BACKENDS.md) for copy-paste setups. The model id + dimension get stamped into the schema and a later mismatch hard-fails instead of silently returning garbage.
|
|
161
161
|
|
|
@@ -170,12 +170,16 @@ Everything is env, all optional (defaults suit a single-user embed). Full list i
|
|
|
170
170
|
| `MEMGRES_MAX_BODY_BYTES` | `262144` | ceiling for a whole record body (256 KB) |
|
|
171
171
|
| `MEMGRES_MAX_WRITE_BYTES` | `16384` | ceiling for one write/diff payload (≤ body) |
|
|
172
172
|
| `MEMGRES_MAX_SOURCE_BYTES` / `_MAX_REASON_BYTES` | `2048` / `1024` | ceilings for a write's `source` / `reason` provenance |
|
|
173
|
-
| `MEMGRES_RETENTION_DAYS` | `0` | `0` =
|
|
173
|
+
| `MEMGRES_RETENTION_DAYS` | `0` | how long data is kept: `0` = forever (off); `>0` = expire N days after **last touch**. Operator-only — there is no per-write TTL. Note the clock restarts on a touch, and a read counts as one unless `MEMGRES_RENEW_ON_READ=false`: the window covers data nobody uses |
|
|
174
174
|
| `MEMGRES_RENEW_ON_READ` | `true` | a read pushes the expiry clock forward |
|
|
175
|
+
| `MEMGRES_RETENTION_SWEEP` | `true` | this process runs the retention sweep. Every server process starts one; set `false` where a dedicated sweeper already runs |
|
|
176
|
+
| `MEMGRES_RETENTION_SWEEP_INTERVAL` | `3600` | seconds between sweeps that DELETE expired rows (and their vectors). Only runs when `RETENTION_DAYS > 0` |
|
|
177
|
+
| `MEMGRES_USAGE_COUNTERS` | `true` | count how often each memory surfaces in search and is read in full (`memory_usage`, a separate table — never the `memory` row, never the hash chain). Off makes reads pure again, for a read-only replica |
|
|
175
178
|
| `MEMGRES_KEY_MODE` | `single` | `single` (no auth, one space) · `open` (bring-your-own token, self-registers) · `managed` (admin-provisioned). See [docs/TENANCY.md](docs/TENANCY.md) |
|
|
176
179
|
| `MEMGRES_ADMIN_TOKEN` | — | global admin bearer for provisioning (managed mode) |
|
|
177
180
|
| `MEMGRES_TOKEN` | — | default token used when a call passes none (single-tenant endpoints) |
|
|
178
181
|
| `MEMGRES_TREE` | `true` | `ltree` path column + GiST index (fast subtree select) |
|
|
182
|
+
| `MEMGRES_REQUIRE_TITLE` | `true` | `true` = a write that stores content must supply `title`. Captions are what name a memory in results and what title-weighted ranking weighs; `move`/`retag` are exempt (they store no content) |
|
|
179
183
|
| `MEMGRES_REQUIRE_PARENT` | `false` | `true` = a node's parent path must already exist |
|
|
180
184
|
| `MEMGRES_HISTORY` | `true` | keep the hash-chained diff history (deleted with the record) |
|
|
181
185
|
| `MEMGRES_FTS_LANGUAGE` | `simple` | Postgres FTS dictionary (`simple`/`english`/…) |
|
|
@@ -291,8 +295,8 @@ pip install "memgres[mcp]"
|
|
|
291
295
|
```
|
|
292
296
|
|
|
293
297
|
Either way the model gets tools `memory_write`, `memory_recall`, `memory_get`,
|
|
294
|
-
`memory_list`, `
|
|
295
|
-
`memory_server_info`. Tell it *"remember X"* / *"what do you know about Y?"* and it
|
|
298
|
+
`memory_list`, `memory_tags`, `memory_links`, `memory_blame`, `memory_history`,
|
|
299
|
+
`memory_move`, `memory_forget`, `memory_server_info`. Tell it *"remember X"* / *"what do you know about Y?"* and it
|
|
296
300
|
calls them. (For semantic recall add the embedding env vars — see
|
|
297
301
|
[docs/BACKENDS.md](docs/BACKENDS.md).)
|
|
298
302
|
|
|
@@ -22,7 +22,8 @@ from .embeddings import Embedder, get_embedder
|
|
|
22
22
|
from .schema import migrate, SchemaMismatch, SCHEMA_VERSION
|
|
23
23
|
from .search import Hit, recall
|
|
24
24
|
from .blame import annotate, annotate_grouped, reconstruct, replay
|
|
25
|
-
from .store import Store, Memory, Conflict, NotFound, TooLarge, NoParent
|
|
25
|
+
from .store import (Store, Memory, Conflict, NotFound, TooLarge, NoParent,
|
|
26
|
+
MissingTitle)
|
|
26
27
|
from .identity import (
|
|
27
28
|
Principal, AuthError, SpaceNotFound,
|
|
28
29
|
resolve, resolve_space, new_token, valid_format,
|
|
@@ -36,6 +37,7 @@ __all__ = [
|
|
|
36
37
|
"__version__",
|
|
37
38
|
"Config", "load_config",
|
|
38
39
|
"Store", "Memory", "Conflict", "NotFound", "TooLarge", "NoParent",
|
|
40
|
+
"MissingTitle",
|
|
39
41
|
"make_diff", "apply_diff", "content_hash", "DiffConflict",
|
|
40
42
|
"Embedder", "get_embedder",
|
|
41
43
|
"migrate", "SchemaMismatch", "SCHEMA_VERSION",
|
|
@@ -72,6 +72,7 @@ class Config:
|
|
|
72
72
|
# user_manager (default) | superadmin
|
|
73
73
|
# organization
|
|
74
74
|
tree_enabled: bool # ltree path column + GiST index for fast subtree selection
|
|
75
|
+
require_title: bool # True = a write that stores CONTENT must caption it
|
|
75
76
|
require_parent: bool # False = sparse paths (create food.apple with no food row);
|
|
76
77
|
# True = a node's parent path must already exist as a memory
|
|
77
78
|
# history
|
|
@@ -106,6 +107,12 @@ class Config:
|
|
|
106
107
|
# deployment where an external worker embeds.
|
|
107
108
|
embed_worker: bool # a server process runs an in-process embed worker
|
|
108
109
|
embed_worker_interval: float # seconds the idle worker sleeps between drains
|
|
110
|
+
usage_counters: bool # count how often each memory surfaces in search and
|
|
111
|
+
# is read in full. Off makes reads pure again — for
|
|
112
|
+
# a read-only replica, or a deployment unwilling to
|
|
113
|
+
# pay one small write per read.
|
|
114
|
+
retention_sweep: bool # this process runs the retention sweep
|
|
115
|
+
retention_sweep_interval: float # seconds between retention sweeps (see retention_days)
|
|
109
116
|
embed_max_attempts: int # after this many failed embed attempts a row is a
|
|
110
117
|
# dead letter — left flagged but out of the claim
|
|
111
118
|
# rotation (logged), so one poison body can't wedge
|
|
@@ -151,6 +158,8 @@ class Config:
|
|
|
151
158
|
raise ValueError("MEMGRES_FULL_BODY_MAX_CHARS must be >= 0")
|
|
152
159
|
if self.embed_worker_interval <= 0:
|
|
153
160
|
raise ValueError("MEMGRES_EMBED_WORKER_INTERVAL must be > 0")
|
|
161
|
+
if self.retention_sweep_interval <= 0:
|
|
162
|
+
raise ValueError("MEMGRES_RETENTION_SWEEP_INTERVAL must be > 0")
|
|
154
163
|
if self.embed_max_attempts < 1:
|
|
155
164
|
raise ValueError("MEMGRES_EMBED_MAX_ATTEMPTS must be >= 1")
|
|
156
165
|
if self.embed_retry_backoff_s < 0:
|
|
@@ -200,6 +209,7 @@ def load() -> Config:
|
|
|
200
209
|
admin_token_file=_str("MEMGRES_ADMIN_TOKEN_FILE", ""),
|
|
201
210
|
admin_role=_str("MEMGRES_ADMIN_ROLE", "user_manager"),
|
|
202
211
|
tree_enabled=_bool("MEMGRES_TREE", True),
|
|
212
|
+
require_title=_bool("MEMGRES_REQUIRE_TITLE", True),
|
|
203
213
|
require_parent=_bool("MEMGRES_REQUIRE_PARENT", False),
|
|
204
214
|
history_enabled=_bool("MEMGRES_HISTORY", True),
|
|
205
215
|
fts_language=_str("MEMGRES_FTS_LANGUAGE", "simple"),
|
|
@@ -216,6 +226,9 @@ def load() -> Config:
|
|
|
216
226
|
embed_dispatch=_str("MEMGRES_EMBED_DISPATCH", "inline"),
|
|
217
227
|
embed_worker=_bool("MEMGRES_EMBED_WORKER", True),
|
|
218
228
|
embed_worker_interval=_float("MEMGRES_EMBED_WORKER_INTERVAL", 1.0),
|
|
229
|
+
usage_counters=_bool("MEMGRES_USAGE_COUNTERS", True),
|
|
230
|
+
retention_sweep=_bool("MEMGRES_RETENTION_SWEEP", True),
|
|
231
|
+
retention_sweep_interval=_float("MEMGRES_RETENTION_SWEEP_INTERVAL", 3600.0),
|
|
219
232
|
embed_max_attempts=_int("MEMGRES_EMBED_MAX_ATTEMPTS", 5),
|
|
220
233
|
embed_retry_backoff_s=_float("MEMGRES_EMBED_RETRY_BACKOFF_S", 60.0),
|
|
221
234
|
list_preview_chars=_int("MEMGRES_LIST_PREVIEW_CHARS", 120),
|
|
@@ -9,84 +9,46 @@ One daemon thread, one dedicated connection. It backfills on start (so a restart
|
|
|
9
9
|
catches up any rows left pending, including the one-time re-chunk after the
|
|
10
10
|
schema upgrade), then polls. ``drain_once`` is the same code the loop runs and is
|
|
11
11
|
directly callable from a test or a CLI. The real work lives in
|
|
12
|
-
:func:`memgres.indexing.drain`;
|
|
12
|
+
:func:`memgres.indexing.drain`; the lifecycle lives in
|
|
13
|
+
:class:`memgres.periodic.PeriodicWorker`, shared with the retention sweep.
|
|
13
14
|
"""
|
|
14
15
|
|
|
15
16
|
from __future__ import annotations
|
|
16
17
|
|
|
17
18
|
import logging
|
|
18
|
-
import threading
|
|
19
19
|
from typing import Callable, Optional
|
|
20
20
|
|
|
21
21
|
from .indexing import drain
|
|
22
|
+
from .periodic import PeriodicWorker, maybe_start_sweeper
|
|
22
23
|
|
|
23
24
|
_log = logging.getLogger("memgres.embed_worker")
|
|
24
25
|
|
|
25
26
|
|
|
26
|
-
class EmbedWorker:
|
|
27
|
+
class EmbedWorker(PeriodicWorker):
|
|
28
|
+
"""Drains ``embed_pending`` — one tick is one drain pass.
|
|
29
|
+
|
|
30
|
+
The first tick IS the backfill (rows left pending across a restart or the
|
|
31
|
+
schema upgrade). It happens in the thread rather than in ``start()``, so
|
|
32
|
+
building a server never blocks on embedding a backlog."""
|
|
33
|
+
|
|
34
|
+
name = "memgres-embed"
|
|
35
|
+
|
|
27
36
|
def __init__(self, cfg, embedder, backend,
|
|
28
37
|
connect: Callable[[], "object"]):
|
|
29
|
-
|
|
38
|
+
super().__init__(cfg, connect, cfg.embed_worker_interval)
|
|
30
39
|
self.embedder = embedder
|
|
31
40
|
self.backend = backend
|
|
32
|
-
self._connect = connect # () -> a fresh psycopg connection
|
|
33
|
-
self._conn = None
|
|
34
|
-
self._stop = threading.Event()
|
|
35
|
-
self._thread: Optional[threading.Thread] = None
|
|
36
|
-
|
|
37
|
-
def _conn_ok(self):
|
|
38
|
-
if self._conn is None or getattr(self._conn, "closed", False):
|
|
39
|
-
self._conn = self._connect()
|
|
40
|
-
return self._conn
|
|
41
41
|
|
|
42
42
|
def drain_once(self) -> int:
|
|
43
43
|
"""One synchronous drain pass over all currently-pending rows. Returns the
|
|
44
44
|
count embedded. Used by the loop and directly by tests."""
|
|
45
45
|
return drain(self._conn_ok(), self.cfg, self.embedder, self.backend)
|
|
46
46
|
|
|
47
|
-
def
|
|
48
|
-
|
|
49
|
-
# restart / the schema upgrade). Done in the thread, not in start(), so
|
|
50
|
-
# building a server never blocks on embedding a backlog.
|
|
51
|
-
while not self._stop.is_set():
|
|
52
|
-
try:
|
|
53
|
-
self.drain_once()
|
|
54
|
-
except Exception:
|
|
55
|
-
_log.exception("embed worker drain failed; dropping connection, retrying")
|
|
56
|
-
self._reset_conn()
|
|
57
|
-
self._stop.wait(self.cfg.embed_worker_interval)
|
|
58
|
-
|
|
59
|
-
def _reset_conn(self) -> None:
|
|
60
|
-
try:
|
|
61
|
-
if self._conn is not None:
|
|
62
|
-
self._conn.close()
|
|
63
|
-
except Exception:
|
|
64
|
-
pass
|
|
65
|
-
self._conn = None
|
|
66
|
-
|
|
67
|
-
def start(self) -> "EmbedWorker":
|
|
68
|
-
if self._thread is not None:
|
|
69
|
-
return self
|
|
70
|
-
self._thread = threading.Thread(target=self._run, name="memgres-embed",
|
|
71
|
-
daemon=True)
|
|
72
|
-
self._thread.start()
|
|
73
|
-
return self
|
|
74
|
-
|
|
75
|
-
def serve(self) -> None:
|
|
76
|
-
"""Run the drain loop in the CURRENT thread, blocking until ``stop()``.
|
|
77
|
-
Used by the standalone ``memgres-worker`` process (a signal handler calls
|
|
78
|
-
``stop()``); the in-process server path uses ``start()`` instead."""
|
|
79
|
-
self._run()
|
|
80
|
-
self._reset_conn()
|
|
47
|
+
def _tick(self) -> None:
|
|
48
|
+
self.drain_once()
|
|
81
49
|
|
|
82
50
|
def stop(self) -> None:
|
|
83
51
|
self._stop.set()
|
|
84
|
-
if self._thread is not None:
|
|
85
|
-
self._thread.join(timeout=5)
|
|
86
|
-
self._thread = None
|
|
87
|
-
self._reset_conn()
|
|
88
|
-
|
|
89
|
-
|
|
90
52
|
def maybe_start_worker(cfg, embedder, backend,
|
|
91
53
|
connect: Callable[[], "object"]) -> Optional[EmbedWorker]:
|
|
92
54
|
"""Start an :class:`EmbedWorker` when there's something to embed and the
|
|
@@ -99,12 +61,17 @@ def maybe_start_worker(cfg, embedder, backend,
|
|
|
99
61
|
|
|
100
62
|
|
|
101
63
|
def wire_server(cfg, embedder):
|
|
102
|
-
"""Server-side setup shared by the HTTP and MCP entrypoints: build
|
|
103
|
-
backend ONCE, start the in-process embed worker if warranted,
|
|
64
|
+
"""Server-side background setup shared by the HTTP and MCP entrypoints: build
|
|
65
|
+
the vector backend ONCE, start the in-process embed worker if warranted, start
|
|
66
|
+
the retention sweep if the deployment has a retention policy, and return
|
|
104
67
|
``(worker, cfg, backend)`` with ``cfg.embed_dispatch`` set to what actually
|
|
105
68
|
holds. The caller injects ``backend`` into every per-request ``Store`` so a
|
|
106
69
|
qdrant client isn't rebuilt each call.
|
|
107
70
|
|
|
71
|
+
The sweep is started here and not returned: like the embed worker at both
|
|
72
|
+
call sites, nothing holds it — it is a daemon thread that dies with the
|
|
73
|
+
process. Drive it directly (``RetentionSweeper.sweep_once``) in a test.
|
|
74
|
+
|
|
108
75
|
Dispatch resolution:
|
|
109
76
|
* an in-process worker started (``embed_worker`` on, embedder present) →
|
|
110
77
|
writes defer to it → ``async``. The all-in-one server default: fast writes
|
|
@@ -120,9 +87,15 @@ def wire_server(cfg, embedder):
|
|
|
120
87
|
from .vector.base import make_backend
|
|
121
88
|
|
|
122
89
|
backend = make_backend(cfg, embedder)
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
90
|
+
connect = lambda: psycopg.connect(cfg.database_url or "") # noqa: E731
|
|
91
|
+
worker = maybe_start_worker(cfg, embedder, backend, connect=connect)
|
|
92
|
+
# Retention answers to its own policy, not to whether embeddings are on: a
|
|
93
|
+
# lexical-only deployment must still stop holding expired data.
|
|
94
|
+
maybe_start_sweeper(cfg, connect, embedder, backend)
|
|
95
|
+
# One-time, synchronous: a server that starts serving `memory_links` before
|
|
96
|
+
# the graph exists would answer "nothing points here" for every memory.
|
|
97
|
+
from .relink import maybe_backfill
|
|
98
|
+
maybe_backfill(cfg, connect)
|
|
126
99
|
dispatch = "async" if worker is not None else cfg.embed_dispatch
|
|
127
100
|
if dispatch == "async" and worker is None and backend is not None:
|
|
128
101
|
# async + no local worker: writes will flag embed_pending and this process
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
"""`[[wiki links]]` in a body, parsed into edges.
|
|
2
|
+
|
|
3
|
+
The convention arrived on its own: this repo's reference corpus already carried
|
|
4
|
+
238 `[[…]]` links across 97 memories with no tool support at all, 91% of them
|
|
5
|
+
resolving. What it did not carry was a way to ask "what points HERE" — 42 of
|
|
6
|
+
those 97 memories had no inbound link and were reachable only by search — or any
|
|
7
|
+
way to notice when a link stopped resolving.
|
|
8
|
+
|
|
9
|
+
## The syntax
|
|
10
|
+
|
|
11
|
+
[[target]] a link
|
|
12
|
+
[[target#anchor]] …to a section of it
|
|
13
|
+
[[target|label]] …shown as something else
|
|
14
|
+
[[target#anchor|label]] both
|
|
15
|
+
|
|
16
|
+
`#` before `|`, as in every wiki dialect since MediaWiki and in Obsidian today,
|
|
17
|
+
so a link written from habit parses. The split is positional and not clever:
|
|
18
|
+
first `|` ends the target part, first `#` inside that part starts the anchor. A
|
|
19
|
+
`#` in the LABEL is therefore just text, which is the reading that surprises
|
|
20
|
+
nobody.
|
|
21
|
+
|
|
22
|
+
## What is a link, and what is merely square brackets
|
|
23
|
+
|
|
24
|
+
Only two shapes are treated as links, and everything else is left alone:
|
|
25
|
+
|
|
26
|
+
* a **path** — an ltree path in this memory's own namespace (`ops.x402.deploy`);
|
|
27
|
+
* a **scheme** we know (`idea:some-slug`, `file:some-note`) — a pointer at
|
|
28
|
+
another store entirely, recorded as an edge but never resolved here.
|
|
29
|
+
|
|
30
|
+
A URL is not a link for these purposes and neither is prose. The rule is stated
|
|
31
|
+
POSITIVELY — recognise a path or a known scheme — rather than as a list of things
|
|
32
|
+
to exclude, because the exclusions are unbounded and the inclusions are two.
|
|
33
|
+
|
|
34
|
+
Two exclusions are still needed, and both come straight from the corpus:
|
|
35
|
+
|
|
36
|
+
* **code spans and fenced blocks are not scanned.** Documentation that explains
|
|
37
|
+
the syntax writes ``[[path]]`` in backticks, and a validator that flags its own
|
|
38
|
+
documentation trains everyone to ignore it;
|
|
39
|
+
* single-segment targets that name nothing here are still recorded as edges
|
|
40
|
+
rather than dropped. A link to something not yet written is a deliberate move
|
|
41
|
+
("this deserves a memory") and the tooling must let it stand — dangling, and
|
|
42
|
+
visible as dangling.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
from __future__ import annotations
|
|
46
|
+
|
|
47
|
+
import re
|
|
48
|
+
from dataclasses import dataclass
|
|
49
|
+
from typing import Dict, List, Optional
|
|
50
|
+
|
|
51
|
+
# Stores other than this one that a memory may legitimately point at. Recorded as
|
|
52
|
+
# edges so they are visible, never resolved — we do not own the address space.
|
|
53
|
+
KNOWN_SCHEMES = ("idea", "file")
|
|
54
|
+
|
|
55
|
+
# An ltree path: labels of [A-Za-z0-9_] joined by dots. Deliberately strict —
|
|
56
|
+
# this is what tells a path apart from a slug belonging to another store, and
|
|
57
|
+
# from prose that happens to sit in double brackets.
|
|
58
|
+
_PATH = re.compile(r"^[A-Za-z0-9_]+(\.[A-Za-z0-9_]+)*$")
|
|
59
|
+
|
|
60
|
+
_LINK = re.compile(r"\[\[([^\[\]\n]+)\]\]")
|
|
61
|
+
_FENCE = re.compile(r"```.*?```|~~~.*?~~~", re.S)
|
|
62
|
+
_CODE_SPAN = re.compile(r"`[^`\n]*`")
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
@dataclass
|
|
66
|
+
class Link:
|
|
67
|
+
"""One parsed link. `target` is verbatim as written; `scheme` is None for an
|
|
68
|
+
ordinary path and the scheme name for a pointer at another store.
|
|
69
|
+
|
|
70
|
+
`start`/`end` are the link's half-open span in the ORIGINAL body — the whole
|
|
71
|
+
`[[…]]`, brackets included. They are what lets a rewriter replace one link
|
|
72
|
+
without touching the rest of the text, and in particular without touching a
|
|
73
|
+
`[[path]]` written inside backticks: the parser never saw it, so it has no
|
|
74
|
+
span, so a rewrite cannot reach it.
|
|
75
|
+
"""
|
|
76
|
+
raw_target: str
|
|
77
|
+
label: Optional[str] = None
|
|
78
|
+
anchor: Optional[str] = None
|
|
79
|
+
scheme: Optional[str] = None
|
|
80
|
+
ord: int = 0
|
|
81
|
+
start: int = 0
|
|
82
|
+
end: int = 0
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _blank_code(body: str) -> str:
|
|
86
|
+
"""Replace code spans and fenced blocks with spaces of the same length.
|
|
87
|
+
|
|
88
|
+
Same length, not removal, so every offset in the returned text still lines up
|
|
89
|
+
with the original — the parser does not need that today, but an anchor
|
|
90
|
+
resolver or a renderer will, and a scanner that quietly shifts positions is a
|
|
91
|
+
bug waiting for the feature that depends on them.
|
|
92
|
+
"""
|
|
93
|
+
def blank(m: re.Match) -> str:
|
|
94
|
+
return "".join(" " if c != "\n" else "\n" for c in m.group(0))
|
|
95
|
+
return _CODE_SPAN.sub(blank, _FENCE.sub(blank, body))
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _classify(target: str) -> Optional[str]:
|
|
99
|
+
"""None for a path in this namespace, the scheme name for a known foreign
|
|
100
|
+
store, and the string ``"ignore"`` for everything else."""
|
|
101
|
+
if ":" in target:
|
|
102
|
+
scheme = target.split(":", 1)[0].strip().lower()
|
|
103
|
+
return scheme if scheme in KNOWN_SCHEMES else "ignore"
|
|
104
|
+
return None if _PATH.match(target) else "ignore"
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def parse_links(body: Optional[str]) -> List[Link]:
|
|
108
|
+
"""Every link in `body`, in the order it appears.
|
|
109
|
+
|
|
110
|
+
Matched against the ORIGINAL text, with the blanked copy used only to decide
|
|
111
|
+
whether a match sits inside code. Reading the fields out of the blanked copy
|
|
112
|
+
instead was harmless while nothing wrote them back — and destructive the
|
|
113
|
+
moment something did: a label containing an inline `code` span came back as
|
|
114
|
+
that many spaces, and a rewrite then committed the loss as a well-formed diff.
|
|
115
|
+
A match may also STRADDLE a blanked region (backticks that swallow a `]]` and
|
|
116
|
+
the `[[` after it), which in the blanked copy looks like one enormous link;
|
|
117
|
+
requiring the span to be untouched by blanking rejects that as well.
|
|
118
|
+
"""
|
|
119
|
+
body = body or ""
|
|
120
|
+
blanked = _blank_code(body)
|
|
121
|
+
out: List[Link] = []
|
|
122
|
+
for m in _LINK.finditer(body):
|
|
123
|
+
# Is THIS `[[` inside code? Its own opening brackets are blanked if so.
|
|
124
|
+
# Deliberately not "is any part of the span blanked": a label may legally
|
|
125
|
+
# contain an inline `code` span, and rejecting the whole link for that
|
|
126
|
+
# would silently drop a real edge.
|
|
127
|
+
if blanked[m.start():m.start() + 2] != "[[":
|
|
128
|
+
continue
|
|
129
|
+
inner = m.group(1).strip()
|
|
130
|
+
target_part, _, label = inner.partition("|")
|
|
131
|
+
target, _, anchor = target_part.partition("#")
|
|
132
|
+
target, label, anchor = (target.strip(), label.strip(), anchor.strip())
|
|
133
|
+
if not target:
|
|
134
|
+
continue
|
|
135
|
+
kind = _classify(target)
|
|
136
|
+
if kind == "ignore":
|
|
137
|
+
continue
|
|
138
|
+
out.append(Link(raw_target=target, label=label or None,
|
|
139
|
+
anchor=anchor or None,
|
|
140
|
+
scheme=None if kind is None else kind,
|
|
141
|
+
ord=len(out), start=m.start(), end=m.end()))
|
|
142
|
+
return out
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def render(target: str, anchor: Optional[str] = None,
|
|
146
|
+
label: Optional[str] = None) -> str:
|
|
147
|
+
"""The canonical text for a link. Inverse of the parse, in its normal form:
|
|
148
|
+
whitespace the author put inside the brackets is not preserved, because there
|
|
149
|
+
is nothing to preserve it for and a single spelling is easier to read."""
|
|
150
|
+
inner = target
|
|
151
|
+
if anchor:
|
|
152
|
+
inner += "#" + anchor
|
|
153
|
+
if label:
|
|
154
|
+
inner += "|" + label
|
|
155
|
+
return f"[[{inner}]]"
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def rewrite_targets(body: str, new_targets: Dict[int, str]) -> str:
|
|
159
|
+
"""Point selected links somewhere else, leaving the rest of the body alone.
|
|
160
|
+
|
|
161
|
+
`new_targets` maps a link's `ord` — its index among the links this parser
|
|
162
|
+
RECOGNISES, which is exactly what the edge table stores — to the target it
|
|
163
|
+
should now carry. Anchor and label ride along unchanged: a rename moves the
|
|
164
|
+
address, not what the author called it or which section they meant.
|
|
165
|
+
|
|
166
|
+
Addressed by ord rather than by text on purpose. A body that links to the same
|
|
167
|
+
path twice has two edges, and only one of them may be the one to change; and a
|
|
168
|
+
body that mentions the old path in prose or in a code span must not be touched
|
|
169
|
+
at all. Substituting text would get both of those wrong.
|
|
170
|
+
|
|
171
|
+
Assembled in one forward pass over the spans rather than by slicing the whole
|
|
172
|
+
body once per link: a memory at the default 256 KB ceiling can hold tens of
|
|
173
|
+
thousands of links, and rebuilding the string each time made a single rewrite
|
|
174
|
+
quadratic — seconds of CPU while holding row locks.
|
|
175
|
+
"""
|
|
176
|
+
if not new_targets:
|
|
177
|
+
return body
|
|
178
|
+
links = sorted((l for l in parse_links(body) if l.ord in new_targets),
|
|
179
|
+
key=lambda l: l.start)
|
|
180
|
+
if not links:
|
|
181
|
+
return body
|
|
182
|
+
parts, at = [], 0
|
|
183
|
+
for l in links:
|
|
184
|
+
parts.append(body[at:l.start])
|
|
185
|
+
parts.append(render(new_targets[l.ord], l.anchor, l.label))
|
|
186
|
+
at = l.end
|
|
187
|
+
parts.append(body[at:])
|
|
188
|
+
return "".join(parts)
|