memgres 0.11.0__tar.gz → 0.12.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {memgres-0.11.0 → memgres-0.12.0}/PKG-INFO +2 -1
- {memgres-0.11.0 → memgres-0.12.0}/README.md +1 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/__init__.py +2 -1
- {memgres-0.11.0 → memgres-0.12.0}/memgres/_version.py +1 -1
- {memgres-0.11.0 → memgres-0.12.0}/memgres/config.py +6 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/info.py +8 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/links.py +63 -4
- memgres-0.12.0/memgres/migrations/0023_relink_after_parser_fix.sql +16 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/schema.py +5 -1
- {memgres-0.11.0 → memgres-0.12.0}/memgres/store.py +69 -2
- {memgres-0.11.0 → memgres-0.12.0}/memgres.egg-info/PKG-INFO +2 -1
- {memgres-0.11.0 → memgres-0.12.0}/memgres.egg-info/SOURCES.txt +2 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_links.py +43 -1
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_list.py +1 -1
- memgres-0.12.0/tests/test_required_fields.py +195 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_server_info.py +20 -1
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_server_integration.py +2 -1
- {memgres-0.11.0 → memgres-0.12.0}/LICENSE +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/admin.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/admin_cli.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/blame.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/bootstrap.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/delimiters.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/diffing.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/embed_worker.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/embeddings.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/healthcheck.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/identity.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/indexing.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/lines.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/mcp_server.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0001_core.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0002_identity.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0003_history_author.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0004_title.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0005_chunk_index.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0006_reader_floor.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0007_embed_retry.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0008_service_roles.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0009_create_namespace_right.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0010_namespace_alias.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0011_drop_default_namespace.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0012_user_profile.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0013_hash_version.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0014_access_request_no_fk.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0015_normalize_tags.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0016_valid_at.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0017_memory_link.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0018_links_built.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0019_memory_usage.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0020_memory_usage_no_fk.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0021_enrollment_key.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/migrations/0022_user_disabled.sql +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/periodic.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/reembed.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/relink.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/search.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/segments.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/server.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/tags.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/token_cli.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/vector/__init__.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/vector/base.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/vector/pgvector.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/vector/qdrant.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres/worker.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres.egg-info/dependency_links.txt +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres.egg-info/entry_points.txt +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres.egg-info/requires.txt +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/memgres.egg-info/top_level.txt +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/pyproject.toml +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/setup.cfg +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_admin_two_way.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_blame_integration.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_chunk_index.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_claim_and_reembed.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_config.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_diffing.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_embed_worker.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_embeddings.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_enrollment.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_healthcheck.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_identity_integration.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_lexical_match.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_limits.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_mcp_admin_tools.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_mcp_http_transport.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_mcp_instructions.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_mcp_recall_schema.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_mcp_tool_visibility.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_mcp_tool_visibility_http.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_migration_upgrade.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_multi_space_search.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_path_addressing.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_qdrant_ca.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_qdrant_integration.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_replace_build.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_require_title.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_retention.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_roles_bootstrap.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_search_integration.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_security_followups.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_security_integration.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_segments.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_segments_store.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_snippets.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_store_integration.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_tags.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_token_sink.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_usage.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_valid_at.py +0 -0
- {memgres-0.11.0 → memgres-0.12.0}/tests/test_write_ergonomics.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: memgres
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.12.0
|
|
4
4
|
Summary: Drop-in memory for AI agents: one Postgres, lexical + semantic recall, diff-versioned history, GDPR-erasable.
|
|
5
5
|
Author: mozgsml
|
|
6
6
|
License-Expression: MIT
|
|
@@ -216,6 +216,7 @@ Everything is env, all optional (defaults suit a single-user embed). Full list i
|
|
|
216
216
|
| `MEMGRES_TOKEN` | — | default token used when a call passes none (single-tenant endpoints) |
|
|
217
217
|
| `MEMGRES_TOKEN_SINK` | — | absolute directory a minted secret is **written to** (`<token-id>.token`, `0600`) instead of being returned. Set it when provisioning is done by an agent — a secret in a tool result is a secret in a transcript. See [docs/TENANCY.md](docs/TENANCY.md) |
|
|
218
218
|
| `MEMGRES_TREE` | `true` | `ltree` path column + GiST index (fast subtree select) |
|
|
219
|
+
| `MEMGRES_REQUIRED_FIELDS` | — | comma-separated fields a content-storing write must carry (`source`, `reason`), refused by name when absent. A policy about a corpus, not a property of the software: a scratch database wants none, a corporate one wants `source`. `move`/`retag` are exempt. Announced in `server_info.write_requirements` |
|
|
219
220
|
| `MEMGRES_REQUIRE_TITLE` | `true` | `true` = a write that stores content must supply `title`. Captions are what name a memory in results and what title-weighted ranking weighs; `move`/`retag` are exempt (they store no content) |
|
|
220
221
|
| `MEMGRES_REQUIRE_PARENT` | `false` | `true` = a node's parent path must already exist |
|
|
221
222
|
| `MEMGRES_HISTORY` | `true` | keep the hash-chained diff history (deleted with the record) |
|
|
@@ -180,6 +180,7 @@ Everything is env, all optional (defaults suit a single-user embed). Full list i
|
|
|
180
180
|
| `MEMGRES_TOKEN` | — | default token used when a call passes none (single-tenant endpoints) |
|
|
181
181
|
| `MEMGRES_TOKEN_SINK` | — | absolute directory a minted secret is **written to** (`<token-id>.token`, `0600`) instead of being returned. Set it when provisioning is done by an agent — a secret in a tool result is a secret in a transcript. See [docs/TENANCY.md](docs/TENANCY.md) |
|
|
182
182
|
| `MEMGRES_TREE` | `true` | `ltree` path column + GiST index (fast subtree select) |
|
|
183
|
+
| `MEMGRES_REQUIRED_FIELDS` | — | comma-separated fields a content-storing write must carry (`source`, `reason`), refused by name when absent. A policy about a corpus, not a property of the software: a scratch database wants none, a corporate one wants `source`. `move`/`retag` are exempt. Announced in `server_info.write_requirements` |
|
|
183
184
|
| `MEMGRES_REQUIRE_TITLE` | `true` | `true` = a write that stores content must supply `title`. Captions are what name a memory in results and what title-weighted ranking weighs; `move`/`retag` are exempt (they store no content) |
|
|
184
185
|
| `MEMGRES_REQUIRE_PARENT` | `false` | `true` = a node's parent path must already exist |
|
|
185
186
|
| `MEMGRES_HISTORY` | `true` | keep the hash-chained diff history (deleted with the record) |
|
|
@@ -23,7 +23,7 @@ from .schema import migrate, SchemaMismatch, SCHEMA_VERSION
|
|
|
23
23
|
from .search import Hit, recall
|
|
24
24
|
from .blame import annotate, annotate_grouped, reconstruct, replay
|
|
25
25
|
from .store import (Store, Memory, Conflict, NotFound, TooLarge, NoParent,
|
|
26
|
-
MissingTitle)
|
|
26
|
+
MissingTitle, MissingField)
|
|
27
27
|
from .identity import (
|
|
28
28
|
Principal, AuthError, SpaceNotFound,
|
|
29
29
|
resolve, resolve_space, new_token, valid_format,
|
|
@@ -38,6 +38,7 @@ __all__ = [
|
|
|
38
38
|
"Config", "load_config",
|
|
39
39
|
"Store", "Memory", "Conflict", "NotFound", "TooLarge", "NoParent",
|
|
40
40
|
"MissingTitle",
|
|
41
|
+
"MissingField",
|
|
41
42
|
"make_diff", "apply_diff", "content_hash", "DiffConflict",
|
|
42
43
|
"Embedder", "get_embedder",
|
|
43
44
|
"migrate", "SchemaMismatch", "SCHEMA_VERSION",
|
|
@@ -79,6 +79,8 @@ class Config:
|
|
|
79
79
|
# organization
|
|
80
80
|
tree_enabled: bool # ltree path column + GiST index for fast subtree selection
|
|
81
81
|
require_title: bool # True = a write that stores CONTENT must caption it
|
|
82
|
+
required_fields: tuple # extra fields a content-storing write must carry
|
|
83
|
+
# (MEMGRES_REQUIRED_FIELDS=source), refused when absent
|
|
82
84
|
require_parent: bool # False = sparse paths (create food.apple with no food row);
|
|
83
85
|
# True = a node's parent path must already exist as a memory
|
|
84
86
|
# history
|
|
@@ -222,6 +224,10 @@ def load() -> Config:
|
|
|
222
224
|
token_sink=_str("MEMGRES_TOKEN_SINK", ""),
|
|
223
225
|
tree_enabled=_bool("MEMGRES_TREE", True),
|
|
224
226
|
require_title=_bool("MEMGRES_REQUIRE_TITLE", True),
|
|
227
|
+
required_fields=tuple(
|
|
228
|
+
f for f in (x.strip().lower()
|
|
229
|
+
for x in _str("MEMGRES_REQUIRED_FIELDS", "").split(","))
|
|
230
|
+
if f),
|
|
225
231
|
require_parent=_bool("MEMGRES_REQUIRE_PARENT", False),
|
|
226
232
|
history_enabled=_bool("MEMGRES_HISTORY", True),
|
|
227
233
|
fts_language=_str("MEMGRES_FTS_LANGUAGE", "simple"),
|
|
@@ -65,6 +65,14 @@ def server_info(cfg: Config, embed_dim: Optional[int] = None) -> dict:
|
|
|
65
65
|
"renew_on_read": bool(days > 0 and cfg.renew_on_read),
|
|
66
66
|
"policy": _retention_policy(days, cfg.renew_on_read),
|
|
67
67
|
},
|
|
68
|
+
# What a write MUST carry here. Announced rather than discovered from a
|
|
69
|
+
# refusal: a client that learns the rule by being rejected has already
|
|
70
|
+
# composed the memory, and a rule nobody can see before writing is a rule
|
|
71
|
+
# that gets satisfied with junk on the second attempt.
|
|
72
|
+
"write_requirements": {
|
|
73
|
+
"title": cfg.require_title,
|
|
74
|
+
"fields": list(cfg.required_fields),
|
|
75
|
+
},
|
|
68
76
|
"recall_modes": ["lexical"] if lexical_only
|
|
69
77
|
else ["lexical", "semantic", "hybrid", "auto"],
|
|
70
78
|
"vector_backend": cfg.vector_backend,
|
|
@@ -52,10 +52,19 @@ from typing import Dict, List, Optional
|
|
|
52
52
|
# edges so they are visible, never resolved — we do not own the address space.
|
|
53
53
|
KNOWN_SCHEMES = ("idea", "file")
|
|
54
54
|
|
|
55
|
-
# An ltree path: labels of [A-Za-z0-9_] joined by dots. Deliberately strict —
|
|
55
|
+
# An ltree path: labels of [A-Za-z0-9_-] joined by dots. Deliberately strict —
|
|
56
56
|
# this is what tells a path apart from a slug belonging to another store, and
|
|
57
57
|
# from prose that happens to sit in double brackets.
|
|
58
|
-
|
|
58
|
+
#
|
|
59
|
+
# The hyphen is there because PATHS MAY CONTAIN ONE. Postgres has allowed `-` in
|
|
60
|
+
# ltree labels since 13 (the minimum this project supports), `write` accepts such
|
|
61
|
+
# a path without complaint, and real corpora are full of them —
|
|
62
|
+
# `infra.servers.video-production`. While this pattern rejected the hyphen the
|
|
63
|
+
# two halves of the product disagreed about what a path is: the link was stored
|
|
64
|
+
# as prose, `_classify` returned "ignore", and the edge did not even become a
|
|
65
|
+
# DANGLING one. It vanished, and `memory_links` answered "nothing points here",
|
|
66
|
+
# which reads as a fact about the corpus rather than as a parser that quit.
|
|
67
|
+
_PATH = re.compile(r"^[A-Za-z0-9_-]+(\.[A-Za-z0-9_-]+)*$")
|
|
59
68
|
|
|
60
69
|
_LINK = re.compile(r"\[\[([^\[\]\n]+)\]\]")
|
|
61
70
|
_FENCE = re.compile(r"```.*?```|~~~.*?~~~", re.S)
|
|
@@ -82,8 +91,58 @@ class Link:
|
|
|
82
91
|
end: int = 0
|
|
83
92
|
|
|
84
93
|
|
|
94
|
+
def _blank_indented(original: str, text: str) -> str:
|
|
95
|
+
"""Blank markdown's OTHER code block: a run of lines indented by four spaces
|
|
96
|
+
(or a tab), opened by a blank line.
|
|
97
|
+
|
|
98
|
+
Fences are not the only way to show code, and the indented form is what a
|
|
99
|
+
hand-written example tends to use. A frp config pasted that way put
|
|
100
|
+
``[[proxies]]`` — a TOML array-of-tables header, and a perfectly well-formed
|
|
101
|
+
path — into the link graph as a dangling edge to a memory nobody will ever
|
|
102
|
+
write. The parser must ignore code wherever markdown says code is.
|
|
103
|
+
|
|
104
|
+
A blank line inside the block does not end it; the first non-blank line back
|
|
105
|
+
at the margin does. Opening on a blank line is what keeps an ordinary list
|
|
106
|
+
item or a wrapped line — indented, but not preceded by a blank — from being
|
|
107
|
+
read as code.
|
|
108
|
+
|
|
109
|
+
🔴 Which lines are code is decided from `original`, and the blanking applied
|
|
110
|
+
to `text` — the copy where spans and fences are already spaces. Deciding from
|
|
111
|
+
the blanked copy instead made every line that STARTED with an inline code
|
|
112
|
+
span look indented: "`[[a]]` but [[b]] counts" became four-plus leading
|
|
113
|
+
spaces, the whole line was swallowed as a block, and a real link next to a
|
|
114
|
+
code span disappeared. Both strings are length-preserving, so their lines
|
|
115
|
+
correspond one to one.
|
|
116
|
+
"""
|
|
117
|
+
def blank(line: str) -> str:
|
|
118
|
+
return "".join(" " if c != "\n" else "\n" for c in line)
|
|
119
|
+
|
|
120
|
+
src = original.splitlines(keepends=True)
|
|
121
|
+
dst = text.splitlines(keepends=True)
|
|
122
|
+
out, in_block, prev_blank = [], False, True
|
|
123
|
+
for i, line in enumerate(src):
|
|
124
|
+
stripped = line.strip("\n")
|
|
125
|
+
is_blank = not stripped.strip()
|
|
126
|
+
indented = stripped.startswith(" ") or stripped.startswith("\t")
|
|
127
|
+
current = dst[i] if i < len(dst) else line
|
|
128
|
+
if in_block:
|
|
129
|
+
if is_blank or indented:
|
|
130
|
+
out.append(current if is_blank else blank(current))
|
|
131
|
+
else:
|
|
132
|
+
in_block = False
|
|
133
|
+
out.append(current)
|
|
134
|
+
elif indented and prev_blank:
|
|
135
|
+
in_block = True
|
|
136
|
+
out.append(blank(current))
|
|
137
|
+
else:
|
|
138
|
+
out.append(current)
|
|
139
|
+
prev_blank = is_blank
|
|
140
|
+
return "".join(out)
|
|
141
|
+
|
|
142
|
+
|
|
85
143
|
def _blank_code(body: str) -> str:
|
|
86
|
-
"""Replace code spans and
|
|
144
|
+
"""Replace code spans and code blocks — fenced AND indented — with spaces of
|
|
145
|
+
the same length.
|
|
87
146
|
|
|
88
147
|
Same length, not removal, so every offset in the returned text still lines up
|
|
89
148
|
with the original — the parser does not need that today, but an anchor
|
|
@@ -92,7 +151,7 @@ def _blank_code(body: str) -> str:
|
|
|
92
151
|
"""
|
|
93
152
|
def blank(m: re.Match) -> str:
|
|
94
153
|
return "".join(" " if c != "\n" else "\n" for c in m.group(0))
|
|
95
|
-
return _CODE_SPAN.sub(blank, _FENCE.sub(blank, body))
|
|
154
|
+
return _blank_indented(body, _CODE_SPAN.sub(blank, _FENCE.sub(blank, body)))
|
|
96
155
|
|
|
97
156
|
|
|
98
157
|
def _classify(target: str) -> Optional[str]:
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
-- Re-derive the link graph, because the PARSER changed, not the schema.
|
|
2
|
+
--
|
|
3
|
+
-- Until now `_PATH` rejected the hyphen, so every `[[infra.servers.video-production]]`
|
|
4
|
+
-- was classified as prose and dropped — not stored as a dangling edge, dropped —
|
|
5
|
+
-- while `[[proxies]]` from a TOML example indented as code was stored as a real
|
|
6
|
+
-- one. Both halves are fixed in `links.py`; neither fix reaches a body that is
|
|
7
|
+
-- already written, since edges are derived on write.
|
|
8
|
+
--
|
|
9
|
+
-- Clearing the flag makes the existing one-time backfill (`relink.rebuild`, run
|
|
10
|
+
-- at startup by `maybe_backfill`) do the pass again over every body. It rewrites
|
|
11
|
+
-- only the edge table: bodies, history and the hash chain are untouched, so this
|
|
12
|
+
-- is repeatable and costs nothing but a scan.
|
|
13
|
+
--
|
|
14
|
+
-- Additive by nature — an older client reading this database is no worse off
|
|
15
|
+
-- than it was, so the compatibility floor does not move.
|
|
16
|
+
UPDATE memgres_meta SET links_built = false;
|
|
@@ -18,7 +18,7 @@ from pathlib import Path
|
|
|
18
18
|
from .config import Config
|
|
19
19
|
|
|
20
20
|
# The version this build migrates the database TO (the latest migration it carries).
|
|
21
|
-
SCHEMA_VERSION =
|
|
21
|
+
SCHEMA_VERSION = 24
|
|
22
22
|
|
|
23
23
|
# The compatibility FLOOR: the schema version of the most recent backward-
|
|
24
24
|
# INCOMPATIBLE migration — one that changed the shape/semantics old code relied on
|
|
@@ -74,6 +74,10 @@ SCHEMA_VERSION = 23
|
|
|
74
74
|
# v21 (0020): dropped 0019's foreign key from databases that already ran it →
|
|
75
75
|
# additive, floor stays 16. Nothing reads the constraint; removing it only
|
|
76
76
|
# stops a counted read from locking the memory row it counts.
|
|
77
|
+
# v24 (0023): cleared links_built so the link backfill runs again after the
|
|
78
|
+
# link PARSER was fixed (hyphens in paths were dropped, indented code was
|
|
79
|
+
# parsed) → additive, floor stays 16. It rebuilds a derived index from text
|
|
80
|
+
# that is already stored; an older client neither writes nor reads differently.
|
|
77
81
|
SCHEMA_BREAKING_VERSION = 16
|
|
78
82
|
|
|
79
83
|
# Dev layout: repo/migrations next to the package. When packaged, migrations are
|
|
@@ -95,6 +95,15 @@ def _as_date(value: object):
|
|
|
95
95
|
f"the content was last known to be accurate")
|
|
96
96
|
|
|
97
97
|
|
|
98
|
+
class MissingField(ValueError):
|
|
99
|
+
"""A write left out a field this deployment declares mandatory.
|
|
100
|
+
|
|
101
|
+
Separate from `MissingTitle` on purpose: the title is required by a setting
|
|
102
|
+
of its own and has been since long before this, and folding it in would
|
|
103
|
+
change the exception a client already catches.
|
|
104
|
+
"""
|
|
105
|
+
|
|
106
|
+
|
|
98
107
|
class MissingTitle(ValueError):
|
|
99
108
|
"""A write stored content without a caption, and the deployment requires one
|
|
100
109
|
(``MEMGRES_REQUIRE_TITLE``, default on).
|
|
@@ -238,6 +247,15 @@ class Memory:
|
|
|
238
247
|
# how much this memory is used — set only by `get`, and only when the
|
|
239
248
|
# deployment counts (see `_count_usage`). A write does not have it.
|
|
240
249
|
usage: Optional[dict] = None
|
|
250
|
+
# Provenance of the REVISION this call just wrote — never of the memory,
|
|
251
|
+
# which has no single source: `source`/`reason`/`valid_at` live on the
|
|
252
|
+
# history row because different edits speak to different origins and dates.
|
|
253
|
+
# So these are set by `write` and stay None on a read, where the question
|
|
254
|
+
# belongs to `history`/`blame`. Echoed back because a required field that the
|
|
255
|
+
# answer does not confirm is a field whose absence nobody notices: four edits
|
|
256
|
+
# in a row went out with an empty `source` and every reply looked fine.
|
|
257
|
+
source: Optional[str] = None
|
|
258
|
+
valid_at: object = None
|
|
241
259
|
|
|
242
260
|
def to_dict(self, *, stringify_dates: bool = False) -> dict:
|
|
243
261
|
"""Serialize for an API layer. ``stringify_dates`` str()-coerces the
|
|
@@ -247,6 +265,7 @@ class Memory:
|
|
|
247
265
|
return (str(v) if v is not None else None) if stringify_dates else v
|
|
248
266
|
return {"id": self.id, "content_hash": self.content_hash, "body": self.body,
|
|
249
267
|
"title": self.title, "tags": self.tags, "path": self.path,
|
|
268
|
+
"source": self.source, "valid_at": d(self.valid_at),
|
|
250
269
|
"seq": self.seq, "created_at": d(self.created_at),
|
|
251
270
|
"updated_at": d(self.updated_at), "expires_at": d(self.expires_at),
|
|
252
271
|
"created": self.created, "moved_from": self.moved_from,
|
|
@@ -595,6 +614,12 @@ class Store:
|
|
|
595
614
|
need="write", for_write=True)
|
|
596
615
|
self._check_provenance_size(source, reason)
|
|
597
616
|
self._check_title_size(title)
|
|
617
|
+
# Content-storing is what the title rule already means by it: a body,
|
|
618
|
+
# a diff or a substring edit. A create always stores content.
|
|
619
|
+
self._require_fields(
|
|
620
|
+
stores_content=(id is None and at is None) or body is not None
|
|
621
|
+
or diff is not None or replace is not None,
|
|
622
|
+
source=source, reason=reason, title=title)
|
|
598
623
|
# One spelling per tag, decided here rather than in `_create` and
|
|
599
624
|
# `_update` separately — two normalisation sites is how a tag ends up
|
|
600
625
|
# stored one way and filtered another.
|
|
@@ -615,6 +640,8 @@ class Store:
|
|
|
615
640
|
# Checked on the STORED body, not the request: a substring edit or a
|
|
616
641
|
# diff can introduce the stray tag just as a whole body can.
|
|
617
642
|
m.warnings = write_warnings(m.body)
|
|
643
|
+
# The revision's own provenance, straight back to whoever wrote it.
|
|
644
|
+
m.source, m.valid_at = source, valid_at
|
|
618
645
|
return m
|
|
619
646
|
|
|
620
647
|
def _check_path_free(self, ns: str, path: Optional[str],
|
|
@@ -699,6 +726,36 @@ class Store:
|
|
|
699
726
|
f"(it is what names the memory in results and what title search "
|
|
700
727
|
f"matches). Its first line is: {first!r}")
|
|
701
728
|
|
|
729
|
+
# What each declarable field is FOR — the refusal has to say this, or the
|
|
730
|
+
# requirement degenerates into filling the box: "from the email", "the user
|
|
731
|
+
# said", and a corpus of assertions nobody can check.
|
|
732
|
+
_FIELD_HELP = {
|
|
733
|
+
"source": ("where this knowledge came from, as an ADDRESS someone else "
|
|
734
|
+
"can follow back to the original: host + absolute path + "
|
|
735
|
+
"date; mailbox, sender -> recipient, date, subject; full URL "
|
|
736
|
+
"+ date read. 'from the email' or 'the user said' is not one"),
|
|
737
|
+
"reason": ("why this write happened — what changed and why, in one line"),
|
|
738
|
+
"title": ("a short caption; it names the memory in results"),
|
|
739
|
+
}
|
|
740
|
+
|
|
741
|
+
def _require_fields(self, *, stores_content: bool, **values) -> None:
|
|
742
|
+
"""Enforce `MEMGRES_REQUIRED_FIELDS` on a write that stores CONTENT.
|
|
743
|
+
|
|
744
|
+
Metadata-only edits are exempt, exactly as they are for the title: a move
|
|
745
|
+
or a retag asserts nothing new, and demanding provenance for one would
|
|
746
|
+
make re-filing a memory harder than writing one — friction unrelated to
|
|
747
|
+
the point, and the surest way to get a required field filled with junk.
|
|
748
|
+
"""
|
|
749
|
+
if not stores_content:
|
|
750
|
+
return
|
|
751
|
+
for field in self.cfg.required_fields:
|
|
752
|
+
if (values.get(field) or "").strip():
|
|
753
|
+
continue
|
|
754
|
+
help_text = self._FIELD_HELP.get(field, "required by this deployment")
|
|
755
|
+
raise MissingField(
|
|
756
|
+
f"this deployment requires `{field}` on every write that stores "
|
|
757
|
+
f"content — {help_text}")
|
|
758
|
+
|
|
702
759
|
def _check_title_size(self, title: Optional[str]):
|
|
703
760
|
if title is not None and byte_len(title) > self.cfg.max_title_bytes:
|
|
704
761
|
raise TooLarge(
|
|
@@ -1323,13 +1380,13 @@ class Store:
|
|
|
1323
1380
|
"created_at, updated_at, namespace, "
|
|
1324
1381
|
# Browsing is where usage becomes actionable: it is how you find the
|
|
1325
1382
|
# subtree nobody reads. No row yet means never used, which is zero.
|
|
1326
|
-
"COALESCE(u.recall_count, 0), COALESCE(u.get_count, 0) "
|
|
1383
|
+
"COALESCE(u.recall_count, 0), COALESCE(u.get_count, 0), seq "
|
|
1327
1384
|
"FROM memory LEFT JOIN memory_usage u ON u.memory_id = memory.id "
|
|
1328
1385
|
f"WHERE {where} ORDER BY namespace, path, id "
|
|
1329
1386
|
"LIMIT %s OFFSET %s",
|
|
1330
1387
|
head + params + [limit, offset])
|
|
1331
1388
|
cols = ["id", "path", "tags", "title", "shown", "created_at",
|
|
1332
|
-
"updated_at", "space_id", "recalled", "gets"]
|
|
1389
|
+
"updated_at", "space_id", "recalled", "gets", "seq"]
|
|
1333
1390
|
budget = self.cfg.list_bodies_max_bytes
|
|
1334
1391
|
rows, wanted = [], []
|
|
1335
1392
|
for r in cur.fetchall():
|
|
@@ -1338,6 +1395,9 @@ class Store:
|
|
|
1338
1395
|
d["tags"] = list(d["tags"]) if d["tags"] is not None else []
|
|
1339
1396
|
d["space_id"] = str(d["space_id"])
|
|
1340
1397
|
d["space"] = names.get(d["space_id"])
|
|
1398
|
+
# Revisions after the creation. Comes off the row itself, so unlike
|
|
1399
|
+
# the read counters it is exact and costs nothing.
|
|
1400
|
+
d["edits"] = max(0, int(d.pop("seq")) - 1)
|
|
1341
1401
|
if not self.cfg.usage_counters:
|
|
1342
1402
|
# Null, not zero. Nothing is being counted here, and reporting
|
|
1343
1403
|
# "0 recalls, 0 reads" would read as a measurement — on a
|
|
@@ -1543,6 +1603,13 @@ class Store:
|
|
|
1543
1603
|
# the store working rather than the memory being used.
|
|
1544
1604
|
if _count:
|
|
1545
1605
|
m.usage = self._count_usage("get", [m.id], want=True)
|
|
1606
|
+
if m.usage is not None:
|
|
1607
|
+
# How often this memory has been REWRITTEN, which is a different
|
|
1608
|
+
# question from how often it is read: a much-edited memory is a
|
|
1609
|
+
# live one, a much-read one is a useful one, and ranking "what is
|
|
1610
|
+
# hot" wants both. Free — `seq` counts the revisions already, and
|
|
1611
|
+
# the first one is the creation, so edits are one fewer.
|
|
1612
|
+
m.usage["edits"] = max(0, int(m.seq) - 1)
|
|
1546
1613
|
return _slice_lines(m, lines) if lines else m
|
|
1547
1614
|
|
|
1548
1615
|
def history(self, token: Optional[str], id: Optional[str] = None, *,
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: memgres
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.12.0
|
|
4
4
|
Summary: Drop-in memory for AI agents: one Postgres, lexical + semantic recall, diff-versioned history, GDPR-erasable.
|
|
5
5
|
Author: mozgsml
|
|
6
6
|
License-Expression: MIT
|
|
@@ -216,6 +216,7 @@ Everything is env, all optional (defaults suit a single-user embed). Full list i
|
|
|
216
216
|
| `MEMGRES_TOKEN` | — | default token used when a call passes none (single-tenant endpoints) |
|
|
217
217
|
| `MEMGRES_TOKEN_SINK` | — | absolute directory a minted secret is **written to** (`<token-id>.token`, `0600`) instead of being returned. Set it when provisioning is done by an agent — a secret in a tool result is a secret in a transcript. See [docs/TENANCY.md](docs/TENANCY.md) |
|
|
218
218
|
| `MEMGRES_TREE` | `true` | `ltree` path column + GiST index (fast subtree select) |
|
|
219
|
+
| `MEMGRES_REQUIRED_FIELDS` | — | comma-separated fields a content-storing write must carry (`source`, `reason`), refused by name when absent. A policy about a corpus, not a property of the software: a scratch database wants none, a corporate one wants `source`. `move`/`retag` are exempt. Announced in `server_info.write_requirements` |
|
|
219
220
|
| `MEMGRES_REQUIRE_TITLE` | `true` | `true` = a write that stores content must supply `title`. Captions are what name a memory in results and what title-weighted ranking weighs; `move`/`retag` are exempt (they store no content) |
|
|
220
221
|
| `MEMGRES_REQUIRE_PARENT` | `false` | `true` = a node's parent path must already exist |
|
|
221
222
|
| `MEMGRES_HISTORY` | `true` | keep the hash-chained diff history (deleted with the record) |
|
|
@@ -58,6 +58,7 @@ memgres/migrations/0019_memory_usage.sql
|
|
|
58
58
|
memgres/migrations/0020_memory_usage_no_fk.sql
|
|
59
59
|
memgres/migrations/0021_enrollment_key.sql
|
|
60
60
|
memgres/migrations/0022_user_disabled.sql
|
|
61
|
+
memgres/migrations/0023_relink_after_parser_fix.sql
|
|
61
62
|
memgres/vector/__init__.py
|
|
62
63
|
memgres/vector/base.py
|
|
63
64
|
memgres/vector/pgvector.py
|
|
@@ -90,6 +91,7 @@ tests/test_qdrant_ca.py
|
|
|
90
91
|
tests/test_qdrant_integration.py
|
|
91
92
|
tests/test_replace_build.py
|
|
92
93
|
tests/test_require_title.py
|
|
94
|
+
tests/test_required_fields.py
|
|
93
95
|
tests/test_retention.py
|
|
94
96
|
tests/test_roles_bootstrap.py
|
|
95
97
|
tests/test_search_integration.py
|
|
@@ -147,7 +147,49 @@ def test_urls_and_prose_are_left_alone():
|
|
|
147
147
|
assert parse_links("[[https://example.com/x]]") == []
|
|
148
148
|
assert parse_links("[[mailto:someone@example.com]]") == []
|
|
149
149
|
assert parse_links("[[some thing with spaces]]") == []
|
|
150
|
-
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def test_a_hyphen_belongs_to_a_path():
|
|
153
|
+
"""This assertion used to say the opposite, on the belief that a hyphen is
|
|
154
|
+
not an ltree label. It is: Postgres has allowed one since 13, `write` stores
|
|
155
|
+
such paths without complaint, and a real corpus is full of them. While the
|
|
156
|
+
parser disagreed, every `[[infra.servers.video-production]]` was silently
|
|
157
|
+
dropped — not kept as dangling, dropped — so `memory_links` reported an empty
|
|
158
|
+
graph and it read like a fact about the corpus."""
|
|
159
|
+
links = parse_links("see [[infra.servers.video-production]] and [[ops.frp.tunnel]]")
|
|
160
|
+
assert [l.raw_target for l in links] == ["infra.servers.video-production",
|
|
161
|
+
"ops.frp.tunnel"]
|
|
162
|
+
assert all(l.scheme is None for l in links)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def test_code_indented_by_four_spaces_is_still_code():
|
|
166
|
+
"""Fences are not the only way to show code. A frp config pasted with an
|
|
167
|
+
indent put `[[proxies]]` — a TOML array-of-tables header that happens to be a
|
|
168
|
+
well-formed path — into the graph as an edge to a memory nobody will write."""
|
|
169
|
+
body = ("Add a section:\n"
|
|
170
|
+
"\n"
|
|
171
|
+
" [[proxies]]\n"
|
|
172
|
+
" name = \"memgres\"\n"
|
|
173
|
+
"\n"
|
|
174
|
+
"then restart, see [[ops.frp.tunnel]].\n")
|
|
175
|
+
assert [l.raw_target for l in parse_links(body)] == ["ops.frp.tunnel"]
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def test_a_blank_line_inside_the_block_does_not_end_it():
|
|
179
|
+
body = ("Example:\n\n"
|
|
180
|
+
" [[first]]\n"
|
|
181
|
+
"\n"
|
|
182
|
+
" [[second]]\n"
|
|
183
|
+
"\n"
|
|
184
|
+
"back at the margin: [[real.one]]\n")
|
|
185
|
+
assert [l.raw_target for l in parse_links(body)] == ["real.one"]
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def test_an_indented_list_item_is_not_code():
|
|
189
|
+
"""Two spaces is a list, not a block — and the link in it is a real one.
|
|
190
|
+
Blanking by indentation alone would have eaten it."""
|
|
191
|
+
body = "Пункты:\n - см. [[ops.memory.onboarding]]\n"
|
|
192
|
+
assert [l.raw_target for l in parse_links(body)] == ["ops.memory.onboarding"]
|
|
151
193
|
|
|
152
194
|
|
|
153
195
|
def test_known_schemes_point_at_other_stores():
|
|
@@ -66,7 +66,7 @@ def test_lists_subtree_ordered_by_path(store):
|
|
|
66
66
|
# the ops row is NOT in the decisions subtree
|
|
67
67
|
assert all(r["path"].startswith("decisions") for r in rows)
|
|
68
68
|
# shape of each row
|
|
69
|
-
assert set(rows[0]) == {"id", "path", "tags", "title", "preview",
|
|
69
|
+
assert set(rows[0]) == {"id", "path", "tags", "title", "preview", "edits",
|
|
70
70
|
"created_at", "updated_at", "space_id", "space",
|
|
71
71
|
"recalled", "gets"}
|
|
72
72
|
|
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
"""What a write MUST carry, and what the answer tells you about what it carried.
|
|
2
|
+
|
|
3
|
+
Two failures of the same shape sit behind this file.
|
|
4
|
+
|
|
5
|
+
The first: `source` is described to every client as mandatory — an address by
|
|
6
|
+
which someone else finds the original — but nothing enforced it, so a corpus
|
|
7
|
+
fills with assertions nobody can check. The requirement is now declarable per
|
|
8
|
+
deployment (`MEMGRES_REQUIRED_FIELDS=source`), because it is a policy about a
|
|
9
|
+
corpus, not a property of the software: a scratch database has no use for it.
|
|
10
|
+
|
|
11
|
+
The second is subtler and is why the first went unnoticed for so long. The reply
|
|
12
|
+
to a write did not echo the provenance it had just recorded, so four edits in a
|
|
13
|
+
row went out with an empty `source` and every reply looked perfectly healthy.
|
|
14
|
+
A required field that the answer does not confirm is a field whose absence
|
|
15
|
+
nobody notices. It is echoed on WRITE only: provenance belongs to the revision,
|
|
16
|
+
not to the memory, so a read has no business claiming one — that question is
|
|
17
|
+
`history`/`blame`.
|
|
18
|
+
|
|
19
|
+
Alongside them, `edits`: how often a memory has been rewritten, which is a
|
|
20
|
+
different question from how often it is read, and the half of "what is hot"
|
|
21
|
+
the usage counters were missing.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
import os
|
|
25
|
+
import sys
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
|
|
28
|
+
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
29
|
+
|
|
30
|
+
import pytest # noqa: E402
|
|
31
|
+
|
|
32
|
+
psycopg = pytest.importorskip("psycopg")
|
|
33
|
+
|
|
34
|
+
from memgres.config import load # noqa: E402
|
|
35
|
+
from memgres.schema import migrate # noqa: E402
|
|
36
|
+
from memgres.store import MissingField, Store # noqa: E402
|
|
37
|
+
|
|
38
|
+
DSN = os.environ.get("MEMGRES_TEST_DSN",
|
|
39
|
+
"postgresql://memgres:memgres@localhost:55432/memgres")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _reachable() -> bool:
|
|
43
|
+
try:
|
|
44
|
+
psycopg.connect(DSN, connect_timeout=2).close()
|
|
45
|
+
return True
|
|
46
|
+
except Exception:
|
|
47
|
+
return False
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
pytestmark = pytest.mark.skipif(not _reachable(), reason="no test Postgres")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@pytest.fixture
|
|
54
|
+
def conn(monkeypatch):
|
|
55
|
+
with psycopg.connect(DSN, autocommit=True) as c, c.cursor() as cur:
|
|
56
|
+
cur.execute("DROP SCHEMA public CASCADE; CREATE SCHEMA public;")
|
|
57
|
+
for k in list(os.environ):
|
|
58
|
+
if k.startswith("MEMGRES_"):
|
|
59
|
+
monkeypatch.delenv(k, raising=False)
|
|
60
|
+
monkeypatch.setenv("MEMGRES_DATABASE_URL", DSN)
|
|
61
|
+
monkeypatch.setenv("MEMGRES_FTS_LANGUAGE", "simple")
|
|
62
|
+
monkeypatch.setenv("MEMGRES_EMBED_PROVIDER", "none")
|
|
63
|
+
c = psycopg.connect(DSN)
|
|
64
|
+
migrate(c, load())
|
|
65
|
+
yield c
|
|
66
|
+
c.close()
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _store(conn, monkeypatch, required: str = "") -> Store:
|
|
70
|
+
"""A store whose deployment declares — or doesn't — extra required fields.
|
|
71
|
+
Built on the SAME connection as its predecessor on purpose: that is how a
|
|
72
|
+
corpus written before the policy is modelled, old rows real, new rule live."""
|
|
73
|
+
monkeypatch.setenv("MEMGRES_REQUIRED_FIELDS", required)
|
|
74
|
+
return Store(load(), conn=conn)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# ─── the default: nothing extra is required ──────────────────────────────────
|
|
78
|
+
def test_nothing_is_required_unless_a_deployment_says_so(monkeypatch):
|
|
79
|
+
for k in list(os.environ):
|
|
80
|
+
if k.startswith("MEMGRES_"):
|
|
81
|
+
monkeypatch.delenv(k, raising=False)
|
|
82
|
+
monkeypatch.setenv("MEMGRES_DATABASE_URL", DSN)
|
|
83
|
+
assert load().required_fields == ()
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def test_without_the_policy_a_sourceless_write_is_fine(conn, monkeypatch):
|
|
87
|
+
s = _store(conn, monkeypatch, "")
|
|
88
|
+
m = s.write(body="один", path="a.b", title="A")
|
|
89
|
+
assert m.source is None
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
# ─── the refusal ─────────────────────────────────────────────────────────────
|
|
93
|
+
def test_creating_without_source_is_refused(conn, monkeypatch):
|
|
94
|
+
s = _store(conn, monkeypatch, "source")
|
|
95
|
+
with pytest.raises(MissingField) as e:
|
|
96
|
+
s.write(body="лимит имени OKX — 25", path="ops.okx", title="OKX")
|
|
97
|
+
msg = str(e.value)
|
|
98
|
+
assert "source" in msg
|
|
99
|
+
# The refusal must say what the field is FOR, or the requirement degenerates
|
|
100
|
+
# into filling the box: the second attempt would read "from the email".
|
|
101
|
+
assert "ADDRESS" in msg
|
|
102
|
+
assert "the user said" in msg
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def test_whitespace_is_not_a_source(conn, monkeypatch):
|
|
106
|
+
s = _store(conn, monkeypatch, "source")
|
|
107
|
+
with pytest.raises(MissingField):
|
|
108
|
+
s.write(body="один", path="a.b", title="A", source=" ")
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def test_an_edit_that_changes_the_body_needs_one_too(conn, monkeypatch):
|
|
112
|
+
s = _store(conn, monkeypatch, "")
|
|
113
|
+
s.write(body="один", path="a.b", title="A")
|
|
114
|
+
strict = _store(conn, monkeypatch, "source")
|
|
115
|
+
with pytest.raises(MissingField):
|
|
116
|
+
strict.write(at="a.b", body="два")
|
|
117
|
+
with pytest.raises(MissingField):
|
|
118
|
+
strict.write(at="a.b", replace=("два", "три"))
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def test_a_declared_field_can_be_something_other_than_source(conn, monkeypatch):
|
|
122
|
+
s = _store(conn, monkeypatch, "reason")
|
|
123
|
+
with pytest.raises(MissingField) as e:
|
|
124
|
+
s.write(body="один", path="a.b", title="A", source="host:/path 2026-08-27")
|
|
125
|
+
assert "reason" in str(e.value)
|
|
126
|
+
s.write(body="один", path="a.b", title="A",
|
|
127
|
+
source="host:/path 2026-08-27", reason="первая запись")
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
# ─── what is exempt ──────────────────────────────────────────────────────────
|
|
131
|
+
def test_moving_and_retagging_are_exempt(conn, monkeypatch):
|
|
132
|
+
"""They store no content. Requiring provenance for re-filing would make a
|
|
133
|
+
memory harder to organise than to write — friction unrelated to the point,
|
|
134
|
+
and the surest way to get the field filled with junk."""
|
|
135
|
+
s = _store(conn, monkeypatch, "")
|
|
136
|
+
s.write(body="один", path="a.b", title="A")
|
|
137
|
+
strict = _store(conn, monkeypatch, "source")
|
|
138
|
+
strict.write(at="a.b", path="a.c") # move
|
|
139
|
+
strict.write(at="a.c", tags=["x"]) # retag
|
|
140
|
+
assert strict.get(None, at="a.c").tags == ["x"]
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
# ─── the echo ────────────────────────────────────────────────────────────────
|
|
144
|
+
def test_the_answer_confirms_the_provenance_it_recorded(conn, monkeypatch):
|
|
145
|
+
s = _store(conn, monkeypatch, "source")
|
|
146
|
+
m = s.write(body="один", path="a.b", title="A",
|
|
147
|
+
source="192.168.1.121:/var/www/memgres 2026-08-27",
|
|
148
|
+
valid_at="2026-08-01")
|
|
149
|
+
assert m.source == "192.168.1.121:/var/www/memgres 2026-08-27"
|
|
150
|
+
assert str(m.valid_at) == "2026-08-01"
|
|
151
|
+
d = m.to_dict(stringify_dates=True)
|
|
152
|
+
assert d["source"] == "192.168.1.121:/var/www/memgres 2026-08-27"
|
|
153
|
+
assert d["valid_at"] == "2026-08-01"
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def test_a_read_claims_no_provenance(conn, monkeypatch):
|
|
157
|
+
"""A memory has no single source — its revisions do. Reporting one on a read
|
|
158
|
+
would attribute the whole record to whichever edit happened to be last."""
|
|
159
|
+
s = _store(conn, monkeypatch, "")
|
|
160
|
+
s.write(body="один", path="a.b", title="A", source="первый источник")
|
|
161
|
+
s.write(at="a.b", body="два", source="второй источник")
|
|
162
|
+
got = s.get(None, at="a.b")
|
|
163
|
+
assert got.source is None and got.valid_at is None
|
|
164
|
+
# Oldest first, as history reads.
|
|
165
|
+
assert [h["source"] for h in s.history(None, at="a.b")] == [
|
|
166
|
+
"первый источник", "второй источник"]
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
# ─── the edit counter ────────────────────────────────────────────────────────
|
|
170
|
+
def test_edits_count_revisions_after_the_creation(conn, monkeypatch):
|
|
171
|
+
s = _store(conn, monkeypatch, "")
|
|
172
|
+
s.write(body="один", path="a.b", title="A")
|
|
173
|
+
assert s.get(None, at="a.b").usage["edits"] == 0
|
|
174
|
+
s.write(at="a.b", body="два")
|
|
175
|
+
s.write(at="a.b", body="три")
|
|
176
|
+
assert s.get(None, at="a.b").usage["edits"] == 2
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def test_browsing_reports_the_same_count(conn, monkeypatch):
|
|
180
|
+
"""`list` is where "what is hot" gets asked over a whole subtree, so the
|
|
181
|
+
number has to be there too — and has to agree with the one `get` reports."""
|
|
182
|
+
s = _store(conn, monkeypatch, "")
|
|
183
|
+
s.write(body="один", path="a.b", title="A")
|
|
184
|
+
s.write(at="a.b", body="два")
|
|
185
|
+
row = [r for r in s.list(None) if r["path"] == "a.b"][0]
|
|
186
|
+
assert row["edits"] == 1 == s.get(None, at="a.b").usage["edits"]
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def test_a_move_is_a_revision_too(conn, monkeypatch):
|
|
190
|
+
"""Deliberate: `seq` counts what the history holds, and a move IS an entry
|
|
191
|
+
there. A number that skipped them would disagree with `history`."""
|
|
192
|
+
s = _store(conn, monkeypatch, "")
|
|
193
|
+
s.write(body="один", path="a.b", title="A")
|
|
194
|
+
s.write(at="a.b", path="a.c")
|
|
195
|
+
assert s.get(None, at="a.c").usage["edits"] == 1
|
|
@@ -34,7 +34,8 @@ def test_top_level_keys_and_limits(monkeypatch):
|
|
|
34
34
|
monkeypatch.setenv("MEMGRES_MAX_TITLE_BYTES", "128")
|
|
35
35
|
info = server_info(load())
|
|
36
36
|
assert set(info) == {"version", "schema_version", "limits", "embed",
|
|
37
|
-
"retention", "
|
|
37
|
+
"retention", "write_requirements",
|
|
38
|
+
"recall_modes", "vector_backend",
|
|
38
39
|
"key_mode", "fts_language"}
|
|
39
40
|
monkeypatch.setenv("MEMGRES_LIST_BODIES_MAX_BYTES", "4096")
|
|
40
41
|
info = server_info(load())
|
|
@@ -129,6 +130,24 @@ def test_renew_on_read_is_not_advertised_when_nothing_expires(monkeypatch):
|
|
|
129
130
|
assert info["retention"]["renew_on_read"] is False
|
|
130
131
|
|
|
131
132
|
|
|
133
|
+
def test_what_a_write_must_carry_is_announced(monkeypatch):
|
|
134
|
+
"""A rule a client can only learn from a refusal is a rule it satisfies with
|
|
135
|
+
junk on the second attempt: it has already composed the memory by then."""
|
|
136
|
+
_clear(monkeypatch)
|
|
137
|
+
monkeypatch.setenv("MEMGRES_REQUIRED_FIELDS", "source")
|
|
138
|
+
info = server_info(load())
|
|
139
|
+
assert info["write_requirements"]["fields"] == ["source"]
|
|
140
|
+
# `_clear` turns captions off for this file, so the flag reports the
|
|
141
|
+
# deployment as configured — which is exactly the point of announcing it.
|
|
142
|
+
assert info["write_requirements"]["title"] is False
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def test_a_deployment_requiring_nothing_extra_says_so(monkeypatch):
|
|
146
|
+
_clear(monkeypatch)
|
|
147
|
+
info = server_info(load())
|
|
148
|
+
assert info["write_requirements"]["fields"] == []
|
|
149
|
+
|
|
150
|
+
|
|
132
151
|
def test_no_secrets_leak(monkeypatch):
|
|
133
152
|
_clear(monkeypatch)
|
|
134
153
|
monkeypatch.setenv("MEMGRES_TOKEN", "mgk_supersecret_token")
|
|
@@ -166,7 +166,8 @@ def test_list_memories_route(client):
|
|
|
166
166
|
def test_info_route(client):
|
|
167
167
|
info = client.get("/info").json()
|
|
168
168
|
assert set(info) == {"version", "schema_version", "limits", "embed",
|
|
169
|
-
"retention", "
|
|
169
|
+
"retention", "write_requirements",
|
|
170
|
+
"recall_modes", "vector_backend",
|
|
170
171
|
"key_mode", "fts_language"}
|
|
171
172
|
# the fixture keeps everything, and the route has to say so rather than
|
|
172
173
|
# leaving a client to infer it from a missing field
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|