memgres 0.12.1__tar.gz → 0.12.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {memgres-0.12.1 → memgres-0.12.3}/PKG-INFO +1 -1
- {memgres-0.12.1 → memgres-0.12.3}/memgres/_version.py +1 -1
- {memgres-0.12.1 → memgres-0.12.3}/memgres/mcp_server.py +9 -1
- {memgres-0.12.1 → memgres-0.12.3}/memgres/store.py +107 -12
- {memgres-0.12.1 → memgres-0.12.3}/memgres/vector/base.py +12 -2
- {memgres-0.12.1 → memgres-0.12.3}/memgres.egg-info/PKG-INFO +1 -1
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_require_title.py +34 -1
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_required_fields.py +20 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_store_integration.py +43 -0
- {memgres-0.12.1 → memgres-0.12.3}/LICENSE +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/README.md +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/__init__.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/admin.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/admin_cli.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/blame.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/bootstrap.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/config.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/delimiters.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/diffing.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/embed_worker.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/embeddings.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/healthcheck.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/identity.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/indexing.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/info.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/lines.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/links.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0001_core.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0002_identity.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0003_history_author.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0004_title.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0005_chunk_index.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0006_reader_floor.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0007_embed_retry.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0008_service_roles.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0009_create_namespace_right.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0010_namespace_alias.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0011_drop_default_namespace.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0012_user_profile.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0013_hash_version.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0014_access_request_no_fk.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0015_normalize_tags.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0016_valid_at.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0017_memory_link.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0018_links_built.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0019_memory_usage.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0020_memory_usage_no_fk.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0021_enrollment_key.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0022_user_disabled.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/migrations/0023_relink_after_parser_fix.sql +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/paths.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/periodic.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/reembed.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/relink.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/schema.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/search.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/segments.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/server.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/tags.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/token_cli.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/vector/__init__.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/vector/pgvector.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/vector/qdrant.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres/worker.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres.egg-info/SOURCES.txt +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres.egg-info/dependency_links.txt +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres.egg-info/entry_points.txt +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres.egg-info/requires.txt +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/memgres.egg-info/top_level.txt +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/pyproject.toml +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/setup.cfg +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_admin_two_way.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_blame_integration.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_chunk_index.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_claim_and_reembed.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_config.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_diffing.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_embed_worker.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_embeddings.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_enrollment.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_healthcheck.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_identity_integration.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_lexical_match.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_limits.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_links.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_list.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_mcp_admin_tools.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_mcp_error_messages.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_mcp_http_transport.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_mcp_instructions.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_mcp_recall_schema.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_mcp_tool_visibility.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_mcp_tool_visibility_http.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_migration_upgrade.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_multi_space_search.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_path_addressing.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_qdrant_ca.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_qdrant_integration.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_replace_build.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_retention.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_roles_bootstrap.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_search_integration.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_security_followups.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_security_integration.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_segments.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_segments_store.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_server_info.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_server_integration.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_snippets.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_tags.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_token_sink.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_usage.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_valid_at.py +0 -0
- {memgres-0.12.1 → memgres-0.12.3}/tests/test_write_ergonomics.py +0 -0
|
@@ -61,7 +61,11 @@ SourceArg = Annotated[Optional[str], Field(
|
|
|
61
61
|
"sender -> recipient, date, subject; messenger, who with whom, date; "
|
|
62
62
|
"machine + project + session/transcript for an agent run; full URL + "
|
|
63
63
|
"date read. 'email', 'the meeting', 'the user said' is not one — "
|
|
64
|
-
"nothing can be reached through it, so the fact can only be believed."
|
|
64
|
+
"nothing can be reached through it, so the fact can only be believed. "
|
|
65
|
+
"It is recorded on THIS revision, not on the memory: a memory has no "
|
|
66
|
+
"single origin, its edits do. So a later read does not carry it — "
|
|
67
|
+
"`memory_blame` says where a given line came from, `memory_history` "
|
|
68
|
+
"where each revision did.")]
|
|
65
69
|
ReasonArg = Annotated[Optional[str], Field(
|
|
66
70
|
default=None,
|
|
67
71
|
description="Why this write happened — what changed and why, in one line. Kept "
|
|
@@ -503,6 +507,10 @@ def build_server(cfg: Optional[Config] = None):
|
|
|
503
507
|
The answer carries `usage`: how often this memory has surfaced in search
|
|
504
508
|
(`recalled`) and been fetched (`gets`), and when each last happened.
|
|
505
509
|
|
|
510
|
+
It carries NO `source`/`valid_at`, and that is deliberate: provenance
|
|
511
|
+
belongs to a revision, not to the document. Ask `memory_blame` where a
|
|
512
|
+
particular line came from, or `memory_history` where each revision did.
|
|
513
|
+
|
|
506
514
|
`lines` ("40-80", "5", "1,10-12") returns only part of a long body. The
|
|
507
515
|
answer is then marked `partial`, carries `total_lines`, and has NO
|
|
508
516
|
`content_hash` — do not send a slice back as a whole `body`, or
|
|
@@ -251,12 +251,21 @@ class Memory:
|
|
|
251
251
|
# Provenance of the REVISION this call just wrote — never of the memory,
|
|
252
252
|
# which has no single source: `source`/`reason`/`valid_at` live on the
|
|
253
253
|
# history row because different edits speak to different origins and dates.
|
|
254
|
-
#
|
|
255
|
-
#
|
|
256
|
-
#
|
|
257
|
-
#
|
|
254
|
+
# Echoed back on a write because a required field the answer does not confirm
|
|
255
|
+
# is a field whose absence nobody notices: four edits in a row went out with
|
|
256
|
+
# an empty `source` and every reply looked fine.
|
|
257
|
+
#
|
|
258
|
+
# `is_write` decides whether they are SERIALIZED at all, and that is the
|
|
259
|
+
# point: a read used to emit `"source": null`, which does not read as "this
|
|
260
|
+
# memory has no such field" but as "the value was there and is now gone".
|
|
261
|
+
# Two readers in a row concluded the field was being silently dropped and
|
|
262
|
+
# went looking for the bug — one of them spent a day on it and wrote the
|
|
263
|
+
# workaround into memory. So on a read the keys are ABSENT, and the question
|
|
264
|
+
# they answer belongs to `history`/`blame`, where the answer is per-revision
|
|
265
|
+
# and per-line rather than one value pretending to describe the document.
|
|
258
266
|
source: Optional[str] = None
|
|
259
267
|
valid_at: object = None
|
|
268
|
+
is_write: bool = False
|
|
260
269
|
|
|
261
270
|
def to_dict(self, *, stringify_dates: bool = False) -> dict:
|
|
262
271
|
"""Serialize for an API layer. ``stringify_dates`` str()-coerces the
|
|
@@ -264,9 +273,8 @@ class Memory:
|
|
|
264
273
|
datetimes itself, so the HTTP layer passes them through raw)."""
|
|
265
274
|
def d(v):
|
|
266
275
|
return (str(v) if v is not None else None) if stringify_dates else v
|
|
267
|
-
|
|
276
|
+
out = {"id": self.id, "content_hash": self.content_hash, "body": self.body,
|
|
268
277
|
"title": self.title, "tags": self.tags, "path": self.path,
|
|
269
|
-
"source": self.source, "valid_at": d(self.valid_at),
|
|
270
278
|
"seq": self.seq, "created_at": d(self.created_at),
|
|
271
279
|
"updated_at": d(self.updated_at), "expires_at": d(self.expires_at),
|
|
272
280
|
"created": self.created, "moved_from": self.moved_from,
|
|
@@ -276,6 +284,12 @@ class Memory:
|
|
|
276
284
|
"last_recall_at": d(self.usage["last_recall_at"]),
|
|
277
285
|
"last_get_at": d(self.usage["last_get_at"])}
|
|
278
286
|
if self.usage else None)}
|
|
287
|
+
if self.is_write:
|
|
288
|
+
# Only here: the provenance of the revision just written, confirmed
|
|
289
|
+
# back to its author. A read carries no such keys (see `is_write`).
|
|
290
|
+
out["source"] = self.source
|
|
291
|
+
out["valid_at"] = d(self.valid_at)
|
|
292
|
+
return out
|
|
279
293
|
|
|
280
294
|
|
|
281
295
|
def _sha(text: str) -> str:
|
|
@@ -437,6 +451,57 @@ def _slice_lines(m: "Memory", spec: str) -> "Memory":
|
|
|
437
451
|
return m
|
|
438
452
|
|
|
439
453
|
|
|
454
|
+
|
|
455
|
+
# How much body text a "why did my replace miss" hint may quote, per side.
|
|
456
|
+
_HINT_CONTEXT = 48
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
def _why_replace_missed(body: str, old: str) -> str:
|
|
460
|
+
"""Say WHY a substring edit found nothing — the refusal alone sends the author
|
|
461
|
+
to re-read the whole record, when the useful answer is almost always "your
|
|
462
|
+
quote is not what the body says, here is what it says".
|
|
463
|
+
|
|
464
|
+
Three causes, in the order they actually happen:
|
|
465
|
+
|
|
466
|
+
1. **Whitespace.** Bodies are hard-wrapped, so a phrase that reads as one line
|
|
467
|
+
on screen contains a newline. Retyped as a space, it can never match.
|
|
468
|
+
2. **Case.**
|
|
469
|
+
3. **Everything else** — then the most useful thing is the point where the
|
|
470
|
+
quote stops agreeing with the body, and what the body has instead.
|
|
471
|
+
|
|
472
|
+
The quote itself is usually reconstructed by eye rather than copied, which is
|
|
473
|
+
why the hint ends by naming the one reliable source: `memory_get`. A recall
|
|
474
|
+
snippet is a slice of the body (`lines` says which) EXCEPT on the ts_headline
|
|
475
|
+
path, where Postgres rebuilds the text from tokens and `lines` is null — that
|
|
476
|
+
one is not quotable at all.
|
|
477
|
+
"""
|
|
478
|
+
squashed_old = " ".join(old.split())
|
|
479
|
+
if squashed_old and squashed_old in " ".join(body.split()):
|
|
480
|
+
return (" — the text IS in the body, but the whitespace differs: bodies "
|
|
481
|
+
"wrap, so a line break in the record reads as a space on screen. "
|
|
482
|
+
"Copy the line from `memory_get`, not from a recall snippet")
|
|
483
|
+
if old.lower() in body.lower():
|
|
484
|
+
return (" — it is there in a different case. Copy it from `memory_get`, "
|
|
485
|
+
"not from a recall snippet")
|
|
486
|
+
# Longest prefix of `old` the body still agrees with: the point of divergence
|
|
487
|
+
# is what the author needs to see. Binary search — bodies are capped, but a
|
|
488
|
+
# linear scan would re-scan the whole body once per character.
|
|
489
|
+
lo, hi = 0, len(old)
|
|
490
|
+
while lo < hi:
|
|
491
|
+
mid = (lo + hi + 1) // 2
|
|
492
|
+
if old[:mid] in body:
|
|
493
|
+
lo = mid
|
|
494
|
+
else:
|
|
495
|
+
hi = mid - 1
|
|
496
|
+
if lo == 0:
|
|
497
|
+
return (" — not one character of it is in the body; wrong memory, or the "
|
|
498
|
+
"quote was written from memory rather than copied from `memory_get`")
|
|
499
|
+
at = body.find(old[:lo])
|
|
500
|
+
reads = body[at:at + lo + _HINT_CONTEXT]
|
|
501
|
+
return (f" — the first {lo} chars match, then they part: the body reads "
|
|
502
|
+
f"{reads!r}. Copy from `memory_get`, not from a recall snippet")
|
|
503
|
+
|
|
504
|
+
|
|
440
505
|
class Store:
|
|
441
506
|
def __init__(self, cfg: Config, embedder: Optional[Embedder] = None,
|
|
442
507
|
conn: Optional["psycopg.Connection"] = None,
|
|
@@ -643,7 +708,8 @@ class Store:
|
|
|
643
708
|
# diff can introduce the stray tag just as a whole body can.
|
|
644
709
|
m.warnings = write_warnings(m.body)
|
|
645
710
|
# The revision's own provenance, straight back to whoever wrote it.
|
|
646
|
-
|
|
711
|
+
# `is_write` is what puts these on the wire; a read leaves them off.
|
|
712
|
+
m.source, m.valid_at, m.is_write = source, valid_at, True
|
|
647
713
|
return m
|
|
648
714
|
|
|
649
715
|
def _check_path_free(self, ns: str, path: Optional[str],
|
|
@@ -758,11 +824,38 @@ class Store:
|
|
|
758
824
|
f"this deployment requires `{field}` on every write that stores "
|
|
759
825
|
f"content — {help_text}")
|
|
760
826
|
|
|
827
|
+
# How much of the title to quote either side of the cut. Enough to recognise
|
|
828
|
+
# the words, short enough that a pathological title cannot turn the refusal
|
|
829
|
+
# into a wall of text.
|
|
830
|
+
_TITLE_CUT_CONTEXT = 40
|
|
831
|
+
|
|
761
832
|
def _check_title_size(self, title: Optional[str]):
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
833
|
+
"""Refuse an oversized title, and SHOW WHERE IT STOPS FITTING.
|
|
834
|
+
|
|
835
|
+
The ceiling is in bytes because that is what storage and the wire count,
|
|
836
|
+
but the author is writing characters — and in UTF-8 the exchange rate
|
|
837
|
+
depends on the script: Latin runs ~1 B/char, Cyrillic ~2, so the same 256
|
|
838
|
+
B is ~256 characters in one language and ~148 in another. A refusal that
|
|
839
|
+
names only bytes leaves the author trimming blind, one attempt at a time
|
|
840
|
+
(observed: four rejected writes in a row on one record). So the message
|
|
841
|
+
carries the character counts and quotes the cut itself, marked with ✂ —
|
|
842
|
+
what survives on the left, what has to go on the right."""
|
|
843
|
+
if title is None:
|
|
844
|
+
return
|
|
845
|
+
size, cap = byte_len(title), self.cfg.max_title_bytes
|
|
846
|
+
if size <= cap:
|
|
847
|
+
return
|
|
848
|
+
# Truncating BYTES can land mid-character; decoding with "ignore" drops
|
|
849
|
+
# that partial character, which is exactly the last one that does NOT fit.
|
|
850
|
+
fits = title.encode("utf-8")[:cap].decode("utf-8", "ignore")
|
|
851
|
+
over = title[len(fits):]
|
|
852
|
+
ctx = self._TITLE_CUT_CONTEXT
|
|
853
|
+
left = ("…" if len(fits) > ctx else "") + fits[-ctx:]
|
|
854
|
+
right = over[:ctx] + ("…" if len(over) > ctx else "")
|
|
855
|
+
raise TooLarge(
|
|
856
|
+
f"title is {size}B > MEMGRES_MAX_TITLE_BYTES {cap}: "
|
|
857
|
+
f"{len(title)} chars, {len(fits)} fit, drop {len(over)} — "
|
|
858
|
+
f"{left}[✂]{right}")
|
|
766
859
|
|
|
767
860
|
def _create(self, ns, author, body, path, tags, source, reason,
|
|
768
861
|
title=None, valid_at=None) -> Memory:
|
|
@@ -853,7 +946,9 @@ class Store:
|
|
|
853
946
|
raise Conflict(f"stale replace: base {base_hash[:12]} != current {cur_hash[:12]}")
|
|
854
947
|
count = cur_body.count(old)
|
|
855
948
|
if count == 0:
|
|
856
|
-
raise ReplaceNotFound(
|
|
949
|
+
raise ReplaceNotFound(
|
|
950
|
+
f"replace text not found in body: {old[:60]!r}"
|
|
951
|
+
f"{_why_replace_missed(cur_body, old)}")
|
|
857
952
|
if count > 1 and not replace_all:
|
|
858
953
|
raise AmbiguousReplace(
|
|
859
954
|
f"replace text occurs {count}× — pass replace_all, or add "
|
|
@@ -66,6 +66,12 @@ class Hit:
|
|
|
66
66
|
# set by a backend's grouped search: the (start, end) char offsets of the
|
|
67
67
|
# winning chunk, so attach_snippets slices the snippet with no re-embedding.
|
|
68
68
|
chunk_span: Optional[Tuple[int, int]] = None
|
|
69
|
+
# When the memory was last changed. Carried because recall is where staleness
|
|
70
|
+
# is decided: a caller picks which hit to read from the ranked list, and
|
|
71
|
+
# without a date the choice is made on wording alone. Every other read
|
|
72
|
+
# (`get`, `list`, `history`) already says when — this was the one that did
|
|
73
|
+
# not, and it is the one that ranks.
|
|
74
|
+
updated_at: object = None
|
|
69
75
|
|
|
70
76
|
def to_recall_dict(self) -> dict:
|
|
71
77
|
"""The recall wire shape — one definition, so the HTTP and MCP layers
|
|
@@ -74,6 +80,10 @@ class Hit:
|
|
|
74
80
|
return {"id": self.id, "title": self.title, "tags": self.tags,
|
|
75
81
|
"path": self.path, "score": self.score, "snippet": self.snippet,
|
|
76
82
|
"kind": self.kind, "lines": self.lines,
|
|
83
|
+
# str() rather than a datetime: the MCP layer needs plain
|
|
84
|
+
# strings, and FastAPI passes a string through unchanged, so one
|
|
85
|
+
# shape serves both transports.
|
|
86
|
+
"updated_at": str(self.updated_at) if self.updated_at else None,
|
|
77
87
|
"space_id": self.namespace, "space": self.space}
|
|
78
88
|
|
|
79
89
|
|
|
@@ -86,13 +96,13 @@ def _vec_literal(vec: Sequence[float]) -> str:
|
|
|
86
96
|
# definition so lexical and both chunk backends SELECT the same set and adding a
|
|
87
97
|
# field is one edit. When a backend ranks in SQL and appends its score column
|
|
88
98
|
# AFTER these, the score is row[len(HIT_COLUMNS)].
|
|
89
|
-
HIT_COLUMNS = "id, body, tags, path::text, title, namespace"
|
|
99
|
+
HIT_COLUMNS = "id, body, tags, path::text, title, namespace, updated_at"
|
|
90
100
|
|
|
91
101
|
|
|
92
102
|
def row_to_hit(row, score: float) -> "Hit":
|
|
93
103
|
"""Build a Hit from a (HIT_COLUMNS) row plus a separately-supplied score."""
|
|
94
104
|
return Hit(str(row[0]), row[1], list(row[2]), row[3], float(score),
|
|
95
|
-
title=row[4], namespace=str(row[5]))
|
|
105
|
+
title=row[4], namespace=str(row[5]), updated_at=row[6])
|
|
96
106
|
|
|
97
107
|
|
|
98
108
|
def as_namespaces(ns) -> List[str]:
|
|
@@ -26,7 +26,7 @@ psycopg = pytest.importorskip("psycopg")
|
|
|
26
26
|
|
|
27
27
|
from memgres.config import load # noqa: E402
|
|
28
28
|
from memgres.schema import migrate # noqa: E402
|
|
29
|
-
from memgres.store import MissingTitle, Store # noqa: E402
|
|
29
|
+
from memgres.store import MissingTitle, Store, TooLarge # noqa: E402
|
|
30
30
|
|
|
31
31
|
DSN = os.environ.get("MEMGRES_TEST_DSN",
|
|
32
32
|
"postgresql://memgres:memgres@localhost:55432/memgres")
|
|
@@ -181,3 +181,36 @@ def test_a_create_with_no_body_blames_the_body(conn, monkeypatch):
|
|
|
181
181
|
s = _store(conn, monkeypatch, True)
|
|
182
182
|
with pytest.raises(ValueError, match="needs a body"):
|
|
183
183
|
s.write(path="a.b")
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
# ─── the size ceiling explains itself ────────────────────────────────────────
|
|
187
|
+
def test_an_oversized_title_shows_where_it_stops_fitting(conn, monkeypatch):
|
|
188
|
+
"""The ceiling counts bytes, the author counts characters, and in UTF-8 the
|
|
189
|
+
rate depends on the script — so a refusal naming only bytes leaves them
|
|
190
|
+
trimming blind. The message quotes the cut: what fits, what has to go."""
|
|
191
|
+
monkeypatch.setenv("MEMGRES_MAX_TITLE_BYTES", "40")
|
|
192
|
+
s = _store(conn, monkeypatch, True)
|
|
193
|
+
title = "Лимит заголовка меряется в байтах, а не в символах"
|
|
194
|
+
with pytest.raises(TooLarge) as e:
|
|
195
|
+
s.write(body="one", path="a.b", title=title)
|
|
196
|
+
msg = str(e.value)
|
|
197
|
+
# counts in BOTH units, and how much to drop
|
|
198
|
+
assert f"{len(title.encode())}B" in msg and "40" in msg
|
|
199
|
+
assert f"{len(title)} chars" in msg
|
|
200
|
+
# the cut is quoted, and the two sides really are the two sides
|
|
201
|
+
assert "[✂]" in msg
|
|
202
|
+
fits = title.encode()[:40].decode("utf-8", "ignore")
|
|
203
|
+
assert f"{len(fits)} fit" in msg
|
|
204
|
+
assert f"drop {len(title) - len(fits)}" in msg
|
|
205
|
+
assert fits[-10:] + "[✂]" in msg # left of the cut survives
|
|
206
|
+
assert "[✂]" + title[len(fits):][:10] in msg # right of it is what to drop
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def test_a_title_that_only_just_fits_is_accepted(conn, monkeypatch):
|
|
210
|
+
"""The boundary itself is not an error — byte_len == cap must pass, or the
|
|
211
|
+
limit silently means one byte less than it says."""
|
|
212
|
+
monkeypatch.setenv("MEMGRES_MAX_TITLE_BYTES", "40")
|
|
213
|
+
s = _store(conn, monkeypatch, True)
|
|
214
|
+
title = "я" * 20 # exactly 40 bytes
|
|
215
|
+
assert len(title.encode()) == 40
|
|
216
|
+
assert s.write(body="one", path="a.b", title=title).title == title
|
|
@@ -166,6 +166,26 @@ def test_a_read_claims_no_provenance(conn, monkeypatch):
|
|
|
166
166
|
"первый источник", "второй источник"]
|
|
167
167
|
|
|
168
168
|
|
|
169
|
+
def test_a_read_does_not_even_carry_the_keys(conn, monkeypatch):
|
|
170
|
+
"""`"source": null` on a read does not say "no such field here" — it reads as
|
|
171
|
+
"the value was there and is gone". Two readers concluded exactly that and went
|
|
172
|
+
hunting for a bug that did not exist; one spent a day on it. So a read omits
|
|
173
|
+
the keys outright, and the question belongs to `history`/`blame`."""
|
|
174
|
+
s = _store(conn, monkeypatch, "")
|
|
175
|
+
s.write(body="один", path="a.b", title="A", source="источник")
|
|
176
|
+
d = s.get(None, at="a.b").to_dict(stringify_dates=True)
|
|
177
|
+
assert "source" not in d and "valid_at" not in d
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def test_a_write_carries_them_even_when_empty(conn, monkeypatch):
|
|
181
|
+
"""The other half: on a write the keys are always there, because a required
|
|
182
|
+
field the answer does not confirm is one whose absence nobody notices. Null
|
|
183
|
+
here means "you sent nothing", which is exactly what the writer must see."""
|
|
184
|
+
s = _store(conn, monkeypatch, "")
|
|
185
|
+
d = s.write(body="один", path="a.b", title="A").to_dict(stringify_dates=True)
|
|
186
|
+
assert d["source"] is None and d["valid_at"] is None
|
|
187
|
+
|
|
188
|
+
|
|
169
189
|
# ─── the edit counter ────────────────────────────────────────────────────────
|
|
170
190
|
def test_edits_count_revisions_after_the_creation(conn, monkeypatch):
|
|
171
191
|
s = _store(conn, monkeypatch, "")
|
|
@@ -230,6 +230,38 @@ def test_replace_not_found_leaves_record_untouched(store):
|
|
|
230
230
|
assert again.body == "alpha\nbeta\n" and again.seq == 1
|
|
231
231
|
|
|
232
232
|
|
|
233
|
+
def test_a_missed_replace_says_the_quote_crossed_a_line_break(store):
|
|
234
|
+
"""The commonest miss: the body wraps, the phrase reads as one line on
|
|
235
|
+
screen, and the quote comes back with a space where the record has \n. The
|
|
236
|
+
refusal has to name that, or the author goes re-reading the whole record."""
|
|
237
|
+
store.write(body="хвост ниши по деньгам — перепродажа\nExa через x402\n",
|
|
238
|
+
path="a.b", title="A")
|
|
239
|
+
with pytest.raises(ReplaceNotFound) as e:
|
|
240
|
+
store.write(at="a.b", replace=("перепродажа Exa", "перепродажа Exa Search"))
|
|
241
|
+
msg = str(e.value)
|
|
242
|
+
assert "whitespace differs" in msg and "memory_get" in msg
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def test_a_missed_replace_shows_where_the_quote_parts_from_the_body(store):
|
|
246
|
+
"""When it is not whitespace, the useful answer is the point of divergence
|
|
247
|
+
and what the body actually says there."""
|
|
248
|
+
store.write(body="верх ниши — **«rail dead»** значит другое\n",
|
|
249
|
+
path="a.c", title="A")
|
|
250
|
+
with pytest.raises(ReplaceNotFound) as e:
|
|
251
|
+
store.write(at="a.c", replace=("верх ниши — «rail dead»", "x"))
|
|
252
|
+
msg = str(e.value)
|
|
253
|
+
assert "chars match" in msg and "«rail dead»**" in msg
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def test_a_replace_of_something_absent_says_so_plainly(store):
|
|
257
|
+
"""No prefix at all is a different mistake — wrong record, or a quote written
|
|
258
|
+
from memory — and must not be dressed up as a near miss."""
|
|
259
|
+
store.write(body="alpha\nbeta\n", path="a.d", title="A")
|
|
260
|
+
with pytest.raises(ReplaceNotFound) as e:
|
|
261
|
+
store.write(at="a.d", replace=("zzz", "x"))
|
|
262
|
+
assert "not one character" in str(e.value)
|
|
263
|
+
|
|
264
|
+
|
|
233
265
|
def test_replace_ambiguous_requires_all_or_context(store):
|
|
234
266
|
m = store.write(body="x x x\n")
|
|
235
267
|
with pytest.raises(AmbiguousReplace):
|
|
@@ -362,6 +394,17 @@ def test_the_light_pass_returns_no_text(store):
|
|
|
362
394
|
assert row["snippet"] is None and row["title"] == "Fruit Notes"
|
|
363
395
|
|
|
364
396
|
|
|
397
|
+
def test_a_recall_hit_says_when_the_memory_last_changed(store):
|
|
398
|
+
"""Recall is where staleness is decided — the caller picks what to read from
|
|
399
|
+
the ranked list. Without a date that choice is made on wording alone, and a
|
|
400
|
+
fact from July reads exactly like one from yesterday."""
|
|
401
|
+
m = store.write(body="the body mentions apples\n", title="Fruit Notes",
|
|
402
|
+
path="notes.fruit")
|
|
403
|
+
[hit] = store.recall(None, "fruit")
|
|
404
|
+
assert hit.updated_at is not None
|
|
405
|
+
assert hit.to_recall_dict()["updated_at"] == str(m.updated_at)
|
|
406
|
+
|
|
407
|
+
|
|
365
408
|
def test_recall_respects_tag_filter(store):
|
|
366
409
|
store.write(body="x\n", title="alpha report", tags=["keep"])
|
|
367
410
|
store.write(body="y\n", title="alpha summary", tags=["drop"])
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|