dlightrag-memory 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,260 @@
1
+ # Copyright 2025-2026 Hanlian Lu. SPDX-License-Identifier: Apache-2.0
2
+ """The host-neutral Profile Memory façade."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import asyncio
7
+ from dataclasses import dataclass
8
+ from datetime import datetime
9
+
10
+ from dlightrag_memory.fusion import rrf_fuse
11
+ from dlightrag_memory.models import (
12
+ MemoryKind,
13
+ MemoryOperation,
14
+ MemoryOperationReceipt,
15
+ MemoryProvenance,
16
+ MemoryRecord,
17
+ )
18
+ from dlightrag_memory.policy import (
19
+ RECALL_CHAR_BUDGET,
20
+ RECALL_TOP_K,
21
+ evaluate_memory_operation,
22
+ )
23
+ from dlightrag_memory.ports import SearchCandidate
24
+ from dlightrag_memory.recall import recall_recency
25
+ from dlightrag_memory.store import MemoryStore, OperationGuard
26
+
27
+ _SEARCH_DEADLINE_SECONDS = 2.0
28
+ _HEADER_CHARS = 160
29
+
30
+
31
+ @dataclass(frozen=True, slots=True)
32
+ class RecallResult:
33
+ """Structured query-aware recall result."""
34
+
35
+ records: tuple[MemoryRecord, ...]
36
+ strategy: str
37
+ candidates: tuple[SearchCandidate, ...] = ()
38
+ degraded: tuple[str, ...] = ()
39
+ content_chars: int = 0
40
+
41
+ @property
42
+ def skipped(self) -> bool:
43
+ return bool(self.degraded)
44
+
45
+
46
+ class Memory:
47
+ """Cross-conversation owner Profile Memory behind one deep mutation seam."""
48
+
49
+ def __init__(self, store: MemoryStore) -> None:
50
+ self._store = store
51
+
52
+ async def apply(
53
+ self,
54
+ operation: MemoryOperation,
55
+ *,
56
+ guard: OperationGuard | None = None,
57
+ ) -> MemoryOperationReceipt:
58
+ """Validate and atomically settle one idempotent operation."""
59
+ evaluate_memory_operation(operation)
60
+ return await self._store.apply_operation(operation, guard=guard)
61
+
62
+ async def remember(
63
+ self,
64
+ *,
65
+ owner_id: str,
66
+ kind: MemoryKind,
67
+ body: str,
68
+ provenance: MemoryProvenance,
69
+ idempotency_key: str,
70
+ supersedes_id: str | None = None,
71
+ mutation_scope: str | None = None,
72
+ mutation_limit: int | None = None,
73
+ guard: OperationGuard | None = None,
74
+ ) -> MemoryOperationReceipt:
75
+ return await self.apply(
76
+ MemoryOperation(
77
+ owner_id=owner_id,
78
+ idempotency_key=idempotency_key,
79
+ action="remember",
80
+ provenance=provenance,
81
+ kind=kind,
82
+ body=body,
83
+ supersedes_id=supersedes_id,
84
+ mutation_scope=mutation_scope,
85
+ mutation_limit=mutation_limit,
86
+ ),
87
+ guard=guard,
88
+ )
89
+
90
+ async def forget(
91
+ self,
92
+ *,
93
+ owner_id: str,
94
+ provenance: MemoryProvenance,
95
+ idempotency_key: str,
96
+ memory_id: str | None = None,
97
+ body: str | None = None,
98
+ mutation_scope: str | None = None,
99
+ mutation_limit: int | None = None,
100
+ guard: OperationGuard | None = None,
101
+ ) -> MemoryOperationReceipt:
102
+ return await self.apply(
103
+ MemoryOperation(
104
+ owner_id=owner_id,
105
+ idempotency_key=idempotency_key,
106
+ action="forget",
107
+ provenance=provenance,
108
+ memory_id=memory_id,
109
+ body=body or "",
110
+ mutation_scope=mutation_scope,
111
+ mutation_limit=mutation_limit,
112
+ ),
113
+ guard=guard,
114
+ )
115
+
116
+ async def undo(
117
+ self,
118
+ *,
119
+ owner_id: str,
120
+ change_id: str,
121
+ provenance: MemoryProvenance,
122
+ idempotency_key: str,
123
+ guard: OperationGuard | None = None,
124
+ ) -> MemoryOperationReceipt:
125
+ return await self.apply(
126
+ MemoryOperation(
127
+ owner_id=owner_id,
128
+ idempotency_key=idempotency_key,
129
+ action="undo",
130
+ provenance=provenance,
131
+ target_change_id=change_id,
132
+ ),
133
+ guard=guard,
134
+ )
135
+
136
+ async def clear(
137
+ self,
138
+ *,
139
+ owner_id: str,
140
+ guard: OperationGuard | None = None,
141
+ ) -> int:
142
+ """Physically erase one owner's complete Profile Memory schema state."""
143
+ return await self._store.clear_owner(owner_id=owner_id, guard=guard)
144
+
145
+ async def count_active(self, *, owner_id: str) -> int:
146
+ return await self._store.count_active(owner_id=owner_id)
147
+
148
+ async def browse(
149
+ self,
150
+ *,
151
+ owner_id: str,
152
+ cursor: tuple[datetime, str] | None = None,
153
+ limit: int = 50,
154
+ ) -> tuple[tuple[MemoryRecord, ...], tuple[datetime, str] | None]:
155
+ return await self._store.list_active_page(owner_id=owner_id, after=cursor, limit=limit)
156
+
157
+ async def recall(
158
+ self,
159
+ *,
160
+ owner_id: str,
161
+ query: str,
162
+ top_k: int = RECALL_TOP_K,
163
+ char_budget: int = RECALL_CHAR_BUDGET,
164
+ ) -> RecallResult:
165
+ cap = max(1, min(int(top_k), 100))
166
+ budget = max(_HEADER_CHARS, int(char_budget))
167
+ try:
168
+ candidates = await asyncio.wait_for(
169
+ self._store.search_candidates(owner_id=owner_id, query=query, limit=cap),
170
+ timeout=_SEARCH_DEADLINE_SECONDS,
171
+ )
172
+ except TimeoutError:
173
+ page, _ = await self._store.list_active_page(owner_id=owner_id, limit=cap)
174
+ recent = _truncate_to_budget(list(page), budget=budget)
175
+ recent.sort(key=lambda record: (recall_recency(record), record.memory_id))
176
+ return RecallResult(
177
+ records=tuple(recent),
178
+ strategy="recent_fallback",
179
+ degraded=("search_timeout",),
180
+ content_chars=sum(len(record.body) for record in recent),
181
+ )
182
+
183
+ exact_ids: list[str] = []
184
+ fused_ids: dict[str, float] = {}
185
+ rankings: list[list[str]] = []
186
+ for leg in ("exact", "sparse", "dense"):
187
+ ranking: list[str] = []
188
+ for candidate in candidates:
189
+ if candidate.leg != leg or candidate.record.memory_id in ranking:
190
+ continue
191
+ ranking.append(candidate.record.memory_id)
192
+ if leg == "exact":
193
+ exact_ids = ranking
194
+ else:
195
+ rankings.append(ranking)
196
+ fused_ids.update(rrf_fuse(rankings))
197
+
198
+ by_id = {candidate.record.memory_id: candidate.record for candidate in candidates}
199
+ exact_records = [by_id[memory_id] for memory_id in exact_ids if memory_id in by_id]
200
+ ranked_records = [
201
+ by_id[memory_id]
202
+ for memory_id, _ in sorted(
203
+ (
204
+ (memory_id, score)
205
+ for memory_id, score in fused_ids.items()
206
+ if memory_id not in exact_ids
207
+ ),
208
+ key=lambda item: item[1],
209
+ reverse=True,
210
+ )
211
+ if memory_id in by_id
212
+ ]
213
+ ordered = _packing_prior([*exact_records, *ranked_records])[:cap]
214
+ ordered = _truncate_to_budget(ordered, budget=budget)
215
+ exact_set = set(exact_ids)
216
+ exact_block = [record for record in ordered if record.memory_id in exact_set]
217
+ rest = [record for record in ordered if record.memory_id not in exact_set]
218
+ exact_block.sort(key=lambda record: (recall_recency(record), record.memory_id))
219
+ rest.sort(key=lambda record: (recall_recency(record), record.memory_id))
220
+ ordered = [*exact_block, *rest]
221
+ return RecallResult(
222
+ records=tuple(ordered),
223
+ strategy="query_search",
224
+ candidates=tuple(candidates),
225
+ content_chars=sum(len(record.body) for record in ordered),
226
+ )
227
+
228
+ async def purge_superseded(self, *, older_than: datetime) -> int:
229
+ return await self._store.purge_superseded(older_than=older_than)
230
+
231
+
232
+ def _packing_prior(records: list[MemoryRecord]) -> list[MemoryRecord]:
233
+ if not records:
234
+ return records
235
+ first_preference = next((record for record in records if record.kind == "preference"), None)
236
+ first_fact = next((record for record in records if record.kind == "fact"), None)
237
+ if first_preference is None or first_fact is None:
238
+ return records
239
+ kept = [first_preference, first_fact]
240
+ kept.extend(
241
+ record
242
+ for record in records
243
+ if record.memory_id not in {first_preference.memory_id, first_fact.memory_id}
244
+ )
245
+ return kept
246
+
247
+
248
+ def _truncate_to_budget(records: list[MemoryRecord], *, budget: int) -> list[MemoryRecord]:
249
+ kept: list[MemoryRecord] = []
250
+ used = _HEADER_CHARS
251
+ for record in records:
252
+ cost = len(record.body)
253
+ if used + cost > budget and kept:
254
+ break
255
+ kept.append(record)
256
+ used += cost
257
+ return kept
258
+
259
+
260
+ __all__ = ["Memory", "RecallResult"]
@@ -0,0 +1,95 @@
1
+ # Copyright 2025-2026 Hanlian Lu. SPDX-License-Identifier: Apache-2.0
2
+ """Owner-scoped Profile Memory records, operations, and receipts."""
3
+
4
+ from __future__ import annotations
5
+
6
+ from dataclasses import dataclass
7
+ from datetime import datetime
8
+ from typing import Literal
9
+
10
+ MemoryKind = Literal["preference", "fact"]
11
+ MemoryStatus = Literal["active", "superseded", "forgotten"]
12
+ MemoryOriginKind = Literal["answer_run", "management", "mcp", "undo"]
13
+ MemoryOperationAction = Literal["remember", "forget", "undo"]
14
+ MemoryOperationOutcome = Literal["changed", "unchanged", "conflict"]
15
+
16
+
17
+ @dataclass(frozen=True, slots=True)
18
+ class MemoryProvenance:
19
+ """Trusted host-bound source of one Memory operation."""
20
+
21
+ origin_kind: MemoryOriginKind
22
+ origin_id: str
23
+ run_id: str | None = None
24
+ session_id: str | None = None
25
+
26
+
27
+ @dataclass(frozen=True, slots=True)
28
+ class MemoryRecord:
29
+ """One owner-scoped, non-citable remembered preference or fact."""
30
+
31
+ owner_id: str
32
+ memory_id: str
33
+ kind: MemoryKind
34
+ body: str
35
+ provenance: MemoryProvenance
36
+ status: MemoryStatus = "active"
37
+ supersedes_id: str | None = None
38
+ created_at: datetime | None = None
39
+ updated_at: datetime | None = None
40
+
41
+
42
+ @dataclass(frozen=True, slots=True)
43
+ class MemoryOperation:
44
+ """One canonical, idempotent mutation request at the storage seam."""
45
+
46
+ owner_id: str
47
+ idempotency_key: str
48
+ action: MemoryOperationAction
49
+ provenance: MemoryProvenance
50
+ kind: MemoryKind | None = None
51
+ body: str = ""
52
+ memory_id: str | None = None
53
+ supersedes_id: str | None = None
54
+ target_change_id: str | None = None
55
+ mutation_scope: str | None = None
56
+ mutation_limit: int | None = None
57
+
58
+
59
+ @dataclass(frozen=True, slots=True)
60
+ class MemoryOperationReceipt:
61
+ """Replay-stable result of one settled Memory operation."""
62
+
63
+ change_id: str
64
+ action: MemoryOperationAction
65
+ outcome: MemoryOperationOutcome
66
+ memory_ids: tuple[str, ...]
67
+ provenance: MemoryProvenance
68
+ kind: MemoryKind | None = None
69
+ body: str = ""
70
+ supersedes_id: str | None = None
71
+ target_change_id: str | None = None
72
+ mutation_scope: str | None = None
73
+ created_at: datetime | None = None
74
+
75
+ @property
76
+ def memory_id(self) -> str | None:
77
+ """The primary affected record, when the operation has one."""
78
+ return self.memory_ids[0] if self.memory_ids else None
79
+
80
+ @property
81
+ def changed(self) -> bool:
82
+ return self.outcome == "changed"
83
+
84
+
85
+ __all__ = [
86
+ "MemoryKind",
87
+ "MemoryOperation",
88
+ "MemoryOperationAction",
89
+ "MemoryOperationOutcome",
90
+ "MemoryOperationReceipt",
91
+ "MemoryOriginKind",
92
+ "MemoryProvenance",
93
+ "MemoryRecord",
94
+ "MemoryStatus",
95
+ ]
@@ -0,0 +1,24 @@
1
+ # Copyright 2025-2026 Hanlian Lu. SPDX-License-Identifier: Apache-2.0
2
+ """Canonical body normalization for the exact-match recall leg.
3
+
4
+ One pure function shared by every storage adapter so equality semantics never
5
+ drift between backends: NFKC (fullwidth/halfwidth and compatibility forms),
6
+ casefold, and internal-whitespace collapse. Language-agnostic — CJK fullwidth
7
+ and Latin case both normalize here.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import unicodedata
13
+
14
+ _WHITESPACE = " \t\r\n"
15
+
16
+
17
+ def normalized_body(text: str) -> str:
18
+ """Return one canonical comparison key for a memory body."""
19
+ folded = unicodedata.normalize("NFKC", text).casefold()
20
+ stripped = "".join(" " if char in _WHITESPACE else char for char in folded)
21
+ return " ".join(stripped.split())
22
+
23
+
24
+ __all__ = ["normalized_body"]
@@ -0,0 +1,82 @@
1
+ # Copyright 2025-2026 Hanlian Lu. SPDX-License-Identifier: Apache-2.0
2
+ """Closed Profile Memory operation checklist and fixed safety bounds."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import re
7
+
8
+ from dlightrag_memory.errors import MemoryWriteRejectedError
9
+ from dlightrag_memory.models import MemoryOperation
10
+
11
+ MEMORY_BODY_LIMIT = 500
12
+ RECALL_TOP_K = 10
13
+ RECALL_CHAR_BUDGET = 4000
14
+ MEMORY_SUPERSEDE_RETENTION_DAYS = 365
15
+
16
+ _CITATION_MARK = re.compile(r"\[\d+(?:-\d+)?\]")
17
+ _PRIVATE_KEY_MARK = re.compile(r"-----BEGIN (?:[A-Z0-9 ]+ )?PRIVATE KEY-----", re.IGNORECASE)
18
+ _TOKEN_MARK = re.compile(
19
+ r"(?:\bAKIA[0-9A-Z]{16}\b|\bgh[opusr]_[A-Za-z0-9_]{20,}\b|"
20
+ r"\bgithub_pat_[A-Za-z0-9_]{20,}\b|\bxox[baprs]-[A-Za-z0-9-]{20,}\b|"
21
+ r"\bsk-[A-Za-z0-9_-]{20,}\b)"
22
+ )
23
+
24
+
25
+ def evaluate_memory_operation(operation: MemoryOperation) -> None:
26
+ """Validate one operation before any storage mutation."""
27
+ if not operation.owner_id.strip():
28
+ raise MemoryWriteRejectedError("A Memory operation needs an owner.")
29
+ if not operation.idempotency_key.strip():
30
+ raise MemoryWriteRejectedError("A Memory operation needs an idempotency key.")
31
+ if len(operation.idempotency_key) > 255:
32
+ raise MemoryWriteRejectedError("Memory idempotency keys cannot exceed 255 characters.")
33
+ if not operation.provenance.origin_id.strip():
34
+ raise MemoryWriteRejectedError("A Memory operation needs trusted provenance.")
35
+ if operation.mutation_limit is not None and operation.mutation_limit < 1:
36
+ raise MemoryWriteRejectedError("A Memory mutation limit must be positive.")
37
+ if (operation.mutation_scope is None) != (operation.mutation_limit is None):
38
+ raise MemoryWriteRejectedError("Memory mutation scope and limit must be provided together.")
39
+
40
+ body = operation.body.strip()
41
+ if operation.action == "remember":
42
+ if operation.kind not in {"preference", "fact"}:
43
+ raise MemoryWriteRejectedError("Memory kind must be preference or fact.")
44
+ if not body:
45
+ raise MemoryWriteRejectedError("Memory body cannot be empty.")
46
+ if len(body) > MEMORY_BODY_LIMIT:
47
+ raise MemoryWriteRejectedError(
48
+ f"Memory body cannot exceed {MEMORY_BODY_LIMIT} characters."
49
+ )
50
+ if operation.memory_id is not None or operation.target_change_id is not None:
51
+ raise MemoryWriteRejectedError("Remember received an incompatible target.")
52
+ if _CITATION_MARK.search(body):
53
+ raise MemoryWriteRejectedError("Memory body cannot carry citation markers.")
54
+ if _PRIVATE_KEY_MARK.search(body) or _TOKEN_MARK.search(body):
55
+ raise MemoryWriteRejectedError("Credentials and private keys cannot be remembered.")
56
+ return
57
+
58
+ if operation.action == "forget":
59
+ selectors = sum(bool(value and value.strip()) for value in (operation.memory_id, body))
60
+ if selectors != 1:
61
+ raise MemoryWriteRejectedError("Forget needs exactly one memory id or exact body.")
62
+ if any((operation.kind, operation.supersedes_id, operation.target_change_id)):
63
+ raise MemoryWriteRejectedError("Forget received an incompatible target.")
64
+ return
65
+
66
+ if operation.action == "undo":
67
+ if not (operation.target_change_id or "").strip():
68
+ raise MemoryWriteRejectedError("Undo needs a change id.")
69
+ if any((operation.kind, body, operation.memory_id, operation.supersedes_id)):
70
+ raise MemoryWriteRejectedError("Undo received an incompatible target.")
71
+ return
72
+
73
+ raise MemoryWriteRejectedError("Unknown Memory operation.")
74
+
75
+
76
+ __all__ = [
77
+ "MEMORY_BODY_LIMIT",
78
+ "MEMORY_SUPERSEDE_RETENTION_DAYS",
79
+ "RECALL_CHAR_BUDGET",
80
+ "RECALL_TOP_K",
81
+ "evaluate_memory_operation",
82
+ ]
@@ -0,0 +1,79 @@
1
+ # Copyright 2025-2026 Hanlian Lu. SPDX-License-Identifier: Apache-2.0
2
+ """Storage-neutral ports: text embedding and recall candidate shapes.
3
+
4
+ These ports are the P4 substrate. ``TextEmbedder`` keeps dense recall optional
5
+ and backend-independent; ``NullEmbedder`` is the zero-configuration default
6
+ for standalone hosts (sparse + exact legs only). ``SearchCandidate`` is the
7
+ leg-tagged candidate a storage adapter returns in per-leg rank order.
8
+ PostgreSQL-specific connection and migration shapes live in ``_storage.pg``,
9
+ not here.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from collections.abc import Sequence
15
+ from typing import Literal, Protocol
16
+
17
+ from dlightrag_memory.models import MemoryRecord
18
+
19
+ type Vector = Sequence[float]
20
+
21
+
22
+ class TextEmbedder(Protocol):
23
+ """Produce one embedding space for memory bodies.
24
+
25
+ ``embedding_fingerprint`` identifies the embedding model; an adapter
26
+ stores it with every vector so a model change invalidates the dense index
27
+ instead of silently comparing across spaces.
28
+ """
29
+
30
+ @property
31
+ def embedding_fingerprint(self) -> str: ...
32
+
33
+ dim: int
34
+
35
+ async def embed_documents(self, texts: Sequence[str]) -> Sequence[Vector]: ...
36
+
37
+ async def embed_query(self, text: str) -> Vector: ...
38
+
39
+
40
+ class NullEmbedder:
41
+ """The zero-configuration embedder: dense recall stays off."""
42
+
43
+ dim = 0
44
+
45
+ @property
46
+ def embedding_fingerprint(self) -> str:
47
+ return "none"
48
+
49
+ async def aclose(self) -> None:
50
+ return None
51
+
52
+ async def embed_documents(self, texts: Sequence[str]) -> Sequence[Vector]:
53
+ raise RuntimeError("NullEmbedder produces no vectors; disable the dense leg")
54
+
55
+ async def embed_query(self, text: str) -> Vector:
56
+ raise RuntimeError("NullEmbedder produces no vectors; disable the dense leg")
57
+
58
+
59
+ SearchLeg = Literal["dense", "sparse", "exact"]
60
+
61
+
62
+ class SearchCandidate:
63
+ """One recalled record with its source leg and a comparable score."""
64
+
65
+ __slots__ = ("record", "leg", "score")
66
+
67
+ def __init__(self, *, record: MemoryRecord, leg: SearchLeg, score: float) -> None:
68
+ self.record = record
69
+ self.leg = leg
70
+ self.score = score
71
+
72
+
73
+ __all__ = [
74
+ "NullEmbedder",
75
+ "SearchCandidate",
76
+ "SearchLeg",
77
+ "TextEmbedder",
78
+ "Vector",
79
+ ]
@@ -0,0 +1,13 @@
1
+ # Copyright 2025-2026 Hanlian Lu. SPDX-License-Identifier: Apache-2.0
2
+ """PostgreSQL storage entry point for the Memory package.
3
+
4
+ Kept out of the package ``__init__`` so importing the core never drags the
5
+ database adapter into an embedding host's import graph; hosts that use
6
+ PostgreSQL import it directly:
7
+
8
+ from dlightrag_memory.postgres import PostgresMemoryStore
9
+ """
10
+
11
+ from dlightrag_memory._storage.pg import PostgresMemoryStore
12
+
13
+ __all__ = ["PostgresMemoryStore"]
File without changes
@@ -0,0 +1,25 @@
1
+ # Copyright 2025-2026 Hanlian Lu. SPDX-License-Identifier: Apache-2.0
2
+ """Recency ordering for record tie-breaks.
3
+
4
+ Query relevance decides the candidate set (fusion in ``fusion.py``); time is
5
+ never a score component — it only orders presentation and breaks ties, the
6
+ same design MemMachine uses.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from datetime import UTC, datetime
12
+
13
+ from dlightrag_memory.models import MemoryRecord
14
+
15
+
16
+ def recall_recency(record: MemoryRecord) -> datetime:
17
+ """One record's standing-ordering timestamp."""
18
+ if record.updated_at is not None:
19
+ return record.updated_at
20
+ if record.created_at is not None:
21
+ return record.created_at
22
+ return datetime.min.replace(tzinfo=UTC)
23
+
24
+
25
+ __all__ = ["recall_recency"]