dlightrag-memory 2.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dlightrag_memory/__init__.py +37 -0
- dlightrag_memory/_storage/__init__.py +0 -0
- dlightrag_memory/_storage/pg.py +1311 -0
- dlightrag_memory/_storage/pg_bm25.py +186 -0
- dlightrag_memory/errors.py +25 -0
- dlightrag_memory/fusion.py +28 -0
- dlightrag_memory/mcp_server.py +241 -0
- dlightrag_memory/memory.py +260 -0
- dlightrag_memory/models.py +95 -0
- dlightrag_memory/normalize.py +24 -0
- dlightrag_memory/policy.py +82 -0
- dlightrag_memory/ports.py +79 -0
- dlightrag_memory/postgres.py +13 -0
- dlightrag_memory/py.typed +0 -0
- dlightrag_memory/recall.py +25 -0
- dlightrag_memory/store.py +676 -0
- dlightrag_memory-2.0.0.dist-info/METADATA +12 -0
- dlightrag_memory-2.0.0.dist-info/RECORD +22 -0
- dlightrag_memory-2.0.0.dist-info/WHEEL +4 -0
- dlightrag_memory-2.0.0.dist-info/entry_points.txt +2 -0
- dlightrag_memory-2.0.0.dist-info/licenses/LICENSE +199 -0
- dlightrag_memory-2.0.0.dist-info/licenses/NOTICE +14 -0
|
@@ -0,0 +1,260 @@
|
|
|
1
|
+
# Copyright 2025-2026 Hanlian Lu. SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""The host-neutral Profile Memory façade."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import asyncio
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from datetime import datetime
|
|
9
|
+
|
|
10
|
+
from dlightrag_memory.fusion import rrf_fuse
|
|
11
|
+
from dlightrag_memory.models import (
|
|
12
|
+
MemoryKind,
|
|
13
|
+
MemoryOperation,
|
|
14
|
+
MemoryOperationReceipt,
|
|
15
|
+
MemoryProvenance,
|
|
16
|
+
MemoryRecord,
|
|
17
|
+
)
|
|
18
|
+
from dlightrag_memory.policy import (
|
|
19
|
+
RECALL_CHAR_BUDGET,
|
|
20
|
+
RECALL_TOP_K,
|
|
21
|
+
evaluate_memory_operation,
|
|
22
|
+
)
|
|
23
|
+
from dlightrag_memory.ports import SearchCandidate
|
|
24
|
+
from dlightrag_memory.recall import recall_recency
|
|
25
|
+
from dlightrag_memory.store import MemoryStore, OperationGuard
|
|
26
|
+
|
|
27
|
+
_SEARCH_DEADLINE_SECONDS = 2.0
|
|
28
|
+
_HEADER_CHARS = 160
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass(frozen=True, slots=True)
|
|
32
|
+
class RecallResult:
|
|
33
|
+
"""Structured query-aware recall result."""
|
|
34
|
+
|
|
35
|
+
records: tuple[MemoryRecord, ...]
|
|
36
|
+
strategy: str
|
|
37
|
+
candidates: tuple[SearchCandidate, ...] = ()
|
|
38
|
+
degraded: tuple[str, ...] = ()
|
|
39
|
+
content_chars: int = 0
|
|
40
|
+
|
|
41
|
+
@property
|
|
42
|
+
def skipped(self) -> bool:
|
|
43
|
+
return bool(self.degraded)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class Memory:
|
|
47
|
+
"""Cross-conversation owner Profile Memory behind one deep mutation seam."""
|
|
48
|
+
|
|
49
|
+
def __init__(self, store: MemoryStore) -> None:
|
|
50
|
+
self._store = store
|
|
51
|
+
|
|
52
|
+
async def apply(
|
|
53
|
+
self,
|
|
54
|
+
operation: MemoryOperation,
|
|
55
|
+
*,
|
|
56
|
+
guard: OperationGuard | None = None,
|
|
57
|
+
) -> MemoryOperationReceipt:
|
|
58
|
+
"""Validate and atomically settle one idempotent operation."""
|
|
59
|
+
evaluate_memory_operation(operation)
|
|
60
|
+
return await self._store.apply_operation(operation, guard=guard)
|
|
61
|
+
|
|
62
|
+
async def remember(
|
|
63
|
+
self,
|
|
64
|
+
*,
|
|
65
|
+
owner_id: str,
|
|
66
|
+
kind: MemoryKind,
|
|
67
|
+
body: str,
|
|
68
|
+
provenance: MemoryProvenance,
|
|
69
|
+
idempotency_key: str,
|
|
70
|
+
supersedes_id: str | None = None,
|
|
71
|
+
mutation_scope: str | None = None,
|
|
72
|
+
mutation_limit: int | None = None,
|
|
73
|
+
guard: OperationGuard | None = None,
|
|
74
|
+
) -> MemoryOperationReceipt:
|
|
75
|
+
return await self.apply(
|
|
76
|
+
MemoryOperation(
|
|
77
|
+
owner_id=owner_id,
|
|
78
|
+
idempotency_key=idempotency_key,
|
|
79
|
+
action="remember",
|
|
80
|
+
provenance=provenance,
|
|
81
|
+
kind=kind,
|
|
82
|
+
body=body,
|
|
83
|
+
supersedes_id=supersedes_id,
|
|
84
|
+
mutation_scope=mutation_scope,
|
|
85
|
+
mutation_limit=mutation_limit,
|
|
86
|
+
),
|
|
87
|
+
guard=guard,
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
async def forget(
|
|
91
|
+
self,
|
|
92
|
+
*,
|
|
93
|
+
owner_id: str,
|
|
94
|
+
provenance: MemoryProvenance,
|
|
95
|
+
idempotency_key: str,
|
|
96
|
+
memory_id: str | None = None,
|
|
97
|
+
body: str | None = None,
|
|
98
|
+
mutation_scope: str | None = None,
|
|
99
|
+
mutation_limit: int | None = None,
|
|
100
|
+
guard: OperationGuard | None = None,
|
|
101
|
+
) -> MemoryOperationReceipt:
|
|
102
|
+
return await self.apply(
|
|
103
|
+
MemoryOperation(
|
|
104
|
+
owner_id=owner_id,
|
|
105
|
+
idempotency_key=idempotency_key,
|
|
106
|
+
action="forget",
|
|
107
|
+
provenance=provenance,
|
|
108
|
+
memory_id=memory_id,
|
|
109
|
+
body=body or "",
|
|
110
|
+
mutation_scope=mutation_scope,
|
|
111
|
+
mutation_limit=mutation_limit,
|
|
112
|
+
),
|
|
113
|
+
guard=guard,
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
async def undo(
|
|
117
|
+
self,
|
|
118
|
+
*,
|
|
119
|
+
owner_id: str,
|
|
120
|
+
change_id: str,
|
|
121
|
+
provenance: MemoryProvenance,
|
|
122
|
+
idempotency_key: str,
|
|
123
|
+
guard: OperationGuard | None = None,
|
|
124
|
+
) -> MemoryOperationReceipt:
|
|
125
|
+
return await self.apply(
|
|
126
|
+
MemoryOperation(
|
|
127
|
+
owner_id=owner_id,
|
|
128
|
+
idempotency_key=idempotency_key,
|
|
129
|
+
action="undo",
|
|
130
|
+
provenance=provenance,
|
|
131
|
+
target_change_id=change_id,
|
|
132
|
+
),
|
|
133
|
+
guard=guard,
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
async def clear(
|
|
137
|
+
self,
|
|
138
|
+
*,
|
|
139
|
+
owner_id: str,
|
|
140
|
+
guard: OperationGuard | None = None,
|
|
141
|
+
) -> int:
|
|
142
|
+
"""Physically erase one owner's complete Profile Memory schema state."""
|
|
143
|
+
return await self._store.clear_owner(owner_id=owner_id, guard=guard)
|
|
144
|
+
|
|
145
|
+
async def count_active(self, *, owner_id: str) -> int:
|
|
146
|
+
return await self._store.count_active(owner_id=owner_id)
|
|
147
|
+
|
|
148
|
+
async def browse(
|
|
149
|
+
self,
|
|
150
|
+
*,
|
|
151
|
+
owner_id: str,
|
|
152
|
+
cursor: tuple[datetime, str] | None = None,
|
|
153
|
+
limit: int = 50,
|
|
154
|
+
) -> tuple[tuple[MemoryRecord, ...], tuple[datetime, str] | None]:
|
|
155
|
+
return await self._store.list_active_page(owner_id=owner_id, after=cursor, limit=limit)
|
|
156
|
+
|
|
157
|
+
async def recall(
|
|
158
|
+
self,
|
|
159
|
+
*,
|
|
160
|
+
owner_id: str,
|
|
161
|
+
query: str,
|
|
162
|
+
top_k: int = RECALL_TOP_K,
|
|
163
|
+
char_budget: int = RECALL_CHAR_BUDGET,
|
|
164
|
+
) -> RecallResult:
|
|
165
|
+
cap = max(1, min(int(top_k), 100))
|
|
166
|
+
budget = max(_HEADER_CHARS, int(char_budget))
|
|
167
|
+
try:
|
|
168
|
+
candidates = await asyncio.wait_for(
|
|
169
|
+
self._store.search_candidates(owner_id=owner_id, query=query, limit=cap),
|
|
170
|
+
timeout=_SEARCH_DEADLINE_SECONDS,
|
|
171
|
+
)
|
|
172
|
+
except TimeoutError:
|
|
173
|
+
page, _ = await self._store.list_active_page(owner_id=owner_id, limit=cap)
|
|
174
|
+
recent = _truncate_to_budget(list(page), budget=budget)
|
|
175
|
+
recent.sort(key=lambda record: (recall_recency(record), record.memory_id))
|
|
176
|
+
return RecallResult(
|
|
177
|
+
records=tuple(recent),
|
|
178
|
+
strategy="recent_fallback",
|
|
179
|
+
degraded=("search_timeout",),
|
|
180
|
+
content_chars=sum(len(record.body) for record in recent),
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
exact_ids: list[str] = []
|
|
184
|
+
fused_ids: dict[str, float] = {}
|
|
185
|
+
rankings: list[list[str]] = []
|
|
186
|
+
for leg in ("exact", "sparse", "dense"):
|
|
187
|
+
ranking: list[str] = []
|
|
188
|
+
for candidate in candidates:
|
|
189
|
+
if candidate.leg != leg or candidate.record.memory_id in ranking:
|
|
190
|
+
continue
|
|
191
|
+
ranking.append(candidate.record.memory_id)
|
|
192
|
+
if leg == "exact":
|
|
193
|
+
exact_ids = ranking
|
|
194
|
+
else:
|
|
195
|
+
rankings.append(ranking)
|
|
196
|
+
fused_ids.update(rrf_fuse(rankings))
|
|
197
|
+
|
|
198
|
+
by_id = {candidate.record.memory_id: candidate.record for candidate in candidates}
|
|
199
|
+
exact_records = [by_id[memory_id] for memory_id in exact_ids if memory_id in by_id]
|
|
200
|
+
ranked_records = [
|
|
201
|
+
by_id[memory_id]
|
|
202
|
+
for memory_id, _ in sorted(
|
|
203
|
+
(
|
|
204
|
+
(memory_id, score)
|
|
205
|
+
for memory_id, score in fused_ids.items()
|
|
206
|
+
if memory_id not in exact_ids
|
|
207
|
+
),
|
|
208
|
+
key=lambda item: item[1],
|
|
209
|
+
reverse=True,
|
|
210
|
+
)
|
|
211
|
+
if memory_id in by_id
|
|
212
|
+
]
|
|
213
|
+
ordered = _packing_prior([*exact_records, *ranked_records])[:cap]
|
|
214
|
+
ordered = _truncate_to_budget(ordered, budget=budget)
|
|
215
|
+
exact_set = set(exact_ids)
|
|
216
|
+
exact_block = [record for record in ordered if record.memory_id in exact_set]
|
|
217
|
+
rest = [record for record in ordered if record.memory_id not in exact_set]
|
|
218
|
+
exact_block.sort(key=lambda record: (recall_recency(record), record.memory_id))
|
|
219
|
+
rest.sort(key=lambda record: (recall_recency(record), record.memory_id))
|
|
220
|
+
ordered = [*exact_block, *rest]
|
|
221
|
+
return RecallResult(
|
|
222
|
+
records=tuple(ordered),
|
|
223
|
+
strategy="query_search",
|
|
224
|
+
candidates=tuple(candidates),
|
|
225
|
+
content_chars=sum(len(record.body) for record in ordered),
|
|
226
|
+
)
|
|
227
|
+
|
|
228
|
+
async def purge_superseded(self, *, older_than: datetime) -> int:
|
|
229
|
+
return await self._store.purge_superseded(older_than=older_than)
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _packing_prior(records: list[MemoryRecord]) -> list[MemoryRecord]:
|
|
233
|
+
if not records:
|
|
234
|
+
return records
|
|
235
|
+
first_preference = next((record for record in records if record.kind == "preference"), None)
|
|
236
|
+
first_fact = next((record for record in records if record.kind == "fact"), None)
|
|
237
|
+
if first_preference is None or first_fact is None:
|
|
238
|
+
return records
|
|
239
|
+
kept = [first_preference, first_fact]
|
|
240
|
+
kept.extend(
|
|
241
|
+
record
|
|
242
|
+
for record in records
|
|
243
|
+
if record.memory_id not in {first_preference.memory_id, first_fact.memory_id}
|
|
244
|
+
)
|
|
245
|
+
return kept
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _truncate_to_budget(records: list[MemoryRecord], *, budget: int) -> list[MemoryRecord]:
|
|
249
|
+
kept: list[MemoryRecord] = []
|
|
250
|
+
used = _HEADER_CHARS
|
|
251
|
+
for record in records:
|
|
252
|
+
cost = len(record.body)
|
|
253
|
+
if used + cost > budget and kept:
|
|
254
|
+
break
|
|
255
|
+
kept.append(record)
|
|
256
|
+
used += cost
|
|
257
|
+
return kept
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
__all__ = ["Memory", "RecallResult"]
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
# Copyright 2025-2026 Hanlian Lu. SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""Owner-scoped Profile Memory records, operations, and receipts."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from datetime import datetime
|
|
8
|
+
from typing import Literal
|
|
9
|
+
|
|
10
|
+
MemoryKind = Literal["preference", "fact"]
|
|
11
|
+
MemoryStatus = Literal["active", "superseded", "forgotten"]
|
|
12
|
+
MemoryOriginKind = Literal["answer_run", "management", "mcp", "undo"]
|
|
13
|
+
MemoryOperationAction = Literal["remember", "forget", "undo"]
|
|
14
|
+
MemoryOperationOutcome = Literal["changed", "unchanged", "conflict"]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(frozen=True, slots=True)
|
|
18
|
+
class MemoryProvenance:
|
|
19
|
+
"""Trusted host-bound source of one Memory operation."""
|
|
20
|
+
|
|
21
|
+
origin_kind: MemoryOriginKind
|
|
22
|
+
origin_id: str
|
|
23
|
+
run_id: str | None = None
|
|
24
|
+
session_id: str | None = None
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass(frozen=True, slots=True)
|
|
28
|
+
class MemoryRecord:
|
|
29
|
+
"""One owner-scoped, non-citable remembered preference or fact."""
|
|
30
|
+
|
|
31
|
+
owner_id: str
|
|
32
|
+
memory_id: str
|
|
33
|
+
kind: MemoryKind
|
|
34
|
+
body: str
|
|
35
|
+
provenance: MemoryProvenance
|
|
36
|
+
status: MemoryStatus = "active"
|
|
37
|
+
supersedes_id: str | None = None
|
|
38
|
+
created_at: datetime | None = None
|
|
39
|
+
updated_at: datetime | None = None
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass(frozen=True, slots=True)
|
|
43
|
+
class MemoryOperation:
|
|
44
|
+
"""One canonical, idempotent mutation request at the storage seam."""
|
|
45
|
+
|
|
46
|
+
owner_id: str
|
|
47
|
+
idempotency_key: str
|
|
48
|
+
action: MemoryOperationAction
|
|
49
|
+
provenance: MemoryProvenance
|
|
50
|
+
kind: MemoryKind | None = None
|
|
51
|
+
body: str = ""
|
|
52
|
+
memory_id: str | None = None
|
|
53
|
+
supersedes_id: str | None = None
|
|
54
|
+
target_change_id: str | None = None
|
|
55
|
+
mutation_scope: str | None = None
|
|
56
|
+
mutation_limit: int | None = None
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass(frozen=True, slots=True)
|
|
60
|
+
class MemoryOperationReceipt:
|
|
61
|
+
"""Replay-stable result of one settled Memory operation."""
|
|
62
|
+
|
|
63
|
+
change_id: str
|
|
64
|
+
action: MemoryOperationAction
|
|
65
|
+
outcome: MemoryOperationOutcome
|
|
66
|
+
memory_ids: tuple[str, ...]
|
|
67
|
+
provenance: MemoryProvenance
|
|
68
|
+
kind: MemoryKind | None = None
|
|
69
|
+
body: str = ""
|
|
70
|
+
supersedes_id: str | None = None
|
|
71
|
+
target_change_id: str | None = None
|
|
72
|
+
mutation_scope: str | None = None
|
|
73
|
+
created_at: datetime | None = None
|
|
74
|
+
|
|
75
|
+
@property
|
|
76
|
+
def memory_id(self) -> str | None:
|
|
77
|
+
"""The primary affected record, when the operation has one."""
|
|
78
|
+
return self.memory_ids[0] if self.memory_ids else None
|
|
79
|
+
|
|
80
|
+
@property
|
|
81
|
+
def changed(self) -> bool:
|
|
82
|
+
return self.outcome == "changed"
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
__all__ = [
|
|
86
|
+
"MemoryKind",
|
|
87
|
+
"MemoryOperation",
|
|
88
|
+
"MemoryOperationAction",
|
|
89
|
+
"MemoryOperationOutcome",
|
|
90
|
+
"MemoryOperationReceipt",
|
|
91
|
+
"MemoryOriginKind",
|
|
92
|
+
"MemoryProvenance",
|
|
93
|
+
"MemoryRecord",
|
|
94
|
+
"MemoryStatus",
|
|
95
|
+
]
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# Copyright 2025-2026 Hanlian Lu. SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""Canonical body normalization for the exact-match recall leg.
|
|
3
|
+
|
|
4
|
+
One pure function shared by every storage adapter so equality semantics never
|
|
5
|
+
drift between backends: NFKC (fullwidth/halfwidth and compatibility forms),
|
|
6
|
+
casefold, and internal-whitespace collapse. Language-agnostic — CJK fullwidth
|
|
7
|
+
and Latin case both normalize here.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import unicodedata
|
|
13
|
+
|
|
14
|
+
_WHITESPACE = " \t\r\n"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def normalized_body(text: str) -> str:
|
|
18
|
+
"""Return one canonical comparison key for a memory body."""
|
|
19
|
+
folded = unicodedata.normalize("NFKC", text).casefold()
|
|
20
|
+
stripped = "".join(" " if char in _WHITESPACE else char for char in folded)
|
|
21
|
+
return " ".join(stripped.split())
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
__all__ = ["normalized_body"]
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
# Copyright 2025-2026 Hanlian Lu. SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""Closed Profile Memory operation checklist and fixed safety bounds."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import re
|
|
7
|
+
|
|
8
|
+
from dlightrag_memory.errors import MemoryWriteRejectedError
|
|
9
|
+
from dlightrag_memory.models import MemoryOperation
|
|
10
|
+
|
|
11
|
+
MEMORY_BODY_LIMIT = 500
|
|
12
|
+
RECALL_TOP_K = 10
|
|
13
|
+
RECALL_CHAR_BUDGET = 4000
|
|
14
|
+
MEMORY_SUPERSEDE_RETENTION_DAYS = 365
|
|
15
|
+
|
|
16
|
+
_CITATION_MARK = re.compile(r"\[\d+(?:-\d+)?\]")
|
|
17
|
+
_PRIVATE_KEY_MARK = re.compile(r"-----BEGIN (?:[A-Z0-9 ]+ )?PRIVATE KEY-----", re.IGNORECASE)
|
|
18
|
+
_TOKEN_MARK = re.compile(
|
|
19
|
+
r"(?:\bAKIA[0-9A-Z]{16}\b|\bgh[opusr]_[A-Za-z0-9_]{20,}\b|"
|
|
20
|
+
r"\bgithub_pat_[A-Za-z0-9_]{20,}\b|\bxox[baprs]-[A-Za-z0-9-]{20,}\b|"
|
|
21
|
+
r"\bsk-[A-Za-z0-9_-]{20,}\b)"
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def evaluate_memory_operation(operation: MemoryOperation) -> None:
|
|
26
|
+
"""Validate one operation before any storage mutation."""
|
|
27
|
+
if not operation.owner_id.strip():
|
|
28
|
+
raise MemoryWriteRejectedError("A Memory operation needs an owner.")
|
|
29
|
+
if not operation.idempotency_key.strip():
|
|
30
|
+
raise MemoryWriteRejectedError("A Memory operation needs an idempotency key.")
|
|
31
|
+
if len(operation.idempotency_key) > 255:
|
|
32
|
+
raise MemoryWriteRejectedError("Memory idempotency keys cannot exceed 255 characters.")
|
|
33
|
+
if not operation.provenance.origin_id.strip():
|
|
34
|
+
raise MemoryWriteRejectedError("A Memory operation needs trusted provenance.")
|
|
35
|
+
if operation.mutation_limit is not None and operation.mutation_limit < 1:
|
|
36
|
+
raise MemoryWriteRejectedError("A Memory mutation limit must be positive.")
|
|
37
|
+
if (operation.mutation_scope is None) != (operation.mutation_limit is None):
|
|
38
|
+
raise MemoryWriteRejectedError("Memory mutation scope and limit must be provided together.")
|
|
39
|
+
|
|
40
|
+
body = operation.body.strip()
|
|
41
|
+
if operation.action == "remember":
|
|
42
|
+
if operation.kind not in {"preference", "fact"}:
|
|
43
|
+
raise MemoryWriteRejectedError("Memory kind must be preference or fact.")
|
|
44
|
+
if not body:
|
|
45
|
+
raise MemoryWriteRejectedError("Memory body cannot be empty.")
|
|
46
|
+
if len(body) > MEMORY_BODY_LIMIT:
|
|
47
|
+
raise MemoryWriteRejectedError(
|
|
48
|
+
f"Memory body cannot exceed {MEMORY_BODY_LIMIT} characters."
|
|
49
|
+
)
|
|
50
|
+
if operation.memory_id is not None or operation.target_change_id is not None:
|
|
51
|
+
raise MemoryWriteRejectedError("Remember received an incompatible target.")
|
|
52
|
+
if _CITATION_MARK.search(body):
|
|
53
|
+
raise MemoryWriteRejectedError("Memory body cannot carry citation markers.")
|
|
54
|
+
if _PRIVATE_KEY_MARK.search(body) or _TOKEN_MARK.search(body):
|
|
55
|
+
raise MemoryWriteRejectedError("Credentials and private keys cannot be remembered.")
|
|
56
|
+
return
|
|
57
|
+
|
|
58
|
+
if operation.action == "forget":
|
|
59
|
+
selectors = sum(bool(value and value.strip()) for value in (operation.memory_id, body))
|
|
60
|
+
if selectors != 1:
|
|
61
|
+
raise MemoryWriteRejectedError("Forget needs exactly one memory id or exact body.")
|
|
62
|
+
if any((operation.kind, operation.supersedes_id, operation.target_change_id)):
|
|
63
|
+
raise MemoryWriteRejectedError("Forget received an incompatible target.")
|
|
64
|
+
return
|
|
65
|
+
|
|
66
|
+
if operation.action == "undo":
|
|
67
|
+
if not (operation.target_change_id or "").strip():
|
|
68
|
+
raise MemoryWriteRejectedError("Undo needs a change id.")
|
|
69
|
+
if any((operation.kind, body, operation.memory_id, operation.supersedes_id)):
|
|
70
|
+
raise MemoryWriteRejectedError("Undo received an incompatible target.")
|
|
71
|
+
return
|
|
72
|
+
|
|
73
|
+
raise MemoryWriteRejectedError("Unknown Memory operation.")
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
__all__ = [
|
|
77
|
+
"MEMORY_BODY_LIMIT",
|
|
78
|
+
"MEMORY_SUPERSEDE_RETENTION_DAYS",
|
|
79
|
+
"RECALL_CHAR_BUDGET",
|
|
80
|
+
"RECALL_TOP_K",
|
|
81
|
+
"evaluate_memory_operation",
|
|
82
|
+
]
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# Copyright 2025-2026 Hanlian Lu. SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""Storage-neutral ports: text embedding and recall candidate shapes.
|
|
3
|
+
|
|
4
|
+
These ports are the P4 substrate. ``TextEmbedder`` keeps dense recall optional
|
|
5
|
+
and backend-independent; ``NullEmbedder`` is the zero-configuration default
|
|
6
|
+
for standalone hosts (sparse + exact legs only). ``SearchCandidate`` is the
|
|
7
|
+
leg-tagged candidate a storage adapter returns in per-leg rank order.
|
|
8
|
+
PostgreSQL-specific connection and migration shapes live in ``_storage.pg``,
|
|
9
|
+
not here.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from collections.abc import Sequence
|
|
15
|
+
from typing import Literal, Protocol
|
|
16
|
+
|
|
17
|
+
from dlightrag_memory.models import MemoryRecord
|
|
18
|
+
|
|
19
|
+
type Vector = Sequence[float]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class TextEmbedder(Protocol):
|
|
23
|
+
"""Produce one embedding space for memory bodies.
|
|
24
|
+
|
|
25
|
+
``embedding_fingerprint`` identifies the embedding model; an adapter
|
|
26
|
+
stores it with every vector so a model change invalidates the dense index
|
|
27
|
+
instead of silently comparing across spaces.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
@property
|
|
31
|
+
def embedding_fingerprint(self) -> str: ...
|
|
32
|
+
|
|
33
|
+
dim: int
|
|
34
|
+
|
|
35
|
+
async def embed_documents(self, texts: Sequence[str]) -> Sequence[Vector]: ...
|
|
36
|
+
|
|
37
|
+
async def embed_query(self, text: str) -> Vector: ...
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class NullEmbedder:
|
|
41
|
+
"""The zero-configuration embedder: dense recall stays off."""
|
|
42
|
+
|
|
43
|
+
dim = 0
|
|
44
|
+
|
|
45
|
+
@property
|
|
46
|
+
def embedding_fingerprint(self) -> str:
|
|
47
|
+
return "none"
|
|
48
|
+
|
|
49
|
+
async def aclose(self) -> None:
|
|
50
|
+
return None
|
|
51
|
+
|
|
52
|
+
async def embed_documents(self, texts: Sequence[str]) -> Sequence[Vector]:
|
|
53
|
+
raise RuntimeError("NullEmbedder produces no vectors; disable the dense leg")
|
|
54
|
+
|
|
55
|
+
async def embed_query(self, text: str) -> Vector:
|
|
56
|
+
raise RuntimeError("NullEmbedder produces no vectors; disable the dense leg")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
SearchLeg = Literal["dense", "sparse", "exact"]
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class SearchCandidate:
|
|
63
|
+
"""One recalled record with its source leg and a comparable score."""
|
|
64
|
+
|
|
65
|
+
__slots__ = ("record", "leg", "score")
|
|
66
|
+
|
|
67
|
+
def __init__(self, *, record: MemoryRecord, leg: SearchLeg, score: float) -> None:
|
|
68
|
+
self.record = record
|
|
69
|
+
self.leg = leg
|
|
70
|
+
self.score = score
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
__all__ = [
|
|
74
|
+
"NullEmbedder",
|
|
75
|
+
"SearchCandidate",
|
|
76
|
+
"SearchLeg",
|
|
77
|
+
"TextEmbedder",
|
|
78
|
+
"Vector",
|
|
79
|
+
]
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Copyright 2025-2026 Hanlian Lu. SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""PostgreSQL storage entry point for the Memory package.
|
|
3
|
+
|
|
4
|
+
Kept out of the package ``__init__`` so importing the core never drags the
|
|
5
|
+
database adapter into an embedding host's import graph; hosts that use
|
|
6
|
+
PostgreSQL import it directly:
|
|
7
|
+
|
|
8
|
+
from dlightrag_memory.postgres import PostgresMemoryStore
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from dlightrag_memory._storage.pg import PostgresMemoryStore
|
|
12
|
+
|
|
13
|
+
__all__ = ["PostgresMemoryStore"]
|
|
File without changes
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
# Copyright 2025-2026 Hanlian Lu. SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
"""Recency ordering for record tie-breaks.
|
|
3
|
+
|
|
4
|
+
Query relevance decides the candidate set (fusion in ``fusion.py``); time is
|
|
5
|
+
never a score component — it only orders presentation and breaks ties, the
|
|
6
|
+
same design MemMachine uses.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from datetime import UTC, datetime
|
|
12
|
+
|
|
13
|
+
from dlightrag_memory.models import MemoryRecord
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def recall_recency(record: MemoryRecord) -> datetime:
|
|
17
|
+
"""One record's standing-ordering timestamp."""
|
|
18
|
+
if record.updated_at is not None:
|
|
19
|
+
return record.updated_at
|
|
20
|
+
if record.created_at is not None:
|
|
21
|
+
return record.created_at
|
|
22
|
+
return datetime.min.replace(tzinfo=UTC)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
__all__ = ["recall_recency"]
|