spomory 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. memory_core/__init__.py +0 -0
  2. memory_core/audit.py +69 -0
  3. memory_core/export/__init__.py +0 -0
  4. memory_core/export/exporter.py +69 -0
  5. memory_core/export/schema.py +90 -0
  6. memory_core/graph/__init__.py +0 -0
  7. memory_core/graph/extract.py +16 -0
  8. memory_core/graph/incremental.py +174 -0
  9. memory_core/graph/local_store.py +287 -0
  10. memory_core/graph/models.py +98 -0
  11. memory_core/graph/postgres_store.py +202 -0
  12. memory_core/graph/store.py +104 -0
  13. memory_core/llm/__init__.py +0 -0
  14. memory_core/llm/base.py +35 -0
  15. memory_core/llm/embedding_base.py +16 -0
  16. memory_core/llm/local_sentence_transformer.py +26 -0
  17. memory_core/llm/openai_compatible.py +72 -0
  18. memory_core/llm/redact.py +44 -0
  19. memory_core/mcp_server/__init__.py +0 -0
  20. memory_core/mcp_server/__main__.py +13 -0
  21. memory_core/mcp_server/remote.py +235 -0
  22. memory_core/mcp_server/server.py +311 -0
  23. memory_core/memory_manager/__init__.py +0 -0
  24. memory_core/memory_manager/actions.py +95 -0
  25. memory_core/memory_manager/policy.py +120 -0
  26. memory_core/memory_manager/reward.py +28 -0
  27. memory_core/memory_manager/train_grpo.py +142 -0
  28. memory_core/multimodal/__init__.py +0 -0
  29. memory_core/multimodal/clip_verification.py +47 -0
  30. memory_core/multimodal/image_captioning.py +82 -0
  31. memory_core/onboarding.py +57 -0
  32. memory_core/retrieval/__init__.py +0 -0
  33. memory_core/retrieval/ppr.py +38 -0
  34. memory_core/retrieval/query_match.py +81 -0
  35. memory_core/retrieval/ranker.py +152 -0
  36. memory_core/usage.py +57 -0
  37. spomory-0.1.0.dist-info/METADATA +334 -0
  38. spomory-0.1.0.dist-info/RECORD +41 -0
  39. spomory-0.1.0.dist-info/WHEEL +4 -0
  40. spomory-0.1.0.dist-info/entry_points.txt +3 -0
  41. spomory-0.1.0.dist-info/licenses/LICENSE +201 -0
File without changes
memory_core/audit.py ADDED
@@ -0,0 +1,69 @@
1
+ """Append-only audit log for data-deletion events (Epic 10.3).
2
+
3
+ Backs the "记忆主权" promise's verifiability: every true-delete (Epic 7.3)
4
+ should leave behind a tamper-evident record of when it happened and what
5
+ was removed, even though the underlying data itself is gone.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ import sqlite3
12
+ import uuid
13
+ from dataclasses import dataclass
14
+ from datetime import UTC, datetime
15
+ from pathlib import Path
16
+
17
+ _SCHEMA = """
18
+ CREATE TABLE IF NOT EXISTS audit_log (
19
+ id TEXT PRIMARY KEY,
20
+ event_type TEXT NOT NULL,
21
+ user_id TEXT NOT NULL,
22
+ detail TEXT NOT NULL,
23
+ created_at TEXT NOT NULL
24
+ );
25
+ """
26
+
27
+
28
+ @dataclass
29
+ class AuditRecord:
30
+ id: str
31
+ event_type: str
32
+ user_id: str
33
+ detail: dict[str, object]
34
+ created_at: str
35
+
36
+
37
+ class AuditLog:
38
+ """Append-only by convention: this class exposes no update/delete methods."""
39
+
40
+ def __init__(self, db_path: str | Path = ":memory:") -> None:
41
+ self._conn = sqlite3.connect(str(db_path), check_same_thread=False)
42
+ self._conn.executescript(_SCHEMA)
43
+ self._conn.commit()
44
+
45
+ def record_deletion(self, user_id: str, entities_deleted: int, relations_deleted: int) -> AuditRecord:
46
+ record = AuditRecord(
47
+ id=str(uuid.uuid4()),
48
+ event_type="true_delete",
49
+ user_id=user_id,
50
+ detail={"entities_deleted": entities_deleted, "relations_deleted": relations_deleted},
51
+ created_at=datetime.now(UTC).isoformat(),
52
+ )
53
+ self._conn.execute(
54
+ "INSERT INTO audit_log (id, event_type, user_id, detail, created_at) VALUES (?, ?, ?, ?, ?)",
55
+ (record.id, record.event_type, record.user_id, json.dumps(record.detail), record.created_at),
56
+ )
57
+ self._conn.commit()
58
+ return record
59
+
60
+ def query(self, user_id: str) -> list[AuditRecord]:
61
+ rows = self._conn.execute(
62
+ "SELECT id, event_type, user_id, detail, created_at FROM audit_log "
63
+ "WHERE user_id = ? ORDER BY created_at",
64
+ (user_id,),
65
+ ).fetchall()
66
+ return [
67
+ AuditRecord(id=r[0], event_type=r[1], user_id=r[2], detail=json.loads(r[3]), created_at=r[4])
68
+ for r in rows
69
+ ]
File without changes
@@ -0,0 +1,69 @@
1
+ """Full export and true-delete: the technical backing for the "记忆护照"
2
+ promise — export the whole graph, or physically erase it, on demand.
3
+ """
4
+
5
+ from __future__ import annotations
6
+
7
+ from dataclasses import dataclass
8
+
9
+ from memory_core.audit import AuditLog
10
+ from memory_core.graph.store import GraphStoreBase
11
+
12
+ from .schema import ExportedEntity, ExportedRelation, MemoryExport
13
+
14
+
15
+ def export_all(store: GraphStoreBase, subject_id: str) -> MemoryExport:
16
+ entities = [ExportedEntity.from_entity(e) for e in store.all_entities()]
17
+ relations = [ExportedRelation.from_relation(r) for r in store.all_relations()]
18
+ return MemoryExport(subject_id=subject_id, entities=entities, relations=relations)
19
+
20
+
21
+ @dataclass
22
+ class DeletionReceipt:
23
+ """A verifiable confirmation that data was physically removed, not just hidden."""
24
+
25
+ subject_id: str
26
+ entities_deleted: int
27
+ relations_deleted: int
28
+ entities_remaining: int
29
+ relations_remaining: int
30
+
31
+ @property
32
+ def fully_deleted(self) -> bool:
33
+ return self.entities_remaining == 0 and self.relations_remaining == 0
34
+
35
+
36
+ def delete_all(
37
+ store: GraphStoreBase, subject_id: str, audit_log: AuditLog | None = None
38
+ ) -> DeletionReceipt:
39
+ """Physically delete every entity (and, transitively, every relation touching one).
40
+
41
+ If ``audit_log`` is given, the deletion is recorded there (Epic 10.3) —
42
+ a tamper-evident record that the deletion happened, kept separately
43
+ from the data that was actually removed.
44
+ """
45
+ entities_before = store.all_entities()
46
+ relations_before = len(store.all_relations())
47
+
48
+ for entity in entities_before:
49
+ store.delete_entity(entity.id)
50
+
51
+ remaining_entities = store.all_entities()
52
+ remaining_relations = store.all_relations()
53
+
54
+ receipt = DeletionReceipt(
55
+ subject_id=subject_id,
56
+ entities_deleted=len(entities_before) - len(remaining_entities),
57
+ relations_deleted=relations_before - len(remaining_relations),
58
+ entities_remaining=len(remaining_entities),
59
+ relations_remaining=len(remaining_relations),
60
+ )
61
+
62
+ if audit_log is not None:
63
+ audit_log.record_deletion(
64
+ user_id=subject_id,
65
+ entities_deleted=receipt.entities_deleted,
66
+ relations_deleted=receipt.relations_deleted,
67
+ )
68
+
69
+ return receipt
@@ -0,0 +1,90 @@
1
+ """The "记忆护照" (memory passport) export schema.
2
+
3
+ Design follows the MIF/PAM early-draft conventions cited in the business
4
+ plan: semantic/episodic/procedural memory classification, W3C PROV-O-style
5
+ provenance (not the full spec, just its shape — source, derivation span,
6
+ extractor), and a JSON-LD-flavored top-level ``@context``/``@type`` so the
7
+ file is self-describing without inventing a private format from scratch.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from datetime import UTC, datetime
13
+ from typing import Any
14
+
15
+ from pydantic import BaseModel, Field
16
+
17
+ from memory_core.graph.models import Entity, Relation
18
+
19
+ JSONLD_CONTEXT: dict[str, Any] = {
20
+ "@vocab": "https://schema.org/",
21
+ "memory": "https://memory-core.dev/schema/memory#",
22
+ "prov": "http://www.w3.org/ns/prov#",
23
+ }
24
+
25
+
26
+ class ExportedEntity(BaseModel):
27
+ """An entity as it appears in an export file — same fields as ``Entity``,
28
+ kept as a distinct model so the export format can evolve independently
29
+ of the internal storage model."""
30
+
31
+ id: str
32
+ name: str
33
+ type: str
34
+ memory_type: str
35
+ attributes: dict[str, str]
36
+ aliases: list[str]
37
+ provenance: list[dict[str, str]]
38
+ created_at: datetime
39
+ updated_at: datetime
40
+
41
+ @classmethod
42
+ def from_entity(cls, entity: Entity) -> ExportedEntity:
43
+ return cls(
44
+ id=entity.id,
45
+ name=entity.name,
46
+ type=entity.type,
47
+ memory_type=entity.memory_type,
48
+ attributes=entity.attributes,
49
+ aliases=entity.aliases,
50
+ provenance=[p.model_dump() for p in entity.provenance],
51
+ created_at=entity.created_at,
52
+ updated_at=entity.updated_at,
53
+ )
54
+
55
+
56
+ class ExportedRelation(BaseModel):
57
+ id: str
58
+ subject_id: str
59
+ predicate: str
60
+ object_id: str
61
+ confidence: float
62
+ memory_type: str
63
+ provenance: list[dict[str, str]]
64
+ created_at: datetime
65
+ updated_at: datetime
66
+
67
+ @classmethod
68
+ def from_relation(cls, relation: Relation) -> ExportedRelation:
69
+ return cls(
70
+ id=relation.id,
71
+ subject_id=relation.subject_id,
72
+ predicate=relation.predicate,
73
+ object_id=relation.object_id,
74
+ confidence=relation.confidence,
75
+ memory_type=relation.memory_type,
76
+ provenance=[p.model_dump() for p in relation.provenance],
77
+ created_at=relation.created_at,
78
+ updated_at=relation.updated_at,
79
+ )
80
+
81
+
82
+ class MemoryExport(BaseModel):
83
+ context: dict[str, Any] = Field(default=JSONLD_CONTEXT, alias="@context")
84
+ type: str = Field(default="memory:MemoryPassport", alias="@type")
85
+ exported_at: datetime = Field(default_factory=lambda: datetime.now(UTC))
86
+ subject_id: str
87
+ entities: list[ExportedEntity]
88
+ relations: list[ExportedRelation]
89
+
90
+ model_config = {"populate_by_name": True}
File without changes
@@ -0,0 +1,16 @@
1
+ """Text -> candidate triples extraction pipeline.
2
+
3
+ Thin wrapper over ``LLMProvider.extract_triples`` today; kept as its own
4
+ module because Epic 1.5's incremental merge logic needs a stable seam to
5
+ call into (and to swap in more elaborate chunking/prompting later without
6
+ touching callers).
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from memory_core.llm.base import LLMProvider, TripleCandidate
12
+
13
+
14
+ def extract_candidate_triples(text: str, llm: LLMProvider) -> list[TripleCandidate]:
15
+ """Extract candidate (subject, predicate, object) triples with source spans."""
16
+ return llm.extract_triples(text)
@@ -0,0 +1,174 @@
1
+ """Incremental write/merge logic.
2
+
3
+ Only the newly-extracted triples are touched on each call — there is no
4
+ step here that re-reads or re-derives the rest of the graph, which is the
5
+ property that makes this "incremental" rather than an indexer that
6
+ rebuilds everything (the LightRAG-style design choice called out in the
7
+ business plan, as opposed to GraphRAG's full-rebuild-per-update approach).
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from dataclasses import dataclass, field
13
+ from typing import TYPE_CHECKING
14
+
15
+ from memory_core.llm.base import LLMProvider
16
+ from memory_core.llm.redact import redact_secrets
17
+
18
+ from .models import Entity, Provenance, Relation
19
+ from .store import GraphStoreBase
20
+
21
+ if TYPE_CHECKING:
22
+ from memory_core.memory_manager.policy import MemoryPolicy
23
+
24
+
25
+ # Epic 12.4: exact-match, case/punctuation-insensitive filler phrases that
26
+ # carry no extractable fact -- deliberately a small, conservative allowlist
27
+ # rather than a length threshold or fuzzy match, so a short but meaningful
28
+ # sentence ("I quit.") is never mistaken for filler. Skipping these avoids
29
+ # an LLM extraction call (real money, and the only thing standing between a
30
+ # public endpoint like /demo/try and a free way to burn API budget) for
31
+ # input that could never contain a triple anyway.
32
+ _LOW_INFORMATION_PHRASES = {
33
+ "thanks",
34
+ "thank you",
35
+ "ok",
36
+ "okay",
37
+ "got it",
38
+ "sounds good",
39
+ "sure",
40
+ "yes",
41
+ "no",
42
+ "hi",
43
+ "hello",
44
+ "bye",
45
+ "goodbye",
46
+ "谢谢",
47
+ "谢谢你",
48
+ "好的",
49
+ "好",
50
+ "嗯",
51
+ "在吗",
52
+ "在",
53
+ "明白",
54
+ "明白了",
55
+ "知道了",
56
+ "收到",
57
+ }
58
+
59
+ _STRIP_CHARS = " \t\n\r!?。!?.,,、~~"
60
+
61
+
62
+ def _is_low_information(text: str) -> bool:
63
+ normalized = text.strip(_STRIP_CHARS).lower()
64
+ return normalized in _LOW_INFORMATION_PHRASES
65
+
66
+
67
+ @dataclass
68
+ class IngestResult:
69
+ new_entities: int = 0
70
+ merged_entities: int = 0
71
+ new_relations: int = 0
72
+ updated_relations: int = 0
73
+ noop_relations: int = 0
74
+ entity_ids_by_name: dict[str, str] = field(default_factory=dict)
75
+
76
+
77
+ class IncrementalIngestor:
78
+ """Extracts triples from new text and merges them into an existing ``GraphStoreBase``.
79
+
80
+ Entity resolution policy: an incoming subject/object name is merged into
81
+ an existing entity only on an exact (case/whitespace-insensitive) match
82
+ against that entity's name or a known alias. Ambiguous cases (multiple
83
+ existing entities share the name) are resolved conservatively by
84
+ creating a new entity rather than risking a false merge between two
85
+ distinct same-named entities — a false split is easier to fix later
86
+ than a false merge that silently conflates two people/things.
87
+ """
88
+
89
+ def __init__(
90
+ self, store: GraphStoreBase, llm: LLMProvider, policy: MemoryPolicy | None = None
91
+ ) -> None:
92
+ """``policy`` (Epic 3.2/3.5): when given, every extracted candidate is
93
+ routed through ``policy.decide()`` (ADD/UPDATE/DELETE/NOOP) instead of
94
+ being unconditionally written — this is what actually wires Epic 3's
95
+ memory-management layer into the ingestion path Epic 1+2 use, rather
96
+ than leaving it a standalone, never-called module. Defaults to
97
+ ``None`` (unconditional add/merge) to keep existing callers'
98
+ behavior unchanged.
99
+ """
100
+ self.store = store
101
+ self.llm = llm
102
+ self.policy = policy
103
+
104
+ def ingest(self, text: str, source_id: str) -> IngestResult:
105
+ result = IngestResult()
106
+ if _is_low_information(text):
107
+ return result
108
+
109
+ text = redact_secrets(text)
110
+ candidates = self.llm.extract_triples(text)
111
+
112
+ new_entities: list[Entity] = []
113
+ pending_relations: list[Relation] = []
114
+
115
+ def resolve(name: str) -> str:
116
+ if name in result.entity_ids_by_name:
117
+ return result.entity_ids_by_name[name]
118
+
119
+ matches = self.store.find_entities_by_name(name)
120
+ if len(matches) == 1:
121
+ entity = matches[0]
122
+ result.merged_entities += 1
123
+ else:
124
+ entity = Entity(name=name, type="unknown")
125
+ new_entities.append(entity)
126
+ result.new_entities += 1
127
+
128
+ result.entity_ids_by_name[name] = entity.id
129
+ return entity.id
130
+
131
+ for candidate in candidates:
132
+ subject_id = resolve(candidate.subject)
133
+ object_id = resolve(candidate.object)
134
+ relation = Relation(
135
+ subject_id=subject_id,
136
+ predicate=candidate.predicate,
137
+ object_id=object_id,
138
+ provenance=[Provenance(source_id=source_id, source_span=candidate.source_span)],
139
+ )
140
+
141
+ if self.policy is None:
142
+ pending_relations.append(relation)
143
+ result.new_relations += 1
144
+ continue
145
+
146
+ # New entities must be visible to the store before the policy can
147
+ # meaningfully query "what do we already know about this subject"
148
+ # (get_neighbors), so flush them immediately rather than batching.
149
+ if new_entities:
150
+ self.store.add_entities(new_entities)
151
+ new_entities = []
152
+ self._apply_via_policy(relation, result)
153
+
154
+ if new_entities:
155
+ self.store.add_entities(new_entities)
156
+ if pending_relations:
157
+ self.store.add_relations(pending_relations)
158
+
159
+ return result
160
+
161
+ def _apply_via_policy(self, relation: Relation, result: IngestResult) -> None:
162
+ from memory_core.memory_manager.actions import ActionType, apply_action
163
+
164
+ action = self.policy.decide(relation, self.store)
165
+ apply_action(action, self.store)
166
+
167
+ if action.action_type is ActionType.ADD:
168
+ result.new_relations += 1
169
+ elif action.action_type is ActionType.UPDATE:
170
+ result.updated_relations += 1
171
+ elif action.action_type is ActionType.NOOP:
172
+ result.noop_relations += 1
173
+ # DELETE isn't reachable from RuleBasedPolicy's own candidate-vs-existing
174
+ # comparison today, but is handled uniformly by apply_action() either way.