contextos-memory-runtime 1.0.0rc2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contextos/__init__.py +3 -0
- contextos/__main__.py +6 -0
- contextos/api/__init__.py +1 -0
- contextos/api/routes/__init__.py +1 -0
- contextos/api/routes/desktop.py +322 -0
- contextos/api/routes/ingest.py +17 -0
- contextos/api/routes/memories.py +84 -0
- contextos/api/routes/models.py +81 -0
- contextos/api/routes/retrieval.py +89 -0
- contextos/api/routes/system.py +216 -0
- contextos/api/server.py +195 -0
- contextos/benchmarks/__init__.py +1 -0
- contextos/benchmarks/compilation.py +245 -0
- contextos/benchmarks/connectors.py +423 -0
- contextos/benchmarks/explainability.py +103 -0
- contextos/benchmarks/final.py +406 -0
- contextos/benchmarks/graph.py +310 -0
- contextos/benchmarks/graph_adversarial.py +525 -0
- contextos/benchmarks/mcp.py +324 -0
- contextos/benchmarks/model_routing.py +203 -0
- contextos/benchmarks/optimization.py +305 -0
- contextos/benchmarks/rescue_integration.py +127 -0
- contextos/benchmarks/retrieval.py +266 -0
- contextos/benchmarks/temporal.py +377 -0
- contextos/benchmarks/temporal_hotpath.py +76 -0
- contextos/benchmarks/terminal.py +62 -0
- contextos/cli/__init__.py +1 -0
- contextos/cli/app.py +932 -0
- contextos/cli/dashboard.py +174 -0
- contextos/cli/formatters.py +299 -0
- contextos/config/__init__.py +1 -0
- contextos/config/settings.py +160 -0
- contextos/connectors/__init__.py +6 -0
- contextos/connectors/fake.py +11 -0
- contextos/connectors/json_import.py +125 -0
- contextos/connectors/local_files.py +102 -0
- contextos/connectors/manager.py +293 -0
- contextos/connectors/models.py +62 -0
- contextos/connectors/protocols.py +11 -0
- contextos/core/__init__.py +103 -0
- contextos/core/enums.py +489 -0
- contextos/core/exceptions.py +293 -0
- contextos/core/models.py +1147 -0
- contextos/core/protocols.py +549 -0
- contextos/daemon/__init__.py +1 -0
- contextos/daemon/manager.py +510 -0
- contextos/daemon/state.py +127 -0
- contextos/daemon/wiring.py +296 -0
- contextos/demo.py +217 -0
- contextos/embedding/__init__.py +1 -0
- contextos/embedding/deterministic.py +76 -0
- contextos/embedding/sentence_transformers.py +80 -0
- contextos/mcp/__init__.py +5 -0
- contextos/mcp/server.py +269 -0
- contextos/providers/__init__.py +13 -0
- contextos/providers/fake.py +217 -0
- contextos/providers/ollama.py +297 -0
- contextos/providers/openai_compatible.py +337 -0
- contextos/services/__init__.py +1 -0
- contextos/services/compilation.py +535 -0
- contextos/services/explainability.py +553 -0
- contextos/services/extraction.py +311 -0
- contextos/services/graph.py +524 -0
- contextos/services/graph_retrieval.py +143 -0
- contextos/services/ingestion.py +143 -0
- contextos/services/inspection.py +174 -0
- contextos/services/memory.py +291 -0
- contextos/services/model_service.py +409 -0
- contextos/services/optimization.py +426 -0
- contextos/services/privacy.py +331 -0
- contextos/services/retrieval.py +302 -0
- contextos/services/retrieval_index.py +88 -0
- contextos/services/router.py +302 -0
- contextos/services/secret_scanner.py +207 -0
- contextos/services/telemetry_query.py +102 -0
- contextos/services/temporal.py +500 -0
- contextos/services/token_counter.py +222 -0
- contextos/storage/__init__.py +1 -0
- contextos/storage/connector_repo.py +67 -0
- contextos/storage/database.py +497 -0
- contextos/storage/event_repo.py +137 -0
- contextos/storage/graph_repo.py +228 -0
- contextos/storage/lexical/__init__.py +1 -0
- contextos/storage/lexical/bm25.py +134 -0
- contextos/storage/memory_repo.py +589 -0
- contextos/storage/relation_repo.py +80 -0
- contextos/storage/telemetry_repo.py +481 -0
- contextos/storage/vector/__init__.py +1 -0
- contextos/storage/vector/in_memory.py +162 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,311 @@
|
|
|
1
|
+
"""Deterministic candidate extraction for ContextOS Phase 2.
|
|
2
|
+
|
|
3
|
+
The extractor is deliberately local and side-effect free. It turns raw user
|
|
4
|
+
text into unaccepted CandidateMemory objects; it never writes long-term memory.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import hashlib
|
|
10
|
+
import re
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
|
|
13
|
+
from contextos.core.enums import (
|
|
14
|
+
CandidateAction,
|
|
15
|
+
CandidateTemporalStatus,
|
|
16
|
+
MemoryType,
|
|
17
|
+
SourceRole,
|
|
18
|
+
)
|
|
19
|
+
from contextos.core.models import CandidateMemory
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
MAX_INPUT_CHARS = 100_000
|
|
23
|
+
MAX_CANDIDATES = 100
|
|
24
|
+
MAX_CLAUSE_CHARS = 9_000
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass(frozen=True)
|
|
28
|
+
class _Clause:
|
|
29
|
+
text: str
|
|
30
|
+
evidence: str
|
|
31
|
+
start: int | None = None
|
|
32
|
+
end: int | None = None
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
_TYPE_RULES: tuple[tuple[MemoryType, re.Pattern[str]], ...] = (
|
|
36
|
+
(MemoryType.PREFERENCE, re.compile(
|
|
37
|
+
r"\b(?:prefer|preference|favorite|always use|never use|keep (?:your )?answers)\b", re.I
|
|
38
|
+
)),
|
|
39
|
+
(MemoryType.GOAL, re.compile(
|
|
40
|
+
r"\b(?:want to|would like to|plan(?:ning)? to|going to|goal|prepar(?:e|ing) for|"
|
|
41
|
+
r"focus(?:ing)?(?: mainly)? on|switch(?:ing)? (?:from|to)|"
|
|
42
|
+
r"(?:i(?:'ll| will| might| may)|maybe i(?:'ll| will)) learn|might learn|may learn|will learn)\b",
|
|
43
|
+
re.I,
|
|
44
|
+
)),
|
|
45
|
+
(MemoryType.PROJECT, re.compile(
|
|
46
|
+
r"\b(?:building|working on|developing|creating|maintaining|my project|current project)\b", re.I
|
|
47
|
+
)),
|
|
48
|
+
(MemoryType.SKILL, re.compile(
|
|
49
|
+
r"\b(?:proficient|experienced|skilled|fluent|know|learning|learned|experience with)\b", re.I
|
|
50
|
+
)),
|
|
51
|
+
(MemoryType.RELATIONSHIP, re.compile(
|
|
52
|
+
r"\b(?:manager|boss|lead|mentor|colleague|friend|partner|wife|husband)\b", re.I
|
|
53
|
+
)),
|
|
54
|
+
(MemoryType.PROCEDURE, re.compile(
|
|
55
|
+
r"\b(?:workflow|process|routine|setup|to (?:deploy|build|test|run|install|configure))\b", re.I
|
|
56
|
+
)),
|
|
57
|
+
(MemoryType.OPINION, re.compile(r"\b(?:i think|i believe|in my opinion|in my view)\b", re.I)),
|
|
58
|
+
(MemoryType.FACT, re.compile(
|
|
59
|
+
r"\b(?:work(?:ing)? at|live in|study at|teach at|my (?:name|role|job|company)|i use)\b", re.I
|
|
60
|
+
)),
|
|
61
|
+
(MemoryType.TEMPORAL, re.compile(
|
|
62
|
+
r"\b(?:today|tomorrow|this (?:week|month|sprint)|next (?:week|month)|right now)\b", re.I
|
|
63
|
+
)),
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
_FILLER = re.compile(
|
|
67
|
+
r"^(?:ok(?:ay)?|sure|yes|no|thanks|thank you|nice|lol|continue|got it|understood|"
|
|
68
|
+
r"hi|hello|hey|hmm|huh)[.!?]*$",
|
|
69
|
+
re.I,
|
|
70
|
+
)
|
|
71
|
+
_CASUAL = re.compile(r"^(?:the )?weather (?:looks|is) (?:good|nice|great)[.!?]*$", re.I)
|
|
72
|
+
_PERSONAL_SIGNAL = re.compile(r"\b(?:i|i'm|i've|i'll|i'd|my|me|user|the user)\b", re.I)
|
|
73
|
+
_UNCERTAIN = re.compile(r"\b(?:maybe|might|may|perhaps|possibly|i guess|not sure)\b", re.I)
|
|
74
|
+
_HISTORICAL = re.compile(r"\b(?:used to|previously|in the past|before)\b", re.I)
|
|
75
|
+
_CHANGE = re.compile(r"\b(?:stopped|no longer|not anymore|anymore|switching from|used to)\b", re.I)
|
|
76
|
+
_FUTURE = re.compile(
|
|
77
|
+
r"\b(?:plan(?:ning)? to|going to|will|might|may|next (?:week|month|year)|later|someday)\b",
|
|
78
|
+
re.I,
|
|
79
|
+
)
|
|
80
|
+
_CURRENT = re.compile(r"\b(?:now|currently|right now|recently|today)\b", re.I)
|
|
81
|
+
_NEGATION = re.compile(r"\b(?:don't|do not|doesn't|does not|never|no longer|stopped)\b", re.I)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _normalize_input(text: str) -> str:
|
|
85
|
+
text = text.replace("\r\n", "\n").replace("\r", "\n")
|
|
86
|
+
text = text.replace("’", "'").replace("“", '"').replace("”", '"')
|
|
87
|
+
return re.sub(r"[ \t]+", " ", text).strip()
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _locate(raw_text: str, evidence: str) -> tuple[int | None, int | None]:
|
|
91
|
+
start = raw_text.casefold().find(evidence.casefold())
|
|
92
|
+
return (start, start + len(evidence)) if start >= 0 else (None, None)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _expand_switch_statement(sentence: str, raw_text: str) -> list[_Clause] | None:
|
|
96
|
+
match = re.fullmatch(
|
|
97
|
+
r"i(?:'m| am) switching from (?P<old>.+?) to (?P<new>.+?)[.!?]?",
|
|
98
|
+
sentence.strip(),
|
|
99
|
+
re.I,
|
|
100
|
+
)
|
|
101
|
+
if not match:
|
|
102
|
+
return None
|
|
103
|
+
start, end = _locate(raw_text, sentence)
|
|
104
|
+
return [
|
|
105
|
+
_Clause(f"I used to focus on {match.group('old')}", sentence, start, end),
|
|
106
|
+
_Clause(f"I am switching to {match.group('new')}", sentence, start, end),
|
|
107
|
+
]
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _candidate_clauses(text: str) -> list[_Clause]:
|
|
111
|
+
clauses: list[_Clause] = []
|
|
112
|
+
sentences = re.split(r"(?<=[.!?])\s+|\n+", text)
|
|
113
|
+
for sentence in sentences:
|
|
114
|
+
sentence = sentence.strip()
|
|
115
|
+
if not sentence:
|
|
116
|
+
continue
|
|
117
|
+
expanded = _expand_switch_statement(sentence, text)
|
|
118
|
+
if expanded is not None:
|
|
119
|
+
clauses.extend(expanded)
|
|
120
|
+
continue
|
|
121
|
+
|
|
122
|
+
contrast_parts = re.split(r"\s*,?\s*\b(?:but|however)\b\s*", sentence, flags=re.I)
|
|
123
|
+
for part in contrast_parts:
|
|
124
|
+
atomic_parts = re.split(
|
|
125
|
+
r"\s*;\s*|\s+\band\b\s+(?=(?:i\b|i'm\b|i am\b|i've\b|i'll\b|i'd\b|my\b))",
|
|
126
|
+
part,
|
|
127
|
+
flags=re.I,
|
|
128
|
+
)
|
|
129
|
+
for atomic in atomic_parts:
|
|
130
|
+
atomic = atomic.strip(" ,")
|
|
131
|
+
if not atomic:
|
|
132
|
+
continue
|
|
133
|
+
if len(atomic) > MAX_CLAUSE_CHARS:
|
|
134
|
+
atomic = atomic[:MAX_CLAUSE_CHARS].rstrip()
|
|
135
|
+
start, end = _locate(text, atomic)
|
|
136
|
+
clauses.append(_Clause(atomic, atomic, start, end))
|
|
137
|
+
return clauses
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _is_memory_worthy(text: str, suggested_type: MemoryType | None) -> bool:
|
|
141
|
+
stripped = text.strip()
|
|
142
|
+
if not stripped or _FILLER.fullmatch(stripped) or _CASUAL.fullmatch(stripped):
|
|
143
|
+
return False
|
|
144
|
+
if len(stripped.strip(".!? ")) < 4:
|
|
145
|
+
return False
|
|
146
|
+
return suggested_type is not None or bool(_PERSONAL_SIGNAL.search(stripped))
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _classify(text: str, suggested_type: MemoryType | None) -> MemoryType:
|
|
150
|
+
if suggested_type is not None:
|
|
151
|
+
return suggested_type
|
|
152
|
+
for memory_type, pattern in _TYPE_RULES:
|
|
153
|
+
if pattern.search(text):
|
|
154
|
+
return memory_type
|
|
155
|
+
return MemoryType.CONTEXT
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _temporal_analysis(text: str) -> tuple[CandidateTemporalStatus, str | None, CandidateAction]:
|
|
159
|
+
hint_match: re.Match[str] | None
|
|
160
|
+
if (hint_match := _HISTORICAL.search(text)) is not None:
|
|
161
|
+
return CandidateTemporalStatus.HISTORICAL, hint_match.group(0), CandidateAction.SUPERSEDE
|
|
162
|
+
if (hint_match := _CHANGE.search(text)) is not None:
|
|
163
|
+
return CandidateTemporalStatus.HISTORICAL, hint_match.group(0), CandidateAction.SUPERSEDE
|
|
164
|
+
if (hint_match := _FUTURE.search(text)) is not None:
|
|
165
|
+
return CandidateTemporalStatus.FUTURE, hint_match.group(0), CandidateAction.ADD
|
|
166
|
+
if (hint_match := _CURRENT.search(text)) is not None:
|
|
167
|
+
return CandidateTemporalStatus.CURRENT, hint_match.group(0), CandidateAction.ADD
|
|
168
|
+
return CandidateTemporalStatus.CURRENT, None, CandidateAction.ADD
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _canonicalize(text: str) -> str:
|
|
172
|
+
value = text.strip().rstrip(".!?").strip()
|
|
173
|
+
value = re.sub(r"^also\s+", "", value, flags=re.I)
|
|
174
|
+
value = re.sub(r"^i also\s+", "I ", value, flags=re.I)
|
|
175
|
+
value = re.sub(
|
|
176
|
+
r"^keep (?:your )?answers (?:short|concise)$",
|
|
177
|
+
"User prefers concise answers",
|
|
178
|
+
value,
|
|
179
|
+
flags=re.I,
|
|
180
|
+
)
|
|
181
|
+
value = re.sub(r"^maybe\s+i(?:'ll| will)\s+", "User may ", value, flags=re.I)
|
|
182
|
+
value = re.sub(r"^now\s+i(?:'m| am)\s+", "User is now ", value, flags=re.I)
|
|
183
|
+
replacements = (
|
|
184
|
+
(r"^i don't\s+", "User does not "),
|
|
185
|
+
(r"^i do not\s+", "User does not "),
|
|
186
|
+
(r"^i prefer\s+", "User prefers "),
|
|
187
|
+
(r"^i use\s+", "User uses "),
|
|
188
|
+
(r"^i want\s+", "User wants "),
|
|
189
|
+
(r"^i plan\s+", "User plans "),
|
|
190
|
+
(r"^i like\s+", "User likes "),
|
|
191
|
+
(r"^i work\s+", "User works "),
|
|
192
|
+
(r"^i think\s+", "User thinks "),
|
|
193
|
+
(r"^i(?:'m| am)\s+", "User is "),
|
|
194
|
+
(r"^i(?:'ve| have)\s+", "User has "),
|
|
195
|
+
(r"^i(?:'ll| will)\s+", "User will "),
|
|
196
|
+
(r"^i(?:'d| would)\s+", "User would "),
|
|
197
|
+
(r"^i\s+", "User "),
|
|
198
|
+
(r"^my\s+", "User's "),
|
|
199
|
+
)
|
|
200
|
+
for pattern, replacement in replacements:
|
|
201
|
+
updated = re.sub(pattern, replacement, value, count=1, flags=re.I)
|
|
202
|
+
if updated != value:
|
|
203
|
+
value = updated
|
|
204
|
+
break
|
|
205
|
+
value = re.sub(r"\bshort answers\b", "concise answers", value, flags=re.I)
|
|
206
|
+
return re.sub(r"\s+", " ", value).strip()
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _confidence(text: str) -> float:
|
|
210
|
+
score = 0.88
|
|
211
|
+
if _UNCERTAIN.search(text):
|
|
212
|
+
score -= 0.32
|
|
213
|
+
if re.search(r"\b(?:think|guess|could)\b", text, re.I):
|
|
214
|
+
score -= 0.12
|
|
215
|
+
if re.search(r"\b(?:always|never|definitely|absolutely|certainly)\b", text, re.I):
|
|
216
|
+
score += 0.07
|
|
217
|
+
return round(min(1.0, max(0.0, score)), 2)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _importance(memory_type: MemoryType, text: str) -> float:
|
|
221
|
+
scores = {
|
|
222
|
+
MemoryType.GOAL: 0.85,
|
|
223
|
+
MemoryType.PROJECT: 0.85,
|
|
224
|
+
MemoryType.PREFERENCE: 0.8,
|
|
225
|
+
MemoryType.PROCEDURE: 0.75,
|
|
226
|
+
MemoryType.SKILL: 0.7,
|
|
227
|
+
MemoryType.FACT: 0.65,
|
|
228
|
+
MemoryType.RELATIONSHIP: 0.65,
|
|
229
|
+
MemoryType.OPINION: 0.55,
|
|
230
|
+
MemoryType.CONTEXT: 0.5,
|
|
231
|
+
MemoryType.TEMPORAL: 0.4,
|
|
232
|
+
}
|
|
233
|
+
score = scores[memory_type]
|
|
234
|
+
if re.search(r"\b(?:today|tomorrow|this week|right now)\b", text, re.I):
|
|
235
|
+
score -= 0.15
|
|
236
|
+
return round(min(1.0, max(0.0, score)), 2)
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _dedup_key(candidate: CandidateMemory) -> str:
|
|
240
|
+
return re.sub(r"[^a-z0-9]+", " ", candidate.content.casefold()).strip()
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
class RuleBasedMemoryExtractor:
|
|
244
|
+
"""Side-effect-free deterministic baseline implementing MemoryExtractor."""
|
|
245
|
+
|
|
246
|
+
def __init__(self, *, min_confidence: float = 0.3, max_candidates: int = MAX_CANDIDATES) -> None:
|
|
247
|
+
self._min_confidence = min_confidence
|
|
248
|
+
self._max_candidates = max_candidates
|
|
249
|
+
|
|
250
|
+
async def extract(
|
|
251
|
+
self,
|
|
252
|
+
text: str,
|
|
253
|
+
source_type: str = "cli_input",
|
|
254
|
+
source_uri: str | None = None,
|
|
255
|
+
suggested_type: MemoryType | None = None,
|
|
256
|
+
tags: list[str] | None = None,
|
|
257
|
+
source_role: SourceRole = SourceRole.USER,
|
|
258
|
+
confirmed_user_information: bool = False,
|
|
259
|
+
) -> list[CandidateMemory]:
|
|
260
|
+
if not text or not text.strip():
|
|
261
|
+
return []
|
|
262
|
+
if source_role != SourceRole.USER and not confirmed_user_information:
|
|
263
|
+
return []
|
|
264
|
+
|
|
265
|
+
normalized = _normalize_input(text[:MAX_INPUT_CHARS])
|
|
266
|
+
input_hash = hashlib.sha256(normalized.encode("utf-8")).hexdigest()
|
|
267
|
+
candidates: list[CandidateMemory] = []
|
|
268
|
+
seen: set[str] = set()
|
|
269
|
+
|
|
270
|
+
for clause in _candidate_clauses(normalized):
|
|
271
|
+
if len(candidates) >= self._max_candidates:
|
|
272
|
+
break
|
|
273
|
+
if not _is_memory_worthy(clause.text, suggested_type):
|
|
274
|
+
continue
|
|
275
|
+
memory_type = _classify(clause.text, suggested_type)
|
|
276
|
+
temporal_status, temporal_hint, action_hint = _temporal_analysis(clause.text)
|
|
277
|
+
confidence = _confidence(clause.text)
|
|
278
|
+
if confidence < self._min_confidence:
|
|
279
|
+
continue
|
|
280
|
+
candidate = CandidateMemory(
|
|
281
|
+
content=_canonicalize(clause.text),
|
|
282
|
+
memory_type=memory_type,
|
|
283
|
+
confidence=confidence,
|
|
284
|
+
importance=_importance(memory_type, clause.text),
|
|
285
|
+
temporal_status=temporal_status,
|
|
286
|
+
temporal_hint=temporal_hint,
|
|
287
|
+
action_hint=action_hint,
|
|
288
|
+
uncertain=bool(_UNCERTAIN.search(clause.text)),
|
|
289
|
+
negated=bool(_NEGATION.search(clause.text)),
|
|
290
|
+
source_type=source_type,
|
|
291
|
+
source_uri=source_uri,
|
|
292
|
+
source_role=source_role,
|
|
293
|
+
evidence=clause.evidence,
|
|
294
|
+
evidence_start=clause.start,
|
|
295
|
+
evidence_end=clause.end,
|
|
296
|
+
tags=list(tags or []),
|
|
297
|
+
metadata={
|
|
298
|
+
"input_hash": input_hash,
|
|
299
|
+
"negated": bool(_NEGATION.search(clause.text)),
|
|
300
|
+
"confirmed_user_information": confirmed_user_information,
|
|
301
|
+
"input_truncated": len(text) > MAX_INPUT_CHARS,
|
|
302
|
+
"extractor": "deterministic-v2",
|
|
303
|
+
},
|
|
304
|
+
)
|
|
305
|
+
key = _dedup_key(candidate)
|
|
306
|
+
if key in seen:
|
|
307
|
+
continue
|
|
308
|
+
seen.add(key)
|
|
309
|
+
candidates.append(candidate)
|
|
310
|
+
|
|
311
|
+
return candidates
|