contextos-memory-runtime 1.0.0rc2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. contextos/__init__.py +3 -0
  2. contextos/__main__.py +6 -0
  3. contextos/api/__init__.py +1 -0
  4. contextos/api/routes/__init__.py +1 -0
  5. contextos/api/routes/desktop.py +322 -0
  6. contextos/api/routes/ingest.py +17 -0
  7. contextos/api/routes/memories.py +84 -0
  8. contextos/api/routes/models.py +81 -0
  9. contextos/api/routes/retrieval.py +89 -0
  10. contextos/api/routes/system.py +216 -0
  11. contextos/api/server.py +195 -0
  12. contextos/benchmarks/__init__.py +1 -0
  13. contextos/benchmarks/compilation.py +245 -0
  14. contextos/benchmarks/connectors.py +423 -0
  15. contextos/benchmarks/explainability.py +103 -0
  16. contextos/benchmarks/final.py +406 -0
  17. contextos/benchmarks/graph.py +310 -0
  18. contextos/benchmarks/graph_adversarial.py +525 -0
  19. contextos/benchmarks/mcp.py +324 -0
  20. contextos/benchmarks/model_routing.py +203 -0
  21. contextos/benchmarks/optimization.py +305 -0
  22. contextos/benchmarks/rescue_integration.py +127 -0
  23. contextos/benchmarks/retrieval.py +266 -0
  24. contextos/benchmarks/temporal.py +377 -0
  25. contextos/benchmarks/temporal_hotpath.py +76 -0
  26. contextos/benchmarks/terminal.py +62 -0
  27. contextos/cli/__init__.py +1 -0
  28. contextos/cli/app.py +932 -0
  29. contextos/cli/dashboard.py +174 -0
  30. contextos/cli/formatters.py +299 -0
  31. contextos/config/__init__.py +1 -0
  32. contextos/config/settings.py +160 -0
  33. contextos/connectors/__init__.py +6 -0
  34. contextos/connectors/fake.py +11 -0
  35. contextos/connectors/json_import.py +125 -0
  36. contextos/connectors/local_files.py +102 -0
  37. contextos/connectors/manager.py +293 -0
  38. contextos/connectors/models.py +62 -0
  39. contextos/connectors/protocols.py +11 -0
  40. contextos/core/__init__.py +103 -0
  41. contextos/core/enums.py +489 -0
  42. contextos/core/exceptions.py +293 -0
  43. contextos/core/models.py +1147 -0
  44. contextos/core/protocols.py +549 -0
  45. contextos/daemon/__init__.py +1 -0
  46. contextos/daemon/manager.py +510 -0
  47. contextos/daemon/state.py +127 -0
  48. contextos/daemon/wiring.py +296 -0
  49. contextos/demo.py +217 -0
  50. contextos/embedding/__init__.py +1 -0
  51. contextos/embedding/deterministic.py +76 -0
  52. contextos/embedding/sentence_transformers.py +80 -0
  53. contextos/mcp/__init__.py +5 -0
  54. contextos/mcp/server.py +269 -0
  55. contextos/providers/__init__.py +13 -0
  56. contextos/providers/fake.py +217 -0
  57. contextos/providers/ollama.py +297 -0
  58. contextos/providers/openai_compatible.py +337 -0
  59. contextos/services/__init__.py +1 -0
  60. contextos/services/compilation.py +535 -0
  61. contextos/services/explainability.py +553 -0
  62. contextos/services/extraction.py +311 -0
  63. contextos/services/graph.py +524 -0
  64. contextos/services/graph_retrieval.py +143 -0
  65. contextos/services/ingestion.py +143 -0
  66. contextos/services/inspection.py +174 -0
  67. contextos/services/memory.py +291 -0
  68. contextos/services/model_service.py +409 -0
  69. contextos/services/optimization.py +426 -0
  70. contextos/services/privacy.py +331 -0
  71. contextos/services/retrieval.py +302 -0
  72. contextos/services/retrieval_index.py +88 -0
  73. contextos/services/router.py +302 -0
  74. contextos/services/secret_scanner.py +207 -0
  75. contextos/services/telemetry_query.py +102 -0
  76. contextos/services/temporal.py +500 -0
  77. contextos/services/token_counter.py +222 -0
  78. contextos/storage/__init__.py +1 -0
  79. contextos/storage/connector_repo.py +67 -0
  80. contextos/storage/database.py +497 -0
  81. contextos/storage/event_repo.py +137 -0
  82. contextos/storage/graph_repo.py +228 -0
  83. contextos/storage/lexical/__init__.py +1 -0
  84. contextos/storage/lexical/bm25.py +134 -0
  85. contextos/storage/memory_repo.py +589 -0
  86. contextos/storage/relation_repo.py +80 -0
  87. contextos/storage/telemetry_repo.py +481 -0
  88. contextos/storage/vector/__init__.py +1 -0
  89. contextos/storage/vector/in_memory.py +162 -0
  90. contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
  91. contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
  92. contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
  93. contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,311 @@
1
+ """Deterministic candidate extraction for ContextOS Phase 2.
2
+
3
+ The extractor is deliberately local and side-effect free. It turns raw user
4
+ text into unaccepted CandidateMemory objects; it never writes long-term memory.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import hashlib
10
+ import re
11
+ from dataclasses import dataclass
12
+
13
+ from contextos.core.enums import (
14
+ CandidateAction,
15
+ CandidateTemporalStatus,
16
+ MemoryType,
17
+ SourceRole,
18
+ )
19
+ from contextos.core.models import CandidateMemory
20
+
21
+
22
+ MAX_INPUT_CHARS = 100_000
23
+ MAX_CANDIDATES = 100
24
+ MAX_CLAUSE_CHARS = 9_000
25
+
26
+
27
+ @dataclass(frozen=True)
28
+ class _Clause:
29
+ text: str
30
+ evidence: str
31
+ start: int | None = None
32
+ end: int | None = None
33
+
34
+
35
+ _TYPE_RULES: tuple[tuple[MemoryType, re.Pattern[str]], ...] = (
36
+ (MemoryType.PREFERENCE, re.compile(
37
+ r"\b(?:prefer|preference|favorite|always use|never use|keep (?:your )?answers)\b", re.I
38
+ )),
39
+ (MemoryType.GOAL, re.compile(
40
+ r"\b(?:want to|would like to|plan(?:ning)? to|going to|goal|prepar(?:e|ing) for|"
41
+ r"focus(?:ing)?(?: mainly)? on|switch(?:ing)? (?:from|to)|"
42
+ r"(?:i(?:'ll| will| might| may)|maybe i(?:'ll| will)) learn|might learn|may learn|will learn)\b",
43
+ re.I,
44
+ )),
45
+ (MemoryType.PROJECT, re.compile(
46
+ r"\b(?:building|working on|developing|creating|maintaining|my project|current project)\b", re.I
47
+ )),
48
+ (MemoryType.SKILL, re.compile(
49
+ r"\b(?:proficient|experienced|skilled|fluent|know|learning|learned|experience with)\b", re.I
50
+ )),
51
+ (MemoryType.RELATIONSHIP, re.compile(
52
+ r"\b(?:manager|boss|lead|mentor|colleague|friend|partner|wife|husband)\b", re.I
53
+ )),
54
+ (MemoryType.PROCEDURE, re.compile(
55
+ r"\b(?:workflow|process|routine|setup|to (?:deploy|build|test|run|install|configure))\b", re.I
56
+ )),
57
+ (MemoryType.OPINION, re.compile(r"\b(?:i think|i believe|in my opinion|in my view)\b", re.I)),
58
+ (MemoryType.FACT, re.compile(
59
+ r"\b(?:work(?:ing)? at|live in|study at|teach at|my (?:name|role|job|company)|i use)\b", re.I
60
+ )),
61
+ (MemoryType.TEMPORAL, re.compile(
62
+ r"\b(?:today|tomorrow|this (?:week|month|sprint)|next (?:week|month)|right now)\b", re.I
63
+ )),
64
+ )
65
+
66
+ _FILLER = re.compile(
67
+ r"^(?:ok(?:ay)?|sure|yes|no|thanks|thank you|nice|lol|continue|got it|understood|"
68
+ r"hi|hello|hey|hmm|huh)[.!?]*$",
69
+ re.I,
70
+ )
71
+ _CASUAL = re.compile(r"^(?:the )?weather (?:looks|is) (?:good|nice|great)[.!?]*$", re.I)
72
+ _PERSONAL_SIGNAL = re.compile(r"\b(?:i|i'm|i've|i'll|i'd|my|me|user|the user)\b", re.I)
73
+ _UNCERTAIN = re.compile(r"\b(?:maybe|might|may|perhaps|possibly|i guess|not sure)\b", re.I)
74
+ _HISTORICAL = re.compile(r"\b(?:used to|previously|in the past|before)\b", re.I)
75
+ _CHANGE = re.compile(r"\b(?:stopped|no longer|not anymore|anymore|switching from|used to)\b", re.I)
76
+ _FUTURE = re.compile(
77
+ r"\b(?:plan(?:ning)? to|going to|will|might|may|next (?:week|month|year)|later|someday)\b",
78
+ re.I,
79
+ )
80
+ _CURRENT = re.compile(r"\b(?:now|currently|right now|recently|today)\b", re.I)
81
+ _NEGATION = re.compile(r"\b(?:don't|do not|doesn't|does not|never|no longer|stopped)\b", re.I)
82
+
83
+
84
+ def _normalize_input(text: str) -> str:
85
+ text = text.replace("\r\n", "\n").replace("\r", "\n")
86
+ text = text.replace("’", "'").replace("“", '"').replace("”", '"')
87
+ return re.sub(r"[ \t]+", " ", text).strip()
88
+
89
+
90
+ def _locate(raw_text: str, evidence: str) -> tuple[int | None, int | None]:
91
+ start = raw_text.casefold().find(evidence.casefold())
92
+ return (start, start + len(evidence)) if start >= 0 else (None, None)
93
+
94
+
95
+ def _expand_switch_statement(sentence: str, raw_text: str) -> list[_Clause] | None:
96
+ match = re.fullmatch(
97
+ r"i(?:'m| am) switching from (?P<old>.+?) to (?P<new>.+?)[.!?]?",
98
+ sentence.strip(),
99
+ re.I,
100
+ )
101
+ if not match:
102
+ return None
103
+ start, end = _locate(raw_text, sentence)
104
+ return [
105
+ _Clause(f"I used to focus on {match.group('old')}", sentence, start, end),
106
+ _Clause(f"I am switching to {match.group('new')}", sentence, start, end),
107
+ ]
108
+
109
+
110
+ def _candidate_clauses(text: str) -> list[_Clause]:
111
+ clauses: list[_Clause] = []
112
+ sentences = re.split(r"(?<=[.!?])\s+|\n+", text)
113
+ for sentence in sentences:
114
+ sentence = sentence.strip()
115
+ if not sentence:
116
+ continue
117
+ expanded = _expand_switch_statement(sentence, text)
118
+ if expanded is not None:
119
+ clauses.extend(expanded)
120
+ continue
121
+
122
+ contrast_parts = re.split(r"\s*,?\s*\b(?:but|however)\b\s*", sentence, flags=re.I)
123
+ for part in contrast_parts:
124
+ atomic_parts = re.split(
125
+ r"\s*;\s*|\s+\band\b\s+(?=(?:i\b|i'm\b|i am\b|i've\b|i'll\b|i'd\b|my\b))",
126
+ part,
127
+ flags=re.I,
128
+ )
129
+ for atomic in atomic_parts:
130
+ atomic = atomic.strip(" ,")
131
+ if not atomic:
132
+ continue
133
+ if len(atomic) > MAX_CLAUSE_CHARS:
134
+ atomic = atomic[:MAX_CLAUSE_CHARS].rstrip()
135
+ start, end = _locate(text, atomic)
136
+ clauses.append(_Clause(atomic, atomic, start, end))
137
+ return clauses
138
+
139
+
140
+ def _is_memory_worthy(text: str, suggested_type: MemoryType | None) -> bool:
141
+ stripped = text.strip()
142
+ if not stripped or _FILLER.fullmatch(stripped) or _CASUAL.fullmatch(stripped):
143
+ return False
144
+ if len(stripped.strip(".!? ")) < 4:
145
+ return False
146
+ return suggested_type is not None or bool(_PERSONAL_SIGNAL.search(stripped))
147
+
148
+
149
+ def _classify(text: str, suggested_type: MemoryType | None) -> MemoryType:
150
+ if suggested_type is not None:
151
+ return suggested_type
152
+ for memory_type, pattern in _TYPE_RULES:
153
+ if pattern.search(text):
154
+ return memory_type
155
+ return MemoryType.CONTEXT
156
+
157
+
158
+ def _temporal_analysis(text: str) -> tuple[CandidateTemporalStatus, str | None, CandidateAction]:
159
+ hint_match: re.Match[str] | None
160
+ if (hint_match := _HISTORICAL.search(text)) is not None:
161
+ return CandidateTemporalStatus.HISTORICAL, hint_match.group(0), CandidateAction.SUPERSEDE
162
+ if (hint_match := _CHANGE.search(text)) is not None:
163
+ return CandidateTemporalStatus.HISTORICAL, hint_match.group(0), CandidateAction.SUPERSEDE
164
+ if (hint_match := _FUTURE.search(text)) is not None:
165
+ return CandidateTemporalStatus.FUTURE, hint_match.group(0), CandidateAction.ADD
166
+ if (hint_match := _CURRENT.search(text)) is not None:
167
+ return CandidateTemporalStatus.CURRENT, hint_match.group(0), CandidateAction.ADD
168
+ return CandidateTemporalStatus.CURRENT, None, CandidateAction.ADD
169
+
170
+
171
+ def _canonicalize(text: str) -> str:
172
+ value = text.strip().rstrip(".!?").strip()
173
+ value = re.sub(r"^also\s+", "", value, flags=re.I)
174
+ value = re.sub(r"^i also\s+", "I ", value, flags=re.I)
175
+ value = re.sub(
176
+ r"^keep (?:your )?answers (?:short|concise)$",
177
+ "User prefers concise answers",
178
+ value,
179
+ flags=re.I,
180
+ )
181
+ value = re.sub(r"^maybe\s+i(?:'ll| will)\s+", "User may ", value, flags=re.I)
182
+ value = re.sub(r"^now\s+i(?:'m| am)\s+", "User is now ", value, flags=re.I)
183
+ replacements = (
184
+ (r"^i don't\s+", "User does not "),
185
+ (r"^i do not\s+", "User does not "),
186
+ (r"^i prefer\s+", "User prefers "),
187
+ (r"^i use\s+", "User uses "),
188
+ (r"^i want\s+", "User wants "),
189
+ (r"^i plan\s+", "User plans "),
190
+ (r"^i like\s+", "User likes "),
191
+ (r"^i work\s+", "User works "),
192
+ (r"^i think\s+", "User thinks "),
193
+ (r"^i(?:'m| am)\s+", "User is "),
194
+ (r"^i(?:'ve| have)\s+", "User has "),
195
+ (r"^i(?:'ll| will)\s+", "User will "),
196
+ (r"^i(?:'d| would)\s+", "User would "),
197
+ (r"^i\s+", "User "),
198
+ (r"^my\s+", "User's "),
199
+ )
200
+ for pattern, replacement in replacements:
201
+ updated = re.sub(pattern, replacement, value, count=1, flags=re.I)
202
+ if updated != value:
203
+ value = updated
204
+ break
205
+ value = re.sub(r"\bshort answers\b", "concise answers", value, flags=re.I)
206
+ return re.sub(r"\s+", " ", value).strip()
207
+
208
+
209
+ def _confidence(text: str) -> float:
210
+ score = 0.88
211
+ if _UNCERTAIN.search(text):
212
+ score -= 0.32
213
+ if re.search(r"\b(?:think|guess|could)\b", text, re.I):
214
+ score -= 0.12
215
+ if re.search(r"\b(?:always|never|definitely|absolutely|certainly)\b", text, re.I):
216
+ score += 0.07
217
+ return round(min(1.0, max(0.0, score)), 2)
218
+
219
+
220
+ def _importance(memory_type: MemoryType, text: str) -> float:
221
+ scores = {
222
+ MemoryType.GOAL: 0.85,
223
+ MemoryType.PROJECT: 0.85,
224
+ MemoryType.PREFERENCE: 0.8,
225
+ MemoryType.PROCEDURE: 0.75,
226
+ MemoryType.SKILL: 0.7,
227
+ MemoryType.FACT: 0.65,
228
+ MemoryType.RELATIONSHIP: 0.65,
229
+ MemoryType.OPINION: 0.55,
230
+ MemoryType.CONTEXT: 0.5,
231
+ MemoryType.TEMPORAL: 0.4,
232
+ }
233
+ score = scores[memory_type]
234
+ if re.search(r"\b(?:today|tomorrow|this week|right now)\b", text, re.I):
235
+ score -= 0.15
236
+ return round(min(1.0, max(0.0, score)), 2)
237
+
238
+
239
+ def _dedup_key(candidate: CandidateMemory) -> str:
240
+ return re.sub(r"[^a-z0-9]+", " ", candidate.content.casefold()).strip()
241
+
242
+
243
+ class RuleBasedMemoryExtractor:
244
+ """Side-effect-free deterministic baseline implementing MemoryExtractor."""
245
+
246
+ def __init__(self, *, min_confidence: float = 0.3, max_candidates: int = MAX_CANDIDATES) -> None:
247
+ self._min_confidence = min_confidence
248
+ self._max_candidates = max_candidates
249
+
250
+ async def extract(
251
+ self,
252
+ text: str,
253
+ source_type: str = "cli_input",
254
+ source_uri: str | None = None,
255
+ suggested_type: MemoryType | None = None,
256
+ tags: list[str] | None = None,
257
+ source_role: SourceRole = SourceRole.USER,
258
+ confirmed_user_information: bool = False,
259
+ ) -> list[CandidateMemory]:
260
+ if not text or not text.strip():
261
+ return []
262
+ if source_role != SourceRole.USER and not confirmed_user_information:
263
+ return []
264
+
265
+ normalized = _normalize_input(text[:MAX_INPUT_CHARS])
266
+ input_hash = hashlib.sha256(normalized.encode("utf-8")).hexdigest()
267
+ candidates: list[CandidateMemory] = []
268
+ seen: set[str] = set()
269
+
270
+ for clause in _candidate_clauses(normalized):
271
+ if len(candidates) >= self._max_candidates:
272
+ break
273
+ if not _is_memory_worthy(clause.text, suggested_type):
274
+ continue
275
+ memory_type = _classify(clause.text, suggested_type)
276
+ temporal_status, temporal_hint, action_hint = _temporal_analysis(clause.text)
277
+ confidence = _confidence(clause.text)
278
+ if confidence < self._min_confidence:
279
+ continue
280
+ candidate = CandidateMemory(
281
+ content=_canonicalize(clause.text),
282
+ memory_type=memory_type,
283
+ confidence=confidence,
284
+ importance=_importance(memory_type, clause.text),
285
+ temporal_status=temporal_status,
286
+ temporal_hint=temporal_hint,
287
+ action_hint=action_hint,
288
+ uncertain=bool(_UNCERTAIN.search(clause.text)),
289
+ negated=bool(_NEGATION.search(clause.text)),
290
+ source_type=source_type,
291
+ source_uri=source_uri,
292
+ source_role=source_role,
293
+ evidence=clause.evidence,
294
+ evidence_start=clause.start,
295
+ evidence_end=clause.end,
296
+ tags=list(tags or []),
297
+ metadata={
298
+ "input_hash": input_hash,
299
+ "negated": bool(_NEGATION.search(clause.text)),
300
+ "confirmed_user_information": confirmed_user_information,
301
+ "input_truncated": len(text) > MAX_INPUT_CHARS,
302
+ "extractor": "deterministic-v2",
303
+ },
304
+ )
305
+ key = _dedup_key(candidate)
306
+ if key in seen:
307
+ continue
308
+ seen.add(key)
309
+ candidates.append(candidate)
310
+
311
+ return candidates