contextos-memory-runtime 1.0.0rc2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. contextos/__init__.py +3 -0
  2. contextos/__main__.py +6 -0
  3. contextos/api/__init__.py +1 -0
  4. contextos/api/routes/__init__.py +1 -0
  5. contextos/api/routes/desktop.py +322 -0
  6. contextos/api/routes/ingest.py +17 -0
  7. contextos/api/routes/memories.py +84 -0
  8. contextos/api/routes/models.py +81 -0
  9. contextos/api/routes/retrieval.py +89 -0
  10. contextos/api/routes/system.py +216 -0
  11. contextos/api/server.py +195 -0
  12. contextos/benchmarks/__init__.py +1 -0
  13. contextos/benchmarks/compilation.py +245 -0
  14. contextos/benchmarks/connectors.py +423 -0
  15. contextos/benchmarks/explainability.py +103 -0
  16. contextos/benchmarks/final.py +406 -0
  17. contextos/benchmarks/graph.py +310 -0
  18. contextos/benchmarks/graph_adversarial.py +525 -0
  19. contextos/benchmarks/mcp.py +324 -0
  20. contextos/benchmarks/model_routing.py +203 -0
  21. contextos/benchmarks/optimization.py +305 -0
  22. contextos/benchmarks/rescue_integration.py +127 -0
  23. contextos/benchmarks/retrieval.py +266 -0
  24. contextos/benchmarks/temporal.py +377 -0
  25. contextos/benchmarks/temporal_hotpath.py +76 -0
  26. contextos/benchmarks/terminal.py +62 -0
  27. contextos/cli/__init__.py +1 -0
  28. contextos/cli/app.py +932 -0
  29. contextos/cli/dashboard.py +174 -0
  30. contextos/cli/formatters.py +299 -0
  31. contextos/config/__init__.py +1 -0
  32. contextos/config/settings.py +160 -0
  33. contextos/connectors/__init__.py +6 -0
  34. contextos/connectors/fake.py +11 -0
  35. contextos/connectors/json_import.py +125 -0
  36. contextos/connectors/local_files.py +102 -0
  37. contextos/connectors/manager.py +293 -0
  38. contextos/connectors/models.py +62 -0
  39. contextos/connectors/protocols.py +11 -0
  40. contextos/core/__init__.py +103 -0
  41. contextos/core/enums.py +489 -0
  42. contextos/core/exceptions.py +293 -0
  43. contextos/core/models.py +1147 -0
  44. contextos/core/protocols.py +549 -0
  45. contextos/daemon/__init__.py +1 -0
  46. contextos/daemon/manager.py +510 -0
  47. contextos/daemon/state.py +127 -0
  48. contextos/daemon/wiring.py +296 -0
  49. contextos/demo.py +217 -0
  50. contextos/embedding/__init__.py +1 -0
  51. contextos/embedding/deterministic.py +76 -0
  52. contextos/embedding/sentence_transformers.py +80 -0
  53. contextos/mcp/__init__.py +5 -0
  54. contextos/mcp/server.py +269 -0
  55. contextos/providers/__init__.py +13 -0
  56. contextos/providers/fake.py +217 -0
  57. contextos/providers/ollama.py +297 -0
  58. contextos/providers/openai_compatible.py +337 -0
  59. contextos/services/__init__.py +1 -0
  60. contextos/services/compilation.py +535 -0
  61. contextos/services/explainability.py +553 -0
  62. contextos/services/extraction.py +311 -0
  63. contextos/services/graph.py +524 -0
  64. contextos/services/graph_retrieval.py +143 -0
  65. contextos/services/ingestion.py +143 -0
  66. contextos/services/inspection.py +174 -0
  67. contextos/services/memory.py +291 -0
  68. contextos/services/model_service.py +409 -0
  69. contextos/services/optimization.py +426 -0
  70. contextos/services/privacy.py +331 -0
  71. contextos/services/retrieval.py +302 -0
  72. contextos/services/retrieval_index.py +88 -0
  73. contextos/services/router.py +302 -0
  74. contextos/services/secret_scanner.py +207 -0
  75. contextos/services/telemetry_query.py +102 -0
  76. contextos/services/temporal.py +500 -0
  77. contextos/services/token_counter.py +222 -0
  78. contextos/storage/__init__.py +1 -0
  79. contextos/storage/connector_repo.py +67 -0
  80. contextos/storage/database.py +497 -0
  81. contextos/storage/event_repo.py +137 -0
  82. contextos/storage/graph_repo.py +228 -0
  83. contextos/storage/lexical/__init__.py +1 -0
  84. contextos/storage/lexical/bm25.py +134 -0
  85. contextos/storage/memory_repo.py +589 -0
  86. contextos/storage/relation_repo.py +80 -0
  87. contextos/storage/telemetry_repo.py +481 -0
  88. contextos/storage/vector/__init__.py +1 -0
  89. contextos/storage/vector/in_memory.py +162 -0
  90. contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
  91. contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
  92. contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
  93. contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,426 @@
1
+ """Whole-memory token-aware selection after retrieval."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ import time
7
+ from dataclasses import dataclass
8
+ from uuid import UUID
9
+
10
+ from contextos.core.enums import (
11
+ ExclusionReason,
12
+ MemoryStatus,
13
+ OptimizationStrategy,
14
+ PrivacyLevel,
15
+ )
16
+ from contextos.core.models import (
17
+ CandidateDecision,
18
+ ContextBudget,
19
+ OptimizationTrace,
20
+ ScoredMemory,
21
+ SelectionResult,
22
+ )
23
+ from contextos.core.protocols import TokenCounter
24
+
25
+
26
+ _STOP_WORDS = {
27
+ "a", "an", "and", "are", "as", "at", "be", "because", "for", "from",
28
+ "has", "in", "is", "it", "of", "on", "or", "that", "the", "to", "user",
29
+ "was", "with",
30
+ }
31
+ _CANONICAL = {
32
+ "answers": "response",
33
+ "answer": "response",
34
+ "brief": "concise",
35
+ "short": "concise",
36
+ "responses": "response",
37
+ "explanation": "response",
38
+ "explanations": "response",
39
+ "likes": "prefer",
40
+ "prefers": "prefer",
41
+ "uses": "use",
42
+ "used": "use",
43
+ "using": "use",
44
+ }
45
+ _VALID_STATUSES = {
46
+ MemoryStatus.ACTIVE,
47
+ MemoryStatus.HISTORICAL,
48
+ MemoryStatus.SUPERSEDED,
49
+ MemoryStatus.CONTRADICTED,
50
+ }
51
+ _LIFECYCLE_FACTOR = {
52
+ MemoryStatus.ACTIVE: 1.0,
53
+ MemoryStatus.HISTORICAL: 0.95,
54
+ MemoryStatus.SUPERSEDED: 0.90,
55
+ MemoryStatus.CONTRADICTED: 0.80,
56
+ }
57
+
58
+
59
+ def information_tokens(text: str) -> frozenset[str]:
60
+ """Extract deterministic concept tokens for overlap and novelty."""
61
+ values: set[str] = set()
62
+ for raw in re.findall(r"[a-z0-9]+(?:[+#._-][a-z0-9]+)*", text.casefold()):
63
+ if raw in _STOP_WORDS:
64
+ continue
65
+ values.add(_CANONICAL.get(raw, raw))
66
+ return frozenset(values)
67
+
68
+
69
+ def jaccard_similarity(left: frozenset[str], right: frozenset[str]) -> float:
70
+ if not left and not right:
71
+ return 1.0
72
+ union = left | right
73
+ return len(left & right) / len(union) if union else 0.0
74
+
75
+
76
+ def redundancy_similarity(left: frozenset[str], right: frozenset[str]) -> float:
77
+ """Combine Jaccard with containment to catch concise paraphrase subsets."""
78
+ if not left and not right:
79
+ return 1.0
80
+ if not left or not right:
81
+ return 0.0
82
+ containment = len(left & right) / min(len(left), len(right))
83
+ return max(jaccard_similarity(left, right), containment)
84
+
85
+
86
+ @dataclass
87
+ class _Candidate:
88
+ scored: ScoredMemory
89
+ index: int
90
+ content_tokens: int
91
+ cost: int
92
+ concepts: frozenset[str]
93
+ normalized_relevance: float = 0.0
94
+ importance_contribution: float = 0.0
95
+ confidence_contribution: float = 0.0
96
+ support_contribution: float = 0.0
97
+ lifecycle_multiplier: float = 0.0
98
+ base_utility: float = 0.0
99
+ marginal_utility: float = 0.0
100
+ redundancy: float = 0.0
101
+ selected: bool = False
102
+ exclusion_reason: ExclusionReason | None = None
103
+ redundant_with: UUID | None = None
104
+
105
+
106
+ class MemoryContextOptimizer:
107
+ """Select whole memories using bounded utility, novelty, and token cost."""
108
+
109
+ redundancy_threshold = 0.60
110
+ minimum_relative_relevance = 0.15
111
+
112
+ def __init__(self, *, token_counter: TokenCounter) -> None:
113
+ self._token_counter = token_counter
114
+
115
+ def optimize(
116
+ self,
117
+ query: str,
118
+ candidates: list[ScoredMemory],
119
+ budget: ContextBudget,
120
+ strategy: OptimizationStrategy = OptimizationStrategy.CONTEXTOS,
121
+ ) -> SelectionResult:
122
+ del query # Phase 4 scores already encode query relevance.
123
+ started = time.perf_counter()
124
+ prepared = self._prepare(candidates, budget)
125
+ eligible = [
126
+ candidate for candidate in prepared
127
+ if candidate.exclusion_reason is None
128
+ ]
129
+
130
+ if strategy == OptimizationStrategy.TOP_RANK_STOP:
131
+ selected = self._top_rank(eligible, budget.available_tokens)
132
+ elif strategy == OptimizationStrategy.TOP_RANK_SKIP:
133
+ selected = self._top_rank_skip(eligible, budget.available_tokens)
134
+ elif strategy == OptimizationStrategy.GREEDY:
135
+ selected = self._greedy(eligible, budget.available_tokens)
136
+ else:
137
+ selected = self._contextos(eligible, budget.available_tokens)
138
+
139
+ selected.sort(key=self._output_order)
140
+ content_tokens = sum(candidate.content_tokens for candidate in selected)
141
+ overhead_tokens = len(selected) * budget.overhead_per_memory
142
+ total_tokens = content_tokens + overhead_tokens
143
+ remaining = budget.available_tokens - total_tokens
144
+ rescue_candidates = self._compiler_rescue_candidates(prepared, selected)
145
+ decisions = [self._decision(candidate, budget) for candidate in prepared]
146
+ latency_ms = (time.perf_counter() - started) * 1000
147
+ utilization = (
148
+ total_tokens / budget.available_tokens
149
+ if budget.available_tokens
150
+ else 0.0
151
+ )
152
+ trace = OptimizationTrace(
153
+ strategy=strategy,
154
+ candidate_count=len(candidates),
155
+ eligible_count=len(eligible),
156
+ selected_count=len(selected),
157
+ compiler_rescue_candidate_count=len(rescue_candidates),
158
+ budget_tokens=budget.max_tokens,
159
+ reserved_tokens=budget.reserved_tokens,
160
+ available_tokens=budget.available_tokens,
161
+ tokens_used=total_tokens,
162
+ remaining_tokens=remaining,
163
+ latency_ms=latency_ms,
164
+ decisions=decisions,
165
+ )
166
+ return SelectionResult(
167
+ strategy=strategy,
168
+ selected_memories=[candidate.scored for candidate in selected],
169
+ compiler_rescue_candidates=[
170
+ candidate.scored for candidate in rescue_candidates
171
+ ],
172
+ total_tokens=total_tokens,
173
+ content_tokens=content_tokens,
174
+ overhead_tokens=overhead_tokens,
175
+ budget=budget,
176
+ remaining_tokens=remaining,
177
+ utilization=utilization,
178
+ trace=trace,
179
+ )
180
+
181
+ def _compiler_rescue_candidates(
182
+ self,
183
+ candidates: list[_Candidate],
184
+ selected: list[_Candidate],
185
+ ) -> list[_Candidate]:
186
+ """Expose only eligible whole-memory rejects for extractive compilation."""
187
+ rescued: list[_Candidate] = []
188
+ for candidate in candidates:
189
+ if candidate.exclusion_reason != ExclusionReason.OVERSIZED:
190
+ continue
191
+ if candidate.normalized_relevance < self.minimum_relative_relevance:
192
+ continue
193
+ if candidate.scored.memory.status not in _VALID_STATUSES:
194
+ continue
195
+ if candidate.scored.memory.privacy_level == PrivacyLevel.RESTRICTED:
196
+ continue
197
+ redundancy, _ = self._maximum_redundancy(candidate, selected)
198
+ if redundancy >= self.redundancy_threshold:
199
+ continue
200
+ rescued.append(candidate)
201
+ return sorted(rescued, key=self._input_order)
202
+
203
+ def _prepare(
204
+ self, candidates: list[ScoredMemory], budget: ContextBudget
205
+ ) -> list[_Candidate]:
206
+ prepared = [
207
+ _Candidate(
208
+ scored=scored,
209
+ index=index,
210
+ content_tokens=self._token_counter.count(scored.memory.content),
211
+ cost=self._token_counter.count(scored.memory.content)
212
+ + budget.overhead_per_memory,
213
+ concepts=information_tokens(scored.memory.content),
214
+ )
215
+ for index, scored in enumerate(candidates)
216
+ ]
217
+ best_by_id: dict[UUID, _Candidate] = {}
218
+ for candidate in prepared:
219
+ existing = best_by_id.get(candidate.scored.memory.id)
220
+ if existing is None or self._input_order(candidate) < self._input_order(existing):
221
+ if existing is not None:
222
+ existing.exclusion_reason = ExclusionReason.DUPLICATE_ID
223
+ best_by_id[candidate.scored.memory.id] = candidate
224
+ else:
225
+ candidate.exclusion_reason = ExclusionReason.DUPLICATE_ID
226
+
227
+ eligible = [
228
+ candidate
229
+ for candidate in prepared
230
+ if candidate.exclusion_reason is None
231
+ and candidate.scored.memory.status in _VALID_STATUSES
232
+ ]
233
+ maximum_score = max(
234
+ (candidate.scored.final_score for candidate in eligible),
235
+ default=0.0,
236
+ )
237
+ for candidate in prepared:
238
+ status = candidate.scored.memory.status
239
+ if candidate.exclusion_reason is not None:
240
+ continue
241
+ if status not in _VALID_STATUSES:
242
+ candidate.exclusion_reason = ExclusionReason.INVALID_LIFECYCLE
243
+ continue
244
+ candidate.normalized_relevance = (
245
+ candidate.scored.final_score / maximum_score if maximum_score else 0.0
246
+ )
247
+ candidate.importance_contribution = 0.10 * candidate.scored.memory.importance
248
+ candidate.confidence_contribution = 0.10 * candidate.scored.memory.confidence
249
+ source_count = len(set(candidate.scored.retrieval_sources))
250
+ candidate.support_contribution = 0.05 * min(source_count, 2) / 2
251
+ candidate.lifecycle_multiplier = _LIFECYCLE_FACTOR[status]
252
+ utility = (
253
+ 0.75 * candidate.normalized_relevance
254
+ + candidate.importance_contribution
255
+ + candidate.confidence_contribution
256
+ + candidate.support_contribution
257
+ )
258
+ candidate.base_utility = min(1.0, utility) * candidate.lifecycle_multiplier
259
+ return prepared
260
+
261
+ @staticmethod
262
+ def _input_order(candidate: _Candidate) -> tuple[float, float, str, int]:
263
+ rank = candidate.scored.rank if candidate.scored.rank > 0 else float("inf")
264
+ return (
265
+ rank,
266
+ -candidate.scored.final_score,
267
+ str(candidate.scored.memory.id),
268
+ candidate.index,
269
+ )
270
+
271
+ @staticmethod
272
+ def _output_order(candidate: _Candidate) -> tuple[float, float, str]:
273
+ rank = candidate.scored.rank if candidate.scored.rank > 0 else float("inf")
274
+ return (rank, -candidate.scored.final_score, str(candidate.scored.memory.id))
275
+
276
+ def _top_rank(
277
+ self, candidates: list[_Candidate], available: int
278
+ ) -> list[_Candidate]:
279
+ selected: list[_Candidate] = []
280
+ used = 0
281
+ stopped = False
282
+ for candidate in sorted(candidates, key=self._input_order):
283
+ if stopped:
284
+ candidate.exclusion_reason = self._budget_reason(candidate, available)
285
+ continue
286
+ if used + candidate.cost > available:
287
+ candidate.exclusion_reason = self._budget_reason(candidate, available)
288
+ stopped = True
289
+ continue
290
+ candidate.selected = True
291
+ candidate.marginal_utility = candidate.base_utility
292
+ selected.append(candidate)
293
+ used += candidate.cost
294
+ return selected
295
+
296
+ def _greedy(
297
+ self, candidates: list[_Candidate], available: int
298
+ ) -> list[_Candidate]:
299
+ selected: list[_Candidate] = []
300
+ used = 0
301
+ ranked = sorted(
302
+ candidates,
303
+ key=lambda candidate: (
304
+ -(candidate.base_utility / max(candidate.cost, 1)),
305
+ *self._input_order(candidate),
306
+ ),
307
+ )
308
+ for candidate in ranked:
309
+ candidate.marginal_utility = candidate.base_utility
310
+ if used + candidate.cost <= available:
311
+ candidate.selected = True
312
+ selected.append(candidate)
313
+ used += candidate.cost
314
+ else:
315
+ candidate.exclusion_reason = self._budget_reason(candidate, available)
316
+ return selected
317
+
318
+ def _top_rank_skip(
319
+ self, candidates: list[_Candidate], available: int
320
+ ) -> list[_Candidate]:
321
+ """Select in retrieval order, skipping non-fitting candidates."""
322
+ selected: list[_Candidate] = []
323
+ used = 0
324
+ for candidate in sorted(candidates, key=self._input_order):
325
+ candidate.marginal_utility = candidate.base_utility
326
+ if used + candidate.cost <= available:
327
+ candidate.selected = True
328
+ selected.append(candidate)
329
+ used += candidate.cost
330
+ else:
331
+ candidate.exclusion_reason = self._budget_reason(candidate, available)
332
+ return selected
333
+
334
+ def _contextos(
335
+ self, candidates: list[_Candidate], available: int
336
+ ) -> list[_Candidate]:
337
+ selected: list[_Candidate] = []
338
+ remaining = list(candidates)
339
+ used = 0
340
+ covered: set[str] = set()
341
+ while remaining:
342
+ viable: list[tuple[float, _Candidate]] = []
343
+ for candidate in list(remaining):
344
+ if candidate.normalized_relevance < self.minimum_relative_relevance:
345
+ candidate.exclusion_reason = ExclusionReason.LOW_RELEVANCE
346
+ remaining.remove(candidate)
347
+ continue
348
+ redundancy, redundant_with = self._maximum_redundancy(candidate, selected)
349
+ candidate.redundancy = redundancy
350
+ candidate.redundant_with = redundant_with
351
+ if redundancy >= self.redundancy_threshold:
352
+ candidate.exclusion_reason = ExclusionReason.REDUNDANT
353
+ remaining.remove(candidate)
354
+ continue
355
+ novelty = (
356
+ len(candidate.concepts - covered) / len(candidate.concepts)
357
+ if candidate.concepts
358
+ else 0.0
359
+ )
360
+ candidate.marginal_utility = (
361
+ candidate.base_utility
362
+ * (0.65 + 0.35 * novelty)
363
+ * (1.0 - 0.70 * redundancy)
364
+ )
365
+ density = candidate.marginal_utility / max(candidate.cost, 1)
366
+ viable.append((density, candidate))
367
+ if not viable:
368
+ break
369
+ viable.sort(
370
+ key=lambda item: (
371
+ -item[0],
372
+ *self._input_order(item[1]),
373
+ )
374
+ )
375
+ candidate = viable[0][1]
376
+ remaining.remove(candidate)
377
+ if used + candidate.cost > available:
378
+ candidate.exclusion_reason = self._budget_reason(candidate, available)
379
+ continue
380
+ candidate.selected = True
381
+ selected.append(candidate)
382
+ used += candidate.cost
383
+ covered.update(candidate.concepts)
384
+ return selected
385
+
386
+ @staticmethod
387
+ def _maximum_redundancy(
388
+ candidate: _Candidate, selected: list[_Candidate]
389
+ ) -> tuple[float, UUID | None]:
390
+ best = 0.0
391
+ matched: UUID | None = None
392
+ for existing in selected:
393
+ similarity = redundancy_similarity(candidate.concepts, existing.concepts)
394
+ if similarity > best:
395
+ best = similarity
396
+ matched = existing.scored.memory.id
397
+ return best, matched
398
+
399
+ @staticmethod
400
+ def _budget_reason(candidate: _Candidate, available: int) -> ExclusionReason:
401
+ return (
402
+ ExclusionReason.OVERSIZED
403
+ if candidate.cost > available
404
+ else ExclusionReason.BUDGET_EXHAUSTED
405
+ )
406
+
407
+ @staticmethod
408
+ def _decision(candidate: _Candidate, budget: ContextBudget) -> CandidateDecision:
409
+ return CandidateDecision(
410
+ memory_id=candidate.scored.memory.id,
411
+ token_cost=candidate.cost,
412
+ content_tokens=candidate.content_tokens,
413
+ overhead_tokens=budget.overhead_per_memory,
414
+ retrieval_score=candidate.scored.final_score,
415
+ normalized_relevance=candidate.normalized_relevance,
416
+ importance_contribution=candidate.importance_contribution,
417
+ confidence_contribution=candidate.confidence_contribution,
418
+ support_contribution=candidate.support_contribution,
419
+ lifecycle_multiplier=candidate.lifecycle_multiplier,
420
+ base_utility=candidate.base_utility,
421
+ marginal_utility=candidate.marginal_utility,
422
+ redundancy=candidate.redundancy,
423
+ selected=candidate.selected,
424
+ exclusion_reason=candidate.exclusion_reason,
425
+ redundant_with=candidate.redundant_with,
426
+ )