contextos-memory-runtime 1.0.0rc2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contextos/__init__.py +3 -0
- contextos/__main__.py +6 -0
- contextos/api/__init__.py +1 -0
- contextos/api/routes/__init__.py +1 -0
- contextos/api/routes/desktop.py +322 -0
- contextos/api/routes/ingest.py +17 -0
- contextos/api/routes/memories.py +84 -0
- contextos/api/routes/models.py +81 -0
- contextos/api/routes/retrieval.py +89 -0
- contextos/api/routes/system.py +216 -0
- contextos/api/server.py +195 -0
- contextos/benchmarks/__init__.py +1 -0
- contextos/benchmarks/compilation.py +245 -0
- contextos/benchmarks/connectors.py +423 -0
- contextos/benchmarks/explainability.py +103 -0
- contextos/benchmarks/final.py +406 -0
- contextos/benchmarks/graph.py +310 -0
- contextos/benchmarks/graph_adversarial.py +525 -0
- contextos/benchmarks/mcp.py +324 -0
- contextos/benchmarks/model_routing.py +203 -0
- contextos/benchmarks/optimization.py +305 -0
- contextos/benchmarks/rescue_integration.py +127 -0
- contextos/benchmarks/retrieval.py +266 -0
- contextos/benchmarks/temporal.py +377 -0
- contextos/benchmarks/temporal_hotpath.py +76 -0
- contextos/benchmarks/terminal.py +62 -0
- contextos/cli/__init__.py +1 -0
- contextos/cli/app.py +932 -0
- contextos/cli/dashboard.py +174 -0
- contextos/cli/formatters.py +299 -0
- contextos/config/__init__.py +1 -0
- contextos/config/settings.py +160 -0
- contextos/connectors/__init__.py +6 -0
- contextos/connectors/fake.py +11 -0
- contextos/connectors/json_import.py +125 -0
- contextos/connectors/local_files.py +102 -0
- contextos/connectors/manager.py +293 -0
- contextos/connectors/models.py +62 -0
- contextos/connectors/protocols.py +11 -0
- contextos/core/__init__.py +103 -0
- contextos/core/enums.py +489 -0
- contextos/core/exceptions.py +293 -0
- contextos/core/models.py +1147 -0
- contextos/core/protocols.py +549 -0
- contextos/daemon/__init__.py +1 -0
- contextos/daemon/manager.py +510 -0
- contextos/daemon/state.py +127 -0
- contextos/daemon/wiring.py +296 -0
- contextos/demo.py +217 -0
- contextos/embedding/__init__.py +1 -0
- contextos/embedding/deterministic.py +76 -0
- contextos/embedding/sentence_transformers.py +80 -0
- contextos/mcp/__init__.py +5 -0
- contextos/mcp/server.py +269 -0
- contextos/providers/__init__.py +13 -0
- contextos/providers/fake.py +217 -0
- contextos/providers/ollama.py +297 -0
- contextos/providers/openai_compatible.py +337 -0
- contextos/services/__init__.py +1 -0
- contextos/services/compilation.py +535 -0
- contextos/services/explainability.py +553 -0
- contextos/services/extraction.py +311 -0
- contextos/services/graph.py +524 -0
- contextos/services/graph_retrieval.py +143 -0
- contextos/services/ingestion.py +143 -0
- contextos/services/inspection.py +174 -0
- contextos/services/memory.py +291 -0
- contextos/services/model_service.py +409 -0
- contextos/services/optimization.py +426 -0
- contextos/services/privacy.py +331 -0
- contextos/services/retrieval.py +302 -0
- contextos/services/retrieval_index.py +88 -0
- contextos/services/router.py +302 -0
- contextos/services/secret_scanner.py +207 -0
- contextos/services/telemetry_query.py +102 -0
- contextos/services/temporal.py +500 -0
- contextos/services/token_counter.py +222 -0
- contextos/storage/__init__.py +1 -0
- contextos/storage/connector_repo.py +67 -0
- contextos/storage/database.py +497 -0
- contextos/storage/event_repo.py +137 -0
- contextos/storage/graph_repo.py +228 -0
- contextos/storage/lexical/__init__.py +1 -0
- contextos/storage/lexical/bm25.py +134 -0
- contextos/storage/memory_repo.py +589 -0
- contextos/storage/relation_repo.py +80 -0
- contextos/storage/telemetry_repo.py +481 -0
- contextos/storage/vector/__init__.py +1 -0
- contextos/storage/vector/in_memory.py +162 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,305 @@
|
|
|
1
|
+
"""Offline Phase 5 token-selection evaluation and scale smoke test."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import time
|
|
6
|
+
from collections import Counter
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from uuid import UUID
|
|
9
|
+
|
|
10
|
+
from contextos.core.enums import MemoryStatus, MemoryType, OptimizationStrategy
|
|
11
|
+
from contextos.core.models import ContextBudget, Memory, ScoredMemory, SelectionResult
|
|
12
|
+
from contextos.services.optimization import (
|
|
13
|
+
MemoryContextOptimizer,
|
|
14
|
+
information_tokens,
|
|
15
|
+
redundancy_similarity,
|
|
16
|
+
)
|
|
17
|
+
from contextos.services.token_counter import DeterministicWordTokenCounter
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
SMOKE_QUERY = "Which local model should I use given my machine and previous attempts?"
|
|
21
|
+
INFORMATION_UNITS = {"runtime", "failed_large", "successful_small", "hardware_limit"}
|
|
22
|
+
UNIT_WEIGHTS = {
|
|
23
|
+
"runtime": 1.0,
|
|
24
|
+
"failed_large": 1.2,
|
|
25
|
+
"successful_small": 1.0,
|
|
26
|
+
"hardware_limit": 1.2,
|
|
27
|
+
}
|
|
28
|
+
STRATEGY_LABELS = {
|
|
29
|
+
OptimizationStrategy.TOP_RANK_STOP: "top_rank_stop",
|
|
30
|
+
OptimizationStrategy.TOP_RANK_SKIP: "top_rank_skip",
|
|
31
|
+
OptimizationStrategy.GREEDY: "greedy",
|
|
32
|
+
OptimizationStrategy.CONTEXTOS: "contextos",
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass(frozen=True)
|
|
37
|
+
class OptimizationMetrics:
|
|
38
|
+
tokens: int
|
|
39
|
+
utilization: float
|
|
40
|
+
coverage: float
|
|
41
|
+
weighted_coverage: float
|
|
42
|
+
relevance_retained: float
|
|
43
|
+
redundancy: float
|
|
44
|
+
efficiency: float
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def information_unit_coverage(
|
|
48
|
+
selected_ids: list[str],
|
|
49
|
+
annotations: dict[str, set[str]],
|
|
50
|
+
required_units: set[str],
|
|
51
|
+
) -> float:
|
|
52
|
+
if not required_units:
|
|
53
|
+
return 0.0
|
|
54
|
+
covered = set().union(*(annotations.get(identifier, set()) for identifier in selected_ids))
|
|
55
|
+
return len(covered & required_units) / len(required_units)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def weighted_information_unit_coverage(
|
|
59
|
+
selected_ids: list[str],
|
|
60
|
+
annotations: dict[str, set[str]],
|
|
61
|
+
unit_weights: dict[str, float],
|
|
62
|
+
) -> float:
|
|
63
|
+
total = sum(unit_weights.values())
|
|
64
|
+
if total <= 0:
|
|
65
|
+
return 0.0
|
|
66
|
+
covered = set().union(*(annotations.get(identifier, set()) for identifier in selected_ids))
|
|
67
|
+
return sum(weight for unit, weight in unit_weights.items() if unit in covered) / total
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def retained_relevance(
|
|
71
|
+
selected: list[ScoredMemory], candidates: list[ScoredMemory]
|
|
72
|
+
) -> float:
|
|
73
|
+
total = sum(item.final_score for item in candidates)
|
|
74
|
+
return sum(item.final_score for item in selected) / total if total else 0.0
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def redundancy_rate(selected: list[ScoredMemory]) -> float:
|
|
78
|
+
if len(selected) < 2:
|
|
79
|
+
return 0.0
|
|
80
|
+
similarities: list[float] = []
|
|
81
|
+
for index, left in enumerate(selected):
|
|
82
|
+
for right in selected[index + 1:]:
|
|
83
|
+
similarities.append(redundancy_similarity(
|
|
84
|
+
information_tokens(left.memory.content),
|
|
85
|
+
information_tokens(right.memory.content),
|
|
86
|
+
))
|
|
87
|
+
return sum(similarities) / len(similarities)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def budget_utilization(tokens: int, available_tokens: int) -> float:
|
|
91
|
+
return tokens / available_tokens if available_tokens > 0 else 0.0
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def coverage_efficiency(coverage: float, tokens: int) -> float:
|
|
95
|
+
return coverage / tokens if tokens > 0 else 0.0
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _padded(base: str, target: int, repeated_token: str) -> str:
|
|
99
|
+
counter = DeterministicWordTokenCounter()
|
|
100
|
+
current = counter.count(base)
|
|
101
|
+
if current > target:
|
|
102
|
+
raise ValueError("Base content already exceeds target")
|
|
103
|
+
return base + (" " + repeated_token) * (target - current)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def smoke_candidates() -> tuple[list[ScoredMemory], dict[str, set[str]]]:
|
|
107
|
+
specs = [
|
|
108
|
+
(1, _padded("User uses Ollama.", 12, "Ollama"), 0.95, {"runtime"}),
|
|
109
|
+
(
|
|
110
|
+
2,
|
|
111
|
+
_padded(
|
|
112
|
+
"Qwen 30B failed due to insufficient available memory.",
|
|
113
|
+
20,
|
|
114
|
+
"memory",
|
|
115
|
+
),
|
|
116
|
+
0.93,
|
|
117
|
+
{"failed_large"},
|
|
118
|
+
),
|
|
119
|
+
(
|
|
120
|
+
3,
|
|
121
|
+
_padded("User successfully runs Qwen 9B locally.", 18, "Qwen"),
|
|
122
|
+
0.90,
|
|
123
|
+
{"successful_small"},
|
|
124
|
+
),
|
|
125
|
+
(
|
|
126
|
+
4,
|
|
127
|
+
_padded(
|
|
128
|
+
"Historically Qwen 30B failed due to insufficient available memory.",
|
|
129
|
+
180,
|
|
130
|
+
"memory",
|
|
131
|
+
),
|
|
132
|
+
0.92,
|
|
133
|
+
{"failed_large"},
|
|
134
|
+
),
|
|
135
|
+
(
|
|
136
|
+
5,
|
|
137
|
+
_padded("User prefers concise responses.", 10, "responses"),
|
|
138
|
+
0.10,
|
|
139
|
+
set(),
|
|
140
|
+
),
|
|
141
|
+
(
|
|
142
|
+
6,
|
|
143
|
+
_padded("User uses Ollama for local model inference.", 16, "Ollama"),
|
|
144
|
+
0.88,
|
|
145
|
+
{"runtime"},
|
|
146
|
+
),
|
|
147
|
+
(
|
|
148
|
+
7,
|
|
149
|
+
_padded(
|
|
150
|
+
"User's machine has limited memory for very large models.",
|
|
151
|
+
17,
|
|
152
|
+
"memory",
|
|
153
|
+
),
|
|
154
|
+
0.87,
|
|
155
|
+
{"hardware_limit"},
|
|
156
|
+
),
|
|
157
|
+
]
|
|
158
|
+
candidates: list[ScoredMemory] = []
|
|
159
|
+
annotations: dict[str, set[str]] = {}
|
|
160
|
+
for rank, (number, content, score, units) in enumerate(specs, 1):
|
|
161
|
+
identifier = UUID(f"10000000-0000-0000-0000-{number:012d}")
|
|
162
|
+
status = MemoryStatus.HISTORICAL if number == 4 else MemoryStatus.ACTIVE
|
|
163
|
+
scored = ScoredMemory(
|
|
164
|
+
memory=Memory(
|
|
165
|
+
id=identifier,
|
|
166
|
+
content=content,
|
|
167
|
+
status=status,
|
|
168
|
+
type=MemoryType.FACT,
|
|
169
|
+
importance=0.8,
|
|
170
|
+
confidence=0.9,
|
|
171
|
+
),
|
|
172
|
+
final_score=score,
|
|
173
|
+
rank=rank,
|
|
174
|
+
retrieval_sources=["lexical", "dense"],
|
|
175
|
+
)
|
|
176
|
+
candidates.append(scored)
|
|
177
|
+
annotations[str(identifier)] = units
|
|
178
|
+
return candidates, annotations
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def measure(
|
|
182
|
+
result: SelectionResult,
|
|
183
|
+
candidates: list[ScoredMemory],
|
|
184
|
+
annotations: dict[str, set[str]],
|
|
185
|
+
) -> OptimizationMetrics:
|
|
186
|
+
identifiers = [str(item.memory.id) for item in result.selected_memories]
|
|
187
|
+
coverage = information_unit_coverage(identifiers, annotations, INFORMATION_UNITS)
|
|
188
|
+
return OptimizationMetrics(
|
|
189
|
+
tokens=result.total_tokens,
|
|
190
|
+
utilization=result.utilization,
|
|
191
|
+
coverage=coverage,
|
|
192
|
+
weighted_coverage=weighted_information_unit_coverage(
|
|
193
|
+
identifiers, annotations, UNIT_WEIGHTS
|
|
194
|
+
),
|
|
195
|
+
relevance_retained=retained_relevance(result.selected_memories, candidates),
|
|
196
|
+
redundancy=redundancy_rate(result.selected_memories),
|
|
197
|
+
efficiency=coverage_efficiency(coverage, result.total_tokens),
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def synthetic_candidates(count: int = 1_000) -> list[ScoredMemory]:
|
|
202
|
+
return [
|
|
203
|
+
ScoredMemory(
|
|
204
|
+
memory=Memory(
|
|
205
|
+
id=UUID(f"20000000-0000-0000-0000-{index:012d}"),
|
|
206
|
+
content=(
|
|
207
|
+
f"Memory topic{index} records constraint group{index % 37} "
|
|
208
|
+
f"tool{index % 83} outcome{index % 19}."
|
|
209
|
+
),
|
|
210
|
+
status=MemoryStatus.ACTIVE,
|
|
211
|
+
importance=(index % 10) / 10,
|
|
212
|
+
confidence=0.8,
|
|
213
|
+
),
|
|
214
|
+
final_score=1.0 / (1 + index % 100),
|
|
215
|
+
rank=index + 1,
|
|
216
|
+
retrieval_sources=["dense"] if index % 2 else ["lexical", "dense"],
|
|
217
|
+
)
|
|
218
|
+
for index in range(count)
|
|
219
|
+
]
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def main() -> None:
|
|
223
|
+
counter = DeterministicWordTokenCounter()
|
|
224
|
+
optimizer = MemoryContextOptimizer(token_counter=counter)
|
|
225
|
+
candidates, annotations = smoke_candidates()
|
|
226
|
+
aggregate: dict[OptimizationStrategy, list[OptimizationMetrics]] = {
|
|
227
|
+
strategy: [] for strategy in OptimizationStrategy
|
|
228
|
+
}
|
|
229
|
+
print("Smoke scenario")
|
|
230
|
+
print(
|
|
231
|
+
"Budget Strategy IDs Tokens Util Coverage "
|
|
232
|
+
"Weighted Relevance Redundancy Efficiency Exclusions"
|
|
233
|
+
)
|
|
234
|
+
for budget_size in (40, 60, 100, 250):
|
|
235
|
+
for strategy in OptimizationStrategy:
|
|
236
|
+
result = optimizer.optimize(
|
|
237
|
+
SMOKE_QUERY,
|
|
238
|
+
candidates,
|
|
239
|
+
ContextBudget(max_tokens=budget_size),
|
|
240
|
+
strategy,
|
|
241
|
+
)
|
|
242
|
+
metrics = measure(result, candidates, annotations)
|
|
243
|
+
aggregate[strategy].append(metrics)
|
|
244
|
+
ids = ",".join(str(item.memory.id)[-2:] for item in result.selected_memories) or "-"
|
|
245
|
+
exclusions = Counter(
|
|
246
|
+
decision.exclusion_reason.value
|
|
247
|
+
for decision in result.trace.decisions
|
|
248
|
+
if decision.exclusion_reason is not None
|
|
249
|
+
)
|
|
250
|
+
exclusion_text = ",".join(
|
|
251
|
+
f"{reason}:{count}" for reason, count in sorted(exclusions.items())
|
|
252
|
+
) or "-"
|
|
253
|
+
print(
|
|
254
|
+
f"{budget_size:>6} {STRATEGY_LABELS[strategy]:<14} {ids:<19} "
|
|
255
|
+
f"{metrics.tokens:>6} {metrics.utilization:>5.3f} "
|
|
256
|
+
f"{metrics.coverage:>8.3f} {metrics.weighted_coverage:>8.3f} "
|
|
257
|
+
f"{metrics.relevance_retained:>9.3f} "
|
|
258
|
+
f"{metrics.redundancy:>10.3f} {metrics.efficiency:>10.4f} "
|
|
259
|
+
f"{exclusion_text}"
|
|
260
|
+
)
|
|
261
|
+
|
|
262
|
+
print("\nAverage evaluation metrics across budgets")
|
|
263
|
+
print(
|
|
264
|
+
"Strategy Tokens Utilization Coverage Weighted Relevance "
|
|
265
|
+
"Redundancy Efficiency"
|
|
266
|
+
)
|
|
267
|
+
for strategy, rows in aggregate.items():
|
|
268
|
+
count = len(rows)
|
|
269
|
+
print(
|
|
270
|
+
f"{STRATEGY_LABELS[strategy]:<13} "
|
|
271
|
+
f"{sum(row.tokens for row in rows) / count:>6.1f} "
|
|
272
|
+
f"{sum(row.utilization for row in rows) / count:>11.3f} "
|
|
273
|
+
f"{sum(row.coverage for row in rows) / count:>8.3f} "
|
|
274
|
+
f"{sum(row.weighted_coverage for row in rows) / count:>8.3f} "
|
|
275
|
+
f"{sum(row.relevance_retained for row in rows) / count:>9.3f} "
|
|
276
|
+
f"{sum(row.redundancy for row in rows) / count:>10.3f} "
|
|
277
|
+
f"{sum(row.efficiency for row in rows) / count:>10.4f}"
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
scale = synthetic_candidates()
|
|
281
|
+
started = time.perf_counter()
|
|
282
|
+
first = optimizer.optimize(
|
|
283
|
+
"Find machine tools and constraints",
|
|
284
|
+
scale,
|
|
285
|
+
ContextBudget(max_tokens=500),
|
|
286
|
+
)
|
|
287
|
+
elapsed_ms = (time.perf_counter() - started) * 1000
|
|
288
|
+
second = optimizer.optimize(
|
|
289
|
+
"Find machine tools and constraints",
|
|
290
|
+
scale,
|
|
291
|
+
ContextBudget(max_tokens=500),
|
|
292
|
+
)
|
|
293
|
+
deterministic = [
|
|
294
|
+
item.memory.id for item in first.selected_memories
|
|
295
|
+
] == [item.memory.id for item in second.selected_memories]
|
|
296
|
+
print("\nSynthetic scale")
|
|
297
|
+
print(
|
|
298
|
+
f"candidates=1000 selected={len(first.selected_memories)} "
|
|
299
|
+
f"tokens={first.total_tokens}/500 latency_ms={elapsed_ms:.3f} "
|
|
300
|
+
f"deterministic={deterministic}"
|
|
301
|
+
)
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
if __name__ == "__main__":
|
|
305
|
+
main()
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
"""Deterministic Phase 4 -> 5 -> 6 oversized-rescue smoke scenario."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import json
|
|
7
|
+
import tempfile
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from uuid import UUID
|
|
10
|
+
|
|
11
|
+
from contextos.core.enums import MemoryStatus, MemoryType
|
|
12
|
+
from contextos.core.models import CompilationConfig, ContextBudget, Memory, RetrievalQuery
|
|
13
|
+
from contextos.embedding.deterministic import DeterministicEmbedding
|
|
14
|
+
from contextos.services.compilation import QueryAwareContextCompiler
|
|
15
|
+
from contextos.services.optimization import MemoryContextOptimizer
|
|
16
|
+
from contextos.services.retrieval import HybridRetrievalEngine
|
|
17
|
+
from contextos.services.retrieval_index import RetrievalIndexSynchronizer
|
|
18
|
+
from contextos.services.token_counter import DeterministicWordTokenCounter
|
|
19
|
+
from contextos.storage.database import Database
|
|
20
|
+
from contextos.storage.lexical.bm25 import BM25Index
|
|
21
|
+
from contextos.storage.memory_repo import SqliteMemoryRepository
|
|
22
|
+
from contextos.storage.vector.in_memory import InMemoryVectorStore
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _memory(number: int, content: str, status: MemoryStatus = MemoryStatus.ACTIVE) -> Memory:
|
|
26
|
+
return Memory(
|
|
27
|
+
id=UUID(f"80000000-0000-0000-0000-{number:012d}"),
|
|
28
|
+
content=content,
|
|
29
|
+
status=status,
|
|
30
|
+
type=MemoryType.FACT,
|
|
31
|
+
provenance_event_id=UUID(f"81000000-0000-0000-0000-{number:012d}"),
|
|
32
|
+
confidence=0.95,
|
|
33
|
+
importance=0.9,
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
async def run_smoke() -> dict[str, object]:
|
|
38
|
+
padding = " ".join(f"diagnostic{index}" for index in range(320)) + "."
|
|
39
|
+
unrelated = " ".join(f"gardening{index}" for index in range(320)) + "."
|
|
40
|
+
memories = [
|
|
41
|
+
_memory(1, "Ollama is the selected local model runner."),
|
|
42
|
+
_memory(
|
|
43
|
+
2,
|
|
44
|
+
"Qwen30B failed on the local machine because available VRAM was insufficient. "
|
|
45
|
+
+ padding,
|
|
46
|
+
),
|
|
47
|
+
_memory(3, "Qwen9B works on the local machine."),
|
|
48
|
+
_memory(4, "A historical garden watering archive. " + unrelated, MemoryStatus.HISTORICAL),
|
|
49
|
+
]
|
|
50
|
+
with tempfile.TemporaryDirectory(prefix="contextos-rescue-") as directory:
|
|
51
|
+
database_path = Path(directory) / "contextos.db"
|
|
52
|
+
database = Database(database_path)
|
|
53
|
+
await database.initialize()
|
|
54
|
+
try:
|
|
55
|
+
repository = SqliteMemoryRepository(database.connection())
|
|
56
|
+
for memory in memories:
|
|
57
|
+
await repository.create(memory)
|
|
58
|
+
|
|
59
|
+
embedding = DeterministicEmbedding(64)
|
|
60
|
+
lexical = BM25Index()
|
|
61
|
+
vector = InMemoryVectorStore(embedding.dimension)
|
|
62
|
+
synchronizer = RetrievalIndexSynchronizer(
|
|
63
|
+
memory_repo=repository,
|
|
64
|
+
lexical_index=lexical,
|
|
65
|
+
vector_store=vector,
|
|
66
|
+
embedding_service=embedding,
|
|
67
|
+
)
|
|
68
|
+
retrieval = HybridRetrievalEngine(
|
|
69
|
+
memory_repo=repository,
|
|
70
|
+
lexical_index=lexical,
|
|
71
|
+
vector_store=vector,
|
|
72
|
+
embedding_service=embedding,
|
|
73
|
+
index_synchronizer=synchronizer,
|
|
74
|
+
)
|
|
75
|
+
query = "local model machine attempts"
|
|
76
|
+
retrieved = await retrieval.retrieve(RetrievalQuery(text=query, k=10))
|
|
77
|
+
counter = DeterministicWordTokenCounter()
|
|
78
|
+
optimizer = MemoryContextOptimizer(token_counter=counter)
|
|
79
|
+
selection = optimizer.optimize(
|
|
80
|
+
query, retrieved.memories, ContextBudget(max_tokens=24)
|
|
81
|
+
)
|
|
82
|
+
compiler = QueryAwareContextCompiler(token_counter=counter)
|
|
83
|
+
compiled = await compiler.compile(
|
|
84
|
+
query, selection, CompilationConfig(budget=48)
|
|
85
|
+
)
|
|
86
|
+
return {
|
|
87
|
+
"database_on_disk": database_path.exists(),
|
|
88
|
+
"retrieved_ids": [str(value.memory.id) for value in retrieved.memories],
|
|
89
|
+
"selected_ids": [
|
|
90
|
+
str(value.memory.id) for value in selection.selected_memories
|
|
91
|
+
],
|
|
92
|
+
"rescue_ids": [
|
|
93
|
+
str(value.memory.id)
|
|
94
|
+
for value in selection.compiler_rescue_candidates
|
|
95
|
+
],
|
|
96
|
+
"emitted_facts": [
|
|
97
|
+
{
|
|
98
|
+
"text": fact.text,
|
|
99
|
+
"input_kind": fact.input_kind.value,
|
|
100
|
+
"source_memory_ids": [
|
|
101
|
+
str(value) for value in fact.source_memory_ids
|
|
102
|
+
],
|
|
103
|
+
"provenance_event_ids": [
|
|
104
|
+
str(value) for value in fact.provenance_event_ids
|
|
105
|
+
],
|
|
106
|
+
}
|
|
107
|
+
for fact in compiled.facts
|
|
108
|
+
],
|
|
109
|
+
"tokens": compiled.total_tokens,
|
|
110
|
+
"budget": compiled.budget,
|
|
111
|
+
"oversized_rescue_inputs": compiled.trace.oversized_rescue_inputs,
|
|
112
|
+
"rescued_facts_included": compiled.trace.rescued_facts_included,
|
|
113
|
+
"unsupported_fact_rate": compiled.unsupported_fact_rate,
|
|
114
|
+
"historical_unrelated_rescued": memories[3].id in {
|
|
115
|
+
value.memory.id for value in selection.compiler_rescue_candidates
|
|
116
|
+
},
|
|
117
|
+
}
|
|
118
|
+
finally:
|
|
119
|
+
await database.close()
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def main() -> None:
|
|
123
|
+
print(json.dumps(asyncio.run(run_smoke()), indent=2, sort_keys=True))
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
if __name__ == "__main__":
|
|
127
|
+
main()
|
|
@@ -0,0 +1,266 @@
|
|
|
1
|
+
"""Deterministic, offline Phase 4 retrieval evaluation.
|
|
2
|
+
|
|
3
|
+
Run with: python -m contextos.benchmarks.retrieval
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import asyncio
|
|
9
|
+
import math
|
|
10
|
+
import tempfile
|
|
11
|
+
import time
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from contextos.core.enums import MemoryStatus, MemoryType, RetrievalMode, TemporalScope
|
|
16
|
+
from contextos.core.models import Memory, RetrievalQuery
|
|
17
|
+
from contextos.embedding.deterministic import DeterministicEmbedding
|
|
18
|
+
from contextos.services.retrieval import HybridRetrievalEngine
|
|
19
|
+
from contextos.services.retrieval_index import RetrievalIndexSynchronizer
|
|
20
|
+
from contextos.storage.database import Database
|
|
21
|
+
from contextos.storage.lexical.bm25 import BM25Index
|
|
22
|
+
from contextos.storage.memory_repo import SqliteMemoryRepository
|
|
23
|
+
from contextos.storage.vector.in_memory import InMemoryVectorStore
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(frozen=True)
|
|
27
|
+
class EvaluationCase:
|
|
28
|
+
query_id: str
|
|
29
|
+
text: str
|
|
30
|
+
qrels: dict[str, int]
|
|
31
|
+
temporal_scope: TemporalScope = TemporalScope.CURRENT
|
|
32
|
+
memory_types: set[MemoryType] | None = None
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass
|
|
36
|
+
class EvaluationSummary:
|
|
37
|
+
recall: float
|
|
38
|
+
precision: float
|
|
39
|
+
hit_rate: float
|
|
40
|
+
mrr: float
|
|
41
|
+
ndcg: float
|
|
42
|
+
average_latency_ms: float
|
|
43
|
+
stage_latency_ms: dict[str, float] = field(default_factory=dict)
|
|
44
|
+
rankings: dict[str, list[str]] = field(default_factory=dict)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def recall_at_k(retrieved: list[str], qrels: dict[str, int], k: int) -> float:
|
|
48
|
+
relevant = {identifier for identifier, grade in qrels.items() if grade > 0}
|
|
49
|
+
if not relevant:
|
|
50
|
+
return 0.0
|
|
51
|
+
return len(set(retrieved[:k]) & relevant) / len(relevant)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def precision_at_k(retrieved: list[str], qrels: dict[str, int], k: int) -> float:
|
|
55
|
+
if k <= 0:
|
|
56
|
+
return 0.0
|
|
57
|
+
relevant = {identifier for identifier, grade in qrels.items() if grade > 0}
|
|
58
|
+
return len(set(retrieved[:k]) & relevant) / k
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def hit_rate_at_k(retrieved: list[str], qrels: dict[str, int], k: int) -> float:
|
|
62
|
+
return float(any(qrels.get(identifier, 0) > 0 for identifier in retrieved[:k]))
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def reciprocal_rank(retrieved: list[str], qrels: dict[str, int]) -> float:
|
|
66
|
+
for rank, identifier in enumerate(retrieved, 1):
|
|
67
|
+
if qrels.get(identifier, 0) > 0:
|
|
68
|
+
return 1.0 / rank
|
|
69
|
+
return 0.0
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def ndcg_at_k(retrieved: list[str], qrels: dict[str, int], k: int) -> float:
|
|
73
|
+
def dcg(grades: list[int]) -> float:
|
|
74
|
+
return sum((2**grade - 1) / math.log2(rank + 1) for rank, grade in enumerate(grades, 1))
|
|
75
|
+
|
|
76
|
+
actual = dcg([qrels.get(identifier, 0) for identifier in retrieved[:k]])
|
|
77
|
+
ideal = dcg(sorted(qrels.values(), reverse=True)[:k])
|
|
78
|
+
return actual / ideal if ideal else 0.0
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def evaluation_memories() -> list[Memory]:
|
|
82
|
+
specs = [
|
|
83
|
+
("m01", "User currently uses Ollama for local model inference.",
|
|
84
|
+
MemoryStatus.ACTIVE, MemoryType.FACT),
|
|
85
|
+
("m02", "User previously focused primarily on Python.",
|
|
86
|
+
MemoryStatus.HISTORICAL, MemoryType.SKILL),
|
|
87
|
+
("m03", "User currently focuses on C++17 for systems interviews.",
|
|
88
|
+
MemoryStatus.ACTIVE, MemoryType.GOAL),
|
|
89
|
+
("m04", "User prefers concise technical responses.",
|
|
90
|
+
MemoryStatus.ACTIVE, MemoryType.PREFERENCE),
|
|
91
|
+
("m05", "User enjoys biryani.", MemoryStatus.ACTIVE, MemoryType.PREFERENCE),
|
|
92
|
+
(
|
|
93
|
+
"m06",
|
|
94
|
+
"Qwen 30B, the much larger model, failed locally because available "
|
|
95
|
+
"memory was insufficient.",
|
|
96
|
+
MemoryStatus.ACTIVE,
|
|
97
|
+
MemoryType.FACT,
|
|
98
|
+
),
|
|
99
|
+
("m07", "User has successfully run Qwen 9B locally.",
|
|
100
|
+
MemoryStatus.ACTIVE, MemoryType.FACT),
|
|
101
|
+
("m08", "The Atlas project uses PostgreSQL for durable storage.",
|
|
102
|
+
MemoryStatus.ACTIVE, MemoryType.PROJECT),
|
|
103
|
+
("m09", "User tests Python services with pytest.",
|
|
104
|
+
MemoryStatus.ACTIVE, MemoryType.PROCEDURE),
|
|
105
|
+
("m10", "User likes detailed travel stories.",
|
|
106
|
+
MemoryStatus.ACTIVE, MemoryType.PREFERENCE),
|
|
107
|
+
("m11", "A deleted note claimed the runtime was Docker.",
|
|
108
|
+
MemoryStatus.DELETED, MemoryType.FACT),
|
|
109
|
+
("m12", "An expired goal was to study Java.",
|
|
110
|
+
MemoryStatus.EXPIRED, MemoryType.GOAL),
|
|
111
|
+
("m13", "User builds local search tools with SQLite.",
|
|
112
|
+
MemoryStatus.ACTIVE, MemoryType.PROJECT),
|
|
113
|
+
("m14", "User previously used llama.cpp for inference.",
|
|
114
|
+
MemoryStatus.SUPERSEDED, MemoryType.FACT),
|
|
115
|
+
]
|
|
116
|
+
from uuid import UUID
|
|
117
|
+
|
|
118
|
+
return [
|
|
119
|
+
Memory(
|
|
120
|
+
id=UUID(f"00000000-0000-0000-0000-{index:012d}"),
|
|
121
|
+
content=content,
|
|
122
|
+
status=status,
|
|
123
|
+
type=memory_type,
|
|
124
|
+
source_type="benchmark",
|
|
125
|
+
confidence=0.9,
|
|
126
|
+
importance=0.7,
|
|
127
|
+
tags=[short_id],
|
|
128
|
+
)
|
|
129
|
+
for index, (short_id, content, status, memory_type) in enumerate(specs, 1)
|
|
130
|
+
]
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def evaluation_cases() -> list[EvaluationCase]:
|
|
134
|
+
def uid(number: int) -> str:
|
|
135
|
+
return f"00000000-0000-0000-0000-{number:012d}"
|
|
136
|
+
|
|
137
|
+
return [
|
|
138
|
+
EvaluationCase(
|
|
139
|
+
"q1", "Which local AI runtime does the user use?", {uid(1): 2, uid(7): 1}
|
|
140
|
+
),
|
|
141
|
+
EvaluationCase("q2", "What programming language is the user focusing on now?", {uid(3): 2}),
|
|
142
|
+
EvaluationCase(
|
|
143
|
+
"q3", "What language did the user focus on previously?", {uid(2): 2},
|
|
144
|
+
TemporalScope.HISTORICAL,
|
|
145
|
+
),
|
|
146
|
+
EvaluationCase("q4", "What response style does the user prefer?", {uid(4): 2}),
|
|
147
|
+
EvaluationCase(
|
|
148
|
+
"q5",
|
|
149
|
+
"What happened when the user tried a much larger Qwen model?",
|
|
150
|
+
{uid(6): 2},
|
|
151
|
+
),
|
|
152
|
+
EvaluationCase("q6", "Which database backs the Atlas project?", {uid(8): 2}),
|
|
153
|
+
EvaluationCase(
|
|
154
|
+
"q7", "What active career learning objective involves systems?", {uid(3): 2},
|
|
155
|
+
memory_types={MemoryType.GOAL},
|
|
156
|
+
),
|
|
157
|
+
]
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
async def build_engine(path: Path) -> tuple[Database, HybridRetrievalEngine]:
|
|
161
|
+
database = Database(path)
|
|
162
|
+
await database.initialize()
|
|
163
|
+
repository = SqliteMemoryRepository(database.connection())
|
|
164
|
+
for memory in evaluation_memories():
|
|
165
|
+
await repository.create(memory)
|
|
166
|
+
embedding = DeterministicEmbedding()
|
|
167
|
+
lexical = BM25Index()
|
|
168
|
+
vector = InMemoryVectorStore(embedding.dimension)
|
|
169
|
+
synchronizer = RetrievalIndexSynchronizer(
|
|
170
|
+
memory_repo=repository,
|
|
171
|
+
lexical_index=lexical,
|
|
172
|
+
vector_store=vector,
|
|
173
|
+
embedding_service=embedding,
|
|
174
|
+
)
|
|
175
|
+
engine = HybridRetrievalEngine(
|
|
176
|
+
memory_repo=repository,
|
|
177
|
+
lexical_index=lexical,
|
|
178
|
+
vector_store=vector,
|
|
179
|
+
embedding_service=embedding,
|
|
180
|
+
index_synchronizer=synchronizer,
|
|
181
|
+
)
|
|
182
|
+
return database, engine
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
async def evaluate(
|
|
186
|
+
engine: HybridRetrievalEngine, mode: RetrievalMode, *, k: int = 5
|
|
187
|
+
) -> EvaluationSummary:
|
|
188
|
+
totals = {"recall": 0.0, "precision": 0.0, "hit": 0.0, "mrr": 0.0, "ndcg": 0.0}
|
|
189
|
+
latencies: list[float] = []
|
|
190
|
+
stage_totals: dict[str, float] = {}
|
|
191
|
+
stage_counts: dict[str, int] = {}
|
|
192
|
+
rankings: dict[str, list[str]] = {}
|
|
193
|
+
cases = evaluation_cases()
|
|
194
|
+
for case in cases:
|
|
195
|
+
started = time.perf_counter()
|
|
196
|
+
result = await engine.retrieve(RetrievalQuery(
|
|
197
|
+
text=case.text,
|
|
198
|
+
k=k,
|
|
199
|
+
mode=mode,
|
|
200
|
+
temporal_scope=case.temporal_scope,
|
|
201
|
+
allowed_memory_types=case.memory_types,
|
|
202
|
+
))
|
|
203
|
+
latencies.append((time.perf_counter() - started) * 1000)
|
|
204
|
+
for stage in result.trace.stages:
|
|
205
|
+
stage_totals[stage.stage_name] = (
|
|
206
|
+
stage_totals.get(stage.stage_name, 0.0) + stage.latency_ms
|
|
207
|
+
)
|
|
208
|
+
stage_counts[stage.stage_name] = stage_counts.get(stage.stage_name, 0) + 1
|
|
209
|
+
identifiers = [str(item.memory.id) for item in result.memories]
|
|
210
|
+
rankings[case.query_id] = [item.memory.content for item in result.memories]
|
|
211
|
+
totals["recall"] += recall_at_k(identifiers, case.qrels, k)
|
|
212
|
+
totals["precision"] += precision_at_k(identifiers, case.qrels, k)
|
|
213
|
+
totals["hit"] += hit_rate_at_k(identifiers, case.qrels, k)
|
|
214
|
+
totals["mrr"] += reciprocal_rank(identifiers, case.qrels)
|
|
215
|
+
totals["ndcg"] += ndcg_at_k(identifiers, case.qrels, k)
|
|
216
|
+
count = len(cases)
|
|
217
|
+
return EvaluationSummary(
|
|
218
|
+
recall=totals["recall"] / count,
|
|
219
|
+
precision=totals["precision"] / count,
|
|
220
|
+
hit_rate=totals["hit"] / count,
|
|
221
|
+
mrr=totals["mrr"] / count,
|
|
222
|
+
ndcg=totals["ndcg"] / count,
|
|
223
|
+
average_latency_ms=sum(latencies) / count,
|
|
224
|
+
stage_latency_ms={
|
|
225
|
+
name: total / stage_counts[name] for name, total in stage_totals.items()
|
|
226
|
+
},
|
|
227
|
+
rankings=rankings,
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
async def _main() -> None:
|
|
232
|
+
with tempfile.TemporaryDirectory(prefix="contextos-retrieval-") as directory:
|
|
233
|
+
database, engine = await build_engine(Path(directory) / "benchmark.db")
|
|
234
|
+
try:
|
|
235
|
+
print("Mode Recall@5 Precision@5 HitRate@5 MRR NDCG@5 Avg ms")
|
|
236
|
+
summaries: dict[RetrievalMode, EvaluationSummary] = {}
|
|
237
|
+
for mode in RetrievalMode:
|
|
238
|
+
summaries[mode] = await evaluate(engine, mode)
|
|
239
|
+
item = summaries[mode]
|
|
240
|
+
print(
|
|
241
|
+
f"{mode.value:<10} {item.recall:>8.3f} {item.precision:>11.3f} "
|
|
242
|
+
f"{item.hit_rate:>9.3f} {item.mrr:>6.3f} {item.ndcg:>6.3f} "
|
|
243
|
+
f"{item.average_latency_ms:>6.3f}"
|
|
244
|
+
)
|
|
245
|
+
print("\nAverage stage latency (ms):")
|
|
246
|
+
print("Mode Lexical Dense Fusion Total")
|
|
247
|
+
for mode, item in summaries.items():
|
|
248
|
+
print(
|
|
249
|
+
f"{mode.value:<10} "
|
|
250
|
+
f"{item.stage_latency_ms.get('lexical_search', 0.0):>7.3f} "
|
|
251
|
+
f"{item.stage_latency_ms.get('dense_search', 0.0):>7.3f} "
|
|
252
|
+
f"{item.stage_latency_ms.get('fusion_rerank', 0.0):>7.3f} "
|
|
253
|
+
f"{item.average_latency_ms:>7.3f}"
|
|
254
|
+
)
|
|
255
|
+
print("\nSmoke rankings (top 3):")
|
|
256
|
+
for case in evaluation_cases()[:5]:
|
|
257
|
+
print(f"\n{case.query_id}: {case.text}")
|
|
258
|
+
for mode in RetrievalMode:
|
|
259
|
+
ranking = summaries[mode].rankings[case.query_id][:3]
|
|
260
|
+
print(f" {mode.value:<8} " + " | ".join(ranking))
|
|
261
|
+
finally:
|
|
262
|
+
await database.close()
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
if __name__ == "__main__":
|
|
266
|
+
asyncio.run(_main())
|