contextos-memory-runtime 1.0.0rc2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contextos/__init__.py +3 -0
- contextos/__main__.py +6 -0
- contextos/api/__init__.py +1 -0
- contextos/api/routes/__init__.py +1 -0
- contextos/api/routes/desktop.py +322 -0
- contextos/api/routes/ingest.py +17 -0
- contextos/api/routes/memories.py +84 -0
- contextos/api/routes/models.py +81 -0
- contextos/api/routes/retrieval.py +89 -0
- contextos/api/routes/system.py +216 -0
- contextos/api/server.py +195 -0
- contextos/benchmarks/__init__.py +1 -0
- contextos/benchmarks/compilation.py +245 -0
- contextos/benchmarks/connectors.py +423 -0
- contextos/benchmarks/explainability.py +103 -0
- contextos/benchmarks/final.py +406 -0
- contextos/benchmarks/graph.py +310 -0
- contextos/benchmarks/graph_adversarial.py +525 -0
- contextos/benchmarks/mcp.py +324 -0
- contextos/benchmarks/model_routing.py +203 -0
- contextos/benchmarks/optimization.py +305 -0
- contextos/benchmarks/rescue_integration.py +127 -0
- contextos/benchmarks/retrieval.py +266 -0
- contextos/benchmarks/temporal.py +377 -0
- contextos/benchmarks/temporal_hotpath.py +76 -0
- contextos/benchmarks/terminal.py +62 -0
- contextos/cli/__init__.py +1 -0
- contextos/cli/app.py +932 -0
- contextos/cli/dashboard.py +174 -0
- contextos/cli/formatters.py +299 -0
- contextos/config/__init__.py +1 -0
- contextos/config/settings.py +160 -0
- contextos/connectors/__init__.py +6 -0
- contextos/connectors/fake.py +11 -0
- contextos/connectors/json_import.py +125 -0
- contextos/connectors/local_files.py +102 -0
- contextos/connectors/manager.py +293 -0
- contextos/connectors/models.py +62 -0
- contextos/connectors/protocols.py +11 -0
- contextos/core/__init__.py +103 -0
- contextos/core/enums.py +489 -0
- contextos/core/exceptions.py +293 -0
- contextos/core/models.py +1147 -0
- contextos/core/protocols.py +549 -0
- contextos/daemon/__init__.py +1 -0
- contextos/daemon/manager.py +510 -0
- contextos/daemon/state.py +127 -0
- contextos/daemon/wiring.py +296 -0
- contextos/demo.py +217 -0
- contextos/embedding/__init__.py +1 -0
- contextos/embedding/deterministic.py +76 -0
- contextos/embedding/sentence_transformers.py +80 -0
- contextos/mcp/__init__.py +5 -0
- contextos/mcp/server.py +269 -0
- contextos/providers/__init__.py +13 -0
- contextos/providers/fake.py +217 -0
- contextos/providers/ollama.py +297 -0
- contextos/providers/openai_compatible.py +337 -0
- contextos/services/__init__.py +1 -0
- contextos/services/compilation.py +535 -0
- contextos/services/explainability.py +553 -0
- contextos/services/extraction.py +311 -0
- contextos/services/graph.py +524 -0
- contextos/services/graph_retrieval.py +143 -0
- contextos/services/ingestion.py +143 -0
- contextos/services/inspection.py +174 -0
- contextos/services/memory.py +291 -0
- contextos/services/model_service.py +409 -0
- contextos/services/optimization.py +426 -0
- contextos/services/privacy.py +331 -0
- contextos/services/retrieval.py +302 -0
- contextos/services/retrieval_index.py +88 -0
- contextos/services/router.py +302 -0
- contextos/services/secret_scanner.py +207 -0
- contextos/services/telemetry_query.py +102 -0
- contextos/services/temporal.py +500 -0
- contextos/services/token_counter.py +222 -0
- contextos/storage/__init__.py +1 -0
- contextos/storage/connector_repo.py +67 -0
- contextos/storage/database.py +497 -0
- contextos/storage/event_repo.py +137 -0
- contextos/storage/graph_repo.py +228 -0
- contextos/storage/lexical/__init__.py +1 -0
- contextos/storage/lexical/bm25.py +134 -0
- contextos/storage/memory_repo.py +589 -0
- contextos/storage/relation_repo.py +80 -0
- contextos/storage/telemetry_repo.py +481 -0
- contextos/storage/vector/__init__.py +1 -0
- contextos/storage/vector/in_memory.py +162 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,324 @@
|
|
|
1
|
+
"""Deterministic local MCP adapter measurement helpers.
|
|
2
|
+
|
|
3
|
+
Callers provide the already-wired deterministic services used by tests; this
|
|
4
|
+
module never creates a provider or sends network traffic.
|
|
5
|
+
|
|
6
|
+
Label: LOCAL DEVELOPMENT MACHINE SYNTHETIC INFRASTRUCTURE BENCHMARK
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
import statistics
|
|
13
|
+
import time
|
|
14
|
+
from dataclasses import dataclass, field
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from contextos.core.models import RetrievalQuery
|
|
18
|
+
from contextos.mcp.server import ContextOSMCPApplication, MCPPermissions
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True)
|
|
22
|
+
class OperationResult:
|
|
23
|
+
"""Timing and size for a single benchmark operation."""
|
|
24
|
+
operation: str
|
|
25
|
+
iterations: int
|
|
26
|
+
mean_ms: float
|
|
27
|
+
median_ms: float
|
|
28
|
+
p95_ms: float | None # Only with sufficient samples (>= 20)
|
|
29
|
+
serialized_bytes: int
|
|
30
|
+
# Operation-specific metrics
|
|
31
|
+
selected_memories: int | None = None
|
|
32
|
+
compiled_facts: int | None = None
|
|
33
|
+
compiled_tokens: int | None = None
|
|
34
|
+
graph_nodes: int | None = None
|
|
35
|
+
privacy_decision: str | None = None
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True)
|
|
39
|
+
class MCPBenchmarkReport:
|
|
40
|
+
"""Complete Phase 10 benchmark report."""
|
|
41
|
+
label: str = "LOCAL DEVELOPMENT MACHINE SYNTHETIC INFRASTRUCTURE BENCHMARK"
|
|
42
|
+
direct_results: list[OperationResult] = field(default_factory=list)
|
|
43
|
+
mcp_results: list[OperationResult] = field(default_factory=list)
|
|
44
|
+
stdio_results: list[OperationResult] = field(default_factory=list)
|
|
45
|
+
overheads: dict[str, float] = field(default_factory=dict)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _p95(samples: list[float]) -> float | None:
|
|
49
|
+
if len(samples) < 20:
|
|
50
|
+
return None
|
|
51
|
+
return sorted(samples)[int(len(samples) * 0.95)]
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _operation_result(
|
|
55
|
+
operation: str, samples: list[float], response: dict[str, object],
|
|
56
|
+
) -> OperationResult:
|
|
57
|
+
serialized = len(json.dumps(response, sort_keys=True, default=str).encode("utf-8"))
|
|
58
|
+
return OperationResult(
|
|
59
|
+
operation=operation,
|
|
60
|
+
iterations=len(samples),
|
|
61
|
+
mean_ms=statistics.mean(samples),
|
|
62
|
+
median_ms=statistics.median(samples),
|
|
63
|
+
p95_ms=_p95(samples),
|
|
64
|
+
serialized_bytes=serialized,
|
|
65
|
+
selected_memories=int(response.get("selected_memory_count", 0)) if "selected_memory_count" in response else (int(response.get("result_count", 0)) if response.get("ok") else None),
|
|
66
|
+
compiled_facts=int(response.get("compiled_fact_count", 0)) if "compiled_fact_count" in response else None,
|
|
67
|
+
compiled_tokens=int(response.get("token_count", 0)) if "token_count" in response else None,
|
|
68
|
+
graph_nodes=int(response.get("graph_node_count", 0)) if "graph_node_count" in response else None,
|
|
69
|
+
privacy_decision=str(response.get("error_code")) if response.get("error_code") == "PRIVACY_REJECTED" else None,
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
async def _measure(coro_factory, iterations: int) -> tuple[list[float], Any]:
|
|
74
|
+
"""Run a coroutine factory N times and collect timings."""
|
|
75
|
+
samples: list[float] = []
|
|
76
|
+
last_result = None
|
|
77
|
+
for _ in range(iterations):
|
|
78
|
+
started = time.perf_counter()
|
|
79
|
+
last_result = await coro_factory()
|
|
80
|
+
elapsed = (time.perf_counter() - started) * 1000
|
|
81
|
+
samples.append(elapsed)
|
|
82
|
+
return samples, last_result
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
async def run_phase10_mcp_benchmark(
|
|
86
|
+
services: dict[str, Any],
|
|
87
|
+
query: str = "ContextOS memory",
|
|
88
|
+
iterations: int = 5,
|
|
89
|
+
) -> MCPBenchmarkReport:
|
|
90
|
+
"""Measure direct retrieval versus the local adapter using identical services.
|
|
91
|
+
|
|
92
|
+
Exercises: search, compile, graph, remember success, remember privacy
|
|
93
|
+
rejection, and temporal lookup.
|
|
94
|
+
"""
|
|
95
|
+
app = ContextOSMCPApplication(services, MCPPermissions(allow_write=True))
|
|
96
|
+
direct_results: list[OperationResult] = []
|
|
97
|
+
mcp_results: list[OperationResult] = []
|
|
98
|
+
overheads: dict[str, float] = {}
|
|
99
|
+
|
|
100
|
+
# --- Direct service measurements ---
|
|
101
|
+
|
|
102
|
+
# Direct search
|
|
103
|
+
samples, last = await _measure(
|
|
104
|
+
lambda: services["retrieval"].retrieve(RetrievalQuery(text=query, k=5)),
|
|
105
|
+
iterations,
|
|
106
|
+
)
|
|
107
|
+
direct_results.append(OperationResult(
|
|
108
|
+
operation="search", iterations=len(samples),
|
|
109
|
+
mean_ms=statistics.mean(samples), median_ms=statistics.median(samples),
|
|
110
|
+
p95_ms=_p95(samples), serialized_bytes=0,
|
|
111
|
+
selected_memories=len(last.memories) if last else 0,
|
|
112
|
+
))
|
|
113
|
+
|
|
114
|
+
# --- MCP adapter measurements ---
|
|
115
|
+
|
|
116
|
+
# MCP search
|
|
117
|
+
samples, last = await _measure(
|
|
118
|
+
lambda: app.invoke("contextos_search_memory", None,
|
|
119
|
+
lambda: app.search(query, 5, "hybrid", False)),
|
|
120
|
+
iterations,
|
|
121
|
+
)
|
|
122
|
+
mcp_results.append(_operation_result("search", samples, last))
|
|
123
|
+
overheads["search_adapter_ms"] = max(0.0, mcp_results[-1].mean_ms - direct_results[0].mean_ms)
|
|
124
|
+
|
|
125
|
+
# MCP compile
|
|
126
|
+
samples, last = await _measure(
|
|
127
|
+
lambda: app.invoke("contextos_compile_context", None,
|
|
128
|
+
lambda: app.compile(query, 1000, "hybrid")),
|
|
129
|
+
iterations,
|
|
130
|
+
)
|
|
131
|
+
mcp_results.append(_operation_result("compile", samples, last))
|
|
132
|
+
|
|
133
|
+
# MCP graph
|
|
134
|
+
samples, last = await _measure(
|
|
135
|
+
lambda: app.invoke("contextos_graph_neighbors", None,
|
|
136
|
+
lambda: app.graph_neighbors(query, 1, 50, 100)),
|
|
137
|
+
iterations,
|
|
138
|
+
)
|
|
139
|
+
mcp_results.append(_operation_result("graph", samples, last))
|
|
140
|
+
|
|
141
|
+
# MCP remember (success)
|
|
142
|
+
samples, last = await _measure(
|
|
143
|
+
lambda: app.invoke("contextos_remember", None,
|
|
144
|
+
lambda: app.remember(f"Benchmark memory about {query}")),
|
|
145
|
+
iterations,
|
|
146
|
+
)
|
|
147
|
+
mcp_results.append(_operation_result("remember_success", samples, last))
|
|
148
|
+
|
|
149
|
+
# MCP temporal lookup
|
|
150
|
+
samples, last = await _measure(
|
|
151
|
+
lambda: app.invoke("contextos_current_state", None,
|
|
152
|
+
lambda: app.current_state("programming_language", "user", "global")),
|
|
153
|
+
iterations,
|
|
154
|
+
)
|
|
155
|
+
mcp_results.append(_operation_result("temporal_lookup", samples, last))
|
|
156
|
+
|
|
157
|
+
# MCP telemetry
|
|
158
|
+
samples, last = await _measure(
|
|
159
|
+
lambda: app.invoke("contextos_telemetry_summary", None,
|
|
160
|
+
app.telemetry_summary),
|
|
161
|
+
iterations,
|
|
162
|
+
)
|
|
163
|
+
mcp_results.append(_operation_result("telemetry", samples, last))
|
|
164
|
+
|
|
165
|
+
return MCPBenchmarkReport(
|
|
166
|
+
direct_results=direct_results,
|
|
167
|
+
mcp_results=mcp_results,
|
|
168
|
+
overheads=overheads,
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
async def main() -> None:
|
|
173
|
+
"""Run the canonical Phase 10 MCP benchmark against a temporary SQLite DB."""
|
|
174
|
+
import asyncio
|
|
175
|
+
import tempfile
|
|
176
|
+
from pathlib import Path
|
|
177
|
+
from mcp import Client
|
|
178
|
+
from contextos.core.enums import SecretDetectionMode
|
|
179
|
+
from contextos.embedding.deterministic import DeterministicEmbedding
|
|
180
|
+
from contextos.mcp.server import MCPPermissions, create_mcp_server
|
|
181
|
+
from contextos.services.compilation import QueryAwareContextCompiler
|
|
182
|
+
from contextos.services.extraction import RuleBasedMemoryExtractor
|
|
183
|
+
from contextos.services.graph import MemoryGraphService
|
|
184
|
+
from contextos.services.graph_retrieval import GraphAugmentedRetrievalEngine
|
|
185
|
+
from contextos.services.ingestion import IngestionPipeline
|
|
186
|
+
from contextos.services.optimization import MemoryContextOptimizer
|
|
187
|
+
from contextos.services.retrieval import HybridRetrievalEngine
|
|
188
|
+
from contextos.services.retrieval_index import RetrievalIndexSynchronizer
|
|
189
|
+
from contextos.services.secret_scanner import PatternSecretScanner
|
|
190
|
+
from contextos.services.telemetry_query import TelemetryQueryService
|
|
191
|
+
from contextos.services.temporal import TemporalMemoryService
|
|
192
|
+
from contextos.services.token_counter import DeterministicWordTokenCounter
|
|
193
|
+
from contextos.storage.database import Database
|
|
194
|
+
from contextos.storage.event_repo import SqliteEventRepository
|
|
195
|
+
from contextos.storage.graph_repo import SqliteGraphRepository
|
|
196
|
+
from contextos.storage.lexical.bm25 import BM25Index
|
|
197
|
+
from contextos.storage.memory_repo import SqliteMemoryRepository
|
|
198
|
+
from contextos.storage.relation_repo import SqliteRelationRepository
|
|
199
|
+
from contextos.storage.telemetry_repo import SqliteTelemetryRepository
|
|
200
|
+
from contextos.storage.vector.in_memory import InMemoryVectorStore
|
|
201
|
+
|
|
202
|
+
tmp_dir = Path(tempfile.mkdtemp())
|
|
203
|
+
db_path = tmp_dir / "bench.db"
|
|
204
|
+
db = Database(db_path)
|
|
205
|
+
await db.initialize()
|
|
206
|
+
conn = db.connection()
|
|
207
|
+
memory_repo = SqliteMemoryRepository(conn)
|
|
208
|
+
event_repo = SqliteEventRepository(conn)
|
|
209
|
+
relation_repo = SqliteRelationRepository(conn)
|
|
210
|
+
graph_repo = SqliteGraphRepository(conn)
|
|
211
|
+
telemetry_repo = SqliteTelemetryRepository(conn)
|
|
212
|
+
embedding = DeterministicEmbedding(16)
|
|
213
|
+
lexical, vector = BM25Index(), InMemoryVectorStore(16)
|
|
214
|
+
index = RetrievalIndexSynchronizer(
|
|
215
|
+
memory_repo=memory_repo, embedding_service=embedding,
|
|
216
|
+
vector_store=vector, lexical_index=lexical,
|
|
217
|
+
)
|
|
218
|
+
graph = MemoryGraphService(
|
|
219
|
+
memory_repo=memory_repo, relation_repo=relation_repo,
|
|
220
|
+
graph_repo=graph_repo,
|
|
221
|
+
)
|
|
222
|
+
retrieval = GraphAugmentedRetrievalEngine(
|
|
223
|
+
base_engine=HybridRetrievalEngine(
|
|
224
|
+
memory_repo=memory_repo, vector_store=vector,
|
|
225
|
+
lexical_index=lexical, embedding_service=embedding,
|
|
226
|
+
index_synchronizer=index,
|
|
227
|
+
),
|
|
228
|
+
graph_service=graph, memory_repo=memory_repo,
|
|
229
|
+
)
|
|
230
|
+
token_counter = DeterministicWordTokenCounter()
|
|
231
|
+
temporal = TemporalMemoryService(memory_repo)
|
|
232
|
+
ingestion = IngestionPipeline(
|
|
233
|
+
secret_scanner=PatternSecretScanner(),
|
|
234
|
+
memory_extractor=RuleBasedMemoryExtractor(),
|
|
235
|
+
memory_repo=memory_repo, event_repo=event_repo,
|
|
236
|
+
embedding_service=embedding, vector_store=vector,
|
|
237
|
+
lexical_index=lexical, token_counter=token_counter,
|
|
238
|
+
secret_detection_mode=SecretDetectionMode.STRICT,
|
|
239
|
+
)
|
|
240
|
+
services = {
|
|
241
|
+
"database": db,
|
|
242
|
+
"retrieval": retrieval,
|
|
243
|
+
"optimizer": MemoryContextOptimizer(token_counter=token_counter),
|
|
244
|
+
"compilation": QueryAwareContextCompiler(token_counter=token_counter),
|
|
245
|
+
"graph": graph,
|
|
246
|
+
"temporal": temporal,
|
|
247
|
+
"ingestion": ingestion,
|
|
248
|
+
"telemetry_query": TelemetryQueryService(telemetry_repo),
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
# Seed data using MCP adapter
|
|
252
|
+
server = create_mcp_server(services, MCPPermissions(allow_write=True))
|
|
253
|
+
async with Client(server) as client:
|
|
254
|
+
seed_data = [
|
|
255
|
+
"I am working on Project Atlas. Project Atlas uses Python for machine learning.",
|
|
256
|
+
"I am working on Project Beta. Project Beta uses Rust for systems programming.",
|
|
257
|
+
"I currently use Ollama for local inference.",
|
|
258
|
+
"I am working on Project Atlas. Project Atlas uses llama.cpp.",
|
|
259
|
+
"I prefer concise answers for technical questions.",
|
|
260
|
+
"I use Docker for containerization.",
|
|
261
|
+
"I am working on Project Gamma. Project Gamma uses PostgreSQL.",
|
|
262
|
+
"I am learning TypeScript for web development.",
|
|
263
|
+
]
|
|
264
|
+
for text in seed_data:
|
|
265
|
+
await client.call_tool("contextos_remember", {"text": text})
|
|
266
|
+
|
|
267
|
+
# Run benchmark
|
|
268
|
+
report = await run_phase10_mcp_benchmark(services, query="Atlas Python machine learning", iterations=5)
|
|
269
|
+
|
|
270
|
+
# Print results
|
|
271
|
+
print("=" * 70)
|
|
272
|
+
print("LOCAL DEVELOPMENT MACHINE SYNTHETIC INFRASTRUCTURE BENCHMARK")
|
|
273
|
+
print("=" * 70)
|
|
274
|
+
print()
|
|
275
|
+
print("DIRECT SERVICE RESULTS:")
|
|
276
|
+
print("-" * 50)
|
|
277
|
+
for r in report.direct_results:
|
|
278
|
+
print(f" {r.operation}:")
|
|
279
|
+
print(f" iterations: {r.iterations}")
|
|
280
|
+
print(f" mean latency: {r.mean_ms:.3f} ms")
|
|
281
|
+
print(f" median latency: {r.median_ms:.3f} ms")
|
|
282
|
+
if r.p95_ms is not None:
|
|
283
|
+
print(f" p95 latency: {r.p95_ms:.3f} ms")
|
|
284
|
+
print(f" serialized bytes: {r.serialized_bytes}")
|
|
285
|
+
if r.selected_memories is not None:
|
|
286
|
+
print(f" selected memories: {r.selected_memories}")
|
|
287
|
+
print()
|
|
288
|
+
|
|
289
|
+
print("MCP ADAPTER RESULTS:")
|
|
290
|
+
print("-" * 50)
|
|
291
|
+
for r in report.mcp_results:
|
|
292
|
+
print(f" {r.operation}:")
|
|
293
|
+
print(f" iterations: {r.iterations}")
|
|
294
|
+
print(f" mean latency: {r.mean_ms:.3f} ms")
|
|
295
|
+
print(f" median latency: {r.median_ms:.3f} ms")
|
|
296
|
+
if r.p95_ms is not None:
|
|
297
|
+
print(f" p95 latency: {r.p95_ms:.3f} ms")
|
|
298
|
+
print(f" serialized bytes: {r.serialized_bytes}")
|
|
299
|
+
if r.selected_memories is not None:
|
|
300
|
+
print(f" selected memories: {r.selected_memories}")
|
|
301
|
+
if r.compiled_facts is not None:
|
|
302
|
+
print(f" compiled facts: {r.compiled_facts}")
|
|
303
|
+
if r.compiled_tokens is not None:
|
|
304
|
+
print(f" compiled tokens: {r.compiled_tokens}")
|
|
305
|
+
if r.graph_nodes is not None:
|
|
306
|
+
print(f" graph nodes: {r.graph_nodes}")
|
|
307
|
+
if r.privacy_decision is not None:
|
|
308
|
+
print(f" privacy decision: {r.privacy_decision}")
|
|
309
|
+
print()
|
|
310
|
+
|
|
311
|
+
print("OVERHEAD CALCULATIONS:")
|
|
312
|
+
print("-" * 50)
|
|
313
|
+
for key, value in report.overheads.items():
|
|
314
|
+
print(f" {key}: {value:.3f} ms")
|
|
315
|
+
print()
|
|
316
|
+
|
|
317
|
+
await db.close()
|
|
318
|
+
print("Benchmark complete.")
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
if __name__ == "__main__":
|
|
322
|
+
import asyncio
|
|
323
|
+
asyncio.run(main())
|
|
324
|
+
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
"""Deterministic Phase 9 benchmark: Model routing, multi-model token comparison, and telemetry overhead."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import time
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from uuid import uuid4
|
|
10
|
+
|
|
11
|
+
from contextos.core.enums import (
|
|
12
|
+
MemoryStatus,
|
|
13
|
+
ModelFinishReason,
|
|
14
|
+
RoutingPolicy,
|
|
15
|
+
TokenMeasurementSource,
|
|
16
|
+
)
|
|
17
|
+
from contextos.core.models import (
|
|
18
|
+
CompilationConfig,
|
|
19
|
+
ContextBudget,
|
|
20
|
+
Memory,
|
|
21
|
+
ModelCapabilities,
|
|
22
|
+
ModelInvocationTelemetry,
|
|
23
|
+
ModelRequest,
|
|
24
|
+
ScoredMemory,
|
|
25
|
+
)
|
|
26
|
+
from contextos.providers.fake import DeterministicFakeProvider
|
|
27
|
+
from contextos.services.compilation import QueryAwareContextCompiler
|
|
28
|
+
from contextos.services.optimization import MemoryContextOptimizer
|
|
29
|
+
from contextos.services.router import DeterministicModelRouter
|
|
30
|
+
from contextos.services.token_counter import (
|
|
31
|
+
ClaudeProfileTokenCounter,
|
|
32
|
+
QwenProfileTokenCounter,
|
|
33
|
+
TiktokenCounter,
|
|
34
|
+
get_token_counter_for_model,
|
|
35
|
+
recount_cross_model,
|
|
36
|
+
)
|
|
37
|
+
from contextos.storage.database import Database
|
|
38
|
+
from contextos.storage.telemetry_repo import SqliteTelemetryRepository
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass(frozen=True)
|
|
42
|
+
class ModelProfileResult:
|
|
43
|
+
profile_name: str
|
|
44
|
+
target_token_estimate: int
|
|
45
|
+
measurement_source: str
|
|
46
|
+
reduction_ratio: float
|
|
47
|
+
context_tokens_avoided: int
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass(frozen=True)
|
|
51
|
+
class Phase9BenchmarkReport:
|
|
52
|
+
candidate_context_tokens: int
|
|
53
|
+
compiled_context_tokens: int
|
|
54
|
+
profiles: dict[str, ModelProfileResult]
|
|
55
|
+
router_overhead_ms: float
|
|
56
|
+
telemetry_overhead_ms: float
|
|
57
|
+
token_counting_overhead_ms: float
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def benchmark_memories() -> list[ScoredMemory]:
|
|
61
|
+
"""Diverse benchmark corpus containing code, technical preferences, and project facts."""
|
|
62
|
+
texts = [
|
|
63
|
+
"User currently uses Ollama for local model inference with llama3.2.",
|
|
64
|
+
"User prefers concise technical responses with typed Python 3.12+ syntax and pydantic models.",
|
|
65
|
+
"Project Atlas architecture uses SQLite with WAL mode, graph projection, and reciprocal rank fusion.",
|
|
66
|
+
"User's machine is an M2 Max with 64GB unified memory running macOS Sonoma.",
|
|
67
|
+
"User previously attempted running Qwen-72B locally, but inference failed due to high memory pressure.",
|
|
68
|
+
"Development setup: pytest for testing, ruff for linting, and mypy in strict mode.",
|
|
69
|
+
"ContextOS compilation pipeline achieves token reduction while preserving exact fact attribution.",
|
|
70
|
+
"The local LLM endpoint is hosted at http://127.0.0.1:11434 with zero external telemetry transmission.",
|
|
71
|
+
]
|
|
72
|
+
memories = [
|
|
73
|
+
Memory(
|
|
74
|
+
id=uuid4(),
|
|
75
|
+
content=text,
|
|
76
|
+
status=MemoryStatus.ACTIVE,
|
|
77
|
+
token_count=len(text.split()),
|
|
78
|
+
)
|
|
79
|
+
for text in texts
|
|
80
|
+
]
|
|
81
|
+
return [
|
|
82
|
+
ScoredMemory(
|
|
83
|
+
memory=m,
|
|
84
|
+
final_score=0.95 - (i * 0.05),
|
|
85
|
+
retrieval_sources=["lexical", "dense"] if i % 2 == 0 else ["lexical", "graph"],
|
|
86
|
+
)
|
|
87
|
+
for i, m in enumerate(memories)
|
|
88
|
+
]
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
async def run_phase9_benchmark(db_path: Path | None = None) -> Phase9BenchmarkReport:
|
|
92
|
+
"""Run deterministic Phase 9 benchmark and return structured metrics."""
|
|
93
|
+
ref_counter = TiktokenCounter("cl100k_base")
|
|
94
|
+
optimizer = MemoryContextOptimizer(token_counter=ref_counter)
|
|
95
|
+
compiler = QueryAwareContextCompiler(token_counter=ref_counter)
|
|
96
|
+
|
|
97
|
+
candidates = benchmark_memories()
|
|
98
|
+
candidate_tokens = sum(ref_counter.count(c.memory.content) for c in candidates)
|
|
99
|
+
|
|
100
|
+
# 1. Optimize and Compile
|
|
101
|
+
query = "What is the recommended local model and hardware configuration for Project Atlas?"
|
|
102
|
+
selection = optimizer.optimize(
|
|
103
|
+
query=query,
|
|
104
|
+
candidates=candidates,
|
|
105
|
+
budget=ContextBudget(max_tokens=2000),
|
|
106
|
+
)
|
|
107
|
+
compiled = await compiler.compile(
|
|
108
|
+
query=query,
|
|
109
|
+
memories=selection,
|
|
110
|
+
config=CompilationConfig(budget=1500),
|
|
111
|
+
)
|
|
112
|
+
compiled_tokens = compiled.total_tokens
|
|
113
|
+
|
|
114
|
+
# 2. Multi-Model Token Recounting (Profiles)
|
|
115
|
+
tc_start = time.perf_counter()
|
|
116
|
+
recounts = recount_cross_model(
|
|
117
|
+
compiled.context_text,
|
|
118
|
+
["FAKE_CLAUDE_LIKE", "FAKE_OPENAI_LIKE", "FAKE_QWEN_LIKE"],
|
|
119
|
+
)
|
|
120
|
+
tc_overhead = (time.perf_counter() - tc_start) * 1000.0
|
|
121
|
+
|
|
122
|
+
profiles: dict[str, ModelProfileResult] = {}
|
|
123
|
+
for prof_name, (count, src) in recounts.items():
|
|
124
|
+
# Context tokens avoided from perspective of this model profile
|
|
125
|
+
prof_counter = get_token_counter_for_model(prof_name, prof_name)
|
|
126
|
+
cand_for_model = sum(prof_counter.count(c.memory.content) for c in candidates)
|
|
127
|
+
avoided = max(0, cand_for_model - count)
|
|
128
|
+
red_ratio = max(0.0, 1.0 - (count / cand_for_model)) if cand_for_model > 0 else 0.0
|
|
129
|
+
profiles[prof_name] = ModelProfileResult(
|
|
130
|
+
profile_name=prof_name,
|
|
131
|
+
target_token_estimate=count,
|
|
132
|
+
measurement_source=src.value,
|
|
133
|
+
reduction_ratio=red_ratio,
|
|
134
|
+
context_tokens_avoided=avoided,
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
# 3. Router Overhead Benchmark
|
|
138
|
+
router = DeterministicModelRouter()
|
|
139
|
+
fake_prov = DeterministicFakeProvider()
|
|
140
|
+
providers = {fake_prov.provider_id: fake_prov}
|
|
141
|
+
req = ModelRequest(user_prompt=query, compiled_context=compiled)
|
|
142
|
+
|
|
143
|
+
r_start = time.perf_counter()
|
|
144
|
+
iterations = 50
|
|
145
|
+
for _ in range(iterations):
|
|
146
|
+
await router.route(req, providers, policy=RoutingPolicy.LOCAL_FIRST)
|
|
147
|
+
router_overhead = ((time.perf_counter() - r_start) * 1000.0) / iterations
|
|
148
|
+
|
|
149
|
+
# 4. Telemetry Persistence Overhead Benchmark
|
|
150
|
+
import tempfile
|
|
151
|
+
test_db_path = db_path or Path(tempfile.mkdtemp()) / "bench_telemetry.db"
|
|
152
|
+
db = Database(test_db_path)
|
|
153
|
+
await db.initialize()
|
|
154
|
+
repo = SqliteTelemetryRepository(db.connection())
|
|
155
|
+
|
|
156
|
+
t_start = time.perf_counter()
|
|
157
|
+
tel_iterations = 20
|
|
158
|
+
for i in range(tel_iterations):
|
|
159
|
+
sample_tel = ModelInvocationTelemetry(
|
|
160
|
+
invocation_id=uuid4(),
|
|
161
|
+
provider_id="fake",
|
|
162
|
+
model_id="fake-default",
|
|
163
|
+
is_local=True,
|
|
164
|
+
candidate_context_tokens=candidate_tokens,
|
|
165
|
+
compiled_context_tokens=compiled_tokens,
|
|
166
|
+
context_tokens_avoided=max(0, candidate_tokens - compiled_tokens),
|
|
167
|
+
reduction_ratio=1.0 - (compiled_tokens / candidate_tokens) if candidate_tokens > 0 else 0.0,
|
|
168
|
+
token_measurement_source=TokenMeasurementSource.PROVIDER_REPORTED,
|
|
169
|
+
routing_policy=RoutingPolicy.LOCAL_FIRST,
|
|
170
|
+
routing_reason="benchmark",
|
|
171
|
+
selected_provider="fake",
|
|
172
|
+
selected_model="fake-default",
|
|
173
|
+
finish_reason=ModelFinishReason.STOP,
|
|
174
|
+
)
|
|
175
|
+
await repo.record(sample_tel)
|
|
176
|
+
telemetry_overhead = ((time.perf_counter() - t_start) * 1000.0) / tel_iterations
|
|
177
|
+
await db.close()
|
|
178
|
+
|
|
179
|
+
return Phase9BenchmarkReport(
|
|
180
|
+
candidate_context_tokens=candidate_tokens,
|
|
181
|
+
compiled_context_tokens=compiled_tokens,
|
|
182
|
+
profiles=profiles,
|
|
183
|
+
router_overhead_ms=router_overhead,
|
|
184
|
+
telemetry_overhead_ms=telemetry_overhead,
|
|
185
|
+
token_counting_overhead_ms=tc_overhead,
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
if __name__ == "__main__":
|
|
190
|
+
report = asyncio.run(run_phase9_benchmark())
|
|
191
|
+
print("\n=== CONTEXTOS PHASE 9 BENCHMARK REPORT ===")
|
|
192
|
+
print(f"Candidate Context Tokens: {report.candidate_context_tokens}")
|
|
193
|
+
print(f"Compiled Context Tokens: {report.compiled_context_tokens}")
|
|
194
|
+
print(f"Router Overhead: {report.router_overhead_ms:.3f} ms")
|
|
195
|
+
print(f"Telemetry Write Overhead: {report.telemetry_overhead_ms:.3f} ms")
|
|
196
|
+
print(f"Token Counting Overhead: {report.token_counting_overhead_ms:.3f} ms")
|
|
197
|
+
print("\n--- Target Model Tokenizer Profiles (Synthetic Profiles — NOT Actual Vendor Benchmarks) ---")
|
|
198
|
+
for name, p in report.profiles.items():
|
|
199
|
+
print(
|
|
200
|
+
f" {name:20s}: tokens={p.target_token_estimate:4d} | "
|
|
201
|
+
f"reduction={p.reduction_ratio * 100.0:.1f}% | "
|
|
202
|
+
f"avoided={p.context_tokens_avoided:4d} | source={p.measurement_source}"
|
|
203
|
+
)
|