contextos-memory-runtime 1.0.0rc2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. contextos/__init__.py +3 -0
  2. contextos/__main__.py +6 -0
  3. contextos/api/__init__.py +1 -0
  4. contextos/api/routes/__init__.py +1 -0
  5. contextos/api/routes/desktop.py +322 -0
  6. contextos/api/routes/ingest.py +17 -0
  7. contextos/api/routes/memories.py +84 -0
  8. contextos/api/routes/models.py +81 -0
  9. contextos/api/routes/retrieval.py +89 -0
  10. contextos/api/routes/system.py +216 -0
  11. contextos/api/server.py +195 -0
  12. contextos/benchmarks/__init__.py +1 -0
  13. contextos/benchmarks/compilation.py +245 -0
  14. contextos/benchmarks/connectors.py +423 -0
  15. contextos/benchmarks/explainability.py +103 -0
  16. contextos/benchmarks/final.py +406 -0
  17. contextos/benchmarks/graph.py +310 -0
  18. contextos/benchmarks/graph_adversarial.py +525 -0
  19. contextos/benchmarks/mcp.py +324 -0
  20. contextos/benchmarks/model_routing.py +203 -0
  21. contextos/benchmarks/optimization.py +305 -0
  22. contextos/benchmarks/rescue_integration.py +127 -0
  23. contextos/benchmarks/retrieval.py +266 -0
  24. contextos/benchmarks/temporal.py +377 -0
  25. contextos/benchmarks/temporal_hotpath.py +76 -0
  26. contextos/benchmarks/terminal.py +62 -0
  27. contextos/cli/__init__.py +1 -0
  28. contextos/cli/app.py +932 -0
  29. contextos/cli/dashboard.py +174 -0
  30. contextos/cli/formatters.py +299 -0
  31. contextos/config/__init__.py +1 -0
  32. contextos/config/settings.py +160 -0
  33. contextos/connectors/__init__.py +6 -0
  34. contextos/connectors/fake.py +11 -0
  35. contextos/connectors/json_import.py +125 -0
  36. contextos/connectors/local_files.py +102 -0
  37. contextos/connectors/manager.py +293 -0
  38. contextos/connectors/models.py +62 -0
  39. contextos/connectors/protocols.py +11 -0
  40. contextos/core/__init__.py +103 -0
  41. contextos/core/enums.py +489 -0
  42. contextos/core/exceptions.py +293 -0
  43. contextos/core/models.py +1147 -0
  44. contextos/core/protocols.py +549 -0
  45. contextos/daemon/__init__.py +1 -0
  46. contextos/daemon/manager.py +510 -0
  47. contextos/daemon/state.py +127 -0
  48. contextos/daemon/wiring.py +296 -0
  49. contextos/demo.py +217 -0
  50. contextos/embedding/__init__.py +1 -0
  51. contextos/embedding/deterministic.py +76 -0
  52. contextos/embedding/sentence_transformers.py +80 -0
  53. contextos/mcp/__init__.py +5 -0
  54. contextos/mcp/server.py +269 -0
  55. contextos/providers/__init__.py +13 -0
  56. contextos/providers/fake.py +217 -0
  57. contextos/providers/ollama.py +297 -0
  58. contextos/providers/openai_compatible.py +337 -0
  59. contextos/services/__init__.py +1 -0
  60. contextos/services/compilation.py +535 -0
  61. contextos/services/explainability.py +553 -0
  62. contextos/services/extraction.py +311 -0
  63. contextos/services/graph.py +524 -0
  64. contextos/services/graph_retrieval.py +143 -0
  65. contextos/services/ingestion.py +143 -0
  66. contextos/services/inspection.py +174 -0
  67. contextos/services/memory.py +291 -0
  68. contextos/services/model_service.py +409 -0
  69. contextos/services/optimization.py +426 -0
  70. contextos/services/privacy.py +331 -0
  71. contextos/services/retrieval.py +302 -0
  72. contextos/services/retrieval_index.py +88 -0
  73. contextos/services/router.py +302 -0
  74. contextos/services/secret_scanner.py +207 -0
  75. contextos/services/telemetry_query.py +102 -0
  76. contextos/services/temporal.py +500 -0
  77. contextos/services/token_counter.py +222 -0
  78. contextos/storage/__init__.py +1 -0
  79. contextos/storage/connector_repo.py +67 -0
  80. contextos/storage/database.py +497 -0
  81. contextos/storage/event_repo.py +137 -0
  82. contextos/storage/graph_repo.py +228 -0
  83. contextos/storage/lexical/__init__.py +1 -0
  84. contextos/storage/lexical/bm25.py +134 -0
  85. contextos/storage/memory_repo.py +589 -0
  86. contextos/storage/relation_repo.py +80 -0
  87. contextos/storage/telemetry_repo.py +481 -0
  88. contextos/storage/vector/__init__.py +1 -0
  89. contextos/storage/vector/in_memory.py +162 -0
  90. contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
  91. contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
  92. contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
  93. contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,324 @@
1
+ """Deterministic local MCP adapter measurement helpers.
2
+
3
+ Callers provide the already-wired deterministic services used by tests; this
4
+ module never creates a provider or sends network traffic.
5
+
6
+ Label: LOCAL DEVELOPMENT MACHINE SYNTHETIC INFRASTRUCTURE BENCHMARK
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ import statistics
13
+ import time
14
+ from dataclasses import dataclass, field
15
+ from typing import Any
16
+
17
+ from contextos.core.models import RetrievalQuery
18
+ from contextos.mcp.server import ContextOSMCPApplication, MCPPermissions
19
+
20
+
21
+ @dataclass(frozen=True)
22
+ class OperationResult:
23
+ """Timing and size for a single benchmark operation."""
24
+ operation: str
25
+ iterations: int
26
+ mean_ms: float
27
+ median_ms: float
28
+ p95_ms: float | None # Only with sufficient samples (>= 20)
29
+ serialized_bytes: int
30
+ # Operation-specific metrics
31
+ selected_memories: int | None = None
32
+ compiled_facts: int | None = None
33
+ compiled_tokens: int | None = None
34
+ graph_nodes: int | None = None
35
+ privacy_decision: str | None = None
36
+
37
+
38
+ @dataclass(frozen=True)
39
+ class MCPBenchmarkReport:
40
+ """Complete Phase 10 benchmark report."""
41
+ label: str = "LOCAL DEVELOPMENT MACHINE SYNTHETIC INFRASTRUCTURE BENCHMARK"
42
+ direct_results: list[OperationResult] = field(default_factory=list)
43
+ mcp_results: list[OperationResult] = field(default_factory=list)
44
+ stdio_results: list[OperationResult] = field(default_factory=list)
45
+ overheads: dict[str, float] = field(default_factory=dict)
46
+
47
+
48
+ def _p95(samples: list[float]) -> float | None:
49
+ if len(samples) < 20:
50
+ return None
51
+ return sorted(samples)[int(len(samples) * 0.95)]
52
+
53
+
54
+ def _operation_result(
55
+ operation: str, samples: list[float], response: dict[str, object],
56
+ ) -> OperationResult:
57
+ serialized = len(json.dumps(response, sort_keys=True, default=str).encode("utf-8"))
58
+ return OperationResult(
59
+ operation=operation,
60
+ iterations=len(samples),
61
+ mean_ms=statistics.mean(samples),
62
+ median_ms=statistics.median(samples),
63
+ p95_ms=_p95(samples),
64
+ serialized_bytes=serialized,
65
+ selected_memories=int(response.get("selected_memory_count", 0)) if "selected_memory_count" in response else (int(response.get("result_count", 0)) if response.get("ok") else None),
66
+ compiled_facts=int(response.get("compiled_fact_count", 0)) if "compiled_fact_count" in response else None,
67
+ compiled_tokens=int(response.get("token_count", 0)) if "token_count" in response else None,
68
+ graph_nodes=int(response.get("graph_node_count", 0)) if "graph_node_count" in response else None,
69
+ privacy_decision=str(response.get("error_code")) if response.get("error_code") == "PRIVACY_REJECTED" else None,
70
+ )
71
+
72
+
73
+ async def _measure(coro_factory, iterations: int) -> tuple[list[float], Any]:
74
+ """Run a coroutine factory N times and collect timings."""
75
+ samples: list[float] = []
76
+ last_result = None
77
+ for _ in range(iterations):
78
+ started = time.perf_counter()
79
+ last_result = await coro_factory()
80
+ elapsed = (time.perf_counter() - started) * 1000
81
+ samples.append(elapsed)
82
+ return samples, last_result
83
+
84
+
85
+ async def run_phase10_mcp_benchmark(
86
+ services: dict[str, Any],
87
+ query: str = "ContextOS memory",
88
+ iterations: int = 5,
89
+ ) -> MCPBenchmarkReport:
90
+ """Measure direct retrieval versus the local adapter using identical services.
91
+
92
+ Exercises: search, compile, graph, remember success, remember privacy
93
+ rejection, and temporal lookup.
94
+ """
95
+ app = ContextOSMCPApplication(services, MCPPermissions(allow_write=True))
96
+ direct_results: list[OperationResult] = []
97
+ mcp_results: list[OperationResult] = []
98
+ overheads: dict[str, float] = {}
99
+
100
+ # --- Direct service measurements ---
101
+
102
+ # Direct search
103
+ samples, last = await _measure(
104
+ lambda: services["retrieval"].retrieve(RetrievalQuery(text=query, k=5)),
105
+ iterations,
106
+ )
107
+ direct_results.append(OperationResult(
108
+ operation="search", iterations=len(samples),
109
+ mean_ms=statistics.mean(samples), median_ms=statistics.median(samples),
110
+ p95_ms=_p95(samples), serialized_bytes=0,
111
+ selected_memories=len(last.memories) if last else 0,
112
+ ))
113
+
114
+ # --- MCP adapter measurements ---
115
+
116
+ # MCP search
117
+ samples, last = await _measure(
118
+ lambda: app.invoke("contextos_search_memory", None,
119
+ lambda: app.search(query, 5, "hybrid", False)),
120
+ iterations,
121
+ )
122
+ mcp_results.append(_operation_result("search", samples, last))
123
+ overheads["search_adapter_ms"] = max(0.0, mcp_results[-1].mean_ms - direct_results[0].mean_ms)
124
+
125
+ # MCP compile
126
+ samples, last = await _measure(
127
+ lambda: app.invoke("contextos_compile_context", None,
128
+ lambda: app.compile(query, 1000, "hybrid")),
129
+ iterations,
130
+ )
131
+ mcp_results.append(_operation_result("compile", samples, last))
132
+
133
+ # MCP graph
134
+ samples, last = await _measure(
135
+ lambda: app.invoke("contextos_graph_neighbors", None,
136
+ lambda: app.graph_neighbors(query, 1, 50, 100)),
137
+ iterations,
138
+ )
139
+ mcp_results.append(_operation_result("graph", samples, last))
140
+
141
+ # MCP remember (success)
142
+ samples, last = await _measure(
143
+ lambda: app.invoke("contextos_remember", None,
144
+ lambda: app.remember(f"Benchmark memory about {query}")),
145
+ iterations,
146
+ )
147
+ mcp_results.append(_operation_result("remember_success", samples, last))
148
+
149
+ # MCP temporal lookup
150
+ samples, last = await _measure(
151
+ lambda: app.invoke("contextos_current_state", None,
152
+ lambda: app.current_state("programming_language", "user", "global")),
153
+ iterations,
154
+ )
155
+ mcp_results.append(_operation_result("temporal_lookup", samples, last))
156
+
157
+ # MCP telemetry
158
+ samples, last = await _measure(
159
+ lambda: app.invoke("contextos_telemetry_summary", None,
160
+ app.telemetry_summary),
161
+ iterations,
162
+ )
163
+ mcp_results.append(_operation_result("telemetry", samples, last))
164
+
165
+ return MCPBenchmarkReport(
166
+ direct_results=direct_results,
167
+ mcp_results=mcp_results,
168
+ overheads=overheads,
169
+ )
170
+
171
+
172
+ async def main() -> None:
173
+ """Run the canonical Phase 10 MCP benchmark against a temporary SQLite DB."""
174
+ import asyncio
175
+ import tempfile
176
+ from pathlib import Path
177
+ from mcp import Client
178
+ from contextos.core.enums import SecretDetectionMode
179
+ from contextos.embedding.deterministic import DeterministicEmbedding
180
+ from contextos.mcp.server import MCPPermissions, create_mcp_server
181
+ from contextos.services.compilation import QueryAwareContextCompiler
182
+ from contextos.services.extraction import RuleBasedMemoryExtractor
183
+ from contextos.services.graph import MemoryGraphService
184
+ from contextos.services.graph_retrieval import GraphAugmentedRetrievalEngine
185
+ from contextos.services.ingestion import IngestionPipeline
186
+ from contextos.services.optimization import MemoryContextOptimizer
187
+ from contextos.services.retrieval import HybridRetrievalEngine
188
+ from contextos.services.retrieval_index import RetrievalIndexSynchronizer
189
+ from contextos.services.secret_scanner import PatternSecretScanner
190
+ from contextos.services.telemetry_query import TelemetryQueryService
191
+ from contextos.services.temporal import TemporalMemoryService
192
+ from contextos.services.token_counter import DeterministicWordTokenCounter
193
+ from contextos.storage.database import Database
194
+ from contextos.storage.event_repo import SqliteEventRepository
195
+ from contextos.storage.graph_repo import SqliteGraphRepository
196
+ from contextos.storage.lexical.bm25 import BM25Index
197
+ from contextos.storage.memory_repo import SqliteMemoryRepository
198
+ from contextos.storage.relation_repo import SqliteRelationRepository
199
+ from contextos.storage.telemetry_repo import SqliteTelemetryRepository
200
+ from contextos.storage.vector.in_memory import InMemoryVectorStore
201
+
202
+ tmp_dir = Path(tempfile.mkdtemp())
203
+ db_path = tmp_dir / "bench.db"
204
+ db = Database(db_path)
205
+ await db.initialize()
206
+ conn = db.connection()
207
+ memory_repo = SqliteMemoryRepository(conn)
208
+ event_repo = SqliteEventRepository(conn)
209
+ relation_repo = SqliteRelationRepository(conn)
210
+ graph_repo = SqliteGraphRepository(conn)
211
+ telemetry_repo = SqliteTelemetryRepository(conn)
212
+ embedding = DeterministicEmbedding(16)
213
+ lexical, vector = BM25Index(), InMemoryVectorStore(16)
214
+ index = RetrievalIndexSynchronizer(
215
+ memory_repo=memory_repo, embedding_service=embedding,
216
+ vector_store=vector, lexical_index=lexical,
217
+ )
218
+ graph = MemoryGraphService(
219
+ memory_repo=memory_repo, relation_repo=relation_repo,
220
+ graph_repo=graph_repo,
221
+ )
222
+ retrieval = GraphAugmentedRetrievalEngine(
223
+ base_engine=HybridRetrievalEngine(
224
+ memory_repo=memory_repo, vector_store=vector,
225
+ lexical_index=lexical, embedding_service=embedding,
226
+ index_synchronizer=index,
227
+ ),
228
+ graph_service=graph, memory_repo=memory_repo,
229
+ )
230
+ token_counter = DeterministicWordTokenCounter()
231
+ temporal = TemporalMemoryService(memory_repo)
232
+ ingestion = IngestionPipeline(
233
+ secret_scanner=PatternSecretScanner(),
234
+ memory_extractor=RuleBasedMemoryExtractor(),
235
+ memory_repo=memory_repo, event_repo=event_repo,
236
+ embedding_service=embedding, vector_store=vector,
237
+ lexical_index=lexical, token_counter=token_counter,
238
+ secret_detection_mode=SecretDetectionMode.STRICT,
239
+ )
240
+ services = {
241
+ "database": db,
242
+ "retrieval": retrieval,
243
+ "optimizer": MemoryContextOptimizer(token_counter=token_counter),
244
+ "compilation": QueryAwareContextCompiler(token_counter=token_counter),
245
+ "graph": graph,
246
+ "temporal": temporal,
247
+ "ingestion": ingestion,
248
+ "telemetry_query": TelemetryQueryService(telemetry_repo),
249
+ }
250
+
251
+ # Seed data using MCP adapter
252
+ server = create_mcp_server(services, MCPPermissions(allow_write=True))
253
+ async with Client(server) as client:
254
+ seed_data = [
255
+ "I am working on Project Atlas. Project Atlas uses Python for machine learning.",
256
+ "I am working on Project Beta. Project Beta uses Rust for systems programming.",
257
+ "I currently use Ollama for local inference.",
258
+ "I am working on Project Atlas. Project Atlas uses llama.cpp.",
259
+ "I prefer concise answers for technical questions.",
260
+ "I use Docker for containerization.",
261
+ "I am working on Project Gamma. Project Gamma uses PostgreSQL.",
262
+ "I am learning TypeScript for web development.",
263
+ ]
264
+ for text in seed_data:
265
+ await client.call_tool("contextos_remember", {"text": text})
266
+
267
+ # Run benchmark
268
+ report = await run_phase10_mcp_benchmark(services, query="Atlas Python machine learning", iterations=5)
269
+
270
+ # Print results
271
+ print("=" * 70)
272
+ print("LOCAL DEVELOPMENT MACHINE SYNTHETIC INFRASTRUCTURE BENCHMARK")
273
+ print("=" * 70)
274
+ print()
275
+ print("DIRECT SERVICE RESULTS:")
276
+ print("-" * 50)
277
+ for r in report.direct_results:
278
+ print(f" {r.operation}:")
279
+ print(f" iterations: {r.iterations}")
280
+ print(f" mean latency: {r.mean_ms:.3f} ms")
281
+ print(f" median latency: {r.median_ms:.3f} ms")
282
+ if r.p95_ms is not None:
283
+ print(f" p95 latency: {r.p95_ms:.3f} ms")
284
+ print(f" serialized bytes: {r.serialized_bytes}")
285
+ if r.selected_memories is not None:
286
+ print(f" selected memories: {r.selected_memories}")
287
+ print()
288
+
289
+ print("MCP ADAPTER RESULTS:")
290
+ print("-" * 50)
291
+ for r in report.mcp_results:
292
+ print(f" {r.operation}:")
293
+ print(f" iterations: {r.iterations}")
294
+ print(f" mean latency: {r.mean_ms:.3f} ms")
295
+ print(f" median latency: {r.median_ms:.3f} ms")
296
+ if r.p95_ms is not None:
297
+ print(f" p95 latency: {r.p95_ms:.3f} ms")
298
+ print(f" serialized bytes: {r.serialized_bytes}")
299
+ if r.selected_memories is not None:
300
+ print(f" selected memories: {r.selected_memories}")
301
+ if r.compiled_facts is not None:
302
+ print(f" compiled facts: {r.compiled_facts}")
303
+ if r.compiled_tokens is not None:
304
+ print(f" compiled tokens: {r.compiled_tokens}")
305
+ if r.graph_nodes is not None:
306
+ print(f" graph nodes: {r.graph_nodes}")
307
+ if r.privacy_decision is not None:
308
+ print(f" privacy decision: {r.privacy_decision}")
309
+ print()
310
+
311
+ print("OVERHEAD CALCULATIONS:")
312
+ print("-" * 50)
313
+ for key, value in report.overheads.items():
314
+ print(f" {key}: {value:.3f} ms")
315
+ print()
316
+
317
+ await db.close()
318
+ print("Benchmark complete.")
319
+
320
+
321
+ if __name__ == "__main__":
322
+ import asyncio
323
+ asyncio.run(main())
324
+
@@ -0,0 +1,203 @@
1
+ """Deterministic Phase 9 benchmark: Model routing, multi-model token comparison, and telemetry overhead."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ import time
7
+ from dataclasses import dataclass
8
+ from pathlib import Path
9
+ from uuid import uuid4
10
+
11
+ from contextos.core.enums import (
12
+ MemoryStatus,
13
+ ModelFinishReason,
14
+ RoutingPolicy,
15
+ TokenMeasurementSource,
16
+ )
17
+ from contextos.core.models import (
18
+ CompilationConfig,
19
+ ContextBudget,
20
+ Memory,
21
+ ModelCapabilities,
22
+ ModelInvocationTelemetry,
23
+ ModelRequest,
24
+ ScoredMemory,
25
+ )
26
+ from contextos.providers.fake import DeterministicFakeProvider
27
+ from contextos.services.compilation import QueryAwareContextCompiler
28
+ from contextos.services.optimization import MemoryContextOptimizer
29
+ from contextos.services.router import DeterministicModelRouter
30
+ from contextos.services.token_counter import (
31
+ ClaudeProfileTokenCounter,
32
+ QwenProfileTokenCounter,
33
+ TiktokenCounter,
34
+ get_token_counter_for_model,
35
+ recount_cross_model,
36
+ )
37
+ from contextos.storage.database import Database
38
+ from contextos.storage.telemetry_repo import SqliteTelemetryRepository
39
+
40
+
41
+ @dataclass(frozen=True)
42
+ class ModelProfileResult:
43
+ profile_name: str
44
+ target_token_estimate: int
45
+ measurement_source: str
46
+ reduction_ratio: float
47
+ context_tokens_avoided: int
48
+
49
+
50
+ @dataclass(frozen=True)
51
+ class Phase9BenchmarkReport:
52
+ candidate_context_tokens: int
53
+ compiled_context_tokens: int
54
+ profiles: dict[str, ModelProfileResult]
55
+ router_overhead_ms: float
56
+ telemetry_overhead_ms: float
57
+ token_counting_overhead_ms: float
58
+
59
+
60
+ def benchmark_memories() -> list[ScoredMemory]:
61
+ """Diverse benchmark corpus containing code, technical preferences, and project facts."""
62
+ texts = [
63
+ "User currently uses Ollama for local model inference with llama3.2.",
64
+ "User prefers concise technical responses with typed Python 3.12+ syntax and pydantic models.",
65
+ "Project Atlas architecture uses SQLite with WAL mode, graph projection, and reciprocal rank fusion.",
66
+ "User's machine is an M2 Max with 64GB unified memory running macOS Sonoma.",
67
+ "User previously attempted running Qwen-72B locally, but inference failed due to high memory pressure.",
68
+ "Development setup: pytest for testing, ruff for linting, and mypy in strict mode.",
69
+ "ContextOS compilation pipeline achieves token reduction while preserving exact fact attribution.",
70
+ "The local LLM endpoint is hosted at http://127.0.0.1:11434 with zero external telemetry transmission.",
71
+ ]
72
+ memories = [
73
+ Memory(
74
+ id=uuid4(),
75
+ content=text,
76
+ status=MemoryStatus.ACTIVE,
77
+ token_count=len(text.split()),
78
+ )
79
+ for text in texts
80
+ ]
81
+ return [
82
+ ScoredMemory(
83
+ memory=m,
84
+ final_score=0.95 - (i * 0.05),
85
+ retrieval_sources=["lexical", "dense"] if i % 2 == 0 else ["lexical", "graph"],
86
+ )
87
+ for i, m in enumerate(memories)
88
+ ]
89
+
90
+
91
+ async def run_phase9_benchmark(db_path: Path | None = None) -> Phase9BenchmarkReport:
92
+ """Run deterministic Phase 9 benchmark and return structured metrics."""
93
+ ref_counter = TiktokenCounter("cl100k_base")
94
+ optimizer = MemoryContextOptimizer(token_counter=ref_counter)
95
+ compiler = QueryAwareContextCompiler(token_counter=ref_counter)
96
+
97
+ candidates = benchmark_memories()
98
+ candidate_tokens = sum(ref_counter.count(c.memory.content) for c in candidates)
99
+
100
+ # 1. Optimize and Compile
101
+ query = "What is the recommended local model and hardware configuration for Project Atlas?"
102
+ selection = optimizer.optimize(
103
+ query=query,
104
+ candidates=candidates,
105
+ budget=ContextBudget(max_tokens=2000),
106
+ )
107
+ compiled = await compiler.compile(
108
+ query=query,
109
+ memories=selection,
110
+ config=CompilationConfig(budget=1500),
111
+ )
112
+ compiled_tokens = compiled.total_tokens
113
+
114
+ # 2. Multi-Model Token Recounting (Profiles)
115
+ tc_start = time.perf_counter()
116
+ recounts = recount_cross_model(
117
+ compiled.context_text,
118
+ ["FAKE_CLAUDE_LIKE", "FAKE_OPENAI_LIKE", "FAKE_QWEN_LIKE"],
119
+ )
120
+ tc_overhead = (time.perf_counter() - tc_start) * 1000.0
121
+
122
+ profiles: dict[str, ModelProfileResult] = {}
123
+ for prof_name, (count, src) in recounts.items():
124
+ # Context tokens avoided from perspective of this model profile
125
+ prof_counter = get_token_counter_for_model(prof_name, prof_name)
126
+ cand_for_model = sum(prof_counter.count(c.memory.content) for c in candidates)
127
+ avoided = max(0, cand_for_model - count)
128
+ red_ratio = max(0.0, 1.0 - (count / cand_for_model)) if cand_for_model > 0 else 0.0
129
+ profiles[prof_name] = ModelProfileResult(
130
+ profile_name=prof_name,
131
+ target_token_estimate=count,
132
+ measurement_source=src.value,
133
+ reduction_ratio=red_ratio,
134
+ context_tokens_avoided=avoided,
135
+ )
136
+
137
+ # 3. Router Overhead Benchmark
138
+ router = DeterministicModelRouter()
139
+ fake_prov = DeterministicFakeProvider()
140
+ providers = {fake_prov.provider_id: fake_prov}
141
+ req = ModelRequest(user_prompt=query, compiled_context=compiled)
142
+
143
+ r_start = time.perf_counter()
144
+ iterations = 50
145
+ for _ in range(iterations):
146
+ await router.route(req, providers, policy=RoutingPolicy.LOCAL_FIRST)
147
+ router_overhead = ((time.perf_counter() - r_start) * 1000.0) / iterations
148
+
149
+ # 4. Telemetry Persistence Overhead Benchmark
150
+ import tempfile
151
+ test_db_path = db_path or Path(tempfile.mkdtemp()) / "bench_telemetry.db"
152
+ db = Database(test_db_path)
153
+ await db.initialize()
154
+ repo = SqliteTelemetryRepository(db.connection())
155
+
156
+ t_start = time.perf_counter()
157
+ tel_iterations = 20
158
+ for i in range(tel_iterations):
159
+ sample_tel = ModelInvocationTelemetry(
160
+ invocation_id=uuid4(),
161
+ provider_id="fake",
162
+ model_id="fake-default",
163
+ is_local=True,
164
+ candidate_context_tokens=candidate_tokens,
165
+ compiled_context_tokens=compiled_tokens,
166
+ context_tokens_avoided=max(0, candidate_tokens - compiled_tokens),
167
+ reduction_ratio=1.0 - (compiled_tokens / candidate_tokens) if candidate_tokens > 0 else 0.0,
168
+ token_measurement_source=TokenMeasurementSource.PROVIDER_REPORTED,
169
+ routing_policy=RoutingPolicy.LOCAL_FIRST,
170
+ routing_reason="benchmark",
171
+ selected_provider="fake",
172
+ selected_model="fake-default",
173
+ finish_reason=ModelFinishReason.STOP,
174
+ )
175
+ await repo.record(sample_tel)
176
+ telemetry_overhead = ((time.perf_counter() - t_start) * 1000.0) / tel_iterations
177
+ await db.close()
178
+
179
+ return Phase9BenchmarkReport(
180
+ candidate_context_tokens=candidate_tokens,
181
+ compiled_context_tokens=compiled_tokens,
182
+ profiles=profiles,
183
+ router_overhead_ms=router_overhead,
184
+ telemetry_overhead_ms=telemetry_overhead,
185
+ token_counting_overhead_ms=tc_overhead,
186
+ )
187
+
188
+
189
+ if __name__ == "__main__":
190
+ report = asyncio.run(run_phase9_benchmark())
191
+ print("\n=== CONTEXTOS PHASE 9 BENCHMARK REPORT ===")
192
+ print(f"Candidate Context Tokens: {report.candidate_context_tokens}")
193
+ print(f"Compiled Context Tokens: {report.compiled_context_tokens}")
194
+ print(f"Router Overhead: {report.router_overhead_ms:.3f} ms")
195
+ print(f"Telemetry Write Overhead: {report.telemetry_overhead_ms:.3f} ms")
196
+ print(f"Token Counting Overhead: {report.token_counting_overhead_ms:.3f} ms")
197
+ print("\n--- Target Model Tokenizer Profiles (Synthetic Profiles — NOT Actual Vendor Benchmarks) ---")
198
+ for name, p in report.profiles.items():
199
+ print(
200
+ f" {name:20s}: tokens={p.target_token_estimate:4d} | "
201
+ f"reduction={p.reduction_ratio * 100.0:.1f}% | "
202
+ f"avoided={p.context_tokens_avoided:4d} | source={p.measurement_source}"
203
+ )