contextos-memory-runtime 1.0.0rc2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. contextos/__init__.py +3 -0
  2. contextos/__main__.py +6 -0
  3. contextos/api/__init__.py +1 -0
  4. contextos/api/routes/__init__.py +1 -0
  5. contextos/api/routes/desktop.py +322 -0
  6. contextos/api/routes/ingest.py +17 -0
  7. contextos/api/routes/memories.py +84 -0
  8. contextos/api/routes/models.py +81 -0
  9. contextos/api/routes/retrieval.py +89 -0
  10. contextos/api/routes/system.py +216 -0
  11. contextos/api/server.py +195 -0
  12. contextos/benchmarks/__init__.py +1 -0
  13. contextos/benchmarks/compilation.py +245 -0
  14. contextos/benchmarks/connectors.py +423 -0
  15. contextos/benchmarks/explainability.py +103 -0
  16. contextos/benchmarks/final.py +406 -0
  17. contextos/benchmarks/graph.py +310 -0
  18. contextos/benchmarks/graph_adversarial.py +525 -0
  19. contextos/benchmarks/mcp.py +324 -0
  20. contextos/benchmarks/model_routing.py +203 -0
  21. contextos/benchmarks/optimization.py +305 -0
  22. contextos/benchmarks/rescue_integration.py +127 -0
  23. contextos/benchmarks/retrieval.py +266 -0
  24. contextos/benchmarks/temporal.py +377 -0
  25. contextos/benchmarks/temporal_hotpath.py +76 -0
  26. contextos/benchmarks/terminal.py +62 -0
  27. contextos/cli/__init__.py +1 -0
  28. contextos/cli/app.py +932 -0
  29. contextos/cli/dashboard.py +174 -0
  30. contextos/cli/formatters.py +299 -0
  31. contextos/config/__init__.py +1 -0
  32. contextos/config/settings.py +160 -0
  33. contextos/connectors/__init__.py +6 -0
  34. contextos/connectors/fake.py +11 -0
  35. contextos/connectors/json_import.py +125 -0
  36. contextos/connectors/local_files.py +102 -0
  37. contextos/connectors/manager.py +293 -0
  38. contextos/connectors/models.py +62 -0
  39. contextos/connectors/protocols.py +11 -0
  40. contextos/core/__init__.py +103 -0
  41. contextos/core/enums.py +489 -0
  42. contextos/core/exceptions.py +293 -0
  43. contextos/core/models.py +1147 -0
  44. contextos/core/protocols.py +549 -0
  45. contextos/daemon/__init__.py +1 -0
  46. contextos/daemon/manager.py +510 -0
  47. contextos/daemon/state.py +127 -0
  48. contextos/daemon/wiring.py +296 -0
  49. contextos/demo.py +217 -0
  50. contextos/embedding/__init__.py +1 -0
  51. contextos/embedding/deterministic.py +76 -0
  52. contextos/embedding/sentence_transformers.py +80 -0
  53. contextos/mcp/__init__.py +5 -0
  54. contextos/mcp/server.py +269 -0
  55. contextos/providers/__init__.py +13 -0
  56. contextos/providers/fake.py +217 -0
  57. contextos/providers/ollama.py +297 -0
  58. contextos/providers/openai_compatible.py +337 -0
  59. contextos/services/__init__.py +1 -0
  60. contextos/services/compilation.py +535 -0
  61. contextos/services/explainability.py +553 -0
  62. contextos/services/extraction.py +311 -0
  63. contextos/services/graph.py +524 -0
  64. contextos/services/graph_retrieval.py +143 -0
  65. contextos/services/ingestion.py +143 -0
  66. contextos/services/inspection.py +174 -0
  67. contextos/services/memory.py +291 -0
  68. contextos/services/model_service.py +409 -0
  69. contextos/services/optimization.py +426 -0
  70. contextos/services/privacy.py +331 -0
  71. contextos/services/retrieval.py +302 -0
  72. contextos/services/retrieval_index.py +88 -0
  73. contextos/services/router.py +302 -0
  74. contextos/services/secret_scanner.py +207 -0
  75. contextos/services/telemetry_query.py +102 -0
  76. contextos/services/temporal.py +500 -0
  77. contextos/services/token_counter.py +222 -0
  78. contextos/storage/__init__.py +1 -0
  79. contextos/storage/connector_repo.py +67 -0
  80. contextos/storage/database.py +497 -0
  81. contextos/storage/event_repo.py +137 -0
  82. contextos/storage/graph_repo.py +228 -0
  83. contextos/storage/lexical/__init__.py +1 -0
  84. contextos/storage/lexical/bm25.py +134 -0
  85. contextos/storage/memory_repo.py +589 -0
  86. contextos/storage/relation_repo.py +80 -0
  87. contextos/storage/telemetry_repo.py +481 -0
  88. contextos/storage/vector/__init__.py +1 -0
  89. contextos/storage/vector/in_memory.py +162 -0
  90. contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
  91. contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
  92. contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
  93. contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,216 @@
1
+ """System status and diagnostics API routes."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ import time
7
+ import asyncio
8
+
9
+ from fastapi import APIRouter
10
+
11
+ from contextos import __version__
12
+ from contextos.api.server import get_service
13
+ from contextos.core.models import MemoryFilters, SystemStatus, TokenStats
14
+ from contextos.core.enums import MemoryStatus
15
+
16
+ router = APIRouter(tags=["system"])
17
+
18
+ _start_time = time.time()
19
+
20
+
21
+ @router.get("/status", response_model=SystemStatus)
22
+ async def status() -> SystemStatus:
23
+ """Get system health and status."""
24
+ database = get_service("database")
25
+ embedding_service = get_service("embedding")
26
+ vector_store = get_service("vector_store")
27
+ bm25_index = get_service("bm25_index")
28
+
29
+ # Get counts
30
+ from contextos.core.protocols import MemoryRepository
31
+ memory_repo: MemoryRepository = get_service("memory_repo")
32
+ total_count = await memory_repo.count()
33
+ active_count = await memory_repo.count(MemoryFilters(status=MemoryStatus.ACTIVE))
34
+
35
+ from contextos.core.protocols import EventRepository
36
+ event_repo: EventRepository = get_service("event_repo")
37
+ event_count = await event_repo.count()
38
+
39
+ vector_count = await vector_store.count()
40
+ bm25_count = await bm25_index.count()
41
+
42
+ db_size = await database.get_size_bytes()
43
+
44
+ return SystemStatus(
45
+ daemon_running=True,
46
+ pid=os.getpid(),
47
+ uptime_seconds=time.time() - _start_time,
48
+ total_memories=total_count,
49
+ active_memories=active_count,
50
+ total_events=event_count,
51
+ embedding_model=embedding_service.model_name,
52
+ embedding_model_loaded=True,
53
+ vector_index_size=vector_count,
54
+ bm25_index_size=bm25_count,
55
+ database_size_bytes=db_size,
56
+ data_directory="[LOCAL]",
57
+ )
58
+
59
+
60
+ @router.get("/stats", response_model=TokenStats)
61
+ async def stats() -> TokenStats:
62
+ """Report stored counts and observed model compilations with provenance."""
63
+ conn = get_service("database").connection()
64
+ cursor = await conn.execute(
65
+ "SELECT COALESCE(SUM(token_count), 0), COUNT(*) FROM memories WHERE status = ?",
66
+ (MemoryStatus.ACTIVE.value,),
67
+ )
68
+ total_tokens, memory_count = await cursor.fetchone()
69
+ bases = await get_service("telemetry_query").context_measurement_bases()
70
+ basis = bases[0] if len(bases) == 1 and bases[0]["source"] != "unknown" else None
71
+ if basis is not None:
72
+ cursor = await conn.execute(
73
+ "SELECT COUNT(*), COALESCE(SUM(compiled_context_tokens), 0), "
74
+ "COALESCE(SUM(context_tokens_avoided), 0), "
75
+ "COALESCE(SUM(candidate_context_tokens), 0) "
76
+ "FROM model_invocations WHERE status = 'success'"
77
+ )
78
+ runs, compiled, avoided, candidates = await cursor.fetchone()
79
+ else:
80
+ runs = compiled = avoided = candidates = None
81
+ from contextos.api.routes.desktop import _public_label
82
+
83
+ return TokenStats(
84
+ total_tokens_stored=total_tokens,
85
+ tokens_per_memory=total_tokens / memory_count if memory_count else 0,
86
+ # Standalone compilations are not persisted; these are successful model runs.
87
+ total_compilations=runs,
88
+ total_tokens_compiled=compiled,
89
+ total_tokens_saved=avoided,
90
+ average_compression_ratio=avoided / candidates if candidates else None,
91
+ measurement_basis=f"{basis['source']}:{_public_label(basis['tokenizer'])}" if basis else None,
92
+ )
93
+
94
+
95
+ @router.post("/doctor")
96
+ async def doctor() -> dict:
97
+ """Run diagnostic checks."""
98
+ results: dict = {
99
+ "version": __version__,
100
+ "checks": {},
101
+ }
102
+
103
+ # Database integrity
104
+ database = get_service("database")
105
+ ok, msg = await database.integrity_check()
106
+ results["checks"]["database_integrity"] = {"ok": ok, "message": msg}
107
+
108
+ from contextos.storage.database import SCHEMA_VERSION
109
+ conn = database.connection()
110
+ cursor = await conn.execute("SELECT MAX(version) FROM schema_version")
111
+ version_row = await cursor.fetchone()
112
+ schema_version = version_row[0] if version_row else None
113
+ results["checks"]["schema"] = {
114
+ "ok": schema_version == SCHEMA_VERSION,
115
+ "version": schema_version, "expected": SCHEMA_VERSION,
116
+ }
117
+ cursor = await conn.execute(
118
+ "SELECT name FROM sqlite_master WHERE type = 'table' AND name IN "
119
+ "('memories', 'memory_relations', 'model_invocations', 'connector_state', 'graph_nodes', 'graph_edges')"
120
+ )
121
+ tables = {row[0] for row in await cursor.fetchall()}
122
+ results["checks"]["tables"] = {"ok": len(tables) == 6, "present": sorted(tables)}
123
+ cursor = await conn.execute("PRAGMA index_list('memories')")
124
+ indexes = {row[1] for row in await cursor.fetchall()}
125
+ required_indexes = {"idx_memories_status", "idx_memories_slot_key"}
126
+ results["checks"]["indexes"] = {
127
+ "ok": required_indexes <= indexes,
128
+ "required_present": sorted(indexes & required_indexes),
129
+ }
130
+ graph_repo = get_service("graph_repo")
131
+ graph_nodes, graph_edges, _ = await graph_repo.counts()
132
+ graph_dirty = await graph_repo.source_is_dirty()
133
+ results["checks"]["graph_projection"] = {
134
+ "ok": True, "dirty": graph_dirty, "nodes": graph_nodes, "edges": graph_edges,
135
+ "suggestion": "Graph rebuilds on the next graph retrieval" if graph_dirty else None,
136
+ }
137
+ results["checks"]["telemetry"] = {
138
+ "ok": True, "recorded_invocations": await get_service("telemetry_repo").count(),
139
+ }
140
+ connector_ids = get_service("connectors").list_connectors()
141
+ connector_states = await get_service("connector_repo").list_states()
142
+ results["checks"]["connectors"] = {
143
+ "ok": True, "configured": len(connector_ids),
144
+ "failed": sum(state.status == "failed" for state in connector_states),
145
+ }
146
+ settings = get_service("settings")
147
+ results["checks"]["local_configuration"] = {
148
+ "ok": settings.daemon.host in {"127.0.0.1", "localhost", "::1"}
149
+ and os.access(database.path.parent, os.R_OK | os.W_OK),
150
+ "data_directory_accessible": os.access(database.path.parent, os.R_OK | os.W_OK),
151
+ }
152
+ scan = get_service("secret_scanner").scan("Authorization: Bearer privatefixture1234567890")
153
+ results["checks"]["privacy_scanner"] = {"ok": bool(scan.matches)}
154
+ providers = get_service("providers")
155
+
156
+ async def available(provider):
157
+ try:
158
+ return bool(await asyncio.wait_for(provider.health(), timeout=1.0))
159
+ except Exception:
160
+ return False
161
+
162
+ healths = await asyncio.gather(*(available(provider) for provider in providers.values()))
163
+ results["checks"]["providers"] = {
164
+ "ok": any(healths), "configured": len(providers), "available": sum(healths),
165
+ "optional_unavailable": len(providers) - sum(healths),
166
+ }
167
+
168
+ # Embedding model
169
+ try:
170
+ embedding = get_service("embedding")
171
+ dim = embedding.dimension
172
+ results["checks"]["embedding_model"] = {
173
+ "ok": True,
174
+ "model": embedding.model_name,
175
+ "dimension": dim,
176
+ }
177
+ except Exception as e:
178
+ results["checks"]["embedding_model"] = {"ok": False, "error": e.__class__.__name__}
179
+
180
+ # Index consistency
181
+ memory_repo = get_service("memory_repo")
182
+ vector_store = get_service("vector_store")
183
+ bm25_index = get_service("bm25_index")
184
+
185
+ from contextos.services.retrieval_index import INDEXED_STATUSES
186
+ indexed_count = 0
187
+ for memory_status in INDEXED_STATUSES:
188
+ indexed_count += await memory_repo.count(MemoryFilters(status=memory_status))
189
+ vector_count = await vector_store.count()
190
+ bm25_count = await bm25_index.count()
191
+
192
+ results["checks"]["index_consistency"] = {
193
+ "ok": True,
194
+ "indexable_memories": indexed_count,
195
+ "vector_index": vector_count,
196
+ "bm25_index": bm25_count,
197
+ "warnings": [],
198
+ }
199
+
200
+ if vector_count != indexed_count:
201
+ results["checks"]["index_consistency"]["warnings"].append(
202
+ f"Vector index ({vector_count}) != indexable memories ({indexed_count})"
203
+ )
204
+ if bm25_count != indexed_count:
205
+ results["checks"]["index_consistency"]["warnings"].append(
206
+ f"BM25 index ({bm25_count}) != indexable memories ({indexed_count})"
207
+ )
208
+
209
+ if results["checks"]["index_consistency"]["warnings"]:
210
+ results["checks"]["index_consistency"]["ok"] = False
211
+
212
+ results["overall"] = all(
213
+ c.get("ok", False) for c in results["checks"].values()
214
+ )
215
+
216
+ return results
@@ -0,0 +1,195 @@
1
+ """FastAPI server setup for ContextOS.
2
+
3
+ This is the HTTP API that the CLI and future MCP server communicate with.
4
+ All business logic lives in the service layer — this is a thin translation
5
+ layer between HTTP and service protocol calls.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import logging
11
+ import time
12
+ from contextlib import asynccontextmanager
13
+ from typing import Any
14
+
15
+ from fastapi import FastAPI, HTTPException, Request
16
+ from fastapi.exceptions import RequestValidationError
17
+ from fastapi.responses import JSONResponse
18
+
19
+ from contextos import __version__
20
+ from contextos.core.exceptions import (
21
+ ContextOSError,
22
+ InvalidTransitionError,
23
+ MemoryNotFoundError,
24
+ SecretDetectedError,
25
+ )
26
+
27
+ logger = logging.getLogger(__name__)
28
+
29
+ # Service instances — set by daemon wiring before startup
30
+ _services: dict[str, Any] = {}
31
+
32
+
33
+ def set_services(services: dict[str, Any]) -> None:
34
+ """Set service instances. Called by daemon wiring at startup."""
35
+ _services.update(services)
36
+
37
+
38
+ def get_service(name: str) -> Any:
39
+ """Get a service by name. Raises if not set."""
40
+ if name not in _services:
41
+ raise RuntimeError(f"Service '{name}' not initialized")
42
+ return _services[name]
43
+
44
+
45
+ def _dispatch_error_payload(exc: Exception) -> dict[str, Any]:
46
+ """Expose bounded dispatch evidence without echoing provider request content."""
47
+ evidence = getattr(exc, "dispatch_evidence", None)
48
+ if evidence is None:
49
+ return {}
50
+ return {"dispatch_evidence": evidence.model_dump(mode="json")}
51
+
52
+
53
+ @asynccontextmanager
54
+ async def lifespan(app: FastAPI):
55
+ """Application lifespan handler."""
56
+ logger.info("ContextOS API server starting (v%s)", __version__)
57
+ yield
58
+ logger.info("ContextOS API server shutting down")
59
+
60
+
61
+ def create_app() -> FastAPI:
62
+ """Create and configure the FastAPI application."""
63
+ app = FastAPI(
64
+ title="ContextOS",
65
+ description="Local-first personal AI memory runtime",
66
+ version=__version__,
67
+ lifespan=lifespan,
68
+ docs_url="/api/docs",
69
+ redoc_url="/api/redoc",
70
+ )
71
+
72
+ # --- Middleware ---
73
+
74
+ @app.middleware("http")
75
+ async def request_timing(request: Request, call_next):
76
+ """Log request timing."""
77
+ start = time.perf_counter()
78
+ response = await call_next(request)
79
+ elapsed = (time.perf_counter() - start) * 1000
80
+ logger.debug(
81
+ "%s %s — %d (%.1fms)",
82
+ request.method, request.url.path, response.status_code, elapsed,
83
+ )
84
+ return response
85
+
86
+ # --- Exception Handlers ---
87
+
88
+ @app.exception_handler(RequestValidationError)
89
+ async def request_validation_handler(request: Request, exc: RequestValidationError):
90
+ """Return validation structure without echoing attacker-controlled values."""
91
+ safe_errors = [
92
+ {key: value for key, value in error.items() if key not in {"input", "ctx"}}
93
+ for error in exc.errors()
94
+ ]
95
+ return JSONResponse(status_code=422, content={"detail": safe_errors})
96
+
97
+ @app.exception_handler(MemoryNotFoundError)
98
+ async def memory_not_found_handler(request: Request, exc: MemoryNotFoundError):
99
+ return JSONResponse(status_code=404, content={"error": str(exc)})
100
+
101
+ @app.exception_handler(InvalidTransitionError)
102
+ async def invalid_transition_handler(request: Request, exc: InvalidTransitionError):
103
+ return JSONResponse(status_code=422, content={"error": str(exc)})
104
+
105
+ @app.exception_handler(SecretDetectedError)
106
+ async def secret_detected_handler(request: Request, exc: SecretDetectedError):
107
+ return JSONResponse(
108
+ status_code=422,
109
+ content={
110
+ "error": str(exc),
111
+ "secret_types": exc.secret_types,
112
+ },
113
+ )
114
+
115
+ from contextos.core.exceptions import (
116
+ ContextWindowExceededError,
117
+ MalformedProviderResponseError,
118
+ ModelUnavailableError,
119
+ ProviderAuthenticationError,
120
+ ProviderRateLimitError,
121
+ ProviderTimeoutError,
122
+ ProviderUnavailableError,
123
+ RoutingFailureError,
124
+ )
125
+
126
+ @app.exception_handler(ProviderUnavailableError)
127
+ async def provider_unavailable_handler(request: Request, exc: ProviderUnavailableError):
128
+ return JSONResponse(status_code=503, content={"error": str(exc), "provider_id": exc.provider_id, **_dispatch_error_payload(exc)})
129
+
130
+ @app.exception_handler(ModelUnavailableError)
131
+ async def model_unavailable_handler(request: Request, exc: ModelUnavailableError):
132
+ return JSONResponse(status_code=404, content={"error": str(exc), "model_id": exc.model_id, **_dispatch_error_payload(exc)})
133
+
134
+ @app.exception_handler(ContextWindowExceededError)
135
+ async def context_window_handler(request: Request, exc: ContextWindowExceededError):
136
+ return JSONResponse(
137
+ status_code=400,
138
+ content={
139
+ "error": str(exc),
140
+ "model_id": exc.model_id,
141
+ "required_tokens": exc.required_tokens,
142
+ "context_window": exc.context_window,
143
+ "prompt_tokens": exc.prompt_tokens,
144
+ "compiled_context_tokens": exc.compiled_context_tokens,
145
+ "reserved_output_tokens": exc.reserved_output_tokens,
146
+ **_dispatch_error_payload(exc),
147
+ },
148
+ )
149
+
150
+ @app.exception_handler(MalformedProviderResponseError)
151
+ async def malformed_provider_handler(request: Request, exc: MalformedProviderResponseError):
152
+ return JSONResponse(
153
+ status_code=502,
154
+ content={"error": str(exc), "provider_id": exc.provider_id, **_dispatch_error_payload(exc)},
155
+ )
156
+
157
+ @app.exception_handler(ProviderTimeoutError)
158
+ async def provider_timeout_handler(request: Request, exc: ProviderTimeoutError):
159
+ return JSONResponse(status_code=504, content={"error": str(exc), **_dispatch_error_payload(exc)})
160
+
161
+ @app.exception_handler(ProviderAuthenticationError)
162
+ async def provider_auth_handler(request: Request, exc: ProviderAuthenticationError):
163
+ return JSONResponse(status_code=401, content={"error": str(exc), **_dispatch_error_payload(exc)})
164
+
165
+ @app.exception_handler(ProviderRateLimitError)
166
+ async def provider_rate_limit_handler(request: Request, exc: ProviderRateLimitError):
167
+ headers = {}
168
+ if exc.retry_after is not None:
169
+ headers["Retry-After"] = str(int(exc.retry_after))
170
+ return JSONResponse(status_code=429, content={"error": str(exc), **_dispatch_error_payload(exc)}, headers=headers)
171
+
172
+ @app.exception_handler(RoutingFailureError)
173
+ async def routing_failure_handler(request: Request, exc: RoutingFailureError):
174
+ return JSONResponse(status_code=400, content={"error": str(exc), "policy": exc.policy})
175
+
176
+ @app.exception_handler(ContextOSError)
177
+ async def contextos_error_handler(request: Request, exc: ContextOSError):
178
+ return JSONResponse(status_code=500, content={"error": str(exc)})
179
+
180
+ # --- Register Routes ---
181
+ from contextos.api.routes.ingest import router as ingest_router
182
+ from contextos.api.routes.memories import router as memories_router
183
+ from contextos.api.routes.models import router as models_router
184
+ from contextos.api.routes.retrieval import router as retrieval_router
185
+ from contextos.api.routes.system import router as system_router
186
+ from contextos.api.routes.desktop import router as desktop_router
187
+
188
+ app.include_router(ingest_router, prefix="/api/v1")
189
+ app.include_router(memories_router, prefix="/api/v1")
190
+ app.include_router(retrieval_router, prefix="/api/v1")
191
+ app.include_router(models_router, prefix="/api/v1")
192
+ app.include_router(system_router, prefix="/api/v1")
193
+ app.include_router(desktop_router, prefix="/api/v1")
194
+
195
+ return app
@@ -0,0 +1 @@
1
+ """Reproducible ContextOS benchmarks."""
@@ -0,0 +1,245 @@
1
+ """Deterministic Phase 6 compiler evaluation and scale benchmark."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ import time
7
+ from dataclasses import dataclass
8
+ from uuid import UUID
9
+
10
+ from contextos.core.enums import (
11
+ CompilationStrategy,
12
+ CompressionLevel,
13
+ MemoryStatus,
14
+ MemoryType,
15
+ )
16
+ from contextos.core.models import CompilationConfig, CompiledContext, Memory, ScoredMemory
17
+ from contextos.services.compilation import QueryAwareContextCompiler, fact_is_supported
18
+ from contextos.services.optimization import information_tokens, redundancy_similarity
19
+ from contextos.services.token_counter import DeterministicWordTokenCounter
20
+
21
+
22
+ SMOKE_QUERY = "Which local model should I use given my machine and previous attempts?"
23
+ UNIT_TERMS = {
24
+ "runtime": {"ollama"},
25
+ "failed_large": {"qwen", "30b", "failed", "memory"},
26
+ "successful_small": {"qwen", "9b", "successfully"},
27
+ }
28
+ UNIT_WEIGHTS = {"runtime": 1.0, "failed_large": 1.3, "successful_small": 1.0}
29
+
30
+
31
+ @dataclass(frozen=True)
32
+ class CompilationMetrics:
33
+ input_tokens: int
34
+ output_tokens: int
35
+ compression_ratio: float
36
+ information_recall: float
37
+ weighted_preservation: float
38
+ unsupported_fact_rate: float
39
+ redundancy: float
40
+ budget_violations: int
41
+ provenance_coverage: float
42
+
43
+
44
+ def smoke_memories() -> list[ScoredMemory]:
45
+ contents = [
46
+ "User currently uses Ollama for local model inference.",
47
+ (
48
+ "During a previous attempt the user tried Qwen 30B locally, but model "
49
+ "loading failed because the machine did not have enough available memory."
50
+ ),
51
+ "User successfully runs Qwen 9B locally.",
52
+ "User prefers concise technical responses.",
53
+ (
54
+ "Historical debugging notes discussed terminal colors and repeated setup steps. "
55
+ "The logs included unrelated package installation details and console output. "
56
+ "During a previous attempt the user tried Qwen 30B locally, but model "
57
+ "loading failed because the machine did not have enough available memory. "
58
+ "Additional historical notes repeated the same failure without new evidence. "
59
+ "The session ended after reviewing unrelated shell configuration."
60
+ ),
61
+ ]
62
+ return [
63
+ ScoredMemory(
64
+ memory=Memory(
65
+ id=UUID(f"40000000-0000-0000-0000-{index:012d}"),
66
+ content=content,
67
+ status=MemoryStatus.HISTORICAL if index == 5 else MemoryStatus.ACTIVE,
68
+ type=MemoryType.PREFERENCE if index == 4 else MemoryType.FACT,
69
+ confidence=0.9,
70
+ importance=0.8,
71
+ ),
72
+ final_score=1.0 - index * 0.05,
73
+ rank=index,
74
+ retrieval_sources=["lexical", "dense"],
75
+ )
76
+ for index, content in enumerate(contents, 1)
77
+ ]
78
+
79
+
80
+ def information_unit_recall(context: CompiledContext) -> float:
81
+ emitted = information_tokens(" ".join(fact.text for fact in context.facts))
82
+ covered = sum(terms <= emitted for terms in UNIT_TERMS.values())
83
+ return covered / len(UNIT_TERMS)
84
+
85
+
86
+ def weighted_preservation(context: CompiledContext) -> float:
87
+ emitted = information_tokens(" ".join(fact.text for fact in context.facts))
88
+ total = sum(UNIT_WEIGHTS.values())
89
+ preserved = sum(
90
+ weight for unit, weight in UNIT_WEIGHTS.items() if UNIT_TERMS[unit] <= emitted
91
+ )
92
+ return preserved / total
93
+
94
+
95
+ def unsupported_fact_rate(
96
+ context: CompiledContext, source_memories: list[ScoredMemory]
97
+ ) -> float:
98
+ if not context.facts:
99
+ return 0.0
100
+ sources = {item.memory.id: item.memory.content for item in source_memories}
101
+ unsupported = 0
102
+ for fact in context.facts:
103
+ supported = any(
104
+ fact_is_supported(fact.text, sources.get(source_id, ""))
105
+ for source_id in fact.source_memory_ids
106
+ )
107
+ unsupported += not supported
108
+ return unsupported / len(context.facts)
109
+
110
+
111
+ def fact_redundancy(context: CompiledContext) -> float:
112
+ if len(context.facts) < 2:
113
+ return 0.0
114
+ redundant_pairs = 0
115
+ pair_count = 0
116
+ for index, left in enumerate(context.facts):
117
+ for right in context.facts[index + 1:]:
118
+ pair_count += 1
119
+ similarity = redundancy_similarity(
120
+ information_tokens(left.text),
121
+ information_tokens(right.text),
122
+ )
123
+ redundant_pairs += similarity >= 0.75
124
+ return redundant_pairs / pair_count
125
+
126
+
127
+ def provenance_coverage(context: CompiledContext) -> float:
128
+ if not context.facts:
129
+ return 1.0
130
+ return sum(
131
+ bool(context.provenance_map.get(fact.fact_id)) for fact in context.facts
132
+ ) / len(context.facts)
133
+
134
+
135
+ def budget_violation_rate(contexts: list[CompiledContext]) -> float:
136
+ if not contexts:
137
+ return 0.0
138
+ return sum(context.total_tokens > context.budget for context in contexts) / len(contexts)
139
+
140
+
141
+ def measure(
142
+ context: CompiledContext, source_memories: list[ScoredMemory]
143
+ ) -> CompilationMetrics:
144
+ return CompilationMetrics(
145
+ input_tokens=context.input_tokens,
146
+ output_tokens=context.total_tokens,
147
+ compression_ratio=context.compression_ratio,
148
+ information_recall=information_unit_recall(context),
149
+ weighted_preservation=weighted_preservation(context),
150
+ unsupported_fact_rate=unsupported_fact_rate(context, source_memories),
151
+ redundancy=fact_redundancy(context),
152
+ budget_violations=int(context.total_tokens > context.budget),
153
+ provenance_coverage=provenance_coverage(context),
154
+ )
155
+
156
+
157
+ def synthetic_memories(count: int = 1_000) -> list[ScoredMemory]:
158
+ return [
159
+ ScoredMemory(
160
+ memory=Memory(
161
+ id=UUID(f"50000000-0000-0000-0000-{index:012d}"),
162
+ content=(
163
+ f"Project topic{index} currently has constraint group{index % 41}. "
164
+ f"Unrelated note category{index % 73} is archived."
165
+ ),
166
+ status=MemoryStatus.ACTIVE,
167
+ type=MemoryType.PROJECT,
168
+ ),
169
+ final_score=1.0 / (index + 1),
170
+ rank=index + 1,
171
+ )
172
+ for index in range(count)
173
+ ]
174
+
175
+
176
+ async def main() -> None:
177
+ compiler = QueryAwareContextCompiler(
178
+ token_counter=DeterministicWordTokenCounter()
179
+ )
180
+ memories = smoke_memories()
181
+ strategies = (
182
+ CompilationStrategy.RAW_CONCAT,
183
+ CompilationStrategy.DEDUP_ONLY,
184
+ CompilationStrategy.CONTEXTOS_COMPILER,
185
+ )
186
+ print(
187
+ "Budget Strategy In Out Ratio Recall Weighted "
188
+ "Unsupported Redundancy Violations Provenance"
189
+ )
190
+ results: dict[tuple[int, CompilationStrategy], CompiledContext] = {}
191
+ for budget in (40, 60, 100, 200):
192
+ for strategy in strategies:
193
+ context = await compiler.compile(
194
+ SMOKE_QUERY,
195
+ memories,
196
+ CompilationConfig(
197
+ budget=budget,
198
+ strategy=strategy,
199
+ compression_level=CompressionLevel.LIGHT,
200
+ ),
201
+ )
202
+ results[(budget, strategy)] = context
203
+ metrics = measure(context, memories)
204
+ print(
205
+ f"{budget:>6} {strategy.value:<20} "
206
+ f"{metrics.input_tokens:>3} {metrics.output_tokens:>3} "
207
+ f"{metrics.compression_ratio:>5.3f} "
208
+ f"{metrics.information_recall:>6.3f} "
209
+ f"{metrics.weighted_preservation:>8.3f} "
210
+ f"{metrics.unsupported_fact_rate:>11.3f} "
211
+ f"{metrics.redundancy:>10.3f} "
212
+ f"{metrics.budget_violations:>10} "
213
+ f"{metrics.provenance_coverage:>10.3f}"
214
+ )
215
+
216
+ print("\nExact compiled contexts")
217
+ for budget in (40, 60, 100, 200):
218
+ print(f"\nBudget {budget}")
219
+ for strategy in strategies:
220
+ text = results[(budget, strategy)].context_text or "<empty>"
221
+ print(f"[{strategy.value}]\n{text}")
222
+
223
+ scale = synthetic_memories()
224
+ started = time.perf_counter()
225
+ first = await compiler.compile(
226
+ "What project constraints are current?",
227
+ scale,
228
+ CompilationConfig(budget=500),
229
+ )
230
+ latency_ms = (time.perf_counter() - started) * 1000
231
+ second = await compiler.compile(
232
+ "What project constraints are current?",
233
+ scale,
234
+ CompilationConfig(budget=500),
235
+ )
236
+ print("\nSynthetic scale")
237
+ print(
238
+ f"inputs=1000 facts={len(first.facts)} tokens={first.total_tokens}/500 "
239
+ f"latency_ms={latency_ms:.3f} "
240
+ f"deterministic={first.context_text == second.context_text}"
241
+ )
242
+
243
+
244
+ if __name__ == "__main__":
245
+ asyncio.run(main())