contextos-memory-runtime 1.0.0rc2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contextos/__init__.py +3 -0
- contextos/__main__.py +6 -0
- contextos/api/__init__.py +1 -0
- contextos/api/routes/__init__.py +1 -0
- contextos/api/routes/desktop.py +322 -0
- contextos/api/routes/ingest.py +17 -0
- contextos/api/routes/memories.py +84 -0
- contextos/api/routes/models.py +81 -0
- contextos/api/routes/retrieval.py +89 -0
- contextos/api/routes/system.py +216 -0
- contextos/api/server.py +195 -0
- contextos/benchmarks/__init__.py +1 -0
- contextos/benchmarks/compilation.py +245 -0
- contextos/benchmarks/connectors.py +423 -0
- contextos/benchmarks/explainability.py +103 -0
- contextos/benchmarks/final.py +406 -0
- contextos/benchmarks/graph.py +310 -0
- contextos/benchmarks/graph_adversarial.py +525 -0
- contextos/benchmarks/mcp.py +324 -0
- contextos/benchmarks/model_routing.py +203 -0
- contextos/benchmarks/optimization.py +305 -0
- contextos/benchmarks/rescue_integration.py +127 -0
- contextos/benchmarks/retrieval.py +266 -0
- contextos/benchmarks/temporal.py +377 -0
- contextos/benchmarks/temporal_hotpath.py +76 -0
- contextos/benchmarks/terminal.py +62 -0
- contextos/cli/__init__.py +1 -0
- contextos/cli/app.py +932 -0
- contextos/cli/dashboard.py +174 -0
- contextos/cli/formatters.py +299 -0
- contextos/config/__init__.py +1 -0
- contextos/config/settings.py +160 -0
- contextos/connectors/__init__.py +6 -0
- contextos/connectors/fake.py +11 -0
- contextos/connectors/json_import.py +125 -0
- contextos/connectors/local_files.py +102 -0
- contextos/connectors/manager.py +293 -0
- contextos/connectors/models.py +62 -0
- contextos/connectors/protocols.py +11 -0
- contextos/core/__init__.py +103 -0
- contextos/core/enums.py +489 -0
- contextos/core/exceptions.py +293 -0
- contextos/core/models.py +1147 -0
- contextos/core/protocols.py +549 -0
- contextos/daemon/__init__.py +1 -0
- contextos/daemon/manager.py +510 -0
- contextos/daemon/state.py +127 -0
- contextos/daemon/wiring.py +296 -0
- contextos/demo.py +217 -0
- contextos/embedding/__init__.py +1 -0
- contextos/embedding/deterministic.py +76 -0
- contextos/embedding/sentence_transformers.py +80 -0
- contextos/mcp/__init__.py +5 -0
- contextos/mcp/server.py +269 -0
- contextos/providers/__init__.py +13 -0
- contextos/providers/fake.py +217 -0
- contextos/providers/ollama.py +297 -0
- contextos/providers/openai_compatible.py +337 -0
- contextos/services/__init__.py +1 -0
- contextos/services/compilation.py +535 -0
- contextos/services/explainability.py +553 -0
- contextos/services/extraction.py +311 -0
- contextos/services/graph.py +524 -0
- contextos/services/graph_retrieval.py +143 -0
- contextos/services/ingestion.py +143 -0
- contextos/services/inspection.py +174 -0
- contextos/services/memory.py +291 -0
- contextos/services/model_service.py +409 -0
- contextos/services/optimization.py +426 -0
- contextos/services/privacy.py +331 -0
- contextos/services/retrieval.py +302 -0
- contextos/services/retrieval_index.py +88 -0
- contextos/services/router.py +302 -0
- contextos/services/secret_scanner.py +207 -0
- contextos/services/telemetry_query.py +102 -0
- contextos/services/temporal.py +500 -0
- contextos/services/token_counter.py +222 -0
- contextos/storage/__init__.py +1 -0
- contextos/storage/connector_repo.py +67 -0
- contextos/storage/database.py +497 -0
- contextos/storage/event_repo.py +137 -0
- contextos/storage/graph_repo.py +228 -0
- contextos/storage/lexical/__init__.py +1 -0
- contextos/storage/lexical/bm25.py +134 -0
- contextos/storage/memory_repo.py +589 -0
- contextos/storage/relation_repo.py +80 -0
- contextos/storage/telemetry_repo.py +481 -0
- contextos/storage/vector/__init__.py +1 -0
- contextos/storage/vector/in_memory.py +162 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
"""System status and diagnostics API routes."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
import time
|
|
7
|
+
import asyncio
|
|
8
|
+
|
|
9
|
+
from fastapi import APIRouter
|
|
10
|
+
|
|
11
|
+
from contextos import __version__
|
|
12
|
+
from contextos.api.server import get_service
|
|
13
|
+
from contextos.core.models import MemoryFilters, SystemStatus, TokenStats
|
|
14
|
+
from contextos.core.enums import MemoryStatus
|
|
15
|
+
|
|
16
|
+
router = APIRouter(tags=["system"])
|
|
17
|
+
|
|
18
|
+
_start_time = time.time()
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@router.get("/status", response_model=SystemStatus)
|
|
22
|
+
async def status() -> SystemStatus:
|
|
23
|
+
"""Get system health and status."""
|
|
24
|
+
database = get_service("database")
|
|
25
|
+
embedding_service = get_service("embedding")
|
|
26
|
+
vector_store = get_service("vector_store")
|
|
27
|
+
bm25_index = get_service("bm25_index")
|
|
28
|
+
|
|
29
|
+
# Get counts
|
|
30
|
+
from contextos.core.protocols import MemoryRepository
|
|
31
|
+
memory_repo: MemoryRepository = get_service("memory_repo")
|
|
32
|
+
total_count = await memory_repo.count()
|
|
33
|
+
active_count = await memory_repo.count(MemoryFilters(status=MemoryStatus.ACTIVE))
|
|
34
|
+
|
|
35
|
+
from contextos.core.protocols import EventRepository
|
|
36
|
+
event_repo: EventRepository = get_service("event_repo")
|
|
37
|
+
event_count = await event_repo.count()
|
|
38
|
+
|
|
39
|
+
vector_count = await vector_store.count()
|
|
40
|
+
bm25_count = await bm25_index.count()
|
|
41
|
+
|
|
42
|
+
db_size = await database.get_size_bytes()
|
|
43
|
+
|
|
44
|
+
return SystemStatus(
|
|
45
|
+
daemon_running=True,
|
|
46
|
+
pid=os.getpid(),
|
|
47
|
+
uptime_seconds=time.time() - _start_time,
|
|
48
|
+
total_memories=total_count,
|
|
49
|
+
active_memories=active_count,
|
|
50
|
+
total_events=event_count,
|
|
51
|
+
embedding_model=embedding_service.model_name,
|
|
52
|
+
embedding_model_loaded=True,
|
|
53
|
+
vector_index_size=vector_count,
|
|
54
|
+
bm25_index_size=bm25_count,
|
|
55
|
+
database_size_bytes=db_size,
|
|
56
|
+
data_directory="[LOCAL]",
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@router.get("/stats", response_model=TokenStats)
|
|
61
|
+
async def stats() -> TokenStats:
|
|
62
|
+
"""Report stored counts and observed model compilations with provenance."""
|
|
63
|
+
conn = get_service("database").connection()
|
|
64
|
+
cursor = await conn.execute(
|
|
65
|
+
"SELECT COALESCE(SUM(token_count), 0), COUNT(*) FROM memories WHERE status = ?",
|
|
66
|
+
(MemoryStatus.ACTIVE.value,),
|
|
67
|
+
)
|
|
68
|
+
total_tokens, memory_count = await cursor.fetchone()
|
|
69
|
+
bases = await get_service("telemetry_query").context_measurement_bases()
|
|
70
|
+
basis = bases[0] if len(bases) == 1 and bases[0]["source"] != "unknown" else None
|
|
71
|
+
if basis is not None:
|
|
72
|
+
cursor = await conn.execute(
|
|
73
|
+
"SELECT COUNT(*), COALESCE(SUM(compiled_context_tokens), 0), "
|
|
74
|
+
"COALESCE(SUM(context_tokens_avoided), 0), "
|
|
75
|
+
"COALESCE(SUM(candidate_context_tokens), 0) "
|
|
76
|
+
"FROM model_invocations WHERE status = 'success'"
|
|
77
|
+
)
|
|
78
|
+
runs, compiled, avoided, candidates = await cursor.fetchone()
|
|
79
|
+
else:
|
|
80
|
+
runs = compiled = avoided = candidates = None
|
|
81
|
+
from contextos.api.routes.desktop import _public_label
|
|
82
|
+
|
|
83
|
+
return TokenStats(
|
|
84
|
+
total_tokens_stored=total_tokens,
|
|
85
|
+
tokens_per_memory=total_tokens / memory_count if memory_count else 0,
|
|
86
|
+
# Standalone compilations are not persisted; these are successful model runs.
|
|
87
|
+
total_compilations=runs,
|
|
88
|
+
total_tokens_compiled=compiled,
|
|
89
|
+
total_tokens_saved=avoided,
|
|
90
|
+
average_compression_ratio=avoided / candidates if candidates else None,
|
|
91
|
+
measurement_basis=f"{basis['source']}:{_public_label(basis['tokenizer'])}" if basis else None,
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
@router.post("/doctor")
|
|
96
|
+
async def doctor() -> dict:
|
|
97
|
+
"""Run diagnostic checks."""
|
|
98
|
+
results: dict = {
|
|
99
|
+
"version": __version__,
|
|
100
|
+
"checks": {},
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
# Database integrity
|
|
104
|
+
database = get_service("database")
|
|
105
|
+
ok, msg = await database.integrity_check()
|
|
106
|
+
results["checks"]["database_integrity"] = {"ok": ok, "message": msg}
|
|
107
|
+
|
|
108
|
+
from contextos.storage.database import SCHEMA_VERSION
|
|
109
|
+
conn = database.connection()
|
|
110
|
+
cursor = await conn.execute("SELECT MAX(version) FROM schema_version")
|
|
111
|
+
version_row = await cursor.fetchone()
|
|
112
|
+
schema_version = version_row[0] if version_row else None
|
|
113
|
+
results["checks"]["schema"] = {
|
|
114
|
+
"ok": schema_version == SCHEMA_VERSION,
|
|
115
|
+
"version": schema_version, "expected": SCHEMA_VERSION,
|
|
116
|
+
}
|
|
117
|
+
cursor = await conn.execute(
|
|
118
|
+
"SELECT name FROM sqlite_master WHERE type = 'table' AND name IN "
|
|
119
|
+
"('memories', 'memory_relations', 'model_invocations', 'connector_state', 'graph_nodes', 'graph_edges')"
|
|
120
|
+
)
|
|
121
|
+
tables = {row[0] for row in await cursor.fetchall()}
|
|
122
|
+
results["checks"]["tables"] = {"ok": len(tables) == 6, "present": sorted(tables)}
|
|
123
|
+
cursor = await conn.execute("PRAGMA index_list('memories')")
|
|
124
|
+
indexes = {row[1] for row in await cursor.fetchall()}
|
|
125
|
+
required_indexes = {"idx_memories_status", "idx_memories_slot_key"}
|
|
126
|
+
results["checks"]["indexes"] = {
|
|
127
|
+
"ok": required_indexes <= indexes,
|
|
128
|
+
"required_present": sorted(indexes & required_indexes),
|
|
129
|
+
}
|
|
130
|
+
graph_repo = get_service("graph_repo")
|
|
131
|
+
graph_nodes, graph_edges, _ = await graph_repo.counts()
|
|
132
|
+
graph_dirty = await graph_repo.source_is_dirty()
|
|
133
|
+
results["checks"]["graph_projection"] = {
|
|
134
|
+
"ok": True, "dirty": graph_dirty, "nodes": graph_nodes, "edges": graph_edges,
|
|
135
|
+
"suggestion": "Graph rebuilds on the next graph retrieval" if graph_dirty else None,
|
|
136
|
+
}
|
|
137
|
+
results["checks"]["telemetry"] = {
|
|
138
|
+
"ok": True, "recorded_invocations": await get_service("telemetry_repo").count(),
|
|
139
|
+
}
|
|
140
|
+
connector_ids = get_service("connectors").list_connectors()
|
|
141
|
+
connector_states = await get_service("connector_repo").list_states()
|
|
142
|
+
results["checks"]["connectors"] = {
|
|
143
|
+
"ok": True, "configured": len(connector_ids),
|
|
144
|
+
"failed": sum(state.status == "failed" for state in connector_states),
|
|
145
|
+
}
|
|
146
|
+
settings = get_service("settings")
|
|
147
|
+
results["checks"]["local_configuration"] = {
|
|
148
|
+
"ok": settings.daemon.host in {"127.0.0.1", "localhost", "::1"}
|
|
149
|
+
and os.access(database.path.parent, os.R_OK | os.W_OK),
|
|
150
|
+
"data_directory_accessible": os.access(database.path.parent, os.R_OK | os.W_OK),
|
|
151
|
+
}
|
|
152
|
+
scan = get_service("secret_scanner").scan("Authorization: Bearer privatefixture1234567890")
|
|
153
|
+
results["checks"]["privacy_scanner"] = {"ok": bool(scan.matches)}
|
|
154
|
+
providers = get_service("providers")
|
|
155
|
+
|
|
156
|
+
async def available(provider):
|
|
157
|
+
try:
|
|
158
|
+
return bool(await asyncio.wait_for(provider.health(), timeout=1.0))
|
|
159
|
+
except Exception:
|
|
160
|
+
return False
|
|
161
|
+
|
|
162
|
+
healths = await asyncio.gather(*(available(provider) for provider in providers.values()))
|
|
163
|
+
results["checks"]["providers"] = {
|
|
164
|
+
"ok": any(healths), "configured": len(providers), "available": sum(healths),
|
|
165
|
+
"optional_unavailable": len(providers) - sum(healths),
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
# Embedding model
|
|
169
|
+
try:
|
|
170
|
+
embedding = get_service("embedding")
|
|
171
|
+
dim = embedding.dimension
|
|
172
|
+
results["checks"]["embedding_model"] = {
|
|
173
|
+
"ok": True,
|
|
174
|
+
"model": embedding.model_name,
|
|
175
|
+
"dimension": dim,
|
|
176
|
+
}
|
|
177
|
+
except Exception as e:
|
|
178
|
+
results["checks"]["embedding_model"] = {"ok": False, "error": e.__class__.__name__}
|
|
179
|
+
|
|
180
|
+
# Index consistency
|
|
181
|
+
memory_repo = get_service("memory_repo")
|
|
182
|
+
vector_store = get_service("vector_store")
|
|
183
|
+
bm25_index = get_service("bm25_index")
|
|
184
|
+
|
|
185
|
+
from contextos.services.retrieval_index import INDEXED_STATUSES
|
|
186
|
+
indexed_count = 0
|
|
187
|
+
for memory_status in INDEXED_STATUSES:
|
|
188
|
+
indexed_count += await memory_repo.count(MemoryFilters(status=memory_status))
|
|
189
|
+
vector_count = await vector_store.count()
|
|
190
|
+
bm25_count = await bm25_index.count()
|
|
191
|
+
|
|
192
|
+
results["checks"]["index_consistency"] = {
|
|
193
|
+
"ok": True,
|
|
194
|
+
"indexable_memories": indexed_count,
|
|
195
|
+
"vector_index": vector_count,
|
|
196
|
+
"bm25_index": bm25_count,
|
|
197
|
+
"warnings": [],
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
if vector_count != indexed_count:
|
|
201
|
+
results["checks"]["index_consistency"]["warnings"].append(
|
|
202
|
+
f"Vector index ({vector_count}) != indexable memories ({indexed_count})"
|
|
203
|
+
)
|
|
204
|
+
if bm25_count != indexed_count:
|
|
205
|
+
results["checks"]["index_consistency"]["warnings"].append(
|
|
206
|
+
f"BM25 index ({bm25_count}) != indexable memories ({indexed_count})"
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
if results["checks"]["index_consistency"]["warnings"]:
|
|
210
|
+
results["checks"]["index_consistency"]["ok"] = False
|
|
211
|
+
|
|
212
|
+
results["overall"] = all(
|
|
213
|
+
c.get("ok", False) for c in results["checks"].values()
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
return results
|
contextos/api/server.py
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
"""FastAPI server setup for ContextOS.
|
|
2
|
+
|
|
3
|
+
This is the HTTP API that the CLI and future MCP server communicate with.
|
|
4
|
+
All business logic lives in the service layer — this is a thin translation
|
|
5
|
+
layer between HTTP and service protocol calls.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import logging
|
|
11
|
+
import time
|
|
12
|
+
from contextlib import asynccontextmanager
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
from fastapi import FastAPI, HTTPException, Request
|
|
16
|
+
from fastapi.exceptions import RequestValidationError
|
|
17
|
+
from fastapi.responses import JSONResponse
|
|
18
|
+
|
|
19
|
+
from contextos import __version__
|
|
20
|
+
from contextos.core.exceptions import (
|
|
21
|
+
ContextOSError,
|
|
22
|
+
InvalidTransitionError,
|
|
23
|
+
MemoryNotFoundError,
|
|
24
|
+
SecretDetectedError,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
logger = logging.getLogger(__name__)
|
|
28
|
+
|
|
29
|
+
# Service instances — set by daemon wiring before startup
|
|
30
|
+
_services: dict[str, Any] = {}
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def set_services(services: dict[str, Any]) -> None:
|
|
34
|
+
"""Set service instances. Called by daemon wiring at startup."""
|
|
35
|
+
_services.update(services)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def get_service(name: str) -> Any:
|
|
39
|
+
"""Get a service by name. Raises if not set."""
|
|
40
|
+
if name not in _services:
|
|
41
|
+
raise RuntimeError(f"Service '{name}' not initialized")
|
|
42
|
+
return _services[name]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _dispatch_error_payload(exc: Exception) -> dict[str, Any]:
|
|
46
|
+
"""Expose bounded dispatch evidence without echoing provider request content."""
|
|
47
|
+
evidence = getattr(exc, "dispatch_evidence", None)
|
|
48
|
+
if evidence is None:
|
|
49
|
+
return {}
|
|
50
|
+
return {"dispatch_evidence": evidence.model_dump(mode="json")}
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@asynccontextmanager
|
|
54
|
+
async def lifespan(app: FastAPI):
|
|
55
|
+
"""Application lifespan handler."""
|
|
56
|
+
logger.info("ContextOS API server starting (v%s)", __version__)
|
|
57
|
+
yield
|
|
58
|
+
logger.info("ContextOS API server shutting down")
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def create_app() -> FastAPI:
|
|
62
|
+
"""Create and configure the FastAPI application."""
|
|
63
|
+
app = FastAPI(
|
|
64
|
+
title="ContextOS",
|
|
65
|
+
description="Local-first personal AI memory runtime",
|
|
66
|
+
version=__version__,
|
|
67
|
+
lifespan=lifespan,
|
|
68
|
+
docs_url="/api/docs",
|
|
69
|
+
redoc_url="/api/redoc",
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
# --- Middleware ---
|
|
73
|
+
|
|
74
|
+
@app.middleware("http")
|
|
75
|
+
async def request_timing(request: Request, call_next):
|
|
76
|
+
"""Log request timing."""
|
|
77
|
+
start = time.perf_counter()
|
|
78
|
+
response = await call_next(request)
|
|
79
|
+
elapsed = (time.perf_counter() - start) * 1000
|
|
80
|
+
logger.debug(
|
|
81
|
+
"%s %s — %d (%.1fms)",
|
|
82
|
+
request.method, request.url.path, response.status_code, elapsed,
|
|
83
|
+
)
|
|
84
|
+
return response
|
|
85
|
+
|
|
86
|
+
# --- Exception Handlers ---
|
|
87
|
+
|
|
88
|
+
@app.exception_handler(RequestValidationError)
|
|
89
|
+
async def request_validation_handler(request: Request, exc: RequestValidationError):
|
|
90
|
+
"""Return validation structure without echoing attacker-controlled values."""
|
|
91
|
+
safe_errors = [
|
|
92
|
+
{key: value for key, value in error.items() if key not in {"input", "ctx"}}
|
|
93
|
+
for error in exc.errors()
|
|
94
|
+
]
|
|
95
|
+
return JSONResponse(status_code=422, content={"detail": safe_errors})
|
|
96
|
+
|
|
97
|
+
@app.exception_handler(MemoryNotFoundError)
|
|
98
|
+
async def memory_not_found_handler(request: Request, exc: MemoryNotFoundError):
|
|
99
|
+
return JSONResponse(status_code=404, content={"error": str(exc)})
|
|
100
|
+
|
|
101
|
+
@app.exception_handler(InvalidTransitionError)
|
|
102
|
+
async def invalid_transition_handler(request: Request, exc: InvalidTransitionError):
|
|
103
|
+
return JSONResponse(status_code=422, content={"error": str(exc)})
|
|
104
|
+
|
|
105
|
+
@app.exception_handler(SecretDetectedError)
|
|
106
|
+
async def secret_detected_handler(request: Request, exc: SecretDetectedError):
|
|
107
|
+
return JSONResponse(
|
|
108
|
+
status_code=422,
|
|
109
|
+
content={
|
|
110
|
+
"error": str(exc),
|
|
111
|
+
"secret_types": exc.secret_types,
|
|
112
|
+
},
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
from contextos.core.exceptions import (
|
|
116
|
+
ContextWindowExceededError,
|
|
117
|
+
MalformedProviderResponseError,
|
|
118
|
+
ModelUnavailableError,
|
|
119
|
+
ProviderAuthenticationError,
|
|
120
|
+
ProviderRateLimitError,
|
|
121
|
+
ProviderTimeoutError,
|
|
122
|
+
ProviderUnavailableError,
|
|
123
|
+
RoutingFailureError,
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
@app.exception_handler(ProviderUnavailableError)
|
|
127
|
+
async def provider_unavailable_handler(request: Request, exc: ProviderUnavailableError):
|
|
128
|
+
return JSONResponse(status_code=503, content={"error": str(exc), "provider_id": exc.provider_id, **_dispatch_error_payload(exc)})
|
|
129
|
+
|
|
130
|
+
@app.exception_handler(ModelUnavailableError)
|
|
131
|
+
async def model_unavailable_handler(request: Request, exc: ModelUnavailableError):
|
|
132
|
+
return JSONResponse(status_code=404, content={"error": str(exc), "model_id": exc.model_id, **_dispatch_error_payload(exc)})
|
|
133
|
+
|
|
134
|
+
@app.exception_handler(ContextWindowExceededError)
|
|
135
|
+
async def context_window_handler(request: Request, exc: ContextWindowExceededError):
|
|
136
|
+
return JSONResponse(
|
|
137
|
+
status_code=400,
|
|
138
|
+
content={
|
|
139
|
+
"error": str(exc),
|
|
140
|
+
"model_id": exc.model_id,
|
|
141
|
+
"required_tokens": exc.required_tokens,
|
|
142
|
+
"context_window": exc.context_window,
|
|
143
|
+
"prompt_tokens": exc.prompt_tokens,
|
|
144
|
+
"compiled_context_tokens": exc.compiled_context_tokens,
|
|
145
|
+
"reserved_output_tokens": exc.reserved_output_tokens,
|
|
146
|
+
**_dispatch_error_payload(exc),
|
|
147
|
+
},
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
@app.exception_handler(MalformedProviderResponseError)
|
|
151
|
+
async def malformed_provider_handler(request: Request, exc: MalformedProviderResponseError):
|
|
152
|
+
return JSONResponse(
|
|
153
|
+
status_code=502,
|
|
154
|
+
content={"error": str(exc), "provider_id": exc.provider_id, **_dispatch_error_payload(exc)},
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
@app.exception_handler(ProviderTimeoutError)
|
|
158
|
+
async def provider_timeout_handler(request: Request, exc: ProviderTimeoutError):
|
|
159
|
+
return JSONResponse(status_code=504, content={"error": str(exc), **_dispatch_error_payload(exc)})
|
|
160
|
+
|
|
161
|
+
@app.exception_handler(ProviderAuthenticationError)
|
|
162
|
+
async def provider_auth_handler(request: Request, exc: ProviderAuthenticationError):
|
|
163
|
+
return JSONResponse(status_code=401, content={"error": str(exc), **_dispatch_error_payload(exc)})
|
|
164
|
+
|
|
165
|
+
@app.exception_handler(ProviderRateLimitError)
|
|
166
|
+
async def provider_rate_limit_handler(request: Request, exc: ProviderRateLimitError):
|
|
167
|
+
headers = {}
|
|
168
|
+
if exc.retry_after is not None:
|
|
169
|
+
headers["Retry-After"] = str(int(exc.retry_after))
|
|
170
|
+
return JSONResponse(status_code=429, content={"error": str(exc), **_dispatch_error_payload(exc)}, headers=headers)
|
|
171
|
+
|
|
172
|
+
@app.exception_handler(RoutingFailureError)
|
|
173
|
+
async def routing_failure_handler(request: Request, exc: RoutingFailureError):
|
|
174
|
+
return JSONResponse(status_code=400, content={"error": str(exc), "policy": exc.policy})
|
|
175
|
+
|
|
176
|
+
@app.exception_handler(ContextOSError)
|
|
177
|
+
async def contextos_error_handler(request: Request, exc: ContextOSError):
|
|
178
|
+
return JSONResponse(status_code=500, content={"error": str(exc)})
|
|
179
|
+
|
|
180
|
+
# --- Register Routes ---
|
|
181
|
+
from contextos.api.routes.ingest import router as ingest_router
|
|
182
|
+
from contextos.api.routes.memories import router as memories_router
|
|
183
|
+
from contextos.api.routes.models import router as models_router
|
|
184
|
+
from contextos.api.routes.retrieval import router as retrieval_router
|
|
185
|
+
from contextos.api.routes.system import router as system_router
|
|
186
|
+
from contextos.api.routes.desktop import router as desktop_router
|
|
187
|
+
|
|
188
|
+
app.include_router(ingest_router, prefix="/api/v1")
|
|
189
|
+
app.include_router(memories_router, prefix="/api/v1")
|
|
190
|
+
app.include_router(retrieval_router, prefix="/api/v1")
|
|
191
|
+
app.include_router(models_router, prefix="/api/v1")
|
|
192
|
+
app.include_router(system_router, prefix="/api/v1")
|
|
193
|
+
app.include_router(desktop_router, prefix="/api/v1")
|
|
194
|
+
|
|
195
|
+
return app
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Reproducible ContextOS benchmarks."""
|
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
"""Deterministic Phase 6 compiler evaluation and scale benchmark."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import time
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from uuid import UUID
|
|
9
|
+
|
|
10
|
+
from contextos.core.enums import (
|
|
11
|
+
CompilationStrategy,
|
|
12
|
+
CompressionLevel,
|
|
13
|
+
MemoryStatus,
|
|
14
|
+
MemoryType,
|
|
15
|
+
)
|
|
16
|
+
from contextos.core.models import CompilationConfig, CompiledContext, Memory, ScoredMemory
|
|
17
|
+
from contextos.services.compilation import QueryAwareContextCompiler, fact_is_supported
|
|
18
|
+
from contextos.services.optimization import information_tokens, redundancy_similarity
|
|
19
|
+
from contextos.services.token_counter import DeterministicWordTokenCounter
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
SMOKE_QUERY = "Which local model should I use given my machine and previous attempts?"
|
|
23
|
+
UNIT_TERMS = {
|
|
24
|
+
"runtime": {"ollama"},
|
|
25
|
+
"failed_large": {"qwen", "30b", "failed", "memory"},
|
|
26
|
+
"successful_small": {"qwen", "9b", "successfully"},
|
|
27
|
+
}
|
|
28
|
+
UNIT_WEIGHTS = {"runtime": 1.0, "failed_large": 1.3, "successful_small": 1.0}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass(frozen=True)
|
|
32
|
+
class CompilationMetrics:
|
|
33
|
+
input_tokens: int
|
|
34
|
+
output_tokens: int
|
|
35
|
+
compression_ratio: float
|
|
36
|
+
information_recall: float
|
|
37
|
+
weighted_preservation: float
|
|
38
|
+
unsupported_fact_rate: float
|
|
39
|
+
redundancy: float
|
|
40
|
+
budget_violations: int
|
|
41
|
+
provenance_coverage: float
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def smoke_memories() -> list[ScoredMemory]:
|
|
45
|
+
contents = [
|
|
46
|
+
"User currently uses Ollama for local model inference.",
|
|
47
|
+
(
|
|
48
|
+
"During a previous attempt the user tried Qwen 30B locally, but model "
|
|
49
|
+
"loading failed because the machine did not have enough available memory."
|
|
50
|
+
),
|
|
51
|
+
"User successfully runs Qwen 9B locally.",
|
|
52
|
+
"User prefers concise technical responses.",
|
|
53
|
+
(
|
|
54
|
+
"Historical debugging notes discussed terminal colors and repeated setup steps. "
|
|
55
|
+
"The logs included unrelated package installation details and console output. "
|
|
56
|
+
"During a previous attempt the user tried Qwen 30B locally, but model "
|
|
57
|
+
"loading failed because the machine did not have enough available memory. "
|
|
58
|
+
"Additional historical notes repeated the same failure without new evidence. "
|
|
59
|
+
"The session ended after reviewing unrelated shell configuration."
|
|
60
|
+
),
|
|
61
|
+
]
|
|
62
|
+
return [
|
|
63
|
+
ScoredMemory(
|
|
64
|
+
memory=Memory(
|
|
65
|
+
id=UUID(f"40000000-0000-0000-0000-{index:012d}"),
|
|
66
|
+
content=content,
|
|
67
|
+
status=MemoryStatus.HISTORICAL if index == 5 else MemoryStatus.ACTIVE,
|
|
68
|
+
type=MemoryType.PREFERENCE if index == 4 else MemoryType.FACT,
|
|
69
|
+
confidence=0.9,
|
|
70
|
+
importance=0.8,
|
|
71
|
+
),
|
|
72
|
+
final_score=1.0 - index * 0.05,
|
|
73
|
+
rank=index,
|
|
74
|
+
retrieval_sources=["lexical", "dense"],
|
|
75
|
+
)
|
|
76
|
+
for index, content in enumerate(contents, 1)
|
|
77
|
+
]
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def information_unit_recall(context: CompiledContext) -> float:
|
|
81
|
+
emitted = information_tokens(" ".join(fact.text for fact in context.facts))
|
|
82
|
+
covered = sum(terms <= emitted for terms in UNIT_TERMS.values())
|
|
83
|
+
return covered / len(UNIT_TERMS)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def weighted_preservation(context: CompiledContext) -> float:
|
|
87
|
+
emitted = information_tokens(" ".join(fact.text for fact in context.facts))
|
|
88
|
+
total = sum(UNIT_WEIGHTS.values())
|
|
89
|
+
preserved = sum(
|
|
90
|
+
weight for unit, weight in UNIT_WEIGHTS.items() if UNIT_TERMS[unit] <= emitted
|
|
91
|
+
)
|
|
92
|
+
return preserved / total
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def unsupported_fact_rate(
|
|
96
|
+
context: CompiledContext, source_memories: list[ScoredMemory]
|
|
97
|
+
) -> float:
|
|
98
|
+
if not context.facts:
|
|
99
|
+
return 0.0
|
|
100
|
+
sources = {item.memory.id: item.memory.content for item in source_memories}
|
|
101
|
+
unsupported = 0
|
|
102
|
+
for fact in context.facts:
|
|
103
|
+
supported = any(
|
|
104
|
+
fact_is_supported(fact.text, sources.get(source_id, ""))
|
|
105
|
+
for source_id in fact.source_memory_ids
|
|
106
|
+
)
|
|
107
|
+
unsupported += not supported
|
|
108
|
+
return unsupported / len(context.facts)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def fact_redundancy(context: CompiledContext) -> float:
|
|
112
|
+
if len(context.facts) < 2:
|
|
113
|
+
return 0.0
|
|
114
|
+
redundant_pairs = 0
|
|
115
|
+
pair_count = 0
|
|
116
|
+
for index, left in enumerate(context.facts):
|
|
117
|
+
for right in context.facts[index + 1:]:
|
|
118
|
+
pair_count += 1
|
|
119
|
+
similarity = redundancy_similarity(
|
|
120
|
+
information_tokens(left.text),
|
|
121
|
+
information_tokens(right.text),
|
|
122
|
+
)
|
|
123
|
+
redundant_pairs += similarity >= 0.75
|
|
124
|
+
return redundant_pairs / pair_count
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def provenance_coverage(context: CompiledContext) -> float:
|
|
128
|
+
if not context.facts:
|
|
129
|
+
return 1.0
|
|
130
|
+
return sum(
|
|
131
|
+
bool(context.provenance_map.get(fact.fact_id)) for fact in context.facts
|
|
132
|
+
) / len(context.facts)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def budget_violation_rate(contexts: list[CompiledContext]) -> float:
|
|
136
|
+
if not contexts:
|
|
137
|
+
return 0.0
|
|
138
|
+
return sum(context.total_tokens > context.budget for context in contexts) / len(contexts)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def measure(
|
|
142
|
+
context: CompiledContext, source_memories: list[ScoredMemory]
|
|
143
|
+
) -> CompilationMetrics:
|
|
144
|
+
return CompilationMetrics(
|
|
145
|
+
input_tokens=context.input_tokens,
|
|
146
|
+
output_tokens=context.total_tokens,
|
|
147
|
+
compression_ratio=context.compression_ratio,
|
|
148
|
+
information_recall=information_unit_recall(context),
|
|
149
|
+
weighted_preservation=weighted_preservation(context),
|
|
150
|
+
unsupported_fact_rate=unsupported_fact_rate(context, source_memories),
|
|
151
|
+
redundancy=fact_redundancy(context),
|
|
152
|
+
budget_violations=int(context.total_tokens > context.budget),
|
|
153
|
+
provenance_coverage=provenance_coverage(context),
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def synthetic_memories(count: int = 1_000) -> list[ScoredMemory]:
|
|
158
|
+
return [
|
|
159
|
+
ScoredMemory(
|
|
160
|
+
memory=Memory(
|
|
161
|
+
id=UUID(f"50000000-0000-0000-0000-{index:012d}"),
|
|
162
|
+
content=(
|
|
163
|
+
f"Project topic{index} currently has constraint group{index % 41}. "
|
|
164
|
+
f"Unrelated note category{index % 73} is archived."
|
|
165
|
+
),
|
|
166
|
+
status=MemoryStatus.ACTIVE,
|
|
167
|
+
type=MemoryType.PROJECT,
|
|
168
|
+
),
|
|
169
|
+
final_score=1.0 / (index + 1),
|
|
170
|
+
rank=index + 1,
|
|
171
|
+
)
|
|
172
|
+
for index in range(count)
|
|
173
|
+
]
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
async def main() -> None:
|
|
177
|
+
compiler = QueryAwareContextCompiler(
|
|
178
|
+
token_counter=DeterministicWordTokenCounter()
|
|
179
|
+
)
|
|
180
|
+
memories = smoke_memories()
|
|
181
|
+
strategies = (
|
|
182
|
+
CompilationStrategy.RAW_CONCAT,
|
|
183
|
+
CompilationStrategy.DEDUP_ONLY,
|
|
184
|
+
CompilationStrategy.CONTEXTOS_COMPILER,
|
|
185
|
+
)
|
|
186
|
+
print(
|
|
187
|
+
"Budget Strategy In Out Ratio Recall Weighted "
|
|
188
|
+
"Unsupported Redundancy Violations Provenance"
|
|
189
|
+
)
|
|
190
|
+
results: dict[tuple[int, CompilationStrategy], CompiledContext] = {}
|
|
191
|
+
for budget in (40, 60, 100, 200):
|
|
192
|
+
for strategy in strategies:
|
|
193
|
+
context = await compiler.compile(
|
|
194
|
+
SMOKE_QUERY,
|
|
195
|
+
memories,
|
|
196
|
+
CompilationConfig(
|
|
197
|
+
budget=budget,
|
|
198
|
+
strategy=strategy,
|
|
199
|
+
compression_level=CompressionLevel.LIGHT,
|
|
200
|
+
),
|
|
201
|
+
)
|
|
202
|
+
results[(budget, strategy)] = context
|
|
203
|
+
metrics = measure(context, memories)
|
|
204
|
+
print(
|
|
205
|
+
f"{budget:>6} {strategy.value:<20} "
|
|
206
|
+
f"{metrics.input_tokens:>3} {metrics.output_tokens:>3} "
|
|
207
|
+
f"{metrics.compression_ratio:>5.3f} "
|
|
208
|
+
f"{metrics.information_recall:>6.3f} "
|
|
209
|
+
f"{metrics.weighted_preservation:>8.3f} "
|
|
210
|
+
f"{metrics.unsupported_fact_rate:>11.3f} "
|
|
211
|
+
f"{metrics.redundancy:>10.3f} "
|
|
212
|
+
f"{metrics.budget_violations:>10} "
|
|
213
|
+
f"{metrics.provenance_coverage:>10.3f}"
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
print("\nExact compiled contexts")
|
|
217
|
+
for budget in (40, 60, 100, 200):
|
|
218
|
+
print(f"\nBudget {budget}")
|
|
219
|
+
for strategy in strategies:
|
|
220
|
+
text = results[(budget, strategy)].context_text or "<empty>"
|
|
221
|
+
print(f"[{strategy.value}]\n{text}")
|
|
222
|
+
|
|
223
|
+
scale = synthetic_memories()
|
|
224
|
+
started = time.perf_counter()
|
|
225
|
+
first = await compiler.compile(
|
|
226
|
+
"What project constraints are current?",
|
|
227
|
+
scale,
|
|
228
|
+
CompilationConfig(budget=500),
|
|
229
|
+
)
|
|
230
|
+
latency_ms = (time.perf_counter() - started) * 1000
|
|
231
|
+
second = await compiler.compile(
|
|
232
|
+
"What project constraints are current?",
|
|
233
|
+
scale,
|
|
234
|
+
CompilationConfig(budget=500),
|
|
235
|
+
)
|
|
236
|
+
print("\nSynthetic scale")
|
|
237
|
+
print(
|
|
238
|
+
f"inputs=1000 facts={len(first.facts)} tokens={first.total_tokens}/500 "
|
|
239
|
+
f"latency_ms={latency_ms:.3f} "
|
|
240
|
+
f"deterministic={first.context_text == second.context_text}"
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
if __name__ == "__main__":
|
|
245
|
+
asyncio.run(main())
|