contextos-memory-runtime 1.0.0rc2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contextos/__init__.py +3 -0
- contextos/__main__.py +6 -0
- contextos/api/__init__.py +1 -0
- contextos/api/routes/__init__.py +1 -0
- contextos/api/routes/desktop.py +322 -0
- contextos/api/routes/ingest.py +17 -0
- contextos/api/routes/memories.py +84 -0
- contextos/api/routes/models.py +81 -0
- contextos/api/routes/retrieval.py +89 -0
- contextos/api/routes/system.py +216 -0
- contextos/api/server.py +195 -0
- contextos/benchmarks/__init__.py +1 -0
- contextos/benchmarks/compilation.py +245 -0
- contextos/benchmarks/connectors.py +423 -0
- contextos/benchmarks/explainability.py +103 -0
- contextos/benchmarks/final.py +406 -0
- contextos/benchmarks/graph.py +310 -0
- contextos/benchmarks/graph_adversarial.py +525 -0
- contextos/benchmarks/mcp.py +324 -0
- contextos/benchmarks/model_routing.py +203 -0
- contextos/benchmarks/optimization.py +305 -0
- contextos/benchmarks/rescue_integration.py +127 -0
- contextos/benchmarks/retrieval.py +266 -0
- contextos/benchmarks/temporal.py +377 -0
- contextos/benchmarks/temporal_hotpath.py +76 -0
- contextos/benchmarks/terminal.py +62 -0
- contextos/cli/__init__.py +1 -0
- contextos/cli/app.py +932 -0
- contextos/cli/dashboard.py +174 -0
- contextos/cli/formatters.py +299 -0
- contextos/config/__init__.py +1 -0
- contextos/config/settings.py +160 -0
- contextos/connectors/__init__.py +6 -0
- contextos/connectors/fake.py +11 -0
- contextos/connectors/json_import.py +125 -0
- contextos/connectors/local_files.py +102 -0
- contextos/connectors/manager.py +293 -0
- contextos/connectors/models.py +62 -0
- contextos/connectors/protocols.py +11 -0
- contextos/core/__init__.py +103 -0
- contextos/core/enums.py +489 -0
- contextos/core/exceptions.py +293 -0
- contextos/core/models.py +1147 -0
- contextos/core/protocols.py +549 -0
- contextos/daemon/__init__.py +1 -0
- contextos/daemon/manager.py +510 -0
- contextos/daemon/state.py +127 -0
- contextos/daemon/wiring.py +296 -0
- contextos/demo.py +217 -0
- contextos/embedding/__init__.py +1 -0
- contextos/embedding/deterministic.py +76 -0
- contextos/embedding/sentence_transformers.py +80 -0
- contextos/mcp/__init__.py +5 -0
- contextos/mcp/server.py +269 -0
- contextos/providers/__init__.py +13 -0
- contextos/providers/fake.py +217 -0
- contextos/providers/ollama.py +297 -0
- contextos/providers/openai_compatible.py +337 -0
- contextos/services/__init__.py +1 -0
- contextos/services/compilation.py +535 -0
- contextos/services/explainability.py +553 -0
- contextos/services/extraction.py +311 -0
- contextos/services/graph.py +524 -0
- contextos/services/graph_retrieval.py +143 -0
- contextos/services/ingestion.py +143 -0
- contextos/services/inspection.py +174 -0
- contextos/services/memory.py +291 -0
- contextos/services/model_service.py +409 -0
- contextos/services/optimization.py +426 -0
- contextos/services/privacy.py +331 -0
- contextos/services/retrieval.py +302 -0
- contextos/services/retrieval_index.py +88 -0
- contextos/services/router.py +302 -0
- contextos/services/secret_scanner.py +207 -0
- contextos/services/telemetry_query.py +102 -0
- contextos/services/temporal.py +500 -0
- contextos/services/token_counter.py +222 -0
- contextos/storage/__init__.py +1 -0
- contextos/storage/connector_repo.py +67 -0
- contextos/storage/database.py +497 -0
- contextos/storage/event_repo.py +137 -0
- contextos/storage/graph_repo.py +228 -0
- contextos/storage/lexical/__init__.py +1 -0
- contextos/storage/lexical/bm25.py +134 -0
- contextos/storage/memory_repo.py +589 -0
- contextos/storage/relation_repo.py +80 -0
- contextos/storage/telemetry_repo.py +481 -0
- contextos/storage/vector/__init__.py +1 -0
- contextos/storage/vector/in_memory.py +162 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,481 @@
|
|
|
1
|
+
"""SQLite persistence for model invocation telemetry."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import logging
|
|
7
|
+
import re
|
|
8
|
+
from datetime import datetime
|
|
9
|
+
from typing import Any
|
|
10
|
+
from uuid import UUID
|
|
11
|
+
|
|
12
|
+
import aiosqlite
|
|
13
|
+
|
|
14
|
+
from contextos.core.enums import ModelFinishReason, RoutingPolicy, TokenMeasurementSource
|
|
15
|
+
from contextos.core.models import ModelInvocationTelemetry, TelemetrySummary
|
|
16
|
+
|
|
17
|
+
logger = logging.getLogger(__name__)
|
|
18
|
+
|
|
19
|
+
_SECRET_PATTERNS = [
|
|
20
|
+
re.compile(r"Bearer\s+[A-Za-z0-9_\-\.~+/]+=*", re.IGNORECASE),
|
|
21
|
+
re.compile(r"sk-[A-Za-z0-9_\-]{8,}", re.IGNORECASE),
|
|
22
|
+
re.compile(r"gh[pousr]-[A-Za-z0-9_]{16,}", re.IGNORECASE),
|
|
23
|
+
re.compile(r"(?:api[-_]?key|secret|password|token)\s*[:=]\s*['\"]?[A-Za-z0-9_\-\.~+/]+['\"]?", re.IGNORECASE),
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
_SENSITIVE_KEY_SUBSTRINGS = {
|
|
27
|
+
"api_key",
|
|
28
|
+
"apikey",
|
|
29
|
+
"secret",
|
|
30
|
+
"password",
|
|
31
|
+
"token",
|
|
32
|
+
"auth",
|
|
33
|
+
"authorization",
|
|
34
|
+
"raw_prompt",
|
|
35
|
+
"raw_response",
|
|
36
|
+
"cookie",
|
|
37
|
+
"credential",
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def sanitize_telemetry_metadata(val: Any) -> Any:
|
|
42
|
+
"""Recursively sanitize metadata dicts/lists to strip secrets and raw payloads."""
|
|
43
|
+
if isinstance(val, dict):
|
|
44
|
+
cleaned: dict[str, Any] = {}
|
|
45
|
+
for k, v in val.items():
|
|
46
|
+
key_lower = str(k).lower()
|
|
47
|
+
if any(s in key_lower for s in _SENSITIVE_KEY_SUBSTRINGS):
|
|
48
|
+
cleaned[str(k)] = "[REDACTED]"
|
|
49
|
+
else:
|
|
50
|
+
cleaned[str(k)] = sanitize_telemetry_metadata(v)
|
|
51
|
+
return cleaned
|
|
52
|
+
elif isinstance(val, list):
|
|
53
|
+
return [sanitize_telemetry_metadata(item) for item in val]
|
|
54
|
+
elif isinstance(val, str):
|
|
55
|
+
sanitized_str = val
|
|
56
|
+
for pat in _SECRET_PATTERNS:
|
|
57
|
+
sanitized_str = pat.sub("[REDACTED_SECRET]", sanitized_str)
|
|
58
|
+
return sanitized_str
|
|
59
|
+
return val
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class SqliteTelemetryRepository:
|
|
64
|
+
"""Stores and queries model invocation metrics in SQLite."""
|
|
65
|
+
|
|
66
|
+
def __init__(self, conn: aiosqlite.Connection) -> None:
|
|
67
|
+
self._conn = conn
|
|
68
|
+
|
|
69
|
+
async def record(self, telemetry: ModelInvocationTelemetry) -> None:
|
|
70
|
+
"""Persist a single model invocation telemetry record."""
|
|
71
|
+
row_id = str(telemetry.invocation_id)
|
|
72
|
+
safe_meta = sanitize_telemetry_metadata(telemetry.metadata)
|
|
73
|
+
metadata_json = json.dumps(safe_meta)
|
|
74
|
+
timestamp_str = telemetry.timestamp.isoformat()
|
|
75
|
+
|
|
76
|
+
sql = """
|
|
77
|
+
INSERT INTO model_invocations (
|
|
78
|
+
id,
|
|
79
|
+
invocation_id,
|
|
80
|
+
session_id,
|
|
81
|
+
provider_id,
|
|
82
|
+
model_id,
|
|
83
|
+
is_local,
|
|
84
|
+
timestamp,
|
|
85
|
+
candidate_context_tokens,
|
|
86
|
+
retrieved_context_tokens,
|
|
87
|
+
optimized_context_tokens,
|
|
88
|
+
compiled_context_tokens,
|
|
89
|
+
prompt_tokens_before_context,
|
|
90
|
+
final_input_tokens,
|
|
91
|
+
provider_input_tokens,
|
|
92
|
+
provider_output_tokens,
|
|
93
|
+
provider_total_tokens,
|
|
94
|
+
token_measurement_source,
|
|
95
|
+
context_tokens_avoided,
|
|
96
|
+
reduction_ratio,
|
|
97
|
+
lexical_candidate_count,
|
|
98
|
+
dense_candidate_count,
|
|
99
|
+
hybrid_candidate_count,
|
|
100
|
+
graph_expanded_count,
|
|
101
|
+
temporal_filtered_count,
|
|
102
|
+
selected_memory_count,
|
|
103
|
+
compiled_fact_count,
|
|
104
|
+
retrieval_ms,
|
|
105
|
+
optimization_ms,
|
|
106
|
+
compilation_ms,
|
|
107
|
+
routing_ms,
|
|
108
|
+
token_counting_ms,
|
|
109
|
+
provider_latency_ms,
|
|
110
|
+
end_to_end_ms,
|
|
111
|
+
routing_policy,
|
|
112
|
+
routing_reason,
|
|
113
|
+
selected_provider,
|
|
114
|
+
selected_model,
|
|
115
|
+
fallback_used,
|
|
116
|
+
fallback_reason,
|
|
117
|
+
finish_reason,
|
|
118
|
+
status,
|
|
119
|
+
error_code,
|
|
120
|
+
metadata,
|
|
121
|
+
context_token_measurement_source,
|
|
122
|
+
context_tokenizer
|
|
123
|
+
) VALUES (
|
|
124
|
+
?, ?, ?, ?, ?, ?, ?, ?, ?, ?,
|
|
125
|
+
?, ?, ?, ?, ?, ?, ?, ?, ?, ?,
|
|
126
|
+
?, ?, ?, ?, ?, ?, ?, ?, ?, ?,
|
|
127
|
+
?, ?, ?, ?, ?, ?, ?, ?, ?, ?,
|
|
128
|
+
?, ?, ?, ?, ?
|
|
129
|
+
)
|
|
130
|
+
"""
|
|
131
|
+
params = (
|
|
132
|
+
row_id,
|
|
133
|
+
str(telemetry.invocation_id),
|
|
134
|
+
telemetry.session_id,
|
|
135
|
+
telemetry.provider_id,
|
|
136
|
+
telemetry.model_id,
|
|
137
|
+
1 if telemetry.is_local else 0,
|
|
138
|
+
timestamp_str,
|
|
139
|
+
telemetry.candidate_context_tokens,
|
|
140
|
+
telemetry.retrieved_context_tokens,
|
|
141
|
+
telemetry.optimized_context_tokens,
|
|
142
|
+
telemetry.compiled_context_tokens,
|
|
143
|
+
telemetry.prompt_tokens_before_context,
|
|
144
|
+
telemetry.final_input_tokens,
|
|
145
|
+
telemetry.provider_input_tokens,
|
|
146
|
+
telemetry.provider_output_tokens,
|
|
147
|
+
telemetry.provider_total_tokens,
|
|
148
|
+
telemetry.token_measurement_source.value,
|
|
149
|
+
telemetry.context_tokens_avoided,
|
|
150
|
+
telemetry.reduction_ratio,
|
|
151
|
+
telemetry.lexical_candidate_count,
|
|
152
|
+
telemetry.dense_candidate_count,
|
|
153
|
+
telemetry.hybrid_candidate_count,
|
|
154
|
+
telemetry.graph_expanded_count,
|
|
155
|
+
telemetry.temporal_filtered_count,
|
|
156
|
+
telemetry.selected_memory_count,
|
|
157
|
+
telemetry.compiled_fact_count,
|
|
158
|
+
telemetry.retrieval_ms,
|
|
159
|
+
telemetry.optimization_ms,
|
|
160
|
+
telemetry.compilation_ms,
|
|
161
|
+
telemetry.routing_ms,
|
|
162
|
+
telemetry.token_counting_ms,
|
|
163
|
+
telemetry.provider_latency_ms,
|
|
164
|
+
telemetry.end_to_end_ms,
|
|
165
|
+
telemetry.routing_policy.value,
|
|
166
|
+
telemetry.routing_reason,
|
|
167
|
+
telemetry.selected_provider,
|
|
168
|
+
telemetry.selected_model,
|
|
169
|
+
1 if telemetry.fallback_used else 0,
|
|
170
|
+
telemetry.fallback_reason,
|
|
171
|
+
telemetry.finish_reason.value,
|
|
172
|
+
telemetry.status,
|
|
173
|
+
telemetry.error_code,
|
|
174
|
+
metadata_json,
|
|
175
|
+
telemetry.context_token_measurement_source.value if telemetry.context_token_measurement_source else None,
|
|
176
|
+
telemetry.context_tokenizer,
|
|
177
|
+
)
|
|
178
|
+
await self._conn.execute(sql, params)
|
|
179
|
+
await self._conn.commit()
|
|
180
|
+
|
|
181
|
+
async def get(self, invocation_id: UUID) -> ModelInvocationTelemetry | None:
|
|
182
|
+
"""Retrieve a telemetry record by invocation UUID."""
|
|
183
|
+
sql = "SELECT * FROM model_invocations WHERE invocation_id = ?"
|
|
184
|
+
async with self._conn.execute(sql, (str(invocation_id),)) as cursor:
|
|
185
|
+
row = await cursor.fetchone()
|
|
186
|
+
if row is None:
|
|
187
|
+
return None
|
|
188
|
+
return self._row_to_model(row)
|
|
189
|
+
|
|
190
|
+
async def list_recent(
|
|
191
|
+
self, limit: int = 50, model_id: str | None = None,
|
|
192
|
+
provider_id: str | None = None, start: datetime | None = None,
|
|
193
|
+
) -> list[ModelInvocationTelemetry]:
|
|
194
|
+
"""List recent invocations in descending chronological order."""
|
|
195
|
+
sql = "SELECT * FROM model_invocations WHERE 1=1"
|
|
196
|
+
params: list[Any] = []
|
|
197
|
+
if model_id is not None:
|
|
198
|
+
sql += " AND model_id = ?"
|
|
199
|
+
params.append(model_id)
|
|
200
|
+
if provider_id is not None:
|
|
201
|
+
sql += " AND provider_id = ?"
|
|
202
|
+
params.append(provider_id)
|
|
203
|
+
if start is not None:
|
|
204
|
+
sql += " AND timestamp >= ?"
|
|
205
|
+
params.append(start.isoformat())
|
|
206
|
+
sql += " ORDER BY timestamp DESC LIMIT ?"
|
|
207
|
+
params.append(max(1, min(limit, 100)))
|
|
208
|
+
async with self._conn.execute(sql, params) as cursor:
|
|
209
|
+
rows = await cursor.fetchall()
|
|
210
|
+
return [self._row_to_model(r) for r in rows]
|
|
211
|
+
|
|
212
|
+
async def provider_model_breakdown(
|
|
213
|
+
self, start: datetime | None = None, provider_id: str | None = None,
|
|
214
|
+
model_id: str | None = None,
|
|
215
|
+
) -> list[dict[str, Any]]:
|
|
216
|
+
"""Group successful context counts by provider, model and tokenizer basis."""
|
|
217
|
+
conditions = ["1=1"]
|
|
218
|
+
params: list[Any] = []
|
|
219
|
+
if start is not None:
|
|
220
|
+
conditions.append("timestamp >= ?")
|
|
221
|
+
params.append(start.isoformat())
|
|
222
|
+
if provider_id is not None:
|
|
223
|
+
conditions.append("provider_id = ?")
|
|
224
|
+
params.append(provider_id)
|
|
225
|
+
if model_id is not None:
|
|
226
|
+
conditions.append("model_id = ?")
|
|
227
|
+
params.append(model_id)
|
|
228
|
+
sql = f"""
|
|
229
|
+
SELECT provider_id, model_id, is_local,
|
|
230
|
+
COALESCE(context_token_measurement_source, 'unknown') AS basis,
|
|
231
|
+
COALESCE(context_tokenizer, 'unknown') AS tokenizer,
|
|
232
|
+
SUM(CASE WHEN status = 'success' THEN 1 ELSE 0 END) AS successful,
|
|
233
|
+
SUM(CASE WHEN status != 'success' THEN 1 ELSE 0 END) AS errors,
|
|
234
|
+
COALESCE(SUM(CASE WHEN status = 'success' THEN candidate_context_tokens END), 0),
|
|
235
|
+
COALESCE(SUM(CASE WHEN status = 'success' THEN compiled_context_tokens END), 0),
|
|
236
|
+
COALESCE(SUM(CASE WHEN status = 'success' THEN context_tokens_avoided END), 0),
|
|
237
|
+
COALESCE(SUM(CASE WHEN status = 'success' THEN final_input_tokens END), 0),
|
|
238
|
+
COALESCE(SUM(CASE WHEN status = 'success' THEN provider_input_tokens END), 0),
|
|
239
|
+
COALESCE(SUM(CASE WHEN status = 'success' THEN provider_output_tokens END), 0),
|
|
240
|
+
COALESCE(SUM(graph_expanded_count), 0),
|
|
241
|
+
COALESCE(SUM(selected_memory_count), 0),
|
|
242
|
+
COALESCE(AVG(retrieval_ms), 0),
|
|
243
|
+
COALESCE(AVG(compilation_ms), 0),
|
|
244
|
+
COALESCE(AVG(provider_latency_ms), 0)
|
|
245
|
+
FROM model_invocations WHERE {' AND '.join(conditions)}
|
|
246
|
+
GROUP BY provider_id, model_id, is_local, basis, tokenizer
|
|
247
|
+
ORDER BY successful DESC, provider_id, model_id LIMIT 50
|
|
248
|
+
"""
|
|
249
|
+
async with self._conn.execute(sql, params) as cursor:
|
|
250
|
+
rows = await cursor.fetchall()
|
|
251
|
+
return [{
|
|
252
|
+
"provider": row[0], "model": row[1], "local": bool(row[2]),
|
|
253
|
+
"context_measurement_source": row[3], "context_tokenizer": row[4],
|
|
254
|
+
"invocations": row[5], "errors": row[6],
|
|
255
|
+
"candidate_context_tokens": row[7], "compiled_context_tokens": row[8],
|
|
256
|
+
"context_tokens_avoided": row[9], "preflight_input_tokens": row[10],
|
|
257
|
+
"provider_input_tokens": row[11], "provider_output_tokens": row[12],
|
|
258
|
+
"graph_expanded_count": row[13], "selected_memory_count": row[14],
|
|
259
|
+
"average_retrieval_ms": row[15], "average_compilation_ms": row[16],
|
|
260
|
+
"average_provider_ms": row[17],
|
|
261
|
+
"weighted_reduction_ratio": row[9] / row[7] if row[7] else None,
|
|
262
|
+
} for row in rows]
|
|
263
|
+
|
|
264
|
+
async def count(self) -> int:
|
|
265
|
+
"""Count total recorded invocations."""
|
|
266
|
+
sql = "SELECT COUNT(*) FROM model_invocations"
|
|
267
|
+
async with self._conn.execute(sql) as cursor:
|
|
268
|
+
row = await cursor.fetchone()
|
|
269
|
+
return row[0] if row else 0
|
|
270
|
+
|
|
271
|
+
async def context_measurement_bases(
|
|
272
|
+
self, model_id: str | None = None, provider_id: str | None = None,
|
|
273
|
+
start: datetime | None = None,
|
|
274
|
+
) -> list[dict[str, str]]:
|
|
275
|
+
"""Distinct provenance bases; unknown legacy rows are preserved as unknown."""
|
|
276
|
+
sql = """SELECT DISTINCT COALESCE(context_token_measurement_source, 'unknown'),
|
|
277
|
+
COALESCE(context_tokenizer, 'unknown') FROM model_invocations WHERE status = 'success'"""
|
|
278
|
+
params: list[str] = []
|
|
279
|
+
if model_id is not None:
|
|
280
|
+
sql += " AND model_id = ?"
|
|
281
|
+
params.append(model_id)
|
|
282
|
+
if provider_id is not None:
|
|
283
|
+
sql += " AND provider_id = ?"
|
|
284
|
+
params.append(provider_id)
|
|
285
|
+
if start is not None:
|
|
286
|
+
sql += " AND timestamp >= ?"
|
|
287
|
+
params.append(start.isoformat())
|
|
288
|
+
async with self._conn.execute(sql, params) as cursor:
|
|
289
|
+
rows = await cursor.fetchall()
|
|
290
|
+
return [{"source": row[0], "tokenizer": row[1]} for row in rows]
|
|
291
|
+
|
|
292
|
+
async def summary(
|
|
293
|
+
self,
|
|
294
|
+
start: datetime | None = None,
|
|
295
|
+
end: datetime | None = None,
|
|
296
|
+
provider_id: str | None = None,
|
|
297
|
+
model_id: str | None = None,
|
|
298
|
+
success_only: bool = False,
|
|
299
|
+
) -> TelemetrySummary:
|
|
300
|
+
"""Compute aggregated token usage, avoidance, and latency statistics."""
|
|
301
|
+
conditions: list[str] = []
|
|
302
|
+
params: list[Any] = []
|
|
303
|
+
|
|
304
|
+
if start is not None:
|
|
305
|
+
conditions.append("timestamp >= ?")
|
|
306
|
+
params.append(start.isoformat())
|
|
307
|
+
if end is not None:
|
|
308
|
+
conditions.append("timestamp <= ?")
|
|
309
|
+
params.append(end.isoformat())
|
|
310
|
+
if provider_id is not None:
|
|
311
|
+
conditions.append("provider_id = ?")
|
|
312
|
+
params.append(provider_id)
|
|
313
|
+
if model_id is not None:
|
|
314
|
+
conditions.append("model_id = ?")
|
|
315
|
+
params.append(model_id)
|
|
316
|
+
if success_only:
|
|
317
|
+
conditions.append("status = 'success'")
|
|
318
|
+
|
|
319
|
+
where_clause = f" WHERE {' AND '.join(conditions)}" if conditions else ""
|
|
320
|
+
|
|
321
|
+
agg_sql = f"""
|
|
322
|
+
SELECT
|
|
323
|
+
COUNT(*),
|
|
324
|
+
COALESCE(SUM(provider_input_tokens), 0),
|
|
325
|
+
COALESCE(SUM(provider_output_tokens), 0),
|
|
326
|
+
COALESCE(SUM(context_tokens_avoided), 0),
|
|
327
|
+
COALESCE(AVG(reduction_ratio), 0.0),
|
|
328
|
+
COALESCE(AVG(provider_latency_ms), 0.0),
|
|
329
|
+
COALESCE(SUM(CASE WHEN is_local = 1 THEN 1 ELSE 0 END), 0),
|
|
330
|
+
COALESCE(SUM(CASE WHEN is_local = 0 THEN 1 ELSE 0 END), 0),
|
|
331
|
+
COALESCE(SUM(candidate_context_tokens), 0)
|
|
332
|
+
FROM model_invocations{where_clause}
|
|
333
|
+
"""
|
|
334
|
+
|
|
335
|
+
async with self._conn.execute(agg_sql, params) as cursor:
|
|
336
|
+
row = await cursor.fetchone()
|
|
337
|
+
|
|
338
|
+
total_invocations = row[0] if row else 0
|
|
339
|
+
total_input_tokens = row[1] if row else 0
|
|
340
|
+
total_output_tokens = row[2] if row else 0
|
|
341
|
+
total_tokens_avoided = row[3] if row else 0
|
|
342
|
+
average_reduction_ratio = float(row[4]) if row else 0.0
|
|
343
|
+
average_provider_latency_ms = float(row[5]) if row else 0.0
|
|
344
|
+
local_invocations = row[6] if row else 0
|
|
345
|
+
remote_invocations = row[7] if row else 0
|
|
346
|
+
total_candidate_tokens = row[8] if row else 0
|
|
347
|
+
weighted_reduction_ratio = (
|
|
348
|
+
float(total_tokens_avoided / total_candidate_tokens)
|
|
349
|
+
if total_candidate_tokens > 0
|
|
350
|
+
else 0.0
|
|
351
|
+
)
|
|
352
|
+
|
|
353
|
+
# By provider breakdown
|
|
354
|
+
by_provider: dict[str, Any] = {}
|
|
355
|
+
prov_sql = f"""
|
|
356
|
+
SELECT
|
|
357
|
+
provider_id,
|
|
358
|
+
COUNT(*),
|
|
359
|
+
COALESCE(SUM(provider_input_tokens), 0),
|
|
360
|
+
COALESCE(SUM(provider_output_tokens), 0),
|
|
361
|
+
COALESCE(SUM(context_tokens_avoided), 0),
|
|
362
|
+
COALESCE(AVG(reduction_ratio), 0.0),
|
|
363
|
+
COALESCE(AVG(provider_latency_ms), 0.0),
|
|
364
|
+
COALESCE(SUM(candidate_context_tokens), 0)
|
|
365
|
+
FROM model_invocations{where_clause}
|
|
366
|
+
GROUP BY provider_id
|
|
367
|
+
"""
|
|
368
|
+
async with self._conn.execute(prov_sql, params) as cursor:
|
|
369
|
+
async for p_row in cursor:
|
|
370
|
+
p_avoided = p_row[4]
|
|
371
|
+
p_cand = p_row[7]
|
|
372
|
+
by_provider[p_row[0]] = {
|
|
373
|
+
"invocations": p_row[1],
|
|
374
|
+
"input_tokens": p_row[2],
|
|
375
|
+
"output_tokens": p_row[3],
|
|
376
|
+
"tokens_avoided": p_avoided,
|
|
377
|
+
"average_reduction_ratio": float(p_row[5]),
|
|
378
|
+
"weighted_reduction_ratio": float(p_avoided / p_cand) if p_cand > 0 else 0.0,
|
|
379
|
+
"average_latency_ms": float(p_row[6]),
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
# By model breakdown
|
|
383
|
+
by_model: dict[str, Any] = {}
|
|
384
|
+
model_sql = f"""
|
|
385
|
+
SELECT
|
|
386
|
+
model_id,
|
|
387
|
+
COUNT(*),
|
|
388
|
+
COALESCE(SUM(provider_input_tokens), 0),
|
|
389
|
+
COALESCE(SUM(provider_output_tokens), 0),
|
|
390
|
+
COALESCE(SUM(context_tokens_avoided), 0),
|
|
391
|
+
COALESCE(AVG(reduction_ratio), 0.0),
|
|
392
|
+
COALESCE(AVG(provider_latency_ms), 0.0),
|
|
393
|
+
COALESCE(SUM(candidate_context_tokens), 0)
|
|
394
|
+
FROM model_invocations{where_clause}
|
|
395
|
+
GROUP BY model_id
|
|
396
|
+
"""
|
|
397
|
+
async with self._conn.execute(model_sql, params) as cursor:
|
|
398
|
+
async for m_row in cursor:
|
|
399
|
+
m_avoided = m_row[4]
|
|
400
|
+
m_cand = m_row[7]
|
|
401
|
+
by_model[m_row[0]] = {
|
|
402
|
+
"invocations": m_row[1],
|
|
403
|
+
"input_tokens": m_row[2],
|
|
404
|
+
"output_tokens": m_row[3],
|
|
405
|
+
"tokens_avoided": m_avoided,
|
|
406
|
+
"average_reduction_ratio": float(m_row[5]),
|
|
407
|
+
"weighted_reduction_ratio": float(m_avoided / m_cand) if m_cand > 0 else 0.0,
|
|
408
|
+
"average_latency_ms": float(m_row[6]),
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
return TelemetrySummary(
|
|
412
|
+
total_invocations=total_invocations,
|
|
413
|
+
total_input_tokens=total_input_tokens,
|
|
414
|
+
total_output_tokens=total_output_tokens,
|
|
415
|
+
total_tokens_avoided=total_tokens_avoided,
|
|
416
|
+
average_reduction_ratio=average_reduction_ratio,
|
|
417
|
+
weighted_reduction_ratio=weighted_reduction_ratio,
|
|
418
|
+
average_provider_latency_ms=average_provider_latency_ms,
|
|
419
|
+
local_invocations=local_invocations,
|
|
420
|
+
remote_invocations=remote_invocations,
|
|
421
|
+
by_provider=by_provider,
|
|
422
|
+
by_model=by_model,
|
|
423
|
+
)
|
|
424
|
+
|
|
425
|
+
def _row_to_model(self, row: Any) -> ModelInvocationTelemetry:
|
|
426
|
+
metadata_raw = row["metadata"] if "metadata" in row.keys() else "{}"
|
|
427
|
+
try:
|
|
428
|
+
metadata = json.loads(metadata_raw) if isinstance(metadata_raw, str) else metadata_raw
|
|
429
|
+
except Exception:
|
|
430
|
+
metadata = {}
|
|
431
|
+
|
|
432
|
+
return ModelInvocationTelemetry(
|
|
433
|
+
invocation_id=UUID(row["invocation_id"]),
|
|
434
|
+
session_id=row["session_id"],
|
|
435
|
+
provider_id=row["provider_id"],
|
|
436
|
+
model_id=row["model_id"],
|
|
437
|
+
is_local=bool(row["is_local"]),
|
|
438
|
+
timestamp=datetime.fromisoformat(row["timestamp"]),
|
|
439
|
+
candidate_context_tokens=row["candidate_context_tokens"],
|
|
440
|
+
retrieved_context_tokens=row["retrieved_context_tokens"],
|
|
441
|
+
optimized_context_tokens=row["optimized_context_tokens"],
|
|
442
|
+
compiled_context_tokens=row["compiled_context_tokens"],
|
|
443
|
+
prompt_tokens_before_context=row["prompt_tokens_before_context"],
|
|
444
|
+
preflight_input_tokens=row["final_input_tokens"],
|
|
445
|
+
final_input_tokens=row["final_input_tokens"],
|
|
446
|
+
provider_input_tokens=row["provider_input_tokens"],
|
|
447
|
+
provider_output_tokens=row["provider_output_tokens"],
|
|
448
|
+
provider_total_tokens=row["provider_total_tokens"],
|
|
449
|
+
token_measurement_source=TokenMeasurementSource(row["token_measurement_source"]),
|
|
450
|
+
context_token_measurement_source=(
|
|
451
|
+
TokenMeasurementSource(row["context_token_measurement_source"])
|
|
452
|
+
if row["context_token_measurement_source"] else None
|
|
453
|
+
),
|
|
454
|
+
context_tokenizer=row["context_tokenizer"],
|
|
455
|
+
context_tokens_avoided=row["context_tokens_avoided"],
|
|
456
|
+
reduction_ratio=float(row["reduction_ratio"]),
|
|
457
|
+
lexical_candidate_count=row["lexical_candidate_count"],
|
|
458
|
+
dense_candidate_count=row["dense_candidate_count"],
|
|
459
|
+
hybrid_candidate_count=row["hybrid_candidate_count"],
|
|
460
|
+
graph_expanded_count=row["graph_expanded_count"],
|
|
461
|
+
temporal_filtered_count=row["temporal_filtered_count"],
|
|
462
|
+
selected_memory_count=row["selected_memory_count"],
|
|
463
|
+
compiled_fact_count=row["compiled_fact_count"],
|
|
464
|
+
retrieval_ms=float(row["retrieval_ms"]),
|
|
465
|
+
optimization_ms=float(row["optimization_ms"]),
|
|
466
|
+
compilation_ms=float(row["compilation_ms"]),
|
|
467
|
+
routing_ms=float(row["routing_ms"]),
|
|
468
|
+
token_counting_ms=float(row["token_counting_ms"]),
|
|
469
|
+
provider_latency_ms=float(row["provider_latency_ms"]),
|
|
470
|
+
end_to_end_ms=float(row["end_to_end_ms"]),
|
|
471
|
+
routing_policy=RoutingPolicy(row["routing_policy"]),
|
|
472
|
+
routing_reason=row["routing_reason"],
|
|
473
|
+
selected_provider=row["selected_provider"],
|
|
474
|
+
selected_model=row["selected_model"],
|
|
475
|
+
fallback_used=bool(row["fallback_used"]),
|
|
476
|
+
fallback_reason=row["fallback_reason"],
|
|
477
|
+
finish_reason=ModelFinishReason(row["finish_reason"]),
|
|
478
|
+
status=row["status"],
|
|
479
|
+
error_code=row["error_code"],
|
|
480
|
+
metadata=metadata if isinstance(metadata, dict) else {},
|
|
481
|
+
)
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Vector storage sub-package for ContextOS."""
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
"""In-memory vector store using numpy for ContextOS.
|
|
2
|
+
|
|
3
|
+
Phase 1 implementation: simple brute-force cosine similarity search.
|
|
4
|
+
No ANN index — at Phase 1 scale (< 10K vectors), brute force is fast enough
|
|
5
|
+
and avoids external library complexity.
|
|
6
|
+
|
|
7
|
+
This will be replaced by sqlite-vec or LanceDB when we need ANN performance.
|
|
8
|
+
It satisfies the VectorStore protocol and passes the same contract tests.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import logging
|
|
14
|
+
from typing import Any
|
|
15
|
+
|
|
16
|
+
import numpy as np
|
|
17
|
+
|
|
18
|
+
from contextos.core.models import VectorResult
|
|
19
|
+
|
|
20
|
+
logger = logging.getLogger(__name__)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class InMemoryVectorStore:
|
|
24
|
+
"""Brute-force cosine similarity vector store.
|
|
25
|
+
|
|
26
|
+
Implements the VectorStore protocol.
|
|
27
|
+
|
|
28
|
+
Suitable for Phase 1 (< 10K vectors). O(n) search, but n is small.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
def __init__(self, dimension: int = 384) -> None:
|
|
32
|
+
self._dimension = dimension
|
|
33
|
+
self._ids: list[str] = []
|
|
34
|
+
self._vectors: np.ndarray = np.empty((0, dimension), dtype=np.float32)
|
|
35
|
+
self._metadata: dict[str, dict[str, Any]] = {}
|
|
36
|
+
|
|
37
|
+
def contains(self, doc_id: str | Any) -> bool:
|
|
38
|
+
"""Check whether a document ID is present in the vector store."""
|
|
39
|
+
return str(doc_id) in self._metadata
|
|
40
|
+
|
|
41
|
+
def __contains__(self, doc_id: str | Any) -> bool:
|
|
42
|
+
return str(doc_id) in self._metadata
|
|
43
|
+
|
|
44
|
+
async def add(
|
|
45
|
+
self,
|
|
46
|
+
ids: list[str],
|
|
47
|
+
vectors: list[list[float]],
|
|
48
|
+
metadata: list[dict[str, Any]],
|
|
49
|
+
) -> None:
|
|
50
|
+
"""Add vectors to the store."""
|
|
51
|
+
if not ids:
|
|
52
|
+
return
|
|
53
|
+
if len(ids) != len(vectors) or len(ids) != len(metadata):
|
|
54
|
+
raise ValueError("ids, vectors, and metadata must have equal lengths")
|
|
55
|
+
|
|
56
|
+
new_vectors = np.array(vectors, dtype=np.float32)
|
|
57
|
+
if new_vectors.ndim == 1:
|
|
58
|
+
new_vectors = new_vectors.reshape(1, -1)
|
|
59
|
+
if new_vectors.ndim != 2 or new_vectors.shape[1] != self._dimension:
|
|
60
|
+
raise ValueError(
|
|
61
|
+
f"Expected vectors with dimension {self._dimension}, got shape {new_vectors.shape}"
|
|
62
|
+
)
|
|
63
|
+
if not np.isfinite(new_vectors).all():
|
|
64
|
+
raise ValueError("Vectors must contain only finite values")
|
|
65
|
+
|
|
66
|
+
# Normalize for cosine similarity
|
|
67
|
+
norms = np.linalg.norm(new_vectors, axis=1, keepdims=True)
|
|
68
|
+
norms[norms == 0] = 1.0 # Avoid division by zero
|
|
69
|
+
new_vectors = new_vectors / norms
|
|
70
|
+
|
|
71
|
+
for i, doc_id in enumerate(ids):
|
|
72
|
+
if doc_id in self._metadata:
|
|
73
|
+
# Update: replace existing
|
|
74
|
+
idx = self._ids.index(doc_id)
|
|
75
|
+
self._vectors[idx] = new_vectors[i]
|
|
76
|
+
self._metadata[doc_id] = metadata[i] if i < len(metadata) else {}
|
|
77
|
+
else:
|
|
78
|
+
# Add new
|
|
79
|
+
self._ids.append(doc_id)
|
|
80
|
+
self._vectors = np.vstack([self._vectors, new_vectors[i:i + 1]])
|
|
81
|
+
self._metadata[doc_id] = metadata[i] if i < len(metadata) else {}
|
|
82
|
+
|
|
83
|
+
async def search(
|
|
84
|
+
self,
|
|
85
|
+
vector: list[float],
|
|
86
|
+
top_k: int = 20,
|
|
87
|
+
filters: dict[str, Any] | None = None,
|
|
88
|
+
) -> list[VectorResult]:
|
|
89
|
+
"""Search by cosine similarity."""
|
|
90
|
+
if len(self._ids) == 0:
|
|
91
|
+
return []
|
|
92
|
+
|
|
93
|
+
query = np.array(vector, dtype=np.float32)
|
|
94
|
+
if query.ndim != 1 or query.shape[0] != self._dimension:
|
|
95
|
+
raise ValueError(
|
|
96
|
+
f"Expected query dimension {self._dimension}, got shape {query.shape}"
|
|
97
|
+
)
|
|
98
|
+
if not np.isfinite(query).all():
|
|
99
|
+
raise ValueError("Query vector must contain only finite values")
|
|
100
|
+
norm = np.linalg.norm(query)
|
|
101
|
+
if norm == 0:
|
|
102
|
+
return []
|
|
103
|
+
query = query / norm
|
|
104
|
+
|
|
105
|
+
# Cosine similarity (vectors are pre-normalized)
|
|
106
|
+
similarities = self._vectors @ query
|
|
107
|
+
|
|
108
|
+
# Build results
|
|
109
|
+
indices = sorted(
|
|
110
|
+
range(len(self._ids)),
|
|
111
|
+
key=lambda index: (-float(similarities[index]), self._ids[index]),
|
|
112
|
+
)
|
|
113
|
+
results: list[VectorResult] = []
|
|
114
|
+
|
|
115
|
+
for idx in indices:
|
|
116
|
+
if len(results) >= top_k:
|
|
117
|
+
break
|
|
118
|
+
|
|
119
|
+
doc_id = self._ids[idx]
|
|
120
|
+
score = float(similarities[idx])
|
|
121
|
+
|
|
122
|
+
if score <= 0:
|
|
123
|
+
continue
|
|
124
|
+
|
|
125
|
+
# Apply filters
|
|
126
|
+
if filters:
|
|
127
|
+
meta = self._metadata.get(doc_id, {})
|
|
128
|
+
if not all(meta.get(k) == v for k, v in filters.items()):
|
|
129
|
+
continue
|
|
130
|
+
|
|
131
|
+
results.append(VectorResult(
|
|
132
|
+
id=doc_id,
|
|
133
|
+
score=score,
|
|
134
|
+
metadata=self._metadata.get(doc_id, {}),
|
|
135
|
+
))
|
|
136
|
+
|
|
137
|
+
return results
|
|
138
|
+
|
|
139
|
+
async def delete(self, ids: list[str]) -> None:
|
|
140
|
+
"""Remove vectors from the store."""
|
|
141
|
+
for doc_id in ids:
|
|
142
|
+
if doc_id in self._metadata:
|
|
143
|
+
idx = self._ids.index(doc_id)
|
|
144
|
+
self._ids.pop(idx)
|
|
145
|
+
self._vectors = np.delete(self._vectors, idx, axis=0)
|
|
146
|
+
del self._metadata[doc_id]
|
|
147
|
+
|
|
148
|
+
async def count(self) -> int:
|
|
149
|
+
return len(self._ids)
|
|
150
|
+
|
|
151
|
+
async def rebuild(
|
|
152
|
+
self,
|
|
153
|
+
ids: list[str],
|
|
154
|
+
vectors: list[list[float]],
|
|
155
|
+
metadata: list[dict[str, Any]],
|
|
156
|
+
) -> None:
|
|
157
|
+
"""Replace the complete index after validating the new corpus."""
|
|
158
|
+
replacement = InMemoryVectorStore(self._dimension)
|
|
159
|
+
await replacement.add(ids, vectors, metadata)
|
|
160
|
+
self._ids = replacement._ids
|
|
161
|
+
self._vectors = replacement._vectors
|
|
162
|
+
self._metadata = replacement._metadata
|