contextos-memory-runtime 1.0.0rc2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contextos/__init__.py +3 -0
- contextos/__main__.py +6 -0
- contextos/api/__init__.py +1 -0
- contextos/api/routes/__init__.py +1 -0
- contextos/api/routes/desktop.py +322 -0
- contextos/api/routes/ingest.py +17 -0
- contextos/api/routes/memories.py +84 -0
- contextos/api/routes/models.py +81 -0
- contextos/api/routes/retrieval.py +89 -0
- contextos/api/routes/system.py +216 -0
- contextos/api/server.py +195 -0
- contextos/benchmarks/__init__.py +1 -0
- contextos/benchmarks/compilation.py +245 -0
- contextos/benchmarks/connectors.py +423 -0
- contextos/benchmarks/explainability.py +103 -0
- contextos/benchmarks/final.py +406 -0
- contextos/benchmarks/graph.py +310 -0
- contextos/benchmarks/graph_adversarial.py +525 -0
- contextos/benchmarks/mcp.py +324 -0
- contextos/benchmarks/model_routing.py +203 -0
- contextos/benchmarks/optimization.py +305 -0
- contextos/benchmarks/rescue_integration.py +127 -0
- contextos/benchmarks/retrieval.py +266 -0
- contextos/benchmarks/temporal.py +377 -0
- contextos/benchmarks/temporal_hotpath.py +76 -0
- contextos/benchmarks/terminal.py +62 -0
- contextos/cli/__init__.py +1 -0
- contextos/cli/app.py +932 -0
- contextos/cli/dashboard.py +174 -0
- contextos/cli/formatters.py +299 -0
- contextos/config/__init__.py +1 -0
- contextos/config/settings.py +160 -0
- contextos/connectors/__init__.py +6 -0
- contextos/connectors/fake.py +11 -0
- contextos/connectors/json_import.py +125 -0
- contextos/connectors/local_files.py +102 -0
- contextos/connectors/manager.py +293 -0
- contextos/connectors/models.py +62 -0
- contextos/connectors/protocols.py +11 -0
- contextos/core/__init__.py +103 -0
- contextos/core/enums.py +489 -0
- contextos/core/exceptions.py +293 -0
- contextos/core/models.py +1147 -0
- contextos/core/protocols.py +549 -0
- contextos/daemon/__init__.py +1 -0
- contextos/daemon/manager.py +510 -0
- contextos/daemon/state.py +127 -0
- contextos/daemon/wiring.py +296 -0
- contextos/demo.py +217 -0
- contextos/embedding/__init__.py +1 -0
- contextos/embedding/deterministic.py +76 -0
- contextos/embedding/sentence_transformers.py +80 -0
- contextos/mcp/__init__.py +5 -0
- contextos/mcp/server.py +269 -0
- contextos/providers/__init__.py +13 -0
- contextos/providers/fake.py +217 -0
- contextos/providers/ollama.py +297 -0
- contextos/providers/openai_compatible.py +337 -0
- contextos/services/__init__.py +1 -0
- contextos/services/compilation.py +535 -0
- contextos/services/explainability.py +553 -0
- contextos/services/extraction.py +311 -0
- contextos/services/graph.py +524 -0
- contextos/services/graph_retrieval.py +143 -0
- contextos/services/ingestion.py +143 -0
- contextos/services/inspection.py +174 -0
- contextos/services/memory.py +291 -0
- contextos/services/model_service.py +409 -0
- contextos/services/optimization.py +426 -0
- contextos/services/privacy.py +331 -0
- contextos/services/retrieval.py +302 -0
- contextos/services/retrieval_index.py +88 -0
- contextos/services/router.py +302 -0
- contextos/services/secret_scanner.py +207 -0
- contextos/services/telemetry_query.py +102 -0
- contextos/services/temporal.py +500 -0
- contextos/services/token_counter.py +222 -0
- contextos/storage/__init__.py +1 -0
- contextos/storage/connector_repo.py +67 -0
- contextos/storage/database.py +497 -0
- contextos/storage/event_repo.py +137 -0
- contextos/storage/graph_repo.py +228 -0
- contextos/storage/lexical/__init__.py +1 -0
- contextos/storage/lexical/bm25.py +134 -0
- contextos/storage/memory_repo.py +589 -0
- contextos/storage/relation_repo.py +80 -0
- contextos/storage/telemetry_repo.py +481 -0
- contextos/storage/vector/__init__.py +1 -0
- contextos/storage/vector/in_memory.py +162 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,406 @@
|
|
|
1
|
+
"""Local synthetic comparison of retrieval and context preparation strategies."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import asyncio
|
|
7
|
+
import json
|
|
8
|
+
import math
|
|
9
|
+
import platform
|
|
10
|
+
import statistics
|
|
11
|
+
import tempfile
|
|
12
|
+
import time
|
|
13
|
+
from datetime import UTC, datetime, timedelta
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any
|
|
16
|
+
from uuid import NAMESPACE_URL, uuid5
|
|
17
|
+
|
|
18
|
+
import psutil # type: ignore[import-untyped]
|
|
19
|
+
|
|
20
|
+
from contextos.config.settings import DaemonConfig, EmbeddingConfig, Settings, TokenCounterConfig
|
|
21
|
+
from contextos.core.enums import MemoryStatus, RetrievalMode
|
|
22
|
+
from contextos.core.models import CompilationConfig, ContextBudget, Memory, RetrievalQuery
|
|
23
|
+
from contextos.daemon.wiring import wire_services
|
|
24
|
+
from contextos.services.inspection import InspectionRequest
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def ranking_metrics(ids: list[str], relevant: set[str]) -> dict[str, float]:
|
|
28
|
+
"""Binary relevance metrics from IDs specified before retrieval runs."""
|
|
29
|
+
if not relevant:
|
|
30
|
+
raise ValueError("Ground truth cannot be empty")
|
|
31
|
+
metrics: dict[str, float] = {}
|
|
32
|
+
for k in (1, 3, 5, 10):
|
|
33
|
+
hits = sum(item in relevant for item in ids[:k])
|
|
34
|
+
metrics[f"recall@{k}"] = hits / len(relevant)
|
|
35
|
+
metrics[f"precision@{k}"] = hits / k
|
|
36
|
+
metrics[f"hit@{k}"] = float(hits > 0)
|
|
37
|
+
first = next((rank for rank, item in enumerate(ids, 1) if item in relevant), None)
|
|
38
|
+
metrics["mrr"] = 1 / first if first else 0.0
|
|
39
|
+
dcg = sum(
|
|
40
|
+
(1 / math.log2(rank + 1)) for rank, item in enumerate(ids[:10], 1) if item in relevant
|
|
41
|
+
)
|
|
42
|
+
ideal = sum(1 / math.log2(rank + 1) for rank in range(1, min(len(relevant), 10) + 1))
|
|
43
|
+
metrics["ndcg@10"] = dcg / ideal if ideal else 0.0
|
|
44
|
+
return metrics
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def distribution(samples: list[float]) -> dict[str, float | int]:
|
|
48
|
+
ordered = sorted(samples)
|
|
49
|
+
return {
|
|
50
|
+
"mean_ms": round(statistics.mean(samples), 3),
|
|
51
|
+
"median_ms": round(statistics.median(samples), 3),
|
|
52
|
+
"p95_ms": round(ordered[min(len(ordered) - 1, int(0.95 * len(ordered)))], 3),
|
|
53
|
+
"samples": len(samples),
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
async def _dataset(
|
|
58
|
+
size: int, services: dict[str, Any]
|
|
59
|
+
) -> tuple[list[Memory], list[tuple[str, set[str]]], set[str]]:
|
|
60
|
+
repo = services["memory_repo"]
|
|
61
|
+
fixed = (
|
|
62
|
+
("Project Atlas uses Python for automation.", MemoryStatus.ACTIVE),
|
|
63
|
+
("Project Boreal uses Rust for its local compiler.", MemoryStatus.ACTIVE),
|
|
64
|
+
("Ollama runs on the local workstation.", MemoryStatus.ACTIVE),
|
|
65
|
+
("My current editor is VS Code.", MemoryStatus.ACTIVE),
|
|
66
|
+
("Previously my editor was Sublime Text.", MemoryStatus.SUPERSEDED),
|
|
67
|
+
("Atlas stores reproducible build settings.", MemoryStatus.ACTIVE),
|
|
68
|
+
("Boreal depends on Ollama for local inference.", MemoryStatus.ACTIVE),
|
|
69
|
+
("I use Python and Rust for different projects.", MemoryStatus.ACTIVE),
|
|
70
|
+
("A long project note: " + "integration detail " * 80, MemoryStatus.ACTIVE),
|
|
71
|
+
("Project Atlas uses Python for automation.", MemoryStatus.ACTIVE),
|
|
72
|
+
("Project Atlas previously used Java for automation.", MemoryStatus.SUPERSEDED),
|
|
73
|
+
("Project Cedar uses PostgreSQL for catalog queries.", MemoryStatus.ACTIVE),
|
|
74
|
+
("Project Cedar uses PostgreSQL for catalog queries.", MemoryStatus.ACTIVE),
|
|
75
|
+
("I prefer concise code review comments.", MemoryStatus.ACTIVE),
|
|
76
|
+
("I prefer detailed architecture review comments.", MemoryStatus.ACTIVE),
|
|
77
|
+
("I do not use Docker for Project Atlas deployment.", MemoryStatus.ACTIVE),
|
|
78
|
+
("Project Birch uses Qwen9B for local summaries.", MemoryStatus.ACTIVE),
|
|
79
|
+
("My travel preference is a window seat.", MemoryStatus.ACTIVE),
|
|
80
|
+
("I prefer an aisle seat on long flights.", MemoryStatus.ACTIVE),
|
|
81
|
+
("I attended Event Cedar in 2025.", MemoryStatus.CONTRADICTED),
|
|
82
|
+
)
|
|
83
|
+
memories: list[Memory] = []
|
|
84
|
+
epoch = datetime(2026, 1, 1, tzinfo=UTC)
|
|
85
|
+
for index, (content, status) in enumerate(fixed):
|
|
86
|
+
instant = epoch + timedelta(seconds=index)
|
|
87
|
+
memories.append(
|
|
88
|
+
await repo.create(
|
|
89
|
+
Memory(
|
|
90
|
+
id=uuid5(NAMESPACE_URL, f"contextos-final-{size}-{index}"),
|
|
91
|
+
content=content,
|
|
92
|
+
status=status,
|
|
93
|
+
source_type="benchmark",
|
|
94
|
+
created_at=instant,
|
|
95
|
+
updated_at=instant,
|
|
96
|
+
observed_at=instant,
|
|
97
|
+
)
|
|
98
|
+
)
|
|
99
|
+
)
|
|
100
|
+
for index in range(size - len(fixed)):
|
|
101
|
+
category = ("preference", "project", "tool", "technical fact", "irrelevant")[index % 5]
|
|
102
|
+
content = (
|
|
103
|
+
f"Synthetic {category} item {index}: local workspace module "
|
|
104
|
+
f"{index % 37} uses feature {index % 19}."
|
|
105
|
+
)
|
|
106
|
+
position = index + len(fixed)
|
|
107
|
+
instant = epoch + timedelta(seconds=position)
|
|
108
|
+
memories.append(
|
|
109
|
+
await repo.create(
|
|
110
|
+
Memory(
|
|
111
|
+
id=uuid5(NAMESPACE_URL, f"contextos-final-{size}-{position}"),
|
|
112
|
+
content=content,
|
|
113
|
+
status=MemoryStatus.ACTIVE,
|
|
114
|
+
source_type="benchmark",
|
|
115
|
+
created_at=instant,
|
|
116
|
+
updated_at=instant,
|
|
117
|
+
observed_at=instant,
|
|
118
|
+
)
|
|
119
|
+
)
|
|
120
|
+
)
|
|
121
|
+
queries = [
|
|
122
|
+
("Which language does Project Atlas use?", {str(memories[0].id), str(memories[9].id)}),
|
|
123
|
+
("What does Project Boreal use?", {str(memories[1].id), str(memories[6].id)}),
|
|
124
|
+
("Where does Ollama run?", {str(memories[2].id)}),
|
|
125
|
+
("What is my current editor?", {str(memories[3].id)}),
|
|
126
|
+
("Which database does Project Cedar use?", {str(memories[11].id), str(memories[12].id)}),
|
|
127
|
+
("What style of code review comments do I prefer?", {str(memories[13].id)}),
|
|
128
|
+
("What style of architecture review comments do I prefer?", {str(memories[14].id)}),
|
|
129
|
+
("Which model does Project Birch use for local summaries?", {str(memories[16].id)}),
|
|
130
|
+
("What is my travel seat preference?", {str(memories[17].id), str(memories[18].id)}),
|
|
131
|
+
("Do I use Docker for Atlas deployment?", {str(memories[15].id)}),
|
|
132
|
+
]
|
|
133
|
+
return (
|
|
134
|
+
memories,
|
|
135
|
+
queries,
|
|
136
|
+
{
|
|
137
|
+
str(memories[4].id),
|
|
138
|
+
str(memories[10].id),
|
|
139
|
+
str(memories[19].id),
|
|
140
|
+
},
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
async def _measure_size(size: int, iterations: int) -> dict[str, Any]:
|
|
145
|
+
with tempfile.TemporaryDirectory(prefix=f"contextos-final-{size}-") as folder:
|
|
146
|
+
services = await wire_services(
|
|
147
|
+
Settings(
|
|
148
|
+
daemon=DaemonConfig(data_dir=Path(folder)),
|
|
149
|
+
embedding=EmbeddingConfig(model="deterministic"),
|
|
150
|
+
# Preserve the published benchmark's explicit, exact tokenizer basis.
|
|
151
|
+
token_counter=TokenCounterConfig(encoding="cl100k_base"),
|
|
152
|
+
)
|
|
153
|
+
)
|
|
154
|
+
try:
|
|
155
|
+
memories, queries, stale_ids = await _dataset(size, services)
|
|
156
|
+
await services["retrieval_index"].ensure_current()
|
|
157
|
+
lexical_count = await services["bm25_index"].count()
|
|
158
|
+
dense_count = await services["vector_store"].count()
|
|
159
|
+
if lexical_count != size or dense_count != size:
|
|
160
|
+
raise RuntimeError(
|
|
161
|
+
f"Incomplete benchmark indexes: expected {size}, "
|
|
162
|
+
f"lexical={lexical_count}, dense={dense_count}"
|
|
163
|
+
)
|
|
164
|
+
graph_started = time.perf_counter()
|
|
165
|
+
await services["graph"].rebuild()
|
|
166
|
+
graph_rebuild_ms = (time.perf_counter() - graph_started) * 1000
|
|
167
|
+
graph_nodes, graph_edges, _ = await services["graph_repo"].counts()
|
|
168
|
+
counter = services["token_counter"]
|
|
169
|
+
modes = {
|
|
170
|
+
"vector": RetrievalMode.DENSE,
|
|
171
|
+
"hybrid": RetrievalMode.HYBRID,
|
|
172
|
+
"hybrid_graph": RetrievalMode.HYBRID_GRAPH,
|
|
173
|
+
"contextos": RetrievalMode.HYBRID,
|
|
174
|
+
}
|
|
175
|
+
metrics: dict[str, list[dict[str, float]]] = {
|
|
176
|
+
name: [] for name in ("full_history", *modes)
|
|
177
|
+
}
|
|
178
|
+
latencies: dict[str, list[float]] = {name: [] for name in modes}
|
|
179
|
+
contextos_stage_ms: dict[str, list[float]] = {
|
|
180
|
+
"retrieval": [],
|
|
181
|
+
"optimizer": [],
|
|
182
|
+
"compiler": [],
|
|
183
|
+
}
|
|
184
|
+
contextos_facts = {
|
|
185
|
+
"emitted": 0,
|
|
186
|
+
"excluded": 0,
|
|
187
|
+
"duplicates_excluded": 0,
|
|
188
|
+
"with_provenance": 0,
|
|
189
|
+
"selected_memories": 0,
|
|
190
|
+
"required_source_hits": 0,
|
|
191
|
+
"required_source_total": 0,
|
|
192
|
+
"stale_source_facts": 0,
|
|
193
|
+
}
|
|
194
|
+
token_rows: dict[str, list[dict[str, int]]] = {
|
|
195
|
+
name: [] for name in ("full_history", *modes)
|
|
196
|
+
}
|
|
197
|
+
stale_inclusion: dict[str, int] = {name: 0 for name in ("full_history", *modes)}
|
|
198
|
+
overlaps: list[float] = []
|
|
199
|
+
inspector_latencies: list[float] = []
|
|
200
|
+
for _ in range(iterations):
|
|
201
|
+
for query, relevant in queries:
|
|
202
|
+
full_ids = [str(memory.id) for memory in memories]
|
|
203
|
+
metrics["full_history"].append(ranking_metrics(full_ids, relevant))
|
|
204
|
+
full_tokens = sum(counter.count(memory.content) for memory in memories)
|
|
205
|
+
token_rows["full_history"].append(
|
|
206
|
+
{"candidate": full_tokens, "compiled": full_tokens}
|
|
207
|
+
)
|
|
208
|
+
stale_inclusion["full_history"] += bool(stale_ids & set(full_ids))
|
|
209
|
+
result_ids: dict[str, list[str]] = {}
|
|
210
|
+
for name, mode in modes.items():
|
|
211
|
+
started = time.perf_counter()
|
|
212
|
+
result = await services["retrieval"].retrieve(
|
|
213
|
+
RetrievalQuery(
|
|
214
|
+
text=query,
|
|
215
|
+
mode=mode,
|
|
216
|
+
k=10,
|
|
217
|
+
)
|
|
218
|
+
)
|
|
219
|
+
candidate_tokens = sum(
|
|
220
|
+
counter.count(row.memory.content) for row in result.memories
|
|
221
|
+
)
|
|
222
|
+
compiled_tokens = candidate_tokens
|
|
223
|
+
optimized_tokens = candidate_tokens
|
|
224
|
+
if name == "contextos":
|
|
225
|
+
contextos_stage_ms["retrieval"].append(
|
|
226
|
+
(time.perf_counter() - started) * 1000
|
|
227
|
+
)
|
|
228
|
+
optimize_started = time.perf_counter()
|
|
229
|
+
selection = services["optimizer"].optimize(
|
|
230
|
+
query, result.memories, ContextBudget(max_tokens=200)
|
|
231
|
+
)
|
|
232
|
+
contextos_stage_ms["optimizer"].append(
|
|
233
|
+
(time.perf_counter() - optimize_started) * 1000
|
|
234
|
+
)
|
|
235
|
+
optimized_tokens = sum(
|
|
236
|
+
counter.count(row.memory.content)
|
|
237
|
+
for row in selection.selected_memories
|
|
238
|
+
)
|
|
239
|
+
compile_started = time.perf_counter()
|
|
240
|
+
compiled = await services["compilation"].compile(
|
|
241
|
+
query, selection, CompilationConfig(budget=200)
|
|
242
|
+
)
|
|
243
|
+
contextos_stage_ms["compiler"].append(
|
|
244
|
+
(time.perf_counter() - compile_started) * 1000
|
|
245
|
+
)
|
|
246
|
+
compiled_tokens = counter.count(compiled.context_text)
|
|
247
|
+
contextos_facts["emitted"] += len(compiled.facts)
|
|
248
|
+
contextos_facts["excluded"] += len(compiled.excluded_facts)
|
|
249
|
+
contextos_facts["duplicates_excluded"] += sum(
|
|
250
|
+
fact.reason.value == "duplicate" for fact in compiled.excluded_facts
|
|
251
|
+
)
|
|
252
|
+
contextos_facts["with_provenance"] += sum(
|
|
253
|
+
bool(fact.provenance_event_ids) for fact in compiled.facts
|
|
254
|
+
)
|
|
255
|
+
contextos_facts["selected_memories"] += len(selection.selected_memories)
|
|
256
|
+
fact_sources = {
|
|
257
|
+
str(source_id)
|
|
258
|
+
for fact in compiled.facts
|
|
259
|
+
for source_id in fact.source_memory_ids
|
|
260
|
+
}
|
|
261
|
+
contextos_facts["required_source_hits"] += len(fact_sources & relevant)
|
|
262
|
+
contextos_facts["required_source_total"] += len(relevant)
|
|
263
|
+
contextos_facts["stale_source_facts"] += sum(
|
|
264
|
+
bool(
|
|
265
|
+
{str(source_id) for source_id in fact.source_memory_ids}
|
|
266
|
+
& stale_ids
|
|
267
|
+
)
|
|
268
|
+
for fact in compiled.facts
|
|
269
|
+
)
|
|
270
|
+
included = set(compiled.included_memory_ids)
|
|
271
|
+
ids = [
|
|
272
|
+
str(row.memory.id)
|
|
273
|
+
for row in selection.selected_memories
|
|
274
|
+
if row.memory.id in included
|
|
275
|
+
]
|
|
276
|
+
else:
|
|
277
|
+
ids = [str(row.memory.id) for row in result.memories]
|
|
278
|
+
latencies[name].append((time.perf_counter() - started) * 1000)
|
|
279
|
+
result_ids[name] = ids
|
|
280
|
+
metrics[name].append(ranking_metrics(ids, relevant))
|
|
281
|
+
token_rows[name].append(
|
|
282
|
+
{
|
|
283
|
+
"candidate": candidate_tokens,
|
|
284
|
+
"optimized": optimized_tokens,
|
|
285
|
+
"compiled": compiled_tokens,
|
|
286
|
+
}
|
|
287
|
+
)
|
|
288
|
+
stale_inclusion[name] += bool(stale_ids & set(ids))
|
|
289
|
+
overlaps.append(
|
|
290
|
+
len(set(result_ids["hybrid"]) & set(result_ids["hybrid_graph"])) / 10
|
|
291
|
+
)
|
|
292
|
+
inspect_started = time.perf_counter()
|
|
293
|
+
await services["inspector"].inspect(
|
|
294
|
+
InspectionRequest(
|
|
295
|
+
query=query,
|
|
296
|
+
limit=10,
|
|
297
|
+
budget=200,
|
|
298
|
+
graph=False,
|
|
299
|
+
)
|
|
300
|
+
)
|
|
301
|
+
inspector_latencies.append((time.perf_counter() - inspect_started) * 1000)
|
|
302
|
+
aggregate: dict[str, Any] = {}
|
|
303
|
+
for name, rows in metrics.items():
|
|
304
|
+
candidate_sum = sum(row["candidate"] for row in token_rows[name])
|
|
305
|
+
optimized_sum = sum(
|
|
306
|
+
row.get("optimized", row["candidate"]) for row in token_rows[name]
|
|
307
|
+
)
|
|
308
|
+
compiled_sum = sum(row["compiled"] for row in token_rows[name])
|
|
309
|
+
aggregate[name] = {
|
|
310
|
+
"retrieval": {
|
|
311
|
+
key: round(statistics.mean(row[key] for row in rows), 4) for key in rows[0]
|
|
312
|
+
},
|
|
313
|
+
"candidate_tokens_total": candidate_sum,
|
|
314
|
+
"optimized_tokens_total": optimized_sum,
|
|
315
|
+
"compiled_tokens_total": compiled_sum,
|
|
316
|
+
"weighted_token_reduction": (
|
|
317
|
+
round(1 - compiled_sum / candidate_sum, 4) if candidate_sum else None
|
|
318
|
+
),
|
|
319
|
+
"stale_inclusion_queries": stale_inclusion[name],
|
|
320
|
+
"latency": distribution(latencies[name]) if name in latencies else None,
|
|
321
|
+
"stage_latency": (
|
|
322
|
+
{
|
|
323
|
+
stage: distribution(samples)
|
|
324
|
+
for stage, samples in contextos_stage_ms.items()
|
|
325
|
+
}
|
|
326
|
+
if name == "contextos"
|
|
327
|
+
else None
|
|
328
|
+
),
|
|
329
|
+
"context_evidence": (
|
|
330
|
+
{
|
|
331
|
+
**contextos_facts,
|
|
332
|
+
"provenance_coverage": (
|
|
333
|
+
contextos_facts["with_provenance"] / contextos_facts["emitted"]
|
|
334
|
+
if contextos_facts["emitted"]
|
|
335
|
+
else None
|
|
336
|
+
),
|
|
337
|
+
"required_source_coverage": (
|
|
338
|
+
contextos_facts["required_source_hits"]
|
|
339
|
+
/ contextos_facts["required_source_total"]
|
|
340
|
+
if contextos_facts["required_source_total"]
|
|
341
|
+
else None
|
|
342
|
+
),
|
|
343
|
+
"budget_utilization": compiled_sum / (200 * len(rows)),
|
|
344
|
+
}
|
|
345
|
+
if name == "contextos"
|
|
346
|
+
else None
|
|
347
|
+
),
|
|
348
|
+
}
|
|
349
|
+
return {
|
|
350
|
+
"memory_count": size,
|
|
351
|
+
"queries": len(queries),
|
|
352
|
+
"iterations": iterations,
|
|
353
|
+
"strategies": aggregate,
|
|
354
|
+
"hybrid_graph_top10_overlap": round(statistics.mean(overlaps), 4),
|
|
355
|
+
"graph_rebuild_ms": round(graph_rebuild_ms, 3),
|
|
356
|
+
"inspector_latency": distribution(inspector_latencies),
|
|
357
|
+
"graph_nodes": graph_nodes,
|
|
358
|
+
"graph_edges": graph_edges,
|
|
359
|
+
"lexical_index_count": lexical_count,
|
|
360
|
+
"dense_index_count": dense_count,
|
|
361
|
+
"sqlite_bytes": await services["database"].get_size_bytes(),
|
|
362
|
+
"process_rss_bytes": psutil.Process().memory_info().rss,
|
|
363
|
+
"token_measurement_source": counter.measurement_source.value,
|
|
364
|
+
"tokenizer": counter.encoding_name,
|
|
365
|
+
"answer_quality": None,
|
|
366
|
+
"answer_quality_reason": "No answer model or independently graded answers",
|
|
367
|
+
"temporal_relation_accuracy": None,
|
|
368
|
+
"temporal_relation_reason": (
|
|
369
|
+
"This retrieval dataset does not grade relation classification"
|
|
370
|
+
),
|
|
371
|
+
"provenance_note": "Direct synthetic fixture rows have no provenance event IDs",
|
|
372
|
+
}
|
|
373
|
+
finally:
|
|
374
|
+
await services["database"].close()
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
async def measure(extended: bool = False, iterations: int = 2) -> dict[str, Any]:
|
|
378
|
+
sizes = (100, 1000, 5000) if extended else (100, 1000)
|
|
379
|
+
from contextos.benchmarks.temporal import run_evaluation
|
|
380
|
+
|
|
381
|
+
return {
|
|
382
|
+
"label": "LOCAL SYNTHETIC DETERMINISTIC BENCHMARK",
|
|
383
|
+
"environment": {"python": platform.python_version(), "platform": platform.platform()},
|
|
384
|
+
"dataset_note": (
|
|
385
|
+
"Fixed queries and relevance IDs are defined before retrieval; no model judge"
|
|
386
|
+
),
|
|
387
|
+
"ground_truth_scope": (
|
|
388
|
+
"Ten fixed questions over synthetic records; not a general quality claim"
|
|
389
|
+
),
|
|
390
|
+
"corpora": {str(size): await _measure_size(size, iterations) for size in sizes},
|
|
391
|
+
"temporal_evaluation": await run_evaluation(),
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def main() -> None:
|
|
396
|
+
parser = argparse.ArgumentParser(description=__doc__)
|
|
397
|
+
parser.add_argument("--extended", action="store_true", help="Also measure 5,000 memories")
|
|
398
|
+
parser.add_argument("--iterations", type=int, default=2)
|
|
399
|
+
args = parser.parse_args()
|
|
400
|
+
if not 1 <= args.iterations <= 10:
|
|
401
|
+
parser.error("iterations must be between 1 and 10")
|
|
402
|
+
print(json.dumps(asyncio.run(measure(args.extended, args.iterations)), indent=2))
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
if __name__ == "__main__":
|
|
406
|
+
main()
|