contextos-memory-runtime 1.0.0rc2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contextos/__init__.py +3 -0
- contextos/__main__.py +6 -0
- contextos/api/__init__.py +1 -0
- contextos/api/routes/__init__.py +1 -0
- contextos/api/routes/desktop.py +322 -0
- contextos/api/routes/ingest.py +17 -0
- contextos/api/routes/memories.py +84 -0
- contextos/api/routes/models.py +81 -0
- contextos/api/routes/retrieval.py +89 -0
- contextos/api/routes/system.py +216 -0
- contextos/api/server.py +195 -0
- contextos/benchmarks/__init__.py +1 -0
- contextos/benchmarks/compilation.py +245 -0
- contextos/benchmarks/connectors.py +423 -0
- contextos/benchmarks/explainability.py +103 -0
- contextos/benchmarks/final.py +406 -0
- contextos/benchmarks/graph.py +310 -0
- contextos/benchmarks/graph_adversarial.py +525 -0
- contextos/benchmarks/mcp.py +324 -0
- contextos/benchmarks/model_routing.py +203 -0
- contextos/benchmarks/optimization.py +305 -0
- contextos/benchmarks/rescue_integration.py +127 -0
- contextos/benchmarks/retrieval.py +266 -0
- contextos/benchmarks/temporal.py +377 -0
- contextos/benchmarks/temporal_hotpath.py +76 -0
- contextos/benchmarks/terminal.py +62 -0
- contextos/cli/__init__.py +1 -0
- contextos/cli/app.py +932 -0
- contextos/cli/dashboard.py +174 -0
- contextos/cli/formatters.py +299 -0
- contextos/config/__init__.py +1 -0
- contextos/config/settings.py +160 -0
- contextos/connectors/__init__.py +6 -0
- contextos/connectors/fake.py +11 -0
- contextos/connectors/json_import.py +125 -0
- contextos/connectors/local_files.py +102 -0
- contextos/connectors/manager.py +293 -0
- contextos/connectors/models.py +62 -0
- contextos/connectors/protocols.py +11 -0
- contextos/core/__init__.py +103 -0
- contextos/core/enums.py +489 -0
- contextos/core/exceptions.py +293 -0
- contextos/core/models.py +1147 -0
- contextos/core/protocols.py +549 -0
- contextos/daemon/__init__.py +1 -0
- contextos/daemon/manager.py +510 -0
- contextos/daemon/state.py +127 -0
- contextos/daemon/wiring.py +296 -0
- contextos/demo.py +217 -0
- contextos/embedding/__init__.py +1 -0
- contextos/embedding/deterministic.py +76 -0
- contextos/embedding/sentence_transformers.py +80 -0
- contextos/mcp/__init__.py +5 -0
- contextos/mcp/server.py +269 -0
- contextos/providers/__init__.py +13 -0
- contextos/providers/fake.py +217 -0
- contextos/providers/ollama.py +297 -0
- contextos/providers/openai_compatible.py +337 -0
- contextos/services/__init__.py +1 -0
- contextos/services/compilation.py +535 -0
- contextos/services/explainability.py +553 -0
- contextos/services/extraction.py +311 -0
- contextos/services/graph.py +524 -0
- contextos/services/graph_retrieval.py +143 -0
- contextos/services/ingestion.py +143 -0
- contextos/services/inspection.py +174 -0
- contextos/services/memory.py +291 -0
- contextos/services/model_service.py +409 -0
- contextos/services/optimization.py +426 -0
- contextos/services/privacy.py +331 -0
- contextos/services/retrieval.py +302 -0
- contextos/services/retrieval_index.py +88 -0
- contextos/services/router.py +302 -0
- contextos/services/secret_scanner.py +207 -0
- contextos/services/telemetry_query.py +102 -0
- contextos/services/temporal.py +500 -0
- contextos/services/token_counter.py +222 -0
- contextos/storage/__init__.py +1 -0
- contextos/storage/connector_repo.py +67 -0
- contextos/storage/database.py +497 -0
- contextos/storage/event_repo.py +137 -0
- contextos/storage/graph_repo.py +228 -0
- contextos/storage/lexical/__init__.py +1 -0
- contextos/storage/lexical/bm25.py +134 -0
- contextos/storage/memory_repo.py +589 -0
- contextos/storage/relation_repo.py +80 -0
- contextos/storage/telemetry_repo.py +481 -0
- contextos/storage/vector/__init__.py +1 -0
- contextos/storage/vector/in_memory.py +162 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,377 @@
|
|
|
1
|
+
"""Deterministic Phase 7 temporal evaluation, smoke, and scale harnesses."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import json
|
|
7
|
+
import tempfile
|
|
8
|
+
import time
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from datetime import datetime, timezone
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from uuid import NAMESPACE_URL, uuid5
|
|
13
|
+
|
|
14
|
+
from contextos.core.enums import (
|
|
15
|
+
CandidateTemporalStatus,
|
|
16
|
+
MemoryStatus,
|
|
17
|
+
MemoryType,
|
|
18
|
+
RetrievalMode,
|
|
19
|
+
TemporalOutcome,
|
|
20
|
+
TemporalScope,
|
|
21
|
+
)
|
|
22
|
+
from contextos.core.models import Memory, MemorySlot, RetrievalQuery
|
|
23
|
+
from contextos.embedding.deterministic import DeterministicEmbedding
|
|
24
|
+
from contextos.services.retrieval import HybridRetrievalEngine
|
|
25
|
+
from contextos.services.retrieval_index import RetrievalIndexSynchronizer
|
|
26
|
+
from contextos.services.temporal import TemporalMemoryService
|
|
27
|
+
from contextos.storage.database import Database
|
|
28
|
+
from contextos.storage.lexical.bm25 import BM25Index
|
|
29
|
+
from contextos.storage.memory_repo import SqliteMemoryRepository
|
|
30
|
+
from contextos.storage.vector.in_memory import InMemoryVectorStore
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
UTC = timezone.utc
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass(frozen=True)
|
|
37
|
+
class TemporalEvaluationCase:
|
|
38
|
+
name: str
|
|
39
|
+
previous: str
|
|
40
|
+
candidate: str
|
|
41
|
+
expected: TemporalOutcome
|
|
42
|
+
previous_observed: datetime
|
|
43
|
+
candidate_observed: datetime
|
|
44
|
+
previous_valid: datetime | None = None
|
|
45
|
+
candidate_valid: datetime | None = None
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def evaluation_cases() -> list[TemporalEvaluationCase]:
|
|
49
|
+
"""Seventeen balanced cases, totaling 34 synthetic memory records."""
|
|
50
|
+
old = datetime(2025, 1, 1, tzinfo=UTC)
|
|
51
|
+
new = datetime(2026, 1, 1, tzinfo=UTC)
|
|
52
|
+
late = datetime(2026, 9, 1, tzinfo=UTC)
|
|
53
|
+
return [
|
|
54
|
+
TemporalEvaluationCase("language_transition", "User uses Python for systems interviews.", "User now uses C++17 for systems interviews.", TemporalOutcome.SUPERSEDE, old, new),
|
|
55
|
+
TemporalEvaluationCase("ram_correction", "My machine has 16 GB RAM.", "Correction: my machine actually has 32 GB RAM.", TemporalOutcome.CORRECT, old, new),
|
|
56
|
+
TemporalEvaluationCase("tool_negation", "I use Docker.", "I no longer use Docker.", TemporalOutcome.SUPERSEDE, old, new),
|
|
57
|
+
TemporalEvaluationCase("response_change", "User prefers concise answers.", "User now prefers detailed answers.", TemporalOutcome.SUPERSEDE, old, new),
|
|
58
|
+
TemporalEvaluationCase("local_model_change", "User uses Ollama as a local model.", "User now uses Qwen9B as a local model.", TemporalOutcome.SUPERSEDE, old, new),
|
|
59
|
+
TemporalEvaluationCase("project_change", "Project Atlas is active.", "Project Atlas is now paused.", TemporalOutcome.SUPERSEDE, old, new),
|
|
60
|
+
TemporalEvaluationCase("skill_progression", "User is a Rust beginner.", "User is now proficient in Rust.", TemporalOutcome.SUPERSEDE, old, new),
|
|
61
|
+
TemporalEvaluationCase("language_scopes", "User uses Python for machine learning.", "User uses C++ for systems programming.", TemporalOutcome.COEXIST, old, new),
|
|
62
|
+
TemporalEvaluationCase("web_scope", "User uses Python for machine learning.", "User uses JavaScript for web development.", TemporalOutcome.COEXIST, old, new),
|
|
63
|
+
TemporalEvaluationCase("future_plan", "User currently uses Python.", "User might learn Rust next year.", TemporalOutcome.COEXIST, old, new),
|
|
64
|
+
TemporalEvaluationCase("uncertain_change", "User uses Python.", "I think I prefer C++ now.", TemporalOutcome.COEXIST, old, new),
|
|
65
|
+
TemporalEvaluationCase("scoped_negation", "I use Docker generally.", "I don't use Docker for Project X.", TemporalOutcome.COEXIST, old, new),
|
|
66
|
+
TemporalEvaluationCase("attendance_conflict", "I attended Event X.", "I did not attend Event X.", TemporalOutcome.CONTRADICT, old, new),
|
|
67
|
+
TemporalEvaluationCase("preference_conflict", "User prefers concise answers.", "User prefers detailed answers.", TemporalOutcome.CONTRADICT, old, new),
|
|
68
|
+
TemporalEvaluationCase("exact_duplicate", "User uses Python for ML.", "User uses Python for ML.", TemporalOutcome.DUPLICATE, old, new),
|
|
69
|
+
TemporalEvaluationCase("semantic_no_change", "User uses Python for ML.", "User currently uses Python for machine learning.", TemporalOutcome.NO_CHANGE, old, new),
|
|
70
|
+
TemporalEvaluationCase("late_historical_import", "User currently uses C++ for systems interviews.", "User used Python for systems interviews in 2025.", TemporalOutcome.ADD_NEW, new, late, new, old),
|
|
71
|
+
]
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def classification_accuracy(expected: list[TemporalOutcome], predicted: list[TemporalOutcome]) -> float:
|
|
75
|
+
return sum(left == right for left, right in zip(expected, predicted, strict=True)) / len(expected)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def outcome_precision(
|
|
79
|
+
expected: list[TemporalOutcome], predicted: list[TemporalOutcome], outcome: TemporalOutcome
|
|
80
|
+
) -> float:
|
|
81
|
+
chosen = [index for index, value in enumerate(predicted) if value == outcome]
|
|
82
|
+
if not chosen:
|
|
83
|
+
return 0.0 if outcome in expected else 1.0
|
|
84
|
+
return sum(expected[index] == outcome for index in chosen) / len(chosen)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def outcome_recall(
|
|
88
|
+
expected: list[TemporalOutcome], predicted: list[TemporalOutcome], outcome: TemporalOutcome
|
|
89
|
+
) -> float:
|
|
90
|
+
actual = [index for index, value in enumerate(expected) if value == outcome]
|
|
91
|
+
return sum(predicted[index] == outcome for index in actual) / len(actual) if actual else 1.0
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def false_supersession_rate(
|
|
95
|
+
expected: list[TemporalOutcome], predicted: list[TemporalOutcome]
|
|
96
|
+
) -> float:
|
|
97
|
+
wrong = [
|
|
98
|
+
index for index, value in enumerate(predicted)
|
|
99
|
+
if value in {TemporalOutcome.SUPERSEDE, TemporalOutcome.CORRECT}
|
|
100
|
+
and expected[index] not in {TemporalOutcome.SUPERSEDE, TemporalOutcome.CORRECT}
|
|
101
|
+
]
|
|
102
|
+
negatives = sum(
|
|
103
|
+
value not in {TemporalOutcome.SUPERSEDE, TemporalOutcome.CORRECT}
|
|
104
|
+
for value in expected
|
|
105
|
+
)
|
|
106
|
+
return len(wrong) / negatives if negatives else 0.0
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def coexistence_accuracy(expected: list[TemporalOutcome], predicted: list[TemporalOutcome]) -> float:
|
|
110
|
+
indices = [index for index, value in enumerate(expected) if value == TemporalOutcome.COEXIST]
|
|
111
|
+
return sum(predicted[index] == TemporalOutcome.COEXIST for index in indices) / len(indices)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def current_state_accuracy(
|
|
115
|
+
expected: list[TemporalOutcome], predicted: list[TemporalOutcome]
|
|
116
|
+
) -> float:
|
|
117
|
+
def family(value: TemporalOutcome) -> str:
|
|
118
|
+
if value in {TemporalOutcome.SUPERSEDE, TemporalOutcome.CORRECT}:
|
|
119
|
+
return "new_current"
|
|
120
|
+
if value in {TemporalOutcome.DUPLICATE, TemporalOutcome.NO_CHANGE}:
|
|
121
|
+
return "old_current"
|
|
122
|
+
return value.value
|
|
123
|
+
return sum(
|
|
124
|
+
family(left) == family(right)
|
|
125
|
+
for left, right in zip(expected, predicted, strict=True)
|
|
126
|
+
) / len(expected)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def historical_state_recall(
|
|
130
|
+
expected: list[TemporalOutcome], predicted: list[TemporalOutcome]
|
|
131
|
+
) -> float:
|
|
132
|
+
indices = [
|
|
133
|
+
index for index, value in enumerate(expected)
|
|
134
|
+
if value in {TemporalOutcome.SUPERSEDE, TemporalOutcome.CORRECT, TemporalOutcome.ADD_NEW}
|
|
135
|
+
]
|
|
136
|
+
return sum(
|
|
137
|
+
predicted[index] in {TemporalOutcome.SUPERSEDE, TemporalOutcome.CORRECT}
|
|
138
|
+
if expected[index] in {TemporalOutcome.SUPERSEDE, TemporalOutcome.CORRECT}
|
|
139
|
+
else predicted[index] == TemporalOutcome.ADD_NEW
|
|
140
|
+
for index in indices
|
|
141
|
+
) / len(indices)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def timeline_consistency_violations(
|
|
145
|
+
expected: list[TemporalOutcome], predicted: list[TemporalOutcome]
|
|
146
|
+
) -> int:
|
|
147
|
+
"""Count decisions that would assign the wrong current/history relationship."""
|
|
148
|
+
return sum(
|
|
149
|
+
left != right
|
|
150
|
+
for left, right in zip(expected, predicted, strict=True)
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def naive_latest_write(cases: list[TemporalEvaluationCase]) -> list[TemporalOutcome]:
|
|
155
|
+
return [
|
|
156
|
+
TemporalOutcome.DUPLICATE if case.previous == case.candidate
|
|
157
|
+
else TemporalOutcome.SUPERSEDE
|
|
158
|
+
for case in cases
|
|
159
|
+
]
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def timestamp_only(cases: list[TemporalEvaluationCase]) -> list[TemporalOutcome]:
|
|
163
|
+
return [
|
|
164
|
+
TemporalOutcome.DUPLICATE if case.previous == case.candidate
|
|
165
|
+
else TemporalOutcome.SUPERSEDE
|
|
166
|
+
if case.candidate_observed >= case.previous_observed
|
|
167
|
+
else TemporalOutcome.COEXIST
|
|
168
|
+
for case in cases
|
|
169
|
+
]
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
async def _contextos_predictions(
|
|
173
|
+
cases: list[TemporalEvaluationCase], directory: Path
|
|
174
|
+
) -> list[TemporalOutcome]:
|
|
175
|
+
predicted: list[TemporalOutcome] = []
|
|
176
|
+
for index, case in enumerate(cases):
|
|
177
|
+
database = Database(directory / f"case-{index}.db")
|
|
178
|
+
await database.initialize()
|
|
179
|
+
try:
|
|
180
|
+
service = TemporalMemoryService(SqliteMemoryRepository(database.connection()))
|
|
181
|
+
previous = Memory(
|
|
182
|
+
id=uuid5(NAMESPACE_URL, f"{case.name}:previous"),
|
|
183
|
+
content=case.previous,
|
|
184
|
+
type=MemoryType.FACT,
|
|
185
|
+
status=MemoryStatus.CANDIDATE,
|
|
186
|
+
observed_at=case.previous_observed,
|
|
187
|
+
valid_from=case.previous_valid,
|
|
188
|
+
)
|
|
189
|
+
candidate = Memory(
|
|
190
|
+
id=uuid5(NAMESPACE_URL, f"{case.name}:candidate"),
|
|
191
|
+
content=case.candidate,
|
|
192
|
+
type=MemoryType.FACT,
|
|
193
|
+
status=MemoryStatus.CANDIDATE,
|
|
194
|
+
observed_at=case.candidate_observed,
|
|
195
|
+
valid_from=case.candidate_valid,
|
|
196
|
+
)
|
|
197
|
+
await service.resolve(previous)
|
|
198
|
+
predicted.append((await service.resolve(candidate)).decision.outcome)
|
|
199
|
+
finally:
|
|
200
|
+
await database.close()
|
|
201
|
+
return predicted
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def metric_summary(
|
|
205
|
+
expected: list[TemporalOutcome], predicted: list[TemporalOutcome]
|
|
206
|
+
) -> dict[str, float]:
|
|
207
|
+
return {
|
|
208
|
+
"relation_classification_accuracy": classification_accuracy(expected, predicted),
|
|
209
|
+
"supersession_precision": outcome_precision(expected, predicted, TemporalOutcome.SUPERSEDE),
|
|
210
|
+
"supersession_recall": outcome_recall(expected, predicted, TemporalOutcome.SUPERSEDE),
|
|
211
|
+
"contradiction_precision": outcome_precision(expected, predicted, TemporalOutcome.CONTRADICT),
|
|
212
|
+
"contradiction_recall": outcome_recall(expected, predicted, TemporalOutcome.CONTRADICT),
|
|
213
|
+
"coexistence_accuracy": coexistence_accuracy(expected, predicted),
|
|
214
|
+
"false_supersession_rate": false_supersession_rate(expected, predicted),
|
|
215
|
+
"current_state_accuracy": current_state_accuracy(expected, predicted),
|
|
216
|
+
"historical_state_recall": historical_state_recall(expected, predicted),
|
|
217
|
+
"timeline_consistency_violations": float(
|
|
218
|
+
timeline_consistency_violations(expected, predicted)
|
|
219
|
+
),
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
async def run_evaluation() -> dict[str, dict[str, float]]:
|
|
224
|
+
cases = evaluation_cases()
|
|
225
|
+
expected = [case.expected for case in cases]
|
|
226
|
+
with tempfile.TemporaryDirectory(prefix="contextos-temporal-eval-") as raw:
|
|
227
|
+
contextos = await _contextos_predictions(cases, Path(raw))
|
|
228
|
+
return {
|
|
229
|
+
"NAIVE_LATEST_WRITE": metric_summary(expected, naive_latest_write(cases)),
|
|
230
|
+
"TIMESTAMP_ONLY": metric_summary(expected, timestamp_only(cases)),
|
|
231
|
+
"CONTEXTOS_TEMPORAL": metric_summary(expected, contextos),
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
async def run_smoke() -> dict[str, object]:
|
|
236
|
+
with tempfile.TemporaryDirectory(prefix="contextos-temporal-smoke-") as raw:
|
|
237
|
+
path = Path(raw) / "timeline.db"
|
|
238
|
+
database = Database(path)
|
|
239
|
+
await database.initialize()
|
|
240
|
+
try:
|
|
241
|
+
repository = SqliteMemoryRepository(database.connection())
|
|
242
|
+
service = TemporalMemoryService(repository)
|
|
243
|
+
statements = [
|
|
244
|
+
"User primarily uses Python for systems interview preparation.",
|
|
245
|
+
"User switched from Python and now primarily uses C++17 for systems interviews.",
|
|
246
|
+
"User might learn Rust later.",
|
|
247
|
+
"User uses Python for machine learning projects.",
|
|
248
|
+
]
|
|
249
|
+
resolutions = []
|
|
250
|
+
for index, content in enumerate(statements):
|
|
251
|
+
resolutions.append(await service.resolve(Memory(
|
|
252
|
+
id=uuid5(NAMESPACE_URL, f"smoke:{index}"),
|
|
253
|
+
content=content,
|
|
254
|
+
type=MemoryType.FACT,
|
|
255
|
+
status=MemoryStatus.CANDIDATE,
|
|
256
|
+
observed_at=datetime(2025 + min(index, 1), index + 1, 1, tzinfo=UTC),
|
|
257
|
+
)))
|
|
258
|
+
|
|
259
|
+
systems_slot = resolutions[1].decision.slot
|
|
260
|
+
ml_slot = resolutions[3].decision.slot
|
|
261
|
+
embedding = DeterministicEmbedding(64)
|
|
262
|
+
lexical = BM25Index()
|
|
263
|
+
vector = InMemoryVectorStore(64)
|
|
264
|
+
sync = RetrievalIndexSynchronizer(
|
|
265
|
+
memory_repo=repository, lexical_index=lexical, vector_store=vector,
|
|
266
|
+
embedding_service=embedding,
|
|
267
|
+
)
|
|
268
|
+
retrieval = HybridRetrievalEngine(
|
|
269
|
+
memory_repo=repository, lexical_index=lexical, vector_store=vector,
|
|
270
|
+
embedding_service=embedding, index_synchronizer=sync,
|
|
271
|
+
)
|
|
272
|
+
queries = {
|
|
273
|
+
"systems_now": RetrievalQuery(
|
|
274
|
+
text="language systems interviews now", mode=RetrievalMode.LEXICAL
|
|
275
|
+
),
|
|
276
|
+
"before_cpp": RetrievalQuery(
|
|
277
|
+
text="Python systems interviews before", mode=RetrievalMode.LEXICAL,
|
|
278
|
+
temporal_scope=TemporalScope.HISTORICAL,
|
|
279
|
+
),
|
|
280
|
+
"ml": RetrievalQuery(
|
|
281
|
+
text="language machine learning", mode=RetrievalMode.LEXICAL
|
|
282
|
+
),
|
|
283
|
+
"future": RetrievalQuery(
|
|
284
|
+
text="learn later Rust", mode=RetrievalMode.LEXICAL,
|
|
285
|
+
temporal_scope=TemporalScope.ALL,
|
|
286
|
+
),
|
|
287
|
+
}
|
|
288
|
+
query_results = {}
|
|
289
|
+
for name, query in queries.items():
|
|
290
|
+
result = await retrieval.retrieve(query)
|
|
291
|
+
query_results[name] = [item.memory.content for item in result.memories]
|
|
292
|
+
|
|
293
|
+
return {
|
|
294
|
+
"database_on_disk": path.exists(),
|
|
295
|
+
"current_state": {
|
|
296
|
+
"systems_interviews": [
|
|
297
|
+
memory.content for memory in await service.get_current_state(systems_slot)
|
|
298
|
+
],
|
|
299
|
+
"machine_learning": [
|
|
300
|
+
memory.content for memory in await service.get_current_state(ml_slot)
|
|
301
|
+
],
|
|
302
|
+
},
|
|
303
|
+
"history": {
|
|
304
|
+
"systems_interviews": [
|
|
305
|
+
memory.content for memory in await service.get_history(systems_slot)
|
|
306
|
+
]
|
|
307
|
+
},
|
|
308
|
+
"future": [memory.content for memory in await service.get_future()],
|
|
309
|
+
"queries": query_results,
|
|
310
|
+
"traces": [result.decision.model_dump(mode="json") for result in resolutions],
|
|
311
|
+
}
|
|
312
|
+
finally:
|
|
313
|
+
await database.close()
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
async def run_scale(count: int = 1_000) -> dict[str, object]:
|
|
317
|
+
with tempfile.TemporaryDirectory(prefix="contextos-temporal-scale-") as raw:
|
|
318
|
+
path = Path(raw) / "scale.db"
|
|
319
|
+
database = Database(path)
|
|
320
|
+
await database.initialize()
|
|
321
|
+
repository = SqliteMemoryRepository(database.connection())
|
|
322
|
+
service = TemporalMemoryService(repository)
|
|
323
|
+
started = time.perf_counter()
|
|
324
|
+
for index in range(count):
|
|
325
|
+
await service.resolve(Memory(
|
|
326
|
+
id=uuid5(NAMESPACE_URL, f"scale:{index}"),
|
|
327
|
+
content=f"Synthetic value {index} for benchmark slot {index}.",
|
|
328
|
+
type=MemoryType.FACT,
|
|
329
|
+
status=MemoryStatus.CANDIDATE,
|
|
330
|
+
observed_at=datetime(2026, 1, 1, tzinfo=UTC),
|
|
331
|
+
slot=MemorySlot(
|
|
332
|
+
subject="synthetic", property="benchmark_value", scope=f"slot_{index}"
|
|
333
|
+
),
|
|
334
|
+
))
|
|
335
|
+
resolution_ms = (time.perf_counter() - started) * 1000
|
|
336
|
+
lookup_started = time.perf_counter()
|
|
337
|
+
for index in range(100):
|
|
338
|
+
await service.get_current_state(
|
|
339
|
+
MemorySlot(
|
|
340
|
+
subject="synthetic", property="benchmark_value", scope=f"slot_{index}"
|
|
341
|
+
)
|
|
342
|
+
)
|
|
343
|
+
lookup_ms = (time.perf_counter() - lookup_started) * 1000
|
|
344
|
+
await database.close()
|
|
345
|
+
|
|
346
|
+
restart_started = time.perf_counter()
|
|
347
|
+
reopened = Database(path)
|
|
348
|
+
await reopened.initialize()
|
|
349
|
+
try:
|
|
350
|
+
reopened_repo = SqliteMemoryRepository(reopened.connection())
|
|
351
|
+
persisted = await reopened_repo.count()
|
|
352
|
+
restart_ms = (time.perf_counter() - restart_started) * 1000
|
|
353
|
+
finally:
|
|
354
|
+
await reopened.close()
|
|
355
|
+
return {
|
|
356
|
+
"memories": count,
|
|
357
|
+
"resolution_total_ms": round(resolution_ms, 3),
|
|
358
|
+
"resolution_average_ms": round(resolution_ms / count, 6),
|
|
359
|
+
"timeline_lookups": 100,
|
|
360
|
+
"timeline_lookup_total_ms": round(lookup_ms, 3),
|
|
361
|
+
"timeline_lookup_average_ms": round(lookup_ms / 100, 6),
|
|
362
|
+
"restart_and_count_ms": round(restart_ms, 3),
|
|
363
|
+
"persisted_after_restart": persisted,
|
|
364
|
+
"claim": "synthetic timing only; not a production scalability claim",
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
async def _main() -> None:
|
|
369
|
+
print(json.dumps({
|
|
370
|
+
"evaluation": await run_evaluation(),
|
|
371
|
+
"smoke": await run_smoke(),
|
|
372
|
+
"scale": await run_scale(),
|
|
373
|
+
}, indent=2, sort_keys=True))
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
if __name__ == "__main__":
|
|
377
|
+
asyncio.run(_main())
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""Isolated temporal peer lookup benchmark for local profiling."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import json
|
|
7
|
+
import statistics
|
|
8
|
+
import tempfile
|
|
9
|
+
import time
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from contextos.core.enums import MemoryStatus
|
|
14
|
+
from contextos.core.models import Memory, MemorySlot
|
|
15
|
+
from contextos.services.temporal import TemporalMemoryService
|
|
16
|
+
from contextos.storage.database import Database
|
|
17
|
+
from contextos.storage.memory_repo import SqliteMemoryRepository
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
async def measure(seed_count: int = 600, iterations: int = 25) -> dict[str, Any]:
|
|
21
|
+
with tempfile.TemporaryDirectory(prefix="contextos-temporal-profile-") as folder:
|
|
22
|
+
database = Database(Path(folder) / "temporal.db")
|
|
23
|
+
await database.initialize()
|
|
24
|
+
try:
|
|
25
|
+
repo = SqliteMemoryRepository(database.connection())
|
|
26
|
+
service = TemporalMemoryService(repo)
|
|
27
|
+
for index in range(seed_count):
|
|
28
|
+
await repo.create(
|
|
29
|
+
Memory(
|
|
30
|
+
content=f"Synthetic tooling fact {index}.",
|
|
31
|
+
status=MemoryStatus.ACTIVE,
|
|
32
|
+
slot=MemorySlot(
|
|
33
|
+
subject="user", property="tool_usage", scope=f"scope_{index}"
|
|
34
|
+
),
|
|
35
|
+
)
|
|
36
|
+
)
|
|
37
|
+
cursor = await database.connection().execute(
|
|
38
|
+
"EXPLAIN QUERY PLAN SELECT * FROM memories WHERE status = ? "
|
|
39
|
+
"AND slot_key IS NOT NULL AND slot_key != ? "
|
|
40
|
+
"AND json_extract(slot_json, '$.subject') = ? "
|
|
41
|
+
"AND json_extract(slot_json, '$.property') = ? "
|
|
42
|
+
"ORDER BY COALESCE(valid_from, observed_at, created_at) DESC, "
|
|
43
|
+
"observed_at DESC, id DESC LIMIT 1",
|
|
44
|
+
("active", "unused", "user", "tool_usage"),
|
|
45
|
+
)
|
|
46
|
+
plan = [row[3] for row in await cursor.fetchall()]
|
|
47
|
+
samples = []
|
|
48
|
+
outcomes = []
|
|
49
|
+
for index in range(iterations):
|
|
50
|
+
candidate = Memory(
|
|
51
|
+
content=f"Synthetic new tooling fact {index}.",
|
|
52
|
+
slot=MemorySlot(
|
|
53
|
+
subject="user", property="tool_usage", scope=f"new_scope_{index}"
|
|
54
|
+
),
|
|
55
|
+
)
|
|
56
|
+
started = time.perf_counter()
|
|
57
|
+
decision = await service.decide(candidate)
|
|
58
|
+
samples.append((time.perf_counter() - started) * 1000)
|
|
59
|
+
outcomes.append(decision.outcome.value)
|
|
60
|
+
return {
|
|
61
|
+
"seed_count": seed_count,
|
|
62
|
+
"iterations": iterations,
|
|
63
|
+
"mean_ms": round(statistics.mean(samples), 3),
|
|
64
|
+
"median_ms": round(statistics.median(samples), 3),
|
|
65
|
+
"p95_ms": round(
|
|
66
|
+
sorted(samples)[min(len(samples) - 1, int(0.95 * len(samples)))], 3
|
|
67
|
+
),
|
|
68
|
+
"outcomes": sorted(set(outcomes)),
|
|
69
|
+
"query_plan": plan,
|
|
70
|
+
}
|
|
71
|
+
finally:
|
|
72
|
+
await database.close()
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
if __name__ == "__main__":
|
|
76
|
+
print(json.dumps(asyncio.run(measure()), indent=2))
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""Reproducible local Phase 12 latency measurements; no token-saving claims."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import json
|
|
7
|
+
import statistics
|
|
8
|
+
import subprocess
|
|
9
|
+
import sys
|
|
10
|
+
import tempfile
|
|
11
|
+
import time
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from httpx import ASGITransport, AsyncClient
|
|
15
|
+
|
|
16
|
+
from contextos.api.server import create_app, set_services
|
|
17
|
+
from contextos.config.settings import Settings
|
|
18
|
+
from contextos.daemon.wiring import wire_services
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
async def measure() -> dict[str, dict[str, float]]:
|
|
22
|
+
with tempfile.TemporaryDirectory(prefix="contextos-terminal-bench-") as folder:
|
|
23
|
+
settings = Settings(daemon={"data_dir": Path(folder)}, embedding={"model": "deterministic"})
|
|
24
|
+
services = await wire_services(settings)
|
|
25
|
+
set_services(services)
|
|
26
|
+
metrics: dict[str, list[float]] = {name: [] for name in (
|
|
27
|
+
"cli_startup_ms", "dashboard_refresh_ms", "telemetry_query_ms",
|
|
28
|
+
"memory_search_ms", "monitoring_request_overhead_ms",
|
|
29
|
+
)}
|
|
30
|
+
try:
|
|
31
|
+
async with AsyncClient(transport=ASGITransport(app=create_app()), base_url="http://localhost") as http:
|
|
32
|
+
await http.post("/api/v1/remember", json={"text": "I prefer concise technical documentation."})
|
|
33
|
+
for _ in range(5):
|
|
34
|
+
started = time.perf_counter()
|
|
35
|
+
result = subprocess.run([sys.executable, "-m", "contextos", "version"],
|
|
36
|
+
capture_output=True, timeout=15, check=True)
|
|
37
|
+
assert result.returncode == 0
|
|
38
|
+
metrics["cli_startup_ms"].append((time.perf_counter() - started) * 1000)
|
|
39
|
+
started = time.perf_counter()
|
|
40
|
+
await http.get("/api/v1/dashboard")
|
|
41
|
+
metrics["dashboard_refresh_ms"].append((time.perf_counter() - started) * 1000)
|
|
42
|
+
started = time.perf_counter()
|
|
43
|
+
await http.get("/api/v1/telemetry/summary")
|
|
44
|
+
metrics["telemetry_query_ms"].append((time.perf_counter() - started) * 1000)
|
|
45
|
+
started = time.perf_counter()
|
|
46
|
+
await http.post("/api/v1/retrieve", json={"query": "technical documentation"})
|
|
47
|
+
metrics["memory_search_ms"].append((time.perf_counter() - started) * 1000)
|
|
48
|
+
# Monitor overhead is its polling request relative to a bare telemetry query.
|
|
49
|
+
metrics["monitoring_request_overhead_ms"] = [
|
|
50
|
+
max(0.0, dashboard - telemetry) for dashboard, telemetry in zip(
|
|
51
|
+
metrics["dashboard_refresh_ms"], metrics["telemetry_query_ms"]
|
|
52
|
+
)
|
|
53
|
+
]
|
|
54
|
+
finally:
|
|
55
|
+
await services["database"].close()
|
|
56
|
+
return {key: {"median_ms": round(statistics.median(values), 3),
|
|
57
|
+
"max_ms": round(max(values), 3), "samples": len(values)}
|
|
58
|
+
for key, values in metrics.items()}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
if __name__ == "__main__":
|
|
62
|
+
print(json.dumps(asyncio.run(measure()), indent=2))
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""CLI package for ContextOS."""
|