contextos-memory-runtime 1.0.0rc2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contextos/__init__.py +3 -0
- contextos/__main__.py +6 -0
- contextos/api/__init__.py +1 -0
- contextos/api/routes/__init__.py +1 -0
- contextos/api/routes/desktop.py +322 -0
- contextos/api/routes/ingest.py +17 -0
- contextos/api/routes/memories.py +84 -0
- contextos/api/routes/models.py +81 -0
- contextos/api/routes/retrieval.py +89 -0
- contextos/api/routes/system.py +216 -0
- contextos/api/server.py +195 -0
- contextos/benchmarks/__init__.py +1 -0
- contextos/benchmarks/compilation.py +245 -0
- contextos/benchmarks/connectors.py +423 -0
- contextos/benchmarks/explainability.py +103 -0
- contextos/benchmarks/final.py +406 -0
- contextos/benchmarks/graph.py +310 -0
- contextos/benchmarks/graph_adversarial.py +525 -0
- contextos/benchmarks/mcp.py +324 -0
- contextos/benchmarks/model_routing.py +203 -0
- contextos/benchmarks/optimization.py +305 -0
- contextos/benchmarks/rescue_integration.py +127 -0
- contextos/benchmarks/retrieval.py +266 -0
- contextos/benchmarks/temporal.py +377 -0
- contextos/benchmarks/temporal_hotpath.py +76 -0
- contextos/benchmarks/terminal.py +62 -0
- contextos/cli/__init__.py +1 -0
- contextos/cli/app.py +932 -0
- contextos/cli/dashboard.py +174 -0
- contextos/cli/formatters.py +299 -0
- contextos/config/__init__.py +1 -0
- contextos/config/settings.py +160 -0
- contextos/connectors/__init__.py +6 -0
- contextos/connectors/fake.py +11 -0
- contextos/connectors/json_import.py +125 -0
- contextos/connectors/local_files.py +102 -0
- contextos/connectors/manager.py +293 -0
- contextos/connectors/models.py +62 -0
- contextos/connectors/protocols.py +11 -0
- contextos/core/__init__.py +103 -0
- contextos/core/enums.py +489 -0
- contextos/core/exceptions.py +293 -0
- contextos/core/models.py +1147 -0
- contextos/core/protocols.py +549 -0
- contextos/daemon/__init__.py +1 -0
- contextos/daemon/manager.py +510 -0
- contextos/daemon/state.py +127 -0
- contextos/daemon/wiring.py +296 -0
- contextos/demo.py +217 -0
- contextos/embedding/__init__.py +1 -0
- contextos/embedding/deterministic.py +76 -0
- contextos/embedding/sentence_transformers.py +80 -0
- contextos/mcp/__init__.py +5 -0
- contextos/mcp/server.py +269 -0
- contextos/providers/__init__.py +13 -0
- contextos/providers/fake.py +217 -0
- contextos/providers/ollama.py +297 -0
- contextos/providers/openai_compatible.py +337 -0
- contextos/services/__init__.py +1 -0
- contextos/services/compilation.py +535 -0
- contextos/services/explainability.py +553 -0
- contextos/services/extraction.py +311 -0
- contextos/services/graph.py +524 -0
- contextos/services/graph_retrieval.py +143 -0
- contextos/services/ingestion.py +143 -0
- contextos/services/inspection.py +174 -0
- contextos/services/memory.py +291 -0
- contextos/services/model_service.py +409 -0
- contextos/services/optimization.py +426 -0
- contextos/services/privacy.py +331 -0
- contextos/services/retrieval.py +302 -0
- contextos/services/retrieval_index.py +88 -0
- contextos/services/router.py +302 -0
- contextos/services/secret_scanner.py +207 -0
- contextos/services/telemetry_query.py +102 -0
- contextos/services/temporal.py +500 -0
- contextos/services/token_counter.py +222 -0
- contextos/storage/__init__.py +1 -0
- contextos/storage/connector_repo.py +67 -0
- contextos/storage/database.py +497 -0
- contextos/storage/event_repo.py +137 -0
- contextos/storage/graph_repo.py +228 -0
- contextos/storage/lexical/__init__.py +1 -0
- contextos/storage/lexical/bm25.py +134 -0
- contextos/storage/memory_repo.py +589 -0
- contextos/storage/relation_repo.py +80 -0
- contextos/storage/telemetry_repo.py +481 -0
- contextos/storage/vector/__init__.py +1 -0
- contextos/storage/vector/in_memory.py +162 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
- contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,525 @@
|
|
|
1
|
+
"""Separate adversarial, offline review benchmark for Phase 8 graph retrieval.
|
|
2
|
+
|
|
3
|
+
This benchmark is deliberately distinct from the original graph.py benchmark.
|
|
4
|
+
It exercises 14 hard categories the original corpus does not cover:
|
|
5
|
+
|
|
6
|
+
DIRECT direct fact lookup
|
|
7
|
+
RELATIONAL explicit relational query
|
|
8
|
+
MULTIHOP two-hop traversal to reach a candidate
|
|
9
|
+
TEMPORAL historical / past-usage lookup
|
|
10
|
+
NEGATIVE explicit negation must not produce a false edge
|
|
11
|
+
SHARED_HUB shared tool (Ollama) across two projects; scoping must
|
|
12
|
+
prevent cross-project leakage
|
|
13
|
+
SAME_TOOL same tool name used by multiple projects
|
|
14
|
+
LEXICAL_OVERLAP query tokens appear in irrelevant memories; ranking must
|
|
15
|
+
prefer relevant ones
|
|
16
|
+
DELETED_SUPPORT support memory deleted; edge must disappear after rebuild
|
|
17
|
+
SUPERSEDED_SUPPORT support memory superseded; edge still appears but with
|
|
18
|
+
reduced confidence signal
|
|
19
|
+
EXPLICIT_NEGATION "does not use" sentence must create no edge
|
|
20
|
+
COMPOUND compound sentence with two independent subjects
|
|
21
|
+
DOTNAME technical identifiers with embedded dots (llama.cpp) or
|
|
22
|
+
hyphens (c++17) must survive canonicalization intact
|
|
23
|
+
AMBIGUOUS project-level vs tool-level ambiguous name resolution
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import asyncio
|
|
29
|
+
import tempfile
|
|
30
|
+
from dataclasses import dataclass
|
|
31
|
+
from pathlib import Path
|
|
32
|
+
from uuid import UUID, uuid5
|
|
33
|
+
|
|
34
|
+
from contextos.benchmarks.retrieval import hit_rate_at_k, ndcg_at_k, recall_at_k, reciprocal_rank
|
|
35
|
+
from contextos.core.enums import MemoryStatus, MemoryType, RetrievalMode, TemporalScope
|
|
36
|
+
from contextos.core.models import Memory, MemoryUpdate, RetrievalQuery
|
|
37
|
+
from contextos.embedding.deterministic import DeterministicEmbedding
|
|
38
|
+
from contextos.services.graph import DeterministicEntityExtractor, MemoryGraphService
|
|
39
|
+
from contextos.services.graph_retrieval import GraphAugmentedRetrievalEngine
|
|
40
|
+
from contextos.services.retrieval import HybridRetrievalEngine
|
|
41
|
+
from contextos.services.retrieval_index import RetrievalIndexSynchronizer
|
|
42
|
+
from contextos.storage.database import Database
|
|
43
|
+
from contextos.storage.graph_repo import SqliteGraphRepository
|
|
44
|
+
from contextos.storage.lexical.bm25 import BM25Index
|
|
45
|
+
from contextos.storage.memory_repo import SqliteMemoryRepository
|
|
46
|
+
from contextos.storage.relation_repo import SqliteRelationRepository
|
|
47
|
+
from contextos.storage.vector.in_memory import InMemoryVectorStore
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
_NAMESPACE = UUID("4a02d5f2-d71a-418c-a7f4-1bc91d6fc4ef")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass(frozen=True)
|
|
54
|
+
class AdversarialCase:
|
|
55
|
+
category: str
|
|
56
|
+
text: str
|
|
57
|
+
relevant: tuple[UUID, ...]
|
|
58
|
+
scope: TemporalScope = TemporalScope.CURRENT
|
|
59
|
+
# If True the expected answer is "no results" (empty retrieval set is correct).
|
|
60
|
+
expect_empty: bool = False
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _id(name: str) -> UUID:
|
|
64
|
+
return uuid5(_NAMESPACE, name)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _mem(name: str, content: str, status: MemoryStatus = MemoryStatus.ACTIVE,
|
|
68
|
+
mem_type: MemoryType = MemoryType.PROJECT) -> Memory:
|
|
69
|
+
return Memory(
|
|
70
|
+
id=_id(name), content=content, status=status, type=mem_type,
|
|
71
|
+
source_type="phase8_adversarial", confidence=0.9,
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def adversarial_corpus() -> tuple[list[Memory], list[AdversarialCase]]: # noqa: C901
|
|
76
|
+
memories: list[Memory] = []
|
|
77
|
+
cases: list[AdversarialCase] = []
|
|
78
|
+
|
|
79
|
+
# ------------------------------------------------------------------
|
|
80
|
+
# 1. DIRECT — plain entity name lookup
|
|
81
|
+
# ------------------------------------------------------------------
|
|
82
|
+
m_direct = _mem("direct:1", "Project Alpha uses Ollama")
|
|
83
|
+
memories.append(m_direct)
|
|
84
|
+
cases.append(AdversarialCase(
|
|
85
|
+
"DIRECT", "Ollama", (m_direct.id,),
|
|
86
|
+
))
|
|
87
|
+
|
|
88
|
+
# ------------------------------------------------------------------
|
|
89
|
+
# 2. RELATIONAL — explicit uses-relation query
|
|
90
|
+
# ------------------------------------------------------------------
|
|
91
|
+
m_rel = _mem("relational:1", "Project Beta uses vLLM")
|
|
92
|
+
memories.append(m_rel)
|
|
93
|
+
cases.append(AdversarialCase(
|
|
94
|
+
"RELATIONAL", "What does Project Beta use?", (m_rel.id,),
|
|
95
|
+
))
|
|
96
|
+
|
|
97
|
+
# ------------------------------------------------------------------
|
|
98
|
+
# 3. MULTIHOP — reach model via runtime
|
|
99
|
+
# ------------------------------------------------------------------
|
|
100
|
+
m_mh_proj = _mem("multihop:proj", "Project Gamma uses Ollama")
|
|
101
|
+
m_mh_model = _mem("multihop:model", "Ollama runs Qwen9B", mem_type=MemoryType.FACT)
|
|
102
|
+
memories.extend([m_mh_proj, m_mh_model])
|
|
103
|
+
cases.append(AdversarialCase(
|
|
104
|
+
"MULTIHOP", "Which model is reachable from Project Gamma?", (m_mh_model.id,),
|
|
105
|
+
))
|
|
106
|
+
|
|
107
|
+
# ------------------------------------------------------------------
|
|
108
|
+
# 4. TEMPORAL — historical / previously-used
|
|
109
|
+
# ------------------------------------------------------------------
|
|
110
|
+
m_old = _mem("temporal:old", "Project Delta previously used Docker", MemoryStatus.HISTORICAL)
|
|
111
|
+
m_new = _mem("temporal:new", "Project Delta uses Podman")
|
|
112
|
+
memories.extend([m_old, m_new])
|
|
113
|
+
cases.append(AdversarialCase(
|
|
114
|
+
"TEMPORAL", "What did Project Delta use before?", (m_old.id,), TemporalScope.HISTORICAL,
|
|
115
|
+
))
|
|
116
|
+
|
|
117
|
+
# ------------------------------------------------------------------
|
|
118
|
+
# 5. NEGATIVE — explicit negation must not produce a USES edge
|
|
119
|
+
# ------------------------------------------------------------------
|
|
120
|
+
m_neg_pos = _mem("negative:positive", "Project Epsilon uses Podman")
|
|
121
|
+
m_neg_neg = _mem("negative:negated", "Project Epsilon does not use Docker",
|
|
122
|
+
mem_type=MemoryType.FACT)
|
|
123
|
+
memories.extend([m_neg_pos, m_neg_neg])
|
|
124
|
+
cases.append(AdversarialCase(
|
|
125
|
+
"NEGATIVE", "Does Project Epsilon use Docker?", (m_neg_pos.id,),
|
|
126
|
+
))
|
|
127
|
+
|
|
128
|
+
# ------------------------------------------------------------------
|
|
129
|
+
# 6. SHARED_HUB — both Project Zeta and Project Eta share Ollama;
|
|
130
|
+
# a query about Zeta must NOT surface Eta's failure memory.
|
|
131
|
+
# ------------------------------------------------------------------
|
|
132
|
+
m_zeta = _mem("hub:zeta", "Project Zeta uses Ollama")
|
|
133
|
+
m_eta = _mem("hub:eta", "Project Eta uses Ollama")
|
|
134
|
+
m_eta_fail = _mem("hub:eta:fail", "Project Eta works on Failure7", mem_type=MemoryType.FACT)
|
|
135
|
+
memories.extend([m_zeta, m_eta, m_eta_fail])
|
|
136
|
+
cases.append(AdversarialCase(
|
|
137
|
+
"SHARED_HUB", "What does Project Zeta use?", (m_zeta.id,),
|
|
138
|
+
))
|
|
139
|
+
|
|
140
|
+
# ------------------------------------------------------------------
|
|
141
|
+
# 7. SAME_TOOL — vLLM used by both Project Theta and Project Iota;
|
|
142
|
+
# query for Theta must not return Iota's memory in top results.
|
|
143
|
+
# ------------------------------------------------------------------
|
|
144
|
+
m_theta = _mem("same_tool:theta", "Project Theta uses vLLM")
|
|
145
|
+
m_iota = _mem("same_tool:iota", "Project Iota uses vLLM")
|
|
146
|
+
memories.extend([m_theta, m_iota])
|
|
147
|
+
cases.append(AdversarialCase(
|
|
148
|
+
"SAME_TOOL", "What runtime does Project Theta use?", (m_theta.id,),
|
|
149
|
+
))
|
|
150
|
+
|
|
151
|
+
# ------------------------------------------------------------------
|
|
152
|
+
# 8. LEXICAL_OVERLAP — query text overlaps with an irrelevant memory;
|
|
153
|
+
# relevant memory must still rank above the noise.
|
|
154
|
+
# ------------------------------------------------------------------
|
|
155
|
+
m_lex_rel = _mem("lexical:relevant", "Project Kappa uses SQLite")
|
|
156
|
+
m_lex_noise = _mem("lexical:noise", "SQLite is a database engine used widely",
|
|
157
|
+
mem_type=MemoryType.CONTEXT)
|
|
158
|
+
memories.extend([m_lex_rel, m_lex_noise])
|
|
159
|
+
cases.append(AdversarialCase(
|
|
160
|
+
"LEXICAL_OVERLAP", "What does Project Kappa use?", (m_lex_rel.id,),
|
|
161
|
+
))
|
|
162
|
+
|
|
163
|
+
# ------------------------------------------------------------------
|
|
164
|
+
# 9. DELETED_SUPPORT — ingested, then marked deleted;
|
|
165
|
+
# graph must not surface deleted support memory after rebuild.
|
|
166
|
+
# We store the deleted memory and expect it NOT to appear.
|
|
167
|
+
# ------------------------------------------------------------------
|
|
168
|
+
m_del = _mem("deleted:mem", "Project Lambda uses Redis", MemoryStatus.DELETED)
|
|
169
|
+
memories.append(m_del)
|
|
170
|
+
# Deleted memories are excluded from projected_statuses, so there should
|
|
171
|
+
# be no graph edge for this. Relevant result is effectively none, but we
|
|
172
|
+
# use expect_empty rather than forcing a false positive.
|
|
173
|
+
cases.append(AdversarialCase(
|
|
174
|
+
"DELETED_SUPPORT", "What does Project Lambda use?",
|
|
175
|
+
(), expect_empty=True,
|
|
176
|
+
))
|
|
177
|
+
|
|
178
|
+
# ------------------------------------------------------------------
|
|
179
|
+
# 10. SUPERSEDED_SUPPORT — previous relation still in graph
|
|
180
|
+
# but status is SUPERSEDED; check that retrieval shows the
|
|
181
|
+
# superseded memory (SUPERSEDED is in projected_statuses and is_retrievable).
|
|
182
|
+
# ------------------------------------------------------------------
|
|
183
|
+
m_sup = _mem("superseded:old", "Project Mu uses RabbitMQ", MemoryStatus.SUPERSEDED,
|
|
184
|
+
MemoryType.FACT)
|
|
185
|
+
m_sup_new = _mem("superseded:new", "Project Mu uses Kafka")
|
|
186
|
+
memories.extend([m_sup, m_sup_new])
|
|
187
|
+
# Both edges exist in the graph (SUPERSEDED is projected); relevant is current memory.
|
|
188
|
+
cases.append(AdversarialCase(
|
|
189
|
+
"SUPERSEDED_SUPPORT", "What messaging system does Project Mu use?", (m_sup_new.id,),
|
|
190
|
+
))
|
|
191
|
+
|
|
192
|
+
# ------------------------------------------------------------------
|
|
193
|
+
# 11. EXPLICIT_NEGATION — "does not use" must create NO edge
|
|
194
|
+
# ------------------------------------------------------------------
|
|
195
|
+
m_neg2_pos = _mem("explicit_neg:pos", "Project Nu uses PostgreSQL")
|
|
196
|
+
m_neg2_neg = _mem("explicit_neg:neg", "Project Nu does not use MySQL",
|
|
197
|
+
mem_type=MemoryType.FACT)
|
|
198
|
+
memories.extend([m_neg2_pos, m_neg2_neg])
|
|
199
|
+
cases.append(AdversarialCase(
|
|
200
|
+
"EXPLICIT_NEGATION", "Does Project Nu use MySQL?", (m_neg2_pos.id,),
|
|
201
|
+
))
|
|
202
|
+
|
|
203
|
+
# ------------------------------------------------------------------
|
|
204
|
+
# 12. COMPOUND — compound sentence must produce two independent
|
|
205
|
+
# source→target edges, never a cross-project edge.
|
|
206
|
+
# ------------------------------------------------------------------
|
|
207
|
+
m_comp_a = _mem("compound:a", "Project Xi uses Python and Project Omicron uses Rust",
|
|
208
|
+
mem_type=MemoryType.FACT)
|
|
209
|
+
memories.append(m_comp_a)
|
|
210
|
+
cases.append(AdversarialCase(
|
|
211
|
+
"COMPOUND", "What does Project Xi use?", (m_comp_a.id,),
|
|
212
|
+
))
|
|
213
|
+
cases.append(AdversarialCase(
|
|
214
|
+
"COMPOUND", "What does Project Omicron use?", (m_comp_a.id,),
|
|
215
|
+
))
|
|
216
|
+
|
|
217
|
+
# ------------------------------------------------------------------
|
|
218
|
+
# 13. DOTNAME — embedded-dot and hyphen technical identifiers must
|
|
219
|
+
# survive canonicalization intact.
|
|
220
|
+
# ------------------------------------------------------------------
|
|
221
|
+
m_dot = _mem("dotname:llama", "Project Pi uses llama.cpp")
|
|
222
|
+
m_dot2 = _mem("dotname:cpp17", "Project Rho uses c++17")
|
|
223
|
+
memories.extend([m_dot, m_dot2])
|
|
224
|
+
cases.append(AdversarialCase(
|
|
225
|
+
"DOTNAME", "What inference engine does Project Pi use?", (m_dot.id,),
|
|
226
|
+
))
|
|
227
|
+
cases.append(AdversarialCase(
|
|
228
|
+
"DOTNAME", "What compiler standard does Project Rho use?", (m_dot2.id,),
|
|
229
|
+
))
|
|
230
|
+
|
|
231
|
+
# ------------------------------------------------------------------
|
|
232
|
+
# 14. AMBIGUOUS — "Atlas" could be a project or tool name;
|
|
233
|
+
# query must resolve to the correct memory.
|
|
234
|
+
# ------------------------------------------------------------------
|
|
235
|
+
m_amb = _mem("ambiguous:atlas", "Project Atlas uses Ollama")
|
|
236
|
+
m_amb2 = _mem("ambiguous:atlas_tool", "Atlas is also the name of a mapping tool",
|
|
237
|
+
MemoryStatus.ACTIVE, MemoryType.CONTEXT)
|
|
238
|
+
memories.extend([m_amb, m_amb2])
|
|
239
|
+
cases.append(AdversarialCase(
|
|
240
|
+
"AMBIGUOUS", "What does Project Atlas use?", (m_amb.id,),
|
|
241
|
+
))
|
|
242
|
+
|
|
243
|
+
# ------------------------------------------------------------------
|
|
244
|
+
# 15. NO-PATH — query for entity with no graph tool connection;
|
|
245
|
+
# should surface only via lexical/dense, not fabricate graph edge.
|
|
246
|
+
# ------------------------------------------------------------------
|
|
247
|
+
m_iso = _mem("nopath:isolated", "Project Sigma is an internal analytics service",
|
|
248
|
+
mem_type=MemoryType.CONTEXT)
|
|
249
|
+
memories.append(m_iso)
|
|
250
|
+
cases.append(AdversarialCase(
|
|
251
|
+
"NEGATIVE",
|
|
252
|
+
"What runtime does Project Sigma depend on?",
|
|
253
|
+
(m_iso.id,),
|
|
254
|
+
))
|
|
255
|
+
|
|
256
|
+
# ------------------------------------------------------------------
|
|
257
|
+
# 16. VERSION_NUMBER — identifier with numeric suffix must survive.
|
|
258
|
+
# ------------------------------------------------------------------
|
|
259
|
+
m_ver = _mem("version:qwen", "Project Tau uses Qwen30B for inference",
|
|
260
|
+
mem_type=MemoryType.FACT)
|
|
261
|
+
memories.append(m_ver)
|
|
262
|
+
cases.append(AdversarialCase(
|
|
263
|
+
"DOTNAME", "Which model does Project Tau use?", (m_ver.id,),
|
|
264
|
+
))
|
|
265
|
+
|
|
266
|
+
# ------------------------------------------------------------------
|
|
267
|
+
# 17. TEMPORAL REPLACEMENT — "used X but now uses Y" keeps only Y
|
|
268
|
+
# as a current USES edge.
|
|
269
|
+
# ------------------------------------------------------------------
|
|
270
|
+
m_rep = _mem("temporal_replace:mem",
|
|
271
|
+
"Project Upsilon used Docker previously but now uses Podman")
|
|
272
|
+
memories.append(m_rep)
|
|
273
|
+
cases.append(AdversarialCase(
|
|
274
|
+
"TEMPORAL", "What does Project Upsilon use now?", (m_rep.id,),
|
|
275
|
+
))
|
|
276
|
+
|
|
277
|
+
# ------------------------------------------------------------------
|
|
278
|
+
# 18. SECOND SHARED-HUB CROSS-CHECK — Project Phi also uses Ollama;
|
|
279
|
+
# querying about Phi's specific task must not surface Eta/Zeta.
|
|
280
|
+
# ------------------------------------------------------------------
|
|
281
|
+
m_phi = _mem("hub:phi", "Project Phi uses Ollama")
|
|
282
|
+
m_phi_task = _mem("hub:phi:task", "Project Phi works on VisionTask",
|
|
283
|
+
mem_type=MemoryType.FACT)
|
|
284
|
+
memories.extend([m_phi, m_phi_task])
|
|
285
|
+
cases.append(AdversarialCase(
|
|
286
|
+
"SHARED_HUB", "What does Project Phi work on?", (m_phi_task.id,),
|
|
287
|
+
))
|
|
288
|
+
|
|
289
|
+
# ------------------------------------------------------------------
|
|
290
|
+
# 19. COMPOUND WITH SEMICOLON — split on semicolon, same rules apply.
|
|
291
|
+
# ------------------------------------------------------------------
|
|
292
|
+
m_semi = _mem("compound:semi",
|
|
293
|
+
"Project Chi uses vLLM; Project Psi uses Ollama",
|
|
294
|
+
mem_type=MemoryType.FACT)
|
|
295
|
+
memories.append(m_semi)
|
|
296
|
+
cases.append(AdversarialCase(
|
|
297
|
+
"COMPOUND", "What does Project Chi use?", (m_semi.id,),
|
|
298
|
+
))
|
|
299
|
+
|
|
300
|
+
# ------------------------------------------------------------------
|
|
301
|
+
# 20. MULTIHOP DEPTH-2 NO PATH — Omega has no tool that runs Qwen9B;
|
|
302
|
+
# multihop expansion must not fabricate the path.
|
|
303
|
+
# ------------------------------------------------------------------
|
|
304
|
+
m_omega = _mem("multihop_neg:omega", "Project Omega uses SQLite")
|
|
305
|
+
memories.append(m_omega)
|
|
306
|
+
cases.append(AdversarialCase(
|
|
307
|
+
"MULTIHOP",
|
|
308
|
+
"What model does Project Omega run through its runtime?",
|
|
309
|
+
(m_omega.id,),
|
|
310
|
+
))
|
|
311
|
+
|
|
312
|
+
# ------------------------------------------------------------------
|
|
313
|
+
# 21. SUPERSEDED + CURRENT — for same project, CURRENT scope must
|
|
314
|
+
# prefer the active memory over the superseded one.
|
|
315
|
+
# ------------------------------------------------------------------
|
|
316
|
+
m_sup2_old = _mem("superseded2:old", "Project Psi uses Redis",
|
|
317
|
+
MemoryStatus.SUPERSEDED, MemoryType.PROJECT)
|
|
318
|
+
m_sup2_new = _mem("superseded2:new", "Project Psi uses Kafka")
|
|
319
|
+
memories.extend([m_sup2_old, m_sup2_new])
|
|
320
|
+
cases.append(AdversarialCase(
|
|
321
|
+
"SUPERSEDED_SUPPORT",
|
|
322
|
+
"What storage does Project Psi currently use?",
|
|
323
|
+
(m_sup2_new.id,),
|
|
324
|
+
))
|
|
325
|
+
|
|
326
|
+
# ------------------------------------------------------------------
|
|
327
|
+
# 22. LEXICAL OVERLAP — many memories mention Docker; structured
|
|
328
|
+
# relation query for Project Omega/Docker must be precise.
|
|
329
|
+
# ------------------------------------------------------------------
|
|
330
|
+
m_stop = _mem("lexical:docker_omega", "Project Omega uses Docker")
|
|
331
|
+
memories.append(m_stop)
|
|
332
|
+
cases.append(AdversarialCase(
|
|
333
|
+
"LEXICAL_OVERLAP",
|
|
334
|
+
"What does Project Omega use for containerization?",
|
|
335
|
+
(m_stop.id,),
|
|
336
|
+
))
|
|
337
|
+
|
|
338
|
+
# ------------------------------------------------------------------
|
|
339
|
+
# 23. DIRECT — bare tool name with multiple project users; retrieval
|
|
340
|
+
# must surface at least one relevant memory.
|
|
341
|
+
# ------------------------------------------------------------------
|
|
342
|
+
m_dir2 = _mem("direct:postgres", "Project Alpha uses PostgreSQL")
|
|
343
|
+
memories.append(m_dir2)
|
|
344
|
+
cases.append(AdversarialCase(
|
|
345
|
+
"DIRECT", "PostgreSQL", (m_dir2.id,),
|
|
346
|
+
))
|
|
347
|
+
|
|
348
|
+
# ------------------------------------------------------------------
|
|
349
|
+
# 24. AMBIGUOUS PROJECT NAME — "Nova" in unrelated content must not
|
|
350
|
+
# pollute the project-level structured query.
|
|
351
|
+
# ------------------------------------------------------------------
|
|
352
|
+
m_nova = _mem("ambiguous:nova", "Project Nova uses Ollama")
|
|
353
|
+
m_nova_noise = _mem("ambiguous:nova_noise",
|
|
354
|
+
"Nova is a constellation in the southern hemisphere",
|
|
355
|
+
MemoryStatus.ACTIVE, MemoryType.CONTEXT)
|
|
356
|
+
memories.extend([m_nova, m_nova_noise])
|
|
357
|
+
cases.append(AdversarialCase(
|
|
358
|
+
"AMBIGUOUS", "What does Project Nova use?", (m_nova.id,),
|
|
359
|
+
))
|
|
360
|
+
|
|
361
|
+
# ------------------------------------------------------------------
|
|
362
|
+
# 25. RELATIONAL + NEGATION COMBO — positive and negative clause in
|
|
363
|
+
# same sentence; only the positive clause must produce an edge.
|
|
364
|
+
# ------------------------------------------------------------------
|
|
365
|
+
m_rel_neg = _mem("relational_neg:mem",
|
|
366
|
+
"Project Omega uses Podman but does not use Docker",
|
|
367
|
+
mem_type=MemoryType.FACT)
|
|
368
|
+
memories.append(m_rel_neg)
|
|
369
|
+
cases.append(AdversarialCase(
|
|
370
|
+
"NEGATIVE",
|
|
371
|
+
"Does Project Omega use Podman?",
|
|
372
|
+
(m_rel_neg.id,),
|
|
373
|
+
))
|
|
374
|
+
|
|
375
|
+
assert len(cases) >= 25, f"Expected >= 25 adversarial cases, got {len(cases)}"
|
|
376
|
+
return memories, cases
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
async def run_adversarial_benchmark(root: Path | None = None) -> dict[str, object]:
|
|
380
|
+
owned = root is None
|
|
381
|
+
temporary = tempfile.TemporaryDirectory() if owned else None
|
|
382
|
+
directory = Path(temporary.name) if temporary else root
|
|
383
|
+
assert directory is not None
|
|
384
|
+
directory.mkdir(parents=True, exist_ok=True)
|
|
385
|
+
database = Database(directory / "graph-adversarial.db")
|
|
386
|
+
await database.initialize()
|
|
387
|
+
try:
|
|
388
|
+
memory_repo = SqliteMemoryRepository(database.connection())
|
|
389
|
+
relation_repo = SqliteRelationRepository(database.connection())
|
|
390
|
+
graph_repo = SqliteGraphRepository(database.connection())
|
|
391
|
+
graph = MemoryGraphService(
|
|
392
|
+
memory_repo=memory_repo, relation_repo=relation_repo, graph_repo=graph_repo,
|
|
393
|
+
)
|
|
394
|
+
memories, cases = adversarial_corpus()
|
|
395
|
+
for memory in memories:
|
|
396
|
+
await memory_repo.create(memory)
|
|
397
|
+
|
|
398
|
+
embedding = DeterministicEmbedding(32)
|
|
399
|
+
lexical = BM25Index()
|
|
400
|
+
vector = InMemoryVectorStore(32)
|
|
401
|
+
base = HybridRetrievalEngine(
|
|
402
|
+
memory_repo=memory_repo, lexical_index=lexical, vector_store=vector,
|
|
403
|
+
embedding_service=embedding,
|
|
404
|
+
index_synchronizer=RetrievalIndexSynchronizer(
|
|
405
|
+
memory_repo=memory_repo, lexical_index=lexical, vector_store=vector,
|
|
406
|
+
embedding_service=embedding,
|
|
407
|
+
),
|
|
408
|
+
)
|
|
409
|
+
engine = GraphAugmentedRetrievalEngine(
|
|
410
|
+
base_engine=base, graph_service=graph, memory_repo=memory_repo,
|
|
411
|
+
)
|
|
412
|
+
|
|
413
|
+
metrics: dict[str, list[float]] = {"recall": [], "mrr": [], "ndcg": [], "hit": []}
|
|
414
|
+
by_category: dict[str, dict[str, list[float]]] = {}
|
|
415
|
+
safety: dict[str, bool] = {}
|
|
416
|
+
|
|
417
|
+
for case in cases:
|
|
418
|
+
result = await engine.retrieve(RetrievalQuery(
|
|
419
|
+
text=case.text, mode=RetrievalMode.HYBRID_GRAPH, k=5,
|
|
420
|
+
temporal_scope=case.scope,
|
|
421
|
+
))
|
|
422
|
+
ranking = [str(item.memory.id) for item in result.memories]
|
|
423
|
+
if case.expect_empty:
|
|
424
|
+
# The deleted memory's ID must not appear in any retrieval result.
|
|
425
|
+
deleted_ids = {str(rel_id) for rel_id in case.relevant if case.relevant}
|
|
426
|
+
# relevant is empty for deleted cases; safety checked below via deleted_id set.
|
|
427
|
+
# Still record that no result has a deleted-status memory.
|
|
428
|
+
safety[f"deleted_support_excluded_{case.category.lower()}"] = not any(
|
|
429
|
+
item.memory.status.value == "deleted" for item in result.memories
|
|
430
|
+
)
|
|
431
|
+
continue
|
|
432
|
+
qrels = {str(identifier): 1 for identifier in case.relevant}
|
|
433
|
+
r = recall_at_k(ranking, qrels, 5)
|
|
434
|
+
m = reciprocal_rank(ranking, qrels)
|
|
435
|
+
n = ndcg_at_k(ranking, qrels, 5)
|
|
436
|
+
h = hit_rate_at_k(ranking, qrels, 5)
|
|
437
|
+
metrics["recall"].append(r)
|
|
438
|
+
metrics["mrr"].append(m)
|
|
439
|
+
metrics["ndcg"].append(n)
|
|
440
|
+
metrics["hit"].append(h)
|
|
441
|
+
cat = case.category
|
|
442
|
+
if cat not in by_category:
|
|
443
|
+
by_category[cat] = {"recall": [], "mrr": [], "ndcg": [], "hit": []}
|
|
444
|
+
by_category[cat]["recall"].append(r)
|
|
445
|
+
by_category[cat]["mrr"].append(m)
|
|
446
|
+
by_category[cat]["ndcg"].append(n)
|
|
447
|
+
by_category[cat]["hit"].append(h)
|
|
448
|
+
|
|
449
|
+
# --- Safety checks ---
|
|
450
|
+
edges = await graph_repo.all_edges()
|
|
451
|
+
nodes = {node.id: node for node in await graph_repo.nodes()}
|
|
452
|
+
|
|
453
|
+
# Negation: Epsilon explicitly negated "does not use Docker" → no Epsilon→Docker USES edge
|
|
454
|
+
epsilon_node_key = "epsilon"
|
|
455
|
+
epsilon_node = next(
|
|
456
|
+
(node for node in nodes.values() if node.canonical_key == epsilon_node_key), None
|
|
457
|
+
)
|
|
458
|
+
safety["negation_epsilon_no_docker_edge"] = not any(
|
|
459
|
+
edge.relation_type.value == "uses"
|
|
460
|
+
and edge.source_node_id == (epsilon_node.id if epsilon_node else None)
|
|
461
|
+
and nodes.get(edge.target_node_id) is not None
|
|
462
|
+
and nodes[edge.target_node_id].canonical_key == "docker"
|
|
463
|
+
for edge in edges
|
|
464
|
+
)
|
|
465
|
+
# Negation: no USES edge targeting MySQL from Project Nu
|
|
466
|
+
safety["explicit_negation_mysql_no_edge"] = not any(
|
|
467
|
+
edge.relation_type.value == "uses"
|
|
468
|
+
and nodes.get(edge.target_node_id) is not None
|
|
469
|
+
and nodes[edge.target_node_id].canonical_key == "mysql"
|
|
470
|
+
for edge in edges
|
|
471
|
+
)
|
|
472
|
+
# Dotname: llama.cpp survives canonicalization
|
|
473
|
+
extractor = DeterministicEntityExtractor()
|
|
474
|
+
rel_dot = extractor.relations("Project Pi uses llama.cpp")
|
|
475
|
+
safety["dotname_llama_cpp_preserved"] = (
|
|
476
|
+
len(rel_dot) == 1 and rel_dot[0].target.key == "llama.cpp"
|
|
477
|
+
)
|
|
478
|
+
# Dotname: c++17 survives canonicalization
|
|
479
|
+
rel_cpp = extractor.relations("Project Rho uses c++17")
|
|
480
|
+
safety["dotname_cpp17_preserved"] = (
|
|
481
|
+
len(rel_cpp) == 1 and rel_cpp[0].target.key == "c++17"
|
|
482
|
+
)
|
|
483
|
+
# Compound: exactly two edges from compound sentence, no cross-project edge
|
|
484
|
+
rel_comp = extractor.relations("Project Xi uses Python and Project Omicron uses Rust")
|
|
485
|
+
pairs = {(r.source.key, r.target.key) for r in rel_comp}
|
|
486
|
+
safety["compound_no_cross_edge"] = pairs == {("xi", "python"), ("omicron", "rust")}
|
|
487
|
+
|
|
488
|
+
# Shared-hub scoping: Project Zeta query must not surface Eta's failure memory
|
|
489
|
+
scope_result = await engine.retrieve(RetrievalQuery(
|
|
490
|
+
text="What does Project Zeta use?",
|
|
491
|
+
mode=RetrievalMode.GRAPH, k=10, graph_max_hops=3,
|
|
492
|
+
))
|
|
493
|
+
safety["shared_hub_no_cross_project_leak"] = all(
|
|
494
|
+
"Failure7" not in item.memory.content for item in scope_result.memories
|
|
495
|
+
)
|
|
496
|
+
|
|
497
|
+
# Deleted-support: Lambda/Redis edge must not appear in graph after rebuild
|
|
498
|
+
# (MemoryStatus.DELETED is excluded from projected_statuses)
|
|
499
|
+
safety["deleted_support_no_edge"] = not any(
|
|
500
|
+
"lambda" in (nodes.get(edge.source_node_id) and nodes[edge.source_node_id].canonical_key or "")
|
|
501
|
+
for edge in edges if edge.relation_type.value == "uses"
|
|
502
|
+
)
|
|
503
|
+
|
|
504
|
+
avg = {key: sum(values) / len(values) for key, values in metrics.items() if values}
|
|
505
|
+
cat_avg = {
|
|
506
|
+
cat: {key: sum(values) / len(values) for key, values in v.items() if values}
|
|
507
|
+
for cat, v in by_category.items()
|
|
508
|
+
}
|
|
509
|
+
return {
|
|
510
|
+
"memories": len(memories),
|
|
511
|
+
"queries": len([c for c in cases if not c.expect_empty]),
|
|
512
|
+
"categories": sorted(cat_avg.keys()),
|
|
513
|
+
"overall": avg,
|
|
514
|
+
"by_category": cat_avg,
|
|
515
|
+
"safety": safety,
|
|
516
|
+
}
|
|
517
|
+
finally:
|
|
518
|
+
await database.close()
|
|
519
|
+
if temporary:
|
|
520
|
+
temporary.cleanup()
|
|
521
|
+
|
|
522
|
+
|
|
523
|
+
if __name__ == "__main__":
|
|
524
|
+
import json
|
|
525
|
+
print(json.dumps(asyncio.run(run_adversarial_benchmark()), indent=2, sort_keys=True))
|