contextos-memory-runtime 1.0.0rc2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. contextos/__init__.py +3 -0
  2. contextos/__main__.py +6 -0
  3. contextos/api/__init__.py +1 -0
  4. contextos/api/routes/__init__.py +1 -0
  5. contextos/api/routes/desktop.py +322 -0
  6. contextos/api/routes/ingest.py +17 -0
  7. contextos/api/routes/memories.py +84 -0
  8. contextos/api/routes/models.py +81 -0
  9. contextos/api/routes/retrieval.py +89 -0
  10. contextos/api/routes/system.py +216 -0
  11. contextos/api/server.py +195 -0
  12. contextos/benchmarks/__init__.py +1 -0
  13. contextos/benchmarks/compilation.py +245 -0
  14. contextos/benchmarks/connectors.py +423 -0
  15. contextos/benchmarks/explainability.py +103 -0
  16. contextos/benchmarks/final.py +406 -0
  17. contextos/benchmarks/graph.py +310 -0
  18. contextos/benchmarks/graph_adversarial.py +525 -0
  19. contextos/benchmarks/mcp.py +324 -0
  20. contextos/benchmarks/model_routing.py +203 -0
  21. contextos/benchmarks/optimization.py +305 -0
  22. contextos/benchmarks/rescue_integration.py +127 -0
  23. contextos/benchmarks/retrieval.py +266 -0
  24. contextos/benchmarks/temporal.py +377 -0
  25. contextos/benchmarks/temporal_hotpath.py +76 -0
  26. contextos/benchmarks/terminal.py +62 -0
  27. contextos/cli/__init__.py +1 -0
  28. contextos/cli/app.py +932 -0
  29. contextos/cli/dashboard.py +174 -0
  30. contextos/cli/formatters.py +299 -0
  31. contextos/config/__init__.py +1 -0
  32. contextos/config/settings.py +160 -0
  33. contextos/connectors/__init__.py +6 -0
  34. contextos/connectors/fake.py +11 -0
  35. contextos/connectors/json_import.py +125 -0
  36. contextos/connectors/local_files.py +102 -0
  37. contextos/connectors/manager.py +293 -0
  38. contextos/connectors/models.py +62 -0
  39. contextos/connectors/protocols.py +11 -0
  40. contextos/core/__init__.py +103 -0
  41. contextos/core/enums.py +489 -0
  42. contextos/core/exceptions.py +293 -0
  43. contextos/core/models.py +1147 -0
  44. contextos/core/protocols.py +549 -0
  45. contextos/daemon/__init__.py +1 -0
  46. contextos/daemon/manager.py +510 -0
  47. contextos/daemon/state.py +127 -0
  48. contextos/daemon/wiring.py +296 -0
  49. contextos/demo.py +217 -0
  50. contextos/embedding/__init__.py +1 -0
  51. contextos/embedding/deterministic.py +76 -0
  52. contextos/embedding/sentence_transformers.py +80 -0
  53. contextos/mcp/__init__.py +5 -0
  54. contextos/mcp/server.py +269 -0
  55. contextos/providers/__init__.py +13 -0
  56. contextos/providers/fake.py +217 -0
  57. contextos/providers/ollama.py +297 -0
  58. contextos/providers/openai_compatible.py +337 -0
  59. contextos/services/__init__.py +1 -0
  60. contextos/services/compilation.py +535 -0
  61. contextos/services/explainability.py +553 -0
  62. contextos/services/extraction.py +311 -0
  63. contextos/services/graph.py +524 -0
  64. contextos/services/graph_retrieval.py +143 -0
  65. contextos/services/ingestion.py +143 -0
  66. contextos/services/inspection.py +174 -0
  67. contextos/services/memory.py +291 -0
  68. contextos/services/model_service.py +409 -0
  69. contextos/services/optimization.py +426 -0
  70. contextos/services/privacy.py +331 -0
  71. contextos/services/retrieval.py +302 -0
  72. contextos/services/retrieval_index.py +88 -0
  73. contextos/services/router.py +302 -0
  74. contextos/services/secret_scanner.py +207 -0
  75. contextos/services/telemetry_query.py +102 -0
  76. contextos/services/temporal.py +500 -0
  77. contextos/services/token_counter.py +222 -0
  78. contextos/storage/__init__.py +1 -0
  79. contextos/storage/connector_repo.py +67 -0
  80. contextos/storage/database.py +497 -0
  81. contextos/storage/event_repo.py +137 -0
  82. contextos/storage/graph_repo.py +228 -0
  83. contextos/storage/lexical/__init__.py +1 -0
  84. contextos/storage/lexical/bm25.py +134 -0
  85. contextos/storage/memory_repo.py +589 -0
  86. contextos/storage/relation_repo.py +80 -0
  87. contextos/storage/telemetry_repo.py +481 -0
  88. contextos/storage/vector/__init__.py +1 -0
  89. contextos/storage/vector/in_memory.py +162 -0
  90. contextos_memory_runtime-1.0.0rc2.dist-info/METADATA +143 -0
  91. contextos_memory_runtime-1.0.0rc2.dist-info/RECORD +93 -0
  92. contextos_memory_runtime-1.0.0rc2.dist-info/WHEEL +4 -0
  93. contextos_memory_runtime-1.0.0rc2.dist-info/entry_points.txt +3 -0
@@ -0,0 +1,525 @@
1
+ """Separate adversarial, offline review benchmark for Phase 8 graph retrieval.
2
+
3
+ This benchmark is deliberately distinct from the original graph.py benchmark.
4
+ It exercises 14 hard categories the original corpus does not cover:
5
+
6
+ DIRECT direct fact lookup
7
+ RELATIONAL explicit relational query
8
+ MULTIHOP two-hop traversal to reach a candidate
9
+ TEMPORAL historical / past-usage lookup
10
+ NEGATIVE explicit negation must not produce a false edge
11
+ SHARED_HUB shared tool (Ollama) across two projects; scoping must
12
+ prevent cross-project leakage
13
+ SAME_TOOL same tool name used by multiple projects
14
+ LEXICAL_OVERLAP query tokens appear in irrelevant memories; ranking must
15
+ prefer relevant ones
16
+ DELETED_SUPPORT support memory deleted; edge must disappear after rebuild
17
+ SUPERSEDED_SUPPORT support memory superseded; edge still appears but with
18
+ reduced confidence signal
19
+ EXPLICIT_NEGATION "does not use" sentence must create no edge
20
+ COMPOUND compound sentence with two independent subjects
21
+ DOTNAME technical identifiers with embedded dots (llama.cpp) or
22
+ hyphens (c++17) must survive canonicalization intact
23
+ AMBIGUOUS project-level vs tool-level ambiguous name resolution
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+ import asyncio
29
+ import tempfile
30
+ from dataclasses import dataclass
31
+ from pathlib import Path
32
+ from uuid import UUID, uuid5
33
+
34
+ from contextos.benchmarks.retrieval import hit_rate_at_k, ndcg_at_k, recall_at_k, reciprocal_rank
35
+ from contextos.core.enums import MemoryStatus, MemoryType, RetrievalMode, TemporalScope
36
+ from contextos.core.models import Memory, MemoryUpdate, RetrievalQuery
37
+ from contextos.embedding.deterministic import DeterministicEmbedding
38
+ from contextos.services.graph import DeterministicEntityExtractor, MemoryGraphService
39
+ from contextos.services.graph_retrieval import GraphAugmentedRetrievalEngine
40
+ from contextos.services.retrieval import HybridRetrievalEngine
41
+ from contextos.services.retrieval_index import RetrievalIndexSynchronizer
42
+ from contextos.storage.database import Database
43
+ from contextos.storage.graph_repo import SqliteGraphRepository
44
+ from contextos.storage.lexical.bm25 import BM25Index
45
+ from contextos.storage.memory_repo import SqliteMemoryRepository
46
+ from contextos.storage.relation_repo import SqliteRelationRepository
47
+ from contextos.storage.vector.in_memory import InMemoryVectorStore
48
+
49
+
50
+ _NAMESPACE = UUID("4a02d5f2-d71a-418c-a7f4-1bc91d6fc4ef")
51
+
52
+
53
+ @dataclass(frozen=True)
54
+ class AdversarialCase:
55
+ category: str
56
+ text: str
57
+ relevant: tuple[UUID, ...]
58
+ scope: TemporalScope = TemporalScope.CURRENT
59
+ # If True the expected answer is "no results" (empty retrieval set is correct).
60
+ expect_empty: bool = False
61
+
62
+
63
+ def _id(name: str) -> UUID:
64
+ return uuid5(_NAMESPACE, name)
65
+
66
+
67
+ def _mem(name: str, content: str, status: MemoryStatus = MemoryStatus.ACTIVE,
68
+ mem_type: MemoryType = MemoryType.PROJECT) -> Memory:
69
+ return Memory(
70
+ id=_id(name), content=content, status=status, type=mem_type,
71
+ source_type="phase8_adversarial", confidence=0.9,
72
+ )
73
+
74
+
75
+ def adversarial_corpus() -> tuple[list[Memory], list[AdversarialCase]]: # noqa: C901
76
+ memories: list[Memory] = []
77
+ cases: list[AdversarialCase] = []
78
+
79
+ # ------------------------------------------------------------------
80
+ # 1. DIRECT — plain entity name lookup
81
+ # ------------------------------------------------------------------
82
+ m_direct = _mem("direct:1", "Project Alpha uses Ollama")
83
+ memories.append(m_direct)
84
+ cases.append(AdversarialCase(
85
+ "DIRECT", "Ollama", (m_direct.id,),
86
+ ))
87
+
88
+ # ------------------------------------------------------------------
89
+ # 2. RELATIONAL — explicit uses-relation query
90
+ # ------------------------------------------------------------------
91
+ m_rel = _mem("relational:1", "Project Beta uses vLLM")
92
+ memories.append(m_rel)
93
+ cases.append(AdversarialCase(
94
+ "RELATIONAL", "What does Project Beta use?", (m_rel.id,),
95
+ ))
96
+
97
+ # ------------------------------------------------------------------
98
+ # 3. MULTIHOP — reach model via runtime
99
+ # ------------------------------------------------------------------
100
+ m_mh_proj = _mem("multihop:proj", "Project Gamma uses Ollama")
101
+ m_mh_model = _mem("multihop:model", "Ollama runs Qwen9B", mem_type=MemoryType.FACT)
102
+ memories.extend([m_mh_proj, m_mh_model])
103
+ cases.append(AdversarialCase(
104
+ "MULTIHOP", "Which model is reachable from Project Gamma?", (m_mh_model.id,),
105
+ ))
106
+
107
+ # ------------------------------------------------------------------
108
+ # 4. TEMPORAL — historical / previously-used
109
+ # ------------------------------------------------------------------
110
+ m_old = _mem("temporal:old", "Project Delta previously used Docker", MemoryStatus.HISTORICAL)
111
+ m_new = _mem("temporal:new", "Project Delta uses Podman")
112
+ memories.extend([m_old, m_new])
113
+ cases.append(AdversarialCase(
114
+ "TEMPORAL", "What did Project Delta use before?", (m_old.id,), TemporalScope.HISTORICAL,
115
+ ))
116
+
117
+ # ------------------------------------------------------------------
118
+ # 5. NEGATIVE — explicit negation must not produce a USES edge
119
+ # ------------------------------------------------------------------
120
+ m_neg_pos = _mem("negative:positive", "Project Epsilon uses Podman")
121
+ m_neg_neg = _mem("negative:negated", "Project Epsilon does not use Docker",
122
+ mem_type=MemoryType.FACT)
123
+ memories.extend([m_neg_pos, m_neg_neg])
124
+ cases.append(AdversarialCase(
125
+ "NEGATIVE", "Does Project Epsilon use Docker?", (m_neg_pos.id,),
126
+ ))
127
+
128
+ # ------------------------------------------------------------------
129
+ # 6. SHARED_HUB — both Project Zeta and Project Eta share Ollama;
130
+ # a query about Zeta must NOT surface Eta's failure memory.
131
+ # ------------------------------------------------------------------
132
+ m_zeta = _mem("hub:zeta", "Project Zeta uses Ollama")
133
+ m_eta = _mem("hub:eta", "Project Eta uses Ollama")
134
+ m_eta_fail = _mem("hub:eta:fail", "Project Eta works on Failure7", mem_type=MemoryType.FACT)
135
+ memories.extend([m_zeta, m_eta, m_eta_fail])
136
+ cases.append(AdversarialCase(
137
+ "SHARED_HUB", "What does Project Zeta use?", (m_zeta.id,),
138
+ ))
139
+
140
+ # ------------------------------------------------------------------
141
+ # 7. SAME_TOOL — vLLM used by both Project Theta and Project Iota;
142
+ # query for Theta must not return Iota's memory in top results.
143
+ # ------------------------------------------------------------------
144
+ m_theta = _mem("same_tool:theta", "Project Theta uses vLLM")
145
+ m_iota = _mem("same_tool:iota", "Project Iota uses vLLM")
146
+ memories.extend([m_theta, m_iota])
147
+ cases.append(AdversarialCase(
148
+ "SAME_TOOL", "What runtime does Project Theta use?", (m_theta.id,),
149
+ ))
150
+
151
+ # ------------------------------------------------------------------
152
+ # 8. LEXICAL_OVERLAP — query text overlaps with an irrelevant memory;
153
+ # relevant memory must still rank above the noise.
154
+ # ------------------------------------------------------------------
155
+ m_lex_rel = _mem("lexical:relevant", "Project Kappa uses SQLite")
156
+ m_lex_noise = _mem("lexical:noise", "SQLite is a database engine used widely",
157
+ mem_type=MemoryType.CONTEXT)
158
+ memories.extend([m_lex_rel, m_lex_noise])
159
+ cases.append(AdversarialCase(
160
+ "LEXICAL_OVERLAP", "What does Project Kappa use?", (m_lex_rel.id,),
161
+ ))
162
+
163
+ # ------------------------------------------------------------------
164
+ # 9. DELETED_SUPPORT — ingested, then marked deleted;
165
+ # graph must not surface deleted support memory after rebuild.
166
+ # We store the deleted memory and expect it NOT to appear.
167
+ # ------------------------------------------------------------------
168
+ m_del = _mem("deleted:mem", "Project Lambda uses Redis", MemoryStatus.DELETED)
169
+ memories.append(m_del)
170
+ # Deleted memories are excluded from projected_statuses, so there should
171
+ # be no graph edge for this. Relevant result is effectively none, but we
172
+ # use expect_empty rather than forcing a false positive.
173
+ cases.append(AdversarialCase(
174
+ "DELETED_SUPPORT", "What does Project Lambda use?",
175
+ (), expect_empty=True,
176
+ ))
177
+
178
+ # ------------------------------------------------------------------
179
+ # 10. SUPERSEDED_SUPPORT — previous relation still in graph
180
+ # but status is SUPERSEDED; check that retrieval shows the
181
+ # superseded memory (SUPERSEDED is in projected_statuses and is_retrievable).
182
+ # ------------------------------------------------------------------
183
+ m_sup = _mem("superseded:old", "Project Mu uses RabbitMQ", MemoryStatus.SUPERSEDED,
184
+ MemoryType.FACT)
185
+ m_sup_new = _mem("superseded:new", "Project Mu uses Kafka")
186
+ memories.extend([m_sup, m_sup_new])
187
+ # Both edges exist in the graph (SUPERSEDED is projected); relevant is current memory.
188
+ cases.append(AdversarialCase(
189
+ "SUPERSEDED_SUPPORT", "What messaging system does Project Mu use?", (m_sup_new.id,),
190
+ ))
191
+
192
+ # ------------------------------------------------------------------
193
+ # 11. EXPLICIT_NEGATION — "does not use" must create NO edge
194
+ # ------------------------------------------------------------------
195
+ m_neg2_pos = _mem("explicit_neg:pos", "Project Nu uses PostgreSQL")
196
+ m_neg2_neg = _mem("explicit_neg:neg", "Project Nu does not use MySQL",
197
+ mem_type=MemoryType.FACT)
198
+ memories.extend([m_neg2_pos, m_neg2_neg])
199
+ cases.append(AdversarialCase(
200
+ "EXPLICIT_NEGATION", "Does Project Nu use MySQL?", (m_neg2_pos.id,),
201
+ ))
202
+
203
+ # ------------------------------------------------------------------
204
+ # 12. COMPOUND — compound sentence must produce two independent
205
+ # source→target edges, never a cross-project edge.
206
+ # ------------------------------------------------------------------
207
+ m_comp_a = _mem("compound:a", "Project Xi uses Python and Project Omicron uses Rust",
208
+ mem_type=MemoryType.FACT)
209
+ memories.append(m_comp_a)
210
+ cases.append(AdversarialCase(
211
+ "COMPOUND", "What does Project Xi use?", (m_comp_a.id,),
212
+ ))
213
+ cases.append(AdversarialCase(
214
+ "COMPOUND", "What does Project Omicron use?", (m_comp_a.id,),
215
+ ))
216
+
217
+ # ------------------------------------------------------------------
218
+ # 13. DOTNAME — embedded-dot and hyphen technical identifiers must
219
+ # survive canonicalization intact.
220
+ # ------------------------------------------------------------------
221
+ m_dot = _mem("dotname:llama", "Project Pi uses llama.cpp")
222
+ m_dot2 = _mem("dotname:cpp17", "Project Rho uses c++17")
223
+ memories.extend([m_dot, m_dot2])
224
+ cases.append(AdversarialCase(
225
+ "DOTNAME", "What inference engine does Project Pi use?", (m_dot.id,),
226
+ ))
227
+ cases.append(AdversarialCase(
228
+ "DOTNAME", "What compiler standard does Project Rho use?", (m_dot2.id,),
229
+ ))
230
+
231
+ # ------------------------------------------------------------------
232
+ # 14. AMBIGUOUS — "Atlas" could be a project or tool name;
233
+ # query must resolve to the correct memory.
234
+ # ------------------------------------------------------------------
235
+ m_amb = _mem("ambiguous:atlas", "Project Atlas uses Ollama")
236
+ m_amb2 = _mem("ambiguous:atlas_tool", "Atlas is also the name of a mapping tool",
237
+ MemoryStatus.ACTIVE, MemoryType.CONTEXT)
238
+ memories.extend([m_amb, m_amb2])
239
+ cases.append(AdversarialCase(
240
+ "AMBIGUOUS", "What does Project Atlas use?", (m_amb.id,),
241
+ ))
242
+
243
+ # ------------------------------------------------------------------
244
+ # 15. NO-PATH — query for entity with no graph tool connection;
245
+ # should surface only via lexical/dense, not fabricate graph edge.
246
+ # ------------------------------------------------------------------
247
+ m_iso = _mem("nopath:isolated", "Project Sigma is an internal analytics service",
248
+ mem_type=MemoryType.CONTEXT)
249
+ memories.append(m_iso)
250
+ cases.append(AdversarialCase(
251
+ "NEGATIVE",
252
+ "What runtime does Project Sigma depend on?",
253
+ (m_iso.id,),
254
+ ))
255
+
256
+ # ------------------------------------------------------------------
257
+ # 16. VERSION_NUMBER — identifier with numeric suffix must survive.
258
+ # ------------------------------------------------------------------
259
+ m_ver = _mem("version:qwen", "Project Tau uses Qwen30B for inference",
260
+ mem_type=MemoryType.FACT)
261
+ memories.append(m_ver)
262
+ cases.append(AdversarialCase(
263
+ "DOTNAME", "Which model does Project Tau use?", (m_ver.id,),
264
+ ))
265
+
266
+ # ------------------------------------------------------------------
267
+ # 17. TEMPORAL REPLACEMENT — "used X but now uses Y" keeps only Y
268
+ # as a current USES edge.
269
+ # ------------------------------------------------------------------
270
+ m_rep = _mem("temporal_replace:mem",
271
+ "Project Upsilon used Docker previously but now uses Podman")
272
+ memories.append(m_rep)
273
+ cases.append(AdversarialCase(
274
+ "TEMPORAL", "What does Project Upsilon use now?", (m_rep.id,),
275
+ ))
276
+
277
+ # ------------------------------------------------------------------
278
+ # 18. SECOND SHARED-HUB CROSS-CHECK — Project Phi also uses Ollama;
279
+ # querying about Phi's specific task must not surface Eta/Zeta.
280
+ # ------------------------------------------------------------------
281
+ m_phi = _mem("hub:phi", "Project Phi uses Ollama")
282
+ m_phi_task = _mem("hub:phi:task", "Project Phi works on VisionTask",
283
+ mem_type=MemoryType.FACT)
284
+ memories.extend([m_phi, m_phi_task])
285
+ cases.append(AdversarialCase(
286
+ "SHARED_HUB", "What does Project Phi work on?", (m_phi_task.id,),
287
+ ))
288
+
289
+ # ------------------------------------------------------------------
290
+ # 19. COMPOUND WITH SEMICOLON — split on semicolon, same rules apply.
291
+ # ------------------------------------------------------------------
292
+ m_semi = _mem("compound:semi",
293
+ "Project Chi uses vLLM; Project Psi uses Ollama",
294
+ mem_type=MemoryType.FACT)
295
+ memories.append(m_semi)
296
+ cases.append(AdversarialCase(
297
+ "COMPOUND", "What does Project Chi use?", (m_semi.id,),
298
+ ))
299
+
300
+ # ------------------------------------------------------------------
301
+ # 20. MULTIHOP DEPTH-2 NO PATH — Omega has no tool that runs Qwen9B;
302
+ # multihop expansion must not fabricate the path.
303
+ # ------------------------------------------------------------------
304
+ m_omega = _mem("multihop_neg:omega", "Project Omega uses SQLite")
305
+ memories.append(m_omega)
306
+ cases.append(AdversarialCase(
307
+ "MULTIHOP",
308
+ "What model does Project Omega run through its runtime?",
309
+ (m_omega.id,),
310
+ ))
311
+
312
+ # ------------------------------------------------------------------
313
+ # 21. SUPERSEDED + CURRENT — for same project, CURRENT scope must
314
+ # prefer the active memory over the superseded one.
315
+ # ------------------------------------------------------------------
316
+ m_sup2_old = _mem("superseded2:old", "Project Psi uses Redis",
317
+ MemoryStatus.SUPERSEDED, MemoryType.PROJECT)
318
+ m_sup2_new = _mem("superseded2:new", "Project Psi uses Kafka")
319
+ memories.extend([m_sup2_old, m_sup2_new])
320
+ cases.append(AdversarialCase(
321
+ "SUPERSEDED_SUPPORT",
322
+ "What storage does Project Psi currently use?",
323
+ (m_sup2_new.id,),
324
+ ))
325
+
326
+ # ------------------------------------------------------------------
327
+ # 22. LEXICAL OVERLAP — many memories mention Docker; structured
328
+ # relation query for Project Omega/Docker must be precise.
329
+ # ------------------------------------------------------------------
330
+ m_stop = _mem("lexical:docker_omega", "Project Omega uses Docker")
331
+ memories.append(m_stop)
332
+ cases.append(AdversarialCase(
333
+ "LEXICAL_OVERLAP",
334
+ "What does Project Omega use for containerization?",
335
+ (m_stop.id,),
336
+ ))
337
+
338
+ # ------------------------------------------------------------------
339
+ # 23. DIRECT — bare tool name with multiple project users; retrieval
340
+ # must surface at least one relevant memory.
341
+ # ------------------------------------------------------------------
342
+ m_dir2 = _mem("direct:postgres", "Project Alpha uses PostgreSQL")
343
+ memories.append(m_dir2)
344
+ cases.append(AdversarialCase(
345
+ "DIRECT", "PostgreSQL", (m_dir2.id,),
346
+ ))
347
+
348
+ # ------------------------------------------------------------------
349
+ # 24. AMBIGUOUS PROJECT NAME — "Nova" in unrelated content must not
350
+ # pollute the project-level structured query.
351
+ # ------------------------------------------------------------------
352
+ m_nova = _mem("ambiguous:nova", "Project Nova uses Ollama")
353
+ m_nova_noise = _mem("ambiguous:nova_noise",
354
+ "Nova is a constellation in the southern hemisphere",
355
+ MemoryStatus.ACTIVE, MemoryType.CONTEXT)
356
+ memories.extend([m_nova, m_nova_noise])
357
+ cases.append(AdversarialCase(
358
+ "AMBIGUOUS", "What does Project Nova use?", (m_nova.id,),
359
+ ))
360
+
361
+ # ------------------------------------------------------------------
362
+ # 25. RELATIONAL + NEGATION COMBO — positive and negative clause in
363
+ # same sentence; only the positive clause must produce an edge.
364
+ # ------------------------------------------------------------------
365
+ m_rel_neg = _mem("relational_neg:mem",
366
+ "Project Omega uses Podman but does not use Docker",
367
+ mem_type=MemoryType.FACT)
368
+ memories.append(m_rel_neg)
369
+ cases.append(AdversarialCase(
370
+ "NEGATIVE",
371
+ "Does Project Omega use Podman?",
372
+ (m_rel_neg.id,),
373
+ ))
374
+
375
+ assert len(cases) >= 25, f"Expected >= 25 adversarial cases, got {len(cases)}"
376
+ return memories, cases
377
+
378
+
379
+ async def run_adversarial_benchmark(root: Path | None = None) -> dict[str, object]:
380
+ owned = root is None
381
+ temporary = tempfile.TemporaryDirectory() if owned else None
382
+ directory = Path(temporary.name) if temporary else root
383
+ assert directory is not None
384
+ directory.mkdir(parents=True, exist_ok=True)
385
+ database = Database(directory / "graph-adversarial.db")
386
+ await database.initialize()
387
+ try:
388
+ memory_repo = SqliteMemoryRepository(database.connection())
389
+ relation_repo = SqliteRelationRepository(database.connection())
390
+ graph_repo = SqliteGraphRepository(database.connection())
391
+ graph = MemoryGraphService(
392
+ memory_repo=memory_repo, relation_repo=relation_repo, graph_repo=graph_repo,
393
+ )
394
+ memories, cases = adversarial_corpus()
395
+ for memory in memories:
396
+ await memory_repo.create(memory)
397
+
398
+ embedding = DeterministicEmbedding(32)
399
+ lexical = BM25Index()
400
+ vector = InMemoryVectorStore(32)
401
+ base = HybridRetrievalEngine(
402
+ memory_repo=memory_repo, lexical_index=lexical, vector_store=vector,
403
+ embedding_service=embedding,
404
+ index_synchronizer=RetrievalIndexSynchronizer(
405
+ memory_repo=memory_repo, lexical_index=lexical, vector_store=vector,
406
+ embedding_service=embedding,
407
+ ),
408
+ )
409
+ engine = GraphAugmentedRetrievalEngine(
410
+ base_engine=base, graph_service=graph, memory_repo=memory_repo,
411
+ )
412
+
413
+ metrics: dict[str, list[float]] = {"recall": [], "mrr": [], "ndcg": [], "hit": []}
414
+ by_category: dict[str, dict[str, list[float]]] = {}
415
+ safety: dict[str, bool] = {}
416
+
417
+ for case in cases:
418
+ result = await engine.retrieve(RetrievalQuery(
419
+ text=case.text, mode=RetrievalMode.HYBRID_GRAPH, k=5,
420
+ temporal_scope=case.scope,
421
+ ))
422
+ ranking = [str(item.memory.id) for item in result.memories]
423
+ if case.expect_empty:
424
+ # The deleted memory's ID must not appear in any retrieval result.
425
+ deleted_ids = {str(rel_id) for rel_id in case.relevant if case.relevant}
426
+ # relevant is empty for deleted cases; safety checked below via deleted_id set.
427
+ # Still record that no result has a deleted-status memory.
428
+ safety[f"deleted_support_excluded_{case.category.lower()}"] = not any(
429
+ item.memory.status.value == "deleted" for item in result.memories
430
+ )
431
+ continue
432
+ qrels = {str(identifier): 1 for identifier in case.relevant}
433
+ r = recall_at_k(ranking, qrels, 5)
434
+ m = reciprocal_rank(ranking, qrels)
435
+ n = ndcg_at_k(ranking, qrels, 5)
436
+ h = hit_rate_at_k(ranking, qrels, 5)
437
+ metrics["recall"].append(r)
438
+ metrics["mrr"].append(m)
439
+ metrics["ndcg"].append(n)
440
+ metrics["hit"].append(h)
441
+ cat = case.category
442
+ if cat not in by_category:
443
+ by_category[cat] = {"recall": [], "mrr": [], "ndcg": [], "hit": []}
444
+ by_category[cat]["recall"].append(r)
445
+ by_category[cat]["mrr"].append(m)
446
+ by_category[cat]["ndcg"].append(n)
447
+ by_category[cat]["hit"].append(h)
448
+
449
+ # --- Safety checks ---
450
+ edges = await graph_repo.all_edges()
451
+ nodes = {node.id: node for node in await graph_repo.nodes()}
452
+
453
+ # Negation: Epsilon explicitly negated "does not use Docker" → no Epsilon→Docker USES edge
454
+ epsilon_node_key = "epsilon"
455
+ epsilon_node = next(
456
+ (node for node in nodes.values() if node.canonical_key == epsilon_node_key), None
457
+ )
458
+ safety["negation_epsilon_no_docker_edge"] = not any(
459
+ edge.relation_type.value == "uses"
460
+ and edge.source_node_id == (epsilon_node.id if epsilon_node else None)
461
+ and nodes.get(edge.target_node_id) is not None
462
+ and nodes[edge.target_node_id].canonical_key == "docker"
463
+ for edge in edges
464
+ )
465
+ # Negation: no USES edge targeting MySQL from Project Nu
466
+ safety["explicit_negation_mysql_no_edge"] = not any(
467
+ edge.relation_type.value == "uses"
468
+ and nodes.get(edge.target_node_id) is not None
469
+ and nodes[edge.target_node_id].canonical_key == "mysql"
470
+ for edge in edges
471
+ )
472
+ # Dotname: llama.cpp survives canonicalization
473
+ extractor = DeterministicEntityExtractor()
474
+ rel_dot = extractor.relations("Project Pi uses llama.cpp")
475
+ safety["dotname_llama_cpp_preserved"] = (
476
+ len(rel_dot) == 1 and rel_dot[0].target.key == "llama.cpp"
477
+ )
478
+ # Dotname: c++17 survives canonicalization
479
+ rel_cpp = extractor.relations("Project Rho uses c++17")
480
+ safety["dotname_cpp17_preserved"] = (
481
+ len(rel_cpp) == 1 and rel_cpp[0].target.key == "c++17"
482
+ )
483
+ # Compound: exactly two edges from compound sentence, no cross-project edge
484
+ rel_comp = extractor.relations("Project Xi uses Python and Project Omicron uses Rust")
485
+ pairs = {(r.source.key, r.target.key) for r in rel_comp}
486
+ safety["compound_no_cross_edge"] = pairs == {("xi", "python"), ("omicron", "rust")}
487
+
488
+ # Shared-hub scoping: Project Zeta query must not surface Eta's failure memory
489
+ scope_result = await engine.retrieve(RetrievalQuery(
490
+ text="What does Project Zeta use?",
491
+ mode=RetrievalMode.GRAPH, k=10, graph_max_hops=3,
492
+ ))
493
+ safety["shared_hub_no_cross_project_leak"] = all(
494
+ "Failure7" not in item.memory.content for item in scope_result.memories
495
+ )
496
+
497
+ # Deleted-support: Lambda/Redis edge must not appear in graph after rebuild
498
+ # (MemoryStatus.DELETED is excluded from projected_statuses)
499
+ safety["deleted_support_no_edge"] = not any(
500
+ "lambda" in (nodes.get(edge.source_node_id) and nodes[edge.source_node_id].canonical_key or "")
501
+ for edge in edges if edge.relation_type.value == "uses"
502
+ )
503
+
504
+ avg = {key: sum(values) / len(values) for key, values in metrics.items() if values}
505
+ cat_avg = {
506
+ cat: {key: sum(values) / len(values) for key, values in v.items() if values}
507
+ for cat, v in by_category.items()
508
+ }
509
+ return {
510
+ "memories": len(memories),
511
+ "queries": len([c for c in cases if not c.expect_empty]),
512
+ "categories": sorted(cat_avg.keys()),
513
+ "overall": avg,
514
+ "by_category": cat_avg,
515
+ "safety": safety,
516
+ }
517
+ finally:
518
+ await database.close()
519
+ if temporary:
520
+ temporary.cleanup()
521
+
522
+
523
+ if __name__ == "__main__":
524
+ import json
525
+ print(json.dumps(asyncio.run(run_adversarial_benchmark()), indent=2, sort_keys=True))