cortexm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. context_m.py +17 -0
  2. cortexm/__init__.py +45 -0
  3. cortexm/accel.py +403 -0
  4. cortexm/api/__init__.py +0 -0
  5. cortexm/api/chaos.py +118 -0
  6. cortexm/api/memory.py +635 -0
  7. cortexm/bench/__init__.py +0 -0
  8. cortexm/bench/abilities.py +311 -0
  9. cortexm/bench/baselines.py +89 -0
  10. cortexm/bench/beam_loader.py +317 -0
  11. cortexm/bench/generator.py +376 -0
  12. cortexm/bench/harness.py +211 -0
  13. cortexm/bench/messy.py +218 -0
  14. cortexm/bench/micro.py +251 -0
  15. cortexm/bench/ood.py +443 -0
  16. cortexm/bench/run.py +137 -0
  17. cortexm/bridge/__init__.py +0 -0
  18. cortexm/bridge/dates.py +178 -0
  19. cortexm/bridge/decoders.py +204 -0
  20. cortexm/bridge/enrich.py +255 -0
  21. cortexm/bridge/extractor.py +316 -0
  22. cortexm/bridge/fallback.py +332 -0
  23. cortexm/bridge/onnx_runtime.py +158 -0
  24. cortexm/bridge/patterns.py +760 -0
  25. cortexm/bridge/ppr.py +104 -0
  26. cortexm/bridge/prefilter.py +188 -0
  27. cortexm/bridge/query_extract.py +420 -0
  28. cortexm/bridge/reader.py +1174 -0
  29. cortexm/bridge/rerank.py +204 -0
  30. cortexm/bridge/writer.py +492 -0
  31. cortexm/cli.py +295 -0
  32. cortexm/cognition/__init__.py +53 -0
  33. cortexm/cognition/abstraction.py +192 -0
  34. cortexm/cognition/analogy.py +159 -0
  35. cortexm/cognition/engine.py +204 -0
  36. cortexm/cognition/gaps.py +365 -0
  37. cortexm/cognition/scanner.py +204 -0
  38. cortexm/config.py +375 -0
  39. cortexm/cortexm.py +8 -0
  40. cortexm/enterprise/__init__.py +0 -0
  41. cortexm/enterprise/audit.py +178 -0
  42. cortexm/enterprise/governance.py +239 -0
  43. cortexm/errors.py +35 -0
  44. cortexm/features/__init__.py +0 -0
  45. cortexm/features/git.py +204 -0
  46. cortexm/features/prefetch.py +88 -0
  47. cortexm/features/zk.py +105 -0
  48. cortexm/federation/__init__.py +39 -0
  49. cortexm/federation/crdt.py +275 -0
  50. cortexm/federation/fabric.py +109 -0
  51. cortexm/federation/hlc.py +80 -0
  52. cortexm/federation/node.py +145 -0
  53. cortexm/federation/schema_report.py +73 -0
  54. cortexm/federation/transport.py +164 -0
  55. cortexm/index/__init__.py +19 -0
  56. cortexm/index/nsg.py +386 -0
  57. cortexm/mcp/__init__.py +0 -0
  58. cortexm/mcp/server.py +985 -0
  59. cortexm/metrics.py +62 -0
  60. cortexm/migrate/__init__.py +0 -0
  61. cortexm/migrate/importers.py +192 -0
  62. cortexm/provenance/__init__.py +78 -0
  63. cortexm/provenance/agent.py +214 -0
  64. cortexm/provenance/cose.py +201 -0
  65. cortexm/provenance/scitt.py +258 -0
  66. cortexm/provenance/vc.py +250 -0
  67. cortexm/security/__init__.py +0 -0
  68. cortexm/security/crypto.py +162 -0
  69. cortexm/security/hashes.py +140 -0
  70. cortexm/security/injection.py +149 -0
  71. cortexm/security/mind.py +154 -0
  72. cortexm/security/pii.py +265 -0
  73. cortexm/security/rbac.py +169 -0
  74. cortexm/security/sandbox.py +131 -0
  75. cortexm/security/zk_hamming.py +142 -0
  76. cortexm/security/zk_sql.py +485 -0
  77. cortexm/server/__init__.py +0 -0
  78. cortexm/server/metrics.py +88 -0
  79. cortexm/server/rest.py +936 -0
  80. cortexm/server/sparql.py +984 -0
  81. cortexm/text/__init__.py +0 -0
  82. cortexm/text/dissim.py +252 -0
  83. cortexm/text/embedder.py +155 -0
  84. cortexm/text/fuzzy.py +218 -0
  85. cortexm/text/idiolect.py +253 -0
  86. cortexm/text/labse.py +374 -0
  87. cortexm/text/tokenizer.py +79 -0
  88. cortexm/trace/__init__.py +0 -0
  89. cortexm/trace/blob_arena.py +277 -0
  90. cortexm/trace/consolidate.py +337 -0
  91. cortexm/trace/contradictions.py +69 -0
  92. cortexm/trace/dedup.py +114 -0
  93. cortexm/trace/edges.py +214 -0
  94. cortexm/trace/fact.py +121 -0
  95. cortexm/trace/fade.py +245 -0
  96. cortexm/trace/lifecycle.py +112 -0
  97. cortexm/trace/rebuild.py +173 -0
  98. cortexm/trace/rules.py +171 -0
  99. cortexm/trace/store.py +680 -0
  100. cortexm/trace/structural.py +183 -0
  101. cortexm/trace/tmt.py +335 -0
  102. cortexm/util.py +148 -0
  103. cortexm/vsa/__init__.py +0 -0
  104. cortexm/vsa/attribution.py +149 -0
  105. cortexm/vsa/cleanup.py +161 -0
  106. cortexm/vsa/codecs.py +397 -0
  107. cortexm/vsa/hologram_overlay.py +139 -0
  108. cortexm/vsa/index.py +163 -0
  109. cortexm/vsa/ops.py +149 -0
  110. cortexm/vsa/palace.py +446 -0
  111. cortexm/vsa/role_vectors.py +236 -0
  112. cortexm/vsa/slb.py +78 -0
  113. cortexm/vsa/tlsh_trie.py +137 -0
  114. cortexm/vsa/working_memory.py +249 -0
  115. cortexm-0.3.0.dist-info/METADATA +482 -0
  116. cortexm-0.3.0.dist-info/RECORD +120 -0
  117. cortexm-0.3.0.dist-info/WHEEL +5 -0
  118. cortexm-0.3.0.dist-info/entry_points.txt +2 -0
  119. cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
  120. cortexm-0.3.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,984 @@
1
+ """SPARQL endpoint — non-LLM decoder path (NSR-inspired, v2 extended).
2
+
3
+ arXiv insight (NSR / ESWEEK24): the VSA core is task-agnostic; only
4
+ the decoder changes. Context-M's reader was hardcoded to format facts
5
+ for an LLM prompt. The Decoders module extracts that formatter into
6
+ a pluggable interface — `RDFDecoder` exports facts as RDF/N3 triples.
7
+
8
+ This module exposes a SPARQL HTTP endpoint that:
9
+ 1. accepts SPARQL SELECT queries via HTTP GET/POST
10
+ 2. parses the query (a hand-rolled parser covering the common
11
+ subset: SELECT [DISTINCT] ?vars WHERE { triple-patterns +
12
+ FILTERs + OPTIONAL } ORDER BY ?var [ASC|DESC] LIMIT N)
13
+ 3. retrieves facts from Context-M's Memory.store (or its
14
+ RDFDecoder if attached — same substrate either way)
15
+ 4. runs the WHERE clause as a join pipeline over those triples
16
+ — naive nested-loop join with binding propagation, OPTIONAL
17
+ handled via LEFT-JOIN semantics, FILTER applied per binding
18
+ 5. resolves blob-arena-stored objects on demand (the sidecar
19
+ blob arena is the Aeon off-graph store; SPARQL "o" may
20
+ transparently dereference long text via Memory.get_chunk_text)
21
+ 6. exposes CAUSAL / REFERS_TO typed edges (Aeon) via the
22
+ `edge/2` predicate family so external graph tools can walk
23
+ the truth-maintenance and episodic-atlas graphs natively
24
+ 7. returns the result as SPARQL 1.1 JSON Results
25
+
26
+ This is a NON-LLM retrieval path. Zero LLM calls. The same palace +
27
+ Trace substrate that powers LLM context-stuffing now serves SPARQL.
28
+
29
+ Usage:
30
+ # standalone SPARQL endpoint (e.g. on port 8910)
31
+ python -m cortexm.server.sparql --port 8910
32
+
33
+ # launched alongside the REST API (recommended for production):
34
+ # `cortexm serve-rest --sparql-port 8910` — both share one Memory
35
+ # instance; the REST API also exposes /v1/sparql for unified access.
36
+
37
+ Example queries (v2 supports a much richer subset than v1):
38
+ SELECT ?s ?p ?o WHERE { ?s ?p ?o } LIMIT 10
39
+ SELECT DISTINCT ?s WHERE { ?s ?p ?o } ORDER BY ?s
40
+ SELECT ?s ?o WHERE { ?s "name" ?o . FILTER regex(?o, "^Jen", "i") }
41
+ SELECT ?cause ?effect WHERE {
42
+ ?cause edge:CAUSAL ?effect .
43
+ ?effect "name" "Alice"
44
+ } LIMIT 5
45
+
46
+ LIMITATIONS (honest, documented):
47
+ - Parser covers the SELECT subset named above. UNION, CONSTRUCT,
48
+ ASK, DESCRIBE, property paths (rdf:type/rdfs:subClassOf*), and
49
+ SPARQL 1.1 aggregates (GROUP BY / COUNT / SUM) are NOT yet
50
+ supported. For those, export via RDFDecoder + use Apache Jena.
51
+ - No inference / reasoning over RDFS/OWL — pattern matching on
52
+ the actual stored facts only.
53
+ - Joins are nested-loop (no query planner). Adequate for the
54
+ ~10^5 facts Context-M is designed to hold per user; for larger
55
+ stores, use a real triple store fed via the RDFDecoder export.
56
+ """
57
+ from __future__ import annotations
58
+
59
+ import json
60
+ import re
61
+ import threading
62
+ import urllib.parse
63
+ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
64
+ from typing import Iterable
65
+
66
+ # Typed-edge vocabulary (Aeon) — used by SPARQL `edge:KIND` predicates.
67
+ # These are the canonical edge kinds actually WRITTEN by writer.py +
68
+ # consolidate.py. Note: supersession is captured via CONTRADICTS edges
69
+ # (writer writes CONTRADICTS with meta indicating kind), so there is no
70
+ # standalone SUPERSEDS constant exported from edges.py.
71
+ from cortexm.trace.edges import (
72
+ CAUSAL, REFERS_TO, CONTRADICTS, MERGED_WITH,
73
+ EXTRACTED_FROM, TEMPORALLY_PRECEDED_BY, NEXT,
74
+ )
75
+ # SPARQL-facing aliases: edge:SUPERSEDES resolves to the actual
76
+ # CONTRADICTS replacement edge. This keeps the user-facing vocabulary
77
+ # stable even though internally we collapse SUPERSEDE → CONTRADICTS.
78
+ SUPERSEDS = CONTRADICTS # alias for `edge:SUPERSEDES` queries
79
+
80
+
81
+ # ----------------------------------------------------------- parser
82
+ def parse_sparql(query: str) -> dict:
83
+ """Parse a SPARQL SELECT query.
84
+
85
+ Supports (v2, extended):
86
+ SELECT [DISTINCT] ?var+ WHERE {
87
+ triple-pattern ( ?s ?p ?o . | ?s "literal" ?o . | ... )
88
+ FILTER ( regex(?v, "pat", "i") | ?v = "lit" | ?v != "lit" )
89
+ OPTIONAL { ... } # left-join semantics
90
+ }
91
+ [ORDER BY ?var [ASC|DESC]]
92
+ [LIMIT N]
93
+ [OFFSET N]
94
+
95
+ Typed-edge shortcut: `edge:KIND` is recognized as a predicate
96
+ where KIND is one of the canonical edge kinds (CAUSAL, REFERS_TO,
97
+ SUPERSEDES, CONTRADICTS, MERGED_WITH, EXTRACTED_FROM,
98
+ TEMPORALLY_PRECEDED_BY, NEXT). The triple (?a edge:CAUSAL ?b)
99
+ queries the Trace's edges table instead of the facts table.
100
+
101
+ Returns:
102
+ {
103
+ "distinct": bool,
104
+ "select": ["?s", "?p", "?o"],
105
+ "patterns": [("?s", "?p", "?o"), ...],
106
+ "optionals":[[pat, ...], ...], # one list per OPTIONAL block
107
+ "filters": [{"type": "regex"/"equals"/"ne", ...}, ...],
108
+ "order_by": {"var": "?s", "desc": False} | None,
109
+ "limit": int | None,
110
+ "offset": int | None,
111
+ }
112
+
113
+ Raises ValueError on unparseable input.
114
+ """
115
+ query = query.strip()
116
+ # strip trailing semicolons / whitespace
117
+ while query.endswith(";"):
118
+ query = query[:-1].strip()
119
+
120
+ # match: SELECT [DISTINCT] <select-vars> WHERE { <where> }
121
+ # select-vars may be one or more ?var tokens, or *
122
+ sel_grp = ""
123
+ where_clause = ""
124
+ tail = ""
125
+ distinct_grp = ""
126
+ matched = False
127
+ for pat in (
128
+ # canonical form with WHERE keyword
129
+ r"SELECT\s+(DISTINCT\s+)?((?:\?\S+(?:\s+|$))+|\*\s*?)\s*WHERE\s*\{(.*)\}\s*(.*)$",
130
+ # without WHERE keyword (some demos)
131
+ r"SELECT\s+(DISTINCT\s+)?((?:\?\S+(?:\s+|$))+|\*\s*?)\s*\{(.*)\}\s*(.*)$",
132
+ ):
133
+ m = re.match(pat, query, re.IGNORECASE | re.DOTALL)
134
+ if m:
135
+ distinct_grp = m.group(1) or ""
136
+ sel_grp = m.group(2) or ""
137
+ where_clause = m.group(3) or ""
138
+ tail = m.group(4) or ""
139
+ matched = True
140
+ break
141
+ if not matched:
142
+ raise ValueError(
143
+ "query must be 'SELECT [DISTINCT] ?vars WHERE { ... }' "
144
+ "— more complex SPARQL not yet supported in v2")
145
+
146
+ # parse SELECT vars (may be multiple ?vars separated by spaces, or *)
147
+ is_distinct = bool(distinct_grp)
148
+ if sel_grp.strip() == "*":
149
+ select_vars: list[str] = ["?s", "?p", "?o"]
150
+ else:
151
+ # find all ?vars in the select clause
152
+ select_vars = re.findall(r"\?\w+", sel_grp)
153
+ if not select_vars:
154
+ raise ValueError("SELECT must list at least one variable")
155
+
156
+ # parse tail (ORDER BY / LIMIT / OFFSET — order-insensitive)
157
+ order_by, limit, offset = _parse_tail(tail + " " + where_clause[-0:])
158
+ # NOTE: the WHERE clause may also have a trailing LIMIT etc — we
159
+ # extract the tail BEFORE the where clause's trailing }. The above
160
+ # regex captures tail as everything AFTER the closing }. Good.
161
+
162
+ # parse WHERE: triple patterns, FILTER, OPTIONAL
163
+ where_clause = where_clause.strip()
164
+ patterns: list[tuple[str, str, str]] = []
165
+ filters: list[dict] = []
166
+ optionals: list[list[tuple[str, str, str]]] = []
167
+
168
+ # tokenize while respecting OPTIONAL { ... } nesting, FILTER(...),
169
+ # and quoted strings. Then merge 'FILTER' + the following
170
+ # parenthesized token so _parse_filter sees the full clause.
171
+ tokens = _merge_filter_tokens(_tokenize_where(where_clause))
172
+
173
+ cur_optional: list[tuple[str, str, str]] | None = None
174
+ i = 0
175
+ while i < len(tokens):
176
+ tok = tokens[i].strip()
177
+ if not tok:
178
+ i += 1
179
+ continue
180
+ upper = tok.upper()
181
+ if upper.startswith("OPTIONAL"):
182
+ # expect '{' next
183
+ i += 1
184
+ if i >= len(tokens) or tokens[i].strip() != "{":
185
+ raise ValueError("OPTIONAL must be followed by '{'")
186
+ # gather until matching '}'
187
+ depth = 1
188
+ inner: list[str] = []
189
+ i += 1
190
+ while i < len(tokens) and depth > 0:
191
+ t = tokens[i].strip()
192
+ if t == "{":
193
+ depth += 1
194
+ inner.append(t)
195
+ elif t == "}":
196
+ depth -= 1
197
+ if depth > 0:
198
+ inner.append(t)
199
+ else:
200
+ inner.append(t)
201
+ i += 1
202
+ # parse inner as a sub-clause (patterns + filters)
203
+ sub_pats, sub_filts = _parse_triples(inner)
204
+ # v2 stores optional patterns separately; filters inside
205
+ # OPTIONAL are also applied within the left-join
206
+ optionals.append(sub_pats)
207
+ filters.extend(sub_filts) # FILTERs are hoisted — applied later
208
+ cur_optional = None
209
+ continue
210
+ if upper.startswith("FILTER"):
211
+ filt = _parse_filter(tok)
212
+ if filt:
213
+ filters.append(filt)
214
+ i += 1
215
+ continue
216
+ # else: triple-pattern token. Look at groups of 3.
217
+ # Easier: re-tokenize by '.' but keep quoted strings together
218
+ # _tokenize_where already split on whitespace, so triple tokens
219
+ # come as a stream. Group until we have 3 non-'.' tokens.
220
+ # Skip '.' separator.
221
+ if tok == ".":
222
+ i += 1
223
+ continue
224
+ # collect 3 terms for a triple
225
+ terms = [tok]
226
+ j = i + 1
227
+ while j < len(tokens) and len(terms) < 3:
228
+ t = tokens[j].strip()
229
+ if t == ".":
230
+ j += 1
231
+ continue
232
+ if t.upper().startswith("FILTER") or t.upper().startswith(
233
+ "OPTIONAL") or t == "}":
234
+ break
235
+ terms.append(t)
236
+ j += 1
237
+ if len(terms) == 3:
238
+ patterns.append(tuple(terms))
239
+ i = j
240
+
241
+ return {
242
+ "distinct": is_distinct,
243
+ "select": select_vars,
244
+ "patterns": patterns,
245
+ "optionals": optionals,
246
+ "filters": filters,
247
+ "order_by": order_by,
248
+ "limit": limit,
249
+ "offset": offset,
250
+ }
251
+
252
+
253
+ def _parse_tail(tail: str) -> tuple[dict | None, int | None, int | None]:
254
+ """Extract ORDER BY / LIMIT / OFFSET from the post-WHERE tail."""
255
+ order_by = None
256
+ limit = None
257
+ offset = None
258
+
259
+ # ORDER BY ?var [ASC|DESC]
260
+ m = re.search(
261
+ r"ORDER\s+BY\s+(\?\w+)(?:\s+(ASC|DESC))?",
262
+ tail, re.IGNORECASE)
263
+ if m:
264
+ order_by = {"var": m.group(1),
265
+ "desc": (m.group(2) or "").upper() == "DESC"}
266
+
267
+ # LIMIT N
268
+ m = re.search(r"LIMIT\s+(\d+)", tail, re.IGNORECASE)
269
+ if m:
270
+ limit = int(m.group(1))
271
+
272
+ # OFFSET N
273
+ m = re.search(r"OFFSET\s+(\d+)", tail, re.IGNORECASE)
274
+ if m:
275
+ offset = int(m.group(1))
276
+
277
+ return order_by, limit, offset
278
+
279
+
280
+ def _tokenize_where(s: str) -> list[str]:
281
+ """Split WHERE body into tokens, preserving quoted strings + parens."""
282
+ tokens: list[str] = []
283
+ cur = ""
284
+ in_string = False
285
+ quote_char = None
286
+ depth = 0
287
+ for ch in s:
288
+ if in_string:
289
+ cur += ch
290
+ if ch == quote_char:
291
+ in_string = False
292
+ continue
293
+ if ch in ('"', "'"):
294
+ in_string = True
295
+ quote_char = ch
296
+ cur += ch
297
+ continue
298
+ if ch == "(":
299
+ depth += 1
300
+ cur += ch
301
+ continue
302
+ if ch == ")":
303
+ depth -= 1
304
+ cur += ch
305
+ continue
306
+ if ch == "{":
307
+ if cur.strip():
308
+ tokens.append(cur.strip())
309
+ cur = ""
310
+ tokens.append("{")
311
+ continue
312
+ if ch == "}":
313
+ if cur.strip():
314
+ tokens.append(cur.strip())
315
+ cur = ""
316
+ tokens.append("}")
317
+ continue
318
+ if depth > 0:
319
+ cur += ch
320
+ continue
321
+ if ch.isspace():
322
+ if cur.strip():
323
+ tokens.append(cur.strip())
324
+ cur = ""
325
+ continue
326
+ cur += ch
327
+ if cur.strip():
328
+ tokens.append(cur.strip())
329
+ return tokens
330
+
331
+
332
+ def _parse_triples(tokens: list[str]) -> tuple[list[tuple], list[dict]]:
333
+ """Parse a flat list of triple tokens + FILTERs into (pats, filts)."""
334
+ tokens = _merge_filter_tokens(tokens)
335
+ pats: list[tuple] = []
336
+ filts: list[dict] = []
337
+ i = 0
338
+ while i < len(tokens):
339
+ tok = tokens[i].strip()
340
+ if not tok:
341
+ i += 1
342
+ continue
343
+ if tok.upper().startswith("FILTER"):
344
+ f = _parse_filter(tok)
345
+ if f:
346
+ filts.append(f)
347
+ i += 1
348
+ continue
349
+ if tok == ".":
350
+ i += 1
351
+ continue
352
+ terms = [tok]
353
+ j = i + 1
354
+ while j < len(tokens) and len(terms) < 3:
355
+ t = tokens[j].strip()
356
+ if t == ".":
357
+ j += 1
358
+ continue
359
+ if t.upper().startswith("FILTER") or t == "}":
360
+ break
361
+ terms.append(t)
362
+ j += 1
363
+ if len(terms) == 3:
364
+ pats.append(tuple(terms))
365
+ i = j
366
+ return pats, filts
367
+
368
+
369
+ def _parse_filter(tok: str) -> dict | None:
370
+ """Parse one FILTER(...) clause into a filter dict.
371
+
372
+ `tok` may be:
373
+ - 'FILTER regex(?v, "pat", "flags")' (merged by tokenizer)
374
+ - 'FILTER(?v = "value")'
375
+ - 'FILTER(?v != "value")'
376
+ - 'FILTER' alone (in which case caller should merge next token)
377
+ """
378
+ # FILTER regex(?v, "pat", "flags")
379
+ m = re.match(
380
+ r"FILTER\s+regex\s*\(\s*(\?\w+)\s*,\s*[\"'](.+?)[\"']\s*"
381
+ r"(?:,\s*[\"'](\w+)[\"']\s*)?\)",
382
+ tok, re.IGNORECASE | re.DOTALL)
383
+ if m:
384
+ return {"type": "regex", "var": m.group(1),
385
+ "pattern": m.group(2), "flags": m.group(3) or ""}
386
+ # FILTER regex without space: FILTERregex(...) — unlikely but tolerated
387
+ m = re.match(
388
+ r"FILTERregex\s*\(\s*(\?\w+)\s*,\s*[\"'](.+?)[\"']\s*"
389
+ r"(?:,\s*[\"'](\w+)[\"']\s*)?\)",
390
+ tok, re.IGNORECASE | re.DOTALL)
391
+ if m:
392
+ return {"type": "regex", "var": m.group(1),
393
+ "pattern": m.group(2), "flags": m.group(3) or ""}
394
+ # FILTER(?v = "value")
395
+ m = re.match(
396
+ r"FILTER\s*\(\s*(\?\w+)\s*=\s*[\"'](.+?)[\"']\s*\)",
397
+ tok, re.IGNORECASE | re.DOTALL)
398
+ if m:
399
+ return {"type": "equals", "var": m.group(1), "value": m.group(2)}
400
+ # FILTER(?v != "value")
401
+ m = re.match(
402
+ r"FILTER\s*\(\s*(\?\w+)\s*!=\s*[\"'](.+?)[\"']\s*\)",
403
+ tok, re.IGNORECASE | re.DOTALL)
404
+ if m:
405
+ return {"type": "ne", "var": m.group(1), "value": m.group(2)}
406
+ return None
407
+
408
+
409
+ def _merge_filter_tokens(tokens: list[str]) -> list[str]:
410
+ """If 'FILTER' appears as a standalone token, merge it with the
411
+ following parenthesized token so _parse_filter sees the full
412
+ 'FILTER regex(...)' or 'FILTER(...)' string.
413
+ """
414
+ out: list[str] = []
415
+ i = 0
416
+ while i < len(tokens):
417
+ t = tokens[i]
418
+ if t.upper() == "FILTER" and i + 1 < len(tokens):
419
+ merged = t + " " + tokens[i + 1]
420
+ out.append(merged)
421
+ i += 2
422
+ continue
423
+ out.append(t)
424
+ i += 1
425
+ return out
426
+
427
+
428
+ # ----------------------------------------------------------- matcher
429
+ def match_triple(triple: tuple, pattern: tuple,
430
+ bindings: dict | None = None) -> dict | None:
431
+ """Try to match a triple against a pattern, returning extended
432
+ bindings or None if no match.
433
+
434
+ triple: (subject, relation, value) or (subject, relation, value,
435
+ source_id) from Context-M facts. If a 4-tuple is passed,
436
+ the source_id is stashed under the synthetic `__source_id`
437
+ key (not projectable) so the executor can resolve arena
438
+ text on demand.
439
+ pattern: ("?s", "?p", "?o") or with literals like
440
+ ("?s", "name", "?o")
441
+ """
442
+ if len(triple) not in (3, 4) or len(pattern) != 3:
443
+ return None
444
+ b = dict(bindings or {})
445
+ # if the triple has a source_id, stash it for the executor
446
+ if len(triple) == 4:
447
+ b["__source_id"] = triple[3]
448
+ for tri_elem, pat_elem in zip(triple[:3], pattern):
449
+ if pat_elem.startswith("?"):
450
+ if pat_elem in b:
451
+ if b[pat_elem] != tri_elem:
452
+ return None
453
+ else:
454
+ b[pat_elem] = tri_elem
455
+ else:
456
+ # literal: strip quotes
457
+ literal = pat_elem.strip('"\'')
458
+ if literal != tri_elem:
459
+ return None
460
+ return b
461
+
462
+
463
+ def apply_filters(bindings: dict, filters: list[dict],
464
+ blob_resolver=None) -> bool:
465
+ """Return True if `bindings` satisfies all `filters`.
466
+
467
+ If `blob_resolver` is provided and a filter targets a variable
468
+ whose value is the synthetic `__source_id` (a chunk_id pointing
469
+ into the sidecar blob arena), we resolve the source text on
470
+ demand and apply the filter against the resolved text. This makes
471
+ `FILTER regex(?src, "pat")` work when ?src is bound to a chunk_id
472
+ via a special `?_source` projection variable.
473
+ """
474
+ for f in filters:
475
+ var = f["var"]
476
+ if var not in bindings:
477
+ # special case: ?_source / ?source_text refers to the
478
+ # arena-resolved chunk text; we resolve lazily here.
479
+ if var in ("?_source", "?source_text") and blob_resolver:
480
+ src_id = bindings.get("__source_id")
481
+ if not src_id:
482
+ return False
483
+ try:
484
+ val = blob_resolver(None, None, src_id) or ""
485
+ except Exception:
486
+ val = ""
487
+ if f["type"] == "regex":
488
+ flags = 0
489
+ if "i" in f.get("flags", ""):
490
+ flags |= re.IGNORECASE
491
+ if not re.search(f["pattern"], val, flags):
492
+ return False
493
+ elif f["type"] == "equals":
494
+ if val != f["value"]:
495
+ return False
496
+ elif f["type"] == "ne":
497
+ if val == f["value"]:
498
+ return False
499
+ continue
500
+ return False
501
+ val = bindings[var]
502
+ if f["type"] == "regex":
503
+ flags = 0
504
+ if "i" in f.get("flags", ""):
505
+ flags |= re.IGNORECASE
506
+ if not re.search(f["pattern"], val, flags):
507
+ return False
508
+ elif f["type"] == "equals":
509
+ if val != f["value"]:
510
+ return False
511
+ elif f["type"] == "ne":
512
+ if val == f["value"]:
513
+ return False
514
+ return True
515
+
516
+
517
+ # ----------------------------------------------------------- edge triples
518
+ # Mapping from SPARQL `edge:KIND` predicates to the actual edge kinds
519
+ # stored in the Trace. External graph tools get a stable SPARQL-facing
520
+ # vocabulary that maps to Aeon's typed-edge graph internally.
521
+ _EDGE_PRED_TO_KIND = {
522
+ "edge:CAUSAL": CAUSAL,
523
+ "edge:REFERS_TO": REFERS_TO,
524
+ "edge:SUPERSEDES": SUPERSEDS,
525
+ "edge:CONTRADICTS": CONTRADICTS,
526
+ "edge:MERGED_WITH": MERGED_WITH,
527
+ "edge:EXTRACTED_FROM": EXTRACTED_FROM,
528
+ "edge:TEMPORALLY_PRECEDED_BY": TEMPORALLY_PRECEDED_BY,
529
+ "edge:NEXT": NEXT,
530
+ }
531
+
532
+
533
+ def is_edge_predicate(p: str) -> bool:
534
+ """Is this a typed-edge predicate (edge:KIND)?"""
535
+ return p in _EDGE_PRED_TO_KIND
536
+
537
+
538
+ def edge_triples(mem, user_id: str | None = None) -> list[tuple]:
539
+ """Yield all typed edges as (src, edge:KIND, dst) triples.
540
+
541
+ Lets SPARQL queries like `?a edge:CAUSAL ?b` walk the typed-edge
542
+ graph natively. We materialize on demand — the edges table is
543
+ usually 10^4-10^5 rows so this is fast.
544
+ """
545
+ rows = mem.store.conn.execute(
546
+ "SELECT src, kind, dst FROM edges").fetchall()
547
+ out = []
548
+ for src, kind, dst in rows:
549
+ # reverse-map kind to SPARQL-facing predicate
550
+ for sp_pred, k in _EDGE_PRED_TO_KIND.items():
551
+ if k == kind:
552
+ out.append((src, sp_pred, dst))
553
+ break
554
+ return out
555
+
556
+
557
+ # ----------------------------------------------------------- executor
558
+ def execute_sparql(query: str, facts: list,
559
+ user_id: str | None = None,
560
+ *, edge_triples: list | None = None,
561
+ blob_resolver=None) -> dict:
562
+ """Execute a parsed SPARQL query against a list of facts.
563
+
564
+ facts: list of (subject, relation, value) triples from the facts table
565
+ edge_triples: optional list of (src, edge:KIND, dst) triples from the
566
+ edges table (for typed-edge queries). If None, edge:
567
+ predicates won't match anything.
568
+ blob_resolver: optional callable(subject, relation, value) -> str that
569
+ resolves long object values stored in the sidecar blob
570
+ arena. If a SPARQL FILTER references the object text
571
+ and the value is a chunk_id pointing into the arena,
572
+ the resolver dereferences on demand.
573
+
574
+ Returns SPARQL JSON Results format dict.
575
+ """
576
+ parsed = parse_sparql(query)
577
+ select_vars = parsed["select"]
578
+ patterns = parsed["patterns"]
579
+ optionals = parsed["optionals"]
580
+ filters = parsed["filters"]
581
+ order_by = parsed["order_by"]
582
+ limit = parsed["limit"]
583
+ offset = parsed["offset"] or 0
584
+
585
+ if not patterns and not optionals:
586
+ # empty WHERE — return all triples
587
+ patterns = [("?s", "?p", "?o")]
588
+ if not select_vars or select_vars == ["*"]:
589
+ select_vars = ["?s", "?p", "?o"]
590
+
591
+ # split patterns into fact-patterns vs edge-patterns so we
592
+ # search the right materialized triple list
593
+ fact_pats = [p for p in patterns if not is_edge_predicate(p[1])]
594
+ edge_pats = [p for p in patterns if is_edge_predicate(p[1])]
595
+ edge_pool = edge_triples or []
596
+
597
+ # naive nested-loop join: for each pattern, extend the binding set
598
+ bindings_list: list[dict] = [{}]
599
+ for pat in fact_pats:
600
+ new_list = []
601
+ for b in bindings_list:
602
+ for tri in facts:
603
+ nb = match_triple(tri, pat, b)
604
+ if nb is not None:
605
+ new_list.append(nb)
606
+ bindings_list = new_list
607
+ if not bindings_list:
608
+ break
609
+ # edge patterns join against the edges materialized view
610
+ for pat in edge_pats:
611
+ new_list = []
612
+ for b in bindings_list:
613
+ for tri in edge_pool:
614
+ nb = match_triple(tri, pat, b)
615
+ if nb is not None:
616
+ new_list.append(nb)
617
+ bindings_list = new_list
618
+ if not bindings_list:
619
+ break
620
+
621
+ # OPTIONAL: left-join each optional pattern group
622
+ for opt_pats in optionals:
623
+ new_list = []
624
+ for b in bindings_list:
625
+ extended = [b]
626
+ for op in opt_pats:
627
+ pool = edge_pool if is_edge_predicate(op[1]) else facts
628
+ next_ext = []
629
+ for bb in extended:
630
+ for tri in pool:
631
+ nb = match_triple(tri, op, bb)
632
+ if nb is not None:
633
+ next_ext.append(nb)
634
+ extended = next_ext if next_ext else [b] # left-join: keep b
635
+ # add all extended bindings (or original if no match)
636
+ new_list.extend(extended)
637
+ bindings_list = new_list
638
+
639
+ # apply filters (pass blob_resolver for ?_source / ?source_text)
640
+ if filters:
641
+ bindings_list = [b for b in bindings_list
642
+ if apply_filters(b, filters, blob_resolver)]
643
+
644
+ # blob resolution: if the SELECT projects ?_source / ?source_text
645
+ # AND a blob_resolver was provided, resolve the chunk text from
646
+ # the sidecar arena for each binding (using the stashed __source_id).
647
+ # This makes the SPARQL surface actually useful for arena-stored
648
+ # content — `SELECT ?s ?source_text WHERE { ?s ?p ?o }` returns
649
+ # the source text of the chunk each fact was extracted from.
650
+ if blob_resolver and select_vars:
651
+ source_vars = [v for v in select_vars
652
+ if v in ("?_source", "?source_text")]
653
+ if source_vars:
654
+ for b in bindings_list:
655
+ src_id = b.get("__source_id")
656
+ if src_id:
657
+ try:
658
+ txt = blob_resolver(None, None, src_id) or ""
659
+ except Exception:
660
+ txt = ""
661
+ for v in source_vars:
662
+ b[v] = txt
663
+
664
+ # ORDER BY
665
+ if order_by:
666
+ var = order_by["var"]
667
+ desc = order_by["desc"]
668
+ bindings_list.sort(
669
+ key=lambda b: (b.get(var) or ""),
670
+ reverse=desc)
671
+
672
+ # DISTINCT
673
+ if parsed["distinct"]:
674
+ seen: set = set()
675
+ deduped = []
676
+ for b in bindings_list:
677
+ key = tuple((v, b.get(v)) for v in select_vars)
678
+ if key not in seen:
679
+ seen.add(key)
680
+ deduped.append(b)
681
+ bindings_list = deduped
682
+
683
+ # OFFSET
684
+ if offset:
685
+ bindings_list = bindings_list[offset:]
686
+
687
+ # LIMIT
688
+ if limit is not None:
689
+ bindings_list = bindings_list[:limit]
690
+
691
+ # project to SELECT vars
692
+ rows = []
693
+ for b in bindings_list:
694
+ row = {v: b.get(v, "") for v in select_vars}
695
+ rows.append(row)
696
+
697
+ return {
698
+ "head": {"vars": select_vars},
699
+ "results": {"bindings": rows},
700
+ "n_results": len(rows),
701
+ }
702
+
703
+
704
+ # Maximum query string length (bytes) — protects against pathological
705
+ # SPARQL queries that would blow up the regex parser or nested-loop
706
+ # joiner. SPARQL queries in practice are <1KB; 64KB is generous.
707
+ MAX_QUERY_BYTES = 64 * 1024
708
+
709
+ # Maximum body size for SPARQL POST (smaller than REST MAX_BODY because
710
+ # SPARQL queries are text-only, no payloads to ingest)
711
+ MAX_BODY_BYTES = 256 * 1024 # 256 KiB
712
+
713
+
714
+ # ----------------------------------------------------------- HTTP server
715
+ def make_handler(mem, user_id: str | None = None,
716
+ *, auth_keys=None, require_auth: bool = False):
717
+ """Build a BaseHTTPRequestHandler bound to a Memory instance.
718
+
719
+ Auth: when `require_auth=True`, the handler validates a Bearer API
720
+ key against `auth_keys` (an APIKeyStore) and checks the
721
+ `sparql.query` permission. This is OFF by default for localhost
722
+ loopback use; the REST server turns it ON when co-hosting the
723
+ SPARQL endpoint so external graph tools go through the same RBAC
724
+ as the REST API.
725
+ """
726
+
727
+ class SparqlHandler(BaseHTTPRequestHandler):
728
+ def _send(self, code: int, body: str,
729
+ content_type: str = "application/json"):
730
+ data = body.encode("utf-8")
731
+ self.send_response(code)
732
+ self.send_header("Content-Type", content_type)
733
+ self.send_header("Content-Length", str(len(data)))
734
+ self.send_header("Access-Control-Allow-Origin", "*")
735
+ self.send_header("Access-Control-Allow-Methods",
736
+ "GET, POST, OPTIONS")
737
+ self.send_header("Access-Control-Allow-Headers",
738
+ "Content-Type, Authorization")
739
+ self.end_headers()
740
+ try:
741
+ self.wfile.write(data)
742
+ except (BrokenPipeError, ConnectionResetError):
743
+ pass
744
+
745
+ def do_OPTIONS(self):
746
+ # CORS preflight — return 204 with headers
747
+ self.send_response(204)
748
+ self.send_header("Access-Control-Allow-Origin", "*")
749
+ self.send_header("Access-Control-Allow-Methods",
750
+ "GET, POST, OPTIONS")
751
+ self.send_header("Access-Control-Allow-Headers",
752
+ "Content-Type, Authorization")
753
+ self.send_header("Content-Length", "0")
754
+ self.end_headers()
755
+
756
+ def _auth(self) -> tuple[dict, str] | None:
757
+ """Validate the Bearer token; return (meta, actor) or None.
758
+
759
+ On None, the caller has already sent a 401/403 response.
760
+ """
761
+ if not require_auth or auth_keys is None:
762
+ # open mode — used for loopback standalone SPARQL.
763
+ # CAUTION: only safe behind a firewall or 127.0.0.1.
764
+ return ({"role": "admin", "label": "open-sparql"},
765
+ "open-sparql")
766
+ hdr = self.headers.get("Authorization") or ""
767
+ if not hdr.startswith("Bearer "):
768
+ self._send(401, json.dumps({
769
+ "error": "SPARQL endpoint requires a Bearer API key "
770
+ "(use the same key as the REST API)"}))
771
+ return None
772
+ key = hdr[7:].strip()
773
+ meta = auth_keys.verify(key)
774
+ if meta is None:
775
+ self._send(401, json.dumps({"error": "invalid or revoked key"}))
776
+ return None
777
+ # check sparql.query permission
778
+ from cortexm.security.rbac import authorize, RBACError
779
+ try:
780
+ authorize(meta, "sparql.query")
781
+ except RBACError as e:
782
+ self._send(403, json.dumps({"error": str(e)}))
783
+ return None
784
+ actor = meta.get("label") or meta.get("id", "key")
785
+ return (meta, actor)
786
+
787
+ def do_GET(self):
788
+ parsed_url = urllib.parse.urlparse(self.path)
789
+ params = urllib.parse.parse_qs(parsed_url.query)
790
+ # URL length guard (414 URI Too Long)
791
+ if len(parsed_url.query) > MAX_QUERY_BYTES:
792
+ self._send(414, json.dumps({
793
+ "error": f"query string exceeds {MAX_QUERY_BYTES} bytes"}))
794
+ return
795
+ auth = self._auth()
796
+ if auth is None:
797
+ return
798
+ query = params.get("query", [""])[0]
799
+ self._handle_query(query, auth)
800
+
801
+ def do_POST(self):
802
+ length = int(self.headers.get("Content-Length", 0))
803
+ if length > MAX_BODY_BYTES:
804
+ self._send(413, json.dumps({
805
+ "error": f"body exceeds {MAX_BODY_BYTES} bytes"}))
806
+ return
807
+ body = self.rfile.read(length).decode("utf-8") if length else ""
808
+ content_type = self.headers.get("Content-Type", "")
809
+ if "x-www-form-urlencoded" in content_type:
810
+ params = urllib.parse.parse_qs(body)
811
+ query = params.get("query", [""])[0]
812
+ else:
813
+ query = body
814
+ auth = self._auth()
815
+ if auth is None:
816
+ return
817
+ self._handle_query(query, auth)
818
+
819
+ def _handle_query(self, query: str, auth: tuple[dict, str]):
820
+ meta, actor = auth
821
+ if not query:
822
+ self._send(400, json.dumps({
823
+ "error": "missing 'query' parameter — send "
824
+ "'?query=SELECT ...' or POST a SPARQL query"}))
825
+ return
826
+ if len(query) > MAX_QUERY_BYTES:
827
+ self._send(413, json.dumps({
828
+ "error": f"query exceeds {MAX_QUERY_BYTES} bytes"}))
829
+ return
830
+ try:
831
+ # pull facts + edges (scope to user_id if SPARQL endpoint
832
+ # was started with --sparql-user-id; otherwise all users)
833
+ fact_objs = mem.store.query_facts(active=True, user_id=user_id)
834
+ # build (s, p, o, source_id) 4-tuples so the blob resolver
835
+ # can dereference source text via fact.source_id — which is
836
+ # the actual chunk_id (NOT fact.value, which holds the
837
+ # literal). This fixes a v2 bug where the resolver never
838
+ # fired because chunk_ids never appeared in `value`.
839
+ facts = [(f.subject, f.relation, f.value, f.source_id)
840
+ for f in fact_objs]
841
+ edges = edge_triples(mem, user_id=user_id)
842
+ # build blob resolver if arena is enabled
843
+ blob_resolver = None
844
+ arena = getattr(mem, "blob_arena", None)
845
+ if arena is not None:
846
+ def _resolve(_s, _r, source_id):
847
+ if not source_id:
848
+ return ""
849
+ from cortexm.trace.blob_arena import get_chunk_text
850
+ return get_chunk_text(mem.store, arena, source_id)
851
+ blob_resolver = _resolve
852
+ result = execute_sparql(query, facts, user_id,
853
+ edge_triples=edges,
854
+ blob_resolver=blob_resolver)
855
+ # audit (if auth is on AND a memory audit_log exists)
856
+ if require_auth and getattr(mem, "audit_log", None):
857
+ mem.audit_log.log(
858
+ "sparql.query", actor=actor,
859
+ role=meta.get("role"),
860
+ meta={"n_results": result.get("n_results", 0),
861
+ "query_head": query[:80]})
862
+ self._send(200, json.dumps(result, indent=2, default=str))
863
+ except ValueError as e:
864
+ self._send(400, json.dumps({"error": str(e)}))
865
+ except Exception as e: # noqa: BLE001
866
+ self._send(500, json.dumps({"error": f"server: {e}"}))
867
+
868
+ def log_message(self, fmt, *args):
869
+ pass # quiet
870
+
871
+ return SparqlHandler
872
+
873
+
874
+ class SparqlServer:
875
+ """A minimal SPARQL HTTP endpoint backed by a Memory instance.
876
+
877
+ The Memory's reader is configured with the RDFDecoder so retrieval
878
+ produces RDF triples instead of an LLM prompt. This server exposes
879
+ those triples via SPARQL — a fully non-LLM retrieval path.
880
+
881
+ Threaded: uses ThreadingHTTPServer so concurrent queries don't
882
+ block each other. Each thread reads from the shared Memory (which
883
+ is itself guarded by an RLock at the store level).
884
+
885
+ Auth: when `require_auth=True` (the default when bound to non-
886
+ loopback interfaces), the server validates a Bearer API key
887
+ against the Memory's APIKeyStore and checks `sparql.query`
888
+ permission. When bound to 127.0.0.1 (default), auth is OFF for
889
+ local-dev convenience — but `--sparql-host 0.0.0.0` automatically
890
+ enables auth to prevent an unauthenticated fact dump.
891
+ """
892
+ def __init__(self, mem, host: str = "127.0.0.1", port: int = 8910,
893
+ user_id: str | None = None,
894
+ require_auth: bool | None = None) -> None:
895
+ self.mem = mem
896
+ self.host = host
897
+ self.port = port
898
+ self.user_id = user_id
899
+ # auth is auto-enabled whenever binding to non-loopback
900
+ if require_auth is None:
901
+ require_auth = host not in ("127.0.0.1", "localhost", "::1")
902
+ self.require_auth = require_auth
903
+ self._server: ThreadingHTTPServer | None = None
904
+ self._thread: threading.Thread | None = None
905
+
906
+ def start(self) -> None:
907
+ auth_keys = self.mem.keys if self.require_auth else None
908
+ handler_cls = make_handler(self.mem, user_id=self.user_id,
909
+ auth_keys=auth_keys,
910
+ require_auth=self.require_auth)
911
+ self._server = ThreadingHTTPServer((self.host, self.port),
912
+ handler_cls)
913
+ self._server.daemon_threads = True
914
+ print(f"[sparql] listening on http://{self.host}:{self.port}/"
915
+ f" (SPARQL 1.1 Protocol; thread-per-request)")
916
+ auth_msg = ("on (Bearer + sparql.query)" if self.require_auth
917
+ else "OFF (loopback only)")
918
+ print(f"[sparql] auth: {auth_msg}")
919
+ print(f"[sparql] user_id: {self.user_id or '(all users)'}")
920
+ print(f"[sparql] try: curl 'http://{self.host}:{self.port}/"
921
+ f"?query=SELECT%20%3Fs%20%3Fp%20%3Fo%20WHERE%20%7B%20%3Fs%20"
922
+ f"%3Fp%20%3Fo%20%7D%20LIMIT%2010'")
923
+
924
+ def start_background(self) -> None:
925
+ """Start the server in a daemon thread (non-blocking).
926
+
927
+ Used by `cortexm serve-rest --sparql-port N` to share one Memory
928
+ instance across both the REST API and the SPARQL endpoint.
929
+ """
930
+ self.start()
931
+ self._thread = threading.Thread(
932
+ target=self._server.serve_forever, daemon=True,
933
+ name=f"sparql-{self.port}")
934
+ self._thread.start()
935
+
936
+ def serve_forever(self) -> None:
937
+ if self._server is None:
938
+ self.start()
939
+ assert self._server is not None
940
+ self._server.serve_forever()
941
+
942
+ def stop(self) -> None:
943
+ if self._server is not None:
944
+ self._server.shutdown()
945
+ self._server.server_close()
946
+ self._server = None
947
+ if self._thread is not None:
948
+ self._thread.join(timeout=2)
949
+ self._thread = None
950
+
951
+
952
+ def main() -> int:
953
+ import argparse
954
+ ap = argparse.ArgumentParser(prog="contextm-sparql",
955
+ description="Context-M SPARQL endpoint")
956
+ ap.add_argument("--host", default="127.0.0.1")
957
+ ap.add_argument("--port", type=int, default=8910)
958
+ ap.add_argument("--db", default=None,
959
+ help="path to the Context-M DB (default: in-memory)")
960
+ ap.add_argument("--user-id", default=None,
961
+ help="scope queries to a single user")
962
+ args = ap.parse_args()
963
+
964
+ from cortexm.api.memory import Memory
965
+ from cortexm.config import Config
966
+ cfg = Config.from_env()
967
+ if args.db:
968
+ cfg.db_path = args.db
969
+ mem = Memory(cfg)
970
+ server = SparqlServer(mem, host=args.host, port=args.port,
971
+ user_id=args.user_id)
972
+ try:
973
+ server.serve_forever()
974
+ except KeyboardInterrupt:
975
+ print("\n[sparql] shutting down")
976
+ finally:
977
+ server.stop()
978
+ mem.close()
979
+ return 0
980
+
981
+
982
+ if __name__ == "__main__":
983
+ import sys
984
+ sys.exit(main())