cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
cortexm/server/sparql.py
ADDED
|
@@ -0,0 +1,984 @@
|
|
|
1
|
+
"""SPARQL endpoint — non-LLM decoder path (NSR-inspired, v2 extended).
|
|
2
|
+
|
|
3
|
+
arXiv insight (NSR / ESWEEK24): the VSA core is task-agnostic; only
|
|
4
|
+
the decoder changes. Context-M's reader was hardcoded to format facts
|
|
5
|
+
for an LLM prompt. The Decoders module extracts that formatter into
|
|
6
|
+
a pluggable interface — `RDFDecoder` exports facts as RDF/N3 triples.
|
|
7
|
+
|
|
8
|
+
This module exposes a SPARQL HTTP endpoint that:
|
|
9
|
+
1. accepts SPARQL SELECT queries via HTTP GET/POST
|
|
10
|
+
2. parses the query (a hand-rolled parser covering the common
|
|
11
|
+
subset: SELECT [DISTINCT] ?vars WHERE { triple-patterns +
|
|
12
|
+
FILTERs + OPTIONAL } ORDER BY ?var [ASC|DESC] LIMIT N)
|
|
13
|
+
3. retrieves facts from Context-M's Memory.store (or its
|
|
14
|
+
RDFDecoder if attached — same substrate either way)
|
|
15
|
+
4. runs the WHERE clause as a join pipeline over those triples
|
|
16
|
+
— naive nested-loop join with binding propagation, OPTIONAL
|
|
17
|
+
handled via LEFT-JOIN semantics, FILTER applied per binding
|
|
18
|
+
5. resolves blob-arena-stored objects on demand (the sidecar
|
|
19
|
+
blob arena is the Aeon off-graph store; SPARQL "o" may
|
|
20
|
+
transparently dereference long text via Memory.get_chunk_text)
|
|
21
|
+
6. exposes CAUSAL / REFERS_TO typed edges (Aeon) via the
|
|
22
|
+
`edge/2` predicate family so external graph tools can walk
|
|
23
|
+
the truth-maintenance and episodic-atlas graphs natively
|
|
24
|
+
7. returns the result as SPARQL 1.1 JSON Results
|
|
25
|
+
|
|
26
|
+
This is a NON-LLM retrieval path. Zero LLM calls. The same palace +
|
|
27
|
+
Trace substrate that powers LLM context-stuffing now serves SPARQL.
|
|
28
|
+
|
|
29
|
+
Usage:
|
|
30
|
+
# standalone SPARQL endpoint (e.g. on port 8910)
|
|
31
|
+
python -m cortexm.server.sparql --port 8910
|
|
32
|
+
|
|
33
|
+
# launched alongside the REST API (recommended for production):
|
|
34
|
+
# `cortexm serve-rest --sparql-port 8910` — both share one Memory
|
|
35
|
+
# instance; the REST API also exposes /v1/sparql for unified access.
|
|
36
|
+
|
|
37
|
+
Example queries (v2 supports a much richer subset than v1):
|
|
38
|
+
SELECT ?s ?p ?o WHERE { ?s ?p ?o } LIMIT 10
|
|
39
|
+
SELECT DISTINCT ?s WHERE { ?s ?p ?o } ORDER BY ?s
|
|
40
|
+
SELECT ?s ?o WHERE { ?s "name" ?o . FILTER regex(?o, "^Jen", "i") }
|
|
41
|
+
SELECT ?cause ?effect WHERE {
|
|
42
|
+
?cause edge:CAUSAL ?effect .
|
|
43
|
+
?effect "name" "Alice"
|
|
44
|
+
} LIMIT 5
|
|
45
|
+
|
|
46
|
+
LIMITATIONS (honest, documented):
|
|
47
|
+
- Parser covers the SELECT subset named above. UNION, CONSTRUCT,
|
|
48
|
+
ASK, DESCRIBE, property paths (rdf:type/rdfs:subClassOf*), and
|
|
49
|
+
SPARQL 1.1 aggregates (GROUP BY / COUNT / SUM) are NOT yet
|
|
50
|
+
supported. For those, export via RDFDecoder + use Apache Jena.
|
|
51
|
+
- No inference / reasoning over RDFS/OWL — pattern matching on
|
|
52
|
+
the actual stored facts only.
|
|
53
|
+
- Joins are nested-loop (no query planner). Adequate for the
|
|
54
|
+
~10^5 facts Context-M is designed to hold per user; for larger
|
|
55
|
+
stores, use a real triple store fed via the RDFDecoder export.
|
|
56
|
+
"""
|
|
57
|
+
from __future__ import annotations
|
|
58
|
+
|
|
59
|
+
import json
|
|
60
|
+
import re
|
|
61
|
+
import threading
|
|
62
|
+
import urllib.parse
|
|
63
|
+
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
64
|
+
from typing import Iterable
|
|
65
|
+
|
|
66
|
+
# Typed-edge vocabulary (Aeon) — used by SPARQL `edge:KIND` predicates.
|
|
67
|
+
# These are the canonical edge kinds actually WRITTEN by writer.py +
|
|
68
|
+
# consolidate.py. Note: supersession is captured via CONTRADICTS edges
|
|
69
|
+
# (writer writes CONTRADICTS with meta indicating kind), so there is no
|
|
70
|
+
# standalone SUPERSEDS constant exported from edges.py.
|
|
71
|
+
from cortexm.trace.edges import (
|
|
72
|
+
CAUSAL, REFERS_TO, CONTRADICTS, MERGED_WITH,
|
|
73
|
+
EXTRACTED_FROM, TEMPORALLY_PRECEDED_BY, NEXT,
|
|
74
|
+
)
|
|
75
|
+
# SPARQL-facing aliases: edge:SUPERSEDES resolves to the actual
|
|
76
|
+
# CONTRADICTS replacement edge. This keeps the user-facing vocabulary
|
|
77
|
+
# stable even though internally we collapse SUPERSEDE → CONTRADICTS.
|
|
78
|
+
SUPERSEDS = CONTRADICTS # alias for `edge:SUPERSEDES` queries
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
# ----------------------------------------------------------- parser
|
|
82
|
+
def parse_sparql(query: str) -> dict:
|
|
83
|
+
"""Parse a SPARQL SELECT query.
|
|
84
|
+
|
|
85
|
+
Supports (v2, extended):
|
|
86
|
+
SELECT [DISTINCT] ?var+ WHERE {
|
|
87
|
+
triple-pattern ( ?s ?p ?o . | ?s "literal" ?o . | ... )
|
|
88
|
+
FILTER ( regex(?v, "pat", "i") | ?v = "lit" | ?v != "lit" )
|
|
89
|
+
OPTIONAL { ... } # left-join semantics
|
|
90
|
+
}
|
|
91
|
+
[ORDER BY ?var [ASC|DESC]]
|
|
92
|
+
[LIMIT N]
|
|
93
|
+
[OFFSET N]
|
|
94
|
+
|
|
95
|
+
Typed-edge shortcut: `edge:KIND` is recognized as a predicate
|
|
96
|
+
where KIND is one of the canonical edge kinds (CAUSAL, REFERS_TO,
|
|
97
|
+
SUPERSEDES, CONTRADICTS, MERGED_WITH, EXTRACTED_FROM,
|
|
98
|
+
TEMPORALLY_PRECEDED_BY, NEXT). The triple (?a edge:CAUSAL ?b)
|
|
99
|
+
queries the Trace's edges table instead of the facts table.
|
|
100
|
+
|
|
101
|
+
Returns:
|
|
102
|
+
{
|
|
103
|
+
"distinct": bool,
|
|
104
|
+
"select": ["?s", "?p", "?o"],
|
|
105
|
+
"patterns": [("?s", "?p", "?o"), ...],
|
|
106
|
+
"optionals":[[pat, ...], ...], # one list per OPTIONAL block
|
|
107
|
+
"filters": [{"type": "regex"/"equals"/"ne", ...}, ...],
|
|
108
|
+
"order_by": {"var": "?s", "desc": False} | None,
|
|
109
|
+
"limit": int | None,
|
|
110
|
+
"offset": int | None,
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
Raises ValueError on unparseable input.
|
|
114
|
+
"""
|
|
115
|
+
query = query.strip()
|
|
116
|
+
# strip trailing semicolons / whitespace
|
|
117
|
+
while query.endswith(";"):
|
|
118
|
+
query = query[:-1].strip()
|
|
119
|
+
|
|
120
|
+
# match: SELECT [DISTINCT] <select-vars> WHERE { <where> }
|
|
121
|
+
# select-vars may be one or more ?var tokens, or *
|
|
122
|
+
sel_grp = ""
|
|
123
|
+
where_clause = ""
|
|
124
|
+
tail = ""
|
|
125
|
+
distinct_grp = ""
|
|
126
|
+
matched = False
|
|
127
|
+
for pat in (
|
|
128
|
+
# canonical form with WHERE keyword
|
|
129
|
+
r"SELECT\s+(DISTINCT\s+)?((?:\?\S+(?:\s+|$))+|\*\s*?)\s*WHERE\s*\{(.*)\}\s*(.*)$",
|
|
130
|
+
# without WHERE keyword (some demos)
|
|
131
|
+
r"SELECT\s+(DISTINCT\s+)?((?:\?\S+(?:\s+|$))+|\*\s*?)\s*\{(.*)\}\s*(.*)$",
|
|
132
|
+
):
|
|
133
|
+
m = re.match(pat, query, re.IGNORECASE | re.DOTALL)
|
|
134
|
+
if m:
|
|
135
|
+
distinct_grp = m.group(1) or ""
|
|
136
|
+
sel_grp = m.group(2) or ""
|
|
137
|
+
where_clause = m.group(3) or ""
|
|
138
|
+
tail = m.group(4) or ""
|
|
139
|
+
matched = True
|
|
140
|
+
break
|
|
141
|
+
if not matched:
|
|
142
|
+
raise ValueError(
|
|
143
|
+
"query must be 'SELECT [DISTINCT] ?vars WHERE { ... }' "
|
|
144
|
+
"— more complex SPARQL not yet supported in v2")
|
|
145
|
+
|
|
146
|
+
# parse SELECT vars (may be multiple ?vars separated by spaces, or *)
|
|
147
|
+
is_distinct = bool(distinct_grp)
|
|
148
|
+
if sel_grp.strip() == "*":
|
|
149
|
+
select_vars: list[str] = ["?s", "?p", "?o"]
|
|
150
|
+
else:
|
|
151
|
+
# find all ?vars in the select clause
|
|
152
|
+
select_vars = re.findall(r"\?\w+", sel_grp)
|
|
153
|
+
if not select_vars:
|
|
154
|
+
raise ValueError("SELECT must list at least one variable")
|
|
155
|
+
|
|
156
|
+
# parse tail (ORDER BY / LIMIT / OFFSET — order-insensitive)
|
|
157
|
+
order_by, limit, offset = _parse_tail(tail + " " + where_clause[-0:])
|
|
158
|
+
# NOTE: the WHERE clause may also have a trailing LIMIT etc — we
|
|
159
|
+
# extract the tail BEFORE the where clause's trailing }. The above
|
|
160
|
+
# regex captures tail as everything AFTER the closing }. Good.
|
|
161
|
+
|
|
162
|
+
# parse WHERE: triple patterns, FILTER, OPTIONAL
|
|
163
|
+
where_clause = where_clause.strip()
|
|
164
|
+
patterns: list[tuple[str, str, str]] = []
|
|
165
|
+
filters: list[dict] = []
|
|
166
|
+
optionals: list[list[tuple[str, str, str]]] = []
|
|
167
|
+
|
|
168
|
+
# tokenize while respecting OPTIONAL { ... } nesting, FILTER(...),
|
|
169
|
+
# and quoted strings. Then merge 'FILTER' + the following
|
|
170
|
+
# parenthesized token so _parse_filter sees the full clause.
|
|
171
|
+
tokens = _merge_filter_tokens(_tokenize_where(where_clause))
|
|
172
|
+
|
|
173
|
+
cur_optional: list[tuple[str, str, str]] | None = None
|
|
174
|
+
i = 0
|
|
175
|
+
while i < len(tokens):
|
|
176
|
+
tok = tokens[i].strip()
|
|
177
|
+
if not tok:
|
|
178
|
+
i += 1
|
|
179
|
+
continue
|
|
180
|
+
upper = tok.upper()
|
|
181
|
+
if upper.startswith("OPTIONAL"):
|
|
182
|
+
# expect '{' next
|
|
183
|
+
i += 1
|
|
184
|
+
if i >= len(tokens) or tokens[i].strip() != "{":
|
|
185
|
+
raise ValueError("OPTIONAL must be followed by '{'")
|
|
186
|
+
# gather until matching '}'
|
|
187
|
+
depth = 1
|
|
188
|
+
inner: list[str] = []
|
|
189
|
+
i += 1
|
|
190
|
+
while i < len(tokens) and depth > 0:
|
|
191
|
+
t = tokens[i].strip()
|
|
192
|
+
if t == "{":
|
|
193
|
+
depth += 1
|
|
194
|
+
inner.append(t)
|
|
195
|
+
elif t == "}":
|
|
196
|
+
depth -= 1
|
|
197
|
+
if depth > 0:
|
|
198
|
+
inner.append(t)
|
|
199
|
+
else:
|
|
200
|
+
inner.append(t)
|
|
201
|
+
i += 1
|
|
202
|
+
# parse inner as a sub-clause (patterns + filters)
|
|
203
|
+
sub_pats, sub_filts = _parse_triples(inner)
|
|
204
|
+
# v2 stores optional patterns separately; filters inside
|
|
205
|
+
# OPTIONAL are also applied within the left-join
|
|
206
|
+
optionals.append(sub_pats)
|
|
207
|
+
filters.extend(sub_filts) # FILTERs are hoisted — applied later
|
|
208
|
+
cur_optional = None
|
|
209
|
+
continue
|
|
210
|
+
if upper.startswith("FILTER"):
|
|
211
|
+
filt = _parse_filter(tok)
|
|
212
|
+
if filt:
|
|
213
|
+
filters.append(filt)
|
|
214
|
+
i += 1
|
|
215
|
+
continue
|
|
216
|
+
# else: triple-pattern token. Look at groups of 3.
|
|
217
|
+
# Easier: re-tokenize by '.' but keep quoted strings together
|
|
218
|
+
# _tokenize_where already split on whitespace, so triple tokens
|
|
219
|
+
# come as a stream. Group until we have 3 non-'.' tokens.
|
|
220
|
+
# Skip '.' separator.
|
|
221
|
+
if tok == ".":
|
|
222
|
+
i += 1
|
|
223
|
+
continue
|
|
224
|
+
# collect 3 terms for a triple
|
|
225
|
+
terms = [tok]
|
|
226
|
+
j = i + 1
|
|
227
|
+
while j < len(tokens) and len(terms) < 3:
|
|
228
|
+
t = tokens[j].strip()
|
|
229
|
+
if t == ".":
|
|
230
|
+
j += 1
|
|
231
|
+
continue
|
|
232
|
+
if t.upper().startswith("FILTER") or t.upper().startswith(
|
|
233
|
+
"OPTIONAL") or t == "}":
|
|
234
|
+
break
|
|
235
|
+
terms.append(t)
|
|
236
|
+
j += 1
|
|
237
|
+
if len(terms) == 3:
|
|
238
|
+
patterns.append(tuple(terms))
|
|
239
|
+
i = j
|
|
240
|
+
|
|
241
|
+
return {
|
|
242
|
+
"distinct": is_distinct,
|
|
243
|
+
"select": select_vars,
|
|
244
|
+
"patterns": patterns,
|
|
245
|
+
"optionals": optionals,
|
|
246
|
+
"filters": filters,
|
|
247
|
+
"order_by": order_by,
|
|
248
|
+
"limit": limit,
|
|
249
|
+
"offset": offset,
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _parse_tail(tail: str) -> tuple[dict | None, int | None, int | None]:
|
|
254
|
+
"""Extract ORDER BY / LIMIT / OFFSET from the post-WHERE tail."""
|
|
255
|
+
order_by = None
|
|
256
|
+
limit = None
|
|
257
|
+
offset = None
|
|
258
|
+
|
|
259
|
+
# ORDER BY ?var [ASC|DESC]
|
|
260
|
+
m = re.search(
|
|
261
|
+
r"ORDER\s+BY\s+(\?\w+)(?:\s+(ASC|DESC))?",
|
|
262
|
+
tail, re.IGNORECASE)
|
|
263
|
+
if m:
|
|
264
|
+
order_by = {"var": m.group(1),
|
|
265
|
+
"desc": (m.group(2) or "").upper() == "DESC"}
|
|
266
|
+
|
|
267
|
+
# LIMIT N
|
|
268
|
+
m = re.search(r"LIMIT\s+(\d+)", tail, re.IGNORECASE)
|
|
269
|
+
if m:
|
|
270
|
+
limit = int(m.group(1))
|
|
271
|
+
|
|
272
|
+
# OFFSET N
|
|
273
|
+
m = re.search(r"OFFSET\s+(\d+)", tail, re.IGNORECASE)
|
|
274
|
+
if m:
|
|
275
|
+
offset = int(m.group(1))
|
|
276
|
+
|
|
277
|
+
return order_by, limit, offset
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def _tokenize_where(s: str) -> list[str]:
|
|
281
|
+
"""Split WHERE body into tokens, preserving quoted strings + parens."""
|
|
282
|
+
tokens: list[str] = []
|
|
283
|
+
cur = ""
|
|
284
|
+
in_string = False
|
|
285
|
+
quote_char = None
|
|
286
|
+
depth = 0
|
|
287
|
+
for ch in s:
|
|
288
|
+
if in_string:
|
|
289
|
+
cur += ch
|
|
290
|
+
if ch == quote_char:
|
|
291
|
+
in_string = False
|
|
292
|
+
continue
|
|
293
|
+
if ch in ('"', "'"):
|
|
294
|
+
in_string = True
|
|
295
|
+
quote_char = ch
|
|
296
|
+
cur += ch
|
|
297
|
+
continue
|
|
298
|
+
if ch == "(":
|
|
299
|
+
depth += 1
|
|
300
|
+
cur += ch
|
|
301
|
+
continue
|
|
302
|
+
if ch == ")":
|
|
303
|
+
depth -= 1
|
|
304
|
+
cur += ch
|
|
305
|
+
continue
|
|
306
|
+
if ch == "{":
|
|
307
|
+
if cur.strip():
|
|
308
|
+
tokens.append(cur.strip())
|
|
309
|
+
cur = ""
|
|
310
|
+
tokens.append("{")
|
|
311
|
+
continue
|
|
312
|
+
if ch == "}":
|
|
313
|
+
if cur.strip():
|
|
314
|
+
tokens.append(cur.strip())
|
|
315
|
+
cur = ""
|
|
316
|
+
tokens.append("}")
|
|
317
|
+
continue
|
|
318
|
+
if depth > 0:
|
|
319
|
+
cur += ch
|
|
320
|
+
continue
|
|
321
|
+
if ch.isspace():
|
|
322
|
+
if cur.strip():
|
|
323
|
+
tokens.append(cur.strip())
|
|
324
|
+
cur = ""
|
|
325
|
+
continue
|
|
326
|
+
cur += ch
|
|
327
|
+
if cur.strip():
|
|
328
|
+
tokens.append(cur.strip())
|
|
329
|
+
return tokens
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def _parse_triples(tokens: list[str]) -> tuple[list[tuple], list[dict]]:
|
|
333
|
+
"""Parse a flat list of triple tokens + FILTERs into (pats, filts)."""
|
|
334
|
+
tokens = _merge_filter_tokens(tokens)
|
|
335
|
+
pats: list[tuple] = []
|
|
336
|
+
filts: list[dict] = []
|
|
337
|
+
i = 0
|
|
338
|
+
while i < len(tokens):
|
|
339
|
+
tok = tokens[i].strip()
|
|
340
|
+
if not tok:
|
|
341
|
+
i += 1
|
|
342
|
+
continue
|
|
343
|
+
if tok.upper().startswith("FILTER"):
|
|
344
|
+
f = _parse_filter(tok)
|
|
345
|
+
if f:
|
|
346
|
+
filts.append(f)
|
|
347
|
+
i += 1
|
|
348
|
+
continue
|
|
349
|
+
if tok == ".":
|
|
350
|
+
i += 1
|
|
351
|
+
continue
|
|
352
|
+
terms = [tok]
|
|
353
|
+
j = i + 1
|
|
354
|
+
while j < len(tokens) and len(terms) < 3:
|
|
355
|
+
t = tokens[j].strip()
|
|
356
|
+
if t == ".":
|
|
357
|
+
j += 1
|
|
358
|
+
continue
|
|
359
|
+
if t.upper().startswith("FILTER") or t == "}":
|
|
360
|
+
break
|
|
361
|
+
terms.append(t)
|
|
362
|
+
j += 1
|
|
363
|
+
if len(terms) == 3:
|
|
364
|
+
pats.append(tuple(terms))
|
|
365
|
+
i = j
|
|
366
|
+
return pats, filts
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
def _parse_filter(tok: str) -> dict | None:
|
|
370
|
+
"""Parse one FILTER(...) clause into a filter dict.
|
|
371
|
+
|
|
372
|
+
`tok` may be:
|
|
373
|
+
- 'FILTER regex(?v, "pat", "flags")' (merged by tokenizer)
|
|
374
|
+
- 'FILTER(?v = "value")'
|
|
375
|
+
- 'FILTER(?v != "value")'
|
|
376
|
+
- 'FILTER' alone (in which case caller should merge next token)
|
|
377
|
+
"""
|
|
378
|
+
# FILTER regex(?v, "pat", "flags")
|
|
379
|
+
m = re.match(
|
|
380
|
+
r"FILTER\s+regex\s*\(\s*(\?\w+)\s*,\s*[\"'](.+?)[\"']\s*"
|
|
381
|
+
r"(?:,\s*[\"'](\w+)[\"']\s*)?\)",
|
|
382
|
+
tok, re.IGNORECASE | re.DOTALL)
|
|
383
|
+
if m:
|
|
384
|
+
return {"type": "regex", "var": m.group(1),
|
|
385
|
+
"pattern": m.group(2), "flags": m.group(3) or ""}
|
|
386
|
+
# FILTER regex without space: FILTERregex(...) — unlikely but tolerated
|
|
387
|
+
m = re.match(
|
|
388
|
+
r"FILTERregex\s*\(\s*(\?\w+)\s*,\s*[\"'](.+?)[\"']\s*"
|
|
389
|
+
r"(?:,\s*[\"'](\w+)[\"']\s*)?\)",
|
|
390
|
+
tok, re.IGNORECASE | re.DOTALL)
|
|
391
|
+
if m:
|
|
392
|
+
return {"type": "regex", "var": m.group(1),
|
|
393
|
+
"pattern": m.group(2), "flags": m.group(3) or ""}
|
|
394
|
+
# FILTER(?v = "value")
|
|
395
|
+
m = re.match(
|
|
396
|
+
r"FILTER\s*\(\s*(\?\w+)\s*=\s*[\"'](.+?)[\"']\s*\)",
|
|
397
|
+
tok, re.IGNORECASE | re.DOTALL)
|
|
398
|
+
if m:
|
|
399
|
+
return {"type": "equals", "var": m.group(1), "value": m.group(2)}
|
|
400
|
+
# FILTER(?v != "value")
|
|
401
|
+
m = re.match(
|
|
402
|
+
r"FILTER\s*\(\s*(\?\w+)\s*!=\s*[\"'](.+?)[\"']\s*\)",
|
|
403
|
+
tok, re.IGNORECASE | re.DOTALL)
|
|
404
|
+
if m:
|
|
405
|
+
return {"type": "ne", "var": m.group(1), "value": m.group(2)}
|
|
406
|
+
return None
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def _merge_filter_tokens(tokens: list[str]) -> list[str]:
|
|
410
|
+
"""If 'FILTER' appears as a standalone token, merge it with the
|
|
411
|
+
following parenthesized token so _parse_filter sees the full
|
|
412
|
+
'FILTER regex(...)' or 'FILTER(...)' string.
|
|
413
|
+
"""
|
|
414
|
+
out: list[str] = []
|
|
415
|
+
i = 0
|
|
416
|
+
while i < len(tokens):
|
|
417
|
+
t = tokens[i]
|
|
418
|
+
if t.upper() == "FILTER" and i + 1 < len(tokens):
|
|
419
|
+
merged = t + " " + tokens[i + 1]
|
|
420
|
+
out.append(merged)
|
|
421
|
+
i += 2
|
|
422
|
+
continue
|
|
423
|
+
out.append(t)
|
|
424
|
+
i += 1
|
|
425
|
+
return out
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
# ----------------------------------------------------------- matcher
|
|
429
|
+
def match_triple(triple: tuple, pattern: tuple,
|
|
430
|
+
bindings: dict | None = None) -> dict | None:
|
|
431
|
+
"""Try to match a triple against a pattern, returning extended
|
|
432
|
+
bindings or None if no match.
|
|
433
|
+
|
|
434
|
+
triple: (subject, relation, value) or (subject, relation, value,
|
|
435
|
+
source_id) from Context-M facts. If a 4-tuple is passed,
|
|
436
|
+
the source_id is stashed under the synthetic `__source_id`
|
|
437
|
+
key (not projectable) so the executor can resolve arena
|
|
438
|
+
text on demand.
|
|
439
|
+
pattern: ("?s", "?p", "?o") or with literals like
|
|
440
|
+
("?s", "name", "?o")
|
|
441
|
+
"""
|
|
442
|
+
if len(triple) not in (3, 4) or len(pattern) != 3:
|
|
443
|
+
return None
|
|
444
|
+
b = dict(bindings or {})
|
|
445
|
+
# if the triple has a source_id, stash it for the executor
|
|
446
|
+
if len(triple) == 4:
|
|
447
|
+
b["__source_id"] = triple[3]
|
|
448
|
+
for tri_elem, pat_elem in zip(triple[:3], pattern):
|
|
449
|
+
if pat_elem.startswith("?"):
|
|
450
|
+
if pat_elem in b:
|
|
451
|
+
if b[pat_elem] != tri_elem:
|
|
452
|
+
return None
|
|
453
|
+
else:
|
|
454
|
+
b[pat_elem] = tri_elem
|
|
455
|
+
else:
|
|
456
|
+
# literal: strip quotes
|
|
457
|
+
literal = pat_elem.strip('"\'')
|
|
458
|
+
if literal != tri_elem:
|
|
459
|
+
return None
|
|
460
|
+
return b
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
def apply_filters(bindings: dict, filters: list[dict],
|
|
464
|
+
blob_resolver=None) -> bool:
|
|
465
|
+
"""Return True if `bindings` satisfies all `filters`.
|
|
466
|
+
|
|
467
|
+
If `blob_resolver` is provided and a filter targets a variable
|
|
468
|
+
whose value is the synthetic `__source_id` (a chunk_id pointing
|
|
469
|
+
into the sidecar blob arena), we resolve the source text on
|
|
470
|
+
demand and apply the filter against the resolved text. This makes
|
|
471
|
+
`FILTER regex(?src, "pat")` work when ?src is bound to a chunk_id
|
|
472
|
+
via a special `?_source` projection variable.
|
|
473
|
+
"""
|
|
474
|
+
for f in filters:
|
|
475
|
+
var = f["var"]
|
|
476
|
+
if var not in bindings:
|
|
477
|
+
# special case: ?_source / ?source_text refers to the
|
|
478
|
+
# arena-resolved chunk text; we resolve lazily here.
|
|
479
|
+
if var in ("?_source", "?source_text") and blob_resolver:
|
|
480
|
+
src_id = bindings.get("__source_id")
|
|
481
|
+
if not src_id:
|
|
482
|
+
return False
|
|
483
|
+
try:
|
|
484
|
+
val = blob_resolver(None, None, src_id) or ""
|
|
485
|
+
except Exception:
|
|
486
|
+
val = ""
|
|
487
|
+
if f["type"] == "regex":
|
|
488
|
+
flags = 0
|
|
489
|
+
if "i" in f.get("flags", ""):
|
|
490
|
+
flags |= re.IGNORECASE
|
|
491
|
+
if not re.search(f["pattern"], val, flags):
|
|
492
|
+
return False
|
|
493
|
+
elif f["type"] == "equals":
|
|
494
|
+
if val != f["value"]:
|
|
495
|
+
return False
|
|
496
|
+
elif f["type"] == "ne":
|
|
497
|
+
if val == f["value"]:
|
|
498
|
+
return False
|
|
499
|
+
continue
|
|
500
|
+
return False
|
|
501
|
+
val = bindings[var]
|
|
502
|
+
if f["type"] == "regex":
|
|
503
|
+
flags = 0
|
|
504
|
+
if "i" in f.get("flags", ""):
|
|
505
|
+
flags |= re.IGNORECASE
|
|
506
|
+
if not re.search(f["pattern"], val, flags):
|
|
507
|
+
return False
|
|
508
|
+
elif f["type"] == "equals":
|
|
509
|
+
if val != f["value"]:
|
|
510
|
+
return False
|
|
511
|
+
elif f["type"] == "ne":
|
|
512
|
+
if val == f["value"]:
|
|
513
|
+
return False
|
|
514
|
+
return True
|
|
515
|
+
|
|
516
|
+
|
|
517
|
+
# ----------------------------------------------------------- edge triples
|
|
518
|
+
# Mapping from SPARQL `edge:KIND` predicates to the actual edge kinds
|
|
519
|
+
# stored in the Trace. External graph tools get a stable SPARQL-facing
|
|
520
|
+
# vocabulary that maps to Aeon's typed-edge graph internally.
|
|
521
|
+
_EDGE_PRED_TO_KIND = {
|
|
522
|
+
"edge:CAUSAL": CAUSAL,
|
|
523
|
+
"edge:REFERS_TO": REFERS_TO,
|
|
524
|
+
"edge:SUPERSEDES": SUPERSEDS,
|
|
525
|
+
"edge:CONTRADICTS": CONTRADICTS,
|
|
526
|
+
"edge:MERGED_WITH": MERGED_WITH,
|
|
527
|
+
"edge:EXTRACTED_FROM": EXTRACTED_FROM,
|
|
528
|
+
"edge:TEMPORALLY_PRECEDED_BY": TEMPORALLY_PRECEDED_BY,
|
|
529
|
+
"edge:NEXT": NEXT,
|
|
530
|
+
}
|
|
531
|
+
|
|
532
|
+
|
|
533
|
+
def is_edge_predicate(p: str) -> bool:
|
|
534
|
+
"""Is this a typed-edge predicate (edge:KIND)?"""
|
|
535
|
+
return p in _EDGE_PRED_TO_KIND
|
|
536
|
+
|
|
537
|
+
|
|
538
|
+
def edge_triples(mem, user_id: str | None = None) -> list[tuple]:
|
|
539
|
+
"""Yield all typed edges as (src, edge:KIND, dst) triples.
|
|
540
|
+
|
|
541
|
+
Lets SPARQL queries like `?a edge:CAUSAL ?b` walk the typed-edge
|
|
542
|
+
graph natively. We materialize on demand — the edges table is
|
|
543
|
+
usually 10^4-10^5 rows so this is fast.
|
|
544
|
+
"""
|
|
545
|
+
rows = mem.store.conn.execute(
|
|
546
|
+
"SELECT src, kind, dst FROM edges").fetchall()
|
|
547
|
+
out = []
|
|
548
|
+
for src, kind, dst in rows:
|
|
549
|
+
# reverse-map kind to SPARQL-facing predicate
|
|
550
|
+
for sp_pred, k in _EDGE_PRED_TO_KIND.items():
|
|
551
|
+
if k == kind:
|
|
552
|
+
out.append((src, sp_pred, dst))
|
|
553
|
+
break
|
|
554
|
+
return out
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
# ----------------------------------------------------------- executor
|
|
558
|
+
def execute_sparql(query: str, facts: list,
|
|
559
|
+
user_id: str | None = None,
|
|
560
|
+
*, edge_triples: list | None = None,
|
|
561
|
+
blob_resolver=None) -> dict:
|
|
562
|
+
"""Execute a parsed SPARQL query against a list of facts.
|
|
563
|
+
|
|
564
|
+
facts: list of (subject, relation, value) triples from the facts table
|
|
565
|
+
edge_triples: optional list of (src, edge:KIND, dst) triples from the
|
|
566
|
+
edges table (for typed-edge queries). If None, edge:
|
|
567
|
+
predicates won't match anything.
|
|
568
|
+
blob_resolver: optional callable(subject, relation, value) -> str that
|
|
569
|
+
resolves long object values stored in the sidecar blob
|
|
570
|
+
arena. If a SPARQL FILTER references the object text
|
|
571
|
+
and the value is a chunk_id pointing into the arena,
|
|
572
|
+
the resolver dereferences on demand.
|
|
573
|
+
|
|
574
|
+
Returns SPARQL JSON Results format dict.
|
|
575
|
+
"""
|
|
576
|
+
parsed = parse_sparql(query)
|
|
577
|
+
select_vars = parsed["select"]
|
|
578
|
+
patterns = parsed["patterns"]
|
|
579
|
+
optionals = parsed["optionals"]
|
|
580
|
+
filters = parsed["filters"]
|
|
581
|
+
order_by = parsed["order_by"]
|
|
582
|
+
limit = parsed["limit"]
|
|
583
|
+
offset = parsed["offset"] or 0
|
|
584
|
+
|
|
585
|
+
if not patterns and not optionals:
|
|
586
|
+
# empty WHERE — return all triples
|
|
587
|
+
patterns = [("?s", "?p", "?o")]
|
|
588
|
+
if not select_vars or select_vars == ["*"]:
|
|
589
|
+
select_vars = ["?s", "?p", "?o"]
|
|
590
|
+
|
|
591
|
+
# split patterns into fact-patterns vs edge-patterns so we
|
|
592
|
+
# search the right materialized triple list
|
|
593
|
+
fact_pats = [p for p in patterns if not is_edge_predicate(p[1])]
|
|
594
|
+
edge_pats = [p for p in patterns if is_edge_predicate(p[1])]
|
|
595
|
+
edge_pool = edge_triples or []
|
|
596
|
+
|
|
597
|
+
# naive nested-loop join: for each pattern, extend the binding set
|
|
598
|
+
bindings_list: list[dict] = [{}]
|
|
599
|
+
for pat in fact_pats:
|
|
600
|
+
new_list = []
|
|
601
|
+
for b in bindings_list:
|
|
602
|
+
for tri in facts:
|
|
603
|
+
nb = match_triple(tri, pat, b)
|
|
604
|
+
if nb is not None:
|
|
605
|
+
new_list.append(nb)
|
|
606
|
+
bindings_list = new_list
|
|
607
|
+
if not bindings_list:
|
|
608
|
+
break
|
|
609
|
+
# edge patterns join against the edges materialized view
|
|
610
|
+
for pat in edge_pats:
|
|
611
|
+
new_list = []
|
|
612
|
+
for b in bindings_list:
|
|
613
|
+
for tri in edge_pool:
|
|
614
|
+
nb = match_triple(tri, pat, b)
|
|
615
|
+
if nb is not None:
|
|
616
|
+
new_list.append(nb)
|
|
617
|
+
bindings_list = new_list
|
|
618
|
+
if not bindings_list:
|
|
619
|
+
break
|
|
620
|
+
|
|
621
|
+
# OPTIONAL: left-join each optional pattern group
|
|
622
|
+
for opt_pats in optionals:
|
|
623
|
+
new_list = []
|
|
624
|
+
for b in bindings_list:
|
|
625
|
+
extended = [b]
|
|
626
|
+
for op in opt_pats:
|
|
627
|
+
pool = edge_pool if is_edge_predicate(op[1]) else facts
|
|
628
|
+
next_ext = []
|
|
629
|
+
for bb in extended:
|
|
630
|
+
for tri in pool:
|
|
631
|
+
nb = match_triple(tri, op, bb)
|
|
632
|
+
if nb is not None:
|
|
633
|
+
next_ext.append(nb)
|
|
634
|
+
extended = next_ext if next_ext else [b] # left-join: keep b
|
|
635
|
+
# add all extended bindings (or original if no match)
|
|
636
|
+
new_list.extend(extended)
|
|
637
|
+
bindings_list = new_list
|
|
638
|
+
|
|
639
|
+
# apply filters (pass blob_resolver for ?_source / ?source_text)
|
|
640
|
+
if filters:
|
|
641
|
+
bindings_list = [b for b in bindings_list
|
|
642
|
+
if apply_filters(b, filters, blob_resolver)]
|
|
643
|
+
|
|
644
|
+
# blob resolution: if the SELECT projects ?_source / ?source_text
|
|
645
|
+
# AND a blob_resolver was provided, resolve the chunk text from
|
|
646
|
+
# the sidecar arena for each binding (using the stashed __source_id).
|
|
647
|
+
# This makes the SPARQL surface actually useful for arena-stored
|
|
648
|
+
# content — `SELECT ?s ?source_text WHERE { ?s ?p ?o }` returns
|
|
649
|
+
# the source text of the chunk each fact was extracted from.
|
|
650
|
+
if blob_resolver and select_vars:
|
|
651
|
+
source_vars = [v for v in select_vars
|
|
652
|
+
if v in ("?_source", "?source_text")]
|
|
653
|
+
if source_vars:
|
|
654
|
+
for b in bindings_list:
|
|
655
|
+
src_id = b.get("__source_id")
|
|
656
|
+
if src_id:
|
|
657
|
+
try:
|
|
658
|
+
txt = blob_resolver(None, None, src_id) or ""
|
|
659
|
+
except Exception:
|
|
660
|
+
txt = ""
|
|
661
|
+
for v in source_vars:
|
|
662
|
+
b[v] = txt
|
|
663
|
+
|
|
664
|
+
# ORDER BY
|
|
665
|
+
if order_by:
|
|
666
|
+
var = order_by["var"]
|
|
667
|
+
desc = order_by["desc"]
|
|
668
|
+
bindings_list.sort(
|
|
669
|
+
key=lambda b: (b.get(var) or ""),
|
|
670
|
+
reverse=desc)
|
|
671
|
+
|
|
672
|
+
# DISTINCT
|
|
673
|
+
if parsed["distinct"]:
|
|
674
|
+
seen: set = set()
|
|
675
|
+
deduped = []
|
|
676
|
+
for b in bindings_list:
|
|
677
|
+
key = tuple((v, b.get(v)) for v in select_vars)
|
|
678
|
+
if key not in seen:
|
|
679
|
+
seen.add(key)
|
|
680
|
+
deduped.append(b)
|
|
681
|
+
bindings_list = deduped
|
|
682
|
+
|
|
683
|
+
# OFFSET
|
|
684
|
+
if offset:
|
|
685
|
+
bindings_list = bindings_list[offset:]
|
|
686
|
+
|
|
687
|
+
# LIMIT
|
|
688
|
+
if limit is not None:
|
|
689
|
+
bindings_list = bindings_list[:limit]
|
|
690
|
+
|
|
691
|
+
# project to SELECT vars
|
|
692
|
+
rows = []
|
|
693
|
+
for b in bindings_list:
|
|
694
|
+
row = {v: b.get(v, "") for v in select_vars}
|
|
695
|
+
rows.append(row)
|
|
696
|
+
|
|
697
|
+
return {
|
|
698
|
+
"head": {"vars": select_vars},
|
|
699
|
+
"results": {"bindings": rows},
|
|
700
|
+
"n_results": len(rows),
|
|
701
|
+
}
|
|
702
|
+
|
|
703
|
+
|
|
704
|
+
# Maximum query string length (bytes) — protects against pathological
|
|
705
|
+
# SPARQL queries that would blow up the regex parser or nested-loop
|
|
706
|
+
# joiner. SPARQL queries in practice are <1KB; 64KB is generous.
|
|
707
|
+
MAX_QUERY_BYTES = 64 * 1024
|
|
708
|
+
|
|
709
|
+
# Maximum body size for SPARQL POST (smaller than REST MAX_BODY because
|
|
710
|
+
# SPARQL queries are text-only, no payloads to ingest)
|
|
711
|
+
MAX_BODY_BYTES = 256 * 1024 # 256 KiB
|
|
712
|
+
|
|
713
|
+
|
|
714
|
+
# ----------------------------------------------------------- HTTP server
|
|
715
|
+
def make_handler(mem, user_id: str | None = None,
|
|
716
|
+
*, auth_keys=None, require_auth: bool = False):
|
|
717
|
+
"""Build a BaseHTTPRequestHandler bound to a Memory instance.
|
|
718
|
+
|
|
719
|
+
Auth: when `require_auth=True`, the handler validates a Bearer API
|
|
720
|
+
key against `auth_keys` (an APIKeyStore) and checks the
|
|
721
|
+
`sparql.query` permission. This is OFF by default for localhost
|
|
722
|
+
loopback use; the REST server turns it ON when co-hosting the
|
|
723
|
+
SPARQL endpoint so external graph tools go through the same RBAC
|
|
724
|
+
as the REST API.
|
|
725
|
+
"""
|
|
726
|
+
|
|
727
|
+
class SparqlHandler(BaseHTTPRequestHandler):
|
|
728
|
+
def _send(self, code: int, body: str,
|
|
729
|
+
content_type: str = "application/json"):
|
|
730
|
+
data = body.encode("utf-8")
|
|
731
|
+
self.send_response(code)
|
|
732
|
+
self.send_header("Content-Type", content_type)
|
|
733
|
+
self.send_header("Content-Length", str(len(data)))
|
|
734
|
+
self.send_header("Access-Control-Allow-Origin", "*")
|
|
735
|
+
self.send_header("Access-Control-Allow-Methods",
|
|
736
|
+
"GET, POST, OPTIONS")
|
|
737
|
+
self.send_header("Access-Control-Allow-Headers",
|
|
738
|
+
"Content-Type, Authorization")
|
|
739
|
+
self.end_headers()
|
|
740
|
+
try:
|
|
741
|
+
self.wfile.write(data)
|
|
742
|
+
except (BrokenPipeError, ConnectionResetError):
|
|
743
|
+
pass
|
|
744
|
+
|
|
745
|
+
def do_OPTIONS(self):
|
|
746
|
+
# CORS preflight — return 204 with headers
|
|
747
|
+
self.send_response(204)
|
|
748
|
+
self.send_header("Access-Control-Allow-Origin", "*")
|
|
749
|
+
self.send_header("Access-Control-Allow-Methods",
|
|
750
|
+
"GET, POST, OPTIONS")
|
|
751
|
+
self.send_header("Access-Control-Allow-Headers",
|
|
752
|
+
"Content-Type, Authorization")
|
|
753
|
+
self.send_header("Content-Length", "0")
|
|
754
|
+
self.end_headers()
|
|
755
|
+
|
|
756
|
+
def _auth(self) -> tuple[dict, str] | None:
|
|
757
|
+
"""Validate the Bearer token; return (meta, actor) or None.
|
|
758
|
+
|
|
759
|
+
On None, the caller has already sent a 401/403 response.
|
|
760
|
+
"""
|
|
761
|
+
if not require_auth or auth_keys is None:
|
|
762
|
+
# open mode — used for loopback standalone SPARQL.
|
|
763
|
+
# CAUTION: only safe behind a firewall or 127.0.0.1.
|
|
764
|
+
return ({"role": "admin", "label": "open-sparql"},
|
|
765
|
+
"open-sparql")
|
|
766
|
+
hdr = self.headers.get("Authorization") or ""
|
|
767
|
+
if not hdr.startswith("Bearer "):
|
|
768
|
+
self._send(401, json.dumps({
|
|
769
|
+
"error": "SPARQL endpoint requires a Bearer API key "
|
|
770
|
+
"(use the same key as the REST API)"}))
|
|
771
|
+
return None
|
|
772
|
+
key = hdr[7:].strip()
|
|
773
|
+
meta = auth_keys.verify(key)
|
|
774
|
+
if meta is None:
|
|
775
|
+
self._send(401, json.dumps({"error": "invalid or revoked key"}))
|
|
776
|
+
return None
|
|
777
|
+
# check sparql.query permission
|
|
778
|
+
from cortexm.security.rbac import authorize, RBACError
|
|
779
|
+
try:
|
|
780
|
+
authorize(meta, "sparql.query")
|
|
781
|
+
except RBACError as e:
|
|
782
|
+
self._send(403, json.dumps({"error": str(e)}))
|
|
783
|
+
return None
|
|
784
|
+
actor = meta.get("label") or meta.get("id", "key")
|
|
785
|
+
return (meta, actor)
|
|
786
|
+
|
|
787
|
+
def do_GET(self):
|
|
788
|
+
parsed_url = urllib.parse.urlparse(self.path)
|
|
789
|
+
params = urllib.parse.parse_qs(parsed_url.query)
|
|
790
|
+
# URL length guard (414 URI Too Long)
|
|
791
|
+
if len(parsed_url.query) > MAX_QUERY_BYTES:
|
|
792
|
+
self._send(414, json.dumps({
|
|
793
|
+
"error": f"query string exceeds {MAX_QUERY_BYTES} bytes"}))
|
|
794
|
+
return
|
|
795
|
+
auth = self._auth()
|
|
796
|
+
if auth is None:
|
|
797
|
+
return
|
|
798
|
+
query = params.get("query", [""])[0]
|
|
799
|
+
self._handle_query(query, auth)
|
|
800
|
+
|
|
801
|
+
def do_POST(self):
|
|
802
|
+
length = int(self.headers.get("Content-Length", 0))
|
|
803
|
+
if length > MAX_BODY_BYTES:
|
|
804
|
+
self._send(413, json.dumps({
|
|
805
|
+
"error": f"body exceeds {MAX_BODY_BYTES} bytes"}))
|
|
806
|
+
return
|
|
807
|
+
body = self.rfile.read(length).decode("utf-8") if length else ""
|
|
808
|
+
content_type = self.headers.get("Content-Type", "")
|
|
809
|
+
if "x-www-form-urlencoded" in content_type:
|
|
810
|
+
params = urllib.parse.parse_qs(body)
|
|
811
|
+
query = params.get("query", [""])[0]
|
|
812
|
+
else:
|
|
813
|
+
query = body
|
|
814
|
+
auth = self._auth()
|
|
815
|
+
if auth is None:
|
|
816
|
+
return
|
|
817
|
+
self._handle_query(query, auth)
|
|
818
|
+
|
|
819
|
+
def _handle_query(self, query: str, auth: tuple[dict, str]):
|
|
820
|
+
meta, actor = auth
|
|
821
|
+
if not query:
|
|
822
|
+
self._send(400, json.dumps({
|
|
823
|
+
"error": "missing 'query' parameter — send "
|
|
824
|
+
"'?query=SELECT ...' or POST a SPARQL query"}))
|
|
825
|
+
return
|
|
826
|
+
if len(query) > MAX_QUERY_BYTES:
|
|
827
|
+
self._send(413, json.dumps({
|
|
828
|
+
"error": f"query exceeds {MAX_QUERY_BYTES} bytes"}))
|
|
829
|
+
return
|
|
830
|
+
try:
|
|
831
|
+
# pull facts + edges (scope to user_id if SPARQL endpoint
|
|
832
|
+
# was started with --sparql-user-id; otherwise all users)
|
|
833
|
+
fact_objs = mem.store.query_facts(active=True, user_id=user_id)
|
|
834
|
+
# build (s, p, o, source_id) 4-tuples so the blob resolver
|
|
835
|
+
# can dereference source text via fact.source_id — which is
|
|
836
|
+
# the actual chunk_id (NOT fact.value, which holds the
|
|
837
|
+
# literal). This fixes a v2 bug where the resolver never
|
|
838
|
+
# fired because chunk_ids never appeared in `value`.
|
|
839
|
+
facts = [(f.subject, f.relation, f.value, f.source_id)
|
|
840
|
+
for f in fact_objs]
|
|
841
|
+
edges = edge_triples(mem, user_id=user_id)
|
|
842
|
+
# build blob resolver if arena is enabled
|
|
843
|
+
blob_resolver = None
|
|
844
|
+
arena = getattr(mem, "blob_arena", None)
|
|
845
|
+
if arena is not None:
|
|
846
|
+
def _resolve(_s, _r, source_id):
|
|
847
|
+
if not source_id:
|
|
848
|
+
return ""
|
|
849
|
+
from cortexm.trace.blob_arena import get_chunk_text
|
|
850
|
+
return get_chunk_text(mem.store, arena, source_id)
|
|
851
|
+
blob_resolver = _resolve
|
|
852
|
+
result = execute_sparql(query, facts, user_id,
|
|
853
|
+
edge_triples=edges,
|
|
854
|
+
blob_resolver=blob_resolver)
|
|
855
|
+
# audit (if auth is on AND a memory audit_log exists)
|
|
856
|
+
if require_auth and getattr(mem, "audit_log", None):
|
|
857
|
+
mem.audit_log.log(
|
|
858
|
+
"sparql.query", actor=actor,
|
|
859
|
+
role=meta.get("role"),
|
|
860
|
+
meta={"n_results": result.get("n_results", 0),
|
|
861
|
+
"query_head": query[:80]})
|
|
862
|
+
self._send(200, json.dumps(result, indent=2, default=str))
|
|
863
|
+
except ValueError as e:
|
|
864
|
+
self._send(400, json.dumps({"error": str(e)}))
|
|
865
|
+
except Exception as e: # noqa: BLE001
|
|
866
|
+
self._send(500, json.dumps({"error": f"server: {e}"}))
|
|
867
|
+
|
|
868
|
+
def log_message(self, fmt, *args):
|
|
869
|
+
pass # quiet
|
|
870
|
+
|
|
871
|
+
return SparqlHandler
|
|
872
|
+
|
|
873
|
+
|
|
874
|
+
class SparqlServer:
|
|
875
|
+
"""A minimal SPARQL HTTP endpoint backed by a Memory instance.
|
|
876
|
+
|
|
877
|
+
The Memory's reader is configured with the RDFDecoder so retrieval
|
|
878
|
+
produces RDF triples instead of an LLM prompt. This server exposes
|
|
879
|
+
those triples via SPARQL — a fully non-LLM retrieval path.
|
|
880
|
+
|
|
881
|
+
Threaded: uses ThreadingHTTPServer so concurrent queries don't
|
|
882
|
+
block each other. Each thread reads from the shared Memory (which
|
|
883
|
+
is itself guarded by an RLock at the store level).
|
|
884
|
+
|
|
885
|
+
Auth: when `require_auth=True` (the default when bound to non-
|
|
886
|
+
loopback interfaces), the server validates a Bearer API key
|
|
887
|
+
against the Memory's APIKeyStore and checks `sparql.query`
|
|
888
|
+
permission. When bound to 127.0.0.1 (default), auth is OFF for
|
|
889
|
+
local-dev convenience — but `--sparql-host 0.0.0.0` automatically
|
|
890
|
+
enables auth to prevent an unauthenticated fact dump.
|
|
891
|
+
"""
|
|
892
|
+
def __init__(self, mem, host: str = "127.0.0.1", port: int = 8910,
|
|
893
|
+
user_id: str | None = None,
|
|
894
|
+
require_auth: bool | None = None) -> None:
|
|
895
|
+
self.mem = mem
|
|
896
|
+
self.host = host
|
|
897
|
+
self.port = port
|
|
898
|
+
self.user_id = user_id
|
|
899
|
+
# auth is auto-enabled whenever binding to non-loopback
|
|
900
|
+
if require_auth is None:
|
|
901
|
+
require_auth = host not in ("127.0.0.1", "localhost", "::1")
|
|
902
|
+
self.require_auth = require_auth
|
|
903
|
+
self._server: ThreadingHTTPServer | None = None
|
|
904
|
+
self._thread: threading.Thread | None = None
|
|
905
|
+
|
|
906
|
+
def start(self) -> None:
|
|
907
|
+
auth_keys = self.mem.keys if self.require_auth else None
|
|
908
|
+
handler_cls = make_handler(self.mem, user_id=self.user_id,
|
|
909
|
+
auth_keys=auth_keys,
|
|
910
|
+
require_auth=self.require_auth)
|
|
911
|
+
self._server = ThreadingHTTPServer((self.host, self.port),
|
|
912
|
+
handler_cls)
|
|
913
|
+
self._server.daemon_threads = True
|
|
914
|
+
print(f"[sparql] listening on http://{self.host}:{self.port}/"
|
|
915
|
+
f" (SPARQL 1.1 Protocol; thread-per-request)")
|
|
916
|
+
auth_msg = ("on (Bearer + sparql.query)" if self.require_auth
|
|
917
|
+
else "OFF (loopback only)")
|
|
918
|
+
print(f"[sparql] auth: {auth_msg}")
|
|
919
|
+
print(f"[sparql] user_id: {self.user_id or '(all users)'}")
|
|
920
|
+
print(f"[sparql] try: curl 'http://{self.host}:{self.port}/"
|
|
921
|
+
f"?query=SELECT%20%3Fs%20%3Fp%20%3Fo%20WHERE%20%7B%20%3Fs%20"
|
|
922
|
+
f"%3Fp%20%3Fo%20%7D%20LIMIT%2010'")
|
|
923
|
+
|
|
924
|
+
def start_background(self) -> None:
|
|
925
|
+
"""Start the server in a daemon thread (non-blocking).
|
|
926
|
+
|
|
927
|
+
Used by `cortexm serve-rest --sparql-port N` to share one Memory
|
|
928
|
+
instance across both the REST API and the SPARQL endpoint.
|
|
929
|
+
"""
|
|
930
|
+
self.start()
|
|
931
|
+
self._thread = threading.Thread(
|
|
932
|
+
target=self._server.serve_forever, daemon=True,
|
|
933
|
+
name=f"sparql-{self.port}")
|
|
934
|
+
self._thread.start()
|
|
935
|
+
|
|
936
|
+
def serve_forever(self) -> None:
|
|
937
|
+
if self._server is None:
|
|
938
|
+
self.start()
|
|
939
|
+
assert self._server is not None
|
|
940
|
+
self._server.serve_forever()
|
|
941
|
+
|
|
942
|
+
def stop(self) -> None:
|
|
943
|
+
if self._server is not None:
|
|
944
|
+
self._server.shutdown()
|
|
945
|
+
self._server.server_close()
|
|
946
|
+
self._server = None
|
|
947
|
+
if self._thread is not None:
|
|
948
|
+
self._thread.join(timeout=2)
|
|
949
|
+
self._thread = None
|
|
950
|
+
|
|
951
|
+
|
|
952
|
+
def main() -> int:
|
|
953
|
+
import argparse
|
|
954
|
+
ap = argparse.ArgumentParser(prog="contextm-sparql",
|
|
955
|
+
description="Context-M SPARQL endpoint")
|
|
956
|
+
ap.add_argument("--host", default="127.0.0.1")
|
|
957
|
+
ap.add_argument("--port", type=int, default=8910)
|
|
958
|
+
ap.add_argument("--db", default=None,
|
|
959
|
+
help="path to the Context-M DB (default: in-memory)")
|
|
960
|
+
ap.add_argument("--user-id", default=None,
|
|
961
|
+
help="scope queries to a single user")
|
|
962
|
+
args = ap.parse_args()
|
|
963
|
+
|
|
964
|
+
from cortexm.api.memory import Memory
|
|
965
|
+
from cortexm.config import Config
|
|
966
|
+
cfg = Config.from_env()
|
|
967
|
+
if args.db:
|
|
968
|
+
cfg.db_path = args.db
|
|
969
|
+
mem = Memory(cfg)
|
|
970
|
+
server = SparqlServer(mem, host=args.host, port=args.port,
|
|
971
|
+
user_id=args.user_id)
|
|
972
|
+
try:
|
|
973
|
+
server.serve_forever()
|
|
974
|
+
except KeyboardInterrupt:
|
|
975
|
+
print("\n[sparql] shutting down")
|
|
976
|
+
finally:
|
|
977
|
+
server.stop()
|
|
978
|
+
mem.close()
|
|
979
|
+
return 0
|
|
980
|
+
|
|
981
|
+
|
|
982
|
+
if __name__ == "__main__":
|
|
983
|
+
import sys
|
|
984
|
+
sys.exit(main())
|