cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Federated schema aggregation — the Semantic Flywheel (privacy-preserving).
|
|
2
|
+
|
|
3
|
+
Opt-in edge nodes contribute ONLY the *schema* of their Trace (relation
|
|
4
|
+
histograms, category mix, temporal density, arity stats) — never raw
|
|
5
|
+
data, never embeddings. The global report feeds ontology learning that
|
|
6
|
+
improves extraction for everyone; a single node's contribution is
|
|
7
|
+
indistinguishable in the aggregate (k-anonymity threshold).
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import math
|
|
13
|
+
from collections import Counter
|
|
14
|
+
|
|
15
|
+
from cortexm.trace.fact import RELATION_CATEGORIES
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def export_schema_report(store, user_id: str | None = None) -> dict:
|
|
19
|
+
facts = store.query_facts(user_id=user_id, active=True,
|
|
20
|
+
include_quarantined=False)
|
|
21
|
+
rel_hist = Counter(f.relation for f in facts)
|
|
22
|
+
cat_hist = Counter()
|
|
23
|
+
for f in facts:
|
|
24
|
+
for cat, rels in RELATION_CATEGORIES.items():
|
|
25
|
+
if f.relation in rels:
|
|
26
|
+
cat_hist[cat] += 1
|
|
27
|
+
break
|
|
28
|
+
else:
|
|
29
|
+
cat_hist["other"] += 1
|
|
30
|
+
dated = [f for f in facts if f.relation == "event"]
|
|
31
|
+
span = None
|
|
32
|
+
if dated:
|
|
33
|
+
span = [min(f.valid_from for f in dated), max(f.valid_from for f in dated)]
|
|
34
|
+
n = max(len(facts), 1)
|
|
35
|
+
return {
|
|
36
|
+
"schema_version": 1,
|
|
37
|
+
"n_facts": len(facts),
|
|
38
|
+
"relation_histogram": dict(rel_hist),
|
|
39
|
+
"category_histogram": dict(cat_hist),
|
|
40
|
+
"single_valued_ratio": round(
|
|
41
|
+
sum(v for k, v in rel_hist.items()
|
|
42
|
+
if k in ("name", "works_at", "role", "lives_in", "prefers")) / n, 4),
|
|
43
|
+
"event_count": len(dated),
|
|
44
|
+
"temporal_span": span,
|
|
45
|
+
"privacy": "no raw data, no embeddings, no entity names",
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def merge_schema_reports(reports: list[dict], k_anonymity: int = 3) -> dict:
|
|
50
|
+
"""Aggregate local reports into a global ontology view."""
|
|
51
|
+
if not reports:
|
|
52
|
+
return {"relations": {}, "categories": {}, "contributors": 0}
|
|
53
|
+
rel = Counter()
|
|
54
|
+
cat = Counter()
|
|
55
|
+
contributors = len(reports)
|
|
56
|
+
for r in reports:
|
|
57
|
+
rel.update(r.get("relation_histogram", {}))
|
|
58
|
+
cat.update(r.get("category_histogram", {}))
|
|
59
|
+
# k-anonymity: only relations observed by >= k contributors (or >= k facts)
|
|
60
|
+
global_rel = {k: v for k, v in rel.items()
|
|
61
|
+
if v >= k_anonymity * max(1, contributors // 4) or v >= 10 * k_anonymity}
|
|
62
|
+
total = sum(rel.values()) or 1
|
|
63
|
+
novelty = sorted(
|
|
64
|
+
((k, v / total) for k, v in rel.items() if k not in global_rel),
|
|
65
|
+
key=lambda kv: -kv[1])[:10]
|
|
66
|
+
return {
|
|
67
|
+
"relations": dict(sorted(global_rel.items(), key=lambda kv: -kv[1])),
|
|
68
|
+
"categories": dict(sorted(cat.items(), key=lambda kv: -kv[1])),
|
|
69
|
+
"contributors": contributors,
|
|
70
|
+
"k_anonymity": k_anonymity,
|
|
71
|
+
"novel_patterns": [{"relation": k, "frequency": round(p, 5)}
|
|
72
|
+
for k, p in novelty],
|
|
73
|
+
}
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""Federation transports.
|
|
2
|
+
|
|
3
|
+
A transport moves signed JSON envelopes between nodes. Two are provided:
|
|
4
|
+
|
|
5
|
+
* ``InMemoryMesh`` — deterministic test/单-process topology (star, line,
|
|
6
|
+
partitioned groups);
|
|
7
|
+
* ``FileTransport`` — offline federation: envelopes are written as JSONL
|
|
8
|
+
files into per-node spool directories. Any shared filesystem, git
|
|
9
|
+
remote, object store or USB stick completes the physical channel; the
|
|
10
|
+
CRDT guarantees convergence no matter the delivery order or duplication.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
from cortexm.federation.node import FederationNode
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class InMemoryMesh:
|
|
22
|
+
"""Direct-connect topology; supports partitions for heal tests."""
|
|
23
|
+
|
|
24
|
+
def __init__(self) -> None:
|
|
25
|
+
self.links: set[tuple[str, str]] = set()
|
|
26
|
+
self.deliveries = 0
|
|
27
|
+
|
|
28
|
+
def link(self, a: str, b: str) -> None:
|
|
29
|
+
self.links.add((a, b))
|
|
30
|
+
self.links.add((b, a))
|
|
31
|
+
|
|
32
|
+
def cut(self, a: str, b: str) -> None:
|
|
33
|
+
self.links.discard((a, b))
|
|
34
|
+
self.links.discard((b, a))
|
|
35
|
+
|
|
36
|
+
def connected(self, a: str, b: str) -> bool:
|
|
37
|
+
return (a, b) in self.links
|
|
38
|
+
|
|
39
|
+
def gossip(self, nodes: dict[str, FederationNode],
|
|
40
|
+
rounds: int = 1) -> int:
|
|
41
|
+
"""Every directly-linked pair runs one two-way sync per round."""
|
|
42
|
+
total = 0
|
|
43
|
+
for _ in range(rounds):
|
|
44
|
+
for a, b in sorted(self.links):
|
|
45
|
+
if a >= b:
|
|
46
|
+
continue
|
|
47
|
+
na, nb = nodes[a], nodes[b]
|
|
48
|
+
na.sync_with(nb)
|
|
49
|
+
self.deliveries += 1
|
|
50
|
+
total += 1
|
|
51
|
+
return total
|
|
52
|
+
|
|
53
|
+
def full_sync(self, nodes: dict[str, FederationNode]) -> int:
|
|
54
|
+
"""All-pairs sync — converges any topology in one shot."""
|
|
55
|
+
ids = sorted(nodes)
|
|
56
|
+
n = 0
|
|
57
|
+
for i, a in enumerate(ids):
|
|
58
|
+
for b in ids[i + 1:]:
|
|
59
|
+
nodes[a].sync_with(nodes[b])
|
|
60
|
+
self.deliveries += 1
|
|
61
|
+
n += 1
|
|
62
|
+
return n
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class FileTransport:
|
|
66
|
+
"""Offline envelope exchange via spool directories.
|
|
67
|
+
|
|
68
|
+
Protocol per node directory:
|
|
69
|
+
outbox/<seq>.jsonl envelopes this node wants others to apply
|
|
70
|
+
inbox/<peer>.jsonl envelopes received from peers (append-only)
|
|
71
|
+
A mule process (cron, rsync, git push, human with a USB drive) moves
|
|
72
|
+
outbox files into the peer's inbox. ``drain`` applies whatever landed
|
|
73
|
+
and replies with reciprocal deltas.
|
|
74
|
+
"""
|
|
75
|
+
|
|
76
|
+
def __init__(self, root: str | Path) -> None:
|
|
77
|
+
self.root = Path(root)
|
|
78
|
+
|
|
79
|
+
def node_dir(self, node_id: str) -> Path:
|
|
80
|
+
d = self.root / node_id
|
|
81
|
+
(d / "outbox").mkdir(parents=True, exist_ok=True)
|
|
82
|
+
(d / "inbox").mkdir(parents=True, exist_ok=True)
|
|
83
|
+
return d
|
|
84
|
+
|
|
85
|
+
def emit_digest(self, node: FederationNode) -> Path:
|
|
86
|
+
env = node.digest_envelope()
|
|
87
|
+
d = self.node_dir(node.node_id)
|
|
88
|
+
p = d / "outbox" / f"digest-{node.clock.now().replace('.', '_')}.json"
|
|
89
|
+
p.write_text(json.dumps(env))
|
|
90
|
+
return p
|
|
91
|
+
|
|
92
|
+
def emit_delta(self, node: FederationNode,
|
|
93
|
+
their_digest_env: dict) -> Path:
|
|
94
|
+
env = node.delta_envelope_for(their_digest_env)
|
|
95
|
+
d = self.node_dir(node.node_id)
|
|
96
|
+
p = d / "outbox" / f"delta-{node.clock.now().replace('.', '_')}.json"
|
|
97
|
+
p.write_text(json.dumps(env))
|
|
98
|
+
return p
|
|
99
|
+
|
|
100
|
+
def emit_delta_and_digest(self, node: FederationNode,
|
|
101
|
+
their_digest_env: dict) -> tuple[Path, Path]:
|
|
102
|
+
"""Answer a peer's digest with BOTH: the delta they are missing and
|
|
103
|
+
our own digest so they can reciprocate. One mule round-trip then
|
|
104
|
+
achieves two-way convergence."""
|
|
105
|
+
delta_path = self.emit_delta(node, their_digest_env)
|
|
106
|
+
digest_path = self.emit_digest(node)
|
|
107
|
+
return delta_path, digest_path
|
|
108
|
+
|
|
109
|
+
def deliver(self, from_node: FederationNode, to_node_id: str,
|
|
110
|
+
envelope_path: Path) -> Path:
|
|
111
|
+
"""The 'mule': copy an outbox file into a peer's inbox."""
|
|
112
|
+
d = self.node_dir(to_node_id)
|
|
113
|
+
dst = d / "inbox" / f"{from_node.node_id}-{envelope_path.name}"
|
|
114
|
+
dst.write_text(envelope_path.read_text())
|
|
115
|
+
return dst
|
|
116
|
+
|
|
117
|
+
def drain(self, node: FederationNode) -> dict:
|
|
118
|
+
"""Apply every inbox envelope; answer digests with reciprocal
|
|
119
|
+
deltas (also placed in outbox for the mule to pick up)."""
|
|
120
|
+
d = self.node_dir(node.node_id)
|
|
121
|
+
applied = digests = 0
|
|
122
|
+
for f in sorted((d / "inbox").glob("*.json")):
|
|
123
|
+
env = json.loads(f.read_text())
|
|
124
|
+
if env.get("type") == "delta":
|
|
125
|
+
node.apply_delta_envelope(env)
|
|
126
|
+
applied += 1
|
|
127
|
+
f.unlink() # envelopes are one-shot
|
|
128
|
+
elif env.get("type") == "digest":
|
|
129
|
+
self.emit_delta_and_digest(node, env)
|
|
130
|
+
digests += 1
|
|
131
|
+
f.unlink()
|
|
132
|
+
return {"deltas_applied": applied, "digests_answered": digests}
|
|
133
|
+
|
|
134
|
+
def mule(self, nodes: dict[str, FederationNode]) -> int:
|
|
135
|
+
"""Move every outbox envelope to its destination's inbox (the
|
|
136
|
+
physical channel — in real deployments this is rsync/git/sneakernet).
|
|
137
|
+
Deltas carry a `to` address and are routed; digests broadcast."""
|
|
138
|
+
moved = 0
|
|
139
|
+
for node_id, node in nodes.items():
|
|
140
|
+
d = self.node_dir(node_id)
|
|
141
|
+
for f in sorted((d / "outbox").glob("*.json")):
|
|
142
|
+
env = json.loads(f.read_text())
|
|
143
|
+
dest = env.get("to") or None
|
|
144
|
+
targets = [dest] if dest else [p for p in nodes
|
|
145
|
+
if p != node_id]
|
|
146
|
+
for peer_id in targets:
|
|
147
|
+
if peer_id != node_id and peer_id in nodes:
|
|
148
|
+
self.deliver(node, peer_id, f)
|
|
149
|
+
moved += 1
|
|
150
|
+
f.unlink()
|
|
151
|
+
return moved
|
|
152
|
+
|
|
153
|
+
def exchange(self, nodes: dict[str, FederationNode],
|
|
154
|
+
rounds: int = 3) -> None:
|
|
155
|
+
"""Offline gossip: post digests, mule, drain — repeat until the
|
|
156
|
+
outboxes stay empty. Converges regardless of delivery order."""
|
|
157
|
+
for node in nodes.values():
|
|
158
|
+
self.emit_digest(node)
|
|
159
|
+
for _ in range(rounds):
|
|
160
|
+
moved = self.mule(nodes)
|
|
161
|
+
if moved == 0:
|
|
162
|
+
break
|
|
163
|
+
for node in nodes.values():
|
|
164
|
+
self.drain(node)
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""Index backends for Context-M's memory palace.
|
|
2
|
+
|
|
3
|
+
The default backend is the page-clustered hierarchical tree (Quadrant)
|
|
4
|
+
shipped as a Rust wheel under ``rust/quadrant/``. This package adds
|
|
5
|
+
alternative backends — currently the NSG (Navigable Spreading-out Graph)
|
|
6
|
+
proximity-graph index, with a pure-numpy fallback when the Rust wheel is
|
|
7
|
+
not installed.
|
|
8
|
+
|
|
9
|
+
All backends expose the same surface:
|
|
10
|
+
build(vectors: np.ndarray, ids: list[str] | None = None) -> None
|
|
11
|
+
search(query: np.ndarray, k: int = 10) -> list[tuple[str, float]]
|
|
12
|
+
stats() -> dict
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from cortexm.index.nsg import NsgBackend
|
|
18
|
+
|
|
19
|
+
__all__ = ["NsgBackend"]
|
cortexm/index/nsg.py
ADDED
|
@@ -0,0 +1,386 @@
|
|
|
1
|
+
"""NSG (Navigating Spreading-out Graph) index backend.
|
|
2
|
+
|
|
3
|
+
Alternative to the default quadrant page-clustered tree. NSG is more
|
|
4
|
+
compact and has better search efficiency at high recall on high-dim
|
|
5
|
+
vectors. Trade-off: build is slower (kNN pass), but query latency is
|
|
6
|
+
lower at the same recall target.
|
|
7
|
+
|
|
8
|
+
This module mirrors the pattern from ``cortexm.accel.QuadrantANN`` —
|
|
9
|
+
when the Rust wheel is installed (``pip install ./rust/quadrant``), the
|
|
10
|
+
hot paths route through the compiled ``NsgIndex``; otherwise a pure-numpy
|
|
11
|
+
implementation produces bit-identical recall behavior (it uses the same
|
|
12
|
+
MRNG pruning rule, the same medoid-selection logic, and the same greedy
|
|
13
|
+
best-first search). The numpy fallback exists so that the NSG backend
|
|
14
|
+
ships in production even before the wheel is built for a target platform.
|
|
15
|
+
|
|
16
|
+
Algorithm references:
|
|
17
|
+
* Fu, Xiang, Wang, Huang — "Fast Approximate Nearest Neighbor Search
|
|
18
|
+
with Navigable Spreading-out Graphs", VLDB 2019.
|
|
19
|
+
* The MRNG (Monotonic Relative Neighborhood Graph) prune rule keeps
|
|
20
|
+
an edge (p, q) iff no other neighbor r of p satisfies
|
|
21
|
+
dist(p, r) < dist(p, q) AND dist(q, r) < dist(p, q).
|
|
22
|
+
|
|
23
|
+
The numpy fallback uses cosine distance = 1 - dot(a, b) on unit vectors
|
|
24
|
+
(which is monotonic in squared Euclidean distance for unit vectors),
|
|
25
|
+
matching the Rust side exactly so recall ordering is preserved.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import os
|
|
31
|
+
from typing import Any
|
|
32
|
+
|
|
33
|
+
import numpy as np
|
|
34
|
+
|
|
35
|
+
# --- optional Rust wheel ---------------------------------------------------
|
|
36
|
+
try: # pragma: no cover
|
|
37
|
+
import quadrant as _quadrant # type: ignore
|
|
38
|
+
except Exception: # pragma: no cover
|
|
39
|
+
_quadrant = None
|
|
40
|
+
|
|
41
|
+
_env = os.environ.get("CONTEXTM_RUST", "auto")
|
|
42
|
+
NSG_RUST_AVAILABLE = _quadrant is not None and hasattr(_quadrant, "NsgIndex")
|
|
43
|
+
NSG_RUST_ENABLED = (_env == "1") or (_env == "auto" and NSG_RUST_AVAILABLE)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _sim_to_dist(sim: float) -> float:
|
|
47
|
+
"""Convert cosine similarity to MRNG distance.
|
|
48
|
+
|
|
49
|
+
For unit vectors, ``1 - dot(a, b)`` is monotonic in squared
|
|
50
|
+
Euclidean distance (``|a-b|^2 = 2 - 2*dot(a, b)``), so it preserves
|
|
51
|
+
the prune-rule ordering the Rust side relies on.
|
|
52
|
+
"""
|
|
53
|
+
return 1.0 - float(sim)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class _NsgNumpyFallback:
|
|
57
|
+
"""Pure-numpy NSG implementation — same recall behavior as Rust.
|
|
58
|
+
|
|
59
|
+
Build is O(N^2 D) (brute kNN), search is greedy best-first with
|
|
60
|
+
frontier width ``ef_search``. Intended for fallback when the wheel
|
|
61
|
+
is not available; the Rust path is preferred for any non-trivial N.
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
def __init__(self, k: int = 200, ef_search: int = 64) -> None:
|
|
65
|
+
self.k_build = int(k)
|
|
66
|
+
self.ef_search = int(ef_search)
|
|
67
|
+
self.vectors: np.ndarray | None = None
|
|
68
|
+
self.dims: int = 0
|
|
69
|
+
self.n: int = 0
|
|
70
|
+
self.edges: list[np.ndarray] = []
|
|
71
|
+
self.nav_node: int = 0
|
|
72
|
+
self.n_edges: int = 0
|
|
73
|
+
# ids map node index -> external string id
|
|
74
|
+
self.ids: list[str] = []
|
|
75
|
+
|
|
76
|
+
def _medoid(self) -> int:
|
|
77
|
+
"""Pick the node with max mean cosine sim to all others.
|
|
78
|
+
|
|
79
|
+
Deterministic and avoids RNG. For N <= 1024 we use the full
|
|
80
|
+
O(N^2 D) pass; for larger N we sample 256 candidates × 1024
|
|
81
|
+
eval rows, mirroring the Rust crate's strategy.
|
|
82
|
+
"""
|
|
83
|
+
if self.n == 0:
|
|
84
|
+
return 0
|
|
85
|
+
if self.n <= 1024:
|
|
86
|
+
cand = np.arange(self.n)
|
|
87
|
+
eval_set = np.arange(self.n)
|
|
88
|
+
else:
|
|
89
|
+
step_c = max(1, self.n // 256)
|
|
90
|
+
step_e = max(1, self.n // 1024)
|
|
91
|
+
cand = np.arange(0, self.n, step_c)[:256]
|
|
92
|
+
eval_set = np.arange(0, self.n, step_e)[:1024]
|
|
93
|
+
eval_vecs = self.vectors[eval_set]
|
|
94
|
+
best = int(cand[0])
|
|
95
|
+
best_mean = -np.inf
|
|
96
|
+
for c in cand:
|
|
97
|
+
sims = self.vectors[c] @ eval_vecs.T
|
|
98
|
+
mask = eval_set != c
|
|
99
|
+
m = float(sims[mask].mean()) if mask.any() else float(sims.mean())
|
|
100
|
+
if m > best_mean:
|
|
101
|
+
best_mean = m
|
|
102
|
+
best = int(c)
|
|
103
|
+
return best
|
|
104
|
+
|
|
105
|
+
def _knn(self, p: int, k: int) -> np.ndarray:
|
|
106
|
+
"""kNN of vector p (excluding itself), ascending distance."""
|
|
107
|
+
sims = self.vectors @ self.vectors[p]
|
|
108
|
+
sims[p] = -np.inf # exclude self
|
|
109
|
+
# Take top-k by sim (highest sim = closest). argpartition + sort
|
|
110
|
+
# gives O(N) partition + O(k log k) sort — same total ordering
|
|
111
|
+
# the Rust side produces via heap selection.
|
|
112
|
+
kk = min(k, self.n - 1)
|
|
113
|
+
if kk <= 0:
|
|
114
|
+
return np.empty(0, dtype=np.int64)
|
|
115
|
+
part = np.argpartition(-sims, kk - 1)[:kk]
|
|
116
|
+
order = part[np.argsort(-sims[part])]
|
|
117
|
+
return order
|
|
118
|
+
|
|
119
|
+
def build(self, vectors: np.ndarray, ids: list[str] | None = None) -> None:
|
|
120
|
+
v = np.ascontiguousarray(vectors, dtype=np.float32)
|
|
121
|
+
if v.ndim != 2:
|
|
122
|
+
raise ValueError(f"vectors must be 2-D, got shape {v.shape}")
|
|
123
|
+
n, d = v.shape
|
|
124
|
+
if n == 0:
|
|
125
|
+
raise ValueError("empty corpus")
|
|
126
|
+
# Normalize rows so cosine == dot product. We DO NOT mutate the
|
|
127
|
+
# caller's array; this is the index's own storage.
|
|
128
|
+
norms = np.linalg.norm(v, axis=1, keepdims=True)
|
|
129
|
+
norms = np.where(norms == 0, 1.0, norms)
|
|
130
|
+
self.vectors = (v / norms).astype(np.float32)
|
|
131
|
+
self.dims = int(d)
|
|
132
|
+
self.n = int(n)
|
|
133
|
+
self.nav_node = self._medoid()
|
|
134
|
+
if ids is None:
|
|
135
|
+
self.ids = [str(i) for i in range(n)]
|
|
136
|
+
else:
|
|
137
|
+
if len(ids) != n:
|
|
138
|
+
raise ValueError(
|
|
139
|
+
f"len(ids)={len(ids)} != n_vectors={n}")
|
|
140
|
+
self.ids = list(ids)
|
|
141
|
+
|
|
142
|
+
# Build pipeline (mirrors Fu et al. VLDB 2019 + the Rust crate):
|
|
143
|
+
# 1. kNN graph (full, pre-prune).
|
|
144
|
+
# 2. MRNG prune rule.
|
|
145
|
+
# 3. Tree-traversal connectivity pass — BFS from the medoid; for
|
|
146
|
+
# any unreachable node i, add an edge from the nearest visited
|
|
147
|
+
# kNN of i to i. This guarantees a greedy search starting from
|
|
148
|
+
# the medoid can reach every node (without it, well-separated
|
|
149
|
+
# clusters can leave the search stuck in the medoid's cluster).
|
|
150
|
+
k_eff = min(self.k_build, self.n - 1)
|
|
151
|
+
if k_eff <= 0:
|
|
152
|
+
k_eff = 1
|
|
153
|
+
|
|
154
|
+
# 1) kNN graph.
|
|
155
|
+
knn_graph: list[np.ndarray] = [self._knn(i, k_eff) for i in range(n)]
|
|
156
|
+
|
|
157
|
+
# 2) MRNG prune.
|
|
158
|
+
self.edges = []
|
|
159
|
+
for i in range(n):
|
|
160
|
+
knn = knn_graph[i]
|
|
161
|
+
kept: list[int] = []
|
|
162
|
+
pv = self.vectors[i]
|
|
163
|
+
for q in knn:
|
|
164
|
+
q = int(q)
|
|
165
|
+
qv = self.vectors[q]
|
|
166
|
+
sim_pq = float(pv @ qv)
|
|
167
|
+
d_pq = _sim_to_dist(sim_pq)
|
|
168
|
+
redundant = False
|
|
169
|
+
for r in kept:
|
|
170
|
+
rv = self.vectors[r]
|
|
171
|
+
sim_pr = float(pv @ rv)
|
|
172
|
+
d_pr = _sim_to_dist(sim_pr)
|
|
173
|
+
if d_pr < d_pq:
|
|
174
|
+
sim_qr = float(qv @ rv)
|
|
175
|
+
d_qr = _sim_to_dist(sim_qr)
|
|
176
|
+
if d_qr < d_pq:
|
|
177
|
+
redundant = True
|
|
178
|
+
break
|
|
179
|
+
if not redundant:
|
|
180
|
+
kept.append(q)
|
|
181
|
+
self.edges.append(np.asarray(kept, dtype=np.int64))
|
|
182
|
+
|
|
183
|
+
# 3) Tree-traversal connectivity pass.
|
|
184
|
+
visited = np.zeros(self.n, dtype=bool)
|
|
185
|
+
visited[self.nav_node] = True
|
|
186
|
+
self._bfs_mark(self.nav_node, visited)
|
|
187
|
+
# Iterate until every node is reachable.
|
|
188
|
+
while True:
|
|
189
|
+
unvisited = np.where(~visited)[0]
|
|
190
|
+
if len(unvisited) == 0:
|
|
191
|
+
break
|
|
192
|
+
orphan = int(unvisited[0])
|
|
193
|
+
# Nearest visited neighbor via kNN list (ascending distance).
|
|
194
|
+
added = None
|
|
195
|
+
for cand in knn_graph[orphan]:
|
|
196
|
+
cand = int(cand)
|
|
197
|
+
if visited[cand]:
|
|
198
|
+
added = cand
|
|
199
|
+
break
|
|
200
|
+
# Pathological: no visited kNN — fall back to medoid edge.
|
|
201
|
+
src = added if added is not None else self.nav_node
|
|
202
|
+
# Idempotent: don't double-add.
|
|
203
|
+
if orphan not in self.edges[src]:
|
|
204
|
+
self.edges[src] = np.append(self.edges[src], orphan)
|
|
205
|
+
visited[orphan] = True
|
|
206
|
+
self._bfs_mark(orphan, visited)
|
|
207
|
+
|
|
208
|
+
self.n_edges = sum(len(e) for e in self.edges)
|
|
209
|
+
|
|
210
|
+
def _bfs_mark(self, start: int, visited: np.ndarray) -> None:
|
|
211
|
+
"""Mark every node reachable from `start` via `self.edges` as visited."""
|
|
212
|
+
stack = [start]
|
|
213
|
+
while stack:
|
|
214
|
+
node = stack.pop()
|
|
215
|
+
for nb in self.edges[node]:
|
|
216
|
+
nb = int(nb)
|
|
217
|
+
if not visited[nb]:
|
|
218
|
+
visited[nb] = True
|
|
219
|
+
stack.append(nb)
|
|
220
|
+
|
|
221
|
+
def search(self, query: np.ndarray, k: int = 10) -> list[tuple[str, float]]:
|
|
222
|
+
if self.vectors is None or self.n == 0:
|
|
223
|
+
return []
|
|
224
|
+
q = np.ascontiguousarray(query, dtype=np.float32).ravel()
|
|
225
|
+
if q.shape[0] != self.dims:
|
|
226
|
+
raise ValueError(
|
|
227
|
+
f"query len {q.shape[0]} != dims {self.dims}")
|
|
228
|
+
nq = float(np.linalg.norm(q))
|
|
229
|
+
if nq == 0.0:
|
|
230
|
+
nq = 1.0
|
|
231
|
+
q = q / nq
|
|
232
|
+
ef = max(self.ef_search, k, 1)
|
|
233
|
+
# Greedy best-first from nav_node — same algorithm as Rust side:
|
|
234
|
+
# min-heap on dist = max-heap on -sim; frontier = top-ef by sim;
|
|
235
|
+
# neighbor enters candidate-expansion queue iff it survives the
|
|
236
|
+
# frontier. Terminates when best candidate is worse than the
|
|
237
|
+
# frontier's worst.
|
|
238
|
+
import heapq
|
|
239
|
+
visited = np.zeros(self.n, dtype=bool)
|
|
240
|
+
start = int(self.nav_node)
|
|
241
|
+
start_sim = float(self.vectors[start] @ q)
|
|
242
|
+
candidates: list[tuple[float, int]] = [] # min-heap on dist (1-sim)
|
|
243
|
+
frontier: list[tuple[float, int]] = [] # min-heap on sim (worst at root)
|
|
244
|
+
visited[start] = True
|
|
245
|
+
heapq.heappush(candidates, (1.0 - start_sim, start))
|
|
246
|
+
heapq.heappush(frontier, (start_sim, start))
|
|
247
|
+
while candidates:
|
|
248
|
+
c_dist, c_id = heapq.heappop(candidates)
|
|
249
|
+
c_sim = 1.0 - c_dist
|
|
250
|
+
if len(frontier) == ef and c_sim < frontier[0][0]:
|
|
251
|
+
break
|
|
252
|
+
for nb in self.edges[c_id]:
|
|
253
|
+
nb = int(nb)
|
|
254
|
+
if visited[nb]:
|
|
255
|
+
continue
|
|
256
|
+
visited[nb] = True
|
|
257
|
+
s = float(self.vectors[nb] @ q)
|
|
258
|
+
# Admit to frontier iff (not full) OR (better than worst)
|
|
259
|
+
admit = (len(frontier) < ef) or (s > frontier[0][0])
|
|
260
|
+
if admit:
|
|
261
|
+
heapq.heappush(frontier, (s, nb))
|
|
262
|
+
if len(frontier) > ef:
|
|
263
|
+
heapq.heappop(frontier)
|
|
264
|
+
heapq.heappush(candidates, (1.0 - s, nb))
|
|
265
|
+
# Return top-k sorted desc by sim, as (id_str, sim) tuples.
|
|
266
|
+
out = sorted(frontier, key=lambda t: -t[0])[:k]
|
|
267
|
+
return [(self.ids[i], float(s)) for s, i in out]
|
|
268
|
+
|
|
269
|
+
def stats(self) -> dict:
|
|
270
|
+
avg_deg = (self.n_edges / self.n) if self.n else 0.0
|
|
271
|
+
return {
|
|
272
|
+
"mode": "numpy-fallback",
|
|
273
|
+
"n_vectors": self.n,
|
|
274
|
+
"dims": self.dims,
|
|
275
|
+
"k_build": self.k_build,
|
|
276
|
+
"n_edges": self.n_edges,
|
|
277
|
+
"avg_degree": round(float(avg_deg), 4),
|
|
278
|
+
"nav_node": self.nav_node,
|
|
279
|
+
"ef_search": self.ef_search,
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
class NsgBackend:
|
|
284
|
+
"""NSG (Navigating Spreading-out Graph) index backend.
|
|
285
|
+
|
|
286
|
+
Alternative to the default quadrant page-clustered tree. NSG is more
|
|
287
|
+
compact and has better search efficiency at high recall on high-dim
|
|
288
|
+
vectors. Trade-off: build is slower (kNN pass), but query latency
|
|
289
|
+
is lower at the same recall target.
|
|
290
|
+
|
|
291
|
+
Mirrors the pattern of ``cortexm.accel.QuadrantANN``: if the Rust
|
|
292
|
+
wheel (``quadrant.NsgIndex``) is installed and enabled, the hot
|
|
293
|
+
paths route through it; otherwise a pure-numpy NSG produces the same
|
|
294
|
+
recall ordering (the brute kNN pass uses the same selection rule,
|
|
295
|
+
the MRNG prune is bit-identical, the search uses the same greedy
|
|
296
|
+
best-first walk with the same termination criterion).
|
|
297
|
+
"""
|
|
298
|
+
|
|
299
|
+
def __init__(self, dims: int, k: int = 200, ef_search: int = 64) -> None:
|
|
300
|
+
self.dims = int(dims)
|
|
301
|
+
self.k_build = int(k)
|
|
302
|
+
self.ef_search = int(ef_search)
|
|
303
|
+
self._rust_idx: Any = None
|
|
304
|
+
self._np_idx: _NsgNumpyFallback | None = None
|
|
305
|
+
self._ids: list[str] = []
|
|
306
|
+
self._vectors: np.ndarray | None = None
|
|
307
|
+
self._mode: str = "uninitialized"
|
|
308
|
+
|
|
309
|
+
def _use_rust(self) -> bool:
|
|
310
|
+
return NSG_RUST_AVAILABLE and NSG_RUST_ENABLED
|
|
311
|
+
|
|
312
|
+
def build(self, vectors: np.ndarray, ids: list[str] | None = None) -> None:
|
|
313
|
+
v = np.ascontiguousarray(vectors, dtype=np.float32)
|
|
314
|
+
if v.ndim != 2:
|
|
315
|
+
raise ValueError(f"vectors must be 2-D, got shape {v.shape}")
|
|
316
|
+
n, d = v.shape
|
|
317
|
+
if d != self.dims:
|
|
318
|
+
raise ValueError(
|
|
319
|
+
f"vectors have dims={d}, expected {self.dims}")
|
|
320
|
+
if n == 0:
|
|
321
|
+
raise ValueError("empty corpus")
|
|
322
|
+
self._vectors = v.copy()
|
|
323
|
+
if ids is None:
|
|
324
|
+
self._ids = [str(i) for i in range(n)]
|
|
325
|
+
else:
|
|
326
|
+
if len(ids) != n:
|
|
327
|
+
raise ValueError(f"len(ids)={len(ids)} != n_vectors={n}")
|
|
328
|
+
self._ids = list(ids)
|
|
329
|
+
if self._use_rust():
|
|
330
|
+
# The Rust wheel owns the vectors and adjacency lists; we
|
|
331
|
+
# keep a reference to the numpy storage so `search()` can
|
|
332
|
+
# resolve ids when callers pass through the Python surface.
|
|
333
|
+
self._rust_idx = _quadrant.NsgIndex.build(v, self.k_build)
|
|
334
|
+
self._np_idx = None
|
|
335
|
+
self._mode = "rust"
|
|
336
|
+
else: # pragma: no cover
|
|
337
|
+
self._rust_idx = None
|
|
338
|
+
self._np_idx = _NsgNumpyFallback(k=self.k_build, ef_search=self.ef_search)
|
|
339
|
+
self._np_idx.build(v, self._ids)
|
|
340
|
+
self._mode = "numpy-fallback"
|
|
341
|
+
|
|
342
|
+
def search(self, query: np.ndarray, k: int = 10) -> list[tuple[str, float]]:
|
|
343
|
+
if self._mode == "uninitialized":
|
|
344
|
+
raise RuntimeError("NsgBackend.build() must be called first")
|
|
345
|
+
q = np.ascontiguousarray(query, dtype=np.float32).ravel()
|
|
346
|
+
if q.shape[0] != self.dims:
|
|
347
|
+
raise ValueError(
|
|
348
|
+
f"query len {q.shape[0]} != dims {self.dims}")
|
|
349
|
+
if self._mode == "rust":
|
|
350
|
+
ids_u32, sims = self._rust_idx.search(q, k, self.ef_search)
|
|
351
|
+
return [(self._ids[int(i)], float(s))
|
|
352
|
+
for i, s in zip(ids_u32, sims)]
|
|
353
|
+
return self._np_idx.search(q, k) # pragma: no cover
|
|
354
|
+
|
|
355
|
+
def stats(self) -> dict:
|
|
356
|
+
if self._mode == "uninitialized":
|
|
357
|
+
return {"mode": "uninitialized", "dims": self.dims,
|
|
358
|
+
"k_build": self.k_build, "ef_search": self.ef_search}
|
|
359
|
+
if self._mode == "rust":
|
|
360
|
+
return {
|
|
361
|
+
"mode": "rust",
|
|
362
|
+
"dims": self.dims,
|
|
363
|
+
"k_build": self.k_build,
|
|
364
|
+
"ef_search": self.ef_search,
|
|
365
|
+
"detail": self._rust_idx.stats(),
|
|
366
|
+
"n_vectors": int(self._rust_idx.n_vectors()),
|
|
367
|
+
"n_edges": int(self._rust_idx.n_edges()),
|
|
368
|
+
"nav_node": int(self._rust_idx.nav_node()),
|
|
369
|
+
}
|
|
370
|
+
return self._np_idx.stats() # pragma: no cover
|
|
371
|
+
|
|
372
|
+
def __repr__(self) -> str:
|
|
373
|
+
return (f"NsgBackend(dims={self.dims}, k={self.k_build}, "
|
|
374
|
+
f"ef_search={self.ef_search}, mode={self._mode})")
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
def nsg_status() -> dict[str, Any]:
|
|
378
|
+
"""Report whether the Rust NSG wheel is wired and active."""
|
|
379
|
+
return {
|
|
380
|
+
"rust_available": bool(NSG_RUST_AVAILABLE),
|
|
381
|
+
"rust_enabled": bool(NSG_RUST_ENABLED),
|
|
382
|
+
"env": _env,
|
|
383
|
+
"hint": ("" if NSG_RUST_AVAILABLE else
|
|
384
|
+
"build with: pip install ./rust/quadrant "
|
|
385
|
+
"(requires cargo + maturin) — numpy fallback active"),
|
|
386
|
+
}
|
cortexm/mcp/__init__.py
ADDED
|
File without changes
|