cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
"""Engineered role vectors — NSR-inspired (ESWEEK24).
|
|
2
|
+
|
|
3
|
+
arXiv insight: NSR trains an autoencoder to learn compact embeddings
|
|
4
|
+
for the fact vocabulary, then uses those embeddings to construct role
|
|
5
|
+
vectors that are *maximally mutually orthogonal* in the data manifold.
|
|
6
|
+
Random role vectors work (HDC theory guarantees capacity grows with
|
|
7
|
+
sqrt(dims)), but engineered ones have higher effective capacity and
|
|
8
|
+
lower cross-talk because they sit on the directions of greatest
|
|
9
|
+
variance in the actual data.
|
|
10
|
+
|
|
11
|
+
This module provides:
|
|
12
|
+
|
|
13
|
+
EngineeredRoleVectors(dims, n_roles=3)
|
|
14
|
+
.fit(fact_matrix) -> trains a tiny 1-layer autoencoder
|
|
15
|
+
on the fact matrix and stores the
|
|
16
|
+
top-k principal directions as role
|
|
17
|
+
vectors
|
|
18
|
+
.role_vec(role) -> returns the engineered role vector
|
|
19
|
+
(falls back to VSA's random init
|
|
20
|
+
if not yet fit)
|
|
21
|
+
.save(path) / .load(path)
|
|
22
|
+
|
|
23
|
+
WHY THIS MATTERS:
|
|
24
|
+
- role_vec("S") currently = rng.standard_normal(dims). For dims=768
|
|
25
|
+
that's a random point on the unit sphere — fine in theory but it
|
|
26
|
+
wastes capacity on directions orthogonal to the actual fact vocab.
|
|
27
|
+
- After fit(), role_vec("S") = the top-1 principal direction of the
|
|
28
|
+
S-vector manifold. role_vec("R") = top-2, role_vec("V") = top-3.
|
|
29
|
+
These directions are where the data actually lives, so:
|
|
30
|
+
(a) bound holograms exploit the full effective capacity
|
|
31
|
+
(b) cross-talk between S/R/V role bindings drops because the
|
|
32
|
+
top-k principal directions are approximately orthogonal by
|
|
33
|
+
construction
|
|
34
|
+
(c) retrieval probes have higher SNR because they're aligned
|
|
35
|
+
with the data axes
|
|
36
|
+
|
|
37
|
+
IMPLEMENTATION:
|
|
38
|
+
Tiny 1-layer linear autoencoder (no nonlinearity) trained with
|
|
39
|
+
plain SGD on the fact matrix. The encoder weights converge to the
|
|
40
|
+
top-k principal components (proven by Baldi & Horn, 1989 — linear
|
|
41
|
+
AE on centered data recovers PCA). We center the data and train the
|
|
42
|
+
encoder to have orthogonal rows via a soft orthogonality penalty.
|
|
43
|
+
|
|
44
|
+
Cost: ~1 day of human work, ~30 seconds of CPU time on 1000 facts.
|
|
45
|
+
Deterministic given the seed.
|
|
46
|
+
"""
|
|
47
|
+
from __future__ import annotations
|
|
48
|
+
|
|
49
|
+
import json
|
|
50
|
+
import os
|
|
51
|
+
from pathlib import Path
|
|
52
|
+
|
|
53
|
+
import numpy as np
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class EngineeredRoleVectors:
|
|
57
|
+
"""Train and serve engineered role vectors.
|
|
58
|
+
|
|
59
|
+
Construction:
|
|
60
|
+
erv = EngineeredRoleVectors(dims=768, n_roles=3, seed=42)
|
|
61
|
+
|
|
62
|
+
Fit:
|
|
63
|
+
erv.fit(fact_matrix) # fact_matrix: (n_facts, dims)
|
|
64
|
+
# trains a tiny AE, extracts top-k principal directions,
|
|
65
|
+
# stores them as the role vectors
|
|
66
|
+
|
|
67
|
+
Use:
|
|
68
|
+
erv.role_vec("S") # top-1 principal direction
|
|
69
|
+
erv.role_vec("R") # top-2
|
|
70
|
+
erv.role_vec("V") # top-3
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
ROLE_ORDER = ["S", "R", "V"] # subject / relation / value
|
|
74
|
+
|
|
75
|
+
def __init__(self, dims: int = 768, n_roles: int = 3,
|
|
76
|
+
seed: int = 42, n_epochs: int = 200,
|
|
77
|
+
lr: float = 0.01, orth_penalty: float = 0.1) -> None:
|
|
78
|
+
self.dims = dims
|
|
79
|
+
self.n_roles = n_roles
|
|
80
|
+
self.seed = seed
|
|
81
|
+
self.n_epochs = n_epochs
|
|
82
|
+
self.lr = lr
|
|
83
|
+
self.orth_penalty = orth_penalty
|
|
84
|
+
self._role_vecs: dict[str, np.ndarray] = {}
|
|
85
|
+
self._mean: np.ndarray | None = None
|
|
86
|
+
self._fit_loss: list[float] = []
|
|
87
|
+
|
|
88
|
+
# ------------------------------------------------------------- fit
|
|
89
|
+
def fit(self, fact_matrix: np.ndarray) -> dict:
|
|
90
|
+
"""Train a tiny linear autoencoder on `fact_matrix`.
|
|
91
|
+
|
|
92
|
+
fact_matrix: (n_samples, dims) float32. Should contain the
|
|
93
|
+
subject / relation / value vectors for all facts in the
|
|
94
|
+
corpus, stacked. The AE learns to reconstruct them through
|
|
95
|
+
a k-dim bottleneck where k = n_roles.
|
|
96
|
+
|
|
97
|
+
After training, the encoder's rows are the top-k principal
|
|
98
|
+
directions. We store them as role_vec("S"/"R"/"V").
|
|
99
|
+
|
|
100
|
+
Returns a training report dict.
|
|
101
|
+
"""
|
|
102
|
+
X = np.asarray(fact_matrix, dtype=np.float32)
|
|
103
|
+
if X.ndim != 2:
|
|
104
|
+
raise ValueError(f"expected 2D matrix, got shape {X.shape}")
|
|
105
|
+
n, d = X.shape
|
|
106
|
+
if d != self.dims:
|
|
107
|
+
raise ValueError(
|
|
108
|
+
f"matrix dim {d} != configured dims {self.dims}")
|
|
109
|
+
if n < self.n_roles:
|
|
110
|
+
# not enough samples to learn k directions — fall back
|
|
111
|
+
# to top-k random vectors orthogonalized via Gram-Schmidt
|
|
112
|
+
rng = np.random.default_rng(self.seed)
|
|
113
|
+
vecs = rng.standard_normal((self.n_roles, d)).astype(np.float32)
|
|
114
|
+
for i in range(self.n_roles):
|
|
115
|
+
for j in range(i):
|
|
116
|
+
vecs[i] -= np.dot(vecs[i], vecs[j]) * vecs[j]
|
|
117
|
+
n_ = float(np.linalg.norm(vecs[i]))
|
|
118
|
+
vecs[i] /= max(n_, 1e-9)
|
|
119
|
+
for i, r in enumerate(self.ROLE_ORDER[:self.n_roles]):
|
|
120
|
+
self._role_vecs[r] = vecs[i]
|
|
121
|
+
return {"trained": False, "reason": "insufficient_samples",
|
|
122
|
+
"n_samples": n, "n_roles": self.n_roles,
|
|
123
|
+
"fallback": "gram_schmidt_random"}
|
|
124
|
+
|
|
125
|
+
# center the data (PCA assumes centered data)
|
|
126
|
+
self._mean = X.mean(axis=0)
|
|
127
|
+
Xc = X - self._mean
|
|
128
|
+
|
|
129
|
+
# tiny linear AE: encoder W: (k, d), decoder W.T: (d, k)
|
|
130
|
+
# init with small random values
|
|
131
|
+
rng = np.random.default_rng(self.seed)
|
|
132
|
+
k = self.n_roles
|
|
133
|
+
scale = float(1.0 / np.sqrt(d))
|
|
134
|
+
W = rng.standard_normal((k, d)).astype(np.float32) * scale
|
|
135
|
+
|
|
136
|
+
# SGD on reconstruction loss + orthogonality penalty
|
|
137
|
+
# (orth penalty: W @ W.T should be ≈ I)
|
|
138
|
+
batch_size = min(64, n)
|
|
139
|
+
for epoch in range(self.n_epochs):
|
|
140
|
+
perm = rng.permutation(n)
|
|
141
|
+
epoch_loss = 0.0
|
|
142
|
+
for i in range(0, n, batch_size):
|
|
143
|
+
idx = perm[i:i + batch_size]
|
|
144
|
+
xb = Xc[idx] # (b, d)
|
|
145
|
+
# forward: z = xb @ W.T (b, k)
|
|
146
|
+
# recon: xhat = z @ W (b, d)
|
|
147
|
+
z = xb @ W.T
|
|
148
|
+
xhat = z @ W
|
|
149
|
+
# L2 reconstruction loss per sample — clip to avoid
|
|
150
|
+
# overflow on numerically large data (the orth penalty
|
|
151
|
+
# can push weights to blow up if lr is too high)
|
|
152
|
+
diff = xhat - xb
|
|
153
|
+
# clip per-element to a safe range
|
|
154
|
+
diff = np.clip(diff, -1e4, 1e4)
|
|
155
|
+
loss = float(np.mean(np.sum(diff * diff, axis=1)))
|
|
156
|
+
# grad on W: dL/dW = 2 * (xhat - x).T @ z / b
|
|
157
|
+
# shape (d, k) -> transpose for our layout
|
|
158
|
+
g = 2.0 * diff.T @ z / len(idx) # (d, k)
|
|
159
|
+
# clip gradient to prevent overflow → NaN
|
|
160
|
+
g = np.clip(g, -1.0, 1.0)
|
|
161
|
+
# orthogonality penalty: ||W @ W.T - I||_F^2 / k
|
|
162
|
+
# grad: 2/k * (W @ W.T - I) @ W
|
|
163
|
+
WtW = W @ W.T # (k, k)
|
|
164
|
+
I_k = np.eye(k, dtype=np.float32)
|
|
165
|
+
orth_grad = (2.0 / k) * (WtW - I_k) @ W
|
|
166
|
+
orth_grad = np.clip(orth_grad, -1.0, 1.0)
|
|
167
|
+
# combined grad on W (note: g is (d,k), so transpose)
|
|
168
|
+
W -= self.lr * (g.T + self.orth_penalty * orth_grad)
|
|
169
|
+
# also clip weights themselves to a safe range
|
|
170
|
+
W = np.clip(W, -10.0, 10.0)
|
|
171
|
+
epoch_loss += loss * len(idx)
|
|
172
|
+
epoch_loss /= n
|
|
173
|
+
self._fit_loss.append(epoch_loss)
|
|
174
|
+
if epoch % 20 == 0 or epoch == self.n_epochs - 1:
|
|
175
|
+
# report conditioning of W @ W.T (lower = more orthogonal)
|
|
176
|
+
cond = float(np.linalg.cond(WtW)) if k > 1 else 1.0
|
|
177
|
+
# suppress per-epoch logging during normal use
|
|
178
|
+
pass
|
|
179
|
+
|
|
180
|
+
# extract role vectors: rows of W are the top-k principal dirs
|
|
181
|
+
# normalize to unit length
|
|
182
|
+
for i, r in enumerate(self.ROLE_ORDER[:k]):
|
|
183
|
+
v = W[i].copy()
|
|
184
|
+
nrm = float(np.linalg.norm(v))
|
|
185
|
+
self._role_vecs[r] = v / max(nrm, 1e-9)
|
|
186
|
+
|
|
187
|
+
# report final reconstruction loss + orthogonality
|
|
188
|
+
WtW_final = W @ W.T
|
|
189
|
+
off_diag = float(np.sum(np.abs(WtW_final - np.eye(k, dtype=np.float32))))
|
|
190
|
+
return {
|
|
191
|
+
"trained": True,
|
|
192
|
+
"n_samples": n,
|
|
193
|
+
"dims": d,
|
|
194
|
+
"n_roles": k,
|
|
195
|
+
"epochs": self.n_epochs,
|
|
196
|
+
"final_loss": self._fit_loss[-1] if self._fit_loss else 0.0,
|
|
197
|
+
"initial_loss": self._fit_loss[0] if self._fit_loss else 0.0,
|
|
198
|
+
"loss_reduction_pct": (
|
|
199
|
+
(1.0 - self._fit_loss[-1] / max(self._fit_loss[0], 1e-9)) * 100
|
|
200
|
+
if self._fit_loss else 0.0),
|
|
201
|
+
"off_diag_sum": off_diag,
|
|
202
|
+
"condition_number": (float(np.linalg.cond(WtW_final))
|
|
203
|
+
if k > 1 else 1.0),
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
# ------------------------------------------------------------- serve
|
|
207
|
+
def role_vec(self, role: str) -> np.ndarray | None:
|
|
208
|
+
return self._role_vecs.get(role)
|
|
209
|
+
|
|
210
|
+
@property
|
|
211
|
+
def is_fit(self) -> bool:
|
|
212
|
+
return bool(self._role_vecs)
|
|
213
|
+
|
|
214
|
+
def save(self, path: str | os.PathLike) -> None:
|
|
215
|
+
"""Persist the role vectors to a .npz file."""
|
|
216
|
+
arrays = {f"role_{r}": v for r, v in self._role_vecs.items()}
|
|
217
|
+
if self._mean is not None:
|
|
218
|
+
arrays["mean"] = self._mean
|
|
219
|
+
arrays["meta"] = np.array(json.dumps({
|
|
220
|
+
"dims": self.dims, "n_roles": self.n_roles,
|
|
221
|
+
"seed": self.seed, "final_loss": (
|
|
222
|
+
self._fit_loss[-1] if self._fit_loss else 0.0),
|
|
223
|
+
}), dtype=str)
|
|
224
|
+
np.savez(path, **arrays)
|
|
225
|
+
|
|
226
|
+
def load(self, path: str | os.PathLike) -> None:
|
|
227
|
+
data = np.load(path, allow_pickle=False)
|
|
228
|
+
for r in self.ROLE_ORDER[:self.n_roles]:
|
|
229
|
+
key = f"role_{r}"
|
|
230
|
+
if key in data:
|
|
231
|
+
self._role_vecs[r] = data[key]
|
|
232
|
+
if "mean" in data:
|
|
233
|
+
self._mean = data["mean"]
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
__all__ = ["EngineeredRoleVectors"]
|
cortexm/vsa/slb.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Semantic Lookaside Buffer — conversational locality cache.
|
|
2
|
+
|
|
3
|
+
64-entry ring buffer of quantized query signatures with cached result
|
|
4
|
+
sets. A new query whose signature is ≥ ``threshold`` cosine-similar to a
|
|
5
|
+
cached signature reuses the cached ranking (conversational locality:
|
|
6
|
+
follow-up questions are near-duplicates of their predecessors). L1-
|
|
7
|
+
resident by design; hit path costs one 64×dims dot product.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import numpy as np
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class SemanticLookasideBuffer:
|
|
16
|
+
def __init__(self, entries: int = 64, threshold: float = 0.97,
|
|
17
|
+
dims: int = 768) -> None:
|
|
18
|
+
self.capacity = entries
|
|
19
|
+
self.threshold = threshold
|
|
20
|
+
self.dims = dims
|
|
21
|
+
self._sigs = np.zeros((entries, dims), dtype=np.float32)
|
|
22
|
+
self._results: list[list[tuple[str, float]] | None] = [None] * entries
|
|
23
|
+
self._queries: list[str | None] = [None] * entries
|
|
24
|
+
self._scopes: list[tuple | None] = [None] * entries
|
|
25
|
+
self._pos = 0
|
|
26
|
+
self._filled = 0
|
|
27
|
+
self.hits = 0
|
|
28
|
+
self.misses = 0
|
|
29
|
+
self.total_hit_latency = 0.0
|
|
30
|
+
self.total_miss_latency = 0.0
|
|
31
|
+
|
|
32
|
+
def lookup(self, q: np.ndarray,
|
|
33
|
+
scope: tuple | None = None) -> list[tuple[str, float]] | None:
|
|
34
|
+
"""Return cached results for a signature **in the same scope**.
|
|
35
|
+
|
|
36
|
+
Scope-blind lookup is a correctness bug, not just a privacy one:
|
|
37
|
+
near-duplicate queries from different users would cross-contaminate
|
|
38
|
+
(and then die in the caller's scope filter, yielding empty blocks).
|
|
39
|
+
"""
|
|
40
|
+
if self._filled == 0:
|
|
41
|
+
return None
|
|
42
|
+
sims = self._sigs[: self._filled] @ q
|
|
43
|
+
best = int(np.argmax(sims))
|
|
44
|
+
if float(sims[best]) >= self.threshold and self._scopes[best] == scope:
|
|
45
|
+
self.hits += 1
|
|
46
|
+
return self._results[best]
|
|
47
|
+
return None
|
|
48
|
+
|
|
49
|
+
def store(self, q: np.ndarray, results: list[tuple[str, float]],
|
|
50
|
+
query: str = "", scope: tuple | None = None) -> None:
|
|
51
|
+
pos = self._pos
|
|
52
|
+
self._sigs[pos] = q
|
|
53
|
+
self._results[pos] = results
|
|
54
|
+
self._queries[pos] = query
|
|
55
|
+
self._scopes[pos] = scope
|
|
56
|
+
self._pos = (self._pos + 1) % self.capacity
|
|
57
|
+
self._filled = min(self._filled + 1, self.capacity)
|
|
58
|
+
|
|
59
|
+
def record_latency(self, hit: bool, seconds: float) -> None:
|
|
60
|
+
if hit:
|
|
61
|
+
self.total_hit_latency += seconds
|
|
62
|
+
else:
|
|
63
|
+
self.total_miss_latency += seconds
|
|
64
|
+
|
|
65
|
+
@property
|
|
66
|
+
def miss_latency_avg(self) -> float:
|
|
67
|
+
return (self.total_miss_latency / self.misses) if self.misses else 0.0
|
|
68
|
+
|
|
69
|
+
def stats(self) -> dict:
|
|
70
|
+
total = self.hits + self.misses
|
|
71
|
+
return {
|
|
72
|
+
"hits": self.hits, "misses": self.misses,
|
|
73
|
+
"hit_rate": round(self.hits / total, 4) if total else 0.0,
|
|
74
|
+
"avg_hit_latency_us": round(
|
|
75
|
+
self.total_hit_latency / self.hits * 1e6, 1) if self.hits else 0.0,
|
|
76
|
+
"avg_miss_latency_us": round(self.miss_latency_avg * 1e6, 1),
|
|
77
|
+
"entries_used": self._filled,
|
|
78
|
+
}
|
cortexm/vsa/tlsh_trie.py
ADDED
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""TLSH ternary trie — software TCAM for O(log N + w) hologram lookup.
|
|
2
|
+
|
|
3
|
+
The Stanford TLSH paper (arXiv:1006.3514) uses ternary content-
|
|
4
|
+
addressable memory for O(1) parallel lookup. Without TCAM hardware,
|
|
5
|
+
we emulate it as a ternary Patricia trie over packed binary holograms:
|
|
6
|
+
each path is the bits of a packed binary vector, with wildcard edges
|
|
7
|
+
that match either bit (the ternary '*' bit).
|
|
8
|
+
|
|
9
|
+
Use as a *pre-filter* in MemoryPalace.search when codec is binary/rabitq:
|
|
10
|
+
the trie returns O(k) candidate fact_ids within max_wildcards bit-flips
|
|
11
|
+
of the query, then codec.scores ranks them. Trades one big argsort for
|
|
12
|
+
a sub-linear trie walk — wins when N grows large and dims is high (16k+).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import numpy as np
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class _TrieNode:
|
|
21
|
+
__slots__ = ("children", "fact_ids", "depth")
|
|
22
|
+
|
|
23
|
+
def __init__(self, depth: int) -> None:
|
|
24
|
+
self.children: dict[int, _TrieNode] = {} # bit 0/1 → node
|
|
25
|
+
self.fact_ids: list[str] = []
|
|
26
|
+
self.depth = depth
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class TernaryTrie:
|
|
30
|
+
"""Software TLSH over packed binary holograms."""
|
|
31
|
+
|
|
32
|
+
def __init__(self, dims: int, max_wildcards: int = 8,
|
|
33
|
+
max_candidates: int = 256) -> None:
|
|
34
|
+
self.dims = dims
|
|
35
|
+
self.max_wildcards = max_wildcards
|
|
36
|
+
self.max_candidates = max_candidates
|
|
37
|
+
self.root = _TrieNode(0)
|
|
38
|
+
self._size = 0
|
|
39
|
+
|
|
40
|
+
def insert(self, fact_id: str, packed: np.ndarray) -> None:
|
|
41
|
+
"""Insert a packed binary vector (uint8 array of packed bits)."""
|
|
42
|
+
bits = _unpack_bits(packed, self.dims)
|
|
43
|
+
node = self.root
|
|
44
|
+
for bit in bits:
|
|
45
|
+
b = int(bit)
|
|
46
|
+
child = node.children.get(b)
|
|
47
|
+
if child is None:
|
|
48
|
+
child = _TrieNode(node.depth + 1)
|
|
49
|
+
node.children[b] = child
|
|
50
|
+
node = child
|
|
51
|
+
node.fact_ids.append(fact_id)
|
|
52
|
+
self._size += 1
|
|
53
|
+
|
|
54
|
+
def lookup(self, packed_q: np.ndarray, k: int = 10,
|
|
55
|
+
max_wildcards: int | None = None) -> list[tuple[str, int]]:
|
|
56
|
+
"""Return up to k (fact_id, hamming_distance) pairs within
|
|
57
|
+
max_wildcards bit-flips of packed_q. O(log N + max_wildcards·branch).
|
|
58
|
+
"""
|
|
59
|
+
mw = max_wildcards if max_wildcards is not None else self.max_wildcards
|
|
60
|
+
bits = _unpack_bits(packed_q, self.dims)
|
|
61
|
+
# candidate (node, position, wildcards_used, distance_so_far)
|
|
62
|
+
# use a stack with priority by wildcards_used
|
|
63
|
+
results: list[tuple[str, int]] = []
|
|
64
|
+
seen: set[str] = set()
|
|
65
|
+
# DFS with wildcard budget
|
|
66
|
+
# stack entries: (node, idx, wildcards_remaining)
|
|
67
|
+
# we don't strictly bound exploration — for small max_wildcards
|
|
68
|
+
# and a sparse trie this is fine
|
|
69
|
+
stack = [(self.root, 0, mw)]
|
|
70
|
+
# use a heap for best-first with priority on wildcards remaining
|
|
71
|
+
import heapq
|
|
72
|
+
# priority: (-wildcards_remaining, idx) so most wildcards remaining
|
|
73
|
+
# (i.e. least used) comes first
|
|
74
|
+
heap: list[tuple[int, int, _TrieNode]] = [(-mw, 0, self.root)]
|
|
75
|
+
while heap and len(results) < self.max_candidates:
|
|
76
|
+
_, idx, node = heapq.heappop(heap)
|
|
77
|
+
if node.fact_ids and idx == self.dims:
|
|
78
|
+
for fid in node.fact_ids:
|
|
79
|
+
if fid not in seen:
|
|
80
|
+
seen.add(fid)
|
|
81
|
+
results.append((fid, mw + _heap_key(0, mw)))
|
|
82
|
+
if len(results) >= self.max_candidates:
|
|
83
|
+
break
|
|
84
|
+
continue
|
|
85
|
+
if idx >= self.dims:
|
|
86
|
+
# at a leaf but bits remaining — these are stored facts
|
|
87
|
+
for fid in node.fact_ids:
|
|
88
|
+
if fid not in seen:
|
|
89
|
+
seen.add(fid)
|
|
90
|
+
results.append((fid, mw))
|
|
91
|
+
continue
|
|
92
|
+
want_bit = int(bits[idx])
|
|
93
|
+
# exact match: free (no wildcard used)
|
|
94
|
+
exact = node.children.get(want_bit)
|
|
95
|
+
if exact is not None:
|
|
96
|
+
heapq.heappush(heap, (-(mw), idx + 1, exact))
|
|
97
|
+
# wildcard match: try the other bit (uses 1 wildcard)
|
|
98
|
+
other = 1 - want_bit
|
|
99
|
+
wild = node.children.get(other)
|
|
100
|
+
if wild is not None and mw > 0:
|
|
101
|
+
heapq.heappush(heap, (-(mw - 1), idx + 1, wild))
|
|
102
|
+
# dedupe and sort by best distance estimate
|
|
103
|
+
# NB: distance estimate is approximate (lower bound by wildcards used)
|
|
104
|
+
dedup: dict[str, int] = {}
|
|
105
|
+
for fid, dist in results:
|
|
106
|
+
if fid not in dedup or dist < dedup[fid]:
|
|
107
|
+
dedup[fid] = dist
|
|
108
|
+
out = sorted(dedup.items(), key=lambda x: x[1])[:k]
|
|
109
|
+
return out
|
|
110
|
+
|
|
111
|
+
def __len__(self) -> int:
|
|
112
|
+
return self._size
|
|
113
|
+
|
|
114
|
+
def stats(self) -> dict:
|
|
115
|
+
return {
|
|
116
|
+
"dims": self.dims,
|
|
117
|
+
"size": self._size,
|
|
118
|
+
"max_wildcards": self.max_wildcards,
|
|
119
|
+
"root_children": len(self.root.children),
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _heap_key(used: int, budget: int) -> int:
|
|
124
|
+
"""Convert 'wildcards used' into a priority value (less used = better)."""
|
|
125
|
+
return budget - used # remaining
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _unpack_bits(packed: np.ndarray, dims: int) -> np.ndarray:
|
|
129
|
+
"""Unpack packed uint8 bits into a 1D array of 0/1 of length dims."""
|
|
130
|
+
arr = np.atleast_1d(packed)
|
|
131
|
+
if arr.dtype == np.uint8 and len(arr) * 8 >= dims:
|
|
132
|
+
return np.unpackbits(arr, count=dims).astype(np.uint8)
|
|
133
|
+
# already a bit array
|
|
134
|
+
return arr.astype(np.uint8)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
__all__ = ["TernaryTrie"]
|