cortexm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. context_m.py +17 -0
  2. cortexm/__init__.py +45 -0
  3. cortexm/accel.py +403 -0
  4. cortexm/api/__init__.py +0 -0
  5. cortexm/api/chaos.py +118 -0
  6. cortexm/api/memory.py +635 -0
  7. cortexm/bench/__init__.py +0 -0
  8. cortexm/bench/abilities.py +311 -0
  9. cortexm/bench/baselines.py +89 -0
  10. cortexm/bench/beam_loader.py +317 -0
  11. cortexm/bench/generator.py +376 -0
  12. cortexm/bench/harness.py +211 -0
  13. cortexm/bench/messy.py +218 -0
  14. cortexm/bench/micro.py +251 -0
  15. cortexm/bench/ood.py +443 -0
  16. cortexm/bench/run.py +137 -0
  17. cortexm/bridge/__init__.py +0 -0
  18. cortexm/bridge/dates.py +178 -0
  19. cortexm/bridge/decoders.py +204 -0
  20. cortexm/bridge/enrich.py +255 -0
  21. cortexm/bridge/extractor.py +316 -0
  22. cortexm/bridge/fallback.py +332 -0
  23. cortexm/bridge/onnx_runtime.py +158 -0
  24. cortexm/bridge/patterns.py +760 -0
  25. cortexm/bridge/ppr.py +104 -0
  26. cortexm/bridge/prefilter.py +188 -0
  27. cortexm/bridge/query_extract.py +420 -0
  28. cortexm/bridge/reader.py +1174 -0
  29. cortexm/bridge/rerank.py +204 -0
  30. cortexm/bridge/writer.py +492 -0
  31. cortexm/cli.py +295 -0
  32. cortexm/cognition/__init__.py +53 -0
  33. cortexm/cognition/abstraction.py +192 -0
  34. cortexm/cognition/analogy.py +159 -0
  35. cortexm/cognition/engine.py +204 -0
  36. cortexm/cognition/gaps.py +365 -0
  37. cortexm/cognition/scanner.py +204 -0
  38. cortexm/config.py +375 -0
  39. cortexm/cortexm.py +8 -0
  40. cortexm/enterprise/__init__.py +0 -0
  41. cortexm/enterprise/audit.py +178 -0
  42. cortexm/enterprise/governance.py +239 -0
  43. cortexm/errors.py +35 -0
  44. cortexm/features/__init__.py +0 -0
  45. cortexm/features/git.py +204 -0
  46. cortexm/features/prefetch.py +88 -0
  47. cortexm/features/zk.py +105 -0
  48. cortexm/federation/__init__.py +39 -0
  49. cortexm/federation/crdt.py +275 -0
  50. cortexm/federation/fabric.py +109 -0
  51. cortexm/federation/hlc.py +80 -0
  52. cortexm/federation/node.py +145 -0
  53. cortexm/federation/schema_report.py +73 -0
  54. cortexm/federation/transport.py +164 -0
  55. cortexm/index/__init__.py +19 -0
  56. cortexm/index/nsg.py +386 -0
  57. cortexm/mcp/__init__.py +0 -0
  58. cortexm/mcp/server.py +985 -0
  59. cortexm/metrics.py +62 -0
  60. cortexm/migrate/__init__.py +0 -0
  61. cortexm/migrate/importers.py +192 -0
  62. cortexm/provenance/__init__.py +78 -0
  63. cortexm/provenance/agent.py +214 -0
  64. cortexm/provenance/cose.py +201 -0
  65. cortexm/provenance/scitt.py +258 -0
  66. cortexm/provenance/vc.py +250 -0
  67. cortexm/security/__init__.py +0 -0
  68. cortexm/security/crypto.py +162 -0
  69. cortexm/security/hashes.py +140 -0
  70. cortexm/security/injection.py +149 -0
  71. cortexm/security/mind.py +154 -0
  72. cortexm/security/pii.py +265 -0
  73. cortexm/security/rbac.py +169 -0
  74. cortexm/security/sandbox.py +131 -0
  75. cortexm/security/zk_hamming.py +142 -0
  76. cortexm/security/zk_sql.py +485 -0
  77. cortexm/server/__init__.py +0 -0
  78. cortexm/server/metrics.py +88 -0
  79. cortexm/server/rest.py +936 -0
  80. cortexm/server/sparql.py +984 -0
  81. cortexm/text/__init__.py +0 -0
  82. cortexm/text/dissim.py +252 -0
  83. cortexm/text/embedder.py +155 -0
  84. cortexm/text/fuzzy.py +218 -0
  85. cortexm/text/idiolect.py +253 -0
  86. cortexm/text/labse.py +374 -0
  87. cortexm/text/tokenizer.py +79 -0
  88. cortexm/trace/__init__.py +0 -0
  89. cortexm/trace/blob_arena.py +277 -0
  90. cortexm/trace/consolidate.py +337 -0
  91. cortexm/trace/contradictions.py +69 -0
  92. cortexm/trace/dedup.py +114 -0
  93. cortexm/trace/edges.py +214 -0
  94. cortexm/trace/fact.py +121 -0
  95. cortexm/trace/fade.py +245 -0
  96. cortexm/trace/lifecycle.py +112 -0
  97. cortexm/trace/rebuild.py +173 -0
  98. cortexm/trace/rules.py +171 -0
  99. cortexm/trace/store.py +680 -0
  100. cortexm/trace/structural.py +183 -0
  101. cortexm/trace/tmt.py +335 -0
  102. cortexm/util.py +148 -0
  103. cortexm/vsa/__init__.py +0 -0
  104. cortexm/vsa/attribution.py +149 -0
  105. cortexm/vsa/cleanup.py +161 -0
  106. cortexm/vsa/codecs.py +397 -0
  107. cortexm/vsa/hologram_overlay.py +139 -0
  108. cortexm/vsa/index.py +163 -0
  109. cortexm/vsa/ops.py +149 -0
  110. cortexm/vsa/palace.py +446 -0
  111. cortexm/vsa/role_vectors.py +236 -0
  112. cortexm/vsa/slb.py +78 -0
  113. cortexm/vsa/tlsh_trie.py +137 -0
  114. cortexm/vsa/working_memory.py +249 -0
  115. cortexm-0.3.0.dist-info/METADATA +482 -0
  116. cortexm-0.3.0.dist-info/RECORD +120 -0
  117. cortexm-0.3.0.dist-info/WHEEL +5 -0
  118. cortexm-0.3.0.dist-info/entry_points.txt +2 -0
  119. cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
  120. cortexm-0.3.0.dist-info/top_level.txt +2 -0
cortexm/vsa/codecs.py ADDED
@@ -0,0 +1,397 @@
1
+ """Vector codecs — the cortexm-compress tier stack.
2
+
3
+ Per-vector storage at 768 dims (plan appendix "From 768 MB down to 8–96 MB"):
4
+
5
+ codec bytes tier binding/similarity
6
+ ------- ----- ---------- ----------------------------------------
7
+ int8 768 baseline scalar symmetric quantization (Aeon-style)
8
+ binary 96 edge bipolar ±1, packed bits; XOR/permutation
9
+ ops map 1:1 to HDC hardware
10
+ rabitq 96 ultra-edge JL-rotation + binarization (RaBitQ-style),
11
+ provable angle preservation
12
+ pq 8 cloud product quantization M=8 x 8-bit codes,
13
+ ADC scoring via L1-resident lookup tables
14
+
15
+ All codecs expose: encode_packed / decoded / query_vec / scores /
16
+ to_bytes / from_bytes / corrupt, so the Palace and the tree index are
17
+ codec-agnostic. INT8 additionally stores a per-vector float16 scale
18
+ (aux). Binary supports triple-modular redundancy (TMR) for the
19
+ self-healing memory feature.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import numpy as np
25
+
26
+ from cortexm.errors import CodecError
27
+ from cortexm.util import h64 as _h64
28
+
29
+
30
+ def _norm(v: np.ndarray) -> np.ndarray:
31
+ n = float(np.linalg.norm(v))
32
+ return (v / n).astype(np.float32) if n > 0 else v.astype(np.float32)
33
+
34
+
35
+ class BaseCodec:
36
+ name = "base"
37
+
38
+ @property
39
+ def bytes_per_vector(self) -> int:
40
+ raise NotImplementedError
41
+
42
+ def encode_packed(self, vec: np.ndarray) -> np.ndarray: ...
43
+ def decoded(self, packed: np.ndarray) -> np.ndarray: ...
44
+ def query_vec(self, q: np.ndarray) -> np.ndarray: ...
45
+ def scores(self, packed: np.ndarray, q: np.ndarray) -> np.ndarray: ...
46
+ def to_bytes(self, packed_row: np.ndarray) -> bytes: ...
47
+ def from_bytes(self, b: bytes) -> np.ndarray: ...
48
+ def corrupt(self, packed_row: np.ndarray, rate: float,
49
+ rng: np.random.Generator) -> np.ndarray: ...
50
+
51
+
52
+ # ---------------------------------------------------------------------------
53
+ class Int8Codec(BaseCodec):
54
+ """Symmetric scalar quantization: q = round(127 * v / max|v|)."""
55
+
56
+ name = "int8"
57
+ uses_aux = True
58
+
59
+ def __init__(self, dims: int) -> None:
60
+ self.dims = dims
61
+
62
+ @property
63
+ def bytes_per_vector(self) -> int:
64
+ return self.dims + 2 # + float16 scale
65
+
66
+ def encode_packed(self, vec: np.ndarray) -> np.ndarray:
67
+ v = np.asarray(vec, dtype=np.float32)
68
+ m = float(np.max(np.abs(v))) or 1.0
69
+ scale = 127.0 / m
70
+ q = np.clip(np.rint(v * scale), -127, 127).astype(np.int8)
71
+ return q
72
+
73
+ @staticmethod
74
+ def encode_scale(vec: np.ndarray) -> np.float16:
75
+ v = np.asarray(vec, dtype=np.float32)
76
+ m = float(np.max(np.abs(v))) or 1.0
77
+ return np.float16(m / 127.0)
78
+
79
+ def decoded(self, packed: np.ndarray, aux: np.ndarray | None = None) -> np.ndarray:
80
+ arr = np.atleast_2d(packed).astype(np.float32)
81
+ if aux is not None:
82
+ arr = arr * np.atleast_1d(aux).astype(np.float32)[:, None]
83
+ return arr
84
+
85
+ def query_vec(self, q: np.ndarray) -> np.ndarray:
86
+ return np.asarray(q, dtype=np.float32)
87
+
88
+ def scores(self, packed: np.ndarray, q: np.ndarray,
89
+ aux: np.ndarray | None = None) -> np.ndarray:
90
+ if aux is None:
91
+ raise CodecError("int8 scoring requires per-vector aux scales")
92
+ arr = packed.astype(np.float32)
93
+ if len(aux):
94
+ arr = arr * np.atleast_1d(aux).astype(np.float32)[:, None]
95
+ return arr @ np.asarray(q, dtype=np.float32)
96
+
97
+ def to_bytes(self, packed_row: np.ndarray, scale: np.float16) -> bytes:
98
+ return packed_row.tobytes() + np.float16(scale).tobytes()
99
+
100
+ def from_bytes(self, b: bytes) -> tuple[np.ndarray, np.float16]:
101
+ q = np.frombuffer(b[: self.dims], dtype=np.int8).copy()
102
+ scale = np.frombuffer(b[self.dims:], dtype=np.float16)[0]
103
+ return q, scale
104
+
105
+ def corrupt(self, packed_row: np.ndarray, rate: float,
106
+ rng: np.random.Generator) -> np.ndarray:
107
+ n = len(packed_row)
108
+ idx = rng.integers(0, n, size=max(1, int(n * rate)))
109
+ out = packed_row.copy()
110
+ noise = rng.integers(-30, 31, size=len(idx)).astype(np.int8)
111
+ out[idx] = np.clip(out[idx].astype(np.int16) + noise, -127, 127).astype(np.int8)
112
+ return out
113
+
114
+
115
+ # ---------------------------------------------------------------------------
116
+ class BinaryCodec(BaseCodec):
117
+ """Bipolar {±1} hypervectors, 1 bit/dim — the HDC hardware target.
118
+
119
+ Sparse embeddings binarize poorly (near-zero components produce
120
+ correlated sign noise), so a fixed JL rotation densifies energy
121
+ before binarization — the RaBitQ insight applied to the MAP model.
122
+ Set ``rotate=False`` for raw bipolar-MAP semantics.
123
+ """
124
+
125
+ name = "binary"
126
+ uses_aux = False
127
+
128
+ def __init__(self, dims: int, tmr: bool = False, rotate: bool = True,
129
+ seed: int = 0x0C0FFEE) -> None:
130
+ self.dims = dims
131
+ self.words = dims // 8
132
+ self.tmr = tmr
133
+ self.rotate = rotate
134
+ self.R = None
135
+ if rotate:
136
+ rng = np.random.default_rng(_h64("binary-rot", seed) & 0xFFFFFFFF)
137
+ g = rng.standard_normal((dims, dims)).astype(np.float32)
138
+ q, _ = np.linalg.qr(g)
139
+ self.R = q.astype(np.float32)
140
+
141
+ @property
142
+ def bytes_per_vector(self) -> int:
143
+ return self.words * (3 if self.tmr else 1)
144
+
145
+ def encode_packed(self, vec: np.ndarray) -> np.ndarray:
146
+ v = np.asarray(vec, dtype=np.float32)
147
+ if self.R is not None:
148
+ v = self.R @ v
149
+ bits = (v > 0).astype(np.uint8)
150
+ packed = np.packbits(bits)
151
+ if not self.tmr:
152
+ return packed
153
+ return np.concatenate([packed, packed, packed])
154
+
155
+ def decoded(self, packed: np.ndarray) -> np.ndarray:
156
+ arr = np.atleast_2d(packed)
157
+ if self.tmr:
158
+ k = arr.shape[1] // 3
159
+ arr = _tmr_majority(arr[:, :k], arr[:, k:2 * k], arr[:, 2 * k:])
160
+ signs = (np.unpackbits(arr, axis=1, count=self.dims).astype(np.float32) * 2 - 1)
161
+ return signs / np.sqrt(self.dims)
162
+
163
+ def query_vec(self, q: np.ndarray) -> np.ndarray:
164
+ v = np.asarray(q, dtype=np.float32)
165
+ if self.R is not None:
166
+ v = self.R @ v
167
+ s = ((v > 0).astype(np.float32) * 2 - 1) / np.sqrt(self.dims)
168
+ return s
169
+
170
+ def scores(self, packed: np.ndarray, q: np.ndarray) -> np.ndarray:
171
+ v = np.asarray(q, dtype=np.float32)
172
+ if self.R is not None:
173
+ v = self.R @ v
174
+ qbits = np.packbits((v > 0).astype(np.uint8))
175
+ arr = np.atleast_2d(packed)
176
+ if self.tmr:
177
+ k = arr.shape[1] // 3
178
+ arr = _tmr_majority(arr[:, :k], arr[:, k:2 * k], arr[:, 2 * k:])
179
+ x = np.bitwise_xor(arr, qbits[None, :])
180
+ ham = np.bitwise_count(x).sum(axis=1).astype(np.float32)
181
+ return 1.0 - 2.0 * ham / self.dims
182
+
183
+ def to_bytes(self, packed_row: np.ndarray) -> bytes:
184
+ return packed_row.tobytes()
185
+
186
+ def from_bytes(self, b: bytes) -> np.ndarray:
187
+ return np.frombuffer(b, dtype=np.uint8).copy()
188
+
189
+ def corrupt(self, packed_row: np.ndarray, rate: float,
190
+ rng: np.random.Generator) -> np.ndarray:
191
+ out = packed_row.copy()
192
+ n = len(out)
193
+ idx = rng.integers(0, n, size=max(1, int(n * rate)))
194
+ out[idx] ^= rng.integers(1, 256, size=len(idx)).astype(np.uint8)
195
+ return out
196
+
197
+ # TMR helpers -------------------------------------------------------
198
+ def tmr_health(self, packed_row: np.ndarray) -> int:
199
+ """Number of bit positions where the 3 copies disagree (corruption)."""
200
+ k = len(packed_row) // 3
201
+ a, b, c = packed_row[:k], packed_row[k:2 * k], packed_row[2 * k:]
202
+ ab = np.bitwise_count(np.bitwise_xor(a, b)).sum()
203
+ ac = np.bitwise_count(np.bitwise_xor(a, c)).sum()
204
+ bc = np.bitwise_count(np.bitwise_xor(b, c)).sum()
205
+ return int((ab + ac + bc) // 6)
206
+
207
+
208
+ def _tmr_majority(a: np.ndarray, b: np.ndarray, c: np.ndarray) -> np.ndarray:
209
+ """Bit-level majority vote across 3 copies (self-healing memory)."""
210
+ ab = a == b
211
+ bc = b == c
212
+ ac = a == c
213
+ maj = np.where(ab, a, np.where(ac, a, c))
214
+ maj = np.where(ab | bc | ac, maj, a)
215
+ return maj.astype(np.uint8)
216
+
217
+
218
+ # ---------------------------------------------------------------------------
219
+ class RaBitQCodec(BaseCodec):
220
+ """RaBitQ-style: fixed JL rotation, then binarization.
221
+
222
+ The random orthogonal rotation concentrates energy so that sign
223
+ binarization preserves angles far better than raw binarization
224
+ (Wang et al., SIGMOD 2024/2025). Simplified here: unit-norm vectors,
225
+ single global rotation, Hamming-based cosine estimate.
226
+ """
227
+
228
+ name = "rabitq"
229
+ uses_aux = False
230
+
231
+ def __init__(self, dims: int, seed: int = 0x0C0FFEE) -> None:
232
+ self.dims = dims
233
+ self.words = dims // 8
234
+ rng = np.random.default_rng(_h64("rabitq-rot", seed) & 0xFFFFFFFF)
235
+ g = rng.standard_normal((dims, dims)).astype(np.float32)
236
+ q, _ = np.linalg.qr(g)
237
+ self.R = q.astype(np.float32)
238
+
239
+ @property
240
+ def bytes_per_vector(self) -> int:
241
+ return self.words
242
+
243
+ def encode_packed(self, vec: np.ndarray) -> np.ndarray:
244
+ rot = self.R @ np.asarray(vec, dtype=np.float32)
245
+ return np.packbits((rot > 0).astype(np.uint8))
246
+
247
+ def decoded(self, packed: np.ndarray) -> np.ndarray:
248
+ signs = np.unpackbits(np.atleast_2d(packed), axis=1,
249
+ count=self.dims).astype(np.float32) * 2 - 1
250
+ return signs / np.sqrt(self.dims)
251
+
252
+ def query_vec(self, q: np.ndarray) -> np.ndarray:
253
+ rot = self.R @ np.asarray(q, dtype=np.float32)
254
+ s = ((rot > 0).astype(np.float32) * 2 - 1) / np.sqrt(self.dims)
255
+ return s
256
+
257
+ def scores(self, packed: np.ndarray, q: np.ndarray) -> np.ndarray:
258
+ return self.scores_packed(packed, self.query_packed(q))
259
+
260
+ def query_packed(self, q: np.ndarray) -> np.ndarray:
261
+ rot = self.R @ np.asarray(q, dtype=np.float32)
262
+ return np.packbits((rot > 0).astype(np.uint8))
263
+
264
+ def scores_packed(self, packed: np.ndarray, qbits: np.ndarray) -> np.ndarray:
265
+ x = np.bitwise_xor(np.atleast_2d(packed), qbits[None, :])
266
+ ham = np.bitwise_count(x).sum(axis=1).astype(np.float32)
267
+ return 1.0 - 2.0 * ham / self.dims
268
+
269
+ def to_bytes(self, packed_row: np.ndarray) -> bytes:
270
+ return packed_row.tobytes()
271
+
272
+ def from_bytes(self, b: bytes) -> np.ndarray:
273
+ return np.frombuffer(b, dtype=np.uint8).copy()
274
+
275
+ def corrupt(self, packed_row: np.ndarray, rate: float,
276
+ rng: np.random.Generator) -> np.ndarray:
277
+ out = packed_row.copy()
278
+ idx = rng.integers(0, len(out), size=max(1, int(len(out) * rate)))
279
+ out[idx] ^= rng.integers(1, 256, size=len(idx)).astype(np.uint8)
280
+ return out
281
+
282
+ def serialize_rotation(self) -> bytes:
283
+ return self.R.tobytes()
284
+
285
+
286
+ # ---------------------------------------------------------------------------
287
+ class PQCodec(BaseCodec):
288
+ """Product quantization: M subspaces x 8-bit codes (8 bytes/vector).
289
+
290
+ Cloud tier. Codebooks trained on the first N vectors (k-means per
291
+ subspace, frozen afterwards); asymmetric distance computation via
292
+ per-query lookup tables that stay L1-resident.
293
+ """
294
+
295
+ name = "pq"
296
+ uses_aux = False
297
+
298
+ def __init__(self, dims: int, m: int = 8, ks: int = 256,
299
+ train_threshold: int = 4096, seed: int = 0x0C0FFEE) -> None:
300
+ if dims % m:
301
+ raise CodecError("dims must be divisible by M")
302
+ self.dims = dims
303
+ self.m = m
304
+ self.ks = ks
305
+ self.sub = dims // m
306
+ self.seed = seed
307
+ self.train_threshold = train_threshold
308
+ self.codebooks: np.ndarray | None = None # (m, ks, sub)
309
+
310
+ @property
311
+ def trained(self) -> bool:
312
+ return self.codebooks is not None
313
+
314
+ @property
315
+ def bytes_per_vector(self) -> int:
316
+ return self.m
317
+
318
+ def train(self, vecs: np.ndarray) -> None:
319
+ from scipy.cluster.vq import kmeans2
320
+ rng = np.random.default_rng(self.seed)
321
+ data = np.asarray(vecs, dtype=np.float32)
322
+ n = min(self.ks * 4, len(data))
323
+ sample = data[rng.choice(len(data), size=n, replace=False)] if len(data) > n else data
324
+ books = np.zeros((self.m, self.ks, self.sub), dtype=np.float32)
325
+ for mi in range(self.m):
326
+ chunk = sample[:, mi * self.sub:(mi + 1) * self.sub]
327
+ k = min(self.ks, max(2, len(np.unique(chunk, axis=0))))
328
+ try:
329
+ cent, _ = kmeans2(chunk, k, iter=12, minit="++", seed=self.seed + mi)
330
+ except Exception:
331
+ cent = chunk[:k]
332
+ books[mi, : len(cent)] = cent
333
+ self.codebooks = books
334
+
335
+ def set_codebooks(self, books: np.ndarray) -> None:
336
+ self.codebooks = np.asarray(books, dtype=np.float32)
337
+
338
+ def encode_packed(self, vec: np.ndarray) -> np.ndarray:
339
+ if self.codebooks is None:
340
+ raise CodecError("PQ codec not trained")
341
+ v = np.asarray(vec, dtype=np.float32)
342
+ codes = np.zeros(self.m, dtype=np.uint8)
343
+ for mi in range(self.m):
344
+ chunk = v[mi * self.sub:(mi + 1) * self.sub]
345
+ sims = self.codebooks[mi] @ chunk
346
+ codes[mi] = int(np.argmax(sims))
347
+ return codes
348
+
349
+ def decoded(self, packed: np.ndarray) -> np.ndarray:
350
+ if self.codebooks is None:
351
+ raise CodecError("PQ codec not trained")
352
+ arr = np.atleast_2d(packed)
353
+ out = np.zeros((arr.shape[0], self.dims), dtype=np.float32)
354
+ for mi in range(self.m):
355
+ out[:, mi * self.sub:(mi + 1) * self.sub] = self.codebooks[mi][arr[:, mi]]
356
+ return out
357
+
358
+ def query_vec(self, q: np.ndarray) -> np.ndarray:
359
+ return np.asarray(q, dtype=np.float32)
360
+
361
+ def scores(self, packed: np.ndarray, q: np.ndarray) -> np.ndarray:
362
+ if self.codebooks is None:
363
+ raise CodecError("PQ codec not trained")
364
+ arr = np.atleast_2d(packed)
365
+ q = np.asarray(q, dtype=np.float32)
366
+ total = np.zeros(arr.shape[0], dtype=np.float32)
367
+ for mi in range(self.m):
368
+ qsub = q[mi * self.sub:(mi + 1) * self.sub]
369
+ lut = self.codebooks[mi] @ qsub # (ks,)
370
+ total += lut[arr[:, mi]]
371
+ return total
372
+
373
+ def to_bytes(self, packed_row: np.ndarray) -> bytes:
374
+ return packed_row.tobytes()
375
+
376
+ def from_bytes(self, b: bytes) -> np.ndarray:
377
+ return np.frombuffer(b, dtype=np.uint8).copy()
378
+
379
+ def corrupt(self, packed_row: np.ndarray, rate: float,
380
+ rng: np.random.Generator) -> np.ndarray:
381
+ out = packed_row.copy()
382
+ idx = rng.integers(0, len(out), size=max(1, int(len(out) * rate)))
383
+ out[idx] = rng.integers(0, self.ks, size=len(idx)).astype(np.uint8)
384
+ return out
385
+
386
+
387
+ def make_codec(name: str, dims: int, seed: int = 0x0C0FFEE,
388
+ tmr: bool = False, **kw) -> BaseCodec:
389
+ if name == "int8":
390
+ return Int8Codec(dims)
391
+ if name == "binary":
392
+ return BinaryCodec(dims, tmr=tmr, seed=seed)
393
+ if name == "rabitq":
394
+ return RaBitQCodec(dims, seed=seed)
395
+ if name == "pq":
396
+ return PQCodec(dims, seed=seed, **kw)
397
+ raise CodecError(f"unknown codec {name!r}")
@@ -0,0 +1,139 @@
1
+ """HRR KG overlay — superposed holographic fact store for O(1) lookup.
2
+
3
+ The Holographic Memory for Knowledge Graphs paper (arXiv:2606.24948,
4
+ 2026) shows that facts bound as (s, r, v) can be superposed into a
5
+ single dense hologram M, and a query (s, r, ?) answered by unbinding
6
+ the V slot from M:
7
+
8
+ M = Σ_f bind_V(v_f) * (bind_S(s_f) + bind_R(r_f)) # superposition
9
+ v_hat = unbind_V(M) → noisy residual ≈ Σ_f (bind_S(s_f) + bind_R(r_f)) * v_f
10
+ cleanup(v_hat) → snapped to nearest stored value vector
11
+
12
+ Capacity bound: ~dims^2 / ln(dims) facts before saturation (Clarkson
13
+ 2023). At d=768 → ~93k facts; at d=16,384 → ~31M facts.
14
+
15
+ Best used as a per-scope overlay on top of MemoryPalace — single-hop
16
+ queries hit the overlay (O(1) + cleanup), multi-hop queries fall back
17
+ to TreeIndex search. The overlay trades noise accumulation for lookup
18
+ speed; saturation is detected by signal-to-noise ratio on the cleanup
19
+ step.
20
+
21
+ Pure numpy. Reuses VSA.bind/unbind and HopfieldCleanup.recall.
22
+
23
+ arxiv research: arXiv:2606.24948 (2026 holographic KG); HolE AAAI 2016
24
+ (Nickel, Rosasco, Poggio); Plate 1995 (HRR).
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import numpy as np
30
+
31
+ from cortexm.vsa.cleanup import HopfieldCleanup
32
+ from cortexm.vsa.ops import VSA
33
+
34
+
35
+ def _norm(v: np.ndarray) -> np.ndarray:
36
+ n = float(np.linalg.norm(v))
37
+ return v / n if n > 0 else v
38
+
39
+
40
+ class HolographicFactOverlay:
41
+ """Per-scope superposed hologram for O(1) single-hop fact lookup.
42
+
43
+ Each (user_id, scope) tuple gets one dense hologram M. Adding a fact
44
+ binds (s, r, v) and adds to M. Querying unbinds the requested role
45
+ and snaps to the nearest stored item via Hopfield cleanup.
46
+
47
+ Falls back gracefully when M saturates — caller can switch to
48
+ MemoryPalace.search for the affected scope.
49
+ """
50
+
51
+ def __init__(self, vsa: VSA, cleanup: HopfieldCleanup,
52
+ saturate_threshold: float = 0.55) -> None:
53
+ self.vsa = vsa
54
+ self.cleanup = cleanup
55
+ self.saturate_threshold = saturate_threshold
56
+ self._M: dict[tuple, np.ndarray] = {} # scope → hologram
57
+ self._counts: dict[tuple, int] = {} # scope → fact count
58
+ self._saturated: set[tuple] = set()
59
+
60
+ def add_fact(self, scope: tuple, s_vec: np.ndarray,
61
+ r_vec: np.ndarray, v_vec: np.ndarray) -> None:
62
+ """Add a fact (s, r, v) into the superposed hologram."""
63
+ # bind each role to its filler
64
+ bound = (self.vsa.bind("S", s_vec)
65
+ + self.vsa.bind("R", r_vec)
66
+ + self.vsa.bind("V", v_vec))
67
+ bound = _norm(bound)
68
+ h = self._M.get(scope)
69
+ if h is None:
70
+ self._M[scope] = bound.copy()
71
+ self._counts[scope] = 1
72
+ else:
73
+ self._M[scope] = _norm(h + bound)
74
+ self._counts[scope] = self._counts.get(scope, 0) + 1
75
+ # populate cleanup codebook with each filler
76
+ self.cleanup.add(f"s:{_hash_vec(s_vec)}", s_vec)
77
+ self.cleanup.add(f"r:{_hash_vec(r_vec)}", r_vec)
78
+ self.cleanup.add(f"v:{_hash_vec(v_vec)}", v_vec)
79
+
80
+ def query(self, scope: tuple, query_vec: np.ndarray,
81
+ target_role: str = "V",
82
+ fallback_embs: list[tuple[str, np.ndarray]] | None = None
83
+ ) -> tuple[str | None, float]:
84
+ """Query (s, r, ?) — unbind target_role and cleanup the residual.
85
+
86
+ Returns (item_key, confidence). Confidence below saturate_threshold
87
+ indicates either saturation or miss; caller should fall back to
88
+ MemoryPalace.search.
89
+
90
+ If fallback_embs is provided, those (key, vec) pairs are temporarily
91
+ added to the cleanup codebook before recall — useful for cross-
92
+ scope queries where the answer may be a value vector not yet in
93
+ the codebook.
94
+ """
95
+ M = self._M.get(scope)
96
+ if M is None:
97
+ return None, 0.0
98
+ if scope in self._saturated:
99
+ # signal saturation — caller should fallback to TreeIndex
100
+ return None, 0.0
101
+ # unbind the target role from M to get a noisy residual
102
+ residual = self.vsa.unbind(target_role, M)
103
+ # add fallback embeddings to cleanup if provided
104
+ if fallback_embs:
105
+ for k, v in fallback_embs:
106
+ self.cleanup.add(k, v)
107
+ self.cleanup.build()
108
+ item_key, conf = self.cleanup.recall(residual)
109
+ if conf < self.saturate_threshold:
110
+ # either miss or saturation — mark saturated if we have many
111
+ # facts and conf is consistently low
112
+ if self._counts.get(scope, 0) > 1000:
113
+ self._saturated.add(scope)
114
+ return None, conf
115
+ return item_key, conf
116
+
117
+ def stats(self) -> dict:
118
+ return {
119
+ "scopes": len(self._M),
120
+ "total_facts": sum(self._counts.values()),
121
+ "saturated_scopes": len(self._saturated),
122
+ "cleanup": self.cleanup.stats(),
123
+ }
124
+
125
+ def reset_scope(self, scope: tuple) -> None:
126
+ """Drop a saturated scope and let it rebuild from new adds."""
127
+ self._M.pop(scope, None)
128
+ self._counts.pop(scope, None)
129
+ self._saturated.discard(scope)
130
+
131
+
132
+ def _hash_vec(v: np.ndarray) -> str:
133
+ """Stable hash for codebook keys."""
134
+ import hashlib
135
+ h = hashlib.blake2b(v.tobytes(), digest_size=8).hexdigest()
136
+ return h[:16]
137
+
138
+
139
+ __all__ = ["HolographicFactOverlay"]
cortexm/vsa/index.py ADDED
@@ -0,0 +1,163 @@
1
+ """Page-Clustered Vector Index — hierarchical tree, O(log N) retrieval.
2
+
3
+ Per the plan (Aeon-style): leaf pages hold up to ``leaf_size`` quantized
4
+ vectors; internal nodes store an fp32 centroid + radius (1 - min cos).
5
+ Search is best-first over the bound ``centroid_sim - radius`` with exact
6
+ scoring inside visited pages only — sub-millisecond at 100K+ vectors
7
+ while brute-force scans the whole palace.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import heapq
13
+ import numpy as np
14
+
15
+ from cortexm.vsa.codecs import BaseCodec
16
+
17
+
18
+ class _Node:
19
+ __slots__ = ("centroid", "radius", "children", "rows", "is_leaf")
20
+
21
+ def __init__(self) -> None:
22
+ self.centroid: np.ndarray | None = None
23
+ self.radius: float = 1.0
24
+ self.children: list["_Node"] = []
25
+ self.rows: np.ndarray | None = None
26
+ self.is_leaf = True
27
+
28
+
29
+ class TreeIndex:
30
+ def __init__(self, codec: BaseCodec, rows_getter, n: int, *,
31
+ branch: int = 8, leaf: int = 512, seed: int = 0x0C0FFEE,
32
+ sample_limit: int = 20000) -> None:
33
+ """``rows_getter(indices) -> (packed, aux)`` fetches codec rows."""
34
+ self.codec = codec
35
+ self.get_rows = rows_getter
36
+ self.n = n
37
+ self.branch = branch
38
+ self.leaf = leaf
39
+ self.seed = seed
40
+ self.sample_limit = sample_limit
41
+ self.root: _Node | None = None
42
+ self.leaf_rows_scanned = 0
43
+
44
+ # ------------------------------------------------------------- build
45
+ def build(self) -> None:
46
+ if self.n == 0:
47
+ return
48
+ rng = np.random.default_rng(self.seed)
49
+ self.root = self._build(np.arange(self.n, dtype=np.int64), rng, depth=0)
50
+
51
+ def _decode(self, idx: np.ndarray):
52
+ packed, aux = self.get_rows(idx)
53
+ return self.codec.decoded(packed, aux) if self.codec.uses_aux else self.codec.decoded(packed)
54
+
55
+ def _build(self, idx: np.ndarray, rng: np.random.Generator, depth: int) -> _Node:
56
+ node = _Node()
57
+ # sample for centroid + kmeans
58
+ take = min(len(idx), self.sample_limit if depth == 0 else 4096)
59
+ sample_idx = idx if len(idx) <= take else idx[rng.choice(len(idx), take, replace=False)]
60
+ sample = self._decode(sample_idx).astype(np.float32)
61
+ centroid = sample.mean(axis=0)
62
+ cn = float(np.linalg.norm(centroid))
63
+ centroid = centroid / cn if cn > 0 else centroid
64
+ node.centroid = centroid
65
+ # EXACT radius w.r.t. this node's own centroid (streamed over all rows)
66
+ max_dist = 0.0
67
+ for i0 in range(0, len(idx), 8192):
68
+ batch = self._decode(idx[i0:i0 + 8192]).astype(np.float32)
69
+ sims = batch @ centroid
70
+ if len(sims):
71
+ max_dist = max(max_dist, float(1.0 - sims.min()))
72
+ node.radius = max_dist + 1e-6
73
+ if len(idx) <= self.leaf or depth > 24:
74
+ node.rows = idx
75
+ node.is_leaf = True
76
+ return node
77
+ k = min(self.branch, len(idx))
78
+ cent = _kmeans(sample, k, rng)
79
+ # assign all rows (streamed)
80
+ parts: list[list[int]] = [[] for _ in range(k)]
81
+ for i0 in range(0, len(idx), 8192):
82
+ batch_idx = idx[i0:i0 + 8192]
83
+ batch = self._decode(batch_idx).astype(np.float32)
84
+ sims = batch @ cent.T # (b, k)
85
+ assign = np.argmax(sims, axis=1)
86
+ for j, a in enumerate(assign):
87
+ parts[int(a)].append(int(batch_idx[j]))
88
+ node.is_leaf = False
89
+ for p in parts:
90
+ if p:
91
+ node.children.append(
92
+ self._build(np.array(p, dtype=np.int64), rng, depth + 1))
93
+ if len(node.children) == 1:
94
+ return node.children[0]
95
+ return node
96
+
97
+ # ------------------------------------------------------------ search
98
+ def search(self, q: np.ndarray, k: int, beam: int = 4) -> tuple[np.ndarray, np.ndarray]:
99
+ """Return (row_indices, scores) of approximate top-k."""
100
+ if self.root is None or self.n == 0:
101
+ return np.array([], dtype=np.int64), np.array([], dtype=np.float32)
102
+ qv = self.codec.query_vec(q)
103
+ heap: list[tuple[float, int, _Node]] = []
104
+ counter = 0
105
+ csim = float(qv @ self.root.centroid)
106
+ heapq.heappush(heap, (-(csim - self.root.radius), counter, self.root))
107
+ best: list[tuple[float, int]] = [] # (score, row)
108
+ visited_leaves = 0
109
+ while heap:
110
+ negbound, _, node = heapq.heappop(heap)
111
+ bound = -negbound
112
+ if len(best) >= k and bound <= best[0][0] and visited_leaves >= beam:
113
+ break
114
+ if node.is_leaf:
115
+ rows = node.rows
116
+ packed, aux = self.get_rows(rows)
117
+ sc = (self.codec.scores(packed, q, aux) if self.codec.uses_aux
118
+ else self.codec.scores(packed, q))
119
+ for r, s in zip(rows.tolist(), sc.tolist()):
120
+ if len(best) < k:
121
+ heapq.heappush(best, (float(s), int(r)))
122
+ elif float(s) > best[0][0]:
123
+ heapq.heapreplace(best, (float(s), int(r)))
124
+ visited_leaves += 1
125
+ self.leaf_rows_scanned += len(rows)
126
+ if visited_leaves >= max(beam, 1) * 8:
127
+ break
128
+ else:
129
+ for ch in node.children:
130
+ c = float(qv @ ch.centroid)
131
+ heapq.heappush(heap, (-(c - ch.radius), counter := counter + 1, ch))
132
+ order = sorted(best, key=lambda t: -t[0])
133
+ return (np.array([r for _, r in order], dtype=np.int64),
134
+ np.array([s for s, _ in order], dtype=np.float32))
135
+
136
+
137
+ def _kmeans(data: np.ndarray, k: int, rng: np.random.Generator,
138
+ iters: int = 8) -> np.ndarray:
139
+ """Small deterministic k-means (kmeans++ style init)."""
140
+ n = len(data)
141
+ if n <= k:
142
+ return data.copy()
143
+ # kmeans++ init
144
+ cent = [data[rng.integers(n)]]
145
+ d = 1.0 - data @ cent[0]
146
+ d = np.maximum(d, 1e-8)
147
+ for _ in range(1, k):
148
+ probs = d / d.sum()
149
+ cent.append(data[rng.choice(n, p=probs)])
150
+ d = np.minimum(d, np.maximum(1.0 - data @ cent[-1], 1e-8))
151
+ C = np.stack(cent).astype(np.float32)
152
+ for _ in range(iters):
153
+ sims = data @ C.T
154
+ assign = np.argmax(sims, axis=1)
155
+ for j in range(k):
156
+ mask = assign == j
157
+ if mask.any():
158
+ C[j] = data[mask].mean(axis=0)
159
+ else:
160
+ C[j] = data[rng.integers(n)]
161
+ cn = np.linalg.norm(C, axis=1, keepdims=True)
162
+ C = C / np.maximum(cn, 1e-8)
163
+ return C