cortexm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. context_m.py +17 -0
  2. cortexm/__init__.py +45 -0
  3. cortexm/accel.py +403 -0
  4. cortexm/api/__init__.py +0 -0
  5. cortexm/api/chaos.py +118 -0
  6. cortexm/api/memory.py +635 -0
  7. cortexm/bench/__init__.py +0 -0
  8. cortexm/bench/abilities.py +311 -0
  9. cortexm/bench/baselines.py +89 -0
  10. cortexm/bench/beam_loader.py +317 -0
  11. cortexm/bench/generator.py +376 -0
  12. cortexm/bench/harness.py +211 -0
  13. cortexm/bench/messy.py +218 -0
  14. cortexm/bench/micro.py +251 -0
  15. cortexm/bench/ood.py +443 -0
  16. cortexm/bench/run.py +137 -0
  17. cortexm/bridge/__init__.py +0 -0
  18. cortexm/bridge/dates.py +178 -0
  19. cortexm/bridge/decoders.py +204 -0
  20. cortexm/bridge/enrich.py +255 -0
  21. cortexm/bridge/extractor.py +316 -0
  22. cortexm/bridge/fallback.py +332 -0
  23. cortexm/bridge/onnx_runtime.py +158 -0
  24. cortexm/bridge/patterns.py +760 -0
  25. cortexm/bridge/ppr.py +104 -0
  26. cortexm/bridge/prefilter.py +188 -0
  27. cortexm/bridge/query_extract.py +420 -0
  28. cortexm/bridge/reader.py +1174 -0
  29. cortexm/bridge/rerank.py +204 -0
  30. cortexm/bridge/writer.py +492 -0
  31. cortexm/cli.py +295 -0
  32. cortexm/cognition/__init__.py +53 -0
  33. cortexm/cognition/abstraction.py +192 -0
  34. cortexm/cognition/analogy.py +159 -0
  35. cortexm/cognition/engine.py +204 -0
  36. cortexm/cognition/gaps.py +365 -0
  37. cortexm/cognition/scanner.py +204 -0
  38. cortexm/config.py +375 -0
  39. cortexm/cortexm.py +8 -0
  40. cortexm/enterprise/__init__.py +0 -0
  41. cortexm/enterprise/audit.py +178 -0
  42. cortexm/enterprise/governance.py +239 -0
  43. cortexm/errors.py +35 -0
  44. cortexm/features/__init__.py +0 -0
  45. cortexm/features/git.py +204 -0
  46. cortexm/features/prefetch.py +88 -0
  47. cortexm/features/zk.py +105 -0
  48. cortexm/federation/__init__.py +39 -0
  49. cortexm/federation/crdt.py +275 -0
  50. cortexm/federation/fabric.py +109 -0
  51. cortexm/federation/hlc.py +80 -0
  52. cortexm/federation/node.py +145 -0
  53. cortexm/federation/schema_report.py +73 -0
  54. cortexm/federation/transport.py +164 -0
  55. cortexm/index/__init__.py +19 -0
  56. cortexm/index/nsg.py +386 -0
  57. cortexm/mcp/__init__.py +0 -0
  58. cortexm/mcp/server.py +985 -0
  59. cortexm/metrics.py +62 -0
  60. cortexm/migrate/__init__.py +0 -0
  61. cortexm/migrate/importers.py +192 -0
  62. cortexm/provenance/__init__.py +78 -0
  63. cortexm/provenance/agent.py +214 -0
  64. cortexm/provenance/cose.py +201 -0
  65. cortexm/provenance/scitt.py +258 -0
  66. cortexm/provenance/vc.py +250 -0
  67. cortexm/security/__init__.py +0 -0
  68. cortexm/security/crypto.py +162 -0
  69. cortexm/security/hashes.py +140 -0
  70. cortexm/security/injection.py +149 -0
  71. cortexm/security/mind.py +154 -0
  72. cortexm/security/pii.py +265 -0
  73. cortexm/security/rbac.py +169 -0
  74. cortexm/security/sandbox.py +131 -0
  75. cortexm/security/zk_hamming.py +142 -0
  76. cortexm/security/zk_sql.py +485 -0
  77. cortexm/server/__init__.py +0 -0
  78. cortexm/server/metrics.py +88 -0
  79. cortexm/server/rest.py +936 -0
  80. cortexm/server/sparql.py +984 -0
  81. cortexm/text/__init__.py +0 -0
  82. cortexm/text/dissim.py +252 -0
  83. cortexm/text/embedder.py +155 -0
  84. cortexm/text/fuzzy.py +218 -0
  85. cortexm/text/idiolect.py +253 -0
  86. cortexm/text/labse.py +374 -0
  87. cortexm/text/tokenizer.py +79 -0
  88. cortexm/trace/__init__.py +0 -0
  89. cortexm/trace/blob_arena.py +277 -0
  90. cortexm/trace/consolidate.py +337 -0
  91. cortexm/trace/contradictions.py +69 -0
  92. cortexm/trace/dedup.py +114 -0
  93. cortexm/trace/edges.py +214 -0
  94. cortexm/trace/fact.py +121 -0
  95. cortexm/trace/fade.py +245 -0
  96. cortexm/trace/lifecycle.py +112 -0
  97. cortexm/trace/rebuild.py +173 -0
  98. cortexm/trace/rules.py +171 -0
  99. cortexm/trace/store.py +680 -0
  100. cortexm/trace/structural.py +183 -0
  101. cortexm/trace/tmt.py +335 -0
  102. cortexm/util.py +148 -0
  103. cortexm/vsa/__init__.py +0 -0
  104. cortexm/vsa/attribution.py +149 -0
  105. cortexm/vsa/cleanup.py +161 -0
  106. cortexm/vsa/codecs.py +397 -0
  107. cortexm/vsa/hologram_overlay.py +139 -0
  108. cortexm/vsa/index.py +163 -0
  109. cortexm/vsa/ops.py +149 -0
  110. cortexm/vsa/palace.py +446 -0
  111. cortexm/vsa/role_vectors.py +236 -0
  112. cortexm/vsa/slb.py +78 -0
  113. cortexm/vsa/tlsh_trie.py +137 -0
  114. cortexm/vsa/working_memory.py +249 -0
  115. cortexm-0.3.0.dist-info/METADATA +482 -0
  116. cortexm-0.3.0.dist-info/RECORD +120 -0
  117. cortexm-0.3.0.dist-info/WHEEL +5 -0
  118. cortexm-0.3.0.dist-info/entry_points.txt +2 -0
  119. cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
  120. cortexm-0.3.0.dist-info/top_level.txt +2 -0
cortexm/vsa/ops.py ADDED
@@ -0,0 +1,149 @@
1
+ """Vector Symbolic Architecture algebra.
2
+
3
+ Modes (config ``vsa_mode``):
4
+ * ``perm`` — permutation binding (default). bind(role, filler) permutes
5
+ the filler with a role-seeded permutation. Similarity-preserving,
6
+ cheap (index shuffle), and directly portable to binary HDC
7
+ hardware (XOR/permutation ops) per the plan's edge roadmap.
8
+ * ``conv`` — Holographic Reduced Representations (Plate 1995):
9
+ circular-convolution binding via FFT, involution-based unbinding.
10
+ * ``bag`` — role-weighted superposition only (ablation baseline).
11
+
12
+ Every fact hologram = role-bound components + λ-weighted lexical
13
+ superposition. The λ term keeps free-text queries effective while the
14
+ bound terms carry structure for probe queries.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import numpy as np
20
+
21
+ from cortexm.util import h64 as _h64 # deterministic seeded hashing
22
+
23
+ ROLE_WEIGHTS = {"S": 1.0, "R": 1.0, "V": 1.0}
24
+
25
+
26
+ def _norm(v: np.ndarray) -> np.ndarray:
27
+ n = float(np.linalg.norm(v))
28
+ return v / n if n > 0 else v
29
+
30
+
31
+ class VSA:
32
+ def __init__(self, dims: int = 768, mode: str = "perm",
33
+ seed: int = 0x0C0FFEE, lexical_lambda: float = 0.6) -> None:
34
+ self.dims = dims
35
+ self.mode = mode
36
+ self.seed = seed
37
+ self.lam = lexical_lambda
38
+ self._perms: dict[str, np.ndarray] = {}
39
+ self._inperms: dict[str, np.ndarray] = {}
40
+ self._roles: dict[str, np.ndarray] = {}
41
+ # NSR-inspired engineered role vectors (optional). When
42
+ # _engineered is set, role_vec() returns the engineered vector
43
+ # instead of the random one. This is opt-in via .use_engineered().
44
+ self._engineered = None
45
+
46
+ def use_engineered(self, erv) -> None:
47
+ """Swap the random role vectors for the engineered ones.
48
+
49
+ erv: an EngineeredRoleVectors instance that has been .fit()
50
+ on the actual fact corpus. After this call:
51
+ role_vec("S") -> erv.role_vec("S") (top-1 principal dir)
52
+ role_vec("R") -> erv.role_vec("R") (top-2)
53
+ role_vec("V") -> erv.role_vec("V") (top-3)
54
+ The existing random role vectors remain available as a fallback
55
+ if erv.role_vec(role) returns None.
56
+ """
57
+ self._engineered = erv
58
+
59
+ # ------------------------------------------------------------- roles
60
+ def perm(self, role: str) -> np.ndarray:
61
+ p = self._perms.get(role)
62
+ if p is None:
63
+ rng = np.random.default_rng(_h64(f"perm:{role}", self.seed) & 0xFFFFFFFF)
64
+ p = rng.permutation(self.dims).astype(np.int32)
65
+ self._perms[role] = p
66
+ inv = np.empty_like(p)
67
+ inv[p] = np.arange(self.dims, dtype=np.int32)
68
+ self._inperms[role] = inv
69
+ return p
70
+
71
+ def inv_perm(self, role: str) -> np.ndarray:
72
+ self.perm(role)
73
+ return self._inperms[role]
74
+
75
+ def role_vec(self, role: str) -> np.ndarray:
76
+ # engineered role vectors (NSR-inspired) take precedence when
77
+ # configured — they sit on the top-k principal directions of
78
+ # the actual fact corpus, giving higher effective capacity and
79
+ # lower cross-talk than random role vectors.
80
+ if self._engineered is not None and self._engineered.is_fit:
81
+ v = self._engineered.role_vec(role)
82
+ if v is not None:
83
+ return v
84
+ v = self._roles.get(role)
85
+ if v is None:
86
+ rng = np.random.default_rng(_h64(f"role:{role}", self.seed) & 0xFFFFFFFF)
87
+ v = rng.standard_normal(self.dims).astype(np.float32)
88
+ v /= max(float(np.linalg.norm(v)), 1e-9)
89
+ self._roles[role] = v
90
+ return v
91
+
92
+ # ------------------------------------------------------- binding ops
93
+ def bind(self, role: str, filler: np.ndarray) -> np.ndarray:
94
+ if self.mode == "perm":
95
+ return filler[self.perm(role)]
96
+ if self.mode == "conv":
97
+ a = np.fft.rfft(filler)
98
+ b = np.fft.rfft(self.role_vec(role))
99
+ return np.fft.irfft(a * b, n=self.dims).astype(np.float32)
100
+ return (ROLE_WEIGHTS.get(role, 1.0) * filler).astype(np.float32)
101
+
102
+ def unbind(self, role: str, h: np.ndarray) -> np.ndarray:
103
+ if self.mode == "perm":
104
+ return h[self.inv_perm(role)]
105
+ if self.mode == "conv":
106
+ r = self.role_vec(role)
107
+ inv = np.concatenate(([r[0]], r[1:][::-1])) # involution
108
+ a = np.fft.rfft(h)
109
+ b = np.fft.rfft(inv)
110
+ return np.fft.irfft(a * b, n=self.dims).astype(np.float32)
111
+ return h / ROLE_WEIGHTS.get(role, 1.0)
112
+
113
+ @staticmethod
114
+ def bundle(vecs: list[np.ndarray]) -> np.ndarray:
115
+ return _norm(np.sum(vecs, axis=0).astype(np.float32))
116
+
117
+ # ---------------------------------------------------- fact encoding
118
+ def encode_fact(self, s_vec: np.ndarray, r_vec: np.ndarray,
119
+ v_vec: np.ndarray) -> np.ndarray:
120
+ lex = _norm(s_vec + r_vec + v_vec)
121
+ if self.mode == "bag":
122
+ return _norm(ROLE_WEIGHTS["S"] * s_vec + ROLE_WEIGHTS["R"] * r_vec
123
+ + ROLE_WEIGHTS["V"] * v_vec)
124
+ bound = self.bundle([
125
+ self.bind("S", s_vec), self.bind("R", r_vec), self.bind("V", v_vec)])
126
+ return _norm(bound + self.lam * lex)
127
+
128
+ def probe(self, s_vec: np.ndarray | None = None,
129
+ r_vec: np.ndarray | None = None,
130
+ v_vec: np.ndarray | None = None,
131
+ lexical: np.ndarray | None = None) -> np.ndarray:
132
+ """Structured probe: approximate fact hologram from known roles."""
133
+ parts: list[np.ndarray] = []
134
+ if s_vec is not None:
135
+ parts.append(self.bind("S", s_vec))
136
+ if r_vec is not None:
137
+ parts.append(self.bind("R", r_vec))
138
+ if v_vec is not None:
139
+ parts.append(self.bind("V", v_vec))
140
+ if not parts:
141
+ return _norm(lexical if lexical is not None else np.zeros(self.dims, np.float32))
142
+ probe = self.bundle(parts)
143
+ if lexical is not None and self.mode != "bag":
144
+ probe = _norm(probe + self.lam * _norm(lexical))
145
+ return probe
146
+
147
+ def unbind_role(self, h: np.ndarray, role: str) -> np.ndarray:
148
+ """Extract the approximate filler bound to ``role`` (audit path)."""
149
+ return _norm(self.unbind(role, h))
cortexm/vsa/palace.py ADDED
@@ -0,0 +1,446 @@
1
+ """The Memory Palace — VSA hologram store with codec-quantized vectors.
2
+
3
+ One hologram per fact: role-bound subject/relation/value fillers plus a
4
+ λ-weighted lexical superposition (see vsa.ops). Storage is codec-
5
+ quantized (INT8 / Binary / RaBitQ / PQ), persisted as BLOBs in SQLite
6
+ and mirrored in RAM as packed numpy matrices for microsecond scoring.
7
+ A page-clustered tree index provides O(log N) retrieval once the
8
+ collection crosses ``index_threshold``; below it, flat scan wins.
9
+
10
+ Also hosts the self-healing machinery: per-record hash checks, TMR
11
+ majority vote (binary codec), and re-encoding from the symbolic Trace
12
+ when corruption exceeds the correction radius.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import base64
18
+ import numpy as np
19
+
20
+ from cortexm.config import Config
21
+ from cortexm.errors import CodecError, StoreError
22
+ from cortexm.text.embedder import HashingEmbedder
23
+ from cortexm.trace.fact import Fact
24
+ from cortexm.trace.store import TraceStore
25
+ from cortexm.util import iso
26
+ import datetime as _dt
27
+ from cortexm.vsa.codecs import make_codec, PQCodec, Int8Codec
28
+ from cortexm.vsa.index import TreeIndex
29
+ from cortexm.vsa.ops import VSA
30
+
31
+ VEC_TABLE = """
32
+ CREATE TABLE IF NOT EXISTS vectors (
33
+ fact_id TEXT PRIMARY KEY, record BLOB NOT NULL,
34
+ vec_hash TEXT DEFAULT '', ts TEXT DEFAULT ''
35
+ )
36
+ """
37
+
38
+
39
+ class MemoryPalace:
40
+ def __init__(self, config: Config, store: TraceStore) -> None:
41
+ self.cfg = config
42
+ self.store = store
43
+ self.dims = config.dims
44
+ self.hasher = store.hasher
45
+ self.codec = make_codec(config.codec, config.dims, config.seed,
46
+ tmr=config.tmr)
47
+ self.vsa = VSA(config.dims, config.vsa_mode, config.seed,
48
+ config.lexical_lambda)
49
+ self.embedder = HashingEmbedder(config.dims, config.seed)
50
+ store.conn.execute(VEC_TABLE)
51
+ store.conn.commit()
52
+
53
+ self._ids: list[str] = []
54
+ self._id2row: dict[str, int] = {}
55
+ self._n = 0
56
+ self._cap = 0
57
+ self._packed: np.ndarray | None = None
58
+ self._aux: np.ndarray | None = None
59
+ self._index: TreeIndex | None = None
60
+ self._index_valid = False
61
+ self._pq_buffer: list[tuple[str, np.ndarray]] = []
62
+ self.searches_flat = 0
63
+ self.searches_indexed = 0
64
+ self._load_pq_codebooks()
65
+ self._load()
66
+
67
+ # ------------------------------------------------------------- loading
68
+ def _load(self) -> None:
69
+ rows = self.store.conn.execute(
70
+ "SELECT fact_id, record FROM vectors").fetchall()
71
+ if not rows:
72
+ return
73
+ recs = [r["record"] for r in rows]
74
+ self._ids = [r["fact_id"] for r in rows]
75
+ self._id2row = {fid: i for i, fid in enumerate(self._ids)}
76
+ packed_rows = [self.codec.from_bytes(b) for b in recs]
77
+ if isinstance(self.codec, Int8Codec):
78
+ q = np.stack([p[0] for p in packed_rows])
79
+ self._aux = np.stack([np.float16(p[1]) for p in packed_rows])
80
+ self._packed = q
81
+ self._cap = self._n = len(q)
82
+ else:
83
+ self._packed = np.stack(packed_rows)
84
+ self._cap = self._n = len(self._packed)
85
+ self._index_valid = False
86
+
87
+ def _load_pq_codebooks(self) -> None:
88
+ if not isinstance(self.codec, PQCodec):
89
+ return
90
+ blob = self.store.kv_get("PQ_CODEBOOKS")
91
+ if blob:
92
+ arr = np.frombuffer(base64.b64decode(blob), dtype=np.float32)
93
+ m = self.codec.m
94
+ ks = arr.size // (m * self.codec.sub)
95
+ self.codec.set_codebooks(arr.reshape(m, ks, self.codec.sub))
96
+
97
+ def _save_pq_codebooks(self) -> None:
98
+ if isinstance(self.codec, PQCodec) and self.codec.trained:
99
+ self.store.kv_set(
100
+ "PQ_CODEBOOKS",
101
+ base64.b64encode(self.codec.codebooks.astype(np.float32).tobytes()).decode())
102
+
103
+ # ------------------------------------------------------------- writing
104
+ def encode_fact(self, fact: Fact) -> np.ndarray:
105
+ return self.vsa.encode_fact(
106
+ self.embedder.embed(fact.subject),
107
+ self.embedder.embed(fact.relation),
108
+ self.embedder.embed(fact.value))
109
+
110
+ def _grow(self, need: int) -> None:
111
+ w = self._row_width()
112
+ new_cap = max(1024, self._cap * 2, need)
113
+ if isinstance(self.codec, Int8Codec):
114
+ packed = np.zeros((new_cap, w), dtype=np.int8)
115
+ aux = np.zeros(new_cap, dtype=np.float16)
116
+ if self._packed is not None and self._n:
117
+ packed[: self._n] = self._packed[: self._n]
118
+ aux[: self._n] = self._aux[: self._n]
119
+ self._packed, self._aux = packed, aux
120
+ else:
121
+ packed = np.zeros((new_cap, w), dtype=np.uint8)
122
+ if self._packed is not None and self._n:
123
+ packed[: self._n] = self._packed[: self._n]
124
+ self._packed = packed
125
+ self._cap = new_cap
126
+
127
+ def _row_width(self) -> int:
128
+ if isinstance(self.codec, Int8Codec):
129
+ return self.dims # scale lives in the aux array
130
+ if isinstance(self.codec, PQCodec):
131
+ return self.codec.m # 8 code bytes
132
+ return self.codec.bytes_per_vector # binary / rabitq packed words
133
+
134
+ def add(self, fact_id: str, vec: np.ndarray) -> None:
135
+ if isinstance(self.codec, PQCodec) and not self.codec.trained:
136
+ self._pq_buffer.append((fact_id, vec))
137
+ if len(self._pq_buffer) >= self.codec.train_threshold:
138
+ self._flush_pq()
139
+ return
140
+ self._append(fact_id, self._encode_row(vec), vec)
141
+
142
+ def _encode_row(self, vec: np.ndarray):
143
+ if isinstance(self.codec, Int8Codec):
144
+ q = self.codec.encode_packed(vec)
145
+ return q, self.codec.encode_scale(vec)
146
+ return self.codec.encode_packed(vec), None
147
+
148
+ def _append(self, fact_id: str, row, vec) -> None:
149
+ packed_row, scale = row
150
+ blob = (self.codec.to_bytes(packed_row, scale)
151
+ if isinstance(self.codec, Int8Codec)
152
+ else self.codec.to_bytes(packed_row))
153
+ self.store.conn.execute(
154
+ "INSERT OR REPLACE INTO vectors(fact_id, record, vec_hash, ts) VALUES(?,?,?,?)",
155
+ (fact_id, blob, self.hasher.hash_bytes(blob), iso(_dt.datetime.now(_dt.timezone.utc))))
156
+ if not self.store.batching:
157
+ self.store.conn.commit()
158
+ if self._n >= self._cap:
159
+ self._grow(self._n + 1)
160
+ if isinstance(self.codec, Int8Codec):
161
+ self._packed[self._n] = packed_row
162
+ self._aux[self._n] = scale
163
+ else:
164
+ self._packed[self._n] = packed_row
165
+ if fact_id in self._id2row:
166
+ old = self._id2row[fact_id]
167
+ # overwrite in place; keep id mapping
168
+ self._ids[old] = fact_id
169
+ else:
170
+ self._id2row[fact_id] = self._n
171
+ self._ids.append(fact_id)
172
+ self._n += 1
173
+ self._index_valid = False
174
+
175
+ def add_many(self, pairs: list[tuple[str, np.ndarray]]) -> None:
176
+ for fid, v in pairs:
177
+ self.add(fid, v)
178
+
179
+ def remove_ids(self, fact_ids: list[str]) -> int:
180
+ """Hard-remove vectors (GDPR erasure). Rebuilds packed state from
181
+ the surviving rows. Returns the number of vectors removed."""
182
+ if not fact_ids:
183
+ return 0
184
+ gone = set(fact_ids)
185
+ cur = self.store.conn.execute(
186
+ "SELECT fact_id FROM vectors").fetchall()
187
+ existing = {r["fact_id"] for r in cur}
188
+ removed = len(gone & existing)
189
+ if removed:
190
+ qmarks = ",".join("?" * len(gone))
191
+ self.store.conn.execute(
192
+ f"DELETE FROM vectors WHERE fact_id IN ({qmarks})",
193
+ list(gone))
194
+ self.store.conn.commit()
195
+ # rebuild in-memory state from disk (source of truth)
196
+ self._ids = []
197
+ self._id2row = {}
198
+ self._n = 0
199
+ self._cap = 0
200
+ self._packed = None
201
+ self._aux = None
202
+ self._index = None
203
+ self._index_valid = False
204
+ self._load()
205
+ return removed
206
+
207
+ def _flush_pq(self) -> None:
208
+ if not self._pq_buffer or not isinstance(self.codec, PQCodec):
209
+ return
210
+ if not self.codec.trained:
211
+ vecs = np.stack([v for _, v in self._pq_buffer])
212
+ self.codec.train(vecs)
213
+ self._save_pq_codebooks()
214
+ for fid, v in self._pq_buffer:
215
+ self._append(fid, self._encode_row(v), v)
216
+ self._pq_buffer.clear()
217
+
218
+ def size(self) -> int:
219
+ return self._n + len(self._pq_buffer)
220
+
221
+ def has(self, fact_id: str) -> bool:
222
+ return fact_id in self._id2row or any(f == fact_id for f, _ in self._pq_buffer)
223
+
224
+ def close(self) -> None:
225
+ self._flush_pq()
226
+ if not self.store.batching:
227
+ self.store.conn.commit()
228
+
229
+ # -------------------------------------------------------------- search
230
+ def _rows_getter(self, rows: np.ndarray):
231
+ if self._packed is None or self._n == 0:
232
+ return np.zeros((0, 1), dtype=np.uint8), None
233
+ packed = self._packed[rows]
234
+ aux = self._aux[rows] if self._aux is not None else None
235
+ return packed, aux
236
+
237
+ def ensure_index(self, force: bool = False) -> None:
238
+ if self._n < self.cfg.index_threshold:
239
+ return
240
+ if self._index is not None and self._index_valid and not force:
241
+ return
242
+ self._flush_pq()
243
+ index = TreeIndex(self.codec, self._rows_getter, self._n,
244
+ branch=self.cfg.index_branch,
245
+ leaf=self.cfg.index_leaf_size,
246
+ seed=self.cfg.seed)
247
+ index.build()
248
+ self._index = index
249
+ self._index_valid = True
250
+
251
+ def search(self, q: np.ndarray, k: int = 10,
252
+ candidate_ids: set[str] | None = None) -> list[tuple[str, float]]:
253
+ self._flush_pq()
254
+ if self._n == 0:
255
+ return []
256
+ use_index = (self._n >= self.cfg.index_threshold and self._index_valid
257
+ and (candidate_ids is None or len(candidate_ids) >= self._n * 0.5))
258
+ if use_index:
259
+ self.searches_indexed += 1
260
+ rows, scores = self._index.search(q, min(self._n, max(k * 3, 24)),
261
+ beam=self.cfg.beam_width)
262
+ out = [(self._ids[int(r)], float(s)) for r, s in zip(rows, scores)]
263
+ if candidate_ids is not None:
264
+ out = [(fid, s) for fid, s in out if fid in candidate_ids]
265
+ out.sort(key=lambda t: -t[1])
266
+ if len(out) < k and candidate_ids is not None and len(candidate_ids) > 0:
267
+ extra = self._flat_search(q, k, candidate_ids, exclude={f for f, _ in out})
268
+ out.extend(extra)
269
+ out.sort(key=lambda t: -t[1])
270
+ return out[:k]
271
+ self.searches_flat += 1
272
+ if candidate_ids is not None:
273
+ return self._flat_search(q, k, candidate_ids)
274
+ sc = self._score_all(q)
275
+ k2 = min(k, self._n)
276
+ idx = np.argpartition(-sc, k2 - 1)[:k2]
277
+ idx = idx[np.argsort(-sc[idx])]
278
+ return [(self._ids[int(i)], float(sc[i])) for i in idx]
279
+
280
+ def _score_all(self, qv: np.ndarray) -> np.ndarray:
281
+ # Rust fast path (Task 6-simd): when cortexm_core is built, route
282
+ # the per-row scoring through `batch_dot_i8` (int8 codec) or
283
+ # `batch_dot` (fp32 path) instead of the codec's numpy `scores`.
284
+ # The Rust kernels are AVX-512 → AVX2+FMA → NEON → scalar and
285
+ # process the whole batch in one Python→Rust boundary crossing.
286
+ # NumPy remains the exact reference; bit parity ≤1e-5 asserted
287
+ # by `tests/test_rust_accel.py::TestSimdKernels::test_batch_dot_*`.
288
+ try:
289
+ from cortexm import accel
290
+ except Exception: # pragma: no cover
291
+ accel = None
292
+ rust_ok = (accel is not None and accel.RUST_AVAILABLE
293
+ and accel.RUST_ENABLED)
294
+ n = self._n
295
+ d = self.dims
296
+ if rust_ok and n > 0 and self._packed is not None:
297
+ qf32 = np.ascontiguousarray(qv, dtype=np.float32)
298
+ if self.codec.uses_aux and self._aux is not None:
299
+ # Int8Codec: Rust returns raw int8·f32 dot products;
300
+ # multiply by per-row aux scales to match the codec's
301
+ # `scores` semantics (`packed.astype(f32) * aux @ q`).
302
+ # Pass the int8 buffer as a flat 1-D view — pyo3 binds
303
+ # to `PyReadonlyArray1<i8>` zero-copy.
304
+ packed_flat = np.ascontiguousarray(
305
+ self._packed[:n]).reshape(-1)
306
+ raw = np.asarray(
307
+ accel._core.batch_dot_i8(packed_flat, qf32, n, d),
308
+ dtype=np.float32)
309
+ aux = np.asarray(self._aux[:n], dtype=np.float32)
310
+ return raw * aux
311
+ if self._packed.dtype == np.float32:
312
+ # FP32-packed codec path. (BinaryCodec stores bipolar
313
+ # ±1 in uint8 — skip; codec.scores handles its own XOR.)
314
+ rows_flat = np.ascontiguousarray(
315
+ self._packed[:n]).reshape(-1)
316
+ raw = np.asarray(
317
+ accel._core.batch_dot(rows_flat, qf32, n, d),
318
+ dtype=np.float32)
319
+ return raw
320
+ if self.codec.uses_aux:
321
+ return self.codec.scores(self._packed[: self._n], qv, self._aux[: self._n])
322
+ return self.codec.scores(self._packed[: self._n], qv)
323
+
324
+ def _flat_search(self, q: np.ndarray, k: int, candidate_ids: set[str],
325
+ exclude: set[str] | None = None) -> list[tuple[str, float]]:
326
+ # iterate candidates in PALACE ROW ORDER (insertion order), never in
327
+ # set order: set iteration is hash-randomized per process, and the
328
+ # argsort below breaks score ties by position — random positions
329
+ # would make identical runs return different facts.
330
+ sel = [self._id2row[f] for f in candidate_ids
331
+ if f in self._id2row
332
+ and (exclude is None or f not in exclude)]
333
+ sel.sort()
334
+ if not sel:
335
+ return []
336
+ arr = np.array(sel, dtype=np.int64)
337
+ packed = self._packed[arr]
338
+ aux = self._aux[arr] if self._aux is not None else None
339
+ sc = (self.codec.scores(packed, q, aux) if self.codec.uses_aux
340
+ else self.codec.scores(packed, q))
341
+ order = np.argsort(-sc, kind="stable")[:k]
342
+ return [(self._ids[int(arr[i])], float(sc[i])) for i in order]
343
+
344
+ # ------------------------------------------------------ self-healing
345
+ def record_hash(self, row: int) -> str:
346
+ blob = self._record_bytes(row)
347
+ return self.hasher.hash_bytes(blob)
348
+
349
+ def _record_bytes(self, row: int) -> bytes:
350
+ if isinstance(self.codec, Int8Codec):
351
+ return self.codec.to_bytes(self._packed[row], self._aux[row])
352
+ return self.codec.to_bytes(self._packed[row])
353
+
354
+ def stored_hash(self, fact_id: str) -> str | None:
355
+ r = self.store.conn.execute(
356
+ "SELECT vec_hash FROM vectors WHERE fact_id=?", (fact_id,)).fetchone()
357
+ return r["vec_hash"] if r else None
358
+
359
+ def corrupt(self, rate: float, seed: int = 0,
360
+ persist: bool = False) -> int:
361
+ """Inject bit flips (cosmic rays / flash degradation). Returns count."""
362
+ if self._n == 0:
363
+ return 0
364
+ rng = np.random.default_rng(seed)
365
+ count = 0
366
+ for row in range(self._n):
367
+ if rng.random() < rate:
368
+ packed_row = self._packed[row]
369
+ self._packed[row] = self.codec.corrupt(packed_row, rate, rng)
370
+ count += 1
371
+ if persist:
372
+ blob = self._record_bytes(row)
373
+ self.store.conn.execute(
374
+ "UPDATE vectors SET record=? WHERE fact_id=?",
375
+ (blob, self._ids[row]))
376
+ if persist:
377
+ self.store.conn.commit()
378
+ self._index_valid = False
379
+ return count
380
+
381
+ def health_check(self, sample: int | None = None) -> dict:
382
+ """Detect corrupt records via stored vec_hash; TMR adds bit-level votes."""
383
+ import random as _random
384
+ rng = _random.Random(7)
385
+ rows = list(range(self._n))
386
+ if sample and sample < self._n:
387
+ rows = rng.sample(rows, sample)
388
+ corrupt_rows, checked = [], 0
389
+ tmr_flips = 0
390
+ for row in rows:
391
+ checked += 1
392
+ fid = self._ids[row]
393
+ want = self.stored_hash(fid)
394
+ got = self.record_hash(row)
395
+ if want and want != got:
396
+ corrupt_rows.append(fid)
397
+ if self.cfg.tmr and self.codec.name == "binary":
398
+ tmr_flips += self.codec.tmr_health(self._packed[row])
399
+ return {"checked": checked, "corrupt": len(corrupt_rows),
400
+ "corrupt_ids": corrupt_rows, "tmr_disagree_bits": int(tmr_flips)}
401
+
402
+ def heal(self, facts_by_id: dict[str, Fact]) -> dict:
403
+ """Re-encode corrupt records from the symbolic Trace (source of truth)."""
404
+ report = self.health_check()
405
+ healed = 0
406
+ for fid in report["corrupt_ids"]:
407
+ fact = facts_by_id.get(fid)
408
+ if fact is None:
409
+ continue
410
+ row = self._id2row.get(fid)
411
+ if row is None:
412
+ continue
413
+ vec = self.encode_fact(fact)
414
+ packed_row, scale = self._encode_row(vec)
415
+ if isinstance(self.codec, Int8Codec):
416
+ self._packed[row] = packed_row
417
+ self._aux[row] = scale
418
+ else:
419
+ self._packed[row] = packed_row
420
+ blob = (self.codec.to_bytes(packed_row, scale)
421
+ if isinstance(self.codec, Int8Codec)
422
+ else self.codec.to_bytes(packed_row))
423
+ self.store.conn.execute(
424
+ "UPDATE vectors SET record=?, vec_hash=? WHERE fact_id=?",
425
+ (blob, self.hasher.hash_bytes(blob), fid))
426
+ healed += 1
427
+ self.store.conn.commit()
428
+ self._index_valid = False
429
+ return {"corrupt": report["corrupt"], "healed": healed,
430
+ "note": "re-encoded from Trace; source hash verified"}
431
+
432
+ # ------------------------------------------------------------- stats
433
+ def storage_stats(self) -> dict:
434
+ vec_bytes = self._n * self.codec.bytes_per_vector if self._n else 0
435
+ return {
436
+ "vectors": self._n,
437
+ "codec": self.codec.name,
438
+ "bytes_per_vector": self.codec.bytes_per_vector,
439
+ "vector_bytes": int(vec_bytes),
440
+ "per_million_memories_mb": round(
441
+ self.codec.bytes_per_vector * 1e6 / 1e6, 1),
442
+ "index": ("tree" if self._index_valid else
443
+ ("flat" if self._n else "empty")),
444
+ "searches_flat": self.searches_flat,
445
+ "searches_indexed": self.searches_indexed,
446
+ }