cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
cortexm/vsa/ops.py
ADDED
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""Vector Symbolic Architecture algebra.
|
|
2
|
+
|
|
3
|
+
Modes (config ``vsa_mode``):
|
|
4
|
+
* ``perm`` — permutation binding (default). bind(role, filler) permutes
|
|
5
|
+
the filler with a role-seeded permutation. Similarity-preserving,
|
|
6
|
+
cheap (index shuffle), and directly portable to binary HDC
|
|
7
|
+
hardware (XOR/permutation ops) per the plan's edge roadmap.
|
|
8
|
+
* ``conv`` — Holographic Reduced Representations (Plate 1995):
|
|
9
|
+
circular-convolution binding via FFT, involution-based unbinding.
|
|
10
|
+
* ``bag`` — role-weighted superposition only (ablation baseline).
|
|
11
|
+
|
|
12
|
+
Every fact hologram = role-bound components + λ-weighted lexical
|
|
13
|
+
superposition. The λ term keeps free-text queries effective while the
|
|
14
|
+
bound terms carry structure for probe queries.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import numpy as np
|
|
20
|
+
|
|
21
|
+
from cortexm.util import h64 as _h64 # deterministic seeded hashing
|
|
22
|
+
|
|
23
|
+
ROLE_WEIGHTS = {"S": 1.0, "R": 1.0, "V": 1.0}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _norm(v: np.ndarray) -> np.ndarray:
|
|
27
|
+
n = float(np.linalg.norm(v))
|
|
28
|
+
return v / n if n > 0 else v
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class VSA:
|
|
32
|
+
def __init__(self, dims: int = 768, mode: str = "perm",
|
|
33
|
+
seed: int = 0x0C0FFEE, lexical_lambda: float = 0.6) -> None:
|
|
34
|
+
self.dims = dims
|
|
35
|
+
self.mode = mode
|
|
36
|
+
self.seed = seed
|
|
37
|
+
self.lam = lexical_lambda
|
|
38
|
+
self._perms: dict[str, np.ndarray] = {}
|
|
39
|
+
self._inperms: dict[str, np.ndarray] = {}
|
|
40
|
+
self._roles: dict[str, np.ndarray] = {}
|
|
41
|
+
# NSR-inspired engineered role vectors (optional). When
|
|
42
|
+
# _engineered is set, role_vec() returns the engineered vector
|
|
43
|
+
# instead of the random one. This is opt-in via .use_engineered().
|
|
44
|
+
self._engineered = None
|
|
45
|
+
|
|
46
|
+
def use_engineered(self, erv) -> None:
|
|
47
|
+
"""Swap the random role vectors for the engineered ones.
|
|
48
|
+
|
|
49
|
+
erv: an EngineeredRoleVectors instance that has been .fit()
|
|
50
|
+
on the actual fact corpus. After this call:
|
|
51
|
+
role_vec("S") -> erv.role_vec("S") (top-1 principal dir)
|
|
52
|
+
role_vec("R") -> erv.role_vec("R") (top-2)
|
|
53
|
+
role_vec("V") -> erv.role_vec("V") (top-3)
|
|
54
|
+
The existing random role vectors remain available as a fallback
|
|
55
|
+
if erv.role_vec(role) returns None.
|
|
56
|
+
"""
|
|
57
|
+
self._engineered = erv
|
|
58
|
+
|
|
59
|
+
# ------------------------------------------------------------- roles
|
|
60
|
+
def perm(self, role: str) -> np.ndarray:
|
|
61
|
+
p = self._perms.get(role)
|
|
62
|
+
if p is None:
|
|
63
|
+
rng = np.random.default_rng(_h64(f"perm:{role}", self.seed) & 0xFFFFFFFF)
|
|
64
|
+
p = rng.permutation(self.dims).astype(np.int32)
|
|
65
|
+
self._perms[role] = p
|
|
66
|
+
inv = np.empty_like(p)
|
|
67
|
+
inv[p] = np.arange(self.dims, dtype=np.int32)
|
|
68
|
+
self._inperms[role] = inv
|
|
69
|
+
return p
|
|
70
|
+
|
|
71
|
+
def inv_perm(self, role: str) -> np.ndarray:
|
|
72
|
+
self.perm(role)
|
|
73
|
+
return self._inperms[role]
|
|
74
|
+
|
|
75
|
+
def role_vec(self, role: str) -> np.ndarray:
|
|
76
|
+
# engineered role vectors (NSR-inspired) take precedence when
|
|
77
|
+
# configured — they sit on the top-k principal directions of
|
|
78
|
+
# the actual fact corpus, giving higher effective capacity and
|
|
79
|
+
# lower cross-talk than random role vectors.
|
|
80
|
+
if self._engineered is not None and self._engineered.is_fit:
|
|
81
|
+
v = self._engineered.role_vec(role)
|
|
82
|
+
if v is not None:
|
|
83
|
+
return v
|
|
84
|
+
v = self._roles.get(role)
|
|
85
|
+
if v is None:
|
|
86
|
+
rng = np.random.default_rng(_h64(f"role:{role}", self.seed) & 0xFFFFFFFF)
|
|
87
|
+
v = rng.standard_normal(self.dims).astype(np.float32)
|
|
88
|
+
v /= max(float(np.linalg.norm(v)), 1e-9)
|
|
89
|
+
self._roles[role] = v
|
|
90
|
+
return v
|
|
91
|
+
|
|
92
|
+
# ------------------------------------------------------- binding ops
|
|
93
|
+
def bind(self, role: str, filler: np.ndarray) -> np.ndarray:
|
|
94
|
+
if self.mode == "perm":
|
|
95
|
+
return filler[self.perm(role)]
|
|
96
|
+
if self.mode == "conv":
|
|
97
|
+
a = np.fft.rfft(filler)
|
|
98
|
+
b = np.fft.rfft(self.role_vec(role))
|
|
99
|
+
return np.fft.irfft(a * b, n=self.dims).astype(np.float32)
|
|
100
|
+
return (ROLE_WEIGHTS.get(role, 1.0) * filler).astype(np.float32)
|
|
101
|
+
|
|
102
|
+
def unbind(self, role: str, h: np.ndarray) -> np.ndarray:
|
|
103
|
+
if self.mode == "perm":
|
|
104
|
+
return h[self.inv_perm(role)]
|
|
105
|
+
if self.mode == "conv":
|
|
106
|
+
r = self.role_vec(role)
|
|
107
|
+
inv = np.concatenate(([r[0]], r[1:][::-1])) # involution
|
|
108
|
+
a = np.fft.rfft(h)
|
|
109
|
+
b = np.fft.rfft(inv)
|
|
110
|
+
return np.fft.irfft(a * b, n=self.dims).astype(np.float32)
|
|
111
|
+
return h / ROLE_WEIGHTS.get(role, 1.0)
|
|
112
|
+
|
|
113
|
+
@staticmethod
|
|
114
|
+
def bundle(vecs: list[np.ndarray]) -> np.ndarray:
|
|
115
|
+
return _norm(np.sum(vecs, axis=0).astype(np.float32))
|
|
116
|
+
|
|
117
|
+
# ---------------------------------------------------- fact encoding
|
|
118
|
+
def encode_fact(self, s_vec: np.ndarray, r_vec: np.ndarray,
|
|
119
|
+
v_vec: np.ndarray) -> np.ndarray:
|
|
120
|
+
lex = _norm(s_vec + r_vec + v_vec)
|
|
121
|
+
if self.mode == "bag":
|
|
122
|
+
return _norm(ROLE_WEIGHTS["S"] * s_vec + ROLE_WEIGHTS["R"] * r_vec
|
|
123
|
+
+ ROLE_WEIGHTS["V"] * v_vec)
|
|
124
|
+
bound = self.bundle([
|
|
125
|
+
self.bind("S", s_vec), self.bind("R", r_vec), self.bind("V", v_vec)])
|
|
126
|
+
return _norm(bound + self.lam * lex)
|
|
127
|
+
|
|
128
|
+
def probe(self, s_vec: np.ndarray | None = None,
|
|
129
|
+
r_vec: np.ndarray | None = None,
|
|
130
|
+
v_vec: np.ndarray | None = None,
|
|
131
|
+
lexical: np.ndarray | None = None) -> np.ndarray:
|
|
132
|
+
"""Structured probe: approximate fact hologram from known roles."""
|
|
133
|
+
parts: list[np.ndarray] = []
|
|
134
|
+
if s_vec is not None:
|
|
135
|
+
parts.append(self.bind("S", s_vec))
|
|
136
|
+
if r_vec is not None:
|
|
137
|
+
parts.append(self.bind("R", r_vec))
|
|
138
|
+
if v_vec is not None:
|
|
139
|
+
parts.append(self.bind("V", v_vec))
|
|
140
|
+
if not parts:
|
|
141
|
+
return _norm(lexical if lexical is not None else np.zeros(self.dims, np.float32))
|
|
142
|
+
probe = self.bundle(parts)
|
|
143
|
+
if lexical is not None and self.mode != "bag":
|
|
144
|
+
probe = _norm(probe + self.lam * _norm(lexical))
|
|
145
|
+
return probe
|
|
146
|
+
|
|
147
|
+
def unbind_role(self, h: np.ndarray, role: str) -> np.ndarray:
|
|
148
|
+
"""Extract the approximate filler bound to ``role`` (audit path)."""
|
|
149
|
+
return _norm(self.unbind(role, h))
|
cortexm/vsa/palace.py
ADDED
|
@@ -0,0 +1,446 @@
|
|
|
1
|
+
"""The Memory Palace — VSA hologram store with codec-quantized vectors.
|
|
2
|
+
|
|
3
|
+
One hologram per fact: role-bound subject/relation/value fillers plus a
|
|
4
|
+
λ-weighted lexical superposition (see vsa.ops). Storage is codec-
|
|
5
|
+
quantized (INT8 / Binary / RaBitQ / PQ), persisted as BLOBs in SQLite
|
|
6
|
+
and mirrored in RAM as packed numpy matrices for microsecond scoring.
|
|
7
|
+
A page-clustered tree index provides O(log N) retrieval once the
|
|
8
|
+
collection crosses ``index_threshold``; below it, flat scan wins.
|
|
9
|
+
|
|
10
|
+
Also hosts the self-healing machinery: per-record hash checks, TMR
|
|
11
|
+
majority vote (binary codec), and re-encoding from the symbolic Trace
|
|
12
|
+
when corruption exceeds the correction radius.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import base64
|
|
18
|
+
import numpy as np
|
|
19
|
+
|
|
20
|
+
from cortexm.config import Config
|
|
21
|
+
from cortexm.errors import CodecError, StoreError
|
|
22
|
+
from cortexm.text.embedder import HashingEmbedder
|
|
23
|
+
from cortexm.trace.fact import Fact
|
|
24
|
+
from cortexm.trace.store import TraceStore
|
|
25
|
+
from cortexm.util import iso
|
|
26
|
+
import datetime as _dt
|
|
27
|
+
from cortexm.vsa.codecs import make_codec, PQCodec, Int8Codec
|
|
28
|
+
from cortexm.vsa.index import TreeIndex
|
|
29
|
+
from cortexm.vsa.ops import VSA
|
|
30
|
+
|
|
31
|
+
VEC_TABLE = """
|
|
32
|
+
CREATE TABLE IF NOT EXISTS vectors (
|
|
33
|
+
fact_id TEXT PRIMARY KEY, record BLOB NOT NULL,
|
|
34
|
+
vec_hash TEXT DEFAULT '', ts TEXT DEFAULT ''
|
|
35
|
+
)
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class MemoryPalace:
|
|
40
|
+
def __init__(self, config: Config, store: TraceStore) -> None:
|
|
41
|
+
self.cfg = config
|
|
42
|
+
self.store = store
|
|
43
|
+
self.dims = config.dims
|
|
44
|
+
self.hasher = store.hasher
|
|
45
|
+
self.codec = make_codec(config.codec, config.dims, config.seed,
|
|
46
|
+
tmr=config.tmr)
|
|
47
|
+
self.vsa = VSA(config.dims, config.vsa_mode, config.seed,
|
|
48
|
+
config.lexical_lambda)
|
|
49
|
+
self.embedder = HashingEmbedder(config.dims, config.seed)
|
|
50
|
+
store.conn.execute(VEC_TABLE)
|
|
51
|
+
store.conn.commit()
|
|
52
|
+
|
|
53
|
+
self._ids: list[str] = []
|
|
54
|
+
self._id2row: dict[str, int] = {}
|
|
55
|
+
self._n = 0
|
|
56
|
+
self._cap = 0
|
|
57
|
+
self._packed: np.ndarray | None = None
|
|
58
|
+
self._aux: np.ndarray | None = None
|
|
59
|
+
self._index: TreeIndex | None = None
|
|
60
|
+
self._index_valid = False
|
|
61
|
+
self._pq_buffer: list[tuple[str, np.ndarray]] = []
|
|
62
|
+
self.searches_flat = 0
|
|
63
|
+
self.searches_indexed = 0
|
|
64
|
+
self._load_pq_codebooks()
|
|
65
|
+
self._load()
|
|
66
|
+
|
|
67
|
+
# ------------------------------------------------------------- loading
|
|
68
|
+
def _load(self) -> None:
|
|
69
|
+
rows = self.store.conn.execute(
|
|
70
|
+
"SELECT fact_id, record FROM vectors").fetchall()
|
|
71
|
+
if not rows:
|
|
72
|
+
return
|
|
73
|
+
recs = [r["record"] for r in rows]
|
|
74
|
+
self._ids = [r["fact_id"] for r in rows]
|
|
75
|
+
self._id2row = {fid: i for i, fid in enumerate(self._ids)}
|
|
76
|
+
packed_rows = [self.codec.from_bytes(b) for b in recs]
|
|
77
|
+
if isinstance(self.codec, Int8Codec):
|
|
78
|
+
q = np.stack([p[0] for p in packed_rows])
|
|
79
|
+
self._aux = np.stack([np.float16(p[1]) for p in packed_rows])
|
|
80
|
+
self._packed = q
|
|
81
|
+
self._cap = self._n = len(q)
|
|
82
|
+
else:
|
|
83
|
+
self._packed = np.stack(packed_rows)
|
|
84
|
+
self._cap = self._n = len(self._packed)
|
|
85
|
+
self._index_valid = False
|
|
86
|
+
|
|
87
|
+
def _load_pq_codebooks(self) -> None:
|
|
88
|
+
if not isinstance(self.codec, PQCodec):
|
|
89
|
+
return
|
|
90
|
+
blob = self.store.kv_get("PQ_CODEBOOKS")
|
|
91
|
+
if blob:
|
|
92
|
+
arr = np.frombuffer(base64.b64decode(blob), dtype=np.float32)
|
|
93
|
+
m = self.codec.m
|
|
94
|
+
ks = arr.size // (m * self.codec.sub)
|
|
95
|
+
self.codec.set_codebooks(arr.reshape(m, ks, self.codec.sub))
|
|
96
|
+
|
|
97
|
+
def _save_pq_codebooks(self) -> None:
|
|
98
|
+
if isinstance(self.codec, PQCodec) and self.codec.trained:
|
|
99
|
+
self.store.kv_set(
|
|
100
|
+
"PQ_CODEBOOKS",
|
|
101
|
+
base64.b64encode(self.codec.codebooks.astype(np.float32).tobytes()).decode())
|
|
102
|
+
|
|
103
|
+
# ------------------------------------------------------------- writing
|
|
104
|
+
def encode_fact(self, fact: Fact) -> np.ndarray:
|
|
105
|
+
return self.vsa.encode_fact(
|
|
106
|
+
self.embedder.embed(fact.subject),
|
|
107
|
+
self.embedder.embed(fact.relation),
|
|
108
|
+
self.embedder.embed(fact.value))
|
|
109
|
+
|
|
110
|
+
def _grow(self, need: int) -> None:
|
|
111
|
+
w = self._row_width()
|
|
112
|
+
new_cap = max(1024, self._cap * 2, need)
|
|
113
|
+
if isinstance(self.codec, Int8Codec):
|
|
114
|
+
packed = np.zeros((new_cap, w), dtype=np.int8)
|
|
115
|
+
aux = np.zeros(new_cap, dtype=np.float16)
|
|
116
|
+
if self._packed is not None and self._n:
|
|
117
|
+
packed[: self._n] = self._packed[: self._n]
|
|
118
|
+
aux[: self._n] = self._aux[: self._n]
|
|
119
|
+
self._packed, self._aux = packed, aux
|
|
120
|
+
else:
|
|
121
|
+
packed = np.zeros((new_cap, w), dtype=np.uint8)
|
|
122
|
+
if self._packed is not None and self._n:
|
|
123
|
+
packed[: self._n] = self._packed[: self._n]
|
|
124
|
+
self._packed = packed
|
|
125
|
+
self._cap = new_cap
|
|
126
|
+
|
|
127
|
+
def _row_width(self) -> int:
|
|
128
|
+
if isinstance(self.codec, Int8Codec):
|
|
129
|
+
return self.dims # scale lives in the aux array
|
|
130
|
+
if isinstance(self.codec, PQCodec):
|
|
131
|
+
return self.codec.m # 8 code bytes
|
|
132
|
+
return self.codec.bytes_per_vector # binary / rabitq packed words
|
|
133
|
+
|
|
134
|
+
def add(self, fact_id: str, vec: np.ndarray) -> None:
|
|
135
|
+
if isinstance(self.codec, PQCodec) and not self.codec.trained:
|
|
136
|
+
self._pq_buffer.append((fact_id, vec))
|
|
137
|
+
if len(self._pq_buffer) >= self.codec.train_threshold:
|
|
138
|
+
self._flush_pq()
|
|
139
|
+
return
|
|
140
|
+
self._append(fact_id, self._encode_row(vec), vec)
|
|
141
|
+
|
|
142
|
+
def _encode_row(self, vec: np.ndarray):
|
|
143
|
+
if isinstance(self.codec, Int8Codec):
|
|
144
|
+
q = self.codec.encode_packed(vec)
|
|
145
|
+
return q, self.codec.encode_scale(vec)
|
|
146
|
+
return self.codec.encode_packed(vec), None
|
|
147
|
+
|
|
148
|
+
def _append(self, fact_id: str, row, vec) -> None:
|
|
149
|
+
packed_row, scale = row
|
|
150
|
+
blob = (self.codec.to_bytes(packed_row, scale)
|
|
151
|
+
if isinstance(self.codec, Int8Codec)
|
|
152
|
+
else self.codec.to_bytes(packed_row))
|
|
153
|
+
self.store.conn.execute(
|
|
154
|
+
"INSERT OR REPLACE INTO vectors(fact_id, record, vec_hash, ts) VALUES(?,?,?,?)",
|
|
155
|
+
(fact_id, blob, self.hasher.hash_bytes(blob), iso(_dt.datetime.now(_dt.timezone.utc))))
|
|
156
|
+
if not self.store.batching:
|
|
157
|
+
self.store.conn.commit()
|
|
158
|
+
if self._n >= self._cap:
|
|
159
|
+
self._grow(self._n + 1)
|
|
160
|
+
if isinstance(self.codec, Int8Codec):
|
|
161
|
+
self._packed[self._n] = packed_row
|
|
162
|
+
self._aux[self._n] = scale
|
|
163
|
+
else:
|
|
164
|
+
self._packed[self._n] = packed_row
|
|
165
|
+
if fact_id in self._id2row:
|
|
166
|
+
old = self._id2row[fact_id]
|
|
167
|
+
# overwrite in place; keep id mapping
|
|
168
|
+
self._ids[old] = fact_id
|
|
169
|
+
else:
|
|
170
|
+
self._id2row[fact_id] = self._n
|
|
171
|
+
self._ids.append(fact_id)
|
|
172
|
+
self._n += 1
|
|
173
|
+
self._index_valid = False
|
|
174
|
+
|
|
175
|
+
def add_many(self, pairs: list[tuple[str, np.ndarray]]) -> None:
|
|
176
|
+
for fid, v in pairs:
|
|
177
|
+
self.add(fid, v)
|
|
178
|
+
|
|
179
|
+
def remove_ids(self, fact_ids: list[str]) -> int:
|
|
180
|
+
"""Hard-remove vectors (GDPR erasure). Rebuilds packed state from
|
|
181
|
+
the surviving rows. Returns the number of vectors removed."""
|
|
182
|
+
if not fact_ids:
|
|
183
|
+
return 0
|
|
184
|
+
gone = set(fact_ids)
|
|
185
|
+
cur = self.store.conn.execute(
|
|
186
|
+
"SELECT fact_id FROM vectors").fetchall()
|
|
187
|
+
existing = {r["fact_id"] for r in cur}
|
|
188
|
+
removed = len(gone & existing)
|
|
189
|
+
if removed:
|
|
190
|
+
qmarks = ",".join("?" * len(gone))
|
|
191
|
+
self.store.conn.execute(
|
|
192
|
+
f"DELETE FROM vectors WHERE fact_id IN ({qmarks})",
|
|
193
|
+
list(gone))
|
|
194
|
+
self.store.conn.commit()
|
|
195
|
+
# rebuild in-memory state from disk (source of truth)
|
|
196
|
+
self._ids = []
|
|
197
|
+
self._id2row = {}
|
|
198
|
+
self._n = 0
|
|
199
|
+
self._cap = 0
|
|
200
|
+
self._packed = None
|
|
201
|
+
self._aux = None
|
|
202
|
+
self._index = None
|
|
203
|
+
self._index_valid = False
|
|
204
|
+
self._load()
|
|
205
|
+
return removed
|
|
206
|
+
|
|
207
|
+
def _flush_pq(self) -> None:
|
|
208
|
+
if not self._pq_buffer or not isinstance(self.codec, PQCodec):
|
|
209
|
+
return
|
|
210
|
+
if not self.codec.trained:
|
|
211
|
+
vecs = np.stack([v for _, v in self._pq_buffer])
|
|
212
|
+
self.codec.train(vecs)
|
|
213
|
+
self._save_pq_codebooks()
|
|
214
|
+
for fid, v in self._pq_buffer:
|
|
215
|
+
self._append(fid, self._encode_row(v), v)
|
|
216
|
+
self._pq_buffer.clear()
|
|
217
|
+
|
|
218
|
+
def size(self) -> int:
|
|
219
|
+
return self._n + len(self._pq_buffer)
|
|
220
|
+
|
|
221
|
+
def has(self, fact_id: str) -> bool:
|
|
222
|
+
return fact_id in self._id2row or any(f == fact_id for f, _ in self._pq_buffer)
|
|
223
|
+
|
|
224
|
+
def close(self) -> None:
|
|
225
|
+
self._flush_pq()
|
|
226
|
+
if not self.store.batching:
|
|
227
|
+
self.store.conn.commit()
|
|
228
|
+
|
|
229
|
+
# -------------------------------------------------------------- search
|
|
230
|
+
def _rows_getter(self, rows: np.ndarray):
|
|
231
|
+
if self._packed is None or self._n == 0:
|
|
232
|
+
return np.zeros((0, 1), dtype=np.uint8), None
|
|
233
|
+
packed = self._packed[rows]
|
|
234
|
+
aux = self._aux[rows] if self._aux is not None else None
|
|
235
|
+
return packed, aux
|
|
236
|
+
|
|
237
|
+
def ensure_index(self, force: bool = False) -> None:
|
|
238
|
+
if self._n < self.cfg.index_threshold:
|
|
239
|
+
return
|
|
240
|
+
if self._index is not None and self._index_valid and not force:
|
|
241
|
+
return
|
|
242
|
+
self._flush_pq()
|
|
243
|
+
index = TreeIndex(self.codec, self._rows_getter, self._n,
|
|
244
|
+
branch=self.cfg.index_branch,
|
|
245
|
+
leaf=self.cfg.index_leaf_size,
|
|
246
|
+
seed=self.cfg.seed)
|
|
247
|
+
index.build()
|
|
248
|
+
self._index = index
|
|
249
|
+
self._index_valid = True
|
|
250
|
+
|
|
251
|
+
def search(self, q: np.ndarray, k: int = 10,
|
|
252
|
+
candidate_ids: set[str] | None = None) -> list[tuple[str, float]]:
|
|
253
|
+
self._flush_pq()
|
|
254
|
+
if self._n == 0:
|
|
255
|
+
return []
|
|
256
|
+
use_index = (self._n >= self.cfg.index_threshold and self._index_valid
|
|
257
|
+
and (candidate_ids is None or len(candidate_ids) >= self._n * 0.5))
|
|
258
|
+
if use_index:
|
|
259
|
+
self.searches_indexed += 1
|
|
260
|
+
rows, scores = self._index.search(q, min(self._n, max(k * 3, 24)),
|
|
261
|
+
beam=self.cfg.beam_width)
|
|
262
|
+
out = [(self._ids[int(r)], float(s)) for r, s in zip(rows, scores)]
|
|
263
|
+
if candidate_ids is not None:
|
|
264
|
+
out = [(fid, s) for fid, s in out if fid in candidate_ids]
|
|
265
|
+
out.sort(key=lambda t: -t[1])
|
|
266
|
+
if len(out) < k and candidate_ids is not None and len(candidate_ids) > 0:
|
|
267
|
+
extra = self._flat_search(q, k, candidate_ids, exclude={f for f, _ in out})
|
|
268
|
+
out.extend(extra)
|
|
269
|
+
out.sort(key=lambda t: -t[1])
|
|
270
|
+
return out[:k]
|
|
271
|
+
self.searches_flat += 1
|
|
272
|
+
if candidate_ids is not None:
|
|
273
|
+
return self._flat_search(q, k, candidate_ids)
|
|
274
|
+
sc = self._score_all(q)
|
|
275
|
+
k2 = min(k, self._n)
|
|
276
|
+
idx = np.argpartition(-sc, k2 - 1)[:k2]
|
|
277
|
+
idx = idx[np.argsort(-sc[idx])]
|
|
278
|
+
return [(self._ids[int(i)], float(sc[i])) for i in idx]
|
|
279
|
+
|
|
280
|
+
def _score_all(self, qv: np.ndarray) -> np.ndarray:
|
|
281
|
+
# Rust fast path (Task 6-simd): when cortexm_core is built, route
|
|
282
|
+
# the per-row scoring through `batch_dot_i8` (int8 codec) or
|
|
283
|
+
# `batch_dot` (fp32 path) instead of the codec's numpy `scores`.
|
|
284
|
+
# The Rust kernels are AVX-512 → AVX2+FMA → NEON → scalar and
|
|
285
|
+
# process the whole batch in one Python→Rust boundary crossing.
|
|
286
|
+
# NumPy remains the exact reference; bit parity ≤1e-5 asserted
|
|
287
|
+
# by `tests/test_rust_accel.py::TestSimdKernels::test_batch_dot_*`.
|
|
288
|
+
try:
|
|
289
|
+
from cortexm import accel
|
|
290
|
+
except Exception: # pragma: no cover
|
|
291
|
+
accel = None
|
|
292
|
+
rust_ok = (accel is not None and accel.RUST_AVAILABLE
|
|
293
|
+
and accel.RUST_ENABLED)
|
|
294
|
+
n = self._n
|
|
295
|
+
d = self.dims
|
|
296
|
+
if rust_ok and n > 0 and self._packed is not None:
|
|
297
|
+
qf32 = np.ascontiguousarray(qv, dtype=np.float32)
|
|
298
|
+
if self.codec.uses_aux and self._aux is not None:
|
|
299
|
+
# Int8Codec: Rust returns raw int8·f32 dot products;
|
|
300
|
+
# multiply by per-row aux scales to match the codec's
|
|
301
|
+
# `scores` semantics (`packed.astype(f32) * aux @ q`).
|
|
302
|
+
# Pass the int8 buffer as a flat 1-D view — pyo3 binds
|
|
303
|
+
# to `PyReadonlyArray1<i8>` zero-copy.
|
|
304
|
+
packed_flat = np.ascontiguousarray(
|
|
305
|
+
self._packed[:n]).reshape(-1)
|
|
306
|
+
raw = np.asarray(
|
|
307
|
+
accel._core.batch_dot_i8(packed_flat, qf32, n, d),
|
|
308
|
+
dtype=np.float32)
|
|
309
|
+
aux = np.asarray(self._aux[:n], dtype=np.float32)
|
|
310
|
+
return raw * aux
|
|
311
|
+
if self._packed.dtype == np.float32:
|
|
312
|
+
# FP32-packed codec path. (BinaryCodec stores bipolar
|
|
313
|
+
# ±1 in uint8 — skip; codec.scores handles its own XOR.)
|
|
314
|
+
rows_flat = np.ascontiguousarray(
|
|
315
|
+
self._packed[:n]).reshape(-1)
|
|
316
|
+
raw = np.asarray(
|
|
317
|
+
accel._core.batch_dot(rows_flat, qf32, n, d),
|
|
318
|
+
dtype=np.float32)
|
|
319
|
+
return raw
|
|
320
|
+
if self.codec.uses_aux:
|
|
321
|
+
return self.codec.scores(self._packed[: self._n], qv, self._aux[: self._n])
|
|
322
|
+
return self.codec.scores(self._packed[: self._n], qv)
|
|
323
|
+
|
|
324
|
+
def _flat_search(self, q: np.ndarray, k: int, candidate_ids: set[str],
|
|
325
|
+
exclude: set[str] | None = None) -> list[tuple[str, float]]:
|
|
326
|
+
# iterate candidates in PALACE ROW ORDER (insertion order), never in
|
|
327
|
+
# set order: set iteration is hash-randomized per process, and the
|
|
328
|
+
# argsort below breaks score ties by position — random positions
|
|
329
|
+
# would make identical runs return different facts.
|
|
330
|
+
sel = [self._id2row[f] for f in candidate_ids
|
|
331
|
+
if f in self._id2row
|
|
332
|
+
and (exclude is None or f not in exclude)]
|
|
333
|
+
sel.sort()
|
|
334
|
+
if not sel:
|
|
335
|
+
return []
|
|
336
|
+
arr = np.array(sel, dtype=np.int64)
|
|
337
|
+
packed = self._packed[arr]
|
|
338
|
+
aux = self._aux[arr] if self._aux is not None else None
|
|
339
|
+
sc = (self.codec.scores(packed, q, aux) if self.codec.uses_aux
|
|
340
|
+
else self.codec.scores(packed, q))
|
|
341
|
+
order = np.argsort(-sc, kind="stable")[:k]
|
|
342
|
+
return [(self._ids[int(arr[i])], float(sc[i])) for i in order]
|
|
343
|
+
|
|
344
|
+
# ------------------------------------------------------ self-healing
|
|
345
|
+
def record_hash(self, row: int) -> str:
|
|
346
|
+
blob = self._record_bytes(row)
|
|
347
|
+
return self.hasher.hash_bytes(blob)
|
|
348
|
+
|
|
349
|
+
def _record_bytes(self, row: int) -> bytes:
|
|
350
|
+
if isinstance(self.codec, Int8Codec):
|
|
351
|
+
return self.codec.to_bytes(self._packed[row], self._aux[row])
|
|
352
|
+
return self.codec.to_bytes(self._packed[row])
|
|
353
|
+
|
|
354
|
+
def stored_hash(self, fact_id: str) -> str | None:
|
|
355
|
+
r = self.store.conn.execute(
|
|
356
|
+
"SELECT vec_hash FROM vectors WHERE fact_id=?", (fact_id,)).fetchone()
|
|
357
|
+
return r["vec_hash"] if r else None
|
|
358
|
+
|
|
359
|
+
def corrupt(self, rate: float, seed: int = 0,
|
|
360
|
+
persist: bool = False) -> int:
|
|
361
|
+
"""Inject bit flips (cosmic rays / flash degradation). Returns count."""
|
|
362
|
+
if self._n == 0:
|
|
363
|
+
return 0
|
|
364
|
+
rng = np.random.default_rng(seed)
|
|
365
|
+
count = 0
|
|
366
|
+
for row in range(self._n):
|
|
367
|
+
if rng.random() < rate:
|
|
368
|
+
packed_row = self._packed[row]
|
|
369
|
+
self._packed[row] = self.codec.corrupt(packed_row, rate, rng)
|
|
370
|
+
count += 1
|
|
371
|
+
if persist:
|
|
372
|
+
blob = self._record_bytes(row)
|
|
373
|
+
self.store.conn.execute(
|
|
374
|
+
"UPDATE vectors SET record=? WHERE fact_id=?",
|
|
375
|
+
(blob, self._ids[row]))
|
|
376
|
+
if persist:
|
|
377
|
+
self.store.conn.commit()
|
|
378
|
+
self._index_valid = False
|
|
379
|
+
return count
|
|
380
|
+
|
|
381
|
+
def health_check(self, sample: int | None = None) -> dict:
|
|
382
|
+
"""Detect corrupt records via stored vec_hash; TMR adds bit-level votes."""
|
|
383
|
+
import random as _random
|
|
384
|
+
rng = _random.Random(7)
|
|
385
|
+
rows = list(range(self._n))
|
|
386
|
+
if sample and sample < self._n:
|
|
387
|
+
rows = rng.sample(rows, sample)
|
|
388
|
+
corrupt_rows, checked = [], 0
|
|
389
|
+
tmr_flips = 0
|
|
390
|
+
for row in rows:
|
|
391
|
+
checked += 1
|
|
392
|
+
fid = self._ids[row]
|
|
393
|
+
want = self.stored_hash(fid)
|
|
394
|
+
got = self.record_hash(row)
|
|
395
|
+
if want and want != got:
|
|
396
|
+
corrupt_rows.append(fid)
|
|
397
|
+
if self.cfg.tmr and self.codec.name == "binary":
|
|
398
|
+
tmr_flips += self.codec.tmr_health(self._packed[row])
|
|
399
|
+
return {"checked": checked, "corrupt": len(corrupt_rows),
|
|
400
|
+
"corrupt_ids": corrupt_rows, "tmr_disagree_bits": int(tmr_flips)}
|
|
401
|
+
|
|
402
|
+
def heal(self, facts_by_id: dict[str, Fact]) -> dict:
|
|
403
|
+
"""Re-encode corrupt records from the symbolic Trace (source of truth)."""
|
|
404
|
+
report = self.health_check()
|
|
405
|
+
healed = 0
|
|
406
|
+
for fid in report["corrupt_ids"]:
|
|
407
|
+
fact = facts_by_id.get(fid)
|
|
408
|
+
if fact is None:
|
|
409
|
+
continue
|
|
410
|
+
row = self._id2row.get(fid)
|
|
411
|
+
if row is None:
|
|
412
|
+
continue
|
|
413
|
+
vec = self.encode_fact(fact)
|
|
414
|
+
packed_row, scale = self._encode_row(vec)
|
|
415
|
+
if isinstance(self.codec, Int8Codec):
|
|
416
|
+
self._packed[row] = packed_row
|
|
417
|
+
self._aux[row] = scale
|
|
418
|
+
else:
|
|
419
|
+
self._packed[row] = packed_row
|
|
420
|
+
blob = (self.codec.to_bytes(packed_row, scale)
|
|
421
|
+
if isinstance(self.codec, Int8Codec)
|
|
422
|
+
else self.codec.to_bytes(packed_row))
|
|
423
|
+
self.store.conn.execute(
|
|
424
|
+
"UPDATE vectors SET record=?, vec_hash=? WHERE fact_id=?",
|
|
425
|
+
(blob, self.hasher.hash_bytes(blob), fid))
|
|
426
|
+
healed += 1
|
|
427
|
+
self.store.conn.commit()
|
|
428
|
+
self._index_valid = False
|
|
429
|
+
return {"corrupt": report["corrupt"], "healed": healed,
|
|
430
|
+
"note": "re-encoded from Trace; source hash verified"}
|
|
431
|
+
|
|
432
|
+
# ------------------------------------------------------------- stats
|
|
433
|
+
def storage_stats(self) -> dict:
|
|
434
|
+
vec_bytes = self._n * self.codec.bytes_per_vector if self._n else 0
|
|
435
|
+
return {
|
|
436
|
+
"vectors": self._n,
|
|
437
|
+
"codec": self.codec.name,
|
|
438
|
+
"bytes_per_vector": self.codec.bytes_per_vector,
|
|
439
|
+
"vector_bytes": int(vec_bytes),
|
|
440
|
+
"per_million_memories_mb": round(
|
|
441
|
+
self.codec.bytes_per_vector * 1e6 / 1e6, 1),
|
|
442
|
+
"index": ("tree" if self._index_valid else
|
|
443
|
+
("flat" if self._n else "empty")),
|
|
444
|
+
"searches_flat": self.searches_flat,
|
|
445
|
+
"searches_indexed": self.searches_indexed,
|
|
446
|
+
}
|