cortexm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. context_m.py +17 -0
  2. cortexm/__init__.py +45 -0
  3. cortexm/accel.py +403 -0
  4. cortexm/api/__init__.py +0 -0
  5. cortexm/api/chaos.py +118 -0
  6. cortexm/api/memory.py +635 -0
  7. cortexm/bench/__init__.py +0 -0
  8. cortexm/bench/abilities.py +311 -0
  9. cortexm/bench/baselines.py +89 -0
  10. cortexm/bench/beam_loader.py +317 -0
  11. cortexm/bench/generator.py +376 -0
  12. cortexm/bench/harness.py +211 -0
  13. cortexm/bench/messy.py +218 -0
  14. cortexm/bench/micro.py +251 -0
  15. cortexm/bench/ood.py +443 -0
  16. cortexm/bench/run.py +137 -0
  17. cortexm/bridge/__init__.py +0 -0
  18. cortexm/bridge/dates.py +178 -0
  19. cortexm/bridge/decoders.py +204 -0
  20. cortexm/bridge/enrich.py +255 -0
  21. cortexm/bridge/extractor.py +316 -0
  22. cortexm/bridge/fallback.py +332 -0
  23. cortexm/bridge/onnx_runtime.py +158 -0
  24. cortexm/bridge/patterns.py +760 -0
  25. cortexm/bridge/ppr.py +104 -0
  26. cortexm/bridge/prefilter.py +188 -0
  27. cortexm/bridge/query_extract.py +420 -0
  28. cortexm/bridge/reader.py +1174 -0
  29. cortexm/bridge/rerank.py +204 -0
  30. cortexm/bridge/writer.py +492 -0
  31. cortexm/cli.py +295 -0
  32. cortexm/cognition/__init__.py +53 -0
  33. cortexm/cognition/abstraction.py +192 -0
  34. cortexm/cognition/analogy.py +159 -0
  35. cortexm/cognition/engine.py +204 -0
  36. cortexm/cognition/gaps.py +365 -0
  37. cortexm/cognition/scanner.py +204 -0
  38. cortexm/config.py +375 -0
  39. cortexm/cortexm.py +8 -0
  40. cortexm/enterprise/__init__.py +0 -0
  41. cortexm/enterprise/audit.py +178 -0
  42. cortexm/enterprise/governance.py +239 -0
  43. cortexm/errors.py +35 -0
  44. cortexm/features/__init__.py +0 -0
  45. cortexm/features/git.py +204 -0
  46. cortexm/features/prefetch.py +88 -0
  47. cortexm/features/zk.py +105 -0
  48. cortexm/federation/__init__.py +39 -0
  49. cortexm/federation/crdt.py +275 -0
  50. cortexm/federation/fabric.py +109 -0
  51. cortexm/federation/hlc.py +80 -0
  52. cortexm/federation/node.py +145 -0
  53. cortexm/federation/schema_report.py +73 -0
  54. cortexm/federation/transport.py +164 -0
  55. cortexm/index/__init__.py +19 -0
  56. cortexm/index/nsg.py +386 -0
  57. cortexm/mcp/__init__.py +0 -0
  58. cortexm/mcp/server.py +985 -0
  59. cortexm/metrics.py +62 -0
  60. cortexm/migrate/__init__.py +0 -0
  61. cortexm/migrate/importers.py +192 -0
  62. cortexm/provenance/__init__.py +78 -0
  63. cortexm/provenance/agent.py +214 -0
  64. cortexm/provenance/cose.py +201 -0
  65. cortexm/provenance/scitt.py +258 -0
  66. cortexm/provenance/vc.py +250 -0
  67. cortexm/security/__init__.py +0 -0
  68. cortexm/security/crypto.py +162 -0
  69. cortexm/security/hashes.py +140 -0
  70. cortexm/security/injection.py +149 -0
  71. cortexm/security/mind.py +154 -0
  72. cortexm/security/pii.py +265 -0
  73. cortexm/security/rbac.py +169 -0
  74. cortexm/security/sandbox.py +131 -0
  75. cortexm/security/zk_hamming.py +142 -0
  76. cortexm/security/zk_sql.py +485 -0
  77. cortexm/server/__init__.py +0 -0
  78. cortexm/server/metrics.py +88 -0
  79. cortexm/server/rest.py +936 -0
  80. cortexm/server/sparql.py +984 -0
  81. cortexm/text/__init__.py +0 -0
  82. cortexm/text/dissim.py +252 -0
  83. cortexm/text/embedder.py +155 -0
  84. cortexm/text/fuzzy.py +218 -0
  85. cortexm/text/idiolect.py +253 -0
  86. cortexm/text/labse.py +374 -0
  87. cortexm/text/tokenizer.py +79 -0
  88. cortexm/trace/__init__.py +0 -0
  89. cortexm/trace/blob_arena.py +277 -0
  90. cortexm/trace/consolidate.py +337 -0
  91. cortexm/trace/contradictions.py +69 -0
  92. cortexm/trace/dedup.py +114 -0
  93. cortexm/trace/edges.py +214 -0
  94. cortexm/trace/fact.py +121 -0
  95. cortexm/trace/fade.py +245 -0
  96. cortexm/trace/lifecycle.py +112 -0
  97. cortexm/trace/rebuild.py +173 -0
  98. cortexm/trace/rules.py +171 -0
  99. cortexm/trace/store.py +680 -0
  100. cortexm/trace/structural.py +183 -0
  101. cortexm/trace/tmt.py +335 -0
  102. cortexm/util.py +148 -0
  103. cortexm/vsa/__init__.py +0 -0
  104. cortexm/vsa/attribution.py +149 -0
  105. cortexm/vsa/cleanup.py +161 -0
  106. cortexm/vsa/codecs.py +397 -0
  107. cortexm/vsa/hologram_overlay.py +139 -0
  108. cortexm/vsa/index.py +163 -0
  109. cortexm/vsa/ops.py +149 -0
  110. cortexm/vsa/palace.py +446 -0
  111. cortexm/vsa/role_vectors.py +236 -0
  112. cortexm/vsa/slb.py +78 -0
  113. cortexm/vsa/tlsh_trie.py +137 -0
  114. cortexm/vsa/working_memory.py +249 -0
  115. cortexm-0.3.0.dist-info/METADATA +482 -0
  116. cortexm-0.3.0.dist-info/RECORD +120 -0
  117. cortexm-0.3.0.dist-info/WHEEL +5 -0
  118. cortexm-0.3.0.dist-info/entry_points.txt +2 -0
  119. cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
  120. cortexm-0.3.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,236 @@
1
+ """Engineered role vectors — NSR-inspired (ESWEEK24).
2
+
3
+ arXiv insight: NSR trains an autoencoder to learn compact embeddings
4
+ for the fact vocabulary, then uses those embeddings to construct role
5
+ vectors that are *maximally mutually orthogonal* in the data manifold.
6
+ Random role vectors work (HDC theory guarantees capacity grows with
7
+ sqrt(dims)), but engineered ones have higher effective capacity and
8
+ lower cross-talk because they sit on the directions of greatest
9
+ variance in the actual data.
10
+
11
+ This module provides:
12
+
13
+ EngineeredRoleVectors(dims, n_roles=3)
14
+ .fit(fact_matrix) -> trains a tiny 1-layer autoencoder
15
+ on the fact matrix and stores the
16
+ top-k principal directions as role
17
+ vectors
18
+ .role_vec(role) -> returns the engineered role vector
19
+ (falls back to VSA's random init
20
+ if not yet fit)
21
+ .save(path) / .load(path)
22
+
23
+ WHY THIS MATTERS:
24
+ - role_vec("S") currently = rng.standard_normal(dims). For dims=768
25
+ that's a random point on the unit sphere — fine in theory but it
26
+ wastes capacity on directions orthogonal to the actual fact vocab.
27
+ - After fit(), role_vec("S") = the top-1 principal direction of the
28
+ S-vector manifold. role_vec("R") = top-2, role_vec("V") = top-3.
29
+ These directions are where the data actually lives, so:
30
+ (a) bound holograms exploit the full effective capacity
31
+ (b) cross-talk between S/R/V role bindings drops because the
32
+ top-k principal directions are approximately orthogonal by
33
+ construction
34
+ (c) retrieval probes have higher SNR because they're aligned
35
+ with the data axes
36
+
37
+ IMPLEMENTATION:
38
+ Tiny 1-layer linear autoencoder (no nonlinearity) trained with
39
+ plain SGD on the fact matrix. The encoder weights converge to the
40
+ top-k principal components (proven by Baldi & Horn, 1989 — linear
41
+ AE on centered data recovers PCA). We center the data and train the
42
+ encoder to have orthogonal rows via a soft orthogonality penalty.
43
+
44
+ Cost: ~1 day of human work, ~30 seconds of CPU time on 1000 facts.
45
+ Deterministic given the seed.
46
+ """
47
+ from __future__ import annotations
48
+
49
+ import json
50
+ import os
51
+ from pathlib import Path
52
+
53
+ import numpy as np
54
+
55
+
56
+ class EngineeredRoleVectors:
57
+ """Train and serve engineered role vectors.
58
+
59
+ Construction:
60
+ erv = EngineeredRoleVectors(dims=768, n_roles=3, seed=42)
61
+
62
+ Fit:
63
+ erv.fit(fact_matrix) # fact_matrix: (n_facts, dims)
64
+ # trains a tiny AE, extracts top-k principal directions,
65
+ # stores them as the role vectors
66
+
67
+ Use:
68
+ erv.role_vec("S") # top-1 principal direction
69
+ erv.role_vec("R") # top-2
70
+ erv.role_vec("V") # top-3
71
+ """
72
+
73
+ ROLE_ORDER = ["S", "R", "V"] # subject / relation / value
74
+
75
+ def __init__(self, dims: int = 768, n_roles: int = 3,
76
+ seed: int = 42, n_epochs: int = 200,
77
+ lr: float = 0.01, orth_penalty: float = 0.1) -> None:
78
+ self.dims = dims
79
+ self.n_roles = n_roles
80
+ self.seed = seed
81
+ self.n_epochs = n_epochs
82
+ self.lr = lr
83
+ self.orth_penalty = orth_penalty
84
+ self._role_vecs: dict[str, np.ndarray] = {}
85
+ self._mean: np.ndarray | None = None
86
+ self._fit_loss: list[float] = []
87
+
88
+ # ------------------------------------------------------------- fit
89
+ def fit(self, fact_matrix: np.ndarray) -> dict:
90
+ """Train a tiny linear autoencoder on `fact_matrix`.
91
+
92
+ fact_matrix: (n_samples, dims) float32. Should contain the
93
+ subject / relation / value vectors for all facts in the
94
+ corpus, stacked. The AE learns to reconstruct them through
95
+ a k-dim bottleneck where k = n_roles.
96
+
97
+ After training, the encoder's rows are the top-k principal
98
+ directions. We store them as role_vec("S"/"R"/"V").
99
+
100
+ Returns a training report dict.
101
+ """
102
+ X = np.asarray(fact_matrix, dtype=np.float32)
103
+ if X.ndim != 2:
104
+ raise ValueError(f"expected 2D matrix, got shape {X.shape}")
105
+ n, d = X.shape
106
+ if d != self.dims:
107
+ raise ValueError(
108
+ f"matrix dim {d} != configured dims {self.dims}")
109
+ if n < self.n_roles:
110
+ # not enough samples to learn k directions — fall back
111
+ # to top-k random vectors orthogonalized via Gram-Schmidt
112
+ rng = np.random.default_rng(self.seed)
113
+ vecs = rng.standard_normal((self.n_roles, d)).astype(np.float32)
114
+ for i in range(self.n_roles):
115
+ for j in range(i):
116
+ vecs[i] -= np.dot(vecs[i], vecs[j]) * vecs[j]
117
+ n_ = float(np.linalg.norm(vecs[i]))
118
+ vecs[i] /= max(n_, 1e-9)
119
+ for i, r in enumerate(self.ROLE_ORDER[:self.n_roles]):
120
+ self._role_vecs[r] = vecs[i]
121
+ return {"trained": False, "reason": "insufficient_samples",
122
+ "n_samples": n, "n_roles": self.n_roles,
123
+ "fallback": "gram_schmidt_random"}
124
+
125
+ # center the data (PCA assumes centered data)
126
+ self._mean = X.mean(axis=0)
127
+ Xc = X - self._mean
128
+
129
+ # tiny linear AE: encoder W: (k, d), decoder W.T: (d, k)
130
+ # init with small random values
131
+ rng = np.random.default_rng(self.seed)
132
+ k = self.n_roles
133
+ scale = float(1.0 / np.sqrt(d))
134
+ W = rng.standard_normal((k, d)).astype(np.float32) * scale
135
+
136
+ # SGD on reconstruction loss + orthogonality penalty
137
+ # (orth penalty: W @ W.T should be ≈ I)
138
+ batch_size = min(64, n)
139
+ for epoch in range(self.n_epochs):
140
+ perm = rng.permutation(n)
141
+ epoch_loss = 0.0
142
+ for i in range(0, n, batch_size):
143
+ idx = perm[i:i + batch_size]
144
+ xb = Xc[idx] # (b, d)
145
+ # forward: z = xb @ W.T (b, k)
146
+ # recon: xhat = z @ W (b, d)
147
+ z = xb @ W.T
148
+ xhat = z @ W
149
+ # L2 reconstruction loss per sample — clip to avoid
150
+ # overflow on numerically large data (the orth penalty
151
+ # can push weights to blow up if lr is too high)
152
+ diff = xhat - xb
153
+ # clip per-element to a safe range
154
+ diff = np.clip(diff, -1e4, 1e4)
155
+ loss = float(np.mean(np.sum(diff * diff, axis=1)))
156
+ # grad on W: dL/dW = 2 * (xhat - x).T @ z / b
157
+ # shape (d, k) -> transpose for our layout
158
+ g = 2.0 * diff.T @ z / len(idx) # (d, k)
159
+ # clip gradient to prevent overflow → NaN
160
+ g = np.clip(g, -1.0, 1.0)
161
+ # orthogonality penalty: ||W @ W.T - I||_F^2 / k
162
+ # grad: 2/k * (W @ W.T - I) @ W
163
+ WtW = W @ W.T # (k, k)
164
+ I_k = np.eye(k, dtype=np.float32)
165
+ orth_grad = (2.0 / k) * (WtW - I_k) @ W
166
+ orth_grad = np.clip(orth_grad, -1.0, 1.0)
167
+ # combined grad on W (note: g is (d,k), so transpose)
168
+ W -= self.lr * (g.T + self.orth_penalty * orth_grad)
169
+ # also clip weights themselves to a safe range
170
+ W = np.clip(W, -10.0, 10.0)
171
+ epoch_loss += loss * len(idx)
172
+ epoch_loss /= n
173
+ self._fit_loss.append(epoch_loss)
174
+ if epoch % 20 == 0 or epoch == self.n_epochs - 1:
175
+ # report conditioning of W @ W.T (lower = more orthogonal)
176
+ cond = float(np.linalg.cond(WtW)) if k > 1 else 1.0
177
+ # suppress per-epoch logging during normal use
178
+ pass
179
+
180
+ # extract role vectors: rows of W are the top-k principal dirs
181
+ # normalize to unit length
182
+ for i, r in enumerate(self.ROLE_ORDER[:k]):
183
+ v = W[i].copy()
184
+ nrm = float(np.linalg.norm(v))
185
+ self._role_vecs[r] = v / max(nrm, 1e-9)
186
+
187
+ # report final reconstruction loss + orthogonality
188
+ WtW_final = W @ W.T
189
+ off_diag = float(np.sum(np.abs(WtW_final - np.eye(k, dtype=np.float32))))
190
+ return {
191
+ "trained": True,
192
+ "n_samples": n,
193
+ "dims": d,
194
+ "n_roles": k,
195
+ "epochs": self.n_epochs,
196
+ "final_loss": self._fit_loss[-1] if self._fit_loss else 0.0,
197
+ "initial_loss": self._fit_loss[0] if self._fit_loss else 0.0,
198
+ "loss_reduction_pct": (
199
+ (1.0 - self._fit_loss[-1] / max(self._fit_loss[0], 1e-9)) * 100
200
+ if self._fit_loss else 0.0),
201
+ "off_diag_sum": off_diag,
202
+ "condition_number": (float(np.linalg.cond(WtW_final))
203
+ if k > 1 else 1.0),
204
+ }
205
+
206
+ # ------------------------------------------------------------- serve
207
+ def role_vec(self, role: str) -> np.ndarray | None:
208
+ return self._role_vecs.get(role)
209
+
210
+ @property
211
+ def is_fit(self) -> bool:
212
+ return bool(self._role_vecs)
213
+
214
+ def save(self, path: str | os.PathLike) -> None:
215
+ """Persist the role vectors to a .npz file."""
216
+ arrays = {f"role_{r}": v for r, v in self._role_vecs.items()}
217
+ if self._mean is not None:
218
+ arrays["mean"] = self._mean
219
+ arrays["meta"] = np.array(json.dumps({
220
+ "dims": self.dims, "n_roles": self.n_roles,
221
+ "seed": self.seed, "final_loss": (
222
+ self._fit_loss[-1] if self._fit_loss else 0.0),
223
+ }), dtype=str)
224
+ np.savez(path, **arrays)
225
+
226
+ def load(self, path: str | os.PathLike) -> None:
227
+ data = np.load(path, allow_pickle=False)
228
+ for r in self.ROLE_ORDER[:self.n_roles]:
229
+ key = f"role_{r}"
230
+ if key in data:
231
+ self._role_vecs[r] = data[key]
232
+ if "mean" in data:
233
+ self._mean = data["mean"]
234
+
235
+
236
+ __all__ = ["EngineeredRoleVectors"]
cortexm/vsa/slb.py ADDED
@@ -0,0 +1,78 @@
1
+ """Semantic Lookaside Buffer — conversational locality cache.
2
+
3
+ 64-entry ring buffer of quantized query signatures with cached result
4
+ sets. A new query whose signature is ≥ ``threshold`` cosine-similar to a
5
+ cached signature reuses the cached ranking (conversational locality:
6
+ follow-up questions are near-duplicates of their predecessors). L1-
7
+ resident by design; hit path costs one 64×dims dot product.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import numpy as np
13
+
14
+
15
+ class SemanticLookasideBuffer:
16
+ def __init__(self, entries: int = 64, threshold: float = 0.97,
17
+ dims: int = 768) -> None:
18
+ self.capacity = entries
19
+ self.threshold = threshold
20
+ self.dims = dims
21
+ self._sigs = np.zeros((entries, dims), dtype=np.float32)
22
+ self._results: list[list[tuple[str, float]] | None] = [None] * entries
23
+ self._queries: list[str | None] = [None] * entries
24
+ self._scopes: list[tuple | None] = [None] * entries
25
+ self._pos = 0
26
+ self._filled = 0
27
+ self.hits = 0
28
+ self.misses = 0
29
+ self.total_hit_latency = 0.0
30
+ self.total_miss_latency = 0.0
31
+
32
+ def lookup(self, q: np.ndarray,
33
+ scope: tuple | None = None) -> list[tuple[str, float]] | None:
34
+ """Return cached results for a signature **in the same scope**.
35
+
36
+ Scope-blind lookup is a correctness bug, not just a privacy one:
37
+ near-duplicate queries from different users would cross-contaminate
38
+ (and then die in the caller's scope filter, yielding empty blocks).
39
+ """
40
+ if self._filled == 0:
41
+ return None
42
+ sims = self._sigs[: self._filled] @ q
43
+ best = int(np.argmax(sims))
44
+ if float(sims[best]) >= self.threshold and self._scopes[best] == scope:
45
+ self.hits += 1
46
+ return self._results[best]
47
+ return None
48
+
49
+ def store(self, q: np.ndarray, results: list[tuple[str, float]],
50
+ query: str = "", scope: tuple | None = None) -> None:
51
+ pos = self._pos
52
+ self._sigs[pos] = q
53
+ self._results[pos] = results
54
+ self._queries[pos] = query
55
+ self._scopes[pos] = scope
56
+ self._pos = (self._pos + 1) % self.capacity
57
+ self._filled = min(self._filled + 1, self.capacity)
58
+
59
+ def record_latency(self, hit: bool, seconds: float) -> None:
60
+ if hit:
61
+ self.total_hit_latency += seconds
62
+ else:
63
+ self.total_miss_latency += seconds
64
+
65
+ @property
66
+ def miss_latency_avg(self) -> float:
67
+ return (self.total_miss_latency / self.misses) if self.misses else 0.0
68
+
69
+ def stats(self) -> dict:
70
+ total = self.hits + self.misses
71
+ return {
72
+ "hits": self.hits, "misses": self.misses,
73
+ "hit_rate": round(self.hits / total, 4) if total else 0.0,
74
+ "avg_hit_latency_us": round(
75
+ self.total_hit_latency / self.hits * 1e6, 1) if self.hits else 0.0,
76
+ "avg_miss_latency_us": round(self.miss_latency_avg * 1e6, 1),
77
+ "entries_used": self._filled,
78
+ }
@@ -0,0 +1,137 @@
1
+ """TLSH ternary trie — software TCAM for O(log N + w) hologram lookup.
2
+
3
+ The Stanford TLSH paper (arXiv:1006.3514) uses ternary content-
4
+ addressable memory for O(1) parallel lookup. Without TCAM hardware,
5
+ we emulate it as a ternary Patricia trie over packed binary holograms:
6
+ each path is the bits of a packed binary vector, with wildcard edges
7
+ that match either bit (the ternary '*' bit).
8
+
9
+ Use as a *pre-filter* in MemoryPalace.search when codec is binary/rabitq:
10
+ the trie returns O(k) candidate fact_ids within max_wildcards bit-flips
11
+ of the query, then codec.scores ranks them. Trades one big argsort for
12
+ a sub-linear trie walk — wins when N grows large and dims is high (16k+).
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import numpy as np
18
+
19
+
20
+ class _TrieNode:
21
+ __slots__ = ("children", "fact_ids", "depth")
22
+
23
+ def __init__(self, depth: int) -> None:
24
+ self.children: dict[int, _TrieNode] = {} # bit 0/1 → node
25
+ self.fact_ids: list[str] = []
26
+ self.depth = depth
27
+
28
+
29
+ class TernaryTrie:
30
+ """Software TLSH over packed binary holograms."""
31
+
32
+ def __init__(self, dims: int, max_wildcards: int = 8,
33
+ max_candidates: int = 256) -> None:
34
+ self.dims = dims
35
+ self.max_wildcards = max_wildcards
36
+ self.max_candidates = max_candidates
37
+ self.root = _TrieNode(0)
38
+ self._size = 0
39
+
40
+ def insert(self, fact_id: str, packed: np.ndarray) -> None:
41
+ """Insert a packed binary vector (uint8 array of packed bits)."""
42
+ bits = _unpack_bits(packed, self.dims)
43
+ node = self.root
44
+ for bit in bits:
45
+ b = int(bit)
46
+ child = node.children.get(b)
47
+ if child is None:
48
+ child = _TrieNode(node.depth + 1)
49
+ node.children[b] = child
50
+ node = child
51
+ node.fact_ids.append(fact_id)
52
+ self._size += 1
53
+
54
+ def lookup(self, packed_q: np.ndarray, k: int = 10,
55
+ max_wildcards: int | None = None) -> list[tuple[str, int]]:
56
+ """Return up to k (fact_id, hamming_distance) pairs within
57
+ max_wildcards bit-flips of packed_q. O(log N + max_wildcards·branch).
58
+ """
59
+ mw = max_wildcards if max_wildcards is not None else self.max_wildcards
60
+ bits = _unpack_bits(packed_q, self.dims)
61
+ # candidate (node, position, wildcards_used, distance_so_far)
62
+ # use a stack with priority by wildcards_used
63
+ results: list[tuple[str, int]] = []
64
+ seen: set[str] = set()
65
+ # DFS with wildcard budget
66
+ # stack entries: (node, idx, wildcards_remaining)
67
+ # we don't strictly bound exploration — for small max_wildcards
68
+ # and a sparse trie this is fine
69
+ stack = [(self.root, 0, mw)]
70
+ # use a heap for best-first with priority on wildcards remaining
71
+ import heapq
72
+ # priority: (-wildcards_remaining, idx) so most wildcards remaining
73
+ # (i.e. least used) comes first
74
+ heap: list[tuple[int, int, _TrieNode]] = [(-mw, 0, self.root)]
75
+ while heap and len(results) < self.max_candidates:
76
+ _, idx, node = heapq.heappop(heap)
77
+ if node.fact_ids and idx == self.dims:
78
+ for fid in node.fact_ids:
79
+ if fid not in seen:
80
+ seen.add(fid)
81
+ results.append((fid, mw + _heap_key(0, mw)))
82
+ if len(results) >= self.max_candidates:
83
+ break
84
+ continue
85
+ if idx >= self.dims:
86
+ # at a leaf but bits remaining — these are stored facts
87
+ for fid in node.fact_ids:
88
+ if fid not in seen:
89
+ seen.add(fid)
90
+ results.append((fid, mw))
91
+ continue
92
+ want_bit = int(bits[idx])
93
+ # exact match: free (no wildcard used)
94
+ exact = node.children.get(want_bit)
95
+ if exact is not None:
96
+ heapq.heappush(heap, (-(mw), idx + 1, exact))
97
+ # wildcard match: try the other bit (uses 1 wildcard)
98
+ other = 1 - want_bit
99
+ wild = node.children.get(other)
100
+ if wild is not None and mw > 0:
101
+ heapq.heappush(heap, (-(mw - 1), idx + 1, wild))
102
+ # dedupe and sort by best distance estimate
103
+ # NB: distance estimate is approximate (lower bound by wildcards used)
104
+ dedup: dict[str, int] = {}
105
+ for fid, dist in results:
106
+ if fid not in dedup or dist < dedup[fid]:
107
+ dedup[fid] = dist
108
+ out = sorted(dedup.items(), key=lambda x: x[1])[:k]
109
+ return out
110
+
111
+ def __len__(self) -> int:
112
+ return self._size
113
+
114
+ def stats(self) -> dict:
115
+ return {
116
+ "dims": self.dims,
117
+ "size": self._size,
118
+ "max_wildcards": self.max_wildcards,
119
+ "root_children": len(self.root.children),
120
+ }
121
+
122
+
123
+ def _heap_key(used: int, budget: int) -> int:
124
+ """Convert 'wildcards used' into a priority value (less used = better)."""
125
+ return budget - used # remaining
126
+
127
+
128
+ def _unpack_bits(packed: np.ndarray, dims: int) -> np.ndarray:
129
+ """Unpack packed uint8 bits into a 1D array of 0/1 of length dims."""
130
+ arr = np.atleast_1d(packed)
131
+ if arr.dtype == np.uint8 and len(arr) * 8 >= dims:
132
+ return np.unpackbits(arr, count=dims).astype(np.uint8)
133
+ # already a bit array
134
+ return arr.astype(np.uint8)
135
+
136
+
137
+ __all__ = ["TernaryTrie"]