cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""ProtoDash attribution — submodular prototype selection for retrieval.
|
|
2
|
+
|
|
3
|
+
Given a query embedding and a candidate set of retrieved chunks,
|
|
4
|
+
ProtoDash (arXiv:1707.01212) selects up to m prototypes that best
|
|
5
|
+
reconstruct the query in kernel space, with non-negative weights.
|
|
6
|
+
|
|
7
|
+
Produces an audit trail: for every retrieved fact, a weight in [0,1]
|
|
8
|
+
indicating its contribution to the query reconstruction. Stored in the
|
|
9
|
+
fact's provenance dict under {"protodash_weight": 0.32}.
|
|
10
|
+
|
|
11
|
+
Pure Python + numpy + scipy.optimize.nnls. Greedy submodular selection
|
|
12
|
+
gives a (1-1/e) approximation guarantee of the optimal selection.
|
|
13
|
+
|
|
14
|
+
Also provides sentence-level cosine similarity scoring — classifies
|
|
15
|
+
each retrieved sentence's contribution as Very High (>0.8), High (>0.6),
|
|
16
|
+
Medium (>0.4), Low (>0.2), or Negligible.
|
|
17
|
+
|
|
18
|
+
arxiv research: arXiv:1707.01212 (ProtoDash, 2017).
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import numpy as np
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class ProtoDashAttributer:
|
|
27
|
+
"""Source attribution via submodular prototype selection."""
|
|
28
|
+
|
|
29
|
+
def __init__(self, kernel: str = "linear", gamma: float = 0.1) -> None:
|
|
30
|
+
self.kernel = kernel
|
|
31
|
+
self.gamma = gamma
|
|
32
|
+
|
|
33
|
+
def _k(self, A: np.ndarray, B: np.ndarray) -> np.ndarray:
|
|
34
|
+
"""Kernel matrix between two sets of embeddings."""
|
|
35
|
+
if self.kernel == "linear":
|
|
36
|
+
return A @ B.T
|
|
37
|
+
# RBF kernel: exp(-gamma ||a-b||^2)
|
|
38
|
+
sq = (np.sum(A ** 2, axis=1)[:, None]
|
|
39
|
+
+ np.sum(B ** 2, axis=1)[None, :]
|
|
40
|
+
- 2 * A @ B.T)
|
|
41
|
+
return np.exp(-self.gamma * np.maximum(sq, 0))
|
|
42
|
+
|
|
43
|
+
def attribute(self, query_emb: np.ndarray, candidate_embs: np.ndarray,
|
|
44
|
+
candidate_ids: list[str], m: int = 5
|
|
45
|
+
) -> list[tuple[str, float]]:
|
|
46
|
+
"""Return up to m (fact_id, weight) pairs that best reconstruct
|
|
47
|
+
the query in kernel space. Weights are non-negative and
|
|
48
|
+
normalized to sum to ~1.
|
|
49
|
+
"""
|
|
50
|
+
if not candidate_ids:
|
|
51
|
+
return []
|
|
52
|
+
X = np.atleast_2d(query_emb.astype(np.float32))
|
|
53
|
+
Y = np.atleast_2d(candidate_embs.astype(np.float32))
|
|
54
|
+
n = len(candidate_ids)
|
|
55
|
+
m = min(m, n)
|
|
56
|
+
# kernel values
|
|
57
|
+
Kxy = self._k(Y, X)[:, 0] # (n,)
|
|
58
|
+
Kyy_sum = float(self._k(X, X).sum()) # constant
|
|
59
|
+
S_idx: list[int] = []
|
|
60
|
+
for _ in range(m):
|
|
61
|
+
best, best_gain = -1, -np.inf
|
|
62
|
+
for j in range(n):
|
|
63
|
+
if j in S_idx:
|
|
64
|
+
continue
|
|
65
|
+
sj = Y[j:j + 1]
|
|
66
|
+
# marginal gain: 2*k(y_j, x) - 2*sum_{s in S} k(y_j, s) - k(y_j, y_j)
|
|
67
|
+
if S_idx:
|
|
68
|
+
Kss = self._k(sj, Y[S_idx]).sum()
|
|
69
|
+
else:
|
|
70
|
+
Kss = 0.0
|
|
71
|
+
Kjj = float(self._k(sj, sj)[0, 0])
|
|
72
|
+
gain = 2.0 * Kxy[j] - 2.0 * Kss - Kjj
|
|
73
|
+
if gain > best_gain:
|
|
74
|
+
best_gain, best = gain, j
|
|
75
|
+
if best < 0 or best_gain <= 0:
|
|
76
|
+
break
|
|
77
|
+
S_idx.append(best)
|
|
78
|
+
if not S_idx:
|
|
79
|
+
return []
|
|
80
|
+
# NNLS weights: solve K_SS w = K_SX, w >= 0
|
|
81
|
+
try:
|
|
82
|
+
from scipy.optimize import nnls
|
|
83
|
+
S = Y[S_idx]
|
|
84
|
+
K_SS = self._k(S, S) + 1e-6 * np.eye(len(S_idx))
|
|
85
|
+
K_SX = self._k(S, X)[:, 0]
|
|
86
|
+
w, _ = nnls(K_SS, K_SX)
|
|
87
|
+
w = w / max(w.sum(), 1e-9)
|
|
88
|
+
except Exception:
|
|
89
|
+
# fallback to uniform
|
|
90
|
+
w = np.ones(len(S_idx)) / len(S_idx)
|
|
91
|
+
return [(candidate_ids[S_idx[i]], float(w[i])) for i in range(len(S_idx))]
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def sentence_level_score(query: str, sentences: list[str],
|
|
95
|
+
embedder) -> list[dict]:
|
|
96
|
+
"""Score each sentence's contribution to the query.
|
|
97
|
+
|
|
98
|
+
Returns list of {sentence, score, classification} dicts.
|
|
99
|
+
Classification buckets: Very High (>0.8), High (>0.6), Medium (>0.4),
|
|
100
|
+
Low (>0.2), Negligible.
|
|
101
|
+
"""
|
|
102
|
+
if not sentences:
|
|
103
|
+
return []
|
|
104
|
+
q = embedder.embed(query)
|
|
105
|
+
out = []
|
|
106
|
+
for sent in sentences:
|
|
107
|
+
if not sent.strip():
|
|
108
|
+
continue
|
|
109
|
+
e = embedder.embed(sent)
|
|
110
|
+
cos = float(np.dot(q, e))
|
|
111
|
+
if cos > 0.8:
|
|
112
|
+
cls = "Very High"
|
|
113
|
+
elif cos > 0.6:
|
|
114
|
+
cls = "High"
|
|
115
|
+
elif cos > 0.4:
|
|
116
|
+
cls = "Medium"
|
|
117
|
+
elif cos > 0.2:
|
|
118
|
+
cls = "Low"
|
|
119
|
+
else:
|
|
120
|
+
cls = "Negligible"
|
|
121
|
+
out.append({"sentence": sent, "score": cos, "classification": cls})
|
|
122
|
+
return out
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
# Retrieval path tag enum — assigned to every retrieved fact in the audit trail
|
|
126
|
+
RETRIEVAL_PATHS = (
|
|
127
|
+
"vsa_unbind", # holographic overlay direct unbind
|
|
128
|
+
"pattern_match", # deterministic pattern extractor
|
|
129
|
+
"neural_fallback", # LLM enrichment path (opt-in)
|
|
130
|
+
"raw_chunk", # raw text chunk fallback
|
|
131
|
+
"tree_index", # TreeIndex search
|
|
132
|
+
"tlsh_trie", # TernaryTrie lookup
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def tag_retrieval_path(fact_dict: dict, path: str) -> dict:
|
|
137
|
+
"""Add retrieval_path to a fact's provenance for audit trail."""
|
|
138
|
+
if "provenance" not in fact_dict or fact_dict["provenance"] is None:
|
|
139
|
+
fact_dict["provenance"] = {}
|
|
140
|
+
fact_dict["provenance"]["retrieval_path"] = path
|
|
141
|
+
return fact_dict
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
__all__ = [
|
|
145
|
+
"ProtoDashAttributer",
|
|
146
|
+
"sentence_level_score",
|
|
147
|
+
"tag_retrieval_path",
|
|
148
|
+
"RETRIEVAL_PATHS",
|
|
149
|
+
]
|
cortexm/vsa/cleanup.py
ADDED
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
"""Modern Hopfield cleanup memory for VSA unbound residuals.
|
|
2
|
+
|
|
3
|
+
After `vsa.unbind(role, h)` produces a noisy residual h' ≈ true_filler +
|
|
4
|
+
noise, this snaps h' to the nearest stored item vector via one-step
|
|
5
|
+
modern Hopfield retrieval:
|
|
6
|
+
|
|
7
|
+
x ← Σ_i softmax(β · (X^T x)_i) · X_i
|
|
8
|
+
|
|
9
|
+
Capacity bound (Ramsauer 2020): ~0.14 d^2 patterns for one-shot perfect
|
|
10
|
+
recall — overkill for typical codebooks, so we bound the codebook by
|
|
11
|
+
memory and run a 1-2 step fixed-point iteration.
|
|
12
|
+
|
|
13
|
+
Pure numpy, no trained model. The codebook is the universe of subject/
|
|
14
|
+
relation/value strings the HashingEmbedder has seen.
|
|
15
|
+
|
|
16
|
+
arxiv research: arXiv:2409.16408 (HEN), arXiv:2301.10352 (capacity).
|
|
17
|
+
|
|
18
|
+
HMS-style improvement: when `sparse_softmax=True` (default), only the
|
|
19
|
+
top-k highest attention weights are kept per recall step. Sparse softmax
|
|
20
|
+
is more robust to outlier codebook entries — instead of letting one
|
|
21
|
+
dominant item wash out the noise via plain softmax, we keep only the
|
|
22
|
+
top-k competitors and renormalize. This recovers clean fillers even
|
|
23
|
+
when the codebook has near-duplicate entries that would otherwise
|
|
24
|
+
attenuate the signal under plain softmax.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import numpy as np
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class HopfieldCleanup:
|
|
33
|
+
"""Sparse associative cleanup for unbound VSA residuals.
|
|
34
|
+
|
|
35
|
+
Stores a codebook of clean item vectors (subjects, relations, values)
|
|
36
|
+
and snaps noisy residuals back to the nearest stored item.
|
|
37
|
+
|
|
38
|
+
When sparse_softmax=True (default), the attention weights are
|
|
39
|
+
sparsified to the top-k before renormalization — HMS-style
|
|
40
|
+
improvement for robustness against outlier codebook entries.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
def __init__(self, dims: int, beta: float = 8.0, iters: int = 1,
|
|
44
|
+
max_items: int = 50_000,
|
|
45
|
+
sparse_softmax: bool = True,
|
|
46
|
+
sparse_topk: int = 16) -> None:
|
|
47
|
+
self.dims = dims
|
|
48
|
+
self.beta = beta
|
|
49
|
+
self.iters = iters
|
|
50
|
+
self.max_items = max_items
|
|
51
|
+
self.sparse_softmax = sparse_softmax
|
|
52
|
+
self.sparse_topk = sparse_topk
|
|
53
|
+
self._items: list[np.ndarray] = []
|
|
54
|
+
self._ids: list[str] = []
|
|
55
|
+
self._mat: np.ndarray | None = None # (N, d) float32 L2-normed
|
|
56
|
+
self._dirty = False
|
|
57
|
+
|
|
58
|
+
def add(self, key: str, vec: np.ndarray) -> None:
|
|
59
|
+
"""Add an item to the cleanup codebook. Idempotent on key."""
|
|
60
|
+
if key in set(self._ids):
|
|
61
|
+
return
|
|
62
|
+
if len(self._items) >= self.max_items:
|
|
63
|
+
# LRU-ish: drop oldest 10%
|
|
64
|
+
drop = max(1, len(self._items) // 10)
|
|
65
|
+
self._items = self._items[drop:]
|
|
66
|
+
self._ids = self._ids[drop:]
|
|
67
|
+
v = np.asarray(vec, dtype=np.float32)
|
|
68
|
+
n = float(np.linalg.norm(v))
|
|
69
|
+
v = v / n if n > 0 else v
|
|
70
|
+
self._items.append(v)
|
|
71
|
+
self._ids.append(key)
|
|
72
|
+
self._dirty = True
|
|
73
|
+
|
|
74
|
+
def build(self) -> None:
|
|
75
|
+
"""Finalize the codebook as a contiguous matrix."""
|
|
76
|
+
if not self._items:
|
|
77
|
+
self._mat = None
|
|
78
|
+
return
|
|
79
|
+
self._mat = np.stack(self._items).astype(np.float32)
|
|
80
|
+
self._dirty = False
|
|
81
|
+
|
|
82
|
+
def _recall_inner(self, x: np.ndarray) -> np.ndarray:
|
|
83
|
+
"""One Hopfield recall step: x ← Σ_i attn_i * X_i."""
|
|
84
|
+
sims = self._mat @ x # (N,)
|
|
85
|
+
if self.sparse_softmax and self.sparse_topk < len(sims):
|
|
86
|
+
# sparse softmax: keep only top-k weights, zero out the rest
|
|
87
|
+
k = min(self.sparse_topk, len(sims))
|
|
88
|
+
# find top-k indices
|
|
89
|
+
topk_idx = np.argpartition(-sims, k - 1)[:k]
|
|
90
|
+
mask = np.zeros_like(sims)
|
|
91
|
+
mask[topk_idx] = 1.0
|
|
92
|
+
sims_topk = np.where(mask > 0, sims, -np.inf)
|
|
93
|
+
# softmax over only the topk entries
|
|
94
|
+
sims_max = np.max(sims_topk)
|
|
95
|
+
# avoid -inf in exp
|
|
96
|
+
sims_topk = np.where(mask > 0, sims_topk, 0.0)
|
|
97
|
+
w = np.exp(self.beta * (sims_topk - sims_max)) * mask
|
|
98
|
+
w_sum = w.sum()
|
|
99
|
+
if w_sum > 0:
|
|
100
|
+
w = w / w_sum
|
|
101
|
+
else:
|
|
102
|
+
# plain softmax (Ramsauer 2020 original)
|
|
103
|
+
w = np.exp(self.beta * (sims - sims.max()))
|
|
104
|
+
w = w / w.sum()
|
|
105
|
+
x_new = self._mat.T @ w # (d,)
|
|
106
|
+
nn = float(np.linalg.norm(x_new))
|
|
107
|
+
return x_new / nn if nn > 0 else x_new
|
|
108
|
+
|
|
109
|
+
def recall(self, noisy: np.ndarray) -> tuple[str | None, float]:
|
|
110
|
+
"""Snap noisy residual to nearest stored item.
|
|
111
|
+
|
|
112
|
+
Returns (item_key, similarity). Returns (None, 0.0) if codebook
|
|
113
|
+
is empty.
|
|
114
|
+
"""
|
|
115
|
+
if self._mat is None or (self._dirty and self._items):
|
|
116
|
+
self.build()
|
|
117
|
+
if self._mat is None or len(self._mat) == 0:
|
|
118
|
+
return None, 0.0
|
|
119
|
+
x = np.asarray(noisy, dtype=np.float32)
|
|
120
|
+
n = float(np.linalg.norm(x))
|
|
121
|
+
x = x / n if n > 0 else x
|
|
122
|
+
for _ in range(self.iters):
|
|
123
|
+
x = self._recall_inner(x)
|
|
124
|
+
# final nearest neighbor
|
|
125
|
+
sims = self._mat @ x
|
|
126
|
+
idx = int(np.argmax(sims))
|
|
127
|
+
return self._ids[idx], float(sims[idx])
|
|
128
|
+
|
|
129
|
+
def recall_topk(self, noisy: np.ndarray, k: int = 5
|
|
130
|
+
) -> list[tuple[str, float]]:
|
|
131
|
+
"""Return top-k candidates after cleanup."""
|
|
132
|
+
if self._mat is None or (self._dirty and self._items):
|
|
133
|
+
self.build()
|
|
134
|
+
if self._mat is None or len(self._mat) == 0:
|
|
135
|
+
return []
|
|
136
|
+
x = np.asarray(noisy, dtype=np.float32)
|
|
137
|
+
n = float(np.linalg.norm(x))
|
|
138
|
+
x = x / n if n > 0 else x
|
|
139
|
+
for _ in range(self.iters):
|
|
140
|
+
x = self._recall_inner(x)
|
|
141
|
+
sims = self._mat @ x
|
|
142
|
+
order = np.argsort(-sims)[:k]
|
|
143
|
+
return [(self._ids[int(i)], float(sims[i])) for i in order]
|
|
144
|
+
|
|
145
|
+
def __len__(self) -> int:
|
|
146
|
+
return len(self._items)
|
|
147
|
+
|
|
148
|
+
def stats(self) -> dict:
|
|
149
|
+
return {
|
|
150
|
+
"items": len(self._items),
|
|
151
|
+
"dims": self.dims,
|
|
152
|
+
"beta": self.beta,
|
|
153
|
+
"iters": self.iters,
|
|
154
|
+
"built": self._mat is not None and not self._dirty,
|
|
155
|
+
"bytes": (self._mat.nbytes if self._mat is not None else 0),
|
|
156
|
+
"sparse_softmax": self.sparse_softmax,
|
|
157
|
+
"sparse_topk": self.sparse_topk,
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
__all__ = ["HopfieldCleanup"]
|