cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
cortexm/bridge/rerank.py
ADDED
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
"""Cross-encoder-style reranking for μ=0 retrieval.
|
|
2
|
+
|
|
3
|
+
SOTA insight (web search 2026-08):
|
|
4
|
+
* Hybrid retrieval (BM25 + dense) is the dominant precision@k lever.
|
|
5
|
+
* Cross-encoder reranking takes top-N (e.g. 50) from a bi-encoder and
|
|
6
|
+
re-scores with a model that reads (query, doc) jointly. This lifts
|
|
7
|
+
precision@5 by 10-20pp on MS-MARCO and similar benchmarks.
|
|
8
|
+
* HippoRAG 2 and Mem0 both do some form of two-stage retrieval.
|
|
9
|
+
* PRF (Pseudo-Relevance Feedback / Rocchio) takes top-3 hits, averages
|
|
10
|
+
their embeddings with the query, and re-retrieves — a 2-5pp lift
|
|
11
|
+
on TREC benchmarks.
|
|
12
|
+
|
|
13
|
+
We cannot ship a learned cross-encoder (μ=0 mandate). What we CAN do:
|
|
14
|
+
* Render each fact (subject, relation, value) into a SHORT natural-
|
|
15
|
+
language string ("the name of beam_1 is Jennifer Mccall") and
|
|
16
|
+
embed THAT with our HashingEmbedder. The chunk text vectors in
|
|
17
|
+
the palace are long, fact-dense, and dilute lexical similarity —
|
|
18
|
+
a fact-level embedding is focused and cosine sim is much sharper.
|
|
19
|
+
* Use the cosine sim between the query embedding and the fact NL
|
|
20
|
+
embedding as a RE-RANK signal on the top-K candidates after the
|
|
21
|
+
initial fusion pass. This is exactly the architecture SlopFilter/
|
|
22
|
+
ColBERT/MS-MARCO cross-encoders use, just with a lexical embedder.
|
|
23
|
+
* PRF: average top-3 fact NL embeddings with query emb (Rocchio
|
|
24
|
+
alpha=0.6 / beta=0.4) and re-rank the wider candidate pool.
|
|
25
|
+
|
|
26
|
+
This module is imported lazily by MemoryReader.search() so the rest
|
|
27
|
+
of the fabric is unaffected. The bench config "+rerank" enables it.
|
|
28
|
+
"""
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
from typing import Iterable
|
|
32
|
+
|
|
33
|
+
import numpy as np
|
|
34
|
+
|
|
35
|
+
from cortexm.trace.fact import Fact
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
# ---------------------------------------------------------------- NL rendering
|
|
39
|
+
# The fact is structured (subject, relation, value). For cross-encoder
|
|
40
|
+
# reranking we need a natural-language string that the HashingEmbedder
|
|
41
|
+
# can lexically match against the query. The exact template matters
|
|
42
|
+
# because the embedder hashes char n-grams (3,4,5) — surface forms
|
|
43
|
+
# drive similarity.
|
|
44
|
+
#
|
|
45
|
+
# Templates picked from inspection of BEAM-10M query patterns:
|
|
46
|
+
# "What is the name of beam_1?" → "the name of beam_1 is Jennifer"
|
|
47
|
+
# "Where does beam_1 live?" → "beam_1 lives_in Seattle"
|
|
48
|
+
# "What is beam_1's age?" → "beam_1 age is 59"
|
|
49
|
+
# We pick the SUBJECT-centric form because that's how the bench query
|
|
50
|
+
# is phrased ("the {relation} of {subject}").
|
|
51
|
+
|
|
52
|
+
_TEMPLATES: dict[str, str] = {
|
|
53
|
+
# identity relations
|
|
54
|
+
"name": "the name of {s} is {v}",
|
|
55
|
+
"age": "the age of {s} is {v}",
|
|
56
|
+
"gender": "the gender of {s} is {v}",
|
|
57
|
+
"location": "the location of {s} is {v}",
|
|
58
|
+
"profession": "the profession of {s} is {v}",
|
|
59
|
+
"birthday": "the birthday of {s} is {v}",
|
|
60
|
+
# kinship
|
|
61
|
+
"parent": "the parent of {s} is {v}",
|
|
62
|
+
"partner": "the partner of {s} is {v}",
|
|
63
|
+
"spouse": "the spouse of {s} is {v}",
|
|
64
|
+
"child": "the child of {s} is {v}",
|
|
65
|
+
"sibling": "the sibling of {s} is {v}",
|
|
66
|
+
"friend": "the friend of {s} is {v}",
|
|
67
|
+
"colleague": "the colleague of {s} is {v}",
|
|
68
|
+
# work / education
|
|
69
|
+
"works_at": "{s} works at {v}",
|
|
70
|
+
"role": "{s} works as {v}",
|
|
71
|
+
"studied": "{s} studied {v}",
|
|
72
|
+
"studied_at": "{s} studied at {v}",
|
|
73
|
+
# misc
|
|
74
|
+
"lives_in": "{s} lives in {v}",
|
|
75
|
+
"moved_to": "{s} moved to {v}",
|
|
76
|
+
"prefers": "{s} prefers {v}",
|
|
77
|
+
"likes": "{s} likes {v}",
|
|
78
|
+
"dislikes": "{s} dislikes {v}",
|
|
79
|
+
"has_skill": "{s} has skill {v}",
|
|
80
|
+
"speaks": "{s} speaks {v}",
|
|
81
|
+
"has_pet": "{s} has pet {v}",
|
|
82
|
+
"hobby": "{s} hobby is {v}",
|
|
83
|
+
"alias": "{s} also known as {v}",
|
|
84
|
+
"goal": "{s} goal is {v}",
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
_DEFAULT_TEMPLATE = "{s} | {r} | {v}" # raw 3-tuple fallback
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def fact_nl(fact: Fact) -> str:
|
|
91
|
+
"""Render a fact into a short natural-language string.
|
|
92
|
+
|
|
93
|
+
The template is keyed by relation; unknown relations fall back to
|
|
94
|
+
the raw 3-tuple which the embedder will still lex-match against.
|
|
95
|
+
"""
|
|
96
|
+
tpl = _TEMPLATES.get(fact.relation, _DEFAULT_TEMPLATE)
|
|
97
|
+
return tpl.format(s=fact.subject, r=fact.relation, v=fact.value).lower()
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
# ---------------------------------------------------------------- reranker
|
|
101
|
+
class FactReranker:
|
|
102
|
+
"""Cross-encoder-style reranker over fact NL strings.
|
|
103
|
+
|
|
104
|
+
Stateless (no learned weights) — the only parameter is the embedder
|
|
105
|
+
used to embed query and fact NL. Default = the palace's HashingEmbedder
|
|
106
|
+
so the rerank score is in the same space as the initial VSA hits.
|
|
107
|
+
|
|
108
|
+
Usage:
|
|
109
|
+
reranker = FactReranker(palace.embedder)
|
|
110
|
+
reranked = reranker.rerank(query_vec, facts, top_k=5)
|
|
111
|
+
"""
|
|
112
|
+
|
|
113
|
+
def __init__(self, embedder, *,
|
|
114
|
+
alpha: float = 0.55, # weight on the rerank score
|
|
115
|
+
beta: float = 0.45, # weight on the original score
|
|
116
|
+
prf_alpha: float = 0.6, # query weight in PRF
|
|
117
|
+
prf_beta: float = 0.4, # top-3 mean weight in PRF
|
|
118
|
+
prf_topn: int = 3,
|
|
119
|
+
cache_cap: int = 8192) -> None:
|
|
120
|
+
self.embedder = embedder
|
|
121
|
+
self.alpha = alpha
|
|
122
|
+
self.beta = beta
|
|
123
|
+
self.prf_alpha = prf_alpha
|
|
124
|
+
self.prf_beta = prf_beta
|
|
125
|
+
self.prf_topn = prf_topn
|
|
126
|
+
self._cache: dict[str, np.ndarray] = {}
|
|
127
|
+
self._cache_cap = cache_cap
|
|
128
|
+
|
|
129
|
+
def _fact_emb(self, fact: Fact) -> np.ndarray:
|
|
130
|
+
"""Embed the fact's NL rendering, with a small LRU cache."""
|
|
131
|
+
nl = fact_nl(fact)
|
|
132
|
+
v = self._cache.get(nl)
|
|
133
|
+
if v is not None:
|
|
134
|
+
return v
|
|
135
|
+
v = self.embedder.embed(nl)
|
|
136
|
+
if len(self._cache) < self._cache_cap:
|
|
137
|
+
self._cache[nl] = v
|
|
138
|
+
return v
|
|
139
|
+
|
|
140
|
+
def rerank(self, query_vec: np.ndarray,
|
|
141
|
+
facts: list[Fact],
|
|
142
|
+
scores: dict[str, float],
|
|
143
|
+
top_k: int = 5,
|
|
144
|
+
*, enable_prf: bool = True) -> tuple[list[Fact], dict[str, float]]:
|
|
145
|
+
"""Rerank facts by cosine(query, fact_nl) and return top_k.
|
|
146
|
+
|
|
147
|
+
Returns (reranked_facts, new_scores). The new scores are
|
|
148
|
+
blended: alpha * rerank_score + beta * original_score, where
|
|
149
|
+
both are first min-max normalized to [0,1] across the candidate
|
|
150
|
+
pool so the blend is scale-invariant.
|
|
151
|
+
|
|
152
|
+
If enable_prf, run a 2nd pass where the query embedding is
|
|
153
|
+
shifted toward the mean of the top-3 fact NL embeddings (Rocchio
|
|
154
|
+
PRF) — this lifts precision@k on TREC by 2-5pp.
|
|
155
|
+
"""
|
|
156
|
+
if not facts:
|
|
157
|
+
return facts, scores
|
|
158
|
+
# embed all candidates (cached)
|
|
159
|
+
embs = np.stack([self._fact_emb(f) for f in facts]) # (N, D)
|
|
160
|
+
# cosine sim with query (all L2-normalized at embedder level)
|
|
161
|
+
rr = embs @ query_vec # (N,)
|
|
162
|
+
|
|
163
|
+
# PRF: shift query toward mean of top-3 fact NL embeddings
|
|
164
|
+
if enable_prf and len(facts) >= 2:
|
|
165
|
+
n_prf = min(self.prf_topn, len(facts))
|
|
166
|
+
top_idx = np.argsort(-rr)[:n_prf]
|
|
167
|
+
prf_vec = embs[top_idx].mean(axis=0)
|
|
168
|
+
# renormalize
|
|
169
|
+
n = float(np.linalg.norm(prf_vec))
|
|
170
|
+
if n > 0:
|
|
171
|
+
prf_vec = prf_vec / n
|
|
172
|
+
new_q = self.prf_alpha * query_vec + self.prf_beta * prf_vec
|
|
173
|
+
n2 = float(np.linalg.norm(new_q))
|
|
174
|
+
if n2 > 0:
|
|
175
|
+
new_q = new_q / n2
|
|
176
|
+
# blend the two rerank signals
|
|
177
|
+
rr_prf = embs @ new_q
|
|
178
|
+
rr = 0.5 * rr + 0.5 * rr_prf
|
|
179
|
+
|
|
180
|
+
# min-max normalize rr and original scores
|
|
181
|
+
def _norm(x: np.ndarray) -> np.ndarray:
|
|
182
|
+
mn, mx = float(x.min()), float(x.max())
|
|
183
|
+
if mx - mn < 1e-9:
|
|
184
|
+
return np.ones_like(x) * 0.5
|
|
185
|
+
return (x - mn) / (mx - mn)
|
|
186
|
+
rr_n = _norm(rr)
|
|
187
|
+
orig = np.array([scores.get(f.id, 0.0) for f in facts],
|
|
188
|
+
dtype=np.float32)
|
|
189
|
+
orig_n = _norm(orig)
|
|
190
|
+
|
|
191
|
+
blended = self.alpha * rr_n + self.beta * orig_n
|
|
192
|
+
# sort descending; tie-break on fact content (deterministic)
|
|
193
|
+
order = sorted(range(len(facts)),
|
|
194
|
+
key=lambda i: (-float(blended[i]),
|
|
195
|
+
facts[i].subject,
|
|
196
|
+
facts[i].relation,
|
|
197
|
+
facts[i].value))
|
|
198
|
+
top_idx = order[:top_k]
|
|
199
|
+
new_facts = [facts[i] for i in top_idx]
|
|
200
|
+
new_scores = {facts[i].id: float(blended[i]) for i in top_idx}
|
|
201
|
+
return new_facts, new_scores
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
__all__ = ["FactReranker", "fact_nl"]
|