cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
cortexm/bridge/ppr.py
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""Personalized PageRank read mode (HippoRAG 2 lineage).
|
|
2
|
+
|
|
3
|
+
The default reader treats retrieval as ranking; multi-hop questions
|
|
4
|
+
("what language does the team of Alice's manager use?") need GRAPH
|
|
5
|
+
diffusion: evidence two hops away should inherit activation mass from
|
|
6
|
+
the query-matched seeds. HippoRAG (arXiv:2502.14802) showed PPR over
|
|
7
|
+
an entity-fact graph is the neuro-symbolic analogue of hippocampal
|
|
8
|
+
spreading activation.
|
|
9
|
+
|
|
10
|
+
Design notes:
|
|
11
|
+
* The graph is built LOCALLY from the candidate facts of one query
|
|
12
|
+
(bounded), not globally — μ=0 intact, no offline index needed.
|
|
13
|
+
* Deterministic: nodes are sorted, power iteration runs a fixed
|
|
14
|
+
number of steps; floats do not depend on dict order.
|
|
15
|
+
* Blended into fusion as an additive boost, gated by
|
|
16
|
+
``config.ppr_enabled`` (default on — it only fires for
|
|
17
|
+
multihop/recall intents).
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def build_fact_graph(facts: list) -> tuple[dict[str, list[str]], set[str]]:
|
|
24
|
+
"""Bipartite graph: entity nodes <-> fact nodes.
|
|
25
|
+
|
|
26
|
+
Edges: entity -- fact (subject), fact -- entity (value), plus
|
|
27
|
+
fact -- fact (CONTRADICTS / TEMPORALLY_PRECEDED_BY already live in
|
|
28
|
+
the trace; the caller passes edges separately).
|
|
29
|
+
Returns (adjacency, fact_node_ids).
|
|
30
|
+
"""
|
|
31
|
+
adj: dict[str, list[str]] = {}
|
|
32
|
+
fact_ids: set[str] = set()
|
|
33
|
+
|
|
34
|
+
def add_edge(a: str, b: str) -> None:
|
|
35
|
+
adj.setdefault(a, []).append(b)
|
|
36
|
+
adj.setdefault(b, []).append(a)
|
|
37
|
+
|
|
38
|
+
for f in facts:
|
|
39
|
+
fact_ids.add(f.id)
|
|
40
|
+
if f.subject:
|
|
41
|
+
add_edge(f"e:{f.subject}", f.id)
|
|
42
|
+
if f.value:
|
|
43
|
+
add_edge(f.id, f"e:{f.value}")
|
|
44
|
+
return adj, fact_ids
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def personalized_pagerank(adj: dict[str, list[str]], seeds: list[str],
|
|
48
|
+
damping: float = 0.85, iters: int = 12,
|
|
49
|
+
) -> dict[str, float]:
|
|
50
|
+
"""Power iteration with teleport ONLY to seed nodes.
|
|
51
|
+
|
|
52
|
+
Deterministic: iteration visits nodes in sorted order.
|
|
53
|
+
"""
|
|
54
|
+
nodes = sorted(adj.keys())
|
|
55
|
+
if not nodes:
|
|
56
|
+
return {}
|
|
57
|
+
seed_set = {s for s in seeds if s in adj}
|
|
58
|
+
if not seed_set:
|
|
59
|
+
return {}
|
|
60
|
+
n = len(nodes)
|
|
61
|
+
rank = {u: (1.0 / len(seed_set)) if u in seed_set else 0.0 for u in nodes}
|
|
62
|
+
teleport = {u: (1.0 / len(seed_set)) if u in seed_set else 0.0 for u in nodes}
|
|
63
|
+
out_deg = {u: max(1, len(adj[u])) for u in nodes}
|
|
64
|
+
for _ in range(iters):
|
|
65
|
+
nxt = {u: 0.0 for u in nodes}
|
|
66
|
+
base = (1.0 - damping)
|
|
67
|
+
for u in nodes:
|
|
68
|
+
nxt[u] += base * teleport[u]
|
|
69
|
+
for u in nodes: # sorted order → deterministic float summation
|
|
70
|
+
share = damping * rank[u] / out_deg[u]
|
|
71
|
+
if share == 0.0:
|
|
72
|
+
continue
|
|
73
|
+
for v in adj[u]:
|
|
74
|
+
nxt[v] += share
|
|
75
|
+
rank = nxt
|
|
76
|
+
return rank
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def ppr_boost(facts: list, seed_ids: list[str], edges: list[dict] | None = None,
|
|
80
|
+
damping: float = 0.85, iters: int = 12) -> dict[str, float]:
|
|
81
|
+
"""PPR over the local fact graph. Returns {fact_id: normalized mass}.
|
|
82
|
+
|
|
83
|
+
``edges`` are trace edges among the given facts (dicts with src/dst);
|
|
84
|
+
they connect fact nodes directly (contradiction chains etc.).
|
|
85
|
+
"""
|
|
86
|
+
if not facts:
|
|
87
|
+
return {}
|
|
88
|
+
adj, fact_ids = build_fact_graph(facts)
|
|
89
|
+
if edges:
|
|
90
|
+
for e in edges:
|
|
91
|
+
s, d = e.get("src"), e.get("dst")
|
|
92
|
+
if s in fact_ids and d in fact_ids:
|
|
93
|
+
adj.setdefault(s, []).append(d)
|
|
94
|
+
adj.setdefault(d, []).append(s)
|
|
95
|
+
seeds = [fid for fid in seed_ids if fid in adj]
|
|
96
|
+
if not seeds:
|
|
97
|
+
return {}
|
|
98
|
+
rank = personalized_pagerank(adj, seeds, damping, iters)
|
|
99
|
+
raw = {fid: rank.get(fid, 0.0) for fid in fact_ids}
|
|
100
|
+
mx = max(raw.values(), default=0.0)
|
|
101
|
+
if mx <= 0:
|
|
102
|
+
return {}
|
|
103
|
+
# normalize to [0, 1]; seeds stay near 1.0, two-hop evidence decays
|
|
104
|
+
return {fid: v / mx for fid, v in raw.items() if v > 0}
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
"""Query-aware triple pre-filter (HippoRAG 2 lineage).
|
|
2
|
+
|
|
3
|
+
arXiv:2505.14832 (HippoRAG 2) reports a 7% F1 gain from query-aware
|
|
4
|
+
passage/triple filtering BEFORE the symbolic-vs-neural fusion step.
|
|
5
|
+
The insight: when you bind candidate triples into a holographic
|
|
6
|
+
superposition for VSA retrieval, every IRRELEVANT triple injects noise
|
|
7
|
+
into the superposition. Pre-filtering with a cheap lexical+semantic
|
|
8
|
+
scorer drops the noise and lifts precision at the fusion output.
|
|
9
|
+
|
|
10
|
+
Context-M's bridge/reader.py currently:
|
|
11
|
+
1. VSA probe → top-K candidate facts by cosine sim
|
|
12
|
+
2. symbolic dereference → full fact rows + edges
|
|
13
|
+
3. PPR over the local fact graph → multi-hop boost
|
|
14
|
+
4. fusion (VSA + PPR + symbolic) → final rank
|
|
15
|
+
|
|
16
|
+
This module inserts between steps 1 and 3:
|
|
17
|
+
1.5 QUERY-AWARE TRIPLE PRE-FILTER
|
|
18
|
+
For each candidate fact, compute:
|
|
19
|
+
* lexical_score — Jaccard of content words in (query, fact text)
|
|
20
|
+
* semantic_score — cosine(query_emb, fact_text_emb)
|
|
21
|
+
* relation_match — +0.2 if the query's RELATION_HINTS include
|
|
22
|
+
the fact's relation (e.g. "where does X live"
|
|
23
|
+
→ relation_hint=lives_in → match)
|
|
24
|
+
Combined weighted score; drop facts below threshold.
|
|
25
|
+
|
|
26
|
+
This is μ=0: deterministic scorer, no LLM. The cost is O(K) cosine sims
|
|
27
|
+
on the top-K candidates (K=20-50 typically), ~50μs total per query.
|
|
28
|
+
|
|
29
|
+
The 7% F1 gain HippoRAG 2 reports is for the LLM-triple-filter variant;
|
|
30
|
+
our deterministic variant should capture most of the lift on natural-
|
|
31
|
+
language queries where lexical+relation overlap is a strong signal.
|
|
32
|
+
"""
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
from dataclasses import dataclass
|
|
36
|
+
|
|
37
|
+
import numpy as np
|
|
38
|
+
|
|
39
|
+
from cortexm.text.tokenizer import STOPWORDS, words
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class PrefilterStats:
|
|
44
|
+
n_in: int
|
|
45
|
+
n_kept: int
|
|
46
|
+
n_dropped: int
|
|
47
|
+
min_score: float
|
|
48
|
+
max_score: float
|
|
49
|
+
mean_score: float
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _jaccard(a: set[str], b: set[str]) -> float:
|
|
53
|
+
if not a and not b:
|
|
54
|
+
return 0.0
|
|
55
|
+
inter = len(a & b)
|
|
56
|
+
union = len(a | b)
|
|
57
|
+
return inter / union if union else 0.0
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _content_word_set(text: str) -> set[str]:
|
|
61
|
+
return {w.lower() for w in words(text) if w.lower() not in STOPWORDS}
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def prefilter_triples(
|
|
65
|
+
candidates: list,
|
|
66
|
+
query: str,
|
|
67
|
+
*,
|
|
68
|
+
query_emb: np.ndarray | None = None,
|
|
69
|
+
fact_text_fn=None,
|
|
70
|
+
relation_hints: list[str] | None = None,
|
|
71
|
+
embedder=None,
|
|
72
|
+
threshold: float = 0.08,
|
|
73
|
+
weights: tuple[float, float, float] = (0.45, 0.45, 0.10),
|
|
74
|
+
min_keep: int = 3,
|
|
75
|
+
) -> tuple[list, PrefilterStats]:
|
|
76
|
+
"""Filter candidate facts by query relevance BEFORE fusion.
|
|
77
|
+
|
|
78
|
+
Parameters
|
|
79
|
+
----------
|
|
80
|
+
candidates : list of Fact-like objects (must have .subject, .relation,
|
|
81
|
+
.value, optionally .id and .memory/.note)
|
|
82
|
+
query : the user's raw query string
|
|
83
|
+
query_emb : pre-computed query embedding (np.ndarray). If None and an
|
|
84
|
+
embedder is provided, we compute it here.
|
|
85
|
+
fact_text_fn : callable(fact) -> str, the natural-language rendering
|
|
86
|
+
of the fact to score against. Defaults to
|
|
87
|
+
f"{subject} {relation} {value}".
|
|
88
|
+
relation_hints : list of relation names the query seems to be asking
|
|
89
|
+
about (from RELATION_HINTS in reader.py). Each
|
|
90
|
+
candidate whose .relation is in this list gets a
|
|
91
|
+
+0.2 boost.
|
|
92
|
+
embedder : optional object with .embed(text) -> np.ndarray, used to
|
|
93
|
+
compute fact_text embeddings for semantic scoring.
|
|
94
|
+
threshold : combined score below this → drop. Conservative default
|
|
95
|
+
(0.08) keeps most candidates; tune up for higher precision.
|
|
96
|
+
weights : (lexical, semantic, relation_match) weights summing to ~1.
|
|
97
|
+
min_keep : always keep at least this many candidates (top-K by score),
|
|
98
|
+
even if all are below threshold — guarantees fusion has
|
|
99
|
+
something to rank.
|
|
100
|
+
|
|
101
|
+
Returns (filtered_list, stats).
|
|
102
|
+
"""
|
|
103
|
+
if not candidates:
|
|
104
|
+
return [], PrefilterStats(0, 0, 0, 0.0, 0.0, 0.0)
|
|
105
|
+
|
|
106
|
+
q_words = _content_word_set(query)
|
|
107
|
+
if query_emb is None and embedder is not None:
|
|
108
|
+
try:
|
|
109
|
+
query_emb = embedder.embed(query)
|
|
110
|
+
except Exception:
|
|
111
|
+
query_emb = None
|
|
112
|
+
|
|
113
|
+
rel_set = set(relation_hints) if relation_hints else set()
|
|
114
|
+
w_lex, w_sem, w_rel = weights
|
|
115
|
+
out: list[tuple[float, int]] = [] # (score, original_idx)
|
|
116
|
+
n = len(candidates)
|
|
117
|
+
scores = [0.0] * n
|
|
118
|
+
|
|
119
|
+
# pre-compute fact embeddings if we have an embedder (batched)
|
|
120
|
+
fact_embs: list[np.ndarray | None] = [None] * n
|
|
121
|
+
if embedder is not None and w_sem > 0:
|
|
122
|
+
for i, c in enumerate(candidates):
|
|
123
|
+
try:
|
|
124
|
+
txt = (fact_text_fn(c) if fact_text_fn
|
|
125
|
+
else _default_fact_text(c))
|
|
126
|
+
fact_embs[i] = embedder.embed(txt)
|
|
127
|
+
except Exception:
|
|
128
|
+
fact_embs[i] = None
|
|
129
|
+
|
|
130
|
+
for i, c in enumerate(candidates):
|
|
131
|
+
txt = (fact_text_fn(c) if fact_text_fn else _default_fact_text(c))
|
|
132
|
+
c_words = _content_word_set(txt)
|
|
133
|
+
lex = _jaccard(q_words, c_words)
|
|
134
|
+
sem = 0.0
|
|
135
|
+
if query_emb is not None and fact_embs[i] is not None:
|
|
136
|
+
try:
|
|
137
|
+
sem = float(np.dot(query_emb, fact_embs[i]))
|
|
138
|
+
# cosine sim in [-1, 1] → normalize to [0, 1]
|
|
139
|
+
sem = max(0.0, (sem + 1.0) / 2.0)
|
|
140
|
+
except Exception:
|
|
141
|
+
sem = 0.0
|
|
142
|
+
rel = 1.0 if (rel_set and getattr(c, "relation", "") in rel_set) else 0.0
|
|
143
|
+
score = w_lex * lex + w_sem * sem + w_rel * rel
|
|
144
|
+
scores[i] = score
|
|
145
|
+
out.append((score, i))
|
|
146
|
+
|
|
147
|
+
# always keep at least min_keep — sort desc, take top min_keep
|
|
148
|
+
out.sort(key=lambda x: -x[0])
|
|
149
|
+
kept_idx = [i for s, i in out if s >= threshold]
|
|
150
|
+
if len(kept_idx) < min_keep:
|
|
151
|
+
for s, i in out:
|
|
152
|
+
if i not in kept_idx:
|
|
153
|
+
kept_idx.append(i)
|
|
154
|
+
if len(kept_idx) >= min_keep:
|
|
155
|
+
break
|
|
156
|
+
kept_set = set(kept_idx)
|
|
157
|
+
filtered = [candidates[i] for i in range(n) if i in kept_set]
|
|
158
|
+
kept_scores = [scores[i] for i in range(n) if i in kept_set]
|
|
159
|
+
stats = PrefilterStats(
|
|
160
|
+
n_in=n,
|
|
161
|
+
n_kept=len(filtered),
|
|
162
|
+
n_dropped=n - len(filtered),
|
|
163
|
+
min_score=min(kept_scores) if kept_scores else 0.0,
|
|
164
|
+
max_score=max(kept_scores) if kept_scores else 0.0,
|
|
165
|
+
mean_score=(sum(kept_scores) / len(kept_scores)) if kept_scores else 0.0,
|
|
166
|
+
)
|
|
167
|
+
return filtered, stats
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _default_fact_text(f) -> str:
|
|
171
|
+
"""Natural-language rendering of a fact for scoring.
|
|
172
|
+
|
|
173
|
+
The pattern library produces facts with .subject, .relation, .value.
|
|
174
|
+
Reader.RetrievalResult wraps them with .memory (a NL string). We
|
|
175
|
+
prefer .memory if available, else fall back to the triple.
|
|
176
|
+
"""
|
|
177
|
+
mem = getattr(f, "memory", None)
|
|
178
|
+
if mem:
|
|
179
|
+
return mem
|
|
180
|
+
subj = getattr(f, "subject", "") or ""
|
|
181
|
+
rel = getattr(f, "relation", "") or ""
|
|
182
|
+
val = getattr(f, "value", "") or ""
|
|
183
|
+
# turn "lives_in" into "lives in" for slightly better lexical overlap
|
|
184
|
+
rel_nl = rel.replace("_", " ")
|
|
185
|
+
return f"{subj} {rel_nl} {val}".strip()
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
__all__ = ["prefilter_triples", "PrefilterStats"]
|