cortexm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. context_m.py +17 -0
  2. cortexm/__init__.py +45 -0
  3. cortexm/accel.py +403 -0
  4. cortexm/api/__init__.py +0 -0
  5. cortexm/api/chaos.py +118 -0
  6. cortexm/api/memory.py +635 -0
  7. cortexm/bench/__init__.py +0 -0
  8. cortexm/bench/abilities.py +311 -0
  9. cortexm/bench/baselines.py +89 -0
  10. cortexm/bench/beam_loader.py +317 -0
  11. cortexm/bench/generator.py +376 -0
  12. cortexm/bench/harness.py +211 -0
  13. cortexm/bench/messy.py +218 -0
  14. cortexm/bench/micro.py +251 -0
  15. cortexm/bench/ood.py +443 -0
  16. cortexm/bench/run.py +137 -0
  17. cortexm/bridge/__init__.py +0 -0
  18. cortexm/bridge/dates.py +178 -0
  19. cortexm/bridge/decoders.py +204 -0
  20. cortexm/bridge/enrich.py +255 -0
  21. cortexm/bridge/extractor.py +316 -0
  22. cortexm/bridge/fallback.py +332 -0
  23. cortexm/bridge/onnx_runtime.py +158 -0
  24. cortexm/bridge/patterns.py +760 -0
  25. cortexm/bridge/ppr.py +104 -0
  26. cortexm/bridge/prefilter.py +188 -0
  27. cortexm/bridge/query_extract.py +420 -0
  28. cortexm/bridge/reader.py +1174 -0
  29. cortexm/bridge/rerank.py +204 -0
  30. cortexm/bridge/writer.py +492 -0
  31. cortexm/cli.py +295 -0
  32. cortexm/cognition/__init__.py +53 -0
  33. cortexm/cognition/abstraction.py +192 -0
  34. cortexm/cognition/analogy.py +159 -0
  35. cortexm/cognition/engine.py +204 -0
  36. cortexm/cognition/gaps.py +365 -0
  37. cortexm/cognition/scanner.py +204 -0
  38. cortexm/config.py +375 -0
  39. cortexm/cortexm.py +8 -0
  40. cortexm/enterprise/__init__.py +0 -0
  41. cortexm/enterprise/audit.py +178 -0
  42. cortexm/enterprise/governance.py +239 -0
  43. cortexm/errors.py +35 -0
  44. cortexm/features/__init__.py +0 -0
  45. cortexm/features/git.py +204 -0
  46. cortexm/features/prefetch.py +88 -0
  47. cortexm/features/zk.py +105 -0
  48. cortexm/federation/__init__.py +39 -0
  49. cortexm/federation/crdt.py +275 -0
  50. cortexm/federation/fabric.py +109 -0
  51. cortexm/federation/hlc.py +80 -0
  52. cortexm/federation/node.py +145 -0
  53. cortexm/federation/schema_report.py +73 -0
  54. cortexm/federation/transport.py +164 -0
  55. cortexm/index/__init__.py +19 -0
  56. cortexm/index/nsg.py +386 -0
  57. cortexm/mcp/__init__.py +0 -0
  58. cortexm/mcp/server.py +985 -0
  59. cortexm/metrics.py +62 -0
  60. cortexm/migrate/__init__.py +0 -0
  61. cortexm/migrate/importers.py +192 -0
  62. cortexm/provenance/__init__.py +78 -0
  63. cortexm/provenance/agent.py +214 -0
  64. cortexm/provenance/cose.py +201 -0
  65. cortexm/provenance/scitt.py +258 -0
  66. cortexm/provenance/vc.py +250 -0
  67. cortexm/security/__init__.py +0 -0
  68. cortexm/security/crypto.py +162 -0
  69. cortexm/security/hashes.py +140 -0
  70. cortexm/security/injection.py +149 -0
  71. cortexm/security/mind.py +154 -0
  72. cortexm/security/pii.py +265 -0
  73. cortexm/security/rbac.py +169 -0
  74. cortexm/security/sandbox.py +131 -0
  75. cortexm/security/zk_hamming.py +142 -0
  76. cortexm/security/zk_sql.py +485 -0
  77. cortexm/server/__init__.py +0 -0
  78. cortexm/server/metrics.py +88 -0
  79. cortexm/server/rest.py +936 -0
  80. cortexm/server/sparql.py +984 -0
  81. cortexm/text/__init__.py +0 -0
  82. cortexm/text/dissim.py +252 -0
  83. cortexm/text/embedder.py +155 -0
  84. cortexm/text/fuzzy.py +218 -0
  85. cortexm/text/idiolect.py +253 -0
  86. cortexm/text/labse.py +374 -0
  87. cortexm/text/tokenizer.py +79 -0
  88. cortexm/trace/__init__.py +0 -0
  89. cortexm/trace/blob_arena.py +277 -0
  90. cortexm/trace/consolidate.py +337 -0
  91. cortexm/trace/contradictions.py +69 -0
  92. cortexm/trace/dedup.py +114 -0
  93. cortexm/trace/edges.py +214 -0
  94. cortexm/trace/fact.py +121 -0
  95. cortexm/trace/fade.py +245 -0
  96. cortexm/trace/lifecycle.py +112 -0
  97. cortexm/trace/rebuild.py +173 -0
  98. cortexm/trace/rules.py +171 -0
  99. cortexm/trace/store.py +680 -0
  100. cortexm/trace/structural.py +183 -0
  101. cortexm/trace/tmt.py +335 -0
  102. cortexm/util.py +148 -0
  103. cortexm/vsa/__init__.py +0 -0
  104. cortexm/vsa/attribution.py +149 -0
  105. cortexm/vsa/cleanup.py +161 -0
  106. cortexm/vsa/codecs.py +397 -0
  107. cortexm/vsa/hologram_overlay.py +139 -0
  108. cortexm/vsa/index.py +163 -0
  109. cortexm/vsa/ops.py +149 -0
  110. cortexm/vsa/palace.py +446 -0
  111. cortexm/vsa/role_vectors.py +236 -0
  112. cortexm/vsa/slb.py +78 -0
  113. cortexm/vsa/tlsh_trie.py +137 -0
  114. cortexm/vsa/working_memory.py +249 -0
  115. cortexm-0.3.0.dist-info/METADATA +482 -0
  116. cortexm-0.3.0.dist-info/RECORD +120 -0
  117. cortexm-0.3.0.dist-info/WHEEL +5 -0
  118. cortexm-0.3.0.dist-info/entry_points.txt +2 -0
  119. cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
  120. cortexm-0.3.0.dist-info/top_level.txt +2 -0
cortexm/bridge/ppr.py ADDED
@@ -0,0 +1,104 @@
1
+ """Personalized PageRank read mode (HippoRAG 2 lineage).
2
+
3
+ The default reader treats retrieval as ranking; multi-hop questions
4
+ ("what language does the team of Alice's manager use?") need GRAPH
5
+ diffusion: evidence two hops away should inherit activation mass from
6
+ the query-matched seeds. HippoRAG (arXiv:2502.14802) showed PPR over
7
+ an entity-fact graph is the neuro-symbolic analogue of hippocampal
8
+ spreading activation.
9
+
10
+ Design notes:
11
+ * The graph is built LOCALLY from the candidate facts of one query
12
+ (bounded), not globally — μ=0 intact, no offline index needed.
13
+ * Deterministic: nodes are sorted, power iteration runs a fixed
14
+ number of steps; floats do not depend on dict order.
15
+ * Blended into fusion as an additive boost, gated by
16
+ ``config.ppr_enabled`` (default on — it only fires for
17
+ multihop/recall intents).
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+
23
+ def build_fact_graph(facts: list) -> tuple[dict[str, list[str]], set[str]]:
24
+ """Bipartite graph: entity nodes <-> fact nodes.
25
+
26
+ Edges: entity -- fact (subject), fact -- entity (value), plus
27
+ fact -- fact (CONTRADICTS / TEMPORALLY_PRECEDED_BY already live in
28
+ the trace; the caller passes edges separately).
29
+ Returns (adjacency, fact_node_ids).
30
+ """
31
+ adj: dict[str, list[str]] = {}
32
+ fact_ids: set[str] = set()
33
+
34
+ def add_edge(a: str, b: str) -> None:
35
+ adj.setdefault(a, []).append(b)
36
+ adj.setdefault(b, []).append(a)
37
+
38
+ for f in facts:
39
+ fact_ids.add(f.id)
40
+ if f.subject:
41
+ add_edge(f"e:{f.subject}", f.id)
42
+ if f.value:
43
+ add_edge(f.id, f"e:{f.value}")
44
+ return adj, fact_ids
45
+
46
+
47
+ def personalized_pagerank(adj: dict[str, list[str]], seeds: list[str],
48
+ damping: float = 0.85, iters: int = 12,
49
+ ) -> dict[str, float]:
50
+ """Power iteration with teleport ONLY to seed nodes.
51
+
52
+ Deterministic: iteration visits nodes in sorted order.
53
+ """
54
+ nodes = sorted(adj.keys())
55
+ if not nodes:
56
+ return {}
57
+ seed_set = {s for s in seeds if s in adj}
58
+ if not seed_set:
59
+ return {}
60
+ n = len(nodes)
61
+ rank = {u: (1.0 / len(seed_set)) if u in seed_set else 0.0 for u in nodes}
62
+ teleport = {u: (1.0 / len(seed_set)) if u in seed_set else 0.0 for u in nodes}
63
+ out_deg = {u: max(1, len(adj[u])) for u in nodes}
64
+ for _ in range(iters):
65
+ nxt = {u: 0.0 for u in nodes}
66
+ base = (1.0 - damping)
67
+ for u in nodes:
68
+ nxt[u] += base * teleport[u]
69
+ for u in nodes: # sorted order → deterministic float summation
70
+ share = damping * rank[u] / out_deg[u]
71
+ if share == 0.0:
72
+ continue
73
+ for v in adj[u]:
74
+ nxt[v] += share
75
+ rank = nxt
76
+ return rank
77
+
78
+
79
+ def ppr_boost(facts: list, seed_ids: list[str], edges: list[dict] | None = None,
80
+ damping: float = 0.85, iters: int = 12) -> dict[str, float]:
81
+ """PPR over the local fact graph. Returns {fact_id: normalized mass}.
82
+
83
+ ``edges`` are trace edges among the given facts (dicts with src/dst);
84
+ they connect fact nodes directly (contradiction chains etc.).
85
+ """
86
+ if not facts:
87
+ return {}
88
+ adj, fact_ids = build_fact_graph(facts)
89
+ if edges:
90
+ for e in edges:
91
+ s, d = e.get("src"), e.get("dst")
92
+ if s in fact_ids and d in fact_ids:
93
+ adj.setdefault(s, []).append(d)
94
+ adj.setdefault(d, []).append(s)
95
+ seeds = [fid for fid in seed_ids if fid in adj]
96
+ if not seeds:
97
+ return {}
98
+ rank = personalized_pagerank(adj, seeds, damping, iters)
99
+ raw = {fid: rank.get(fid, 0.0) for fid in fact_ids}
100
+ mx = max(raw.values(), default=0.0)
101
+ if mx <= 0:
102
+ return {}
103
+ # normalize to [0, 1]; seeds stay near 1.0, two-hop evidence decays
104
+ return {fid: v / mx for fid, v in raw.items() if v > 0}
@@ -0,0 +1,188 @@
1
+ """Query-aware triple pre-filter (HippoRAG 2 lineage).
2
+
3
+ arXiv:2505.14832 (HippoRAG 2) reports a 7% F1 gain from query-aware
4
+ passage/triple filtering BEFORE the symbolic-vs-neural fusion step.
5
+ The insight: when you bind candidate triples into a holographic
6
+ superposition for VSA retrieval, every IRRELEVANT triple injects noise
7
+ into the superposition. Pre-filtering with a cheap lexical+semantic
8
+ scorer drops the noise and lifts precision at the fusion output.
9
+
10
+ Context-M's bridge/reader.py currently:
11
+ 1. VSA probe → top-K candidate facts by cosine sim
12
+ 2. symbolic dereference → full fact rows + edges
13
+ 3. PPR over the local fact graph → multi-hop boost
14
+ 4. fusion (VSA + PPR + symbolic) → final rank
15
+
16
+ This module inserts between steps 1 and 3:
17
+ 1.5 QUERY-AWARE TRIPLE PRE-FILTER
18
+ For each candidate fact, compute:
19
+ * lexical_score — Jaccard of content words in (query, fact text)
20
+ * semantic_score — cosine(query_emb, fact_text_emb)
21
+ * relation_match — +0.2 if the query's RELATION_HINTS include
22
+ the fact's relation (e.g. "where does X live"
23
+ → relation_hint=lives_in → match)
24
+ Combined weighted score; drop facts below threshold.
25
+
26
+ This is μ=0: deterministic scorer, no LLM. The cost is O(K) cosine sims
27
+ on the top-K candidates (K=20-50 typically), ~50μs total per query.
28
+
29
+ The 7% F1 gain HippoRAG 2 reports is for the LLM-triple-filter variant;
30
+ our deterministic variant should capture most of the lift on natural-
31
+ language queries where lexical+relation overlap is a strong signal.
32
+ """
33
+ from __future__ import annotations
34
+
35
+ from dataclasses import dataclass
36
+
37
+ import numpy as np
38
+
39
+ from cortexm.text.tokenizer import STOPWORDS, words
40
+
41
+
42
+ @dataclass
43
+ class PrefilterStats:
44
+ n_in: int
45
+ n_kept: int
46
+ n_dropped: int
47
+ min_score: float
48
+ max_score: float
49
+ mean_score: float
50
+
51
+
52
+ def _jaccard(a: set[str], b: set[str]) -> float:
53
+ if not a and not b:
54
+ return 0.0
55
+ inter = len(a & b)
56
+ union = len(a | b)
57
+ return inter / union if union else 0.0
58
+
59
+
60
+ def _content_word_set(text: str) -> set[str]:
61
+ return {w.lower() for w in words(text) if w.lower() not in STOPWORDS}
62
+
63
+
64
+ def prefilter_triples(
65
+ candidates: list,
66
+ query: str,
67
+ *,
68
+ query_emb: np.ndarray | None = None,
69
+ fact_text_fn=None,
70
+ relation_hints: list[str] | None = None,
71
+ embedder=None,
72
+ threshold: float = 0.08,
73
+ weights: tuple[float, float, float] = (0.45, 0.45, 0.10),
74
+ min_keep: int = 3,
75
+ ) -> tuple[list, PrefilterStats]:
76
+ """Filter candidate facts by query relevance BEFORE fusion.
77
+
78
+ Parameters
79
+ ----------
80
+ candidates : list of Fact-like objects (must have .subject, .relation,
81
+ .value, optionally .id and .memory/.note)
82
+ query : the user's raw query string
83
+ query_emb : pre-computed query embedding (np.ndarray). If None and an
84
+ embedder is provided, we compute it here.
85
+ fact_text_fn : callable(fact) -> str, the natural-language rendering
86
+ of the fact to score against. Defaults to
87
+ f"{subject} {relation} {value}".
88
+ relation_hints : list of relation names the query seems to be asking
89
+ about (from RELATION_HINTS in reader.py). Each
90
+ candidate whose .relation is in this list gets a
91
+ +0.2 boost.
92
+ embedder : optional object with .embed(text) -> np.ndarray, used to
93
+ compute fact_text embeddings for semantic scoring.
94
+ threshold : combined score below this → drop. Conservative default
95
+ (0.08) keeps most candidates; tune up for higher precision.
96
+ weights : (lexical, semantic, relation_match) weights summing to ~1.
97
+ min_keep : always keep at least this many candidates (top-K by score),
98
+ even if all are below threshold — guarantees fusion has
99
+ something to rank.
100
+
101
+ Returns (filtered_list, stats).
102
+ """
103
+ if not candidates:
104
+ return [], PrefilterStats(0, 0, 0, 0.0, 0.0, 0.0)
105
+
106
+ q_words = _content_word_set(query)
107
+ if query_emb is None and embedder is not None:
108
+ try:
109
+ query_emb = embedder.embed(query)
110
+ except Exception:
111
+ query_emb = None
112
+
113
+ rel_set = set(relation_hints) if relation_hints else set()
114
+ w_lex, w_sem, w_rel = weights
115
+ out: list[tuple[float, int]] = [] # (score, original_idx)
116
+ n = len(candidates)
117
+ scores = [0.0] * n
118
+
119
+ # pre-compute fact embeddings if we have an embedder (batched)
120
+ fact_embs: list[np.ndarray | None] = [None] * n
121
+ if embedder is not None and w_sem > 0:
122
+ for i, c in enumerate(candidates):
123
+ try:
124
+ txt = (fact_text_fn(c) if fact_text_fn
125
+ else _default_fact_text(c))
126
+ fact_embs[i] = embedder.embed(txt)
127
+ except Exception:
128
+ fact_embs[i] = None
129
+
130
+ for i, c in enumerate(candidates):
131
+ txt = (fact_text_fn(c) if fact_text_fn else _default_fact_text(c))
132
+ c_words = _content_word_set(txt)
133
+ lex = _jaccard(q_words, c_words)
134
+ sem = 0.0
135
+ if query_emb is not None and fact_embs[i] is not None:
136
+ try:
137
+ sem = float(np.dot(query_emb, fact_embs[i]))
138
+ # cosine sim in [-1, 1] → normalize to [0, 1]
139
+ sem = max(0.0, (sem + 1.0) / 2.0)
140
+ except Exception:
141
+ sem = 0.0
142
+ rel = 1.0 if (rel_set and getattr(c, "relation", "") in rel_set) else 0.0
143
+ score = w_lex * lex + w_sem * sem + w_rel * rel
144
+ scores[i] = score
145
+ out.append((score, i))
146
+
147
+ # always keep at least min_keep — sort desc, take top min_keep
148
+ out.sort(key=lambda x: -x[0])
149
+ kept_idx = [i for s, i in out if s >= threshold]
150
+ if len(kept_idx) < min_keep:
151
+ for s, i in out:
152
+ if i not in kept_idx:
153
+ kept_idx.append(i)
154
+ if len(kept_idx) >= min_keep:
155
+ break
156
+ kept_set = set(kept_idx)
157
+ filtered = [candidates[i] for i in range(n) if i in kept_set]
158
+ kept_scores = [scores[i] for i in range(n) if i in kept_set]
159
+ stats = PrefilterStats(
160
+ n_in=n,
161
+ n_kept=len(filtered),
162
+ n_dropped=n - len(filtered),
163
+ min_score=min(kept_scores) if kept_scores else 0.0,
164
+ max_score=max(kept_scores) if kept_scores else 0.0,
165
+ mean_score=(sum(kept_scores) / len(kept_scores)) if kept_scores else 0.0,
166
+ )
167
+ return filtered, stats
168
+
169
+
170
+ def _default_fact_text(f) -> str:
171
+ """Natural-language rendering of a fact for scoring.
172
+
173
+ The pattern library produces facts with .subject, .relation, .value.
174
+ Reader.RetrievalResult wraps them with .memory (a NL string). We
175
+ prefer .memory if available, else fall back to the triple.
176
+ """
177
+ mem = getattr(f, "memory", None)
178
+ if mem:
179
+ return mem
180
+ subj = getattr(f, "subject", "") or ""
181
+ rel = getattr(f, "relation", "") or ""
182
+ val = getattr(f, "value", "") or ""
183
+ # turn "lives_in" into "lives in" for slightly better lexical overlap
184
+ rel_nl = rel.replace("_", " ")
185
+ return f"{subj} {rel_nl} {val}".strip()
186
+
187
+
188
+ __all__ = ["prefilter_triples", "PrefilterStats"]