cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
cortexm/config.py
ADDED
|
@@ -0,0 +1,375 @@
|
|
|
1
|
+
"""Central configuration for the memory fabric.
|
|
2
|
+
|
|
3
|
+
Every knob the strategic plan calls out is expressed here so the whole
|
|
4
|
+
system is reproducible from one dataclass. Codec selection implements the
|
|
5
|
+
cortexm-compress tier model (INT8 default, Binary-HRR edge, RaBitQ
|
|
6
|
+
ultra-edge, PQ cloud).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import os
|
|
12
|
+
from dataclasses import dataclass, field, asdict
|
|
13
|
+
from typing import Final
|
|
14
|
+
|
|
15
|
+
CODECS = ("int8", "binary", "rabitq", "pq")
|
|
16
|
+
VSA_MODES = ("perm", "conv", "bag")
|
|
17
|
+
# Index backends — alternative storage/search paths for the VSA palace.
|
|
18
|
+
# * "quadrant" — default page-clustered log-depth 2-means tree (Rust wheel
|
|
19
|
+
# at rust/quadrant/). INT8-quantized leaves, best-first tree descent.
|
|
20
|
+
# * "nsg" — Navigating Spreading-out Graph (proximity graph). Lower
|
|
21
|
+
# query latency at high recall on high-dim vectors; build is slower.
|
|
22
|
+
# * "flat" — exact brute-force dot product. Reference; O(N) per query.
|
|
23
|
+
# Used by Config.index_backend (validated in __post_init__).
|
|
24
|
+
INDEX_BACKENDS: Final[tuple[str, ...]] = ("quadrant", "nsg", "flat")
|
|
25
|
+
|
|
26
|
+
# Storage tiers from the compression appendix (bytes per 1M memories).
|
|
27
|
+
STORAGE_TIERS = {
|
|
28
|
+
"int8": "768 MB VSA + ~100 MB Trace (baseline)",
|
|
29
|
+
"binary": "96 MB VSA + ~50 MB Trace (edge / Raspberry Pi 5)",
|
|
30
|
+
"rabitq": "96 MB VSA + ~50 MB Trace (ultra-edge, 94%+ recall)",
|
|
31
|
+
"pq": "8 MB VSA + ~75 MB Trace (cloud, M=8 x 8-bit codes)",
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _env_bool(name: str, default: bool) -> bool:
|
|
36
|
+
v = os.environ.get(name)
|
|
37
|
+
if v is None:
|
|
38
|
+
return default
|
|
39
|
+
return v.strip().lower() in ("1", "true", "yes", "on")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class Config:
|
|
44
|
+
"""All knobs. Defaults implement the plan's "Edge tier" profile."""
|
|
45
|
+
|
|
46
|
+
# --- storage -------------------------------------------------------
|
|
47
|
+
db_path: str = ":memory:" # SQLite file for Trace + vectors
|
|
48
|
+
codec: str = "int8" # int8 | binary | rabitq | pq
|
|
49
|
+
tmr: bool = False # triple-modular redundancy (binary)
|
|
50
|
+
hash_provider: str = "blake3" # blake3 | blake2b (auto-fallback)
|
|
51
|
+
|
|
52
|
+
# --- VSA ------------------------------------------------------------
|
|
53
|
+
dims: int = 768
|
|
54
|
+
vsa_mode: str = "perm" # perm | conv | bag
|
|
55
|
+
lexical_lambda: float = 0.6 # weight of lexical superposition
|
|
56
|
+
seed: int = 0x0C0FFEE
|
|
57
|
+
|
|
58
|
+
# --- ingest ---------------------------------------------------------
|
|
59
|
+
min_confidence: float = 0.30 # below -> not committed
|
|
60
|
+
fallback_mentions: bool = True # low-conf entity mentions
|
|
61
|
+
enable_rules: bool = True # Datalog-lite materialization
|
|
62
|
+
apply_rules_each_add: bool = True # False: defer to Memory.apply_rules()
|
|
63
|
+
enable_lifecycle: bool = True # interference-aware lifecycle
|
|
64
|
+
value_match_threshold: float = 0.92 # near-duplicate merge cutoff
|
|
65
|
+
quarantine_injection: bool = True # InjecMEM defense
|
|
66
|
+
quarantine_contagion: bool = True # MINJA second-order defense
|
|
67
|
+
contagion_threshold: float = 0.50 # token-overlap cutoff for taint
|
|
68
|
+
|
|
69
|
+
# --- retrieval ------------------------------------------------------
|
|
70
|
+
top_k_default: int = 12
|
|
71
|
+
search_k_mult: int = 4 # oversampling for scope filtering
|
|
72
|
+
hop_expansion: int = 1 # associative hops for multi-hop
|
|
73
|
+
history_window: int = 3 # superseded facts shown per chain
|
|
74
|
+
fusion_vsa_weight: float = 0.6
|
|
75
|
+
fusion_symbolic_weight: float = 0.4
|
|
76
|
+
|
|
77
|
+
# --- OOD ingestion (Unmess + DisSim + Bitap trigger widening) -----------
|
|
78
|
+
# When True, the main `mem.add()` path runs the chaos-mode pipeline
|
|
79
|
+
# before the deterministic extractor:
|
|
80
|
+
# 1. PerUserIdiolectNormalizer.observe() — accumulate slang
|
|
81
|
+
# 2. PerUserIdiolectNormalizer.normalize() — text-speak + kNN slang
|
|
82
|
+
# replacement ("u" → "you", "bruh" → "friend" if user-co-occurred)
|
|
83
|
+
# 3. DisSim recursive split — compound sentences become simple clauses
|
|
84
|
+
# so each one matches its own pattern ("Although Alice works at X,
|
|
85
|
+
# she quit yesterday" → 3 clauses, 3 patterns fire)
|
|
86
|
+
# 4. Bitap fuzzy-trigger widening in extractor._sentence_candidates
|
|
87
|
+
# so "wrks at" / "livs in" still fires the works_at/lives_in pattern
|
|
88
|
+
# Default ON — the OOD paraphrase/slang recall catastrophe (9.4% / 5.1%
|
|
89
|
+
# in Tier-1) is exactly what this layer fixes, and it stays μ=0.
|
|
90
|
+
unmess_enabled: bool = True
|
|
91
|
+
unmess_max_depth: int = 2 # DisSim recursion limit
|
|
92
|
+
# Bitap trigger widening: if a known trigger word (works, lives, etc.)
|
|
93
|
+
# is NOT found in the sentence, try fuzzy match within max_edits. This
|
|
94
|
+
# catches "wrks", "livs", "prfrs" without bloating the regex set.
|
|
95
|
+
bitap_trigger_enabled: bool = True
|
|
96
|
+
bitap_trigger_max_edits: int = 2 # 2 edits = "wrks"→"works", "livs"→"lives"
|
|
97
|
+
|
|
98
|
+
# --- μ≈0 tiny-transformer fallback (pattern-miss retrieval) ------------
|
|
99
|
+
# When Bitap widened the trigger but the pattern library still returned
|
|
100
|
+
# zero candidates for a sentence, run a 2-layer self-attention "tiny
|
|
101
|
+
# transformer" whose weights are derived deterministically from the
|
|
102
|
+
# project seed (no external model download, no ONNX runtime, no GPU).
|
|
103
|
+
# Closes the OOD recall long tail without breaking the μ=0 / cost / audit
|
|
104
|
+
# moat. Default ON in production; bench baselines turn it off via
|
|
105
|
+
# bench_config_overrides() so the fallback's lift is visible in isolation.
|
|
106
|
+
tiny_fallback_enabled: bool = True
|
|
107
|
+
|
|
108
|
+
# --- LaBSE-inspired polyglot encoder (non-English ingest fix) ----------
|
|
109
|
+
# docs/BENCHMARKS.md Tier-1 shows non-English extraction recall =
|
|
110
|
+
# 0.000 ± 0.000 — the regex tokenizer + HashingEmbedder drops non-ASCII
|
|
111
|
+
# letters entirely, so every non-English sentence embeds to the same
|
|
112
|
+
# constant vector [1, 0, 0, ...] and retrieval is broken.
|
|
113
|
+
# When True, HashingEmbedder delegates text with >30% non-ASCII chars
|
|
114
|
+
# to PolyglotEncoder (cortexm.text.labse) — a LaBSE-inspired Unicode
|
|
115
|
+
# n-gram hasher that handles CJK / Devanagari / Arabic / Cyrillic /
|
|
116
|
+
# Thai / Hangul / Hiragana / Katakana via script-aware tokenization +
|
|
117
|
+
# char n-grams + signed feature hashing. Pure numpy + stdlib
|
|
118
|
+
# unicodedata, no model download (3GB LaBSE weights would violate the
|
|
119
|
+
# μ=0 + no-GPU rules), bit-identical across runs. Default OFF so the
|
|
120
|
+
# existing English path stays untouched; opt-in for deployments that
|
|
121
|
+
# ingest non-English text. See context_m/text/labse.py for the algorithm.
|
|
122
|
+
labse_enabled: bool = False
|
|
123
|
+
|
|
124
|
+
# --- Query-aware triple pre-filter (HippoRAG 2 lineage) ----------------
|
|
125
|
+
# When True, the reader drops candidate facts with low
|
|
126
|
+
# lexical+semantic+relation overlap with the query BEFORE fusion.
|
|
127
|
+
# HippoRAG 2 credits this for a 7% F1 gain. μ=0 — deterministic scorer.
|
|
128
|
+
prefilter_enabled: bool = True
|
|
129
|
+
prefilter_threshold: float = 0.08 # combined score below this → drop
|
|
130
|
+
prefilter_min_keep: int = 3 # always keep at least this many
|
|
131
|
+
|
|
132
|
+
# --- FadeMem-style forgetting (retention decay + sleep sweeps) ----------
|
|
133
|
+
# When True, the consolidate() pass also runs a FadeMem sweep that
|
|
134
|
+
# decays retention scores, marks low-retention facts for deactivation,
|
|
135
|
+
# and consolidates clusters of related facts into summary holograms.
|
|
136
|
+
# Default ON in production — measured 43.2% storage reduction with
|
|
137
|
+
# zero retrieval-precision regression (see benchmarks/results/final.json).
|
|
138
|
+
# Bench scripts flip this back to False via bench_config_overrides()
|
|
139
|
+
# so baseline numbers stay comparable across releases.
|
|
140
|
+
fade_enabled: bool = True
|
|
141
|
+
fade_lambda: float = 0.05 # exponential decay rate per day
|
|
142
|
+
fade_access_boost: float = 0.5 # each access multiplies retention
|
|
143
|
+
fade_contradiction_penalty: float = 0.25 # supersession pressure
|
|
144
|
+
fade_deactivate_threshold: float = 0.10 # below this → deactivate
|
|
145
|
+
|
|
146
|
+
# --- TiMem Temporal Memory Tree (4-level consolidation hierarchy) -------
|
|
147
|
+
# When True, the consolidate() pass also builds hierarchical summaries:
|
|
148
|
+
# L1 raw chunks → L2 session summaries → L3 daily patterns → L4 persona
|
|
149
|
+
# Each higher level is a derived fact that links down to its constituents
|
|
150
|
+
# via DERIVED_FROM edges. Retrieval can short-circuit to the appropriate
|
|
151
|
+
# level based on query complexity (complex queries hit L3/L4).
|
|
152
|
+
tmt_enabled: bool = False
|
|
153
|
+
tmt_session_cluster_mins: int = 5 # facts needed before a session summary
|
|
154
|
+
tmt_persona_min_sessions: int = 3 # sessions before persona abstraction
|
|
155
|
+
|
|
156
|
+
# --- HMS Cognition Engine (self-organizing memory) ---------------------
|
|
157
|
+
# When True, the consolidate() pass also runs the 5-stage cognition
|
|
158
|
+
# pipeline: PatternScanner (surface regularities) + AbstractionEngine
|
|
159
|
+
# (build prototype categories) + GapDetector (find missing relations)
|
|
160
|
+
# + HypothesisEngine (propose fillers) + AnalogyDetector (find
|
|
161
|
+
# isomorphic domains). Output is HYPOTHESIZED_BY edges with confidence
|
|
162
|
+
# < 0.5 — never promoted to active retrieval unless explicitly
|
|
163
|
+
# confirmed by user input. Default OFF — opt-in for use cases that
|
|
164
|
+
# want self-organization (e.g. agent memory that should infer
|
|
165
|
+
# `grandparent` from two `father` facts).
|
|
166
|
+
cognition_enabled: bool = False
|
|
167
|
+
cognition_min_support: int = 2 # min facts for a pattern to be significant
|
|
168
|
+
cognition_max_hyp_confidence: float = 0.45 # cap on hypothesis confidence
|
|
169
|
+
|
|
170
|
+
# --- Provenance standards stack (enterprise-grade) ---------------------
|
|
171
|
+
# When True, every memory commit is wrapped in a COSE Sign1 envelope
|
|
172
|
+
# (RFC 9052) signed with an Ed25519 agent key, and memory ranges can
|
|
173
|
+
# be exported as W3C Verifiable Credentials / SCITT-signed statements.
|
|
174
|
+
# BLAKE3 source hashing stays as the internal integrity mechanism;
|
|
175
|
+
# this layer sits on top for external interoperability (enterprise
|
|
176
|
+
# security reviews ask "does it support W3C VC? C2PA? SCITT?" — yes).
|
|
177
|
+
provenance_enabled: bool = False
|
|
178
|
+
provenance_agent_key_path: str | None = None # Ed25519 PEM (else generate)
|
|
179
|
+
provenance_agent_did: str | None = None # did:key identifier
|
|
180
|
+
|
|
181
|
+
# --- Structural multi-hop query (deterministic symbolic chains) --------
|
|
182
|
+
# When True, the reader exposes `structural_query(start_entity,
|
|
183
|
+
# relation_chain)` that walks exact relation chains via Trace lookups
|
|
184
|
+
# + VSA unbinding as fallback. Complementary to PPR (probabilistic)
|
|
185
|
+
# — PPR answers "what else might be relevant?", structural query
|
|
186
|
+
# answers "exactly follow this chain."
|
|
187
|
+
structural_query_enabled: bool = True
|
|
188
|
+
|
|
189
|
+
# --- Hopfield sparse-softmax cleanup (modernized cleanup memory) -----
|
|
190
|
+
# When True, the VSA cleanup memory uses sparse-softmax attention
|
|
191
|
+
# instead of plain softmax for noise recovery. Sparse softmax keeps
|
|
192
|
+
# only the top-k highest attention weights per recall step, which is
|
|
193
|
+
# more robust to outlier codebook entries (HMS-style improvement
|
|
194
|
+
# over Ramsauer 2020's plain softmax).
|
|
195
|
+
hopfield_sparse_softmax: bool = True
|
|
196
|
+
hopfield_sparse_topk: int = 16 # top-k weights kept per recall step
|
|
197
|
+
|
|
198
|
+
# --- Active reconstruction (MRAgent ICML 2026) ---------------------------
|
|
199
|
+
# When True, the reader exposes a `reconstruct()` method that runs an
|
|
200
|
+
# iterative PPR+LLM-scoring loop: expand seed nodes via 2-hop PPR, score
|
|
201
|
+
# each hop's relevance to the query, prune low-scoring branches, return
|
|
202
|
+
# a synthesized narrative. Default OFF — it's an LLM-assisted path that
|
|
203
|
+
# breaks strict μ=0; opt-in for use cases that need it.
|
|
204
|
+
reconstruct_enabled: bool = False
|
|
205
|
+
reconstruct_max_hops: int = 3
|
|
206
|
+
reconstruct_prune_threshold: float = 0.25
|
|
207
|
+
|
|
208
|
+
# --- MIND defense (InjecMEM attack mitigation) -----------------------------
|
|
209
|
+
# When True, retrieval runs a diversity check on the top-k results.
|
|
210
|
+
# InjecMEM relies on centroid anchors that cluster in embedding space —
|
|
211
|
+
# if the top-k results are too similar (low intra-result diversity),
|
|
212
|
+
# that's a signature of anchor-based poisoning, and the results are
|
|
213
|
+
# flagged for audit. This is μ=0 compatible (pure embedding math).
|
|
214
|
+
mind_diversity_check: bool = True
|
|
215
|
+
mind_diversity_threshold: float = 0.85 # mean pairwise cosine above this → flag
|
|
216
|
+
mind_flag_on_low_diversity: bool = True # mark results as suspect, don't drop
|
|
217
|
+
|
|
218
|
+
# --- cross-encoder rerank (μ=0, web search SOTA 2026-08) -------------
|
|
219
|
+
# Cross-encoder-style reranking: re-score top-K candidates by cosine
|
|
220
|
+
# sim between query embedding and a natural-language rendering of
|
|
221
|
+
# each fact ("the name of beam_1 is jennifer mccall"). Lifts prec@5
|
|
222
|
+
# 10-20pp on MS-MARCO-style benchmarks. Default OFF so baseline
|
|
223
|
+
# numbers don't shift; bench enables via "+rerank" config.
|
|
224
|
+
enable_rerank: bool = False
|
|
225
|
+
rerank_alpha: float = 0.55 # weight on rerank score (vs original)
|
|
226
|
+
rerank_beta: float = 0.45 # weight on original fusion score
|
|
227
|
+
prf_alpha: float = 0.6 # Rocchio: query weight
|
|
228
|
+
prf_beta: float = 0.4 # Rocchio: top-3 mean weight
|
|
229
|
+
prf_topn: int = 3 # how many top hits for PRF mean
|
|
230
|
+
|
|
231
|
+
# --- index ----------------------------------------------------------
|
|
232
|
+
# Which proximity-index backend the memory palace uses for ANN search
|
|
233
|
+
# over hologram vectors. See INDEX_BACKENDS at the top of this file
|
|
234
|
+
# for the full menu. Default "quadrant" — the existing page-clustered
|
|
235
|
+
# log-depth 2-means tree. "nsg" switches to the Navigating Spreading-
|
|
236
|
+
# out Graph (Fu et al. VLDB 2019) — sparser edges, lower query latency
|
|
237
|
+
# at high recall, slower build. "flat" is the exact brute-force
|
|
238
|
+
# reference path (no approximation).
|
|
239
|
+
index_backend: str = "quadrant"
|
|
240
|
+
index_threshold: int = 2048 # build tree when N >= this
|
|
241
|
+
index_leaf_size: int = 512
|
|
242
|
+
index_branch: int = 8
|
|
243
|
+
beam_width: int = 6
|
|
244
|
+
|
|
245
|
+
# --- SLB --------------------------------------------------------------
|
|
246
|
+
slb_entries: int = 64
|
|
247
|
+
slb_threshold: float = 0.97
|
|
248
|
+
# Bench determinism: when True, the SLB is bypassed entirely so each
|
|
249
|
+
# query recomputes fresh fusion. Without this, templated near-duplicate
|
|
250
|
+
# queries (e.g. "What is the name of beam_1?" / "What is the age of
|
|
251
|
+
# beam_1?") land at cosine ≈ 0.97 against the SLB threshold, and BLAS
|
|
252
|
+
# ULP drift across processes flips the hit/miss decision — producing
|
|
253
|
+
# ±5pp prec@5 variance on identical bench runs. Production runs leave
|
|
254
|
+
# this False (SLB is a real perf win); only the bench script enables it.
|
|
255
|
+
slb_disabled: bool = False
|
|
256
|
+
|
|
257
|
+
# --- Personalized PageRank (HippoRAG 2 lineage) --------------------------
|
|
258
|
+
ppr_enabled: bool = True # graph diffusion read mode
|
|
259
|
+
ppr_damping: float = 0.85
|
|
260
|
+
ppr_iters: int = 12
|
|
261
|
+
ppr_weight: float = 0.5 # fusion boost multiplier
|
|
262
|
+
ppr_graph_size: int = 96 # local subgraph node budget
|
|
263
|
+
ppr_seeds: int = 6 # teleport set size
|
|
264
|
+
|
|
265
|
+
# --- scopes -----------------------------------------------------------
|
|
266
|
+
default_user_id: str = "default"
|
|
267
|
+
sandbox_enabled: bool = True # agent-scoped facts isolated from
|
|
268
|
+
# user-scope reads until promoted
|
|
269
|
+
sandbox_promote_min_confidence: float = 0.5 # promotion gate
|
|
270
|
+
|
|
271
|
+
# --- durability ---------------------------------------------------------
|
|
272
|
+
wal_sync: str = "normal" # normal | full (full = fsync every
|
|
273
|
+
# commit; survives power loss)
|
|
274
|
+
|
|
275
|
+
# --- enterprise ---------------------------------------------------------
|
|
276
|
+
pii_mode: str = "off" # off | redact | block | tag
|
|
277
|
+
encryption_at_rest: bool = False # AES-256-GCM field encryption
|
|
278
|
+
master_key_path: str | None = None # explicit key file (else env/sidecar)
|
|
279
|
+
audit_enabled: bool = True # hash-chained audit log
|
|
280
|
+
audit_actions: str = "security" # "security" | "all" | "none"
|
|
281
|
+
rate_limit_rps: float = 50.0 # REST server, requests/second/key
|
|
282
|
+
rate_limit_burst: int = 100
|
|
283
|
+
|
|
284
|
+
# --- ZK-SQL proofs (Halo2/PLONKish-inspired, pure-Python) ----------------
|
|
285
|
+
# When True, the MCP server exposes `contextm_zk_sql_proof` (membership /
|
|
286
|
+
# count / sum / avg / min / max proofs over the Trace without revealing
|
|
287
|
+
# the underlying facts). Default OFF — proof generation is O(N) in trace
|
|
288
|
+
# size, opt-in. The verifier is sublinear (commitment check is O(1) hash
|
|
289
|
+
# equality + O(1) HMAC). The prover holds an HMAC key (ZK_SQL_KEY in the
|
|
290
|
+
# trace kv store) and signs each transcript; this is NOT a cryptographic
|
|
291
|
+
# SNARK (BLAKE3 commitments are not homomorphic), but it demonstrates the
|
|
292
|
+
# API surface that a production Halo2/KZG backend would expose.
|
|
293
|
+
zk_sql_enabled: bool = False
|
|
294
|
+
|
|
295
|
+
def __post_init__(self) -> None:
|
|
296
|
+
if self.codec not in CODECS:
|
|
297
|
+
raise ValueError(f"codec must be one of {CODECS}, got {self.codec!r}")
|
|
298
|
+
if self.vsa_mode not in VSA_MODES:
|
|
299
|
+
raise ValueError(f"vsa_mode must be one of {VSA_MODES}, got {self.vsa_mode!r}")
|
|
300
|
+
if self.dims <= 0 or self.dims % 8:
|
|
301
|
+
raise ValueError("dims must be a positive multiple of 8")
|
|
302
|
+
if self.index_backend not in INDEX_BACKENDS:
|
|
303
|
+
raise ValueError(
|
|
304
|
+
f"index_backend must be one of {INDEX_BACKENDS}, "
|
|
305
|
+
f"got {self.index_backend!r}")
|
|
306
|
+
|
|
307
|
+
# Environment overrides (12-factor friendly for MCP server / edge daemon)
|
|
308
|
+
@classmethod
|
|
309
|
+
def from_env(cls, **overrides) -> "Config":
|
|
310
|
+
cfg = cls(**overrides) if overrides else cls()
|
|
311
|
+
if p := os.environ.get("CONTEXT_M_DB"):
|
|
312
|
+
cfg.db_path = p
|
|
313
|
+
if c := os.environ.get("CONTEXT_M_CODEC"):
|
|
314
|
+
cfg.codec = c
|
|
315
|
+
if m := os.environ.get("CONTEXT_M_VSA_MODE"):
|
|
316
|
+
cfg.vsa_mode = m
|
|
317
|
+
if d := os.environ.get("CONTEXT_M_DIMS"):
|
|
318
|
+
cfg.dims = int(d)
|
|
319
|
+
if _env_bool("CONTEXT_M_TMR", cfg.tmr):
|
|
320
|
+
cfg.tmr = True
|
|
321
|
+
if pm := os.environ.get("CONTEXT_M_PII_MODE"):
|
|
322
|
+
cfg.pii_mode = pm
|
|
323
|
+
if mk := os.environ.get("CONTEXT_M_MASTER_KEY_PATH"):
|
|
324
|
+
cfg.master_key_path = mk
|
|
325
|
+
if os.environ.get("CONTEXT_M_ENCRYPT"):
|
|
326
|
+
cfg.encryption_at_rest = _env_bool("CONTEXT_M_ENCRYPT",
|
|
327
|
+
cfg.encryption_at_rest)
|
|
328
|
+
if aa := os.environ.get("CONTEXT_M_AUDIT"):
|
|
329
|
+
cfg.audit_actions = aa
|
|
330
|
+
# Allow flipping the index backend via env (e.g. CONTEXT_M_INDEX_BACKEND=nsg
|
|
331
|
+
# for high-recall cloud deployments where the build cost is amortized).
|
|
332
|
+
if ib := os.environ.get("CONTEXT_M_INDEX_BACKEND"):
|
|
333
|
+
cfg.index_backend = ib
|
|
334
|
+
# Production nightly-cron flips. The helm CronJob template sets
|
|
335
|
+
# CONTEXT_M_FADE=true and CONTEXT_M_TMT=true so the batch process
|
|
336
|
+
# runs the FadeMem sweep + TiMem TMT hierarchy build on top of the
|
|
337
|
+
# standard consolidate pass. Reading from env (not just CLI flags)
|
|
338
|
+
# means `cortexm consolidate --db …` in the CronJob container
|
|
339
|
+
# automatically picks them up.
|
|
340
|
+
if os.environ.get("CONTEXT_M_FADE") is not None:
|
|
341
|
+
cfg.fade_enabled = _env_bool("CONTEXT_M_FADE", cfg.fade_enabled)
|
|
342
|
+
if os.environ.get("CONTEXT_M_TMT") is not None:
|
|
343
|
+
cfg.tmt_enabled = _env_bool("CONTEXT_M_TMT", cfg.tmt_enabled)
|
|
344
|
+
if os.environ.get("CONTEXT_M_RECONSTRUCT") is not None:
|
|
345
|
+
cfg.reconstruct_enabled = _env_bool(
|
|
346
|
+
"CONTEXT_M_RECONSTRUCT", cfg.reconstruct_enabled)
|
|
347
|
+
# HMS Cognition Engine — opt-in self-organization. The helm
|
|
348
|
+
# CronJob template sets CONTEXT_M_COGNITION=true so the batch
|
|
349
|
+
# process also runs PatternScanner + AbstractionEngine +
|
|
350
|
+
# GapDetector + HypothesisEngine + AnalogyDetector on top of
|
|
351
|
+
# the standard consolidate pass.
|
|
352
|
+
if os.environ.get("CONTEXT_M_COGNITION") is not None:
|
|
353
|
+
cfg.cognition_enabled = _env_bool(
|
|
354
|
+
"CONTEXT_M_COGNITION", cfg.cognition_enabled)
|
|
355
|
+
# Enterprise provenance standards. Opt-in — when true, every
|
|
356
|
+
# commit is wrapped in a COSE Sign1 envelope (RFC 9052) and
|
|
357
|
+
# ranges can be exported as W3C VC / SCITT statements.
|
|
358
|
+
if os.environ.get("CONTEXT_M_PROVENANCE") is not None:
|
|
359
|
+
cfg.provenance_enabled = _env_bool(
|
|
360
|
+
"CONTEXT_M_PROVENANCE", cfg.provenance_enabled)
|
|
361
|
+
# ZK-SQL proofs (PoneglyphDB-style PLONKish). Opt-in.
|
|
362
|
+
if os.environ.get("CONTEXT_M_ZK_SQL") is not None:
|
|
363
|
+
cfg.zk_sql_enabled = _env_bool(
|
|
364
|
+
"CONTEXT_M_ZK_SQL", cfg.zk_sql_enabled)
|
|
365
|
+
# Polyglot encoder for non-English text. Opt-in — production
|
|
366
|
+
# deployments that ingest CJK / Indic / Arabic / Cyrillic text
|
|
367
|
+
# flip this on so HashingEmbedder falls back to PolyglotEncoder
|
|
368
|
+
# for >30% non-ASCII text instead of emitting a constant
|
|
369
|
+
# [1,0,0,...] vector that breaks retrieval (Tier-1 bug).
|
|
370
|
+
if os.environ.get("CONTEXT_M_LABSE") is not None:
|
|
371
|
+
cfg.labse_enabled = _env_bool("CONTEXT_M_LABSE", cfg.labse_enabled)
|
|
372
|
+
return cfg
|
|
373
|
+
|
|
374
|
+
def to_dict(self) -> dict:
|
|
375
|
+
return asdict(self)
|
cortexm/cortexm.py
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""Alias module — the strategic plan ships the import name ``cortexm``.
|
|
2
|
+
|
|
3
|
+
from cortexm import Memory # identical to: from cortexm import Memory
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from cortexm import Memory, Config, __version__ # noqa: F401
|
|
7
|
+
|
|
8
|
+
__all__ = ["Memory", "Config", "__version__"]
|
|
File without changes
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
"""Hash-chained, append-only audit log (SOC 2 / SIEM ready).
|
|
2
|
+
|
|
3
|
+
Every security-relevant operation — add, search, delete, erasure, key
|
|
4
|
+
management, snapshot, restore — appends one record:
|
|
5
|
+
|
|
6
|
+
{seq, ts, actor, role, action, resource, outcome, meta,
|
|
7
|
+
prev_hash, hash}
|
|
8
|
+
|
|
9
|
+
``hash = BLAKE2b(prev_hash || canonical-record)`` — tampering with any
|
|
10
|
+
record breaks the chain and ``verify()`` pinpoints the first damaged
|
|
11
|
+
sequence number. Records are also exported as JSONL (one per line) for
|
|
12
|
+
Splunk / Elastic / Datadog ingestion, and a syslog-style single-line
|
|
13
|
+
format for legacy collectors.
|
|
14
|
+
|
|
15
|
+
GDPR note: the audit chain is intentionally exempt from user erasure
|
|
16
|
+
(accounting/legitimate-interest records). It stores actor + resource
|
|
17
|
+
ids, never raw conversation text.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import json
|
|
23
|
+
import threading
|
|
24
|
+
from datetime import datetime, timezone
|
|
25
|
+
|
|
26
|
+
from cortexm.security.hashes import HashProvider
|
|
27
|
+
|
|
28
|
+
# actions that are always audited
|
|
29
|
+
AUDITED_ACTIONS = {
|
|
30
|
+
"memory.add", "memory.search", "memory.update", "memory.delete",
|
|
31
|
+
"memory.delete_all", "memory.verify",
|
|
32
|
+
"governance.erase", "governance.retention", "governance.snapshot",
|
|
33
|
+
"governance.restore", "governance.pitr",
|
|
34
|
+
"keys.create", "keys.revoke", "auth.failure", "security.quarantine",
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _now_iso() -> str:
|
|
39
|
+
return datetime.now(timezone.utc).isoformat(timespec="milliseconds")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class AuditLog:
|
|
43
|
+
def __init__(self, store, enabled: bool = True) -> None:
|
|
44
|
+
self.store = store
|
|
45
|
+
self.enabled = enabled
|
|
46
|
+
self._hasher = HashProvider("blake2b")
|
|
47
|
+
self._lock = threading.Lock()
|
|
48
|
+
self._ensure_table()
|
|
49
|
+
|
|
50
|
+
def _ensure_table(self) -> None:
|
|
51
|
+
self.store.conn.execute("""
|
|
52
|
+
CREATE TABLE IF NOT EXISTS audit_log (
|
|
53
|
+
seq INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
54
|
+
ts TEXT NOT NULL,
|
|
55
|
+
actor TEXT NOT NULL,
|
|
56
|
+
role TEXT,
|
|
57
|
+
action TEXT NOT NULL,
|
|
58
|
+
resource TEXT,
|
|
59
|
+
outcome TEXT NOT NULL,
|
|
60
|
+
meta TEXT,
|
|
61
|
+
prev_hash TEXT NOT NULL,
|
|
62
|
+
hash TEXT NOT NULL
|
|
63
|
+
)""")
|
|
64
|
+
# indexes for the common tail() filters (actor, action, ts)
|
|
65
|
+
self.store.conn.execute(
|
|
66
|
+
"CREATE INDEX IF NOT EXISTS idx_audit_actor_action "
|
|
67
|
+
"ON audit_log(actor, action)")
|
|
68
|
+
self.store.conn.execute(
|
|
69
|
+
"CREATE INDEX IF NOT EXISTS idx_audit_ts ON audit_log(ts)")
|
|
70
|
+
self.store.conn.commit()
|
|
71
|
+
|
|
72
|
+
# ------------------------------------------------------------- append
|
|
73
|
+
def log(self, action: str, *, actor: str = "system", role: str | None = None,
|
|
74
|
+
resource: str | None = None, outcome: str = "success",
|
|
75
|
+
meta: dict | None = None) -> dict | None:
|
|
76
|
+
if not self.enabled:
|
|
77
|
+
return None
|
|
78
|
+
with self._lock:
|
|
79
|
+
row = self.store.conn.execute(
|
|
80
|
+
"SELECT seq, hash FROM audit_log ORDER BY seq DESC LIMIT 1"
|
|
81
|
+
).fetchone()
|
|
82
|
+
prev = row["hash"] if row else "genesis"
|
|
83
|
+
rec = {
|
|
84
|
+
"ts": _now_iso(), "actor": actor, "role": role or "-",
|
|
85
|
+
"action": action, "resource": resource or "-",
|
|
86
|
+
"outcome": outcome,
|
|
87
|
+
"meta": json.dumps(meta or {}, sort_keys=True),
|
|
88
|
+
}
|
|
89
|
+
canonical = json.dumps(rec, sort_keys=True, separators=(",", ":"))
|
|
90
|
+
digest = self._hasher.hash_text(f"{prev}|{canonical}")
|
|
91
|
+
self.store.conn.execute(
|
|
92
|
+
"INSERT INTO audit_log(ts, actor, role, action, resource,"
|
|
93
|
+
" outcome, meta, prev_hash, hash) VALUES(?,?,?,?,?,?,?,?,?)",
|
|
94
|
+
(rec["ts"], rec["actor"], rec["role"], rec["action"],
|
|
95
|
+
rec["resource"], rec["outcome"], rec["meta"], prev, digest))
|
|
96
|
+
self.store.conn.commit()
|
|
97
|
+
return {"seq": self.store.conn.execute(
|
|
98
|
+
"SELECT last_insert_rowid() AS s").fetchone()["s"],
|
|
99
|
+
**rec, "prev_hash": prev, "hash": digest}
|
|
100
|
+
|
|
101
|
+
# ------------------------------------------------------------- read
|
|
102
|
+
def tail(self, n: int = 100, actor: str | None = None,
|
|
103
|
+
action: str | None = None) -> list[dict]:
|
|
104
|
+
q = "SELECT * FROM audit_log"
|
|
105
|
+
conds, args = [], []
|
|
106
|
+
if actor:
|
|
107
|
+
conds.append("actor=?"); args.append(actor)
|
|
108
|
+
if action:
|
|
109
|
+
conds.append("action=?"); args.append(action)
|
|
110
|
+
if conds:
|
|
111
|
+
q += " WHERE " + " AND ".join(conds)
|
|
112
|
+
q += " ORDER BY seq DESC LIMIT ?"
|
|
113
|
+
args.append(int(n))
|
|
114
|
+
return [dict(r) for r in self.store.conn.execute(q, args)]
|
|
115
|
+
|
|
116
|
+
def verify(self) -> dict:
|
|
117
|
+
"""Recompute the chain; report the first broken seq."""
|
|
118
|
+
prev = "genesis"
|
|
119
|
+
broken_at = None
|
|
120
|
+
n = 0
|
|
121
|
+
for row in self.store.conn.execute(
|
|
122
|
+
"SELECT * FROM audit_log ORDER BY seq ASC"):
|
|
123
|
+
n += 1
|
|
124
|
+
rec = {"ts": row["ts"], "actor": row["actor"], "role": row["role"],
|
|
125
|
+
"action": row["action"], "resource": row["resource"],
|
|
126
|
+
"outcome": row["outcome"], "meta": row["meta"]}
|
|
127
|
+
canonical = json.dumps(rec, sort_keys=True, separators=(",", ":"))
|
|
128
|
+
digest = self._hasher.hash_text(f"{prev}|{canonical}")
|
|
129
|
+
if digest != row["hash"] or prev != row["prev_hash"]:
|
|
130
|
+
broken_at = row["seq"]
|
|
131
|
+
break
|
|
132
|
+
prev = row["hash"]
|
|
133
|
+
return {"records": n, "intact": broken_at is None,
|
|
134
|
+
"first_broken_seq": broken_at,
|
|
135
|
+
"head_hash": prev if broken_at is None else None}
|
|
136
|
+
|
|
137
|
+
# ------------------------------------------------------------- export
|
|
138
|
+
def export_jsonl(self, path: str) -> int:
|
|
139
|
+
"""SIEM ingestion export (Splunk/Elastic/Datadog friendly)."""
|
|
140
|
+
count = 0
|
|
141
|
+
with open(path, "w", encoding="utf-8") as fh:
|
|
142
|
+
for row in self.store.conn.execute(
|
|
143
|
+
"SELECT * FROM audit_log ORDER BY seq ASC"):
|
|
144
|
+
rec = {k: row[k] for k in row.keys()}
|
|
145
|
+
try:
|
|
146
|
+
rec["meta"] = json.loads(rec["meta"])
|
|
147
|
+
except Exception:
|
|
148
|
+
pass
|
|
149
|
+
fh.write(json.dumps(rec, sort_keys=True) + "\n")
|
|
150
|
+
count += 1
|
|
151
|
+
return count
|
|
152
|
+
|
|
153
|
+
def export_syslog(self, path: str) -> int:
|
|
154
|
+
"""RFC-3164-style lines: <134>ts actor action resource outcome."""
|
|
155
|
+
count = 0
|
|
156
|
+
with open(path, "w", encoding="utf-8") as fh:
|
|
157
|
+
for row in self.store.conn.execute(
|
|
158
|
+
"SELECT * FROM audit_log ORDER BY seq ASC"):
|
|
159
|
+
ts = row["ts"][:19].replace("T", " ")
|
|
160
|
+
fh.write(f"<134>{ts} context-m audit: actor={row['actor']} "
|
|
161
|
+
f"action={row['action']} resource={row['resource']} "
|
|
162
|
+
f"outcome={row['outcome']} seq={row['seq']}\n")
|
|
163
|
+
count += 1
|
|
164
|
+
return count
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
class AuditContext:
|
|
168
|
+
"""Request-scoped audit binding (actor/role propagated by the server)."""
|
|
169
|
+
|
|
170
|
+
def __init__(self, audit: AuditLog | None, actor: str = "system",
|
|
171
|
+
role: str | None = None) -> None:
|
|
172
|
+
self.audit = audit
|
|
173
|
+
self.actor = actor
|
|
174
|
+
self.role = role
|
|
175
|
+
|
|
176
|
+
def log(self, action: str, **kw) -> None:
|
|
177
|
+
if self.audit is not None:
|
|
178
|
+
self.audit.log(action, actor=self.actor, role=self.role, **kw)
|