cortexm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. context_m.py +17 -0
  2. cortexm/__init__.py +45 -0
  3. cortexm/accel.py +403 -0
  4. cortexm/api/__init__.py +0 -0
  5. cortexm/api/chaos.py +118 -0
  6. cortexm/api/memory.py +635 -0
  7. cortexm/bench/__init__.py +0 -0
  8. cortexm/bench/abilities.py +311 -0
  9. cortexm/bench/baselines.py +89 -0
  10. cortexm/bench/beam_loader.py +317 -0
  11. cortexm/bench/generator.py +376 -0
  12. cortexm/bench/harness.py +211 -0
  13. cortexm/bench/messy.py +218 -0
  14. cortexm/bench/micro.py +251 -0
  15. cortexm/bench/ood.py +443 -0
  16. cortexm/bench/run.py +137 -0
  17. cortexm/bridge/__init__.py +0 -0
  18. cortexm/bridge/dates.py +178 -0
  19. cortexm/bridge/decoders.py +204 -0
  20. cortexm/bridge/enrich.py +255 -0
  21. cortexm/bridge/extractor.py +316 -0
  22. cortexm/bridge/fallback.py +332 -0
  23. cortexm/bridge/onnx_runtime.py +158 -0
  24. cortexm/bridge/patterns.py +760 -0
  25. cortexm/bridge/ppr.py +104 -0
  26. cortexm/bridge/prefilter.py +188 -0
  27. cortexm/bridge/query_extract.py +420 -0
  28. cortexm/bridge/reader.py +1174 -0
  29. cortexm/bridge/rerank.py +204 -0
  30. cortexm/bridge/writer.py +492 -0
  31. cortexm/cli.py +295 -0
  32. cortexm/cognition/__init__.py +53 -0
  33. cortexm/cognition/abstraction.py +192 -0
  34. cortexm/cognition/analogy.py +159 -0
  35. cortexm/cognition/engine.py +204 -0
  36. cortexm/cognition/gaps.py +365 -0
  37. cortexm/cognition/scanner.py +204 -0
  38. cortexm/config.py +375 -0
  39. cortexm/cortexm.py +8 -0
  40. cortexm/enterprise/__init__.py +0 -0
  41. cortexm/enterprise/audit.py +178 -0
  42. cortexm/enterprise/governance.py +239 -0
  43. cortexm/errors.py +35 -0
  44. cortexm/features/__init__.py +0 -0
  45. cortexm/features/git.py +204 -0
  46. cortexm/features/prefetch.py +88 -0
  47. cortexm/features/zk.py +105 -0
  48. cortexm/federation/__init__.py +39 -0
  49. cortexm/federation/crdt.py +275 -0
  50. cortexm/federation/fabric.py +109 -0
  51. cortexm/federation/hlc.py +80 -0
  52. cortexm/federation/node.py +145 -0
  53. cortexm/federation/schema_report.py +73 -0
  54. cortexm/federation/transport.py +164 -0
  55. cortexm/index/__init__.py +19 -0
  56. cortexm/index/nsg.py +386 -0
  57. cortexm/mcp/__init__.py +0 -0
  58. cortexm/mcp/server.py +985 -0
  59. cortexm/metrics.py +62 -0
  60. cortexm/migrate/__init__.py +0 -0
  61. cortexm/migrate/importers.py +192 -0
  62. cortexm/provenance/__init__.py +78 -0
  63. cortexm/provenance/agent.py +214 -0
  64. cortexm/provenance/cose.py +201 -0
  65. cortexm/provenance/scitt.py +258 -0
  66. cortexm/provenance/vc.py +250 -0
  67. cortexm/security/__init__.py +0 -0
  68. cortexm/security/crypto.py +162 -0
  69. cortexm/security/hashes.py +140 -0
  70. cortexm/security/injection.py +149 -0
  71. cortexm/security/mind.py +154 -0
  72. cortexm/security/pii.py +265 -0
  73. cortexm/security/rbac.py +169 -0
  74. cortexm/security/sandbox.py +131 -0
  75. cortexm/security/zk_hamming.py +142 -0
  76. cortexm/security/zk_sql.py +485 -0
  77. cortexm/server/__init__.py +0 -0
  78. cortexm/server/metrics.py +88 -0
  79. cortexm/server/rest.py +936 -0
  80. cortexm/server/sparql.py +984 -0
  81. cortexm/text/__init__.py +0 -0
  82. cortexm/text/dissim.py +252 -0
  83. cortexm/text/embedder.py +155 -0
  84. cortexm/text/fuzzy.py +218 -0
  85. cortexm/text/idiolect.py +253 -0
  86. cortexm/text/labse.py +374 -0
  87. cortexm/text/tokenizer.py +79 -0
  88. cortexm/trace/__init__.py +0 -0
  89. cortexm/trace/blob_arena.py +277 -0
  90. cortexm/trace/consolidate.py +337 -0
  91. cortexm/trace/contradictions.py +69 -0
  92. cortexm/trace/dedup.py +114 -0
  93. cortexm/trace/edges.py +214 -0
  94. cortexm/trace/fact.py +121 -0
  95. cortexm/trace/fade.py +245 -0
  96. cortexm/trace/lifecycle.py +112 -0
  97. cortexm/trace/rebuild.py +173 -0
  98. cortexm/trace/rules.py +171 -0
  99. cortexm/trace/store.py +680 -0
  100. cortexm/trace/structural.py +183 -0
  101. cortexm/trace/tmt.py +335 -0
  102. cortexm/util.py +148 -0
  103. cortexm/vsa/__init__.py +0 -0
  104. cortexm/vsa/attribution.py +149 -0
  105. cortexm/vsa/cleanup.py +161 -0
  106. cortexm/vsa/codecs.py +397 -0
  107. cortexm/vsa/hologram_overlay.py +139 -0
  108. cortexm/vsa/index.py +163 -0
  109. cortexm/vsa/ops.py +149 -0
  110. cortexm/vsa/palace.py +446 -0
  111. cortexm/vsa/role_vectors.py +236 -0
  112. cortexm/vsa/slb.py +78 -0
  113. cortexm/vsa/tlsh_trie.py +137 -0
  114. cortexm/vsa/working_memory.py +249 -0
  115. cortexm-0.3.0.dist-info/METADATA +482 -0
  116. cortexm-0.3.0.dist-info/RECORD +120 -0
  117. cortexm-0.3.0.dist-info/WHEEL +5 -0
  118. cortexm-0.3.0.dist-info/entry_points.txt +2 -0
  119. cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
  120. cortexm-0.3.0.dist-info/top_level.txt +2 -0
cortexm/config.py ADDED
@@ -0,0 +1,375 @@
1
+ """Central configuration for the memory fabric.
2
+
3
+ Every knob the strategic plan calls out is expressed here so the whole
4
+ system is reproducible from one dataclass. Codec selection implements the
5
+ cortexm-compress tier model (INT8 default, Binary-HRR edge, RaBitQ
6
+ ultra-edge, PQ cloud).
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import os
12
+ from dataclasses import dataclass, field, asdict
13
+ from typing import Final
14
+
15
+ CODECS = ("int8", "binary", "rabitq", "pq")
16
+ VSA_MODES = ("perm", "conv", "bag")
17
+ # Index backends — alternative storage/search paths for the VSA palace.
18
+ # * "quadrant" — default page-clustered log-depth 2-means tree (Rust wheel
19
+ # at rust/quadrant/). INT8-quantized leaves, best-first tree descent.
20
+ # * "nsg" — Navigating Spreading-out Graph (proximity graph). Lower
21
+ # query latency at high recall on high-dim vectors; build is slower.
22
+ # * "flat" — exact brute-force dot product. Reference; O(N) per query.
23
+ # Used by Config.index_backend (validated in __post_init__).
24
+ INDEX_BACKENDS: Final[tuple[str, ...]] = ("quadrant", "nsg", "flat")
25
+
26
+ # Storage tiers from the compression appendix (bytes per 1M memories).
27
+ STORAGE_TIERS = {
28
+ "int8": "768 MB VSA + ~100 MB Trace (baseline)",
29
+ "binary": "96 MB VSA + ~50 MB Trace (edge / Raspberry Pi 5)",
30
+ "rabitq": "96 MB VSA + ~50 MB Trace (ultra-edge, 94%+ recall)",
31
+ "pq": "8 MB VSA + ~75 MB Trace (cloud, M=8 x 8-bit codes)",
32
+ }
33
+
34
+
35
+ def _env_bool(name: str, default: bool) -> bool:
36
+ v = os.environ.get(name)
37
+ if v is None:
38
+ return default
39
+ return v.strip().lower() in ("1", "true", "yes", "on")
40
+
41
+
42
+ @dataclass
43
+ class Config:
44
+ """All knobs. Defaults implement the plan's "Edge tier" profile."""
45
+
46
+ # --- storage -------------------------------------------------------
47
+ db_path: str = ":memory:" # SQLite file for Trace + vectors
48
+ codec: str = "int8" # int8 | binary | rabitq | pq
49
+ tmr: bool = False # triple-modular redundancy (binary)
50
+ hash_provider: str = "blake3" # blake3 | blake2b (auto-fallback)
51
+
52
+ # --- VSA ------------------------------------------------------------
53
+ dims: int = 768
54
+ vsa_mode: str = "perm" # perm | conv | bag
55
+ lexical_lambda: float = 0.6 # weight of lexical superposition
56
+ seed: int = 0x0C0FFEE
57
+
58
+ # --- ingest ---------------------------------------------------------
59
+ min_confidence: float = 0.30 # below -> not committed
60
+ fallback_mentions: bool = True # low-conf entity mentions
61
+ enable_rules: bool = True # Datalog-lite materialization
62
+ apply_rules_each_add: bool = True # False: defer to Memory.apply_rules()
63
+ enable_lifecycle: bool = True # interference-aware lifecycle
64
+ value_match_threshold: float = 0.92 # near-duplicate merge cutoff
65
+ quarantine_injection: bool = True # InjecMEM defense
66
+ quarantine_contagion: bool = True # MINJA second-order defense
67
+ contagion_threshold: float = 0.50 # token-overlap cutoff for taint
68
+
69
+ # --- retrieval ------------------------------------------------------
70
+ top_k_default: int = 12
71
+ search_k_mult: int = 4 # oversampling for scope filtering
72
+ hop_expansion: int = 1 # associative hops for multi-hop
73
+ history_window: int = 3 # superseded facts shown per chain
74
+ fusion_vsa_weight: float = 0.6
75
+ fusion_symbolic_weight: float = 0.4
76
+
77
+ # --- OOD ingestion (Unmess + DisSim + Bitap trigger widening) -----------
78
+ # When True, the main `mem.add()` path runs the chaos-mode pipeline
79
+ # before the deterministic extractor:
80
+ # 1. PerUserIdiolectNormalizer.observe() — accumulate slang
81
+ # 2. PerUserIdiolectNormalizer.normalize() — text-speak + kNN slang
82
+ # replacement ("u" → "you", "bruh" → "friend" if user-co-occurred)
83
+ # 3. DisSim recursive split — compound sentences become simple clauses
84
+ # so each one matches its own pattern ("Although Alice works at X,
85
+ # she quit yesterday" → 3 clauses, 3 patterns fire)
86
+ # 4. Bitap fuzzy-trigger widening in extractor._sentence_candidates
87
+ # so "wrks at" / "livs in" still fires the works_at/lives_in pattern
88
+ # Default ON — the OOD paraphrase/slang recall catastrophe (9.4% / 5.1%
89
+ # in Tier-1) is exactly what this layer fixes, and it stays μ=0.
90
+ unmess_enabled: bool = True
91
+ unmess_max_depth: int = 2 # DisSim recursion limit
92
+ # Bitap trigger widening: if a known trigger word (works, lives, etc.)
93
+ # is NOT found in the sentence, try fuzzy match within max_edits. This
94
+ # catches "wrks", "livs", "prfrs" without bloating the regex set.
95
+ bitap_trigger_enabled: bool = True
96
+ bitap_trigger_max_edits: int = 2 # 2 edits = "wrks"→"works", "livs"→"lives"
97
+
98
+ # --- μ≈0 tiny-transformer fallback (pattern-miss retrieval) ------------
99
+ # When Bitap widened the trigger but the pattern library still returned
100
+ # zero candidates for a sentence, run a 2-layer self-attention "tiny
101
+ # transformer" whose weights are derived deterministically from the
102
+ # project seed (no external model download, no ONNX runtime, no GPU).
103
+ # Closes the OOD recall long tail without breaking the μ=0 / cost / audit
104
+ # moat. Default ON in production; bench baselines turn it off via
105
+ # bench_config_overrides() so the fallback's lift is visible in isolation.
106
+ tiny_fallback_enabled: bool = True
107
+
108
+ # --- LaBSE-inspired polyglot encoder (non-English ingest fix) ----------
109
+ # docs/BENCHMARKS.md Tier-1 shows non-English extraction recall =
110
+ # 0.000 ± 0.000 — the regex tokenizer + HashingEmbedder drops non-ASCII
111
+ # letters entirely, so every non-English sentence embeds to the same
112
+ # constant vector [1, 0, 0, ...] and retrieval is broken.
113
+ # When True, HashingEmbedder delegates text with >30% non-ASCII chars
114
+ # to PolyglotEncoder (cortexm.text.labse) — a LaBSE-inspired Unicode
115
+ # n-gram hasher that handles CJK / Devanagari / Arabic / Cyrillic /
116
+ # Thai / Hangul / Hiragana / Katakana via script-aware tokenization +
117
+ # char n-grams + signed feature hashing. Pure numpy + stdlib
118
+ # unicodedata, no model download (3GB LaBSE weights would violate the
119
+ # μ=0 + no-GPU rules), bit-identical across runs. Default OFF so the
120
+ # existing English path stays untouched; opt-in for deployments that
121
+ # ingest non-English text. See context_m/text/labse.py for the algorithm.
122
+ labse_enabled: bool = False
123
+
124
+ # --- Query-aware triple pre-filter (HippoRAG 2 lineage) ----------------
125
+ # When True, the reader drops candidate facts with low
126
+ # lexical+semantic+relation overlap with the query BEFORE fusion.
127
+ # HippoRAG 2 credits this for a 7% F1 gain. μ=0 — deterministic scorer.
128
+ prefilter_enabled: bool = True
129
+ prefilter_threshold: float = 0.08 # combined score below this → drop
130
+ prefilter_min_keep: int = 3 # always keep at least this many
131
+
132
+ # --- FadeMem-style forgetting (retention decay + sleep sweeps) ----------
133
+ # When True, the consolidate() pass also runs a FadeMem sweep that
134
+ # decays retention scores, marks low-retention facts for deactivation,
135
+ # and consolidates clusters of related facts into summary holograms.
136
+ # Default ON in production — measured 43.2% storage reduction with
137
+ # zero retrieval-precision regression (see benchmarks/results/final.json).
138
+ # Bench scripts flip this back to False via bench_config_overrides()
139
+ # so baseline numbers stay comparable across releases.
140
+ fade_enabled: bool = True
141
+ fade_lambda: float = 0.05 # exponential decay rate per day
142
+ fade_access_boost: float = 0.5 # each access multiplies retention
143
+ fade_contradiction_penalty: float = 0.25 # supersession pressure
144
+ fade_deactivate_threshold: float = 0.10 # below this → deactivate
145
+
146
+ # --- TiMem Temporal Memory Tree (4-level consolidation hierarchy) -------
147
+ # When True, the consolidate() pass also builds hierarchical summaries:
148
+ # L1 raw chunks → L2 session summaries → L3 daily patterns → L4 persona
149
+ # Each higher level is a derived fact that links down to its constituents
150
+ # via DERIVED_FROM edges. Retrieval can short-circuit to the appropriate
151
+ # level based on query complexity (complex queries hit L3/L4).
152
+ tmt_enabled: bool = False
153
+ tmt_session_cluster_mins: int = 5 # facts needed before a session summary
154
+ tmt_persona_min_sessions: int = 3 # sessions before persona abstraction
155
+
156
+ # --- HMS Cognition Engine (self-organizing memory) ---------------------
157
+ # When True, the consolidate() pass also runs the 5-stage cognition
158
+ # pipeline: PatternScanner (surface regularities) + AbstractionEngine
159
+ # (build prototype categories) + GapDetector (find missing relations)
160
+ # + HypothesisEngine (propose fillers) + AnalogyDetector (find
161
+ # isomorphic domains). Output is HYPOTHESIZED_BY edges with confidence
162
+ # < 0.5 — never promoted to active retrieval unless explicitly
163
+ # confirmed by user input. Default OFF — opt-in for use cases that
164
+ # want self-organization (e.g. agent memory that should infer
165
+ # `grandparent` from two `father` facts).
166
+ cognition_enabled: bool = False
167
+ cognition_min_support: int = 2 # min facts for a pattern to be significant
168
+ cognition_max_hyp_confidence: float = 0.45 # cap on hypothesis confidence
169
+
170
+ # --- Provenance standards stack (enterprise-grade) ---------------------
171
+ # When True, every memory commit is wrapped in a COSE Sign1 envelope
172
+ # (RFC 9052) signed with an Ed25519 agent key, and memory ranges can
173
+ # be exported as W3C Verifiable Credentials / SCITT-signed statements.
174
+ # BLAKE3 source hashing stays as the internal integrity mechanism;
175
+ # this layer sits on top for external interoperability (enterprise
176
+ # security reviews ask "does it support W3C VC? C2PA? SCITT?" — yes).
177
+ provenance_enabled: bool = False
178
+ provenance_agent_key_path: str | None = None # Ed25519 PEM (else generate)
179
+ provenance_agent_did: str | None = None # did:key identifier
180
+
181
+ # --- Structural multi-hop query (deterministic symbolic chains) --------
182
+ # When True, the reader exposes `structural_query(start_entity,
183
+ # relation_chain)` that walks exact relation chains via Trace lookups
184
+ # + VSA unbinding as fallback. Complementary to PPR (probabilistic)
185
+ # — PPR answers "what else might be relevant?", structural query
186
+ # answers "exactly follow this chain."
187
+ structural_query_enabled: bool = True
188
+
189
+ # --- Hopfield sparse-softmax cleanup (modernized cleanup memory) -----
190
+ # When True, the VSA cleanup memory uses sparse-softmax attention
191
+ # instead of plain softmax for noise recovery. Sparse softmax keeps
192
+ # only the top-k highest attention weights per recall step, which is
193
+ # more robust to outlier codebook entries (HMS-style improvement
194
+ # over Ramsauer 2020's plain softmax).
195
+ hopfield_sparse_softmax: bool = True
196
+ hopfield_sparse_topk: int = 16 # top-k weights kept per recall step
197
+
198
+ # --- Active reconstruction (MRAgent ICML 2026) ---------------------------
199
+ # When True, the reader exposes a `reconstruct()` method that runs an
200
+ # iterative PPR+LLM-scoring loop: expand seed nodes via 2-hop PPR, score
201
+ # each hop's relevance to the query, prune low-scoring branches, return
202
+ # a synthesized narrative. Default OFF — it's an LLM-assisted path that
203
+ # breaks strict μ=0; opt-in for use cases that need it.
204
+ reconstruct_enabled: bool = False
205
+ reconstruct_max_hops: int = 3
206
+ reconstruct_prune_threshold: float = 0.25
207
+
208
+ # --- MIND defense (InjecMEM attack mitigation) -----------------------------
209
+ # When True, retrieval runs a diversity check on the top-k results.
210
+ # InjecMEM relies on centroid anchors that cluster in embedding space —
211
+ # if the top-k results are too similar (low intra-result diversity),
212
+ # that's a signature of anchor-based poisoning, and the results are
213
+ # flagged for audit. This is μ=0 compatible (pure embedding math).
214
+ mind_diversity_check: bool = True
215
+ mind_diversity_threshold: float = 0.85 # mean pairwise cosine above this → flag
216
+ mind_flag_on_low_diversity: bool = True # mark results as suspect, don't drop
217
+
218
+ # --- cross-encoder rerank (μ=0, web search SOTA 2026-08) -------------
219
+ # Cross-encoder-style reranking: re-score top-K candidates by cosine
220
+ # sim between query embedding and a natural-language rendering of
221
+ # each fact ("the name of beam_1 is jennifer mccall"). Lifts prec@5
222
+ # 10-20pp on MS-MARCO-style benchmarks. Default OFF so baseline
223
+ # numbers don't shift; bench enables via "+rerank" config.
224
+ enable_rerank: bool = False
225
+ rerank_alpha: float = 0.55 # weight on rerank score (vs original)
226
+ rerank_beta: float = 0.45 # weight on original fusion score
227
+ prf_alpha: float = 0.6 # Rocchio: query weight
228
+ prf_beta: float = 0.4 # Rocchio: top-3 mean weight
229
+ prf_topn: int = 3 # how many top hits for PRF mean
230
+
231
+ # --- index ----------------------------------------------------------
232
+ # Which proximity-index backend the memory palace uses for ANN search
233
+ # over hologram vectors. See INDEX_BACKENDS at the top of this file
234
+ # for the full menu. Default "quadrant" — the existing page-clustered
235
+ # log-depth 2-means tree. "nsg" switches to the Navigating Spreading-
236
+ # out Graph (Fu et al. VLDB 2019) — sparser edges, lower query latency
237
+ # at high recall, slower build. "flat" is the exact brute-force
238
+ # reference path (no approximation).
239
+ index_backend: str = "quadrant"
240
+ index_threshold: int = 2048 # build tree when N >= this
241
+ index_leaf_size: int = 512
242
+ index_branch: int = 8
243
+ beam_width: int = 6
244
+
245
+ # --- SLB --------------------------------------------------------------
246
+ slb_entries: int = 64
247
+ slb_threshold: float = 0.97
248
+ # Bench determinism: when True, the SLB is bypassed entirely so each
249
+ # query recomputes fresh fusion. Without this, templated near-duplicate
250
+ # queries (e.g. "What is the name of beam_1?" / "What is the age of
251
+ # beam_1?") land at cosine ≈ 0.97 against the SLB threshold, and BLAS
252
+ # ULP drift across processes flips the hit/miss decision — producing
253
+ # ±5pp prec@5 variance on identical bench runs. Production runs leave
254
+ # this False (SLB is a real perf win); only the bench script enables it.
255
+ slb_disabled: bool = False
256
+
257
+ # --- Personalized PageRank (HippoRAG 2 lineage) --------------------------
258
+ ppr_enabled: bool = True # graph diffusion read mode
259
+ ppr_damping: float = 0.85
260
+ ppr_iters: int = 12
261
+ ppr_weight: float = 0.5 # fusion boost multiplier
262
+ ppr_graph_size: int = 96 # local subgraph node budget
263
+ ppr_seeds: int = 6 # teleport set size
264
+
265
+ # --- scopes -----------------------------------------------------------
266
+ default_user_id: str = "default"
267
+ sandbox_enabled: bool = True # agent-scoped facts isolated from
268
+ # user-scope reads until promoted
269
+ sandbox_promote_min_confidence: float = 0.5 # promotion gate
270
+
271
+ # --- durability ---------------------------------------------------------
272
+ wal_sync: str = "normal" # normal | full (full = fsync every
273
+ # commit; survives power loss)
274
+
275
+ # --- enterprise ---------------------------------------------------------
276
+ pii_mode: str = "off" # off | redact | block | tag
277
+ encryption_at_rest: bool = False # AES-256-GCM field encryption
278
+ master_key_path: str | None = None # explicit key file (else env/sidecar)
279
+ audit_enabled: bool = True # hash-chained audit log
280
+ audit_actions: str = "security" # "security" | "all" | "none"
281
+ rate_limit_rps: float = 50.0 # REST server, requests/second/key
282
+ rate_limit_burst: int = 100
283
+
284
+ # --- ZK-SQL proofs (Halo2/PLONKish-inspired, pure-Python) ----------------
285
+ # When True, the MCP server exposes `contextm_zk_sql_proof` (membership /
286
+ # count / sum / avg / min / max proofs over the Trace without revealing
287
+ # the underlying facts). Default OFF — proof generation is O(N) in trace
288
+ # size, opt-in. The verifier is sublinear (commitment check is O(1) hash
289
+ # equality + O(1) HMAC). The prover holds an HMAC key (ZK_SQL_KEY in the
290
+ # trace kv store) and signs each transcript; this is NOT a cryptographic
291
+ # SNARK (BLAKE3 commitments are not homomorphic), but it demonstrates the
292
+ # API surface that a production Halo2/KZG backend would expose.
293
+ zk_sql_enabled: bool = False
294
+
295
+ def __post_init__(self) -> None:
296
+ if self.codec not in CODECS:
297
+ raise ValueError(f"codec must be one of {CODECS}, got {self.codec!r}")
298
+ if self.vsa_mode not in VSA_MODES:
299
+ raise ValueError(f"vsa_mode must be one of {VSA_MODES}, got {self.vsa_mode!r}")
300
+ if self.dims <= 0 or self.dims % 8:
301
+ raise ValueError("dims must be a positive multiple of 8")
302
+ if self.index_backend not in INDEX_BACKENDS:
303
+ raise ValueError(
304
+ f"index_backend must be one of {INDEX_BACKENDS}, "
305
+ f"got {self.index_backend!r}")
306
+
307
+ # Environment overrides (12-factor friendly for MCP server / edge daemon)
308
+ @classmethod
309
+ def from_env(cls, **overrides) -> "Config":
310
+ cfg = cls(**overrides) if overrides else cls()
311
+ if p := os.environ.get("CONTEXT_M_DB"):
312
+ cfg.db_path = p
313
+ if c := os.environ.get("CONTEXT_M_CODEC"):
314
+ cfg.codec = c
315
+ if m := os.environ.get("CONTEXT_M_VSA_MODE"):
316
+ cfg.vsa_mode = m
317
+ if d := os.environ.get("CONTEXT_M_DIMS"):
318
+ cfg.dims = int(d)
319
+ if _env_bool("CONTEXT_M_TMR", cfg.tmr):
320
+ cfg.tmr = True
321
+ if pm := os.environ.get("CONTEXT_M_PII_MODE"):
322
+ cfg.pii_mode = pm
323
+ if mk := os.environ.get("CONTEXT_M_MASTER_KEY_PATH"):
324
+ cfg.master_key_path = mk
325
+ if os.environ.get("CONTEXT_M_ENCRYPT"):
326
+ cfg.encryption_at_rest = _env_bool("CONTEXT_M_ENCRYPT",
327
+ cfg.encryption_at_rest)
328
+ if aa := os.environ.get("CONTEXT_M_AUDIT"):
329
+ cfg.audit_actions = aa
330
+ # Allow flipping the index backend via env (e.g. CONTEXT_M_INDEX_BACKEND=nsg
331
+ # for high-recall cloud deployments where the build cost is amortized).
332
+ if ib := os.environ.get("CONTEXT_M_INDEX_BACKEND"):
333
+ cfg.index_backend = ib
334
+ # Production nightly-cron flips. The helm CronJob template sets
335
+ # CONTEXT_M_FADE=true and CONTEXT_M_TMT=true so the batch process
336
+ # runs the FadeMem sweep + TiMem TMT hierarchy build on top of the
337
+ # standard consolidate pass. Reading from env (not just CLI flags)
338
+ # means `cortexm consolidate --db …` in the CronJob container
339
+ # automatically picks them up.
340
+ if os.environ.get("CONTEXT_M_FADE") is not None:
341
+ cfg.fade_enabled = _env_bool("CONTEXT_M_FADE", cfg.fade_enabled)
342
+ if os.environ.get("CONTEXT_M_TMT") is not None:
343
+ cfg.tmt_enabled = _env_bool("CONTEXT_M_TMT", cfg.tmt_enabled)
344
+ if os.environ.get("CONTEXT_M_RECONSTRUCT") is not None:
345
+ cfg.reconstruct_enabled = _env_bool(
346
+ "CONTEXT_M_RECONSTRUCT", cfg.reconstruct_enabled)
347
+ # HMS Cognition Engine — opt-in self-organization. The helm
348
+ # CronJob template sets CONTEXT_M_COGNITION=true so the batch
349
+ # process also runs PatternScanner + AbstractionEngine +
350
+ # GapDetector + HypothesisEngine + AnalogyDetector on top of
351
+ # the standard consolidate pass.
352
+ if os.environ.get("CONTEXT_M_COGNITION") is not None:
353
+ cfg.cognition_enabled = _env_bool(
354
+ "CONTEXT_M_COGNITION", cfg.cognition_enabled)
355
+ # Enterprise provenance standards. Opt-in — when true, every
356
+ # commit is wrapped in a COSE Sign1 envelope (RFC 9052) and
357
+ # ranges can be exported as W3C VC / SCITT statements.
358
+ if os.environ.get("CONTEXT_M_PROVENANCE") is not None:
359
+ cfg.provenance_enabled = _env_bool(
360
+ "CONTEXT_M_PROVENANCE", cfg.provenance_enabled)
361
+ # ZK-SQL proofs (PoneglyphDB-style PLONKish). Opt-in.
362
+ if os.environ.get("CONTEXT_M_ZK_SQL") is not None:
363
+ cfg.zk_sql_enabled = _env_bool(
364
+ "CONTEXT_M_ZK_SQL", cfg.zk_sql_enabled)
365
+ # Polyglot encoder for non-English text. Opt-in — production
366
+ # deployments that ingest CJK / Indic / Arabic / Cyrillic text
367
+ # flip this on so HashingEmbedder falls back to PolyglotEncoder
368
+ # for >30% non-ASCII text instead of emitting a constant
369
+ # [1,0,0,...] vector that breaks retrieval (Tier-1 bug).
370
+ if os.environ.get("CONTEXT_M_LABSE") is not None:
371
+ cfg.labse_enabled = _env_bool("CONTEXT_M_LABSE", cfg.labse_enabled)
372
+ return cfg
373
+
374
+ def to_dict(self) -> dict:
375
+ return asdict(self)
cortexm/cortexm.py ADDED
@@ -0,0 +1,8 @@
1
+ """Alias module — the strategic plan ships the import name ``cortexm``.
2
+
3
+ from cortexm import Memory # identical to: from cortexm import Memory
4
+ """
5
+
6
+ from cortexm import Memory, Config, __version__ # noqa: F401
7
+
8
+ __all__ = ["Memory", "Config", "__version__"]
File without changes
@@ -0,0 +1,178 @@
1
+ """Hash-chained, append-only audit log (SOC 2 / SIEM ready).
2
+
3
+ Every security-relevant operation — add, search, delete, erasure, key
4
+ management, snapshot, restore — appends one record:
5
+
6
+ {seq, ts, actor, role, action, resource, outcome, meta,
7
+ prev_hash, hash}
8
+
9
+ ``hash = BLAKE2b(prev_hash || canonical-record)`` — tampering with any
10
+ record breaks the chain and ``verify()`` pinpoints the first damaged
11
+ sequence number. Records are also exported as JSONL (one per line) for
12
+ Splunk / Elastic / Datadog ingestion, and a syslog-style single-line
13
+ format for legacy collectors.
14
+
15
+ GDPR note: the audit chain is intentionally exempt from user erasure
16
+ (accounting/legitimate-interest records). It stores actor + resource
17
+ ids, never raw conversation text.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import json
23
+ import threading
24
+ from datetime import datetime, timezone
25
+
26
+ from cortexm.security.hashes import HashProvider
27
+
28
+ # actions that are always audited
29
+ AUDITED_ACTIONS = {
30
+ "memory.add", "memory.search", "memory.update", "memory.delete",
31
+ "memory.delete_all", "memory.verify",
32
+ "governance.erase", "governance.retention", "governance.snapshot",
33
+ "governance.restore", "governance.pitr",
34
+ "keys.create", "keys.revoke", "auth.failure", "security.quarantine",
35
+ }
36
+
37
+
38
+ def _now_iso() -> str:
39
+ return datetime.now(timezone.utc).isoformat(timespec="milliseconds")
40
+
41
+
42
+ class AuditLog:
43
+ def __init__(self, store, enabled: bool = True) -> None:
44
+ self.store = store
45
+ self.enabled = enabled
46
+ self._hasher = HashProvider("blake2b")
47
+ self._lock = threading.Lock()
48
+ self._ensure_table()
49
+
50
+ def _ensure_table(self) -> None:
51
+ self.store.conn.execute("""
52
+ CREATE TABLE IF NOT EXISTS audit_log (
53
+ seq INTEGER PRIMARY KEY AUTOINCREMENT,
54
+ ts TEXT NOT NULL,
55
+ actor TEXT NOT NULL,
56
+ role TEXT,
57
+ action TEXT NOT NULL,
58
+ resource TEXT,
59
+ outcome TEXT NOT NULL,
60
+ meta TEXT,
61
+ prev_hash TEXT NOT NULL,
62
+ hash TEXT NOT NULL
63
+ )""")
64
+ # indexes for the common tail() filters (actor, action, ts)
65
+ self.store.conn.execute(
66
+ "CREATE INDEX IF NOT EXISTS idx_audit_actor_action "
67
+ "ON audit_log(actor, action)")
68
+ self.store.conn.execute(
69
+ "CREATE INDEX IF NOT EXISTS idx_audit_ts ON audit_log(ts)")
70
+ self.store.conn.commit()
71
+
72
+ # ------------------------------------------------------------- append
73
+ def log(self, action: str, *, actor: str = "system", role: str | None = None,
74
+ resource: str | None = None, outcome: str = "success",
75
+ meta: dict | None = None) -> dict | None:
76
+ if not self.enabled:
77
+ return None
78
+ with self._lock:
79
+ row = self.store.conn.execute(
80
+ "SELECT seq, hash FROM audit_log ORDER BY seq DESC LIMIT 1"
81
+ ).fetchone()
82
+ prev = row["hash"] if row else "genesis"
83
+ rec = {
84
+ "ts": _now_iso(), "actor": actor, "role": role or "-",
85
+ "action": action, "resource": resource or "-",
86
+ "outcome": outcome,
87
+ "meta": json.dumps(meta or {}, sort_keys=True),
88
+ }
89
+ canonical = json.dumps(rec, sort_keys=True, separators=(",", ":"))
90
+ digest = self._hasher.hash_text(f"{prev}|{canonical}")
91
+ self.store.conn.execute(
92
+ "INSERT INTO audit_log(ts, actor, role, action, resource,"
93
+ " outcome, meta, prev_hash, hash) VALUES(?,?,?,?,?,?,?,?,?)",
94
+ (rec["ts"], rec["actor"], rec["role"], rec["action"],
95
+ rec["resource"], rec["outcome"], rec["meta"], prev, digest))
96
+ self.store.conn.commit()
97
+ return {"seq": self.store.conn.execute(
98
+ "SELECT last_insert_rowid() AS s").fetchone()["s"],
99
+ **rec, "prev_hash": prev, "hash": digest}
100
+
101
+ # ------------------------------------------------------------- read
102
+ def tail(self, n: int = 100, actor: str | None = None,
103
+ action: str | None = None) -> list[dict]:
104
+ q = "SELECT * FROM audit_log"
105
+ conds, args = [], []
106
+ if actor:
107
+ conds.append("actor=?"); args.append(actor)
108
+ if action:
109
+ conds.append("action=?"); args.append(action)
110
+ if conds:
111
+ q += " WHERE " + " AND ".join(conds)
112
+ q += " ORDER BY seq DESC LIMIT ?"
113
+ args.append(int(n))
114
+ return [dict(r) for r in self.store.conn.execute(q, args)]
115
+
116
+ def verify(self) -> dict:
117
+ """Recompute the chain; report the first broken seq."""
118
+ prev = "genesis"
119
+ broken_at = None
120
+ n = 0
121
+ for row in self.store.conn.execute(
122
+ "SELECT * FROM audit_log ORDER BY seq ASC"):
123
+ n += 1
124
+ rec = {"ts": row["ts"], "actor": row["actor"], "role": row["role"],
125
+ "action": row["action"], "resource": row["resource"],
126
+ "outcome": row["outcome"], "meta": row["meta"]}
127
+ canonical = json.dumps(rec, sort_keys=True, separators=(",", ":"))
128
+ digest = self._hasher.hash_text(f"{prev}|{canonical}")
129
+ if digest != row["hash"] or prev != row["prev_hash"]:
130
+ broken_at = row["seq"]
131
+ break
132
+ prev = row["hash"]
133
+ return {"records": n, "intact": broken_at is None,
134
+ "first_broken_seq": broken_at,
135
+ "head_hash": prev if broken_at is None else None}
136
+
137
+ # ------------------------------------------------------------- export
138
+ def export_jsonl(self, path: str) -> int:
139
+ """SIEM ingestion export (Splunk/Elastic/Datadog friendly)."""
140
+ count = 0
141
+ with open(path, "w", encoding="utf-8") as fh:
142
+ for row in self.store.conn.execute(
143
+ "SELECT * FROM audit_log ORDER BY seq ASC"):
144
+ rec = {k: row[k] for k in row.keys()}
145
+ try:
146
+ rec["meta"] = json.loads(rec["meta"])
147
+ except Exception:
148
+ pass
149
+ fh.write(json.dumps(rec, sort_keys=True) + "\n")
150
+ count += 1
151
+ return count
152
+
153
+ def export_syslog(self, path: str) -> int:
154
+ """RFC-3164-style lines: <134>ts actor action resource outcome."""
155
+ count = 0
156
+ with open(path, "w", encoding="utf-8") as fh:
157
+ for row in self.store.conn.execute(
158
+ "SELECT * FROM audit_log ORDER BY seq ASC"):
159
+ ts = row["ts"][:19].replace("T", " ")
160
+ fh.write(f"<134>{ts} context-m audit: actor={row['actor']} "
161
+ f"action={row['action']} resource={row['resource']} "
162
+ f"outcome={row['outcome']} seq={row['seq']}\n")
163
+ count += 1
164
+ return count
165
+
166
+
167
+ class AuditContext:
168
+ """Request-scoped audit binding (actor/role propagated by the server)."""
169
+
170
+ def __init__(self, audit: AuditLog | None, actor: str = "system",
171
+ role: str | None = None) -> None:
172
+ self.audit = audit
173
+ self.actor = actor
174
+ self.role = role
175
+
176
+ def log(self, action: str, **kw) -> None:
177
+ if self.audit is not None:
178
+ self.audit.log(action, actor=self.actor, role=self.role, **kw)