cortexm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. context_m.py +17 -0
  2. cortexm/__init__.py +45 -0
  3. cortexm/accel.py +403 -0
  4. cortexm/api/__init__.py +0 -0
  5. cortexm/api/chaos.py +118 -0
  6. cortexm/api/memory.py +635 -0
  7. cortexm/bench/__init__.py +0 -0
  8. cortexm/bench/abilities.py +311 -0
  9. cortexm/bench/baselines.py +89 -0
  10. cortexm/bench/beam_loader.py +317 -0
  11. cortexm/bench/generator.py +376 -0
  12. cortexm/bench/harness.py +211 -0
  13. cortexm/bench/messy.py +218 -0
  14. cortexm/bench/micro.py +251 -0
  15. cortexm/bench/ood.py +443 -0
  16. cortexm/bench/run.py +137 -0
  17. cortexm/bridge/__init__.py +0 -0
  18. cortexm/bridge/dates.py +178 -0
  19. cortexm/bridge/decoders.py +204 -0
  20. cortexm/bridge/enrich.py +255 -0
  21. cortexm/bridge/extractor.py +316 -0
  22. cortexm/bridge/fallback.py +332 -0
  23. cortexm/bridge/onnx_runtime.py +158 -0
  24. cortexm/bridge/patterns.py +760 -0
  25. cortexm/bridge/ppr.py +104 -0
  26. cortexm/bridge/prefilter.py +188 -0
  27. cortexm/bridge/query_extract.py +420 -0
  28. cortexm/bridge/reader.py +1174 -0
  29. cortexm/bridge/rerank.py +204 -0
  30. cortexm/bridge/writer.py +492 -0
  31. cortexm/cli.py +295 -0
  32. cortexm/cognition/__init__.py +53 -0
  33. cortexm/cognition/abstraction.py +192 -0
  34. cortexm/cognition/analogy.py +159 -0
  35. cortexm/cognition/engine.py +204 -0
  36. cortexm/cognition/gaps.py +365 -0
  37. cortexm/cognition/scanner.py +204 -0
  38. cortexm/config.py +375 -0
  39. cortexm/cortexm.py +8 -0
  40. cortexm/enterprise/__init__.py +0 -0
  41. cortexm/enterprise/audit.py +178 -0
  42. cortexm/enterprise/governance.py +239 -0
  43. cortexm/errors.py +35 -0
  44. cortexm/features/__init__.py +0 -0
  45. cortexm/features/git.py +204 -0
  46. cortexm/features/prefetch.py +88 -0
  47. cortexm/features/zk.py +105 -0
  48. cortexm/federation/__init__.py +39 -0
  49. cortexm/federation/crdt.py +275 -0
  50. cortexm/federation/fabric.py +109 -0
  51. cortexm/federation/hlc.py +80 -0
  52. cortexm/federation/node.py +145 -0
  53. cortexm/federation/schema_report.py +73 -0
  54. cortexm/federation/transport.py +164 -0
  55. cortexm/index/__init__.py +19 -0
  56. cortexm/index/nsg.py +386 -0
  57. cortexm/mcp/__init__.py +0 -0
  58. cortexm/mcp/server.py +985 -0
  59. cortexm/metrics.py +62 -0
  60. cortexm/migrate/__init__.py +0 -0
  61. cortexm/migrate/importers.py +192 -0
  62. cortexm/provenance/__init__.py +78 -0
  63. cortexm/provenance/agent.py +214 -0
  64. cortexm/provenance/cose.py +201 -0
  65. cortexm/provenance/scitt.py +258 -0
  66. cortexm/provenance/vc.py +250 -0
  67. cortexm/security/__init__.py +0 -0
  68. cortexm/security/crypto.py +162 -0
  69. cortexm/security/hashes.py +140 -0
  70. cortexm/security/injection.py +149 -0
  71. cortexm/security/mind.py +154 -0
  72. cortexm/security/pii.py +265 -0
  73. cortexm/security/rbac.py +169 -0
  74. cortexm/security/sandbox.py +131 -0
  75. cortexm/security/zk_hamming.py +142 -0
  76. cortexm/security/zk_sql.py +485 -0
  77. cortexm/server/__init__.py +0 -0
  78. cortexm/server/metrics.py +88 -0
  79. cortexm/server/rest.py +936 -0
  80. cortexm/server/sparql.py +984 -0
  81. cortexm/text/__init__.py +0 -0
  82. cortexm/text/dissim.py +252 -0
  83. cortexm/text/embedder.py +155 -0
  84. cortexm/text/fuzzy.py +218 -0
  85. cortexm/text/idiolect.py +253 -0
  86. cortexm/text/labse.py +374 -0
  87. cortexm/text/tokenizer.py +79 -0
  88. cortexm/trace/__init__.py +0 -0
  89. cortexm/trace/blob_arena.py +277 -0
  90. cortexm/trace/consolidate.py +337 -0
  91. cortexm/trace/contradictions.py +69 -0
  92. cortexm/trace/dedup.py +114 -0
  93. cortexm/trace/edges.py +214 -0
  94. cortexm/trace/fact.py +121 -0
  95. cortexm/trace/fade.py +245 -0
  96. cortexm/trace/lifecycle.py +112 -0
  97. cortexm/trace/rebuild.py +173 -0
  98. cortexm/trace/rules.py +171 -0
  99. cortexm/trace/store.py +680 -0
  100. cortexm/trace/structural.py +183 -0
  101. cortexm/trace/tmt.py +335 -0
  102. cortexm/util.py +148 -0
  103. cortexm/vsa/__init__.py +0 -0
  104. cortexm/vsa/attribution.py +149 -0
  105. cortexm/vsa/cleanup.py +161 -0
  106. cortexm/vsa/codecs.py +397 -0
  107. cortexm/vsa/hologram_overlay.py +139 -0
  108. cortexm/vsa/index.py +163 -0
  109. cortexm/vsa/ops.py +149 -0
  110. cortexm/vsa/palace.py +446 -0
  111. cortexm/vsa/role_vectors.py +236 -0
  112. cortexm/vsa/slb.py +78 -0
  113. cortexm/vsa/tlsh_trie.py +137 -0
  114. cortexm/vsa/working_memory.py +249 -0
  115. cortexm-0.3.0.dist-info/METADATA +482 -0
  116. cortexm-0.3.0.dist-info/RECORD +120 -0
  117. cortexm-0.3.0.dist-info/WHEEL +5 -0
  118. cortexm-0.3.0.dist-info/entry_points.txt +2 -0
  119. cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
  120. cortexm-0.3.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,239 @@
1
+ """Data governance: GDPR erasure, retention, backup/restore, PITR.
2
+
3
+ These are the four operations an enterprise buyer's security review
4
+ actually blocks on:
5
+
6
+ erase_user() — Art. 17 right-to-erasure: hard-delete every trace
7
+ of a subject (facts, chunks, vectors, lexicon,
8
+ aliases, vault entries), crypto-shred the PII vault,
9
+ and leave the audit chain intact (legal exemption)
10
+ with an erasure ATTESTATION record.
11
+ apply_retention() — Art. 5(1)(e) storage-limitation: expire facts and
12
+ chunks older than the policy window, keeping the
13
+ bi-temporal tombstones so history remains honest.
14
+ snapshot()/restore() — atomic, integrity-checked backup envelope
15
+ (SQLite backup API + manifest + Merkle-style digest
16
+ of the artifact).
17
+ state_at() — point-in-time recovery read: the bi-temporal Trace
18
+ replays "what did the system believe at T?" using
19
+ transaction times (tx_from/tx_to) — no snapshot
20
+ needed, the database IS the WAL.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import hashlib
26
+ import json
27
+ import os
28
+ import time
29
+ from datetime import datetime, timezone
30
+
31
+ from cortexm.security.hashes import HashProvider
32
+
33
+
34
+ class Governance:
35
+ def __init__(self, memory) -> None:
36
+ self.memory = memory
37
+ self.store = memory.store
38
+ self.palace = memory.palace
39
+ self.audit = getattr(memory, "audit_log", None)
40
+
41
+ def _log(self, action: str, resource: str | None = None,
42
+ outcome: str = "success", meta: dict | None = None) -> None:
43
+ if self.audit is not None:
44
+ self.audit.log(action, resource=resource, outcome=outcome,
45
+ meta=meta)
46
+
47
+ # ------------------------------------------------------------ erasure
48
+ def erase_user(self, user_id: str, *, hard: bool = True,
49
+ crypto_shred: bool = True) -> dict:
50
+ """GDPR Art. 17 — remove the subject from every layer."""
51
+ t0 = time.time()
52
+ conn = self.store.conn
53
+ counts = {}
54
+ counts["facts"] = conn.execute(
55
+ "SELECT COUNT(*) AS c FROM facts WHERE user_id=?",
56
+ (user_id,)).fetchone()["c"]
57
+ counts["chunks"] = conn.execute(
58
+ "SELECT COUNT(*) AS c FROM chunks WHERE user_id=?",
59
+ (user_id,)).fetchone()["c"]
60
+ # orphaned edges touching removed facts
61
+ fact_ids = [r["id"] for r in conn.execute(
62
+ "SELECT id FROM facts WHERE user_id=?", (user_id,))]
63
+ if fact_ids:
64
+ qmarks = ",".join("?" * len(fact_ids))
65
+ counts["edges"] = conn.execute(
66
+ f"SELECT COUNT(*) AS c FROM edges WHERE src IN ({qmarks}) "
67
+ f"OR dst IN ({qmarks})", fact_ids + fact_ids).fetchone()["c"]
68
+ conn.execute(f"DELETE FROM edges WHERE src IN ({qmarks}) "
69
+ f"OR dst IN ({qmarks})", fact_ids + fact_ids)
70
+ conn.execute("DELETE FROM facts WHERE user_id=?", (user_id,))
71
+ conn.execute("DELETE FROM chunks WHERE user_id=?", (user_id,))
72
+ # kv residue: lexicon, names, alias caches
73
+ kv_removed = 0
74
+ for k, _v in list(self.store.iter_kv(f"lexicon:{user_id}")):
75
+ self.store.kv_delete(k); kv_removed += 1
76
+ for k, _v in list(self.store.iter_kv(f"name:{user_id}")):
77
+ self.store.kv_delete(k); kv_removed += 1
78
+ counts["kv"] = kv_removed
79
+ # PII vault: crypto-shred if enabled and configured
80
+ vault = getattr(self.memory, "pii_vault", None)
81
+ if vault is not None and crypto_shred:
82
+ counts["vault_shredded"] = vault.crypto_shred()
83
+ # palace vectors for those facts
84
+ vec_removed = 0
85
+ if fact_ids:
86
+ vec_removed = self.palace.remove_ids(fact_ids)
87
+ counts["vectors"] = vec_removed
88
+ conn.commit()
89
+ # erasure attestation in the tamper-evident chain (Art. 30 records)
90
+ self._log("governance.erase", resource=user_id, outcome="erased",
91
+ meta={"counts": counts,
92
+ "duration_ms": round((time.time() - t0) * 1e3, 1),
93
+ "hard": hard, "crypto_shred": crypto_shred})
94
+ # residual scan: no row may reference the subject anywhere
95
+ residual = {
96
+ "facts": conn.execute(
97
+ "SELECT COUNT(*) AS c FROM facts WHERE user_id=?",
98
+ (user_id,)).fetchone()["c"],
99
+ "chunks": conn.execute(
100
+ "SELECT COUNT(*) AS c FROM chunks WHERE user_id=?",
101
+ (user_id,)).fetchone()["c"],
102
+ }
103
+ return {"user_id": user_id, "erased": residual["facts"] == 0
104
+ and residual["chunks"] == 0,
105
+ "counts": counts, "residual": residual,
106
+ "duration_ms": round((time.time() - t0) * 1e3, 1)}
107
+
108
+ # ------------------------------------------------------------ retention
109
+ def apply_retention(self, days: int, *, user_id: str | None = None,
110
+ dry_run: bool = False) -> dict:
111
+ """Expire facts whose last transaction time is older than ``days``."""
112
+ if days <= 0:
113
+ raise ValueError("days must be positive")
114
+ cutoff = datetime.now(timezone.utc).timestamp() - days * 86400
115
+ cutoff_iso = datetime.fromtimestamp(cutoff, tz=timezone.utc).isoformat()
116
+ conn = self.store.conn
117
+ q = ("SELECT COUNT(*) AS c FROM facts WHERE tx_from < ? "
118
+ "AND tx_to IS NULL")
119
+ args = [cutoff_iso]
120
+ if user_id:
121
+ q += " AND user_id=?"
122
+ args.append(user_id)
123
+ stale = conn.execute(q, args).fetchone()["c"]
124
+ if dry_run:
125
+ return {"stale_facts": stale, "applied": False}
126
+ q2 = "UPDATE facts SET tx_to=?, is_active=0 WHERE tx_from < ? AND tx_to IS NULL"
127
+ args2 = [datetime.now(timezone.utc).isoformat(), cutoff_iso]
128
+ if user_id:
129
+ q2 += " AND user_id=?"
130
+ args2.append(user_id)
131
+ cur = conn.execute(q2, args2)
132
+ conn.commit()
133
+ self._log("governance.retention",
134
+ meta={"days": days, "expired": cur.rowcount,
135
+ "user_id": user_id or "all"})
136
+ return {"stale_facts": cur.rowcount, "applied": True,
137
+ "cutoff": cutoff_iso}
138
+
139
+ # ------------------------------------------------------------ snapshot
140
+ def snapshot(self, path: str) -> dict:
141
+ """Atomic backup: online-backup the SQLite file + write a manifest.
142
+ The backup API copies page-by-page under a read transaction —
143
+ safe while writers are active."""
144
+ os.makedirs(os.path.dirname(os.path.abspath(path)) or ".", exist_ok=True)
145
+ if path.endswith("/"):
146
+ raise ValueError("path must be a file, not a directory")
147
+ target = sqlite_backup(self.store.db_path, path)
148
+ digest = _file_digest(target)
149
+ manifest = {
150
+ "format": "context-m-snapshot/1",
151
+ "created_at": datetime.now(timezone.utc).isoformat(),
152
+ "source_db": self.store.db_path,
153
+ "file": os.path.basename(target),
154
+ "size_bytes": os.path.getsize(target),
155
+ "sha256": digest,
156
+ "facts": self.store.conn.execute(
157
+ "SELECT COUNT(*) AS c FROM facts").fetchone()["c"],
158
+ "audit_head": (self.audit.verify()["head_hash"]
159
+ if self.audit else None),
160
+ }
161
+ mpath = target + ".manifest.json"
162
+ with open(mpath, "w", encoding="utf-8") as fh:
163
+ json.dump(manifest, fh, indent=2)
164
+ self._log("governance.snapshot",
165
+ resource=path, meta={"sha256": digest,
166
+ "size": manifest["size_bytes"]})
167
+ return {"path": target, "manifest": manifest, "manifest_path": mpath}
168
+
169
+ def restore(self, snapshot_path: str, *, verify: bool = True) -> dict:
170
+ """Restore from a snapshot manifest pair. Refuses mismatched digests."""
171
+ mpath = snapshot_path + ".manifest.json"
172
+ if verify and not os.path.exists(mpath):
173
+ raise FileNotFoundError("manifest missing — cannot verify integrity")
174
+ if os.path.exists(mpath):
175
+ with open(mpath, encoding="utf-8") as fh:
176
+ manifest = json.load(fh)
177
+ if verify:
178
+ digest = _file_digest(snapshot_path)
179
+ if digest != manifest["sha256"]:
180
+ raise ValueError(
181
+ f"integrity check failed: {digest} != {manifest['sha256']}")
182
+ # close current handles, replace file, reopen
183
+ db_path = self.store.db_path
184
+ self.memory.close()
185
+ import shutil
186
+ shutil.copyfile(snapshot_path, db_path)
187
+ self.memory._reopen()
188
+ # _reopen() built a NEW Governance object; this method is running
189
+ # on the old one — rebind self so the audit attestation below
190
+ # writes through the fresh store, not the closed one.
191
+ self.store = self.memory.store
192
+ self.palace = self.memory.palace
193
+ self.audit = self.memory.audit_log
194
+ self._log("governance.restore", resource=snapshot_path)
195
+ return {"restored": db_path, "verified": verify}
196
+
197
+ # ------------------------------------------------------------ PITR
198
+ def state_at(self, when, *, user_id: str | None = None,
199
+ limit: int = 500) -> list[dict]:
200
+ """Point-in-time read: facts the system believed true at ``when``
201
+ (transaction-time replay — the database is its own WAL)."""
202
+ from cortexm.api.memory import parse_ts
203
+ ts = parse_ts(when) if not isinstance(when, datetime) else when
204
+ if ts.tzinfo is None:
205
+ ts = ts.replace(tzinfo=timezone.utc)
206
+ # normalize to the Z-suffix format facts are stored with, so the
207
+ # string comparison in SQL matches the stored tx_from exactly
208
+ iso = ts.strftime("%Y-%m-%dT%H:%M:%SZ")
209
+ q = ("SELECT * FROM facts WHERE tx_from <= ? AND "
210
+ "(tx_to IS NULL OR tx_to > ?) AND quarantined=0")
211
+ args: list = [iso, iso]
212
+ if user_id:
213
+ q += " AND user_id=?"
214
+ args.append(user_id)
215
+ q += " ORDER BY valid_from DESC LIMIT ?"
216
+ args.append(limit)
217
+ rows = [dict(r) for r in self.store.conn.execute(q, args)]
218
+ self._log("governance.pitr", meta={"when": iso, "rows": len(rows)})
219
+ return rows
220
+
221
+
222
+ # ------------------------------------------------------------------ helpers
223
+ def sqlite_backup(src_db: str, dst_path: str) -> str:
224
+ import sqlite3
225
+ src = sqlite3.connect(src_db)
226
+ dst = sqlite3.connect(dst_path)
227
+ with dst:
228
+ src.backup(dst)
229
+ dst.close()
230
+ src.close()
231
+ return dst_path
232
+
233
+
234
+ def _file_digest(path: str) -> str:
235
+ h = hashlib.sha256()
236
+ with open(path, "rb") as fh:
237
+ for chunk in iter(lambda: fh.read(1 << 20), b""):
238
+ h.update(chunk)
239
+ return h.hexdigest()
cortexm/errors.py ADDED
@@ -0,0 +1,35 @@
1
+ """Exception hierarchy for Context-M."""
2
+
3
+ from __future__ import annotations
4
+
5
+
6
+ class ContextMError(Exception):
7
+ """Base class."""
8
+
9
+
10
+ class StoreError(ContextMError):
11
+ """Trace / SQLite layer failure."""
12
+
13
+
14
+ class ExtractionError(ContextMError):
15
+ """Deterministic extraction pipeline failure."""
16
+
17
+
18
+ class SecurityError(ContextMError):
19
+ """Provenance / hash verification failure."""
20
+
21
+
22
+ class VerificationError(SecurityError):
23
+ """Hash or Merkle proof did not verify."""
24
+
25
+
26
+ class BranchError(ContextMError):
27
+ """Memory-Git operation failure (unknown branch, dirty state...)."""
28
+
29
+
30
+ class MigrationError(ContextMError):
31
+ """Import from a foreign memory system failed."""
32
+
33
+
34
+ class CodecError(ContextMError):
35
+ """Vector codec failure (untrained PQ, corrupt record...)."""
File without changes
@@ -0,0 +1,204 @@
1
+ """Memory Git — version control for agent memory.
2
+
3
+ Every add() is a hash-chained commit on a Merkle-style DAG. Branches
4
+ fork memory state (A/B-test an agent personality), merges replay
5
+ theirs-onto-ours with deterministic conflict resolution via the
6
+ lifecycle engine, diffs show exactly which facts changed, and blame
7
+ traces which commit introduced a fact. Enterprises get regulatory
8
+ rollback and forensic audit as first-class operations.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import json
14
+
15
+ from cortexm.errors import BranchError
16
+ from cortexm.trace.fact import Fact, SINGLE_VALUED
17
+ from cortexm.trace.store import TraceStore
18
+ from cortexm.util import new_id, similarity
19
+
20
+
21
+ class MemoryGit:
22
+ def __init__(self, store: TraceStore, palace=None) -> None:
23
+ self.store = store
24
+ self.palace = palace
25
+
26
+ # ------------------------------------------------------------- basics
27
+ def branch(self, name: str, from_commit: str | None = None,
28
+ switch: bool = True) -> str:
29
+ base = self.store.create_branch(name, from_commit, switch=switch)
30
+ return base
31
+
32
+ def checkout(self, name: str) -> None:
33
+ self.store.checkout(name)
34
+
35
+ def branches(self) -> list[dict]:
36
+ return self.store.branches()
37
+
38
+ def log(self, branch: str | None = None, limit: int = 50) -> list[dict]:
39
+ return self.store.log(branch, limit)
40
+
41
+ # ------------------------------------------------------------- sets
42
+ def active_at(self, commit_id: str) -> dict[str, Fact]:
43
+ anc = self.store.ancestry(commit_id)
44
+ rows = self.store.conn.execute(
45
+ "SELECT id, birth_commit, retired_commit FROM facts "
46
+ "WHERE birth_commit IS NOT NULL").fetchall()
47
+ ids = [r["id"] for r in rows
48
+ if r["birth_commit"] in anc
49
+ and (not r["retired_commit"] or r["retired_commit"] not in anc)]
50
+ return {f.id: f for f in self.store.get_facts(ids)}
51
+
52
+ def _triple_key(self, f: Fact) -> tuple:
53
+ return (f.subject, f.relation, f.value.lower())
54
+
55
+ # ------------------------------------------------------------- diff
56
+ def diff(self, a: str, b: str) -> dict:
57
+ fa, fb = self.active_at(a), self.active_at(b)
58
+ ka = {self._triple_key(f): f for f in fa.values()}
59
+ kb = {self._triple_key(f): f for f in fb.values()}
60
+ added = [kb[k] for k in kb.keys() - ka.keys()]
61
+ removed = [ka[k] for k in ka.keys() - kb.keys()]
62
+ return {
63
+ "from": a, "to": b,
64
+ "added": [{"id": f.id, "fact": f.display(),
65
+ "valid_from": f.valid_from} for f in added],
66
+ "removed": [{"id": f.id, "fact": f.display(),
67
+ "valid_from": f.valid_from} for f in removed],
68
+ "n_added": len(added), "n_removed": len(removed),
69
+ }
70
+
71
+ # ------------------------------------------------------------- blame
72
+ def blame(self, subject: str, relation: str | None = None,
73
+ user_id: str | None = None) -> list[dict]:
74
+ clauses = ["subject=?"]
75
+ params: list = [subject]
76
+ if relation:
77
+ clauses.append("relation=?")
78
+ params.append(relation)
79
+ if user_id:
80
+ clauses.append("user_id=?")
81
+ params.append(user_id)
82
+ rows = self.store.conn.execute(
83
+ f"SELECT id, relation, value, valid_from, valid_to, is_active, "
84
+ f"birth_commit, tx_from FROM facts WHERE {' AND '.join(clauses)} "
85
+ f"ORDER BY tx_from", params).fetchall()
86
+ out = []
87
+ for r in rows:
88
+ commit = self.store.commit(r["birth_commit"]) if r["birth_commit"] else None
89
+ out.append({
90
+ "fact_id": r["id"],
91
+ "fact": f"({subject}, {r['relation']}, {r['value']})",
92
+ "valid": f"{r['valid_from']}→{r['valid_to'] or '∞'}",
93
+ "active": bool(r["is_active"]),
94
+ "commit": r["birth_commit"],
95
+ "commit_message": commit["message"] if commit else None,
96
+ "recorded_at": r["tx_from"],
97
+ })
98
+ return out
99
+
100
+ # ------------------------------------------------------------- merge
101
+ def merge(self, name: str, strategy: str = "latest-wins",
102
+ message: str = "") -> dict:
103
+ """3-way merge of branch ``name`` into the current branch."""
104
+ if strategy not in ("latest-wins", "union"):
105
+ raise BranchError(f"unknown strategy {strategy!r}")
106
+ cur = self.store.current_branch()
107
+ ours_head = self.store.head(cur)
108
+ theirs_head = self.store.head(name)
109
+ if not theirs_head:
110
+ raise BranchError(f"branch {name!r} has no commits")
111
+ if theirs_head == ours_head:
112
+ return {"status": "already-merged", "commit": ours_head,
113
+ "applied": 0, "conflicts": 0}
114
+
115
+ ours = self.active_at(ours_head)
116
+ theirs = self.active_at(theirs_head)
117
+ base_head = self._common_ancestor(ours_head, theirs_head)
118
+ base = self.active_at(base_head) if base_head else {}
119
+
120
+ kb = {self._triple_key(f): f for f in base.values()}
121
+ ko = {self._triple_key(f): f for f in ours.values()}
122
+ kt = {self._triple_key(f): f for f in theirs.values()}
123
+
124
+ added_by_theirs = [kt[k] for k in kt.keys() - kb.keys()]
125
+ retired_by_theirs = [kb[k] for k in kb.keys() - kt.keys()]
126
+
127
+ merge_commit = self.store.create_commit(
128
+ message or f"merge {name} into {cur} ({strategy})",
129
+ branch=cur, parents=[ours_head, theirs_head])
130
+ applied, conflicts = 0, 0
131
+
132
+ ours_by_sr: dict[tuple, list[Fact]] = {}
133
+ for f in ours.values():
134
+ ours_by_sr.setdefault((f.subject, f.relation), []).append(f)
135
+
136
+ for f in added_by_theirs:
137
+ key = self._triple_key(f)
138
+ if key in ko:
139
+ continue # already present on our side
140
+ conflict = any(g.value.lower() != f.value.lower()
141
+ for g in ours_by_sr.get((f.subject, f.relation), [])
142
+ if g.is_active)
143
+ if conflict and f.relation in SINGLE_VALUED and strategy == "latest-wins":
144
+ for g in ours_by_sr.get((f.subject, f.relation), []):
145
+ if not g.is_active:
146
+ continue
147
+ if (g.tx_from or "") <= (f.tx_from or ""):
148
+ self.store.update_fact(
149
+ g.id, is_active=0, retired_commit=merge_commit,
150
+ tx_to=f.tx_from,
151
+ provenance={**g.provenance,
152
+ "merged_away": merge_commit})
153
+ conflicts += 1
154
+ self._copy_fact(f, merge_commit)
155
+ applied += 1
156
+ elif conflict:
157
+ self._copy_fact(f, merge_commit) # union: keep both
158
+ conflicts += 1
159
+ applied += 1
160
+ else:
161
+ self._copy_fact(f, merge_commit)
162
+ applied += 1
163
+
164
+ for f in retired_by_theirs:
165
+ key = self._triple_key(f)
166
+ if key in ko and ko[key].is_active:
167
+ self.store.update_fact(
168
+ ko[key].id, is_active=0, retired_commit=merge_commit,
169
+ provenance={**ko[key].provenance,
170
+ "merged_away": f"retired in {name}"})
171
+ applied += 1
172
+
173
+ self.store.conn.commit()
174
+ return {"status": "merged", "commit": merge_commit,
175
+ "base": base_head, "applied": applied, "conflicts": conflicts,
176
+ "strategy": strategy}
177
+
178
+ def _copy_fact(self, f: Fact, commit_id: str) -> Fact:
179
+ nf = Fact(
180
+ id=new_id(), subject=f.subject, relation=f.relation, value=f.value,
181
+ valid_from=f.valid_from, valid_to=f.valid_to, tx_from=f.tx_from,
182
+ confidence=f.confidence, source_hash=f.source_hash,
183
+ source_id=f.source_id, user_id=f.user_id, agent_id=f.agent_id,
184
+ run_id=f.run_id, memory_type=f.memory_type,
185
+ access_count=f.access_count, reinforcement=f.reinforcement,
186
+ is_active=True, is_derived=f.is_derived,
187
+ provenance={**f.provenance, "merged_from": f.id})
188
+ self.store.insert_fact(nf, commit_id)
189
+ if self.palace is not None:
190
+ self.palace.add(nf.id, self.palace.encode_fact(nf))
191
+ return nf
192
+
193
+ def _common_ancestor(self, a: str, b: str) -> str | None:
194
+ anc_a = self.store.ancestry(a)
195
+ anc_b = self.store.ancestry(b)
196
+ common = anc_a & anc_b
197
+ if not common:
198
+ return None
199
+ best, best_ts = None, ""
200
+ for cid in common:
201
+ c = self.store.commit(cid)
202
+ if c and c["ts"] > best_ts:
203
+ best, best_ts = cid, c["ts"]
204
+ return best
@@ -0,0 +1,88 @@
1
+ """Predictive Memory Prefetching — the Memory Branch Target Buffer.
2
+
3
+ Learns co-access patterns across retrievals ("agents that asked about X
4
+ next asked about Y") and prefetches predicted facts into the fusion
5
+ boost set before the next query lands — branch prediction for agent
6
+ memory. Wrong predictions cost nothing (a tiny score boost); right
7
+ predictions cut effective retrieval latency toward the SLB hit path.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+
13
+ class Prefetcher:
14
+ def __init__(self, max_pairs: int = 200_000, decay: float = 0.98,
15
+ min_weight: float = 0.05, window: int = 6) -> None:
16
+ self.max_pairs = max_pairs
17
+ self.decay = decay
18
+ self.min_weight = min_weight
19
+ self.window = window
20
+ self._by_fid: dict[str, dict[str, float]] = {}
21
+ self._recent: list[str] = []
22
+ self._last_predict: dict[str, float] = {}
23
+ self._total_pairs = 0
24
+ self.hits = 0
25
+ self.predictions = 0
26
+ self.predicted_total = 0
27
+
28
+ # ------------------------------------------------------------------
29
+ def _bump(self, a: str, b: str) -> None:
30
+ row = self._by_fid.setdefault(a, {})
31
+ prev = row.get(b, 0.0)
32
+ row[b] = prev + 1.0
33
+ if prev == 0.0:
34
+ self._total_pairs += 1
35
+
36
+ def observe(self, fact_ids: list[str]) -> None:
37
+ """Called after each retrieval with the delivered fact set."""
38
+ combined = list(dict.fromkeys(
39
+ (self._recent[-self.window:] or []) + list(fact_ids)))
40
+ for i, a in enumerate(combined):
41
+ for b in combined[i + 1:i + 5]:
42
+ if a != b:
43
+ self._bump(a, b)
44
+ self._bump(b, a)
45
+ self._recent = list(fact_ids)
46
+ if self._total_pairs > self.max_pairs:
47
+ self._prune()
48
+
49
+ def _prune(self) -> None:
50
+ for fid in list(self._by_fid):
51
+ row = self._by_fid[fid]
52
+ for other in list(row):
53
+ row[other] *= self.decay
54
+ if row[other] < self.min_weight:
55
+ del row[other]
56
+ self._total_pairs -= 1
57
+ if not row:
58
+ del self._by_fid[fid]
59
+
60
+ # ------------------------------------------------------------------
61
+ def predict(self) -> dict[str, float]:
62
+ """Predict next-access facts from recent history (MBTB lookup)."""
63
+ out: dict[str, float] = {}
64
+ for fid in self._recent[-self.window:]:
65
+ row = self._by_fid.get(fid)
66
+ if not row:
67
+ continue
68
+ for other, w in row.items():
69
+ if w >= 1.0:
70
+ out[other] = max(out.get(other, 0.0), min(1.0, w / 8.0))
71
+ self._last_predict = out
72
+ self.predictions += 1
73
+ self.predicted_total += len(out)
74
+ return out
75
+
76
+ def note_hits(self, delivered_ids: list[str]) -> int:
77
+ hits = sum(1 for fid in delivered_ids if fid in self._last_predict)
78
+ self.hits += hits
79
+ return hits
80
+
81
+ def stats(self) -> dict:
82
+ return {
83
+ "pairs": self._total_pairs,
84
+ "predictions": self.predictions,
85
+ "prefetch_hits": self.hits,
86
+ "prefetch_hit_ratio": round(self.hits / self.predicted_total, 4)
87
+ if self.predicted_total else 0.0,
88
+ }
cortexm/features/zk.py ADDED
@@ -0,0 +1,105 @@
1
+ """Zero-Knowledge Memory Proofs (ZK-lite).
2
+
3
+ Proves a retrieved fact satisfies a query WITHOUT revealing its content
4
+ to the LLM: a Merkle membership proof over the tamper-evident leaf set
5
+ (blake3 source hashes) plus an HMAC attestation binding
6
+ {statement, root, timestamp}. The LLM receives only
7
+ ``[ZK-Proof: match on <relation> verified. Content redacted.]``.
8
+
9
+ Honest scope: this is a commit-and-prove membership attestation — full
10
+ ZK-SNARKs over the similarity predicate are on the roadmap (the binary
11
+ codec's Hamming-distance similarity is a natural circuit candidate;
12
+ HRR circular convolution is a group operation, the algebraic
13
+ requirement standard cosine similarity cannot satisfy).
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import secrets
19
+ import time
20
+
21
+ from cortexm.errors import VerificationError
22
+ from cortexm.security.hashes import (HashProvider, attest, merkle_proof,
23
+ merkle_verify, verify_attest)
24
+
25
+
26
+ class ZKProver:
27
+ def __init__(self, store, reader, provider: HashProvider | None = None) -> None:
28
+ self.store = store
29
+ self.reader = reader
30
+ self.hasher = provider or store.hasher
31
+ key = self.store.kv_get("ZK_KEY")
32
+ if not key:
33
+ key = secrets.token_hex(32)
34
+ self.store.kv_set("ZK_KEY", key)
35
+ self._key = bytes.fromhex(key)
36
+ self._leaf_cache: tuple[str, list[str]] = ("", [])
37
+
38
+ # ------------------------------------------------------------------
39
+ def _leaves(self) -> list[str]:
40
+ head = self.store.head() or ""
41
+ if self._leaf_cache[0] != head or not self._leaf_cache[1]:
42
+ hashes = self.store.active_fact_hashes()
43
+ leaves = [f"{fid}:{h}" for fid, h in hashes]
44
+ # leaf hash -> hex digest for merkle
45
+ leaves = [self.hasher.hash_text(l) for l in leaves]
46
+ self._leaf_cache = (head, leaves)
47
+ return self._leaf_cache[1]
48
+
49
+ # ------------------------------------------------------------------
50
+ def prove(self, query: str, *, user_id: str = "default",
51
+ threshold: float = 0.2) -> dict:
52
+ """Retrieve top match and produce a content-free proof."""
53
+ result = self.reader.search(query, user_id=user_id, k=1)
54
+ if not result.facts:
55
+ raise VerificationError("no matching fact to prove")
56
+ f = result.facts[0]
57
+ leaves = self._leaves()
58
+ idx = None
59
+ want = self.hasher.hash_text(f"{f.id}:{f.source_hash}")
60
+ for i, leaf in enumerate(leaves):
61
+ if leaf == want:
62
+ idx = i
63
+ break
64
+ if idx is None:
65
+ raise VerificationError("fact not in active leaf set")
66
+ root, path = merkle_proof(self.hasher, leaves, idx)
67
+ score = result.scores.get(f.id, 0.0)
68
+ if score < threshold:
69
+ raise VerificationError(
70
+ f"similarity {score:.3f} below threshold {threshold}")
71
+ statement = (f"EXISTS fact f in Trace: sim(query, f) >= {threshold} "
72
+ f"AND blake3(source(f)) = {f.source_hash[:16]}... "
73
+ f"(content redacted)")
74
+ ts = time.time()
75
+ tag = attest(self.hasher, self._key, f"{statement}|{root}|{ts}")
76
+ return {
77
+ "statement": statement,
78
+ "fact_commitment": f.source_hash, # content-free commitment
79
+ "leaf_commitment": want, # merkle leaf hash
80
+ "relation": f.relation, # safe to disclose
81
+ "sim_score": round(score, 4),
82
+ "merkle_root": root,
83
+ "merkle_path": path,
84
+ "timestamp": ts,
85
+ "attestation": tag,
86
+ "llm_view": f"[ZK-Proof: high-confidence match on '{f.relation}' "
87
+ f"verified (score {score:.2f}). Content redacted.]",
88
+ }
89
+
90
+ # ------------------------------------------------------------------
91
+ def verify(self, proof: dict) -> bool:
92
+ """Verify membership + attestation. Returns True if sound."""
93
+ if not verify_attest(self.hasher, self._key,
94
+ f"{proof['statement']}|{proof['merkle_root']}|"
95
+ f"{proof['timestamp']}",
96
+ proof["attestation"]):
97
+ return False
98
+ leaf = proof.get("leaf_commitment") or proof["fact_commitment"]
99
+ return merkle_verify(self.hasher, leaf,
100
+ proof["merkle_path"], proof["merkle_root"])
101
+
102
+ def verify_membership(self, leaf_hex: str, proof: dict) -> bool:
103
+ """Full membership verification for a known leaf commitment."""
104
+ return merkle_verify(self.hasher, leaf_hex, proof["merkle_path"],
105
+ proof["merkle_root"])