cortexm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. context_m.py +17 -0
  2. cortexm/__init__.py +45 -0
  3. cortexm/accel.py +403 -0
  4. cortexm/api/__init__.py +0 -0
  5. cortexm/api/chaos.py +118 -0
  6. cortexm/api/memory.py +635 -0
  7. cortexm/bench/__init__.py +0 -0
  8. cortexm/bench/abilities.py +311 -0
  9. cortexm/bench/baselines.py +89 -0
  10. cortexm/bench/beam_loader.py +317 -0
  11. cortexm/bench/generator.py +376 -0
  12. cortexm/bench/harness.py +211 -0
  13. cortexm/bench/messy.py +218 -0
  14. cortexm/bench/micro.py +251 -0
  15. cortexm/bench/ood.py +443 -0
  16. cortexm/bench/run.py +137 -0
  17. cortexm/bridge/__init__.py +0 -0
  18. cortexm/bridge/dates.py +178 -0
  19. cortexm/bridge/decoders.py +204 -0
  20. cortexm/bridge/enrich.py +255 -0
  21. cortexm/bridge/extractor.py +316 -0
  22. cortexm/bridge/fallback.py +332 -0
  23. cortexm/bridge/onnx_runtime.py +158 -0
  24. cortexm/bridge/patterns.py +760 -0
  25. cortexm/bridge/ppr.py +104 -0
  26. cortexm/bridge/prefilter.py +188 -0
  27. cortexm/bridge/query_extract.py +420 -0
  28. cortexm/bridge/reader.py +1174 -0
  29. cortexm/bridge/rerank.py +204 -0
  30. cortexm/bridge/writer.py +492 -0
  31. cortexm/cli.py +295 -0
  32. cortexm/cognition/__init__.py +53 -0
  33. cortexm/cognition/abstraction.py +192 -0
  34. cortexm/cognition/analogy.py +159 -0
  35. cortexm/cognition/engine.py +204 -0
  36. cortexm/cognition/gaps.py +365 -0
  37. cortexm/cognition/scanner.py +204 -0
  38. cortexm/config.py +375 -0
  39. cortexm/cortexm.py +8 -0
  40. cortexm/enterprise/__init__.py +0 -0
  41. cortexm/enterprise/audit.py +178 -0
  42. cortexm/enterprise/governance.py +239 -0
  43. cortexm/errors.py +35 -0
  44. cortexm/features/__init__.py +0 -0
  45. cortexm/features/git.py +204 -0
  46. cortexm/features/prefetch.py +88 -0
  47. cortexm/features/zk.py +105 -0
  48. cortexm/federation/__init__.py +39 -0
  49. cortexm/federation/crdt.py +275 -0
  50. cortexm/federation/fabric.py +109 -0
  51. cortexm/federation/hlc.py +80 -0
  52. cortexm/federation/node.py +145 -0
  53. cortexm/federation/schema_report.py +73 -0
  54. cortexm/federation/transport.py +164 -0
  55. cortexm/index/__init__.py +19 -0
  56. cortexm/index/nsg.py +386 -0
  57. cortexm/mcp/__init__.py +0 -0
  58. cortexm/mcp/server.py +985 -0
  59. cortexm/metrics.py +62 -0
  60. cortexm/migrate/__init__.py +0 -0
  61. cortexm/migrate/importers.py +192 -0
  62. cortexm/provenance/__init__.py +78 -0
  63. cortexm/provenance/agent.py +214 -0
  64. cortexm/provenance/cose.py +201 -0
  65. cortexm/provenance/scitt.py +258 -0
  66. cortexm/provenance/vc.py +250 -0
  67. cortexm/security/__init__.py +0 -0
  68. cortexm/security/crypto.py +162 -0
  69. cortexm/security/hashes.py +140 -0
  70. cortexm/security/injection.py +149 -0
  71. cortexm/security/mind.py +154 -0
  72. cortexm/security/pii.py +265 -0
  73. cortexm/security/rbac.py +169 -0
  74. cortexm/security/sandbox.py +131 -0
  75. cortexm/security/zk_hamming.py +142 -0
  76. cortexm/security/zk_sql.py +485 -0
  77. cortexm/server/__init__.py +0 -0
  78. cortexm/server/metrics.py +88 -0
  79. cortexm/server/rest.py +936 -0
  80. cortexm/server/sparql.py +984 -0
  81. cortexm/text/__init__.py +0 -0
  82. cortexm/text/dissim.py +252 -0
  83. cortexm/text/embedder.py +155 -0
  84. cortexm/text/fuzzy.py +218 -0
  85. cortexm/text/idiolect.py +253 -0
  86. cortexm/text/labse.py +374 -0
  87. cortexm/text/tokenizer.py +79 -0
  88. cortexm/trace/__init__.py +0 -0
  89. cortexm/trace/blob_arena.py +277 -0
  90. cortexm/trace/consolidate.py +337 -0
  91. cortexm/trace/contradictions.py +69 -0
  92. cortexm/trace/dedup.py +114 -0
  93. cortexm/trace/edges.py +214 -0
  94. cortexm/trace/fact.py +121 -0
  95. cortexm/trace/fade.py +245 -0
  96. cortexm/trace/lifecycle.py +112 -0
  97. cortexm/trace/rebuild.py +173 -0
  98. cortexm/trace/rules.py +171 -0
  99. cortexm/trace/store.py +680 -0
  100. cortexm/trace/structural.py +183 -0
  101. cortexm/trace/tmt.py +335 -0
  102. cortexm/util.py +148 -0
  103. cortexm/vsa/__init__.py +0 -0
  104. cortexm/vsa/attribution.py +149 -0
  105. cortexm/vsa/cleanup.py +161 -0
  106. cortexm/vsa/codecs.py +397 -0
  107. cortexm/vsa/hologram_overlay.py +139 -0
  108. cortexm/vsa/index.py +163 -0
  109. cortexm/vsa/ops.py +149 -0
  110. cortexm/vsa/palace.py +446 -0
  111. cortexm/vsa/role_vectors.py +236 -0
  112. cortexm/vsa/slb.py +78 -0
  113. cortexm/vsa/tlsh_trie.py +137 -0
  114. cortexm/vsa/working_memory.py +249 -0
  115. cortexm-0.3.0.dist-info/METADATA +482 -0
  116. cortexm-0.3.0.dist-info/RECORD +120 -0
  117. cortexm-0.3.0.dist-info/WHEEL +5 -0
  118. cortexm-0.3.0.dist-info/entry_points.txt +2 -0
  119. cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
  120. cortexm-0.3.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,79 @@
1
+ """Deterministic tokenizer & sentence segmentation (μ=0 building block)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+
7
+ WORD_RE = re.compile(r"[a-zA-Z][a-zA-Z'’-]*|\d+(?:\.\d+)?")
8
+ ABBREV = {"mr", "mrs", "ms", "dr", "prof", "sr", "jr", "st", "vs", "etc",
9
+ "e.g", "i.e", "fig", "inc", "ltd", "co", "u.s", "u.k"}
10
+
11
+ _SENT_SPLIT = re.compile(r"(?<=[.!?])[\"')\]]*\s+|\n{2,}")
12
+ _SENT_GUARD = re.compile(r"^(?:[A-Z0-9\"'(]|…)")
13
+
14
+
15
+ def words(text: str) -> list[str]:
16
+ return [w.lower().replace("’", "'") for w in WORD_RE.findall(text)]
17
+
18
+
19
+ def sentences(text: str) -> list[tuple[int, int, str]]:
20
+ """Sentence segmentation with (start, end, sentence) spans."""
21
+ out: list[tuple[int, int, str]] = []
22
+ if not text or not text.strip():
23
+ return out
24
+ pos = 0
25
+ for raw in _SENT_SPLIT.split(text):
26
+ s = raw.strip()
27
+ if not s:
28
+ continue
29
+ start = text.find(s[:24], pos)
30
+ if start < 0:
31
+ start = pos
32
+ end = start + len(s)
33
+ out.append((start, end, s))
34
+ pos = end
35
+ # merge fragments ending in abbreviations ("I met Dr. Chen.")
36
+ merged: list[tuple[int, int, str]] = []
37
+ for item in out:
38
+ if merged:
39
+ ps, pe, ptext = merged[-1]
40
+ tail = re.sub(r"[^\w.]", "", ptext.split()[-1] if ptext.split() else "")
41
+ if tail.rstrip(".").lower() in ABBREV:
42
+ nxt = text[pe:item[1]]
43
+ merged[-1] = (ps, item[1], text[ps:item[1]])
44
+ continue
45
+ merged.append(item)
46
+ return merged
47
+
48
+
49
+ def strip_punct(s: str) -> str:
50
+ return re.sub(r"[^\w\s'-]", " ", s).strip()
51
+
52
+
53
+ STOPWORDS = {
54
+ "a", "an", "the", "and", "or", "but", "if", "of", "to", "in", "on", "at",
55
+ "for", "with", "is", "are", "was", "were", "be", "been", "am", "do",
56
+ "does", "did", "have", "has", "had", "i", "you", "he", "she", "it", "we",
57
+ "they", "me", "my", "your", "his", "her", "its", "our", "their", "this",
58
+ "that", "these", "those", "as", "so", "than", "then", "there", "here",
59
+ "what", "which", "who", "whom", "when", "where", "why", "how", "will",
60
+ "would", "can", "could", "should", "shall", "may", "might", "just",
61
+ "about", "from", "by", "up", "out", "not", "no", "yes", "oh", "well",
62
+ }
63
+
64
+
65
+ def content_words(text: str) -> list[str]:
66
+ return [w for w in words(text) if w not in STOPWORDS and len(w) > 1]
67
+
68
+
69
+ def cap_sequences(text: str) -> list[str]:
70
+ """Capitalized multi-word sequences — poor-man's NER (μ=0)."""
71
+ seqs = re.findall(
72
+ r"\b([A-Z][a-zA-Z'&-]*(?:[ ](?:of|the|and|de|van|for)[ ])?[ ]*[A-Z][a-zA-Z'&-]*)+\b",
73
+ text)
74
+ out = []
75
+ for s in seqs:
76
+ s = " ".join(s.split())
77
+ if len(s) > 2 and not s.islower() and "." not in s:
78
+ out.append(s)
79
+ return out
File without changes
@@ -0,0 +1,277 @@
1
+ """Sidecar blob arena — Aeon-inspired off-graph text storage.
2
+
3
+ arXiv:2601.15311 (Aeon): large text is stored off-graph in an
4
+ append-only mmap-backed blob file with generational GC. Graph nodes
5
+ hold only a 64-byte preview + offset.
6
+
7
+ Context-M's chunks table currently stores full text inline
8
+ (`text TEXT NOT NULL`). For typical personas this is fine — chunks
9
+ are 200-400 bytes. But for use cases that ingest long documents
10
+ (research papers, meeting transcripts, issue threads), the chunks
11
+ table balloons and the working set of the SQLite page cache gets
12
+ polluted by long text the graph rarely needs.
13
+
14
+ This module provides an OPT-IN sidecar blob arena:
15
+
16
+ BlobArena(path) — opens/creates an mmap'd blob file at `path`
17
+ .put(text) -> (blob_id, offset, length)
18
+ .get(offset, length) -> bytes
19
+ .preview(text, n=64) -> first n bytes (the in-graph preview)
20
+
21
+ The host migrates chunks by:
22
+ 1. creating a BlobArena
23
+ 2. for each chunk: arena.put(text) -> (blob_id, offset, len)
24
+ 3. updating the chunks row: text = preview, blob_offset = offset,
25
+ blob_len = len (schema migration adds the columns)
26
+ 4. on retrieval: chunks.text gives the preview (fast); full text
27
+ is fetched via arena.get(offset, len) only when the audit
28
+ / retrieval path actually needs it
29
+
30
+ Generational GC is out of scope for v1 (chunks are append-mostly in
31
+ practice — retired facts are deactivated but the chunk text is kept
32
+ for audit). The arena is a single mmap'd file; concurrent writers
33
+ must hold a lock (the arena serializes via flock).
34
+
35
+ BACKWARD COMPATIBILITY: the chunks schema migration is OPT-IN. The
36
+ default Memory() path still stores text inline. To enable the
37
+ sidecar, call Memory.enable_blob_arena(path) after construction.
38
+ """
39
+ from __future__ import annotations
40
+
41
+ import io
42
+ import mmap
43
+ import os
44
+ import threading
45
+ import zlib
46
+ from pathlib import Path
47
+ from typing import Iterator
48
+
49
+
50
+ # Schema additions for the chunks table (applied by TraceStore when
51
+ # the arena is enabled). The columns are nullable so existing rows
52
+ # (with inline text) keep working — a NULL blob_offset means "use the
53
+ # inline text column".
54
+ SCHEMA_MIGRATION = """
55
+ ALTER TABLE chunks ADD COLUMN blob_offset INTEGER DEFAULT NULL;
56
+ ALTER TABLE chunks ADD COLUMN blob_len INTEGER DEFAULT NULL;
57
+ ALTER TABLE chunks ADD COLUMN blob_compressed INTEGER DEFAULT 0;
58
+ """
59
+
60
+
61
+ class BlobArena:
62
+ """Append-only mmap-backed blob file.
63
+
64
+ Layout:
65
+ - 8-byte header: magic + version + length
66
+ - records: [8-byte length | length bytes of payload]
67
+ (payload may be zlib-compressed if compressed=1 in the chunk
68
+ row; the arena itself is format-agnostic to the bytes)
69
+
70
+ The arena serializes writes via a process-local threading.Lock
71
+ AND an flock on the file for cross-process safety. Concurrent
72
+ readers don't need the lock — mmap pages are CoW.
73
+ """
74
+
75
+ MAGIC = b"BLB1" # Blob Layer v1
76
+ HEADER_LEN = 16
77
+
78
+ def __init__(self, path: str | os.PathLike) -> None:
79
+ self.path = Path(path)
80
+ self.path.parent.mkdir(parents=True, exist_ok=True)
81
+ # create or open
82
+ is_new = not self.path.exists()
83
+ if is_new:
84
+ with open(self.path, "wb") as f:
85
+ f.write(self.MAGIC + b"\x00" * 12) # magic + 12 bytes pad
86
+ self._fd = open(self.path, "r+b")
87
+ # write-lock for cross-process safety
88
+ try:
89
+ import fcntl
90
+ fcntl.flock(self._fd.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
91
+ except (ImportError, OSError):
92
+ pass # Windows or already locked by same process
93
+ self._mmap: mmap.mmap | None = None
94
+ self._lock = threading.Lock()
95
+ self._open_mmap()
96
+
97
+ def _open_mmap(self) -> None:
98
+ if self._mmap is None:
99
+ try:
100
+ self._mmap = mmap.mmap(
101
+ self._fd.fileno(), 0, access=mmap.ACCESS_WRITE)
102
+ except (ValueError, OSError):
103
+ # empty file (just header) — fall back to file I/O
104
+ self._mmap = None
105
+
106
+ def put(self, data: bytes, compress: bool = True) -> tuple[int, int, int, bool]:
107
+ """Append `data` to the blob file.
108
+
109
+ Returns (blob_id, offset, length, was_compressed).
110
+ blob_id — monotonically increasing int (records / appended)
111
+ offset — byte offset in the file where the record starts
112
+ length — payload length in bytes
113
+ was_compressed — whether zlib compression was applied
114
+
115
+ Compression: enabled by default for text payloads > 256 bytes;
116
+ small payloads are stored raw because the zlib header (5+ bytes)
117
+ makes small blobs larger.
118
+ """
119
+ if isinstance(data, str):
120
+ data = data.encode("utf-8")
121
+ if compress and len(data) > 256:
122
+ compressed = zlib.compress(data, level=6)
123
+ if len(compressed) < len(data) - 8: # only if meaningful savings
124
+ payload = compressed
125
+ was_compressed = True
126
+ else:
127
+ payload = data
128
+ was_compressed = False
129
+ else:
130
+ payload = data
131
+ was_compressed = False
132
+
133
+ with self._lock:
134
+ # we need to write 8 bytes (length) + len(payload) bytes
135
+ rec_len = len(payload)
136
+ # current file size = where this record starts
137
+ self._fd.seek(0, io.SEEK_END)
138
+ offset = self._fd.tell()
139
+ # write 8-byte length + payload
140
+ self._fd.write(rec_len.to_bytes(8, "little"))
141
+ self._fd.write(payload)
142
+ self._fd.flush()
143
+ # invalidate the mmap view so the next get() re-reads
144
+ if self._mmap is not None:
145
+ try:
146
+ self._mmap.close()
147
+ except Exception:
148
+ pass
149
+ self._mmap = None
150
+ self._open_mmap()
151
+ blob_id = offset # offset IS the id (unique within file)
152
+ return blob_id, offset, rec_len, was_compressed
153
+
154
+ def get(self, offset: int, length: int,
155
+ was_compressed: bool = False) -> bytes:
156
+ """Read a record from the blob file.
157
+
158
+ `offset` and `length` come from the chunks table. If
159
+ `was_compressed` is True (stored in the chunk row), the
160
+ returned bytes are zlib-decompressed before return.
161
+ """
162
+ with self._lock:
163
+ self._fd.seek(offset)
164
+ raw_len = self._fd.read(8)
165
+ if len(raw_len) != 8:
166
+ raise IOError(f"short read at offset {offset}")
167
+ rec_len = int.from_bytes(raw_len, "little")
168
+ payload = self._fd.read(rec_len)
169
+ if len(payload) != rec_len:
170
+ raise IOError(
171
+ f"short payload read at offset {offset}: "
172
+ f"expected {rec_len}, got {len(payload)}")
173
+ if was_compressed:
174
+ try:
175
+ return zlib.decompress(payload)
176
+ except zlib.error:
177
+ return payload # return raw on decompression failure
178
+ return payload
179
+
180
+ def get_text(self, offset: int, length: int,
181
+ was_compressed: bool = False) -> str:
182
+ """Convenience: get bytes and decode as UTF-8."""
183
+ return self.get(offset, length, was_compressed).decode(
184
+ "utf-8", errors="replace")
185
+
186
+ def close(self) -> None:
187
+ with self._lock:
188
+ if self._mmap is not None:
189
+ try:
190
+ self._mmap.close()
191
+ except Exception:
192
+ pass
193
+ self._mmap = None
194
+ try:
195
+ self._fd.close()
196
+ except Exception:
197
+ pass
198
+
199
+ def __enter__(self) -> "BlobArena":
200
+ return self
201
+
202
+ def __exit__(self, *exc) -> None:
203
+ self.close()
204
+
205
+
206
+ # ---------- Migration helper ------------------------------------------
207
+
208
+ def migrate_chunks_to_arena(store, arena: BlobArena,
209
+ batch_size: int = 1000) -> dict:
210
+ """Migrate existing chunks.text rows into the sidecar arena.
211
+
212
+ For each chunk:
213
+ - put text into arena -> (blob_id, offset, len, compressed)
214
+ - update the row: blob_offset=offset, blob_len=len,
215
+ blob_compressed=compressed, text=preview (first 64 bytes)
216
+ The original text column is REPLACED with the preview — the full
217
+ text lives only in the arena. Existing readers that use chunks.text
218
+ will see the preview (good for display); readers that need full
219
+ text must call arena.get_text(offset, len, compressed).
220
+
221
+ Idempotent: chunks with blob_offset IS NOT NULL are skipped.
222
+ """
223
+ # ensure the schema migration has run
224
+ cols = {r[1] for r in store.conn.execute("PRAGMA table_info(chunks)").fetchall()}
225
+ if "blob_offset" not in cols:
226
+ for stmt in SCHEMA_MIGRATION.strip().split(";"):
227
+ stmt = stmt.strip()
228
+ if stmt:
229
+ store.conn.execute(stmt)
230
+ store.conn.commit()
231
+
232
+ migrated = 0
233
+ skipped = 0
234
+ while True:
235
+ rows = store.conn.execute(
236
+ "SELECT id, text FROM chunks WHERE blob_offset IS NULL "
237
+ f"LIMIT {batch_size}").fetchall()
238
+ if not rows:
239
+ break
240
+ for chunk_id, text in rows:
241
+ if not text:
242
+ skipped += 1
243
+ continue
244
+ data = text.encode("utf-8")
245
+ _, offset, length, compressed = arena.put(data, compress=True)
246
+ preview = text[:64] + ("..." if len(text) > 64 else "")
247
+ store.conn.execute(
248
+ "UPDATE chunks SET text=?, blob_offset=?, blob_len=?, "
249
+ "blob_compressed=? WHERE id=?",
250
+ (preview, offset, length, 1 if compressed else 0, chunk_id))
251
+ migrated += 1
252
+ store.conn.commit()
253
+ return {"migrated": migrated, "skipped": skipped}
254
+
255
+
256
+ def get_chunk_text(store, arena: BlobArena, chunk_id: str) -> str:
257
+ """Fetch full text for a chunk — from the arena if blob_offset is
258
+ set, otherwise fall back to inline text.
259
+
260
+ Use this in the audit / retrieval path when the 64-byte preview
261
+ in chunks.text isn't enough and you need the full source.
262
+ """
263
+ row = store.conn.execute(
264
+ "SELECT text, blob_offset, blob_len, blob_compressed "
265
+ "FROM chunks WHERE id=?", (chunk_id,)).fetchone()
266
+ if not row:
267
+ return ""
268
+ text, offset, length, compressed = row
269
+ if offset is None:
270
+ return text or ""
271
+ return arena.get_text(offset, length, bool(compressed))
272
+
273
+
274
+ __all__ = [
275
+ "BlobArena", "SCHEMA_MIGRATION",
276
+ "migrate_chunks_to_arena", "get_chunk_text",
277
+ ]
@@ -0,0 +1,337 @@
1
+ """Dreaming / consolidation — Aeon-inspired idle-time optimization.
2
+
3
+ arXiv:2601.15311 (Aeon) describes a background task analogous to
4
+ biological sleep: defragmentation, GC, and consolidation of the
5
+ verbose Trace into compressed long-term episodic summaries.
6
+
7
+ Context-M's Memory Git is currently passive — it version-controls
8
+ facts but doesn't optimize them. This module turns it into an ACTIVE
9
+ memory optimizer that runs during idle periods:
10
+
11
+ consolidate(store, palace, prefetcher, ...) ->
12
+ - merges redundant triples in the Trace (same subject/relation,
13
+ near-duplicate values) via MERGED_WITH edges
14
+ - retires facts past their valid_to + grace period
15
+ - defragments the palace by rebuilding the in-memory packed matrix
16
+ from active facts only (drops retired/merged IDs)
17
+ - re-trains the MBTB prefetcher from recent query access patterns
18
+
19
+ The pass is IDEMPOTENT and SAFE — every change is a commit on the
20
+ current branch, every retired fact is still queryable via
21
+ allow_inactive=True, every merged fact keeps a MERGED_WITH edge so
22
+ the original triple is recoverable for audit. A failed pass rolls back
23
+ via the existing Memory Git ancestry.
24
+
25
+ NOTE: this is NOT a separate daemon — it's a function the host app
26
+ calls when idle (e.g.overnight cron, `cortexm consolidate` CLI, or
27
+ an `on_idle` hook in the MCP server).
28
+ """
29
+ from __future__ import annotations
30
+
31
+ import datetime as _dt
32
+ from datetime import datetime, timezone
33
+ from typing import Iterable
34
+
35
+ from cortexm.trace.edges import MERGED_WITH, RETRACTED_BY
36
+ from cortexm.util import iso, similarity
37
+
38
+
39
+ def _now() -> datetime:
40
+ return datetime.now(timezone.utc)
41
+
42
+
43
+ def consolidate(store, palace=None, prefetcher=None, *,
44
+ user_id: str | None = None,
45
+ merge_threshold: float = 0.92,
46
+ retire_grace_days: int = 365,
47
+ defrag_palace: bool = True,
48
+ retrain_prefetcher: bool = True,
49
+ run_fade: bool = True,
50
+ run_tmt: bool = False,
51
+ run_cognition: bool = False,
52
+ dry_run: bool = False,
53
+ fade_cfg: dict | None = None,
54
+ tmt_cfg: dict | None = None,
55
+ cognition_cfg: dict | None = None) -> dict:
56
+ """Run a single consolidation pass.
57
+
58
+ Returns a stats dict:
59
+ {merged_pairs, retired_facts, palace_defragged,
60
+ prefetcher_retrained, fade_stats, tmt_stats,
61
+ cognition_stats, commit_id, dry_run}
62
+
63
+ Parameters:
64
+ store — TraceStore
65
+ palace — MemoryPalace (optional; only needed for defrag)
66
+ prefetcher — Prefetcher (optional; only needed for retrain)
67
+ user_id — restrict pass to one user (None = all users)
68
+ merge_threshold — jaccard similarity above which two facts
69
+ (same subject+relation) are merged
70
+ retire_grace_days — facts with valid_to older than this many
71
+ days are retired (deactivated)
72
+ defrag_palace — rebuild the palace packed matrix from active facts
73
+ retrain_prefetcher — rebuild the MBTB from access_count stats
74
+ run_fade — also run FadeMem sweep (decay + deactivate + merge)
75
+ run_tmt — also run TiMem TMT hierarchy build
76
+ run_cognition — also run the HMS-style Cognition Engine pass
77
+ (PatternScanner + AbstractionEngine + GapDetector
78
+ + HypothesisEngine + AnalogyDetector). Emits
79
+ HYPOTHESIZED_BY edges with confidence < 0.5 —
80
+ never active in retrieval unless promoted.
81
+ fade_cfg — kwargs for fade_sweep (lambda_, thresholds, etc.)
82
+ tmt_cfg — kwargs for tmt_build (cluster mins, etc.)
83
+ cognition_cfg — kwargs for run_cognition_pass
84
+ dry_run — compute the changes but don't apply them
85
+ """
86
+ stats = {
87
+ "merged_pairs": 0,
88
+ "retired_facts": 0,
89
+ "palace_defragged": False,
90
+ "prefetcher_retrained": False,
91
+ "fade_stats": None,
92
+ "tmt_stats": None,
93
+ "cognition_stats": None,
94
+ "commit_id": None,
95
+ "dry_run": dry_run,
96
+ }
97
+
98
+ # ---------- 1. Merge redundant triples -------------------------------
99
+ # Group active facts by (user_id, subject, relation) and find
100
+ # near-duplicate values within each group.
101
+ where = "is_active=1 AND quarantined=0"
102
+ args: tuple = ()
103
+ if user_id is not None:
104
+ where += " AND user_id=?"
105
+ args = (user_id,)
106
+ rows = store.conn.execute(
107
+ f"SELECT id, subject, relation, value, user_id, confidence, "
108
+ f"access_count FROM facts WHERE {where} ORDER BY user_id, subject, "
109
+ f"relation, valid_from", args).fetchall()
110
+ groups: dict[tuple, list[dict]] = {}
111
+ for r in rows:
112
+ r = dict(r)
113
+ key = (r["user_id"], r["subject"], r["relation"])
114
+ groups.setdefault(key, []).append(r)
115
+
116
+ merge_pairs: list[tuple[str, str, float]] = []
117
+ for key, group in groups.items():
118
+ if len(group) < 2:
119
+ continue
120
+ # pairwise similarity (small groups, so O(n^2) is fine)
121
+ for i, a in enumerate(group):
122
+ for b in group[i + 1:]:
123
+ sim = similarity(a["value"], b["value"])
124
+ if sim >= merge_threshold:
125
+ # keep the higher-confidence one; merge the other into it
126
+ if a["confidence"] >= b["confidence"]:
127
+ keep, drop = a, b
128
+ else:
129
+ keep, drop = b, a
130
+ merge_pairs.append((keep["id"], drop["id"], sim))
131
+
132
+ if not dry_run:
133
+ store.begin_batch()
134
+ commit = store.create_commit(
135
+ f"consolidate: merge {len(merge_pairs)} pairs, "
136
+ f"retire stale", n_facts=0)
137
+ stats["commit_id"] = commit
138
+
139
+ # apply merges
140
+ for keep_id, drop_id, sim in merge_pairs:
141
+ # deactivate the drop, wire MERGED_WITH edge so audits can
142
+ # always recover the original triple
143
+ store.update_fact(
144
+ drop_id, is_active=0, tx_to=iso(_now()),
145
+ retired_commit=commit,
146
+ provenance={"merged_into": keep_id,
147
+ "merge_sim": round(sim, 4),
148
+ "consolidated_at": iso(_now())})
149
+ store.add_edge(keep_id, drop_id, MERGED_WITH,
150
+ {"sim": round(sim, 4),
151
+ "consolidated_at": iso(_now())})
152
+
153
+ stats["merged_pairs"] = len(merge_pairs)
154
+
155
+ # ---------- 2. Retire stale facts -----------------------------------
156
+ # Facts whose valid_to is older than retire_grace_days and which
157
+ # haven't been accessed recently (access_count == 0) — these are
158
+ # safe to retire; the bi-temporal model still serves them via
159
+ # allow_inactive=True.
160
+ cutoff = (_now() - _dt.timedelta(days=retire_grace_days)).strftime(
161
+ "%Y-%m-%d")
162
+ retire_where = (
163
+ f"is_active=1 AND valid_to IS NOT NULL AND valid_to < ? "
164
+ f"AND valid_to != '' AND access_count = 0")
165
+ retire_args: tuple = (cutoff,)
166
+ if user_id is not None:
167
+ retire_where += " AND user_id=?"
168
+ retire_args = (cutoff, user_id)
169
+ retire_rows = store.conn.execute(
170
+ f"SELECT id FROM facts WHERE {retire_where}", retire_args).fetchall()
171
+ retire_ids = [r[0] for r in retire_rows]
172
+
173
+ if not dry_run:
174
+ for fid in retire_ids:
175
+ store.update_fact(
176
+ fid, is_active=0, tx_to=iso(_now()),
177
+ retired_commit=commit,
178
+ provenance={"retired_by_consolidate": True,
179
+ "consolidated_at": iso(_now())})
180
+ stats["retired_facts"] = len(retire_ids)
181
+
182
+ # ---------- 3. Palace defrag ----------------------------------------
183
+ # Rebuild the palace's packed matrix from active facts only.
184
+ # This drops retired / merged IDs from the in-memory index and
185
+ # re-tightens the page-clustered tree (currently a no-op for the
186
+ # SQLite-backed palace; for the in-memory palace it compacts).
187
+ if defrag_palace and palace is not None and not dry_run:
188
+ try:
189
+ # the palace's _n tracks active entries — a defrag pass
190
+ # re-builds the matrix by re-adding only active fact IDs.
191
+ # For the SQLite-backed palace this is a no-op (vectors
192
+ # are stored in a BLOB, not a packed matrix), but the
193
+ # call still flushes any in-memory dirty state.
194
+ if hasattr(palace, "defrag"):
195
+ palace.defrag()
196
+ elif hasattr(palace, "close"):
197
+ palace.close()
198
+ stats["palace_defragged"] = True
199
+ except Exception:
200
+ pass
201
+
202
+ # ---------- 4. Prefetcher retrain ------------------------------------
203
+ # The MBTB prefetcher tracks co-access patterns. Re-training from
204
+ # access_count stats lets it pick up shifts in user behavior.
205
+ if retrain_prefetcher and prefetcher is not None and not dry_run:
206
+ try:
207
+ if hasattr(prefetcher, "retrain"):
208
+ prefetcher.retrain(store)
209
+ elif hasattr(prefetcher, "rebuild"):
210
+ prefetcher.rebuild(store)
211
+ stats["prefetcher_retrained"] = True
212
+ except Exception:
213
+ pass
214
+
215
+ # ---------- 5. FadeMem sweep (decay + deactivate + cluster merge) ----
216
+ # Biologically-inspired forgetting: exponential decay on retention
217
+ # scores, with access-driven reconsolidation (frequently-retrieved
218
+ # facts resist decay). Deactivates facts whose retention_score drops
219
+ # below fade_deactivate_threshold; merges clusters of low-retention
220
+ # siblings. Bi-temporal safe: only is_active flips.
221
+ if run_fade and not dry_run:
222
+ try:
223
+ from cortexm.trace.fade import fade_sweep
224
+ fade_kwargs = dict(
225
+ lambda_=0.05,
226
+ access_boost=0.5,
227
+ contradiction_penalty=0.25,
228
+ deactivate_threshold=0.10,
229
+ merge_threshold=0.30,
230
+ merge_similarity=merge_threshold,
231
+ user_id=user_id,
232
+ dry_run=dry_run,
233
+ )
234
+ if fade_cfg:
235
+ fade_kwargs.update(fade_cfg)
236
+ stats["fade_stats"] = fade_sweep(store, palace, **fade_kwargs)
237
+ except Exception as e:
238
+ stats["fade_stats"] = {"error": str(e)}
239
+
240
+ # ---------- 6. TiMem TMT hierarchy build -----------------------------
241
+ # Episodic → session → day → persona abstraction. Each higher level
242
+ # is a derived fact with DERIVED_FROM edges back to its constituents.
243
+ # Retrieval can short-circuit to the appropriate level based on
244
+ # query complexity.
245
+ if run_tmt and not dry_run:
246
+ try:
247
+ from cortexm.trace.tmt import tmt_build
248
+ tmt_kwargs = dict(
249
+ session_cluster_mins=5,
250
+ persona_min_sessions=3,
251
+ user_id=user_id,
252
+ )
253
+ if tmt_cfg:
254
+ tmt_kwargs.update(tmt_cfg)
255
+ stats["tmt_stats"] = tmt_build(store, palace, **tmt_kwargs)
256
+ except Exception as e:
257
+ stats["tmt_stats"] = {"error": str(e)}
258
+
259
+ # ---------- 7. HMS Cognition Engine pass ----------------------------
260
+ # PatternScanner + AbstractionEngine + GapDetector + HypothesisEngine
261
+ # + AnalogyDetector. Surfaces structural regularities, builds
262
+ # prototype categories, fills in missing relations via hypotheses,
263
+ # finds cross-domain analogies. Output is derived facts with
264
+ # confidence < 0.5 and is_derived=1, never promoted to active
265
+ # retrieval unless explicitly confirmed by user input.
266
+ if run_cognition:
267
+ try:
268
+ from cortexm.cognition import run_cognition_pass
269
+ cog_kwargs = dict(
270
+ dry_run=dry_run,
271
+ user_id=user_id,
272
+ )
273
+ if cognition_cfg:
274
+ cog_kwargs.update(cognition_cfg)
275
+ cog_report = run_cognition_pass(store, palace=palace,
276
+ **cog_kwargs)
277
+ stats["cognition_stats"] = {
278
+ "scan": cog_report.scan,
279
+ "abstraction": cog_report.abstraction,
280
+ "gaps": cog_report.gaps,
281
+ "hypotheses": cog_report.hypotheses,
282
+ "analogies": cog_report.analogies,
283
+ "total_derived_facts": cog_report.total_derived_facts,
284
+ "duration_ms": round(cog_report.duration_ms, 2),
285
+ "cognition_commit_id": cog_report.commit_id,
286
+ }
287
+ except Exception as e:
288
+ stats["cognition_stats"] = {"error": str(e)}
289
+
290
+ if not dry_run:
291
+ store.end_batch()
292
+ if palace is not None and hasattr(palace, "close"):
293
+ palace.close()
294
+
295
+ return stats
296
+
297
+
298
+ # ---------- Audit / inspection -----------------------------------------
299
+
300
+ def consolidation_report(store, since: str | None = None) -> dict:
301
+ """Return a summary of consolidation activity since `since` (ISO).
302
+
303
+ Counts merged pairs, retired facts, and lists the commit IDs that
304
+ ran consolidation. Useful for the audit / governance dashboard.
305
+ """
306
+ where = "is_active=0 AND provenance LIKE ?"
307
+ pat = '%"consolidated_at"%'
308
+ if since:
309
+ where += " AND tx_to >= ?"
310
+ args = (pat, since)
311
+ else:
312
+ args = (pat,)
313
+ rows = store.conn.execute(
314
+ f"SELECT COUNT(*) FROM facts WHERE {where}", args).fetchone()
315
+ merged = 0
316
+ retired = 0
317
+ for r in store.conn.execute(
318
+ "SELECT provenance FROM facts WHERE " + where, args).fetchall():
319
+ prov = r[0] or ""
320
+ if "merged_into" in prov:
321
+ merged += 1
322
+ if "retired_by_consolidate" in prov:
323
+ retired += 1
324
+ commits = []
325
+ try:
326
+ for r in store.conn.execute(
327
+ "SELECT id, message FROM commits WHERE message LIKE ?",
328
+ ("%consolidate%",)).fetchall():
329
+ commits.append({"id": r[0], "message": r[1]})
330
+ except Exception:
331
+ pass
332
+ return {"merged_facts": merged, "retired_facts": retired,
333
+ "consolidation_commits": len(commits),
334
+ "recent_commits": commits[:10]}
335
+
336
+
337
+ __all__ = ["consolidate", "consolidation_report"]