cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""Deterministic tokenizer & sentence segmentation (μ=0 building block)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
WORD_RE = re.compile(r"[a-zA-Z][a-zA-Z'’-]*|\d+(?:\.\d+)?")
|
|
8
|
+
ABBREV = {"mr", "mrs", "ms", "dr", "prof", "sr", "jr", "st", "vs", "etc",
|
|
9
|
+
"e.g", "i.e", "fig", "inc", "ltd", "co", "u.s", "u.k"}
|
|
10
|
+
|
|
11
|
+
_SENT_SPLIT = re.compile(r"(?<=[.!?])[\"')\]]*\s+|\n{2,}")
|
|
12
|
+
_SENT_GUARD = re.compile(r"^(?:[A-Z0-9\"'(]|…)")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def words(text: str) -> list[str]:
|
|
16
|
+
return [w.lower().replace("’", "'") for w in WORD_RE.findall(text)]
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def sentences(text: str) -> list[tuple[int, int, str]]:
|
|
20
|
+
"""Sentence segmentation with (start, end, sentence) spans."""
|
|
21
|
+
out: list[tuple[int, int, str]] = []
|
|
22
|
+
if not text or not text.strip():
|
|
23
|
+
return out
|
|
24
|
+
pos = 0
|
|
25
|
+
for raw in _SENT_SPLIT.split(text):
|
|
26
|
+
s = raw.strip()
|
|
27
|
+
if not s:
|
|
28
|
+
continue
|
|
29
|
+
start = text.find(s[:24], pos)
|
|
30
|
+
if start < 0:
|
|
31
|
+
start = pos
|
|
32
|
+
end = start + len(s)
|
|
33
|
+
out.append((start, end, s))
|
|
34
|
+
pos = end
|
|
35
|
+
# merge fragments ending in abbreviations ("I met Dr. Chen.")
|
|
36
|
+
merged: list[tuple[int, int, str]] = []
|
|
37
|
+
for item in out:
|
|
38
|
+
if merged:
|
|
39
|
+
ps, pe, ptext = merged[-1]
|
|
40
|
+
tail = re.sub(r"[^\w.]", "", ptext.split()[-1] if ptext.split() else "")
|
|
41
|
+
if tail.rstrip(".").lower() in ABBREV:
|
|
42
|
+
nxt = text[pe:item[1]]
|
|
43
|
+
merged[-1] = (ps, item[1], text[ps:item[1]])
|
|
44
|
+
continue
|
|
45
|
+
merged.append(item)
|
|
46
|
+
return merged
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def strip_punct(s: str) -> str:
|
|
50
|
+
return re.sub(r"[^\w\s'-]", " ", s).strip()
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
STOPWORDS = {
|
|
54
|
+
"a", "an", "the", "and", "or", "but", "if", "of", "to", "in", "on", "at",
|
|
55
|
+
"for", "with", "is", "are", "was", "were", "be", "been", "am", "do",
|
|
56
|
+
"does", "did", "have", "has", "had", "i", "you", "he", "she", "it", "we",
|
|
57
|
+
"they", "me", "my", "your", "his", "her", "its", "our", "their", "this",
|
|
58
|
+
"that", "these", "those", "as", "so", "than", "then", "there", "here",
|
|
59
|
+
"what", "which", "who", "whom", "when", "where", "why", "how", "will",
|
|
60
|
+
"would", "can", "could", "should", "shall", "may", "might", "just",
|
|
61
|
+
"about", "from", "by", "up", "out", "not", "no", "yes", "oh", "well",
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def content_words(text: str) -> list[str]:
|
|
66
|
+
return [w for w in words(text) if w not in STOPWORDS and len(w) > 1]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def cap_sequences(text: str) -> list[str]:
|
|
70
|
+
"""Capitalized multi-word sequences — poor-man's NER (μ=0)."""
|
|
71
|
+
seqs = re.findall(
|
|
72
|
+
r"\b([A-Z][a-zA-Z'&-]*(?:[ ](?:of|the|and|de|van|for)[ ])?[ ]*[A-Z][a-zA-Z'&-]*)+\b",
|
|
73
|
+
text)
|
|
74
|
+
out = []
|
|
75
|
+
for s in seqs:
|
|
76
|
+
s = " ".join(s.split())
|
|
77
|
+
if len(s) > 2 and not s.islower() and "." not in s:
|
|
78
|
+
out.append(s)
|
|
79
|
+
return out
|
|
File without changes
|
|
@@ -0,0 +1,277 @@
|
|
|
1
|
+
"""Sidecar blob arena — Aeon-inspired off-graph text storage.
|
|
2
|
+
|
|
3
|
+
arXiv:2601.15311 (Aeon): large text is stored off-graph in an
|
|
4
|
+
append-only mmap-backed blob file with generational GC. Graph nodes
|
|
5
|
+
hold only a 64-byte preview + offset.
|
|
6
|
+
|
|
7
|
+
Context-M's chunks table currently stores full text inline
|
|
8
|
+
(`text TEXT NOT NULL`). For typical personas this is fine — chunks
|
|
9
|
+
are 200-400 bytes. But for use cases that ingest long documents
|
|
10
|
+
(research papers, meeting transcripts, issue threads), the chunks
|
|
11
|
+
table balloons and the working set of the SQLite page cache gets
|
|
12
|
+
polluted by long text the graph rarely needs.
|
|
13
|
+
|
|
14
|
+
This module provides an OPT-IN sidecar blob arena:
|
|
15
|
+
|
|
16
|
+
BlobArena(path) — opens/creates an mmap'd blob file at `path`
|
|
17
|
+
.put(text) -> (blob_id, offset, length)
|
|
18
|
+
.get(offset, length) -> bytes
|
|
19
|
+
.preview(text, n=64) -> first n bytes (the in-graph preview)
|
|
20
|
+
|
|
21
|
+
The host migrates chunks by:
|
|
22
|
+
1. creating a BlobArena
|
|
23
|
+
2. for each chunk: arena.put(text) -> (blob_id, offset, len)
|
|
24
|
+
3. updating the chunks row: text = preview, blob_offset = offset,
|
|
25
|
+
blob_len = len (schema migration adds the columns)
|
|
26
|
+
4. on retrieval: chunks.text gives the preview (fast); full text
|
|
27
|
+
is fetched via arena.get(offset, len) only when the audit
|
|
28
|
+
/ retrieval path actually needs it
|
|
29
|
+
|
|
30
|
+
Generational GC is out of scope for v1 (chunks are append-mostly in
|
|
31
|
+
practice — retired facts are deactivated but the chunk text is kept
|
|
32
|
+
for audit). The arena is a single mmap'd file; concurrent writers
|
|
33
|
+
must hold a lock (the arena serializes via flock).
|
|
34
|
+
|
|
35
|
+
BACKWARD COMPATIBILITY: the chunks schema migration is OPT-IN. The
|
|
36
|
+
default Memory() path still stores text inline. To enable the
|
|
37
|
+
sidecar, call Memory.enable_blob_arena(path) after construction.
|
|
38
|
+
"""
|
|
39
|
+
from __future__ import annotations
|
|
40
|
+
|
|
41
|
+
import io
|
|
42
|
+
import mmap
|
|
43
|
+
import os
|
|
44
|
+
import threading
|
|
45
|
+
import zlib
|
|
46
|
+
from pathlib import Path
|
|
47
|
+
from typing import Iterator
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
# Schema additions for the chunks table (applied by TraceStore when
|
|
51
|
+
# the arena is enabled). The columns are nullable so existing rows
|
|
52
|
+
# (with inline text) keep working — a NULL blob_offset means "use the
|
|
53
|
+
# inline text column".
|
|
54
|
+
SCHEMA_MIGRATION = """
|
|
55
|
+
ALTER TABLE chunks ADD COLUMN blob_offset INTEGER DEFAULT NULL;
|
|
56
|
+
ALTER TABLE chunks ADD COLUMN blob_len INTEGER DEFAULT NULL;
|
|
57
|
+
ALTER TABLE chunks ADD COLUMN blob_compressed INTEGER DEFAULT 0;
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class BlobArena:
|
|
62
|
+
"""Append-only mmap-backed blob file.
|
|
63
|
+
|
|
64
|
+
Layout:
|
|
65
|
+
- 8-byte header: magic + version + length
|
|
66
|
+
- records: [8-byte length | length bytes of payload]
|
|
67
|
+
(payload may be zlib-compressed if compressed=1 in the chunk
|
|
68
|
+
row; the arena itself is format-agnostic to the bytes)
|
|
69
|
+
|
|
70
|
+
The arena serializes writes via a process-local threading.Lock
|
|
71
|
+
AND an flock on the file for cross-process safety. Concurrent
|
|
72
|
+
readers don't need the lock — mmap pages are CoW.
|
|
73
|
+
"""
|
|
74
|
+
|
|
75
|
+
MAGIC = b"BLB1" # Blob Layer v1
|
|
76
|
+
HEADER_LEN = 16
|
|
77
|
+
|
|
78
|
+
def __init__(self, path: str | os.PathLike) -> None:
|
|
79
|
+
self.path = Path(path)
|
|
80
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
81
|
+
# create or open
|
|
82
|
+
is_new = not self.path.exists()
|
|
83
|
+
if is_new:
|
|
84
|
+
with open(self.path, "wb") as f:
|
|
85
|
+
f.write(self.MAGIC + b"\x00" * 12) # magic + 12 bytes pad
|
|
86
|
+
self._fd = open(self.path, "r+b")
|
|
87
|
+
# write-lock for cross-process safety
|
|
88
|
+
try:
|
|
89
|
+
import fcntl
|
|
90
|
+
fcntl.flock(self._fd.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
|
|
91
|
+
except (ImportError, OSError):
|
|
92
|
+
pass # Windows or already locked by same process
|
|
93
|
+
self._mmap: mmap.mmap | None = None
|
|
94
|
+
self._lock = threading.Lock()
|
|
95
|
+
self._open_mmap()
|
|
96
|
+
|
|
97
|
+
def _open_mmap(self) -> None:
|
|
98
|
+
if self._mmap is None:
|
|
99
|
+
try:
|
|
100
|
+
self._mmap = mmap.mmap(
|
|
101
|
+
self._fd.fileno(), 0, access=mmap.ACCESS_WRITE)
|
|
102
|
+
except (ValueError, OSError):
|
|
103
|
+
# empty file (just header) — fall back to file I/O
|
|
104
|
+
self._mmap = None
|
|
105
|
+
|
|
106
|
+
def put(self, data: bytes, compress: bool = True) -> tuple[int, int, int, bool]:
|
|
107
|
+
"""Append `data` to the blob file.
|
|
108
|
+
|
|
109
|
+
Returns (blob_id, offset, length, was_compressed).
|
|
110
|
+
blob_id — monotonically increasing int (records / appended)
|
|
111
|
+
offset — byte offset in the file where the record starts
|
|
112
|
+
length — payload length in bytes
|
|
113
|
+
was_compressed — whether zlib compression was applied
|
|
114
|
+
|
|
115
|
+
Compression: enabled by default for text payloads > 256 bytes;
|
|
116
|
+
small payloads are stored raw because the zlib header (5+ bytes)
|
|
117
|
+
makes small blobs larger.
|
|
118
|
+
"""
|
|
119
|
+
if isinstance(data, str):
|
|
120
|
+
data = data.encode("utf-8")
|
|
121
|
+
if compress and len(data) > 256:
|
|
122
|
+
compressed = zlib.compress(data, level=6)
|
|
123
|
+
if len(compressed) < len(data) - 8: # only if meaningful savings
|
|
124
|
+
payload = compressed
|
|
125
|
+
was_compressed = True
|
|
126
|
+
else:
|
|
127
|
+
payload = data
|
|
128
|
+
was_compressed = False
|
|
129
|
+
else:
|
|
130
|
+
payload = data
|
|
131
|
+
was_compressed = False
|
|
132
|
+
|
|
133
|
+
with self._lock:
|
|
134
|
+
# we need to write 8 bytes (length) + len(payload) bytes
|
|
135
|
+
rec_len = len(payload)
|
|
136
|
+
# current file size = where this record starts
|
|
137
|
+
self._fd.seek(0, io.SEEK_END)
|
|
138
|
+
offset = self._fd.tell()
|
|
139
|
+
# write 8-byte length + payload
|
|
140
|
+
self._fd.write(rec_len.to_bytes(8, "little"))
|
|
141
|
+
self._fd.write(payload)
|
|
142
|
+
self._fd.flush()
|
|
143
|
+
# invalidate the mmap view so the next get() re-reads
|
|
144
|
+
if self._mmap is not None:
|
|
145
|
+
try:
|
|
146
|
+
self._mmap.close()
|
|
147
|
+
except Exception:
|
|
148
|
+
pass
|
|
149
|
+
self._mmap = None
|
|
150
|
+
self._open_mmap()
|
|
151
|
+
blob_id = offset # offset IS the id (unique within file)
|
|
152
|
+
return blob_id, offset, rec_len, was_compressed
|
|
153
|
+
|
|
154
|
+
def get(self, offset: int, length: int,
|
|
155
|
+
was_compressed: bool = False) -> bytes:
|
|
156
|
+
"""Read a record from the blob file.
|
|
157
|
+
|
|
158
|
+
`offset` and `length` come from the chunks table. If
|
|
159
|
+
`was_compressed` is True (stored in the chunk row), the
|
|
160
|
+
returned bytes are zlib-decompressed before return.
|
|
161
|
+
"""
|
|
162
|
+
with self._lock:
|
|
163
|
+
self._fd.seek(offset)
|
|
164
|
+
raw_len = self._fd.read(8)
|
|
165
|
+
if len(raw_len) != 8:
|
|
166
|
+
raise IOError(f"short read at offset {offset}")
|
|
167
|
+
rec_len = int.from_bytes(raw_len, "little")
|
|
168
|
+
payload = self._fd.read(rec_len)
|
|
169
|
+
if len(payload) != rec_len:
|
|
170
|
+
raise IOError(
|
|
171
|
+
f"short payload read at offset {offset}: "
|
|
172
|
+
f"expected {rec_len}, got {len(payload)}")
|
|
173
|
+
if was_compressed:
|
|
174
|
+
try:
|
|
175
|
+
return zlib.decompress(payload)
|
|
176
|
+
except zlib.error:
|
|
177
|
+
return payload # return raw on decompression failure
|
|
178
|
+
return payload
|
|
179
|
+
|
|
180
|
+
def get_text(self, offset: int, length: int,
|
|
181
|
+
was_compressed: bool = False) -> str:
|
|
182
|
+
"""Convenience: get bytes and decode as UTF-8."""
|
|
183
|
+
return self.get(offset, length, was_compressed).decode(
|
|
184
|
+
"utf-8", errors="replace")
|
|
185
|
+
|
|
186
|
+
def close(self) -> None:
|
|
187
|
+
with self._lock:
|
|
188
|
+
if self._mmap is not None:
|
|
189
|
+
try:
|
|
190
|
+
self._mmap.close()
|
|
191
|
+
except Exception:
|
|
192
|
+
pass
|
|
193
|
+
self._mmap = None
|
|
194
|
+
try:
|
|
195
|
+
self._fd.close()
|
|
196
|
+
except Exception:
|
|
197
|
+
pass
|
|
198
|
+
|
|
199
|
+
def __enter__(self) -> "BlobArena":
|
|
200
|
+
return self
|
|
201
|
+
|
|
202
|
+
def __exit__(self, *exc) -> None:
|
|
203
|
+
self.close()
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
# ---------- Migration helper ------------------------------------------
|
|
207
|
+
|
|
208
|
+
def migrate_chunks_to_arena(store, arena: BlobArena,
|
|
209
|
+
batch_size: int = 1000) -> dict:
|
|
210
|
+
"""Migrate existing chunks.text rows into the sidecar arena.
|
|
211
|
+
|
|
212
|
+
For each chunk:
|
|
213
|
+
- put text into arena -> (blob_id, offset, len, compressed)
|
|
214
|
+
- update the row: blob_offset=offset, blob_len=len,
|
|
215
|
+
blob_compressed=compressed, text=preview (first 64 bytes)
|
|
216
|
+
The original text column is REPLACED with the preview — the full
|
|
217
|
+
text lives only in the arena. Existing readers that use chunks.text
|
|
218
|
+
will see the preview (good for display); readers that need full
|
|
219
|
+
text must call arena.get_text(offset, len, compressed).
|
|
220
|
+
|
|
221
|
+
Idempotent: chunks with blob_offset IS NOT NULL are skipped.
|
|
222
|
+
"""
|
|
223
|
+
# ensure the schema migration has run
|
|
224
|
+
cols = {r[1] for r in store.conn.execute("PRAGMA table_info(chunks)").fetchall()}
|
|
225
|
+
if "blob_offset" not in cols:
|
|
226
|
+
for stmt in SCHEMA_MIGRATION.strip().split(";"):
|
|
227
|
+
stmt = stmt.strip()
|
|
228
|
+
if stmt:
|
|
229
|
+
store.conn.execute(stmt)
|
|
230
|
+
store.conn.commit()
|
|
231
|
+
|
|
232
|
+
migrated = 0
|
|
233
|
+
skipped = 0
|
|
234
|
+
while True:
|
|
235
|
+
rows = store.conn.execute(
|
|
236
|
+
"SELECT id, text FROM chunks WHERE blob_offset IS NULL "
|
|
237
|
+
f"LIMIT {batch_size}").fetchall()
|
|
238
|
+
if not rows:
|
|
239
|
+
break
|
|
240
|
+
for chunk_id, text in rows:
|
|
241
|
+
if not text:
|
|
242
|
+
skipped += 1
|
|
243
|
+
continue
|
|
244
|
+
data = text.encode("utf-8")
|
|
245
|
+
_, offset, length, compressed = arena.put(data, compress=True)
|
|
246
|
+
preview = text[:64] + ("..." if len(text) > 64 else "")
|
|
247
|
+
store.conn.execute(
|
|
248
|
+
"UPDATE chunks SET text=?, blob_offset=?, blob_len=?, "
|
|
249
|
+
"blob_compressed=? WHERE id=?",
|
|
250
|
+
(preview, offset, length, 1 if compressed else 0, chunk_id))
|
|
251
|
+
migrated += 1
|
|
252
|
+
store.conn.commit()
|
|
253
|
+
return {"migrated": migrated, "skipped": skipped}
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def get_chunk_text(store, arena: BlobArena, chunk_id: str) -> str:
|
|
257
|
+
"""Fetch full text for a chunk — from the arena if blob_offset is
|
|
258
|
+
set, otherwise fall back to inline text.
|
|
259
|
+
|
|
260
|
+
Use this in the audit / retrieval path when the 64-byte preview
|
|
261
|
+
in chunks.text isn't enough and you need the full source.
|
|
262
|
+
"""
|
|
263
|
+
row = store.conn.execute(
|
|
264
|
+
"SELECT text, blob_offset, blob_len, blob_compressed "
|
|
265
|
+
"FROM chunks WHERE id=?", (chunk_id,)).fetchone()
|
|
266
|
+
if not row:
|
|
267
|
+
return ""
|
|
268
|
+
text, offset, length, compressed = row
|
|
269
|
+
if offset is None:
|
|
270
|
+
return text or ""
|
|
271
|
+
return arena.get_text(offset, length, bool(compressed))
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
__all__ = [
|
|
275
|
+
"BlobArena", "SCHEMA_MIGRATION",
|
|
276
|
+
"migrate_chunks_to_arena", "get_chunk_text",
|
|
277
|
+
]
|
|
@@ -0,0 +1,337 @@
|
|
|
1
|
+
"""Dreaming / consolidation — Aeon-inspired idle-time optimization.
|
|
2
|
+
|
|
3
|
+
arXiv:2601.15311 (Aeon) describes a background task analogous to
|
|
4
|
+
biological sleep: defragmentation, GC, and consolidation of the
|
|
5
|
+
verbose Trace into compressed long-term episodic summaries.
|
|
6
|
+
|
|
7
|
+
Context-M's Memory Git is currently passive — it version-controls
|
|
8
|
+
facts but doesn't optimize them. This module turns it into an ACTIVE
|
|
9
|
+
memory optimizer that runs during idle periods:
|
|
10
|
+
|
|
11
|
+
consolidate(store, palace, prefetcher, ...) ->
|
|
12
|
+
- merges redundant triples in the Trace (same subject/relation,
|
|
13
|
+
near-duplicate values) via MERGED_WITH edges
|
|
14
|
+
- retires facts past their valid_to + grace period
|
|
15
|
+
- defragments the palace by rebuilding the in-memory packed matrix
|
|
16
|
+
from active facts only (drops retired/merged IDs)
|
|
17
|
+
- re-trains the MBTB prefetcher from recent query access patterns
|
|
18
|
+
|
|
19
|
+
The pass is IDEMPOTENT and SAFE — every change is a commit on the
|
|
20
|
+
current branch, every retired fact is still queryable via
|
|
21
|
+
allow_inactive=True, every merged fact keeps a MERGED_WITH edge so
|
|
22
|
+
the original triple is recoverable for audit. A failed pass rolls back
|
|
23
|
+
via the existing Memory Git ancestry.
|
|
24
|
+
|
|
25
|
+
NOTE: this is NOT a separate daemon — it's a function the host app
|
|
26
|
+
calls when idle (e.g.overnight cron, `cortexm consolidate` CLI, or
|
|
27
|
+
an `on_idle` hook in the MCP server).
|
|
28
|
+
"""
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
import datetime as _dt
|
|
32
|
+
from datetime import datetime, timezone
|
|
33
|
+
from typing import Iterable
|
|
34
|
+
|
|
35
|
+
from cortexm.trace.edges import MERGED_WITH, RETRACTED_BY
|
|
36
|
+
from cortexm.util import iso, similarity
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _now() -> datetime:
|
|
40
|
+
return datetime.now(timezone.utc)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def consolidate(store, palace=None, prefetcher=None, *,
|
|
44
|
+
user_id: str | None = None,
|
|
45
|
+
merge_threshold: float = 0.92,
|
|
46
|
+
retire_grace_days: int = 365,
|
|
47
|
+
defrag_palace: bool = True,
|
|
48
|
+
retrain_prefetcher: bool = True,
|
|
49
|
+
run_fade: bool = True,
|
|
50
|
+
run_tmt: bool = False,
|
|
51
|
+
run_cognition: bool = False,
|
|
52
|
+
dry_run: bool = False,
|
|
53
|
+
fade_cfg: dict | None = None,
|
|
54
|
+
tmt_cfg: dict | None = None,
|
|
55
|
+
cognition_cfg: dict | None = None) -> dict:
|
|
56
|
+
"""Run a single consolidation pass.
|
|
57
|
+
|
|
58
|
+
Returns a stats dict:
|
|
59
|
+
{merged_pairs, retired_facts, palace_defragged,
|
|
60
|
+
prefetcher_retrained, fade_stats, tmt_stats,
|
|
61
|
+
cognition_stats, commit_id, dry_run}
|
|
62
|
+
|
|
63
|
+
Parameters:
|
|
64
|
+
store — TraceStore
|
|
65
|
+
palace — MemoryPalace (optional; only needed for defrag)
|
|
66
|
+
prefetcher — Prefetcher (optional; only needed for retrain)
|
|
67
|
+
user_id — restrict pass to one user (None = all users)
|
|
68
|
+
merge_threshold — jaccard similarity above which two facts
|
|
69
|
+
(same subject+relation) are merged
|
|
70
|
+
retire_grace_days — facts with valid_to older than this many
|
|
71
|
+
days are retired (deactivated)
|
|
72
|
+
defrag_palace — rebuild the palace packed matrix from active facts
|
|
73
|
+
retrain_prefetcher — rebuild the MBTB from access_count stats
|
|
74
|
+
run_fade — also run FadeMem sweep (decay + deactivate + merge)
|
|
75
|
+
run_tmt — also run TiMem TMT hierarchy build
|
|
76
|
+
run_cognition — also run the HMS-style Cognition Engine pass
|
|
77
|
+
(PatternScanner + AbstractionEngine + GapDetector
|
|
78
|
+
+ HypothesisEngine + AnalogyDetector). Emits
|
|
79
|
+
HYPOTHESIZED_BY edges with confidence < 0.5 —
|
|
80
|
+
never active in retrieval unless promoted.
|
|
81
|
+
fade_cfg — kwargs for fade_sweep (lambda_, thresholds, etc.)
|
|
82
|
+
tmt_cfg — kwargs for tmt_build (cluster mins, etc.)
|
|
83
|
+
cognition_cfg — kwargs for run_cognition_pass
|
|
84
|
+
dry_run — compute the changes but don't apply them
|
|
85
|
+
"""
|
|
86
|
+
stats = {
|
|
87
|
+
"merged_pairs": 0,
|
|
88
|
+
"retired_facts": 0,
|
|
89
|
+
"palace_defragged": False,
|
|
90
|
+
"prefetcher_retrained": False,
|
|
91
|
+
"fade_stats": None,
|
|
92
|
+
"tmt_stats": None,
|
|
93
|
+
"cognition_stats": None,
|
|
94
|
+
"commit_id": None,
|
|
95
|
+
"dry_run": dry_run,
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
# ---------- 1. Merge redundant triples -------------------------------
|
|
99
|
+
# Group active facts by (user_id, subject, relation) and find
|
|
100
|
+
# near-duplicate values within each group.
|
|
101
|
+
where = "is_active=1 AND quarantined=0"
|
|
102
|
+
args: tuple = ()
|
|
103
|
+
if user_id is not None:
|
|
104
|
+
where += " AND user_id=?"
|
|
105
|
+
args = (user_id,)
|
|
106
|
+
rows = store.conn.execute(
|
|
107
|
+
f"SELECT id, subject, relation, value, user_id, confidence, "
|
|
108
|
+
f"access_count FROM facts WHERE {where} ORDER BY user_id, subject, "
|
|
109
|
+
f"relation, valid_from", args).fetchall()
|
|
110
|
+
groups: dict[tuple, list[dict]] = {}
|
|
111
|
+
for r in rows:
|
|
112
|
+
r = dict(r)
|
|
113
|
+
key = (r["user_id"], r["subject"], r["relation"])
|
|
114
|
+
groups.setdefault(key, []).append(r)
|
|
115
|
+
|
|
116
|
+
merge_pairs: list[tuple[str, str, float]] = []
|
|
117
|
+
for key, group in groups.items():
|
|
118
|
+
if len(group) < 2:
|
|
119
|
+
continue
|
|
120
|
+
# pairwise similarity (small groups, so O(n^2) is fine)
|
|
121
|
+
for i, a in enumerate(group):
|
|
122
|
+
for b in group[i + 1:]:
|
|
123
|
+
sim = similarity(a["value"], b["value"])
|
|
124
|
+
if sim >= merge_threshold:
|
|
125
|
+
# keep the higher-confidence one; merge the other into it
|
|
126
|
+
if a["confidence"] >= b["confidence"]:
|
|
127
|
+
keep, drop = a, b
|
|
128
|
+
else:
|
|
129
|
+
keep, drop = b, a
|
|
130
|
+
merge_pairs.append((keep["id"], drop["id"], sim))
|
|
131
|
+
|
|
132
|
+
if not dry_run:
|
|
133
|
+
store.begin_batch()
|
|
134
|
+
commit = store.create_commit(
|
|
135
|
+
f"consolidate: merge {len(merge_pairs)} pairs, "
|
|
136
|
+
f"retire stale", n_facts=0)
|
|
137
|
+
stats["commit_id"] = commit
|
|
138
|
+
|
|
139
|
+
# apply merges
|
|
140
|
+
for keep_id, drop_id, sim in merge_pairs:
|
|
141
|
+
# deactivate the drop, wire MERGED_WITH edge so audits can
|
|
142
|
+
# always recover the original triple
|
|
143
|
+
store.update_fact(
|
|
144
|
+
drop_id, is_active=0, tx_to=iso(_now()),
|
|
145
|
+
retired_commit=commit,
|
|
146
|
+
provenance={"merged_into": keep_id,
|
|
147
|
+
"merge_sim": round(sim, 4),
|
|
148
|
+
"consolidated_at": iso(_now())})
|
|
149
|
+
store.add_edge(keep_id, drop_id, MERGED_WITH,
|
|
150
|
+
{"sim": round(sim, 4),
|
|
151
|
+
"consolidated_at": iso(_now())})
|
|
152
|
+
|
|
153
|
+
stats["merged_pairs"] = len(merge_pairs)
|
|
154
|
+
|
|
155
|
+
# ---------- 2. Retire stale facts -----------------------------------
|
|
156
|
+
# Facts whose valid_to is older than retire_grace_days and which
|
|
157
|
+
# haven't been accessed recently (access_count == 0) — these are
|
|
158
|
+
# safe to retire; the bi-temporal model still serves them via
|
|
159
|
+
# allow_inactive=True.
|
|
160
|
+
cutoff = (_now() - _dt.timedelta(days=retire_grace_days)).strftime(
|
|
161
|
+
"%Y-%m-%d")
|
|
162
|
+
retire_where = (
|
|
163
|
+
f"is_active=1 AND valid_to IS NOT NULL AND valid_to < ? "
|
|
164
|
+
f"AND valid_to != '' AND access_count = 0")
|
|
165
|
+
retire_args: tuple = (cutoff,)
|
|
166
|
+
if user_id is not None:
|
|
167
|
+
retire_where += " AND user_id=?"
|
|
168
|
+
retire_args = (cutoff, user_id)
|
|
169
|
+
retire_rows = store.conn.execute(
|
|
170
|
+
f"SELECT id FROM facts WHERE {retire_where}", retire_args).fetchall()
|
|
171
|
+
retire_ids = [r[0] for r in retire_rows]
|
|
172
|
+
|
|
173
|
+
if not dry_run:
|
|
174
|
+
for fid in retire_ids:
|
|
175
|
+
store.update_fact(
|
|
176
|
+
fid, is_active=0, tx_to=iso(_now()),
|
|
177
|
+
retired_commit=commit,
|
|
178
|
+
provenance={"retired_by_consolidate": True,
|
|
179
|
+
"consolidated_at": iso(_now())})
|
|
180
|
+
stats["retired_facts"] = len(retire_ids)
|
|
181
|
+
|
|
182
|
+
# ---------- 3. Palace defrag ----------------------------------------
|
|
183
|
+
# Rebuild the palace's packed matrix from active facts only.
|
|
184
|
+
# This drops retired / merged IDs from the in-memory index and
|
|
185
|
+
# re-tightens the page-clustered tree (currently a no-op for the
|
|
186
|
+
# SQLite-backed palace; for the in-memory palace it compacts).
|
|
187
|
+
if defrag_palace and palace is not None and not dry_run:
|
|
188
|
+
try:
|
|
189
|
+
# the palace's _n tracks active entries — a defrag pass
|
|
190
|
+
# re-builds the matrix by re-adding only active fact IDs.
|
|
191
|
+
# For the SQLite-backed palace this is a no-op (vectors
|
|
192
|
+
# are stored in a BLOB, not a packed matrix), but the
|
|
193
|
+
# call still flushes any in-memory dirty state.
|
|
194
|
+
if hasattr(palace, "defrag"):
|
|
195
|
+
palace.defrag()
|
|
196
|
+
elif hasattr(palace, "close"):
|
|
197
|
+
palace.close()
|
|
198
|
+
stats["palace_defragged"] = True
|
|
199
|
+
except Exception:
|
|
200
|
+
pass
|
|
201
|
+
|
|
202
|
+
# ---------- 4. Prefetcher retrain ------------------------------------
|
|
203
|
+
# The MBTB prefetcher tracks co-access patterns. Re-training from
|
|
204
|
+
# access_count stats lets it pick up shifts in user behavior.
|
|
205
|
+
if retrain_prefetcher and prefetcher is not None and not dry_run:
|
|
206
|
+
try:
|
|
207
|
+
if hasattr(prefetcher, "retrain"):
|
|
208
|
+
prefetcher.retrain(store)
|
|
209
|
+
elif hasattr(prefetcher, "rebuild"):
|
|
210
|
+
prefetcher.rebuild(store)
|
|
211
|
+
stats["prefetcher_retrained"] = True
|
|
212
|
+
except Exception:
|
|
213
|
+
pass
|
|
214
|
+
|
|
215
|
+
# ---------- 5. FadeMem sweep (decay + deactivate + cluster merge) ----
|
|
216
|
+
# Biologically-inspired forgetting: exponential decay on retention
|
|
217
|
+
# scores, with access-driven reconsolidation (frequently-retrieved
|
|
218
|
+
# facts resist decay). Deactivates facts whose retention_score drops
|
|
219
|
+
# below fade_deactivate_threshold; merges clusters of low-retention
|
|
220
|
+
# siblings. Bi-temporal safe: only is_active flips.
|
|
221
|
+
if run_fade and not dry_run:
|
|
222
|
+
try:
|
|
223
|
+
from cortexm.trace.fade import fade_sweep
|
|
224
|
+
fade_kwargs = dict(
|
|
225
|
+
lambda_=0.05,
|
|
226
|
+
access_boost=0.5,
|
|
227
|
+
contradiction_penalty=0.25,
|
|
228
|
+
deactivate_threshold=0.10,
|
|
229
|
+
merge_threshold=0.30,
|
|
230
|
+
merge_similarity=merge_threshold,
|
|
231
|
+
user_id=user_id,
|
|
232
|
+
dry_run=dry_run,
|
|
233
|
+
)
|
|
234
|
+
if fade_cfg:
|
|
235
|
+
fade_kwargs.update(fade_cfg)
|
|
236
|
+
stats["fade_stats"] = fade_sweep(store, palace, **fade_kwargs)
|
|
237
|
+
except Exception as e:
|
|
238
|
+
stats["fade_stats"] = {"error": str(e)}
|
|
239
|
+
|
|
240
|
+
# ---------- 6. TiMem TMT hierarchy build -----------------------------
|
|
241
|
+
# Episodic → session → day → persona abstraction. Each higher level
|
|
242
|
+
# is a derived fact with DERIVED_FROM edges back to its constituents.
|
|
243
|
+
# Retrieval can short-circuit to the appropriate level based on
|
|
244
|
+
# query complexity.
|
|
245
|
+
if run_tmt and not dry_run:
|
|
246
|
+
try:
|
|
247
|
+
from cortexm.trace.tmt import tmt_build
|
|
248
|
+
tmt_kwargs = dict(
|
|
249
|
+
session_cluster_mins=5,
|
|
250
|
+
persona_min_sessions=3,
|
|
251
|
+
user_id=user_id,
|
|
252
|
+
)
|
|
253
|
+
if tmt_cfg:
|
|
254
|
+
tmt_kwargs.update(tmt_cfg)
|
|
255
|
+
stats["tmt_stats"] = tmt_build(store, palace, **tmt_kwargs)
|
|
256
|
+
except Exception as e:
|
|
257
|
+
stats["tmt_stats"] = {"error": str(e)}
|
|
258
|
+
|
|
259
|
+
# ---------- 7. HMS Cognition Engine pass ----------------------------
|
|
260
|
+
# PatternScanner + AbstractionEngine + GapDetector + HypothesisEngine
|
|
261
|
+
# + AnalogyDetector. Surfaces structural regularities, builds
|
|
262
|
+
# prototype categories, fills in missing relations via hypotheses,
|
|
263
|
+
# finds cross-domain analogies. Output is derived facts with
|
|
264
|
+
# confidence < 0.5 and is_derived=1, never promoted to active
|
|
265
|
+
# retrieval unless explicitly confirmed by user input.
|
|
266
|
+
if run_cognition:
|
|
267
|
+
try:
|
|
268
|
+
from cortexm.cognition import run_cognition_pass
|
|
269
|
+
cog_kwargs = dict(
|
|
270
|
+
dry_run=dry_run,
|
|
271
|
+
user_id=user_id,
|
|
272
|
+
)
|
|
273
|
+
if cognition_cfg:
|
|
274
|
+
cog_kwargs.update(cognition_cfg)
|
|
275
|
+
cog_report = run_cognition_pass(store, palace=palace,
|
|
276
|
+
**cog_kwargs)
|
|
277
|
+
stats["cognition_stats"] = {
|
|
278
|
+
"scan": cog_report.scan,
|
|
279
|
+
"abstraction": cog_report.abstraction,
|
|
280
|
+
"gaps": cog_report.gaps,
|
|
281
|
+
"hypotheses": cog_report.hypotheses,
|
|
282
|
+
"analogies": cog_report.analogies,
|
|
283
|
+
"total_derived_facts": cog_report.total_derived_facts,
|
|
284
|
+
"duration_ms": round(cog_report.duration_ms, 2),
|
|
285
|
+
"cognition_commit_id": cog_report.commit_id,
|
|
286
|
+
}
|
|
287
|
+
except Exception as e:
|
|
288
|
+
stats["cognition_stats"] = {"error": str(e)}
|
|
289
|
+
|
|
290
|
+
if not dry_run:
|
|
291
|
+
store.end_batch()
|
|
292
|
+
if palace is not None and hasattr(palace, "close"):
|
|
293
|
+
palace.close()
|
|
294
|
+
|
|
295
|
+
return stats
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
# ---------- Audit / inspection -----------------------------------------
|
|
299
|
+
|
|
300
|
+
def consolidation_report(store, since: str | None = None) -> dict:
|
|
301
|
+
"""Return a summary of consolidation activity since `since` (ISO).
|
|
302
|
+
|
|
303
|
+
Counts merged pairs, retired facts, and lists the commit IDs that
|
|
304
|
+
ran consolidation. Useful for the audit / governance dashboard.
|
|
305
|
+
"""
|
|
306
|
+
where = "is_active=0 AND provenance LIKE ?"
|
|
307
|
+
pat = '%"consolidated_at"%'
|
|
308
|
+
if since:
|
|
309
|
+
where += " AND tx_to >= ?"
|
|
310
|
+
args = (pat, since)
|
|
311
|
+
else:
|
|
312
|
+
args = (pat,)
|
|
313
|
+
rows = store.conn.execute(
|
|
314
|
+
f"SELECT COUNT(*) FROM facts WHERE {where}", args).fetchone()
|
|
315
|
+
merged = 0
|
|
316
|
+
retired = 0
|
|
317
|
+
for r in store.conn.execute(
|
|
318
|
+
"SELECT provenance FROM facts WHERE " + where, args).fetchall():
|
|
319
|
+
prov = r[0] or ""
|
|
320
|
+
if "merged_into" in prov:
|
|
321
|
+
merged += 1
|
|
322
|
+
if "retired_by_consolidate" in prov:
|
|
323
|
+
retired += 1
|
|
324
|
+
commits = []
|
|
325
|
+
try:
|
|
326
|
+
for r in store.conn.execute(
|
|
327
|
+
"SELECT id, message FROM commits WHERE message LIKE ?",
|
|
328
|
+
("%consolidate%",)).fetchall():
|
|
329
|
+
commits.append({"id": r[0], "message": r[1]})
|
|
330
|
+
except Exception:
|
|
331
|
+
pass
|
|
332
|
+
return {"merged_facts": merged, "retired_facts": retired,
|
|
333
|
+
"consolidation_commits": len(commits),
|
|
334
|
+
"recent_commits": commits[:10]}
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
__all__ = ["consolidate", "consolidation_report"]
|