cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
cortexm/cli.py
ADDED
|
@@ -0,0 +1,295 @@
|
|
|
1
|
+
"""cortexm — the Context-M command line.
|
|
2
|
+
|
|
3
|
+
cortexm serve # MCP server (stdio JSON-RPC)
|
|
4
|
+
cortexm stats [--db PATH]
|
|
5
|
+
cortexm verify [--db PATH] # integrity audit (hashes + vectors)
|
|
6
|
+
cortexm consolidate [--db PATH]
|
|
7
|
+
cortexm migrate --from mem0 --path mem0.db [--db PATH]
|
|
8
|
+
cortexm cost --memories 1000000 # μ=0 cost calculator
|
|
9
|
+
cortexm bench --buckets 128k,1m # BEAM-style benchmark
|
|
10
|
+
cortexm export-schema [--db PATH]
|
|
11
|
+
cortexm git log|branches|diff|blame
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import argparse
|
|
17
|
+
import json
|
|
18
|
+
import sys
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _memory(args):
|
|
22
|
+
from cortexm.api.memory import Memory
|
|
23
|
+
from cortexm.config import Config
|
|
24
|
+
cfg = Config.from_env()
|
|
25
|
+
if getattr(args, "db", None):
|
|
26
|
+
cfg.db_path = args.db
|
|
27
|
+
return Memory(cfg)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def main(argv=None) -> int:
|
|
31
|
+
ap = argparse.ArgumentParser(prog="cortexm",
|
|
32
|
+
description="Context-M memory fabric CLI")
|
|
33
|
+
sub = ap.add_subparsers(dest="cmd")
|
|
34
|
+
|
|
35
|
+
p = sub.add_parser("serve", help="run the MCP server on stdio")
|
|
36
|
+
p.add_argument("--db", default=None)
|
|
37
|
+
|
|
38
|
+
p = sub.add_parser("serve-rest", help="run the REST API server (HTTP)")
|
|
39
|
+
p.add_argument("--db", default=":memory:")
|
|
40
|
+
p.add_argument("--host", default="0.0.0.0")
|
|
41
|
+
p.add_argument("--port", type=int, default=8900)
|
|
42
|
+
p.add_argument("--pii", default=None, choices=[None, "off", "redact",
|
|
43
|
+
"block", "tag"])
|
|
44
|
+
p.add_argument("--admin-key", default=None,
|
|
45
|
+
help="mint an admin API key at boot (printed once)")
|
|
46
|
+
p.add_argument("--sparql-port", type=int, default=None,
|
|
47
|
+
help="co-host a SPARQL endpoint on this port "
|
|
48
|
+
"(shares one Memory instance with the REST API)")
|
|
49
|
+
p.add_argument("--sparql-host", default="0.0.0.0",
|
|
50
|
+
help="bind SPARQL endpoint to this host")
|
|
51
|
+
p.add_argument("--sparql-user-id", default=None,
|
|
52
|
+
help="scope SPARQL queries to a single user")
|
|
53
|
+
|
|
54
|
+
for name, help_ in (("stats", "memory statistics"),
|
|
55
|
+
("verify", "integrity audit"),
|
|
56
|
+
("consolidate", "lifecycle consolidation"),
|
|
57
|
+
("export-schema", "federation schema report")):
|
|
58
|
+
p = sub.add_parser(name, help=help_)
|
|
59
|
+
p.add_argument("--db", default=None)
|
|
60
|
+
|
|
61
|
+
# extend consolidate with toggles for each pass
|
|
62
|
+
# (the loop above already created 'consolidate'; we fetch it
|
|
63
|
+
# by name and add extra flags)
|
|
64
|
+
con = [a for a in sub.choices.values() if a.prog.endswith("consolidate")][0]
|
|
65
|
+
con.add_argument("--dry-run", action="store_true",
|
|
66
|
+
help="don't commit changes — just report what would happen")
|
|
67
|
+
con.add_argument("--no-lifecycle", action="store_true",
|
|
68
|
+
help="skip the lifecycle pass (short→long promotion / decay)")
|
|
69
|
+
con.add_argument("--no-dreaming", action="store_true",
|
|
70
|
+
help="skip the Aeon dreaming pass (merge triples, retire stale, "
|
|
71
|
+
"defrag palace, retrain prefetcher)")
|
|
72
|
+
con.add_argument("--user-id", default=None,
|
|
73
|
+
help="scope the dreaming pass to one user")
|
|
74
|
+
|
|
75
|
+
p = sub.add_parser("keys", help="API key management (RBAC)")
|
|
76
|
+
p.add_argument("op", choices=["create", "list", "revoke"])
|
|
77
|
+
p.add_argument("--db", default=None)
|
|
78
|
+
p.add_argument("--role", default="reader",
|
|
79
|
+
choices=["admin", "operator", "reader", "auditor"])
|
|
80
|
+
p.add_argument("--label", default="")
|
|
81
|
+
p.add_argument("--id", default=None, help="key id to revoke")
|
|
82
|
+
|
|
83
|
+
p = sub.add_parser("audit", help="audit log tail / verify / export")
|
|
84
|
+
p.add_argument("op", choices=["tail", "verify", "export"])
|
|
85
|
+
p.add_argument("--db", default=None)
|
|
86
|
+
p.add_argument("-n", type=int, default=30)
|
|
87
|
+
p.add_argument("--out", default="audit.jsonl",
|
|
88
|
+
help="export path (op=export)")
|
|
89
|
+
|
|
90
|
+
p = sub.add_parser("snapshot", help="atomic backup with manifest")
|
|
91
|
+
p.add_argument("--db", default=None)
|
|
92
|
+
p.add_argument("--path", required=True)
|
|
93
|
+
|
|
94
|
+
p = sub.add_parser("erase", help="GDPR right-to-erasure")
|
|
95
|
+
p.add_argument("--db", default=None)
|
|
96
|
+
p.add_argument("--user-id", required=True)
|
|
97
|
+
|
|
98
|
+
p = sub.add_parser("governance", help="governance ops")
|
|
99
|
+
p.add_argument("op", choices=["retention", "state-at"])
|
|
100
|
+
p.add_argument("--db", default=None)
|
|
101
|
+
p.add_argument("--days", type=int, default=365)
|
|
102
|
+
p.add_argument("--when", default=None,
|
|
103
|
+
help="ISO datetime for state-at")
|
|
104
|
+
p.add_argument("--user-id", default=None)
|
|
105
|
+
|
|
106
|
+
p = sub.add_parser("migrate", help="import from mem0/zep/chroma")
|
|
107
|
+
p.add_argument("--from", dest="source", required=True,
|
|
108
|
+
choices=["mem0", "zep", "chroma"])
|
|
109
|
+
p.add_argument("--path", required=True)
|
|
110
|
+
p.add_argument("--db", default=None)
|
|
111
|
+
p.add_argument("--user-id", default="migrated")
|
|
112
|
+
|
|
113
|
+
p = sub.add_parser("cost", help="μ=0 cost calculator")
|
|
114
|
+
p.add_argument("--memories", type=int, default=1_000_000)
|
|
115
|
+
p.add_argument("--ingest-per-memory", type=float, default=0.001,
|
|
116
|
+
help="competitor LLM-extract cost per memory (USD)")
|
|
117
|
+
|
|
118
|
+
p = sub.add_parser("bench", help="BEAM-style benchmark")
|
|
119
|
+
p.add_argument("--buckets", default="128k")
|
|
120
|
+
p.add_argument("--micro", action="store_true")
|
|
121
|
+
|
|
122
|
+
p = sub.add_parser("git", help="Memory Git operations")
|
|
123
|
+
p.add_argument("op", choices=["log", "branches", "diff", "blame"])
|
|
124
|
+
p.add_argument("rest", nargs="*")
|
|
125
|
+
p.add_argument("--db", default=None)
|
|
126
|
+
|
|
127
|
+
args = ap.parse_args(argv)
|
|
128
|
+
if not args.cmd:
|
|
129
|
+
ap.print_help()
|
|
130
|
+
return 0
|
|
131
|
+
|
|
132
|
+
if args.cmd == "serve":
|
|
133
|
+
from cortexm.mcp.server import serve
|
|
134
|
+
serve(args.db)
|
|
135
|
+
return 0
|
|
136
|
+
|
|
137
|
+
if args.cmd == "serve-rest":
|
|
138
|
+
from cortexm.server.rest import main as rest_main
|
|
139
|
+
# argparse passthrough for the REST server
|
|
140
|
+
rest_argv = ["--db", args.db, "--host", args.host,
|
|
141
|
+
"--port", str(args.port)]
|
|
142
|
+
if args.pii:
|
|
143
|
+
rest_argv += ["--pii", args.pii]
|
|
144
|
+
if args.admin_key:
|
|
145
|
+
rest_argv += ["--admin-key", args.admin_key]
|
|
146
|
+
if getattr(args, "sparql_port", None):
|
|
147
|
+
rest_argv += ["--sparql-port", str(args.sparql_port),
|
|
148
|
+
"--sparql-host", args.sparql_host]
|
|
149
|
+
if args.sparql_user_id:
|
|
150
|
+
rest_argv += ["--sparql-user-id", args.sparql_user_id]
|
|
151
|
+
import sys as _sys
|
|
152
|
+
_sys.argv = ["contextm-serve"] + rest_argv
|
|
153
|
+
rest_main()
|
|
154
|
+
return 0
|
|
155
|
+
|
|
156
|
+
if args.cmd == "keys":
|
|
157
|
+
m = _memory(args)
|
|
158
|
+
try:
|
|
159
|
+
if args.op == "create":
|
|
160
|
+
out = m.keys.create(args.role, label=args.label)
|
|
161
|
+
print(json.dumps(out, indent=2))
|
|
162
|
+
elif args.op == "list":
|
|
163
|
+
print(json.dumps({"keys": m.keys.list_keys()}, indent=2))
|
|
164
|
+
elif args.op == "revoke":
|
|
165
|
+
print(json.dumps({"revoked": m.keys.revoke(args.id or "")}))
|
|
166
|
+
finally:
|
|
167
|
+
m.close()
|
|
168
|
+
return 0
|
|
169
|
+
|
|
170
|
+
if args.cmd == "audit":
|
|
171
|
+
m = _memory(args)
|
|
172
|
+
try:
|
|
173
|
+
if args.op == "verify":
|
|
174
|
+
print(json.dumps(m.audit_log.verify(), indent=2))
|
|
175
|
+
elif args.op == "export":
|
|
176
|
+
n = m.audit_log.export_jsonl(args.out)
|
|
177
|
+
print(json.dumps({"exported": n, "path": args.out}))
|
|
178
|
+
else:
|
|
179
|
+
print(json.dumps({"events": m.audit_log.tail(args.n)}, indent=2))
|
|
180
|
+
finally:
|
|
181
|
+
m.close()
|
|
182
|
+
return 0
|
|
183
|
+
|
|
184
|
+
if args.cmd == "snapshot":
|
|
185
|
+
m = _memory(args)
|
|
186
|
+
try:
|
|
187
|
+
print(json.dumps(m.governance.snapshot(args.path), indent=2))
|
|
188
|
+
finally:
|
|
189
|
+
m.close()
|
|
190
|
+
return 0
|
|
191
|
+
|
|
192
|
+
if args.cmd == "erase":
|
|
193
|
+
m = _memory(args)
|
|
194
|
+
try:
|
|
195
|
+
print(json.dumps(m.governance.erase_user(args.user_id), indent=2))
|
|
196
|
+
finally:
|
|
197
|
+
m.close()
|
|
198
|
+
return 0
|
|
199
|
+
|
|
200
|
+
if args.cmd == "governance":
|
|
201
|
+
m = _memory(args)
|
|
202
|
+
try:
|
|
203
|
+
if args.op == "retention":
|
|
204
|
+
out = m.governance.apply_retention(args.days,
|
|
205
|
+
user_id=args.user_id)
|
|
206
|
+
else:
|
|
207
|
+
out = {"facts": m.governance.state_at(args.when or "",
|
|
208
|
+
user_id=args.user_id)}
|
|
209
|
+
print(json.dumps(out, indent=2, default=str))
|
|
210
|
+
finally:
|
|
211
|
+
m.close()
|
|
212
|
+
return 0
|
|
213
|
+
|
|
214
|
+
if args.cmd == "cost":
|
|
215
|
+
print(cost_report(args.memories, args.ingest_per_memory))
|
|
216
|
+
return 0
|
|
217
|
+
|
|
218
|
+
if args.cmd == "bench":
|
|
219
|
+
from cortexm.bench.run import main as bench_main
|
|
220
|
+
argv = ["--buckets", args.buckets]
|
|
221
|
+
if args.micro:
|
|
222
|
+
argv.append("--micro")
|
|
223
|
+
bench_main()
|
|
224
|
+
return 0
|
|
225
|
+
|
|
226
|
+
m = _memory(args)
|
|
227
|
+
try:
|
|
228
|
+
if args.cmd == "stats":
|
|
229
|
+
print(json.dumps(m.stats(), indent=2, default=str))
|
|
230
|
+
elif args.cmd == "verify":
|
|
231
|
+
print(json.dumps(m.verify_integrity(), indent=2, default=str))
|
|
232
|
+
elif args.cmd == "consolidate":
|
|
233
|
+
print(json.dumps(m.consolidate(
|
|
234
|
+
dry_run=getattr(args, "dry_run", False),
|
|
235
|
+
lifecycle=not getattr(args, "no_lifecycle", False),
|
|
236
|
+
dreaming=not getattr(args, "no_dreaming", False),
|
|
237
|
+
user_id=getattr(args, "user_id", None),
|
|
238
|
+
), indent=2, default=str))
|
|
239
|
+
elif args.cmd == "export-schema":
|
|
240
|
+
print(json.dumps(m.export_schema_report(), indent=2, default=str))
|
|
241
|
+
elif args.cmd == "migrate":
|
|
242
|
+
from cortexm.migrate.importers import MIGRATORS
|
|
243
|
+
out = MIGRATORS[args.source](m, args.path, user_id=args.user_id)
|
|
244
|
+
print(json.dumps(out, indent=2))
|
|
245
|
+
elif args.cmd == "git":
|
|
246
|
+
if args.op == "log":
|
|
247
|
+
for c in m.log():
|
|
248
|
+
print(f"{c['id'][:8]} {c['ts']} {c['message']} "
|
|
249
|
+
f"({json.loads(c['parents'])})")
|
|
250
|
+
elif args.op == "branches":
|
|
251
|
+
for b in m.branches():
|
|
252
|
+
print(f"{b['name']}: {b['head'][:8]}")
|
|
253
|
+
elif args.op == "diff":
|
|
254
|
+
if len(args.rest) < 2:
|
|
255
|
+
print("usage: cortexm git diff <commitA> <commitB>")
|
|
256
|
+
return 2
|
|
257
|
+
print(json.dumps(m.diff(args.rest[0], args.rest[1]),
|
|
258
|
+
indent=2))
|
|
259
|
+
elif args.op == "blame":
|
|
260
|
+
if not args.rest:
|
|
261
|
+
print("usage: cortexm git blame <subject> [relation]")
|
|
262
|
+
return 2
|
|
263
|
+
rel = args.rest[1] if len(args.rest) > 1 else None
|
|
264
|
+
for row in m.blame(args.rest[0], rel):
|
|
265
|
+
print(f"{row['commit'][:8] if row['commit'] else '--------'} "
|
|
266
|
+
f"{row['recorded_at'][:10]} {row['fact']} "
|
|
267
|
+
f"{'[active]' if row['active'] else '[retired]'}")
|
|
268
|
+
return 0
|
|
269
|
+
finally:
|
|
270
|
+
m.close()
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def cost_report(memories: int, competitor_per_memory: float) -> str:
|
|
274
|
+
"""The μ=0 cost asymmetry (Section: MOAT 6)."""
|
|
275
|
+
ours = memories * 0.00001 # deterministic CPU-only ingest
|
|
276
|
+
theirs = memories * competitor_per_memory
|
|
277
|
+
lines = [
|
|
278
|
+
"Context-M μ=0 Cost Calculator",
|
|
279
|
+
"============================",
|
|
280
|
+
f"memories: {memories:,}",
|
|
281
|
+
f"cortex-m ingest (CPU): ${ours:,.2f}",
|
|
282
|
+
f"LLM-in-loop ingest: ${theirs:,.2f}",
|
|
283
|
+
f"cost advantage: {theirs / max(ours, 1e-9):,.0f}x",
|
|
284
|
+
"",
|
|
285
|
+
"storage tiers (per million memories):",
|
|
286
|
+
" int8 770 MB baseline",
|
|
287
|
+
" binary 96 MB edge (Raspberry Pi 5 → 10M memories)",
|
|
288
|
+
" rabitq 96 MB ultra-edge",
|
|
289
|
+
" pq 8 MB cloud",
|
|
290
|
+
]
|
|
291
|
+
return "\n".join(lines)
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
if __name__ == "__main__":
|
|
295
|
+
sys.exit(main())
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""HMS-style Cognition Engine — background self-organization.
|
|
2
|
+
|
|
3
|
+
The standout feature of holographic-memory (HMS): a background thread
|
|
4
|
+
that surfaces patterns, builds abstractions, detects knowledge gaps,
|
|
5
|
+
hypothesizes fillers, and finds analogies across domains.
|
|
6
|
+
|
|
7
|
+
This is the active part of memory — it turns a passive store into a
|
|
8
|
+
self-organizing knowledge base. When a user says "Alice works at Google"
|
|
9
|
+
and later "Alice moved to Mountain View," the engine should hypothesize
|
|
10
|
+
(Alice, lives_in, Mountain_View) and flag it for confirmation.
|
|
11
|
+
|
|
12
|
+
Architecture:
|
|
13
|
+
PatternScanner — surfaces structural regularities across triples
|
|
14
|
+
AbstractionEngine — bundles atom vectors into prototypes when N
|
|
15
|
+
entities share a relation pattern
|
|
16
|
+
GapDetector — finds missing relations by comparing an entity's
|
|
17
|
+
profile to peers
|
|
18
|
+
HypothesisEngine — proposes fillers for gaps via Hopfield cleanup
|
|
19
|
+
AnalogyDetector — finds structurally isomorphic domains via
|
|
20
|
+
bipartite relation mapping
|
|
21
|
+
|
|
22
|
+
Unlike HMS (which runs in a background thread), we trigger the engine
|
|
23
|
+
from `cortexm consolidate` — deterministic, auditable, no surprise
|
|
24
|
+
writes. Output is HYPOTHESIZED_BY edges in the Trace with confidence
|
|
25
|
+
< 0.5; never active in retrieval unless explicitly promoted.
|
|
26
|
+
|
|
27
|
+
The 5 stages run in a fixed pipeline so a single consolidate() call
|
|
28
|
+
produces a full self-organization sweep. Each stage reads the prior
|
|
29
|
+
stage's outputs via the Trace, so the engine is composable.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
from cortexm.cognition.scanner import PatternScanner
|
|
33
|
+
from cortexm.cognition.abstraction import AbstractionEngine
|
|
34
|
+
from cortexm.cognition.gaps import GapDetector, HypothesisEngine
|
|
35
|
+
from cortexm.cognition.analogy import AnalogyDetector
|
|
36
|
+
from cortexm.cognition.engine import (
|
|
37
|
+
CognitionEngine,
|
|
38
|
+
run_cognition_pass,
|
|
39
|
+
HYPOTHESIZED_BY,
|
|
40
|
+
PROMOTED_FROM,
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
__all__ = [
|
|
44
|
+
"PatternScanner",
|
|
45
|
+
"AbstractionEngine",
|
|
46
|
+
"GapDetector",
|
|
47
|
+
"HypothesisEngine",
|
|
48
|
+
"AnalogyDetector",
|
|
49
|
+
"CognitionEngine",
|
|
50
|
+
"run_cognition_pass",
|
|
51
|
+
"HYPOTHESIZED_BY",
|
|
52
|
+
"PROMOTED_FROM",
|
|
53
|
+
]
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
"""AbstractionEngine — bundles atom vectors into prototype categories.
|
|
2
|
+
|
|
3
|
+
Second stage of the HMS cognition engine. Consumes the patterns
|
|
4
|
+
surfaced by PatternScanner and builds prototype categories:
|
|
5
|
+
|
|
6
|
+
- When N entities share a relation pattern (e.g. works_at + lives_in),
|
|
7
|
+
create a prototype "person" category and tag those entities as
|
|
8
|
+
members.
|
|
9
|
+
- When N values appear for the same relation across subjects
|
|
10
|
+
(value_fanout), those values become "category centroids" — e.g.
|
|
11
|
+
if (Google, Stripe, Anthropic) all appear as works_at values,
|
|
12
|
+
they form a "tech_company" prototype.
|
|
13
|
+
|
|
14
|
+
The prototype is stored as a derived fact in the Trace:
|
|
15
|
+
(proto:<name>, member_of, <entity>)
|
|
16
|
+
with provenance.kind = "abstraction" and confidence < 0.5
|
|
17
|
+
|
|
18
|
+
This stage does NOT delete facts — it adds new derived facts that
|
|
19
|
+
link entities to abstractions. The derived facts are tagged
|
|
20
|
+
`is_derived=1` so they don't pollute the active fact count.
|
|
21
|
+
|
|
22
|
+
Prototype vectors are also computed in the VSA palace (superposition
|
|
23
|
+
of members' holograms) so retrieval can match against the prototype
|
|
24
|
+
directly — but this is the palace's responsibility, not the
|
|
25
|
+
abstraction engine's. We only emit the membership edges here.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
from collections import defaultdict
|
|
31
|
+
from dataclasses import dataclass, field
|
|
32
|
+
from typing import Any
|
|
33
|
+
|
|
34
|
+
from cortexm.cognition.scanner import Pattern, ScanResult
|
|
35
|
+
from cortexm.trace.store import TraceStore
|
|
36
|
+
from cortexm.util import iso, new_id
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass
|
|
40
|
+
class Abstraction:
|
|
41
|
+
"""A prototype category discovered by AbstractionEngine."""
|
|
42
|
+
name: str # e.g. "person" or "tech_company"
|
|
43
|
+
kind: str # "subject_role" | "value_cluster"
|
|
44
|
+
members: list[str] = field(default_factory=list)
|
|
45
|
+
prototype_relations: list[str] = field(default_factory=list)
|
|
46
|
+
confidence: float = 0.0
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass
|
|
50
|
+
class AbstractionResult:
|
|
51
|
+
abstractions: list[Abstraction] = field(default_factory=list)
|
|
52
|
+
membership_edges_added: int = 0
|
|
53
|
+
duration_ms: float = 0.0
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class AbstractionEngine:
|
|
57
|
+
"""Builds prototype categories from surfaced patterns."""
|
|
58
|
+
|
|
59
|
+
def __init__(self, store: TraceStore,
|
|
60
|
+
min_members: int = 2,
|
|
61
|
+
min_co_occur_support: int = 2) -> None:
|
|
62
|
+
self.store = store
|
|
63
|
+
self.min_members = min_members
|
|
64
|
+
self.min_co_occur_support = min_co_occur_support
|
|
65
|
+
|
|
66
|
+
def run(self, scan: ScanResult, *,
|
|
67
|
+
dry_run: bool = False,
|
|
68
|
+
commit_id: str | None = None,
|
|
69
|
+
user_id: str | None = None) -> AbstractionResult:
|
|
70
|
+
"""Build abstractions from scan results."""
|
|
71
|
+
import time
|
|
72
|
+
t0 = time.perf_counter()
|
|
73
|
+
|
|
74
|
+
# subject-role abstractions: subjects with the same set of
|
|
75
|
+
# relations (e.g. {works_at, lives_in, has_skill}) form a
|
|
76
|
+
# "person" prototype. Group subjects by frozenset(relations).
|
|
77
|
+
rel_groups: dict[frozenset, list[str]] = defaultdict(list)
|
|
78
|
+
for p in scan.patterns:
|
|
79
|
+
if p.kind != "subject_fanout":
|
|
80
|
+
continue
|
|
81
|
+
if p.support < self.min_co_occur_support:
|
|
82
|
+
continue
|
|
83
|
+
rels = frozenset(p.payload.get("relations", []))
|
|
84
|
+
if len(rels) >= 2:
|
|
85
|
+
rel_groups[rels].append(p.payload["subject"])
|
|
86
|
+
|
|
87
|
+
# value-cluster abstractions: values that appear for the same
|
|
88
|
+
# relation across N subjects form a "value_cluster" prototype
|
|
89
|
+
# (e.g. tech_company for {Google, Stripe, Anthropic, OpenAI})
|
|
90
|
+
val_groups: dict[tuple[str, list[str]], list[str]] = defaultdict(list)
|
|
91
|
+
for p in scan.patterns:
|
|
92
|
+
if p.kind != "value_fanout":
|
|
93
|
+
continue
|
|
94
|
+
if p.support < self.min_members:
|
|
95
|
+
continue
|
|
96
|
+
# which relation does this value cluster under? We don't
|
|
97
|
+
# know directly from the pattern — look it up.
|
|
98
|
+
val = p.payload["value"]
|
|
99
|
+
subjs = p.payload["subjects"]
|
|
100
|
+
# find the relation(s) that produce this value across subjects
|
|
101
|
+
rel_rows = self.store.conn.execute(
|
|
102
|
+
"SELECT DISTINCT relation FROM facts WHERE value=? "
|
|
103
|
+
"AND is_active=1 AND quarantined=0 LIMIT 3",
|
|
104
|
+
(val,)).fetchall()
|
|
105
|
+
for r in rel_rows:
|
|
106
|
+
rel = r[0]
|
|
107
|
+
key = (rel, [val])
|
|
108
|
+
val_groups[key].extend(subjs)
|
|
109
|
+
|
|
110
|
+
abstractions: list[Abstraction] = []
|
|
111
|
+
# synthesize subject-role abstractions
|
|
112
|
+
proto_idx = 0
|
|
113
|
+
for rels, members in rel_groups.items():
|
|
114
|
+
if len(members) < self.min_members:
|
|
115
|
+
continue
|
|
116
|
+
name = f"proto:role_{proto_idx}"
|
|
117
|
+
proto_idx += 1
|
|
118
|
+
abstractions.append(Abstraction(
|
|
119
|
+
name=name,
|
|
120
|
+
kind="subject_role",
|
|
121
|
+
members=members,
|
|
122
|
+
prototype_relations=sorted(rels),
|
|
123
|
+
confidence=min(0.49, 0.10 + 0.05 * len(members))))
|
|
124
|
+
|
|
125
|
+
# synthesize value-cluster abstractions
|
|
126
|
+
cluster_idx = 0
|
|
127
|
+
seen_clusters: set[frozenset] = set()
|
|
128
|
+
for (rel, vals), members in val_groups.items():
|
|
129
|
+
# collapse duplicates (same values for same relation)
|
|
130
|
+
key = frozenset(vals + [rel])
|
|
131
|
+
if key in seen_clusters:
|
|
132
|
+
continue
|
|
133
|
+
seen_clusters.add(key)
|
|
134
|
+
# collect unique members (subjects that have these values)
|
|
135
|
+
unique_members = sorted(set(members))
|
|
136
|
+
if len(unique_members) < self.min_members:
|
|
137
|
+
continue
|
|
138
|
+
name = f"proto:value_cluster_{cluster_idx}"
|
|
139
|
+
cluster_idx += 1
|
|
140
|
+
abstractions.append(Abstraction(
|
|
141
|
+
name=name,
|
|
142
|
+
kind="value_cluster",
|
|
143
|
+
members=unique_members,
|
|
144
|
+
prototype_relations=[rel],
|
|
145
|
+
confidence=min(0.49, 0.10 + 0.05 * len(unique_members))))
|
|
146
|
+
|
|
147
|
+
# emit membership edges as derived facts
|
|
148
|
+
n_added = 0
|
|
149
|
+
if not dry_run and abstractions:
|
|
150
|
+
ts = iso(_now())
|
|
151
|
+
for ab in abstractions:
|
|
152
|
+
for member in ab.members:
|
|
153
|
+
fid = new_id()
|
|
154
|
+
self.store.conn.execute(
|
|
155
|
+
"INSERT INTO facts "
|
|
156
|
+
"(id, subject, relation, value, valid_from, "
|
|
157
|
+
" tx_from, confidence, user_id, memory_type, "
|
|
158
|
+
" is_derived, is_active, birth_commit, provenance) "
|
|
159
|
+
"VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?)",
|
|
160
|
+
(fid, ab.name, "member_of", member, ts, ts,
|
|
161
|
+
ab.confidence, user_id or "default",
|
|
162
|
+
"long_term", 1, 1,
|
|
163
|
+
commit_id,
|
|
164
|
+
_prov_json(ab)))
|
|
165
|
+
n_added += 1
|
|
166
|
+
if commit_id:
|
|
167
|
+
self.store.update_commit_n_facts(commit_id, n_added)
|
|
168
|
+
|
|
169
|
+
return AbstractionResult(
|
|
170
|
+
abstractions=abstractions,
|
|
171
|
+
membership_edges_added=n_added,
|
|
172
|
+
duration_ms=(time.perf_counter() - t0) * 1000.0)
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _now():
|
|
176
|
+
from datetime import datetime, timezone
|
|
177
|
+
return datetime.now(timezone.utc)
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _prov_json(ab: Abstraction) -> str:
|
|
181
|
+
import json
|
|
182
|
+
return json.dumps({
|
|
183
|
+
"kind": "abstraction",
|
|
184
|
+
"abstraction_kind": ab.kind,
|
|
185
|
+
"abstraction_name": ab.name,
|
|
186
|
+
"n_members": len(ab.members),
|
|
187
|
+
"prototype_relations": ab.prototype_relations,
|
|
188
|
+
"generated_by": "cognition.abstraction",
|
|
189
|
+
})
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
__all__ = ["AbstractionEngine", "Abstraction", "AbstractionResult"]
|