cortexm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. context_m.py +17 -0
  2. cortexm/__init__.py +45 -0
  3. cortexm/accel.py +403 -0
  4. cortexm/api/__init__.py +0 -0
  5. cortexm/api/chaos.py +118 -0
  6. cortexm/api/memory.py +635 -0
  7. cortexm/bench/__init__.py +0 -0
  8. cortexm/bench/abilities.py +311 -0
  9. cortexm/bench/baselines.py +89 -0
  10. cortexm/bench/beam_loader.py +317 -0
  11. cortexm/bench/generator.py +376 -0
  12. cortexm/bench/harness.py +211 -0
  13. cortexm/bench/messy.py +218 -0
  14. cortexm/bench/micro.py +251 -0
  15. cortexm/bench/ood.py +443 -0
  16. cortexm/bench/run.py +137 -0
  17. cortexm/bridge/__init__.py +0 -0
  18. cortexm/bridge/dates.py +178 -0
  19. cortexm/bridge/decoders.py +204 -0
  20. cortexm/bridge/enrich.py +255 -0
  21. cortexm/bridge/extractor.py +316 -0
  22. cortexm/bridge/fallback.py +332 -0
  23. cortexm/bridge/onnx_runtime.py +158 -0
  24. cortexm/bridge/patterns.py +760 -0
  25. cortexm/bridge/ppr.py +104 -0
  26. cortexm/bridge/prefilter.py +188 -0
  27. cortexm/bridge/query_extract.py +420 -0
  28. cortexm/bridge/reader.py +1174 -0
  29. cortexm/bridge/rerank.py +204 -0
  30. cortexm/bridge/writer.py +492 -0
  31. cortexm/cli.py +295 -0
  32. cortexm/cognition/__init__.py +53 -0
  33. cortexm/cognition/abstraction.py +192 -0
  34. cortexm/cognition/analogy.py +159 -0
  35. cortexm/cognition/engine.py +204 -0
  36. cortexm/cognition/gaps.py +365 -0
  37. cortexm/cognition/scanner.py +204 -0
  38. cortexm/config.py +375 -0
  39. cortexm/cortexm.py +8 -0
  40. cortexm/enterprise/__init__.py +0 -0
  41. cortexm/enterprise/audit.py +178 -0
  42. cortexm/enterprise/governance.py +239 -0
  43. cortexm/errors.py +35 -0
  44. cortexm/features/__init__.py +0 -0
  45. cortexm/features/git.py +204 -0
  46. cortexm/features/prefetch.py +88 -0
  47. cortexm/features/zk.py +105 -0
  48. cortexm/federation/__init__.py +39 -0
  49. cortexm/federation/crdt.py +275 -0
  50. cortexm/federation/fabric.py +109 -0
  51. cortexm/federation/hlc.py +80 -0
  52. cortexm/federation/node.py +145 -0
  53. cortexm/federation/schema_report.py +73 -0
  54. cortexm/federation/transport.py +164 -0
  55. cortexm/index/__init__.py +19 -0
  56. cortexm/index/nsg.py +386 -0
  57. cortexm/mcp/__init__.py +0 -0
  58. cortexm/mcp/server.py +985 -0
  59. cortexm/metrics.py +62 -0
  60. cortexm/migrate/__init__.py +0 -0
  61. cortexm/migrate/importers.py +192 -0
  62. cortexm/provenance/__init__.py +78 -0
  63. cortexm/provenance/agent.py +214 -0
  64. cortexm/provenance/cose.py +201 -0
  65. cortexm/provenance/scitt.py +258 -0
  66. cortexm/provenance/vc.py +250 -0
  67. cortexm/security/__init__.py +0 -0
  68. cortexm/security/crypto.py +162 -0
  69. cortexm/security/hashes.py +140 -0
  70. cortexm/security/injection.py +149 -0
  71. cortexm/security/mind.py +154 -0
  72. cortexm/security/pii.py +265 -0
  73. cortexm/security/rbac.py +169 -0
  74. cortexm/security/sandbox.py +131 -0
  75. cortexm/security/zk_hamming.py +142 -0
  76. cortexm/security/zk_sql.py +485 -0
  77. cortexm/server/__init__.py +0 -0
  78. cortexm/server/metrics.py +88 -0
  79. cortexm/server/rest.py +936 -0
  80. cortexm/server/sparql.py +984 -0
  81. cortexm/text/__init__.py +0 -0
  82. cortexm/text/dissim.py +252 -0
  83. cortexm/text/embedder.py +155 -0
  84. cortexm/text/fuzzy.py +218 -0
  85. cortexm/text/idiolect.py +253 -0
  86. cortexm/text/labse.py +374 -0
  87. cortexm/text/tokenizer.py +79 -0
  88. cortexm/trace/__init__.py +0 -0
  89. cortexm/trace/blob_arena.py +277 -0
  90. cortexm/trace/consolidate.py +337 -0
  91. cortexm/trace/contradictions.py +69 -0
  92. cortexm/trace/dedup.py +114 -0
  93. cortexm/trace/edges.py +214 -0
  94. cortexm/trace/fact.py +121 -0
  95. cortexm/trace/fade.py +245 -0
  96. cortexm/trace/lifecycle.py +112 -0
  97. cortexm/trace/rebuild.py +173 -0
  98. cortexm/trace/rules.py +171 -0
  99. cortexm/trace/store.py +680 -0
  100. cortexm/trace/structural.py +183 -0
  101. cortexm/trace/tmt.py +335 -0
  102. cortexm/util.py +148 -0
  103. cortexm/vsa/__init__.py +0 -0
  104. cortexm/vsa/attribution.py +149 -0
  105. cortexm/vsa/cleanup.py +161 -0
  106. cortexm/vsa/codecs.py +397 -0
  107. cortexm/vsa/hologram_overlay.py +139 -0
  108. cortexm/vsa/index.py +163 -0
  109. cortexm/vsa/ops.py +149 -0
  110. cortexm/vsa/palace.py +446 -0
  111. cortexm/vsa/role_vectors.py +236 -0
  112. cortexm/vsa/slb.py +78 -0
  113. cortexm/vsa/tlsh_trie.py +137 -0
  114. cortexm/vsa/working_memory.py +249 -0
  115. cortexm-0.3.0.dist-info/METADATA +482 -0
  116. cortexm-0.3.0.dist-info/RECORD +120 -0
  117. cortexm-0.3.0.dist-info/WHEEL +5 -0
  118. cortexm-0.3.0.dist-info/entry_points.txt +2 -0
  119. cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
  120. cortexm-0.3.0.dist-info/top_level.txt +2 -0
cortexm/cli.py ADDED
@@ -0,0 +1,295 @@
1
+ """cortexm — the Context-M command line.
2
+
3
+ cortexm serve # MCP server (stdio JSON-RPC)
4
+ cortexm stats [--db PATH]
5
+ cortexm verify [--db PATH] # integrity audit (hashes + vectors)
6
+ cortexm consolidate [--db PATH]
7
+ cortexm migrate --from mem0 --path mem0.db [--db PATH]
8
+ cortexm cost --memories 1000000 # μ=0 cost calculator
9
+ cortexm bench --buckets 128k,1m # BEAM-style benchmark
10
+ cortexm export-schema [--db PATH]
11
+ cortexm git log|branches|diff|blame
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import argparse
17
+ import json
18
+ import sys
19
+
20
+
21
+ def _memory(args):
22
+ from cortexm.api.memory import Memory
23
+ from cortexm.config import Config
24
+ cfg = Config.from_env()
25
+ if getattr(args, "db", None):
26
+ cfg.db_path = args.db
27
+ return Memory(cfg)
28
+
29
+
30
+ def main(argv=None) -> int:
31
+ ap = argparse.ArgumentParser(prog="cortexm",
32
+ description="Context-M memory fabric CLI")
33
+ sub = ap.add_subparsers(dest="cmd")
34
+
35
+ p = sub.add_parser("serve", help="run the MCP server on stdio")
36
+ p.add_argument("--db", default=None)
37
+
38
+ p = sub.add_parser("serve-rest", help="run the REST API server (HTTP)")
39
+ p.add_argument("--db", default=":memory:")
40
+ p.add_argument("--host", default="0.0.0.0")
41
+ p.add_argument("--port", type=int, default=8900)
42
+ p.add_argument("--pii", default=None, choices=[None, "off", "redact",
43
+ "block", "tag"])
44
+ p.add_argument("--admin-key", default=None,
45
+ help="mint an admin API key at boot (printed once)")
46
+ p.add_argument("--sparql-port", type=int, default=None,
47
+ help="co-host a SPARQL endpoint on this port "
48
+ "(shares one Memory instance with the REST API)")
49
+ p.add_argument("--sparql-host", default="0.0.0.0",
50
+ help="bind SPARQL endpoint to this host")
51
+ p.add_argument("--sparql-user-id", default=None,
52
+ help="scope SPARQL queries to a single user")
53
+
54
+ for name, help_ in (("stats", "memory statistics"),
55
+ ("verify", "integrity audit"),
56
+ ("consolidate", "lifecycle consolidation"),
57
+ ("export-schema", "federation schema report")):
58
+ p = sub.add_parser(name, help=help_)
59
+ p.add_argument("--db", default=None)
60
+
61
+ # extend consolidate with toggles for each pass
62
+ # (the loop above already created 'consolidate'; we fetch it
63
+ # by name and add extra flags)
64
+ con = [a for a in sub.choices.values() if a.prog.endswith("consolidate")][0]
65
+ con.add_argument("--dry-run", action="store_true",
66
+ help="don't commit changes — just report what would happen")
67
+ con.add_argument("--no-lifecycle", action="store_true",
68
+ help="skip the lifecycle pass (short→long promotion / decay)")
69
+ con.add_argument("--no-dreaming", action="store_true",
70
+ help="skip the Aeon dreaming pass (merge triples, retire stale, "
71
+ "defrag palace, retrain prefetcher)")
72
+ con.add_argument("--user-id", default=None,
73
+ help="scope the dreaming pass to one user")
74
+
75
+ p = sub.add_parser("keys", help="API key management (RBAC)")
76
+ p.add_argument("op", choices=["create", "list", "revoke"])
77
+ p.add_argument("--db", default=None)
78
+ p.add_argument("--role", default="reader",
79
+ choices=["admin", "operator", "reader", "auditor"])
80
+ p.add_argument("--label", default="")
81
+ p.add_argument("--id", default=None, help="key id to revoke")
82
+
83
+ p = sub.add_parser("audit", help="audit log tail / verify / export")
84
+ p.add_argument("op", choices=["tail", "verify", "export"])
85
+ p.add_argument("--db", default=None)
86
+ p.add_argument("-n", type=int, default=30)
87
+ p.add_argument("--out", default="audit.jsonl",
88
+ help="export path (op=export)")
89
+
90
+ p = sub.add_parser("snapshot", help="atomic backup with manifest")
91
+ p.add_argument("--db", default=None)
92
+ p.add_argument("--path", required=True)
93
+
94
+ p = sub.add_parser("erase", help="GDPR right-to-erasure")
95
+ p.add_argument("--db", default=None)
96
+ p.add_argument("--user-id", required=True)
97
+
98
+ p = sub.add_parser("governance", help="governance ops")
99
+ p.add_argument("op", choices=["retention", "state-at"])
100
+ p.add_argument("--db", default=None)
101
+ p.add_argument("--days", type=int, default=365)
102
+ p.add_argument("--when", default=None,
103
+ help="ISO datetime for state-at")
104
+ p.add_argument("--user-id", default=None)
105
+
106
+ p = sub.add_parser("migrate", help="import from mem0/zep/chroma")
107
+ p.add_argument("--from", dest="source", required=True,
108
+ choices=["mem0", "zep", "chroma"])
109
+ p.add_argument("--path", required=True)
110
+ p.add_argument("--db", default=None)
111
+ p.add_argument("--user-id", default="migrated")
112
+
113
+ p = sub.add_parser("cost", help="μ=0 cost calculator")
114
+ p.add_argument("--memories", type=int, default=1_000_000)
115
+ p.add_argument("--ingest-per-memory", type=float, default=0.001,
116
+ help="competitor LLM-extract cost per memory (USD)")
117
+
118
+ p = sub.add_parser("bench", help="BEAM-style benchmark")
119
+ p.add_argument("--buckets", default="128k")
120
+ p.add_argument("--micro", action="store_true")
121
+
122
+ p = sub.add_parser("git", help="Memory Git operations")
123
+ p.add_argument("op", choices=["log", "branches", "diff", "blame"])
124
+ p.add_argument("rest", nargs="*")
125
+ p.add_argument("--db", default=None)
126
+
127
+ args = ap.parse_args(argv)
128
+ if not args.cmd:
129
+ ap.print_help()
130
+ return 0
131
+
132
+ if args.cmd == "serve":
133
+ from cortexm.mcp.server import serve
134
+ serve(args.db)
135
+ return 0
136
+
137
+ if args.cmd == "serve-rest":
138
+ from cortexm.server.rest import main as rest_main
139
+ # argparse passthrough for the REST server
140
+ rest_argv = ["--db", args.db, "--host", args.host,
141
+ "--port", str(args.port)]
142
+ if args.pii:
143
+ rest_argv += ["--pii", args.pii]
144
+ if args.admin_key:
145
+ rest_argv += ["--admin-key", args.admin_key]
146
+ if getattr(args, "sparql_port", None):
147
+ rest_argv += ["--sparql-port", str(args.sparql_port),
148
+ "--sparql-host", args.sparql_host]
149
+ if args.sparql_user_id:
150
+ rest_argv += ["--sparql-user-id", args.sparql_user_id]
151
+ import sys as _sys
152
+ _sys.argv = ["contextm-serve"] + rest_argv
153
+ rest_main()
154
+ return 0
155
+
156
+ if args.cmd == "keys":
157
+ m = _memory(args)
158
+ try:
159
+ if args.op == "create":
160
+ out = m.keys.create(args.role, label=args.label)
161
+ print(json.dumps(out, indent=2))
162
+ elif args.op == "list":
163
+ print(json.dumps({"keys": m.keys.list_keys()}, indent=2))
164
+ elif args.op == "revoke":
165
+ print(json.dumps({"revoked": m.keys.revoke(args.id or "")}))
166
+ finally:
167
+ m.close()
168
+ return 0
169
+
170
+ if args.cmd == "audit":
171
+ m = _memory(args)
172
+ try:
173
+ if args.op == "verify":
174
+ print(json.dumps(m.audit_log.verify(), indent=2))
175
+ elif args.op == "export":
176
+ n = m.audit_log.export_jsonl(args.out)
177
+ print(json.dumps({"exported": n, "path": args.out}))
178
+ else:
179
+ print(json.dumps({"events": m.audit_log.tail(args.n)}, indent=2))
180
+ finally:
181
+ m.close()
182
+ return 0
183
+
184
+ if args.cmd == "snapshot":
185
+ m = _memory(args)
186
+ try:
187
+ print(json.dumps(m.governance.snapshot(args.path), indent=2))
188
+ finally:
189
+ m.close()
190
+ return 0
191
+
192
+ if args.cmd == "erase":
193
+ m = _memory(args)
194
+ try:
195
+ print(json.dumps(m.governance.erase_user(args.user_id), indent=2))
196
+ finally:
197
+ m.close()
198
+ return 0
199
+
200
+ if args.cmd == "governance":
201
+ m = _memory(args)
202
+ try:
203
+ if args.op == "retention":
204
+ out = m.governance.apply_retention(args.days,
205
+ user_id=args.user_id)
206
+ else:
207
+ out = {"facts": m.governance.state_at(args.when or "",
208
+ user_id=args.user_id)}
209
+ print(json.dumps(out, indent=2, default=str))
210
+ finally:
211
+ m.close()
212
+ return 0
213
+
214
+ if args.cmd == "cost":
215
+ print(cost_report(args.memories, args.ingest_per_memory))
216
+ return 0
217
+
218
+ if args.cmd == "bench":
219
+ from cortexm.bench.run import main as bench_main
220
+ argv = ["--buckets", args.buckets]
221
+ if args.micro:
222
+ argv.append("--micro")
223
+ bench_main()
224
+ return 0
225
+
226
+ m = _memory(args)
227
+ try:
228
+ if args.cmd == "stats":
229
+ print(json.dumps(m.stats(), indent=2, default=str))
230
+ elif args.cmd == "verify":
231
+ print(json.dumps(m.verify_integrity(), indent=2, default=str))
232
+ elif args.cmd == "consolidate":
233
+ print(json.dumps(m.consolidate(
234
+ dry_run=getattr(args, "dry_run", False),
235
+ lifecycle=not getattr(args, "no_lifecycle", False),
236
+ dreaming=not getattr(args, "no_dreaming", False),
237
+ user_id=getattr(args, "user_id", None),
238
+ ), indent=2, default=str))
239
+ elif args.cmd == "export-schema":
240
+ print(json.dumps(m.export_schema_report(), indent=2, default=str))
241
+ elif args.cmd == "migrate":
242
+ from cortexm.migrate.importers import MIGRATORS
243
+ out = MIGRATORS[args.source](m, args.path, user_id=args.user_id)
244
+ print(json.dumps(out, indent=2))
245
+ elif args.cmd == "git":
246
+ if args.op == "log":
247
+ for c in m.log():
248
+ print(f"{c['id'][:8]} {c['ts']} {c['message']} "
249
+ f"({json.loads(c['parents'])})")
250
+ elif args.op == "branches":
251
+ for b in m.branches():
252
+ print(f"{b['name']}: {b['head'][:8]}")
253
+ elif args.op == "diff":
254
+ if len(args.rest) < 2:
255
+ print("usage: cortexm git diff <commitA> <commitB>")
256
+ return 2
257
+ print(json.dumps(m.diff(args.rest[0], args.rest[1]),
258
+ indent=2))
259
+ elif args.op == "blame":
260
+ if not args.rest:
261
+ print("usage: cortexm git blame <subject> [relation]")
262
+ return 2
263
+ rel = args.rest[1] if len(args.rest) > 1 else None
264
+ for row in m.blame(args.rest[0], rel):
265
+ print(f"{row['commit'][:8] if row['commit'] else '--------'} "
266
+ f"{row['recorded_at'][:10]} {row['fact']} "
267
+ f"{'[active]' if row['active'] else '[retired]'}")
268
+ return 0
269
+ finally:
270
+ m.close()
271
+
272
+
273
+ def cost_report(memories: int, competitor_per_memory: float) -> str:
274
+ """The μ=0 cost asymmetry (Section: MOAT 6)."""
275
+ ours = memories * 0.00001 # deterministic CPU-only ingest
276
+ theirs = memories * competitor_per_memory
277
+ lines = [
278
+ "Context-M μ=0 Cost Calculator",
279
+ "============================",
280
+ f"memories: {memories:,}",
281
+ f"cortex-m ingest (CPU): ${ours:,.2f}",
282
+ f"LLM-in-loop ingest: ${theirs:,.2f}",
283
+ f"cost advantage: {theirs / max(ours, 1e-9):,.0f}x",
284
+ "",
285
+ "storage tiers (per million memories):",
286
+ " int8 770 MB baseline",
287
+ " binary 96 MB edge (Raspberry Pi 5 → 10M memories)",
288
+ " rabitq 96 MB ultra-edge",
289
+ " pq 8 MB cloud",
290
+ ]
291
+ return "\n".join(lines)
292
+
293
+
294
+ if __name__ == "__main__":
295
+ sys.exit(main())
@@ -0,0 +1,53 @@
1
+ """HMS-style Cognition Engine — background self-organization.
2
+
3
+ The standout feature of holographic-memory (HMS): a background thread
4
+ that surfaces patterns, builds abstractions, detects knowledge gaps,
5
+ hypothesizes fillers, and finds analogies across domains.
6
+
7
+ This is the active part of memory — it turns a passive store into a
8
+ self-organizing knowledge base. When a user says "Alice works at Google"
9
+ and later "Alice moved to Mountain View," the engine should hypothesize
10
+ (Alice, lives_in, Mountain_View) and flag it for confirmation.
11
+
12
+ Architecture:
13
+ PatternScanner — surfaces structural regularities across triples
14
+ AbstractionEngine — bundles atom vectors into prototypes when N
15
+ entities share a relation pattern
16
+ GapDetector — finds missing relations by comparing an entity's
17
+ profile to peers
18
+ HypothesisEngine — proposes fillers for gaps via Hopfield cleanup
19
+ AnalogyDetector — finds structurally isomorphic domains via
20
+ bipartite relation mapping
21
+
22
+ Unlike HMS (which runs in a background thread), we trigger the engine
23
+ from `cortexm consolidate` — deterministic, auditable, no surprise
24
+ writes. Output is HYPOTHESIZED_BY edges in the Trace with confidence
25
+ < 0.5; never active in retrieval unless explicitly promoted.
26
+
27
+ The 5 stages run in a fixed pipeline so a single consolidate() call
28
+ produces a full self-organization sweep. Each stage reads the prior
29
+ stage's outputs via the Trace, so the engine is composable.
30
+ """
31
+
32
+ from cortexm.cognition.scanner import PatternScanner
33
+ from cortexm.cognition.abstraction import AbstractionEngine
34
+ from cortexm.cognition.gaps import GapDetector, HypothesisEngine
35
+ from cortexm.cognition.analogy import AnalogyDetector
36
+ from cortexm.cognition.engine import (
37
+ CognitionEngine,
38
+ run_cognition_pass,
39
+ HYPOTHESIZED_BY,
40
+ PROMOTED_FROM,
41
+ )
42
+
43
+ __all__ = [
44
+ "PatternScanner",
45
+ "AbstractionEngine",
46
+ "GapDetector",
47
+ "HypothesisEngine",
48
+ "AnalogyDetector",
49
+ "CognitionEngine",
50
+ "run_cognition_pass",
51
+ "HYPOTHESIZED_BY",
52
+ "PROMOTED_FROM",
53
+ ]
@@ -0,0 +1,192 @@
1
+ """AbstractionEngine — bundles atom vectors into prototype categories.
2
+
3
+ Second stage of the HMS cognition engine. Consumes the patterns
4
+ surfaced by PatternScanner and builds prototype categories:
5
+
6
+ - When N entities share a relation pattern (e.g. works_at + lives_in),
7
+ create a prototype "person" category and tag those entities as
8
+ members.
9
+ - When N values appear for the same relation across subjects
10
+ (value_fanout), those values become "category centroids" — e.g.
11
+ if (Google, Stripe, Anthropic) all appear as works_at values,
12
+ they form a "tech_company" prototype.
13
+
14
+ The prototype is stored as a derived fact in the Trace:
15
+ (proto:<name>, member_of, <entity>)
16
+ with provenance.kind = "abstraction" and confidence < 0.5
17
+
18
+ This stage does NOT delete facts — it adds new derived facts that
19
+ link entities to abstractions. The derived facts are tagged
20
+ `is_derived=1` so they don't pollute the active fact count.
21
+
22
+ Prototype vectors are also computed in the VSA palace (superposition
23
+ of members' holograms) so retrieval can match against the prototype
24
+ directly — but this is the palace's responsibility, not the
25
+ abstraction engine's. We only emit the membership edges here.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ from collections import defaultdict
31
+ from dataclasses import dataclass, field
32
+ from typing import Any
33
+
34
+ from cortexm.cognition.scanner import Pattern, ScanResult
35
+ from cortexm.trace.store import TraceStore
36
+ from cortexm.util import iso, new_id
37
+
38
+
39
+ @dataclass
40
+ class Abstraction:
41
+ """A prototype category discovered by AbstractionEngine."""
42
+ name: str # e.g. "person" or "tech_company"
43
+ kind: str # "subject_role" | "value_cluster"
44
+ members: list[str] = field(default_factory=list)
45
+ prototype_relations: list[str] = field(default_factory=list)
46
+ confidence: float = 0.0
47
+
48
+
49
+ @dataclass
50
+ class AbstractionResult:
51
+ abstractions: list[Abstraction] = field(default_factory=list)
52
+ membership_edges_added: int = 0
53
+ duration_ms: float = 0.0
54
+
55
+
56
+ class AbstractionEngine:
57
+ """Builds prototype categories from surfaced patterns."""
58
+
59
+ def __init__(self, store: TraceStore,
60
+ min_members: int = 2,
61
+ min_co_occur_support: int = 2) -> None:
62
+ self.store = store
63
+ self.min_members = min_members
64
+ self.min_co_occur_support = min_co_occur_support
65
+
66
+ def run(self, scan: ScanResult, *,
67
+ dry_run: bool = False,
68
+ commit_id: str | None = None,
69
+ user_id: str | None = None) -> AbstractionResult:
70
+ """Build abstractions from scan results."""
71
+ import time
72
+ t0 = time.perf_counter()
73
+
74
+ # subject-role abstractions: subjects with the same set of
75
+ # relations (e.g. {works_at, lives_in, has_skill}) form a
76
+ # "person" prototype. Group subjects by frozenset(relations).
77
+ rel_groups: dict[frozenset, list[str]] = defaultdict(list)
78
+ for p in scan.patterns:
79
+ if p.kind != "subject_fanout":
80
+ continue
81
+ if p.support < self.min_co_occur_support:
82
+ continue
83
+ rels = frozenset(p.payload.get("relations", []))
84
+ if len(rels) >= 2:
85
+ rel_groups[rels].append(p.payload["subject"])
86
+
87
+ # value-cluster abstractions: values that appear for the same
88
+ # relation across N subjects form a "value_cluster" prototype
89
+ # (e.g. tech_company for {Google, Stripe, Anthropic, OpenAI})
90
+ val_groups: dict[tuple[str, list[str]], list[str]] = defaultdict(list)
91
+ for p in scan.patterns:
92
+ if p.kind != "value_fanout":
93
+ continue
94
+ if p.support < self.min_members:
95
+ continue
96
+ # which relation does this value cluster under? We don't
97
+ # know directly from the pattern — look it up.
98
+ val = p.payload["value"]
99
+ subjs = p.payload["subjects"]
100
+ # find the relation(s) that produce this value across subjects
101
+ rel_rows = self.store.conn.execute(
102
+ "SELECT DISTINCT relation FROM facts WHERE value=? "
103
+ "AND is_active=1 AND quarantined=0 LIMIT 3",
104
+ (val,)).fetchall()
105
+ for r in rel_rows:
106
+ rel = r[0]
107
+ key = (rel, [val])
108
+ val_groups[key].extend(subjs)
109
+
110
+ abstractions: list[Abstraction] = []
111
+ # synthesize subject-role abstractions
112
+ proto_idx = 0
113
+ for rels, members in rel_groups.items():
114
+ if len(members) < self.min_members:
115
+ continue
116
+ name = f"proto:role_{proto_idx}"
117
+ proto_idx += 1
118
+ abstractions.append(Abstraction(
119
+ name=name,
120
+ kind="subject_role",
121
+ members=members,
122
+ prototype_relations=sorted(rels),
123
+ confidence=min(0.49, 0.10 + 0.05 * len(members))))
124
+
125
+ # synthesize value-cluster abstractions
126
+ cluster_idx = 0
127
+ seen_clusters: set[frozenset] = set()
128
+ for (rel, vals), members in val_groups.items():
129
+ # collapse duplicates (same values for same relation)
130
+ key = frozenset(vals + [rel])
131
+ if key in seen_clusters:
132
+ continue
133
+ seen_clusters.add(key)
134
+ # collect unique members (subjects that have these values)
135
+ unique_members = sorted(set(members))
136
+ if len(unique_members) < self.min_members:
137
+ continue
138
+ name = f"proto:value_cluster_{cluster_idx}"
139
+ cluster_idx += 1
140
+ abstractions.append(Abstraction(
141
+ name=name,
142
+ kind="value_cluster",
143
+ members=unique_members,
144
+ prototype_relations=[rel],
145
+ confidence=min(0.49, 0.10 + 0.05 * len(unique_members))))
146
+
147
+ # emit membership edges as derived facts
148
+ n_added = 0
149
+ if not dry_run and abstractions:
150
+ ts = iso(_now())
151
+ for ab in abstractions:
152
+ for member in ab.members:
153
+ fid = new_id()
154
+ self.store.conn.execute(
155
+ "INSERT INTO facts "
156
+ "(id, subject, relation, value, valid_from, "
157
+ " tx_from, confidence, user_id, memory_type, "
158
+ " is_derived, is_active, birth_commit, provenance) "
159
+ "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?)",
160
+ (fid, ab.name, "member_of", member, ts, ts,
161
+ ab.confidence, user_id or "default",
162
+ "long_term", 1, 1,
163
+ commit_id,
164
+ _prov_json(ab)))
165
+ n_added += 1
166
+ if commit_id:
167
+ self.store.update_commit_n_facts(commit_id, n_added)
168
+
169
+ return AbstractionResult(
170
+ abstractions=abstractions,
171
+ membership_edges_added=n_added,
172
+ duration_ms=(time.perf_counter() - t0) * 1000.0)
173
+
174
+
175
+ def _now():
176
+ from datetime import datetime, timezone
177
+ return datetime.now(timezone.utc)
178
+
179
+
180
+ def _prov_json(ab: Abstraction) -> str:
181
+ import json
182
+ return json.dumps({
183
+ "kind": "abstraction",
184
+ "abstraction_kind": ab.kind,
185
+ "abstraction_name": ab.name,
186
+ "n_members": len(ab.members),
187
+ "prototype_relations": ab.prototype_relations,
188
+ "generated_by": "cognition.abstraction",
189
+ })
190
+
191
+
192
+ __all__ = ["AbstractionEngine", "Abstraction", "AbstractionResult"]