cortexm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. context_m.py +17 -0
  2. cortexm/__init__.py +45 -0
  3. cortexm/accel.py +403 -0
  4. cortexm/api/__init__.py +0 -0
  5. cortexm/api/chaos.py +118 -0
  6. cortexm/api/memory.py +635 -0
  7. cortexm/bench/__init__.py +0 -0
  8. cortexm/bench/abilities.py +311 -0
  9. cortexm/bench/baselines.py +89 -0
  10. cortexm/bench/beam_loader.py +317 -0
  11. cortexm/bench/generator.py +376 -0
  12. cortexm/bench/harness.py +211 -0
  13. cortexm/bench/messy.py +218 -0
  14. cortexm/bench/micro.py +251 -0
  15. cortexm/bench/ood.py +443 -0
  16. cortexm/bench/run.py +137 -0
  17. cortexm/bridge/__init__.py +0 -0
  18. cortexm/bridge/dates.py +178 -0
  19. cortexm/bridge/decoders.py +204 -0
  20. cortexm/bridge/enrich.py +255 -0
  21. cortexm/bridge/extractor.py +316 -0
  22. cortexm/bridge/fallback.py +332 -0
  23. cortexm/bridge/onnx_runtime.py +158 -0
  24. cortexm/bridge/patterns.py +760 -0
  25. cortexm/bridge/ppr.py +104 -0
  26. cortexm/bridge/prefilter.py +188 -0
  27. cortexm/bridge/query_extract.py +420 -0
  28. cortexm/bridge/reader.py +1174 -0
  29. cortexm/bridge/rerank.py +204 -0
  30. cortexm/bridge/writer.py +492 -0
  31. cortexm/cli.py +295 -0
  32. cortexm/cognition/__init__.py +53 -0
  33. cortexm/cognition/abstraction.py +192 -0
  34. cortexm/cognition/analogy.py +159 -0
  35. cortexm/cognition/engine.py +204 -0
  36. cortexm/cognition/gaps.py +365 -0
  37. cortexm/cognition/scanner.py +204 -0
  38. cortexm/config.py +375 -0
  39. cortexm/cortexm.py +8 -0
  40. cortexm/enterprise/__init__.py +0 -0
  41. cortexm/enterprise/audit.py +178 -0
  42. cortexm/enterprise/governance.py +239 -0
  43. cortexm/errors.py +35 -0
  44. cortexm/features/__init__.py +0 -0
  45. cortexm/features/git.py +204 -0
  46. cortexm/features/prefetch.py +88 -0
  47. cortexm/features/zk.py +105 -0
  48. cortexm/federation/__init__.py +39 -0
  49. cortexm/federation/crdt.py +275 -0
  50. cortexm/federation/fabric.py +109 -0
  51. cortexm/federation/hlc.py +80 -0
  52. cortexm/federation/node.py +145 -0
  53. cortexm/federation/schema_report.py +73 -0
  54. cortexm/federation/transport.py +164 -0
  55. cortexm/index/__init__.py +19 -0
  56. cortexm/index/nsg.py +386 -0
  57. cortexm/mcp/__init__.py +0 -0
  58. cortexm/mcp/server.py +985 -0
  59. cortexm/metrics.py +62 -0
  60. cortexm/migrate/__init__.py +0 -0
  61. cortexm/migrate/importers.py +192 -0
  62. cortexm/provenance/__init__.py +78 -0
  63. cortexm/provenance/agent.py +214 -0
  64. cortexm/provenance/cose.py +201 -0
  65. cortexm/provenance/scitt.py +258 -0
  66. cortexm/provenance/vc.py +250 -0
  67. cortexm/security/__init__.py +0 -0
  68. cortexm/security/crypto.py +162 -0
  69. cortexm/security/hashes.py +140 -0
  70. cortexm/security/injection.py +149 -0
  71. cortexm/security/mind.py +154 -0
  72. cortexm/security/pii.py +265 -0
  73. cortexm/security/rbac.py +169 -0
  74. cortexm/security/sandbox.py +131 -0
  75. cortexm/security/zk_hamming.py +142 -0
  76. cortexm/security/zk_sql.py +485 -0
  77. cortexm/server/__init__.py +0 -0
  78. cortexm/server/metrics.py +88 -0
  79. cortexm/server/rest.py +936 -0
  80. cortexm/server/sparql.py +984 -0
  81. cortexm/text/__init__.py +0 -0
  82. cortexm/text/dissim.py +252 -0
  83. cortexm/text/embedder.py +155 -0
  84. cortexm/text/fuzzy.py +218 -0
  85. cortexm/text/idiolect.py +253 -0
  86. cortexm/text/labse.py +374 -0
  87. cortexm/text/tokenizer.py +79 -0
  88. cortexm/trace/__init__.py +0 -0
  89. cortexm/trace/blob_arena.py +277 -0
  90. cortexm/trace/consolidate.py +337 -0
  91. cortexm/trace/contradictions.py +69 -0
  92. cortexm/trace/dedup.py +114 -0
  93. cortexm/trace/edges.py +214 -0
  94. cortexm/trace/fact.py +121 -0
  95. cortexm/trace/fade.py +245 -0
  96. cortexm/trace/lifecycle.py +112 -0
  97. cortexm/trace/rebuild.py +173 -0
  98. cortexm/trace/rules.py +171 -0
  99. cortexm/trace/store.py +680 -0
  100. cortexm/trace/structural.py +183 -0
  101. cortexm/trace/tmt.py +335 -0
  102. cortexm/util.py +148 -0
  103. cortexm/vsa/__init__.py +0 -0
  104. cortexm/vsa/attribution.py +149 -0
  105. cortexm/vsa/cleanup.py +161 -0
  106. cortexm/vsa/codecs.py +397 -0
  107. cortexm/vsa/hologram_overlay.py +139 -0
  108. cortexm/vsa/index.py +163 -0
  109. cortexm/vsa/ops.py +149 -0
  110. cortexm/vsa/palace.py +446 -0
  111. cortexm/vsa/role_vectors.py +236 -0
  112. cortexm/vsa/slb.py +78 -0
  113. cortexm/vsa/tlsh_trie.py +137 -0
  114. cortexm/vsa/working_memory.py +249 -0
  115. cortexm-0.3.0.dist-info/METADATA +482 -0
  116. cortexm-0.3.0.dist-info/RECORD +120 -0
  117. cortexm-0.3.0.dist-info/WHEEL +5 -0
  118. cortexm-0.3.0.dist-info/entry_points.txt +2 -0
  119. cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
  120. cortexm-0.3.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,365 @@
1
+ """GapDetector + HypothesisEngine — fill in missing relations.
2
+
3
+ Third stage of the HMS cognition engine. Two stages in one module
4
+ because they're tightly coupled:
5
+
6
+ GapDetector — for each entity, compare its set of relations
7
+ to its peers' sets. Missing relations that
8
+ peers have are gaps. For example:
9
+ Alice has {works_at, lives_in, has_skill}
10
+ Bob has {works_at, lives_in, has_skill, age}
11
+ → Alice is missing an `age` fact — gap.
12
+
13
+ HypothesisEngine — for each gap, propose a filler. The filler can
14
+ come from:
15
+ (a) Majority vote: what value do peers of
16
+ the same abstraction have?
17
+ (b) Structural transitivity: if Alice's father
18
+ is Bob and Bob's father is Charles, the
19
+ engine hypothesizes Alice's grandfather is
20
+ Charles via the father→father chain.
21
+ (c) Hopfield cleanup: if a VSA palace is
22
+ available, query the superposition of
23
+ peers' values for that relation.
24
+
25
+ Hypotheses are written as facts with:
26
+ is_derived=1
27
+ confidence < 0.5
28
+ provenance.kind = "hypothesis"
29
+ provenance.source = the basis of the guess
30
+
31
+ The HYPOTHESIZED_BY edge links the supporting fact(s) to the new
32
+ hypothesis fact, so audits can trace the reasoning chain.
33
+ """
34
+
35
+ from __future__ import annotations
36
+
37
+ import json
38
+ from collections import defaultdict
39
+ from dataclasses import dataclass, field
40
+ from datetime import datetime, timezone
41
+ from typing import Any
42
+
43
+ from cortexm.cognition.abstraction import Abstraction
44
+ from cortexm.cognition.scanner import ScanResult
45
+ from cortexm.trace.store import TraceStore
46
+ from cortexm.util import iso, new_id
47
+
48
+
49
+ # Edge kind used to wire supporting facts to hypotheses
50
+ HYPOTHESIZED_BY = "HYPOTHESIZED_BY"
51
+
52
+
53
+ @dataclass
54
+ class Gap:
55
+ """A missing relation for an entity."""
56
+ subject: str
57
+ missing_relation: str
58
+ peer_count: int = 0 # how many peers have this relation
59
+ peer_values: list[str] = field(default_factory=list)
60
+ basis: str = "" # "majority" | "structural" | "cleanup"
61
+
62
+
63
+ @dataclass
64
+ class Hypothesis:
65
+ """A proposed filler for a gap."""
66
+ subject: str
67
+ relation: str
68
+ proposed_value: str
69
+ confidence: float = 0.0
70
+ basis: str = "" # which strategy proposed this
71
+ supporting_facts: list[str] = field(default_factory=list)
72
+ fact_id: str = ""
73
+
74
+
75
+ @dataclass
76
+ class GapResult:
77
+ gaps: list[Gap] = field(default_factory=list)
78
+ n_subjects_compared: int = 0
79
+ duration_ms: float = 0.0
80
+
81
+
82
+ @dataclass
83
+ class HypothesisResult:
84
+ hypotheses: list[Hypothesis] = field(default_factory=list)
85
+ facts_added: int = 0
86
+ duration_ms: float = 0.0
87
+
88
+
89
+ class GapDetector:
90
+ """Finds missing relations by comparing entity profiles to peers."""
91
+
92
+ def __init__(self, store: TraceStore,
93
+ min_peers_with_relation: int = 2) -> None:
94
+ self.store = store
95
+ self.min_peers_with_relation = min_peers_with_relation
96
+
97
+ def run(self, scan: ScanResult,
98
+ user_id: str | None = None) -> GapResult:
99
+ """Compare each subject's relation set to its peers' sets."""
100
+ import time
101
+ t0 = time.perf_counter()
102
+
103
+ # build per-subject relation sets from the trace
104
+ where = "is_active=1 AND quarantined=0"
105
+ args: tuple = ()
106
+ if user_id is not None:
107
+ where += " AND user_id=?"
108
+ args = (user_id,)
109
+ rows = self.store.conn.execute(
110
+ f"SELECT subject, relation, value FROM facts WHERE {where}",
111
+ args).fetchall()
112
+ subj_rels: dict[str, set[str]] = defaultdict(set)
113
+ subj_vals: dict[str, dict[str, str]] = defaultdict(dict)
114
+ rel_subjects: dict[str, set[str]] = defaultdict(set)
115
+ rel_values: dict[str, list[str]] = defaultdict(list)
116
+ for r in rows:
117
+ s, rel, v = r[0], r[1], r[2]
118
+ subj_rels[s].add(rel)
119
+ subj_vals[s][rel] = v
120
+ rel_subjects[rel].add(s)
121
+ rel_values[rel].append(v)
122
+
123
+ # for each relation that exists in the trace, find subjects
124
+ # that DON'T have it but have at least one co-occurring relation
125
+ # with peers that DO have it.
126
+ gaps: list[Gap] = []
127
+ all_subjects = set(subj_rels.keys())
128
+ for rel, peers in rel_subjects.items():
129
+ if len(peers) < self.min_peers_with_relation:
130
+ continue
131
+ missing_subjects = all_subjects - peers
132
+ for s in missing_subjects:
133
+ # check overlap with peers — at least one shared relation
134
+ my_rels = subj_rels[s]
135
+ peer_rels_union = set()
136
+ for p in peers:
137
+ peer_rels_union |= subj_rels[p]
138
+ overlap = my_rels & peer_rels_union
139
+ if len(overlap) < 1:
140
+ continue # not a peer of these — different category
141
+ gaps.append(Gap(
142
+ subject=s,
143
+ missing_relation=rel,
144
+ peer_count=len(peers),
145
+ peer_values=sorted(set(rel_values[rel]))[:5],
146
+ basis="majority"))
147
+
148
+ # structural gaps: for relations that form chain signatures
149
+ # (e.g. father → father = grandfather), check each subject that
150
+ # has the first hop — does it have the result of the second hop
151
+ # recorded? If not, that's a structural gap.
152
+ for p in scan.patterns:
153
+ if p.kind != "relation_pair":
154
+ continue
155
+ chain_rel = p.payload["relation"]
156
+ for ex in p.payload.get("examples", []):
157
+ # ex = (start, mid, end). The subject `start` has
158
+ # chain_rel=start→mid, mid→end. Hypothesize that
159
+ # start has a chain_rel_2 value of end.
160
+ # This is reported as a structural gap so the
161
+ # HypothesisEngine can propose (start, chain_rel*2, end).
162
+ if len(ex) != 3:
163
+ continue
164
+ start, mid, end = ex
165
+ # check if start already has the inferred composite
166
+ # relation — we model the composite as the chain
167
+ # relation with the value being the chain endpoint.
168
+ # Specifically: a hypothesis fact (start, chain_rel,
169
+ # end) is misleading because start already has
170
+ # chain_rel=mid. We use a synthetic relation name.
171
+ composite_rel = f"{chain_rel}*{chain_rel}"
172
+ existing = self.store.conn.execute(
173
+ "SELECT 1 FROM facts WHERE subject=? AND relation=? "
174
+ "AND value=? AND is_active=1 LIMIT 1",
175
+ (start, composite_rel, end)).fetchone()
176
+ if existing:
177
+ continue # already known — not a gap
178
+ gaps.append(Gap(
179
+ subject=start,
180
+ missing_relation=composite_rel,
181
+ peer_count=p.support,
182
+ peer_values=[end],
183
+ basis="structural"))
184
+
185
+ return GapResult(
186
+ gaps=gaps,
187
+ n_subjects_compared=len(all_subjects),
188
+ duration_ms=(time.perf_counter() - t0) * 1000.0)
189
+
190
+
191
+ class HypothesisEngine:
192
+ """Proposes fillers for gaps via three strategies."""
193
+
194
+ MAX_HYPOTHESES_PER_GAP: int = 1
195
+ MAX_HYPOTHESIS_CONFIDENCE: float = 0.45
196
+
197
+ def __init__(self, store: TraceStore,
198
+ palace=None,
199
+ max_confidence: float | None = None) -> None:
200
+ self.store = store
201
+ self.palace = palace # optional, for Hopfield cleanup path
202
+ self.max_conf = max_confidence or self.MAX_HYPOTHESIS_CONFIDENCE
203
+
204
+ def run(self, gaps: list[Gap], *,
205
+ dry_run: bool = False,
206
+ commit_id: str | None = None,
207
+ user_id: str | None = None) -> HypothesisResult:
208
+ """Propose and write hypothesis facts for each gap."""
209
+ import time
210
+ t0 = time.perf_counter()
211
+
212
+ hypotheses: list[Hypothesis] = []
213
+ facts_added = 0
214
+ ts = iso(_now()) if not dry_run else ""
215
+
216
+ for gap in gaps:
217
+ proposed = self._propose(gap, user_id)
218
+ if not proposed:
219
+ continue
220
+ hypotheses.append(proposed)
221
+
222
+ if dry_run:
223
+ continue
224
+
225
+ # write the hypothesis as a derived fact
226
+ fid = new_id()
227
+ self.store.conn.execute(
228
+ "INSERT INTO facts "
229
+ "(id, subject, relation, value, valid_from, tx_from, "
230
+ " confidence, user_id, memory_type, is_derived, "
231
+ " is_active, birth_commit, provenance) "
232
+ "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?)",
233
+ (fid, proposed.subject, proposed.relation,
234
+ proposed.proposed_value, ts, ts,
235
+ proposed.confidence,
236
+ user_id or "default",
237
+ "long_term", 1, 1, commit_id,
238
+ json.dumps({
239
+ "kind": "hypothesis",
240
+ "basis": proposed.basis,
241
+ "supporting_facts": proposed.supporting_facts,
242
+ "peer_count": gap.peer_count,
243
+ "peer_values_sample": gap.peer_values,
244
+ "generated_by": "cognition.hypothesis",
245
+ })))
246
+ # wire HYPOTHESIZED_BY edges from each supporting fact
247
+ for support_fid in proposed.supporting_facts:
248
+ if support_fid:
249
+ try:
250
+ self.store.add_edge(
251
+ fid, support_fid, HYPOTHESIZED_BY,
252
+ {"basis": proposed.basis})
253
+ except Exception:
254
+ pass
255
+ proposed.fact_id = fid
256
+ facts_added += 1
257
+
258
+ if not dry_run and commit_id and facts_added:
259
+ self.store.update_commit_n_facts(commit_id, facts_added)
260
+
261
+ return HypothesisResult(
262
+ hypotheses=hypotheses,
263
+ facts_added=facts_added,
264
+ duration_ms=(time.perf_counter() - t0) * 1000.0)
265
+
266
+ # ---------- strategies -------------------------------------------
267
+ def _propose(self, gap: Gap, user_id: str | None) -> Hypothesis | None:
268
+ # structural: peer_values are exact (end of chain)
269
+ if gap.basis == "structural" and gap.peer_values:
270
+ value = gap.peer_values[0]
271
+ supporting = self._lookup_supporting_facts(
272
+ gap.subject, gap.missing_relation, value, user_id)
273
+ return Hypothesis(
274
+ subject=gap.subject,
275
+ relation=gap.missing_relation,
276
+ proposed_value=value,
277
+ confidence=min(self.max_conf, 0.20 + 0.05 * gap.peer_count),
278
+ basis="structural",
279
+ supporting_facts=supporting,
280
+ )
281
+
282
+ # majority: most common value among peers
283
+ if gap.basis == "majority" and gap.peer_values:
284
+ # majority among peers — for categorical relations like
285
+ # `has_skill:rust` or `lives_in:toronto`, the most common
286
+ # peer value is a reasonable guess. For unique values
287
+ # like `works_at` (each person works at one company), the
288
+ # majority strategy is weaker — we still emit a hypothesis
289
+ # but at much lower confidence.
290
+ counter: dict[str, int] = defaultdict(int)
291
+ where = "is_active=1 AND quarantined=0 AND relation=?"
292
+ args: tuple = (gap.missing_relation,)
293
+ if user_id is not None:
294
+ where += " AND user_id=?"
295
+ args = (gap.missing_relation, user_id)
296
+ rows = self.store.conn.execute(
297
+ f"SELECT value FROM facts WHERE {where}", args).fetchall()
298
+ for r in rows:
299
+ counter[r[0]] += 1
300
+ if not counter:
301
+ return None
302
+ most_common = max(counter.items(), key=lambda x: x[1])
303
+ value = most_common[0]
304
+ n_peers_with_value = most_common[1]
305
+ total_peers = sum(counter.values())
306
+ conf = min(self.max_conf,
307
+ 0.05 + 0.05 * (n_peers_with_value / max(1, total_peers)))
308
+ supporting = self._lookup_supporting_facts(
309
+ gap.subject, gap.missing_relation, value, user_id)
310
+ return Hypothesis(
311
+ subject=gap.subject,
312
+ relation=gap.missing_relation,
313
+ proposed_value=value,
314
+ confidence=conf,
315
+ basis="majority",
316
+ supporting_facts=supporting,
317
+ )
318
+
319
+ return None
320
+
321
+ def _lookup_supporting_facts(self, subject: str, relation: str,
322
+ value: str,
323
+ user_id: str | None) -> list[str]:
324
+ """Look up fact IDs that support a hypothesis.
325
+
326
+ For structural hypotheses (father→father), supporting facts
327
+ are the two hops (Alice father Bob, Bob father Charles).
328
+ """
329
+ ids: list[str] = []
330
+ if relation.endswith("*" + relation.split("*")[0]):
331
+ # composite chain — look up the two hops
332
+ base_rel = relation.split("*")[0]
333
+ cur = subject
334
+ for _ in range(2):
335
+ rows = self.store.conn.execute(
336
+ "SELECT id, value FROM facts WHERE subject=? "
337
+ "AND relation=? AND is_active=1 "
338
+ "AND quarantined=0 LIMIT 1",
339
+ (cur, base_rel)).fetchall()
340
+ if not rows:
341
+ break
342
+ ids.append(rows[0][0])
343
+ cur = rows[0][1]
344
+ if cur == value:
345
+ break
346
+ else:
347
+ rows = self.store.conn.execute(
348
+ "SELECT id FROM facts WHERE relation=? AND value=? "
349
+ "AND is_active=1 AND quarantined=0 LIMIT 5",
350
+ (relation, value)).fetchall()
351
+ ids = [r[0] for r in rows]
352
+ return ids
353
+
354
+
355
+ def _now():
356
+ from datetime import datetime, timezone
357
+ return datetime.now(timezone.utc)
358
+
359
+
360
+ __all__ = [
361
+ "GapDetector", "HypothesisEngine",
362
+ "Gap", "Hypothesis",
363
+ "GapResult", "HypothesisResult",
364
+ "HYPOTHESIZED_BY",
365
+ ]
@@ -0,0 +1,204 @@
1
+ """PatternScanner — surfaces structural regularities across triples.
2
+
3
+ The first stage of the HMS cognition engine. Scans the Trace for
4
+ recurring patterns in the (subject, relation, value) graph and reports:
5
+
6
+ - Relation frequency histogram (which relations are common)
7
+ - Subject fan-out (entities with many relations — hubs)
8
+ - Value fan-out (values that appear for many subjects — categories)
9
+ - Co-occurring relations (e.g. works_at + lives_in co-occur on
10
+ 80% of subjects → suggest a 'person' abstraction)
11
+ - Relation pair signatures (e.g. father→father appears 12 times
12
+ → suggest a 'grandfather' composite relation)
13
+
14
+ Output is a list of `Pattern` records stored in the engine's working
15
+ state. The next stage (AbstractionEngine) consumes these patterns to
16
+ build prototype categories.
17
+
18
+ This module is read-only against the Trace — it never writes. The
19
+ HypothesisEngine later writes HYPOTHESIZED_BY edges; the scanner only
20
+ OBSERVES.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ from collections import Counter, defaultdict
26
+ from dataclasses import dataclass, field
27
+ from typing import Any
28
+
29
+ from cortexm.trace.store import TraceStore
30
+
31
+
32
+ @dataclass
33
+ class Pattern:
34
+ """A structural regularity the scanner surfaced."""
35
+ kind: str # "relation_freq" | "subject_fanout" |
36
+ # "value_fanout" | "co_occur" | "relation_pair"
37
+ payload: dict # kind-specific fields
38
+ support: int # how many facts support this pattern
39
+ confidence: float = 0.0 # 0..1 — strength of the regularity
40
+
41
+
42
+ @dataclass
43
+ class ScanResult:
44
+ """Result of a single PatternScanner.run() pass."""
45
+ patterns: list[Pattern] = field(default_factory=list)
46
+ n_facts_scanned: int = 0
47
+ n_relations: int = 0
48
+ n_subjects: int = 0
49
+ n_values: int = 0
50
+ duration_ms: float = 0.0
51
+
52
+
53
+ class PatternScanner:
54
+ """Surfaces structural regularities across the Trace."""
55
+
56
+ # minimum support for a pattern to be considered significant
57
+ MIN_SUPPORT: int = 2
58
+ # minimum co-occurrence fraction (0..1) for a co-occur pattern
59
+ CO_OCCUR_MIN: float = 0.30
60
+
61
+ def __init__(self, store: TraceStore,
62
+ min_support: int | None = None,
63
+ co_occur_min: float | None = None) -> None:
64
+ self.store = store
65
+ self.min_support = min_support or self.MIN_SUPPORT
66
+ self.co_occur_min = co_occur_min or self.CO_OCCUR_MIN
67
+
68
+ def run(self, user_id: str | None = None) -> ScanResult:
69
+ """Scan the Trace and return structural patterns."""
70
+ import time
71
+ t0 = time.perf_counter()
72
+ where = "is_active=1 AND quarantined=0"
73
+ args: tuple = ()
74
+ if user_id is not None:
75
+ where += " AND user_id=?"
76
+ args = (user_id,)
77
+ rows = self.store.conn.execute(
78
+ f"SELECT subject, relation, value FROM facts WHERE {where}",
79
+ args).fetchall()
80
+
81
+ rel_counter: Counter = Counter()
82
+ subj_rels: defaultdict[str, set[str]] = defaultdict(set)
83
+ subj_vals: defaultdict[str, set[str]] = defaultdict(set)
84
+ val_subjs: defaultdict[str, set[str]] = defaultdict(set)
85
+ pair_counter: Counter = Counter()
86
+ pair_examples: dict[tuple[str, str], list[str]] = {}
87
+
88
+ subjects: list[str] = []
89
+ for r in rows:
90
+ s, rel, v = r[0], r[1], r[2]
91
+ rel_counter[rel] += 1
92
+ subj_rels[s].add(rel)
93
+ subj_vals[s].add(v)
94
+ val_subjs[v].add(s)
95
+ subjects.append(s)
96
+
97
+ # relation pair signatures — for each subject, look at all
98
+ # ordered pairs of relations it has, then for each pair count
99
+ # how often a subject has both (this is the basis for the
100
+ # "co_occur" pattern). We also count explicit pair chains like
101
+ # (father, father) which signals a grandfather composite.
102
+ seen_subj: set[str] = set()
103
+ for s in subjects:
104
+ if s in seen_subj:
105
+ continue
106
+ seen_subj.add(s)
107
+ rels = list(subj_rels[s])
108
+ for i, r1 in enumerate(rels):
109
+ for r2 in rels[i + 1:]:
110
+ pair_counter[(r1, r2)] += 1
111
+ pair_counter[(r2, r1)] += 1
112
+ pair_examples.setdefault((r1, r2), []).append(s)
113
+
114
+ # check for self-pair chains (r1 followed by r1 on a different
115
+ # subject — e.g. father(A, B), father(B, C) means A's grandfather
116
+ # is C). This requires walking the value-as-next-subject chain.
117
+ # Implementation: for each (subj, rel, val), check if there's
118
+ # another fact with subj=val, rel=rel. If so, we have a 2-hop
119
+ # chain and (rel, rel) is a relation_pair pattern.
120
+ chain_examples: dict[str, list[tuple[str, str, str]]] = {}
121
+ chain_counter: Counter = Counter()
122
+ for r in rows:
123
+ s, rel, v = r[0], r[1], r[2]
124
+ # does (v, rel, ?) exist?
125
+ sub = self.store.conn.execute(
126
+ "SELECT value FROM facts WHERE subject=? AND relation=? "
127
+ "AND is_active=1 AND quarantined=0 LIMIT 5",
128
+ (v, rel)).fetchall()
129
+ for sr in sub:
130
+ chain_counter[rel] += 1
131
+ chain_examples.setdefault(rel, []).append(
132
+ (s, v, sr[0]))
133
+
134
+ patterns: list[Pattern] = []
135
+
136
+ # ---- 1. relation_freq ----
137
+ for rel, cnt in rel_counter.most_common():
138
+ if cnt >= self.min_support:
139
+ patterns.append(Pattern(
140
+ kind="relation_freq",
141
+ payload={"relation": rel},
142
+ support=cnt,
143
+ confidence=min(1.0, cnt / max(1, len(rows)))))
144
+
145
+ # ---- 2. subject_fanout ----
146
+ subj_fanout = sorted(
147
+ ((s, len(rels)) for s, rels in subj_rels.items()),
148
+ key=lambda x: -x[1])
149
+ for s, cnt in subj_fanout:
150
+ if cnt >= max(2, self.min_support):
151
+ patterns.append(Pattern(
152
+ kind="subject_fanout",
153
+ payload={"subject": s, "relations": sorted(subj_rels[s])},
154
+ support=cnt,
155
+ confidence=min(1.0, cnt / 10.0))) # 10 rels = max
156
+
157
+ # ---- 3. value_fanout ----
158
+ val_fanout = sorted(
159
+ ((v, len(subjs)) for v, subjs in val_subjs.items()),
160
+ key=lambda x: -x[1])
161
+ for v, cnt in val_fanout:
162
+ if cnt >= max(2, self.min_support):
163
+ patterns.append(Pattern(
164
+ kind="value_fanout",
165
+ payload={"value": v, "subjects": sorted(val_subjs[v])},
166
+ support=cnt,
167
+ confidence=min(1.0, cnt / 20.0)))
168
+
169
+ # ---- 4. co_occur (relation pairs on same subject) ----
170
+ n_subjects = len(seen_subj)
171
+ for (r1, r2), cnt in pair_counter.most_common(20):
172
+ if cnt < self.min_support:
173
+ continue
174
+ frac = cnt / max(1, n_subjects)
175
+ if frac >= self.co_occur_min:
176
+ patterns.append(Pattern(
177
+ kind="co_occur",
178
+ payload={"rel_a": r1, "rel_b": r2,
179
+ "examples": pair_examples.get((r1, r2), [])[:5]},
180
+ support=cnt,
181
+ confidence=frac))
182
+
183
+ # ---- 5. relation_pair (chain signatures) ----
184
+ for rel, cnt in chain_counter.most_common():
185
+ if cnt >= self.min_support:
186
+ ex = chain_examples.get(rel, [])[:5]
187
+ patterns.append(Pattern(
188
+ kind="relation_pair",
189
+ payload={"relation": rel, "chain": [rel, rel],
190
+ "examples": ex},
191
+ support=cnt,
192
+ confidence=min(1.0, cnt / 10.0)))
193
+
194
+ return ScanResult(
195
+ patterns=patterns,
196
+ n_facts_scanned=len(rows),
197
+ n_relations=len(rel_counter),
198
+ n_subjects=n_subjects,
199
+ n_values=len(val_subjs),
200
+ duration_ms=(time.perf_counter() - t0) * 1000.0,
201
+ )
202
+
203
+
204
+ __all__ = ["PatternScanner", "Pattern", "ScanResult"]