cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,365 @@
|
|
|
1
|
+
"""GapDetector + HypothesisEngine — fill in missing relations.
|
|
2
|
+
|
|
3
|
+
Third stage of the HMS cognition engine. Two stages in one module
|
|
4
|
+
because they're tightly coupled:
|
|
5
|
+
|
|
6
|
+
GapDetector — for each entity, compare its set of relations
|
|
7
|
+
to its peers' sets. Missing relations that
|
|
8
|
+
peers have are gaps. For example:
|
|
9
|
+
Alice has {works_at, lives_in, has_skill}
|
|
10
|
+
Bob has {works_at, lives_in, has_skill, age}
|
|
11
|
+
→ Alice is missing an `age` fact — gap.
|
|
12
|
+
|
|
13
|
+
HypothesisEngine — for each gap, propose a filler. The filler can
|
|
14
|
+
come from:
|
|
15
|
+
(a) Majority vote: what value do peers of
|
|
16
|
+
the same abstraction have?
|
|
17
|
+
(b) Structural transitivity: if Alice's father
|
|
18
|
+
is Bob and Bob's father is Charles, the
|
|
19
|
+
engine hypothesizes Alice's grandfather is
|
|
20
|
+
Charles via the father→father chain.
|
|
21
|
+
(c) Hopfield cleanup: if a VSA palace is
|
|
22
|
+
available, query the superposition of
|
|
23
|
+
peers' values for that relation.
|
|
24
|
+
|
|
25
|
+
Hypotheses are written as facts with:
|
|
26
|
+
is_derived=1
|
|
27
|
+
confidence < 0.5
|
|
28
|
+
provenance.kind = "hypothesis"
|
|
29
|
+
provenance.source = the basis of the guess
|
|
30
|
+
|
|
31
|
+
The HYPOTHESIZED_BY edge links the supporting fact(s) to the new
|
|
32
|
+
hypothesis fact, so audits can trace the reasoning chain.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
from __future__ import annotations
|
|
36
|
+
|
|
37
|
+
import json
|
|
38
|
+
from collections import defaultdict
|
|
39
|
+
from dataclasses import dataclass, field
|
|
40
|
+
from datetime import datetime, timezone
|
|
41
|
+
from typing import Any
|
|
42
|
+
|
|
43
|
+
from cortexm.cognition.abstraction import Abstraction
|
|
44
|
+
from cortexm.cognition.scanner import ScanResult
|
|
45
|
+
from cortexm.trace.store import TraceStore
|
|
46
|
+
from cortexm.util import iso, new_id
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
# Edge kind used to wire supporting facts to hypotheses
|
|
50
|
+
HYPOTHESIZED_BY = "HYPOTHESIZED_BY"
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass
|
|
54
|
+
class Gap:
|
|
55
|
+
"""A missing relation for an entity."""
|
|
56
|
+
subject: str
|
|
57
|
+
missing_relation: str
|
|
58
|
+
peer_count: int = 0 # how many peers have this relation
|
|
59
|
+
peer_values: list[str] = field(default_factory=list)
|
|
60
|
+
basis: str = "" # "majority" | "structural" | "cleanup"
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass
|
|
64
|
+
class Hypothesis:
|
|
65
|
+
"""A proposed filler for a gap."""
|
|
66
|
+
subject: str
|
|
67
|
+
relation: str
|
|
68
|
+
proposed_value: str
|
|
69
|
+
confidence: float = 0.0
|
|
70
|
+
basis: str = "" # which strategy proposed this
|
|
71
|
+
supporting_facts: list[str] = field(default_factory=list)
|
|
72
|
+
fact_id: str = ""
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass
|
|
76
|
+
class GapResult:
|
|
77
|
+
gaps: list[Gap] = field(default_factory=list)
|
|
78
|
+
n_subjects_compared: int = 0
|
|
79
|
+
duration_ms: float = 0.0
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass
|
|
83
|
+
class HypothesisResult:
|
|
84
|
+
hypotheses: list[Hypothesis] = field(default_factory=list)
|
|
85
|
+
facts_added: int = 0
|
|
86
|
+
duration_ms: float = 0.0
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
class GapDetector:
|
|
90
|
+
"""Finds missing relations by comparing entity profiles to peers."""
|
|
91
|
+
|
|
92
|
+
def __init__(self, store: TraceStore,
|
|
93
|
+
min_peers_with_relation: int = 2) -> None:
|
|
94
|
+
self.store = store
|
|
95
|
+
self.min_peers_with_relation = min_peers_with_relation
|
|
96
|
+
|
|
97
|
+
def run(self, scan: ScanResult,
|
|
98
|
+
user_id: str | None = None) -> GapResult:
|
|
99
|
+
"""Compare each subject's relation set to its peers' sets."""
|
|
100
|
+
import time
|
|
101
|
+
t0 = time.perf_counter()
|
|
102
|
+
|
|
103
|
+
# build per-subject relation sets from the trace
|
|
104
|
+
where = "is_active=1 AND quarantined=0"
|
|
105
|
+
args: tuple = ()
|
|
106
|
+
if user_id is not None:
|
|
107
|
+
where += " AND user_id=?"
|
|
108
|
+
args = (user_id,)
|
|
109
|
+
rows = self.store.conn.execute(
|
|
110
|
+
f"SELECT subject, relation, value FROM facts WHERE {where}",
|
|
111
|
+
args).fetchall()
|
|
112
|
+
subj_rels: dict[str, set[str]] = defaultdict(set)
|
|
113
|
+
subj_vals: dict[str, dict[str, str]] = defaultdict(dict)
|
|
114
|
+
rel_subjects: dict[str, set[str]] = defaultdict(set)
|
|
115
|
+
rel_values: dict[str, list[str]] = defaultdict(list)
|
|
116
|
+
for r in rows:
|
|
117
|
+
s, rel, v = r[0], r[1], r[2]
|
|
118
|
+
subj_rels[s].add(rel)
|
|
119
|
+
subj_vals[s][rel] = v
|
|
120
|
+
rel_subjects[rel].add(s)
|
|
121
|
+
rel_values[rel].append(v)
|
|
122
|
+
|
|
123
|
+
# for each relation that exists in the trace, find subjects
|
|
124
|
+
# that DON'T have it but have at least one co-occurring relation
|
|
125
|
+
# with peers that DO have it.
|
|
126
|
+
gaps: list[Gap] = []
|
|
127
|
+
all_subjects = set(subj_rels.keys())
|
|
128
|
+
for rel, peers in rel_subjects.items():
|
|
129
|
+
if len(peers) < self.min_peers_with_relation:
|
|
130
|
+
continue
|
|
131
|
+
missing_subjects = all_subjects - peers
|
|
132
|
+
for s in missing_subjects:
|
|
133
|
+
# check overlap with peers — at least one shared relation
|
|
134
|
+
my_rels = subj_rels[s]
|
|
135
|
+
peer_rels_union = set()
|
|
136
|
+
for p in peers:
|
|
137
|
+
peer_rels_union |= subj_rels[p]
|
|
138
|
+
overlap = my_rels & peer_rels_union
|
|
139
|
+
if len(overlap) < 1:
|
|
140
|
+
continue # not a peer of these — different category
|
|
141
|
+
gaps.append(Gap(
|
|
142
|
+
subject=s,
|
|
143
|
+
missing_relation=rel,
|
|
144
|
+
peer_count=len(peers),
|
|
145
|
+
peer_values=sorted(set(rel_values[rel]))[:5],
|
|
146
|
+
basis="majority"))
|
|
147
|
+
|
|
148
|
+
# structural gaps: for relations that form chain signatures
|
|
149
|
+
# (e.g. father → father = grandfather), check each subject that
|
|
150
|
+
# has the first hop — does it have the result of the second hop
|
|
151
|
+
# recorded? If not, that's a structural gap.
|
|
152
|
+
for p in scan.patterns:
|
|
153
|
+
if p.kind != "relation_pair":
|
|
154
|
+
continue
|
|
155
|
+
chain_rel = p.payload["relation"]
|
|
156
|
+
for ex in p.payload.get("examples", []):
|
|
157
|
+
# ex = (start, mid, end). The subject `start` has
|
|
158
|
+
# chain_rel=start→mid, mid→end. Hypothesize that
|
|
159
|
+
# start has a chain_rel_2 value of end.
|
|
160
|
+
# This is reported as a structural gap so the
|
|
161
|
+
# HypothesisEngine can propose (start, chain_rel*2, end).
|
|
162
|
+
if len(ex) != 3:
|
|
163
|
+
continue
|
|
164
|
+
start, mid, end = ex
|
|
165
|
+
# check if start already has the inferred composite
|
|
166
|
+
# relation — we model the composite as the chain
|
|
167
|
+
# relation with the value being the chain endpoint.
|
|
168
|
+
# Specifically: a hypothesis fact (start, chain_rel,
|
|
169
|
+
# end) is misleading because start already has
|
|
170
|
+
# chain_rel=mid. We use a synthetic relation name.
|
|
171
|
+
composite_rel = f"{chain_rel}*{chain_rel}"
|
|
172
|
+
existing = self.store.conn.execute(
|
|
173
|
+
"SELECT 1 FROM facts WHERE subject=? AND relation=? "
|
|
174
|
+
"AND value=? AND is_active=1 LIMIT 1",
|
|
175
|
+
(start, composite_rel, end)).fetchone()
|
|
176
|
+
if existing:
|
|
177
|
+
continue # already known — not a gap
|
|
178
|
+
gaps.append(Gap(
|
|
179
|
+
subject=start,
|
|
180
|
+
missing_relation=composite_rel,
|
|
181
|
+
peer_count=p.support,
|
|
182
|
+
peer_values=[end],
|
|
183
|
+
basis="structural"))
|
|
184
|
+
|
|
185
|
+
return GapResult(
|
|
186
|
+
gaps=gaps,
|
|
187
|
+
n_subjects_compared=len(all_subjects),
|
|
188
|
+
duration_ms=(time.perf_counter() - t0) * 1000.0)
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
class HypothesisEngine:
|
|
192
|
+
"""Proposes fillers for gaps via three strategies."""
|
|
193
|
+
|
|
194
|
+
MAX_HYPOTHESES_PER_GAP: int = 1
|
|
195
|
+
MAX_HYPOTHESIS_CONFIDENCE: float = 0.45
|
|
196
|
+
|
|
197
|
+
def __init__(self, store: TraceStore,
|
|
198
|
+
palace=None,
|
|
199
|
+
max_confidence: float | None = None) -> None:
|
|
200
|
+
self.store = store
|
|
201
|
+
self.palace = palace # optional, for Hopfield cleanup path
|
|
202
|
+
self.max_conf = max_confidence or self.MAX_HYPOTHESIS_CONFIDENCE
|
|
203
|
+
|
|
204
|
+
def run(self, gaps: list[Gap], *,
|
|
205
|
+
dry_run: bool = False,
|
|
206
|
+
commit_id: str | None = None,
|
|
207
|
+
user_id: str | None = None) -> HypothesisResult:
|
|
208
|
+
"""Propose and write hypothesis facts for each gap."""
|
|
209
|
+
import time
|
|
210
|
+
t0 = time.perf_counter()
|
|
211
|
+
|
|
212
|
+
hypotheses: list[Hypothesis] = []
|
|
213
|
+
facts_added = 0
|
|
214
|
+
ts = iso(_now()) if not dry_run else ""
|
|
215
|
+
|
|
216
|
+
for gap in gaps:
|
|
217
|
+
proposed = self._propose(gap, user_id)
|
|
218
|
+
if not proposed:
|
|
219
|
+
continue
|
|
220
|
+
hypotheses.append(proposed)
|
|
221
|
+
|
|
222
|
+
if dry_run:
|
|
223
|
+
continue
|
|
224
|
+
|
|
225
|
+
# write the hypothesis as a derived fact
|
|
226
|
+
fid = new_id()
|
|
227
|
+
self.store.conn.execute(
|
|
228
|
+
"INSERT INTO facts "
|
|
229
|
+
"(id, subject, relation, value, valid_from, tx_from, "
|
|
230
|
+
" confidence, user_id, memory_type, is_derived, "
|
|
231
|
+
" is_active, birth_commit, provenance) "
|
|
232
|
+
"VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?)",
|
|
233
|
+
(fid, proposed.subject, proposed.relation,
|
|
234
|
+
proposed.proposed_value, ts, ts,
|
|
235
|
+
proposed.confidence,
|
|
236
|
+
user_id or "default",
|
|
237
|
+
"long_term", 1, 1, commit_id,
|
|
238
|
+
json.dumps({
|
|
239
|
+
"kind": "hypothesis",
|
|
240
|
+
"basis": proposed.basis,
|
|
241
|
+
"supporting_facts": proposed.supporting_facts,
|
|
242
|
+
"peer_count": gap.peer_count,
|
|
243
|
+
"peer_values_sample": gap.peer_values,
|
|
244
|
+
"generated_by": "cognition.hypothesis",
|
|
245
|
+
})))
|
|
246
|
+
# wire HYPOTHESIZED_BY edges from each supporting fact
|
|
247
|
+
for support_fid in proposed.supporting_facts:
|
|
248
|
+
if support_fid:
|
|
249
|
+
try:
|
|
250
|
+
self.store.add_edge(
|
|
251
|
+
fid, support_fid, HYPOTHESIZED_BY,
|
|
252
|
+
{"basis": proposed.basis})
|
|
253
|
+
except Exception:
|
|
254
|
+
pass
|
|
255
|
+
proposed.fact_id = fid
|
|
256
|
+
facts_added += 1
|
|
257
|
+
|
|
258
|
+
if not dry_run and commit_id and facts_added:
|
|
259
|
+
self.store.update_commit_n_facts(commit_id, facts_added)
|
|
260
|
+
|
|
261
|
+
return HypothesisResult(
|
|
262
|
+
hypotheses=hypotheses,
|
|
263
|
+
facts_added=facts_added,
|
|
264
|
+
duration_ms=(time.perf_counter() - t0) * 1000.0)
|
|
265
|
+
|
|
266
|
+
# ---------- strategies -------------------------------------------
|
|
267
|
+
def _propose(self, gap: Gap, user_id: str | None) -> Hypothesis | None:
|
|
268
|
+
# structural: peer_values are exact (end of chain)
|
|
269
|
+
if gap.basis == "structural" and gap.peer_values:
|
|
270
|
+
value = gap.peer_values[0]
|
|
271
|
+
supporting = self._lookup_supporting_facts(
|
|
272
|
+
gap.subject, gap.missing_relation, value, user_id)
|
|
273
|
+
return Hypothesis(
|
|
274
|
+
subject=gap.subject,
|
|
275
|
+
relation=gap.missing_relation,
|
|
276
|
+
proposed_value=value,
|
|
277
|
+
confidence=min(self.max_conf, 0.20 + 0.05 * gap.peer_count),
|
|
278
|
+
basis="structural",
|
|
279
|
+
supporting_facts=supporting,
|
|
280
|
+
)
|
|
281
|
+
|
|
282
|
+
# majority: most common value among peers
|
|
283
|
+
if gap.basis == "majority" and gap.peer_values:
|
|
284
|
+
# majority among peers — for categorical relations like
|
|
285
|
+
# `has_skill:rust` or `lives_in:toronto`, the most common
|
|
286
|
+
# peer value is a reasonable guess. For unique values
|
|
287
|
+
# like `works_at` (each person works at one company), the
|
|
288
|
+
# majority strategy is weaker — we still emit a hypothesis
|
|
289
|
+
# but at much lower confidence.
|
|
290
|
+
counter: dict[str, int] = defaultdict(int)
|
|
291
|
+
where = "is_active=1 AND quarantined=0 AND relation=?"
|
|
292
|
+
args: tuple = (gap.missing_relation,)
|
|
293
|
+
if user_id is not None:
|
|
294
|
+
where += " AND user_id=?"
|
|
295
|
+
args = (gap.missing_relation, user_id)
|
|
296
|
+
rows = self.store.conn.execute(
|
|
297
|
+
f"SELECT value FROM facts WHERE {where}", args).fetchall()
|
|
298
|
+
for r in rows:
|
|
299
|
+
counter[r[0]] += 1
|
|
300
|
+
if not counter:
|
|
301
|
+
return None
|
|
302
|
+
most_common = max(counter.items(), key=lambda x: x[1])
|
|
303
|
+
value = most_common[0]
|
|
304
|
+
n_peers_with_value = most_common[1]
|
|
305
|
+
total_peers = sum(counter.values())
|
|
306
|
+
conf = min(self.max_conf,
|
|
307
|
+
0.05 + 0.05 * (n_peers_with_value / max(1, total_peers)))
|
|
308
|
+
supporting = self._lookup_supporting_facts(
|
|
309
|
+
gap.subject, gap.missing_relation, value, user_id)
|
|
310
|
+
return Hypothesis(
|
|
311
|
+
subject=gap.subject,
|
|
312
|
+
relation=gap.missing_relation,
|
|
313
|
+
proposed_value=value,
|
|
314
|
+
confidence=conf,
|
|
315
|
+
basis="majority",
|
|
316
|
+
supporting_facts=supporting,
|
|
317
|
+
)
|
|
318
|
+
|
|
319
|
+
return None
|
|
320
|
+
|
|
321
|
+
def _lookup_supporting_facts(self, subject: str, relation: str,
|
|
322
|
+
value: str,
|
|
323
|
+
user_id: str | None) -> list[str]:
|
|
324
|
+
"""Look up fact IDs that support a hypothesis.
|
|
325
|
+
|
|
326
|
+
For structural hypotheses (father→father), supporting facts
|
|
327
|
+
are the two hops (Alice father Bob, Bob father Charles).
|
|
328
|
+
"""
|
|
329
|
+
ids: list[str] = []
|
|
330
|
+
if relation.endswith("*" + relation.split("*")[0]):
|
|
331
|
+
# composite chain — look up the two hops
|
|
332
|
+
base_rel = relation.split("*")[0]
|
|
333
|
+
cur = subject
|
|
334
|
+
for _ in range(2):
|
|
335
|
+
rows = self.store.conn.execute(
|
|
336
|
+
"SELECT id, value FROM facts WHERE subject=? "
|
|
337
|
+
"AND relation=? AND is_active=1 "
|
|
338
|
+
"AND quarantined=0 LIMIT 1",
|
|
339
|
+
(cur, base_rel)).fetchall()
|
|
340
|
+
if not rows:
|
|
341
|
+
break
|
|
342
|
+
ids.append(rows[0][0])
|
|
343
|
+
cur = rows[0][1]
|
|
344
|
+
if cur == value:
|
|
345
|
+
break
|
|
346
|
+
else:
|
|
347
|
+
rows = self.store.conn.execute(
|
|
348
|
+
"SELECT id FROM facts WHERE relation=? AND value=? "
|
|
349
|
+
"AND is_active=1 AND quarantined=0 LIMIT 5",
|
|
350
|
+
(relation, value)).fetchall()
|
|
351
|
+
ids = [r[0] for r in rows]
|
|
352
|
+
return ids
|
|
353
|
+
|
|
354
|
+
|
|
355
|
+
def _now():
|
|
356
|
+
from datetime import datetime, timezone
|
|
357
|
+
return datetime.now(timezone.utc)
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
__all__ = [
|
|
361
|
+
"GapDetector", "HypothesisEngine",
|
|
362
|
+
"Gap", "Hypothesis",
|
|
363
|
+
"GapResult", "HypothesisResult",
|
|
364
|
+
"HYPOTHESIZED_BY",
|
|
365
|
+
]
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
"""PatternScanner — surfaces structural regularities across triples.
|
|
2
|
+
|
|
3
|
+
The first stage of the HMS cognition engine. Scans the Trace for
|
|
4
|
+
recurring patterns in the (subject, relation, value) graph and reports:
|
|
5
|
+
|
|
6
|
+
- Relation frequency histogram (which relations are common)
|
|
7
|
+
- Subject fan-out (entities with many relations — hubs)
|
|
8
|
+
- Value fan-out (values that appear for many subjects — categories)
|
|
9
|
+
- Co-occurring relations (e.g. works_at + lives_in co-occur on
|
|
10
|
+
80% of subjects → suggest a 'person' abstraction)
|
|
11
|
+
- Relation pair signatures (e.g. father→father appears 12 times
|
|
12
|
+
→ suggest a 'grandfather' composite relation)
|
|
13
|
+
|
|
14
|
+
Output is a list of `Pattern` records stored in the engine's working
|
|
15
|
+
state. The next stage (AbstractionEngine) consumes these patterns to
|
|
16
|
+
build prototype categories.
|
|
17
|
+
|
|
18
|
+
This module is read-only against the Trace — it never writes. The
|
|
19
|
+
HypothesisEngine later writes HYPOTHESIZED_BY edges; the scanner only
|
|
20
|
+
OBSERVES.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
from collections import Counter, defaultdict
|
|
26
|
+
from dataclasses import dataclass, field
|
|
27
|
+
from typing import Any
|
|
28
|
+
|
|
29
|
+
from cortexm.trace.store import TraceStore
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class Pattern:
|
|
34
|
+
"""A structural regularity the scanner surfaced."""
|
|
35
|
+
kind: str # "relation_freq" | "subject_fanout" |
|
|
36
|
+
# "value_fanout" | "co_occur" | "relation_pair"
|
|
37
|
+
payload: dict # kind-specific fields
|
|
38
|
+
support: int # how many facts support this pattern
|
|
39
|
+
confidence: float = 0.0 # 0..1 — strength of the regularity
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class ScanResult:
|
|
44
|
+
"""Result of a single PatternScanner.run() pass."""
|
|
45
|
+
patterns: list[Pattern] = field(default_factory=list)
|
|
46
|
+
n_facts_scanned: int = 0
|
|
47
|
+
n_relations: int = 0
|
|
48
|
+
n_subjects: int = 0
|
|
49
|
+
n_values: int = 0
|
|
50
|
+
duration_ms: float = 0.0
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class PatternScanner:
|
|
54
|
+
"""Surfaces structural regularities across the Trace."""
|
|
55
|
+
|
|
56
|
+
# minimum support for a pattern to be considered significant
|
|
57
|
+
MIN_SUPPORT: int = 2
|
|
58
|
+
# minimum co-occurrence fraction (0..1) for a co-occur pattern
|
|
59
|
+
CO_OCCUR_MIN: float = 0.30
|
|
60
|
+
|
|
61
|
+
def __init__(self, store: TraceStore,
|
|
62
|
+
min_support: int | None = None,
|
|
63
|
+
co_occur_min: float | None = None) -> None:
|
|
64
|
+
self.store = store
|
|
65
|
+
self.min_support = min_support or self.MIN_SUPPORT
|
|
66
|
+
self.co_occur_min = co_occur_min or self.CO_OCCUR_MIN
|
|
67
|
+
|
|
68
|
+
def run(self, user_id: str | None = None) -> ScanResult:
|
|
69
|
+
"""Scan the Trace and return structural patterns."""
|
|
70
|
+
import time
|
|
71
|
+
t0 = time.perf_counter()
|
|
72
|
+
where = "is_active=1 AND quarantined=0"
|
|
73
|
+
args: tuple = ()
|
|
74
|
+
if user_id is not None:
|
|
75
|
+
where += " AND user_id=?"
|
|
76
|
+
args = (user_id,)
|
|
77
|
+
rows = self.store.conn.execute(
|
|
78
|
+
f"SELECT subject, relation, value FROM facts WHERE {where}",
|
|
79
|
+
args).fetchall()
|
|
80
|
+
|
|
81
|
+
rel_counter: Counter = Counter()
|
|
82
|
+
subj_rels: defaultdict[str, set[str]] = defaultdict(set)
|
|
83
|
+
subj_vals: defaultdict[str, set[str]] = defaultdict(set)
|
|
84
|
+
val_subjs: defaultdict[str, set[str]] = defaultdict(set)
|
|
85
|
+
pair_counter: Counter = Counter()
|
|
86
|
+
pair_examples: dict[tuple[str, str], list[str]] = {}
|
|
87
|
+
|
|
88
|
+
subjects: list[str] = []
|
|
89
|
+
for r in rows:
|
|
90
|
+
s, rel, v = r[0], r[1], r[2]
|
|
91
|
+
rel_counter[rel] += 1
|
|
92
|
+
subj_rels[s].add(rel)
|
|
93
|
+
subj_vals[s].add(v)
|
|
94
|
+
val_subjs[v].add(s)
|
|
95
|
+
subjects.append(s)
|
|
96
|
+
|
|
97
|
+
# relation pair signatures — for each subject, look at all
|
|
98
|
+
# ordered pairs of relations it has, then for each pair count
|
|
99
|
+
# how often a subject has both (this is the basis for the
|
|
100
|
+
# "co_occur" pattern). We also count explicit pair chains like
|
|
101
|
+
# (father, father) which signals a grandfather composite.
|
|
102
|
+
seen_subj: set[str] = set()
|
|
103
|
+
for s in subjects:
|
|
104
|
+
if s in seen_subj:
|
|
105
|
+
continue
|
|
106
|
+
seen_subj.add(s)
|
|
107
|
+
rels = list(subj_rels[s])
|
|
108
|
+
for i, r1 in enumerate(rels):
|
|
109
|
+
for r2 in rels[i + 1:]:
|
|
110
|
+
pair_counter[(r1, r2)] += 1
|
|
111
|
+
pair_counter[(r2, r1)] += 1
|
|
112
|
+
pair_examples.setdefault((r1, r2), []).append(s)
|
|
113
|
+
|
|
114
|
+
# check for self-pair chains (r1 followed by r1 on a different
|
|
115
|
+
# subject — e.g. father(A, B), father(B, C) means A's grandfather
|
|
116
|
+
# is C). This requires walking the value-as-next-subject chain.
|
|
117
|
+
# Implementation: for each (subj, rel, val), check if there's
|
|
118
|
+
# another fact with subj=val, rel=rel. If so, we have a 2-hop
|
|
119
|
+
# chain and (rel, rel) is a relation_pair pattern.
|
|
120
|
+
chain_examples: dict[str, list[tuple[str, str, str]]] = {}
|
|
121
|
+
chain_counter: Counter = Counter()
|
|
122
|
+
for r in rows:
|
|
123
|
+
s, rel, v = r[0], r[1], r[2]
|
|
124
|
+
# does (v, rel, ?) exist?
|
|
125
|
+
sub = self.store.conn.execute(
|
|
126
|
+
"SELECT value FROM facts WHERE subject=? AND relation=? "
|
|
127
|
+
"AND is_active=1 AND quarantined=0 LIMIT 5",
|
|
128
|
+
(v, rel)).fetchall()
|
|
129
|
+
for sr in sub:
|
|
130
|
+
chain_counter[rel] += 1
|
|
131
|
+
chain_examples.setdefault(rel, []).append(
|
|
132
|
+
(s, v, sr[0]))
|
|
133
|
+
|
|
134
|
+
patterns: list[Pattern] = []
|
|
135
|
+
|
|
136
|
+
# ---- 1. relation_freq ----
|
|
137
|
+
for rel, cnt in rel_counter.most_common():
|
|
138
|
+
if cnt >= self.min_support:
|
|
139
|
+
patterns.append(Pattern(
|
|
140
|
+
kind="relation_freq",
|
|
141
|
+
payload={"relation": rel},
|
|
142
|
+
support=cnt,
|
|
143
|
+
confidence=min(1.0, cnt / max(1, len(rows)))))
|
|
144
|
+
|
|
145
|
+
# ---- 2. subject_fanout ----
|
|
146
|
+
subj_fanout = sorted(
|
|
147
|
+
((s, len(rels)) for s, rels in subj_rels.items()),
|
|
148
|
+
key=lambda x: -x[1])
|
|
149
|
+
for s, cnt in subj_fanout:
|
|
150
|
+
if cnt >= max(2, self.min_support):
|
|
151
|
+
patterns.append(Pattern(
|
|
152
|
+
kind="subject_fanout",
|
|
153
|
+
payload={"subject": s, "relations": sorted(subj_rels[s])},
|
|
154
|
+
support=cnt,
|
|
155
|
+
confidence=min(1.0, cnt / 10.0))) # 10 rels = max
|
|
156
|
+
|
|
157
|
+
# ---- 3. value_fanout ----
|
|
158
|
+
val_fanout = sorted(
|
|
159
|
+
((v, len(subjs)) for v, subjs in val_subjs.items()),
|
|
160
|
+
key=lambda x: -x[1])
|
|
161
|
+
for v, cnt in val_fanout:
|
|
162
|
+
if cnt >= max(2, self.min_support):
|
|
163
|
+
patterns.append(Pattern(
|
|
164
|
+
kind="value_fanout",
|
|
165
|
+
payload={"value": v, "subjects": sorted(val_subjs[v])},
|
|
166
|
+
support=cnt,
|
|
167
|
+
confidence=min(1.0, cnt / 20.0)))
|
|
168
|
+
|
|
169
|
+
# ---- 4. co_occur (relation pairs on same subject) ----
|
|
170
|
+
n_subjects = len(seen_subj)
|
|
171
|
+
for (r1, r2), cnt in pair_counter.most_common(20):
|
|
172
|
+
if cnt < self.min_support:
|
|
173
|
+
continue
|
|
174
|
+
frac = cnt / max(1, n_subjects)
|
|
175
|
+
if frac >= self.co_occur_min:
|
|
176
|
+
patterns.append(Pattern(
|
|
177
|
+
kind="co_occur",
|
|
178
|
+
payload={"rel_a": r1, "rel_b": r2,
|
|
179
|
+
"examples": pair_examples.get((r1, r2), [])[:5]},
|
|
180
|
+
support=cnt,
|
|
181
|
+
confidence=frac))
|
|
182
|
+
|
|
183
|
+
# ---- 5. relation_pair (chain signatures) ----
|
|
184
|
+
for rel, cnt in chain_counter.most_common():
|
|
185
|
+
if cnt >= self.min_support:
|
|
186
|
+
ex = chain_examples.get(rel, [])[:5]
|
|
187
|
+
patterns.append(Pattern(
|
|
188
|
+
kind="relation_pair",
|
|
189
|
+
payload={"relation": rel, "chain": [rel, rel],
|
|
190
|
+
"examples": ex},
|
|
191
|
+
support=cnt,
|
|
192
|
+
confidence=min(1.0, cnt / 10.0)))
|
|
193
|
+
|
|
194
|
+
return ScanResult(
|
|
195
|
+
patterns=patterns,
|
|
196
|
+
n_facts_scanned=len(rows),
|
|
197
|
+
n_relations=len(rel_counter),
|
|
198
|
+
n_subjects=n_subjects,
|
|
199
|
+
n_values=len(val_subjs),
|
|
200
|
+
duration_ms=(time.perf_counter() - t0) * 1000.0,
|
|
201
|
+
)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
__all__ = ["PatternScanner", "Pattern", "ScanResult"]
|