cortexm 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- context_m.py +17 -0
- cortexm/__init__.py +45 -0
- cortexm/accel.py +403 -0
- cortexm/api/__init__.py +0 -0
- cortexm/api/chaos.py +118 -0
- cortexm/api/memory.py +635 -0
- cortexm/bench/__init__.py +0 -0
- cortexm/bench/abilities.py +311 -0
- cortexm/bench/baselines.py +89 -0
- cortexm/bench/beam_loader.py +317 -0
- cortexm/bench/generator.py +376 -0
- cortexm/bench/harness.py +211 -0
- cortexm/bench/messy.py +218 -0
- cortexm/bench/micro.py +251 -0
- cortexm/bench/ood.py +443 -0
- cortexm/bench/run.py +137 -0
- cortexm/bridge/__init__.py +0 -0
- cortexm/bridge/dates.py +178 -0
- cortexm/bridge/decoders.py +204 -0
- cortexm/bridge/enrich.py +255 -0
- cortexm/bridge/extractor.py +316 -0
- cortexm/bridge/fallback.py +332 -0
- cortexm/bridge/onnx_runtime.py +158 -0
- cortexm/bridge/patterns.py +760 -0
- cortexm/bridge/ppr.py +104 -0
- cortexm/bridge/prefilter.py +188 -0
- cortexm/bridge/query_extract.py +420 -0
- cortexm/bridge/reader.py +1174 -0
- cortexm/bridge/rerank.py +204 -0
- cortexm/bridge/writer.py +492 -0
- cortexm/cli.py +295 -0
- cortexm/cognition/__init__.py +53 -0
- cortexm/cognition/abstraction.py +192 -0
- cortexm/cognition/analogy.py +159 -0
- cortexm/cognition/engine.py +204 -0
- cortexm/cognition/gaps.py +365 -0
- cortexm/cognition/scanner.py +204 -0
- cortexm/config.py +375 -0
- cortexm/cortexm.py +8 -0
- cortexm/enterprise/__init__.py +0 -0
- cortexm/enterprise/audit.py +178 -0
- cortexm/enterprise/governance.py +239 -0
- cortexm/errors.py +35 -0
- cortexm/features/__init__.py +0 -0
- cortexm/features/git.py +204 -0
- cortexm/features/prefetch.py +88 -0
- cortexm/features/zk.py +105 -0
- cortexm/federation/__init__.py +39 -0
- cortexm/federation/crdt.py +275 -0
- cortexm/federation/fabric.py +109 -0
- cortexm/federation/hlc.py +80 -0
- cortexm/federation/node.py +145 -0
- cortexm/federation/schema_report.py +73 -0
- cortexm/federation/transport.py +164 -0
- cortexm/index/__init__.py +19 -0
- cortexm/index/nsg.py +386 -0
- cortexm/mcp/__init__.py +0 -0
- cortexm/mcp/server.py +985 -0
- cortexm/metrics.py +62 -0
- cortexm/migrate/__init__.py +0 -0
- cortexm/migrate/importers.py +192 -0
- cortexm/provenance/__init__.py +78 -0
- cortexm/provenance/agent.py +214 -0
- cortexm/provenance/cose.py +201 -0
- cortexm/provenance/scitt.py +258 -0
- cortexm/provenance/vc.py +250 -0
- cortexm/security/__init__.py +0 -0
- cortexm/security/crypto.py +162 -0
- cortexm/security/hashes.py +140 -0
- cortexm/security/injection.py +149 -0
- cortexm/security/mind.py +154 -0
- cortexm/security/pii.py +265 -0
- cortexm/security/rbac.py +169 -0
- cortexm/security/sandbox.py +131 -0
- cortexm/security/zk_hamming.py +142 -0
- cortexm/security/zk_sql.py +485 -0
- cortexm/server/__init__.py +0 -0
- cortexm/server/metrics.py +88 -0
- cortexm/server/rest.py +936 -0
- cortexm/server/sparql.py +984 -0
- cortexm/text/__init__.py +0 -0
- cortexm/text/dissim.py +252 -0
- cortexm/text/embedder.py +155 -0
- cortexm/text/fuzzy.py +218 -0
- cortexm/text/idiolect.py +253 -0
- cortexm/text/labse.py +374 -0
- cortexm/text/tokenizer.py +79 -0
- cortexm/trace/__init__.py +0 -0
- cortexm/trace/blob_arena.py +277 -0
- cortexm/trace/consolidate.py +337 -0
- cortexm/trace/contradictions.py +69 -0
- cortexm/trace/dedup.py +114 -0
- cortexm/trace/edges.py +214 -0
- cortexm/trace/fact.py +121 -0
- cortexm/trace/fade.py +245 -0
- cortexm/trace/lifecycle.py +112 -0
- cortexm/trace/rebuild.py +173 -0
- cortexm/trace/rules.py +171 -0
- cortexm/trace/store.py +680 -0
- cortexm/trace/structural.py +183 -0
- cortexm/trace/tmt.py +335 -0
- cortexm/util.py +148 -0
- cortexm/vsa/__init__.py +0 -0
- cortexm/vsa/attribution.py +149 -0
- cortexm/vsa/cleanup.py +161 -0
- cortexm/vsa/codecs.py +397 -0
- cortexm/vsa/hologram_overlay.py +139 -0
- cortexm/vsa/index.py +163 -0
- cortexm/vsa/ops.py +149 -0
- cortexm/vsa/palace.py +446 -0
- cortexm/vsa/role_vectors.py +236 -0
- cortexm/vsa/slb.py +78 -0
- cortexm/vsa/tlsh_trie.py +137 -0
- cortexm/vsa/working_memory.py +249 -0
- cortexm-0.3.0.dist-info/METADATA +482 -0
- cortexm-0.3.0.dist-info/RECORD +120 -0
- cortexm-0.3.0.dist-info/WHEEL +5 -0
- cortexm-0.3.0.dist-info/entry_points.txt +2 -0
- cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
- cortexm-0.3.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Deterministic contradiction detection & truth maintenance.
|
|
2
|
+
|
|
3
|
+
Phase rules from Section 1.1:
|
|
4
|
+
1. Classification — relation → semantic category, single/multi valued
|
|
5
|
+
2. Contradiction — exact + fuzzy (Jaccard/Levenshtein) match on
|
|
6
|
+
subject-relation pairs; latest-value-wins for
|
|
7
|
+
single-valued relations, append for multi-valued
|
|
8
|
+
3. Interference-aware lifecycle — see lifecycle.py
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
from enum import Enum
|
|
15
|
+
|
|
16
|
+
from cortexm.trace.fact import Fact, SINGLE_VALUED
|
|
17
|
+
from cortexm.trace.store import TraceStore
|
|
18
|
+
from cortexm.util import similarity
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class Action(str, Enum):
|
|
22
|
+
COMMIT = "commit" # brand-new fact
|
|
23
|
+
MERGE = "merge" # near-duplicate: reinforce existing
|
|
24
|
+
SUPERSEDE = "supersede" # contradiction on single-valued relation
|
|
25
|
+
COEXIST = "coexist" # contradiction on multi-valued relation
|
|
26
|
+
SKIP = "skip" # exact duplicate
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class Conflict:
|
|
31
|
+
action: Action
|
|
32
|
+
existing: list[Fact]
|
|
33
|
+
note: str = ""
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def find_conflicts(store: TraceStore, candidate: Fact) -> Conflict:
|
|
37
|
+
"""Decide how ``candidate`` interacts with active memory."""
|
|
38
|
+
existing = [
|
|
39
|
+
f for f in store.query_facts(
|
|
40
|
+
subject=candidate.subject, relation=candidate.relation,
|
|
41
|
+
user_id=candidate.user_id, active=True)
|
|
42
|
+
if f.id != candidate.id
|
|
43
|
+
]
|
|
44
|
+
if not existing:
|
|
45
|
+
return Conflict(Action.COMMIT, [])
|
|
46
|
+
|
|
47
|
+
exact = [f for f in existing if f.value.strip().lower() == candidate.value.strip().lower()]
|
|
48
|
+
if exact:
|
|
49
|
+
return Conflict(Action.SKIP, exact, "exact duplicate")
|
|
50
|
+
|
|
51
|
+
# low-salience mention anchors: exact-dup semantics only (no fuzzy
|
|
52
|
+
# quadratic scans — mention streams are high-volume, low-signal)
|
|
53
|
+
if candidate.relation in ("mentioned", "event", "instruction"):
|
|
54
|
+
if candidate.relation == "mentioned":
|
|
55
|
+
return Conflict(Action.COEXIST, existing, "mention anchor recorded")
|
|
56
|
+
|
|
57
|
+
near = [f for f in existing
|
|
58
|
+
if similarity(f.value, candidate.value) >= 0.92]
|
|
59
|
+
if near:
|
|
60
|
+
return Conflict(Action.MERGE, near, "near-duplicate merged; reinforcement +1")
|
|
61
|
+
|
|
62
|
+
single = candidate.relation in SINGLE_VALUED
|
|
63
|
+
if single:
|
|
64
|
+
# newest valid_from wins reality; old fact gets valid_to
|
|
65
|
+
target = max(existing, key=lambda f: (f.valid_from, f.tx_from))
|
|
66
|
+
return Conflict(Action.SUPERSEDE, [target],
|
|
67
|
+
f"contradiction on single-valued '{candidate.relation}'")
|
|
68
|
+
return Conflict(Action.COEXIST, existing,
|
|
69
|
+
f"conflicting values on multi-valued '{candidate.relation}' coexist")
|
cortexm/trace/dedup.py
ADDED
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""Deduplication + holographic compression audit.
|
|
2
|
+
|
|
3
|
+
User concern (Con #4): "Storage Bloat — wrong. Deduplication and
|
|
4
|
+
holographic compression solve this."
|
|
5
|
+
|
|
6
|
+
This module audits the MemoryPalace and TraceStore for:
|
|
7
|
+
1. **Dedup ratio**: how many input tokens map to how many derived
|
|
8
|
+
facts. The README claims 10M tokens → ~590 facts. This formalizes
|
|
9
|
+
that audit.
|
|
10
|
+
2. **Holographic compression ratio**: dense vector bytes vs stored
|
|
11
|
+
bytes per codec. PQ codec achieves 96x at 768 dims (8 B vs 768 B);
|
|
12
|
+
binary codec achieves 8x (96 B vs 768 B with TMR).
|
|
13
|
+
3. **Effective memory budget per million memories**: the metric that
|
|
14
|
+
matters for edge deployment — "fits on a Raspberry Pi 5".
|
|
15
|
+
|
|
16
|
+
Pure Python + numpy. Runs as part of palace.storage_stats() and exposed
|
|
17
|
+
via the MCP server's `contextm_stats` tool.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
from dataclasses import dataclass, asdict
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass
|
|
26
|
+
class CompressionReport:
|
|
27
|
+
"""Audit of dedup + holographic compression."""
|
|
28
|
+
raw_token_count: int
|
|
29
|
+
unique_fact_count: int
|
|
30
|
+
dedup_ratio: float # raw / unique
|
|
31
|
+
codec: str
|
|
32
|
+
dims: int
|
|
33
|
+
bytes_per_vector: int
|
|
34
|
+
dense_bytes_per_vector: int # FP32 dense = dims * 4
|
|
35
|
+
compression_ratio: float # dense / stored
|
|
36
|
+
total_stored_bytes: int
|
|
37
|
+
per_million_mb: float # MB to store 1M memories
|
|
38
|
+
fits_raspberry_pi_5: bool # 8GB RAM target
|
|
39
|
+
notes: list[str]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class DedupAuditor:
|
|
43
|
+
"""Audit dedup + compression for a MemoryPalace."""
|
|
44
|
+
|
|
45
|
+
DENSE_BYTES_PER_DIM = 4 # FP32
|
|
46
|
+
|
|
47
|
+
def __init__(self, palace, store) -> None:
|
|
48
|
+
self.palace = palace
|
|
49
|
+
self.store = store
|
|
50
|
+
|
|
51
|
+
def audit(self, raw_token_count: int | None = None) -> CompressionReport:
|
|
52
|
+
"""Compute dedup ratio + compression ratio for the palace."""
|
|
53
|
+
facts = self.store.query_facts(active=True)
|
|
54
|
+
unique_facts = len(facts)
|
|
55
|
+
# raw token count: heuristic from stored chunks (rows in trace text table)
|
|
56
|
+
if raw_token_count is None:
|
|
57
|
+
raw_token_count = self._estimate_raw_tokens()
|
|
58
|
+
dedup_ratio = (raw_token_count / unique_facts
|
|
59
|
+
if unique_facts > 0 else 0.0)
|
|
60
|
+
# compression ratio
|
|
61
|
+
codec_name = self.palace.codec.name
|
|
62
|
+
dims = self.palace.dims
|
|
63
|
+
bytes_per_vec = self.palace.codec.bytes_per_vector
|
|
64
|
+
dense_bytes = dims * self.DENSE_BYTES_PER_DIM
|
|
65
|
+
comp_ratio = dense_bytes / bytes_per_vec if bytes_per_vec > 0 else 0.0
|
|
66
|
+
total_stored = unique_facts * bytes_per_vec
|
|
67
|
+
per_million_mb = (bytes_per_vec * 1_000_000) / (1024 * 1024)
|
|
68
|
+
fits_pi = per_million_mb < 1024 # 1GB threshold
|
|
69
|
+
notes = []
|
|
70
|
+
if codec_name == "pq":
|
|
71
|
+
notes.append("PQ achieves ~96x compression at 768 dims (8 B/v)")
|
|
72
|
+
notes.append("Codebook adds ~16KB overhead (amortized at >2k facts)")
|
|
73
|
+
elif codec_name == "binary":
|
|
74
|
+
notes.append("Binary + TMR: 96 B/v at 768 dims, 8x compression")
|
|
75
|
+
notes.append("TMR adds 3x storage but enables self-healing")
|
|
76
|
+
elif codec_name == "rabitq":
|
|
77
|
+
notes.append("RaBitQ: 96 B/v at 768 dims, JL rotation + binarization")
|
|
78
|
+
elif codec_name == "int8":
|
|
79
|
+
notes.append("INT8: 770 B/v at 768 dims (baseline, 4x compression)")
|
|
80
|
+
notes.append(f"Dedup ratio {dedup_ratio:.1f}x "
|
|
81
|
+
f"({raw_token_count} tokens → {unique_facts} facts)")
|
|
82
|
+
if dedup_ratio < 5:
|
|
83
|
+
notes.append("⚠ Dedup ratio <5x — check for redundant patterns")
|
|
84
|
+
return CompressionReport(
|
|
85
|
+
raw_token_count=raw_token_count,
|
|
86
|
+
unique_fact_count=unique_facts,
|
|
87
|
+
dedup_ratio=dedup_ratio,
|
|
88
|
+
codec=codec_name,
|
|
89
|
+
dims=dims,
|
|
90
|
+
bytes_per_vector=bytes_per_vec,
|
|
91
|
+
dense_bytes_per_vector=dense_bytes,
|
|
92
|
+
compression_ratio=comp_ratio,
|
|
93
|
+
total_stored_bytes=total_stored,
|
|
94
|
+
per_million_mb=per_million_mb,
|
|
95
|
+
fits_raspberry_pi_5=fits_pi,
|
|
96
|
+
notes=notes,
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
def _estimate_raw_tokens(self) -> int:
|
|
100
|
+
"""Estimate raw token count from stored chunk text rows."""
|
|
101
|
+
try:
|
|
102
|
+
# trace store has raw_text rows; count chars / 4
|
|
103
|
+
row = self.store.conn.execute(
|
|
104
|
+
"SELECT COUNT(*) as n, SUM(LENGTH(text)) as chars FROM raw_text"
|
|
105
|
+
).fetchone()
|
|
106
|
+
if row and row["chars"]:
|
|
107
|
+
return int(row["chars"] / 4) # ~4 chars/token heuristic
|
|
108
|
+
except Exception:
|
|
109
|
+
pass
|
|
110
|
+
# fallback: estimate from fact count * 16 (heuristic)
|
|
111
|
+
return self.palace.size() * 16
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
__all__ = ["DedupAuditor", "CompressionReport"]
|
cortexm/trace/edges.py
ADDED
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
"""Typed edge vocabulary for the Trace.
|
|
2
|
+
|
|
3
|
+
arXiv:2601.15311 (Aeon) formalizes an episodic Trace with typed edges:
|
|
4
|
+
CAUSAL, NEXT, REFERS_TO. Context-M already had EXTRACTED_FROM /
|
|
5
|
+
CONTRADICTS / TEMPORALLY_PRECEDED_BY — this module adds the missing
|
|
6
|
+
types and centralizes the vocabulary so:
|
|
7
|
+
|
|
8
|
+
- write-side code (writer.py / query_extract.py / consolidate.py) has
|
|
9
|
+
one place to discover the canonical edge kinds
|
|
10
|
+
- read-side code (reader.py / ppr.py) can enumerate them by purpose
|
|
11
|
+
("causal", "temporal", "provenance", "semantic_ref") without hard-
|
|
12
|
+
coding strings scattered across the codebase
|
|
13
|
+
- the dreaming / consolidation pass can walk the typed graph
|
|
14
|
+
intentionally (e.g. merge facts joined by NEXT, summarize chains
|
|
15
|
+
linked by CAUSAL)
|
|
16
|
+
|
|
17
|
+
Edge kind -> direction convention:
|
|
18
|
+
src -> dst means "src is the cause / antecedent / referring node"
|
|
19
|
+
e.g. CAUSAL(fact_A, fact_B) reads "fact_A caused fact_B" (A is
|
|
20
|
+
the cause, B is the effect). This matches Aeon's convention.
|
|
21
|
+
|
|
22
|
+
The kind column on `edges` is TEXT — adding new kinds requires no
|
|
23
|
+
schema migration, only a helper here that knows how to wire them.
|
|
24
|
+
"""
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
from typing import Final
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# ---------- Canonical edge vocabulary (centralized) --------------------
|
|
31
|
+
|
|
32
|
+
# provenance: where a fact came from
|
|
33
|
+
EXTRACTED_FROM: Final[str] = "EXTRACTED_FROM" # fact -> chunk
|
|
34
|
+
MENTIONS: Final[str] = "MENTIONS" # fact -> entity mention
|
|
35
|
+
|
|
36
|
+
# temporal / causal ordering
|
|
37
|
+
TEMPORALLY_PRECEDED_BY: Final[str] = "TEMPORALLY_PRECEDED_BY" # fact -> fact (next is newer)
|
|
38
|
+
CAUSAL: Final[str] = "CAUSAL" # fact -> fact (cause -> effect)
|
|
39
|
+
NEXT: Final[str] = "NEXT" # fact -> fact (narrative next, no causal claim)
|
|
40
|
+
|
|
41
|
+
# truth maintenance
|
|
42
|
+
CONTRADICTS: Final[str] = "CONTRADICTS" # fact -> fact (new contradicts old)
|
|
43
|
+
SUPERSEDES: Final[str] = "SUPERSEDES" # fact -> fact (new replaces old)
|
|
44
|
+
RETRACTED_BY: Final[str] = "RETRACTED_BY" # fact -> fact (retired by retraction)
|
|
45
|
+
MERGED_WITH: Final[str] = "MERGED_WITH" # fact -> fact (merged into target)
|
|
46
|
+
|
|
47
|
+
# semantic cross-references (Aeon's REFERS_TO)
|
|
48
|
+
REFERS_TO: Final[str] = "REFERS_TO" # fact -> fact/concept (episodic → atlas)
|
|
49
|
+
SAME_PERSON: Final[str] = "SAME_PERSON" # alias fact -> canonical fact (e.g. "Priya" → "Priya Johnson")
|
|
50
|
+
|
|
51
|
+
# HMS cognition engine edges — produced by cortexm.cognition.*.
|
|
52
|
+
# HYPOTHESIZED_BY: hypothesis fact -> supporting fact(s) (reasoning chain)
|
|
53
|
+
# PROMOTED_FROM: user-confirmed fact -> hypothesis origin (truth
|
|
54
|
+
# maintenance — when a hypothesis is verified by
|
|
55
|
+
# user input, the new fact links back to its origin)
|
|
56
|
+
# ABSTRACTS: abstraction prototype -> member (categorization)
|
|
57
|
+
# INSTANTIATES: member -> abstraction prototype (inverse)
|
|
58
|
+
# ANALOGOUS_TO: relation_a -> relation_b (structural isomorphism)
|
|
59
|
+
HYPOTHESIZED_BY: Final[str] = "HYPOTHESIZED_BY"
|
|
60
|
+
PROMOTED_FROM: Final[str] = "PROMOTED_FROM"
|
|
61
|
+
ABSTRACTS: Final[str] = "ABSTRACTS"
|
|
62
|
+
INSTANTIATES: Final[str] = "INSTANTIATES"
|
|
63
|
+
ANALOGOUS_TO: Final[str] = "ANALOGOUS_TO"
|
|
64
|
+
|
|
65
|
+
# All kinds in one place for enumeration / display
|
|
66
|
+
ALL_KINDS: Final[tuple[str, ...]] = (
|
|
67
|
+
EXTRACTED_FROM, MENTIONS,
|
|
68
|
+
TEMPORALLY_PRECEDED_BY, CAUSAL, NEXT,
|
|
69
|
+
CONTRADICTS, SUPERSEDES, RETRACTED_BY, MERGED_WITH,
|
|
70
|
+
REFERS_TO, SAME_PERSON,
|
|
71
|
+
HYPOTHESIZED_BY, PROMOTED_FROM,
|
|
72
|
+
ABSTRACTS, INSTANTIATES, ANALOGOUS_TO,
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
# Group by purpose — used by the reader / PPR / consolidator
|
|
76
|
+
PROVENANCE_EDGES: Final[tuple[str, ...]] = (EXTRACTED_FROM, MENTIONS)
|
|
77
|
+
TEMPORAL_EDGES: Final[tuple[str, ...]] = (TEMPORALLY_PRECEDED_BY, NEXT)
|
|
78
|
+
CAUSAL_EDGES: Final[tuple[str, ...]] = (CAUSAL,)
|
|
79
|
+
TRUTH_MAINTENANCE_EDGES: Final[tuple[str, ...]] = (
|
|
80
|
+
CONTRADICTS, SUPERSEDES, RETRACTED_BY, MERGED_WITH,
|
|
81
|
+
PROMOTED_FROM)
|
|
82
|
+
SEMANTIC_REF_EDGES: Final[tuple[str, ...]] = (
|
|
83
|
+
REFERS_TO, SAME_PERSON,
|
|
84
|
+
ABSTRACTS, INSTANTIATES, ANALOGOUS_TO)
|
|
85
|
+
COGNITION_EDGES: Final[tuple[str, ...]] = (
|
|
86
|
+
HYPOTHESIZED_BY, PROMOTED_FROM,
|
|
87
|
+
ABSTRACTS, INSTANTIATES, ANALOGOUS_TO)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
# ---------- Helpers -----------------------------------------------------
|
|
91
|
+
|
|
92
|
+
def is_causal(kind: str) -> bool:
|
|
93
|
+
"""Does this edge type assert a causal / temporal antecedent?"""
|
|
94
|
+
return kind in (CAUSAL, TEMPORALLY_PRECEDED_BY, NEXT)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def is_truth_maintenance(kind: str) -> bool:
|
|
98
|
+
"""Does this edge type mark a fact as no-longer-current?"""
|
|
99
|
+
return kind in (CONTRADICTS, SUPERSEDES, RETRACTED_BY, MERGED_WITH)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def is_semantic_ref(kind: str) -> bool:
|
|
103
|
+
"""Does this edge type link two facts semantically (vs. structurally)?"""
|
|
104
|
+
return kind in (REFERS_TO, SAME_PERSON)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def direction_convention(kind: str) -> str:
|
|
108
|
+
"""Return 'cause_to_effect' or 'new_to_old' or 'ref_to_target'.
|
|
109
|
+
|
|
110
|
+
Used by reader/PPR to know which way to walk the edge for a given
|
|
111
|
+
query intent (e.g. "why" queries walk CAUSAL src->dst; "what
|
|
112
|
+
replaced this" queries walk SUPERSEDES dst->src).
|
|
113
|
+
"""
|
|
114
|
+
if kind in (CAUSAL, TEMPORALLY_PRECEDED_BY, NEXT):
|
|
115
|
+
return "cause_to_effect" # src causes/precedes dst
|
|
116
|
+
if kind in (CONTRADICTS, SUPERSEDES, RETRACTED_BY, MERGED_WITH):
|
|
117
|
+
return "new_to_old" # src is the newer fact, dst is the older
|
|
118
|
+
if kind in (REFERS_TO, SAME_PERSON, MENTIONS):
|
|
119
|
+
return "ref_to_target"
|
|
120
|
+
return "out"
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
# ---------- CAUSAL / REFERS_TO writers --------------------------------
|
|
124
|
+
|
|
125
|
+
def wire_causal_edge(store, cause_fact_id: str, effect_fact_id: str,
|
|
126
|
+
reason: str = "") -> None:
|
|
127
|
+
"""Wire a CAUSAL edge from cause to effect.
|
|
128
|
+
|
|
129
|
+
Used by the writer when a SUPERSEDE / retraction pattern fires —
|
|
130
|
+
e.g. "I left Google in March" (cause) caused the older works_at
|
|
131
|
+
fact to be retired (effect). Both edges are kept:
|
|
132
|
+
SUPERSEDES marks the truth-maintenance side, CAUSAL marks the
|
|
133
|
+
narrative cause. Reader can answer "why did X happen?" via CAUSAL
|
|
134
|
+
traversal; it can answer "what's the current truth about X?" via
|
|
135
|
+
SUPERSEDES traversal.
|
|
136
|
+
"""
|
|
137
|
+
if cause_fact_id == effect_fact_id:
|
|
138
|
+
return
|
|
139
|
+
store.add_edge(cause_fact_id, effect_fact_id, CAUSAL,
|
|
140
|
+
{"reason": reason} if reason else None)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def wire_refers_to(store, ref_fact_id: str, target_id: str,
|
|
144
|
+
target_kind: str = "fact") -> None:
|
|
145
|
+
"""Wire a REFERS_TO edge from a referring fact to its target.
|
|
146
|
+
|
|
147
|
+
Used by the writer / consolidator to link an episodic fact back to
|
|
148
|
+
the atlas concept it refers to. For Context-M the "atlas" is the
|
|
149
|
+
palace (the VSA vector space) — so REFERS_TO currently points at
|
|
150
|
+
other facts (semantic cross-references) and may also point at
|
|
151
|
+
chunk_ids (episodic → raw source) when called from the
|
|
152
|
+
consolidator's branch-compression pass.
|
|
153
|
+
"""
|
|
154
|
+
if ref_fact_id == target_id:
|
|
155
|
+
return
|
|
156
|
+
store.add_edge(ref_fact_id, target_id, REFERS_TO,
|
|
157
|
+
{"target_kind": target_kind})
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def find_causal_chain(store, fact_id: str,
|
|
161
|
+
direction: str = "ancestors",
|
|
162
|
+
max_depth: int = 8) -> list[str]:
|
|
163
|
+
"""Walk CAUSAL edges from fact_id.
|
|
164
|
+
|
|
165
|
+
direction='ancestors': walk src->dst backwards — return the chain
|
|
166
|
+
of causes that led to this fact (effect).
|
|
167
|
+
direction='descendants': walk src->dst forwards — return the chain
|
|
168
|
+
of effects this fact caused.
|
|
169
|
+
"""
|
|
170
|
+
seen: set[str] = {fact_id}
|
|
171
|
+
out: list[str] = []
|
|
172
|
+
frontier = [fact_id]
|
|
173
|
+
for _ in range(max_depth):
|
|
174
|
+
next_frontier: list[str] = []
|
|
175
|
+
for fid in frontier:
|
|
176
|
+
if direction == "ancestors":
|
|
177
|
+
# find edges where dst=fid (this fact is the effect)
|
|
178
|
+
rows = store.conn.execute(
|
|
179
|
+
"SELECT src FROM edges WHERE dst=? AND kind=?",
|
|
180
|
+
(fid, CAUSAL)).fetchall()
|
|
181
|
+
else:
|
|
182
|
+
# find edges where src=fid (this fact is the cause)
|
|
183
|
+
rows = store.conn.execute(
|
|
184
|
+
"SELECT dst FROM edges WHERE src=? AND kind=?",
|
|
185
|
+
(fid, CAUSAL)).fetchall()
|
|
186
|
+
for r in rows:
|
|
187
|
+
other = r[0]
|
|
188
|
+
if other in seen:
|
|
189
|
+
continue
|
|
190
|
+
seen.add(other)
|
|
191
|
+
out.append(other)
|
|
192
|
+
next_frontier.append(other)
|
|
193
|
+
if not next_frontier:
|
|
194
|
+
break
|
|
195
|
+
frontier = next_frontier
|
|
196
|
+
return out
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
__all__ = [
|
|
200
|
+
# kinds
|
|
201
|
+
"EXTRACTED_FROM", "MENTIONS", "TEMPORALLY_PRECEDED_BY",
|
|
202
|
+
"CAUSAL", "NEXT", "CONTRADICTS", "SUPERSEDES",
|
|
203
|
+
"RETRACTED_BY", "MERGED_WITH", "REFERS_TO", "SAME_PERSON",
|
|
204
|
+
"HYPOTHESIZED_BY", "PROMOTED_FROM",
|
|
205
|
+
"ABSTRACTS", "INSTANTIATES", "ANALOGOUS_TO",
|
|
206
|
+
"ALL_KINDS", "PROVENANCE_EDGES", "TEMPORAL_EDGES",
|
|
207
|
+
"CAUSAL_EDGES", "TRUTH_MAINTENANCE_EDGES", "SEMANTIC_REF_EDGES",
|
|
208
|
+
"COGNITION_EDGES",
|
|
209
|
+
# helpers
|
|
210
|
+
"is_causal", "is_truth_maintenance", "is_semantic_ref",
|
|
211
|
+
"direction_convention",
|
|
212
|
+
# writers
|
|
213
|
+
"wire_causal_edge", "wire_refers_to", "find_causal_chain",
|
|
214
|
+
]
|
cortexm/trace/fact.py
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
"""Fact model & relation taxonomy for the Symbolic Trace.
|
|
2
|
+
|
|
3
|
+
Mirrors the Section 1.1 schema: Subject-Relation-Value triples with
|
|
4
|
+
bi-temporal timestamps (valid time + transaction time), confidence,
|
|
5
|
+
BLAKE3 source hash, scope (user/agent/flow), memory type, access count
|
|
6
|
+
and active flag — plus CONTRADICTS / TEMPORALLY_PRECEDED_BY /
|
|
7
|
+
EXTRACTED_FROM edges materialized in the store.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
from dataclasses import dataclass, field, asdict
|
|
14
|
+
from datetime import datetime
|
|
15
|
+
|
|
16
|
+
from cortexm.util import parse_ts, iso
|
|
17
|
+
|
|
18
|
+
# Relations where reality has exactly one value at a time: a new value
|
|
19
|
+
# SUPERSEDES the old one (contradiction resolution by truth maintenance).
|
|
20
|
+
SINGLE_VALUED = {
|
|
21
|
+
"name", "works_at", "role", "lives_in", "birthday", "age",
|
|
22
|
+
"reports_to", "member_of", "studied_at", "team",
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
# Relations that accumulate: duplicates are merged, conflicts coexist.
|
|
26
|
+
MULTI_VALUED = {
|
|
27
|
+
"likes", "dislikes", "prefers", "has_skill", "works_on", "completed", "alias",
|
|
28
|
+
"sibling", "spouse", "parent", "child", "friend", "uses", "manages",
|
|
29
|
+
"studied", "owns", "goal", "event", "mentioned", "joined", "left",
|
|
30
|
+
"moved_to", "allergy", "has_pet", "speaks", "hobby",
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
RELATION_CATEGORIES = {
|
|
34
|
+
"personal": {"name", "birthday", "age", "sibling", "spouse", "parent",
|
|
35
|
+
"child", "friend", "alias", "lives_in", "speaks", "has_pet"},
|
|
36
|
+
"work": {"works_at", "role", "reports_to", "member_of", "manages",
|
|
37
|
+
"team", "joined", "left", "works_on", "completed"},
|
|
38
|
+
"preference": {"likes", "dislikes", "prefers"},
|
|
39
|
+
"skill": {"has_skill", "studied", "studied_at"},
|
|
40
|
+
"task": {"event", "goal", "mentioned", "uses", "owns"},
|
|
41
|
+
"temporal": {"moved_to"},
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass
|
|
46
|
+
class Fact:
|
|
47
|
+
id: str
|
|
48
|
+
subject: str
|
|
49
|
+
relation: str
|
|
50
|
+
value: str
|
|
51
|
+
valid_from: str # when it became true in reality
|
|
52
|
+
valid_to: str | None = None # when it stopped being true (None = now)
|
|
53
|
+
tx_from: str = "" # when we recorded it
|
|
54
|
+
tx_to: str | None = None
|
|
55
|
+
confidence: float = 0.8
|
|
56
|
+
source_hash: str = ""
|
|
57
|
+
source_id: str = ""
|
|
58
|
+
user_id: str = "default"
|
|
59
|
+
agent_id: str | None = None
|
|
60
|
+
run_id: str | None = None
|
|
61
|
+
memory_type: str = "short_term"
|
|
62
|
+
access_count: int = 0
|
|
63
|
+
reinforcement: int = 1
|
|
64
|
+
is_active: bool = True
|
|
65
|
+
is_derived: bool = False
|
|
66
|
+
quarantined: bool = False
|
|
67
|
+
birth_commit: str | None = None
|
|
68
|
+
retired_commit: str | None = None
|
|
69
|
+
provenance: dict = field(default_factory=dict)
|
|
70
|
+
|
|
71
|
+
# ------------------------------------------------------------------
|
|
72
|
+
def text(self) -> str:
|
|
73
|
+
return f"{self.subject} | {self.relation} | {self.value}"
|
|
74
|
+
|
|
75
|
+
def display(self) -> str:
|
|
76
|
+
return f"({self.subject}, {self.relation}, {self.value})"
|
|
77
|
+
|
|
78
|
+
def valid_window(self) -> str:
|
|
79
|
+
return f"{self.valid_from}→{self.valid_to or '∞'}"
|
|
80
|
+
|
|
81
|
+
def scope_dict(self) -> dict:
|
|
82
|
+
return {"user_id": self.user_id, "agent_id": self.agent_id, "run_id": self.run_id}
|
|
83
|
+
|
|
84
|
+
def to_row(self) -> dict:
|
|
85
|
+
d = asdict(self)
|
|
86
|
+
d["is_active"] = int(self.is_active)
|
|
87
|
+
d["is_derived"] = int(self.is_derived)
|
|
88
|
+
d["quarantined"] = int(self.quarantined)
|
|
89
|
+
d["provenance"] = json.dumps(self.provenance, default=str)
|
|
90
|
+
return d
|
|
91
|
+
|
|
92
|
+
@staticmethod
|
|
93
|
+
def from_row(row: dict) -> "Fact":
|
|
94
|
+
row = dict(row)
|
|
95
|
+
row["is_active"] = bool(row["is_active"])
|
|
96
|
+
row["is_derived"] = bool(row["is_derived"])
|
|
97
|
+
row["quarantined"] = bool(row.get("quarantined", 0))
|
|
98
|
+
row["provenance"] = json.loads(row.get("provenance") or "{}")
|
|
99
|
+
return Fact(**row)
|
|
100
|
+
|
|
101
|
+
def matches_scope(self, user_id: str | None, agent_id: str | None = None,
|
|
102
|
+
run_id: str | None = None) -> bool:
|
|
103
|
+
if user_id is not None and self.user_id != user_id:
|
|
104
|
+
return False
|
|
105
|
+
if agent_id is not None and self.agent_id != agent_id:
|
|
106
|
+
return False
|
|
107
|
+
if run_id is not None and self.run_id != run_id:
|
|
108
|
+
return False
|
|
109
|
+
return True
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def make_fact(subject: str, relation: str, value: str, *,
|
|
113
|
+
now: datetime, valid_from: datetime | str | None = None,
|
|
114
|
+
valid_to: datetime | str | None = None, **kwargs) -> Fact:
|
|
115
|
+
"""Convenience constructor normalizing timestamps to ISO strings."""
|
|
116
|
+
vf = iso(parse_ts(valid_from) or now)[:10] if valid_from else iso(now)[:10]
|
|
117
|
+
vt = iso(parse_ts(valid_to))[:10] if valid_to else None
|
|
118
|
+
return Fact(
|
|
119
|
+
id=kwargs.pop("id", "") or __import__("uuid").uuid4().hex,
|
|
120
|
+
subject=subject, relation=relation, value=value,
|
|
121
|
+
valid_from=vf, valid_to=vt, tx_from=iso(now), **kwargs)
|