cortexm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. context_m.py +17 -0
  2. cortexm/__init__.py +45 -0
  3. cortexm/accel.py +403 -0
  4. cortexm/api/__init__.py +0 -0
  5. cortexm/api/chaos.py +118 -0
  6. cortexm/api/memory.py +635 -0
  7. cortexm/bench/__init__.py +0 -0
  8. cortexm/bench/abilities.py +311 -0
  9. cortexm/bench/baselines.py +89 -0
  10. cortexm/bench/beam_loader.py +317 -0
  11. cortexm/bench/generator.py +376 -0
  12. cortexm/bench/harness.py +211 -0
  13. cortexm/bench/messy.py +218 -0
  14. cortexm/bench/micro.py +251 -0
  15. cortexm/bench/ood.py +443 -0
  16. cortexm/bench/run.py +137 -0
  17. cortexm/bridge/__init__.py +0 -0
  18. cortexm/bridge/dates.py +178 -0
  19. cortexm/bridge/decoders.py +204 -0
  20. cortexm/bridge/enrich.py +255 -0
  21. cortexm/bridge/extractor.py +316 -0
  22. cortexm/bridge/fallback.py +332 -0
  23. cortexm/bridge/onnx_runtime.py +158 -0
  24. cortexm/bridge/patterns.py +760 -0
  25. cortexm/bridge/ppr.py +104 -0
  26. cortexm/bridge/prefilter.py +188 -0
  27. cortexm/bridge/query_extract.py +420 -0
  28. cortexm/bridge/reader.py +1174 -0
  29. cortexm/bridge/rerank.py +204 -0
  30. cortexm/bridge/writer.py +492 -0
  31. cortexm/cli.py +295 -0
  32. cortexm/cognition/__init__.py +53 -0
  33. cortexm/cognition/abstraction.py +192 -0
  34. cortexm/cognition/analogy.py +159 -0
  35. cortexm/cognition/engine.py +204 -0
  36. cortexm/cognition/gaps.py +365 -0
  37. cortexm/cognition/scanner.py +204 -0
  38. cortexm/config.py +375 -0
  39. cortexm/cortexm.py +8 -0
  40. cortexm/enterprise/__init__.py +0 -0
  41. cortexm/enterprise/audit.py +178 -0
  42. cortexm/enterprise/governance.py +239 -0
  43. cortexm/errors.py +35 -0
  44. cortexm/features/__init__.py +0 -0
  45. cortexm/features/git.py +204 -0
  46. cortexm/features/prefetch.py +88 -0
  47. cortexm/features/zk.py +105 -0
  48. cortexm/federation/__init__.py +39 -0
  49. cortexm/federation/crdt.py +275 -0
  50. cortexm/federation/fabric.py +109 -0
  51. cortexm/federation/hlc.py +80 -0
  52. cortexm/federation/node.py +145 -0
  53. cortexm/federation/schema_report.py +73 -0
  54. cortexm/federation/transport.py +164 -0
  55. cortexm/index/__init__.py +19 -0
  56. cortexm/index/nsg.py +386 -0
  57. cortexm/mcp/__init__.py +0 -0
  58. cortexm/mcp/server.py +985 -0
  59. cortexm/metrics.py +62 -0
  60. cortexm/migrate/__init__.py +0 -0
  61. cortexm/migrate/importers.py +192 -0
  62. cortexm/provenance/__init__.py +78 -0
  63. cortexm/provenance/agent.py +214 -0
  64. cortexm/provenance/cose.py +201 -0
  65. cortexm/provenance/scitt.py +258 -0
  66. cortexm/provenance/vc.py +250 -0
  67. cortexm/security/__init__.py +0 -0
  68. cortexm/security/crypto.py +162 -0
  69. cortexm/security/hashes.py +140 -0
  70. cortexm/security/injection.py +149 -0
  71. cortexm/security/mind.py +154 -0
  72. cortexm/security/pii.py +265 -0
  73. cortexm/security/rbac.py +169 -0
  74. cortexm/security/sandbox.py +131 -0
  75. cortexm/security/zk_hamming.py +142 -0
  76. cortexm/security/zk_sql.py +485 -0
  77. cortexm/server/__init__.py +0 -0
  78. cortexm/server/metrics.py +88 -0
  79. cortexm/server/rest.py +936 -0
  80. cortexm/server/sparql.py +984 -0
  81. cortexm/text/__init__.py +0 -0
  82. cortexm/text/dissim.py +252 -0
  83. cortexm/text/embedder.py +155 -0
  84. cortexm/text/fuzzy.py +218 -0
  85. cortexm/text/idiolect.py +253 -0
  86. cortexm/text/labse.py +374 -0
  87. cortexm/text/tokenizer.py +79 -0
  88. cortexm/trace/__init__.py +0 -0
  89. cortexm/trace/blob_arena.py +277 -0
  90. cortexm/trace/consolidate.py +337 -0
  91. cortexm/trace/contradictions.py +69 -0
  92. cortexm/trace/dedup.py +114 -0
  93. cortexm/trace/edges.py +214 -0
  94. cortexm/trace/fact.py +121 -0
  95. cortexm/trace/fade.py +245 -0
  96. cortexm/trace/lifecycle.py +112 -0
  97. cortexm/trace/rebuild.py +173 -0
  98. cortexm/trace/rules.py +171 -0
  99. cortexm/trace/store.py +680 -0
  100. cortexm/trace/structural.py +183 -0
  101. cortexm/trace/tmt.py +335 -0
  102. cortexm/util.py +148 -0
  103. cortexm/vsa/__init__.py +0 -0
  104. cortexm/vsa/attribution.py +149 -0
  105. cortexm/vsa/cleanup.py +161 -0
  106. cortexm/vsa/codecs.py +397 -0
  107. cortexm/vsa/hologram_overlay.py +139 -0
  108. cortexm/vsa/index.py +163 -0
  109. cortexm/vsa/ops.py +149 -0
  110. cortexm/vsa/palace.py +446 -0
  111. cortexm/vsa/role_vectors.py +236 -0
  112. cortexm/vsa/slb.py +78 -0
  113. cortexm/vsa/tlsh_trie.py +137 -0
  114. cortexm/vsa/working_memory.py +249 -0
  115. cortexm-0.3.0.dist-info/METADATA +482 -0
  116. cortexm-0.3.0.dist-info/RECORD +120 -0
  117. cortexm-0.3.0.dist-info/WHEEL +5 -0
  118. cortexm-0.3.0.dist-info/entry_points.txt +2 -0
  119. cortexm-0.3.0.dist-info/licenses/LICENSE +190 -0
  120. cortexm-0.3.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,69 @@
1
+ """Deterministic contradiction detection & truth maintenance.
2
+
3
+ Phase rules from Section 1.1:
4
+ 1. Classification — relation → semantic category, single/multi valued
5
+ 2. Contradiction — exact + fuzzy (Jaccard/Levenshtein) match on
6
+ subject-relation pairs; latest-value-wins for
7
+ single-valued relations, append for multi-valued
8
+ 3. Interference-aware lifecycle — see lifecycle.py
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from dataclasses import dataclass
14
+ from enum import Enum
15
+
16
+ from cortexm.trace.fact import Fact, SINGLE_VALUED
17
+ from cortexm.trace.store import TraceStore
18
+ from cortexm.util import similarity
19
+
20
+
21
+ class Action(str, Enum):
22
+ COMMIT = "commit" # brand-new fact
23
+ MERGE = "merge" # near-duplicate: reinforce existing
24
+ SUPERSEDE = "supersede" # contradiction on single-valued relation
25
+ COEXIST = "coexist" # contradiction on multi-valued relation
26
+ SKIP = "skip" # exact duplicate
27
+
28
+
29
+ @dataclass
30
+ class Conflict:
31
+ action: Action
32
+ existing: list[Fact]
33
+ note: str = ""
34
+
35
+
36
+ def find_conflicts(store: TraceStore, candidate: Fact) -> Conflict:
37
+ """Decide how ``candidate`` interacts with active memory."""
38
+ existing = [
39
+ f for f in store.query_facts(
40
+ subject=candidate.subject, relation=candidate.relation,
41
+ user_id=candidate.user_id, active=True)
42
+ if f.id != candidate.id
43
+ ]
44
+ if not existing:
45
+ return Conflict(Action.COMMIT, [])
46
+
47
+ exact = [f for f in existing if f.value.strip().lower() == candidate.value.strip().lower()]
48
+ if exact:
49
+ return Conflict(Action.SKIP, exact, "exact duplicate")
50
+
51
+ # low-salience mention anchors: exact-dup semantics only (no fuzzy
52
+ # quadratic scans — mention streams are high-volume, low-signal)
53
+ if candidate.relation in ("mentioned", "event", "instruction"):
54
+ if candidate.relation == "mentioned":
55
+ return Conflict(Action.COEXIST, existing, "mention anchor recorded")
56
+
57
+ near = [f for f in existing
58
+ if similarity(f.value, candidate.value) >= 0.92]
59
+ if near:
60
+ return Conflict(Action.MERGE, near, "near-duplicate merged; reinforcement +1")
61
+
62
+ single = candidate.relation in SINGLE_VALUED
63
+ if single:
64
+ # newest valid_from wins reality; old fact gets valid_to
65
+ target = max(existing, key=lambda f: (f.valid_from, f.tx_from))
66
+ return Conflict(Action.SUPERSEDE, [target],
67
+ f"contradiction on single-valued '{candidate.relation}'")
68
+ return Conflict(Action.COEXIST, existing,
69
+ f"conflicting values on multi-valued '{candidate.relation}' coexist")
cortexm/trace/dedup.py ADDED
@@ -0,0 +1,114 @@
1
+ """Deduplication + holographic compression audit.
2
+
3
+ User concern (Con #4): "Storage Bloat — wrong. Deduplication and
4
+ holographic compression solve this."
5
+
6
+ This module audits the MemoryPalace and TraceStore for:
7
+ 1. **Dedup ratio**: how many input tokens map to how many derived
8
+ facts. The README claims 10M tokens → ~590 facts. This formalizes
9
+ that audit.
10
+ 2. **Holographic compression ratio**: dense vector bytes vs stored
11
+ bytes per codec. PQ codec achieves 96x at 768 dims (8 B vs 768 B);
12
+ binary codec achieves 8x (96 B vs 768 B with TMR).
13
+ 3. **Effective memory budget per million memories**: the metric that
14
+ matters for edge deployment — "fits on a Raspberry Pi 5".
15
+
16
+ Pure Python + numpy. Runs as part of palace.storage_stats() and exposed
17
+ via the MCP server's `contextm_stats` tool.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ from dataclasses import dataclass, asdict
23
+
24
+
25
+ @dataclass
26
+ class CompressionReport:
27
+ """Audit of dedup + holographic compression."""
28
+ raw_token_count: int
29
+ unique_fact_count: int
30
+ dedup_ratio: float # raw / unique
31
+ codec: str
32
+ dims: int
33
+ bytes_per_vector: int
34
+ dense_bytes_per_vector: int # FP32 dense = dims * 4
35
+ compression_ratio: float # dense / stored
36
+ total_stored_bytes: int
37
+ per_million_mb: float # MB to store 1M memories
38
+ fits_raspberry_pi_5: bool # 8GB RAM target
39
+ notes: list[str]
40
+
41
+
42
+ class DedupAuditor:
43
+ """Audit dedup + compression for a MemoryPalace."""
44
+
45
+ DENSE_BYTES_PER_DIM = 4 # FP32
46
+
47
+ def __init__(self, palace, store) -> None:
48
+ self.palace = palace
49
+ self.store = store
50
+
51
+ def audit(self, raw_token_count: int | None = None) -> CompressionReport:
52
+ """Compute dedup ratio + compression ratio for the palace."""
53
+ facts = self.store.query_facts(active=True)
54
+ unique_facts = len(facts)
55
+ # raw token count: heuristic from stored chunks (rows in trace text table)
56
+ if raw_token_count is None:
57
+ raw_token_count = self._estimate_raw_tokens()
58
+ dedup_ratio = (raw_token_count / unique_facts
59
+ if unique_facts > 0 else 0.0)
60
+ # compression ratio
61
+ codec_name = self.palace.codec.name
62
+ dims = self.palace.dims
63
+ bytes_per_vec = self.palace.codec.bytes_per_vector
64
+ dense_bytes = dims * self.DENSE_BYTES_PER_DIM
65
+ comp_ratio = dense_bytes / bytes_per_vec if bytes_per_vec > 0 else 0.0
66
+ total_stored = unique_facts * bytes_per_vec
67
+ per_million_mb = (bytes_per_vec * 1_000_000) / (1024 * 1024)
68
+ fits_pi = per_million_mb < 1024 # 1GB threshold
69
+ notes = []
70
+ if codec_name == "pq":
71
+ notes.append("PQ achieves ~96x compression at 768 dims (8 B/v)")
72
+ notes.append("Codebook adds ~16KB overhead (amortized at >2k facts)")
73
+ elif codec_name == "binary":
74
+ notes.append("Binary + TMR: 96 B/v at 768 dims, 8x compression")
75
+ notes.append("TMR adds 3x storage but enables self-healing")
76
+ elif codec_name == "rabitq":
77
+ notes.append("RaBitQ: 96 B/v at 768 dims, JL rotation + binarization")
78
+ elif codec_name == "int8":
79
+ notes.append("INT8: 770 B/v at 768 dims (baseline, 4x compression)")
80
+ notes.append(f"Dedup ratio {dedup_ratio:.1f}x "
81
+ f"({raw_token_count} tokens → {unique_facts} facts)")
82
+ if dedup_ratio < 5:
83
+ notes.append("⚠ Dedup ratio <5x — check for redundant patterns")
84
+ return CompressionReport(
85
+ raw_token_count=raw_token_count,
86
+ unique_fact_count=unique_facts,
87
+ dedup_ratio=dedup_ratio,
88
+ codec=codec_name,
89
+ dims=dims,
90
+ bytes_per_vector=bytes_per_vec,
91
+ dense_bytes_per_vector=dense_bytes,
92
+ compression_ratio=comp_ratio,
93
+ total_stored_bytes=total_stored,
94
+ per_million_mb=per_million_mb,
95
+ fits_raspberry_pi_5=fits_pi,
96
+ notes=notes,
97
+ )
98
+
99
+ def _estimate_raw_tokens(self) -> int:
100
+ """Estimate raw token count from stored chunk text rows."""
101
+ try:
102
+ # trace store has raw_text rows; count chars / 4
103
+ row = self.store.conn.execute(
104
+ "SELECT COUNT(*) as n, SUM(LENGTH(text)) as chars FROM raw_text"
105
+ ).fetchone()
106
+ if row and row["chars"]:
107
+ return int(row["chars"] / 4) # ~4 chars/token heuristic
108
+ except Exception:
109
+ pass
110
+ # fallback: estimate from fact count * 16 (heuristic)
111
+ return self.palace.size() * 16
112
+
113
+
114
+ __all__ = ["DedupAuditor", "CompressionReport"]
cortexm/trace/edges.py ADDED
@@ -0,0 +1,214 @@
1
+ """Typed edge vocabulary for the Trace.
2
+
3
+ arXiv:2601.15311 (Aeon) formalizes an episodic Trace with typed edges:
4
+ CAUSAL, NEXT, REFERS_TO. Context-M already had EXTRACTED_FROM /
5
+ CONTRADICTS / TEMPORALLY_PRECEDED_BY — this module adds the missing
6
+ types and centralizes the vocabulary so:
7
+
8
+ - write-side code (writer.py / query_extract.py / consolidate.py) has
9
+ one place to discover the canonical edge kinds
10
+ - read-side code (reader.py / ppr.py) can enumerate them by purpose
11
+ ("causal", "temporal", "provenance", "semantic_ref") without hard-
12
+ coding strings scattered across the codebase
13
+ - the dreaming / consolidation pass can walk the typed graph
14
+ intentionally (e.g. merge facts joined by NEXT, summarize chains
15
+ linked by CAUSAL)
16
+
17
+ Edge kind -> direction convention:
18
+ src -> dst means "src is the cause / antecedent / referring node"
19
+ e.g. CAUSAL(fact_A, fact_B) reads "fact_A caused fact_B" (A is
20
+ the cause, B is the effect). This matches Aeon's convention.
21
+
22
+ The kind column on `edges` is TEXT — adding new kinds requires no
23
+ schema migration, only a helper here that knows how to wire them.
24
+ """
25
+ from __future__ import annotations
26
+
27
+ from typing import Final
28
+
29
+
30
+ # ---------- Canonical edge vocabulary (centralized) --------------------
31
+
32
+ # provenance: where a fact came from
33
+ EXTRACTED_FROM: Final[str] = "EXTRACTED_FROM" # fact -> chunk
34
+ MENTIONS: Final[str] = "MENTIONS" # fact -> entity mention
35
+
36
+ # temporal / causal ordering
37
+ TEMPORALLY_PRECEDED_BY: Final[str] = "TEMPORALLY_PRECEDED_BY" # fact -> fact (next is newer)
38
+ CAUSAL: Final[str] = "CAUSAL" # fact -> fact (cause -> effect)
39
+ NEXT: Final[str] = "NEXT" # fact -> fact (narrative next, no causal claim)
40
+
41
+ # truth maintenance
42
+ CONTRADICTS: Final[str] = "CONTRADICTS" # fact -> fact (new contradicts old)
43
+ SUPERSEDES: Final[str] = "SUPERSEDES" # fact -> fact (new replaces old)
44
+ RETRACTED_BY: Final[str] = "RETRACTED_BY" # fact -> fact (retired by retraction)
45
+ MERGED_WITH: Final[str] = "MERGED_WITH" # fact -> fact (merged into target)
46
+
47
+ # semantic cross-references (Aeon's REFERS_TO)
48
+ REFERS_TO: Final[str] = "REFERS_TO" # fact -> fact/concept (episodic → atlas)
49
+ SAME_PERSON: Final[str] = "SAME_PERSON" # alias fact -> canonical fact (e.g. "Priya" → "Priya Johnson")
50
+
51
+ # HMS cognition engine edges — produced by cortexm.cognition.*.
52
+ # HYPOTHESIZED_BY: hypothesis fact -> supporting fact(s) (reasoning chain)
53
+ # PROMOTED_FROM: user-confirmed fact -> hypothesis origin (truth
54
+ # maintenance — when a hypothesis is verified by
55
+ # user input, the new fact links back to its origin)
56
+ # ABSTRACTS: abstraction prototype -> member (categorization)
57
+ # INSTANTIATES: member -> abstraction prototype (inverse)
58
+ # ANALOGOUS_TO: relation_a -> relation_b (structural isomorphism)
59
+ HYPOTHESIZED_BY: Final[str] = "HYPOTHESIZED_BY"
60
+ PROMOTED_FROM: Final[str] = "PROMOTED_FROM"
61
+ ABSTRACTS: Final[str] = "ABSTRACTS"
62
+ INSTANTIATES: Final[str] = "INSTANTIATES"
63
+ ANALOGOUS_TO: Final[str] = "ANALOGOUS_TO"
64
+
65
+ # All kinds in one place for enumeration / display
66
+ ALL_KINDS: Final[tuple[str, ...]] = (
67
+ EXTRACTED_FROM, MENTIONS,
68
+ TEMPORALLY_PRECEDED_BY, CAUSAL, NEXT,
69
+ CONTRADICTS, SUPERSEDES, RETRACTED_BY, MERGED_WITH,
70
+ REFERS_TO, SAME_PERSON,
71
+ HYPOTHESIZED_BY, PROMOTED_FROM,
72
+ ABSTRACTS, INSTANTIATES, ANALOGOUS_TO,
73
+ )
74
+
75
+ # Group by purpose — used by the reader / PPR / consolidator
76
+ PROVENANCE_EDGES: Final[tuple[str, ...]] = (EXTRACTED_FROM, MENTIONS)
77
+ TEMPORAL_EDGES: Final[tuple[str, ...]] = (TEMPORALLY_PRECEDED_BY, NEXT)
78
+ CAUSAL_EDGES: Final[tuple[str, ...]] = (CAUSAL,)
79
+ TRUTH_MAINTENANCE_EDGES: Final[tuple[str, ...]] = (
80
+ CONTRADICTS, SUPERSEDES, RETRACTED_BY, MERGED_WITH,
81
+ PROMOTED_FROM)
82
+ SEMANTIC_REF_EDGES: Final[tuple[str, ...]] = (
83
+ REFERS_TO, SAME_PERSON,
84
+ ABSTRACTS, INSTANTIATES, ANALOGOUS_TO)
85
+ COGNITION_EDGES: Final[tuple[str, ...]] = (
86
+ HYPOTHESIZED_BY, PROMOTED_FROM,
87
+ ABSTRACTS, INSTANTIATES, ANALOGOUS_TO)
88
+
89
+
90
+ # ---------- Helpers -----------------------------------------------------
91
+
92
+ def is_causal(kind: str) -> bool:
93
+ """Does this edge type assert a causal / temporal antecedent?"""
94
+ return kind in (CAUSAL, TEMPORALLY_PRECEDED_BY, NEXT)
95
+
96
+
97
+ def is_truth_maintenance(kind: str) -> bool:
98
+ """Does this edge type mark a fact as no-longer-current?"""
99
+ return kind in (CONTRADICTS, SUPERSEDES, RETRACTED_BY, MERGED_WITH)
100
+
101
+
102
+ def is_semantic_ref(kind: str) -> bool:
103
+ """Does this edge type link two facts semantically (vs. structurally)?"""
104
+ return kind in (REFERS_TO, SAME_PERSON)
105
+
106
+
107
+ def direction_convention(kind: str) -> str:
108
+ """Return 'cause_to_effect' or 'new_to_old' or 'ref_to_target'.
109
+
110
+ Used by reader/PPR to know which way to walk the edge for a given
111
+ query intent (e.g. "why" queries walk CAUSAL src->dst; "what
112
+ replaced this" queries walk SUPERSEDES dst->src).
113
+ """
114
+ if kind in (CAUSAL, TEMPORALLY_PRECEDED_BY, NEXT):
115
+ return "cause_to_effect" # src causes/precedes dst
116
+ if kind in (CONTRADICTS, SUPERSEDES, RETRACTED_BY, MERGED_WITH):
117
+ return "new_to_old" # src is the newer fact, dst is the older
118
+ if kind in (REFERS_TO, SAME_PERSON, MENTIONS):
119
+ return "ref_to_target"
120
+ return "out"
121
+
122
+
123
+ # ---------- CAUSAL / REFERS_TO writers --------------------------------
124
+
125
+ def wire_causal_edge(store, cause_fact_id: str, effect_fact_id: str,
126
+ reason: str = "") -> None:
127
+ """Wire a CAUSAL edge from cause to effect.
128
+
129
+ Used by the writer when a SUPERSEDE / retraction pattern fires —
130
+ e.g. "I left Google in March" (cause) caused the older works_at
131
+ fact to be retired (effect). Both edges are kept:
132
+ SUPERSEDES marks the truth-maintenance side, CAUSAL marks the
133
+ narrative cause. Reader can answer "why did X happen?" via CAUSAL
134
+ traversal; it can answer "what's the current truth about X?" via
135
+ SUPERSEDES traversal.
136
+ """
137
+ if cause_fact_id == effect_fact_id:
138
+ return
139
+ store.add_edge(cause_fact_id, effect_fact_id, CAUSAL,
140
+ {"reason": reason} if reason else None)
141
+
142
+
143
+ def wire_refers_to(store, ref_fact_id: str, target_id: str,
144
+ target_kind: str = "fact") -> None:
145
+ """Wire a REFERS_TO edge from a referring fact to its target.
146
+
147
+ Used by the writer / consolidator to link an episodic fact back to
148
+ the atlas concept it refers to. For Context-M the "atlas" is the
149
+ palace (the VSA vector space) — so REFERS_TO currently points at
150
+ other facts (semantic cross-references) and may also point at
151
+ chunk_ids (episodic → raw source) when called from the
152
+ consolidator's branch-compression pass.
153
+ """
154
+ if ref_fact_id == target_id:
155
+ return
156
+ store.add_edge(ref_fact_id, target_id, REFERS_TO,
157
+ {"target_kind": target_kind})
158
+
159
+
160
+ def find_causal_chain(store, fact_id: str,
161
+ direction: str = "ancestors",
162
+ max_depth: int = 8) -> list[str]:
163
+ """Walk CAUSAL edges from fact_id.
164
+
165
+ direction='ancestors': walk src->dst backwards — return the chain
166
+ of causes that led to this fact (effect).
167
+ direction='descendants': walk src->dst forwards — return the chain
168
+ of effects this fact caused.
169
+ """
170
+ seen: set[str] = {fact_id}
171
+ out: list[str] = []
172
+ frontier = [fact_id]
173
+ for _ in range(max_depth):
174
+ next_frontier: list[str] = []
175
+ for fid in frontier:
176
+ if direction == "ancestors":
177
+ # find edges where dst=fid (this fact is the effect)
178
+ rows = store.conn.execute(
179
+ "SELECT src FROM edges WHERE dst=? AND kind=?",
180
+ (fid, CAUSAL)).fetchall()
181
+ else:
182
+ # find edges where src=fid (this fact is the cause)
183
+ rows = store.conn.execute(
184
+ "SELECT dst FROM edges WHERE src=? AND kind=?",
185
+ (fid, CAUSAL)).fetchall()
186
+ for r in rows:
187
+ other = r[0]
188
+ if other in seen:
189
+ continue
190
+ seen.add(other)
191
+ out.append(other)
192
+ next_frontier.append(other)
193
+ if not next_frontier:
194
+ break
195
+ frontier = next_frontier
196
+ return out
197
+
198
+
199
+ __all__ = [
200
+ # kinds
201
+ "EXTRACTED_FROM", "MENTIONS", "TEMPORALLY_PRECEDED_BY",
202
+ "CAUSAL", "NEXT", "CONTRADICTS", "SUPERSEDES",
203
+ "RETRACTED_BY", "MERGED_WITH", "REFERS_TO", "SAME_PERSON",
204
+ "HYPOTHESIZED_BY", "PROMOTED_FROM",
205
+ "ABSTRACTS", "INSTANTIATES", "ANALOGOUS_TO",
206
+ "ALL_KINDS", "PROVENANCE_EDGES", "TEMPORAL_EDGES",
207
+ "CAUSAL_EDGES", "TRUTH_MAINTENANCE_EDGES", "SEMANTIC_REF_EDGES",
208
+ "COGNITION_EDGES",
209
+ # helpers
210
+ "is_causal", "is_truth_maintenance", "is_semantic_ref",
211
+ "direction_convention",
212
+ # writers
213
+ "wire_causal_edge", "wire_refers_to", "find_causal_chain",
214
+ ]
cortexm/trace/fact.py ADDED
@@ -0,0 +1,121 @@
1
+ """Fact model & relation taxonomy for the Symbolic Trace.
2
+
3
+ Mirrors the Section 1.1 schema: Subject-Relation-Value triples with
4
+ bi-temporal timestamps (valid time + transaction time), confidence,
5
+ BLAKE3 source hash, scope (user/agent/flow), memory type, access count
6
+ and active flag — plus CONTRADICTS / TEMPORALLY_PRECEDED_BY /
7
+ EXTRACTED_FROM edges materialized in the store.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ from dataclasses import dataclass, field, asdict
14
+ from datetime import datetime
15
+
16
+ from cortexm.util import parse_ts, iso
17
+
18
+ # Relations where reality has exactly one value at a time: a new value
19
+ # SUPERSEDES the old one (contradiction resolution by truth maintenance).
20
+ SINGLE_VALUED = {
21
+ "name", "works_at", "role", "lives_in", "birthday", "age",
22
+ "reports_to", "member_of", "studied_at", "team",
23
+ }
24
+
25
+ # Relations that accumulate: duplicates are merged, conflicts coexist.
26
+ MULTI_VALUED = {
27
+ "likes", "dislikes", "prefers", "has_skill", "works_on", "completed", "alias",
28
+ "sibling", "spouse", "parent", "child", "friend", "uses", "manages",
29
+ "studied", "owns", "goal", "event", "mentioned", "joined", "left",
30
+ "moved_to", "allergy", "has_pet", "speaks", "hobby",
31
+ }
32
+
33
+ RELATION_CATEGORIES = {
34
+ "personal": {"name", "birthday", "age", "sibling", "spouse", "parent",
35
+ "child", "friend", "alias", "lives_in", "speaks", "has_pet"},
36
+ "work": {"works_at", "role", "reports_to", "member_of", "manages",
37
+ "team", "joined", "left", "works_on", "completed"},
38
+ "preference": {"likes", "dislikes", "prefers"},
39
+ "skill": {"has_skill", "studied", "studied_at"},
40
+ "task": {"event", "goal", "mentioned", "uses", "owns"},
41
+ "temporal": {"moved_to"},
42
+ }
43
+
44
+
45
+ @dataclass
46
+ class Fact:
47
+ id: str
48
+ subject: str
49
+ relation: str
50
+ value: str
51
+ valid_from: str # when it became true in reality
52
+ valid_to: str | None = None # when it stopped being true (None = now)
53
+ tx_from: str = "" # when we recorded it
54
+ tx_to: str | None = None
55
+ confidence: float = 0.8
56
+ source_hash: str = ""
57
+ source_id: str = ""
58
+ user_id: str = "default"
59
+ agent_id: str | None = None
60
+ run_id: str | None = None
61
+ memory_type: str = "short_term"
62
+ access_count: int = 0
63
+ reinforcement: int = 1
64
+ is_active: bool = True
65
+ is_derived: bool = False
66
+ quarantined: bool = False
67
+ birth_commit: str | None = None
68
+ retired_commit: str | None = None
69
+ provenance: dict = field(default_factory=dict)
70
+
71
+ # ------------------------------------------------------------------
72
+ def text(self) -> str:
73
+ return f"{self.subject} | {self.relation} | {self.value}"
74
+
75
+ def display(self) -> str:
76
+ return f"({self.subject}, {self.relation}, {self.value})"
77
+
78
+ def valid_window(self) -> str:
79
+ return f"{self.valid_from}→{self.valid_to or '∞'}"
80
+
81
+ def scope_dict(self) -> dict:
82
+ return {"user_id": self.user_id, "agent_id": self.agent_id, "run_id": self.run_id}
83
+
84
+ def to_row(self) -> dict:
85
+ d = asdict(self)
86
+ d["is_active"] = int(self.is_active)
87
+ d["is_derived"] = int(self.is_derived)
88
+ d["quarantined"] = int(self.quarantined)
89
+ d["provenance"] = json.dumps(self.provenance, default=str)
90
+ return d
91
+
92
+ @staticmethod
93
+ def from_row(row: dict) -> "Fact":
94
+ row = dict(row)
95
+ row["is_active"] = bool(row["is_active"])
96
+ row["is_derived"] = bool(row["is_derived"])
97
+ row["quarantined"] = bool(row.get("quarantined", 0))
98
+ row["provenance"] = json.loads(row.get("provenance") or "{}")
99
+ return Fact(**row)
100
+
101
+ def matches_scope(self, user_id: str | None, agent_id: str | None = None,
102
+ run_id: str | None = None) -> bool:
103
+ if user_id is not None and self.user_id != user_id:
104
+ return False
105
+ if agent_id is not None and self.agent_id != agent_id:
106
+ return False
107
+ if run_id is not None and self.run_id != run_id:
108
+ return False
109
+ return True
110
+
111
+
112
+ def make_fact(subject: str, relation: str, value: str, *,
113
+ now: datetime, valid_from: datetime | str | None = None,
114
+ valid_to: datetime | str | None = None, **kwargs) -> Fact:
115
+ """Convenience constructor normalizing timestamps to ISO strings."""
116
+ vf = iso(parse_ts(valid_from) or now)[:10] if valid_from else iso(now)[:10]
117
+ vt = iso(parse_ts(valid_to))[:10] if valid_to else None
118
+ return Fact(
119
+ id=kwargs.pop("id", "") or __import__("uuid").uuid4().hex,
120
+ subject=subject, relation=relation, value=value,
121
+ valid_from=vf, valid_to=vt, tx_from=iso(now), **kwargs)