cortexm 0.6.0__tar.gz → 0.6.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexm-0.6.0 → cortexm-0.6.1}/PKG-INFO +1 -1
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/__init__.py +1 -1
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/api/memory.py +35 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/patterns.py +18 -1
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/reader.py +42 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/synonyms.py +16 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/writer.py +162 -6
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/config.py +18 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/plugins/verbatim.py +28 -19
- cortexm-0.6.1/cortexm/trace/contradictions.py +126 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/trace/rules.py +18 -3
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/trace/store.py +98 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm.egg-info/PKG-INFO +1 -1
- {cortexm-0.6.0 → cortexm-0.6.1}/pyproject.toml +1 -1
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_v060_ir_pro.py +372 -6
- cortexm-0.6.0/cortexm/trace/contradictions.py +0 -69
- {cortexm-0.6.0 → cortexm-0.6.1}/LICENSE +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/README.md +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/context_m.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/accel.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/api/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/api/chaos.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/api/long_recall.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bench/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bench/abilities.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bench/baselines.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bench/beam_loader.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bench/generator.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bench/harness.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bench/messy.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bench/micro.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bench/ood.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bench/run.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/dates.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/decoders.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/enrich.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/extractor.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/fallback.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/fst.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/fusion.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/ir_pro.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/multilingual.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/negation.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/onnx_runtime.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/ppr.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/prefilter.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/query_extract.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/query_rewrite.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/recognizers.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/rerank.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/bridge/slang.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/cli.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/cognition/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/cognition/abstraction.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/cognition/analogy.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/cognition/engine.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/cognition/gaps.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/cognition/scanner.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/cortexm.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/creator.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/enterprise/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/enterprise/audit.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/enterprise/governance.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/errors.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/features/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/features/git.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/features/prefetch.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/features/zk.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/federation/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/federation/crdt.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/federation/fabric.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/federation/hlc.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/federation/node.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/federation/schema_report.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/federation/transport.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/index/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/index/nsg.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/kernel.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/markdown_io.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/mcp/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/mcp/server.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/metrics.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/migrate/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/migrate/importers.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/pipeline.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/plugins/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/plugins/security.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/plugins/structured.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/provenance/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/provenance/agent.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/provenance/cose.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/provenance/scitt.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/provenance/vc.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/router.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/security/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/security/crypto.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/security/hashes.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/security/injection.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/security/mind.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/security/permission.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/security/pii.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/security/rbac.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/security/sandbox.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/security/zk_hamming.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/security/zk_sql.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/server/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/server/metrics.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/server/rest.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/server/sparql.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/text/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/text/dissim.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/text/embedder.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/text/fuzzy.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/text/idiolect.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/text/labse.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/text/tokenizer.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/trace/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/trace/blob_arena.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/trace/consolidate.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/trace/dedup.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/trace/edges.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/trace/fact.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/trace/fade.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/trace/lifecycle.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/trace/rebuild.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/trace/structural.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/trace/tmt.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/trajectory_view.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/util.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/vsa/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/vsa/attribution.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/vsa/cleanup.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/vsa/codecs.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/vsa/hologram_overlay.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/vsa/index.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/vsa/ops.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/vsa/palace.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/vsa/role_vectors.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/vsa/slb.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/vsa/tlsh_trie.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm/vsa/working_memory.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm.egg-info/SOURCES.txt +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm.egg-info/dependency_links.txt +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm.egg-info/entry_points.txt +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm.egg-info/requires.txt +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/cortexm.egg-info/top_level.txt +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/setup.cfg +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_arxiv_improvements.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_bench_infra.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_bm25_chunk_recall_and_inspect_cli.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_cognition_and_provenance.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_engineering_push_2026_08_28.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_enterprise.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_fabric.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_federation.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_fusion_security.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_kernel.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_kinship_extraction.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_labse.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_list_superseded_intent.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_migration.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_new_modules.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_nsg.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_permission.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_ppr.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_public_api_smoke.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_reddit_steals_round3.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_rerank.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_research_steals.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_research_steals_round2.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_rust_accel.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_sandbox_enrich.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_sparql_rest_v2.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_tier443_abstention_fix.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_verbatim.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_wal_recovery.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.1}/tests/test_zk_sql.py +0 -0
|
@@ -18,7 +18,7 @@ Plugin kernel: ``from cortexm import Context, mount_default``
|
|
|
18
18
|
|
|
19
19
|
from __future__ import annotations
|
|
20
20
|
|
|
21
|
-
__version__ = "0.6.
|
|
21
|
+
__version__ = "0.6.1"
|
|
22
22
|
|
|
23
23
|
# μ=0 protocol counter: number of LLM invocations used by this process.
|
|
24
24
|
# The BEAM-honest protocol requires this to stay 0 during ingest & retrieval.
|
|
@@ -1030,6 +1030,41 @@ class Memory:
|
|
|
1030
1030
|
self.palace.close()
|
|
1031
1031
|
self.store.close()
|
|
1032
1032
|
|
|
1033
|
+
# ------------------------------------------------------------------
|
|
1034
|
+
# v0.6.1: BM25 + index maintenance facade
|
|
1035
|
+
# Exposed on Memory so users can call ``m.tune_bm25(k1=1.2, b=0.5)``
|
|
1036
|
+
# and ``m.optimize_index()`` without reaching into the verbatim
|
|
1037
|
+
# plugin. If the verbatim tier isn't mounted (e.g. disabled in
|
|
1038
|
+
# Config), these are no-ops so callers don't need to defensively
|
|
1039
|
+
# check.
|
|
1040
|
+
def tune_bm25(self, k1: float = 1.2, b: float = 0.75) -> None:
|
|
1041
|
+
"""Tune BM25 k1 (term saturation) + b (length normalization).
|
|
1042
|
+
|
|
1043
|
+
Lucene defaults: k1=1.2, b=0.75. Our corpus (short chat chunks,
|
|
1044
|
+
avg ~12 tokens) tends to prefer slightly higher saturation and
|
|
1045
|
+
weaker length norm; the v0.6.0 defaults were k1=1.5, b=0.75.
|
|
1046
|
+
Run ``scripts/tune_bm25_canonical.py`` to grid-search on the
|
|
1047
|
+
canonical LongMemEval sample and pick the best for your data.
|
|
1048
|
+
|
|
1049
|
+
Takes effect on the next ``search()`` call.
|
|
1050
|
+
"""
|
|
1051
|
+
if self._verbatim is not None:
|
|
1052
|
+
self._verbatim.tune_bm25(k1=k1, b=b)
|
|
1053
|
+
# Persist on the config too so reopens honor the tuning.
|
|
1054
|
+
self.config.bm25_k1 = k1
|
|
1055
|
+
self.config.bm25_b = b
|
|
1056
|
+
|
|
1057
|
+
def optimize_index(self) -> dict:
|
|
1058
|
+
"""VACUUM + FTS5 optimize + WAL checkpoint.
|
|
1059
|
+
|
|
1060
|
+
Run this after large bulk ingests to fold the WAL back into
|
|
1061
|
+
the main .db file, reclaim deleted-page space, and merge the
|
|
1062
|
+
FTS5 b-tree segments. Idempotent; safe to call any time.
|
|
1063
|
+
"""
|
|
1064
|
+
if self._verbatim is not None:
|
|
1065
|
+
return self._verbatim.optimize_index()
|
|
1066
|
+
return {"optimized": False, "reason": "verbatim tier not mounted"}
|
|
1067
|
+
|
|
1033
1068
|
def _reopen(self) -> None:
|
|
1034
1069
|
"""Rebind every component to a freshly-opened store (post-restore)."""
|
|
1035
1070
|
from cortexm.security.pii import PIIGuard, PIIVault
|
|
@@ -156,7 +156,8 @@ def _works(m, ctx, sp, ts, sent):
|
|
|
156
156
|
|
|
157
157
|
|
|
158
158
|
@pattern("joined_org",
|
|
159
|
-
rf"\bi\s+(?:
|
|
159
|
+
rf"\bi\s+(?:(?:just|finally|recently|recently\s+just)?\s+)?"
|
|
160
|
+
rf"(?:joined|started(?:\s+working)?\s+at|got a job at|moved to a job at)\s+{WORK_AT}")
|
|
160
161
|
def _joined(m, ctx, sp, ts, sent):
|
|
161
162
|
v = clean_value(m.group("val"))
|
|
162
163
|
return [Candidate("SELF", "works_at", v, 0.92, "joined_org",
|
|
@@ -389,6 +390,22 @@ def _pet(m, ctx, sp, ts, sent):
|
|
|
389
390
|
0.88, "pet")]
|
|
390
391
|
|
|
391
392
|
|
|
393
|
+
# v0.6.1: "my dog's name is Charlie" / "my dog is named Charlie" /
|
|
394
|
+
# "my dog is called Charlie" — the apostrophe-s + "name is" surface form
|
|
395
|
+
# (LongMemEval canonical Q). The base _pet pattern above requires the
|
|
396
|
+
# is-named/called verb form; this pattern catches the possessive variant.
|
|
397
|
+
@pattern("pet_named",
|
|
398
|
+
rf"\bmy\s+(?P<kind>dog|cat|bird|rabbit)"
|
|
399
|
+
rf"(?:'s\s+(?:name\s+is|is\s+named|is\s+called)"
|
|
400
|
+
rf"|\s+is\s+(?:named|called))"
|
|
401
|
+
rf"\s+(?P<val>{NAME})")
|
|
402
|
+
def _pet_named(m, ctx, sp, ts, sent):
|
|
403
|
+
return [Candidate(
|
|
404
|
+
"SELF", "has_pet",
|
|
405
|
+
f"{m.group('kind')} named {clean_value(m.group('val'))}",
|
|
406
|
+
0.88, "pet_named")]
|
|
407
|
+
|
|
408
|
+
|
|
392
409
|
@pattern("hobby", rf"\bmy hobby is\s+(?P<val>[a-z][a-z ]{{2,40}})|\bin my free time\s+i\s+(?P<val2>[a-z][a-z ]{{2,40}})")
|
|
393
410
|
def _hobby(m, ctx, sp, ts, sent):
|
|
394
411
|
v = clean_value(m.group("val") or m.group("val2") or "")
|
|
@@ -841,6 +841,19 @@ class MemoryReader:
|
|
|
841
841
|
if tc_notes:
|
|
842
842
|
notes = (notes or []) + tc_notes
|
|
843
843
|
|
|
844
|
+
# v0.6.1: Negation-aware retrieval. If the user ingested a
|
|
845
|
+
# negation like "I don't eat meat", the reader surfaces it
|
|
846
|
+
# here so the judge sees the explicit "No — they stated..."
|
|
847
|
+
# signal BEFORE any positive fact lookup. μ=0: pure content-
|
|
848
|
+
# word overlap (cortexm.bridge.negation.is_negation_overlap).
|
|
849
|
+
if getattr(self.cfg, "negation_indexing_enabled", True):
|
|
850
|
+
try:
|
|
851
|
+
neg_notes = self._negation_notes(query, user_id)
|
|
852
|
+
if neg_notes:
|
|
853
|
+
notes = (neg_notes if not notes else notes + neg_notes)
|
|
854
|
+
except Exception:
|
|
855
|
+
pass
|
|
856
|
+
|
|
844
857
|
# --- query-aware triple pre-filter (HippoRAG 2 lineage) ------------
|
|
845
858
|
# Drop candidate facts that have low lexical+semantic+relation
|
|
846
859
|
# overlap with the query BEFORE fusion. HippoRAG 2 credits this
|
|
@@ -1410,6 +1423,35 @@ class MemoryReader:
|
|
|
1410
1423
|
notes.append("\n".join(lines))
|
|
1411
1424
|
return notes
|
|
1412
1425
|
|
|
1426
|
+
# ------------------------------------------------------- negation lookup
|
|
1427
|
+
def _negation_notes(self, query: str, user_id: str) -> list[str]:
|
|
1428
|
+
"""Surface ingested negations that overlap the query.
|
|
1429
|
+
|
|
1430
|
+
The writer wrote sentences like ``"I don't eat meat"`` into
|
|
1431
|
+
``negation_records`` (see writer._store_negations). When the
|
|
1432
|
+
user later asks ``"Do I eat meat?"``, we want the judge to
|
|
1433
|
+
see the explicit "No — they stated..." signal. μ=0: pure
|
|
1434
|
+
content-word overlap (≥2 shared content words) so we never
|
|
1435
|
+
spuriously suppress a real positive answer.
|
|
1436
|
+
"""
|
|
1437
|
+
try:
|
|
1438
|
+
from cortexm.bridge.negation import is_negation_overlap
|
|
1439
|
+
records = self.store.query_negation_records(user_id=user_id)
|
|
1440
|
+
if not records:
|
|
1441
|
+
return []
|
|
1442
|
+
notes: list[str] = []
|
|
1443
|
+
for rec in records:
|
|
1444
|
+
if not is_negation_overlap(query, rec):
|
|
1445
|
+
continue
|
|
1446
|
+
marker = rec.get("marker", "")
|
|
1447
|
+
sentence = rec.get("sentence", "")
|
|
1448
|
+
notes.append(
|
|
1449
|
+
f"NEGATION: user stated — \"{sentence}\" "
|
|
1450
|
+
f"(marker={marker!r})")
|
|
1451
|
+
return notes
|
|
1452
|
+
except Exception:
|
|
1453
|
+
return []
|
|
1454
|
+
|
|
1413
1455
|
# ------------------------------------------------------------- symbolic
|
|
1414
1456
|
def _symbolic_query(self, plan: QueryPlan, user_id, agent_id, run_id,
|
|
1415
1457
|
scope, k, query):
|
|
@@ -122,6 +122,22 @@ DEFAULT_CLUSTERS: Dict[str, List[str]] = {
|
|
|
122
122
|
"grandfather", "grandmother", "grandparent",
|
|
123
123
|
"uncle", "aunt", "cousin", "nephew", "niece",
|
|
124
124
|
],
|
|
125
|
+
# MEDICAL — surface-form variants of the same "this user has X"
|
|
126
|
+
# statement. Each phrase means the user carries a diagnosis or
|
|
127
|
+
# condition. The rewriter emits one expansion per variant so a
|
|
128
|
+
# query phrased as "Does Alice have diabetes?" surfaces chunks
|
|
129
|
+
# like "Alice was diagnosed with diabetes" or "Alice has the
|
|
130
|
+
# condition diabetes". μ=0: pure phrase substitution, no NER.
|
|
131
|
+
"medical": [
|
|
132
|
+
"diagnosed with", "diagnosed as", "diagnosed as having",
|
|
133
|
+
"has condition", "has the condition", "has a condition of",
|
|
134
|
+
"has been diagnosed with", "was diagnosed with",
|
|
135
|
+
"received a diagnosis of", "diagnosis of",
|
|
136
|
+
"suffers from", "suffering from", "afflicted with",
|
|
137
|
+
"has been treated for", "under treatment for",
|
|
138
|
+
"on medication for", "prescribed for",
|
|
139
|
+
"history of", "family history of", # weak variants — match
|
|
140
|
+
],
|
|
125
141
|
}
|
|
126
142
|
|
|
127
143
|
|
|
@@ -15,6 +15,7 @@ from datetime import datetime, timezone
|
|
|
15
15
|
from cortexm import metrics
|
|
16
16
|
from cortexm.bridge.extractor import Extractor
|
|
17
17
|
from cortexm.bridge.patterns import ExtractionContext
|
|
18
|
+
from cortexm.bridge.negation import extract_with_negation
|
|
18
19
|
from cortexm.config import Config
|
|
19
20
|
from cortexm.security.injection import scan as injection_scan
|
|
20
21
|
from cortexm.security.injection import contagion_scan
|
|
@@ -147,6 +148,57 @@ class MemoryWriter:
|
|
|
147
148
|
if text not in corpus:
|
|
148
149
|
corpus.append(text)
|
|
149
150
|
|
|
151
|
+
# ------------------------------------------------------------------
|
|
152
|
+
def _store_negations(self, *, text: str, user_id: str,
|
|
153
|
+
session_id: str | None,
|
|
154
|
+
source_tx_id: str | None,
|
|
155
|
+
agent_id: str | None = None,
|
|
156
|
+
created_at: datetime | None = None) -> int:
|
|
157
|
+
"""μ=0 negation routing.
|
|
158
|
+
|
|
159
|
+
Splits the text into non-negated + negated sentences
|
|
160
|
+
(cortexm.bridge.negation.extract_with_negation) and writes
|
|
161
|
+
each negated sentence into the ``negation_records`` table
|
|
162
|
+
on the trace store. Returns the number of negation rows
|
|
163
|
+
written (best-effort: never blocks ingest).
|
|
164
|
+
|
|
165
|
+
v0.6.1 wiring: this is the follow-up the v0.6.0 detector
|
|
166
|
+
was waiting for. Before this, the extractor saw
|
|
167
|
+
``"I don't eat meat"`` and, if a pattern fired before the
|
|
168
|
+
negation was checked, mis-extracted ``(+user, eats, meat)``
|
|
169
|
+
as a positive fact — the reader then hallucinated "Yes"
|
|
170
|
+
from the very text that denied it.
|
|
171
|
+
"""
|
|
172
|
+
if not getattr(self.cfg, "negation_indexing_enabled", True):
|
|
173
|
+
return 0
|
|
174
|
+
try:
|
|
175
|
+
split = extract_with_negation(text)
|
|
176
|
+
negs = split["negations"]
|
|
177
|
+
if not negs:
|
|
178
|
+
return 0
|
|
179
|
+
ts_s = iso(created_at) if created_at else iso(_now())
|
|
180
|
+
src_hash = self.store.hasher.hash_text(text)
|
|
181
|
+
n = 0
|
|
182
|
+
for rec in negs:
|
|
183
|
+
self.store.insert_negation_record(
|
|
184
|
+
user_id=user_id,
|
|
185
|
+
sentence=rec.get("sentence", ""),
|
|
186
|
+
marker=rec.get("marker", ""),
|
|
187
|
+
implied_subject=rec.get("implied_subject", "") or "",
|
|
188
|
+
agent_id=agent_id,
|
|
189
|
+
session_id=session_id,
|
|
190
|
+
source_tx_id=str(source_tx_id) if source_tx_id is not None else None,
|
|
191
|
+
source_hash=src_hash,
|
|
192
|
+
created_at=ts_s,
|
|
193
|
+
)
|
|
194
|
+
n += 1
|
|
195
|
+
return n
|
|
196
|
+
except Exception as e:
|
|
197
|
+
# best-effort — never block the write path
|
|
198
|
+
import sys as _sys
|
|
199
|
+
print(f"[negation] store failed: {e}", file=_sys.stderr)
|
|
200
|
+
return 0
|
|
201
|
+
|
|
150
202
|
# ------------------------------------------------------------------
|
|
151
203
|
def _unmess_cache(self) -> dict:
|
|
152
204
|
"""Lazy-init the unmess (idiolect + dissim) cache on this writer.
|
|
@@ -225,6 +277,27 @@ class MemoryWriter:
|
|
|
225
277
|
session_id=run_id or agent_id or user_id,
|
|
226
278
|
source_tx_id=_src_tx_id,
|
|
227
279
|
agent_id=agent_id)
|
|
280
|
+
# v0.6.1: split negated sentences out BEFORE the extractor
|
|
281
|
+
# runs. The negated sentences go into a separate
|
|
282
|
+
# ``negation_records`` table (so the reader can return
|
|
283
|
+
# "No — explicitly stated" later); the extractor now sees
|
|
284
|
+
# only the non-negated portion, which kills the
|
|
285
|
+
# "I don't eat meat" → (+user, eats, meat) mis-extraction.
|
|
286
|
+
# Best-effort: on any failure we fall back to the raw text
|
|
287
|
+
# so the write path is never blocked.
|
|
288
|
+
try:
|
|
289
|
+
if getattr(self.cfg, "negation_indexing_enabled", True):
|
|
290
|
+
neg_split = extract_with_negation(text)
|
|
291
|
+
self._store_negations(
|
|
292
|
+
text=text, user_id=user_id,
|
|
293
|
+
session_id=run_id or agent_id or user_id,
|
|
294
|
+
source_tx_id=str(_src_tx_id) if _src_tx_id is not None else None,
|
|
295
|
+
agent_id=agent_id, created_at=msg_time)
|
|
296
|
+
extraction_text = neg_split["positive_text"] or text
|
|
297
|
+
else:
|
|
298
|
+
extraction_text = text
|
|
299
|
+
except Exception:
|
|
300
|
+
extraction_text = text
|
|
228
301
|
verdict = injection_scan(text, self.cfg.quarantine_injection)
|
|
229
302
|
if not verdict.quarantined and self.cfg.quarantine_contagion:
|
|
230
303
|
cv = contagion_scan(text, self._tainted_corpus(user_id),
|
|
@@ -249,8 +322,8 @@ class MemoryWriter:
|
|
|
249
322
|
# The extractor's internal Bitap trigger widening handles
|
|
250
323
|
# misspelled trigger words. When unmess is OFF (bench baseline
|
|
251
324
|
# config), we run the raw text through the extractor unchanged.
|
|
252
|
-
clauses = self._unmess_text(
|
|
253
|
-
if self.cfg.unmess_enabled else [
|
|
325
|
+
clauses = self._unmess_text(extraction_text, user_id) \
|
|
326
|
+
if self.cfg.unmess_enabled else [extraction_text]
|
|
254
327
|
|
|
255
328
|
candidates = []
|
|
256
329
|
for clause in clauses:
|
|
@@ -425,6 +498,56 @@ class MemoryWriter:
|
|
|
425
498
|
self.store.end_batch()
|
|
426
499
|
return derived
|
|
427
500
|
|
|
501
|
+
# ------------------------------------------------------------------
|
|
502
|
+
def _max_vsa_overlap(self, fact: Fact, user_id: str) -> float:
|
|
503
|
+
"""Shannon tiered storage: compute the max cosine similarity
|
|
504
|
+
of this fact's VSA hologram vs. existing facts in the user's
|
|
505
|
+
scope. Returns 0.0 on cold-start (<shannon_min_facts facts)
|
|
506
|
+
or any failure. μ=0: deterministic cosine over the user's
|
|
507
|
+
palace vectors; no LLM, no statistics.
|
|
508
|
+
|
|
509
|
+
Used to gate ``palace.add()`` in the COMMIT/COEXIST and
|
|
510
|
+
SUPERSEDE branches of ``_apply_decision``. If the overlap
|
|
511
|
+
is above ``shannon_overlap_threshold`` (default 0.9), we
|
|
512
|
+
store the structured fact + chunk + edges (still findable
|
|
513
|
+
by BM25 + symbolic query) but SKIP the VSA palace.add — the
|
|
514
|
+
holographic superposition stays clean and retrieval stays
|
|
515
|
+
fast. Verbatim tier is unchanged; "doesn't forget" holds.
|
|
516
|
+
"""
|
|
517
|
+
if not getattr(self.cfg, "shannon_tiered_storage", True):
|
|
518
|
+
return 0.0
|
|
519
|
+
try:
|
|
520
|
+
existing = self.store.query_facts(
|
|
521
|
+
user_id=user_id, active=True, limit=500)
|
|
522
|
+
if len(existing) < int(getattr(
|
|
523
|
+
self.cfg, "shannon_min_facts", 10)):
|
|
524
|
+
return 0.0 # cold-start: not enough signal yet
|
|
525
|
+
new_vec = self.palace.encode_fact(fact)
|
|
526
|
+
best = 0.0
|
|
527
|
+
# Scan in chunks of 64 to bound memory on the 4GB box.
|
|
528
|
+
for f in existing:
|
|
529
|
+
# Skip self (in case fact is already partially committed)
|
|
530
|
+
if f.id == fact.id:
|
|
531
|
+
continue
|
|
532
|
+
try:
|
|
533
|
+
# encode each existing fact once and cosine-compare
|
|
534
|
+
ev = self.palace.encode_fact(f)
|
|
535
|
+
# cosine via dot product on normalized vectors
|
|
536
|
+
a = new_vec.ravel()
|
|
537
|
+
b = ev.ravel()
|
|
538
|
+
na = float((a @ a) ** 0.5) or 1.0
|
|
539
|
+
nb = float((b @ b) ** 0.5) or 1.0
|
|
540
|
+
sim = float((a @ b) / (na * nb))
|
|
541
|
+
if sim > best:
|
|
542
|
+
best = sim
|
|
543
|
+
if best >= 0.99:
|
|
544
|
+
break # near-identical — no need to scan more
|
|
545
|
+
except Exception:
|
|
546
|
+
continue
|
|
547
|
+
return best
|
|
548
|
+
except Exception:
|
|
549
|
+
return 0.0
|
|
550
|
+
|
|
428
551
|
# ------------------------------------------------------------------
|
|
429
552
|
def _apply_decision(self, fact: Fact, decision, commit, chunk_id,
|
|
430
553
|
results) -> int:
|
|
@@ -478,14 +601,47 @@ class MemoryWriter:
|
|
|
478
601
|
reason=f"superseded: {decision.note}")
|
|
479
602
|
self.store.insert_fact(fact, commit)
|
|
480
603
|
self.store.add_edge(fact.id, chunk_id, "EXTRACTED_FROM")
|
|
481
|
-
|
|
482
|
-
|
|
604
|
+
# v0.6.1: Shannon tiered storage. For SUPERSEDE we usually
|
|
605
|
+
# palace.add (the new fact is the canonical value now).
|
|
606
|
+
# But if the new fact has high VSA overlap to existing
|
|
607
|
+
# memory, skip the palace.add — the holographic signal
|
|
608
|
+
# is already there.
|
|
609
|
+
overlap = self._max_vsa_overlap(fact, fact.user_id or "default")
|
|
610
|
+
if overlap > float(getattr(self.cfg, "shannon_overlap_threshold", 0.9)):
|
|
611
|
+
fact.provenance = {
|
|
612
|
+
**fact.provenance,
|
|
613
|
+
"shannon_tier": "verbatim_only",
|
|
614
|
+
"shannon_overlap": round(overlap, 3),
|
|
615
|
+
}
|
|
616
|
+
results.append(self._result(
|
|
617
|
+
fact, "SUPERSEDED",
|
|
618
|
+
f"{decision.note}; shannon_tier=verbatim_only "
|
|
619
|
+
f"(overlap={overlap:.3f})"))
|
|
620
|
+
else:
|
|
621
|
+
self.palace.add(fact.id, self.palace.encode_fact(fact))
|
|
622
|
+
results.append(self._result(fact, "SUPERSEDED", decision.note))
|
|
483
623
|
return 1
|
|
484
624
|
# COMMIT / COEXIST
|
|
485
625
|
self.store.insert_fact(fact, commit)
|
|
486
626
|
self.store.add_edge(fact.id, chunk_id, "EXTRACTED_FROM")
|
|
487
|
-
|
|
488
|
-
|
|
627
|
+
# v0.6.1: Shannon tiered storage. For brand-new COMMIT/COEXIST
|
|
628
|
+
# facts, check VSA overlap before adding the hologram. High
|
|
629
|
+
# overlap (≥0.9) → store the fact + chunk + edges but skip
|
|
630
|
+
# palace.add (verbatim tier still catches exact-match via BM25).
|
|
631
|
+
overlap = self._max_vsa_overlap(fact, fact.user_id or "default")
|
|
632
|
+
if overlap > float(getattr(self.cfg, "shannon_overlap_threshold", 0.9)):
|
|
633
|
+
fact.provenance = {
|
|
634
|
+
**fact.provenance,
|
|
635
|
+
"shannon_tier": "verbatim_only",
|
|
636
|
+
"shannon_overlap": round(overlap, 3),
|
|
637
|
+
}
|
|
638
|
+
results.append(self._result(
|
|
639
|
+
fact, "ADD",
|
|
640
|
+
f"shannon_tier=verbatim_only "
|
|
641
|
+
f"(overlap={overlap:.3f}; verbatim tier still findable)"))
|
|
642
|
+
else:
|
|
643
|
+
self.palace.add(fact.id, self.palace.encode_fact(fact))
|
|
644
|
+
results.append(self._result(fact, "ADD", decision.note))
|
|
489
645
|
return 1
|
|
490
646
|
|
|
491
647
|
# ------------------------------------------------------------------
|
|
@@ -399,6 +399,24 @@ class Config:
|
|
|
399
399
|
# (Config.labse_enabled) handles the embedding for non-English text.
|
|
400
400
|
multilingual_routing_enabled: bool = True
|
|
401
401
|
|
|
402
|
+
# Shannon entropy-weighted storage (tiered precision compromise).
|
|
403
|
+
# Pure entropy filter — "skip storing redundant facts" — violates
|
|
404
|
+
# the "doesn't forget" promise: a fact with VSA overlap > 0.9 to
|
|
405
|
+
# existing memory gets dropped, and later you ask "what's my
|
|
406
|
+
# dog's name?" and the system has forgotten. The safer compromise
|
|
407
|
+
# is TIERED PRECISION: store the verbatim chunk + structured fact
|
|
408
|
+
# (still findable by BM25 + symbolic query) but SKIP the VSA
|
|
409
|
+
# palace.add for high-overlap facts. This deduplicates the
|
|
410
|
+
# holographic superposition (smaller palace = faster retrieval)
|
|
411
|
+
# without losing any information.
|
|
412
|
+
#
|
|
413
|
+
# Cold-start guard: skip the overlap check until the user has at
|
|
414
|
+
# least ``shannon_min_facts`` facts (default 10) so the first few
|
|
415
|
+
# noisy facts don't get spuriously tiered-down.
|
|
416
|
+
shannon_tiered_storage: bool = True
|
|
417
|
+
shannon_overlap_threshold: float = 0.9
|
|
418
|
+
shannon_min_facts: int = 10
|
|
419
|
+
|
|
402
420
|
# IR fundamentals (Lucene/Solr-grade primitives). See
|
|
403
421
|
# cortexm/bridge/ir_pro.py for the implementations.
|
|
404
422
|
# * query_cache: LRU on (query, user_id, k) — invalidated on add()
|
|
@@ -334,35 +334,44 @@ class VerbatimPlugin:
|
|
|
334
334
|
# expansion, so the existing BM25 + cosine path is preserved
|
|
335
335
|
# (additive). Expansions run additional BM25 passes; results
|
|
336
336
|
# are unioned by rowid with dedup.
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
except Exception:
|
|
346
|
-
queries = [query] # rewriter failure → fall back to original
|
|
347
|
-
|
|
348
|
-
# Run the existing search path on the FIRST query (original).
|
|
349
|
-
# Then for any additional expansions, run BM25-only and append
|
|
350
|
-
# new rowids to the result pool. The dense cosine path runs
|
|
351
|
-
# once on the original query (cheaper, and the fused score is
|
|
352
|
-
# most meaningful for the original query embedding).
|
|
337
|
+
#
|
|
338
|
+
# v0.6.1: LAZY expansion. The QueryRewriter adds ~2–4ms to every
|
|
339
|
+
# search() call. The 95% of queries that hit on the first try
|
|
340
|
+
# don't need expansion. So we try the original query FIRST; only
|
|
341
|
+
# if it returns empty do we fall back to the rewritten variants.
|
|
342
|
+
# This restores the v0.5.x 1.6ms read p50 for the common case
|
|
343
|
+
# while keeping the v0.6.0 synonym/FST/slang safety net for the
|
|
344
|
+
# long-tail paraphrase queries.
|
|
353
345
|
primary = self._search_single(
|
|
354
|
-
query=
|
|
346
|
+
query=query, user_id=user_id, k=k,
|
|
355
347
|
session_id=session_id, agent_id=agent_id)
|
|
356
|
-
if
|
|
348
|
+
if primary or not self.query_rewrite_enabled:
|
|
349
|
+
if self.query_cache_enabled:
|
|
350
|
+
self._query_cache.put(cache_key, primary)
|
|
351
|
+
return primary
|
|
352
|
+
# Primary returned empty — fall back to expansions
|
|
353
|
+
if self._query_rewriter is None:
|
|
354
|
+
from cortexm.bridge.query_rewrite import QueryRewriter
|
|
355
|
+
self._query_rewriter = QueryRewriter(
|
|
356
|
+
max_expansions=self.max_expansions)
|
|
357
|
+
try:
|
|
358
|
+
queries = self._query_rewriter.rewrite(query)
|
|
359
|
+
except Exception:
|
|
360
|
+
queries = [query]
|
|
361
|
+
# Drop the original (already tried); expansions only
|
|
362
|
+
if queries and queries[0] == query:
|
|
363
|
+
queries = queries[1:]
|
|
364
|
+
if not queries:
|
|
357
365
|
if self.query_cache_enabled:
|
|
358
366
|
self._query_cache.put(cache_key, primary)
|
|
359
367
|
return primary
|
|
360
368
|
# Additional expansions: BM25-only, dedup against primary rowids
|
|
369
|
+
# (primary is empty here, but keep the guard for safety)
|
|
361
370
|
seen_rowids = {h.chunk_id for h in primary}
|
|
362
371
|
extra: list = []
|
|
363
372
|
# Cap total candidates at 2*k so the union stays bounded.
|
|
364
373
|
budget = max(0, 2 * k - len(primary))
|
|
365
|
-
for q in queries[
|
|
374
|
+
for q in queries[:budget + 1]:
|
|
366
375
|
if budget <= 0:
|
|
367
376
|
break
|
|
368
377
|
extra_hits = self._bm25_only(
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""Deterministic contradiction detection & truth maintenance.
|
|
2
|
+
|
|
3
|
+
Phase rules from Section 1.1:
|
|
4
|
+
1. Classification — relation → semantic category, single/multi valued
|
|
5
|
+
2. Contradiction — exact + fuzzy (Jaccard/Levenshtein) match on
|
|
6
|
+
subject-relation pairs; latest-value-wins for
|
|
7
|
+
single-valued relations, append for multi-valued
|
|
8
|
+
3. Interference-aware lifecycle — see lifecycle.py
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
from enum import Enum
|
|
15
|
+
|
|
16
|
+
from cortexm.trace.fact import Fact, SINGLE_VALUED
|
|
17
|
+
from cortexm.trace.store import TraceStore
|
|
18
|
+
from cortexm.util import similarity
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
# v0.6.1: relation-name aliasing. The extractor emits several surface
|
|
22
|
+
# relations that all mean the same semantic thing:
|
|
23
|
+
# * moved_to / relocated_to / shifted_to — all "where the user lives"
|
|
24
|
+
# * joined / started_at / got_job_at — all "where the user works"
|
|
25
|
+
# Without this map, "I moved to Berlin" (moved_to) and "I live in Munich"
|
|
26
|
+
# (lives_in) sit in DIFFERENT relation slots, so find_conflicts treats
|
|
27
|
+
# them as unrelated facts — the bi-temporal SUPERSEDE chain never fires
|
|
28
|
+
# and the reader answers "Berlin AND Munich" instead of "Munich (latest)".
|
|
29
|
+
# μ=0: a static dict. No model, no inference.
|
|
30
|
+
RELATION_ALIASES: dict[str, str] = {
|
|
31
|
+
# residence cluster
|
|
32
|
+
"moved_to": "lives_in",
|
|
33
|
+
"relocated_to": "lives_in",
|
|
34
|
+
"shifted_to": "lives_in",
|
|
35
|
+
"based_in": "lives_in",
|
|
36
|
+
# employment cluster
|
|
37
|
+
"started_at": "works_at",
|
|
38
|
+
"joined": "works_at",
|
|
39
|
+
"got_job_at": "works_at",
|
|
40
|
+
"hired_at": "works_at",
|
|
41
|
+
# education cluster (aliasing 'study_at' / 'enrolled_at' to 'studies_at')
|
|
42
|
+
"study_at": "studies_at",
|
|
43
|
+
"enrolled_at": "studies_at",
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def canonical_relation(relation: str) -> str:
|
|
48
|
+
"""Return the canonical relation name. Used by find_conflicts and
|
|
49
|
+
the reader's symbolic query path so alias-of-alias queries land in
|
|
50
|
+
the same bucket. Falls through unchanged for relations without an
|
|
51
|
+
alias."""
|
|
52
|
+
return RELATION_ALIASES.get(relation, relation)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class Action(str, Enum):
|
|
56
|
+
COMMIT = "commit" # brand-new fact
|
|
57
|
+
MERGE = "merge" # near-duplicate: reinforce existing
|
|
58
|
+
SUPERSEDE = "supersede" # contradiction on single-valued relation
|
|
59
|
+
COEXIST = "coexist" # contradiction on multi-valued relation
|
|
60
|
+
SKIP = "skip" # exact duplicate
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass
|
|
64
|
+
class Conflict:
|
|
65
|
+
action: Action
|
|
66
|
+
existing: list[Fact]
|
|
67
|
+
note: str = ""
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def find_conflicts(store: TraceStore, candidate: Fact) -> Conflict:
|
|
71
|
+
"""Decide how ``candidate`` interacts with active memory.
|
|
72
|
+
|
|
73
|
+
v0.6.1: relation aliasing. Before querying for existing facts on
|
|
74
|
+
the same (subject, relation) we look up the canonical relation
|
|
75
|
+
so a ``moved_to = Berlin`` candidate sees the prior ``lives_in =
|
|
76
|
+
Munich`` and fires SUPERSEDE (not COEXIST-unknown).
|
|
77
|
+
"""
|
|
78
|
+
canon_rel = canonical_relation(candidate.relation)
|
|
79
|
+
# Look up BOTH the candidate's literal relation AND the canonical
|
|
80
|
+
# form — older facts may be stored under either name. We union the
|
|
81
|
+
# two queries so a partial migration (some facts pre-aliasing, some
|
|
82
|
+
# post-) still resolves correctly.
|
|
83
|
+
matching: list[Fact] = []
|
|
84
|
+
for rel in {candidate.relation, canon_rel}:
|
|
85
|
+
matching.extend(
|
|
86
|
+
f for f in store.query_facts(
|
|
87
|
+
subject=candidate.subject, relation=rel,
|
|
88
|
+
user_id=candidate.user_id, active=True)
|
|
89
|
+
if f.id != candidate.id)
|
|
90
|
+
# de-dup by fact id (a fact stored under the canonical rel could
|
|
91
|
+
# appear in both queries if candidate.relation == canon_rel)
|
|
92
|
+
seen_ids: set[str] = set()
|
|
93
|
+
existing: list[Fact] = []
|
|
94
|
+
for f in matching:
|
|
95
|
+
if f.id in seen_ids:
|
|
96
|
+
continue
|
|
97
|
+
seen_ids.add(f.id)
|
|
98
|
+
existing.append(f)
|
|
99
|
+
if not existing:
|
|
100
|
+
return Conflict(Action.COMMIT, [])
|
|
101
|
+
|
|
102
|
+
exact = [f for f in existing if f.value.strip().lower() == candidate.value.strip().lower()]
|
|
103
|
+
if exact:
|
|
104
|
+
return Conflict(Action.SKIP, exact, "exact duplicate")
|
|
105
|
+
|
|
106
|
+
# low-salience mention anchors: exact-dup semantics only (no fuzzy
|
|
107
|
+
# quadratic scans — mention streams are high-volume, low-signal)
|
|
108
|
+
if candidate.relation in ("mentioned", "event", "instruction"):
|
|
109
|
+
if candidate.relation == "mentioned":
|
|
110
|
+
return Conflict(Action.COEXIST, existing, "mention anchor recorded")
|
|
111
|
+
|
|
112
|
+
near = [f for f in existing
|
|
113
|
+
if similarity(f.value, candidate.value) >= 0.92]
|
|
114
|
+
if near:
|
|
115
|
+
return Conflict(Action.MERGE, near, "near-duplicate merged; reinforcement +1")
|
|
116
|
+
|
|
117
|
+
# SINGLE_VALUED is keyed by canonical relation name; check both
|
|
118
|
+
single = (candidate.relation in SINGLE_VALUED
|
|
119
|
+
or canon_rel in SINGLE_VALUED)
|
|
120
|
+
if single:
|
|
121
|
+
# newest valid_from wins reality; old fact gets valid_to
|
|
122
|
+
target = max(existing, key=lambda f: (f.valid_from, f.tx_from))
|
|
123
|
+
return Conflict(Action.SUPERSEDE, [target],
|
|
124
|
+
f"contradiction on single-valued '{canon_rel}'")
|
|
125
|
+
return Conflict(Action.COEXIST, existing,
|
|
126
|
+
f"conflicting values on multi-valued '{canon_rel}' coexist")
|
|
@@ -132,7 +132,18 @@ class RuleEngine:
|
|
|
132
132
|
Derived facts inherit the USER SCOPE of their premises — without
|
|
133
133
|
this, ``team_uses(X, L) :- member_of(X, T), uses(T, L)`` derives a
|
|
134
134
|
fact under ``default`` that user0's reader can never see (scope
|
|
135
|
-
filter drops it), silently losing every multi-hop answer.
|
|
135
|
+
filter drops it), silently losing every multi-hop answer.
|
|
136
|
+
|
|
137
|
+
v0.6.1 fix: the dedup check now looks at ALL facts (active OR
|
|
138
|
+
retired) with the same (subject, relation, value, scope). Before
|
|
139
|
+
this, the rule ``lives_in(X, C) :- moved_to(X, C)`` would re-
|
|
140
|
+
derive ``lives_in=Berlin`` AFTER the candidate ``lives_in=Munich``
|
|
141
|
+
fired SUPERSEDE on it — because the retired Berlin fact was
|
|
142
|
+
invisible to ``active=True`` filter and the rule happily re-
|
|
143
|
+
created it, undoing the supersession. With this fix, once a
|
|
144
|
+
(subject, relation, value) tuple has been derived (and possibly
|
|
145
|
+
later retired by a SUPERSEDE), the rule won't re-materialize it.
|
|
146
|
+
"""
|
|
136
147
|
now = now or _dt.datetime.now(_dt.timezone.utc)
|
|
137
148
|
derived_new: list[Fact] = []
|
|
138
149
|
seen: set[tuple] = set()
|
|
@@ -149,8 +160,12 @@ class RuleEngine:
|
|
|
149
160
|
if key in seen:
|
|
150
161
|
continue
|
|
151
162
|
seen.add(key)
|
|
152
|
-
|
|
153
|
-
|
|
163
|
+
# v0.6.1: dedup against ALL facts (active OR retired)
|
|
164
|
+
# — re-deriving a fact that was just superseded
|
|
165
|
+
# would silently undo the supersession.
|
|
166
|
+
if self.store.query_facts(subject=s, relation=rel,
|
|
167
|
+
value=v, user_id=scope,
|
|
168
|
+
active=None):
|
|
154
169
|
continue
|
|
155
170
|
f = Fact(
|
|
156
171
|
id=new_id(), subject=s, relation=rel, value=v,
|