cortexm 0.6.0__tar.gz → 0.6.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexm-0.6.0/cortexm.egg-info → cortexm-0.6.4}/PKG-INFO +67 -20
- cortexm-0.6.0/PKG-INFO → cortexm-0.6.4/README.md +43 -39
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/__init__.py +1 -1
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/api/memory.py +96 -7
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/extractor.py +24 -1
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/fallback.py +19 -1
- cortexm-0.6.4/cortexm/bridge/fst_real.py +270 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/patterns.py +18 -1
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/reader.py +113 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/synonyms.py +16 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/writer.py +148 -13
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/config.py +69 -2
- cortexm-0.6.4/cortexm/experimental/__init__.py +29 -0
- cortexm-0.6.4/cortexm/experimental/coherence.py +123 -0
- cortexm-0.6.4/cortexm/experimental/graph_recall.py +193 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/federation/hlc.py +2 -2
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/mcp/server.py +91 -58
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/migrate/importers.py +22 -15
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/plugins/verbatim.py +49 -33
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/provenance/cose.py +14 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/provenance/vc.py +10 -0
- cortexm-0.6.4/cortexm/py.typed +1 -0
- cortexm-0.6.4/cortexm/security/hamming_attestation.py +291 -0
- cortexm-0.6.4/cortexm/security/zk_proofs.py +670 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/text/embedder.py +95 -4
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/text/idiolect.py +38 -6
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/text/labse.py +35 -1
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/text/tokenizer.py +7 -0
- cortexm-0.6.4/cortexm/trace/contradictions.py +126 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/rules.py +18 -3
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/store.py +138 -8
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/palace.py +5 -4
- cortexm-0.6.4/cortexm.egg-info/PKG-INFO +181 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm.egg-info/SOURCES.txt +11 -3
- cortexm-0.6.4/cortexm.egg-info/requires.txt +24 -0
- cortexm-0.6.4/cortexm.egg-info/top_level.txt +1 -0
- cortexm-0.6.4/pyproject.toml +68 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_arxiv_improvements.py +25 -66
- cortexm-0.6.4/tests/test_experimental_v064.py +231 -0
- cortexm-0.6.4/tests/test_infrastructure_regressions.py +70 -0
- cortexm-0.6.4/tests/test_mcp_zk_tools.py +117 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_permission.py +3 -3
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_reddit_steals_round3.py +2 -1
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_research_steals_round2.py +19 -3
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_v060_ir_pro.py +372 -6
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_wal_recovery.py +2 -0
- cortexm-0.6.4/tests/test_zk_soundness.py +190 -0
- cortexm-0.6.4/tests/test_zk_sql.py +144 -0
- cortexm-0.6.0/README.md +0 -99
- cortexm-0.6.0/context_m.py +0 -17
- cortexm-0.6.0/cortexm/security/zk_hamming.py +0 -142
- cortexm-0.6.0/cortexm/security/zk_sql.py +0 -485
- cortexm-0.6.0/cortexm/trace/contradictions.py +0 -69
- cortexm-0.6.0/cortexm.egg-info/requires.txt +0 -10
- cortexm-0.6.0/cortexm.egg-info/top_level.txt +0 -2
- cortexm-0.6.0/pyproject.toml +0 -59
- cortexm-0.6.0/tests/test_zk_sql.py +0 -326
- {cortexm-0.6.0 → cortexm-0.6.4}/LICENSE +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/accel.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/api/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/api/chaos.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/api/long_recall.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/abilities.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/baselines.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/beam_loader.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/generator.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/harness.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/messy.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/micro.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/ood.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/run.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/dates.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/decoders.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/enrich.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/fst.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/fusion.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/ir_pro.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/multilingual.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/negation.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/onnx_runtime.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/ppr.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/prefilter.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/query_extract.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/query_rewrite.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/recognizers.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/rerank.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/slang.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cli.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cognition/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cognition/abstraction.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cognition/analogy.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cognition/engine.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cognition/gaps.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cognition/scanner.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cortexm.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/creator.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/enterprise/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/enterprise/audit.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/enterprise/governance.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/errors.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/features/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/features/git.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/features/prefetch.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/features/zk.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/federation/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/federation/crdt.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/federation/fabric.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/federation/node.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/federation/schema_report.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/federation/transport.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/index/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/index/nsg.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/kernel.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/markdown_io.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/mcp/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/metrics.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/migrate/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/pipeline.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/plugins/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/plugins/security.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/plugins/structured.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/provenance/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/provenance/agent.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/provenance/scitt.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/router.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/crypto.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/hashes.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/injection.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/mind.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/permission.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/pii.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/rbac.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/sandbox.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/server/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/server/metrics.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/server/rest.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/server/sparql.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/text/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/text/dissim.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/text/fuzzy.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/blob_arena.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/consolidate.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/dedup.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/edges.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/fact.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/fade.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/lifecycle.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/rebuild.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/structural.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/tmt.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trajectory_view.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/util.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/__init__.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/attribution.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/cleanup.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/codecs.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/hologram_overlay.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/index.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/ops.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/role_vectors.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/slb.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/tlsh_trie.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/working_memory.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm.egg-info/dependency_links.txt +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/cortexm.egg-info/entry_points.txt +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/setup.cfg +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_bench_infra.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_bm25_chunk_recall_and_inspect_cli.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_cognition_and_provenance.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_engineering_push_2026_08_28.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_enterprise.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_fabric.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_federation.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_fusion_security.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_kernel.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_kinship_extraction.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_labse.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_list_superseded_intent.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_migration.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_new_modules.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_nsg.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_ppr.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_public_api_smoke.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_rerank.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_research_steals.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_rust_accel.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_sandbox_enrich.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_sparql_rest_v2.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_tier443_abstention_fix.py +0 -0
- {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_verbatim.py +0 -0
|
@@ -1,36 +1,44 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cortexm
|
|
3
|
-
Version: 0.6.
|
|
4
|
-
Summary:
|
|
5
|
-
Author: Context-M
|
|
6
|
-
License
|
|
3
|
+
Version: 0.6.4
|
|
4
|
+
Summary: Context-M: deterministic, auditable, zero-cost memory for AI agents
|
|
5
|
+
Author-email: Context-M Team <dev@context-m.ai>
|
|
6
|
+
License: MIT
|
|
7
7
|
Project-URL: Homepage, https://github.com/ssmurfgg04-gif/context-m
|
|
8
|
-
Project-URL: Documentation, https://github.com/ssmurfgg04-gif/context-m
|
|
8
|
+
Project-URL: Documentation, https://github.com/ssmurfgg04-gif/context-m#readme
|
|
9
9
|
Project-URL: Repository, https://github.com/ssmurfgg04-gif/context-m
|
|
10
10
|
Project-URL: Issues, https://github.com/ssmurfgg04-gif/context-m/issues
|
|
11
|
-
|
|
12
|
-
Keywords: agent-memory,llm-memory,long-term-memory,mem0,memgpt,letta,zep,chroma,deterministic-ai,local-first,vector-symbolic-architecture,provenance,bi-temporal,hippocampus,context-engineering,rag,mcp,neuro-symbolic,hrr,hdc,self-hosted
|
|
11
|
+
Keywords: memory,ai,agents,deterministic,audit,zk
|
|
13
12
|
Classifier: Development Status :: 4 - Beta
|
|
14
13
|
Classifier: Intended Audience :: Developers
|
|
15
|
-
Classifier:
|
|
16
|
-
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
16
|
Classifier: Programming Language :: Python :: 3.10
|
|
18
17
|
Classifier: Programming Language :: Python :: 3.11
|
|
19
18
|
Classifier: Programming Language :: Python :: 3.12
|
|
20
19
|
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
21
|
-
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
22
|
-
Classifier: Topic :: Database
|
|
23
|
-
Classifier: Typing :: Typed
|
|
24
20
|
Requires-Python: >=3.10
|
|
25
21
|
Description-Content-Type: text/markdown
|
|
26
22
|
License-File: LICENSE
|
|
27
|
-
Requires-Dist: numpy>=1.24
|
|
23
|
+
Requires-Dist: numpy>=1.24.0
|
|
24
|
+
Requires-Dist: cryptography>=41.0.0
|
|
25
|
+
Requires-Dist: fastecdsa>=2.3.0
|
|
28
26
|
Provides-Extra: blake3
|
|
29
|
-
Requires-Dist: blake3>=
|
|
27
|
+
Requires-Dist: blake3>=0.3.0; extra == "blake3"
|
|
30
28
|
Provides-Extra: crypto
|
|
31
|
-
Requires-Dist: cryptography>=
|
|
29
|
+
Requires-Dist: cryptography>=41.0.0; extra == "crypto"
|
|
30
|
+
Provides-Extra: fst
|
|
31
|
+
Requires-Dist: marisa-trie>=1.2.0; extra == "fst"
|
|
32
32
|
Provides-Extra: test
|
|
33
|
-
Requires-Dist: pytest>=7; extra == "test"
|
|
33
|
+
Requires-Dist: pytest>=7.0; extra == "test"
|
|
34
|
+
Requires-Dist: pytest-cov>=4.0; extra == "test"
|
|
35
|
+
Provides-Extra: all
|
|
36
|
+
Requires-Dist: blake3>=0.3.0; extra == "all"
|
|
37
|
+
Requires-Dist: cryptography>=41.0.0; extra == "all"
|
|
38
|
+
Requires-Dist: fastecdsa>=2.3.0; extra == "all"
|
|
39
|
+
Requires-Dist: marisa-trie>=1.2.0; extra == "all"
|
|
40
|
+
Requires-Dist: pytest>=7.0; extra == "all"
|
|
41
|
+
Requires-Dist: pytest-cov>=4.0; extra == "all"
|
|
34
42
|
Dynamic: license-file
|
|
35
43
|
|
|
36
44
|
<div align="center">
|
|
@@ -72,15 +80,53 @@ m.search("Where does Alice work?", user_id="alice")
|
|
|
72
80
|
|
|
73
81
|
### Canonical LongMemEval — μ=0, $0, on a 4GB laptop
|
|
74
82
|
|
|
75
|
-
| | cortexm v0.
|
|
83
|
+
| | cortexm v0.6.4 | MemPalace (honest E2E) |
|
|
76
84
|
|---|---|---|
|
|
77
|
-
| canonical LongMemEval (
|
|
85
|
+
| **canonical LongMemEval (500-Q full corpus)** | **97.4% (487/500)** | ~96.6% (retrieval-only, no QA) |
|
|
86
|
+
| single_session | **100.0%** | — |
|
|
87
|
+
| knowledge_update | **100.0%** | — |
|
|
88
|
+
| multi_session | 94.74% | — |
|
|
89
|
+
| temporal_reasoning | 95.49% | — |
|
|
78
90
|
| LLM calls (ingest + retrieval + judge) | 0 | 0 |
|
|
79
91
|
| monthly cost | $0 | $0 |
|
|
80
92
|
| determinism | byte-exact across 3× runs | byte-exact |
|
|
81
93
|
| owns your data | ✓ single `.db` file | ✓ |
|
|
82
94
|
|
|
83
|
-
**
|
|
95
|
+
**Full 500-question results** (v0.6.2 baseline; v0.6.4 re-run lands the experimental graph-recall + coherence modules below):
|
|
96
|
+
|
|
97
|
+
| Subtask | Score | Notes |
|
|
98
|
+
|---|---|---|
|
|
99
|
+
| **Overall** | **0.974 (487/500)** | Full corpus, not a proxy sample |
|
|
100
|
+
| single_session | **1.000** | Perfect retrieval across all sessions |
|
|
101
|
+
| knowledge_update | **1.000** | Supersession edges working correctly |
|
|
102
|
+
| temporal_reasoning | 0.9549 | 6 failures on long-distance relative refs (>2 weeks) |
|
|
103
|
+
| multi_session | 0.9474 | 7 failures; 4 retrieval misses, 2–3 arithmetic aggregation gaps |
|
|
104
|
+
|
|
105
|
+
| Strategy | Score |
|
|
106
|
+
|---|---|
|
|
107
|
+
| holiday_date, paren_abbreviation, list, sum_or_diff | **1.000** |
|
|
108
|
+
| nugget | 0.9691 |
|
|
109
|
+
| bool | 0.8571 |
|
|
110
|
+
|
|
111
|
+
**Baseline beaten:** v0.5.5 baseline was 0.948; this is a **+2.6 pp** improvement on the full 500-question corpus.
|
|
112
|
+
|
|
113
|
+
**Known remaining gaps (diagnosed, not guessed):**
|
|
114
|
+
- **Temporal anchoring** — degrades on multi-week relative references ("four weeks ago", "10 days ago"). These 6 failures connect to the `temporal_chain_notes` / supersession-history mechanism in `reader.py`. v0.6.4's `cortexm/experimental/coherence.py` adds a deterministic temporal-coherence rerank signal aimed at exactly these.
|
|
115
|
+
- **Arithmetic aggregation** — the generalized `sum_or_diff` judge (v0.6.2) fixes the 2–3 real computation gaps. The remaining multi_session failures are **retrieval misses** (wrong session pulled: poetry instead of podcasts, marketing facts instead of video views), not judge failures. v0.6.4 wires the previously-dead `percentage`/`numeric_agg` judges and adds `cortexm/experimental/graph_recall.py` (entity-adjacency 2-hop walks) aimed at the wrong-session misses.
|
|
116
|
+
- **BOOL strategy** at 85.7% is the weakest category — needs sign-of-evidence refinement for edge cases.
|
|
117
|
+
|
|
118
|
+
Run the full 500-Q benchmark via GitHub Actions: `.github/workflows/longmemeval.yml` (20 shards, ~30s/q with per-shard DB caching).
|
|
119
|
+
|
|
120
|
+
### Known boundaries (the short list)
|
|
121
|
+
|
|
122
|
+
> Full detail: [`docs/FAILURE_MODES.md`](docs/FAILURE_MODES.md) — every failure tied to a public benchmark question.
|
|
123
|
+
|
|
124
|
+
1. **The extractor is a 61-pattern lookup, not a language model.** Phrasings outside the pattern library are silently dropped at ingest (e.g. "Anna has a cat named Whiskers") — they remain retrievable via verbatim/BM25 chunk recall, but never become structured facts. This is the price of μ=0: no generativity, no fabrication, no drift.
|
|
125
|
+
2. **ZK proofs are trusted-prover attestations.** The v0.6.4 backend (Pedersen + Sigma protocols on secp256k1) is sound at the commitment layer — challenges are bound to announcements, both OR-proof branches verify, H has no known discrete log, thresholds are enforced — but the linkage between committed values and store rows is established at prove-time by the prover. Verify the integration layer before trusting it against a malicious host.
|
|
126
|
+
3. **Set membership reveals the leaf index.** The value stays hidden (random-blinding Pedersen + equality proof); the position in the set does not. Position-hiding needs a ZK-friendly Merkle construction — documented future work.
|
|
127
|
+
4. **No cross-user inference, ever.** Every fact is scoped by `user_id`; the scope sandbox turns empty scopes into empty results (not unrestricted fallbacks). This is a feature, and it also means no "insight across users" stories.
|
|
128
|
+
5. **Compression tiers are documented, not default.** int8/binary quantization trade recall for space (see `docs/COMPRESSION.md`); the default build keeps full-precision embeddings because the benchmark headroom doesn't justify the loss yet.
|
|
129
|
+
6. **Judge coverage is rule-based.** The deterministic judge answers via strategy dispatch (bool/list/nugget/sum_or_diff/percentage/numeric_agg/holiday/paren). Questions outside those strategies score 0 even when retrieval succeeded — the failure is honest, the number is real.
|
|
84
130
|
|
|
85
131
|
### When to use cortexm vs Mem0 / Zep / Chroma
|
|
86
132
|
|
|
@@ -124,7 +170,8 @@ The README is intentionally short. Everything else lives in `docs/`:
|
|
|
124
170
|
### Examples & tests
|
|
125
171
|
|
|
126
172
|
- [`examples/`](examples/) — runnable scripts, offline, no API keys (01_quickstart → 20_agent_session)
|
|
127
|
-
- [`tests/`](tests/) —
|
|
173
|
+
- [`tests/`](tests/) — 698 tests: fabric + enterprise + PPR + concurrency + sandbox + enrichment + WAL crash-recovery + migration + CRDT federation + Rust parity + ZK soundness/forgery + public-API smoke
|
|
174
|
+
- [`cortexm/experimental/`](cortexm/experimental/) — deterministic research borrows (graph recall, coherence) — μ=0 or it doesn't ship
|
|
128
175
|
- [`leaderboard/`](leaderboard/) — self-hosted benchmark site (rebuild: `python leaderboard/build.py`; open `leaderboard/index.html`)
|
|
129
176
|
- [`AGENTS.md`](AGENTS.md) — how AI coding agents should interact with this repo (2026 standard)
|
|
130
177
|
- [`CONTRIBUTING.md`](CONTRIBUTING.md) — contribution guide
|
|
@@ -1,38 +1,3 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: cortexm
|
|
3
|
-
Version: 0.6.0
|
|
4
|
-
Summary: Deterministic agent memory. 96 bytes per fact. Zero LLM at ingest.
|
|
5
|
-
Author: Context-M Contributors
|
|
6
|
-
License-Expression: Apache-2.0
|
|
7
|
-
Project-URL: Homepage, https://github.com/ssmurfgg04-gif/context-m
|
|
8
|
-
Project-URL: Documentation, https://github.com/ssmurfgg04-gif/context-m/tree/main/docs
|
|
9
|
-
Project-URL: Repository, https://github.com/ssmurfgg04-gif/context-m
|
|
10
|
-
Project-URL: Issues, https://github.com/ssmurfgg04-gif/context-m/issues
|
|
11
|
-
Project-URL: Changelog, https://github.com/ssmurfgg04-gif/context-m/releases
|
|
12
|
-
Keywords: agent-memory,llm-memory,long-term-memory,mem0,memgpt,letta,zep,chroma,deterministic-ai,local-first,vector-symbolic-architecture,provenance,bi-temporal,hippocampus,context-engineering,rag,mcp,neuro-symbolic,hrr,hdc,self-hosted
|
|
13
|
-
Classifier: Development Status :: 4 - Beta
|
|
14
|
-
Classifier: Intended Audience :: Developers
|
|
15
|
-
Classifier: Operating System :: OS Independent
|
|
16
|
-
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
-
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
-
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
-
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
-
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
21
|
-
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
22
|
-
Classifier: Topic :: Database
|
|
23
|
-
Classifier: Typing :: Typed
|
|
24
|
-
Requires-Python: >=3.10
|
|
25
|
-
Description-Content-Type: text/markdown
|
|
26
|
-
License-File: LICENSE
|
|
27
|
-
Requires-Dist: numpy>=1.24
|
|
28
|
-
Provides-Extra: blake3
|
|
29
|
-
Requires-Dist: blake3>=1.0; extra == "blake3"
|
|
30
|
-
Provides-Extra: crypto
|
|
31
|
-
Requires-Dist: cryptography>=42.0; extra == "crypto"
|
|
32
|
-
Provides-Extra: test
|
|
33
|
-
Requires-Dist: pytest>=7; extra == "test"
|
|
34
|
-
Dynamic: license-file
|
|
35
|
-
|
|
36
1
|
<div align="center">
|
|
37
2
|
<h1>cortexm</h1>
|
|
38
3
|
<h3>Deterministic agent memory. μ=0. Free, local, forever. Same result every time.</h3>
|
|
@@ -72,15 +37,53 @@ m.search("Where does Alice work?", user_id="alice")
|
|
|
72
37
|
|
|
73
38
|
### Canonical LongMemEval — μ=0, $0, on a 4GB laptop
|
|
74
39
|
|
|
75
|
-
| | cortexm v0.
|
|
40
|
+
| | cortexm v0.6.4 | MemPalace (honest E2E) |
|
|
76
41
|
|---|---|---|
|
|
77
|
-
| canonical LongMemEval (
|
|
42
|
+
| **canonical LongMemEval (500-Q full corpus)** | **97.4% (487/500)** | ~96.6% (retrieval-only, no QA) |
|
|
43
|
+
| single_session | **100.0%** | — |
|
|
44
|
+
| knowledge_update | **100.0%** | — |
|
|
45
|
+
| multi_session | 94.74% | — |
|
|
46
|
+
| temporal_reasoning | 95.49% | — |
|
|
78
47
|
| LLM calls (ingest + retrieval + judge) | 0 | 0 |
|
|
79
48
|
| monthly cost | $0 | $0 |
|
|
80
49
|
| determinism | byte-exact across 3× runs | byte-exact |
|
|
81
50
|
| owns your data | ✓ single `.db` file | ✓ |
|
|
82
51
|
|
|
83
|
-
**
|
|
52
|
+
**Full 500-question results** (v0.6.2 baseline; v0.6.4 re-run lands the experimental graph-recall + coherence modules below):
|
|
53
|
+
|
|
54
|
+
| Subtask | Score | Notes |
|
|
55
|
+
|---|---|---|
|
|
56
|
+
| **Overall** | **0.974 (487/500)** | Full corpus, not a proxy sample |
|
|
57
|
+
| single_session | **1.000** | Perfect retrieval across all sessions |
|
|
58
|
+
| knowledge_update | **1.000** | Supersession edges working correctly |
|
|
59
|
+
| temporal_reasoning | 0.9549 | 6 failures on long-distance relative refs (>2 weeks) |
|
|
60
|
+
| multi_session | 0.9474 | 7 failures; 4 retrieval misses, 2–3 arithmetic aggregation gaps |
|
|
61
|
+
|
|
62
|
+
| Strategy | Score |
|
|
63
|
+
|---|---|
|
|
64
|
+
| holiday_date, paren_abbreviation, list, sum_or_diff | **1.000** |
|
|
65
|
+
| nugget | 0.9691 |
|
|
66
|
+
| bool | 0.8571 |
|
|
67
|
+
|
|
68
|
+
**Baseline beaten:** v0.5.5 baseline was 0.948; this is a **+2.6 pp** improvement on the full 500-question corpus.
|
|
69
|
+
|
|
70
|
+
**Known remaining gaps (diagnosed, not guessed):**
|
|
71
|
+
- **Temporal anchoring** — degrades on multi-week relative references ("four weeks ago", "10 days ago"). These 6 failures connect to the `temporal_chain_notes` / supersession-history mechanism in `reader.py`. v0.6.4's `cortexm/experimental/coherence.py` adds a deterministic temporal-coherence rerank signal aimed at exactly these.
|
|
72
|
+
- **Arithmetic aggregation** — the generalized `sum_or_diff` judge (v0.6.2) fixes the 2–3 real computation gaps. The remaining multi_session failures are **retrieval misses** (wrong session pulled: poetry instead of podcasts, marketing facts instead of video views), not judge failures. v0.6.4 wires the previously-dead `percentage`/`numeric_agg` judges and adds `cortexm/experimental/graph_recall.py` (entity-adjacency 2-hop walks) aimed at the wrong-session misses.
|
|
73
|
+
- **BOOL strategy** at 85.7% is the weakest category — needs sign-of-evidence refinement for edge cases.
|
|
74
|
+
|
|
75
|
+
Run the full 500-Q benchmark via GitHub Actions: `.github/workflows/longmemeval.yml` (20 shards, ~30s/q with per-shard DB caching).
|
|
76
|
+
|
|
77
|
+
### Known boundaries (the short list)
|
|
78
|
+
|
|
79
|
+
> Full detail: [`docs/FAILURE_MODES.md`](docs/FAILURE_MODES.md) — every failure tied to a public benchmark question.
|
|
80
|
+
|
|
81
|
+
1. **The extractor is a 61-pattern lookup, not a language model.** Phrasings outside the pattern library are silently dropped at ingest (e.g. "Anna has a cat named Whiskers") — they remain retrievable via verbatim/BM25 chunk recall, but never become structured facts. This is the price of μ=0: no generativity, no fabrication, no drift.
|
|
82
|
+
2. **ZK proofs are trusted-prover attestations.** The v0.6.4 backend (Pedersen + Sigma protocols on secp256k1) is sound at the commitment layer — challenges are bound to announcements, both OR-proof branches verify, H has no known discrete log, thresholds are enforced — but the linkage between committed values and store rows is established at prove-time by the prover. Verify the integration layer before trusting it against a malicious host.
|
|
83
|
+
3. **Set membership reveals the leaf index.** The value stays hidden (random-blinding Pedersen + equality proof); the position in the set does not. Position-hiding needs a ZK-friendly Merkle construction — documented future work.
|
|
84
|
+
4. **No cross-user inference, ever.** Every fact is scoped by `user_id`; the scope sandbox turns empty scopes into empty results (not unrestricted fallbacks). This is a feature, and it also means no "insight across users" stories.
|
|
85
|
+
5. **Compression tiers are documented, not default.** int8/binary quantization trade recall for space (see `docs/COMPRESSION.md`); the default build keeps full-precision embeddings because the benchmark headroom doesn't justify the loss yet.
|
|
86
|
+
6. **Judge coverage is rule-based.** The deterministic judge answers via strategy dispatch (bool/list/nugget/sum_or_diff/percentage/numeric_agg/holiday/paren). Questions outside those strategies score 0 even when retrieval succeeded — the failure is honest, the number is real.
|
|
84
87
|
|
|
85
88
|
### When to use cortexm vs Mem0 / Zep / Chroma
|
|
86
89
|
|
|
@@ -124,7 +127,8 @@ The README is intentionally short. Everything else lives in `docs/`:
|
|
|
124
127
|
### Examples & tests
|
|
125
128
|
|
|
126
129
|
- [`examples/`](examples/) — runnable scripts, offline, no API keys (01_quickstart → 20_agent_session)
|
|
127
|
-
- [`tests/`](tests/) —
|
|
130
|
+
- [`tests/`](tests/) — 698 tests: fabric + enterprise + PPR + concurrency + sandbox + enrichment + WAL crash-recovery + migration + CRDT federation + Rust parity + ZK soundness/forgery + public-API smoke
|
|
131
|
+
- [`cortexm/experimental/`](cortexm/experimental/) — deterministic research borrows (graph recall, coherence) — μ=0 or it doesn't ship
|
|
128
132
|
- [`leaderboard/`](leaderboard/) — self-hosted benchmark site (rebuild: `python leaderboard/build.py`; open `leaderboard/index.html`)
|
|
129
133
|
- [`AGENTS.md`](AGENTS.md) — how AI coding agents should interact with this repo (2026 standard)
|
|
130
134
|
- [`CONTRIBUTING.md`](CONTRIBUTING.md) — contribution guide
|
|
@@ -18,7 +18,7 @@ Plugin kernel: ``from cortexm import Context, mount_default``
|
|
|
18
18
|
|
|
19
19
|
from __future__ import annotations
|
|
20
20
|
|
|
21
|
-
__version__ = "0.6.
|
|
21
|
+
__version__ = "0.6.4"
|
|
22
22
|
|
|
23
23
|
# μ=0 protocol counter: number of LLM invocations used by this process.
|
|
24
24
|
# The BEAM-honest protocol requires this to stay 0 during ingest & retrieval.
|
|
@@ -53,13 +53,15 @@ class Memory:
|
|
|
53
53
|
config = dataclasses.replace(config, **changes)
|
|
54
54
|
self.config = config
|
|
55
55
|
self.store = TraceStore(config.db_path, HashProvider(config.hash_provider),
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
56
|
+
wal_sync=getattr(config, "wal_sync", "normal"),
|
|
57
|
+
pragma_cache_mb=getattr(config, "pragma_cache_mb", 64),
|
|
58
|
+
pragma_mmap_mb=getattr(config, "pragma_mmap_mb", 256),
|
|
59
|
+
pragma_threads=getattr(config, "pragma_threads", 4),
|
|
60
|
+
pragma_temp_in_memory=getattr(config, "pragma_temp_in_memory", True),
|
|
61
|
+
pragma_locking_exclusive=getattr(config, "pragma_locking_exclusive", False))
|
|
62
|
+
# FIX 2: Allow passing a shared embedder for persistent workers
|
|
63
|
+
shared_embedder = getattr(config, "_shared_embedder", None)
|
|
64
|
+
self.palace = MemoryPalace(config, self.store, embedder=shared_embedder)
|
|
63
65
|
self.extractor = Extractor(config)
|
|
64
66
|
self.prefetcher = Prefetcher()
|
|
65
67
|
self.writer = MemoryWriter(config, self.store, self.palace, self.extractor)
|
|
@@ -157,6 +159,51 @@ class Memory:
|
|
|
157
159
|
self._maybe_run_fade_under_pressure(user_id)
|
|
158
160
|
return out
|
|
159
161
|
|
|
162
|
+
def add_batch(self, messages_list, *, user_id: str | None = None,
|
|
163
|
+
agent_id: str | None = None, run_id: str | None = None,
|
|
164
|
+
metadata: dict | None = None, timestamp=None, **kw) -> dict:
|
|
165
|
+
"""μ=0 batch ingest. Accepts a list of messages (each str | list[str] | dict).
|
|
166
|
+
|
|
167
|
+
Wraps the entire batch in a single transaction — much faster than
|
|
168
|
+
calling add() repeatedly. All messages share the same user_id/agent_id/run_id.
|
|
169
|
+
|
|
170
|
+
Returns the same dict format as add()."""
|
|
171
|
+
user_id = user_id or self.config.default_user_id
|
|
172
|
+
ts = parse_ts(timestamp) if timestamp else None
|
|
173
|
+
if self.pii_guard.mode != "off":
|
|
174
|
+
# Apply PII to each message list
|
|
175
|
+
processed = []
|
|
176
|
+
for msgs in messages_list:
|
|
177
|
+
processed.append(self._apply_pii(msgs))
|
|
178
|
+
if processed[-1] is None:
|
|
179
|
+
self.audit_log.log("memory.add_batch", resource=user_id,
|
|
180
|
+
outcome="blocked_pii",
|
|
181
|
+
meta={"reason": "pii_mode=block"})
|
|
182
|
+
return {"results": [], "blocked": "pii_policy"}
|
|
183
|
+
messages_list = processed
|
|
184
|
+
|
|
185
|
+
# Single transaction for the entire batch
|
|
186
|
+
self.store.begin_batch()
|
|
187
|
+
try:
|
|
188
|
+
all_results = []
|
|
189
|
+
total_facts = 0
|
|
190
|
+
for messages in messages_list:
|
|
191
|
+
out = self.writer.add(messages, user_id=user_id, agent_id=agent_id,
|
|
192
|
+
run_id=run_id, ts=ts, metadata=metadata, **kw)
|
|
193
|
+
all_results.extend(out.get("results", []))
|
|
194
|
+
total_facts += out.get("stats", {}).get("facts_inserted", 0)
|
|
195
|
+
self.store.end_batch()
|
|
196
|
+
self.reader.invalidate_caches()
|
|
197
|
+
if self.config.audit_actions == "all":
|
|
198
|
+
self.audit_log.log("memory.add_batch", resource=user_id,
|
|
199
|
+
meta={"facts": total_facts})
|
|
200
|
+
return {"event": "ADD_BATCH", "results": all_results,
|
|
201
|
+
"stats": {"messages": len(messages_list),
|
|
202
|
+
"facts_inserted": total_facts, "llm_calls": 0}}
|
|
203
|
+
except Exception:
|
|
204
|
+
self.store.rollback()
|
|
205
|
+
raise
|
|
206
|
+
|
|
160
207
|
# ------------------------------------------------------------ mem.edit()
|
|
161
208
|
def edit(self, fact_id: str, new_text: str, *,
|
|
162
209
|
edited_by: str = "user", reason: str | None = None) -> dict:
|
|
@@ -1027,9 +1074,51 @@ class Memory:
|
|
|
1027
1074
|
return {"loaded": False, "reason": "file_empty_or_corrupt"}
|
|
1028
1075
|
|
|
1029
1076
|
def close(self) -> None:
|
|
1077
|
+
# Close sidecar arena first on Windows (mmap holds file lock, prevents unlink)
|
|
1078
|
+
try:
|
|
1079
|
+
arena = getattr(self, "blob_arena", None)
|
|
1080
|
+
if arena is not None:
|
|
1081
|
+
arena.close()
|
|
1082
|
+
except Exception:
|
|
1083
|
+
pass
|
|
1030
1084
|
self.palace.close()
|
|
1031
1085
|
self.store.close()
|
|
1032
1086
|
|
|
1087
|
+
# ------------------------------------------------------------------
|
|
1088
|
+
# v0.6.1: BM25 + index maintenance facade
|
|
1089
|
+
# Exposed on Memory so users can call ``m.tune_bm25(k1=1.2, b=0.5)``
|
|
1090
|
+
# and ``m.optimize_index()`` without reaching into the verbatim
|
|
1091
|
+
# plugin. If the verbatim tier isn't mounted (e.g. disabled in
|
|
1092
|
+
# Config), these are no-ops so callers don't need to defensively
|
|
1093
|
+
# check.
|
|
1094
|
+
def tune_bm25(self, k1: float = 1.2, b: float = 0.75) -> None:
|
|
1095
|
+
"""Tune BM25 k1 (term saturation) + b (length normalization).
|
|
1096
|
+
|
|
1097
|
+
Lucene defaults: k1=1.2, b=0.75. Our corpus (short chat chunks,
|
|
1098
|
+
avg ~12 tokens) tends to prefer slightly higher saturation and
|
|
1099
|
+
weaker length norm; the v0.6.0 defaults were k1=1.5, b=0.75.
|
|
1100
|
+
Run ``scripts/tune_bm25_canonical.py`` to grid-search on the
|
|
1101
|
+
canonical LongMemEval sample and pick the best for your data.
|
|
1102
|
+
|
|
1103
|
+
Takes effect on the next ``search()`` call.
|
|
1104
|
+
"""
|
|
1105
|
+
if self._verbatim is not None:
|
|
1106
|
+
self._verbatim.tune_bm25(k1=k1, b=b)
|
|
1107
|
+
# Persist on the config too so reopens honor the tuning.
|
|
1108
|
+
self.config.bm25_k1 = k1
|
|
1109
|
+
self.config.bm25_b = b
|
|
1110
|
+
|
|
1111
|
+
def optimize_index(self) -> dict:
|
|
1112
|
+
"""VACUUM + FTS5 optimize + WAL checkpoint.
|
|
1113
|
+
|
|
1114
|
+
Run this after large bulk ingests to fold the WAL back into
|
|
1115
|
+
the main .db file, reclaim deleted-page space, and merge the
|
|
1116
|
+
FTS5 b-tree segments. Idempotent; safe to call any time.
|
|
1117
|
+
"""
|
|
1118
|
+
if self._verbatim is not None:
|
|
1119
|
+
return self._verbatim.optimize_index()
|
|
1120
|
+
return {"optimized": False, "reason": "verbatim tier not mounted"}
|
|
1121
|
+
|
|
1033
1122
|
def _reopen(self) -> None:
|
|
1034
1123
|
"""Rebind every component to a freshly-opened store (post-restore)."""
|
|
1035
1124
|
from cortexm.security.pii import PIIGuard, PIIVault
|
|
@@ -113,6 +113,29 @@ def _bitap_trigger_match(sent: str, max_edits: int = 2) -> bool:
|
|
|
113
113
|
class Extractor:
|
|
114
114
|
def __init__(self, config) -> None:
|
|
115
115
|
self.cfg = config
|
|
116
|
+
self._trigger_automaton = None
|
|
117
|
+
try:
|
|
118
|
+
import ahocorasick
|
|
119
|
+
automaton = ahocorasick.Automaton()
|
|
120
|
+
for word in _BITAP_TRIGGERS:
|
|
121
|
+
automaton.add_word(word, word)
|
|
122
|
+
automaton.make_automaton()
|
|
123
|
+
self._trigger_automaton = automaton
|
|
124
|
+
except ImportError:
|
|
125
|
+
# Package is a runtime dependency; retain the regex fallback for
|
|
126
|
+
# constrained embedded deployments with a partial installation.
|
|
127
|
+
pass
|
|
128
|
+
|
|
129
|
+
def _strict_trigger_match(self, sent: str) -> bool:
|
|
130
|
+
if self._trigger_automaton is None:
|
|
131
|
+
return bool(_TRIGGER.search(sent))
|
|
132
|
+
lowered = sent.lower()
|
|
133
|
+
for end, word in self._trigger_automaton.iter(lowered):
|
|
134
|
+
start = end - len(word) + 1
|
|
135
|
+
if (start == 0 or not lowered[start - 1].isalnum()) and \
|
|
136
|
+
(end + 1 == len(lowered) or not lowered[end + 1].isalnum()):
|
|
137
|
+
return True
|
|
138
|
+
return bool(_DATE_TRIGGER.search(sent)) or bool(_TRIGGER.search(sent))
|
|
116
139
|
|
|
117
140
|
# ------------------------------------------------------------------
|
|
118
141
|
def extract(self, text: str, ctx: ExtractionContext) -> list[Candidate]:
|
|
@@ -184,7 +207,7 @@ class Extractor:
|
|
|
184
207
|
# against the sentence with up to N edits. This stays deterministic
|
|
185
208
|
# (Wu-Manber is bitwise, no learned weights) and <50μs on a typical
|
|
186
209
|
# sentence — same order as the regex itself.
|
|
187
|
-
trigger_fired =
|
|
210
|
+
trigger_fired = self._strict_trigger_match(sent)
|
|
188
211
|
bitap_widened = False
|
|
189
212
|
if not trigger_fired:
|
|
190
213
|
if (not getattr(self.cfg, "bitap_trigger_enabled", True)
|
|
@@ -163,6 +163,11 @@ class TinyTransformerFallback:
|
|
|
163
163
|
self.seed = seed & 0xFFFFFFFFFFFFFFFF
|
|
164
164
|
self.max_tokens = max_tokens
|
|
165
165
|
self.tables = _HashedTables(self.seed)
|
|
166
|
+
# Relation labels are drawn from a small, stable vocabulary. The
|
|
167
|
+
# fallback evaluates every label for every pattern miss, so caching
|
|
168
|
+
# their deterministic embeddings avoids repeating the same attention
|
|
169
|
+
# computation dozens of times per sentence.
|
|
170
|
+
self._relation_embeddings: dict[str, np.ndarray] = {}
|
|
166
171
|
|
|
167
172
|
# ------------------------------------------------------------------
|
|
168
173
|
def _tokenize(self, text: str) -> list[str]:
|
|
@@ -213,6 +218,19 @@ class TinyTransformerFallback:
|
|
|
213
218
|
n = float(np.linalg.norm(pooled))
|
|
214
219
|
return pooled / n if n > 0 else pooled
|
|
215
220
|
|
|
221
|
+
def _relation_embedding(self, relation: str) -> np.ndarray:
|
|
222
|
+
"""Return the deterministic embedding for a relation label.
|
|
223
|
+
|
|
224
|
+
Relation labels are immutable strings and ``embed`` has no mutable
|
|
225
|
+
input-dependent state, so retaining this vector is semantically
|
|
226
|
+
equivalent to recomputing it on every fallback invocation.
|
|
227
|
+
"""
|
|
228
|
+
hit = self._relation_embeddings.get(relation)
|
|
229
|
+
if hit is None:
|
|
230
|
+
hit = self.embed(relation.replace("_", " "))
|
|
231
|
+
self._relation_embeddings[relation] = hit
|
|
232
|
+
return hit
|
|
233
|
+
|
|
216
234
|
# ------------------------------------------------------------------
|
|
217
235
|
def extract_candidates(self, sent: str, *,
|
|
218
236
|
subject_hint: str | None = None,
|
|
@@ -251,7 +269,7 @@ class TinyTransformerFallback:
|
|
|
251
269
|
rels = relations or _DEFAULT_RELATIONS
|
|
252
270
|
rel_scores = []
|
|
253
271
|
for r in rels:
|
|
254
|
-
r_emb = self.
|
|
272
|
+
r_emb = self._relation_embedding(r)
|
|
255
273
|
score = float(np.dot(sent_emb, r_emb))
|
|
256
274
|
rel_scores.append((r, score, r_emb))
|
|
257
275
|
rel_scores.sort(key=lambda x: -x[1])
|