cortexm 0.6.0__tar.gz → 0.6.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (194) hide show
  1. {cortexm-0.6.0/cortexm.egg-info → cortexm-0.6.4}/PKG-INFO +67 -20
  2. cortexm-0.6.0/PKG-INFO → cortexm-0.6.4/README.md +43 -39
  3. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/__init__.py +1 -1
  4. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/api/memory.py +96 -7
  5. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/extractor.py +24 -1
  6. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/fallback.py +19 -1
  7. cortexm-0.6.4/cortexm/bridge/fst_real.py +270 -0
  8. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/patterns.py +18 -1
  9. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/reader.py +113 -0
  10. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/synonyms.py +16 -0
  11. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/writer.py +148 -13
  12. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/config.py +69 -2
  13. cortexm-0.6.4/cortexm/experimental/__init__.py +29 -0
  14. cortexm-0.6.4/cortexm/experimental/coherence.py +123 -0
  15. cortexm-0.6.4/cortexm/experimental/graph_recall.py +193 -0
  16. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/federation/hlc.py +2 -2
  17. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/mcp/server.py +91 -58
  18. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/migrate/importers.py +22 -15
  19. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/plugins/verbatim.py +49 -33
  20. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/provenance/cose.py +14 -0
  21. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/provenance/vc.py +10 -0
  22. cortexm-0.6.4/cortexm/py.typed +1 -0
  23. cortexm-0.6.4/cortexm/security/hamming_attestation.py +291 -0
  24. cortexm-0.6.4/cortexm/security/zk_proofs.py +670 -0
  25. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/text/embedder.py +95 -4
  26. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/text/idiolect.py +38 -6
  27. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/text/labse.py +35 -1
  28. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/text/tokenizer.py +7 -0
  29. cortexm-0.6.4/cortexm/trace/contradictions.py +126 -0
  30. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/rules.py +18 -3
  31. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/store.py +138 -8
  32. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/palace.py +5 -4
  33. cortexm-0.6.4/cortexm.egg-info/PKG-INFO +181 -0
  34. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm.egg-info/SOURCES.txt +11 -3
  35. cortexm-0.6.4/cortexm.egg-info/requires.txt +24 -0
  36. cortexm-0.6.4/cortexm.egg-info/top_level.txt +1 -0
  37. cortexm-0.6.4/pyproject.toml +68 -0
  38. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_arxiv_improvements.py +25 -66
  39. cortexm-0.6.4/tests/test_experimental_v064.py +231 -0
  40. cortexm-0.6.4/tests/test_infrastructure_regressions.py +70 -0
  41. cortexm-0.6.4/tests/test_mcp_zk_tools.py +117 -0
  42. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_permission.py +3 -3
  43. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_reddit_steals_round3.py +2 -1
  44. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_research_steals_round2.py +19 -3
  45. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_v060_ir_pro.py +372 -6
  46. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_wal_recovery.py +2 -0
  47. cortexm-0.6.4/tests/test_zk_soundness.py +190 -0
  48. cortexm-0.6.4/tests/test_zk_sql.py +144 -0
  49. cortexm-0.6.0/README.md +0 -99
  50. cortexm-0.6.0/context_m.py +0 -17
  51. cortexm-0.6.0/cortexm/security/zk_hamming.py +0 -142
  52. cortexm-0.6.0/cortexm/security/zk_sql.py +0 -485
  53. cortexm-0.6.0/cortexm/trace/contradictions.py +0 -69
  54. cortexm-0.6.0/cortexm.egg-info/requires.txt +0 -10
  55. cortexm-0.6.0/cortexm.egg-info/top_level.txt +0 -2
  56. cortexm-0.6.0/pyproject.toml +0 -59
  57. cortexm-0.6.0/tests/test_zk_sql.py +0 -326
  58. {cortexm-0.6.0 → cortexm-0.6.4}/LICENSE +0 -0
  59. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/accel.py +0 -0
  60. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/api/__init__.py +0 -0
  61. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/api/chaos.py +0 -0
  62. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/api/long_recall.py +0 -0
  63. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/__init__.py +0 -0
  64. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/abilities.py +0 -0
  65. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/baselines.py +0 -0
  66. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/beam_loader.py +0 -0
  67. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/generator.py +0 -0
  68. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/harness.py +0 -0
  69. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/messy.py +0 -0
  70. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/micro.py +0 -0
  71. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/ood.py +0 -0
  72. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bench/run.py +0 -0
  73. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/__init__.py +0 -0
  74. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/dates.py +0 -0
  75. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/decoders.py +0 -0
  76. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/enrich.py +0 -0
  77. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/fst.py +0 -0
  78. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/fusion.py +0 -0
  79. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/ir_pro.py +0 -0
  80. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/multilingual.py +0 -0
  81. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/negation.py +0 -0
  82. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/onnx_runtime.py +0 -0
  83. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/ppr.py +0 -0
  84. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/prefilter.py +0 -0
  85. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/query_extract.py +0 -0
  86. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/query_rewrite.py +0 -0
  87. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/recognizers.py +0 -0
  88. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/rerank.py +0 -0
  89. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/bridge/slang.py +0 -0
  90. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cli.py +0 -0
  91. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cognition/__init__.py +0 -0
  92. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cognition/abstraction.py +0 -0
  93. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cognition/analogy.py +0 -0
  94. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cognition/engine.py +0 -0
  95. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cognition/gaps.py +0 -0
  96. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cognition/scanner.py +0 -0
  97. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/cortexm.py +0 -0
  98. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/creator.py +0 -0
  99. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/enterprise/__init__.py +0 -0
  100. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/enterprise/audit.py +0 -0
  101. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/enterprise/governance.py +0 -0
  102. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/errors.py +0 -0
  103. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/features/__init__.py +0 -0
  104. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/features/git.py +0 -0
  105. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/features/prefetch.py +0 -0
  106. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/features/zk.py +0 -0
  107. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/federation/__init__.py +0 -0
  108. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/federation/crdt.py +0 -0
  109. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/federation/fabric.py +0 -0
  110. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/federation/node.py +0 -0
  111. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/federation/schema_report.py +0 -0
  112. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/federation/transport.py +0 -0
  113. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/index/__init__.py +0 -0
  114. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/index/nsg.py +0 -0
  115. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/kernel.py +0 -0
  116. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/markdown_io.py +0 -0
  117. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/mcp/__init__.py +0 -0
  118. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/metrics.py +0 -0
  119. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/migrate/__init__.py +0 -0
  120. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/pipeline.py +0 -0
  121. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/plugins/__init__.py +0 -0
  122. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/plugins/security.py +0 -0
  123. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/plugins/structured.py +0 -0
  124. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/provenance/__init__.py +0 -0
  125. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/provenance/agent.py +0 -0
  126. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/provenance/scitt.py +0 -0
  127. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/router.py +0 -0
  128. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/__init__.py +0 -0
  129. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/crypto.py +0 -0
  130. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/hashes.py +0 -0
  131. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/injection.py +0 -0
  132. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/mind.py +0 -0
  133. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/permission.py +0 -0
  134. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/pii.py +0 -0
  135. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/rbac.py +0 -0
  136. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/security/sandbox.py +0 -0
  137. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/server/__init__.py +0 -0
  138. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/server/metrics.py +0 -0
  139. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/server/rest.py +0 -0
  140. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/server/sparql.py +0 -0
  141. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/text/__init__.py +0 -0
  142. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/text/dissim.py +0 -0
  143. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/text/fuzzy.py +0 -0
  144. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/__init__.py +0 -0
  145. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/blob_arena.py +0 -0
  146. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/consolidate.py +0 -0
  147. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/dedup.py +0 -0
  148. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/edges.py +0 -0
  149. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/fact.py +0 -0
  150. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/fade.py +0 -0
  151. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/lifecycle.py +0 -0
  152. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/rebuild.py +0 -0
  153. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/structural.py +0 -0
  154. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trace/tmt.py +0 -0
  155. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/trajectory_view.py +0 -0
  156. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/util.py +0 -0
  157. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/__init__.py +0 -0
  158. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/attribution.py +0 -0
  159. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/cleanup.py +0 -0
  160. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/codecs.py +0 -0
  161. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/hologram_overlay.py +0 -0
  162. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/index.py +0 -0
  163. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/ops.py +0 -0
  164. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/role_vectors.py +0 -0
  165. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/slb.py +0 -0
  166. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/tlsh_trie.py +0 -0
  167. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm/vsa/working_memory.py +0 -0
  168. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm.egg-info/dependency_links.txt +0 -0
  169. {cortexm-0.6.0 → cortexm-0.6.4}/cortexm.egg-info/entry_points.txt +0 -0
  170. {cortexm-0.6.0 → cortexm-0.6.4}/setup.cfg +0 -0
  171. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_bench_infra.py +0 -0
  172. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_bm25_chunk_recall_and_inspect_cli.py +0 -0
  173. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_cognition_and_provenance.py +0 -0
  174. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_engineering_push_2026_08_28.py +0 -0
  175. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_enterprise.py +0 -0
  176. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_fabric.py +0 -0
  177. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_federation.py +0 -0
  178. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_fusion_security.py +0 -0
  179. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_kernel.py +0 -0
  180. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_kinship_extraction.py +0 -0
  181. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_labse.py +0 -0
  182. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_list_superseded_intent.py +0 -0
  183. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_migration.py +0 -0
  184. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_new_modules.py +0 -0
  185. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_nsg.py +0 -0
  186. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_ppr.py +0 -0
  187. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_public_api_smoke.py +0 -0
  188. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_rerank.py +0 -0
  189. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_research_steals.py +0 -0
  190. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_rust_accel.py +0 -0
  191. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_sandbox_enrich.py +0 -0
  192. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_sparql_rest_v2.py +0 -0
  193. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_tier443_abstention_fix.py +0 -0
  194. {cortexm-0.6.0 → cortexm-0.6.4}/tests/test_verbatim.py +0 -0
@@ -1,36 +1,44 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cortexm
3
- Version: 0.6.0
4
- Summary: Deterministic agent memory. 96 bytes per fact. Zero LLM at ingest.
5
- Author: Context-M Contributors
6
- License-Expression: Apache-2.0
3
+ Version: 0.6.4
4
+ Summary: Context-M: deterministic, auditable, zero-cost memory for AI agents
5
+ Author-email: Context-M Team <dev@context-m.ai>
6
+ License: MIT
7
7
  Project-URL: Homepage, https://github.com/ssmurfgg04-gif/context-m
8
- Project-URL: Documentation, https://github.com/ssmurfgg04-gif/context-m/tree/main/docs
8
+ Project-URL: Documentation, https://github.com/ssmurfgg04-gif/context-m#readme
9
9
  Project-URL: Repository, https://github.com/ssmurfgg04-gif/context-m
10
10
  Project-URL: Issues, https://github.com/ssmurfgg04-gif/context-m/issues
11
- Project-URL: Changelog, https://github.com/ssmurfgg04-gif/context-m/releases
12
- Keywords: agent-memory,llm-memory,long-term-memory,mem0,memgpt,letta,zep,chroma,deterministic-ai,local-first,vector-symbolic-architecture,provenance,bi-temporal,hippocampus,context-engineering,rag,mcp,neuro-symbolic,hrr,hdc,self-hosted
11
+ Keywords: memory,ai,agents,deterministic,audit,zk
13
12
  Classifier: Development Status :: 4 - Beta
14
13
  Classifier: Intended Audience :: Developers
15
- Classifier: Operating System :: OS Independent
16
- Classifier: Programming Language :: Python :: 3 :: Only
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Programming Language :: Python :: 3
17
16
  Classifier: Programming Language :: Python :: 3.10
18
17
  Classifier: Programming Language :: Python :: 3.11
19
18
  Classifier: Programming Language :: Python :: 3.12
20
19
  Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
21
- Classifier: Topic :: Software Development :: Libraries :: Python Modules
22
- Classifier: Topic :: Database
23
- Classifier: Typing :: Typed
24
20
  Requires-Python: >=3.10
25
21
  Description-Content-Type: text/markdown
26
22
  License-File: LICENSE
27
- Requires-Dist: numpy>=1.24
23
+ Requires-Dist: numpy>=1.24.0
24
+ Requires-Dist: cryptography>=41.0.0
25
+ Requires-Dist: fastecdsa>=2.3.0
28
26
  Provides-Extra: blake3
29
- Requires-Dist: blake3>=1.0; extra == "blake3"
27
+ Requires-Dist: blake3>=0.3.0; extra == "blake3"
30
28
  Provides-Extra: crypto
31
- Requires-Dist: cryptography>=42.0; extra == "crypto"
29
+ Requires-Dist: cryptography>=41.0.0; extra == "crypto"
30
+ Provides-Extra: fst
31
+ Requires-Dist: marisa-trie>=1.2.0; extra == "fst"
32
32
  Provides-Extra: test
33
- Requires-Dist: pytest>=7; extra == "test"
33
+ Requires-Dist: pytest>=7.0; extra == "test"
34
+ Requires-Dist: pytest-cov>=4.0; extra == "test"
35
+ Provides-Extra: all
36
+ Requires-Dist: blake3>=0.3.0; extra == "all"
37
+ Requires-Dist: cryptography>=41.0.0; extra == "all"
38
+ Requires-Dist: fastecdsa>=2.3.0; extra == "all"
39
+ Requires-Dist: marisa-trie>=1.2.0; extra == "all"
40
+ Requires-Dist: pytest>=7.0; extra == "all"
41
+ Requires-Dist: pytest-cov>=4.0; extra == "all"
34
42
  Dynamic: license-file
35
43
 
36
44
  <div align="center">
@@ -72,15 +80,53 @@ m.search("Where does Alice work?", user_id="alice")
72
80
 
73
81
  ### Canonical LongMemEval — μ=0, $0, on a 4GB laptop
74
82
 
75
- | | cortexm v0.5.6 | MemPalace (honest E2E) |
83
+ | | cortexm v0.6.4 | MemPalace (honest E2E) |
76
84
  |---|---|---|
77
- | canonical LongMemEval (154-Q sample) | **94.8%** → 154/154 after v0.5.5 judges | ~96.6% (retrieval-only, no QA) |
85
+ | **canonical LongMemEval (500-Q full corpus)** | **97.4% (487/500)** | ~96.6% (retrieval-only, no QA) |
86
+ | single_session | **100.0%** | — |
87
+ | knowledge_update | **100.0%** | — |
88
+ | multi_session | 94.74% | — |
89
+ | temporal_reasoning | 95.49% | — |
78
90
  | LLM calls (ingest + retrieval + judge) | 0 | 0 |
79
91
  | monthly cost | $0 | $0 |
80
92
  | determinism | byte-exact across 3× runs | byte-exact |
81
93
  | owns your data | ✓ single `.db` file | ✓ |
82
94
 
83
- **Honest scope.** 154 of 500 canonical questions (single_session + multi_session subtasks; KU + TR subtasks land at different indices in the 500-Q file and were not in this slice). All 154/154 answered correctly after v0.5.5's aggregation + holiday + abbreviation judges. Full 500-Q run needs ≥16GB RAM or GitHub Actions runners (workflow ready at `.github/workflows/longmemeval_canonical_full.yml`). We do **not** claim parity on the full canonical 500.
95
+ **Full 500-question results** (v0.6.2 baseline; v0.6.4 re-run lands the experimental graph-recall + coherence modules below):
96
+
97
+ | Subtask | Score | Notes |
98
+ |---|---|---|
99
+ | **Overall** | **0.974 (487/500)** | Full corpus, not a proxy sample |
100
+ | single_session | **1.000** | Perfect retrieval across all sessions |
101
+ | knowledge_update | **1.000** | Supersession edges working correctly |
102
+ | temporal_reasoning | 0.9549 | 6 failures on long-distance relative refs (>2 weeks) |
103
+ | multi_session | 0.9474 | 7 failures; 4 retrieval misses, 2–3 arithmetic aggregation gaps |
104
+
105
+ | Strategy | Score |
106
+ |---|---|
107
+ | holiday_date, paren_abbreviation, list, sum_or_diff | **1.000** |
108
+ | nugget | 0.9691 |
109
+ | bool | 0.8571 |
110
+
111
+ **Baseline beaten:** v0.5.5 baseline was 0.948; this is a **+2.6 pp** improvement on the full 500-question corpus.
112
+
113
+ **Known remaining gaps (diagnosed, not guessed):**
114
+ - **Temporal anchoring** — degrades on multi-week relative references ("four weeks ago", "10 days ago"). These 6 failures connect to the `temporal_chain_notes` / supersession-history mechanism in `reader.py`. v0.6.4's `cortexm/experimental/coherence.py` adds a deterministic temporal-coherence rerank signal aimed at exactly these.
115
+ - **Arithmetic aggregation** — the generalized `sum_or_diff` judge (v0.6.2) fixes the 2–3 real computation gaps. The remaining multi_session failures are **retrieval misses** (wrong session pulled: poetry instead of podcasts, marketing facts instead of video views), not judge failures. v0.6.4 wires the previously-dead `percentage`/`numeric_agg` judges and adds `cortexm/experimental/graph_recall.py` (entity-adjacency 2-hop walks) aimed at the wrong-session misses.
116
+ - **BOOL strategy** at 85.7% is the weakest category — needs sign-of-evidence refinement for edge cases.
117
+
118
+ Run the full 500-Q benchmark via GitHub Actions: `.github/workflows/longmemeval.yml` (20 shards, ~30s/q with per-shard DB caching).
119
+
120
+ ### Known boundaries (the short list)
121
+
122
+ > Full detail: [`docs/FAILURE_MODES.md`](docs/FAILURE_MODES.md) — every failure tied to a public benchmark question.
123
+
124
+ 1. **The extractor is a 61-pattern lookup, not a language model.** Phrasings outside the pattern library are silently dropped at ingest (e.g. "Anna has a cat named Whiskers") — they remain retrievable via verbatim/BM25 chunk recall, but never become structured facts. This is the price of μ=0: no generativity, no fabrication, no drift.
125
+ 2. **ZK proofs are trusted-prover attestations.** The v0.6.4 backend (Pedersen + Sigma protocols on secp256k1) is sound at the commitment layer — challenges are bound to announcements, both OR-proof branches verify, H has no known discrete log, thresholds are enforced — but the linkage between committed values and store rows is established at prove-time by the prover. Verify the integration layer before trusting it against a malicious host.
126
+ 3. **Set membership reveals the leaf index.** The value stays hidden (random-blinding Pedersen + equality proof); the position in the set does not. Position-hiding needs a ZK-friendly Merkle construction — documented future work.
127
+ 4. **No cross-user inference, ever.** Every fact is scoped by `user_id`; the scope sandbox turns empty scopes into empty results (not unrestricted fallbacks). This is a feature, and it also means no "insight across users" stories.
128
+ 5. **Compression tiers are documented, not default.** int8/binary quantization trade recall for space (see `docs/COMPRESSION.md`); the default build keeps full-precision embeddings because the benchmark headroom doesn't justify the loss yet.
129
+ 6. **Judge coverage is rule-based.** The deterministic judge answers via strategy dispatch (bool/list/nugget/sum_or_diff/percentage/numeric_agg/holiday/paren). Questions outside those strategies score 0 even when retrieval succeeded — the failure is honest, the number is real.
84
130
 
85
131
  ### When to use cortexm vs Mem0 / Zep / Chroma
86
132
 
@@ -124,7 +170,8 @@ The README is intentionally short. Everything else lives in `docs/`:
124
170
  ### Examples & tests
125
171
 
126
172
  - [`examples/`](examples/) — runnable scripts, offline, no API keys (01_quickstart → 20_agent_session)
127
- - [`tests/`](tests/) — 117 tests: fabric + enterprise + PPR + concurrency + sandbox + enrichment + WAL crash-recovery + migration + CRDT federation + Rust parity + public-API smoke
173
+ - [`tests/`](tests/) — 698 tests: fabric + enterprise + PPR + concurrency + sandbox + enrichment + WAL crash-recovery + migration + CRDT federation + Rust parity + ZK soundness/forgery + public-API smoke
174
+ - [`cortexm/experimental/`](cortexm/experimental/) — deterministic research borrows (graph recall, coherence) — μ=0 or it doesn't ship
128
175
  - [`leaderboard/`](leaderboard/) — self-hosted benchmark site (rebuild: `python leaderboard/build.py`; open `leaderboard/index.html`)
129
176
  - [`AGENTS.md`](AGENTS.md) — how AI coding agents should interact with this repo (2026 standard)
130
177
  - [`CONTRIBUTING.md`](CONTRIBUTING.md) — contribution guide
@@ -1,38 +1,3 @@
1
- Metadata-Version: 2.4
2
- Name: cortexm
3
- Version: 0.6.0
4
- Summary: Deterministic agent memory. 96 bytes per fact. Zero LLM at ingest.
5
- Author: Context-M Contributors
6
- License-Expression: Apache-2.0
7
- Project-URL: Homepage, https://github.com/ssmurfgg04-gif/context-m
8
- Project-URL: Documentation, https://github.com/ssmurfgg04-gif/context-m/tree/main/docs
9
- Project-URL: Repository, https://github.com/ssmurfgg04-gif/context-m
10
- Project-URL: Issues, https://github.com/ssmurfgg04-gif/context-m/issues
11
- Project-URL: Changelog, https://github.com/ssmurfgg04-gif/context-m/releases
12
- Keywords: agent-memory,llm-memory,long-term-memory,mem0,memgpt,letta,zep,chroma,deterministic-ai,local-first,vector-symbolic-architecture,provenance,bi-temporal,hippocampus,context-engineering,rag,mcp,neuro-symbolic,hrr,hdc,self-hosted
13
- Classifier: Development Status :: 4 - Beta
14
- Classifier: Intended Audience :: Developers
15
- Classifier: Operating System :: OS Independent
16
- Classifier: Programming Language :: Python :: 3 :: Only
17
- Classifier: Programming Language :: Python :: 3.10
18
- Classifier: Programming Language :: Python :: 3.11
19
- Classifier: Programming Language :: Python :: 3.12
20
- Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
21
- Classifier: Topic :: Software Development :: Libraries :: Python Modules
22
- Classifier: Topic :: Database
23
- Classifier: Typing :: Typed
24
- Requires-Python: >=3.10
25
- Description-Content-Type: text/markdown
26
- License-File: LICENSE
27
- Requires-Dist: numpy>=1.24
28
- Provides-Extra: blake3
29
- Requires-Dist: blake3>=1.0; extra == "blake3"
30
- Provides-Extra: crypto
31
- Requires-Dist: cryptography>=42.0; extra == "crypto"
32
- Provides-Extra: test
33
- Requires-Dist: pytest>=7; extra == "test"
34
- Dynamic: license-file
35
-
36
1
  <div align="center">
37
2
  <h1>cortexm</h1>
38
3
  <h3>Deterministic agent memory. μ=0. Free, local, forever. Same result every time.</h3>
@@ -72,15 +37,53 @@ m.search("Where does Alice work?", user_id="alice")
72
37
 
73
38
  ### Canonical LongMemEval — μ=0, $0, on a 4GB laptop
74
39
 
75
- | | cortexm v0.5.6 | MemPalace (honest E2E) |
40
+ | | cortexm v0.6.4 | MemPalace (honest E2E) |
76
41
  |---|---|---|
77
- | canonical LongMemEval (154-Q sample) | **94.8%** → 154/154 after v0.5.5 judges | ~96.6% (retrieval-only, no QA) |
42
+ | **canonical LongMemEval (500-Q full corpus)** | **97.4% (487/500)** | ~96.6% (retrieval-only, no QA) |
43
+ | single_session | **100.0%** | — |
44
+ | knowledge_update | **100.0%** | — |
45
+ | multi_session | 94.74% | — |
46
+ | temporal_reasoning | 95.49% | — |
78
47
  | LLM calls (ingest + retrieval + judge) | 0 | 0 |
79
48
  | monthly cost | $0 | $0 |
80
49
  | determinism | byte-exact across 3× runs | byte-exact |
81
50
  | owns your data | ✓ single `.db` file | ✓ |
82
51
 
83
- **Honest scope.** 154 of 500 canonical questions (single_session + multi_session subtasks; KU + TR subtasks land at different indices in the 500-Q file and were not in this slice). All 154/154 answered correctly after v0.5.5's aggregation + holiday + abbreviation judges. Full 500-Q run needs ≥16GB RAM or GitHub Actions runners (workflow ready at `.github/workflows/longmemeval_canonical_full.yml`). We do **not** claim parity on the full canonical 500.
52
+ **Full 500-question results** (v0.6.2 baseline; v0.6.4 re-run lands the experimental graph-recall + coherence modules below):
53
+
54
+ | Subtask | Score | Notes |
55
+ |---|---|---|
56
+ | **Overall** | **0.974 (487/500)** | Full corpus, not a proxy sample |
57
+ | single_session | **1.000** | Perfect retrieval across all sessions |
58
+ | knowledge_update | **1.000** | Supersession edges working correctly |
59
+ | temporal_reasoning | 0.9549 | 6 failures on long-distance relative refs (>2 weeks) |
60
+ | multi_session | 0.9474 | 7 failures; 4 retrieval misses, 2–3 arithmetic aggregation gaps |
61
+
62
+ | Strategy | Score |
63
+ |---|---|
64
+ | holiday_date, paren_abbreviation, list, sum_or_diff | **1.000** |
65
+ | nugget | 0.9691 |
66
+ | bool | 0.8571 |
67
+
68
+ **Baseline beaten:** v0.5.5 baseline was 0.948; this is a **+2.6 pp** improvement on the full 500-question corpus.
69
+
70
+ **Known remaining gaps (diagnosed, not guessed):**
71
+ - **Temporal anchoring** — degrades on multi-week relative references ("four weeks ago", "10 days ago"). These 6 failures connect to the `temporal_chain_notes` / supersession-history mechanism in `reader.py`. v0.6.4's `cortexm/experimental/coherence.py` adds a deterministic temporal-coherence rerank signal aimed at exactly these.
72
+ - **Arithmetic aggregation** — the generalized `sum_or_diff` judge (v0.6.2) fixes the 2–3 real computation gaps. The remaining multi_session failures are **retrieval misses** (wrong session pulled: poetry instead of podcasts, marketing facts instead of video views), not judge failures. v0.6.4 wires the previously-dead `percentage`/`numeric_agg` judges and adds `cortexm/experimental/graph_recall.py` (entity-adjacency 2-hop walks) aimed at the wrong-session misses.
73
+ - **BOOL strategy** at 85.7% is the weakest category — needs sign-of-evidence refinement for edge cases.
74
+
75
+ Run the full 500-Q benchmark via GitHub Actions: `.github/workflows/longmemeval.yml` (20 shards, ~30s/q with per-shard DB caching).
76
+
77
+ ### Known boundaries (the short list)
78
+
79
+ > Full detail: [`docs/FAILURE_MODES.md`](docs/FAILURE_MODES.md) — every failure tied to a public benchmark question.
80
+
81
+ 1. **The extractor is a 61-pattern lookup, not a language model.** Phrasings outside the pattern library are silently dropped at ingest (e.g. "Anna has a cat named Whiskers") — they remain retrievable via verbatim/BM25 chunk recall, but never become structured facts. This is the price of μ=0: no generativity, no fabrication, no drift.
82
+ 2. **ZK proofs are trusted-prover attestations.** The v0.6.4 backend (Pedersen + Sigma protocols on secp256k1) is sound at the commitment layer — challenges are bound to announcements, both OR-proof branches verify, H has no known discrete log, thresholds are enforced — but the linkage between committed values and store rows is established at prove-time by the prover. Verify the integration layer before trusting it against a malicious host.
83
+ 3. **Set membership reveals the leaf index.** The value stays hidden (random-blinding Pedersen + equality proof); the position in the set does not. Position-hiding needs a ZK-friendly Merkle construction — documented future work.
84
+ 4. **No cross-user inference, ever.** Every fact is scoped by `user_id`; the scope sandbox turns empty scopes into empty results (not unrestricted fallbacks). This is a feature, and it also means no "insight across users" stories.
85
+ 5. **Compression tiers are documented, not default.** int8/binary quantization trade recall for space (see `docs/COMPRESSION.md`); the default build keeps full-precision embeddings because the benchmark headroom doesn't justify the loss yet.
86
+ 6. **Judge coverage is rule-based.** The deterministic judge answers via strategy dispatch (bool/list/nugget/sum_or_diff/percentage/numeric_agg/holiday/paren). Questions outside those strategies score 0 even when retrieval succeeded — the failure is honest, the number is real.
84
87
 
85
88
  ### When to use cortexm vs Mem0 / Zep / Chroma
86
89
 
@@ -124,7 +127,8 @@ The README is intentionally short. Everything else lives in `docs/`:
124
127
  ### Examples & tests
125
128
 
126
129
  - [`examples/`](examples/) — runnable scripts, offline, no API keys (01_quickstart → 20_agent_session)
127
- - [`tests/`](tests/) — 117 tests: fabric + enterprise + PPR + concurrency + sandbox + enrichment + WAL crash-recovery + migration + CRDT federation + Rust parity + public-API smoke
130
+ - [`tests/`](tests/) — 698 tests: fabric + enterprise + PPR + concurrency + sandbox + enrichment + WAL crash-recovery + migration + CRDT federation + Rust parity + ZK soundness/forgery + public-API smoke
131
+ - [`cortexm/experimental/`](cortexm/experimental/) — deterministic research borrows (graph recall, coherence) — μ=0 or it doesn't ship
128
132
  - [`leaderboard/`](leaderboard/) — self-hosted benchmark site (rebuild: `python leaderboard/build.py`; open `leaderboard/index.html`)
129
133
  - [`AGENTS.md`](AGENTS.md) — how AI coding agents should interact with this repo (2026 standard)
130
134
  - [`CONTRIBUTING.md`](CONTRIBUTING.md) — contribution guide
@@ -18,7 +18,7 @@ Plugin kernel: ``from cortexm import Context, mount_default``
18
18
 
19
19
  from __future__ import annotations
20
20
 
21
- __version__ = "0.6.0"
21
+ __version__ = "0.6.4"
22
22
 
23
23
  # μ=0 protocol counter: number of LLM invocations used by this process.
24
24
  # The BEAM-honest protocol requires this to stay 0 during ingest & retrieval.
@@ -53,13 +53,15 @@ class Memory:
53
53
  config = dataclasses.replace(config, **changes)
54
54
  self.config = config
55
55
  self.store = TraceStore(config.db_path, HashProvider(config.hash_provider),
56
- wal_sync=getattr(config, "wal_sync", "normal"),
57
- pragma_cache_mb=getattr(config, "pragma_cache_mb", 64),
58
- pragma_mmap_mb=getattr(config, "pragma_mmap_mb", 256),
59
- pragma_threads=getattr(config, "pragma_threads", 4),
60
- pragma_temp_in_memory=getattr(config, "pragma_temp_in_memory", True),
61
- pragma_locking_exclusive=getattr(config, "pragma_locking_exclusive", False))
62
- self.palace = MemoryPalace(config, self.store)
56
+ wal_sync=getattr(config, "wal_sync", "normal"),
57
+ pragma_cache_mb=getattr(config, "pragma_cache_mb", 64),
58
+ pragma_mmap_mb=getattr(config, "pragma_mmap_mb", 256),
59
+ pragma_threads=getattr(config, "pragma_threads", 4),
60
+ pragma_temp_in_memory=getattr(config, "pragma_temp_in_memory", True),
61
+ pragma_locking_exclusive=getattr(config, "pragma_locking_exclusive", False))
62
+ # FIX 2: Allow passing a shared embedder for persistent workers
63
+ shared_embedder = getattr(config, "_shared_embedder", None)
64
+ self.palace = MemoryPalace(config, self.store, embedder=shared_embedder)
63
65
  self.extractor = Extractor(config)
64
66
  self.prefetcher = Prefetcher()
65
67
  self.writer = MemoryWriter(config, self.store, self.palace, self.extractor)
@@ -157,6 +159,51 @@ class Memory:
157
159
  self._maybe_run_fade_under_pressure(user_id)
158
160
  return out
159
161
 
162
+ def add_batch(self, messages_list, *, user_id: str | None = None,
163
+ agent_id: str | None = None, run_id: str | None = None,
164
+ metadata: dict | None = None, timestamp=None, **kw) -> dict:
165
+ """μ=0 batch ingest. Accepts a list of messages (each str | list[str] | dict).
166
+
167
+ Wraps the entire batch in a single transaction — much faster than
168
+ calling add() repeatedly. All messages share the same user_id/agent_id/run_id.
169
+
170
+ Returns the same dict format as add()."""
171
+ user_id = user_id or self.config.default_user_id
172
+ ts = parse_ts(timestamp) if timestamp else None
173
+ if self.pii_guard.mode != "off":
174
+ # Apply PII to each message list
175
+ processed = []
176
+ for msgs in messages_list:
177
+ processed.append(self._apply_pii(msgs))
178
+ if processed[-1] is None:
179
+ self.audit_log.log("memory.add_batch", resource=user_id,
180
+ outcome="blocked_pii",
181
+ meta={"reason": "pii_mode=block"})
182
+ return {"results": [], "blocked": "pii_policy"}
183
+ messages_list = processed
184
+
185
+ # Single transaction for the entire batch
186
+ self.store.begin_batch()
187
+ try:
188
+ all_results = []
189
+ total_facts = 0
190
+ for messages in messages_list:
191
+ out = self.writer.add(messages, user_id=user_id, agent_id=agent_id,
192
+ run_id=run_id, ts=ts, metadata=metadata, **kw)
193
+ all_results.extend(out.get("results", []))
194
+ total_facts += out.get("stats", {}).get("facts_inserted", 0)
195
+ self.store.end_batch()
196
+ self.reader.invalidate_caches()
197
+ if self.config.audit_actions == "all":
198
+ self.audit_log.log("memory.add_batch", resource=user_id,
199
+ meta={"facts": total_facts})
200
+ return {"event": "ADD_BATCH", "results": all_results,
201
+ "stats": {"messages": len(messages_list),
202
+ "facts_inserted": total_facts, "llm_calls": 0}}
203
+ except Exception:
204
+ self.store.rollback()
205
+ raise
206
+
160
207
  # ------------------------------------------------------------ mem.edit()
161
208
  def edit(self, fact_id: str, new_text: str, *,
162
209
  edited_by: str = "user", reason: str | None = None) -> dict:
@@ -1027,9 +1074,51 @@ class Memory:
1027
1074
  return {"loaded": False, "reason": "file_empty_or_corrupt"}
1028
1075
 
1029
1076
  def close(self) -> None:
1077
+ # Close sidecar arena first on Windows (mmap holds file lock, prevents unlink)
1078
+ try:
1079
+ arena = getattr(self, "blob_arena", None)
1080
+ if arena is not None:
1081
+ arena.close()
1082
+ except Exception:
1083
+ pass
1030
1084
  self.palace.close()
1031
1085
  self.store.close()
1032
1086
 
1087
+ # ------------------------------------------------------------------
1088
+ # v0.6.1: BM25 + index maintenance facade
1089
+ # Exposed on Memory so users can call ``m.tune_bm25(k1=1.2, b=0.5)``
1090
+ # and ``m.optimize_index()`` without reaching into the verbatim
1091
+ # plugin. If the verbatim tier isn't mounted (e.g. disabled in
1092
+ # Config), these are no-ops so callers don't need to defensively
1093
+ # check.
1094
+ def tune_bm25(self, k1: float = 1.2, b: float = 0.75) -> None:
1095
+ """Tune BM25 k1 (term saturation) + b (length normalization).
1096
+
1097
+ Lucene defaults: k1=1.2, b=0.75. Our corpus (short chat chunks,
1098
+ avg ~12 tokens) tends to prefer slightly higher saturation and
1099
+ weaker length norm; the v0.6.0 defaults were k1=1.5, b=0.75.
1100
+ Run ``scripts/tune_bm25_canonical.py`` to grid-search on the
1101
+ canonical LongMemEval sample and pick the best for your data.
1102
+
1103
+ Takes effect on the next ``search()`` call.
1104
+ """
1105
+ if self._verbatim is not None:
1106
+ self._verbatim.tune_bm25(k1=k1, b=b)
1107
+ # Persist on the config too so reopens honor the tuning.
1108
+ self.config.bm25_k1 = k1
1109
+ self.config.bm25_b = b
1110
+
1111
+ def optimize_index(self) -> dict:
1112
+ """VACUUM + FTS5 optimize + WAL checkpoint.
1113
+
1114
+ Run this after large bulk ingests to fold the WAL back into
1115
+ the main .db file, reclaim deleted-page space, and merge the
1116
+ FTS5 b-tree segments. Idempotent; safe to call any time.
1117
+ """
1118
+ if self._verbatim is not None:
1119
+ return self._verbatim.optimize_index()
1120
+ return {"optimized": False, "reason": "verbatim tier not mounted"}
1121
+
1033
1122
  def _reopen(self) -> None:
1034
1123
  """Rebind every component to a freshly-opened store (post-restore)."""
1035
1124
  from cortexm.security.pii import PIIGuard, PIIVault
@@ -113,6 +113,29 @@ def _bitap_trigger_match(sent: str, max_edits: int = 2) -> bool:
113
113
  class Extractor:
114
114
  def __init__(self, config) -> None:
115
115
  self.cfg = config
116
+ self._trigger_automaton = None
117
+ try:
118
+ import ahocorasick
119
+ automaton = ahocorasick.Automaton()
120
+ for word in _BITAP_TRIGGERS:
121
+ automaton.add_word(word, word)
122
+ automaton.make_automaton()
123
+ self._trigger_automaton = automaton
124
+ except ImportError:
125
+ # Package is a runtime dependency; retain the regex fallback for
126
+ # constrained embedded deployments with a partial installation.
127
+ pass
128
+
129
+ def _strict_trigger_match(self, sent: str) -> bool:
130
+ if self._trigger_automaton is None:
131
+ return bool(_TRIGGER.search(sent))
132
+ lowered = sent.lower()
133
+ for end, word in self._trigger_automaton.iter(lowered):
134
+ start = end - len(word) + 1
135
+ if (start == 0 or not lowered[start - 1].isalnum()) and \
136
+ (end + 1 == len(lowered) or not lowered[end + 1].isalnum()):
137
+ return True
138
+ return bool(_DATE_TRIGGER.search(sent)) or bool(_TRIGGER.search(sent))
116
139
 
117
140
  # ------------------------------------------------------------------
118
141
  def extract(self, text: str, ctx: ExtractionContext) -> list[Candidate]:
@@ -184,7 +207,7 @@ class Extractor:
184
207
  # against the sentence with up to N edits. This stays deterministic
185
208
  # (Wu-Manber is bitwise, no learned weights) and <50μs on a typical
186
209
  # sentence — same order as the regex itself.
187
- trigger_fired = bool(_TRIGGER.search(sent))
210
+ trigger_fired = self._strict_trigger_match(sent)
188
211
  bitap_widened = False
189
212
  if not trigger_fired:
190
213
  if (not getattr(self.cfg, "bitap_trigger_enabled", True)
@@ -163,6 +163,11 @@ class TinyTransformerFallback:
163
163
  self.seed = seed & 0xFFFFFFFFFFFFFFFF
164
164
  self.max_tokens = max_tokens
165
165
  self.tables = _HashedTables(self.seed)
166
+ # Relation labels are drawn from a small, stable vocabulary. The
167
+ # fallback evaluates every label for every pattern miss, so caching
168
+ # their deterministic embeddings avoids repeating the same attention
169
+ # computation dozens of times per sentence.
170
+ self._relation_embeddings: dict[str, np.ndarray] = {}
166
171
 
167
172
  # ------------------------------------------------------------------
168
173
  def _tokenize(self, text: str) -> list[str]:
@@ -213,6 +218,19 @@ class TinyTransformerFallback:
213
218
  n = float(np.linalg.norm(pooled))
214
219
  return pooled / n if n > 0 else pooled
215
220
 
221
+ def _relation_embedding(self, relation: str) -> np.ndarray:
222
+ """Return the deterministic embedding for a relation label.
223
+
224
+ Relation labels are immutable strings and ``embed`` has no mutable
225
+ input-dependent state, so retaining this vector is semantically
226
+ equivalent to recomputing it on every fallback invocation.
227
+ """
228
+ hit = self._relation_embeddings.get(relation)
229
+ if hit is None:
230
+ hit = self.embed(relation.replace("_", " "))
231
+ self._relation_embeddings[relation] = hit
232
+ return hit
233
+
216
234
  # ------------------------------------------------------------------
217
235
  def extract_candidates(self, sent: str, *,
218
236
  subject_hint: str | None = None,
@@ -251,7 +269,7 @@ class TinyTransformerFallback:
251
269
  rels = relations or _DEFAULT_RELATIONS
252
270
  rel_scores = []
253
271
  for r in rels:
254
- r_emb = self.embed(r.replace("_", " "))
272
+ r_emb = self._relation_embedding(r)
255
273
  score = float(np.dot(sent_emb, r_emb))
256
274
  rel_scores.append((r, score, r_emb))
257
275
  rel_scores.sort(key=lambda x: -x[1])