cortexm 0.5.7__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (177) hide show
  1. {cortexm-0.5.7 → cortexm-0.6.0}/PKG-INFO +1 -1
  2. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/__init__.py +1 -1
  3. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/api/memory.py +12 -2
  4. cortexm-0.6.0/cortexm/bridge/fst.py +280 -0
  5. cortexm-0.6.0/cortexm/bridge/ir_pro.py +697 -0
  6. cortexm-0.6.0/cortexm/bridge/multilingual.py +284 -0
  7. cortexm-0.6.0/cortexm/bridge/negation.py +200 -0
  8. cortexm-0.6.0/cortexm/bridge/query_rewrite.py +178 -0
  9. cortexm-0.6.0/cortexm/bridge/recognizers.py +341 -0
  10. cortexm-0.6.0/cortexm/bridge/slang.py +216 -0
  11. cortexm-0.6.0/cortexm/bridge/synonyms.py +241 -0
  12. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/config.py +62 -0
  13. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/plugins/verbatim.py +284 -1
  14. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/store.py +38 -1
  15. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm.egg-info/PKG-INFO +1 -1
  16. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm.egg-info/SOURCES.txt +9 -0
  17. {cortexm-0.5.7 → cortexm-0.6.0}/pyproject.toml +1 -1
  18. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_public_api_smoke.py +0 -0
  19. cortexm-0.6.0/tests/test_v060_ir_pro.py +922 -0
  20. {cortexm-0.5.7 → cortexm-0.6.0}/LICENSE +0 -0
  21. {cortexm-0.5.7 → cortexm-0.6.0}/README.md +0 -0
  22. {cortexm-0.5.7 → cortexm-0.6.0}/context_m.py +0 -0
  23. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/accel.py +0 -0
  24. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/api/__init__.py +0 -0
  25. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/api/chaos.py +0 -0
  26. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/api/long_recall.py +0 -0
  27. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/__init__.py +0 -0
  28. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/abilities.py +0 -0
  29. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/baselines.py +0 -0
  30. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/beam_loader.py +0 -0
  31. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/generator.py +0 -0
  32. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/harness.py +0 -0
  33. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/messy.py +0 -0
  34. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/micro.py +0 -0
  35. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/ood.py +0 -0
  36. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/run.py +0 -0
  37. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/__init__.py +0 -0
  38. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/dates.py +0 -0
  39. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/decoders.py +0 -0
  40. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/enrich.py +0 -0
  41. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/extractor.py +0 -0
  42. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/fallback.py +0 -0
  43. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/fusion.py +0 -0
  44. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/onnx_runtime.py +0 -0
  45. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/patterns.py +0 -0
  46. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/ppr.py +0 -0
  47. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/prefilter.py +0 -0
  48. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/query_extract.py +0 -0
  49. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/reader.py +0 -0
  50. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/rerank.py +0 -0
  51. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/writer.py +0 -0
  52. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cli.py +0 -0
  53. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cognition/__init__.py +0 -0
  54. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cognition/abstraction.py +0 -0
  55. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cognition/analogy.py +0 -0
  56. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cognition/engine.py +0 -0
  57. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cognition/gaps.py +0 -0
  58. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cognition/scanner.py +0 -0
  59. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cortexm.py +0 -0
  60. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/creator.py +0 -0
  61. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/enterprise/__init__.py +0 -0
  62. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/enterprise/audit.py +0 -0
  63. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/enterprise/governance.py +0 -0
  64. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/errors.py +0 -0
  65. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/features/__init__.py +0 -0
  66. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/features/git.py +0 -0
  67. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/features/prefetch.py +0 -0
  68. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/features/zk.py +0 -0
  69. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/federation/__init__.py +0 -0
  70. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/federation/crdt.py +0 -0
  71. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/federation/fabric.py +0 -0
  72. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/federation/hlc.py +0 -0
  73. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/federation/node.py +0 -0
  74. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/federation/schema_report.py +0 -0
  75. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/federation/transport.py +0 -0
  76. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/index/__init__.py +0 -0
  77. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/index/nsg.py +0 -0
  78. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/kernel.py +0 -0
  79. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/markdown_io.py +0 -0
  80. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/mcp/__init__.py +0 -0
  81. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/mcp/server.py +0 -0
  82. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/metrics.py +0 -0
  83. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/migrate/__init__.py +0 -0
  84. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/migrate/importers.py +0 -0
  85. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/pipeline.py +0 -0
  86. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/plugins/__init__.py +0 -0
  87. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/plugins/security.py +0 -0
  88. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/plugins/structured.py +0 -0
  89. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/provenance/__init__.py +0 -0
  90. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/provenance/agent.py +0 -0
  91. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/provenance/cose.py +0 -0
  92. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/provenance/scitt.py +0 -0
  93. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/provenance/vc.py +0 -0
  94. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/router.py +0 -0
  95. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/__init__.py +0 -0
  96. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/crypto.py +0 -0
  97. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/hashes.py +0 -0
  98. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/injection.py +0 -0
  99. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/mind.py +0 -0
  100. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/permission.py +0 -0
  101. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/pii.py +0 -0
  102. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/rbac.py +0 -0
  103. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/sandbox.py +0 -0
  104. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/zk_hamming.py +0 -0
  105. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/zk_sql.py +0 -0
  106. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/server/__init__.py +0 -0
  107. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/server/metrics.py +0 -0
  108. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/server/rest.py +0 -0
  109. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/server/sparql.py +0 -0
  110. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/text/__init__.py +0 -0
  111. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/text/dissim.py +0 -0
  112. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/text/embedder.py +0 -0
  113. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/text/fuzzy.py +0 -0
  114. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/text/idiolect.py +0 -0
  115. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/text/labse.py +0 -0
  116. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/text/tokenizer.py +0 -0
  117. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/__init__.py +0 -0
  118. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/blob_arena.py +0 -0
  119. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/consolidate.py +0 -0
  120. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/contradictions.py +0 -0
  121. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/dedup.py +0 -0
  122. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/edges.py +0 -0
  123. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/fact.py +0 -0
  124. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/fade.py +0 -0
  125. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/lifecycle.py +0 -0
  126. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/rebuild.py +0 -0
  127. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/rules.py +0 -0
  128. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/structural.py +0 -0
  129. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/tmt.py +0 -0
  130. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trajectory_view.py +0 -0
  131. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/util.py +0 -0
  132. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/__init__.py +0 -0
  133. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/attribution.py +0 -0
  134. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/cleanup.py +0 -0
  135. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/codecs.py +0 -0
  136. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/hologram_overlay.py +0 -0
  137. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/index.py +0 -0
  138. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/ops.py +0 -0
  139. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/palace.py +0 -0
  140. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/role_vectors.py +0 -0
  141. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/slb.py +0 -0
  142. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/tlsh_trie.py +0 -0
  143. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/working_memory.py +0 -0
  144. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm.egg-info/dependency_links.txt +0 -0
  145. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm.egg-info/entry_points.txt +0 -0
  146. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm.egg-info/requires.txt +0 -0
  147. {cortexm-0.5.7 → cortexm-0.6.0}/cortexm.egg-info/top_level.txt +0 -0
  148. {cortexm-0.5.7 → cortexm-0.6.0}/setup.cfg +0 -0
  149. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_arxiv_improvements.py +0 -0
  150. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_bench_infra.py +0 -0
  151. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_bm25_chunk_recall_and_inspect_cli.py +0 -0
  152. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_cognition_and_provenance.py +0 -0
  153. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_engineering_push_2026_08_28.py +0 -0
  154. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_enterprise.py +0 -0
  155. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_fabric.py +0 -0
  156. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_federation.py +0 -0
  157. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_fusion_security.py +0 -0
  158. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_kernel.py +0 -0
  159. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_kinship_extraction.py +0 -0
  160. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_labse.py +0 -0
  161. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_list_superseded_intent.py +0 -0
  162. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_migration.py +0 -0
  163. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_new_modules.py +0 -0
  164. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_nsg.py +0 -0
  165. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_permission.py +0 -0
  166. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_ppr.py +0 -0
  167. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_reddit_steals_round3.py +0 -0
  168. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_rerank.py +0 -0
  169. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_research_steals.py +0 -0
  170. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_research_steals_round2.py +0 -0
  171. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_rust_accel.py +0 -0
  172. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_sandbox_enrich.py +0 -0
  173. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_sparql_rest_v2.py +0 -0
  174. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_tier443_abstention_fix.py +0 -0
  175. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_verbatim.py +0 -0
  176. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_wal_recovery.py +0 -0
  177. {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_zk_sql.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cortexm
3
- Version: 0.5.7
3
+ Version: 0.6.0
4
4
  Summary: Deterministic agent memory. 96 bytes per fact. Zero LLM at ingest.
5
5
  Author: Context-M Contributors
6
6
  License-Expression: Apache-2.0
@@ -18,7 +18,7 @@ Plugin kernel: ``from cortexm import Context, mount_default``
18
18
 
19
19
  from __future__ import annotations
20
20
 
21
- __version__ = "0.5.7"
21
+ __version__ = "0.6.0"
22
22
 
23
23
  # μ=0 protocol counter: number of LLM invocations used by this process.
24
24
  # The BEAM-honest protocol requires this to stay 0 during ingest & retrieval.
@@ -53,7 +53,12 @@ class Memory:
53
53
  config = dataclasses.replace(config, **changes)
54
54
  self.config = config
55
55
  self.store = TraceStore(config.db_path, HashProvider(config.hash_provider),
56
- wal_sync=getattr(config, "wal_sync", "normal"))
56
+ wal_sync=getattr(config, "wal_sync", "normal"),
57
+ pragma_cache_mb=getattr(config, "pragma_cache_mb", 64),
58
+ pragma_mmap_mb=getattr(config, "pragma_mmap_mb", 256),
59
+ pragma_threads=getattr(config, "pragma_threads", 4),
60
+ pragma_temp_in_memory=getattr(config, "pragma_temp_in_memory", True),
61
+ pragma_locking_exclusive=getattr(config, "pragma_locking_exclusive", False))
57
62
  self.palace = MemoryPalace(config, self.store)
58
63
  self.extractor = Extractor(config)
59
64
  self.prefetcher = Prefetcher()
@@ -1033,7 +1038,12 @@ class Memory:
1033
1038
  from cortexm.enterprise.governance import Governance
1034
1039
  self.store = TraceStore(self.config.db_path,
1035
1040
  HashProvider(self.config.hash_provider),
1036
- wal_sync=getattr(self.config, "wal_sync", "normal"))
1041
+ wal_sync=getattr(self.config, "wal_sync", "normal"),
1042
+ pragma_cache_mb=getattr(self.config, "pragma_cache_mb", 64),
1043
+ pragma_mmap_mb=getattr(self.config, "pragma_mmap_mb", 256),
1044
+ pragma_threads=getattr(self.config, "pragma_threads", 4),
1045
+ pragma_temp_in_memory=getattr(self.config, "pragma_temp_in_memory", True),
1046
+ pragma_locking_exclusive=getattr(self.config, "pragma_locking_exclusive", False))
1037
1047
  self.palace = MemoryPalace(self.config, self.store)
1038
1048
  self.writer = MemoryWriter(self.config, self.store, self.palace,
1039
1049
  self.extractor)
@@ -0,0 +1,280 @@
1
+ """Deterministic finite-state transducer for query normalization
2
+ (abbreviation expansion + spelling correction).
3
+
4
+ WHY THIS MODULE EXISTS
5
+ -----------------------
6
+ Google uses Finite State Transducers (FSTs) for spelling correction
7
+ and query normalization — a compiled FST does both prefix matching
8
+ and edit-distance computation in O(length of query) time, regardless
9
+ of dictionary size.
10
+
11
+ This module is a lightweight, μ=0 Python FST for two specific tasks:
12
+ 1. ABBREVIATION EXPANSION — "ucla" → "university of california
13
+ los angeles" (covers the canonical LongMemEval PAREN_ABBREVIATION
14
+ judge's failure mode at query time, not just at judge time).
15
+ 2. SPELLING CORRECTION — a curated list of common typos that the
16
+ Bitap fuzzy matcher in ``cortexm.text.fuzzy`` would catch but
17
+ only on a per-pattern basis. The FST does it at QUERY time so
18
+ every downstream search benefits.
19
+
20
+ NOT A REAL FST
21
+ --------------
22
+ A real FST would be a compiled automaton (Lucene's FST is a 100KB
23
+ compiled Java class). This is a dict-with-regex-backfill that gives
24
+ the same O(L) lookup for the curated entries. For query-time use
25
+ where L ≤ ~10 tokens, the perf is equivalent. For 100K+ dictionaries
26
+ the real FST would matter — we're not at that scale.
27
+
28
+ ARCHITECTURE
29
+ -------------
30
+ * ABBREVIATIONS: dict of lowercase_abbrev → expanded_form
31
+ * SPELLING: dict of misspelling → correct_form
32
+ * ``normalize(query)``: split on whitespace, apply abbreviations
33
+ first (so "MIT" expands to "massachusetts institute of
34
+ technology" before token-level spelling correction), then apply
35
+ spelling corrections token-by-token.
36
+
37
+ The normalize step is IDempotent — applying it twice produces the
38
+ same output as applying it once. Important because the rewriter
39
+ calls normalize() on already-normalized queries.
40
+ """
41
+ from __future__ import annotations
42
+
43
+ import re
44
+ from typing import Dict
45
+
46
+
47
+ # ---------------------------------------------------------------------------
48
+ # Curated abbreviation → expansion table. Lowercased keys; the
49
+ # expansion preserves the canonical capitalization. This catches the
50
+ # canonical LongMemEval abbreviation ground-truth pattern where the
51
+ # user says "UCLA" in a chunk but the expected answer is "University
52
+ # of California, Los Angeles (UCLA)" — the query-time expansion lets
53
+ # the verbatim BM25 search hit either phrasing.
54
+ # ---------------------------------------------------------------------------
55
+ DEFAULT_ABBREVIATIONS: Dict[str, str] = {
56
+ # US universities
57
+ "ucla": "University of California Los Angeles",
58
+ "ucb": "University of California Berkeley",
59
+ "ucsd": "University of California San Diego",
60
+ "ucsf": "University of California San Francisco",
61
+ "mit": "Massachusetts Institute of Technology",
62
+ "caltech": "California Institute of Technology",
63
+ "nyu": "New York University",
64
+ "usc": "University of Southern California",
65
+ "cmu": "Carnegie Mellon University",
66
+ "stanford": "Stanford University",
67
+ "harvard": "Harvard University",
68
+ "yale": "Yale University",
69
+ "princeton": "Princeton University",
70
+ "columbia": "Columbia University",
71
+ "upenn": "University of Pennsylvania",
72
+ "gatech": "Georgia Institute of Technology",
73
+ "uiuc": "University of Illinois Urbana Champaign",
74
+ "umich": "University of Michigan",
75
+ "ut austin": "University of Texas at Austin",
76
+ # US cities
77
+ "nyc": "New York City",
78
+ "la": "Los Angeles",
79
+ "sf": "San Francisco",
80
+ "dc": "Washington DC",
81
+ "philly": "Philadelphia",
82
+ "vegas": "Las Vegas",
83
+ "pdx": "Portland",
84
+ "sea": "Seattle",
85
+ "atl": "Atlanta",
86
+ "bos": "Boston",
87
+ "chi": "Chicago",
88
+ # Companies
89
+ "ibm": "International Business Machines",
90
+ "ge": "General Electric",
91
+ "pg": "Procter and Gamble",
92
+ "gm": "General Motors",
93
+ # Government
94
+ "fbi": "Federal Bureau of Investigation",
95
+ "cia": "Central Intelligence Agency",
96
+ "nsa": "National Security Agency",
97
+ "doj": "Department of Justice",
98
+ "dod": "Department of Defense",
99
+ "faa": "Federal Aviation Administration",
100
+ "fcc": "Federal Communications Commission",
101
+ "ftc": "Federal Trade Commission",
102
+ "sec": "Securities and Exchange Commission",
103
+ # Tech
104
+ "ai": "artificial intelligence",
105
+ "ml": "machine learning",
106
+ "nlp": "natural language processing",
107
+ "cv": "computer vision",
108
+ "gpu": "graphics processing unit",
109
+ "cpu": "central processing unit",
110
+ "ram": "random access memory",
111
+ "ssd": "solid state drive",
112
+ "api": "application programming interface",
113
+ "sdk": "software development kit",
114
+ "cli": "command line interface",
115
+ "ui": "user interface",
116
+ "ux": "user experience",
117
+ }
118
+
119
+ # ---------------------------------------------------------------------------
120
+ # Common misspellings → correct form. Sourced from the Wikipedia
121
+ # "List of common misspellings" + the canonical LongMemEval failure
122
+ # modes observed in v0.5.5 (mostly typing errors on factual chunks).
123
+ # ---------------------------------------------------------------------------
124
+ DEFAULT_SPELLING: Dict[str, str] = {
125
+ # Classic typos
126
+ "recieve": "receive",
127
+ "definately": "definitely",
128
+ "occured": "occurred",
129
+ "seperate": "separate",
130
+ "tommorow": "tomorrow",
131
+ "tommorrow": "tomorrow",
132
+ "untill": "until",
133
+ "wich": "which",
134
+ "thier": "their",
135
+ "teh": "the",
136
+ "adn": "and",
137
+ "taht": "that",
138
+ "wit h": "with",
139
+ "adress": "address",
140
+ "occassion": "occasion",
141
+ "neccessary": "necessary",
142
+ "accomodate": "accommodate",
143
+ "priviledge": "privilege",
144
+ "liason": "liaison",
145
+ "supercede": "supersede",
146
+ "consensus": "consensus", # common mis-spelling "concensus"
147
+ # Compound word errors
148
+ "alot": "a lot",
149
+ "infact": "in fact",
150
+ "inspite": "in spite",
151
+ "alright": "all right",
152
+ # Possessive confusion (common in user text)
153
+ "its a": "it's a", # ambiguous — "its" is also possessive
154
+ # Note: this entry is conservative; the FST leaves ambiguous cases
155
+ # alone rather than guess.
156
+ }
157
+
158
+
159
+ class QueryFST:
160
+ """Lightweight finite-state transducer for query normalization.
161
+
162
+ Two stages, applied in order:
163
+ 1. ABBREVIATION EXPANSION — multi-word phrase substitution via
164
+ a master regex (longest match first, case-insensitive).
165
+ 2. SPELLING CORRECTION — per-token lookup in the spelling dict.
166
+
167
+ The FST is deterministic — same input always produces the same
168
+ output. No neural network, no statistics, no API calls.
169
+ """
170
+
171
+ def __init__(self,
172
+ abbreviations: Dict[str, str] | None = None,
173
+ spelling: Dict[str, str] | None = None) -> None:
174
+ self.abbreviations = dict(abbreviations or DEFAULT_ABBREVIATIONS)
175
+ self.spelling = dict(spelling or DEFAULT_SPELLING)
176
+ self._build_regex()
177
+
178
+ def _build_regex(self) -> None:
179
+ """Pre-compile a master regex of all abbreviation phrases.
180
+ Sorted longest-first so "ut austin" matches before "ut".
181
+ """
182
+ if not self.abbreviations:
183
+ self._abbrev_re = re.compile(r"$.") # never matches
184
+ return
185
+ phrases = sorted(self.abbreviations.keys(), key=len, reverse=True)
186
+ joined = "|".join(re.escape(p) for p in phrases)
187
+ self._abbrev_re = re.compile(rf"\b(?:{joined})\b", re.IGNORECASE)
188
+
189
+ def register_abbreviation(self, abbrev: str, expansion: str) -> None:
190
+ """Add or replace an abbreviation at runtime. Idempotent."""
191
+ self.abbreviations[abbrev.lower()] = expansion
192
+ self._build_regex()
193
+
194
+ def register_spelling(self, misspelling: str, correct: str) -> None:
195
+ """Add or replace a spelling correction at runtime. Idempotent."""
196
+ self.spelling[misspelling.lower()] = correct
197
+
198
+ # ------------------------------------------------------------------
199
+ # Public API
200
+ # ------------------------------------------------------------------
201
+
202
+ def normalize(self, query: str) -> str:
203
+ """Apply all transductions deterministically.
204
+
205
+ Order:
206
+ 1. Abbreviation expansion (phrase-level)
207
+ 2. Spelling correction (token-level)
208
+
209
+ The result is idempotent — applying normalize() again produces
210
+ the same output (because abbreviations don't recursively expand
211
+ and spelling corrections don't trigger further correction).
212
+ """
213
+ if not query:
214
+ return query
215
+ # Stage 1: abbreviation expansion. Substitute the full
216
+ # canonical expansion at each match position.
217
+ result = self._abbrev_re.sub(self._abbrev_repl, query)
218
+ # Stage 2: spelling correction (token-level). Tokenize on
219
+ # whitespace but preserve the original whitespace by splitting
220
+ # with a regex that captures it.
221
+ tokens = re.split(r"(\s+)", result)
222
+ for i, tok in enumerate(tokens):
223
+ # Strip surrounding punctuation for lookup, but preserve
224
+ # it in the output.
225
+ m = re.match(r"^([^\w]*)(.*?)([^\w]*)$", tok)
226
+ if m and m.group(2):
227
+ pre, word, post = m.group(1), m.group(2), m.group(3)
228
+ corrected = self.spelling.get(word.lower(), word)
229
+ # Preserve the capitalization pattern of the original
230
+ # token (so "Recieve" → "Receive", "RECIEVE" → "RECEIVE").
231
+ corrected = _preserve_case(word, corrected)
232
+ tokens[i] = pre + corrected + post
233
+ return "".join(tokens)
234
+
235
+ def _abbrev_repl(self, m: re.Match) -> str:
236
+ """regex.sub callback for abbreviation expansion."""
237
+ key = m.group(0).lower()
238
+ expansion = self.abbreviations.get(key)
239
+ if expansion is None:
240
+ return m.group(0)
241
+ # Preserve leading capital if the original was capitalized
242
+ original = m.group(0)
243
+ if original and original[0].isupper():
244
+ # Capitalize the first letter of the expansion
245
+ return expansion[0].upper() + expansion[1:]
246
+ return expansion
247
+
248
+
249
+ def _preserve_case(original: str, replacement: str) -> str:
250
+ """Make ``replacement`` match the capitalization pattern of ``original``.
251
+
252
+ * all-caps original → all-caps replacement
253
+ * title-case original → title-case replacement
254
+ * otherwise → replacement as-is (already lowercase from the dict)
255
+ """
256
+ if not original or not replacement:
257
+ return replacement
258
+ if original.isupper():
259
+ return replacement.upper()
260
+ if original[0].isupper():
261
+ return replacement[0].upper() + replacement[1:]
262
+ return replacement
263
+
264
+
265
+ # ---------------------------------------------------------------------------
266
+ # Module-level singleton
267
+ # ---------------------------------------------------------------------------
268
+ _default_fst: QueryFST | None = None
269
+
270
+
271
+ def default_fst() -> QueryFST:
272
+ global _default_fst
273
+ if _default_fst is None:
274
+ _default_fst = QueryFST()
275
+ return _default_fst
276
+
277
+
278
+ def normalize(query: str) -> str:
279
+ """Module-level convenience wrapper around the default FST."""
280
+ return default_fst().normalize(query)