cortexm 0.5.7__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexm-0.5.7 → cortexm-0.6.0}/PKG-INFO +1 -1
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/__init__.py +1 -1
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/api/memory.py +12 -2
- cortexm-0.6.0/cortexm/bridge/fst.py +280 -0
- cortexm-0.6.0/cortexm/bridge/ir_pro.py +697 -0
- cortexm-0.6.0/cortexm/bridge/multilingual.py +284 -0
- cortexm-0.6.0/cortexm/bridge/negation.py +200 -0
- cortexm-0.6.0/cortexm/bridge/query_rewrite.py +178 -0
- cortexm-0.6.0/cortexm/bridge/recognizers.py +341 -0
- cortexm-0.6.0/cortexm/bridge/slang.py +216 -0
- cortexm-0.6.0/cortexm/bridge/synonyms.py +241 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/config.py +62 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/plugins/verbatim.py +284 -1
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/store.py +38 -1
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm.egg-info/PKG-INFO +1 -1
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm.egg-info/SOURCES.txt +9 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/pyproject.toml +1 -1
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_public_api_smoke.py +0 -0
- cortexm-0.6.0/tests/test_v060_ir_pro.py +922 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/LICENSE +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/README.md +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/context_m.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/accel.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/api/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/api/chaos.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/api/long_recall.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/abilities.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/baselines.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/beam_loader.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/generator.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/harness.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/messy.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/micro.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/ood.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bench/run.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/dates.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/decoders.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/enrich.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/extractor.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/fallback.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/fusion.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/onnx_runtime.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/patterns.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/ppr.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/prefilter.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/query_extract.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/reader.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/rerank.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/bridge/writer.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cli.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cognition/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cognition/abstraction.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cognition/analogy.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cognition/engine.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cognition/gaps.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cognition/scanner.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/cortexm.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/creator.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/enterprise/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/enterprise/audit.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/enterprise/governance.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/errors.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/features/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/features/git.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/features/prefetch.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/features/zk.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/federation/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/federation/crdt.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/federation/fabric.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/federation/hlc.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/federation/node.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/federation/schema_report.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/federation/transport.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/index/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/index/nsg.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/kernel.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/markdown_io.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/mcp/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/mcp/server.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/metrics.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/migrate/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/migrate/importers.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/pipeline.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/plugins/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/plugins/security.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/plugins/structured.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/provenance/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/provenance/agent.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/provenance/cose.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/provenance/scitt.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/provenance/vc.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/router.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/crypto.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/hashes.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/injection.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/mind.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/permission.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/pii.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/rbac.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/sandbox.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/zk_hamming.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/security/zk_sql.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/server/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/server/metrics.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/server/rest.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/server/sparql.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/text/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/text/dissim.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/text/embedder.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/text/fuzzy.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/text/idiolect.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/text/labse.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/text/tokenizer.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/blob_arena.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/consolidate.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/contradictions.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/dedup.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/edges.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/fact.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/fade.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/lifecycle.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/rebuild.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/rules.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/structural.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trace/tmt.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/trajectory_view.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/util.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/__init__.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/attribution.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/cleanup.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/codecs.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/hologram_overlay.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/index.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/ops.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/palace.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/role_vectors.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/slb.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/tlsh_trie.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm/vsa/working_memory.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm.egg-info/dependency_links.txt +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm.egg-info/entry_points.txt +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm.egg-info/requires.txt +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/cortexm.egg-info/top_level.txt +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/setup.cfg +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_arxiv_improvements.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_bench_infra.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_bm25_chunk_recall_and_inspect_cli.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_cognition_and_provenance.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_engineering_push_2026_08_28.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_enterprise.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_fabric.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_federation.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_fusion_security.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_kernel.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_kinship_extraction.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_labse.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_list_superseded_intent.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_migration.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_new_modules.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_nsg.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_permission.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_ppr.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_reddit_steals_round3.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_rerank.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_research_steals.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_research_steals_round2.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_rust_accel.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_sandbox_enrich.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_sparql_rest_v2.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_tier443_abstention_fix.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_verbatim.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_wal_recovery.py +0 -0
- {cortexm-0.5.7 → cortexm-0.6.0}/tests/test_zk_sql.py +0 -0
|
@@ -18,7 +18,7 @@ Plugin kernel: ``from cortexm import Context, mount_default``
|
|
|
18
18
|
|
|
19
19
|
from __future__ import annotations
|
|
20
20
|
|
|
21
|
-
__version__ = "0.
|
|
21
|
+
__version__ = "0.6.0"
|
|
22
22
|
|
|
23
23
|
# μ=0 protocol counter: number of LLM invocations used by this process.
|
|
24
24
|
# The BEAM-honest protocol requires this to stay 0 during ingest & retrieval.
|
|
@@ -53,7 +53,12 @@ class Memory:
|
|
|
53
53
|
config = dataclasses.replace(config, **changes)
|
|
54
54
|
self.config = config
|
|
55
55
|
self.store = TraceStore(config.db_path, HashProvider(config.hash_provider),
|
|
56
|
-
wal_sync=getattr(config, "wal_sync", "normal")
|
|
56
|
+
wal_sync=getattr(config, "wal_sync", "normal"),
|
|
57
|
+
pragma_cache_mb=getattr(config, "pragma_cache_mb", 64),
|
|
58
|
+
pragma_mmap_mb=getattr(config, "pragma_mmap_mb", 256),
|
|
59
|
+
pragma_threads=getattr(config, "pragma_threads", 4),
|
|
60
|
+
pragma_temp_in_memory=getattr(config, "pragma_temp_in_memory", True),
|
|
61
|
+
pragma_locking_exclusive=getattr(config, "pragma_locking_exclusive", False))
|
|
57
62
|
self.palace = MemoryPalace(config, self.store)
|
|
58
63
|
self.extractor = Extractor(config)
|
|
59
64
|
self.prefetcher = Prefetcher()
|
|
@@ -1033,7 +1038,12 @@ class Memory:
|
|
|
1033
1038
|
from cortexm.enterprise.governance import Governance
|
|
1034
1039
|
self.store = TraceStore(self.config.db_path,
|
|
1035
1040
|
HashProvider(self.config.hash_provider),
|
|
1036
|
-
wal_sync=getattr(self.config, "wal_sync", "normal")
|
|
1041
|
+
wal_sync=getattr(self.config, "wal_sync", "normal"),
|
|
1042
|
+
pragma_cache_mb=getattr(self.config, "pragma_cache_mb", 64),
|
|
1043
|
+
pragma_mmap_mb=getattr(self.config, "pragma_mmap_mb", 256),
|
|
1044
|
+
pragma_threads=getattr(self.config, "pragma_threads", 4),
|
|
1045
|
+
pragma_temp_in_memory=getattr(self.config, "pragma_temp_in_memory", True),
|
|
1046
|
+
pragma_locking_exclusive=getattr(self.config, "pragma_locking_exclusive", False))
|
|
1037
1047
|
self.palace = MemoryPalace(self.config, self.store)
|
|
1038
1048
|
self.writer = MemoryWriter(self.config, self.store, self.palace,
|
|
1039
1049
|
self.extractor)
|
|
@@ -0,0 +1,280 @@
|
|
|
1
|
+
"""Deterministic finite-state transducer for query normalization
|
|
2
|
+
(abbreviation expansion + spelling correction).
|
|
3
|
+
|
|
4
|
+
WHY THIS MODULE EXISTS
|
|
5
|
+
-----------------------
|
|
6
|
+
Google uses Finite State Transducers (FSTs) for spelling correction
|
|
7
|
+
and query normalization — a compiled FST does both prefix matching
|
|
8
|
+
and edit-distance computation in O(length of query) time, regardless
|
|
9
|
+
of dictionary size.
|
|
10
|
+
|
|
11
|
+
This module is a lightweight, μ=0 Python FST for two specific tasks:
|
|
12
|
+
1. ABBREVIATION EXPANSION — "ucla" → "university of california
|
|
13
|
+
los angeles" (covers the canonical LongMemEval PAREN_ABBREVIATION
|
|
14
|
+
judge's failure mode at query time, not just at judge time).
|
|
15
|
+
2. SPELLING CORRECTION — a curated list of common typos that the
|
|
16
|
+
Bitap fuzzy matcher in ``cortexm.text.fuzzy`` would catch but
|
|
17
|
+
only on a per-pattern basis. The FST does it at QUERY time so
|
|
18
|
+
every downstream search benefits.
|
|
19
|
+
|
|
20
|
+
NOT A REAL FST
|
|
21
|
+
--------------
|
|
22
|
+
A real FST would be a compiled automaton (Lucene's FST is a 100KB
|
|
23
|
+
compiled Java class). This is a dict-with-regex-backfill that gives
|
|
24
|
+
the same O(L) lookup for the curated entries. For query-time use
|
|
25
|
+
where L ≤ ~10 tokens, the perf is equivalent. For 100K+ dictionaries
|
|
26
|
+
the real FST would matter — we're not at that scale.
|
|
27
|
+
|
|
28
|
+
ARCHITECTURE
|
|
29
|
+
-------------
|
|
30
|
+
* ABBREVIATIONS: dict of lowercase_abbrev → expanded_form
|
|
31
|
+
* SPELLING: dict of misspelling → correct_form
|
|
32
|
+
* ``normalize(query)``: split on whitespace, apply abbreviations
|
|
33
|
+
first (so "MIT" expands to "massachusetts institute of
|
|
34
|
+
technology" before token-level spelling correction), then apply
|
|
35
|
+
spelling corrections token-by-token.
|
|
36
|
+
|
|
37
|
+
The normalize step is IDempotent — applying it twice produces the
|
|
38
|
+
same output as applying it once. Important because the rewriter
|
|
39
|
+
calls normalize() on already-normalized queries.
|
|
40
|
+
"""
|
|
41
|
+
from __future__ import annotations
|
|
42
|
+
|
|
43
|
+
import re
|
|
44
|
+
from typing import Dict
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
# ---------------------------------------------------------------------------
|
|
48
|
+
# Curated abbreviation → expansion table. Lowercased keys; the
|
|
49
|
+
# expansion preserves the canonical capitalization. This catches the
|
|
50
|
+
# canonical LongMemEval abbreviation ground-truth pattern where the
|
|
51
|
+
# user says "UCLA" in a chunk but the expected answer is "University
|
|
52
|
+
# of California, Los Angeles (UCLA)" — the query-time expansion lets
|
|
53
|
+
# the verbatim BM25 search hit either phrasing.
|
|
54
|
+
# ---------------------------------------------------------------------------
|
|
55
|
+
DEFAULT_ABBREVIATIONS: Dict[str, str] = {
|
|
56
|
+
# US universities
|
|
57
|
+
"ucla": "University of California Los Angeles",
|
|
58
|
+
"ucb": "University of California Berkeley",
|
|
59
|
+
"ucsd": "University of California San Diego",
|
|
60
|
+
"ucsf": "University of California San Francisco",
|
|
61
|
+
"mit": "Massachusetts Institute of Technology",
|
|
62
|
+
"caltech": "California Institute of Technology",
|
|
63
|
+
"nyu": "New York University",
|
|
64
|
+
"usc": "University of Southern California",
|
|
65
|
+
"cmu": "Carnegie Mellon University",
|
|
66
|
+
"stanford": "Stanford University",
|
|
67
|
+
"harvard": "Harvard University",
|
|
68
|
+
"yale": "Yale University",
|
|
69
|
+
"princeton": "Princeton University",
|
|
70
|
+
"columbia": "Columbia University",
|
|
71
|
+
"upenn": "University of Pennsylvania",
|
|
72
|
+
"gatech": "Georgia Institute of Technology",
|
|
73
|
+
"uiuc": "University of Illinois Urbana Champaign",
|
|
74
|
+
"umich": "University of Michigan",
|
|
75
|
+
"ut austin": "University of Texas at Austin",
|
|
76
|
+
# US cities
|
|
77
|
+
"nyc": "New York City",
|
|
78
|
+
"la": "Los Angeles",
|
|
79
|
+
"sf": "San Francisco",
|
|
80
|
+
"dc": "Washington DC",
|
|
81
|
+
"philly": "Philadelphia",
|
|
82
|
+
"vegas": "Las Vegas",
|
|
83
|
+
"pdx": "Portland",
|
|
84
|
+
"sea": "Seattle",
|
|
85
|
+
"atl": "Atlanta",
|
|
86
|
+
"bos": "Boston",
|
|
87
|
+
"chi": "Chicago",
|
|
88
|
+
# Companies
|
|
89
|
+
"ibm": "International Business Machines",
|
|
90
|
+
"ge": "General Electric",
|
|
91
|
+
"pg": "Procter and Gamble",
|
|
92
|
+
"gm": "General Motors",
|
|
93
|
+
# Government
|
|
94
|
+
"fbi": "Federal Bureau of Investigation",
|
|
95
|
+
"cia": "Central Intelligence Agency",
|
|
96
|
+
"nsa": "National Security Agency",
|
|
97
|
+
"doj": "Department of Justice",
|
|
98
|
+
"dod": "Department of Defense",
|
|
99
|
+
"faa": "Federal Aviation Administration",
|
|
100
|
+
"fcc": "Federal Communications Commission",
|
|
101
|
+
"ftc": "Federal Trade Commission",
|
|
102
|
+
"sec": "Securities and Exchange Commission",
|
|
103
|
+
# Tech
|
|
104
|
+
"ai": "artificial intelligence",
|
|
105
|
+
"ml": "machine learning",
|
|
106
|
+
"nlp": "natural language processing",
|
|
107
|
+
"cv": "computer vision",
|
|
108
|
+
"gpu": "graphics processing unit",
|
|
109
|
+
"cpu": "central processing unit",
|
|
110
|
+
"ram": "random access memory",
|
|
111
|
+
"ssd": "solid state drive",
|
|
112
|
+
"api": "application programming interface",
|
|
113
|
+
"sdk": "software development kit",
|
|
114
|
+
"cli": "command line interface",
|
|
115
|
+
"ui": "user interface",
|
|
116
|
+
"ux": "user experience",
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
# ---------------------------------------------------------------------------
|
|
120
|
+
# Common misspellings → correct form. Sourced from the Wikipedia
|
|
121
|
+
# "List of common misspellings" + the canonical LongMemEval failure
|
|
122
|
+
# modes observed in v0.5.5 (mostly typing errors on factual chunks).
|
|
123
|
+
# ---------------------------------------------------------------------------
|
|
124
|
+
DEFAULT_SPELLING: Dict[str, str] = {
|
|
125
|
+
# Classic typos
|
|
126
|
+
"recieve": "receive",
|
|
127
|
+
"definately": "definitely",
|
|
128
|
+
"occured": "occurred",
|
|
129
|
+
"seperate": "separate",
|
|
130
|
+
"tommorow": "tomorrow",
|
|
131
|
+
"tommorrow": "tomorrow",
|
|
132
|
+
"untill": "until",
|
|
133
|
+
"wich": "which",
|
|
134
|
+
"thier": "their",
|
|
135
|
+
"teh": "the",
|
|
136
|
+
"adn": "and",
|
|
137
|
+
"taht": "that",
|
|
138
|
+
"wit h": "with",
|
|
139
|
+
"adress": "address",
|
|
140
|
+
"occassion": "occasion",
|
|
141
|
+
"neccessary": "necessary",
|
|
142
|
+
"accomodate": "accommodate",
|
|
143
|
+
"priviledge": "privilege",
|
|
144
|
+
"liason": "liaison",
|
|
145
|
+
"supercede": "supersede",
|
|
146
|
+
"consensus": "consensus", # common mis-spelling "concensus"
|
|
147
|
+
# Compound word errors
|
|
148
|
+
"alot": "a lot",
|
|
149
|
+
"infact": "in fact",
|
|
150
|
+
"inspite": "in spite",
|
|
151
|
+
"alright": "all right",
|
|
152
|
+
# Possessive confusion (common in user text)
|
|
153
|
+
"its a": "it's a", # ambiguous — "its" is also possessive
|
|
154
|
+
# Note: this entry is conservative; the FST leaves ambiguous cases
|
|
155
|
+
# alone rather than guess.
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class QueryFST:
|
|
160
|
+
"""Lightweight finite-state transducer for query normalization.
|
|
161
|
+
|
|
162
|
+
Two stages, applied in order:
|
|
163
|
+
1. ABBREVIATION EXPANSION — multi-word phrase substitution via
|
|
164
|
+
a master regex (longest match first, case-insensitive).
|
|
165
|
+
2. SPELLING CORRECTION — per-token lookup in the spelling dict.
|
|
166
|
+
|
|
167
|
+
The FST is deterministic — same input always produces the same
|
|
168
|
+
output. No neural network, no statistics, no API calls.
|
|
169
|
+
"""
|
|
170
|
+
|
|
171
|
+
def __init__(self,
|
|
172
|
+
abbreviations: Dict[str, str] | None = None,
|
|
173
|
+
spelling: Dict[str, str] | None = None) -> None:
|
|
174
|
+
self.abbreviations = dict(abbreviations or DEFAULT_ABBREVIATIONS)
|
|
175
|
+
self.spelling = dict(spelling or DEFAULT_SPELLING)
|
|
176
|
+
self._build_regex()
|
|
177
|
+
|
|
178
|
+
def _build_regex(self) -> None:
|
|
179
|
+
"""Pre-compile a master regex of all abbreviation phrases.
|
|
180
|
+
Sorted longest-first so "ut austin" matches before "ut".
|
|
181
|
+
"""
|
|
182
|
+
if not self.abbreviations:
|
|
183
|
+
self._abbrev_re = re.compile(r"$.") # never matches
|
|
184
|
+
return
|
|
185
|
+
phrases = sorted(self.abbreviations.keys(), key=len, reverse=True)
|
|
186
|
+
joined = "|".join(re.escape(p) for p in phrases)
|
|
187
|
+
self._abbrev_re = re.compile(rf"\b(?:{joined})\b", re.IGNORECASE)
|
|
188
|
+
|
|
189
|
+
def register_abbreviation(self, abbrev: str, expansion: str) -> None:
|
|
190
|
+
"""Add or replace an abbreviation at runtime. Idempotent."""
|
|
191
|
+
self.abbreviations[abbrev.lower()] = expansion
|
|
192
|
+
self._build_regex()
|
|
193
|
+
|
|
194
|
+
def register_spelling(self, misspelling: str, correct: str) -> None:
|
|
195
|
+
"""Add or replace a spelling correction at runtime. Idempotent."""
|
|
196
|
+
self.spelling[misspelling.lower()] = correct
|
|
197
|
+
|
|
198
|
+
# ------------------------------------------------------------------
|
|
199
|
+
# Public API
|
|
200
|
+
# ------------------------------------------------------------------
|
|
201
|
+
|
|
202
|
+
def normalize(self, query: str) -> str:
|
|
203
|
+
"""Apply all transductions deterministically.
|
|
204
|
+
|
|
205
|
+
Order:
|
|
206
|
+
1. Abbreviation expansion (phrase-level)
|
|
207
|
+
2. Spelling correction (token-level)
|
|
208
|
+
|
|
209
|
+
The result is idempotent — applying normalize() again produces
|
|
210
|
+
the same output (because abbreviations don't recursively expand
|
|
211
|
+
and spelling corrections don't trigger further correction).
|
|
212
|
+
"""
|
|
213
|
+
if not query:
|
|
214
|
+
return query
|
|
215
|
+
# Stage 1: abbreviation expansion. Substitute the full
|
|
216
|
+
# canonical expansion at each match position.
|
|
217
|
+
result = self._abbrev_re.sub(self._abbrev_repl, query)
|
|
218
|
+
# Stage 2: spelling correction (token-level). Tokenize on
|
|
219
|
+
# whitespace but preserve the original whitespace by splitting
|
|
220
|
+
# with a regex that captures it.
|
|
221
|
+
tokens = re.split(r"(\s+)", result)
|
|
222
|
+
for i, tok in enumerate(tokens):
|
|
223
|
+
# Strip surrounding punctuation for lookup, but preserve
|
|
224
|
+
# it in the output.
|
|
225
|
+
m = re.match(r"^([^\w]*)(.*?)([^\w]*)$", tok)
|
|
226
|
+
if m and m.group(2):
|
|
227
|
+
pre, word, post = m.group(1), m.group(2), m.group(3)
|
|
228
|
+
corrected = self.spelling.get(word.lower(), word)
|
|
229
|
+
# Preserve the capitalization pattern of the original
|
|
230
|
+
# token (so "Recieve" → "Receive", "RECIEVE" → "RECEIVE").
|
|
231
|
+
corrected = _preserve_case(word, corrected)
|
|
232
|
+
tokens[i] = pre + corrected + post
|
|
233
|
+
return "".join(tokens)
|
|
234
|
+
|
|
235
|
+
def _abbrev_repl(self, m: re.Match) -> str:
|
|
236
|
+
"""regex.sub callback for abbreviation expansion."""
|
|
237
|
+
key = m.group(0).lower()
|
|
238
|
+
expansion = self.abbreviations.get(key)
|
|
239
|
+
if expansion is None:
|
|
240
|
+
return m.group(0)
|
|
241
|
+
# Preserve leading capital if the original was capitalized
|
|
242
|
+
original = m.group(0)
|
|
243
|
+
if original and original[0].isupper():
|
|
244
|
+
# Capitalize the first letter of the expansion
|
|
245
|
+
return expansion[0].upper() + expansion[1:]
|
|
246
|
+
return expansion
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _preserve_case(original: str, replacement: str) -> str:
|
|
250
|
+
"""Make ``replacement`` match the capitalization pattern of ``original``.
|
|
251
|
+
|
|
252
|
+
* all-caps original → all-caps replacement
|
|
253
|
+
* title-case original → title-case replacement
|
|
254
|
+
* otherwise → replacement as-is (already lowercase from the dict)
|
|
255
|
+
"""
|
|
256
|
+
if not original or not replacement:
|
|
257
|
+
return replacement
|
|
258
|
+
if original.isupper():
|
|
259
|
+
return replacement.upper()
|
|
260
|
+
if original[0].isupper():
|
|
261
|
+
return replacement[0].upper() + replacement[1:]
|
|
262
|
+
return replacement
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
# ---------------------------------------------------------------------------
|
|
266
|
+
# Module-level singleton
|
|
267
|
+
# ---------------------------------------------------------------------------
|
|
268
|
+
_default_fst: QueryFST | None = None
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def default_fst() -> QueryFST:
|
|
272
|
+
global _default_fst
|
|
273
|
+
if _default_fst is None:
|
|
274
|
+
_default_fst = QueryFST()
|
|
275
|
+
return _default_fst
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def normalize(query: str) -> str:
|
|
279
|
+
"""Module-level convenience wrapper around the default FST."""
|
|
280
|
+
return default_fst().normalize(query)
|