ahocorasick-ner 0.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,13 @@
1
+ Metadata-Version: 2.1
2
+ Name: ahocorasick-ner
3
+ Version: 0.0.1
4
+ Summary: A fast, dictionary-based Named Entity Recognition system using the Aho-Corasick algorithm.
5
+ Home-page: https://github.com/TigreGotico/ahocorasick-ner
6
+ Author: JarbasAI
7
+ Author-email: jarbasai@mailfence.com
8
+ Classifier: Programming Language :: Python :: 3
9
+ Classifier: License :: OSI Approved :: MIT License
10
+ Classifier: Operating System :: OS Independent
11
+ Classifier: Topic :: Text Processing :: Linguistic
12
+ Classifier: Intended Audience :: Developers
13
+ Requires-Python: >=3.9
@@ -0,0 +1,106 @@
1
+ # AhocorasickNER
2
+
3
+ A fast and simple Named Entity Recognition (NER) tool based on the Aho-Corasick algorithm. This package is ideal for rule-based entity extraction using pre-defined vocabularies, especially when speed and scalability matter.
4
+
5
+ ---
6
+
7
+ ## ✨ Features
8
+
9
+ - ✅ Ultra-fast multi-pattern string matching using [Aho-Corasick](https://en.wikipedia.org/wiki/Aho%E2%80%93Corasick_algorithm)
10
+ - ✅ Word-boundary-aware matching
11
+ - ✅ Case-sensitive or case-insensitive modes
12
+ - ✅ Minimal dependencies
13
+ - ✅ Designed for integration with Hugging Face Datasets and similar sources
14
+
15
+ ---
16
+
17
+ ## 🧠 Theoretical Background
18
+
19
+ The Aho-Corasick algorithm, developed by Alfred V. Aho and Margaret J. Corasick in 1975, constructs a **finite state machine** from a set of keywords to allow **simultaneous pattern matching** in linear time. It's similar to a Trie, but extended with failure transitions that allow it to efficiently handle mismatches and overlapping substrings.
20
+
21
+ This approach is **ideal for dictionary-based NER systems**, where:
22
+ - You have a fixed list of entities (e.g., names, locations, products)
23
+ - You want to search for **many patterns in a single pass**
24
+ - Speed and low memory usage are critical
25
+
26
+ Unlike statistical or neural NER systems, this approach doesn't require training and is fully deterministic. It is particularly useful when:
27
+ - You want **consistent results**
28
+ - You are working with **domain-specific vocabularies**
29
+ - You need to process **large corpora quickly**
30
+
31
+ ---
32
+
33
+ ## 🚀 Installation
34
+
35
+ ```bash
36
+ pip install ahocorasick-ner
37
+ ```
38
+
39
+ ---
40
+
41
+ ## 🛠️ Usage
42
+
43
+ ```python
44
+ from ahocorasick_ner import AhocorasickNER
45
+ from datasets import load_dataset
46
+
47
+ EncyclopediaMetallvm = AhocorasickNER()
48
+
49
+ dataset_name = "Jarbas/metal-archives-tracks"
50
+ dataset = load_dataset(dataset_name)["train"]
51
+ for entry in dataset:
52
+ EncyclopediaMetallvm.add_word("artist_name", entry["band_name"])
53
+ if entry.get("track_name"):
54
+ EncyclopediaMetallvm.add_word("track_name", entry["track_name"])
55
+ if entry.get("album_name"):
56
+ EncyclopediaMetallvm.add_word("album_name", entry["album_name"])
57
+ EncyclopediaMetallvm.add_word("album_type", entry["album_type"])
58
+
59
+ dataset_name = "Jarbas/metal-archives-bands"
60
+ dataset = load_dataset(dataset_name)["train"]
61
+ for entry in dataset:
62
+ EncyclopediaMetallvm.add_word("artist_name", entry["name"])
63
+ if entry.get("genre"):
64
+ EncyclopediaMetallvm.add_word("music_genre", entry["genre"])
65
+ if entry.get("label"):
66
+ EncyclopediaMetallvm.add_word("record_label", entry["label"])
67
+ if entry.get("country"):
68
+ EncyclopediaMetallvm.add_word("country", entry["country"])
69
+
70
+ for entity in EncyclopediaMetallvm.tag("I fucking love black metal from Norway"):
71
+ print(entity)
72
+ ```
73
+
74
+ ### Output:
75
+ ```python
76
+ {'start': 15, 'end': 25, 'word': 'black metal', 'label': 'genre'}
77
+ {'start': 32, 'end': 37, 'word': 'Norway', 'label': 'country'}
78
+ ```
79
+
80
+ ---
81
+
82
+ ## 🧪 Benchmarks
83
+
84
+ With 100k+ known phrases, this tool can tag documents in milliseconds thanks to the Aho-Corasick FSM structure. It scales gracefully with both the number of patterns and the size of input text.
85
+
86
+ ---
87
+
88
+ ## 🧩 Limitations
89
+
90
+ - Does not handle nested or overlapping entities well (greedy, longest match wins)
91
+ - No fuzzy matching (e.g., typos or misspellings won't match)
92
+ - Requires all entities to be known beforehand
93
+
94
+
95
+ ---
96
+
97
+ ## 📄 License
98
+
99
+ MIT — free for commercial and non-commercial use.
100
+
101
+ ---
102
+
103
+ ## 🙏 Acknowledgements
104
+
105
+ - [pyahocorasick](https://github.com/WojciechMula/pyahocorasick) — The underlying C-based Aho-Corasick implementation
106
+ - [Hugging Face Datasets](https://huggingface.co/docs/datasets) — For loading domain-specific corpora
@@ -0,0 +1,99 @@
1
+ import ahocorasick
2
+ import pickle
3
+ import re
4
+ from typing import Dict, Iterable, List, Tuple, Set
5
+
6
+
7
+ class AhocorasickNER:
8
+ """
9
+ A simple Named Entity Recognition system using the Aho-Corasick algorithm.
10
+ Supports matching pre-defined entities in a given string with word boundary filtering.
11
+ """
12
+
13
+ def __init__(self, case_sensitive: bool = False):
14
+ """
15
+ Initialize the NER system.
16
+
17
+ Args:
18
+ case_sensitive (bool): Whether matching should be case-sensitive. Defaults to False.
19
+ """
20
+ self.automaton = ahocorasick.Automaton()
21
+ self.case_sensitive = case_sensitive
22
+ self._fitted = False
23
+
24
+ def save(self, path: str):
25
+ self.automaton.save(path, pickle.dumps)
26
+
27
+ def load(self, path: str):
28
+ self.automaton = ahocorasick.load(path, pickle.loads)
29
+
30
+ def add_word(self, label: str, example: str) -> None:
31
+ """
32
+ Add a labeled word or phrase to the automaton.
33
+
34
+ Args:
35
+ label (str): The label to associate with the word (e.g., 'artist_name').
36
+ example (str): The word or phrase to match.
37
+ """
38
+ key = example if self.case_sensitive else example.lower()
39
+ self.automaton.add_word(key, (label, key))
40
+ self._fitted = False
41
+
42
+ def fit(self) -> None:
43
+ """
44
+ Finalize the automaton. This must be called after all words are added.
45
+ """
46
+ if not self._fitted:
47
+ self.automaton.make_automaton()
48
+ self._fitted = True
49
+
50
+ def tag(self, haystack: str, min_word_len: int = 5) -> Iterable[Dict[str, str]]:
51
+ """
52
+ Search for labeled entities in the given string.
53
+
54
+ Args:
55
+ haystack (str): The input string to search.
56
+ min_word_len (int): Minimum word length to consider a match. Defaults to 5.
57
+
58
+ Yields:
59
+ Dict[str, str]: A dictionary with keys 'start', 'end', 'word', and 'label'.
60
+ """
61
+ if not self._fitted:
62
+ self.fit()
63
+
64
+ processed_haystack = haystack if self.case_sensitive else haystack.lower()
65
+ matches: List[Tuple[int, int, str, str]] = []
66
+
67
+ for idx, (label, word) in self.automaton.iter(processed_haystack):
68
+ if len(word) < min_word_len:
69
+ continue
70
+
71
+ start = idx - len(word) + 1
72
+ end = idx
73
+
74
+ # Respect word boundaries
75
+ before = processed_haystack[start - 1] if start > 0 else ' '
76
+ after = processed_haystack[end + 1] if end + 1 < len(processed_haystack) else ' '
77
+ if re.match(r'\w', before) or re.match(r'\w', after):
78
+ continue # skip partial word matches
79
+
80
+ matches.append((start, end, word, label))
81
+
82
+ # Sort by descending length, then by start position
83
+ matches.sort(key=lambda x: (-(x[1] - x[0] + 1), x[0]))
84
+
85
+ selected: List[Tuple[int, int, str, str]] = []
86
+ used: Set[int] = set()
87
+
88
+ for start, end, word, label in matches:
89
+ if all(i not in used for i in range(start, end + 1)):
90
+ selected.append((start, end, word, label))
91
+ used.update(range(start, end + 1))
92
+
93
+ for start, end, word, label in sorted(selected, key=lambda x: x[0]):
94
+ yield {
95
+ "start": start,
96
+ "end": end,
97
+ "word": haystack[start:end + 1],
98
+ "label": label
99
+ }
@@ -0,0 +1,180 @@
1
+ import ahocorasick
2
+ import os
3
+ from ahocorasick_ner import AhocorasickNER
4
+
5
+ try:
6
+ from datasets import load_dataset
7
+ except ImportError as e:
8
+ # only used in demo classes, not a hard requirement
9
+ def load_dataset(*args, **kwargs):
10
+ raise e
11
+
12
+
13
+ class EncyclopediaMetallvmNER(AhocorasickNER):
14
+ def __init__(self, path: str | None = None, case_sensitive: bool = False):
15
+ super().__init__(case_sensitive)
16
+ if path and os.path.exists(path):
17
+ self.load(path)
18
+ else:
19
+ self.train(path)
20
+
21
+ def train(self, path: str | None = None):
22
+ self.load_huggingface()
23
+ if path:
24
+ self.save(path)
25
+
26
+ def load_huggingface(self):
27
+ dataset_name = "Jarbas/metal-archives-tracks"
28
+ dataset = load_dataset(dataset_name)["train"]
29
+ for entry in dataset:
30
+ self.add_word("artist_name", entry["band_name"])
31
+ if entry.get("track_name"):
32
+ self.add_word("track_name", entry["track_name"])
33
+ if entry.get("album_name"):
34
+ self.add_word("album_name", entry["album_name"])
35
+ self.add_word("album_type", entry["album_type"])
36
+
37
+ dataset_name = "Jarbas/metal-archives-bands"
38
+ dataset = load_dataset(dataset_name)["train"]
39
+ for entry in dataset:
40
+ self.add_word("artist_name", entry["name"])
41
+ if entry.get("genre"):
42
+ self.add_word("music_genre", entry["genre"])
43
+ if entry.get("label"):
44
+ self.add_word("record_label", entry["label"])
45
+
46
+
47
+ class MusicNER(AhocorasickNER):
48
+ def __init__(self, path: str | None = None,
49
+ case_sensitive: bool = False):
50
+ super().__init__(case_sensitive)
51
+ if path and os.path.exists(path):
52
+ self.load(path)
53
+ else:
54
+ self.train(path)
55
+
56
+ def train(self, path: str | None = None):
57
+ self.load_huggingface()
58
+ if path:
59
+ self.save(path)
60
+
61
+ def load_huggingface(self):
62
+ self.load_metallvm()
63
+ self.load_jazz()
64
+ self.load_prog()
65
+ self.load_classical()
66
+ self.load_trance()
67
+
68
+ def load_trance(self):
69
+ dataset_name = "Jarbas/trance_tracks"
70
+ dataset = load_dataset(dataset_name)["train"]
71
+ for entry in dataset:
72
+ if entry.get("ARTIST(S)"):
73
+ self.add_word("artist_name", entry["ARTIST(S)"])
74
+ if entry.get("TRACK"):
75
+ self.add_word("track_name", entry["TRACK"])
76
+ if entry.get("STYLE"):
77
+ self.add_word("music_genre", entry["STYLE"])
78
+
79
+ def load_classical(self):
80
+ dataset_name = "Jarbas/classic-composers"
81
+ dataset = load_dataset(dataset_name)["train"]
82
+ for entry in dataset:
83
+ if entry.get("name"):
84
+ self.add_word("artist_name", entry["name"])
85
+
86
+ def load_prog(self):
87
+ dataset_name = "Jarbas/prog-archives"
88
+ dataset = load_dataset(dataset_name)["train"]
89
+ for entry in dataset:
90
+ if entry.get("artist"):
91
+ self.add_word("artist_name", entry["artist"])
92
+ if entry.get("genre"):
93
+ self.add_word("music_genre", entry["genre"])
94
+
95
+ def load_jazz(self):
96
+ dataset_name = "Jarbas/jazz-music-archives"
97
+ dataset = load_dataset(dataset_name)["train"]
98
+ for entry in dataset:
99
+ if entry.get("artist"):
100
+ self.add_word("artist_name", entry["artist"])
101
+ if entry.get("genre"):
102
+ self.add_word("music_genre", entry["genre"])
103
+
104
+ def load_metallvm(self):
105
+ dataset_name = "Jarbas/metal-archives-tracks"
106
+ dataset = load_dataset(dataset_name)["train"]
107
+ for entry in dataset:
108
+ self.add_word("artist_name", entry["band_name"])
109
+ if entry.get("track_name"):
110
+ self.add_word("track_name", entry["track_name"])
111
+ if entry.get("album_name"):
112
+ self.add_word("album_name", entry["album_name"])
113
+ self.add_word("album_type", entry["album_type"])
114
+
115
+ dataset_name = "Jarbas/metal-archives-bands"
116
+ dataset = load_dataset(dataset_name)["train"]
117
+ for entry in dataset:
118
+ self.add_word("artist_name", entry["name"])
119
+ if entry.get("genre"):
120
+ self.add_word("music_genre", entry["genre"])
121
+ if entry.get("label"):
122
+ self.add_word("record_label", entry["label"])
123
+
124
+
125
+ class ImdbNER(AhocorasickNER):
126
+ def __init__(self, path: str | None = None, case_sensitive: bool = False):
127
+ super().__init__(case_sensitive)
128
+ if path and os.path.exists(path):
129
+ self.load(path)
130
+ else:
131
+ self.train(path)
132
+
133
+ def train(self, path: str | None = None):
134
+ self.load_huggingface()
135
+ if path:
136
+ self.save(path)
137
+
138
+ def load_huggingface(self):
139
+ dataset_name = "Jarbas/movie_actors"
140
+ dataset = load_dataset(dataset_name)["train"]
141
+ for entry in dataset:
142
+ if entry.get("name"):
143
+ self.add_word("movie_actor", entry["name"])
144
+
145
+ dataset_name = "Jarbas/movie_directors"
146
+ dataset = load_dataset(dataset_name)["train"]
147
+ for entry in dataset:
148
+ if entry.get("name"):
149
+ self.add_word("movie_director", entry["name"])
150
+
151
+ dataset_name = "Jarbas/movie_producers"
152
+ dataset = load_dataset(dataset_name)["train"]
153
+ for entry in dataset:
154
+ if entry.get("name"):
155
+ self.add_word("movie_producer", entry["name"])
156
+
157
+ dataset_name = "Jarbas/movie_writers"
158
+ dataset = load_dataset(dataset_name)["train"]
159
+ for entry in dataset:
160
+ if entry.get("name"):
161
+ self.add_word("movie_writer", entry["name"])
162
+
163
+ dataset_name = "Jarbas/movie_composers"
164
+ dataset = load_dataset(dataset_name)["train"]
165
+ for entry in dataset:
166
+ if entry.get("name"):
167
+ self.add_word("movie_composer", entry["name"])
168
+
169
+
170
+ if __name__ == "__main__":
171
+ import time
172
+
173
+ # e = MusicNER("media_net.ahocorasick")
174
+ # e = ImdbNER("imdb.ahocorasick")
175
+ e = EncyclopediaMetallvmNER("metallvm.ahocorasick")
176
+
177
+ s = time.monotonic()
178
+ for entity in e.tag("I fucking love black metal"):
179
+ print(entity)
180
+ print(time.monotonic() - s)
@@ -0,0 +1,6 @@
1
+ # START_VERSION_BLOCK
2
+ VERSION_MAJOR = 0
3
+ VERSION_MINOR = 0
4
+ VERSION_BUILD = 1
5
+ VERSION_ALPHA = 0
6
+ # END_VERSION_BLOCK
@@ -0,0 +1,13 @@
1
+ Metadata-Version: 2.1
2
+ Name: ahocorasick-ner
3
+ Version: 0.0.1
4
+ Summary: A fast, dictionary-based Named Entity Recognition system using the Aho-Corasick algorithm.
5
+ Home-page: https://github.com/TigreGotico/ahocorasick-ner
6
+ Author: JarbasAI
7
+ Author-email: jarbasai@mailfence.com
8
+ Classifier: Programming Language :: Python :: 3
9
+ Classifier: License :: OSI Approved :: MIT License
10
+ Classifier: Operating System :: OS Independent
11
+ Classifier: Topic :: Text Processing :: Linguistic
12
+ Classifier: Intended Audience :: Developers
13
+ Requires-Python: >=3.9
@@ -0,0 +1,10 @@
1
+ README.md
2
+ setup.py
3
+ ahocorasick_ner/__init__.py
4
+ ahocorasick_ner/datasets.py
5
+ ahocorasick_ner/version.py
6
+ ahocorasick_ner.egg-info/PKG-INFO
7
+ ahocorasick_ner.egg-info/SOURCES.txt
8
+ ahocorasick_ner.egg-info/dependency_links.txt
9
+ ahocorasick_ner.egg-info/requires.txt
10
+ ahocorasick_ner.egg-info/top_level.txt
@@ -0,0 +1 @@
1
+ pyahocorasick
@@ -0,0 +1 @@
1
+ ahocorasick_ner
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,60 @@
1
+ import os.path
2
+
3
+ import os
4
+ from setuptools import setup
5
+
6
+ BASEDIR = os.path.abspath(os.path.dirname(__file__))
7
+
8
+
9
+ def get_version():
10
+ """ Find the version"""
11
+ version_file = os.path.join(BASEDIR, 'ahocorasick_ner', 'version.py')
12
+ major, minor, build, alpha = (None, None, None, None)
13
+ with open(version_file) as f:
14
+ for line in f:
15
+ if 'VERSION_MAJOR' in line:
16
+ major = line.split('=')[1].strip()
17
+ elif 'VERSION_MINOR' in line:
18
+ minor = line.split('=')[1].strip()
19
+ elif 'VERSION_BUILD' in line:
20
+ build = line.split('=')[1].strip()
21
+ elif 'VERSION_ALPHA' in line:
22
+ alpha = line.split('=')[1].strip()
23
+
24
+ if ((major and minor and build and alpha) or
25
+ '# END_VERSION_BLOCK' in line):
26
+ break
27
+ version = f"{major}.{minor}.{build}"
28
+ if int(alpha):
29
+ version += f"a{alpha}"
30
+ return version
31
+
32
+
33
+ def required(requirements_file):
34
+ """ Read requirements file and remove comments and empty lines. """
35
+ with open(os.path.join(BASEDIR, requirements_file), 'r') as f:
36
+ requirements = f.read().splitlines()
37
+ return [pkg for pkg in requirements
38
+ if pkg.strip() and not pkg.startswith("#")]
39
+
40
+
41
+ setup(
42
+ name="ahocorasick-ner",
43
+ version=get_version(),
44
+ modules=["ahocorasick_ner"],
45
+ install_requires=required("requirements.txt"),
46
+ author="JarbasAI",
47
+ author_email="jarbasai@mailfence.com",
48
+ description="A fast, dictionary-based Named Entity Recognition system using the Aho-Corasick algorithm.",
49
+ # long_description=open("README.md").read(),
50
+ # long_description_content_type="text/markdown",
51
+ url="https://github.com/TigreGotico/ahocorasick-ner",
52
+ classifiers=[
53
+ "Programming Language :: Python :: 3",
54
+ "License :: OSI Approved :: MIT License",
55
+ "Operating System :: OS Independent",
56
+ "Topic :: Text Processing :: Linguistic",
57
+ "Intended Audience :: Developers",
58
+ ],
59
+ python_requires=">=3.9",
60
+ )