ahocorasick-ner 0.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ahocorasick-ner-0.0.1/PKG-INFO +13 -0
- ahocorasick-ner-0.0.1/README.md +106 -0
- ahocorasick-ner-0.0.1/ahocorasick_ner/__init__.py +99 -0
- ahocorasick-ner-0.0.1/ahocorasick_ner/datasets.py +180 -0
- ahocorasick-ner-0.0.1/ahocorasick_ner/version.py +6 -0
- ahocorasick-ner-0.0.1/ahocorasick_ner.egg-info/PKG-INFO +13 -0
- ahocorasick-ner-0.0.1/ahocorasick_ner.egg-info/SOURCES.txt +10 -0
- ahocorasick-ner-0.0.1/ahocorasick_ner.egg-info/dependency_links.txt +1 -0
- ahocorasick-ner-0.0.1/ahocorasick_ner.egg-info/requires.txt +1 -0
- ahocorasick-ner-0.0.1/ahocorasick_ner.egg-info/top_level.txt +1 -0
- ahocorasick-ner-0.0.1/setup.cfg +4 -0
- ahocorasick-ner-0.0.1/setup.py +60 -0
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: ahocorasick-ner
|
|
3
|
+
Version: 0.0.1
|
|
4
|
+
Summary: A fast, dictionary-based Named Entity Recognition system using the Aho-Corasick algorithm.
|
|
5
|
+
Home-page: https://github.com/TigreGotico/ahocorasick-ner
|
|
6
|
+
Author: JarbasAI
|
|
7
|
+
Author-email: jarbasai@mailfence.com
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Operating System :: OS Independent
|
|
11
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Requires-Python: >=3.9
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
# AhocorasickNER
|
|
2
|
+
|
|
3
|
+
A fast and simple Named Entity Recognition (NER) tool based on the Aho-Corasick algorithm. This package is ideal for rule-based entity extraction using pre-defined vocabularies, especially when speed and scalability matter.
|
|
4
|
+
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
## ✨ Features
|
|
8
|
+
|
|
9
|
+
- ✅ Ultra-fast multi-pattern string matching using [Aho-Corasick](https://en.wikipedia.org/wiki/Aho%E2%80%93Corasick_algorithm)
|
|
10
|
+
- ✅ Word-boundary-aware matching
|
|
11
|
+
- ✅ Case-sensitive or case-insensitive modes
|
|
12
|
+
- ✅ Minimal dependencies
|
|
13
|
+
- ✅ Designed for integration with Hugging Face Datasets and similar sources
|
|
14
|
+
|
|
15
|
+
---
|
|
16
|
+
|
|
17
|
+
## 🧠 Theoretical Background
|
|
18
|
+
|
|
19
|
+
The Aho-Corasick algorithm, developed by Alfred V. Aho and Margaret J. Corasick in 1975, constructs a **finite state machine** from a set of keywords to allow **simultaneous pattern matching** in linear time. It's similar to a Trie, but extended with failure transitions that allow it to efficiently handle mismatches and overlapping substrings.
|
|
20
|
+
|
|
21
|
+
This approach is **ideal for dictionary-based NER systems**, where:
|
|
22
|
+
- You have a fixed list of entities (e.g., names, locations, products)
|
|
23
|
+
- You want to search for **many patterns in a single pass**
|
|
24
|
+
- Speed and low memory usage are critical
|
|
25
|
+
|
|
26
|
+
Unlike statistical or neural NER systems, this approach doesn't require training and is fully deterministic. It is particularly useful when:
|
|
27
|
+
- You want **consistent results**
|
|
28
|
+
- You are working with **domain-specific vocabularies**
|
|
29
|
+
- You need to process **large corpora quickly**
|
|
30
|
+
|
|
31
|
+
---
|
|
32
|
+
|
|
33
|
+
## 🚀 Installation
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
pip install ahocorasick-ner
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
---
|
|
40
|
+
|
|
41
|
+
## 🛠️ Usage
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from ahocorasick_ner import AhocorasickNER
|
|
45
|
+
from datasets import load_dataset
|
|
46
|
+
|
|
47
|
+
EncyclopediaMetallvm = AhocorasickNER()
|
|
48
|
+
|
|
49
|
+
dataset_name = "Jarbas/metal-archives-tracks"
|
|
50
|
+
dataset = load_dataset(dataset_name)["train"]
|
|
51
|
+
for entry in dataset:
|
|
52
|
+
EncyclopediaMetallvm.add_word("artist_name", entry["band_name"])
|
|
53
|
+
if entry.get("track_name"):
|
|
54
|
+
EncyclopediaMetallvm.add_word("track_name", entry["track_name"])
|
|
55
|
+
if entry.get("album_name"):
|
|
56
|
+
EncyclopediaMetallvm.add_word("album_name", entry["album_name"])
|
|
57
|
+
EncyclopediaMetallvm.add_word("album_type", entry["album_type"])
|
|
58
|
+
|
|
59
|
+
dataset_name = "Jarbas/metal-archives-bands"
|
|
60
|
+
dataset = load_dataset(dataset_name)["train"]
|
|
61
|
+
for entry in dataset:
|
|
62
|
+
EncyclopediaMetallvm.add_word("artist_name", entry["name"])
|
|
63
|
+
if entry.get("genre"):
|
|
64
|
+
EncyclopediaMetallvm.add_word("music_genre", entry["genre"])
|
|
65
|
+
if entry.get("label"):
|
|
66
|
+
EncyclopediaMetallvm.add_word("record_label", entry["label"])
|
|
67
|
+
if entry.get("country"):
|
|
68
|
+
EncyclopediaMetallvm.add_word("country", entry["country"])
|
|
69
|
+
|
|
70
|
+
for entity in EncyclopediaMetallvm.tag("I fucking love black metal from Norway"):
|
|
71
|
+
print(entity)
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
### Output:
|
|
75
|
+
```python
|
|
76
|
+
{'start': 15, 'end': 25, 'word': 'black metal', 'label': 'genre'}
|
|
77
|
+
{'start': 32, 'end': 37, 'word': 'Norway', 'label': 'country'}
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
---
|
|
81
|
+
|
|
82
|
+
## 🧪 Benchmarks
|
|
83
|
+
|
|
84
|
+
With 100k+ known phrases, this tool can tag documents in milliseconds thanks to the Aho-Corasick FSM structure. It scales gracefully with both the number of patterns and the size of input text.
|
|
85
|
+
|
|
86
|
+
---
|
|
87
|
+
|
|
88
|
+
## 🧩 Limitations
|
|
89
|
+
|
|
90
|
+
- Does not handle nested or overlapping entities well (greedy, longest match wins)
|
|
91
|
+
- No fuzzy matching (e.g., typos or misspellings won't match)
|
|
92
|
+
- Requires all entities to be known beforehand
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
---
|
|
96
|
+
|
|
97
|
+
## 📄 License
|
|
98
|
+
|
|
99
|
+
MIT — free for commercial and non-commercial use.
|
|
100
|
+
|
|
101
|
+
---
|
|
102
|
+
|
|
103
|
+
## 🙏 Acknowledgements
|
|
104
|
+
|
|
105
|
+
- [pyahocorasick](https://github.com/WojciechMula/pyahocorasick) — The underlying C-based Aho-Corasick implementation
|
|
106
|
+
- [Hugging Face Datasets](https://huggingface.co/docs/datasets) — For loading domain-specific corpora
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
import ahocorasick
|
|
2
|
+
import pickle
|
|
3
|
+
import re
|
|
4
|
+
from typing import Dict, Iterable, List, Tuple, Set
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class AhocorasickNER:
|
|
8
|
+
"""
|
|
9
|
+
A simple Named Entity Recognition system using the Aho-Corasick algorithm.
|
|
10
|
+
Supports matching pre-defined entities in a given string with word boundary filtering.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
def __init__(self, case_sensitive: bool = False):
|
|
14
|
+
"""
|
|
15
|
+
Initialize the NER system.
|
|
16
|
+
|
|
17
|
+
Args:
|
|
18
|
+
case_sensitive (bool): Whether matching should be case-sensitive. Defaults to False.
|
|
19
|
+
"""
|
|
20
|
+
self.automaton = ahocorasick.Automaton()
|
|
21
|
+
self.case_sensitive = case_sensitive
|
|
22
|
+
self._fitted = False
|
|
23
|
+
|
|
24
|
+
def save(self, path: str):
|
|
25
|
+
self.automaton.save(path, pickle.dumps)
|
|
26
|
+
|
|
27
|
+
def load(self, path: str):
|
|
28
|
+
self.automaton = ahocorasick.load(path, pickle.loads)
|
|
29
|
+
|
|
30
|
+
def add_word(self, label: str, example: str) -> None:
|
|
31
|
+
"""
|
|
32
|
+
Add a labeled word or phrase to the automaton.
|
|
33
|
+
|
|
34
|
+
Args:
|
|
35
|
+
label (str): The label to associate with the word (e.g., 'artist_name').
|
|
36
|
+
example (str): The word or phrase to match.
|
|
37
|
+
"""
|
|
38
|
+
key = example if self.case_sensitive else example.lower()
|
|
39
|
+
self.automaton.add_word(key, (label, key))
|
|
40
|
+
self._fitted = False
|
|
41
|
+
|
|
42
|
+
def fit(self) -> None:
|
|
43
|
+
"""
|
|
44
|
+
Finalize the automaton. This must be called after all words are added.
|
|
45
|
+
"""
|
|
46
|
+
if not self._fitted:
|
|
47
|
+
self.automaton.make_automaton()
|
|
48
|
+
self._fitted = True
|
|
49
|
+
|
|
50
|
+
def tag(self, haystack: str, min_word_len: int = 5) -> Iterable[Dict[str, str]]:
|
|
51
|
+
"""
|
|
52
|
+
Search for labeled entities in the given string.
|
|
53
|
+
|
|
54
|
+
Args:
|
|
55
|
+
haystack (str): The input string to search.
|
|
56
|
+
min_word_len (int): Minimum word length to consider a match. Defaults to 5.
|
|
57
|
+
|
|
58
|
+
Yields:
|
|
59
|
+
Dict[str, str]: A dictionary with keys 'start', 'end', 'word', and 'label'.
|
|
60
|
+
"""
|
|
61
|
+
if not self._fitted:
|
|
62
|
+
self.fit()
|
|
63
|
+
|
|
64
|
+
processed_haystack = haystack if self.case_sensitive else haystack.lower()
|
|
65
|
+
matches: List[Tuple[int, int, str, str]] = []
|
|
66
|
+
|
|
67
|
+
for idx, (label, word) in self.automaton.iter(processed_haystack):
|
|
68
|
+
if len(word) < min_word_len:
|
|
69
|
+
continue
|
|
70
|
+
|
|
71
|
+
start = idx - len(word) + 1
|
|
72
|
+
end = idx
|
|
73
|
+
|
|
74
|
+
# Respect word boundaries
|
|
75
|
+
before = processed_haystack[start - 1] if start > 0 else ' '
|
|
76
|
+
after = processed_haystack[end + 1] if end + 1 < len(processed_haystack) else ' '
|
|
77
|
+
if re.match(r'\w', before) or re.match(r'\w', after):
|
|
78
|
+
continue # skip partial word matches
|
|
79
|
+
|
|
80
|
+
matches.append((start, end, word, label))
|
|
81
|
+
|
|
82
|
+
# Sort by descending length, then by start position
|
|
83
|
+
matches.sort(key=lambda x: (-(x[1] - x[0] + 1), x[0]))
|
|
84
|
+
|
|
85
|
+
selected: List[Tuple[int, int, str, str]] = []
|
|
86
|
+
used: Set[int] = set()
|
|
87
|
+
|
|
88
|
+
for start, end, word, label in matches:
|
|
89
|
+
if all(i not in used for i in range(start, end + 1)):
|
|
90
|
+
selected.append((start, end, word, label))
|
|
91
|
+
used.update(range(start, end + 1))
|
|
92
|
+
|
|
93
|
+
for start, end, word, label in sorted(selected, key=lambda x: x[0]):
|
|
94
|
+
yield {
|
|
95
|
+
"start": start,
|
|
96
|
+
"end": end,
|
|
97
|
+
"word": haystack[start:end + 1],
|
|
98
|
+
"label": label
|
|
99
|
+
}
|
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
import ahocorasick
|
|
2
|
+
import os
|
|
3
|
+
from ahocorasick_ner import AhocorasickNER
|
|
4
|
+
|
|
5
|
+
try:
|
|
6
|
+
from datasets import load_dataset
|
|
7
|
+
except ImportError as e:
|
|
8
|
+
# only used in demo classes, not a hard requirement
|
|
9
|
+
def load_dataset(*args, **kwargs):
|
|
10
|
+
raise e
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class EncyclopediaMetallvmNER(AhocorasickNER):
|
|
14
|
+
def __init__(self, path: str | None = None, case_sensitive: bool = False):
|
|
15
|
+
super().__init__(case_sensitive)
|
|
16
|
+
if path and os.path.exists(path):
|
|
17
|
+
self.load(path)
|
|
18
|
+
else:
|
|
19
|
+
self.train(path)
|
|
20
|
+
|
|
21
|
+
def train(self, path: str | None = None):
|
|
22
|
+
self.load_huggingface()
|
|
23
|
+
if path:
|
|
24
|
+
self.save(path)
|
|
25
|
+
|
|
26
|
+
def load_huggingface(self):
|
|
27
|
+
dataset_name = "Jarbas/metal-archives-tracks"
|
|
28
|
+
dataset = load_dataset(dataset_name)["train"]
|
|
29
|
+
for entry in dataset:
|
|
30
|
+
self.add_word("artist_name", entry["band_name"])
|
|
31
|
+
if entry.get("track_name"):
|
|
32
|
+
self.add_word("track_name", entry["track_name"])
|
|
33
|
+
if entry.get("album_name"):
|
|
34
|
+
self.add_word("album_name", entry["album_name"])
|
|
35
|
+
self.add_word("album_type", entry["album_type"])
|
|
36
|
+
|
|
37
|
+
dataset_name = "Jarbas/metal-archives-bands"
|
|
38
|
+
dataset = load_dataset(dataset_name)["train"]
|
|
39
|
+
for entry in dataset:
|
|
40
|
+
self.add_word("artist_name", entry["name"])
|
|
41
|
+
if entry.get("genre"):
|
|
42
|
+
self.add_word("music_genre", entry["genre"])
|
|
43
|
+
if entry.get("label"):
|
|
44
|
+
self.add_word("record_label", entry["label"])
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class MusicNER(AhocorasickNER):
|
|
48
|
+
def __init__(self, path: str | None = None,
|
|
49
|
+
case_sensitive: bool = False):
|
|
50
|
+
super().__init__(case_sensitive)
|
|
51
|
+
if path and os.path.exists(path):
|
|
52
|
+
self.load(path)
|
|
53
|
+
else:
|
|
54
|
+
self.train(path)
|
|
55
|
+
|
|
56
|
+
def train(self, path: str | None = None):
|
|
57
|
+
self.load_huggingface()
|
|
58
|
+
if path:
|
|
59
|
+
self.save(path)
|
|
60
|
+
|
|
61
|
+
def load_huggingface(self):
|
|
62
|
+
self.load_metallvm()
|
|
63
|
+
self.load_jazz()
|
|
64
|
+
self.load_prog()
|
|
65
|
+
self.load_classical()
|
|
66
|
+
self.load_trance()
|
|
67
|
+
|
|
68
|
+
def load_trance(self):
|
|
69
|
+
dataset_name = "Jarbas/trance_tracks"
|
|
70
|
+
dataset = load_dataset(dataset_name)["train"]
|
|
71
|
+
for entry in dataset:
|
|
72
|
+
if entry.get("ARTIST(S)"):
|
|
73
|
+
self.add_word("artist_name", entry["ARTIST(S)"])
|
|
74
|
+
if entry.get("TRACK"):
|
|
75
|
+
self.add_word("track_name", entry["TRACK"])
|
|
76
|
+
if entry.get("STYLE"):
|
|
77
|
+
self.add_word("music_genre", entry["STYLE"])
|
|
78
|
+
|
|
79
|
+
def load_classical(self):
|
|
80
|
+
dataset_name = "Jarbas/classic-composers"
|
|
81
|
+
dataset = load_dataset(dataset_name)["train"]
|
|
82
|
+
for entry in dataset:
|
|
83
|
+
if entry.get("name"):
|
|
84
|
+
self.add_word("artist_name", entry["name"])
|
|
85
|
+
|
|
86
|
+
def load_prog(self):
|
|
87
|
+
dataset_name = "Jarbas/prog-archives"
|
|
88
|
+
dataset = load_dataset(dataset_name)["train"]
|
|
89
|
+
for entry in dataset:
|
|
90
|
+
if entry.get("artist"):
|
|
91
|
+
self.add_word("artist_name", entry["artist"])
|
|
92
|
+
if entry.get("genre"):
|
|
93
|
+
self.add_word("music_genre", entry["genre"])
|
|
94
|
+
|
|
95
|
+
def load_jazz(self):
|
|
96
|
+
dataset_name = "Jarbas/jazz-music-archives"
|
|
97
|
+
dataset = load_dataset(dataset_name)["train"]
|
|
98
|
+
for entry in dataset:
|
|
99
|
+
if entry.get("artist"):
|
|
100
|
+
self.add_word("artist_name", entry["artist"])
|
|
101
|
+
if entry.get("genre"):
|
|
102
|
+
self.add_word("music_genre", entry["genre"])
|
|
103
|
+
|
|
104
|
+
def load_metallvm(self):
|
|
105
|
+
dataset_name = "Jarbas/metal-archives-tracks"
|
|
106
|
+
dataset = load_dataset(dataset_name)["train"]
|
|
107
|
+
for entry in dataset:
|
|
108
|
+
self.add_word("artist_name", entry["band_name"])
|
|
109
|
+
if entry.get("track_name"):
|
|
110
|
+
self.add_word("track_name", entry["track_name"])
|
|
111
|
+
if entry.get("album_name"):
|
|
112
|
+
self.add_word("album_name", entry["album_name"])
|
|
113
|
+
self.add_word("album_type", entry["album_type"])
|
|
114
|
+
|
|
115
|
+
dataset_name = "Jarbas/metal-archives-bands"
|
|
116
|
+
dataset = load_dataset(dataset_name)["train"]
|
|
117
|
+
for entry in dataset:
|
|
118
|
+
self.add_word("artist_name", entry["name"])
|
|
119
|
+
if entry.get("genre"):
|
|
120
|
+
self.add_word("music_genre", entry["genre"])
|
|
121
|
+
if entry.get("label"):
|
|
122
|
+
self.add_word("record_label", entry["label"])
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
class ImdbNER(AhocorasickNER):
|
|
126
|
+
def __init__(self, path: str | None = None, case_sensitive: bool = False):
|
|
127
|
+
super().__init__(case_sensitive)
|
|
128
|
+
if path and os.path.exists(path):
|
|
129
|
+
self.load(path)
|
|
130
|
+
else:
|
|
131
|
+
self.train(path)
|
|
132
|
+
|
|
133
|
+
def train(self, path: str | None = None):
|
|
134
|
+
self.load_huggingface()
|
|
135
|
+
if path:
|
|
136
|
+
self.save(path)
|
|
137
|
+
|
|
138
|
+
def load_huggingface(self):
|
|
139
|
+
dataset_name = "Jarbas/movie_actors"
|
|
140
|
+
dataset = load_dataset(dataset_name)["train"]
|
|
141
|
+
for entry in dataset:
|
|
142
|
+
if entry.get("name"):
|
|
143
|
+
self.add_word("movie_actor", entry["name"])
|
|
144
|
+
|
|
145
|
+
dataset_name = "Jarbas/movie_directors"
|
|
146
|
+
dataset = load_dataset(dataset_name)["train"]
|
|
147
|
+
for entry in dataset:
|
|
148
|
+
if entry.get("name"):
|
|
149
|
+
self.add_word("movie_director", entry["name"])
|
|
150
|
+
|
|
151
|
+
dataset_name = "Jarbas/movie_producers"
|
|
152
|
+
dataset = load_dataset(dataset_name)["train"]
|
|
153
|
+
for entry in dataset:
|
|
154
|
+
if entry.get("name"):
|
|
155
|
+
self.add_word("movie_producer", entry["name"])
|
|
156
|
+
|
|
157
|
+
dataset_name = "Jarbas/movie_writers"
|
|
158
|
+
dataset = load_dataset(dataset_name)["train"]
|
|
159
|
+
for entry in dataset:
|
|
160
|
+
if entry.get("name"):
|
|
161
|
+
self.add_word("movie_writer", entry["name"])
|
|
162
|
+
|
|
163
|
+
dataset_name = "Jarbas/movie_composers"
|
|
164
|
+
dataset = load_dataset(dataset_name)["train"]
|
|
165
|
+
for entry in dataset:
|
|
166
|
+
if entry.get("name"):
|
|
167
|
+
self.add_word("movie_composer", entry["name"])
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
if __name__ == "__main__":
|
|
171
|
+
import time
|
|
172
|
+
|
|
173
|
+
# e = MusicNER("media_net.ahocorasick")
|
|
174
|
+
# e = ImdbNER("imdb.ahocorasick")
|
|
175
|
+
e = EncyclopediaMetallvmNER("metallvm.ahocorasick")
|
|
176
|
+
|
|
177
|
+
s = time.monotonic()
|
|
178
|
+
for entity in e.tag("I fucking love black metal"):
|
|
179
|
+
print(entity)
|
|
180
|
+
print(time.monotonic() - s)
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: ahocorasick-ner
|
|
3
|
+
Version: 0.0.1
|
|
4
|
+
Summary: A fast, dictionary-based Named Entity Recognition system using the Aho-Corasick algorithm.
|
|
5
|
+
Home-page: https://github.com/TigreGotico/ahocorasick-ner
|
|
6
|
+
Author: JarbasAI
|
|
7
|
+
Author-email: jarbasai@mailfence.com
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Operating System :: OS Independent
|
|
11
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Requires-Python: >=3.9
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
README.md
|
|
2
|
+
setup.py
|
|
3
|
+
ahocorasick_ner/__init__.py
|
|
4
|
+
ahocorasick_ner/datasets.py
|
|
5
|
+
ahocorasick_ner/version.py
|
|
6
|
+
ahocorasick_ner.egg-info/PKG-INFO
|
|
7
|
+
ahocorasick_ner.egg-info/SOURCES.txt
|
|
8
|
+
ahocorasick_ner.egg-info/dependency_links.txt
|
|
9
|
+
ahocorasick_ner.egg-info/requires.txt
|
|
10
|
+
ahocorasick_ner.egg-info/top_level.txt
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
pyahocorasick
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
ahocorasick_ner
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import os.path
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from setuptools import setup
|
|
5
|
+
|
|
6
|
+
BASEDIR = os.path.abspath(os.path.dirname(__file__))
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def get_version():
|
|
10
|
+
""" Find the version"""
|
|
11
|
+
version_file = os.path.join(BASEDIR, 'ahocorasick_ner', 'version.py')
|
|
12
|
+
major, minor, build, alpha = (None, None, None, None)
|
|
13
|
+
with open(version_file) as f:
|
|
14
|
+
for line in f:
|
|
15
|
+
if 'VERSION_MAJOR' in line:
|
|
16
|
+
major = line.split('=')[1].strip()
|
|
17
|
+
elif 'VERSION_MINOR' in line:
|
|
18
|
+
minor = line.split('=')[1].strip()
|
|
19
|
+
elif 'VERSION_BUILD' in line:
|
|
20
|
+
build = line.split('=')[1].strip()
|
|
21
|
+
elif 'VERSION_ALPHA' in line:
|
|
22
|
+
alpha = line.split('=')[1].strip()
|
|
23
|
+
|
|
24
|
+
if ((major and minor and build and alpha) or
|
|
25
|
+
'# END_VERSION_BLOCK' in line):
|
|
26
|
+
break
|
|
27
|
+
version = f"{major}.{minor}.{build}"
|
|
28
|
+
if int(alpha):
|
|
29
|
+
version += f"a{alpha}"
|
|
30
|
+
return version
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def required(requirements_file):
|
|
34
|
+
""" Read requirements file and remove comments and empty lines. """
|
|
35
|
+
with open(os.path.join(BASEDIR, requirements_file), 'r') as f:
|
|
36
|
+
requirements = f.read().splitlines()
|
|
37
|
+
return [pkg for pkg in requirements
|
|
38
|
+
if pkg.strip() and not pkg.startswith("#")]
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
setup(
|
|
42
|
+
name="ahocorasick-ner",
|
|
43
|
+
version=get_version(),
|
|
44
|
+
modules=["ahocorasick_ner"],
|
|
45
|
+
install_requires=required("requirements.txt"),
|
|
46
|
+
author="JarbasAI",
|
|
47
|
+
author_email="jarbasai@mailfence.com",
|
|
48
|
+
description="A fast, dictionary-based Named Entity Recognition system using the Aho-Corasick algorithm.",
|
|
49
|
+
# long_description=open("README.md").read(),
|
|
50
|
+
# long_description_content_type="text/markdown",
|
|
51
|
+
url="https://github.com/TigreGotico/ahocorasick-ner",
|
|
52
|
+
classifiers=[
|
|
53
|
+
"Programming Language :: Python :: 3",
|
|
54
|
+
"License :: OSI Approved :: MIT License",
|
|
55
|
+
"Operating System :: OS Independent",
|
|
56
|
+
"Topic :: Text Processing :: Linguistic",
|
|
57
|
+
"Intended Audience :: Developers",
|
|
58
|
+
],
|
|
59
|
+
python_requires=">=3.9",
|
|
60
|
+
)
|