lexgrep 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lexgrep/__init__.py +3 -0
- lexgrep/__main__.py +4 -0
- lexgrep/bench.py +113 -0
- lexgrep/cli.py +20 -0
- lexgrep/epub.py +118 -0
- lexgrep/inflect.py +1057 -0
- lexgrep/lem.py +51 -0
- lexgrep/lexgrep.py +2 -0
- lexgrep/sim.py +44 -0
- lexgrep-0.2.0.dist-info/METADATA +9 -0
- lexgrep-0.2.0.dist-info/RECORD +13 -0
- lexgrep-0.2.0.dist-info/WHEEL +4 -0
- lexgrep-0.2.0.dist-info/entry_points.txt +2 -0
lexgrep/__init__.py
ADDED
lexgrep/__main__.py
ADDED
lexgrep/bench.py
ADDED
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
from time import perf_counter
|
|
2
|
+
from io import StringIO
|
|
3
|
+
import re
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def related_words(word):
|
|
7
|
+
"""
|
|
8
|
+
Generate words related to a supplied word
|
|
9
|
+
|
|
10
|
+
This is intended to be an over estimate. not all returned result need to
|
|
11
|
+
be proper words.
|
|
12
|
+
|
|
13
|
+
>>> 'compute' in related_words("computer")
|
|
14
|
+
True
|
|
15
|
+
|
|
16
|
+
>>> 'computes' in related_words("computer")
|
|
17
|
+
True
|
|
18
|
+
|
|
19
|
+
>>> 'computes' in related_words("computing")
|
|
20
|
+
True
|
|
21
|
+
|
|
22
|
+
>>> 'computation' in related_words("computing")
|
|
23
|
+
True
|
|
24
|
+
|
|
25
|
+
>>> 'running' in related_words("run")
|
|
26
|
+
True
|
|
27
|
+
"""
|
|
28
|
+
word = word.strip().lower()
|
|
29
|
+
|
|
30
|
+
# Guess the verb behind agent nouns: computer -> compute.
|
|
31
|
+
for suffix in ["er", "or", "ing", "ers", "ors", "ed"]:
|
|
32
|
+
if word.endswith(suffix):
|
|
33
|
+
base = word[: -len(suffix)]
|
|
34
|
+
if base and not base.endswith(("a", "e", "i", "o", "u")):
|
|
35
|
+
base += "e"
|
|
36
|
+
break
|
|
37
|
+
else:
|
|
38
|
+
base = word
|
|
39
|
+
|
|
40
|
+
if base.endswith("e"):
|
|
41
|
+
# Separate the stem for common -ed and -ing forms.
|
|
42
|
+
stem = base[:-1]
|
|
43
|
+
elif re.search(r"[^aeiou][aeiou][^aeiou]$", base):
|
|
44
|
+
# Apply double consonant rule
|
|
45
|
+
stem = base + base[-1]
|
|
46
|
+
else:
|
|
47
|
+
stem = base
|
|
48
|
+
|
|
49
|
+
if stem.endswith("y") and len(stem) > 1:
|
|
50
|
+
plural = stem[:-1] + "ies"
|
|
51
|
+
else:
|
|
52
|
+
plural = base + "s"
|
|
53
|
+
|
|
54
|
+
return {
|
|
55
|
+
word,
|
|
56
|
+
base,
|
|
57
|
+
plural,
|
|
58
|
+
stem + "er",
|
|
59
|
+
stem + "ers",
|
|
60
|
+
stem + "ed",
|
|
61
|
+
stem + "ing",
|
|
62
|
+
stem + "ings",
|
|
63
|
+
stem + "ation",
|
|
64
|
+
stem + "ion",
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def splitter(text, query):
|
|
69
|
+
keywords = [f" {kw} " for kw in query.split()]
|
|
70
|
+
|
|
71
|
+
matches = []
|
|
72
|
+
|
|
73
|
+
for line in text.splitlines():
|
|
74
|
+
count = 0
|
|
75
|
+
|
|
76
|
+
line = " " + re.sub(r"[^A-Za-z]", " ", line).lower()
|
|
77
|
+
|
|
78
|
+
for kw in keywords:
|
|
79
|
+
count += line.count(kw)
|
|
80
|
+
|
|
81
|
+
if count > 0:
|
|
82
|
+
matches.append((count / max(200, len(line)), line))
|
|
83
|
+
|
|
84
|
+
matches.sort()
|
|
85
|
+
|
|
86
|
+
return matches
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def regex(text, query):
|
|
90
|
+
rx = re.compile(r"\b" + r"\b|\b".join(query.split()) + r"\b", flags=re.I)
|
|
91
|
+
|
|
92
|
+
matches = []
|
|
93
|
+
|
|
94
|
+
for line in text.splitlines():
|
|
95
|
+
count = sum(1 for _ in rx.finditer(line))
|
|
96
|
+
|
|
97
|
+
if count > 0:
|
|
98
|
+
matches.append((count / max(200, len(line)), line))
|
|
99
|
+
|
|
100
|
+
matches.sort()
|
|
101
|
+
|
|
102
|
+
return matches
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
if __name__ == "__main__":
|
|
106
|
+
text = open("artofwar.txt").read()
|
|
107
|
+
|
|
108
|
+
for fn in [splitter, regex]:
|
|
109
|
+
start = perf_counter()
|
|
110
|
+
|
|
111
|
+
print(fn(text, "war info")[-1])
|
|
112
|
+
|
|
113
|
+
print(fn, perf_counter() - start)
|
lexgrep/cli.py
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
|
|
3
|
+
from .lexgrep import lexgrep
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def main(argv=None):
|
|
7
|
+
parser = argparse.ArgumentParser(
|
|
8
|
+
prog="lexgrep",
|
|
9
|
+
description="",
|
|
10
|
+
)
|
|
11
|
+
parser.add_argument("files", nargs="*")
|
|
12
|
+
args = parser.parse_args(argv)
|
|
13
|
+
|
|
14
|
+
lexgrep(args.files)
|
|
15
|
+
|
|
16
|
+
return 0
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
if __name__ == "__main__":
|
|
20
|
+
raise SystemExit(main())
|
lexgrep/epub.py
ADDED
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
from concurrent.futures import ProcessPoolExecutor
|
|
2
|
+
from zipfile import ZipFile
|
|
3
|
+
import re
|
|
4
|
+
import unicodedata
|
|
5
|
+
|
|
6
|
+
ascii_map = str.maketrans(
|
|
7
|
+
{
|
|
8
|
+
"\u2018": "'",
|
|
9
|
+
"\u2019": "'",
|
|
10
|
+
"\u201a": "'",
|
|
11
|
+
"\u201b": "'",
|
|
12
|
+
"\u201c": '"',
|
|
13
|
+
"\u201d": '"',
|
|
14
|
+
"\u201e": '"',
|
|
15
|
+
"\u201f": '"',
|
|
16
|
+
"\u2013": "-",
|
|
17
|
+
"\u2014": "-",
|
|
18
|
+
"\u2212": "-",
|
|
19
|
+
"\u2026": "...",
|
|
20
|
+
"\u2022": "*",
|
|
21
|
+
"\u00a9": "(c)",
|
|
22
|
+
"\u00ae": "(R)",
|
|
23
|
+
"\u2122": "TM",
|
|
24
|
+
"\u00df": "ss",
|
|
25
|
+
"\u00c6": "AE",
|
|
26
|
+
"\u00e6": "ae",
|
|
27
|
+
"\u0152": "OE",
|
|
28
|
+
"\u0153": "oe",
|
|
29
|
+
"\u00d8": "O",
|
|
30
|
+
"\u00f8": "o",
|
|
31
|
+
"\u0141": "L",
|
|
32
|
+
"\u0142": "l",
|
|
33
|
+
"\u0300": "",
|
|
34
|
+
"\u0301": "",
|
|
35
|
+
"\u0302": "",
|
|
36
|
+
"\u0303": "",
|
|
37
|
+
"\u0304": "",
|
|
38
|
+
"\u0306": "",
|
|
39
|
+
"\u0307": "",
|
|
40
|
+
"\u0308": "",
|
|
41
|
+
"\u030a": "",
|
|
42
|
+
"\u030b": "",
|
|
43
|
+
"\u030c": "",
|
|
44
|
+
"\u0327": "",
|
|
45
|
+
"\u0328": "",
|
|
46
|
+
"\n": " ",
|
|
47
|
+
"\t": " ",
|
|
48
|
+
"\r": " ",
|
|
49
|
+
}
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def to_ascii(text):
|
|
54
|
+
"""Transliterate common punctuation and accented letters to ASCII.
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
>>> to_ascii("Café déjà vu")
|
|
58
|
+
'Cafe deja vu'
|
|
59
|
+
"""
|
|
60
|
+
text = unicodedata.normalize("NFKD", text)
|
|
61
|
+
text = text.translate(ascii_map)
|
|
62
|
+
return text.encode("ascii", "replace").decode("ascii")
|
|
63
|
+
return text
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
addbreaks = re.compile(r"<[ph][^>]*>")
|
|
67
|
+
rmtags = re.compile(r" *<[^>]*> *")
|
|
68
|
+
normbreaks = re.compile(r"\s*\n\s*\n\s*")
|
|
69
|
+
normspaces = re.compile(r" +")
|
|
70
|
+
archive = None
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def init_worker(path):
|
|
74
|
+
global archive
|
|
75
|
+
archive = ZipFile(path)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def process_chapter(name):
|
|
79
|
+
with archive.open(name) as file:
|
|
80
|
+
chapter = file.read().decode()
|
|
81
|
+
|
|
82
|
+
chapter = to_ascii(chapter)
|
|
83
|
+
chapter = addbreaks.sub("\n\n", chapter)
|
|
84
|
+
chapter = rmtags.sub(" ", chapter)
|
|
85
|
+
chapter = normbreaks.sub("\n\n", chapter)
|
|
86
|
+
chapter = normspaces.sub(" ", chapter)
|
|
87
|
+
return name, chapter.strip()
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def read_epub_text(filename):
|
|
91
|
+
chapters = []
|
|
92
|
+
|
|
93
|
+
with ZipFile(filename) as source:
|
|
94
|
+
names = sorted(
|
|
95
|
+
(
|
|
96
|
+
item.filename
|
|
97
|
+
for item in source.infolist()
|
|
98
|
+
if not item.is_dir() and item.filename.endswith("xhtml")
|
|
99
|
+
)
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
with ProcessPoolExecutor(
|
|
103
|
+
max_workers=4,
|
|
104
|
+
initializer=init_worker,
|
|
105
|
+
initargs=(filename,),
|
|
106
|
+
) as executor:
|
|
107
|
+
for name, chapter in executor.map(process_chapter, names, chunksize=4):
|
|
108
|
+
chapters.append(chapter)
|
|
109
|
+
|
|
110
|
+
return "\n\n".join(c for c in chapters if c)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def main():
|
|
114
|
+
print(read_epub_text("test.epub"))
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
if __name__ == "__main__":
|
|
118
|
+
main()
|