lexgrep 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,15 @@
1
+ name: Lint
2
+ on:
3
+ push:
4
+ branches:
5
+ - main
6
+ pull_request:
7
+ branches:
8
+ - main
9
+ jobs:
10
+ Lint:
11
+ runs-on: ubuntu-latest
12
+ steps:
13
+ - uses: actions/checkout@v5
14
+ - uses: astral-sh/setup-uv@v5
15
+ - run: make lint
@@ -0,0 +1,30 @@
1
+ name: PyPI
2
+
3
+ on:
4
+ push:
5
+ tags:
6
+ - 'v*.*.*'
7
+
8
+ jobs:
9
+ PyPI:
10
+ runs-on: ubuntu-latest
11
+ permissions:
12
+ contents: write
13
+ id-token: write
14
+ steps:
15
+ - name: Checkout
16
+ uses: actions/checkout@v5
17
+
18
+ - name: Setup uv
19
+ uses: astral-sh/setup-uv@v5
20
+
21
+ - name: Build
22
+ run: uv build
23
+
24
+ - name: Publish
25
+ run: uv publish
26
+
27
+ - name: Release
28
+ uses: softprops/action-gh-release@v2
29
+ with:
30
+ files: |
@@ -0,0 +1,15 @@
1
+ name: Test
2
+ on:
3
+ push:
4
+ branches:
5
+ - main
6
+ pull_request:
7
+ branches:
8
+ - main
9
+ jobs:
10
+ Test:
11
+ runs-on: ubuntu-latest
12
+ steps:
13
+ - uses: actions/checkout@v5
14
+ - uses: astral-sh/setup-uv@v5
15
+ - run: make test
lexgrep-0.2.0/PKG-INFO ADDED
@@ -0,0 +1,9 @@
1
+ Metadata-Version: 2.5
2
+ Name: lexgrep
3
+ Version: 0.2.0
4
+ Requires-Python: >=3.9
5
+ Description-Content-Type: text/markdown
6
+
7
+ https://github.com/clips/pattern/blob/master/pattern/text/en/inflect.py
8
+
9
+ https://github.com/nizarhabash1/catvar
@@ -0,0 +1,3 @@
1
+ from .lexgrep import lexgrep
2
+
3
+ __all__ = ["lexgrep"]
@@ -0,0 +1,4 @@
1
+ from .cli import main
2
+
3
+ if __name__ == "__main__":
4
+ raise SystemExit(main())
@@ -0,0 +1,113 @@
1
+ from time import perf_counter
2
+ from io import StringIO
3
+ import re
4
+
5
+
6
+ def related_words(word):
7
+ """
8
+ Generate words related to a supplied word
9
+
10
+ This is intended to be an over estimate. not all returned result need to
11
+ be proper words.
12
+
13
+ >>> 'compute' in related_words("computer")
14
+ True
15
+
16
+ >>> 'computes' in related_words("computer")
17
+ True
18
+
19
+ >>> 'computes' in related_words("computing")
20
+ True
21
+
22
+ >>> 'computation' in related_words("computing")
23
+ True
24
+
25
+ >>> 'running' in related_words("run")
26
+ True
27
+ """
28
+ word = word.strip().lower()
29
+
30
+ # Guess the verb behind agent nouns: computer -> compute.
31
+ for suffix in ["er", "or", "ing", "ers", "ors", "ed"]:
32
+ if word.endswith(suffix):
33
+ base = word[: -len(suffix)]
34
+ if base and not base.endswith(("a", "e", "i", "o", "u")):
35
+ base += "e"
36
+ break
37
+ else:
38
+ base = word
39
+
40
+ if base.endswith("e"):
41
+ # Separate the stem for common -ed and -ing forms.
42
+ stem = base[:-1]
43
+ elif re.search(r"[^aeiou][aeiou][^aeiou]$", base):
44
+ # Apply double consonant rule
45
+ stem = base + base[-1]
46
+ else:
47
+ stem = base
48
+
49
+ if stem.endswith("y") and len(stem) > 1:
50
+ plural = stem[:-1] + "ies"
51
+ else:
52
+ plural = base + "s"
53
+
54
+ return {
55
+ word,
56
+ base,
57
+ plural,
58
+ stem + "er",
59
+ stem + "ers",
60
+ stem + "ed",
61
+ stem + "ing",
62
+ stem + "ings",
63
+ stem + "ation",
64
+ stem + "ion",
65
+ }
66
+
67
+
68
+ def splitter(text, query):
69
+ keywords = [f" {kw} " for kw in query.split()]
70
+
71
+ matches = []
72
+
73
+ for line in text.splitlines():
74
+ count = 0
75
+
76
+ line = " " + re.sub(r"[^A-Za-z]", " ", line).lower()
77
+
78
+ for kw in keywords:
79
+ count += line.count(kw)
80
+
81
+ if count > 0:
82
+ matches.append((count / max(200, len(line)), line))
83
+
84
+ matches.sort()
85
+
86
+ return matches
87
+
88
+
89
+ def regex(text, query):
90
+ rx = re.compile(r"\b" + r"\b|\b".join(query.split()) + r"\b", flags=re.I)
91
+
92
+ matches = []
93
+
94
+ for line in text.splitlines():
95
+ count = sum(1 for _ in rx.finditer(line))
96
+
97
+ if count > 0:
98
+ matches.append((count / max(200, len(line)), line))
99
+
100
+ matches.sort()
101
+
102
+ return matches
103
+
104
+
105
+ if __name__ == "__main__":
106
+ text = open("artofwar.txt").read()
107
+
108
+ for fn in [splitter, regex]:
109
+ start = perf_counter()
110
+
111
+ print(fn(text, "war info")[-1])
112
+
113
+ print(fn, perf_counter() - start)
@@ -0,0 +1,20 @@
1
+ import argparse
2
+
3
+ from .lexgrep import lexgrep
4
+
5
+
6
+ def main(argv=None):
7
+ parser = argparse.ArgumentParser(
8
+ prog="lexgrep",
9
+ description="",
10
+ )
11
+ parser.add_argument("files", nargs="*")
12
+ args = parser.parse_args(argv)
13
+
14
+ lexgrep(args.files)
15
+
16
+ return 0
17
+
18
+
19
+ if __name__ == "__main__":
20
+ raise SystemExit(main())
@@ -0,0 +1,118 @@
1
+ from concurrent.futures import ProcessPoolExecutor
2
+ from zipfile import ZipFile
3
+ import re
4
+ import unicodedata
5
+
6
+ ascii_map = str.maketrans(
7
+ {
8
+ "\u2018": "'",
9
+ "\u2019": "'",
10
+ "\u201a": "'",
11
+ "\u201b": "'",
12
+ "\u201c": '"',
13
+ "\u201d": '"',
14
+ "\u201e": '"',
15
+ "\u201f": '"',
16
+ "\u2013": "-",
17
+ "\u2014": "-",
18
+ "\u2212": "-",
19
+ "\u2026": "...",
20
+ "\u2022": "*",
21
+ "\u00a9": "(c)",
22
+ "\u00ae": "(R)",
23
+ "\u2122": "TM",
24
+ "\u00df": "ss",
25
+ "\u00c6": "AE",
26
+ "\u00e6": "ae",
27
+ "\u0152": "OE",
28
+ "\u0153": "oe",
29
+ "\u00d8": "O",
30
+ "\u00f8": "o",
31
+ "\u0141": "L",
32
+ "\u0142": "l",
33
+ "\u0300": "",
34
+ "\u0301": "",
35
+ "\u0302": "",
36
+ "\u0303": "",
37
+ "\u0304": "",
38
+ "\u0306": "",
39
+ "\u0307": "",
40
+ "\u0308": "",
41
+ "\u030a": "",
42
+ "\u030b": "",
43
+ "\u030c": "",
44
+ "\u0327": "",
45
+ "\u0328": "",
46
+ "\n": " ",
47
+ "\t": " ",
48
+ "\r": " ",
49
+ }
50
+ )
51
+
52
+
53
+ def to_ascii(text):
54
+ """Transliterate common punctuation and accented letters to ASCII.
55
+
56
+
57
+ >>> to_ascii("Café déjà vu")
58
+ 'Cafe deja vu'
59
+ """
60
+ text = unicodedata.normalize("NFKD", text)
61
+ text = text.translate(ascii_map)
62
+ return text.encode("ascii", "replace").decode("ascii")
63
+ return text
64
+
65
+
66
+ addbreaks = re.compile(r"<[ph][^>]*>")
67
+ rmtags = re.compile(r" *<[^>]*> *")
68
+ normbreaks = re.compile(r"\s*\n\s*\n\s*")
69
+ normspaces = re.compile(r" +")
70
+ archive = None
71
+
72
+
73
+ def init_worker(path):
74
+ global archive
75
+ archive = ZipFile(path)
76
+
77
+
78
+ def process_chapter(name):
79
+ with archive.open(name) as file:
80
+ chapter = file.read().decode()
81
+
82
+ chapter = to_ascii(chapter)
83
+ chapter = addbreaks.sub("\n\n", chapter)
84
+ chapter = rmtags.sub(" ", chapter)
85
+ chapter = normbreaks.sub("\n\n", chapter)
86
+ chapter = normspaces.sub(" ", chapter)
87
+ return name, chapter.strip()
88
+
89
+
90
+ def read_epub_text(filename):
91
+ chapters = []
92
+
93
+ with ZipFile(filename) as source:
94
+ names = sorted(
95
+ (
96
+ item.filename
97
+ for item in source.infolist()
98
+ if not item.is_dir() and item.filename.endswith("xhtml")
99
+ )
100
+ )
101
+
102
+ with ProcessPoolExecutor(
103
+ max_workers=4,
104
+ initializer=init_worker,
105
+ initargs=(filename,),
106
+ ) as executor:
107
+ for name, chapter in executor.map(process_chapter, names, chunksize=4):
108
+ chapters.append(chapter)
109
+
110
+ return "\n\n".join(c for c in chapters if c)
111
+
112
+
113
+ def main():
114
+ print(read_epub_text("test.epub"))
115
+
116
+
117
+ if __name__ == "__main__":
118
+ main()