analp 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
analp-1.0.0/PKG-INFO ADDED
@@ -0,0 +1,20 @@
1
+ Metadata-Version: 2.4
2
+ Name: analp
3
+ Version: 1.0.0
4
+ Author-email: Helmut Konrad Schewe <helmutus@outlook.com>
5
+ License-Expression: MIT
6
+ Project-URL: Homepage, https://github.com/anaticulae/iamraw
7
+ Project-URL: Repository, https://github.com/anaticulae/iamraw
8
+ Classifier: Programming Language :: Python :: 3.12
9
+ Classifier: Programming Language :: Python :: 3.13
10
+ Classifier: Programming Language :: Python :: 3.14
11
+ Requires-Python: >=3.12
12
+ Description-Content-Type: text/markdown
13
+ Requires-Dist: utilo<3.0.0,>=2.107.4
14
+ Requires-Dist: konradus<3.0.0,>=1.0.1
15
+ Requires-Dist: nltk<4.0.0,>=3.9.4
16
+ Requires-Dist: ltk_data<2.0.0,>=1.0.2
17
+ Provides-Extra: dev
18
+ Requires-Dist: utilotest<2.0.0,>=1.0.1; extra == "dev"
19
+
20
+ # analp
analp-1.0.0/README ADDED
@@ -0,0 +1 @@
1
+ # analp
@@ -0,0 +1,33 @@
1
+ #==============================================================================
2
+ # C O P Y R I G H T
3
+ #------------------------------------------------------------------------------
4
+ # Copyright (c) 2021-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ #==============================================================================
9
+
10
+ import importlib.metadata
11
+ import os
12
+
13
+ import analp.__lazy__
14
+ from analp.pos import sent_pos
15
+ from analp.sentence import normalize as normalize_sentence
16
+ from analp.sentence import sent_tokenize
17
+ from analp.sentiment import sent_sentiment
18
+ from analp.word import isstopword
19
+ from analp.word import stopwords
20
+ from analp.word import word_tokenize
21
+
22
+ # import german_data
23
+ # import ltk_data
24
+
25
+ PACKAGE = 'analp'
26
+ __version__ = importlib.metadata.version(PACKAGE)
27
+
28
+ ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), '..'))
29
+ """Load STOPWORDS on access time."""
30
+ # from analp.corpus import STOPWORDS, see __lazy__
31
+ __getattr__ = lambda name: getattr(analp.__lazy__, name)
32
+
33
+ # ltk_data.add_nltk_path(german_data.ROOT)
@@ -0,0 +1,50 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2021-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import collections
11
+ import functools
12
+ import threading
13
+
14
+ import utilo
15
+
16
+ Configure = collections.namedtuple(
17
+ 'Configure',
18
+ 'sent_tokenize word_tokenize pos_tag STOPWORDS STEMMER',
19
+ )
20
+
21
+
22
+ @functools.lru_cache
23
+ def lazy() -> Configure:
24
+ utilo.debug('configure nltk')
25
+ import nltk
26
+ import nltk.corpus
27
+ import nltk.stem
28
+
29
+ nltk.download('stopwords', quiet=True)
30
+ nltk.download('punkt_tab', quiet=True)
31
+ nltk.download('averaged_perceptron_tagger_eng', quiet=True)
32
+
33
+ result = Configure(
34
+ sent_tokenize=nltk.sent_tokenize,
35
+ word_tokenize=nltk.word_tokenize,
36
+ pos_tag=nltk.pos_tag,
37
+ STOPWORDS=nltk.corpus.stopwords.words('german'),
38
+ STEMMER=nltk.stem.SnowballStemmer('german'),
39
+ )
40
+ return result
41
+
42
+
43
+ LOCK = threading.Lock()
44
+
45
+
46
+ def __getattr__(name):
47
+ with LOCK:
48
+ data = lazy()
49
+ result = getattr(data, name)
50
+ return result
@@ -0,0 +1,14 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2021-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ # SEE __lazy__
11
+
12
+ # import nltk.corpus
13
+
14
+ # STOPWORDS = frozenset(nltk.corpus.stopwords.words('german'))
@@ -0,0 +1,35 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2021-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import utilo
11
+
12
+ import analp
13
+ import analp.__lazy__
14
+
15
+
16
+ def sent_pos(text: str, language='german') -> list:
17
+ """\
18
+ >>> sent_pos('Helmut is speaking.', language='eng')
19
+ [('Helmut', 'NNP'), ('is', 'VBZ'), ('speaking', 'VBG'), ('.', '.')]
20
+ >>> sent_pos('Hier spricht Helmut.') # TODO: not supported yet
21
+ """
22
+ language = lang(language)
23
+ tokens = analp.word_tokenize(text)
24
+ try:
25
+ tagged = analp.__lazy__.pos_tag(tokens, lang=language)
26
+ except NotImplementedError:
27
+ utilo.error(f'language not supported: {language}')
28
+ return None
29
+ return tagged
30
+
31
+
32
+ def lang(item: str) -> str:
33
+ item = item.replace('german', 'ger')
34
+ item = item.replace('english', 'eng')
35
+ return item
@@ -0,0 +1,44 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2021-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import string
11
+
12
+ import analp.__lazy__
13
+
14
+
15
+ def normalize(
16
+ raw: str,
17
+ join: bool = True,
18
+ remove_punctation: bool = True,
19
+ ) -> list:
20
+ """\
21
+ >>> normalize('Hier spricht Helmut .', join=False)
22
+ ['hier', 'spricht', 'helmut']
23
+ >>> normalize('Heute sprechen wir doch über ein Thema')
24
+ 'heut sprech wir doch uber ein thema'
25
+ """
26
+ raw = raw.lower()
27
+ splitted = raw.split()
28
+ if remove_punctation:
29
+ splitted = [item for item in splitted if item not in string.punctuation]
30
+ stemmed = [analp.__lazy__.STEMMER.stem(item) for item in splitted]
31
+ if join:
32
+ result = ' '.join(stemmed)
33
+ else:
34
+ result = stemmed
35
+ return result
36
+
37
+
38
+ def sent_tokenize(text: str, language='german') -> list:
39
+ """\
40
+ >>> sent_tokenize('Das ist ein Text. Und ich bin Satz 2.')
41
+ ['Das ist ein Text.', 'Und ich bin Satz 2.']
42
+ """
43
+ tokenized = analp.__lazy__.sent_tokenize(text, language=language)
44
+ return list(tokenized)
@@ -0,0 +1,21 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2021-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import dataclasses
11
+
12
+
13
+ @dataclasses.dataclass
14
+ class Sentiment:
15
+
16
+ polarity: float = None
17
+ subjectivity: float = None
18
+
19
+
20
+ def sent_sentiment(text: str) -> Sentiment: # pylint:disable=W0613
21
+ pass
@@ -0,0 +1,59 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2021-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import unicodedata
11
+
12
+
13
+ def hashed(item: str) -> str:
14
+ """\
15
+ >>> hashed('‘SCHEN')
16
+ 'LEFTSINGLEQUOTATIONMARKSCHEN'
17
+ """
18
+ selected = [
19
+ unicodedata.name(char) if ord(char) > 256 else char for char in item
20
+ ]
21
+ selected = ''.join(selected).replace(' ', '')
22
+ return selected
23
+
24
+
25
+ ESCAPE = """\
26
+ ‘schen
27
+ ‘SCHEN
28
+ ’schen
29
+ ’SCHEN
30
+ 'schen
31
+ 'SCHEN
32
+ ‘sche
33
+ ’sche
34
+ ’SCHE
35
+ 'sche
36
+ 'SCHE
37
+ ‘s
38
+ ‘S
39
+ ’s
40
+ ’S
41
+ 's
42
+ 'S
43
+ """
44
+ ESCAPE: dict = {
45
+ f'{line} ': f'{hashed(line)} ' for line in ESCAPE.strip().splitlines()
46
+ }
47
+
48
+
49
+ def escape(text: str):
50
+ for key, value in ESCAPE.items():
51
+ text = text.replace(key, value)
52
+ return text
53
+
54
+
55
+ def deescape(text: list):
56
+ # escape mann_sche
57
+ for key, value in ESCAPE.items():
58
+ text = [item.replace(value.strip(), key.strip()) for item in text]
59
+ return text
@@ -0,0 +1,80 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2021-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import functools
11
+ import re
12
+
13
+ import konradus
14
+
15
+ import analp
16
+ import analp.utils
17
+
18
+
19
+ def word_tokenize(sentence: str, language: str = 'german') -> list:
20
+ """\
21
+ >>> word_tokenize('Ein Aufzählung z.B. heute oder morgen 1. Frosch.')
22
+ ['Ein', 'Aufzählung', 'z.B.', 'heute', 'oder', 'morgen', '1', '.', 'Frosch', '.']
23
+ >>> word_tokenize('Hier lässt sich auch der Luhmann’sche Personenbegriff angliedern.', language='science')
24
+ ['Hier', 'lässt', 'sich', 'auch', 'der', 'Luhmann’sche', 'Personenbegriff', 'angliedern', '.']
25
+ >>> word_tokenize('Clay Shirky‘s Writings About the Internet.', language='science')
26
+ ['Clay', 'Shirky‘s', 'Writings', 'About', 'the', 'Internet', '.']
27
+ >>> word_tokenize('Kaplan/Haenlein (2009) schreiben dazu: “In our view […] Social Media”12.', language='science')
28
+ ['Kaplan/Haenlein', '(', '2009', ')', 'schreiben', 'dazu', ':', '“', 'In', 'our', 'view', '[', '…', ']', 'Social', 'Media', '”', '12', '.']
29
+ >>> word_tokenize('Phänomen‚Protest‘ angemessen erfassen', language='science')
30
+ ['Phänomen', '‚', 'Protest', '‘', 'angemessen', 'erfassen']
31
+ """
32
+ if language == 'science':
33
+ # TODO: ENABLE SCIENCE LATER
34
+ language = 'german'
35
+ language = konradus.complexlang(language)
36
+ if language == 'unknown':
37
+ # unknown language is not defined yet.
38
+ language = 'science'
39
+ sentence = hack(sentence)
40
+ sentence = analp.utils.escape(sentence)
41
+ tokenized = analp.__lazy__.word_tokenize(sentence, language=language)
42
+ result = analp.utils.deescape(tokenized)
43
+ return result
44
+
45
+
46
+ def hack(line: str) -> str:
47
+ """Remove after improving `Science` parser.
48
+
49
+ >>> hack('Phänomen‚Protest‘ angemessen')
50
+ 'Phänomen ‚ Protest‘ angemessen'
51
+ """
52
+ line = re.sub(r'‚(?=\S)', '‚ ', line)
53
+ line = re.sub(r'(?=\S)‚', ' ‚', line)
54
+ return line
55
+
56
+
57
+ def isstopword(word: str, lang: str = 'german') -> bool:
58
+ """\
59
+ >>> isstopword('der')
60
+ True
61
+ >>> isstopword('House')
62
+ False
63
+ """
64
+ lang = konradus.complexlang(lang)
65
+ word = word.lower()
66
+ return word in stopwords(lang)
67
+
68
+
69
+ @functools.lru_cache(maxsize=None)
70
+ def stopwords(lang: str = 'german') -> set:
71
+ """\
72
+ >>> sorted(stopwords('english'))
73
+ ['a', 'about', 'above',...'your', 'yours', 'yourself', 'yourselves']
74
+ >>> sorted(stopwords('fre'))
75
+ ['ai', 'aie', 'aient', 'aies',...'étées', 'étés', 'êtes']
76
+ """
77
+ import nltk.corpus
78
+ lang = konradus.complexlang(lang)
79
+ result = set(nltk.corpus.stopwords.words(lang))
80
+ return result
@@ -0,0 +1,20 @@
1
+ Metadata-Version: 2.4
2
+ Name: analp
3
+ Version: 1.0.0
4
+ Author-email: Helmut Konrad Schewe <helmutus@outlook.com>
5
+ License-Expression: MIT
6
+ Project-URL: Homepage, https://github.com/anaticulae/iamraw
7
+ Project-URL: Repository, https://github.com/anaticulae/iamraw
8
+ Classifier: Programming Language :: Python :: 3.12
9
+ Classifier: Programming Language :: Python :: 3.13
10
+ Classifier: Programming Language :: Python :: 3.14
11
+ Requires-Python: >=3.12
12
+ Description-Content-Type: text/markdown
13
+ Requires-Dist: utilo<3.0.0,>=2.107.4
14
+ Requires-Dist: konradus<3.0.0,>=1.0.1
15
+ Requires-Dist: nltk<4.0.0,>=3.9.4
16
+ Requires-Dist: ltk_data<2.0.0,>=1.0.2
17
+ Provides-Extra: dev
18
+ Requires-Dist: utilotest<2.0.0,>=1.0.1; extra == "dev"
19
+
20
+ # analp
@@ -0,0 +1,16 @@
1
+ README
2
+ pyproject.toml
3
+ analp/__init__.py
4
+ analp/__lazy__.py
5
+ analp/corpus.py
6
+ analp/pos.py
7
+ analp/sentence.py
8
+ analp/sentiment.py
9
+ analp/utils.py
10
+ analp/word.py
11
+ analp.egg-info/PKG-INFO
12
+ analp.egg-info/SOURCES.txt
13
+ analp.egg-info/dependency_links.txt
14
+ analp.egg-info/requires.txt
15
+ analp.egg-info/top_level.txt
16
+ tests/test_imports.py
@@ -0,0 +1,7 @@
1
+ utilo<3.0.0,>=2.107.4
2
+ konradus<3.0.0,>=1.0.1
3
+ nltk<4.0.0,>=3.9.4
4
+ ltk_data<2.0.0,>=1.0.2
5
+
6
+ [dev]
7
+ utilotest<2.0.0,>=1.0.1
@@ -0,0 +1 @@
1
+ analp
@@ -0,0 +1,81 @@
1
+ [build-system]
2
+ requires = [
3
+ "setuptools>=82.0.1",
4
+ "wheel>=0.47.0",
5
+ ]
6
+ build-backend = "setuptools.build_meta"
7
+
8
+ [project]
9
+ name = "analp"
10
+ version = "1.0.0"
11
+ description = ""
12
+ requires-python = ">=3.12"
13
+ authors = [
14
+ { name = "Helmut Konrad Schewe", email = "helmutus@outlook.com" },
15
+ ]
16
+ dependencies = [
17
+ "utilo>=2.107.4,<3.0.0",
18
+ "konradus>=1.0.1,<3.0.0",
19
+ "nltk>=3.9.4,<4.0.0",
20
+ "ltk_data>=1.0.2,<2.0.0",
21
+ ]
22
+ keywords = []
23
+ classifiers = [
24
+ "Programming Language :: Python :: 3.12",
25
+ "Programming Language :: Python :: 3.13",
26
+ "Programming Language :: Python :: 3.14",
27
+ ]
28
+ license = "MIT"
29
+ license-files = [
30
+ "LICENSE",
31
+ ]
32
+
33
+ [project.readme]
34
+ file = "README"
35
+ content-type = "text/markdown"
36
+
37
+ [project.optional-dependencies]
38
+ dev = [
39
+ "utilotest>=1.0.1,<2.0.0",
40
+ ]
41
+
42
+ [project.urls]
43
+ Homepage = "https://github.com/anaticulae/iamraw"
44
+ Repository = "https://github.com/anaticulae/iamraw"
45
+
46
+ [tool.semantic_release]
47
+ version_toml = [
48
+ "pyproject.toml:project.version",
49
+ ]
50
+
51
+ [tool.semantic_release.changelog]
52
+ mode = "init"
53
+ output_format = "md"
54
+
55
+ [tool.semantic_release.changelog.default_templates]
56
+ changelog_file = "CHANGELOG"
57
+
58
+ [tool.semantic_release.commit_parser_options]
59
+ patch_tags = [
60
+ "fix",
61
+ "perf",
62
+ "build",
63
+ "chore",
64
+ "ci",
65
+ "docs",
66
+ "style",
67
+ "refactor",
68
+ "test",
69
+ "deps",
70
+ ]
71
+
72
+ [tool.setuptools.packages.find]
73
+ where = [
74
+ ".",
75
+ ]
76
+ include = [
77
+ "analp",
78
+ ]
79
+ exclude = [
80
+ "tests*",
81
+ ]
analp-1.0.0/setup.cfg ADDED
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,19 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2021-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+
11
+ def test_import_stopwords_lazy():
12
+ import analp
13
+ assert len(analp.__lazy__.STOPWORDS) > 100
14
+ assert len(analp.STOPWORDS) > 100
15
+
16
+
17
+ def test_import_stopwords_from_module():
18
+ import analp
19
+ assert len(analp.STOPWORDS) > 100