germania 1.30.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. germania-1.30.0/MANIFEST.in +11 -0
  2. germania-1.30.0/PKG-INFO +25 -0
  3. germania-1.30.0/README +1 -0
  4. germania-1.30.0/germania/__init__.py +106 -0
  5. germania-1.30.0/germania/abbrev.py +40 -0
  6. germania-1.30.0/germania/error/__init__.py +8 -0
  7. germania-1.30.0/germania/error/finding.py +43 -0
  8. germania-1.30.0/germania/error/machine.py +84 -0
  9. germania-1.30.0/germania/error/text.py +107 -0
  10. germania-1.30.0/germania/improve/__init__.py +8 -0
  11. germania-1.30.0/germania/improve/abbreviation.py +31 -0
  12. germania-1.30.0/germania/improve/highnote.py +28 -0
  13. germania-1.30.0/germania/improve/href.py +69 -0
  14. germania-1.30.0/germania/improve/magic.py +27 -0
  15. germania-1.30.0/germania/language.py +84 -0
  16. germania-1.30.0/germania/magic.py +227 -0
  17. germania-1.30.0/germania/pattern/__init__.py +36 -0
  18. germania-1.30.0/germania/pattern/access.py +97 -0
  19. germania-1.30.0/germania/pattern/author.py +275 -0
  20. germania-1.30.0/germania/pattern/book.py +237 -0
  21. germania-1.30.0/germania/pattern/date.py +150 -0
  22. germania-1.30.0/germania/pattern/href.py +118 -0
  23. germania-1.30.0/germania/pattern/mail.py +29 -0
  24. germania-1.30.0/germania/pattern/pagination.py +134 -0
  25. germania-1.30.0/germania/quotation.py +107 -0
  26. germania-1.30.0/germania/sentence/__init__.py +342 -0
  27. germania-1.30.0/germania/sentence/escape.py +80 -0
  28. germania-1.30.0/germania/sequence.py +291 -0
  29. germania-1.30.0/germania/tagger.py +27 -0
  30. germania-1.30.0/germania/text.py +22 -0
  31. germania-1.30.0/germania/utils/__init__.py +28 -0
  32. germania-1.30.0/germania/utils/month.py +104 -0
  33. germania-1.30.0/germania/word.py +323 -0
  34. germania-1.30.0/germania.egg-info/PKG-INFO +25 -0
  35. germania-1.30.0/germania.egg-info/SOURCES.txt +60 -0
  36. germania-1.30.0/germania.egg-info/dependency_links.txt +1 -0
  37. germania-1.30.0/germania.egg-info/requires.txt +12 -0
  38. germania-1.30.0/germania.egg-info/top_level.txt +3 -0
  39. germania-1.30.0/germania_data/__init__.py +27 -0
  40. germania-1.30.0/germania_data/institution.dict +14 -0
  41. germania-1.30.0/germania_data/names.dict +182 -0
  42. germania-1.30.0/germania_data/noperson.dict +40 -0
  43. germania-1.30.0/germania_data/press.dict +22 -0
  44. germania-1.30.0/germania_data/utils.py +28 -0
  45. germania-1.30.0/pyproject.toml +93 -0
  46. germania-1.30.0/science_text/__init__.py +12 -0
  47. germania-1.30.0/science_text/config.py +39 -0
  48. germania-1.30.0/science_text/improve.py +48 -0
  49. germania-1.30.0/science_text/train.py +90 -0
  50. germania-1.30.0/setup.cfg +4 -0
  51. germania-1.30.0/tests/test_data.py +15 -0
  52. germania-1.30.0/tests/test_language.py +60 -0
  53. germania-1.30.0/tests/test_magic.py +37 -0
  54. germania-1.30.0/tests/test_nltk.py +22 -0
  55. germania-1.30.0/tests/test_pages_master72.py +39 -0
  56. germania-1.30.0/tests/test_pattern_matched.py +45 -0
  57. germania-1.30.0/tests/test_quotation.py +87 -0
  58. germania-1.30.0/tests/test_sentence_select.py +39 -0
  59. germania-1.30.0/tests/test_sentence_split.py +353 -0
  60. germania-1.30.0/tests/test_sequence.py +63 -0
  61. germania-1.30.0/tests/test_tagger.py +20 -0
  62. germania-1.30.0/tests/test_words_split.py +146 -0
@@ -0,0 +1,11 @@
1
+ #==============================================================================
2
+ # C O P Y R I G H T
3
+ #------------------------------------------------------------------------------
4
+ # Copyright (c) 2020-2023 by Helmut Konrad Schewe. All rights reserved.
5
+ # This file is property of Helmut Konrad Schewe. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ #==============================================================================
9
+
10
+ include germania_data/*.dict
11
+ graft germania_data/ltk_data
@@ -0,0 +1,25 @@
1
+ Metadata-Version: 2.4
2
+ Name: germania
3
+ Version: 1.30.0
4
+ Author-email: Helmut Konrad Schewe <helmutus@outlook.com>
5
+ License-Expression: MIT
6
+ Project-URL: Homepage, https://github.com/anaticulae/germania
7
+ Project-URL: Repository, https://github.com/anaticulae/germania
8
+ Classifier: Programming Language :: Python :: 3.12
9
+ Classifier: Programming Language :: Python :: 3.13
10
+ Classifier: Programming Language :: Python :: 3.14
11
+ Requires-Python: >=3.12
12
+ Description-Content-Type: text/markdown
13
+ Requires-Dist: utilo<3.0.0,>=2.108.0
14
+ Requires-Dist: konradus<2.0.0,>=1.0.1
15
+ Requires-Dist: iamraw<5.0.0,>=4.91.4
16
+ Requires-Dist: configos<2.0.0,>=1.0.4
17
+ Requires-Dist: nltk<4.0.0,>=3.9.4
18
+ Requires-Dist: ltk_data<2.0.0,>=1.0.2
19
+ Requires-Dist: sdatum<2.0.0,>=1.0.0
20
+ Requires-Dist: analp<2.0.0,>=1.0.0
21
+ Provides-Extra: dev
22
+ Requires-Dist: utilotest<2.0.0,>=1.0.1; extra == "dev"
23
+ Requires-Dist: textbone<2.0.0,>=1.0.0; extra == "dev"
24
+
25
+ # germania
germania-1.30.0/README ADDED
@@ -0,0 +1 @@
1
+ # germania
@@ -0,0 +1,106 @@
1
+ #==============================================================================
2
+ # C O P Y R I G H T
3
+ #------------------------------------------------------------------------------
4
+ # Copyright (c) 2020-2023 by Helmut Konrad Schewe. All rights reserved.
5
+ # This file is property of Helmut Konrad Schewe. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ #==============================================================================
9
+ """germania
10
+ ======
11
+
12
+ The `germania` package is a wrapper for external tooling which have some
13
+ `magic` inside to determine the type of a word or split a text into
14
+ sentences into words.
15
+ """
16
+
17
+ import os
18
+
19
+ import ltk_data
20
+ import nltk
21
+
22
+ import germania.sentence
23
+ from germania.abbrev import find_abbrev
24
+ from germania.error.finding import TextError
25
+ from germania.error.finding import TextErrors
26
+ from germania.error.finding import TextErrorType
27
+ from germania.error.machine import TextErrorMachine
28
+ from germania.error.text import TextMachine
29
+ from germania.improve.abbreviation import abbreviation_magic
30
+ from germania.improve.highnote import highnote_magic
31
+ from germania.improve.href import href_magic
32
+ from germania.improve.magic import text_magic
33
+ from germania.language import LanguageResult
34
+ from germania.language import determine as lang
35
+ from germania.language import iseng
36
+ from germania.language import isfre
37
+ from germania.language import isger
38
+ from germania.magic import WordType
39
+ from germania.magic import WordTypes
40
+ from germania.magic import iscity
41
+ from germania.magic import isperson
42
+ from germania.magic import ispress
43
+ from germania.magic import isreference
44
+ from germania.magic import isyear
45
+ from germania.magic import wordtype
46
+ from germania.magic import wordtypes
47
+ from germania.pattern import matched
48
+ from germania.pattern.access import accessed
49
+ from germania.pattern.author import authors
50
+ from germania.pattern.author import authors_decide
51
+ from germania.pattern.book import bibtexts
52
+ from germania.pattern.book import doi
53
+ from germania.pattern.book import isbn
54
+ from germania.pattern.book import issn
55
+ from germania.pattern.book import references
56
+ from germania.pattern.book import volumes
57
+ from germania.pattern.date import dates
58
+ from germania.pattern.date import dates_master
59
+ from germania.pattern.date import dates_month_year
60
+ from germania.pattern.date import years
61
+ from germania.pattern.href import hyperlink
62
+ from germania.pattern.href import links
63
+ from germania.pattern.href import locallink
64
+ from germania.pattern.mail import mails
65
+ from germania.pattern.pagination import page_single
66
+ from germania.pattern.pagination import pagenumbers
67
+ from germania.pattern.pagination import pages_complex
68
+ from germania.quotation import extract_quotes
69
+ from germania.quotation import raw_quotation
70
+ from germania.sentence import Sentences
71
+ from germania.sentence import is_sentence
72
+ from germania.sentence import is_sentence_closed
73
+ from germania.sentence import sentence_select
74
+ from germania.sentence import sentence_tokenize
75
+ from germania.sentence import split_token
76
+ from germania.sequence import init
77
+ from germania.sequence import ngram
78
+ from germania.sequence import search
79
+ from germania.sequence import searches
80
+ from germania.sequence import token_plain
81
+ from germania.tagger import word_tag
82
+ from germania.text import words_fromstr
83
+ from germania.utils import collect_and_replace
84
+ from germania.utils.month import MONTH
85
+ from germania.utils.month import MONTH_REGEX
86
+ from germania.utils.month import month
87
+ from germania.word import Words
88
+ from germania.word import contain_quotation_marks
89
+ from germania.word import word_normalize
90
+ from germania.word import word_tokenize
91
+
92
+ split_words = word_tokenize # pylint:disable=C0103
93
+ split_sentences = sentence_tokenize # pylint:disable=C0103
94
+
95
+ __version__ = '1.30.0'
96
+
97
+ ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), '..'))
98
+
99
+ # REMOVE LATER
100
+ pages = page_single
101
+
102
+ nltk.download('crubadan', quiet=True)
103
+ nltk.download('punkt_tab', quiet=True)
104
+
105
+ # TODO: REMovE LATER
106
+ germania.sentence.language_select = lambda x: 'german'
@@ -0,0 +1,40 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2022-2023 by Helmut Konrad Schewe. All rights reserved.
5
+ # This file is property of Helmut Konrad Schewe. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import sdatum
11
+ import utilo
12
+
13
+ import germania
14
+
15
+
16
+ def find_abbrev(abbrev: str, words: list) -> str:
17
+ """\
18
+ >>> find_abbrev('MNU', 'Steuergebiete zugunsten multinationaler Unternehmen'.split())
19
+ 'multinationaler Unternehmen'
20
+ >>> import germania
21
+ >>> find_abbrev('MNU', germania.words_fromstr('Steuergebiete zugunsten multinationaler Unternehmens(MNU) ende'))
22
+ 'multinationaler Unternehmens'
23
+ """
24
+ lookup = sdatum.abbrev(abbrev)
25
+ if lookup is None:
26
+ return None
27
+ lookup = [germania.word_normalize(item) for item in lookup]
28
+ normalized = [
29
+ germania.word_normalize(item) if isinstance(item, str) else item
30
+ for item in words
31
+ ]
32
+ detected = germania.searches(
33
+ patterns=lookup,
34
+ sentence=normalized,
35
+ tokens_complex=False,
36
+ )
37
+ if not detected:
38
+ return None
39
+ result = ' '.join(words[index] for index in utilo.rlist(*detected[0]))
40
+ return result
@@ -0,0 +1,8 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2022-2023 by Helmut Konrad Schewe. All rights reserved.
5
+ # This file is property of Helmut Konrad Schewe. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
@@ -0,0 +1,43 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2022-2023 by Helmut Konrad Schewe. All rights reserved.
5
+ # This file is property of Helmut Konrad Schewe. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import dataclasses
11
+ import enum
12
+
13
+ import iamraw
14
+
15
+
16
+ class TextErrorType(enum.Enum):
17
+ MISSING = enum.auto()
18
+ """Text token is missing"""
19
+ STYLE = enum.auto()
20
+ """Violation against good style"""
21
+ RULE = enum.auto()
22
+ """Writing is against the writing laws"""
23
+ DUPLICATED = enum.auto()
24
+ """Copy paste error"""
25
+ REPLACEMENT = enum.auto()
26
+ """Replace this content"""
27
+ UNDEFINED = enum.auto()
28
+ """No special state"""
29
+
30
+
31
+ @dataclasses.dataclass
32
+ class TextError:
33
+ title: str = None
34
+ text: str = None
35
+ state: TextErrorType = None
36
+ location: iamraw.Location = None
37
+ raw: str = None
38
+ better: str = None
39
+ debug_method: str = None
40
+ """Name of method which had determined this error."""
41
+
42
+
43
+ TextErrors = list[TextError]
@@ -0,0 +1,84 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2022-2023 by Helmut Konrad Schewe. All rights reserved.
5
+ # This file is property of Helmut Konrad Schewe. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import contextlib
11
+
12
+ import iamraw
13
+ import utilo
14
+
15
+ import germania.error.finding
16
+
17
+
18
+ class TextErrorMachine:
19
+ """\
20
+ >>> empty = TextErrorMachine()
21
+ >>> empty.determine('')
22
+ []
23
+ """
24
+
25
+ def determine(
26
+ self,
27
+ text: str,
28
+ page: int = None,
29
+ ) -> germania.error.finding.TextErrors:
30
+ result = []
31
+ todo = utilo.methods(self, starts='check_')
32
+ for method in todo:
33
+ detected = method(text)
34
+ if not detected:
35
+ continue
36
+ if not utilo.iterable(detected):
37
+ result.append(detected)
38
+ detected.debug_method = method.__name__
39
+ continue
40
+ for item in detected:
41
+ item.debug_method = method.__name__
42
+ result.extend(detected)
43
+ if page is not None:
44
+ for item in result:
45
+ with contextlib.suppress(AttributeError):
46
+ item.location.page = page
47
+ return result
48
+
49
+ def location(self, match) -> iamraw.RangedLocation: # pylint:disable=R0201
50
+ if not match:
51
+ return None
52
+ location = iamraw.RangedLocation(
53
+ char=match.span()[0],
54
+ char_end=match.span()[1],
55
+ )
56
+ return location
57
+
58
+
59
+ class PhysicMachine(TextErrorMachine):
60
+ """\
61
+ >>> machine = PhysicMachine()
62
+ >>> machine.determine('The weight is 200kg. Thats a lot.', page=10)
63
+ [TextError(...state=<TextErrorType.RULE...>, location=RangedLocation(page=10, char=13, char_end=19), raw='200kg', better='200 kg', debug_method='check_physical_spaces')]
64
+ """
65
+
66
+ MISSING_SPACE_BEFORE_UNIT = utilo.compiles(r"""
67
+ \W
68
+ (
69
+ (?P<value>\d+((\.|\,)\d+){0,1})
70
+ (?P<unit>%|‰|kg|km/h|mmHg|mm|ms|mg/km|m|cm|V|W|Hz|mW)
71
+ )
72
+ """)
73
+
74
+ def check_physical_spaces(self, text: str) -> list:
75
+ result = []
76
+ for match in self.MISSING_SPACE_BEFORE_UNIT.finditer(text):
77
+ error = germania.error.finding.TextError(
78
+ state=germania.error.finding.TextErrorType.RULE,
79
+ better=match['value'] + ' ' + match['unit'],
80
+ location=self.location(match),
81
+ raw=match[1],
82
+ )
83
+ result.append(error)
84
+ return result
@@ -0,0 +1,107 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2022-2023 by Helmut Konrad Schewe. All rights reserved.
5
+ # This file is property of Helmut Konrad Schewe. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import re
11
+
12
+ import sdatum
13
+ import utilo
14
+
15
+ import germania
16
+
17
+
18
+ class TextMachine(germania.TextErrorMachine):
19
+
20
+ MISSING_PAGENUMBER = utilo.compiles(r"""
21
+ (
22
+ \s|
23
+ [^\w]\.| # u.s.
24
+ [,;:]
25
+ )
26
+ (
27
+ S\.|
28
+ p\.
29
+ )
30
+ [ ]{0,4}\n?
31
+ (?![\dixv\s]) # s. VII
32
+ """)
33
+
34
+ def check_pagenumber_complete(self, text: str) -> list:
35
+ r"""\
36
+ >>> check = TextMachine().check_pagenumber_complete
37
+ >>> check('Bonn, Herford 1993, S.\nDer Brief')
38
+ [TextError(...<TextErrorType.MISSING...raw='S.'...)]
39
+ >>> check('Hier fehlt wohlx S. die Seitennummer')
40
+ [TextError(...<TextErrorType.MISSING...raw='S.'...)]
41
+ >>> check('Vgl. Dixon, S.: Twitter: distribution of global audiences 2021')
42
+ []
43
+ >>> check('Schols. Hier')
44
+ []
45
+ >>> check('Berlin 19982, s. VII ')
46
+ []
47
+ >>> check('S. Meuschel, Legitimation und Parteiherrschaft ')
48
+ []
49
+ >>> check('in the U.S. | Pew Research')
50
+ []
51
+ >>> check('Meuschel S., Legitimation und Parteiherrschaft ')
52
+ []
53
+
54
+ run large text test
55
+ >>> import textbone;check(textbone.text_improved(1024*1024))
56
+ []
57
+ """
58
+ result = []
59
+ for match in self.MISSING_PAGENUMBER.finditer(text):
60
+ after = text[match.span()[1]:]
61
+ if follows_name(after):
62
+ continue
63
+ start, lookback = match.span()[0], 60
64
+ before = text[max(0, start - lookback):start]
65
+ if name_before(before):
66
+ continue
67
+ error = germania.TextError(
68
+ state=germania.TextErrorType.MISSING,
69
+ location=self.location(match),
70
+ raw=match[0].strip(),
71
+ )
72
+ result.append(error)
73
+ return result
74
+
75
+
76
+ def follows_name(text: str) -> bool:
77
+ """\
78
+ >>> follows_name('Helmut wohnt hier')
79
+ True
80
+ >>> follows_name('Copyright (c) 2022 by Helmut')
81
+ False
82
+ >>> follows_name('Der Brief')
83
+ False
84
+ """
85
+ text = text.strip()
86
+ if re.match(r'^\w\.', text):
87
+ return True
88
+ for name in text.split()[0:4]:
89
+ name = name.strip(':;, ')
90
+ if sdatum.isname(name):
91
+ return True
92
+ return False
93
+
94
+
95
+ def name_before(text: str) -> bool:
96
+ """\
97
+ >>> name_before('Legitimation und Parteiherrschaft Meuschel')
98
+ True
99
+ """
100
+ token = text.rstrip(':;, ').rsplit(maxsplit=1)
101
+ if not token:
102
+ return False
103
+ # select the right one
104
+ name = token[-1].strip(':;, ')
105
+ if sdatum.isname(name):
106
+ return True
107
+ return False
@@ -0,0 +1,8 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2021-2023 by Helmut Konrad Schewe. All rights reserved.
5
+ # This file is property of Helmut Konrad Schewe. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
@@ -0,0 +1,31 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2021-2023 by Helmut Konrad Schewe. All rights reserved.
5
+ # This file is property of Helmut Konrad Schewe. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import konradus
11
+ import utilo
12
+
13
+
14
+ @utilo.cacheme
15
+ def abbreviation_magic(text: str) -> str:
16
+ """\
17
+ >>> abbreviation_magic('Helmut hier u. a. und mehr.')
18
+ 'Helmut hier u.a. und mehr.'
19
+ >>> abbreviation_magic('a. a. o.')
20
+ 'a.a.o.'
21
+ """
22
+ for token, replace in TEXT_MAGIC:
23
+ text = text.replace(token, replace)
24
+ return text
25
+
26
+
27
+ TEXT_MAGIC = [
28
+ ('. '.join(item.split('.')).strip(), item)
29
+ for item in list(konradus.ABBREVIATION) + list(konradus.ABBREVIATION_LOWER)
30
+ if item.count('.') > 1
31
+ ]
@@ -0,0 +1,28 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2021-2023 by Helmut Konrad Schewe. All rights reserved.
5
+ # This file is property of Helmut Konrad Schewe. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import re
11
+
12
+ import utilo
13
+
14
+
15
+ @utilo.cacheme
16
+ def highnote_magic(text: str) -> str:
17
+ """\
18
+ >>> highnote_magic('ohnehin unmöglich.89 So')
19
+ 'ohnehin unmöglich. 89 So'
20
+ """
21
+ # highnote at the end of sentence
22
+ text = re.sub(
23
+ r'([a-z])([\.\!\?])(\d{1,4})([ ]{1,4})',
24
+ r'\1\2 \3\4',
25
+ text,
26
+ flags=re.I,
27
+ )
28
+ return text
@@ -0,0 +1,69 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2021-2023 by Helmut Konrad Schewe. All rights reserved.
5
+ # This file is property of Helmut Konrad Schewe. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import re
11
+
12
+ import utilo
13
+
14
+
15
+ @utilo.cacheme
16
+ def href_magic(text: str) -> str:
17
+ """\
18
+ >>> href_magic('4. url: https : / / www . apache . org / licenses/LICENS')
19
+ '4. url: https://www.apache.org/licenses/LICENS'
20
+ """
21
+ result = link_fink(text)
22
+ return result
23
+
24
+
25
+ @utilo.cacheme
26
+ def link_fink(text: str) -> str:
27
+ """\
28
+ >>> link_fink('url: http : / / www . bitkom . org / files / documents / BITKOM _ Leitfaden')
29
+ 'url: http://www.bitkom.org/files/documents/BITKOM_Leitfaden'
30
+ >>> link_fink('singulären bzw. typischen')
31
+ 'singulären bzw. typischen'
32
+ """
33
+ if not IS_HTTP.search(text):
34
+ return text
35
+ for (token, replacement) in SPACE_PATTERN:
36
+ text = re.sub(token, replacement, text)
37
+ return text
38
+
39
+
40
+ IS_HTTP = utilo.compiles(r'\bhttp')
41
+
42
+ SPACE_PATTERN = (
43
+ ('. org', '.org'),
44
+ ('. de', '.de'),
45
+ ('. com', '.com'),
46
+ ('. net', '.net'),
47
+ (' .org', '.org'),
48
+ (' .de', '.de'),
49
+ (' .com', '.com'),
50
+ (' .net', '.net'),
51
+ (' : /', ':/'),
52
+ (' / ', '/'),
53
+ (':/ ', ':/'),
54
+ (' . ', '.'),
55
+ (' _ ', '_'),
56
+ ('_ ', '_'),
57
+ ('/ ', '/'),
58
+ ('http :', 'http:'),
59
+ ('https :', 'https:'),
60
+ ('http ://', 'http://'),
61
+ ('https ://', 'https://'),
62
+ ('http: //', 'http://'),
63
+ ('https: //', 'https://'),
64
+ )
65
+ SPACE_PATTERN = [(re.escape(left), right) for left, right in SPACE_PATTERN]
66
+ SPACE_PATTERN.append((
67
+ r'(?P<left>[a-z])\s{0,2}\.\s{0,2}(?P<right>[a-z])',
68
+ r'\g<left>.\g<right>',
69
+ ))
@@ -0,0 +1,27 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2021-2023 by Helmut Konrad Schewe. All rights reserved.
5
+ # This file is property of Helmut Konrad Schewe. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import utilo
11
+
12
+ import germania.improve.abbreviation
13
+ import germania.improve.highnote
14
+ import germania.improve.href
15
+
16
+ TODO = (
17
+ germania.improve.abbreviation.abbreviation_magic,
18
+ germania.improve.highnote.highnote_magic,
19
+ germania.improve.href.href_magic,
20
+ )
21
+
22
+
23
+ @utilo.cacheme
24
+ def text_magic(text: str) -> str:
25
+ for pattern in TODO:
26
+ text = pattern(text)
27
+ return text
@@ -0,0 +1,84 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2020-2023 by Helmut Konrad Schewe. All rights reserved.
5
+ # This file is property of Helmut Konrad Schewe. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+ """Language Probability
10
+ ====================
11
+
12
+ Use nltk to determine language where sentence is written in.
13
+ """
14
+
15
+ import collections
16
+ import contextlib
17
+
18
+ import konradus
19
+ import utilo
20
+
21
+ LanguageResult = collections.namedtuple(
22
+ 'LanguageResult',
23
+ 'language, probability',
24
+ )
25
+
26
+
27
+ def determine(text: str) -> LanguageResult:
28
+ if isinstance(text, list):
29
+ text = konradus.remove_marks(text)
30
+ if not isinstance(text, str):
31
+ text = ' '.join(text)
32
+ cat = textcat()
33
+ detected = cat.guess_language(text)
34
+ language = konradus.Language.UNKNOWN
35
+ with contextlib.suppress(KeyError):
36
+ language = MAPPING[detected]
37
+ return LanguageResult(language=language, probability=1.0)
38
+
39
+
40
+ def isfre(tokens: str) -> bool:
41
+ """\
42
+ >>> isfre('Ich bin Helmut')
43
+ False
44
+ >>> isfre('Bonjour monsieur.')
45
+ True
46
+
47
+ verify that interface support tokens
48
+ >>> isfre('Toujour suis Luis.'.split())
49
+ True
50
+ """
51
+ return determine(tokens).language == konradus.Language.FRENCH
52
+
53
+
54
+ def iseng(tokens: str) -> bool:
55
+ """\
56
+ >>> iseng('Ich bin Helmut')
57
+ False
58
+ >>> iseng('i like fish')
59
+ True
60
+ """
61
+ return determine(tokens).language == konradus.Language.ENGLISH
62
+
63
+
64
+ def isger(tokens: str) -> bool:
65
+ """\
66
+ >>> isger('Kartoffelsalat')
67
+ True
68
+ """
69
+ return determine(tokens).language == konradus.Language.GERMAN
70
+
71
+
72
+ MAPPING = {
73
+ 'deu': konradus.Language.GERMAN,
74
+ 'eng': konradus.Language.ENGLISH,
75
+ # 'es': konradus.Language.SPANISH,
76
+ 'fra': konradus.Language.FRENCH,
77
+ }
78
+
79
+
80
+ @utilo.cacheme
81
+ def textcat():
82
+ import nltk.classify.textcat
83
+ result = nltk.classify.textcat.TextCat()
84
+ return result