germania 1.30.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- germania-1.30.0/MANIFEST.in +11 -0
- germania-1.30.0/PKG-INFO +25 -0
- germania-1.30.0/README +1 -0
- germania-1.30.0/germania/__init__.py +106 -0
- germania-1.30.0/germania/abbrev.py +40 -0
- germania-1.30.0/germania/error/__init__.py +8 -0
- germania-1.30.0/germania/error/finding.py +43 -0
- germania-1.30.0/germania/error/machine.py +84 -0
- germania-1.30.0/germania/error/text.py +107 -0
- germania-1.30.0/germania/improve/__init__.py +8 -0
- germania-1.30.0/germania/improve/abbreviation.py +31 -0
- germania-1.30.0/germania/improve/highnote.py +28 -0
- germania-1.30.0/germania/improve/href.py +69 -0
- germania-1.30.0/germania/improve/magic.py +27 -0
- germania-1.30.0/germania/language.py +84 -0
- germania-1.30.0/germania/magic.py +227 -0
- germania-1.30.0/germania/pattern/__init__.py +36 -0
- germania-1.30.0/germania/pattern/access.py +97 -0
- germania-1.30.0/germania/pattern/author.py +275 -0
- germania-1.30.0/germania/pattern/book.py +237 -0
- germania-1.30.0/germania/pattern/date.py +150 -0
- germania-1.30.0/germania/pattern/href.py +118 -0
- germania-1.30.0/germania/pattern/mail.py +29 -0
- germania-1.30.0/germania/pattern/pagination.py +134 -0
- germania-1.30.0/germania/quotation.py +107 -0
- germania-1.30.0/germania/sentence/__init__.py +342 -0
- germania-1.30.0/germania/sentence/escape.py +80 -0
- germania-1.30.0/germania/sequence.py +291 -0
- germania-1.30.0/germania/tagger.py +27 -0
- germania-1.30.0/germania/text.py +22 -0
- germania-1.30.0/germania/utils/__init__.py +28 -0
- germania-1.30.0/germania/utils/month.py +104 -0
- germania-1.30.0/germania/word.py +323 -0
- germania-1.30.0/germania.egg-info/PKG-INFO +25 -0
- germania-1.30.0/germania.egg-info/SOURCES.txt +60 -0
- germania-1.30.0/germania.egg-info/dependency_links.txt +1 -0
- germania-1.30.0/germania.egg-info/requires.txt +12 -0
- germania-1.30.0/germania.egg-info/top_level.txt +3 -0
- germania-1.30.0/germania_data/__init__.py +27 -0
- germania-1.30.0/germania_data/institution.dict +14 -0
- germania-1.30.0/germania_data/names.dict +182 -0
- germania-1.30.0/germania_data/noperson.dict +40 -0
- germania-1.30.0/germania_data/press.dict +22 -0
- germania-1.30.0/germania_data/utils.py +28 -0
- germania-1.30.0/pyproject.toml +93 -0
- germania-1.30.0/science_text/__init__.py +12 -0
- germania-1.30.0/science_text/config.py +39 -0
- germania-1.30.0/science_text/improve.py +48 -0
- germania-1.30.0/science_text/train.py +90 -0
- germania-1.30.0/setup.cfg +4 -0
- germania-1.30.0/tests/test_data.py +15 -0
- germania-1.30.0/tests/test_language.py +60 -0
- germania-1.30.0/tests/test_magic.py +37 -0
- germania-1.30.0/tests/test_nltk.py +22 -0
- germania-1.30.0/tests/test_pages_master72.py +39 -0
- germania-1.30.0/tests/test_pattern_matched.py +45 -0
- germania-1.30.0/tests/test_quotation.py +87 -0
- germania-1.30.0/tests/test_sentence_select.py +39 -0
- germania-1.30.0/tests/test_sentence_split.py +353 -0
- germania-1.30.0/tests/test_sequence.py +63 -0
- germania-1.30.0/tests/test_tagger.py +20 -0
- germania-1.30.0/tests/test_words_split.py +146 -0
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
#==============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
#------------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2020-2023 by Helmut Konrad Schewe. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Schewe. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
#==============================================================================
|
|
9
|
+
|
|
10
|
+
include germania_data/*.dict
|
|
11
|
+
graft germania_data/ltk_data
|
germania-1.30.0/PKG-INFO
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: germania
|
|
3
|
+
Version: 1.30.0
|
|
4
|
+
Author-email: Helmut Konrad Schewe <helmutus@outlook.com>
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Project-URL: Homepage, https://github.com/anaticulae/germania
|
|
7
|
+
Project-URL: Repository, https://github.com/anaticulae/germania
|
|
8
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
11
|
+
Requires-Python: >=3.12
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
Requires-Dist: utilo<3.0.0,>=2.108.0
|
|
14
|
+
Requires-Dist: konradus<2.0.0,>=1.0.1
|
|
15
|
+
Requires-Dist: iamraw<5.0.0,>=4.91.4
|
|
16
|
+
Requires-Dist: configos<2.0.0,>=1.0.4
|
|
17
|
+
Requires-Dist: nltk<4.0.0,>=3.9.4
|
|
18
|
+
Requires-Dist: ltk_data<2.0.0,>=1.0.2
|
|
19
|
+
Requires-Dist: sdatum<2.0.0,>=1.0.0
|
|
20
|
+
Requires-Dist: analp<2.0.0,>=1.0.0
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Requires-Dist: utilotest<2.0.0,>=1.0.1; extra == "dev"
|
|
23
|
+
Requires-Dist: textbone<2.0.0,>=1.0.0; extra == "dev"
|
|
24
|
+
|
|
25
|
+
# germania
|
germania-1.30.0/README
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# germania
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
#==============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
#------------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2020-2023 by Helmut Konrad Schewe. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Schewe. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
#==============================================================================
|
|
9
|
+
"""germania
|
|
10
|
+
======
|
|
11
|
+
|
|
12
|
+
The `germania` package is a wrapper for external tooling which have some
|
|
13
|
+
`magic` inside to determine the type of a word or split a text into
|
|
14
|
+
sentences into words.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
import os
|
|
18
|
+
|
|
19
|
+
import ltk_data
|
|
20
|
+
import nltk
|
|
21
|
+
|
|
22
|
+
import germania.sentence
|
|
23
|
+
from germania.abbrev import find_abbrev
|
|
24
|
+
from germania.error.finding import TextError
|
|
25
|
+
from germania.error.finding import TextErrors
|
|
26
|
+
from germania.error.finding import TextErrorType
|
|
27
|
+
from germania.error.machine import TextErrorMachine
|
|
28
|
+
from germania.error.text import TextMachine
|
|
29
|
+
from germania.improve.abbreviation import abbreviation_magic
|
|
30
|
+
from germania.improve.highnote import highnote_magic
|
|
31
|
+
from germania.improve.href import href_magic
|
|
32
|
+
from germania.improve.magic import text_magic
|
|
33
|
+
from germania.language import LanguageResult
|
|
34
|
+
from germania.language import determine as lang
|
|
35
|
+
from germania.language import iseng
|
|
36
|
+
from germania.language import isfre
|
|
37
|
+
from germania.language import isger
|
|
38
|
+
from germania.magic import WordType
|
|
39
|
+
from germania.magic import WordTypes
|
|
40
|
+
from germania.magic import iscity
|
|
41
|
+
from germania.magic import isperson
|
|
42
|
+
from germania.magic import ispress
|
|
43
|
+
from germania.magic import isreference
|
|
44
|
+
from germania.magic import isyear
|
|
45
|
+
from germania.magic import wordtype
|
|
46
|
+
from germania.magic import wordtypes
|
|
47
|
+
from germania.pattern import matched
|
|
48
|
+
from germania.pattern.access import accessed
|
|
49
|
+
from germania.pattern.author import authors
|
|
50
|
+
from germania.pattern.author import authors_decide
|
|
51
|
+
from germania.pattern.book import bibtexts
|
|
52
|
+
from germania.pattern.book import doi
|
|
53
|
+
from germania.pattern.book import isbn
|
|
54
|
+
from germania.pattern.book import issn
|
|
55
|
+
from germania.pattern.book import references
|
|
56
|
+
from germania.pattern.book import volumes
|
|
57
|
+
from germania.pattern.date import dates
|
|
58
|
+
from germania.pattern.date import dates_master
|
|
59
|
+
from germania.pattern.date import dates_month_year
|
|
60
|
+
from germania.pattern.date import years
|
|
61
|
+
from germania.pattern.href import hyperlink
|
|
62
|
+
from germania.pattern.href import links
|
|
63
|
+
from germania.pattern.href import locallink
|
|
64
|
+
from germania.pattern.mail import mails
|
|
65
|
+
from germania.pattern.pagination import page_single
|
|
66
|
+
from germania.pattern.pagination import pagenumbers
|
|
67
|
+
from germania.pattern.pagination import pages_complex
|
|
68
|
+
from germania.quotation import extract_quotes
|
|
69
|
+
from germania.quotation import raw_quotation
|
|
70
|
+
from germania.sentence import Sentences
|
|
71
|
+
from germania.sentence import is_sentence
|
|
72
|
+
from germania.sentence import is_sentence_closed
|
|
73
|
+
from germania.sentence import sentence_select
|
|
74
|
+
from germania.sentence import sentence_tokenize
|
|
75
|
+
from germania.sentence import split_token
|
|
76
|
+
from germania.sequence import init
|
|
77
|
+
from germania.sequence import ngram
|
|
78
|
+
from germania.sequence import search
|
|
79
|
+
from germania.sequence import searches
|
|
80
|
+
from germania.sequence import token_plain
|
|
81
|
+
from germania.tagger import word_tag
|
|
82
|
+
from germania.text import words_fromstr
|
|
83
|
+
from germania.utils import collect_and_replace
|
|
84
|
+
from germania.utils.month import MONTH
|
|
85
|
+
from germania.utils.month import MONTH_REGEX
|
|
86
|
+
from germania.utils.month import month
|
|
87
|
+
from germania.word import Words
|
|
88
|
+
from germania.word import contain_quotation_marks
|
|
89
|
+
from germania.word import word_normalize
|
|
90
|
+
from germania.word import word_tokenize
|
|
91
|
+
|
|
92
|
+
split_words = word_tokenize # pylint:disable=C0103
|
|
93
|
+
split_sentences = sentence_tokenize # pylint:disable=C0103
|
|
94
|
+
|
|
95
|
+
__version__ = '1.30.0'
|
|
96
|
+
|
|
97
|
+
ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), '..'))
|
|
98
|
+
|
|
99
|
+
# REMOVE LATER
|
|
100
|
+
pages = page_single
|
|
101
|
+
|
|
102
|
+
nltk.download('crubadan', quiet=True)
|
|
103
|
+
nltk.download('punkt_tab', quiet=True)
|
|
104
|
+
|
|
105
|
+
# TODO: REMovE LATER
|
|
106
|
+
germania.sentence.language_select = lambda x: 'german'
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2022-2023 by Helmut Konrad Schewe. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Schewe. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import sdatum
|
|
11
|
+
import utilo
|
|
12
|
+
|
|
13
|
+
import germania
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def find_abbrev(abbrev: str, words: list) -> str:
|
|
17
|
+
"""\
|
|
18
|
+
>>> find_abbrev('MNU', 'Steuergebiete zugunsten multinationaler Unternehmen'.split())
|
|
19
|
+
'multinationaler Unternehmen'
|
|
20
|
+
>>> import germania
|
|
21
|
+
>>> find_abbrev('MNU', germania.words_fromstr('Steuergebiete zugunsten multinationaler Unternehmens(MNU) ende'))
|
|
22
|
+
'multinationaler Unternehmens'
|
|
23
|
+
"""
|
|
24
|
+
lookup = sdatum.abbrev(abbrev)
|
|
25
|
+
if lookup is None:
|
|
26
|
+
return None
|
|
27
|
+
lookup = [germania.word_normalize(item) for item in lookup]
|
|
28
|
+
normalized = [
|
|
29
|
+
germania.word_normalize(item) if isinstance(item, str) else item
|
|
30
|
+
for item in words
|
|
31
|
+
]
|
|
32
|
+
detected = germania.searches(
|
|
33
|
+
patterns=lookup,
|
|
34
|
+
sentence=normalized,
|
|
35
|
+
tokens_complex=False,
|
|
36
|
+
)
|
|
37
|
+
if not detected:
|
|
38
|
+
return None
|
|
39
|
+
result = ' '.join(words[index] for index in utilo.rlist(*detected[0]))
|
|
40
|
+
return result
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2022-2023 by Helmut Konrad Schewe. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Schewe. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2022-2023 by Helmut Konrad Schewe. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Schewe. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import dataclasses
|
|
11
|
+
import enum
|
|
12
|
+
|
|
13
|
+
import iamraw
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class TextErrorType(enum.Enum):
|
|
17
|
+
MISSING = enum.auto()
|
|
18
|
+
"""Text token is missing"""
|
|
19
|
+
STYLE = enum.auto()
|
|
20
|
+
"""Violation against good style"""
|
|
21
|
+
RULE = enum.auto()
|
|
22
|
+
"""Writing is against the writing laws"""
|
|
23
|
+
DUPLICATED = enum.auto()
|
|
24
|
+
"""Copy paste error"""
|
|
25
|
+
REPLACEMENT = enum.auto()
|
|
26
|
+
"""Replace this content"""
|
|
27
|
+
UNDEFINED = enum.auto()
|
|
28
|
+
"""No special state"""
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclasses.dataclass
|
|
32
|
+
class TextError:
|
|
33
|
+
title: str = None
|
|
34
|
+
text: str = None
|
|
35
|
+
state: TextErrorType = None
|
|
36
|
+
location: iamraw.Location = None
|
|
37
|
+
raw: str = None
|
|
38
|
+
better: str = None
|
|
39
|
+
debug_method: str = None
|
|
40
|
+
"""Name of method which had determined this error."""
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
TextErrors = list[TextError]
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2022-2023 by Helmut Konrad Schewe. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Schewe. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import contextlib
|
|
11
|
+
|
|
12
|
+
import iamraw
|
|
13
|
+
import utilo
|
|
14
|
+
|
|
15
|
+
import germania.error.finding
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class TextErrorMachine:
|
|
19
|
+
"""\
|
|
20
|
+
>>> empty = TextErrorMachine()
|
|
21
|
+
>>> empty.determine('')
|
|
22
|
+
[]
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
def determine(
|
|
26
|
+
self,
|
|
27
|
+
text: str,
|
|
28
|
+
page: int = None,
|
|
29
|
+
) -> germania.error.finding.TextErrors:
|
|
30
|
+
result = []
|
|
31
|
+
todo = utilo.methods(self, starts='check_')
|
|
32
|
+
for method in todo:
|
|
33
|
+
detected = method(text)
|
|
34
|
+
if not detected:
|
|
35
|
+
continue
|
|
36
|
+
if not utilo.iterable(detected):
|
|
37
|
+
result.append(detected)
|
|
38
|
+
detected.debug_method = method.__name__
|
|
39
|
+
continue
|
|
40
|
+
for item in detected:
|
|
41
|
+
item.debug_method = method.__name__
|
|
42
|
+
result.extend(detected)
|
|
43
|
+
if page is not None:
|
|
44
|
+
for item in result:
|
|
45
|
+
with contextlib.suppress(AttributeError):
|
|
46
|
+
item.location.page = page
|
|
47
|
+
return result
|
|
48
|
+
|
|
49
|
+
def location(self, match) -> iamraw.RangedLocation: # pylint:disable=R0201
|
|
50
|
+
if not match:
|
|
51
|
+
return None
|
|
52
|
+
location = iamraw.RangedLocation(
|
|
53
|
+
char=match.span()[0],
|
|
54
|
+
char_end=match.span()[1],
|
|
55
|
+
)
|
|
56
|
+
return location
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class PhysicMachine(TextErrorMachine):
|
|
60
|
+
"""\
|
|
61
|
+
>>> machine = PhysicMachine()
|
|
62
|
+
>>> machine.determine('The weight is 200kg. Thats a lot.', page=10)
|
|
63
|
+
[TextError(...state=<TextErrorType.RULE...>, location=RangedLocation(page=10, char=13, char_end=19), raw='200kg', better='200 kg', debug_method='check_physical_spaces')]
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
MISSING_SPACE_BEFORE_UNIT = utilo.compiles(r"""
|
|
67
|
+
\W
|
|
68
|
+
(
|
|
69
|
+
(?P<value>\d+((\.|\,)\d+){0,1})
|
|
70
|
+
(?P<unit>%|‰|kg|km/h|mmHg|mm|ms|mg/km|m|cm|V|W|Hz|mW)
|
|
71
|
+
)
|
|
72
|
+
""")
|
|
73
|
+
|
|
74
|
+
def check_physical_spaces(self, text: str) -> list:
|
|
75
|
+
result = []
|
|
76
|
+
for match in self.MISSING_SPACE_BEFORE_UNIT.finditer(text):
|
|
77
|
+
error = germania.error.finding.TextError(
|
|
78
|
+
state=germania.error.finding.TextErrorType.RULE,
|
|
79
|
+
better=match['value'] + ' ' + match['unit'],
|
|
80
|
+
location=self.location(match),
|
|
81
|
+
raw=match[1],
|
|
82
|
+
)
|
|
83
|
+
result.append(error)
|
|
84
|
+
return result
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2022-2023 by Helmut Konrad Schewe. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Schewe. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
|
|
12
|
+
import sdatum
|
|
13
|
+
import utilo
|
|
14
|
+
|
|
15
|
+
import germania
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class TextMachine(germania.TextErrorMachine):
|
|
19
|
+
|
|
20
|
+
MISSING_PAGENUMBER = utilo.compiles(r"""
|
|
21
|
+
(
|
|
22
|
+
\s|
|
|
23
|
+
[^\w]\.| # u.s.
|
|
24
|
+
[,;:]
|
|
25
|
+
)
|
|
26
|
+
(
|
|
27
|
+
S\.|
|
|
28
|
+
p\.
|
|
29
|
+
)
|
|
30
|
+
[ ]{0,4}\n?
|
|
31
|
+
(?![\dixv\s]) # s. VII
|
|
32
|
+
""")
|
|
33
|
+
|
|
34
|
+
def check_pagenumber_complete(self, text: str) -> list:
|
|
35
|
+
r"""\
|
|
36
|
+
>>> check = TextMachine().check_pagenumber_complete
|
|
37
|
+
>>> check('Bonn, Herford 1993, S.\nDer Brief')
|
|
38
|
+
[TextError(...<TextErrorType.MISSING...raw='S.'...)]
|
|
39
|
+
>>> check('Hier fehlt wohlx S. die Seitennummer')
|
|
40
|
+
[TextError(...<TextErrorType.MISSING...raw='S.'...)]
|
|
41
|
+
>>> check('Vgl. Dixon, S.: Twitter: distribution of global audiences 2021')
|
|
42
|
+
[]
|
|
43
|
+
>>> check('Schols. Hier')
|
|
44
|
+
[]
|
|
45
|
+
>>> check('Berlin 19982, s. VII ')
|
|
46
|
+
[]
|
|
47
|
+
>>> check('S. Meuschel, Legitimation und Parteiherrschaft ')
|
|
48
|
+
[]
|
|
49
|
+
>>> check('in the U.S. | Pew Research')
|
|
50
|
+
[]
|
|
51
|
+
>>> check('Meuschel S., Legitimation und Parteiherrschaft ')
|
|
52
|
+
[]
|
|
53
|
+
|
|
54
|
+
run large text test
|
|
55
|
+
>>> import textbone;check(textbone.text_improved(1024*1024))
|
|
56
|
+
[]
|
|
57
|
+
"""
|
|
58
|
+
result = []
|
|
59
|
+
for match in self.MISSING_PAGENUMBER.finditer(text):
|
|
60
|
+
after = text[match.span()[1]:]
|
|
61
|
+
if follows_name(after):
|
|
62
|
+
continue
|
|
63
|
+
start, lookback = match.span()[0], 60
|
|
64
|
+
before = text[max(0, start - lookback):start]
|
|
65
|
+
if name_before(before):
|
|
66
|
+
continue
|
|
67
|
+
error = germania.TextError(
|
|
68
|
+
state=germania.TextErrorType.MISSING,
|
|
69
|
+
location=self.location(match),
|
|
70
|
+
raw=match[0].strip(),
|
|
71
|
+
)
|
|
72
|
+
result.append(error)
|
|
73
|
+
return result
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def follows_name(text: str) -> bool:
|
|
77
|
+
"""\
|
|
78
|
+
>>> follows_name('Helmut wohnt hier')
|
|
79
|
+
True
|
|
80
|
+
>>> follows_name('Copyright (c) 2022 by Helmut')
|
|
81
|
+
False
|
|
82
|
+
>>> follows_name('Der Brief')
|
|
83
|
+
False
|
|
84
|
+
"""
|
|
85
|
+
text = text.strip()
|
|
86
|
+
if re.match(r'^\w\.', text):
|
|
87
|
+
return True
|
|
88
|
+
for name in text.split()[0:4]:
|
|
89
|
+
name = name.strip(':;, ')
|
|
90
|
+
if sdatum.isname(name):
|
|
91
|
+
return True
|
|
92
|
+
return False
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def name_before(text: str) -> bool:
|
|
96
|
+
"""\
|
|
97
|
+
>>> name_before('Legitimation und Parteiherrschaft Meuschel')
|
|
98
|
+
True
|
|
99
|
+
"""
|
|
100
|
+
token = text.rstrip(':;, ').rsplit(maxsplit=1)
|
|
101
|
+
if not token:
|
|
102
|
+
return False
|
|
103
|
+
# select the right one
|
|
104
|
+
name = token[-1].strip(':;, ')
|
|
105
|
+
if sdatum.isname(name):
|
|
106
|
+
return True
|
|
107
|
+
return False
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2021-2023 by Helmut Konrad Schewe. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Schewe. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2021-2023 by Helmut Konrad Schewe. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Schewe. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import konradus
|
|
11
|
+
import utilo
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@utilo.cacheme
|
|
15
|
+
def abbreviation_magic(text: str) -> str:
|
|
16
|
+
"""\
|
|
17
|
+
>>> abbreviation_magic('Helmut hier u. a. und mehr.')
|
|
18
|
+
'Helmut hier u.a. und mehr.'
|
|
19
|
+
>>> abbreviation_magic('a. a. o.')
|
|
20
|
+
'a.a.o.'
|
|
21
|
+
"""
|
|
22
|
+
for token, replace in TEXT_MAGIC:
|
|
23
|
+
text = text.replace(token, replace)
|
|
24
|
+
return text
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
TEXT_MAGIC = [
|
|
28
|
+
('. '.join(item.split('.')).strip(), item)
|
|
29
|
+
for item in list(konradus.ABBREVIATION) + list(konradus.ABBREVIATION_LOWER)
|
|
30
|
+
if item.count('.') > 1
|
|
31
|
+
]
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2021-2023 by Helmut Konrad Schewe. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Schewe. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
|
|
12
|
+
import utilo
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@utilo.cacheme
|
|
16
|
+
def highnote_magic(text: str) -> str:
|
|
17
|
+
"""\
|
|
18
|
+
>>> highnote_magic('ohnehin unmöglich.89 So')
|
|
19
|
+
'ohnehin unmöglich. 89 So'
|
|
20
|
+
"""
|
|
21
|
+
# highnote at the end of sentence
|
|
22
|
+
text = re.sub(
|
|
23
|
+
r'([a-z])([\.\!\?])(\d{1,4})([ ]{1,4})',
|
|
24
|
+
r'\1\2 \3\4',
|
|
25
|
+
text,
|
|
26
|
+
flags=re.I,
|
|
27
|
+
)
|
|
28
|
+
return text
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2021-2023 by Helmut Konrad Schewe. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Schewe. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
|
|
12
|
+
import utilo
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@utilo.cacheme
|
|
16
|
+
def href_magic(text: str) -> str:
|
|
17
|
+
"""\
|
|
18
|
+
>>> href_magic('4. url: https : / / www . apache . org / licenses/LICENS')
|
|
19
|
+
'4. url: https://www.apache.org/licenses/LICENS'
|
|
20
|
+
"""
|
|
21
|
+
result = link_fink(text)
|
|
22
|
+
return result
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@utilo.cacheme
|
|
26
|
+
def link_fink(text: str) -> str:
|
|
27
|
+
"""\
|
|
28
|
+
>>> link_fink('url: http : / / www . bitkom . org / files / documents / BITKOM _ Leitfaden')
|
|
29
|
+
'url: http://www.bitkom.org/files/documents/BITKOM_Leitfaden'
|
|
30
|
+
>>> link_fink('singulären bzw. typischen')
|
|
31
|
+
'singulären bzw. typischen'
|
|
32
|
+
"""
|
|
33
|
+
if not IS_HTTP.search(text):
|
|
34
|
+
return text
|
|
35
|
+
for (token, replacement) in SPACE_PATTERN:
|
|
36
|
+
text = re.sub(token, replacement, text)
|
|
37
|
+
return text
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
IS_HTTP = utilo.compiles(r'\bhttp')
|
|
41
|
+
|
|
42
|
+
SPACE_PATTERN = (
|
|
43
|
+
('. org', '.org'),
|
|
44
|
+
('. de', '.de'),
|
|
45
|
+
('. com', '.com'),
|
|
46
|
+
('. net', '.net'),
|
|
47
|
+
(' .org', '.org'),
|
|
48
|
+
(' .de', '.de'),
|
|
49
|
+
(' .com', '.com'),
|
|
50
|
+
(' .net', '.net'),
|
|
51
|
+
(' : /', ':/'),
|
|
52
|
+
(' / ', '/'),
|
|
53
|
+
(':/ ', ':/'),
|
|
54
|
+
(' . ', '.'),
|
|
55
|
+
(' _ ', '_'),
|
|
56
|
+
('_ ', '_'),
|
|
57
|
+
('/ ', '/'),
|
|
58
|
+
('http :', 'http:'),
|
|
59
|
+
('https :', 'https:'),
|
|
60
|
+
('http ://', 'http://'),
|
|
61
|
+
('https ://', 'https://'),
|
|
62
|
+
('http: //', 'http://'),
|
|
63
|
+
('https: //', 'https://'),
|
|
64
|
+
)
|
|
65
|
+
SPACE_PATTERN = [(re.escape(left), right) for left, right in SPACE_PATTERN]
|
|
66
|
+
SPACE_PATTERN.append((
|
|
67
|
+
r'(?P<left>[a-z])\s{0,2}\.\s{0,2}(?P<right>[a-z])',
|
|
68
|
+
r'\g<left>.\g<right>',
|
|
69
|
+
))
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2021-2023 by Helmut Konrad Schewe. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Schewe. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import utilo
|
|
11
|
+
|
|
12
|
+
import germania.improve.abbreviation
|
|
13
|
+
import germania.improve.highnote
|
|
14
|
+
import germania.improve.href
|
|
15
|
+
|
|
16
|
+
TODO = (
|
|
17
|
+
germania.improve.abbreviation.abbreviation_magic,
|
|
18
|
+
germania.improve.highnote.highnote_magic,
|
|
19
|
+
germania.improve.href.href_magic,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@utilo.cacheme
|
|
24
|
+
def text_magic(text: str) -> str:
|
|
25
|
+
for pattern in TODO:
|
|
26
|
+
text = pattern(text)
|
|
27
|
+
return text
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2020-2023 by Helmut Konrad Schewe. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Schewe. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
"""Language Probability
|
|
10
|
+
====================
|
|
11
|
+
|
|
12
|
+
Use nltk to determine language where sentence is written in.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import collections
|
|
16
|
+
import contextlib
|
|
17
|
+
|
|
18
|
+
import konradus
|
|
19
|
+
import utilo
|
|
20
|
+
|
|
21
|
+
LanguageResult = collections.namedtuple(
|
|
22
|
+
'LanguageResult',
|
|
23
|
+
'language, probability',
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def determine(text: str) -> LanguageResult:
|
|
28
|
+
if isinstance(text, list):
|
|
29
|
+
text = konradus.remove_marks(text)
|
|
30
|
+
if not isinstance(text, str):
|
|
31
|
+
text = ' '.join(text)
|
|
32
|
+
cat = textcat()
|
|
33
|
+
detected = cat.guess_language(text)
|
|
34
|
+
language = konradus.Language.UNKNOWN
|
|
35
|
+
with contextlib.suppress(KeyError):
|
|
36
|
+
language = MAPPING[detected]
|
|
37
|
+
return LanguageResult(language=language, probability=1.0)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def isfre(tokens: str) -> bool:
|
|
41
|
+
"""\
|
|
42
|
+
>>> isfre('Ich bin Helmut')
|
|
43
|
+
False
|
|
44
|
+
>>> isfre('Bonjour monsieur.')
|
|
45
|
+
True
|
|
46
|
+
|
|
47
|
+
verify that interface support tokens
|
|
48
|
+
>>> isfre('Toujour suis Luis.'.split())
|
|
49
|
+
True
|
|
50
|
+
"""
|
|
51
|
+
return determine(tokens).language == konradus.Language.FRENCH
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def iseng(tokens: str) -> bool:
|
|
55
|
+
"""\
|
|
56
|
+
>>> iseng('Ich bin Helmut')
|
|
57
|
+
False
|
|
58
|
+
>>> iseng('i like fish')
|
|
59
|
+
True
|
|
60
|
+
"""
|
|
61
|
+
return determine(tokens).language == konradus.Language.ENGLISH
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def isger(tokens: str) -> bool:
|
|
65
|
+
"""\
|
|
66
|
+
>>> isger('Kartoffelsalat')
|
|
67
|
+
True
|
|
68
|
+
"""
|
|
69
|
+
return determine(tokens).language == konradus.Language.GERMAN
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
MAPPING = {
|
|
73
|
+
'deu': konradus.Language.GERMAN,
|
|
74
|
+
'eng': konradus.Language.ENGLISH,
|
|
75
|
+
# 'es': konradus.Language.SPANISH,
|
|
76
|
+
'fra': konradus.Language.FRENCH,
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@utilo.cacheme
|
|
81
|
+
def textcat():
|
|
82
|
+
import nltk.classify.textcat
|
|
83
|
+
result = nltk.classify.textcat.TextCat()
|
|
84
|
+
return result
|