doctextstyle 5.20.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- doctextstyle-5.20.5/LICENSE +21 -0
- doctextstyle-5.20.5/PKG-INFO +46 -0
- doctextstyle-5.20.5/README +16 -0
- doctextstyle-5.20.5/doctextstyle/__init__.py +23 -0
- doctextstyle-5.20.5/doctextstyle/__patch__.py +8 -0
- doctextstyle-5.20.5/doctextstyle/cli.py +72 -0
- doctextstyle-5.20.5/doctextstyle/cluster.py +166 -0
- doctextstyle-5.20.5/doctextstyle/extractor.py +178 -0
- doctextstyle-5.20.5/doctextstyle/features/__init__.py +51 -0
- doctextstyle-5.20.5/doctextstyle/features/blockquote.py +69 -0
- doctextstyle-5.20.5/doctextstyle/features/content.py +34 -0
- doctextstyle-5.20.5/doctextstyle/features/footnote.py +64 -0
- doctextstyle-5.20.5/doctextstyle/features/headline.py +161 -0
- doctextstyle-5.20.5/doctextstyle/features/pagesize.py +36 -0
- doctextstyle-5.20.5/doctextstyle/features/paragraph.py +62 -0
- doctextstyle-5.20.5/doctextstyle/features/textbounding.py +67 -0
- doctextstyle-5.20.5/doctextstyle/parser.py +394 -0
- doctextstyle-5.20.5/doctextstyle/path.py +14 -0
- doctextstyle-5.20.5/doctextstyle/utils.py +88 -0
- doctextstyle-5.20.5/doctextstyle/vector/__init__.py +38 -0
- doctextstyle-5.20.5/doctextstyle/vector/decide.py +75 -0
- doctextstyle-5.20.5/doctextstyle/vector/extract.py +155 -0
- doctextstyle-5.20.5/doctextstyle/vector/headlines.py +274 -0
- doctextstyle-5.20.5/doctextstyle/vector/prepare.py +112 -0
- doctextstyle-5.20.5/doctextstyle.egg-info/PKG-INFO +46 -0
- doctextstyle-5.20.5/doctextstyle.egg-info/SOURCES.txt +37 -0
- doctextstyle-5.20.5/doctextstyle.egg-info/dependency_links.txt +1 -0
- doctextstyle-5.20.5/doctextstyle.egg-info/entry_points.txt +3 -0
- doctextstyle-5.20.5/doctextstyle.egg-info/requires.txt +16 -0
- doctextstyle-5.20.5/doctextstyle.egg-info/top_level.txt +3 -0
- doctextstyle-5.20.5/hey/__init__.py +16 -0
- doctextstyle-5.20.5/magic/__init__.py +19 -0
- doctextstyle-5.20.5/magic/cli.py +61 -0
- doctextstyle-5.20.5/magic/feature/__init__.py +8 -0
- doctextstyle-5.20.5/magic/feature/content.py +224 -0
- doctextstyle-5.20.5/magic/feature/oneline.py +28 -0
- doctextstyle-5.20.5/magic/path.py +18 -0
- doctextstyle-5.20.5/pyproject.toml +111 -0
- doctextstyle-5.20.5/setup.cfg +4 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Helmut Konrad Schewe
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: doctextstyle
|
|
3
|
+
Version: 5.20.5
|
|
4
|
+
Author-email: Helmut Konrad Schewe <helmutus@outlook.com>
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Project-URL: Homepage, https://github.com/anaticulae/hey
|
|
7
|
+
Project-URL: Repository, https://github.com/anaticulae/hey
|
|
8
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
11
|
+
Requires-Python: >=3.12
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
License-File: LICENSE
|
|
14
|
+
Requires-Dist: iamraw<5.0.0,>=4.92.0
|
|
15
|
+
Requires-Dist: utilo<3.0.0,>=2.109.1
|
|
16
|
+
Requires-Dist: configos<2.0.0,>=1.0.4
|
|
17
|
+
Requires-Dist: groupmes<2.0.0,>=1.1.0
|
|
18
|
+
Requires-Dist: elementae<2.0.0,>=1.0.1
|
|
19
|
+
Requires-Dist: scipy<2.0.0,>=1.18.0
|
|
20
|
+
Provides-Extra: dev
|
|
21
|
+
Requires-Dist: rawmaker==2.40.5; extra == "dev"
|
|
22
|
+
Requires-Dist: hoverpower==1.5.2; extra == "dev"
|
|
23
|
+
Requires-Dist: jamer==1.0.2; extra == "dev"
|
|
24
|
+
Requires-Dist: indagator==1.0.1; extra == "dev"
|
|
25
|
+
Requires-Dist: gennex==1.0.3; extra == "dev"
|
|
26
|
+
Requires-Dist: utilotest==1.0.4; extra == "dev"
|
|
27
|
+
Requires-Dist: ibidem<2.0.0,>=1.0.2; extra == "dev"
|
|
28
|
+
Requires-Dist: pagenumber<2.0.0,>=1.0.0; extra == "dev"
|
|
29
|
+
Dynamic: license-file
|
|
30
|
+
|
|
31
|
+
# hey
|
|
32
|
+
|
|
33
|
+
`Hey` takes simple data from different data providers and creates `higher
|
|
34
|
+
level` data out of it.
|
|
35
|
+
|
|
36
|
+
## Features
|
|
37
|
+
|
|
38
|
+
### groupme
|
|
39
|
+
|
|
40
|
+
* chapter: split document text by chapter
|
|
41
|
+
* structure: divide document in head(toc) and tail(rest)
|
|
42
|
+
* toc: create table of content out of raw text data
|
|
43
|
+
|
|
44
|
+
## CHECK THIS
|
|
45
|
+
|
|
46
|
+
(inspired from sklearn.pipeline.Pipeline)
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
# hey
|
|
2
|
+
|
|
3
|
+
`Hey` takes simple data from different data providers and creates `higher
|
|
4
|
+
level` data out of it.
|
|
5
|
+
|
|
6
|
+
## Features
|
|
7
|
+
|
|
8
|
+
### groupme
|
|
9
|
+
|
|
10
|
+
* chapter: split document text by chapter
|
|
11
|
+
* structure: divide document in head(toc) and tail(rest)
|
|
12
|
+
* toc: create table of content out of raw text data
|
|
13
|
+
|
|
14
|
+
## CHECK THIS
|
|
15
|
+
|
|
16
|
+
(inspired from sklearn.pipeline.Pipeline)
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2020-2022 by Helmut Konrad Fahrendholz. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import os
|
|
11
|
+
|
|
12
|
+
import doctextstyle.__patch__
|
|
13
|
+
import hey
|
|
14
|
+
from doctextstyle.vector.extract import run as extract_headlines # pylint:disable=C0412
|
|
15
|
+
from doctextstyle.vector.extract import run_fromdata as headlines_fromdata # pylint:disable=C0412
|
|
16
|
+
|
|
17
|
+
__version__ = hey.__version__
|
|
18
|
+
|
|
19
|
+
ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), '..'))
|
|
20
|
+
PROCESS = 'doctextstyle'
|
|
21
|
+
|
|
22
|
+
JUSTIFIED = 1
|
|
23
|
+
NOT_JUSTIFIED = 0
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2020-2022 by Helmut Konrad Fahrendholz. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2020-2022 by Helmut Konrad Fahrendholz. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import os
|
|
11
|
+
|
|
12
|
+
import serializeraw
|
|
13
|
+
import utilo
|
|
14
|
+
import utilo.cli
|
|
15
|
+
|
|
16
|
+
import doctextstyle
|
|
17
|
+
import doctextstyle.extractor
|
|
18
|
+
|
|
19
|
+
DEFAULT_OUTPUT_FILE = 'doctextstyle__textstyle.yaml'
|
|
20
|
+
|
|
21
|
+
DESCRIPTION = """\
|
|
22
|
+
Extracts common text layout properties for text, headlines(h1, h2, h3),
|
|
23
|
+
pagenumber, footnotes, paragraphs and lists.
|
|
24
|
+
|
|
25
|
+
Requires: `rawmaker__oneline` data to create pagetextnavigators
|
|
26
|
+
`rawmaker__oneline` data to create pagecontenttextnavigators
|
|
27
|
+
|
|
28
|
+
Writes: `doctextstyle__textstyle.yaml`
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@utilo.saveme
|
|
33
|
+
def main() -> int:
|
|
34
|
+
commands = []
|
|
35
|
+
parser = utilo.cli.create_parser(
|
|
36
|
+
todo=commands,
|
|
37
|
+
description=DESCRIPTION,
|
|
38
|
+
config=utilo.ParserConfiguration(
|
|
39
|
+
inputparameter=True,
|
|
40
|
+
multiprocessed=True,
|
|
41
|
+
outputparameter=True,
|
|
42
|
+
pages=True,
|
|
43
|
+
prefix=False,
|
|
44
|
+
),
|
|
45
|
+
version=doctextstyle.__version__,
|
|
46
|
+
prog=doctextstyle.PROCESS,
|
|
47
|
+
)
|
|
48
|
+
args = utilo.parse(parser)
|
|
49
|
+
utilo.log('run doctextstyle')
|
|
50
|
+
selected_pages = utilo.pages_fromargs(args)
|
|
51
|
+
|
|
52
|
+
inpath, outpath = utilo.cli.sources(args, singleinput=True) # pylint:disable=W0632
|
|
53
|
+
inpath = inpath[0]
|
|
54
|
+
|
|
55
|
+
if os.path.isdir(outpath):
|
|
56
|
+
outpath = os.path.join(outpath, DEFAULT_OUTPUT_FILE)
|
|
57
|
+
|
|
58
|
+
# ensure that output directory exists
|
|
59
|
+
parent, _ = os.path.split(outpath)
|
|
60
|
+
os.makedirs(parent, exist_ok=True)
|
|
61
|
+
|
|
62
|
+
try:
|
|
63
|
+
extracted = doctextstyle.extractor.extract(inpath, pages=selected_pages)
|
|
64
|
+
except FileNotFoundError as error:
|
|
65
|
+
utilo.error('missing input location')
|
|
66
|
+
utilo.error(f'{error}')
|
|
67
|
+
return utilo.FAILURE
|
|
68
|
+
|
|
69
|
+
dumped = serializeraw.dump_doctextstyle(extracted)
|
|
70
|
+
utilo.file_replace(outpath, dumped)
|
|
71
|
+
utilo.log('completed')
|
|
72
|
+
return utilo.SUCCESS
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2020-2022 by Helmut Konrad Fahrendholz. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import collections
|
|
11
|
+
import enum
|
|
12
|
+
import functools
|
|
13
|
+
|
|
14
|
+
import configos
|
|
15
|
+
import iamraw
|
|
16
|
+
import utilo
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class ClusterProperty(enum.Enum):
|
|
20
|
+
SIZE = enum.auto()
|
|
21
|
+
FONT = enum.auto()
|
|
22
|
+
BEFORE = enum.auto()
|
|
23
|
+
AFTER = enum.auto()
|
|
24
|
+
YPOS = enum.auto()
|
|
25
|
+
LEFT = enum.auto()
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
ClusterPropertySelection = list[ClusterProperty]
|
|
29
|
+
Tol = collections.namedtuple('Tol', 'abs, rel')
|
|
30
|
+
NO_TOLERANCE = Tol(0.0, 0.0)
|
|
31
|
+
|
|
32
|
+
CLUSTER_SIZE_MIN = configos.HV_INT_PLUS(default=5)
|
|
33
|
+
|
|
34
|
+
CLUSTER_SIZE_DIFF_ABS = configos.HV_FLOAT_PLUS(default=0.5)
|
|
35
|
+
|
|
36
|
+
CLUSTER_SIZE_DIFF_REL = configos.HV_FLOAT_PLUS(default=0.1)
|
|
37
|
+
|
|
38
|
+
CLUSTER_AFTER_DIFF_ABS = configos.HV_FLOAT_PLUS(default=2.0)
|
|
39
|
+
|
|
40
|
+
CLUSTER_AFTER_DIFF_REL = configos.HV_FLOAT_PLUS(default=0.1)
|
|
41
|
+
|
|
42
|
+
CLUSTER_BEFORE_DIFF_ABS = configos.HV_FLOAT_PLUS(default=2.0)
|
|
43
|
+
|
|
44
|
+
CLUSTER_BEFORE_DIFF_REL = configos.HV_FLOAT_PLUS(default=0.1)
|
|
45
|
+
|
|
46
|
+
CLUSTER_LEFT_DIFF_ABS = configos.HV_FLOAT_PLUS(default=15.0)
|
|
47
|
+
|
|
48
|
+
CLUSTER_LEFT_DIFF_REL = configos.HV_FLOAT_PLUS(default=0.15)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def cluster(
|
|
52
|
+
items: iamraw.TextProperties,
|
|
53
|
+
selection: ClusterPropertySelection = None,
|
|
54
|
+
validator: callable = None,
|
|
55
|
+
*,
|
|
56
|
+
minsize: int = CLUSTER_SIZE_MIN,
|
|
57
|
+
unique_content: bool = False,
|
|
58
|
+
max_size_diff=Tol(CLUSTER_SIZE_DIFF_ABS, CLUSTER_SIZE_DIFF_REL),
|
|
59
|
+
max_after_diff=Tol(CLUSTER_AFTER_DIFF_ABS, CLUSTER_AFTER_DIFF_REL),
|
|
60
|
+
max_before_diff=Tol(CLUSTER_BEFORE_DIFF_ABS, CLUSTER_BEFORE_DIFF_REL),
|
|
61
|
+
max_left_diff=Tol(CLUSTER_LEFT_DIFF_ABS, CLUSTER_LEFT_DIFF_REL),
|
|
62
|
+
):
|
|
63
|
+
if selection:
|
|
64
|
+
selection = set(selection)
|
|
65
|
+
if validator:
|
|
66
|
+
items = [item for item in items if validator(item)]
|
|
67
|
+
|
|
68
|
+
decider = functools.partial(
|
|
69
|
+
classifier,
|
|
70
|
+
selection=selection,
|
|
71
|
+
max_size_diff=max_size_diff,
|
|
72
|
+
max_after_diff=max_after_diff,
|
|
73
|
+
max_before_diff=max_before_diff,
|
|
74
|
+
max_left_diff=max_left_diff,
|
|
75
|
+
)
|
|
76
|
+
clustered = utilo.determine_cluster(
|
|
77
|
+
todo=items,
|
|
78
|
+
classifier=decider,
|
|
79
|
+
min_elements=minsize,
|
|
80
|
+
)
|
|
81
|
+
if unique_content:
|
|
82
|
+
clustered = [item for item in clustered if iscontent_unique(item)]
|
|
83
|
+
return clustered
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def classifier( # pylint:disable=R1260
|
|
87
|
+
candidat,
|
|
88
|
+
clusteritem,
|
|
89
|
+
selection,
|
|
90
|
+
max_size_diff=None,
|
|
91
|
+
max_after_diff=None,
|
|
92
|
+
max_before_diff=None,
|
|
93
|
+
max_left_diff=None,
|
|
94
|
+
) -> bool:
|
|
95
|
+
if selection is None or ClusterProperty.SIZE in selection:
|
|
96
|
+
if not utilo.pnear(
|
|
97
|
+
candidat.size,
|
|
98
|
+
clusteritem.size,
|
|
99
|
+
abs_tol=max_size_diff.abs,
|
|
100
|
+
rel_tol=max_size_diff.rel,
|
|
101
|
+
):
|
|
102
|
+
return False
|
|
103
|
+
if selection is None or ClusterProperty.BEFORE in selection:
|
|
104
|
+
if not utilo.pnear(
|
|
105
|
+
candidat.before,
|
|
106
|
+
clusteritem.before,
|
|
107
|
+
abs_tol=max_before_diff.abs,
|
|
108
|
+
rel_tol=max_before_diff.rel,
|
|
109
|
+
):
|
|
110
|
+
return False
|
|
111
|
+
if selection is None or ClusterProperty.AFTER in selection:
|
|
112
|
+
if not utilo.pnear(
|
|
113
|
+
candidat.after,
|
|
114
|
+
clusteritem.after,
|
|
115
|
+
abs_tol=max_after_diff.abs,
|
|
116
|
+
rel_tol=max_after_diff.rel,
|
|
117
|
+
):
|
|
118
|
+
return False
|
|
119
|
+
if selection is None or ClusterProperty.LEFT in selection:
|
|
120
|
+
if not utilo.pnear(
|
|
121
|
+
candidat.left,
|
|
122
|
+
clusteritem.left,
|
|
123
|
+
abs_tol=max_left_diff.abs,
|
|
124
|
+
rel_tol=max_left_diff.rel,
|
|
125
|
+
):
|
|
126
|
+
return False
|
|
127
|
+
if selection is None or ClusterProperty.FONT in selection:
|
|
128
|
+
if candidat.font != clusteritem.font:
|
|
129
|
+
return False
|
|
130
|
+
return True
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def iscontent_unique(current) -> bool:
|
|
134
|
+
expected = len(current.content)
|
|
135
|
+
current = {item.hashed for item in current.content}
|
|
136
|
+
return len(current) == expected
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def bestmatch(clustered, number: int = 0):
|
|
140
|
+
try:
|
|
141
|
+
largest_cluster_first_item = clustered[number].content[0]
|
|
142
|
+
except IndexError:
|
|
143
|
+
# no clusted content
|
|
144
|
+
return None
|
|
145
|
+
size = largest_cluster_first_item.size
|
|
146
|
+
font = largest_cluster_first_item.font
|
|
147
|
+
# TODO: PLUS INF OR MINUS INF
|
|
148
|
+
before = [item.before for item in clustered[number].content]
|
|
149
|
+
before = [utilo.INF if item is None else item for item in before]
|
|
150
|
+
before = utilo.mode(before)
|
|
151
|
+
after = [item.after for item in clustered[number].content]
|
|
152
|
+
after = [utilo.INF if item is None else item for item in after]
|
|
153
|
+
after = utilo.mode(after)
|
|
154
|
+
length = len(clustered[0])
|
|
155
|
+
|
|
156
|
+
if before is utilo.INF:
|
|
157
|
+
before = None
|
|
158
|
+
if after is utilo.INF:
|
|
159
|
+
after = None
|
|
160
|
+
return (size, font, length, (before, after))
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def remove(flats, toremove: list):
|
|
164
|
+
toremove = set(toremove)
|
|
165
|
+
flats = [item for item in flats if item not in toremove]
|
|
166
|
+
return flats
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2020-2022 by Helmut Konrad Fahrendholz. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import contextlib
|
|
11
|
+
import os
|
|
12
|
+
|
|
13
|
+
import iamraw
|
|
14
|
+
import serializeraw
|
|
15
|
+
import texmex
|
|
16
|
+
import utilo
|
|
17
|
+
|
|
18
|
+
import doctextstyle
|
|
19
|
+
import doctextstyle.features
|
|
20
|
+
import doctextstyle.features.blockquote
|
|
21
|
+
import doctextstyle.features.content
|
|
22
|
+
import doctextstyle.features.footnote
|
|
23
|
+
import doctextstyle.features.headline
|
|
24
|
+
import doctextstyle.features.pagesize
|
|
25
|
+
import doctextstyle.features.textbounding as dtt
|
|
26
|
+
import doctextstyle.parser
|
|
27
|
+
import doctextstyle.utils
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def extract(path: str, pages: tuple = None) -> iamraw.DocTextStyle: # pylint:disable=R0914,R0915
|
|
31
|
+
navigator, cnavigators, magic = load_data(path, pages=pages)
|
|
32
|
+
|
|
33
|
+
parsed = doctextstyle.parser.parses(navigator, magic)
|
|
34
|
+
flat = doctextstyle.utils.flatten(parsed)
|
|
35
|
+
|
|
36
|
+
text = doctextstyle.features.text(flat)
|
|
37
|
+
if not text or len(text) < 4:
|
|
38
|
+
utilo.error('not enough text data')
|
|
39
|
+
return iamraw.DocTextStyle()
|
|
40
|
+
text_after = text[3][1]
|
|
41
|
+
|
|
42
|
+
result = iamraw.DocTextStyle(
|
|
43
|
+
text_size=text[0],
|
|
44
|
+
text_distance=text_after,
|
|
45
|
+
text_family=text[1],
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
extract_headlines(result, flat)
|
|
49
|
+
|
|
50
|
+
pagenumber = doctextstyle.features.pagenumber(flat)
|
|
51
|
+
if pagenumber:
|
|
52
|
+
result.pagenumber_size = pagenumber[0]
|
|
53
|
+
result.pagenumber_family = pagenumber[1]
|
|
54
|
+
|
|
55
|
+
extract_footnotes(result, flat)
|
|
56
|
+
|
|
57
|
+
pagesizes = doctextstyle.features.pagesize.pagesizes(path, pages=pages)
|
|
58
|
+
|
|
59
|
+
if pagesizes:
|
|
60
|
+
result.page_width, result.page_height = pagesizes[0][0]
|
|
61
|
+
with contextlib.suppress(IndexError):
|
|
62
|
+
result.page_rotated_width, result.page_rotated_height = pagesizes[1][0] # yapf:disable
|
|
63
|
+
else:
|
|
64
|
+
utilo.error('no pagesize, too few pages to run feature')
|
|
65
|
+
|
|
66
|
+
extract_contentborder(result, path, pages)
|
|
67
|
+
extract_textdimension(result, navigator, cnavigators)
|
|
68
|
+
extract_blockquote(result, flat)
|
|
69
|
+
return result
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def load_data(path: str, pages: tuple = None):
|
|
73
|
+
navigator = serializeraw.ptn_frompath(
|
|
74
|
+
path,
|
|
75
|
+
prefix='oneline',
|
|
76
|
+
pages=pages,
|
|
77
|
+
)
|
|
78
|
+
try:
|
|
79
|
+
cnavigators = serializeraw.ptcn_frompath(
|
|
80
|
+
path,
|
|
81
|
+
prefix='oneline',
|
|
82
|
+
pages=pages,
|
|
83
|
+
)
|
|
84
|
+
except FileNotFoundError as error:
|
|
85
|
+
cnavigators = None
|
|
86
|
+
utilo.error(f'missing page text content navigator: {error}')
|
|
87
|
+
|
|
88
|
+
magic = iamraw.path.magic_content(path)
|
|
89
|
+
magic = serializeraw.load_types(magic) if os.path.exists(magic) else []
|
|
90
|
+
return navigator, cnavigators, magic
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def extract_footnotes(result, flat):
|
|
94
|
+
footnotes = doctextstyle.features.footnote.footnote(flat)
|
|
95
|
+
if footnotes:
|
|
96
|
+
footnote_after = footnotes[3][1]
|
|
97
|
+
result.footnote_size = footnotes[0]
|
|
98
|
+
result.footnote_family = footnotes[1]
|
|
99
|
+
result.footnote_distance = footnote_after
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def extract_contentborder(result, path, pages):
|
|
103
|
+
leftright = iamraw.path.groupme_border_leftright(path)
|
|
104
|
+
content = doctextstyle.features.content.content(leftright, pages=pages)
|
|
105
|
+
if content:
|
|
106
|
+
normal = content[0][0]
|
|
107
|
+
result.content_left = normal[0]
|
|
108
|
+
result.content_right = normal[1]
|
|
109
|
+
result.content_top = normal[2]
|
|
110
|
+
result.content_bottom = normal[3]
|
|
111
|
+
if len(content) > 1:
|
|
112
|
+
rotated = content[1][0]
|
|
113
|
+
result.content_rotated_left = rotated[0]
|
|
114
|
+
result.content_rotated_right = rotated[1]
|
|
115
|
+
result.content_rotated_top = rotated[2]
|
|
116
|
+
result.content_rotated_bottom = rotated[3]
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def extract_headlines(result, flat):
|
|
120
|
+
headlines = doctextstyle.features.headline.headlines(flat)
|
|
121
|
+
if not headlines:
|
|
122
|
+
return
|
|
123
|
+
|
|
124
|
+
result.h1_size = headlines[0][0]
|
|
125
|
+
result.h1_family = headlines[0][1]
|
|
126
|
+
result.h1_before = headlines[0][3][0]
|
|
127
|
+
result.h1_after = headlines[0][3][1]
|
|
128
|
+
|
|
129
|
+
if len(headlines) == 1:
|
|
130
|
+
return
|
|
131
|
+
result.h2_size = headlines[1][0]
|
|
132
|
+
result.h2_family = headlines[1][1]
|
|
133
|
+
result.h2_before = headlines[1][3][0]
|
|
134
|
+
result.h2_after = headlines[1][3][1]
|
|
135
|
+
|
|
136
|
+
if len(headlines) == 2:
|
|
137
|
+
return
|
|
138
|
+
result.h3_size = headlines[2][0]
|
|
139
|
+
result.h3_family = headlines[2][1]
|
|
140
|
+
result.h3_before = headlines[2][3][0]
|
|
141
|
+
result.h3_after = headlines[2][3][1]
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def extract_textdimension(result, navigators, cnavigators):
|
|
145
|
+
if not cnavigators:
|
|
146
|
+
return
|
|
147
|
+
twidth = dtt.text_width(cnavigators)
|
|
148
|
+
twidth_min = dtt.text_width_min(cnavigators)
|
|
149
|
+
twidth_max = dtt.text_width_max(cnavigators)
|
|
150
|
+
|
|
151
|
+
result.text_width = twidth
|
|
152
|
+
result.text_width_min = twidth_min
|
|
153
|
+
result.text_width_max = twidth_max
|
|
154
|
+
|
|
155
|
+
result.text_left = texmex.document_textfeed(cnavigators)
|
|
156
|
+
result.text_right = texmex.document_textfeed(navigators, left=False)
|
|
157
|
+
|
|
158
|
+
result.text_left = utilo.roundme(result.text_left)
|
|
159
|
+
result.text_right = utilo.roundme(result.text_right)
|
|
160
|
+
|
|
161
|
+
if result.content_right is None:
|
|
162
|
+
return
|
|
163
|
+
# content_right is None for very short documents, cause there are
|
|
164
|
+
# a minimum number of pages required to extract this properly.
|
|
165
|
+
right = result.page_width - result.content_right
|
|
166
|
+
result.text_alignment = dtt.justified(cnavigators, right)
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def extract_blockquote(result, flat):
|
|
170
|
+
style = doctextstyle.features.blockquote.blockquote_style(flat)
|
|
171
|
+
if not style:
|
|
172
|
+
return
|
|
173
|
+
font, size, left, right = style
|
|
174
|
+
|
|
175
|
+
result.block_size = size
|
|
176
|
+
result.block_family = font
|
|
177
|
+
result.block_left = left
|
|
178
|
+
result.block_right = right
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2020-2022 by Helmut Konrad Fahrendholz. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import doctextstyle.cluster
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def text(flats, returncluster: bool = False):
|
|
14
|
+
clustered = doctextstyle.cluster.cluster(
|
|
15
|
+
flats,
|
|
16
|
+
(
|
|
17
|
+
doctextstyle.cluster.ClusterProperty.SIZE,
|
|
18
|
+
doctextstyle.cluster.ClusterProperty.FONT,
|
|
19
|
+
),
|
|
20
|
+
)
|
|
21
|
+
result = doctextstyle.cluster.bestmatch(clustered)
|
|
22
|
+
if not result:
|
|
23
|
+
# too few data to determine text style information
|
|
24
|
+
return None
|
|
25
|
+
if returncluster:
|
|
26
|
+
return result, clustered[0] if clustered else []
|
|
27
|
+
return result
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def pagenumber(flats, returncluster: bool = False):
|
|
31
|
+
|
|
32
|
+
def validator(item) -> bool:
|
|
33
|
+
if item.top >= 100 and item.bottom >= 100:
|
|
34
|
+
# page number is not in the middle of the page. The page
|
|
35
|
+
# number is located at the top or bottom of the page.
|
|
36
|
+
return False
|
|
37
|
+
return item.length <= 6
|
|
38
|
+
|
|
39
|
+
clustered = doctextstyle.cluster.cluster(
|
|
40
|
+
flats,
|
|
41
|
+
(
|
|
42
|
+
doctextstyle.cluster.ClusterProperty.SIZE,
|
|
43
|
+
doctextstyle.cluster.ClusterProperty.FONT,
|
|
44
|
+
),
|
|
45
|
+
validator=validator,
|
|
46
|
+
)
|
|
47
|
+
# assert len(clustered) == 1, len(clustered)
|
|
48
|
+
result = doctextstyle.cluster.bestmatch(clustered)
|
|
49
|
+
if returncluster:
|
|
50
|
+
return result, clustered[0] if clustered else []
|
|
51
|
+
return result
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2020-2022 by Helmut Konrad Fahrendholz. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import configos
|
|
11
|
+
import utilo
|
|
12
|
+
|
|
13
|
+
BLOCKQUOTE_CLUSTER_SIZE_MIN = configos.HV_INT_PLUS(default=30)
|
|
14
|
+
|
|
15
|
+
BLOCKQUOTE_LEFT_DIFF_MAX = configos.HV_FLOAT_PLUS(default=15.0)
|
|
16
|
+
|
|
17
|
+
BLOCKQUOTE_RIGHT_DIFF_MAX = configos.HV_FLOAT_PLUS(default=15.0)
|
|
18
|
+
|
|
19
|
+
BLOCKQUOTE_TEXT_LENGTH_MIN = configos.HV_INT_PLUS(default=45)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def blockquote_style(flats) -> tuple:
|
|
23
|
+
# TODO: REMOVE TEXT CLUSTER FIRST
|
|
24
|
+
flats = [item for item in flats if item.hashed.count('.') < 10]
|
|
25
|
+
# skip very short text
|
|
26
|
+
flats = [
|
|
27
|
+
item for item in flats if len(item.hashed) > BLOCKQUOTE_TEXT_LENGTH_MIN
|
|
28
|
+
]
|
|
29
|
+
# skip item which start too right
|
|
30
|
+
flats = [item for item in flats if item.left <= 160]
|
|
31
|
+
# cluster text
|
|
32
|
+
clustered = utilo.determine_cluster(
|
|
33
|
+
flats,
|
|
34
|
+
classifier=classifier,
|
|
35
|
+
min_elements=BLOCKQUOTE_CLUSTER_SIZE_MIN,
|
|
36
|
+
strategy=utilo.MatchStrategy.MIN,
|
|
37
|
+
)
|
|
38
|
+
if len(clustered) < 2:
|
|
39
|
+
# too few cluster
|
|
40
|
+
return None
|
|
41
|
+
# largest cluster is text cluster
|
|
42
|
+
blockquote = clustered[1].center
|
|
43
|
+
result = (
|
|
44
|
+
blockquote.font,
|
|
45
|
+
blockquote.size,
|
|
46
|
+
blockquote.left,
|
|
47
|
+
blockquote.right,
|
|
48
|
+
)
|
|
49
|
+
return result
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def classifier(candidat, clusteritem):
|
|
53
|
+
# expected
|
|
54
|
+
x0 = clusteritem.left
|
|
55
|
+
x1 = clusteritem.right
|
|
56
|
+
# current
|
|
57
|
+
x00 = candidat.left
|
|
58
|
+
x11 = candidat.right
|
|
59
|
+
# compare font size
|
|
60
|
+
if candidat.size != clusteritem.size:
|
|
61
|
+
return False
|
|
62
|
+
leftdiff = utilo.near(x00, x0, diff=BLOCKQUOTE_LEFT_DIFF_MAX)
|
|
63
|
+
if not leftdiff:
|
|
64
|
+
return False
|
|
65
|
+
rightdiff = utilo.near(x11, x1, diff=BLOCKQUOTE_RIGHT_DIFF_MAX)
|
|
66
|
+
if not rightdiff:
|
|
67
|
+
return False
|
|
68
|
+
# merge candidat into cluster
|
|
69
|
+
return True
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2020-2022 by Helmut Konrad Fahrendholz. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import configos
|
|
11
|
+
import serializeraw
|
|
12
|
+
import utilo
|
|
13
|
+
|
|
14
|
+
CONTENT_DIFF_MAX = configos.HV_FLOAT_PLUS(default=5.0)
|
|
15
|
+
|
|
16
|
+
CONTENT_ELEMENTS_MIN = configos.HV_INT_PLUS(default=3)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def content(path, pages: tuple = None):
|
|
20
|
+
leftright = serializeraw.load_leftright_border(path, pages)
|
|
21
|
+
|
|
22
|
+
def equals(candidat, clusteritem):
|
|
23
|
+
# left, right, top, down
|
|
24
|
+
distance = utilo.norms(candidat, clusteritem)
|
|
25
|
+
return distance < CONTENT_DIFF_MAX
|
|
26
|
+
|
|
27
|
+
clustered = utilo.determine_cluster(
|
|
28
|
+
leftright.values(),
|
|
29
|
+
classifier=equals,
|
|
30
|
+
min_elements=CONTENT_ELEMENTS_MIN,
|
|
31
|
+
)
|
|
32
|
+
# TODO: SUPPORT LEFT AND RIGHT DIFFERENT PAGE?
|
|
33
|
+
result = [(cluster[0], len(cluster)) for cluster in clustered]
|
|
34
|
+
return result
|