protoerror 3.20.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- protoerror/__init__.py +133 -0
- protoerror/__patch__.py +8 -0
- protoerror/checker.py +78 -0
- protoerror/cli.py +88 -0
- protoerror/config.py +78 -0
- protoerror/control.py +330 -0
- protoerror/finding.py +145 -0
- protoerror/group.py +158 -0
- protoerror/linter.py +428 -0
- protoerror/merger.py +69 -0
- protoerror/messages.py +95 -0
- protoerror/paged.py +109 -0
- protoerror/question.py +138 -0
- protoerror/report.py +210 -0
- protoerror/report_parser.py +58 -0
- protoerror/simple.py +85 -0
- protoerror/solution.py +300 -0
- protoerror/utils.py +49 -0
- protoerror-3.20.2.dist-info/METADATA +21 -0
- protoerror-3.20.2.dist-info/RECORD +22 -0
- protoerror-3.20.2.dist-info/WHEEL +5 -0
- protoerror-3.20.2.dist-info/top_level.txt +1 -0
protoerror/control.py
ADDED
|
@@ -0,0 +1,330 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2020-2022 by Helmut Konrad Fahrendholz. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
"""Control the checker
|
|
10
|
+
===================
|
|
11
|
+
|
|
12
|
+
Use decorators to exclude linter step cause the step has a problem or
|
|
13
|
+
implementation is not finished yet:
|
|
14
|
+
|
|
15
|
+
.. code-block:: python
|
|
16
|
+
|
|
17
|
+
@skip
|
|
18
|
+
def check_1282_skip_step():
|
|
19
|
+
pass
|
|
20
|
+
|
|
21
|
+
Skip linter step for special document length:
|
|
22
|
+
|
|
23
|
+
.. code-block:: python
|
|
24
|
+
|
|
25
|
+
@nosmall
|
|
26
|
+
@nomedium
|
|
27
|
+
def check_1290_long_page_check():
|
|
28
|
+
pass
|
|
29
|
+
|
|
30
|
+
Supported Decorators: @nosmall, @nomedium, @nolarge
|
|
31
|
+
|
|
32
|
+
Run linter for a special document type:
|
|
33
|
+
|
|
34
|
+
.. code-block:: python
|
|
35
|
+
|
|
36
|
+
@book
|
|
37
|
+
def check_1291_page_border():
|
|
38
|
+
pass
|
|
39
|
+
|
|
40
|
+
Supported Decorators: @homework, @bachelor, @master, @dissertation, @diss, @book
|
|
41
|
+
|
|
42
|
+
Examples
|
|
43
|
+
--------
|
|
44
|
+
|
|
45
|
+
>>> @homework
|
|
46
|
+
... @nolarge
|
|
47
|
+
... def check_1234(items):
|
|
48
|
+
... pass
|
|
49
|
+
|
|
50
|
+
Get list of decorators for a linter step:
|
|
51
|
+
|
|
52
|
+
>>> decorators(check_1234)
|
|
53
|
+
['nolarge', 'homework']
|
|
54
|
+
|
|
55
|
+
>>> @homework
|
|
56
|
+
... @disable_perpage(morethan=20)
|
|
57
|
+
... def check_touch_too_much():
|
|
58
|
+
... pass
|
|
59
|
+
|
|
60
|
+
>>> decorators(check_touch_too_much)
|
|
61
|
+
[{'disable_perpage': {'morethan': 20}}, 'homework']
|
|
62
|
+
|
|
63
|
+
Templates: Doctype based replacement
|
|
64
|
+
====================================
|
|
65
|
+
|
|
66
|
+
Add selective advice which are platform dependent, for example give
|
|
67
|
+
different advice when using MSWord instead of Latex:
|
|
68
|
+
|
|
69
|
+
.. code-block:: python
|
|
70
|
+
|
|
71
|
+
{% if MSWORD %}
|
|
72
|
+
...
|
|
73
|
+
{% endif %}
|
|
74
|
+
|
|
75
|
+
{% if LATEX %}
|
|
76
|
+
...
|
|
77
|
+
{% endif %}
|
|
78
|
+
|
|
79
|
+
{% if BASE %}
|
|
80
|
+
...
|
|
81
|
+
{% endif %}
|
|
82
|
+
"""
|
|
83
|
+
|
|
84
|
+
import contextlib
|
|
85
|
+
|
|
86
|
+
import configos
|
|
87
|
+
import iamraw
|
|
88
|
+
import utilo
|
|
89
|
+
|
|
90
|
+
MAX_SMALL_PAGE_LENGTH = configos.HV_INT_PLUS(default=35)
|
|
91
|
+
MAX_MEDIUM_PAGE_LENGTH = configos.HV_INT_PLUS(default=35)
|
|
92
|
+
|
|
93
|
+
DOCTYPES = [item.name.lower() for item in iamraw.DocumentType]
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def filter_checkers(items: list, document: iamraw.DocInfo) -> list:
|
|
97
|
+
current = document.doctype.name.lower() if document.doctype else None
|
|
98
|
+
result = []
|
|
99
|
+
for item in items:
|
|
100
|
+
decorated = decorators(item)
|
|
101
|
+
if should_skip(decorated, document):
|
|
102
|
+
continue
|
|
103
|
+
if current:
|
|
104
|
+
# skipped document type
|
|
105
|
+
if f'no{current}' in decorated:
|
|
106
|
+
# nohome nobachelor etc.
|
|
107
|
+
continue
|
|
108
|
+
# is check decorated for a special doctype
|
|
109
|
+
some = any(item in decorated for item in DOCTYPES)
|
|
110
|
+
if some and current not in decorated:
|
|
111
|
+
# current document is not selected by decorators, but
|
|
112
|
+
# others are. Therefore we have to skip this ckeck,
|
|
113
|
+
# because this check was not made for current document
|
|
114
|
+
# type.
|
|
115
|
+
continue
|
|
116
|
+
result.append(item)
|
|
117
|
+
return result
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def should_skip(decorated, document) -> bool:
|
|
121
|
+
if document.pages is not None:
|
|
122
|
+
small = document.pages < MAX_SMALL_PAGE_LENGTH
|
|
123
|
+
medium = MAX_SMALL_PAGE_LENGTH <= document.pages < MAX_MEDIUM_PAGE_LENGTH
|
|
124
|
+
large = MAX_MEDIUM_PAGE_LENGTH <= document.pages < utilo.INF
|
|
125
|
+
else:
|
|
126
|
+
small, medium, large = False, False, False
|
|
127
|
+
# deactivated method
|
|
128
|
+
if 'skip' in decorated:
|
|
129
|
+
return True
|
|
130
|
+
# verify document length
|
|
131
|
+
if small and 'nosmall' in decorated:
|
|
132
|
+
return True
|
|
133
|
+
if medium and 'nomedium' in decorated:
|
|
134
|
+
return True
|
|
135
|
+
if large and 'nolarge' in decorated:
|
|
136
|
+
return True
|
|
137
|
+
if 'german' in decorated and 'english' not in decorated:
|
|
138
|
+
if document.lang and document.lang != iamraw.Language.GERMAN:
|
|
139
|
+
return True
|
|
140
|
+
return False
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def decorateme(method, value):
|
|
144
|
+
try:
|
|
145
|
+
assert value not in method.__control__, str(method.__control__)
|
|
146
|
+
method.__control__.append(value)
|
|
147
|
+
except AttributeError:
|
|
148
|
+
setattr(method, '__control__', [value])
|
|
149
|
+
return method
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def decorators(method) -> set:
|
|
153
|
+
assert method, str(method)
|
|
154
|
+
with contextlib.suppress(AttributeError):
|
|
155
|
+
return method.__control__
|
|
156
|
+
return []
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
# pylint:disable=C0103
|
|
160
|
+
homework = lambda x: decorateme(x, 'homework')
|
|
161
|
+
bachelor = lambda x: decorateme(x, 'bachelor')
|
|
162
|
+
master = lambda x: decorateme(x, 'master')
|
|
163
|
+
diss = lambda x: decorateme(x, 'diss')
|
|
164
|
+
dissertation = diss
|
|
165
|
+
book = lambda x: decorateme(x, 'book')
|
|
166
|
+
paper = lambda x: decorateme(x, 'paper')
|
|
167
|
+
# exluce length of document
|
|
168
|
+
nosmall = lambda x: decorateme(x, 'nosmall')
|
|
169
|
+
nomedium = lambda x: decorateme(x, 'nomedium')
|
|
170
|
+
nolarge = lambda x: decorateme(x, 'nolarge')
|
|
171
|
+
# exclude types of document
|
|
172
|
+
nohome = lambda x: decorateme(x, 'nohomework')
|
|
173
|
+
nobachelor = lambda x: decorateme(x, 'nobachelor')
|
|
174
|
+
nomaster = lambda x: decorateme(x, 'nomaster')
|
|
175
|
+
nodiss = lambda x: decorateme(x, 'nodiss')
|
|
176
|
+
nobook = lambda x: decorateme(x, 'nobook')
|
|
177
|
+
nopaper = lambda x: decorateme(x, 'nopaper')
|
|
178
|
+
|
|
179
|
+
german = lambda x: decorateme(x, 'german')
|
|
180
|
+
english = lambda x: decorateme(x, 'english')
|
|
181
|
+
|
|
182
|
+
skip = lambda x: decorateme(x, 'skip')
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def section_skip(sections: iamraw.PartOfDocMixin = None):
|
|
186
|
+
return lambda x: decorateme(x, {'section_skip': sections})
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def section_only(sections: iamraw.PartOfDocMixin = None):
|
|
190
|
+
return lambda x: decorateme(x, {'section_only': sections})
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def only_skip(method):
|
|
194
|
+
only, skips = set(), set()
|
|
195
|
+
try:
|
|
196
|
+
control = method.__control__
|
|
197
|
+
except AttributeError:
|
|
198
|
+
return only, skips
|
|
199
|
+
for item in control:
|
|
200
|
+
if 'section_only' in item:
|
|
201
|
+
only.add(item['section_only'])
|
|
202
|
+
if 'section_skip' in item:
|
|
203
|
+
skips.add(item['section_skip'])
|
|
204
|
+
return only, skips
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def disable_perpage(lessthan=None, morethan=None, equal=None):
|
|
208
|
+
values = {}
|
|
209
|
+
if lessthan is not None:
|
|
210
|
+
values['lessthan'] = lessthan
|
|
211
|
+
if morethan is not None:
|
|
212
|
+
values['morethan'] = morethan
|
|
213
|
+
if equal is not None:
|
|
214
|
+
values['equal'] = equal
|
|
215
|
+
return lambda x: decorateme(x, {'disable_perpage': values})
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def enable_perpage(lessthan=None, morethan=None, equal=None):
|
|
219
|
+
values = {}
|
|
220
|
+
if lessthan is not None:
|
|
221
|
+
values['lessthan'] = lessthan
|
|
222
|
+
if morethan is not None:
|
|
223
|
+
values['morethan'] = morethan
|
|
224
|
+
if equal is not None:
|
|
225
|
+
values['equal'] = equal
|
|
226
|
+
return lambda x: decorateme(x, {'enable_perpage': values})
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def get_perpage(methods):
|
|
230
|
+
result = []
|
|
231
|
+
for method in methods:
|
|
232
|
+
decorator = decorators(method)
|
|
233
|
+
if 'perpage' not in str(decorator):
|
|
234
|
+
continue
|
|
235
|
+
result.append(method)
|
|
236
|
+
return result
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def is_disabled_perpage(findings, method) -> bool:
|
|
240
|
+
control = method.__control__
|
|
241
|
+
for item in control:
|
|
242
|
+
try:
|
|
243
|
+
disableperpage = item['disable_perpage']
|
|
244
|
+
except (TypeError, KeyError):
|
|
245
|
+
continue
|
|
246
|
+
with contextlib.suppress(KeyError):
|
|
247
|
+
if disableperpage['equal'] == len(findings):
|
|
248
|
+
return True
|
|
249
|
+
with contextlib.suppress(KeyError):
|
|
250
|
+
if len(findings) <= disableperpage['lessthan']:
|
|
251
|
+
return True
|
|
252
|
+
with contextlib.suppress(KeyError):
|
|
253
|
+
if len(findings) >= disableperpage['morethan']:
|
|
254
|
+
return True
|
|
255
|
+
return False
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
DOCINFO_PATTERN = utilo.compiles(r"""
|
|
259
|
+
^
|
|
260
|
+
(?P<typ>homework|bachelor|master|diss|habil|book|paper)?
|
|
261
|
+
(?P<pages>\d{1,4})?
|
|
262
|
+
(?P<lang>ger|eng|fre|german|english|french)?
|
|
263
|
+
$
|
|
264
|
+
""")
|
|
265
|
+
|
|
266
|
+
DOCINFO = 'bachelor64ger'
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def parse_docinfo(docinfo) -> iamraw.DocInfo:
|
|
270
|
+
"""\
|
|
271
|
+
>>> parse_docinfo('diss215eng')
|
|
272
|
+
DocInfo(pages=215, doctype=...DISS...lang=...ENGLISH...)
|
|
273
|
+
>>> parse_docinfo('15GER')
|
|
274
|
+
DocInfo(pages=15...NONE...lang=<Language.GERMAN...)
|
|
275
|
+
>>> assert parse_docinfo(None) is None
|
|
276
|
+
"""
|
|
277
|
+
if not docinfo:
|
|
278
|
+
return None
|
|
279
|
+
parsed = DOCINFO_PATTERN.match(docinfo)
|
|
280
|
+
if not parsed:
|
|
281
|
+
return None
|
|
282
|
+
doctype = iamraw.DocumentType.NONE
|
|
283
|
+
pages = 256
|
|
284
|
+
lang = iamraw.Language.GERMAN # default lang
|
|
285
|
+
with contextlib.suppress(KeyError):
|
|
286
|
+
if parsed['pages']:
|
|
287
|
+
pages = int(parsed['pages'])
|
|
288
|
+
with contextlib.suppress(KeyError, AttributeError):
|
|
289
|
+
doctype = iamraw.DocumentType[parsed['typ'].upper()]
|
|
290
|
+
with contextlib.suppress(KeyError):
|
|
291
|
+
if parsed['lang']:
|
|
292
|
+
lang = parse_lang(parsed['lang'])
|
|
293
|
+
result = iamraw.DocInfo(
|
|
294
|
+
pages=pages,
|
|
295
|
+
doctype=doctype,
|
|
296
|
+
lang=lang,
|
|
297
|
+
)
|
|
298
|
+
return result
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def parse_lang(lang: str) -> iamraw.Language:
|
|
302
|
+
lang = lang.lower()
|
|
303
|
+
if lang in 'german':
|
|
304
|
+
return iamraw.Language.GERMAN
|
|
305
|
+
if lang in 'english':
|
|
306
|
+
return iamraw.Language.ENGLISH
|
|
307
|
+
if lang in 'french':
|
|
308
|
+
return iamraw.Language.FRENCH
|
|
309
|
+
return iamraw.Language.UNKNOWN
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def integrate_docinfo():
|
|
313
|
+
hook = integrate_cli
|
|
314
|
+
run = evaluate_userchoice
|
|
315
|
+
return hook, run
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def evaluate_userchoice(argv):
|
|
319
|
+
docinfo = argv.get('docinfo', None)
|
|
320
|
+
parsed = parse_docinfo(docinfo)
|
|
321
|
+
if parsed is None:
|
|
322
|
+
parsed = iamraw.DocInfo()
|
|
323
|
+
return {'docinfo': parsed}
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def integrate_cli(parser):
|
|
327
|
+
parser.add_argument(
|
|
328
|
+
'--docinfo',
|
|
329
|
+
help='define docinfo to tell document type, length and language',
|
|
330
|
+
)
|
protoerror/finding.py
ADDED
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2019-2022 by Helmut Konrad Fahrendholz. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import concurrent.futures
|
|
11
|
+
import os
|
|
12
|
+
|
|
13
|
+
import iamraw
|
|
14
|
+
import serializeraw
|
|
15
|
+
import utilo
|
|
16
|
+
|
|
17
|
+
import protoerror
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def findings_from_path(
|
|
21
|
+
path: str,
|
|
22
|
+
worker: int = 10,
|
|
23
|
+
useronly: bool = True,
|
|
24
|
+
msgid: set = None,
|
|
25
|
+
) -> iamraw.PageFindings:
|
|
26
|
+
"""Load Findings from `path` directory and group them by page as
|
|
27
|
+
`PageFindings`."""
|
|
28
|
+
assert os.path.isdir(path), str(path)
|
|
29
|
+
files = utilo.file_list(path, include='yaml', recursive=True)
|
|
30
|
+
if useronly:
|
|
31
|
+
files = [
|
|
32
|
+
item for item in files if utilo.file_name(item).endswith('_user')
|
|
33
|
+
]
|
|
34
|
+
paths = [os.path.join(path, item) for item in files]
|
|
35
|
+
# limit worker by max file count
|
|
36
|
+
worker = utilo.mins(worker, len(files))
|
|
37
|
+
# ensure to have at least one worker when collection now file
|
|
38
|
+
worker = utilo.maxs(1, worker)
|
|
39
|
+
# yaml parsing is cpu bound, therefore we need a process pool instead
|
|
40
|
+
# of thread pool.
|
|
41
|
+
executor = utilo.select_executor()
|
|
42
|
+
with executor(max_workers=worker) as executor:
|
|
43
|
+
todo = {
|
|
44
|
+
executor.submit(serializeraw.load_findings, path): path
|
|
45
|
+
for path in paths
|
|
46
|
+
}
|
|
47
|
+
findings = []
|
|
48
|
+
for job in concurrent.futures.as_completed(todo):
|
|
49
|
+
data = job.result()
|
|
50
|
+
findings.extend(data)
|
|
51
|
+
if msgid:
|
|
52
|
+
# select findings by msgid
|
|
53
|
+
findings = protoerror.select_findings(findings, msgid=msgid)
|
|
54
|
+
result = protoerror.bypage(findings)
|
|
55
|
+
return result
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def iter_findings(path: str):
|
|
59
|
+
files = utilo.file_list(path, include='yaml', recursive=True)
|
|
60
|
+
files = [item for item in files if utilo.file_name(item).endswith('_user')]
|
|
61
|
+
for item in files:
|
|
62
|
+
location = os.path.join(path, item)
|
|
63
|
+
findings = serializeraw.load_findings(location)
|
|
64
|
+
yield location, findings
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def hash_finding(item):
|
|
68
|
+
try:
|
|
69
|
+
return hash(item)
|
|
70
|
+
except TypeError as error:
|
|
71
|
+
utilo.error(f'could not hash finding: {item}')
|
|
72
|
+
raise error
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def make_finding_number_unique(path: str, private: bool = False) -> bool:
|
|
76
|
+
"""Collect all findings from path and replace with unqiue finding
|
|
77
|
+
number.
|
|
78
|
+
|
|
79
|
+
Note: Remove lintings with equal hash cause there seem/must to be
|
|
80
|
+
equal.
|
|
81
|
+
|
|
82
|
+
Args:
|
|
83
|
+
path(str): location where files wither user linter are located
|
|
84
|
+
private(bool): encrypt result
|
|
85
|
+
Returns:
|
|
86
|
+
True if some file was located and replace.
|
|
87
|
+
False if no user file is in `path`.
|
|
88
|
+
"""
|
|
89
|
+
assert os.path.isdir(path), str(path)
|
|
90
|
+
single = utilo.Single()
|
|
91
|
+
replaced = False
|
|
92
|
+
for location, findings in iter_findings(path):
|
|
93
|
+
for finding in findings:
|
|
94
|
+
hashed = hash_finding(finding)
|
|
95
|
+
if single.contains(hashed):
|
|
96
|
+
utilo.error(f'duplicated finding: {finding}')
|
|
97
|
+
finding.number = None # None -> do not dump this finding
|
|
98
|
+
continue
|
|
99
|
+
finding.number = hashed
|
|
100
|
+
findings = [item for item in findings if item.number is not None]
|
|
101
|
+
# TODO: REFACTOR LATER
|
|
102
|
+
dumped = serializeraw.dump_findings(findings)
|
|
103
|
+
utilo.file_replace(location, dumped, private=private)
|
|
104
|
+
replaced = True
|
|
105
|
+
return replaced
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def finding_status_update(
|
|
109
|
+
path: str,
|
|
110
|
+
number: int,
|
|
111
|
+
status: iamraw.ProblemStatus,
|
|
112
|
+
private: bool = False,
|
|
113
|
+
) -> bool:
|
|
114
|
+
assert os.path.isdir(path), str(path)
|
|
115
|
+
assert isinstance(number, int), type(number)
|
|
116
|
+
assert isinstance(status, iamraw.ProblemStatus), type(status)
|
|
117
|
+
# TODO: IMPROVE SPEED LATER? MAY USE A BUFFERED OBJECT ORIENTED APPROACH
|
|
118
|
+
for location, findings in iter_findings(path):
|
|
119
|
+
for finding in findings:
|
|
120
|
+
if finding.number != number:
|
|
121
|
+
continue
|
|
122
|
+
if finding.solution is None:
|
|
123
|
+
utilo.error(f'could not update status: {finding}')
|
|
124
|
+
return False
|
|
125
|
+
finding.solution.status = status
|
|
126
|
+
dumped = serializeraw.dump_findings(findings)
|
|
127
|
+
utilo.debug(f'number: {number}; status: {status};\n'
|
|
128
|
+
f'update: {location}')
|
|
129
|
+
utilo.file_replace(location, dumped, private=private)
|
|
130
|
+
return True
|
|
131
|
+
return False
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def finding_status(path: str, number: int) -> iamraw.ProblemStatus:
|
|
135
|
+
assert os.path.isdir(path), str(path)
|
|
136
|
+
assert isinstance(number, int), type(number)
|
|
137
|
+
for _, findings in iter_findings(path):
|
|
138
|
+
for finding in findings:
|
|
139
|
+
if finding.number != number:
|
|
140
|
+
continue
|
|
141
|
+
if finding.solution is None:
|
|
142
|
+
utilo.error(f'could not get status: {finding}')
|
|
143
|
+
return None
|
|
144
|
+
return finding.solution.status
|
|
145
|
+
return None
|
protoerror/group.py
ADDED
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2019-2022 by Helmut Konrad Fahrendholz. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import collections
|
|
11
|
+
import contextlib
|
|
12
|
+
|
|
13
|
+
import iamraw
|
|
14
|
+
import utilo
|
|
15
|
+
|
|
16
|
+
import protoerror
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def bypage(items: iamraw.Findings) -> iamraw.PageFindings:
|
|
20
|
+
"""Group `items` by location.page of `Finding`. Sort the groups
|
|
21
|
+
ascending by page number."""
|
|
22
|
+
pages = collections.defaultdict(list)
|
|
23
|
+
for item in items:
|
|
24
|
+
assert item.location is not None, f'require location {item}'
|
|
25
|
+
pages[item.location.page].append(item)
|
|
26
|
+
result = [
|
|
27
|
+
iamraw.PageFinding(page=page, content=pages[page])
|
|
28
|
+
for page in sorted(pages.keys())
|
|
29
|
+
]
|
|
30
|
+
return result
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def byid(items: iamraw.Findings) -> dict:
|
|
34
|
+
"""Group findings by `finding.msgid`."""
|
|
35
|
+
grouped = collections.defaultdict(list)
|
|
36
|
+
for item in items:
|
|
37
|
+
grouped[item.msgid].append(item)
|
|
38
|
+
result = dict(grouped)
|
|
39
|
+
return result
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def filter_mark(items: iamraw.Findings, shortcut: str) -> iamraw.Findings:
|
|
43
|
+
"""Filter `Findings` by shortcut and sort them by `location.value`
|
|
44
|
+
afterwards.
|
|
45
|
+
|
|
46
|
+
Args:
|
|
47
|
+
items(iamraw.Findings): list of findings
|
|
48
|
+
shortcut(str): shortcut of protoerror.location, w word, p page,
|
|
49
|
+
ol oneline, etc.
|
|
50
|
+
Returns:
|
|
51
|
+
filtered, sorted list of `Findings`
|
|
52
|
+
"""
|
|
53
|
+
for item in items:
|
|
54
|
+
if item.location:
|
|
55
|
+
continue
|
|
56
|
+
utilo.error(f'missing location: {item}')
|
|
57
|
+
items = [finding for finding in items if finding.location]
|
|
58
|
+
selected = []
|
|
59
|
+
for item in items:
|
|
60
|
+
with contextlib.suppress(AttributeError):
|
|
61
|
+
if not isinstance(value(item.location), int):
|
|
62
|
+
utilo.debug(f'invalid location: {item.location}, require int.')
|
|
63
|
+
continue
|
|
64
|
+
if item.location.shortcut == shortcut:
|
|
65
|
+
selected.append(item)
|
|
66
|
+
selected.sort(key=lambda x: value(x.location))
|
|
67
|
+
return selected
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def value(location) -> int:
|
|
71
|
+
with contextlib.suppress(AttributeError):
|
|
72
|
+
if location.value is not None:
|
|
73
|
+
return location.value
|
|
74
|
+
with contextlib.suppress(AttributeError):
|
|
75
|
+
if location.line is not None:
|
|
76
|
+
return location.line
|
|
77
|
+
if location.page is not None:
|
|
78
|
+
return location.page
|
|
79
|
+
return utilo.INF
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def words(items: iamraw.Findings) -> iamraw.Findings:
|
|
83
|
+
return filter_mark(items, shortcut='w')
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def lines(items: iamraw.Findings) -> iamraw.Findings:
|
|
87
|
+
return filter_mark(items, shortcut='ol')
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def sentences(items: iamraw.Findings) -> iamraw.Findings:
|
|
91
|
+
return filter_mark(items, shortcut='s')
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def ranged(items: iamraw.Findings) -> iamraw.Findings:
|
|
95
|
+
return protoerror.filter_mark(items, shortcut='r')
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def select_findings(
|
|
99
|
+
findings: iamraw.Findings,
|
|
100
|
+
msgid: set,
|
|
101
|
+
) -> iamraw.Findings:
|
|
102
|
+
"""Select `Findings` specified by `msgid`
|
|
103
|
+
|
|
104
|
+
>>> select_findings([iamraw.Finding(msgid=1337), iamraw.Finding(msgid=1338)], msgid=(1337,1400))
|
|
105
|
+
[Finding(...msgid=1337...)]
|
|
106
|
+
>>> select_findings([iamraw.Finding(msgid=1337), iamraw.Finding(msgid=1338)], msgid=1337)
|
|
107
|
+
[Finding(...msgid=1337...)]
|
|
108
|
+
"""
|
|
109
|
+
assert all(isinstance(item, iamraw.Finding) for item in findings)
|
|
110
|
+
if msgid is None:
|
|
111
|
+
return findings
|
|
112
|
+
if isinstance(msgid, int):
|
|
113
|
+
msgid = {msgid}
|
|
114
|
+
elif isinstance(msgid, list):
|
|
115
|
+
msgid = set(msgid)
|
|
116
|
+
return [item for item in findings if item.msgid in msgid]
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def count_findings(findings: iamraw.Findings, msgid: set) -> int:
|
|
120
|
+
counted = len(select_findings(findings, msgid))
|
|
121
|
+
return counted
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def select_pages(
|
|
125
|
+
findings: iamraw.Findings,
|
|
126
|
+
pages: int,
|
|
127
|
+
) -> iamraw.Findings:
|
|
128
|
+
if pages is None:
|
|
129
|
+
return findings
|
|
130
|
+
pages = {pages} if isinstance(pages, int) else pages
|
|
131
|
+
findings = flat(findings)
|
|
132
|
+
findings = [
|
|
133
|
+
finding for finding in findings
|
|
134
|
+
if finding.location and finding.location.page in pages
|
|
135
|
+
]
|
|
136
|
+
return findings
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def flat(pages: iamraw.PageFinding) -> list:
|
|
140
|
+
result = []
|
|
141
|
+
for page in pages:
|
|
142
|
+
try:
|
|
143
|
+
# PageFinding
|
|
144
|
+
result.extend(page.content)
|
|
145
|
+
except AttributeError:
|
|
146
|
+
# findings are already flat
|
|
147
|
+
result.append(page)
|
|
148
|
+
return result
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def select(
|
|
152
|
+
findings: iamraw.Findings,
|
|
153
|
+
pages: int,
|
|
154
|
+
msgid: set,
|
|
155
|
+
) -> iamraw.Findings:
|
|
156
|
+
findings = select_pages(findings, pages=pages)
|
|
157
|
+
findings = select_findings(findings, msgid=msgid)
|
|
158
|
+
return findings
|