protoerror 3.20.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- protoerror/__init__.py +133 -0
- protoerror/__patch__.py +8 -0
- protoerror/checker.py +78 -0
- protoerror/cli.py +88 -0
- protoerror/config.py +78 -0
- protoerror/control.py +330 -0
- protoerror/finding.py +145 -0
- protoerror/group.py +158 -0
- protoerror/linter.py +428 -0
- protoerror/merger.py +69 -0
- protoerror/messages.py +95 -0
- protoerror/paged.py +109 -0
- protoerror/question.py +138 -0
- protoerror/report.py +210 -0
- protoerror/report_parser.py +58 -0
- protoerror/simple.py +85 -0
- protoerror/solution.py +300 -0
- protoerror/utils.py +49 -0
- protoerror-3.20.2.dist-info/METADATA +21 -0
- protoerror-3.20.2.dist-info/RECORD +22 -0
- protoerror-3.20.2.dist-info/WHEEL +5 -0
- protoerror-3.20.2.dist-info/top_level.txt +1 -0
protoerror/linter.py
ADDED
|
@@ -0,0 +1,428 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2019-2022 by Helmut Konrad Fahrendholz. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
"""The `Linter` defines an interface to write and separate `Finding`s
|
|
10
|
+
which are produced due the `Checker`s.
|
|
11
|
+
|
|
12
|
+
There are 2 types of Findings. The first finding type is to deliver
|
|
13
|
+
information to the user. These are findings which are `active` and
|
|
14
|
+
`confident` enough to present them to the user as FAILUREs in there
|
|
15
|
+
document. The other type is to give the devloper more information to
|
|
16
|
+
improve the platform.
|
|
17
|
+
|
|
18
|
+
Note: This class is thread-safe.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
import collections
|
|
22
|
+
import contextlib
|
|
23
|
+
import dataclasses
|
|
24
|
+
import functools
|
|
25
|
+
import importlib
|
|
26
|
+
import os
|
|
27
|
+
import threading
|
|
28
|
+
|
|
29
|
+
import iamraw
|
|
30
|
+
import serializeraw
|
|
31
|
+
import utilo
|
|
32
|
+
|
|
33
|
+
import protoerror.config
|
|
34
|
+
import protoerror.control
|
|
35
|
+
import protoerror.finding
|
|
36
|
+
import protoerror.solution
|
|
37
|
+
import protoerror.utils
|
|
38
|
+
|
|
39
|
+
USER_FILE = 'user_user.yaml'
|
|
40
|
+
DEVELOPER_FILE = 'developer_developer.yaml'
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclasses.dataclass
|
|
44
|
+
class DumpedLinterResult:
|
|
45
|
+
user: str
|
|
46
|
+
developer: str
|
|
47
|
+
|
|
48
|
+
def __getitem__(self, index):
|
|
49
|
+
"""Support tuple-like access.
|
|
50
|
+
|
|
51
|
+
Example:
|
|
52
|
+
user, developer = linter_result
|
|
53
|
+
"""
|
|
54
|
+
if index == 0: # pylint:disable=C2001
|
|
55
|
+
return self.user
|
|
56
|
+
if index == 1:
|
|
57
|
+
return self.developer
|
|
58
|
+
raise IndexError(f'index to high {index}')
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class Linter:
|
|
62
|
+
"""Hint: Messages are activate in default."""
|
|
63
|
+
|
|
64
|
+
def __init__(
|
|
65
|
+
self,
|
|
66
|
+
solver: protoerror.solution.Solver = None,
|
|
67
|
+
active: protoerror.config.MessageStatusList = None,
|
|
68
|
+
checkers: list = None,
|
|
69
|
+
document: iamraw.DocInfo = None,
|
|
70
|
+
):
|
|
71
|
+
# TODO: USE CHECKER DIRECTLY TO REDUCE AMOUT OF CODE
|
|
72
|
+
self.solver = solver
|
|
73
|
+
self.active = {item.msgid: item for item in active} if active else {}
|
|
74
|
+
self.checkerlist = list(checkers) if checkers else []
|
|
75
|
+
self.only, self.skip = only_skip(self.checkerlist)
|
|
76
|
+
self.findings = []
|
|
77
|
+
self.document = document
|
|
78
|
+
self.lock = threading.Lock() # make class thread safe
|
|
79
|
+
|
|
80
|
+
def add_finding(
|
|
81
|
+
self,
|
|
82
|
+
location: iamraw.Location = None,
|
|
83
|
+
msgid: str = None,
|
|
84
|
+
confidence: float = 1.0,
|
|
85
|
+
**kwargs,
|
|
86
|
+
):
|
|
87
|
+
"""Add Finding to store linted result.
|
|
88
|
+
|
|
89
|
+
Args:
|
|
90
|
+
location: locate linting in document
|
|
91
|
+
msgid: use msgid to mark this problem and find a solution.
|
|
92
|
+
confidence: how confident this linting is in range
|
|
93
|
+
lowest(0.0) to highest (1.0). Lower confident
|
|
94
|
+
findings are not presented to the user to avoid
|
|
95
|
+
bad quality lintings.
|
|
96
|
+
kwargs: use key words args to replace values in solution
|
|
97
|
+
template.
|
|
98
|
+
"""
|
|
99
|
+
if self.document and self.document.sections:
|
|
100
|
+
only, skip = set(), set()
|
|
101
|
+
with contextlib.suppress(KeyError):
|
|
102
|
+
only = self.only[msgid]
|
|
103
|
+
with contextlib.suppress(KeyError):
|
|
104
|
+
skip = self.skip[msgid]
|
|
105
|
+
if not self.document.sections(
|
|
106
|
+
location=location,
|
|
107
|
+
only=only,
|
|
108
|
+
skip=skip,
|
|
109
|
+
):
|
|
110
|
+
utilo.debug(f'skip finding in section: {msgid}, {location}')
|
|
111
|
+
# do not add this finding
|
|
112
|
+
return
|
|
113
|
+
# Determine a possible solution
|
|
114
|
+
solution = None
|
|
115
|
+
if self.solver:
|
|
116
|
+
if self.document == iamraw.Generator.MSWORD:
|
|
117
|
+
kwargs['MSWORD'] = True
|
|
118
|
+
if self.document == iamraw.Generator.LATEX:
|
|
119
|
+
kwargs['LATEX'] = True
|
|
120
|
+
if self.document == iamraw.Generator.UNDEFINED:
|
|
121
|
+
kwargs['UNDEFINED'] = True
|
|
122
|
+
solution = self.solver.solution(msgid=msgid, **kwargs)
|
|
123
|
+
|
|
124
|
+
active = self.isactive(msgid, confidence)
|
|
125
|
+
# create finding
|
|
126
|
+
finding = iamraw.Finding(
|
|
127
|
+
confidence=confidence,
|
|
128
|
+
location=location,
|
|
129
|
+
msgid=msgid,
|
|
130
|
+
solution=solution,
|
|
131
|
+
active=active,
|
|
132
|
+
)
|
|
133
|
+
finding.number = protoerror.finding.hash_finding(finding)
|
|
134
|
+
# store finding
|
|
135
|
+
with self.lock:
|
|
136
|
+
self.findings.append(finding)
|
|
137
|
+
|
|
138
|
+
def count_findings(self, msgid: str):
|
|
139
|
+
with self.lock:
|
|
140
|
+
counted = utilo.counts(self.findings, lambda x: x.msgid == msgid)
|
|
141
|
+
return counted
|
|
142
|
+
|
|
143
|
+
def check_findings(self, check: callable):
|
|
144
|
+
"""Run method to rewrite current `findings`."""
|
|
145
|
+
with self.lock:
|
|
146
|
+
self.findings = check(self.findings)
|
|
147
|
+
|
|
148
|
+
@property
|
|
149
|
+
def checkers(self):
|
|
150
|
+
result = self.checkerlist
|
|
151
|
+
if self.document:
|
|
152
|
+
result = protoerror.filter_checkers(result, self.document)
|
|
153
|
+
return result
|
|
154
|
+
|
|
155
|
+
def isactive(self, msgid, confidence):
|
|
156
|
+
if not self.active:
|
|
157
|
+
return True
|
|
158
|
+
active = True
|
|
159
|
+
with contextlib.suppress(KeyError):
|
|
160
|
+
msgstatus = self.active[msgid]
|
|
161
|
+
active = msgstatus.active and msgstatus.confidence >= confidence
|
|
162
|
+
return active
|
|
163
|
+
|
|
164
|
+
def write(self, path: str, unique: bool = False):
|
|
165
|
+
"""Write linter result to `user` and `developer`-file.
|
|
166
|
+
|
|
167
|
+
Args:
|
|
168
|
+
path(str): directory to write both files
|
|
169
|
+
unique(bool): if unique no duplicated user-message are written
|
|
170
|
+
"""
|
|
171
|
+
assert os.path.isdir(path), str(path)
|
|
172
|
+
# create result
|
|
173
|
+
result = self.result(unique=unique)
|
|
174
|
+
write_result(result, path, unique=unique)
|
|
175
|
+
|
|
176
|
+
def run(self, driver=None):
|
|
177
|
+
self.findings = []
|
|
178
|
+
# select document dependend checkers
|
|
179
|
+
for checker in self.checkers:
|
|
180
|
+
call = functools.partial(
|
|
181
|
+
self.add_finding,
|
|
182
|
+
msgid=checker.msgid,
|
|
183
|
+
)
|
|
184
|
+
checker(call, driver)
|
|
185
|
+
|
|
186
|
+
def result(self, unique: bool = False):
|
|
187
|
+
"""Return current linter result of `user`, `developer`"""
|
|
188
|
+
with self.lock:
|
|
189
|
+
result = self.findings[:]
|
|
190
|
+
if unique:
|
|
191
|
+
result = utilo.unique(result)
|
|
192
|
+
return result
|
|
193
|
+
|
|
194
|
+
def register_checker(self, checker):
|
|
195
|
+
"""Required method to auto register this checker."""
|
|
196
|
+
self.checkers.append(checker)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def split_userdeveloper(items: list, checkers: list) -> tuple:
|
|
200
|
+
if not checkers:
|
|
201
|
+
checkers = []
|
|
202
|
+
# move inactive findings to developer
|
|
203
|
+
user, developer = utilo.partition(items=items, key=lambda item: item.active)
|
|
204
|
+
perpage_disabled = perpage_disable(user, checkers)
|
|
205
|
+
# move disabled findings to developer findings, do not show it to the
|
|
206
|
+
# user.
|
|
207
|
+
for item in perpage_disabled:
|
|
208
|
+
user.remove(item)
|
|
209
|
+
developer.append(item)
|
|
210
|
+
return user, developer
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def perpage_disable(findings, checkers):
|
|
214
|
+
"""Determine list of findings which are disabled by
|
|
215
|
+
@disable-decorator."""
|
|
216
|
+
findings = [item for item in findings if item.location is not None]
|
|
217
|
+
grouped = protoerror.bypage(findings)
|
|
218
|
+
# bypage
|
|
219
|
+
result = []
|
|
220
|
+
perpage = protoerror.control.get_perpage(checkers)
|
|
221
|
+
for pageitem in grouped:
|
|
222
|
+
paged = protoerror.byid(pageitem.content)
|
|
223
|
+
for method in perpage:
|
|
224
|
+
msgid = method.msgid
|
|
225
|
+
try:
|
|
226
|
+
findings = paged[msgid]
|
|
227
|
+
except KeyError:
|
|
228
|
+
continue
|
|
229
|
+
if not protoerror.is_disabled_perpage(findings, method):
|
|
230
|
+
# content is not disabled
|
|
231
|
+
continue
|
|
232
|
+
result.extend(findings)
|
|
233
|
+
return result
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def only_skip(checkers):
|
|
237
|
+
only = collections.defaultdict(set)
|
|
238
|
+
skip = collections.defaultdict(set)
|
|
239
|
+
for item in checkers:
|
|
240
|
+
msgid = item.msgid
|
|
241
|
+
item_only, item_skip = protoerror.control.only_skip(item)
|
|
242
|
+
only[msgid] |= item_only
|
|
243
|
+
skip[msgid] |= item_skip
|
|
244
|
+
only: dict = dict(only)
|
|
245
|
+
skip: dict = dict(skip)
|
|
246
|
+
return only, skip
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def dump_result(
|
|
250
|
+
items: iamraw.Findings,
|
|
251
|
+
*,
|
|
252
|
+
unique: bool = False,
|
|
253
|
+
checkers: list = None,
|
|
254
|
+
) -> DumpedLinterResult:
|
|
255
|
+
"""Write linter result to `user` and `developer`-file.
|
|
256
|
+
|
|
257
|
+
Args:
|
|
258
|
+
items(list): list of `Finding`s
|
|
259
|
+
unique(bool): remove duplicated linter findings
|
|
260
|
+
checkers(methods): list of user linters
|
|
261
|
+
Returns:
|
|
262
|
+
Result with dumped user ander developer result in yaml format.
|
|
263
|
+
"""
|
|
264
|
+
if unique:
|
|
265
|
+
items = utilo.unique(items)
|
|
266
|
+
|
|
267
|
+
user, developer = split_userdeveloper(items, checkers=checkers)
|
|
268
|
+
|
|
269
|
+
dumped_user = serializeraw.dump_findings(user)
|
|
270
|
+
dumped_developer = serializeraw.dump_findings(developer)
|
|
271
|
+
|
|
272
|
+
result = DumpedLinterResult(user=dumped_user, developer=dumped_developer)
|
|
273
|
+
return result
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def write_result(
|
|
277
|
+
result: iamraw.Findings,
|
|
278
|
+
path: str,
|
|
279
|
+
*,
|
|
280
|
+
unique: bool = False,
|
|
281
|
+
user_file=USER_FILE,
|
|
282
|
+
dev_file=DEVELOPER_FILE,
|
|
283
|
+
private: bool = False,
|
|
284
|
+
):
|
|
285
|
+
"""Write linter result to `user` and `developer`-file.
|
|
286
|
+
|
|
287
|
+
Args:
|
|
288
|
+
result(list): list of `Finding`s
|
|
289
|
+
path(str): directory to write both files unique(bool): if unique
|
|
290
|
+
no duplicated user-messages are written
|
|
291
|
+
unique(bool): remove duplication out of result
|
|
292
|
+
user_file(str): filename of user linting file. If None, write
|
|
293
|
+
nothing
|
|
294
|
+
dev_file(str): filename of developer linting file. If None,
|
|
295
|
+
write nothing
|
|
296
|
+
private(bool): use encryption
|
|
297
|
+
"""
|
|
298
|
+
assert os.path.isdir(path), str(path)
|
|
299
|
+
dumped_user, dumped_developer = dump_result(result, unique=unique)
|
|
300
|
+
if user_file:
|
|
301
|
+
user_outpath = os.path.join(path, user_file)
|
|
302
|
+
utilo.file_replace(user_outpath, dumped_user, private=private)
|
|
303
|
+
if dev_file:
|
|
304
|
+
developer_outpath = os.path.join(path, dev_file)
|
|
305
|
+
utilo.file_replace(developer_outpath, dumped_developer, private=private)
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def from_file(path: str, document: iamraw.DocInfo = None) -> Linter:
|
|
309
|
+
filename = os.path.basename(path)
|
|
310
|
+
spec = importlib.util.spec_from_file_location(
|
|
311
|
+
filename,
|
|
312
|
+
os.path.join(path),
|
|
313
|
+
)
|
|
314
|
+
module = importlib.util.module_from_spec(spec)
|
|
315
|
+
spec.loader.exec_module(module)
|
|
316
|
+
try:
|
|
317
|
+
solution = module.SOLUTION
|
|
318
|
+
except AttributeError as error:
|
|
319
|
+
msg = f'could not create solver, no SOLUTION: {path}'
|
|
320
|
+
raise ValueError(msg) from error
|
|
321
|
+
try:
|
|
322
|
+
status = module.STATUS
|
|
323
|
+
except AttributeError:
|
|
324
|
+
utilo.debug(f'no `STATUS` provided in {path}')
|
|
325
|
+
status = []
|
|
326
|
+
result = from_solution(
|
|
327
|
+
solution,
|
|
328
|
+
status,
|
|
329
|
+
document=document,
|
|
330
|
+
)
|
|
331
|
+
return result
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def from_solution(
|
|
335
|
+
solutions: iamraw.Solutions,
|
|
336
|
+
statuses: protoerror.config.MessageStatusList,
|
|
337
|
+
checkers: list = None,
|
|
338
|
+
document: iamraw.DocInfo = None,
|
|
339
|
+
) -> Linter:
|
|
340
|
+
solver = protoerror.solution.Solver.fromlist(solutions)
|
|
341
|
+
result = Linter(
|
|
342
|
+
solver,
|
|
343
|
+
active=statuses,
|
|
344
|
+
checkers=checkers,
|
|
345
|
+
document=document,
|
|
346
|
+
)
|
|
347
|
+
return result
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def from_module(
|
|
351
|
+
name: str,
|
|
352
|
+
tests: set = None,
|
|
353
|
+
skips: set = None,
|
|
354
|
+
document: iamraw.DocInfo = None,
|
|
355
|
+
) -> Linter:
|
|
356
|
+
result = from_modules(
|
|
357
|
+
[name],
|
|
358
|
+
tests=tests,
|
|
359
|
+
skips=skips,
|
|
360
|
+
document=document,
|
|
361
|
+
)
|
|
362
|
+
return result
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
def from_modules(
|
|
366
|
+
modules: utilo.Strings,
|
|
367
|
+
tests: set = None,
|
|
368
|
+
skips: set = None,
|
|
369
|
+
document: iamraw.DocInfo = None,
|
|
370
|
+
) -> Linter:
|
|
371
|
+
modules = module_list(modules)
|
|
372
|
+
status = []
|
|
373
|
+
checkers = []
|
|
374
|
+
solutions = []
|
|
375
|
+
for name in modules:
|
|
376
|
+
with contextlib.suppress(AttributeError):
|
|
377
|
+
# support module type, ensure that module name is str
|
|
378
|
+
name = name.__name__
|
|
379
|
+
module = protoerror.utils.module_fromname(name)
|
|
380
|
+
solutions.extend(
|
|
381
|
+
protoerror.solution.parse_solutions(
|
|
382
|
+
module,
|
|
383
|
+
tests=tests,
|
|
384
|
+
skips=skips,
|
|
385
|
+
))
|
|
386
|
+
status.extend(parse_active(module))
|
|
387
|
+
checkers.extend(
|
|
388
|
+
protoerror.parse_checkers(
|
|
389
|
+
module,
|
|
390
|
+
tests=tests,
|
|
391
|
+
skips=skips,
|
|
392
|
+
))
|
|
393
|
+
result = protoerror.from_solution(
|
|
394
|
+
solutions,
|
|
395
|
+
status,
|
|
396
|
+
checkers=checkers,
|
|
397
|
+
document=document,
|
|
398
|
+
)
|
|
399
|
+
return result
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
def module_list(modulename: list) -> list:
|
|
403
|
+
r"""\
|
|
404
|
+
>>> import protoerror.simple; protoerror.simple.run(protoerror.simple)
|
|
405
|
+
('[]\n', '[]\n')
|
|
406
|
+
"""
|
|
407
|
+
if isinstance(modulename, str):
|
|
408
|
+
return [modulename]
|
|
409
|
+
if type(modulename).__name__ == 'module':
|
|
410
|
+
return [modulename]
|
|
411
|
+
return modulename
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def parse_active(module):
|
|
415
|
+
checkers = protoerror.solution.parse_checkers(module)
|
|
416
|
+
result = [
|
|
417
|
+
protoerror.MessageStatus(
|
|
418
|
+
msgid=protoerror.solution.parse_msgid(item.__name__),
|
|
419
|
+
active=item.confidence > 0.0,
|
|
420
|
+
confidence=item.confidence,
|
|
421
|
+
) for item in checkers
|
|
422
|
+
]
|
|
423
|
+
return result
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
# def register(linter):
|
|
427
|
+
# """required method to auto register this checker """
|
|
428
|
+
# linter.register_checker(MisdesignChecker(linter))
|
protoerror/merger.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2021-2022 by Helmut Konrad Fahrendholz. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import utilo
|
|
11
|
+
|
|
12
|
+
import protoerror
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def merge_findings(findings):
|
|
16
|
+
"""Merge lintings which are neighbors and lint the same error.
|
|
17
|
+
|
|
18
|
+
1. Group by page
|
|
19
|
+
2. Group by message id
|
|
20
|
+
3. Merge equal neighbors
|
|
21
|
+
"""
|
|
22
|
+
ranged, notranged = utilo.partition(
|
|
23
|
+
key=lambda x: x.location.shortcut == 'r',
|
|
24
|
+
items=findings,
|
|
25
|
+
)
|
|
26
|
+
result = list(notranged)
|
|
27
|
+
paged = protoerror.bypage(ranged)
|
|
28
|
+
for page in paged:
|
|
29
|
+
msgid = protoerror.byid(page)
|
|
30
|
+
for messagegroup in msgid.values():
|
|
31
|
+
merged = merge_equal_findingid(messagegroup)
|
|
32
|
+
result.extend(merged)
|
|
33
|
+
return result
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def merge_equal_findingid(findings):
|
|
37
|
+
if not findings:
|
|
38
|
+
return []
|
|
39
|
+
# remove findings without lines, cause we carn't merge them
|
|
40
|
+
locations, nolines = utilo.partition(
|
|
41
|
+
lambda x: x.location.line is not None,
|
|
42
|
+
findings,
|
|
43
|
+
)
|
|
44
|
+
if not locations:
|
|
45
|
+
# no findings with lines
|
|
46
|
+
return findings
|
|
47
|
+
findings = sorted(locations, key=lambda x: x.location.line)
|
|
48
|
+
result = [findings[0]]
|
|
49
|
+
for item in findings[1:]:
|
|
50
|
+
before = result[-1]
|
|
51
|
+
linebefore = before.location.line
|
|
52
|
+
if before.location.line_end is not None:
|
|
53
|
+
linebefore = before.location.line_end
|
|
54
|
+
if item.msgid != before.msgid:
|
|
55
|
+
result.append(item)
|
|
56
|
+
continue
|
|
57
|
+
if item.solution.description != before.solution.description:
|
|
58
|
+
result.append(item)
|
|
59
|
+
continue
|
|
60
|
+
if item.location.line != linebefore + 1:
|
|
61
|
+
result.append(item)
|
|
62
|
+
continue
|
|
63
|
+
before.location.line_end = linebefore + 1
|
|
64
|
+
# update finding number
|
|
65
|
+
for finding in result:
|
|
66
|
+
finding.number = protoerror.finding.hash_finding(finding)
|
|
67
|
+
# append no lines
|
|
68
|
+
findings.extend(nolines)
|
|
69
|
+
return result
|
protoerror/messages.py
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2019-2022 by Helmut Konrad Fahrendholz. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
"""The `messages` defines an interface to group problems and sort them by
|
|
10
|
+
weight.
|
|
11
|
+
|
|
12
|
+
MSG_TYPES description:
|
|
13
|
+
|
|
14
|
+
Info: document statistics, wordcount, information about using style,
|
|
15
|
+
document equality to famous writers.
|
|
16
|
+
|
|
17
|
+
Convention: page order - table of content at the end of the
|
|
18
|
+
document, font style, font size, font spacing.
|
|
19
|
+
|
|
20
|
+
Refactor: writing style of paragraph with rewriting hint
|
|
21
|
+
|
|
22
|
+
Warning: Umgangsprache, Black and Write-printing warning, to high/low
|
|
23
|
+
image resolution.
|
|
24
|
+
|
|
25
|
+
Error: Write text over page border, text formatting problem
|
|
26
|
+
"Hurensohn", broken table, wrong citatation, different
|
|
27
|
+
citation styles, missing reference, table of content
|
|
28
|
+
order/level definition.
|
|
29
|
+
|
|
30
|
+
Fatal: An error which blocks further analysis auf the document. In
|
|
31
|
+
general this is an program error which is triggered by pdf state.
|
|
32
|
+
|
|
33
|
+
MSG definition:
|
|
34
|
+
|
|
35
|
+
'ERROR_CODE': (
|
|
36
|
+
'REASON',
|
|
37
|
+
'INTERNAL_SHORTCUT',
|
|
38
|
+
'DESCRIPTION_OF_THE_PROBLEM',
|
|
39
|
+
),
|
|
40
|
+
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
import contextlib
|
|
44
|
+
|
|
45
|
+
import utilo
|
|
46
|
+
|
|
47
|
+
# pylint: disable=consider-using-namedtuple-or-dataclass
|
|
48
|
+
MSGS = {
|
|
49
|
+
'F0000': (
|
|
50
|
+
'Fehler beim Lesen der PDF Datei.',
|
|
51
|
+
'pdf-read-error',
|
|
52
|
+
'Used when pdf-miner is not able to read pdf file.',
|
|
53
|
+
),
|
|
54
|
+
'F0001': (
|
|
55
|
+
'Fehler beim Extrahieren der PDF Datei.',
|
|
56
|
+
'pdf-extract-error',
|
|
57
|
+
'Used when environment is not able to read given pdf file.',
|
|
58
|
+
),
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
MSG_TYPES = {
|
|
62
|
+
"I": "info",
|
|
63
|
+
"C": "convention", # page order
|
|
64
|
+
"R": "refactor", # style of paragraph, page
|
|
65
|
+
"W": "warning", # umgangssprache, image black and white problem
|
|
66
|
+
"E": "error", # writing over border
|
|
67
|
+
"F": "fatal", # pdf analyzing error
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
TYPE_DEFAULT = 'W'
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def parse_msgid(msgid: str, idonly: bool = False) -> tuple[str, int]:
|
|
74
|
+
"""Split `msgid` into `type` and `number`.
|
|
75
|
+
|
|
76
|
+
Args:
|
|
77
|
+
msgid(str): define type and number of used message
|
|
78
|
+
idonly(bool): do not return msg type
|
|
79
|
+
Returns:
|
|
80
|
+
typ(str), number(int) of message
|
|
81
|
+
"""
|
|
82
|
+
if not msgid:
|
|
83
|
+
return msgid
|
|
84
|
+
with contextlib.suppress(ValueError):
|
|
85
|
+
msgid = int(msgid)
|
|
86
|
+
if idonly:
|
|
87
|
+
return msgid
|
|
88
|
+
return TYPE_DEFAULT, msgid
|
|
89
|
+
typ, number = msgid[0], int(msgid[1:])
|
|
90
|
+
typ = typ.upper()
|
|
91
|
+
assert typ in MSG_TYPES, (f'invalid msg type: {typ}; '
|
|
92
|
+
f'use {utilo.from_tuple(MSG_TYPES.keys())}')
|
|
93
|
+
if idonly:
|
|
94
|
+
return number
|
|
95
|
+
return typ, number
|
protoerror/paged.py
ADDED
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
# =============================================================================
|
|
2
|
+
# C O P Y R I G H T
|
|
3
|
+
# -----------------------------------------------------------------------------
|
|
4
|
+
# Copyright (c) 2020-2022 by Helmut Konrad Fahrendholz. All rights reserved.
|
|
5
|
+
# This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
|
|
6
|
+
# use or distribution is an offensive act against international law and may
|
|
7
|
+
# be prosecuted under federal law. Its content is company confidential.
|
|
8
|
+
# =============================================================================
|
|
9
|
+
|
|
10
|
+
import concurrent.futures
|
|
11
|
+
import contextlib
|
|
12
|
+
import os
|
|
13
|
+
|
|
14
|
+
import iamraw
|
|
15
|
+
import serializeraw
|
|
16
|
+
import utilo
|
|
17
|
+
|
|
18
|
+
import protoerror
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def write_grouped(
|
|
22
|
+
findings: iamraw.Findings,
|
|
23
|
+
dest: str,
|
|
24
|
+
overwrite: bool = True,
|
|
25
|
+
private: bool = False,
|
|
26
|
+
) -> list:
|
|
27
|
+
result = []
|
|
28
|
+
grouped = protoerror.bypage(findings)
|
|
29
|
+
writer = utilo.file_replace if overwrite else utilo.file_create
|
|
30
|
+
for item in grouped:
|
|
31
|
+
page = fname(item.page)
|
|
32
|
+
outpath = os.path.join(dest, page)
|
|
33
|
+
dumped = serializeraw.dump_findings(item.content)
|
|
34
|
+
writer(outpath, dumped, private=private)
|
|
35
|
+
result.append(outpath)
|
|
36
|
+
return result
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def load_grouped(
|
|
40
|
+
source: str,
|
|
41
|
+
pages: tuple = None,
|
|
42
|
+
sort: bool = True,
|
|
43
|
+
worker=10,
|
|
44
|
+
) -> iamraw.PageFindings:
|
|
45
|
+
if isinstance(pages, int):
|
|
46
|
+
pages = (pages,)
|
|
47
|
+
if pages is None:
|
|
48
|
+
# load all findings
|
|
49
|
+
pages = [
|
|
50
|
+
pagenumber(item, none=True) for item in utilo.file_list(source)
|
|
51
|
+
]
|
|
52
|
+
# remove invalid file names
|
|
53
|
+
pages = utilo.notnone(pages)
|
|
54
|
+
# yaml parsing is cpu bound, therefore we need a process pool instead
|
|
55
|
+
# of thread pool.
|
|
56
|
+
executor = utilo.select_executor()
|
|
57
|
+
result = []
|
|
58
|
+
with executor(max_workers=worker) as executor:
|
|
59
|
+
todo = {
|
|
60
|
+
executor.submit(load_findings, source, page): page for page in pages
|
|
61
|
+
}
|
|
62
|
+
for job in concurrent.futures.as_completed(todo):
|
|
63
|
+
data = job.result()
|
|
64
|
+
if not data:
|
|
65
|
+
continue
|
|
66
|
+
result.append(data)
|
|
67
|
+
if sort:
|
|
68
|
+
result.sort(key=lambda x: x.page)
|
|
69
|
+
return result
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def load_findings(source: str, page: int) -> iamraw.PageFinding:
|
|
73
|
+
name = fname(page)
|
|
74
|
+
source = os.path.join(source, name)
|
|
75
|
+
if not os.path.exists(source):
|
|
76
|
+
return None
|
|
77
|
+
findings = serializeraw.load_findings(source)
|
|
78
|
+
return iamraw.PageFinding(page=page, content=findings)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def fname(page: int) -> str:
|
|
82
|
+
"""\
|
|
83
|
+
>>> fname(-5)
|
|
84
|
+
'_05'
|
|
85
|
+
>>> fname(1)
|
|
86
|
+
'001'
|
|
87
|
+
>>> fname(333)
|
|
88
|
+
'333'
|
|
89
|
+
"""
|
|
90
|
+
if page < 0:
|
|
91
|
+
return '_' + f'{page*-1}'.zfill(2)
|
|
92
|
+
return f'{page}'.zfill(3)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def pagenumber(page: str, none: bool = True) -> int:
|
|
96
|
+
"""\
|
|
97
|
+
>>> pagenumber('010')
|
|
98
|
+
10
|
|
99
|
+
>>> pagenumber(fname(-10))
|
|
100
|
+
-10
|
|
101
|
+
"""
|
|
102
|
+
if page[0] == '_':
|
|
103
|
+
page = '-' + page[1:]
|
|
104
|
+
with contextlib.suppress(ValueError):
|
|
105
|
+
return int(page)
|
|
106
|
+
if none:
|
|
107
|
+
# handle error case
|
|
108
|
+
return None
|
|
109
|
+
raise ValueError(f'could not convert to int: {page}')
|