protoerror 3.20.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
protoerror/linter.py ADDED
@@ -0,0 +1,428 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2019-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+ """The `Linter` defines an interface to write and separate `Finding`s
10
+ which are produced due the `Checker`s.
11
+
12
+ There are 2 types of Findings. The first finding type is to deliver
13
+ information to the user. These are findings which are `active` and
14
+ `confident` enough to present them to the user as FAILUREs in there
15
+ document. The other type is to give the devloper more information to
16
+ improve the platform.
17
+
18
+ Note: This class is thread-safe.
19
+ """
20
+
21
+ import collections
22
+ import contextlib
23
+ import dataclasses
24
+ import functools
25
+ import importlib
26
+ import os
27
+ import threading
28
+
29
+ import iamraw
30
+ import serializeraw
31
+ import utilo
32
+
33
+ import protoerror.config
34
+ import protoerror.control
35
+ import protoerror.finding
36
+ import protoerror.solution
37
+ import protoerror.utils
38
+
39
+ USER_FILE = 'user_user.yaml'
40
+ DEVELOPER_FILE = 'developer_developer.yaml'
41
+
42
+
43
+ @dataclasses.dataclass
44
+ class DumpedLinterResult:
45
+ user: str
46
+ developer: str
47
+
48
+ def __getitem__(self, index):
49
+ """Support tuple-like access.
50
+
51
+ Example:
52
+ user, developer = linter_result
53
+ """
54
+ if index == 0: # pylint:disable=C2001
55
+ return self.user
56
+ if index == 1:
57
+ return self.developer
58
+ raise IndexError(f'index to high {index}')
59
+
60
+
61
+ class Linter:
62
+ """Hint: Messages are activate in default."""
63
+
64
+ def __init__(
65
+ self,
66
+ solver: protoerror.solution.Solver = None,
67
+ active: protoerror.config.MessageStatusList = None,
68
+ checkers: list = None,
69
+ document: iamraw.DocInfo = None,
70
+ ):
71
+ # TODO: USE CHECKER DIRECTLY TO REDUCE AMOUT OF CODE
72
+ self.solver = solver
73
+ self.active = {item.msgid: item for item in active} if active else {}
74
+ self.checkerlist = list(checkers) if checkers else []
75
+ self.only, self.skip = only_skip(self.checkerlist)
76
+ self.findings = []
77
+ self.document = document
78
+ self.lock = threading.Lock() # make class thread safe
79
+
80
+ def add_finding(
81
+ self,
82
+ location: iamraw.Location = None,
83
+ msgid: str = None,
84
+ confidence: float = 1.0,
85
+ **kwargs,
86
+ ):
87
+ """Add Finding to store linted result.
88
+
89
+ Args:
90
+ location: locate linting in document
91
+ msgid: use msgid to mark this problem and find a solution.
92
+ confidence: how confident this linting is in range
93
+ lowest(0.0) to highest (1.0). Lower confident
94
+ findings are not presented to the user to avoid
95
+ bad quality lintings.
96
+ kwargs: use key words args to replace values in solution
97
+ template.
98
+ """
99
+ if self.document and self.document.sections:
100
+ only, skip = set(), set()
101
+ with contextlib.suppress(KeyError):
102
+ only = self.only[msgid]
103
+ with contextlib.suppress(KeyError):
104
+ skip = self.skip[msgid]
105
+ if not self.document.sections(
106
+ location=location,
107
+ only=only,
108
+ skip=skip,
109
+ ):
110
+ utilo.debug(f'skip finding in section: {msgid}, {location}')
111
+ # do not add this finding
112
+ return
113
+ # Determine a possible solution
114
+ solution = None
115
+ if self.solver:
116
+ if self.document == iamraw.Generator.MSWORD:
117
+ kwargs['MSWORD'] = True
118
+ if self.document == iamraw.Generator.LATEX:
119
+ kwargs['LATEX'] = True
120
+ if self.document == iamraw.Generator.UNDEFINED:
121
+ kwargs['UNDEFINED'] = True
122
+ solution = self.solver.solution(msgid=msgid, **kwargs)
123
+
124
+ active = self.isactive(msgid, confidence)
125
+ # create finding
126
+ finding = iamraw.Finding(
127
+ confidence=confidence,
128
+ location=location,
129
+ msgid=msgid,
130
+ solution=solution,
131
+ active=active,
132
+ )
133
+ finding.number = protoerror.finding.hash_finding(finding)
134
+ # store finding
135
+ with self.lock:
136
+ self.findings.append(finding)
137
+
138
+ def count_findings(self, msgid: str):
139
+ with self.lock:
140
+ counted = utilo.counts(self.findings, lambda x: x.msgid == msgid)
141
+ return counted
142
+
143
+ def check_findings(self, check: callable):
144
+ """Run method to rewrite current `findings`."""
145
+ with self.lock:
146
+ self.findings = check(self.findings)
147
+
148
+ @property
149
+ def checkers(self):
150
+ result = self.checkerlist
151
+ if self.document:
152
+ result = protoerror.filter_checkers(result, self.document)
153
+ return result
154
+
155
+ def isactive(self, msgid, confidence):
156
+ if not self.active:
157
+ return True
158
+ active = True
159
+ with contextlib.suppress(KeyError):
160
+ msgstatus = self.active[msgid]
161
+ active = msgstatus.active and msgstatus.confidence >= confidence
162
+ return active
163
+
164
+ def write(self, path: str, unique: bool = False):
165
+ """Write linter result to `user` and `developer`-file.
166
+
167
+ Args:
168
+ path(str): directory to write both files
169
+ unique(bool): if unique no duplicated user-message are written
170
+ """
171
+ assert os.path.isdir(path), str(path)
172
+ # create result
173
+ result = self.result(unique=unique)
174
+ write_result(result, path, unique=unique)
175
+
176
+ def run(self, driver=None):
177
+ self.findings = []
178
+ # select document dependend checkers
179
+ for checker in self.checkers:
180
+ call = functools.partial(
181
+ self.add_finding,
182
+ msgid=checker.msgid,
183
+ )
184
+ checker(call, driver)
185
+
186
+ def result(self, unique: bool = False):
187
+ """Return current linter result of `user`, `developer`"""
188
+ with self.lock:
189
+ result = self.findings[:]
190
+ if unique:
191
+ result = utilo.unique(result)
192
+ return result
193
+
194
+ def register_checker(self, checker):
195
+ """Required method to auto register this checker."""
196
+ self.checkers.append(checker)
197
+
198
+
199
+ def split_userdeveloper(items: list, checkers: list) -> tuple:
200
+ if not checkers:
201
+ checkers = []
202
+ # move inactive findings to developer
203
+ user, developer = utilo.partition(items=items, key=lambda item: item.active)
204
+ perpage_disabled = perpage_disable(user, checkers)
205
+ # move disabled findings to developer findings, do not show it to the
206
+ # user.
207
+ for item in perpage_disabled:
208
+ user.remove(item)
209
+ developer.append(item)
210
+ return user, developer
211
+
212
+
213
+ def perpage_disable(findings, checkers):
214
+ """Determine list of findings which are disabled by
215
+ @disable-decorator."""
216
+ findings = [item for item in findings if item.location is not None]
217
+ grouped = protoerror.bypage(findings)
218
+ # bypage
219
+ result = []
220
+ perpage = protoerror.control.get_perpage(checkers)
221
+ for pageitem in grouped:
222
+ paged = protoerror.byid(pageitem.content)
223
+ for method in perpage:
224
+ msgid = method.msgid
225
+ try:
226
+ findings = paged[msgid]
227
+ except KeyError:
228
+ continue
229
+ if not protoerror.is_disabled_perpage(findings, method):
230
+ # content is not disabled
231
+ continue
232
+ result.extend(findings)
233
+ return result
234
+
235
+
236
+ def only_skip(checkers):
237
+ only = collections.defaultdict(set)
238
+ skip = collections.defaultdict(set)
239
+ for item in checkers:
240
+ msgid = item.msgid
241
+ item_only, item_skip = protoerror.control.only_skip(item)
242
+ only[msgid] |= item_only
243
+ skip[msgid] |= item_skip
244
+ only: dict = dict(only)
245
+ skip: dict = dict(skip)
246
+ return only, skip
247
+
248
+
249
+ def dump_result(
250
+ items: iamraw.Findings,
251
+ *,
252
+ unique: bool = False,
253
+ checkers: list = None,
254
+ ) -> DumpedLinterResult:
255
+ """Write linter result to `user` and `developer`-file.
256
+
257
+ Args:
258
+ items(list): list of `Finding`s
259
+ unique(bool): remove duplicated linter findings
260
+ checkers(methods): list of user linters
261
+ Returns:
262
+ Result with dumped user ander developer result in yaml format.
263
+ """
264
+ if unique:
265
+ items = utilo.unique(items)
266
+
267
+ user, developer = split_userdeveloper(items, checkers=checkers)
268
+
269
+ dumped_user = serializeraw.dump_findings(user)
270
+ dumped_developer = serializeraw.dump_findings(developer)
271
+
272
+ result = DumpedLinterResult(user=dumped_user, developer=dumped_developer)
273
+ return result
274
+
275
+
276
+ def write_result(
277
+ result: iamraw.Findings,
278
+ path: str,
279
+ *,
280
+ unique: bool = False,
281
+ user_file=USER_FILE,
282
+ dev_file=DEVELOPER_FILE,
283
+ private: bool = False,
284
+ ):
285
+ """Write linter result to `user` and `developer`-file.
286
+
287
+ Args:
288
+ result(list): list of `Finding`s
289
+ path(str): directory to write both files unique(bool): if unique
290
+ no duplicated user-messages are written
291
+ unique(bool): remove duplication out of result
292
+ user_file(str): filename of user linting file. If None, write
293
+ nothing
294
+ dev_file(str): filename of developer linting file. If None,
295
+ write nothing
296
+ private(bool): use encryption
297
+ """
298
+ assert os.path.isdir(path), str(path)
299
+ dumped_user, dumped_developer = dump_result(result, unique=unique)
300
+ if user_file:
301
+ user_outpath = os.path.join(path, user_file)
302
+ utilo.file_replace(user_outpath, dumped_user, private=private)
303
+ if dev_file:
304
+ developer_outpath = os.path.join(path, dev_file)
305
+ utilo.file_replace(developer_outpath, dumped_developer, private=private)
306
+
307
+
308
+ def from_file(path: str, document: iamraw.DocInfo = None) -> Linter:
309
+ filename = os.path.basename(path)
310
+ spec = importlib.util.spec_from_file_location(
311
+ filename,
312
+ os.path.join(path),
313
+ )
314
+ module = importlib.util.module_from_spec(spec)
315
+ spec.loader.exec_module(module)
316
+ try:
317
+ solution = module.SOLUTION
318
+ except AttributeError as error:
319
+ msg = f'could not create solver, no SOLUTION: {path}'
320
+ raise ValueError(msg) from error
321
+ try:
322
+ status = module.STATUS
323
+ except AttributeError:
324
+ utilo.debug(f'no `STATUS` provided in {path}')
325
+ status = []
326
+ result = from_solution(
327
+ solution,
328
+ status,
329
+ document=document,
330
+ )
331
+ return result
332
+
333
+
334
+ def from_solution(
335
+ solutions: iamraw.Solutions,
336
+ statuses: protoerror.config.MessageStatusList,
337
+ checkers: list = None,
338
+ document: iamraw.DocInfo = None,
339
+ ) -> Linter:
340
+ solver = protoerror.solution.Solver.fromlist(solutions)
341
+ result = Linter(
342
+ solver,
343
+ active=statuses,
344
+ checkers=checkers,
345
+ document=document,
346
+ )
347
+ return result
348
+
349
+
350
+ def from_module(
351
+ name: str,
352
+ tests: set = None,
353
+ skips: set = None,
354
+ document: iamraw.DocInfo = None,
355
+ ) -> Linter:
356
+ result = from_modules(
357
+ [name],
358
+ tests=tests,
359
+ skips=skips,
360
+ document=document,
361
+ )
362
+ return result
363
+
364
+
365
+ def from_modules(
366
+ modules: utilo.Strings,
367
+ tests: set = None,
368
+ skips: set = None,
369
+ document: iamraw.DocInfo = None,
370
+ ) -> Linter:
371
+ modules = module_list(modules)
372
+ status = []
373
+ checkers = []
374
+ solutions = []
375
+ for name in modules:
376
+ with contextlib.suppress(AttributeError):
377
+ # support module type, ensure that module name is str
378
+ name = name.__name__
379
+ module = protoerror.utils.module_fromname(name)
380
+ solutions.extend(
381
+ protoerror.solution.parse_solutions(
382
+ module,
383
+ tests=tests,
384
+ skips=skips,
385
+ ))
386
+ status.extend(parse_active(module))
387
+ checkers.extend(
388
+ protoerror.parse_checkers(
389
+ module,
390
+ tests=tests,
391
+ skips=skips,
392
+ ))
393
+ result = protoerror.from_solution(
394
+ solutions,
395
+ status,
396
+ checkers=checkers,
397
+ document=document,
398
+ )
399
+ return result
400
+
401
+
402
+ def module_list(modulename: list) -> list:
403
+ r"""\
404
+ >>> import protoerror.simple; protoerror.simple.run(protoerror.simple)
405
+ ('[]\n', '[]\n')
406
+ """
407
+ if isinstance(modulename, str):
408
+ return [modulename]
409
+ if type(modulename).__name__ == 'module':
410
+ return [modulename]
411
+ return modulename
412
+
413
+
414
+ def parse_active(module):
415
+ checkers = protoerror.solution.parse_checkers(module)
416
+ result = [
417
+ protoerror.MessageStatus(
418
+ msgid=protoerror.solution.parse_msgid(item.__name__),
419
+ active=item.confidence > 0.0,
420
+ confidence=item.confidence,
421
+ ) for item in checkers
422
+ ]
423
+ return result
424
+
425
+
426
+ # def register(linter):
427
+ # """required method to auto register this checker """
428
+ # linter.register_checker(MisdesignChecker(linter))
protoerror/merger.py ADDED
@@ -0,0 +1,69 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2021-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import utilo
11
+
12
+ import protoerror
13
+
14
+
15
+ def merge_findings(findings):
16
+ """Merge lintings which are neighbors and lint the same error.
17
+
18
+ 1. Group by page
19
+ 2. Group by message id
20
+ 3. Merge equal neighbors
21
+ """
22
+ ranged, notranged = utilo.partition(
23
+ key=lambda x: x.location.shortcut == 'r',
24
+ items=findings,
25
+ )
26
+ result = list(notranged)
27
+ paged = protoerror.bypage(ranged)
28
+ for page in paged:
29
+ msgid = protoerror.byid(page)
30
+ for messagegroup in msgid.values():
31
+ merged = merge_equal_findingid(messagegroup)
32
+ result.extend(merged)
33
+ return result
34
+
35
+
36
+ def merge_equal_findingid(findings):
37
+ if not findings:
38
+ return []
39
+ # remove findings without lines, cause we carn't merge them
40
+ locations, nolines = utilo.partition(
41
+ lambda x: x.location.line is not None,
42
+ findings,
43
+ )
44
+ if not locations:
45
+ # no findings with lines
46
+ return findings
47
+ findings = sorted(locations, key=lambda x: x.location.line)
48
+ result = [findings[0]]
49
+ for item in findings[1:]:
50
+ before = result[-1]
51
+ linebefore = before.location.line
52
+ if before.location.line_end is not None:
53
+ linebefore = before.location.line_end
54
+ if item.msgid != before.msgid:
55
+ result.append(item)
56
+ continue
57
+ if item.solution.description != before.solution.description:
58
+ result.append(item)
59
+ continue
60
+ if item.location.line != linebefore + 1:
61
+ result.append(item)
62
+ continue
63
+ before.location.line_end = linebefore + 1
64
+ # update finding number
65
+ for finding in result:
66
+ finding.number = protoerror.finding.hash_finding(finding)
67
+ # append no lines
68
+ findings.extend(nolines)
69
+ return result
protoerror/messages.py ADDED
@@ -0,0 +1,95 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2019-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+ """The `messages` defines an interface to group problems and sort them by
10
+ weight.
11
+
12
+ MSG_TYPES description:
13
+
14
+ Info: document statistics, wordcount, information about using style,
15
+ document equality to famous writers.
16
+
17
+ Convention: page order - table of content at the end of the
18
+ document, font style, font size, font spacing.
19
+
20
+ Refactor: writing style of paragraph with rewriting hint
21
+
22
+ Warning: Umgangsprache, Black and Write-printing warning, to high/low
23
+ image resolution.
24
+
25
+ Error: Write text over page border, text formatting problem
26
+ "Hurensohn", broken table, wrong citatation, different
27
+ citation styles, missing reference, table of content
28
+ order/level definition.
29
+
30
+ Fatal: An error which blocks further analysis auf the document. In
31
+ general this is an program error which is triggered by pdf state.
32
+
33
+ MSG definition:
34
+
35
+ 'ERROR_CODE': (
36
+ 'REASON',
37
+ 'INTERNAL_SHORTCUT',
38
+ 'DESCRIPTION_OF_THE_PROBLEM',
39
+ ),
40
+
41
+ """
42
+
43
+ import contextlib
44
+
45
+ import utilo
46
+
47
+ # pylint: disable=consider-using-namedtuple-or-dataclass
48
+ MSGS = {
49
+ 'F0000': (
50
+ 'Fehler beim Lesen der PDF Datei.',
51
+ 'pdf-read-error',
52
+ 'Used when pdf-miner is not able to read pdf file.',
53
+ ),
54
+ 'F0001': (
55
+ 'Fehler beim Extrahieren der PDF Datei.',
56
+ 'pdf-extract-error',
57
+ 'Used when environment is not able to read given pdf file.',
58
+ ),
59
+ }
60
+
61
+ MSG_TYPES = {
62
+ "I": "info",
63
+ "C": "convention", # page order
64
+ "R": "refactor", # style of paragraph, page
65
+ "W": "warning", # umgangssprache, image black and white problem
66
+ "E": "error", # writing over border
67
+ "F": "fatal", # pdf analyzing error
68
+ }
69
+
70
+ TYPE_DEFAULT = 'W'
71
+
72
+
73
+ def parse_msgid(msgid: str, idonly: bool = False) -> tuple[str, int]:
74
+ """Split `msgid` into `type` and `number`.
75
+
76
+ Args:
77
+ msgid(str): define type and number of used message
78
+ idonly(bool): do not return msg type
79
+ Returns:
80
+ typ(str), number(int) of message
81
+ """
82
+ if not msgid:
83
+ return msgid
84
+ with contextlib.suppress(ValueError):
85
+ msgid = int(msgid)
86
+ if idonly:
87
+ return msgid
88
+ return TYPE_DEFAULT, msgid
89
+ typ, number = msgid[0], int(msgid[1:])
90
+ typ = typ.upper()
91
+ assert typ in MSG_TYPES, (f'invalid msg type: {typ}; '
92
+ f'use {utilo.from_tuple(MSG_TYPES.keys())}')
93
+ if idonly:
94
+ return number
95
+ return typ, number
protoerror/paged.py ADDED
@@ -0,0 +1,109 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2020-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import concurrent.futures
11
+ import contextlib
12
+ import os
13
+
14
+ import iamraw
15
+ import serializeraw
16
+ import utilo
17
+
18
+ import protoerror
19
+
20
+
21
+ def write_grouped(
22
+ findings: iamraw.Findings,
23
+ dest: str,
24
+ overwrite: bool = True,
25
+ private: bool = False,
26
+ ) -> list:
27
+ result = []
28
+ grouped = protoerror.bypage(findings)
29
+ writer = utilo.file_replace if overwrite else utilo.file_create
30
+ for item in grouped:
31
+ page = fname(item.page)
32
+ outpath = os.path.join(dest, page)
33
+ dumped = serializeraw.dump_findings(item.content)
34
+ writer(outpath, dumped, private=private)
35
+ result.append(outpath)
36
+ return result
37
+
38
+
39
+ def load_grouped(
40
+ source: str,
41
+ pages: tuple = None,
42
+ sort: bool = True,
43
+ worker=10,
44
+ ) -> iamraw.PageFindings:
45
+ if isinstance(pages, int):
46
+ pages = (pages,)
47
+ if pages is None:
48
+ # load all findings
49
+ pages = [
50
+ pagenumber(item, none=True) for item in utilo.file_list(source)
51
+ ]
52
+ # remove invalid file names
53
+ pages = utilo.notnone(pages)
54
+ # yaml parsing is cpu bound, therefore we need a process pool instead
55
+ # of thread pool.
56
+ executor = utilo.select_executor()
57
+ result = []
58
+ with executor(max_workers=worker) as executor:
59
+ todo = {
60
+ executor.submit(load_findings, source, page): page for page in pages
61
+ }
62
+ for job in concurrent.futures.as_completed(todo):
63
+ data = job.result()
64
+ if not data:
65
+ continue
66
+ result.append(data)
67
+ if sort:
68
+ result.sort(key=lambda x: x.page)
69
+ return result
70
+
71
+
72
+ def load_findings(source: str, page: int) -> iamraw.PageFinding:
73
+ name = fname(page)
74
+ source = os.path.join(source, name)
75
+ if not os.path.exists(source):
76
+ return None
77
+ findings = serializeraw.load_findings(source)
78
+ return iamraw.PageFinding(page=page, content=findings)
79
+
80
+
81
+ def fname(page: int) -> str:
82
+ """\
83
+ >>> fname(-5)
84
+ '_05'
85
+ >>> fname(1)
86
+ '001'
87
+ >>> fname(333)
88
+ '333'
89
+ """
90
+ if page < 0:
91
+ return '_' + f'{page*-1}'.zfill(2)
92
+ return f'{page}'.zfill(3)
93
+
94
+
95
+ def pagenumber(page: str, none: bool = True) -> int:
96
+ """\
97
+ >>> pagenumber('010')
98
+ 10
99
+ >>> pagenumber(fname(-10))
100
+ -10
101
+ """
102
+ if page[0] == '_':
103
+ page = '-' + page[1:]
104
+ with contextlib.suppress(ValueError):
105
+ return int(page)
106
+ if none:
107
+ # handle error case
108
+ return None
109
+ raise ValueError(f'could not convert to int: {page}')