protoerror 3.20.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
protoerror/control.py ADDED
@@ -0,0 +1,330 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2020-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+ """Control the checker
10
+ ===================
11
+
12
+ Use decorators to exclude linter step cause the step has a problem or
13
+ implementation is not finished yet:
14
+
15
+ .. code-block:: python
16
+
17
+ @skip
18
+ def check_1282_skip_step():
19
+ pass
20
+
21
+ Skip linter step for special document length:
22
+
23
+ .. code-block:: python
24
+
25
+ @nosmall
26
+ @nomedium
27
+ def check_1290_long_page_check():
28
+ pass
29
+
30
+ Supported Decorators: @nosmall, @nomedium, @nolarge
31
+
32
+ Run linter for a special document type:
33
+
34
+ .. code-block:: python
35
+
36
+ @book
37
+ def check_1291_page_border():
38
+ pass
39
+
40
+ Supported Decorators: @homework, @bachelor, @master, @dissertation, @diss, @book
41
+
42
+ Examples
43
+ --------
44
+
45
+ >>> @homework
46
+ ... @nolarge
47
+ ... def check_1234(items):
48
+ ... pass
49
+
50
+ Get list of decorators for a linter step:
51
+
52
+ >>> decorators(check_1234)
53
+ ['nolarge', 'homework']
54
+
55
+ >>> @homework
56
+ ... @disable_perpage(morethan=20)
57
+ ... def check_touch_too_much():
58
+ ... pass
59
+
60
+ >>> decorators(check_touch_too_much)
61
+ [{'disable_perpage': {'morethan': 20}}, 'homework']
62
+
63
+ Templates: Doctype based replacement
64
+ ====================================
65
+
66
+ Add selective advice which are platform dependent, for example give
67
+ different advice when using MSWord instead of Latex:
68
+
69
+ .. code-block:: python
70
+
71
+ {% if MSWORD %}
72
+ ...
73
+ {% endif %}
74
+
75
+ {% if LATEX %}
76
+ ...
77
+ {% endif %}
78
+
79
+ {% if BASE %}
80
+ ...
81
+ {% endif %}
82
+ """
83
+
84
+ import contextlib
85
+
86
+ import configos
87
+ import iamraw
88
+ import utilo
89
+
90
+ MAX_SMALL_PAGE_LENGTH = configos.HV_INT_PLUS(default=35)
91
+ MAX_MEDIUM_PAGE_LENGTH = configos.HV_INT_PLUS(default=35)
92
+
93
+ DOCTYPES = [item.name.lower() for item in iamraw.DocumentType]
94
+
95
+
96
+ def filter_checkers(items: list, document: iamraw.DocInfo) -> list:
97
+ current = document.doctype.name.lower() if document.doctype else None
98
+ result = []
99
+ for item in items:
100
+ decorated = decorators(item)
101
+ if should_skip(decorated, document):
102
+ continue
103
+ if current:
104
+ # skipped document type
105
+ if f'no{current}' in decorated:
106
+ # nohome nobachelor etc.
107
+ continue
108
+ # is check decorated for a special doctype
109
+ some = any(item in decorated for item in DOCTYPES)
110
+ if some and current not in decorated:
111
+ # current document is not selected by decorators, but
112
+ # others are. Therefore we have to skip this ckeck,
113
+ # because this check was not made for current document
114
+ # type.
115
+ continue
116
+ result.append(item)
117
+ return result
118
+
119
+
120
+ def should_skip(decorated, document) -> bool:
121
+ if document.pages is not None:
122
+ small = document.pages < MAX_SMALL_PAGE_LENGTH
123
+ medium = MAX_SMALL_PAGE_LENGTH <= document.pages < MAX_MEDIUM_PAGE_LENGTH
124
+ large = MAX_MEDIUM_PAGE_LENGTH <= document.pages < utilo.INF
125
+ else:
126
+ small, medium, large = False, False, False
127
+ # deactivated method
128
+ if 'skip' in decorated:
129
+ return True
130
+ # verify document length
131
+ if small and 'nosmall' in decorated:
132
+ return True
133
+ if medium and 'nomedium' in decorated:
134
+ return True
135
+ if large and 'nolarge' in decorated:
136
+ return True
137
+ if 'german' in decorated and 'english' not in decorated:
138
+ if document.lang and document.lang != iamraw.Language.GERMAN:
139
+ return True
140
+ return False
141
+
142
+
143
+ def decorateme(method, value):
144
+ try:
145
+ assert value not in method.__control__, str(method.__control__)
146
+ method.__control__.append(value)
147
+ except AttributeError:
148
+ setattr(method, '__control__', [value])
149
+ return method
150
+
151
+
152
+ def decorators(method) -> set:
153
+ assert method, str(method)
154
+ with contextlib.suppress(AttributeError):
155
+ return method.__control__
156
+ return []
157
+
158
+
159
+ # pylint:disable=C0103
160
+ homework = lambda x: decorateme(x, 'homework')
161
+ bachelor = lambda x: decorateme(x, 'bachelor')
162
+ master = lambda x: decorateme(x, 'master')
163
+ diss = lambda x: decorateme(x, 'diss')
164
+ dissertation = diss
165
+ book = lambda x: decorateme(x, 'book')
166
+ paper = lambda x: decorateme(x, 'paper')
167
+ # exluce length of document
168
+ nosmall = lambda x: decorateme(x, 'nosmall')
169
+ nomedium = lambda x: decorateme(x, 'nomedium')
170
+ nolarge = lambda x: decorateme(x, 'nolarge')
171
+ # exclude types of document
172
+ nohome = lambda x: decorateme(x, 'nohomework')
173
+ nobachelor = lambda x: decorateme(x, 'nobachelor')
174
+ nomaster = lambda x: decorateme(x, 'nomaster')
175
+ nodiss = lambda x: decorateme(x, 'nodiss')
176
+ nobook = lambda x: decorateme(x, 'nobook')
177
+ nopaper = lambda x: decorateme(x, 'nopaper')
178
+
179
+ german = lambda x: decorateme(x, 'german')
180
+ english = lambda x: decorateme(x, 'english')
181
+
182
+ skip = lambda x: decorateme(x, 'skip')
183
+
184
+
185
+ def section_skip(sections: iamraw.PartOfDocMixin = None):
186
+ return lambda x: decorateme(x, {'section_skip': sections})
187
+
188
+
189
+ def section_only(sections: iamraw.PartOfDocMixin = None):
190
+ return lambda x: decorateme(x, {'section_only': sections})
191
+
192
+
193
+ def only_skip(method):
194
+ only, skips = set(), set()
195
+ try:
196
+ control = method.__control__
197
+ except AttributeError:
198
+ return only, skips
199
+ for item in control:
200
+ if 'section_only' in item:
201
+ only.add(item['section_only'])
202
+ if 'section_skip' in item:
203
+ skips.add(item['section_skip'])
204
+ return only, skips
205
+
206
+
207
+ def disable_perpage(lessthan=None, morethan=None, equal=None):
208
+ values = {}
209
+ if lessthan is not None:
210
+ values['lessthan'] = lessthan
211
+ if morethan is not None:
212
+ values['morethan'] = morethan
213
+ if equal is not None:
214
+ values['equal'] = equal
215
+ return lambda x: decorateme(x, {'disable_perpage': values})
216
+
217
+
218
+ def enable_perpage(lessthan=None, morethan=None, equal=None):
219
+ values = {}
220
+ if lessthan is not None:
221
+ values['lessthan'] = lessthan
222
+ if morethan is not None:
223
+ values['morethan'] = morethan
224
+ if equal is not None:
225
+ values['equal'] = equal
226
+ return lambda x: decorateme(x, {'enable_perpage': values})
227
+
228
+
229
+ def get_perpage(methods):
230
+ result = []
231
+ for method in methods:
232
+ decorator = decorators(method)
233
+ if 'perpage' not in str(decorator):
234
+ continue
235
+ result.append(method)
236
+ return result
237
+
238
+
239
+ def is_disabled_perpage(findings, method) -> bool:
240
+ control = method.__control__
241
+ for item in control:
242
+ try:
243
+ disableperpage = item['disable_perpage']
244
+ except (TypeError, KeyError):
245
+ continue
246
+ with contextlib.suppress(KeyError):
247
+ if disableperpage['equal'] == len(findings):
248
+ return True
249
+ with contextlib.suppress(KeyError):
250
+ if len(findings) <= disableperpage['lessthan']:
251
+ return True
252
+ with contextlib.suppress(KeyError):
253
+ if len(findings) >= disableperpage['morethan']:
254
+ return True
255
+ return False
256
+
257
+
258
+ DOCINFO_PATTERN = utilo.compiles(r"""
259
+ ^
260
+ (?P<typ>homework|bachelor|master|diss|habil|book|paper)?
261
+ (?P<pages>\d{1,4})?
262
+ (?P<lang>ger|eng|fre|german|english|french)?
263
+ $
264
+ """)
265
+
266
+ DOCINFO = 'bachelor64ger'
267
+
268
+
269
+ def parse_docinfo(docinfo) -> iamraw.DocInfo:
270
+ """\
271
+ >>> parse_docinfo('diss215eng')
272
+ DocInfo(pages=215, doctype=...DISS...lang=...ENGLISH...)
273
+ >>> parse_docinfo('15GER')
274
+ DocInfo(pages=15...NONE...lang=<Language.GERMAN...)
275
+ >>> assert parse_docinfo(None) is None
276
+ """
277
+ if not docinfo:
278
+ return None
279
+ parsed = DOCINFO_PATTERN.match(docinfo)
280
+ if not parsed:
281
+ return None
282
+ doctype = iamraw.DocumentType.NONE
283
+ pages = 256
284
+ lang = iamraw.Language.GERMAN # default lang
285
+ with contextlib.suppress(KeyError):
286
+ if parsed['pages']:
287
+ pages = int(parsed['pages'])
288
+ with contextlib.suppress(KeyError, AttributeError):
289
+ doctype = iamraw.DocumentType[parsed['typ'].upper()]
290
+ with contextlib.suppress(KeyError):
291
+ if parsed['lang']:
292
+ lang = parse_lang(parsed['lang'])
293
+ result = iamraw.DocInfo(
294
+ pages=pages,
295
+ doctype=doctype,
296
+ lang=lang,
297
+ )
298
+ return result
299
+
300
+
301
+ def parse_lang(lang: str) -> iamraw.Language:
302
+ lang = lang.lower()
303
+ if lang in 'german':
304
+ return iamraw.Language.GERMAN
305
+ if lang in 'english':
306
+ return iamraw.Language.ENGLISH
307
+ if lang in 'french':
308
+ return iamraw.Language.FRENCH
309
+ return iamraw.Language.UNKNOWN
310
+
311
+
312
+ def integrate_docinfo():
313
+ hook = integrate_cli
314
+ run = evaluate_userchoice
315
+ return hook, run
316
+
317
+
318
+ def evaluate_userchoice(argv):
319
+ docinfo = argv.get('docinfo', None)
320
+ parsed = parse_docinfo(docinfo)
321
+ if parsed is None:
322
+ parsed = iamraw.DocInfo()
323
+ return {'docinfo': parsed}
324
+
325
+
326
+ def integrate_cli(parser):
327
+ parser.add_argument(
328
+ '--docinfo',
329
+ help='define docinfo to tell document type, length and language',
330
+ )
protoerror/finding.py ADDED
@@ -0,0 +1,145 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2019-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import concurrent.futures
11
+ import os
12
+
13
+ import iamraw
14
+ import serializeraw
15
+ import utilo
16
+
17
+ import protoerror
18
+
19
+
20
+ def findings_from_path(
21
+ path: str,
22
+ worker: int = 10,
23
+ useronly: bool = True,
24
+ msgid: set = None,
25
+ ) -> iamraw.PageFindings:
26
+ """Load Findings from `path` directory and group them by page as
27
+ `PageFindings`."""
28
+ assert os.path.isdir(path), str(path)
29
+ files = utilo.file_list(path, include='yaml', recursive=True)
30
+ if useronly:
31
+ files = [
32
+ item for item in files if utilo.file_name(item).endswith('_user')
33
+ ]
34
+ paths = [os.path.join(path, item) for item in files]
35
+ # limit worker by max file count
36
+ worker = utilo.mins(worker, len(files))
37
+ # ensure to have at least one worker when collection now file
38
+ worker = utilo.maxs(1, worker)
39
+ # yaml parsing is cpu bound, therefore we need a process pool instead
40
+ # of thread pool.
41
+ executor = utilo.select_executor()
42
+ with executor(max_workers=worker) as executor:
43
+ todo = {
44
+ executor.submit(serializeraw.load_findings, path): path
45
+ for path in paths
46
+ }
47
+ findings = []
48
+ for job in concurrent.futures.as_completed(todo):
49
+ data = job.result()
50
+ findings.extend(data)
51
+ if msgid:
52
+ # select findings by msgid
53
+ findings = protoerror.select_findings(findings, msgid=msgid)
54
+ result = protoerror.bypage(findings)
55
+ return result
56
+
57
+
58
+ def iter_findings(path: str):
59
+ files = utilo.file_list(path, include='yaml', recursive=True)
60
+ files = [item for item in files if utilo.file_name(item).endswith('_user')]
61
+ for item in files:
62
+ location = os.path.join(path, item)
63
+ findings = serializeraw.load_findings(location)
64
+ yield location, findings
65
+
66
+
67
+ def hash_finding(item):
68
+ try:
69
+ return hash(item)
70
+ except TypeError as error:
71
+ utilo.error(f'could not hash finding: {item}')
72
+ raise error
73
+
74
+
75
+ def make_finding_number_unique(path: str, private: bool = False) -> bool:
76
+ """Collect all findings from path and replace with unqiue finding
77
+ number.
78
+
79
+ Note: Remove lintings with equal hash cause there seem/must to be
80
+ equal.
81
+
82
+ Args:
83
+ path(str): location where files wither user linter are located
84
+ private(bool): encrypt result
85
+ Returns:
86
+ True if some file was located and replace.
87
+ False if no user file is in `path`.
88
+ """
89
+ assert os.path.isdir(path), str(path)
90
+ single = utilo.Single()
91
+ replaced = False
92
+ for location, findings in iter_findings(path):
93
+ for finding in findings:
94
+ hashed = hash_finding(finding)
95
+ if single.contains(hashed):
96
+ utilo.error(f'duplicated finding: {finding}')
97
+ finding.number = None # None -> do not dump this finding
98
+ continue
99
+ finding.number = hashed
100
+ findings = [item for item in findings if item.number is not None]
101
+ # TODO: REFACTOR LATER
102
+ dumped = serializeraw.dump_findings(findings)
103
+ utilo.file_replace(location, dumped, private=private)
104
+ replaced = True
105
+ return replaced
106
+
107
+
108
+ def finding_status_update(
109
+ path: str,
110
+ number: int,
111
+ status: iamraw.ProblemStatus,
112
+ private: bool = False,
113
+ ) -> bool:
114
+ assert os.path.isdir(path), str(path)
115
+ assert isinstance(number, int), type(number)
116
+ assert isinstance(status, iamraw.ProblemStatus), type(status)
117
+ # TODO: IMPROVE SPEED LATER? MAY USE A BUFFERED OBJECT ORIENTED APPROACH
118
+ for location, findings in iter_findings(path):
119
+ for finding in findings:
120
+ if finding.number != number:
121
+ continue
122
+ if finding.solution is None:
123
+ utilo.error(f'could not update status: {finding}')
124
+ return False
125
+ finding.solution.status = status
126
+ dumped = serializeraw.dump_findings(findings)
127
+ utilo.debug(f'number: {number}; status: {status};\n'
128
+ f'update: {location}')
129
+ utilo.file_replace(location, dumped, private=private)
130
+ return True
131
+ return False
132
+
133
+
134
+ def finding_status(path: str, number: int) -> iamraw.ProblemStatus:
135
+ assert os.path.isdir(path), str(path)
136
+ assert isinstance(number, int), type(number)
137
+ for _, findings in iter_findings(path):
138
+ for finding in findings:
139
+ if finding.number != number:
140
+ continue
141
+ if finding.solution is None:
142
+ utilo.error(f'could not get status: {finding}')
143
+ return None
144
+ return finding.solution.status
145
+ return None
protoerror/group.py ADDED
@@ -0,0 +1,158 @@
1
+ # =============================================================================
2
+ # C O P Y R I G H T
3
+ # -----------------------------------------------------------------------------
4
+ # Copyright (c) 2019-2022 by Helmut Konrad Fahrendholz. All rights reserved.
5
+ # This file is property of Helmut Konrad Fahrendholz. Any unauthorized copy,
6
+ # use or distribution is an offensive act against international law and may
7
+ # be prosecuted under federal law. Its content is company confidential.
8
+ # =============================================================================
9
+
10
+ import collections
11
+ import contextlib
12
+
13
+ import iamraw
14
+ import utilo
15
+
16
+ import protoerror
17
+
18
+
19
+ def bypage(items: iamraw.Findings) -> iamraw.PageFindings:
20
+ """Group `items` by location.page of `Finding`. Sort the groups
21
+ ascending by page number."""
22
+ pages = collections.defaultdict(list)
23
+ for item in items:
24
+ assert item.location is not None, f'require location {item}'
25
+ pages[item.location.page].append(item)
26
+ result = [
27
+ iamraw.PageFinding(page=page, content=pages[page])
28
+ for page in sorted(pages.keys())
29
+ ]
30
+ return result
31
+
32
+
33
+ def byid(items: iamraw.Findings) -> dict:
34
+ """Group findings by `finding.msgid`."""
35
+ grouped = collections.defaultdict(list)
36
+ for item in items:
37
+ grouped[item.msgid].append(item)
38
+ result = dict(grouped)
39
+ return result
40
+
41
+
42
+ def filter_mark(items: iamraw.Findings, shortcut: str) -> iamraw.Findings:
43
+ """Filter `Findings` by shortcut and sort them by `location.value`
44
+ afterwards.
45
+
46
+ Args:
47
+ items(iamraw.Findings): list of findings
48
+ shortcut(str): shortcut of protoerror.location, w word, p page,
49
+ ol oneline, etc.
50
+ Returns:
51
+ filtered, sorted list of `Findings`
52
+ """
53
+ for item in items:
54
+ if item.location:
55
+ continue
56
+ utilo.error(f'missing location: {item}')
57
+ items = [finding for finding in items if finding.location]
58
+ selected = []
59
+ for item in items:
60
+ with contextlib.suppress(AttributeError):
61
+ if not isinstance(value(item.location), int):
62
+ utilo.debug(f'invalid location: {item.location}, require int.')
63
+ continue
64
+ if item.location.shortcut == shortcut:
65
+ selected.append(item)
66
+ selected.sort(key=lambda x: value(x.location))
67
+ return selected
68
+
69
+
70
+ def value(location) -> int:
71
+ with contextlib.suppress(AttributeError):
72
+ if location.value is not None:
73
+ return location.value
74
+ with contextlib.suppress(AttributeError):
75
+ if location.line is not None:
76
+ return location.line
77
+ if location.page is not None:
78
+ return location.page
79
+ return utilo.INF
80
+
81
+
82
+ def words(items: iamraw.Findings) -> iamraw.Findings:
83
+ return filter_mark(items, shortcut='w')
84
+
85
+
86
+ def lines(items: iamraw.Findings) -> iamraw.Findings:
87
+ return filter_mark(items, shortcut='ol')
88
+
89
+
90
+ def sentences(items: iamraw.Findings) -> iamraw.Findings:
91
+ return filter_mark(items, shortcut='s')
92
+
93
+
94
+ def ranged(items: iamraw.Findings) -> iamraw.Findings:
95
+ return protoerror.filter_mark(items, shortcut='r')
96
+
97
+
98
+ def select_findings(
99
+ findings: iamraw.Findings,
100
+ msgid: set,
101
+ ) -> iamraw.Findings:
102
+ """Select `Findings` specified by `msgid`
103
+
104
+ >>> select_findings([iamraw.Finding(msgid=1337), iamraw.Finding(msgid=1338)], msgid=(1337,1400))
105
+ [Finding(...msgid=1337...)]
106
+ >>> select_findings([iamraw.Finding(msgid=1337), iamraw.Finding(msgid=1338)], msgid=1337)
107
+ [Finding(...msgid=1337...)]
108
+ """
109
+ assert all(isinstance(item, iamraw.Finding) for item in findings)
110
+ if msgid is None:
111
+ return findings
112
+ if isinstance(msgid, int):
113
+ msgid = {msgid}
114
+ elif isinstance(msgid, list):
115
+ msgid = set(msgid)
116
+ return [item for item in findings if item.msgid in msgid]
117
+
118
+
119
+ def count_findings(findings: iamraw.Findings, msgid: set) -> int:
120
+ counted = len(select_findings(findings, msgid))
121
+ return counted
122
+
123
+
124
+ def select_pages(
125
+ findings: iamraw.Findings,
126
+ pages: int,
127
+ ) -> iamraw.Findings:
128
+ if pages is None:
129
+ return findings
130
+ pages = {pages} if isinstance(pages, int) else pages
131
+ findings = flat(findings)
132
+ findings = [
133
+ finding for finding in findings
134
+ if finding.location and finding.location.page in pages
135
+ ]
136
+ return findings
137
+
138
+
139
+ def flat(pages: iamraw.PageFinding) -> list:
140
+ result = []
141
+ for page in pages:
142
+ try:
143
+ # PageFinding
144
+ result.extend(page.content)
145
+ except AttributeError:
146
+ # findings are already flat
147
+ result.append(page)
148
+ return result
149
+
150
+
151
+ def select(
152
+ findings: iamraw.Findings,
153
+ pages: int,
154
+ msgid: set,
155
+ ) -> iamraw.Findings:
156
+ findings = select_pages(findings, pages=pages)
157
+ findings = select_findings(findings, msgid=msgid)
158
+ return findings