bauta 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
bauta/__init__.py ADDED
@@ -0,0 +1,93 @@
1
+ from .configuration import (
2
+ BaseJobConfig,
3
+ Configuration,
4
+ ConfigurationError,
5
+ DatabaseConnectionConfig,
6
+ DatabaseType,
7
+ DataJobConfig,
8
+ DataJobsFile,
9
+ InsertStrategy,
10
+ MaskingConfig,
11
+ expandEnvironmentVariables,
12
+ )
13
+ from .databaseDialects import ColumnCategory, DatabaseDialect, ForeignKey, MariaDBDialect, MSSQLDialect, MySQLDialect, OracleDialect, PostgreSQLDialect, SQLiteDialect
14
+ from .database import Database
15
+ from .dependencyGraph import DependencyGraph, JobOutcome, JobStatus
16
+ from .discovery import TableProposal, proposeTable
17
+ from .log import Log
18
+ from .audit import auditJobs, renderAudit
19
+ from .masking import LOCALES, STRATEGIES, MaskingError, MaskingPlan, Strategy, buildMaskingManifest, keyFingerprint, resolveStrategy, sealManifest, \
20
+ verifyManifest
21
+ from .memory import DATABASE_MEMORY_SCHEMA, DatabaseMemory, FileMemory, MemoryBackend, RunInProgressError, exclusiveRun
22
+ from .reporting import DATABASE_HISTORY_SCHEMA, DatabaseHistory, FileHistory, RunHistory, notify, pushMetrics, writeMetricsFile
23
+ from .runner import RunResult, runDataJobs
24
+ from .subset import SubsetError, SubsetPlan, planSubset
25
+ from .synthesize import SynthesisError, planTable, synthesizeTable
26
+ from .transform import Transform, Transformer, TransformError, TransformResolutionError, resolveTransformer
27
+
28
+ __all__ = [
29
+ 'auditJobs',
30
+ 'BaseJobConfig',
31
+ 'buildMaskingManifest',
32
+ 'ColumnCategory',
33
+ 'Configuration',
34
+ 'ConfigurationError',
35
+ 'DATABASE_HISTORY_SCHEMA',
36
+ 'DATABASE_MEMORY_SCHEMA',
37
+ 'DatabaseHistory',
38
+ 'Database',
39
+ 'DatabaseConnectionConfig',
40
+ 'DatabaseDialect',
41
+ 'DatabaseMemory',
42
+ 'DatabaseType',
43
+ 'exclusiveRun',
44
+ 'expandEnvironmentVariables',
45
+ 'DataJobConfig',
46
+ 'DataJobsFile',
47
+ 'DependencyGraph',
48
+ 'FileHistory',
49
+ 'FileMemory',
50
+ 'ForeignKey',
51
+ 'InsertStrategy',
52
+ 'JobOutcome',
53
+ 'JobStatus',
54
+ 'keyFingerprint',
55
+ 'LOCALES',
56
+ 'Log',
57
+ 'MariaDBDialect',
58
+ 'MaskingConfig',
59
+ 'MaskingError',
60
+ 'MaskingPlan',
61
+ 'MemoryBackend',
62
+ 'MSSQLDialect',
63
+ 'notify',
64
+ 'MySQLDialect',
65
+ 'OracleDialect',
66
+ 'planSubset',
67
+ 'PostgreSQLDialect',
68
+ 'proposeTable',
69
+ 'pushMetrics',
70
+ 'RunHistory',
71
+ 'RunInProgressError',
72
+ 'RunResult',
73
+ 'SQLiteDialect',
74
+ 'STRATEGIES',
75
+ 'Strategy',
76
+ 'SubsetError',
77
+ 'SubsetPlan',
78
+ 'SynthesisError',
79
+ 'synthesizeTable',
80
+ 'planTable',
81
+ 'TableProposal',
82
+ 'Transform',
83
+ 'Transformer',
84
+ 'TransformError',
85
+ 'TransformResolutionError',
86
+ 'renderAudit',
87
+ 'resolveStrategy',
88
+ 'resolveTransformer',
89
+ 'sealManifest',
90
+ 'verifyManifest',
91
+ 'writeMetricsFile',
92
+ 'runDataJobs',
93
+ ]
bauta/audit.py ADDED
@@ -0,0 +1,319 @@
1
+ """What a set of jobs does with data, for a reviewer -- `bauta audit`.
2
+
3
+ Findings a valid policy can still deserve -- an `email` column kept as it is,
4
+ a defaultStrategy of `keep` -- for a person, or with --strict a CI gate.
5
+ Works on plain data; the CLI supplies whatever needs a connection.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import datetime
10
+ from typing import Any, Dict, Iterable, List, Mapping, NamedTuple, Optional, Sequence, Tuple
11
+
12
+ from .configuration import DataJobConfig
13
+ from .databaseDialects import ForeignKey, unqualifiedName
14
+ from .discovery import personalDataHint
15
+ from .masking import MaskingError, MaskingPlan, keyFingerprint, resolveStrategy
16
+
17
+ SEVERITIES = ('error', 'warning', 'info')
18
+
19
+
20
+ class Finding(NamedTuple):
21
+
22
+ severity: str
23
+ job: Optional[str]
24
+ message: str
25
+
26
+
27
+ class _Usage(NamedTuple):
28
+ """How one job masks one column: what has to agree for masks to match.
29
+
30
+ `domain` is None for a strategy that doesn't use the key, and `policy` is
31
+ None for a column copied as it is -- `keep`, or a job that doesn't mask.
32
+ """
33
+
34
+ job: str
35
+ column: str
36
+ domain: Optional[str]
37
+ policy: Optional[str]
38
+ keyFingerprint: Optional[str]
39
+
40
+ @property
41
+ def label(self) -> str:
42
+
43
+ return '{}.{}'.format(self.job, self.column)
44
+
45
+ @property
46
+ def signature(self) -> Tuple[Optional[str], Optional[str], Optional[str]]:
47
+
48
+ return self.domain, self.policy, self.keyFingerprint
49
+
50
+ def describe(self) -> str:
51
+
52
+ if self.policy is None:
53
+ return 'not masked'
54
+ if self.domain is None:
55
+ return 'masked with {}'.format(self.policy)
56
+
57
+ return 'masked with {} in domain {} under key {}'.format(self.policy, self.domain, self.keyFingerprint)
58
+
59
+
60
+ def _describePolicy(policy: Mapping[str, Any]) -> str:
61
+ """A strategy and its options, as a reviewer would compare them."""
62
+
63
+ options = ', '.join('{}: {}'.format(name, policy[name]) for name in sorted(policy) if name not in ('strategy', 'domain'))
64
+
65
+ return '{} ({})'.format(policy['strategy'], options) if options else policy['strategy']
66
+
67
+
68
+ def _usages(name: str, plan: MaskingPlan, columns: Sequence[Mapping[str, Any]]) -> List[_Usage]:
69
+
70
+ fingerprint = keyFingerprint(plan.key)
71
+ declared = {column.upper(): policy for column, policy in plan.columns.items()}
72
+ usages = []
73
+
74
+ for entry in columns:
75
+ policy = declared.get(entry['column'].upper()) if entry['source'] == 'column' else plan.defaultStrategy
76
+ assert policy is not None
77
+ if policy['strategy'] == 'keep':
78
+ usages.append(_Usage(name, entry['column'], None, None, None))
79
+ elif entry['domain'] is None:
80
+ usages.append(_Usage(name, entry['column'], None, _describePolicy(policy), None))
81
+ else:
82
+ usages.append(_Usage(name, entry['column'], entry['domain'], _describePolicy(policy), fingerprint))
83
+
84
+ return usages
85
+
86
+
87
+ def _labels(usages: Iterable[_Usage]) -> str:
88
+
89
+ return ', '.join(sorted(usage.label for usage in usages))
90
+
91
+
92
+ def _auditDomains(target: str, usages: Sequence[_Usage], findings: List[Finding]) -> None:
93
+ """Masks agree only between columns masked in the same domain, the same way,
94
+ under the same key. A domain shared in one target database is a promise that
95
+ they do, so each difference is reported. Copies in different target
96
+ databases may deliberately use different keys, so they aren't compared.
97
+ """
98
+
99
+ byDomain: Dict[str, List[_Usage]] = {}
100
+ for usage in usages:
101
+ if usage.domain is not None:
102
+ byDomain.setdefault(usage.domain, []).append(usage)
103
+
104
+ for domain, shared in sorted(byDomain.items()):
105
+ for what, attribute, noun in (('under {} different keys', 'keyFingerprint', 'key {}'), ('{} different ways', 'policy', '{}')):
106
+ variants: Dict[str, List[_Usage]] = {}
107
+ for usage in shared:
108
+ variants.setdefault(getattr(usage, attribute), []).append(usage)
109
+ if len(variants) < 2:
110
+ continue
111
+ findings.append(Finding('warning', None, 'in {}, domain {} is masked {}, so its masks cannot match across them: {}. '
112
+ 'Mask the domain one way, or give columns that should not match a domain of their own'.format(
113
+ target, domain, what.format(len(variants)),
114
+ '; '.join('{} for {}'.format(noun.format(variant), _labels(group)) for variant, group in sorted(variants.items())))))
115
+
116
+
117
+ def _auditForeignKeys(target: str, jobs: Mapping[str, DataJobConfig], usagesByJob: Mapping[str, Sequence[_Usage]],
118
+ targetColumns: Mapping[str, Sequence[str]], foreignKeys: Sequence[ForeignKey], findings: List[Finding]) -> None:
119
+ """A foreign key survives masking only if its columns are masked exactly as
120
+ the columns they reference. Jobs are matched to a key's tables by their
121
+ targetTableFinal's name, and a masked job's target columns to its query's
122
+ by position, as the load matches them.
123
+ """
124
+
125
+ # (table, column) -> how each job loading that table fills that column
126
+ filled: Dict[Tuple[str, str], List[_Usage]] = {}
127
+ for name, job in sorted(jobs.items()):
128
+ table = unqualifiedName(job.targetTableFinal).upper()
129
+ if job.masking is None:
130
+ filled.setdefault((table, '*'), []).append(_Usage(name, '*', None, None, None))
131
+ continue
132
+ columns = targetColumns.get(name)
133
+ usages = usagesByJob.get(name)
134
+ if columns is None or usages is None or len(columns) != len(usages):
135
+ continue
136
+ for column, usage in zip(columns, usages):
137
+ filled.setdefault((table, column.upper()), []).append(usage)
138
+
139
+ def lookup(table: str, column: str) -> List[_Usage]:
140
+ return filled.get((table.upper(), column.upper()), []) + filled.get((table.upper(), '*'), [])
141
+
142
+ for foreignKey in foreignKeys:
143
+ for column, referencedColumn in zip(foreignKey.columns, foreignKey.referencedColumns):
144
+ for child in lookup(foreignKey.table, column):
145
+ for parent in lookup(foreignKey.referencedTable, referencedColumn):
146
+ # A NULL reference points at nothing, so it can't break.
147
+ if child.signature == parent.signature or child.policy == 'null':
148
+ continue
149
+ findings.append(Finding('warning', child.job, 'in {}, {}.{} is {}, but {}.{}, which it references, is {} (by {}), '
150
+ 'so the copied references will not match'.format(
151
+ target, foreignKey.table, column, child.describe(),
152
+ foreignKey.referencedTable, referencedColumn, parent.describe(), parent.job)))
153
+
154
+
155
+ def _declaredColumns(plan: MaskingPlan) -> List[Dict[str, Any]]:
156
+ """The policy as written, for a job whose query wasn't run."""
157
+
158
+ columns = []
159
+ for column, policy in plan.columns.items():
160
+ keyed = resolveStrategy(policy['strategy']).KEYED
161
+ columns.append({'column': column, 'strategy': policy['strategy'], 'domain': policy.get('domain', column.lower()) if keyed else None,
162
+ 'source': 'column'})
163
+
164
+ return columns
165
+
166
+
167
+ def _auditMaskedJob(name: str, job: DataJobConfig, returned: Optional[Sequence[str]], findings: List[Finding],
168
+ usages: Dict[str, List[_Usage]]) -> Dict[str, Any]:
169
+
170
+ assert job.masking is not None
171
+ plan = MaskingPlan(key=job.masking.key.get_secret_value(), columns=job.masking.columns, defaultStrategy=job.masking.defaultStrategy)
172
+ columns = _declaredColumns(plan)
173
+ resolved = False
174
+
175
+ if returned is not None:
176
+ try:
177
+ columns = [entry._asdict() for entry in plan.bind(returned).manifest]
178
+ resolved = True
179
+ except MaskingError as error:
180
+ findings.append(Finding('error', name, 'the policy does not match what sourceQuery returns: {}'.format(error)))
181
+
182
+ for entry in columns:
183
+ hint = personalDataHint(entry['column'])
184
+ entry['personalDataHint'] = hint
185
+ if entry['strategy'] == 'keep' and hint:
186
+ findings.append(Finding('warning', name, 'column {} is kept unmasked, but its {}'.format(entry['column'], hint)))
187
+
188
+ defaultStrategy = plan.defaultStrategy
189
+ if defaultStrategy is not None:
190
+ if defaultStrategy['strategy'] == 'keep':
191
+ findings.append(Finding('warning', name, 'defaultStrategy is keep, so any column added to the source later is copied unmasked'))
192
+ fallen = [entry['column'] for entry in columns if entry['source'] == 'defaultStrategy']
193
+ if fallen:
194
+ findings.append(Finding('info', name, '{} column(s) fall to defaultStrategy {}: {}'.format(
195
+ len(fallen), defaultStrategy['strategy'], ', '.join(fallen))))
196
+
197
+ redacted = sorted(column for column, policy in plan.columns.items() if policy['strategy'] == 'redact')
198
+ if redacted:
199
+ findings.append(Finding('info', name, 'redact on {}: identifiers with a recognisable shape are removed, names are not'.format(
200
+ ', '.join(redacted))))
201
+
202
+ lenient = sorted(column for column, policy in plan.columns.items() if policy['strategy'] == 'fpe' and not policy.get('strict'))
203
+ if lenient:
204
+ findings.append(Finding('info', name, 'fpe without strict on {}: values too short for FF1 are masked with key instead'.format(
205
+ ', '.join(lenient))))
206
+
207
+ usages[name] = _usages(name, plan, columns)
208
+
209
+ if job.watermarkColumn and any(entry['strategy'] == 'shuffle' for entry in columns):
210
+ findings.append(Finding('warning', name, 'shuffle on an incremental job: its small chunks leave values on or near their own rows'))
211
+
212
+ return {
213
+ 'keyFingerprint': keyFingerprint(job.masking.key.get_secret_value()),
214
+ 'defaultStrategy': defaultStrategy,
215
+ 'columnsResolved': resolved,
216
+ 'columns': columns,
217
+ }
218
+
219
+
220
+ def auditJobs(jobs: Mapping[str, DataJobConfig], returnedColumns: Optional[Mapping[str, Sequence[str]]] = None,
221
+ encryption: Optional[Mapping[str, Optional[bool]]] = None, unreachable: Optional[Mapping[str, str]] = None,
222
+ targetColumns: Optional[Mapping[str, Sequence[str]]] = None, foreignKeys: Optional[Mapping[str, Sequence[ForeignKey]]] = None,
223
+ generatedAt: Optional[datetime.datetime] = None) -> Dict[str, Any]:
224
+ """The audit report, as a JSON-ready dict.
225
+
226
+ `returnedColumns` maps a masked job to the columns its query returns, so
227
+ each column's actual policy can be shown -- defaultStrategy included --
228
+ rather than only the declared ones. `encryption` maps a database alias to
229
+ whether its connection is encrypted (None: couldn't tell). `unreachable`
230
+ maps a job to why its query couldn't be checked. `targetColumns` maps a
231
+ masked job to its target's columns in load order, and `foreignKeys` maps a
232
+ target database alias to the foreign keys that apply to its tables, for
233
+ checking that references still match once masked. All of these come from
234
+ connecting, and all are optional.
235
+ """
236
+
237
+ returnedColumns = returnedColumns or {}
238
+ encryption = encryption or {}
239
+ unreachable = unreachable or {}
240
+ findings: List[Finding] = []
241
+ usages: Dict[str, List[_Usage]] = {}
242
+ maskedSources = {job.sourceDatabase for job in jobs.values() if job.masking is not None}
243
+ report = []
244
+
245
+ for name, job in sorted(jobs.items()):
246
+ entry: Dict[str, Any] = {'job': name, 'active': job.active, 'sourceDatabase': job.sourceDatabase, 'targetDatabase': job.targetDatabase,
247
+ 'targetTable': job.targetTableFinal, 'masked': job.masking is not None}
248
+
249
+ if name in unreachable:
250
+ findings.append(Finding('error', name, 'sourceQuery could not be checked: {}'.format(unreachable[name])))
251
+
252
+ if job.masking is not None:
253
+ entry.update(_auditMaskedJob(name, job, returnedColumns.get(name), findings, usages))
254
+ if encryption.get(job.sourceDatabase) is False:
255
+ findings.append(Finding('warning', name, 'reads unmasked data from {} over a connection that is not encrypted'.format(job.sourceDatabase)))
256
+ elif job.sourceDatabase in maskedSources and job.sourceDatabase != job.targetDatabase:
257
+ findings.append(Finding('warning', name, 'copies from {} without masking, though other jobs mask what they read from it'.format(
258
+ job.sourceDatabase)))
259
+
260
+ report.append(entry)
261
+
262
+ for target in sorted({job.targetDatabase for job in jobs.values()}):
263
+ targetJobs = {name: job for name, job in jobs.items() if job.targetDatabase == target}
264
+ _auditDomains(target, [usage for name in sorted(targetJobs) for usage in usages.get(name, [])], findings)
265
+ if foreignKeys and foreignKeys.get(target):
266
+ _auditForeignKeys(target, targetJobs, usages, targetColumns or {}, foreignKeys[target], findings)
267
+
268
+ for alias, encrypted in sorted(encryption.items()):
269
+ if encrypted is None and alias in maskedSources:
270
+ findings.append(Finding('info', None, 'could not tell whether the connection to {} is encrypted'.format(alias)))
271
+
272
+ findings.sort(key=lambda finding: (SEVERITIES.index(finding.severity), finding.job or '', finding.message))
273
+
274
+ return {
275
+ 'generatedAt': (generatedAt or datetime.datetime.now(datetime.timezone.utc)).isoformat(timespec='seconds'),
276
+ 'jobs': report,
277
+ 'connections': {alias: {'encrypted': encrypted} for alias, encrypted in sorted(encryption.items())},
278
+ 'findings': [finding._asdict() for finding in findings],
279
+ 'summary': {severity: sum(1 for finding in findings if finding.severity == severity) for severity in SEVERITIES},
280
+ }
281
+
282
+
283
+ def renderAudit(report: Mapping[str, Any]) -> str:
284
+ """The report for a terminal: each job's columns, then the findings."""
285
+
286
+ lines = []
287
+
288
+ for job in report['jobs']:
289
+ state = '' if job['active'] else ' (inactive)'
290
+ lines.append('{}{}: {} -> {}.{}'.format(job['job'], state, job['sourceDatabase'], job['targetDatabase'], job['targetTable']))
291
+
292
+ if not job['masked']:
293
+ lines.append(' not masked')
294
+ lines.append('')
295
+ continue
296
+
297
+ scope = 'as returned by sourceQuery' if job['columnsResolved'] else 'as declared (run with --connect to resolve)'
298
+ lines.append(' masked under key {}, columns {}:'.format(job['keyFingerprint'], scope))
299
+ for column in job['columns']:
300
+ domain = ' in domain {}'.format(column['domain']) if column['domain'] else ''
301
+ origin = ' (defaultStrategy)' if column['source'] == 'defaultStrategy' else ''
302
+ lines.append(' {:<28} {}{}{}'.format(column['column'], column['strategy'], domain, origin))
303
+ if job['defaultStrategy'] and not job['columnsResolved']:
304
+ lines.append(' {:<28} {} (defaultStrategy)'.format('any other column', job['defaultStrategy']['strategy']))
305
+ lines.append('')
306
+
307
+ if report['connections']:
308
+ lines.append('Connections:')
309
+ for alias, connection in report['connections'].items():
310
+ encrypted = {True: 'encrypted', False: 'NOT encrypted', None: 'encryption unknown'}[connection['encrypted']]
311
+ lines.append(' {:<28} {}'.format(alias, encrypted))
312
+ lines.append('')
313
+
314
+ summary = report['summary']
315
+ lines.append('Findings: {} error(s), {} warning(s), {} note(s)'.format(summary['error'], summary['warning'], summary['info']))
316
+ for finding in report['findings']:
317
+ lines.append(' {:<8} {}{}'.format(finding['severity'].upper(), '{}: '.format(finding['job']) if finding['job'] else '', finding['message']))
318
+
319
+ return '\n'.join(lines) + '\n'