simple-detect-secrets 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. simple_detect_secrets/__init__.py +3 -0
  2. simple_detect_secrets/core/__init__.py +0 -0
  3. simple_detect_secrets/core/baseline.py +249 -0
  4. simple_detect_secrets/core/bidirectional_iterator.py +34 -0
  5. simple_detect_secrets/core/code_snippet.py +101 -0
  6. simple_detect_secrets/core/color.py +14 -0
  7. simple_detect_secrets/core/common.py +11 -0
  8. simple_detect_secrets/core/constants.py +43 -0
  9. simple_detect_secrets/core/log.py +51 -0
  10. simple_detect_secrets/core/potential_secret.py +92 -0
  11. simple_detect_secrets/core/secrets_collection.py +384 -0
  12. simple_detect_secrets/core/usage.py +404 -0
  13. simple_detect_secrets/main.py +164 -0
  14. simple_detect_secrets/plugins/__init__.py +0 -0
  15. simple_detect_secrets/plugins/artifactory.py +16 -0
  16. simple_detect_secrets/plugins/aws.py +161 -0
  17. simple_detect_secrets/plugins/base.py +322 -0
  18. simple_detect_secrets/plugins/basic_auth.py +23 -0
  19. simple_detect_secrets/plugins/cloudant.py +100 -0
  20. simple_detect_secrets/plugins/common/__init__.py +0 -0
  21. simple_detect_secrets/plugins/common/constants.py +30 -0
  22. simple_detect_secrets/plugins/common/filetype.py +48 -0
  23. simple_detect_secrets/plugins/common/filters.py +151 -0
  24. simple_detect_secrets/plugins/common/ini_file_parser.py +179 -0
  25. simple_detect_secrets/plugins/common/initialize.py +218 -0
  26. simple_detect_secrets/plugins/common/util.py +54 -0
  27. simple_detect_secrets/plugins/common/yaml_file_parser.py +151 -0
  28. simple_detect_secrets/plugins/high_entropy_strings.py +423 -0
  29. simple_detect_secrets/plugins/ibm_cloud_iam.py +55 -0
  30. simple_detect_secrets/plugins/jwt.py +58 -0
  31. simple_detect_secrets/plugins/keyword.py +339 -0
  32. simple_detect_secrets/plugins/mailchimp.py +38 -0
  33. simple_detect_secrets/plugins/private_key.py +55 -0
  34. simple_detect_secrets/plugins/slack.py +49 -0
  35. simple_detect_secrets/plugins/softlayer.py +76 -0
  36. simple_detect_secrets/plugins/stripe.py +39 -0
  37. simple_detect_secrets/pre_commit_hook.py +225 -0
  38. simple_detect_secrets/util.py +123 -0
  39. simple_detect_secrets-0.0.1.dist-info/METADATA +70 -0
  40. simple_detect_secrets-0.0.1.dist-info/RECORD +44 -0
  41. simple_detect_secrets-0.0.1.dist-info/WHEEL +4 -0
  42. simple_detect_secrets-0.0.1.dist-info/entry_points.txt +3 -0
  43. simple_detect_secrets-0.0.1.dist-info/licenses/LICENSE +201 -0
  44. simple_detect_secrets-0.0.1.dist-info/licenses/NOTICE +1 -0
@@ -0,0 +1,3 @@
1
+ from importlib.metadata import version
2
+
3
+ VERSION = version('simple-detect-secrets')
File without changes
@@ -0,0 +1,249 @@
1
+ import os
2
+ import re
3
+ import subprocess
4
+
5
+ from simple_detect_secrets import util
6
+ from simple_detect_secrets.core.log import get_logger
7
+ from simple_detect_secrets.core.secrets_collection import SecretsCollection
8
+
9
+ log = get_logger(format_string='%(message)s')
10
+
11
+
12
+ def initialize(
13
+ path,
14
+ plugins,
15
+ exclude_files_regex=None,
16
+ exclude_lines_regex=None,
17
+ word_list_file=None,
18
+ word_list_hash=None,
19
+ should_scan_all_files=False,
20
+ ):
21
+ """Scans the entire codebase for secrets, and returns a
22
+ SecretsCollection object.
23
+
24
+ :type path: list
25
+
26
+ :type plugins: tuple of detect_secrets.plugins.base.BasePlugin
27
+ :param plugins: rules to initialize the SecretsCollection with.
28
+
29
+ :type exclude_files_regex: str|None
30
+ :type exclude_lines_regex: str|None
31
+
32
+ :type word_list_file: str|None
33
+ :param word_list_file: optional word list file for ignoring certain words.
34
+
35
+ :type word_list_hash: str|None
36
+ :param word_list_hash: optional iterated sha1 hash of the words in the word list.
37
+
38
+ :type should_scan_all_files: bool
39
+
40
+ :rtype: SecretsCollection
41
+ """
42
+ output = SecretsCollection(
43
+ plugins,
44
+ exclude_files=exclude_files_regex,
45
+ exclude_lines=exclude_lines_regex,
46
+ word_list_file=word_list_file,
47
+ word_list_hash=word_list_hash,
48
+ )
49
+
50
+ files_to_scan = []
51
+ for element in path:
52
+ if os.path.isdir(element):
53
+ if should_scan_all_files:
54
+ files_to_scan.extend(
55
+ _get_files_recursively(element),
56
+ )
57
+ else:
58
+ files_to_scan.extend(
59
+ _get_git_tracked_files(element),
60
+ )
61
+ elif os.path.isfile(element):
62
+ files_to_scan.append(element)
63
+ else:
64
+ log.error('detect-secrets: %s: No such file or directory', element)
65
+
66
+ if not files_to_scan:
67
+ return output
68
+
69
+ if exclude_files_regex:
70
+ exclude_files_regex = re.compile(exclude_files_regex, re.IGNORECASE)
71
+ files_to_scan = filter(
72
+ lambda file: not exclude_files_regex.search(file),
73
+ files_to_scan,
74
+ )
75
+
76
+ for file in sorted(files_to_scan):
77
+ output.scan_file(file)
78
+
79
+ return output
80
+
81
+
82
+ def get_secrets_not_in_baseline(results, baseline):
83
+ """
84
+ :type results: SecretsCollection
85
+ :param results: SecretsCollection of current results
86
+
87
+ :type baseline: SecretsCollection
88
+ :param baseline: SecretsCollection of baseline results.
89
+ This will be updated accordingly (by reference)
90
+
91
+ :rtype: SecretsCollection
92
+ :returns: SecretsCollection of new results (filtering out baseline)
93
+ """
94
+ exclude_files_regex = None
95
+ if baseline.exclude_files:
96
+ exclude_files_regex = re.compile(baseline.exclude_files, re.IGNORECASE)
97
+
98
+ new_secrets = SecretsCollection()
99
+ for filename in results.data:
100
+ if exclude_files_regex and exclude_files_regex.search(filename):
101
+ continue
102
+
103
+ if filename not in baseline.data:
104
+ # We don't have a previous record of this file, so obviously
105
+ # everything is new.
106
+ new_secrets.data[filename] = results.data[filename]
107
+ continue
108
+
109
+ # The __hash__ method of PotentialSecret makes this work
110
+ filtered_results = {
111
+ secret: secret
112
+ for secret in results.data[filename]
113
+ if secret not in baseline.data[filename]
114
+ }
115
+
116
+ if filtered_results:
117
+ new_secrets.data[filename] = filtered_results
118
+
119
+ return new_secrets
120
+
121
+
122
+ def trim_baseline_of_removed_secrets(results, baseline, filelist):
123
+ """
124
+ NOTE: filelist is not a comprehensive list of all files in the repo
125
+ (because we can't be sure whether --all-files is passed in as a
126
+ parameter to pre-commit).
127
+
128
+ :type results: SecretsCollection
129
+ :type baseline: SecretsCollection
130
+
131
+ :type filelist: list(str)
132
+ :param filelist: filenames that are scanned.
133
+
134
+ :rtype: bool
135
+ :returns: True if baseline was updated
136
+ """
137
+ updated = False
138
+ for filename in filelist:
139
+ if filename not in baseline.data:
140
+ # Nothing to modify, because not even there in the first place.
141
+ continue
142
+
143
+ if filename not in results.data:
144
+ # All secrets relating to that file was removed.
145
+ # We know this because:
146
+ # 1. It's a file that was scanned (in filelist)
147
+ # 2. It was in the baseline
148
+ # 3. It has no results now.
149
+ del baseline.data[filename]
150
+ updated = True
151
+ continue
152
+
153
+ # We clone the baseline, so that we can modify the baseline,
154
+ # without messing up the iteration.
155
+ for baseline_secret in baseline.data[filename].copy():
156
+ new_secret_found = results.get_secret(
157
+ filename,
158
+ baseline_secret.secret_hash,
159
+ baseline_secret.type,
160
+ )
161
+
162
+ if not new_secret_found:
163
+ # No longer in results, so can remove from baseline
164
+ old_secret_to_delete = baseline.get_secret(
165
+ filename,
166
+ baseline_secret.secret_hash,
167
+ baseline_secret.type,
168
+ )
169
+ del baseline.data[filename][old_secret_to_delete]
170
+ updated = True
171
+
172
+ elif new_secret_found.lineno != baseline_secret.lineno:
173
+ # Secret moved around, should update baseline with new location
174
+ old_secret_to_update = baseline.get_secret(
175
+ filename,
176
+ baseline_secret.secret_hash,
177
+ baseline_secret.type,
178
+ )
179
+ old_secret_to_update.lineno = new_secret_found.lineno
180
+ updated = True
181
+
182
+ return updated
183
+
184
+
185
+ def format_baseline_for_output(baseline):
186
+ """
187
+ :type baseline: dict
188
+ :rtype: str
189
+ """
190
+ lines = []
191
+ for filename, secret_list in baseline['results'].items():
192
+ lines_secrets = '\n'.join(
193
+ 'Line %d: %s' % (x['line_number'], x['secret_value']) for x in secret_list
194
+ )
195
+ line = f"""
196
+ Filename: {filename}
197
+ {lines_secrets}
198
+ """
199
+ lines.append(line)
200
+
201
+ return '\n'.join(lines)
202
+
203
+
204
+ def _get_git_tracked_files(rootdir='.'):
205
+ """Parsing .gitignore rules is hard.
206
+
207
+ However, a way we can get around this problem by just listing all
208
+ currently tracked git files, and start our search from there.
209
+ After all, if it isn't in the git repo, we're not concerned about
210
+ it, because secrets aren't being entered in a shared place.
211
+
212
+ :type rootdir: str
213
+ :param rootdir: root directory of where you want to list files from
214
+
215
+ :rtype: set|None
216
+ :returns: filepaths to files which git currently tracks (locally)
217
+ """
218
+ output = []
219
+ try:
220
+ with open(os.devnull, 'w') as fnull:
221
+ git_files = subprocess.check_output(
222
+ [
223
+ 'git',
224
+ '-C',
225
+ rootdir,
226
+ 'ls-files',
227
+ ],
228
+ stderr=fnull,
229
+ )
230
+ for filename in git_files.decode('utf-8').split():
231
+ relative_path = util.get_relative_path_if_in_cwd(rootdir, filename)
232
+ if relative_path:
233
+ output.append(relative_path)
234
+ except subprocess.CalledProcessError:
235
+ pass
236
+ return output
237
+
238
+
239
+ def _get_files_recursively(rootdir):
240
+ """Sometimes, we want to use this tool with non-git repositories.
241
+ This function allows us to do so.
242
+ """
243
+ output = []
244
+ for root, _, files in os.walk(rootdir):
245
+ for filename in files:
246
+ relative_path = util.get_relative_path_if_in_cwd(root, filename)
247
+ if relative_path:
248
+ output.append(relative_path)
249
+ return output
@@ -0,0 +1,34 @@
1
+ class BidirectionalIterator:
2
+ def __init__(self, collection):
3
+ self.collection = collection
4
+ self.index = -1 # Starts on -1, as index is increased _before_ getting result
5
+ self.step_back_once = False
6
+
7
+ def __next__(self):
8
+ if self.step_back_once:
9
+ self.index -= 1
10
+ self.step_back_once = False
11
+ else:
12
+ self.index += 1
13
+
14
+ if self.index < 0:
15
+ raise StopIteration
16
+
17
+ try:
18
+ result = self.collection[self.index]
19
+ except IndexError:
20
+ raise StopIteration
21
+
22
+ return result
23
+
24
+ def next(self): # pragma: no cover
25
+ return self.__next__()
26
+
27
+ def step_back_on_next_iteration(self):
28
+ self.step_back_once = True
29
+
30
+ def can_step_back(self):
31
+ return self.index > 0
32
+
33
+ def __iter__(self): # pragma: no cover
34
+ return self
@@ -0,0 +1,101 @@
1
+ import itertools
2
+
3
+ from .color import AnsiColor, colorize
4
+
5
+
6
+ class CodeSnippetHighlighter:
7
+ def get_code_snippet(self, file_lines, line_number, lines_of_context=5):
8
+ """
9
+ :type file_lines: iterable of str
10
+ :param file_lines: an iterator of lines in the file
11
+
12
+ :type line_number: int
13
+ :param line_number: line which you want to focus on
14
+
15
+ :type lines_of_context: int
16
+ :param lines_of_context: how many lines to display around the line you want
17
+ to focus on.
18
+
19
+ :rtype: CodeSnippet
20
+ """
21
+ secret_line_index = line_number - 1
22
+ end_line = secret_line_index + lines_of_context + 1
23
+
24
+ if secret_line_index <= lines_of_context:
25
+ start_line = 0
26
+ index_of_secret_in_output = secret_line_index
27
+ else:
28
+ start_line = secret_line_index - lines_of_context
29
+ index_of_secret_in_output = lines_of_context
30
+
31
+ return CodeSnippet(
32
+ list(
33
+ itertools.islice(
34
+ file_lines,
35
+ start_line,
36
+ end_line,
37
+ ),
38
+ ),
39
+ start_line,
40
+ index_of_secret_in_output,
41
+ )
42
+
43
+
44
+ class CodeSnippet:
45
+ def __init__(self, snippet, start_line, target_index):
46
+ """
47
+ :type snippet: iterable and indexable of str
48
+ :param snippet: lines of code extracted from file
49
+
50
+ :type start_line: int
51
+ :param start_line: first line number in segment
52
+
53
+ :type target_index: int
54
+ :param target_index: index in snippet of target line
55
+ """
56
+ self.lines = snippet
57
+ self.start_line = start_line
58
+ self.target_index = target_index
59
+
60
+ @property
61
+ def target_line(self):
62
+ return self.lines[self.target_index]
63
+
64
+ @target_line.setter
65
+ def target_line(self, value):
66
+ self.lines[self.target_index] = value
67
+
68
+ def add_line_numbers(self):
69
+ for index, line in enumerate(self.lines):
70
+ self.lines[index] = f'{self.get_line_number(self.start_line + index + 1)}:{line}'
71
+
72
+ return self
73
+
74
+ def highlight_line(self, payload):
75
+ """
76
+ :type payload: str
77
+ :param payload: string to highlight, on chosen line
78
+ """
79
+ index_of_payload = self.target_line.lower().index(payload.lower())
80
+ end_of_payload = index_of_payload + len(payload)
81
+
82
+ self.target_line = f'{self.target_line[:index_of_payload]}{self.apply_highlight(self.target_line[index_of_payload:end_of_payload])}{self.target_line[end_of_payload:]}'
83
+
84
+ return self
85
+
86
+ def get_line_number(self, line_number):
87
+ """Broken out, for custom colorization."""
88
+ return colorize(
89
+ str(line_number),
90
+ AnsiColor.LIGHT_GREEN,
91
+ )
92
+
93
+ def apply_highlight(self, payload):
94
+ """Broken out, for custom colorization."""
95
+ return colorize(
96
+ payload,
97
+ AnsiColor.RED_BACKGROUND,
98
+ )
99
+
100
+ def __str__(self):
101
+ return '\n'.join(self.lines)
@@ -0,0 +1,14 @@
1
+ from enum import Enum
2
+
3
+
4
+ class AnsiColor(Enum):
5
+ RESET = '[0m'
6
+ BOLD = '[1m'
7
+ RED = '[91m'
8
+ RED_BACKGROUND = '[41m'
9
+ LIGHT_GREEN = '[92m'
10
+ PURPLE = '[95m'
11
+
12
+
13
+ def colorize(text, color):
14
+ return f'\x1b{color.value}{text}\x1b{AnsiColor.RESET.value}'
@@ -0,0 +1,11 @@
1
+ from .baseline import format_baseline_for_output
2
+
3
+
4
+ def write_baseline_to_file(filename, data):
5
+ """
6
+ :type filename: str
7
+ :type data: dict
8
+ :rtype: None
9
+ """
10
+ with open(filename, 'w') as f: # pragma: no cover
11
+ f.write(format_baseline_for_output(data) + '\n')
@@ -0,0 +1,43 @@
1
+ from enum import Enum
2
+
3
+ # We don't scan files with these extensions.
4
+ # NOTE: We might be able to do this better with
5
+ # `subprocess.check_output(['file', filename])`
6
+ # and look for "ASCII text", but that might be more expensive.
7
+ #
8
+ # Definitely something to look into, if this list gets unruly long.
9
+ IGNORED_FILE_EXTENSIONS = {
10
+ '.7z',
11
+ '.bmp',
12
+ '.bz2',
13
+ '.dmg',
14
+ '.eot',
15
+ '.exe',
16
+ '.gif',
17
+ '.gz',
18
+ '.ico',
19
+ '.jar',
20
+ '.jpg',
21
+ '.jpeg',
22
+ '.mo',
23
+ '.png',
24
+ '.rar',
25
+ '.realm',
26
+ '.s7z',
27
+ '.svg',
28
+ '.tar',
29
+ '.tif',
30
+ '.tiff',
31
+ '.ttf',
32
+ '.webp',
33
+ '.woff',
34
+ '.xls',
35
+ '.xlsx',
36
+ '.zip',
37
+ }
38
+
39
+
40
+ class VerifiedResult(Enum):
41
+ UNVERIFIED = 1
42
+ VERIFIED_FALSE = 2
43
+ VERIFIED_TRUE = 3
@@ -0,0 +1,51 @@
1
+ import logging
2
+ import sys
3
+
4
+
5
+ def get_logger(name=None, format_string=None):
6
+ """
7
+ :type name: str
8
+ :param name: used for declaring log channels.
9
+
10
+ :type format_string: str
11
+ :param format_string: for custom formatting
12
+ """
13
+ logging.captureWarnings(True)
14
+ log = logging.getLogger(name)
15
+
16
+ # Bind custom method to instance.
17
+ # Source: https://stackoverflow.com/a/2982
18
+ log.set_debug_level = _set_debug_level.__get__(log)
19
+ log.set_debug_level(0)
20
+
21
+ if not format_string:
22
+ format_string = '[%(module)s]\t%(levelname)s\t%(message)s'
23
+
24
+ # Setting up log formats
25
+ log.handlers = []
26
+ handler = logging.StreamHandler(sys.stderr)
27
+ handler.setFormatter(
28
+ logging.Formatter(format_string),
29
+ )
30
+ log.addHandler(handler)
31
+
32
+ return log
33
+
34
+
35
+ def _set_debug_level(self, debug_level):
36
+ """
37
+ :type debug_level: int, between 0-2
38
+ :param debug_level: configure verbosity of log
39
+ """
40
+ mapping = {
41
+ 0: logging.ERROR,
42
+ 1: logging.INFO,
43
+ 2: logging.DEBUG,
44
+ }
45
+
46
+ self.setLevel(
47
+ mapping[min(debug_level, 2)],
48
+ )
49
+
50
+
51
+ log = get_logger()
@@ -0,0 +1,92 @@
1
+ class PotentialSecret:
2
+ """This custom data type represents a string found, matching the
3
+ plugin rules defined in SecretsCollection, that has the potential
4
+ to be a secret that we actually care about.
5
+
6
+ "Potential" is the operative word here, because of the nature of
7
+ false positives.
8
+
9
+ We use this custom class so that we can more easily generate data
10
+ structures and do object-based comparisons with other PotentialSecrets,
11
+ without actually knowing what the secret is.
12
+ """
13
+
14
+ def __init__(
15
+ self,
16
+ typ,
17
+ filename,
18
+ secret,
19
+ lineno=0,
20
+ is_secret=None,
21
+ ):
22
+ """
23
+ :type typ: str
24
+ :param typ: human-readable secret type, defined by the plugin
25
+ that generated this PotentialSecret.
26
+ e.g. "High Entropy String"
27
+
28
+ :type filename: str
29
+ :param filename: name of file that this secret was found
30
+
31
+ :type secret: str
32
+ :param secret: the actual secret identified
33
+
34
+ :type lineno: int
35
+ :param lineno: location of secret, within filename.
36
+ Merely used as a reference for easy triage.
37
+
38
+ :type is_secret: bool|None
39
+ :param is_secret: whether or not the secret is a true- or false- positive
40
+
41
+ :type is_verified: bool
42
+ :param is_verified: whether the secret has been externally verified
43
+ """
44
+ self.type = typ
45
+ self.filename = filename
46
+ self.lineno = lineno
47
+ self.set_secret(secret)
48
+ self.is_secret = is_secret
49
+ self.is_verified = False
50
+
51
+ # If two PotentialSecrets have the same values for these fields,
52
+ # they are considered equal. Note that line numbers aren't included
53
+ # in this, because line numbers are subject to change.
54
+ self.fields_to_compare = ['filename', 'secret_value', 'type']
55
+
56
+ def set_secret(self, secret):
57
+ self.secret_value = secret
58
+
59
+ def json(self):
60
+ """Custom JSON encoder"""
61
+ attributes = {
62
+ 'type': self.type,
63
+ 'filename': self.filename,
64
+ 'line_number': self.lineno,
65
+ 'secret_value': self.secret_value,
66
+ 'is_verified': self.is_verified,
67
+ }
68
+
69
+ if self.is_secret is not None:
70
+ attributes['is_secret'] = self.is_secret
71
+
72
+ return attributes
73
+
74
+ def __eq__(self, other):
75
+ return all(
76
+ getattr(self, field) == getattr(other, field) for field in self.fields_to_compare
77
+ )
78
+
79
+ def __ne__(self, other):
80
+ return not self.__eq__(other)
81
+
82
+ def __hash__(self):
83
+ return hash(
84
+ tuple(getattr(self, x) for x in self.fields_to_compare),
85
+ )
86
+
87
+ def __str__(self): # pragma: no cover
88
+ return ('Secret Type: %s\nLocation: %s:%d\n') % (
89
+ self.type,
90
+ self.filename,
91
+ self.lineno,
92
+ )