commit-shield 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- commit_guard/__init__.py +5 -0
- commit_guard/_toml.py +276 -0
- commit_guard/checkers.py +277 -0
- commit_guard/cli.py +401 -0
- commit_guard/config.py +340 -0
- commit_guard/gitutil.py +106 -0
- commit_shield-0.2.0.dist-info/METADATA +179 -0
- commit_shield-0.2.0.dist-info/RECORD +11 -0
- commit_shield-0.2.0.dist-info/WHEEL +4 -0
- commit_shield-0.2.0.dist-info/entry_points.txt +3 -0
- commit_shield-0.2.0.dist-info/licenses/LICENSE +21 -0
commit_guard/__init__.py
ADDED
commit_guard/_toml.py
ADDED
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Minimal TOML reader for commit-guard configuration files.
|
|
3
|
+
|
|
4
|
+
Python 3.11+ ships ``tomllib``; on older interpreters we fall back to the
|
|
5
|
+
parser in this module. It deliberately supports only the subset of TOML that
|
|
6
|
+
commit-guard configuration actually uses:
|
|
7
|
+
|
|
8
|
+
* comments
|
|
9
|
+
* table headers, including dotted ones (``[tool.commit-guard]``)
|
|
10
|
+
* string, integer, float and boolean values
|
|
11
|
+
* arrays of strings/numbers, written on one line or across several
|
|
12
|
+
|
|
13
|
+
Anything outside that subset raises :class:`TomlError` rather than being
|
|
14
|
+
silently misread. This is not a general-purpose TOML implementation and is not
|
|
15
|
+
intended to become one.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
import re
|
|
19
|
+
import sys
|
|
20
|
+
from typing import Any, Dict, List, Optional, Tuple
|
|
21
|
+
|
|
22
|
+
__all__ = ["loads", "TomlError"]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class TomlError(ValueError):
|
|
26
|
+
"""Raised when configuration text cannot be parsed."""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# Typed as Optional[Any] so both branches type-check: tomllib does not
|
|
30
|
+
# exist below Python 3.11, where this falls back to the parser below.
|
|
31
|
+
_tomllib: Optional[Any]
|
|
32
|
+
if sys.version_info >= (3, 11):
|
|
33
|
+
import tomllib as _tomllib_mod
|
|
34
|
+
|
|
35
|
+
_tomllib = _tomllib_mod
|
|
36
|
+
else:
|
|
37
|
+
_tomllib = None
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
_TABLE_RE = re.compile(r"^\[([^\[\]]+)\]$")
|
|
41
|
+
_KEY_RE = re.compile(r"^([A-Za-z0-9_\-.]+)\s*=\s*(.*)$")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def loads(text: str) -> Dict[str, Any]:
|
|
45
|
+
"""Parse TOML text into nested dictionaries."""
|
|
46
|
+
if _tomllib is not None:
|
|
47
|
+
try:
|
|
48
|
+
return _tomllib.loads(text)
|
|
49
|
+
except Exception as exc: # tomllib raises TOMLDecodeError
|
|
50
|
+
raise TomlError(str(exc)) from exc
|
|
51
|
+
return _fallback_loads(text)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _fallback_loads(text: str) -> Dict[str, Any]:
|
|
55
|
+
root: Dict[str, Any] = {}
|
|
56
|
+
table = root
|
|
57
|
+
lines = text.splitlines()
|
|
58
|
+
index = 0
|
|
59
|
+
|
|
60
|
+
while index < len(lines):
|
|
61
|
+
raw = lines[index]
|
|
62
|
+
index += 1
|
|
63
|
+
line = _strip_comment(raw).strip()
|
|
64
|
+
if not line:
|
|
65
|
+
continue
|
|
66
|
+
|
|
67
|
+
table_match = _TABLE_RE.match(line)
|
|
68
|
+
if table_match:
|
|
69
|
+
table = _resolve_table(root, table_match.group(1).strip())
|
|
70
|
+
continue
|
|
71
|
+
|
|
72
|
+
key_match = _KEY_RE.match(line)
|
|
73
|
+
if not key_match:
|
|
74
|
+
raise TomlError("cannot parse line: {0!r}".format(raw.strip()))
|
|
75
|
+
|
|
76
|
+
key, value_text = key_match.group(1), key_match.group(2).strip()
|
|
77
|
+
|
|
78
|
+
# An array may continue over later lines until brackets balance.
|
|
79
|
+
if value_text.startswith("[") and not _brackets_balanced(value_text):
|
|
80
|
+
parts = [value_text]
|
|
81
|
+
while index < len(lines):
|
|
82
|
+
parts.append(_strip_comment(lines[index]).strip())
|
|
83
|
+
index += 1
|
|
84
|
+
if _brackets_balanced(" ".join(parts)):
|
|
85
|
+
break
|
|
86
|
+
else:
|
|
87
|
+
raise TomlError("unterminated array for key {0!r}".format(key))
|
|
88
|
+
value_text = " ".join(parts)
|
|
89
|
+
|
|
90
|
+
_assign(table, key, _parse_value(value_text))
|
|
91
|
+
|
|
92
|
+
return root
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _strip_comment(line: str) -> str:
|
|
96
|
+
"""Remove a trailing comment, ignoring '#' inside quoted strings."""
|
|
97
|
+
out: List[str] = []
|
|
98
|
+
quote = ""
|
|
99
|
+
for char in line:
|
|
100
|
+
if quote:
|
|
101
|
+
out.append(char)
|
|
102
|
+
if char == quote:
|
|
103
|
+
quote = ""
|
|
104
|
+
continue
|
|
105
|
+
if char == '"' or char == "'":
|
|
106
|
+
quote = char
|
|
107
|
+
out.append(char)
|
|
108
|
+
continue
|
|
109
|
+
if char == "#":
|
|
110
|
+
break
|
|
111
|
+
out.append(char)
|
|
112
|
+
return "".join(out)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _brackets_balanced(text: str) -> bool:
|
|
116
|
+
depth = 0
|
|
117
|
+
quote = ""
|
|
118
|
+
for char in text:
|
|
119
|
+
if quote:
|
|
120
|
+
if char == quote:
|
|
121
|
+
quote = ""
|
|
122
|
+
continue
|
|
123
|
+
if char == '"' or char == "'":
|
|
124
|
+
quote = char
|
|
125
|
+
elif char == "[":
|
|
126
|
+
depth += 1
|
|
127
|
+
elif char == "]":
|
|
128
|
+
depth -= 1
|
|
129
|
+
return depth == 0
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _resolve_table(root: Dict[str, Any], header: str) -> Dict[str, Any]:
|
|
133
|
+
node = root
|
|
134
|
+
for part in _split_dotted(header):
|
|
135
|
+
existing = node.get(part)
|
|
136
|
+
if existing is None:
|
|
137
|
+
existing = {}
|
|
138
|
+
node[part] = existing
|
|
139
|
+
elif not isinstance(existing, dict):
|
|
140
|
+
raise TomlError("cannot redefine {0!r} as a table".format(part))
|
|
141
|
+
node = existing
|
|
142
|
+
return node
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _assign(table: Dict[str, Any], key: str, value: Any) -> None:
|
|
146
|
+
parts = _split_dotted(key)
|
|
147
|
+
node = table
|
|
148
|
+
for part in parts[:-1]:
|
|
149
|
+
existing = node.get(part)
|
|
150
|
+
if existing is None:
|
|
151
|
+
existing = {}
|
|
152
|
+
node[part] = existing
|
|
153
|
+
elif not isinstance(existing, dict):
|
|
154
|
+
raise TomlError("cannot assign into non-table {0!r}".format(part))
|
|
155
|
+
node = existing
|
|
156
|
+
node[parts[-1]] = value
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _split_dotted(name: str) -> List[str]:
|
|
160
|
+
"""Split a dotted key or header, honouring quoted segments."""
|
|
161
|
+
parts: List[str] = []
|
|
162
|
+
current: List[str] = []
|
|
163
|
+
quote = ""
|
|
164
|
+
for char in name:
|
|
165
|
+
if quote:
|
|
166
|
+
if char == quote:
|
|
167
|
+
quote = ""
|
|
168
|
+
else:
|
|
169
|
+
current.append(char)
|
|
170
|
+
continue
|
|
171
|
+
if char == '"' or char == "'":
|
|
172
|
+
quote = char
|
|
173
|
+
continue
|
|
174
|
+
if char == ".":
|
|
175
|
+
parts.append("".join(current).strip())
|
|
176
|
+
current = []
|
|
177
|
+
continue
|
|
178
|
+
current.append(char)
|
|
179
|
+
parts.append("".join(current).strip())
|
|
180
|
+
|
|
181
|
+
cleaned = [part for part in parts if part]
|
|
182
|
+
if not cleaned:
|
|
183
|
+
raise TomlError("empty key or table name: {0!r}".format(name))
|
|
184
|
+
return cleaned
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _parse_value(text: str) -> Any:
|
|
188
|
+
if not text:
|
|
189
|
+
raise TomlError("missing value")
|
|
190
|
+
|
|
191
|
+
if text.startswith("["):
|
|
192
|
+
return _parse_array(text)
|
|
193
|
+
if text[0] == '"' or text[0] == "'":
|
|
194
|
+
value, rest = _parse_string(text)
|
|
195
|
+
if rest.strip():
|
|
196
|
+
raise TomlError("trailing text after string: {0!r}".format(rest))
|
|
197
|
+
return value
|
|
198
|
+
if text == "true" or text == "false":
|
|
199
|
+
return text == "true"
|
|
200
|
+
|
|
201
|
+
try:
|
|
202
|
+
return int(text.replace("_", ""))
|
|
203
|
+
except ValueError:
|
|
204
|
+
pass
|
|
205
|
+
try:
|
|
206
|
+
return float(text.replace("_", ""))
|
|
207
|
+
except ValueError:
|
|
208
|
+
pass
|
|
209
|
+
raise TomlError("unsupported value: {0!r}".format(text))
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _parse_array(text: str) -> List[Any]:
|
|
213
|
+
if not text.startswith("[") or not text.endswith("]"):
|
|
214
|
+
raise TomlError("malformed array: {0!r}".format(text))
|
|
215
|
+
body = text[1:-1].strip()
|
|
216
|
+
if not body:
|
|
217
|
+
return []
|
|
218
|
+
|
|
219
|
+
items: List[Any] = []
|
|
220
|
+
rest = body
|
|
221
|
+
while rest:
|
|
222
|
+
rest = rest.lstrip()
|
|
223
|
+
if not rest:
|
|
224
|
+
break
|
|
225
|
+
if rest[0] == '"' or rest[0] == "'":
|
|
226
|
+
value, rest = _parse_string(rest)
|
|
227
|
+
items.append(value)
|
|
228
|
+
else:
|
|
229
|
+
chunk, _, rest = rest.partition(",")
|
|
230
|
+
chunk = chunk.strip()
|
|
231
|
+
if chunk:
|
|
232
|
+
items.append(_parse_value(chunk))
|
|
233
|
+
continue
|
|
234
|
+
rest = rest.lstrip()
|
|
235
|
+
if rest.startswith(","):
|
|
236
|
+
rest = rest[1:]
|
|
237
|
+
elif rest:
|
|
238
|
+
raise TomlError("expected a comma in array near {0!r}".format(rest))
|
|
239
|
+
return items
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def _parse_string(text: str) -> Tuple[str, str]:
|
|
243
|
+
"""Read one string literal, returning it plus the unconsumed remainder."""
|
|
244
|
+
quote = text[0]
|
|
245
|
+
literal = quote == "'"
|
|
246
|
+
out: List[str] = []
|
|
247
|
+
index = 1
|
|
248
|
+
while index < len(text):
|
|
249
|
+
char = text[index]
|
|
250
|
+
if not literal and char == "\\":
|
|
251
|
+
index += 1
|
|
252
|
+
if index >= len(text):
|
|
253
|
+
raise TomlError("dangling escape in string")
|
|
254
|
+
out.append(_unescape(text[index]))
|
|
255
|
+
elif char == quote:
|
|
256
|
+
return "".join(out), text[index + 1 :]
|
|
257
|
+
else:
|
|
258
|
+
out.append(char)
|
|
259
|
+
index += 1
|
|
260
|
+
raise TomlError("unterminated string: {0!r}".format(text))
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
_ESCAPES = {
|
|
264
|
+
"n": "\n",
|
|
265
|
+
"t": "\t",
|
|
266
|
+
"r": "\r",
|
|
267
|
+
'"': '"',
|
|
268
|
+
"\\": "\\",
|
|
269
|
+
"'": "'",
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _unescape(char: str) -> str:
|
|
274
|
+
if char not in _ESCAPES:
|
|
275
|
+
raise TomlError("unsupported escape sequence")
|
|
276
|
+
return _ESCAPES[char]
|
commit_guard/checkers.py
ADDED
|
@@ -0,0 +1,277 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Core validation logic for commit messages and staged files.
|
|
3
|
+
Zero external dependencies (uses only standard library).
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import fnmatch
|
|
7
|
+
import os
|
|
8
|
+
import re
|
|
9
|
+
from typing import List, Optional, Sequence, Tuple
|
|
10
|
+
|
|
11
|
+
from commit_guard.config import DEFAULT_TYPES, Config
|
|
12
|
+
|
|
13
|
+
# Kept for backward compatibility with code that imported these in v0.1.x.
|
|
14
|
+
# The authoritative values now live in commit_guard.config.
|
|
15
|
+
CONVENTIONAL_TYPES = set(DEFAULT_TYPES)
|
|
16
|
+
|
|
17
|
+
# Regex pattern for Conventional Commits
|
|
18
|
+
CONVENTIONAL_REGEX = re.compile(
|
|
19
|
+
r"^(?P<type>[a-zA-Z]+)" # type
|
|
20
|
+
r"(?:\((?P<scope>[a-zA-Z0-9_\-./,\s]+)\))?" # optional scope
|
|
21
|
+
r"(?P<breaking>!)?" # optional breaking marker
|
|
22
|
+
r":[ ]+" # separator
|
|
23
|
+
r"(?P<subject>.+)$" # commit subject
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
_DEFAULT_CONFIG = Config()
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _config_or_default(config: Optional[Config]) -> Config:
|
|
30
|
+
return config if config is not None else _DEFAULT_CONFIG
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def is_skippable_message(msg: str, config: Optional[Config] = None) -> bool:
|
|
34
|
+
"""
|
|
35
|
+
Report whether a commit message is one git generates rather than one the
|
|
36
|
+
author wrote: merges, reverts, and rebase fixup/squash markers.
|
|
37
|
+
|
|
38
|
+
Validating these blocks ordinary git workflows, since their format is not
|
|
39
|
+
the author's to control.
|
|
40
|
+
"""
|
|
41
|
+
cfg = _config_or_default(config)
|
|
42
|
+
if not cfg.skip_merge_commits:
|
|
43
|
+
return False
|
|
44
|
+
|
|
45
|
+
header = _first_meaningful_line(msg)
|
|
46
|
+
if header is None:
|
|
47
|
+
return False
|
|
48
|
+
|
|
49
|
+
lowered = header.lower()
|
|
50
|
+
return any(lowered.startswith(prefix) for prefix in cfg.skip_prefixes)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _first_meaningful_line(msg: str) -> Optional[str]:
|
|
54
|
+
"""Return the first non-blank, non-comment line, or None."""
|
|
55
|
+
for line in msg.splitlines():
|
|
56
|
+
stripped = line.strip()
|
|
57
|
+
if stripped and not stripped.startswith("#"):
|
|
58
|
+
return stripped
|
|
59
|
+
return None
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def check_commit_message(
|
|
63
|
+
msg: str,
|
|
64
|
+
max_header_len: Optional[int] = None,
|
|
65
|
+
config: Optional[Config] = None,
|
|
66
|
+
) -> Tuple[bool, List[str]]:
|
|
67
|
+
"""
|
|
68
|
+
Validate commit message against Conventional Commits standard.
|
|
69
|
+
Returns (is_valid, list_of_error_messages).
|
|
70
|
+
|
|
71
|
+
``max_header_len`` overrides the configured limit; it is kept as a named
|
|
72
|
+
parameter because v0.1.x callers passed it positionally.
|
|
73
|
+
"""
|
|
74
|
+
cfg = _config_or_default(config)
|
|
75
|
+
header_limit = max_header_len if max_header_len is not None else cfg.max_header_len
|
|
76
|
+
|
|
77
|
+
# Generated merge/revert/fixup messages are not the author's to format.
|
|
78
|
+
if is_skippable_message(msg, cfg):
|
|
79
|
+
return True, []
|
|
80
|
+
|
|
81
|
+
errors: List[str] = []
|
|
82
|
+
|
|
83
|
+
# Strip comments (lines starting with #) and trailing whitespace
|
|
84
|
+
lines = [
|
|
85
|
+
line.strip()
|
|
86
|
+
for line in msg.splitlines()
|
|
87
|
+
if line.strip() and not line.strip().startswith("#")
|
|
88
|
+
]
|
|
89
|
+
|
|
90
|
+
if not lines:
|
|
91
|
+
return False, ["Commit message cannot be empty."]
|
|
92
|
+
|
|
93
|
+
header = lines[0]
|
|
94
|
+
|
|
95
|
+
if header_limit and len(header) > header_limit:
|
|
96
|
+
errors.append(
|
|
97
|
+
"Commit header too long ({0} chars > {1} limit).".format(
|
|
98
|
+
len(header), header_limit
|
|
99
|
+
)
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
match = CONVENTIONAL_REGEX.match(header)
|
|
103
|
+
if not match:
|
|
104
|
+
errors.append(
|
|
105
|
+
"Header does not follow Conventional Commits format: "
|
|
106
|
+
"'<type>(<scope>): <description>'.\n"
|
|
107
|
+
" Received: '{0}'\n"
|
|
108
|
+
" Allowed types: {1}".format(header, ", ".join(sorted(cfg.types)))
|
|
109
|
+
)
|
|
110
|
+
else:
|
|
111
|
+
errors.extend(_check_header_parts(match, cfg))
|
|
112
|
+
|
|
113
|
+
errors.extend(_check_body(msg, cfg))
|
|
114
|
+
return len(errors) == 0, errors
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _check_header_parts(match: "re.Match", cfg: Config) -> List[str]:
|
|
118
|
+
"""Validate the type, scope and subject of a parsed header."""
|
|
119
|
+
errors: List[str] = []
|
|
120
|
+
|
|
121
|
+
ctype = match.group("type")
|
|
122
|
+
if ctype not in cfg.types:
|
|
123
|
+
# Catch the common near-miss of an uppercase or mixed-case type.
|
|
124
|
+
if ctype.lower() in cfg.types:
|
|
125
|
+
errors.append(
|
|
126
|
+
"Commit type must be lowercase: '{0}' should be '{1}'.".format(
|
|
127
|
+
ctype, ctype.lower()
|
|
128
|
+
)
|
|
129
|
+
)
|
|
130
|
+
else:
|
|
131
|
+
errors.append(
|
|
132
|
+
"Unknown commit type: '{0}'. Allowed: {1}".format(
|
|
133
|
+
ctype, ", ".join(sorted(cfg.types))
|
|
134
|
+
)
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
scope = match.group("scope")
|
|
138
|
+
if cfg.require_scope and not scope:
|
|
139
|
+
errors.append(
|
|
140
|
+
"A scope is required. Allowed scopes: {0}".format(
|
|
141
|
+
", ".join(sorted(cfg.scopes))
|
|
142
|
+
)
|
|
143
|
+
)
|
|
144
|
+
elif scope and cfg.scopes:
|
|
145
|
+
# A comma-separated scope list is valid Conventional Commits.
|
|
146
|
+
for part in (p.strip() for p in scope.split(",")):
|
|
147
|
+
if part and part not in cfg.scopes:
|
|
148
|
+
errors.append(
|
|
149
|
+
"Unknown scope: '{0}'. Allowed: {1}".format(
|
|
150
|
+
part, ", ".join(sorted(cfg.scopes))
|
|
151
|
+
)
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
subject = match.group("subject").strip()
|
|
155
|
+
if len(subject) < cfg.subject_min_len:
|
|
156
|
+
errors.append(
|
|
157
|
+
"Commit description is too short (minimum {0} characters).".format(
|
|
158
|
+
cfg.subject_min_len
|
|
159
|
+
)
|
|
160
|
+
)
|
|
161
|
+
if subject.endswith(".") and not cfg.allow_trailing_period:
|
|
162
|
+
errors.append("Commit description should not end with a period ('.').")
|
|
163
|
+
|
|
164
|
+
return errors
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _check_body(msg: str, cfg: Config) -> List[str]:
|
|
168
|
+
"""Validate the blank-line separator and body line lengths."""
|
|
169
|
+
errors: List[str] = []
|
|
170
|
+
|
|
171
|
+
raw_lines = [line for line in msg.splitlines() if not line.strip().startswith("#")]
|
|
172
|
+
|
|
173
|
+
# Drop leading blank lines so the header is at index 0.
|
|
174
|
+
while raw_lines and not raw_lines[0].strip():
|
|
175
|
+
raw_lines.pop(0)
|
|
176
|
+
|
|
177
|
+
if len(raw_lines) > 1 and raw_lines[1].strip() != "":
|
|
178
|
+
errors.append(
|
|
179
|
+
"There must be an empty blank line between commit header and body."
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
if cfg.max_body_line_len:
|
|
183
|
+
for number, line in enumerate(raw_lines[2:], start=3):
|
|
184
|
+
stripped = line.rstrip()
|
|
185
|
+
if len(stripped) <= cfg.max_body_line_len:
|
|
186
|
+
continue
|
|
187
|
+
# Long URLs and footer trailers cannot be wrapped usefully.
|
|
188
|
+
if _is_unwrappable(stripped):
|
|
189
|
+
continue
|
|
190
|
+
errors.append(
|
|
191
|
+
"Body line {0} too long ({1} chars > {2} limit).".format(
|
|
192
|
+
number, len(stripped), cfg.max_body_line_len
|
|
193
|
+
)
|
|
194
|
+
)
|
|
195
|
+
|
|
196
|
+
return errors
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
_TRAILER_RE = re.compile(r"^[A-Za-z][A-Za-z\-]*:\s")
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _is_unwrappable(line: str) -> bool:
|
|
203
|
+
"""Lines that cannot reasonably be wrapped: URLs, trailers, code."""
|
|
204
|
+
if "://" in line:
|
|
205
|
+
return True
|
|
206
|
+
if _TRAILER_RE.match(line):
|
|
207
|
+
return True
|
|
208
|
+
return line.startswith((" ", "\t", "|", ">"))
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def is_allowlisted(path: str, allowlist: Sequence[str]) -> bool:
|
|
212
|
+
"""
|
|
213
|
+
Report whether a path is exempt from sensitive-name matching.
|
|
214
|
+
|
|
215
|
+
Each entry is a glob. It is matched against the full path and against the
|
|
216
|
+
basename, so 'docs/keys/*' and '*.example' both behave as expected.
|
|
217
|
+
"""
|
|
218
|
+
norm = path.replace("\\", "/")
|
|
219
|
+
base = os.path.basename(norm)
|
|
220
|
+
for pattern in allowlist:
|
|
221
|
+
clean = pattern.replace("\\", "/")
|
|
222
|
+
if fnmatch.fnmatch(norm, clean) or fnmatch.fnmatch(base, clean):
|
|
223
|
+
return True
|
|
224
|
+
# A bare directory entry exempts everything beneath it.
|
|
225
|
+
if clean.endswith("/") and norm.startswith(clean):
|
|
226
|
+
return True
|
|
227
|
+
return False
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def check_file_path(
|
|
231
|
+
path: str,
|
|
232
|
+
max_size_mb: Optional[float] = None,
|
|
233
|
+
config: Optional[Config] = None,
|
|
234
|
+
size_bytes: Optional[int] = None,
|
|
235
|
+
) -> Tuple[bool, List[str]]:
|
|
236
|
+
"""
|
|
237
|
+
Validate an individual file path for size and sensitive filename patterns.
|
|
238
|
+
Returns (is_valid, list_of_warning_messages).
|
|
239
|
+
|
|
240
|
+
``size_bytes`` supplies the staged blob size; when omitted the working-tree
|
|
241
|
+
file is measured instead.
|
|
242
|
+
"""
|
|
243
|
+
cfg = _config_or_default(config)
|
|
244
|
+
size_limit = max_size_mb if max_size_mb is not None else cfg.max_size_mb
|
|
245
|
+
|
|
246
|
+
issues: List[str] = []
|
|
247
|
+
norm_path = path.replace("\\", "/")
|
|
248
|
+
|
|
249
|
+
if not is_allowlisted(norm_path, cfg.allowlist):
|
|
250
|
+
for pattern in cfg.sensitive_patterns:
|
|
251
|
+
try:
|
|
252
|
+
compiled = re.compile(pattern, re.IGNORECASE)
|
|
253
|
+
except re.error as exc:
|
|
254
|
+
issues.append(
|
|
255
|
+
"Invalid sensitive pattern {0!r}: {1}".format(pattern, exc)
|
|
256
|
+
)
|
|
257
|
+
continue
|
|
258
|
+
if compiled.search(norm_path):
|
|
259
|
+
issues.append(
|
|
260
|
+
"Potentially sensitive file detected: '{0}'. "
|
|
261
|
+
"Avoid committing secrets.".format(path)
|
|
262
|
+
)
|
|
263
|
+
break
|
|
264
|
+
|
|
265
|
+
measured = size_bytes
|
|
266
|
+
if measured is None and os.path.isfile(path):
|
|
267
|
+
measured = os.path.getsize(path)
|
|
268
|
+
|
|
269
|
+
if measured is not None and size_limit:
|
|
270
|
+
size_mb = measured / (1024 * 1024)
|
|
271
|
+
if size_mb > size_limit:
|
|
272
|
+
issues.append(
|
|
273
|
+
"File '{0}' exceeds max allowed size ({1:.2f} MB > {2} MB limit). "
|
|
274
|
+
"Consider using Git LFS.".format(path, size_mb, size_limit)
|
|
275
|
+
)
|
|
276
|
+
|
|
277
|
+
return len(issues) == 0, issues
|