commit-shield 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,5 @@
1
+ """
2
+ commit-guard: Lightweight, zero-dependency Git commit message and file linter.
3
+ """
4
+
5
+ __version__ = "0.2.0"
commit_guard/_toml.py ADDED
@@ -0,0 +1,276 @@
1
+ """
2
+ Minimal TOML reader for commit-guard configuration files.
3
+
4
+ Python 3.11+ ships ``tomllib``; on older interpreters we fall back to the
5
+ parser in this module. It deliberately supports only the subset of TOML that
6
+ commit-guard configuration actually uses:
7
+
8
+ * comments
9
+ * table headers, including dotted ones (``[tool.commit-guard]``)
10
+ * string, integer, float and boolean values
11
+ * arrays of strings/numbers, written on one line or across several
12
+
13
+ Anything outside that subset raises :class:`TomlError` rather than being
14
+ silently misread. This is not a general-purpose TOML implementation and is not
15
+ intended to become one.
16
+ """
17
+
18
+ import re
19
+ import sys
20
+ from typing import Any, Dict, List, Optional, Tuple
21
+
22
+ __all__ = ["loads", "TomlError"]
23
+
24
+
25
+ class TomlError(ValueError):
26
+ """Raised when configuration text cannot be parsed."""
27
+
28
+
29
+ # Typed as Optional[Any] so both branches type-check: tomllib does not
30
+ # exist below Python 3.11, where this falls back to the parser below.
31
+ _tomllib: Optional[Any]
32
+ if sys.version_info >= (3, 11):
33
+ import tomllib as _tomllib_mod
34
+
35
+ _tomllib = _tomllib_mod
36
+ else:
37
+ _tomllib = None
38
+
39
+
40
+ _TABLE_RE = re.compile(r"^\[([^\[\]]+)\]$")
41
+ _KEY_RE = re.compile(r"^([A-Za-z0-9_\-.]+)\s*=\s*(.*)$")
42
+
43
+
44
+ def loads(text: str) -> Dict[str, Any]:
45
+ """Parse TOML text into nested dictionaries."""
46
+ if _tomllib is not None:
47
+ try:
48
+ return _tomllib.loads(text)
49
+ except Exception as exc: # tomllib raises TOMLDecodeError
50
+ raise TomlError(str(exc)) from exc
51
+ return _fallback_loads(text)
52
+
53
+
54
+ def _fallback_loads(text: str) -> Dict[str, Any]:
55
+ root: Dict[str, Any] = {}
56
+ table = root
57
+ lines = text.splitlines()
58
+ index = 0
59
+
60
+ while index < len(lines):
61
+ raw = lines[index]
62
+ index += 1
63
+ line = _strip_comment(raw).strip()
64
+ if not line:
65
+ continue
66
+
67
+ table_match = _TABLE_RE.match(line)
68
+ if table_match:
69
+ table = _resolve_table(root, table_match.group(1).strip())
70
+ continue
71
+
72
+ key_match = _KEY_RE.match(line)
73
+ if not key_match:
74
+ raise TomlError("cannot parse line: {0!r}".format(raw.strip()))
75
+
76
+ key, value_text = key_match.group(1), key_match.group(2).strip()
77
+
78
+ # An array may continue over later lines until brackets balance.
79
+ if value_text.startswith("[") and not _brackets_balanced(value_text):
80
+ parts = [value_text]
81
+ while index < len(lines):
82
+ parts.append(_strip_comment(lines[index]).strip())
83
+ index += 1
84
+ if _brackets_balanced(" ".join(parts)):
85
+ break
86
+ else:
87
+ raise TomlError("unterminated array for key {0!r}".format(key))
88
+ value_text = " ".join(parts)
89
+
90
+ _assign(table, key, _parse_value(value_text))
91
+
92
+ return root
93
+
94
+
95
+ def _strip_comment(line: str) -> str:
96
+ """Remove a trailing comment, ignoring '#' inside quoted strings."""
97
+ out: List[str] = []
98
+ quote = ""
99
+ for char in line:
100
+ if quote:
101
+ out.append(char)
102
+ if char == quote:
103
+ quote = ""
104
+ continue
105
+ if char == '"' or char == "'":
106
+ quote = char
107
+ out.append(char)
108
+ continue
109
+ if char == "#":
110
+ break
111
+ out.append(char)
112
+ return "".join(out)
113
+
114
+
115
+ def _brackets_balanced(text: str) -> bool:
116
+ depth = 0
117
+ quote = ""
118
+ for char in text:
119
+ if quote:
120
+ if char == quote:
121
+ quote = ""
122
+ continue
123
+ if char == '"' or char == "'":
124
+ quote = char
125
+ elif char == "[":
126
+ depth += 1
127
+ elif char == "]":
128
+ depth -= 1
129
+ return depth == 0
130
+
131
+
132
+ def _resolve_table(root: Dict[str, Any], header: str) -> Dict[str, Any]:
133
+ node = root
134
+ for part in _split_dotted(header):
135
+ existing = node.get(part)
136
+ if existing is None:
137
+ existing = {}
138
+ node[part] = existing
139
+ elif not isinstance(existing, dict):
140
+ raise TomlError("cannot redefine {0!r} as a table".format(part))
141
+ node = existing
142
+ return node
143
+
144
+
145
+ def _assign(table: Dict[str, Any], key: str, value: Any) -> None:
146
+ parts = _split_dotted(key)
147
+ node = table
148
+ for part in parts[:-1]:
149
+ existing = node.get(part)
150
+ if existing is None:
151
+ existing = {}
152
+ node[part] = existing
153
+ elif not isinstance(existing, dict):
154
+ raise TomlError("cannot assign into non-table {0!r}".format(part))
155
+ node = existing
156
+ node[parts[-1]] = value
157
+
158
+
159
+ def _split_dotted(name: str) -> List[str]:
160
+ """Split a dotted key or header, honouring quoted segments."""
161
+ parts: List[str] = []
162
+ current: List[str] = []
163
+ quote = ""
164
+ for char in name:
165
+ if quote:
166
+ if char == quote:
167
+ quote = ""
168
+ else:
169
+ current.append(char)
170
+ continue
171
+ if char == '"' or char == "'":
172
+ quote = char
173
+ continue
174
+ if char == ".":
175
+ parts.append("".join(current).strip())
176
+ current = []
177
+ continue
178
+ current.append(char)
179
+ parts.append("".join(current).strip())
180
+
181
+ cleaned = [part for part in parts if part]
182
+ if not cleaned:
183
+ raise TomlError("empty key or table name: {0!r}".format(name))
184
+ return cleaned
185
+
186
+
187
+ def _parse_value(text: str) -> Any:
188
+ if not text:
189
+ raise TomlError("missing value")
190
+
191
+ if text.startswith("["):
192
+ return _parse_array(text)
193
+ if text[0] == '"' or text[0] == "'":
194
+ value, rest = _parse_string(text)
195
+ if rest.strip():
196
+ raise TomlError("trailing text after string: {0!r}".format(rest))
197
+ return value
198
+ if text == "true" or text == "false":
199
+ return text == "true"
200
+
201
+ try:
202
+ return int(text.replace("_", ""))
203
+ except ValueError:
204
+ pass
205
+ try:
206
+ return float(text.replace("_", ""))
207
+ except ValueError:
208
+ pass
209
+ raise TomlError("unsupported value: {0!r}".format(text))
210
+
211
+
212
+ def _parse_array(text: str) -> List[Any]:
213
+ if not text.startswith("[") or not text.endswith("]"):
214
+ raise TomlError("malformed array: {0!r}".format(text))
215
+ body = text[1:-1].strip()
216
+ if not body:
217
+ return []
218
+
219
+ items: List[Any] = []
220
+ rest = body
221
+ while rest:
222
+ rest = rest.lstrip()
223
+ if not rest:
224
+ break
225
+ if rest[0] == '"' or rest[0] == "'":
226
+ value, rest = _parse_string(rest)
227
+ items.append(value)
228
+ else:
229
+ chunk, _, rest = rest.partition(",")
230
+ chunk = chunk.strip()
231
+ if chunk:
232
+ items.append(_parse_value(chunk))
233
+ continue
234
+ rest = rest.lstrip()
235
+ if rest.startswith(","):
236
+ rest = rest[1:]
237
+ elif rest:
238
+ raise TomlError("expected a comma in array near {0!r}".format(rest))
239
+ return items
240
+
241
+
242
+ def _parse_string(text: str) -> Tuple[str, str]:
243
+ """Read one string literal, returning it plus the unconsumed remainder."""
244
+ quote = text[0]
245
+ literal = quote == "'"
246
+ out: List[str] = []
247
+ index = 1
248
+ while index < len(text):
249
+ char = text[index]
250
+ if not literal and char == "\\":
251
+ index += 1
252
+ if index >= len(text):
253
+ raise TomlError("dangling escape in string")
254
+ out.append(_unescape(text[index]))
255
+ elif char == quote:
256
+ return "".join(out), text[index + 1 :]
257
+ else:
258
+ out.append(char)
259
+ index += 1
260
+ raise TomlError("unterminated string: {0!r}".format(text))
261
+
262
+
263
+ _ESCAPES = {
264
+ "n": "\n",
265
+ "t": "\t",
266
+ "r": "\r",
267
+ '"': '"',
268
+ "\\": "\\",
269
+ "'": "'",
270
+ }
271
+
272
+
273
+ def _unescape(char: str) -> str:
274
+ if char not in _ESCAPES:
275
+ raise TomlError("unsupported escape sequence")
276
+ return _ESCAPES[char]
@@ -0,0 +1,277 @@
1
+ """
2
+ Core validation logic for commit messages and staged files.
3
+ Zero external dependencies (uses only standard library).
4
+ """
5
+
6
+ import fnmatch
7
+ import os
8
+ import re
9
+ from typing import List, Optional, Sequence, Tuple
10
+
11
+ from commit_guard.config import DEFAULT_TYPES, Config
12
+
13
+ # Kept for backward compatibility with code that imported these in v0.1.x.
14
+ # The authoritative values now live in commit_guard.config.
15
+ CONVENTIONAL_TYPES = set(DEFAULT_TYPES)
16
+
17
+ # Regex pattern for Conventional Commits
18
+ CONVENTIONAL_REGEX = re.compile(
19
+ r"^(?P<type>[a-zA-Z]+)" # type
20
+ r"(?:\((?P<scope>[a-zA-Z0-9_\-./,\s]+)\))?" # optional scope
21
+ r"(?P<breaking>!)?" # optional breaking marker
22
+ r":[ ]+" # separator
23
+ r"(?P<subject>.+)$" # commit subject
24
+ )
25
+
26
+ _DEFAULT_CONFIG = Config()
27
+
28
+
29
+ def _config_or_default(config: Optional[Config]) -> Config:
30
+ return config if config is not None else _DEFAULT_CONFIG
31
+
32
+
33
+ def is_skippable_message(msg: str, config: Optional[Config] = None) -> bool:
34
+ """
35
+ Report whether a commit message is one git generates rather than one the
36
+ author wrote: merges, reverts, and rebase fixup/squash markers.
37
+
38
+ Validating these blocks ordinary git workflows, since their format is not
39
+ the author's to control.
40
+ """
41
+ cfg = _config_or_default(config)
42
+ if not cfg.skip_merge_commits:
43
+ return False
44
+
45
+ header = _first_meaningful_line(msg)
46
+ if header is None:
47
+ return False
48
+
49
+ lowered = header.lower()
50
+ return any(lowered.startswith(prefix) for prefix in cfg.skip_prefixes)
51
+
52
+
53
+ def _first_meaningful_line(msg: str) -> Optional[str]:
54
+ """Return the first non-blank, non-comment line, or None."""
55
+ for line in msg.splitlines():
56
+ stripped = line.strip()
57
+ if stripped and not stripped.startswith("#"):
58
+ return stripped
59
+ return None
60
+
61
+
62
+ def check_commit_message(
63
+ msg: str,
64
+ max_header_len: Optional[int] = None,
65
+ config: Optional[Config] = None,
66
+ ) -> Tuple[bool, List[str]]:
67
+ """
68
+ Validate commit message against Conventional Commits standard.
69
+ Returns (is_valid, list_of_error_messages).
70
+
71
+ ``max_header_len`` overrides the configured limit; it is kept as a named
72
+ parameter because v0.1.x callers passed it positionally.
73
+ """
74
+ cfg = _config_or_default(config)
75
+ header_limit = max_header_len if max_header_len is not None else cfg.max_header_len
76
+
77
+ # Generated merge/revert/fixup messages are not the author's to format.
78
+ if is_skippable_message(msg, cfg):
79
+ return True, []
80
+
81
+ errors: List[str] = []
82
+
83
+ # Strip comments (lines starting with #) and trailing whitespace
84
+ lines = [
85
+ line.strip()
86
+ for line in msg.splitlines()
87
+ if line.strip() and not line.strip().startswith("#")
88
+ ]
89
+
90
+ if not lines:
91
+ return False, ["Commit message cannot be empty."]
92
+
93
+ header = lines[0]
94
+
95
+ if header_limit and len(header) > header_limit:
96
+ errors.append(
97
+ "Commit header too long ({0} chars > {1} limit).".format(
98
+ len(header), header_limit
99
+ )
100
+ )
101
+
102
+ match = CONVENTIONAL_REGEX.match(header)
103
+ if not match:
104
+ errors.append(
105
+ "Header does not follow Conventional Commits format: "
106
+ "'<type>(<scope>): <description>'.\n"
107
+ " Received: '{0}'\n"
108
+ " Allowed types: {1}".format(header, ", ".join(sorted(cfg.types)))
109
+ )
110
+ else:
111
+ errors.extend(_check_header_parts(match, cfg))
112
+
113
+ errors.extend(_check_body(msg, cfg))
114
+ return len(errors) == 0, errors
115
+
116
+
117
+ def _check_header_parts(match: "re.Match", cfg: Config) -> List[str]:
118
+ """Validate the type, scope and subject of a parsed header."""
119
+ errors: List[str] = []
120
+
121
+ ctype = match.group("type")
122
+ if ctype not in cfg.types:
123
+ # Catch the common near-miss of an uppercase or mixed-case type.
124
+ if ctype.lower() in cfg.types:
125
+ errors.append(
126
+ "Commit type must be lowercase: '{0}' should be '{1}'.".format(
127
+ ctype, ctype.lower()
128
+ )
129
+ )
130
+ else:
131
+ errors.append(
132
+ "Unknown commit type: '{0}'. Allowed: {1}".format(
133
+ ctype, ", ".join(sorted(cfg.types))
134
+ )
135
+ )
136
+
137
+ scope = match.group("scope")
138
+ if cfg.require_scope and not scope:
139
+ errors.append(
140
+ "A scope is required. Allowed scopes: {0}".format(
141
+ ", ".join(sorted(cfg.scopes))
142
+ )
143
+ )
144
+ elif scope and cfg.scopes:
145
+ # A comma-separated scope list is valid Conventional Commits.
146
+ for part in (p.strip() for p in scope.split(",")):
147
+ if part and part not in cfg.scopes:
148
+ errors.append(
149
+ "Unknown scope: '{0}'. Allowed: {1}".format(
150
+ part, ", ".join(sorted(cfg.scopes))
151
+ )
152
+ )
153
+
154
+ subject = match.group("subject").strip()
155
+ if len(subject) < cfg.subject_min_len:
156
+ errors.append(
157
+ "Commit description is too short (minimum {0} characters).".format(
158
+ cfg.subject_min_len
159
+ )
160
+ )
161
+ if subject.endswith(".") and not cfg.allow_trailing_period:
162
+ errors.append("Commit description should not end with a period ('.').")
163
+
164
+ return errors
165
+
166
+
167
+ def _check_body(msg: str, cfg: Config) -> List[str]:
168
+ """Validate the blank-line separator and body line lengths."""
169
+ errors: List[str] = []
170
+
171
+ raw_lines = [line for line in msg.splitlines() if not line.strip().startswith("#")]
172
+
173
+ # Drop leading blank lines so the header is at index 0.
174
+ while raw_lines and not raw_lines[0].strip():
175
+ raw_lines.pop(0)
176
+
177
+ if len(raw_lines) > 1 and raw_lines[1].strip() != "":
178
+ errors.append(
179
+ "There must be an empty blank line between commit header and body."
180
+ )
181
+
182
+ if cfg.max_body_line_len:
183
+ for number, line in enumerate(raw_lines[2:], start=3):
184
+ stripped = line.rstrip()
185
+ if len(stripped) <= cfg.max_body_line_len:
186
+ continue
187
+ # Long URLs and footer trailers cannot be wrapped usefully.
188
+ if _is_unwrappable(stripped):
189
+ continue
190
+ errors.append(
191
+ "Body line {0} too long ({1} chars > {2} limit).".format(
192
+ number, len(stripped), cfg.max_body_line_len
193
+ )
194
+ )
195
+
196
+ return errors
197
+
198
+
199
+ _TRAILER_RE = re.compile(r"^[A-Za-z][A-Za-z\-]*:\s")
200
+
201
+
202
+ def _is_unwrappable(line: str) -> bool:
203
+ """Lines that cannot reasonably be wrapped: URLs, trailers, code."""
204
+ if "://" in line:
205
+ return True
206
+ if _TRAILER_RE.match(line):
207
+ return True
208
+ return line.startswith((" ", "\t", "|", ">"))
209
+
210
+
211
+ def is_allowlisted(path: str, allowlist: Sequence[str]) -> bool:
212
+ """
213
+ Report whether a path is exempt from sensitive-name matching.
214
+
215
+ Each entry is a glob. It is matched against the full path and against the
216
+ basename, so 'docs/keys/*' and '*.example' both behave as expected.
217
+ """
218
+ norm = path.replace("\\", "/")
219
+ base = os.path.basename(norm)
220
+ for pattern in allowlist:
221
+ clean = pattern.replace("\\", "/")
222
+ if fnmatch.fnmatch(norm, clean) or fnmatch.fnmatch(base, clean):
223
+ return True
224
+ # A bare directory entry exempts everything beneath it.
225
+ if clean.endswith("/") and norm.startswith(clean):
226
+ return True
227
+ return False
228
+
229
+
230
+ def check_file_path(
231
+ path: str,
232
+ max_size_mb: Optional[float] = None,
233
+ config: Optional[Config] = None,
234
+ size_bytes: Optional[int] = None,
235
+ ) -> Tuple[bool, List[str]]:
236
+ """
237
+ Validate an individual file path for size and sensitive filename patterns.
238
+ Returns (is_valid, list_of_warning_messages).
239
+
240
+ ``size_bytes`` supplies the staged blob size; when omitted the working-tree
241
+ file is measured instead.
242
+ """
243
+ cfg = _config_or_default(config)
244
+ size_limit = max_size_mb if max_size_mb is not None else cfg.max_size_mb
245
+
246
+ issues: List[str] = []
247
+ norm_path = path.replace("\\", "/")
248
+
249
+ if not is_allowlisted(norm_path, cfg.allowlist):
250
+ for pattern in cfg.sensitive_patterns:
251
+ try:
252
+ compiled = re.compile(pattern, re.IGNORECASE)
253
+ except re.error as exc:
254
+ issues.append(
255
+ "Invalid sensitive pattern {0!r}: {1}".format(pattern, exc)
256
+ )
257
+ continue
258
+ if compiled.search(norm_path):
259
+ issues.append(
260
+ "Potentially sensitive file detected: '{0}'. "
261
+ "Avoid committing secrets.".format(path)
262
+ )
263
+ break
264
+
265
+ measured = size_bytes
266
+ if measured is None and os.path.isfile(path):
267
+ measured = os.path.getsize(path)
268
+
269
+ if measured is not None and size_limit:
270
+ size_mb = measured / (1024 * 1024)
271
+ if size_mb > size_limit:
272
+ issues.append(
273
+ "File '{0}' exceeds max allowed size ({1:.2f} MB > {2} MB limit). "
274
+ "Consider using Git LFS.".format(path, size_mb, size_limit)
275
+ )
276
+
277
+ return len(issues) == 0, issues