patchnote 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
patchnote/filters.py ADDED
@@ -0,0 +1,199 @@
1
+ """Filter, dedupe, and collapse commits before they become changelog entries."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from collections.abc import Iterable
7
+
8
+ from patchnote.config import Config
9
+ from patchnote.conventional import parse_commit, strip_pr_suffix
10
+ from patchnote.model import Commit, ParsedCommit, PullRequest
11
+
12
+
13
+ def compile_patterns(patterns: Iterable[str]) -> list[re.Pattern[str]]:
14
+ compiled: list[re.Pattern[str]] = []
15
+ for pattern in patterns:
16
+ try:
17
+ compiled.append(re.compile(pattern, re.IGNORECASE))
18
+ except re.error:
19
+ compiled.append(re.compile(re.escape(pattern), re.IGNORECASE))
20
+ return compiled
21
+
22
+
23
+ def matches_any(text: str, patterns: list[re.Pattern[str]]) -> bool:
24
+ return any(pattern.search(text) for pattern in patterns)
25
+
26
+
27
+ def is_merge_noise(commit: Commit) -> bool:
28
+ subject = commit.subject.strip()
29
+ if commit.is_merge:
30
+ return True
31
+ return subject.startswith(
32
+ ("Merge branch ", "Merge pull request ", "Merge remote-tracking branch ")
33
+ )
34
+
35
+
36
+ def is_bot_commit(commit: Commit, patterns: list[re.Pattern[str]]) -> bool:
37
+ haystack = f"{commit.author_name} {commit.author_email}"
38
+ return matches_any(haystack, patterns)
39
+
40
+
41
+ def should_skip_raw(
42
+ commit: Commit,
43
+ config: Config,
44
+ *,
45
+ ignore: list[re.Pattern[str]],
46
+ include_all: bool = False,
47
+ ) -> bool:
48
+ subject = commit.subject
49
+ if matches_any(subject, ignore) or matches_any(commit.message, ignore):
50
+ return True
51
+ return bool(config.skip_merge_commits and is_merge_noise(commit))
52
+
53
+
54
+ def parse_and_filter(
55
+ commits: list[Commit],
56
+ config: Config,
57
+ *,
58
+ include_all: bool = False,
59
+ ) -> tuple[list[ParsedCommit], list[ParsedCommit]]:
60
+ """Return (kept, bot_commits). Merge/ignore-pattern noise is dropped."""
61
+ ignore = compile_patterns(config.ignore_patterns)
62
+ bot_patterns = compile_patterns(config.bots.patterns)
63
+ kept: list[ParsedCommit] = []
64
+ bots: list[ParsedCommit] = []
65
+ for commit in commits:
66
+ if should_skip_raw(commit, config, ignore=ignore, include_all=include_all):
67
+ continue
68
+ parsed = parse_commit(commit)
69
+ if is_bot_commit(commit, bot_patterns):
70
+ if config.bots.skip and not include_all:
71
+ continue
72
+ bots.append(parsed)
73
+ if config.bots.collapse and not include_all and not parsed.breaking:
74
+ continue
75
+ if parsed.breaking:
76
+ bots.pop()
77
+ if parsed.type and parsed.type in config.skip_types and not include_all:
78
+ # A later label can rescue this commit; keep it tagged.
79
+ parsed_skip = parsed.model_copy()
80
+ # stash skip intent via a convention: we still return it and let
81
+ # classify drop it if no label overrides.
82
+ kept.append(parsed_skip)
83
+ continue
84
+ kept.append(parsed)
85
+ return kept, bots
86
+
87
+
88
+ def collapse_reverts(parsed: list[ParsedCommit]) -> list[ParsedCommit]:
89
+ """Drop a commit and its revert when both appear in the range.
90
+
91
+ A revert whose original is *not* in the range is kept (so a later release
92
+ can document the rollback).
93
+ """
94
+ by_hash = {item.raw.hash.lower(): item for item in parsed}
95
+ by_short = {item.raw.short_hash.lower(): item for item in parsed}
96
+ drop: set[str] = set()
97
+ for item in parsed:
98
+ if item.raw.hash in drop:
99
+ continue
100
+ reverted = item.reverted_hash
101
+ is_revert = item.type == "revert" or reverted is not None
102
+ if not is_revert or not reverted:
103
+ continue
104
+ key = reverted.lower()
105
+ original = by_hash.get(key)
106
+ if original is None:
107
+ for short, candidate in by_short.items():
108
+ if key.startswith(short) or short.startswith(key):
109
+ original = candidate
110
+ break
111
+ if original is None:
112
+ # Match by stripped description.
113
+ target = strip_pr_suffix(item.description).lower()
114
+ for candidate in parsed:
115
+ if candidate.raw.hash == item.raw.hash:
116
+ continue
117
+ if strip_pr_suffix(candidate.description).lower() == target:
118
+ original = candidate
119
+ break
120
+ if original is not None:
121
+ drop.add(item.raw.hash)
122
+ drop.add(original.raw.hash)
123
+ return [item for item in parsed if item.raw.hash not in drop]
124
+
125
+
126
+ def attach_pull_requests(
127
+ parsed: list[ParsedCommit],
128
+ pr_by_sha: dict[str, list[PullRequest]],
129
+ ) -> dict[str, list[PullRequest]]:
130
+ """Map commit hash → PRs using API data plus numbers parsed from messages."""
131
+ attached: dict[str, list[PullRequest]] = {}
132
+ for item in parsed:
133
+ sha = item.raw.hash
134
+ prs = list(pr_by_sha.get(sha, []))
135
+ if not prs:
136
+ prs = list(pr_by_sha.get(sha.lower(), []))
137
+ attached[sha] = prs
138
+ return attached
139
+
140
+
141
+ def dedupe_by_pr(
142
+ parsed: list[ParsedCommit],
143
+ prs_for: dict[str, list[PullRequest]],
144
+ ) -> list[tuple[ParsedCommit, PullRequest | None, list[ParsedCommit]]]:
145
+ """Keep one representative per pull request, merging hashes.
146
+
147
+ Returns tuples of (representative, pr_or_none, grouped_commits).
148
+ Commits without a PR stay as individual entries.
149
+ """
150
+ groups: dict[int, list[ParsedCommit]] = {}
151
+ loners: list[ParsedCommit] = []
152
+ pr_objects: dict[int, PullRequest] = {}
153
+ for item in parsed:
154
+ prs = prs_for.get(item.raw.hash) or []
155
+ number: int | None = None
156
+ pr_obj: PullRequest | None = None
157
+ if prs:
158
+ pr_obj = prs[0]
159
+ number = pr_obj.number
160
+ pr_objects[number] = pr_obj
161
+ elif item.pr_numbers:
162
+ number = item.pr_numbers[0]
163
+ if number is None:
164
+ loners.append(item)
165
+ continue
166
+ groups.setdefault(number, []).append(item)
167
+ if pr_obj is not None:
168
+ pr_objects[number] = pr_obj
169
+ result: list[tuple[ParsedCommit, PullRequest | None, list[ParsedCommit]]] = []
170
+ seen_hashes: set[str] = set()
171
+ for item in parsed:
172
+ if item.raw.hash in seen_hashes:
173
+ continue
174
+ prs = prs_for.get(item.raw.hash) or []
175
+ number = prs[0].number if prs else (item.pr_numbers[0] if item.pr_numbers else None)
176
+ if number is None:
177
+ seen_hashes.add(item.raw.hash)
178
+ result.append((item, None, [item]))
179
+ continue
180
+ group = groups.get(number, [item])
181
+ for member in group:
182
+ seen_hashes.add(member.raw.hash)
183
+ representative = _pick_representative(group)
184
+ result.append((representative, pr_objects.get(number), group))
185
+ return result
186
+
187
+
188
+ def _pick_representative(group: list[ParsedCommit]) -> ParsedCommit:
189
+ """Prefer a conventional commit, then the earliest."""
190
+ conventional = [item for item in group if item.is_conventional]
191
+ pool = conventional or group
192
+ return min(pool, key=lambda item: item.raw.author_date)
193
+
194
+
195
+ def author_handle(commit: Commit, pr: PullRequest | None) -> str:
196
+ if pr is not None and pr.author:
197
+ return pr.author if pr.author.startswith("@") else f"@{pr.author}"
198
+ name = commit.author_name.strip()
199
+ return name
patchnote/github.py ADDED
@@ -0,0 +1,388 @@
1
+ """GitHub REST + GraphQL enrichment. Failures never abort a changelog run."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ import re
7
+ from collections.abc import Callable
8
+ from typing import Any
9
+ from urllib.parse import quote, urlparse
10
+
11
+ import httpx
12
+
13
+ from patchnote.config import Config
14
+ from patchnote.model import Commit, PullRequest, RuntimeNetworkError
15
+
16
+ WarnFn = Callable[[str], None]
17
+
18
+
19
+ GITHUB_REMOTE_RE = re.compile(
20
+ r"(?:github\.com[:/])(?P<owner>[^/]+)/(?P<repo>[^/.]+?)(?:\.git)?/?$",
21
+ re.IGNORECASE,
22
+ )
23
+
24
+
25
+ def parse_github_nwo(url: str) -> tuple[str, str] | None:
26
+ """Return (owner, repo) from a git remote URL or ``owner/repo`` slug."""
27
+ text = url.strip()
28
+ if not text:
29
+ return None
30
+ if re.fullmatch(r"[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+", text):
31
+ owner, repo = text.split("/", 1)
32
+ return owner, repo.removesuffix(".git")
33
+ if text.lower().startswith("git@github.com:"):
34
+ return parse_github_nwo(text.split(":", 1)[1])
35
+ try:
36
+ parsed = urlparse(text)
37
+ if parsed.hostname and parsed.hostname.lower() in {"github.com", "www.github.com"}:
38
+ parts = [p for p in parsed.path.split("/") if p]
39
+ if len(parts) == 2:
40
+ return parse_github_nwo("/".join(parts))
41
+ except ValueError:
42
+ return None
43
+ return None
44
+
45
+
46
+ def repository_http_url(owner: str, repo: str) -> str:
47
+ return f"https://github.com/{owner}/{repo}"
48
+
49
+
50
+ def compare_url(owner: str, repo: str, from_ref: str, to_ref: str) -> str:
51
+ left, right = quote(from_ref, safe=""), quote(to_ref, safe="")
52
+ return f"https://github.com/{owner}/{repo}/compare/{left}...{right}"
53
+
54
+
55
+ def pr_url(owner: str, repo: str, number: int) -> str:
56
+ return f"https://github.com/{owner}/{repo}/pull/{number}"
57
+
58
+
59
+ def resolve_token(config: Config, explicit: str | None = None) -> str | None:
60
+ if explicit:
61
+ return explicit
62
+ env_name = config.github.token_env
63
+ return os.environ.get(env_name) or os.environ.get("GITHUB_TOKEN") or None
64
+
65
+
66
+ class GitHubClient:
67
+ """Minimal GitHub API client. Every public method is best-effort except releases."""
68
+
69
+ def __init__(
70
+ self,
71
+ client: httpx.Client,
72
+ token: str | None,
73
+ owner: str,
74
+ repo: str,
75
+ *,
76
+ api_url: str = "https://api.github.com",
77
+ graphql_url: str = "https://api.github.com/graphql",
78
+ warn: WarnFn | None = None,
79
+ ) -> None:
80
+ self._http = client
81
+ self._token = token
82
+ self.owner = owner
83
+ self.repo = repo
84
+ self.api_url = api_url.rstrip("/")
85
+ self.graphql_url = graphql_url
86
+ self._warn = warn or (lambda _msg: None)
87
+ self._gave_up = False
88
+
89
+ def _headers(self, extra: dict[str, str] | None = None) -> dict[str, str]:
90
+ headers = {
91
+ "Accept": "application/vnd.github+json",
92
+ "X-GitHub-Api-Version": "2022-11-28",
93
+ "User-Agent": "patchnote",
94
+ }
95
+ if self._token:
96
+ headers["Authorization"] = f"Bearer {self._token}"
97
+ if extra:
98
+ headers.update(extra)
99
+ return headers
100
+
101
+ def _handle_error(self, response: httpx.Response, action: str) -> None:
102
+ if response.status_code == 429:
103
+ self._gave_up = True
104
+ self._warn(f"GitHub API rate-limited while {action}; skipping remaining enrichment")
105
+ return
106
+ if response.status_code in {401, 403}:
107
+ self._gave_up = True
108
+ self._warn(
109
+ f"GitHub API returned {response.status_code} while {action}; skipping enrichment"
110
+ )
111
+ return
112
+ if response.status_code >= 400:
113
+ self._warn(f"GitHub API {response.status_code} while {action}")
114
+
115
+ def _get(self, path: str, params: dict[str, Any] | None = None) -> httpx.Response | None:
116
+ if self._gave_up:
117
+ return None
118
+ url = path if path.startswith("http") else f"{self.api_url}{path}"
119
+ try:
120
+ response = self._http.get(url, headers=self._headers(), params=params, timeout=20.0)
121
+ except httpx.HTTPError as exc:
122
+ self._warn(f"GitHub request failed ({exc.__class__.__name__}); skipping enrichment")
123
+ self._gave_up = True
124
+ return None
125
+ if response.status_code >= 400:
126
+ self._handle_error(response, f"GET {path}")
127
+ return None
128
+ return response
129
+
130
+ def associated_pulls_rest(self, sha: str) -> list[PullRequest]:
131
+ path = f"/repos/{self.owner}/{self.repo}/commits/{sha}/pulls"
132
+ response = self._get(path)
133
+ if response is None:
134
+ return []
135
+ try:
136
+ payload = response.json()
137
+ except ValueError:
138
+ self._warn("GitHub returned non-JSON for associated pulls")
139
+ return []
140
+ if not isinstance(payload, list):
141
+ return []
142
+ return [
143
+ self._pr_from_rest(item)
144
+ for item in payload
145
+ if isinstance(item, dict) and ("merged_at" not in item or item["merged_at"])
146
+ ]
147
+
148
+ def _pr_from_rest(self, item: dict[str, Any]) -> PullRequest:
149
+ labels_raw = item.get("labels") or []
150
+ labels = [
151
+ str(label.get("name", ""))
152
+ for label in labels_raw
153
+ if isinstance(label, dict) and label.get("name")
154
+ ]
155
+ user = item.get("user") or {}
156
+ author = user.get("login") if isinstance(user, dict) else None
157
+ body = item.get("body")
158
+ number = int(item.get("number", 0))
159
+ return PullRequest(
160
+ number=number,
161
+ title=str(item.get("title") or ""),
162
+ body=body if isinstance(body, str) else None,
163
+ labels=labels,
164
+ author=author,
165
+ url=str(item.get("html_url") or pr_url(self.owner, self.repo, number)),
166
+ merge_commit_sha=item.get("merge_commit_sha"),
167
+ linked_issues=_issues_from_text(f"{item.get('title') or ''}\n{body or ''}"),
168
+ )
169
+
170
+ def associated_pulls_graphql(self, shas: list[str]) -> dict[str, list[PullRequest]] | None:
171
+ if not shas or self._gave_up or not self._token:
172
+ return None
173
+ aliases: list[str] = []
174
+ for index, sha in enumerate(shas):
175
+ if not re.fullmatch(r"[0-9a-fA-F]{7,40}", sha):
176
+ continue
177
+ aliases.append(
178
+ f'c{index}: repository(owner: "{self.owner}", name: "{self.repo}") '
179
+ f'{{ object(oid: "{sha}") {{ ... on Commit {{ oid '
180
+ f"associatedPullRequests(first: 5) {{ nodes {{ "
181
+ f"number title body url mergedAt author {{ login }} "
182
+ f"labels(first: 20) {{ nodes {{ name }} }} "
183
+ f"closingIssuesReferences(first: 10) {{ nodes {{ number }} }} "
184
+ f"}} }} }} }} }}"
185
+ )
186
+ query = "query {\n" + "\n".join(aliases) + "\n}"
187
+ try:
188
+ response = self._http.post(
189
+ self.graphql_url,
190
+ headers=self._headers({"Content-Type": "application/json"}),
191
+ json={"query": query},
192
+ timeout=30.0,
193
+ )
194
+ except httpx.HTTPError as exc:
195
+ self._warn(f"GitHub GraphQL failed ({exc.__class__.__name__}); falling back to REST")
196
+ return None
197
+ if response.status_code >= 400:
198
+ self._handle_error(response, "GraphQL associatedPullRequests")
199
+ return None
200
+ try:
201
+ payload = response.json()
202
+ except ValueError:
203
+ return None
204
+ if not isinstance(payload, dict) or payload.get("errors"):
205
+ self._warn("GitHub GraphQL returned errors; falling back to REST")
206
+ return None
207
+ data = payload.get("data") or {}
208
+ result: dict[str, list[PullRequest]] = {}
209
+ for index, sha in enumerate(shas):
210
+ node = (data.get(f"c{index}") or {}).get("object") if isinstance(data, dict) else None
211
+ if not isinstance(node, dict):
212
+ result[sha] = []
213
+ continue
214
+ pr_nodes = (
215
+ ((node.get("associatedPullRequests") or {}).get("nodes"))
216
+ if isinstance(node.get("associatedPullRequests"), dict)
217
+ else None
218
+ )
219
+ pulls: list[PullRequest] = []
220
+ if isinstance(pr_nodes, list):
221
+ for raw in pr_nodes:
222
+ if isinstance(raw, dict) and ("mergedAt" not in raw or raw["mergedAt"]):
223
+ pulls.append(self._pr_from_graphql(raw))
224
+ result[sha] = pulls
225
+ return result
226
+
227
+ def _pr_from_graphql(self, raw: dict[str, Any]) -> PullRequest:
228
+ labels_nodes = (
229
+ ((raw.get("labels") or {}).get("nodes")) if isinstance(raw.get("labels"), dict) else []
230
+ )
231
+ labels = [
232
+ str(node.get("name"))
233
+ for node in labels_nodes or []
234
+ if isinstance(node, dict) and node.get("name")
235
+ ]
236
+ author_raw = raw.get("author") or {}
237
+ author = author_raw.get("login") if isinstance(author_raw, dict) else None
238
+ issue_nodes = (
239
+ ((raw.get("closingIssuesReferences") or {}).get("nodes"))
240
+ if isinstance(raw.get("closingIssuesReferences"), dict)
241
+ else []
242
+ )
243
+ issues = [
244
+ int(node["number"])
245
+ for node in issue_nodes or []
246
+ if isinstance(node, dict) and "number" in node
247
+ ]
248
+ number = int(raw.get("number", 0))
249
+ body = raw.get("body")
250
+ return PullRequest(
251
+ number=number,
252
+ title=str(raw.get("title") or ""),
253
+ body=body if isinstance(body, str) else None,
254
+ labels=labels,
255
+ author=author,
256
+ url=str(raw.get("url") or pr_url(self.owner, self.repo, number)),
257
+ linked_issues=issues or _issues_from_text(f"{raw.get('title') or ''}\n{body or ''}"),
258
+ )
259
+
260
+ def enrich_commits(self, commits: list[Commit]) -> dict[str, list[PullRequest]]:
261
+ """Map commit SHA → associated PRs. Never raises for API problems."""
262
+ mapping: dict[str, list[PullRequest]] = {commit.hash: [] for commit in commits}
263
+ if not commits or self._gave_up:
264
+ return mapping
265
+ shas = [commit.hash for commit in commits]
266
+ chunk_size = 15
267
+ for start in range(0, len(shas), chunk_size):
268
+ chunk = shas[start : start + chunk_size]
269
+ batched = self.associated_pulls_graphql(chunk)
270
+ if batched is None:
271
+ for sha in chunk:
272
+ if self._gave_up:
273
+ return mapping
274
+ mapping[sha] = self.associated_pulls_rest(sha)
275
+ else:
276
+ for sha, pulls in batched.items():
277
+ mapping[sha] = pulls
278
+ return mapping
279
+
280
+ def first_time_contributors(self, logins: set[str], current_pr_numbers: set[int]) -> set[str]:
281
+ """Return logins whose only merged PRs are in ``current_pr_numbers``."""
282
+ first: set[str] = set()
283
+ for login in logins:
284
+ if self._gave_up:
285
+ break
286
+ query = f"repo:{self.owner}/{self.repo} author:{login} is:pr is:merged"
287
+ response = self._get("/search/issues", params={"q": query, "per_page": 10})
288
+ if response is None:
289
+ continue
290
+ try:
291
+ payload = response.json()
292
+ except ValueError:
293
+ continue
294
+ total = int(payload.get("total_count") or 0) if isinstance(payload, dict) else 0
295
+ items = payload.get("items") if isinstance(payload, dict) else None
296
+ numbers: set[int] = set()
297
+ if isinstance(items, list):
298
+ for item in items:
299
+ if isinstance(item, dict) and "number" in item:
300
+ numbers.add(int(item["number"]))
301
+ if total == 0 or (total <= len(current_pr_numbers) and numbers <= current_pr_numbers):
302
+ first.add(login.lower())
303
+ return first
304
+
305
+ def get_release_by_tag(self, tag: str) -> dict[str, Any] | None:
306
+ response = self._get(f"/repos/{self.owner}/{self.repo}/releases/tags/{tag}")
307
+ if response is None:
308
+ return None
309
+ try:
310
+ payload = response.json()
311
+ except ValueError:
312
+ return None
313
+ return payload if isinstance(payload, dict) else None
314
+
315
+ def create_or_update_release(
316
+ self,
317
+ *,
318
+ tag: str,
319
+ name: str,
320
+ body: str,
321
+ draft: bool = True,
322
+ prerelease: bool = False,
323
+ target: str | None = None,
324
+ ) -> dict[str, Any]:
325
+ if not self._token:
326
+ raise RuntimeNetworkError("A GitHub token is required to create a release")
327
+ existing = self.get_release_by_tag(tag)
328
+ payload: dict[str, Any] = {
329
+ "tag_name": tag,
330
+ "name": name,
331
+ "body": body,
332
+ "draft": draft,
333
+ "prerelease": prerelease,
334
+ }
335
+ if target:
336
+ payload["target_commitish"] = target
337
+ headers = self._headers({"Content-Type": "application/json"})
338
+ try:
339
+ if existing and "id" in existing:
340
+ url = f"{self.api_url}/repos/{self.owner}/{self.repo}/releases/{existing['id']}"
341
+ response = self._http.patch(url, headers=headers, json=payload, timeout=30.0)
342
+ else:
343
+ url = f"{self.api_url}/repos/{self.owner}/{self.repo}/releases"
344
+ response = self._http.post(url, headers=headers, json=payload, timeout=30.0)
345
+ except httpx.HTTPError as exc:
346
+ raise RuntimeNetworkError(
347
+ f"Failed to publish GitHub release: {type(exc).__name__}"
348
+ ) from exc
349
+ if response.status_code >= 400:
350
+ raise RuntimeNetworkError(f"GitHub release API returned {response.status_code}")
351
+ try:
352
+ data = response.json()
353
+ except ValueError as exc:
354
+ raise RuntimeNetworkError("GitHub release API returned non-JSON") from exc
355
+ if not isinstance(data, dict):
356
+ raise RuntimeNetworkError("Unexpected GitHub release payload")
357
+ return data
358
+
359
+
360
+ ISSUE_RE = re.compile(
361
+ r"(?:close[sd]?|fix(?:e[sd])?|resolve[sd]?)\s+#(\d+)",
362
+ re.IGNORECASE,
363
+ )
364
+
365
+
366
+ def _issues_from_text(text: str) -> list[int]:
367
+ found: list[int] = []
368
+ seen: set[int] = set()
369
+ for match in ISSUE_RE.finditer(text):
370
+ number = int(match.group(1))
371
+ if number not in seen:
372
+ found.append(number)
373
+ seen.add(number)
374
+ return found
375
+
376
+
377
+ def enrich_or_empty(
378
+ commits: list[Commit],
379
+ client: GitHubClient | None,
380
+ warn: WarnFn | None = None,
381
+ ) -> dict[str, list[PullRequest]]:
382
+ if client is None:
383
+ return {}
384
+ try:
385
+ return client.enrich_commits(commits)
386
+ except Exception as exc:
387
+ (warn or (lambda _m: None))(f"GitHub enrichment failed: {type(exc).__name__}")
388
+ return {}