patchnote 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
patchnote/classify.py ADDED
@@ -0,0 +1,353 @@
1
+ """Rule-based classification of commits into changelog sections."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from collections.abc import Callable
7
+ from datetime import date, datetime, timezone
8
+
9
+ from patchnote.config import Config
10
+ from patchnote.conventional import parse_commit, strip_pr_suffix
11
+ from patchnote.filters import (
12
+ collapse_reverts,
13
+ dedupe_by_pr,
14
+ parse_and_filter,
15
+ )
16
+ from patchnote.model import (
17
+ Changelog,
18
+ ChangelogEntry,
19
+ Commit,
20
+ Confidence,
21
+ Contributor,
22
+ EntrySource,
23
+ ParsedCommit,
24
+ PullRequest,
25
+ Style,
26
+ )
27
+
28
+ Clock = Callable[[], date]
29
+
30
+ HEURISTIC_RULES: list[tuple[str, tuple[str, ...]]] = [
31
+ ("Security", ("security", "cve-", "xss", "csrf", "vulnerability", "advisory")),
32
+ ("Fixed", ("fix", "bug", "hotfix", "patch", "resolve", "repair")),
33
+ ("Deprecated", ("deprecat",)),
34
+ ("Removed", ("remove", "delete", "drop ", "dropped")),
35
+ ("Added", ("add ", "added", "introduce", "implement", "create", "new ")),
36
+ ("Changed", ("update", "change", "improve", "refactor", "enhance", "tweak")),
37
+ ]
38
+
39
+
40
+ def sentence_case(text: str) -> str:
41
+ if not text:
42
+ return text
43
+ return text[0].upper() + text[1:]
44
+
45
+
46
+ def apply_trailing_period(text: str, keep: bool) -> str:
47
+ stripped = text.rstrip()
48
+ if not stripped:
49
+ return stripped
50
+ if keep:
51
+ return stripped if stripped.endswith(".") else stripped + "."
52
+ return stripped[:-1] if stripped.endswith(".") else stripped
53
+
54
+
55
+ def clean_summary(text: str, config: Config) -> str:
56
+ summary = strip_pr_suffix(text.strip())
57
+ summary = re.sub(r"\s+", " ", summary).strip()
58
+ if config.sentence_case:
59
+ summary = sentence_case(summary)
60
+ summary = apply_trailing_period(summary, config.trailing_period)
61
+ return summary
62
+
63
+
64
+ def heuristic_section(text: str, config: Config) -> tuple[str, Confidence]:
65
+ lowered = text.lower()
66
+ for section, keywords in HEURISTIC_RULES:
67
+ # Map keep-a-changelog names onto conventional section names if needed.
68
+ mapped = _map_section_name(section, config)
69
+ for keyword in keywords:
70
+ if keyword in lowered:
71
+ return mapped, Confidence.LOW
72
+ return config.other_section, Confidence.LOW
73
+
74
+
75
+ def _map_section_name(keepachangelog: str, config: Config) -> str:
76
+ if config.style is Style.KEEPACHANGELOG:
77
+ return keepachangelog
78
+ mapping = {
79
+ "Added": "Features",
80
+ "Fixed": "Bug Fixes",
81
+ "Changed": "Other",
82
+ "Deprecated": "Other",
83
+ "Removed": "Reverts",
84
+ "Security": "Other",
85
+ }
86
+ return mapping.get(keepachangelog, config.other_section)
87
+
88
+
89
+ def _label_section(labels: list[str], config: Config) -> str | None:
90
+ mapping = config.label_to_section()
91
+ for label in labels:
92
+ section = mapping.get(label.lower())
93
+ if section:
94
+ return section
95
+ return None
96
+
97
+
98
+ def _type_section(ctype: str | None, config: Config) -> str | None:
99
+ if not ctype:
100
+ return None
101
+ return config.type_to_section().get(ctype.lower())
102
+
103
+
104
+ def classify_one(
105
+ parsed: ParsedCommit,
106
+ config: Config,
107
+ pr: PullRequest | None,
108
+ group: list[ParsedCommit],
109
+ *,
110
+ include_all: bool = False,
111
+ ) -> ChangelogEntry | None:
112
+ if pr is not None and pr.title.strip():
113
+ titled = parse_commit(parsed.raw.model_copy(update={"subject": pr.title, "body": ""}))
114
+ if titled.is_conventional:
115
+ parsed = titled.model_copy(
116
+ update={
117
+ "breaking": parsed.breaking or titled.breaking,
118
+ "footers": parsed.footers,
119
+ "pr_numbers": parsed.pr_numbers,
120
+ }
121
+ )
122
+ labels = list(pr.labels) if pr is not None else []
123
+ label_section = _label_section(labels, config)
124
+ type_section = _type_section(parsed.type, config)
125
+
126
+ skip_by_type = (
127
+ parsed.type is not None
128
+ and parsed.type in config.skip_types
129
+ and not include_all
130
+ and label_section is None
131
+ )
132
+
133
+ breaking = parsed.breaking or any(item.breaking for item in group)
134
+ if pr is not None:
135
+ lowered_labels = {label.lower() for label in pr.labels}
136
+ if lowered_labels & {"breaking", "breaking-change", "breaking change"}:
137
+ breaking = True
138
+
139
+ if skip_by_type and not breaking:
140
+ return None
141
+
142
+ if label_section and label_section.lower() not in {"breaking changes", "breaking"}:
143
+ section = label_section
144
+ source = EntrySource.LABEL
145
+ confidence = Confidence.HIGH
146
+ elif type_section:
147
+ section = type_section
148
+ source = EntrySource.CONVENTIONAL
149
+ confidence = Confidence.HIGH
150
+ elif parsed.is_conventional:
151
+ section = config.other_section
152
+ source = EntrySource.CONVENTIONAL
153
+ confidence = Confidence.MEDIUM
154
+ else:
155
+ section, confidence = heuristic_section(parsed.description, config)
156
+ source = EntrySource.HEURISTIC if section != config.other_section else EntrySource.OTHER
157
+
158
+ if breaking:
159
+ section = config.breaking_section
160
+
161
+ summary_source = parsed.description
162
+ if pr is not None and pr.title.strip():
163
+ title_parsed = parse_commit(parsed.raw.model_copy(update={"subject": pr.title, "body": ""}))
164
+ summary_source = title_parsed.description
165
+ summary = clean_summary(summary_source, config)
166
+
167
+ breaking_description = None
168
+ for item in group:
169
+ for footer in item.footers:
170
+ if footer.key.upper().replace("-", " ") == "BREAKING CHANGE":
171
+ breaking_description = footer.value
172
+ break
173
+
174
+ hashes = [item.raw.hash for item in group]
175
+ short_hashes = [item.raw.short_hash for item in group]
176
+ authors: list[str] = []
177
+ seen_authors: set[str] = set()
178
+ if pr is not None and pr.author:
179
+ handle = pr.author if pr.author.startswith("@") else f"@{pr.author}"
180
+ authors.append(handle)
181
+ seen_authors.add(handle.lower())
182
+ for item in group:
183
+ name = item.raw.author_name.strip()
184
+ key = name.lower()
185
+ if name and key not in seen_authors:
186
+ authors.append(name)
187
+ seen_authors.add(key)
188
+
189
+ pr_number = (
190
+ pr.number if pr is not None else (parsed.pr_numbers[0] if parsed.pr_numbers else None)
191
+ )
192
+ pr_url = pr.url if pr is not None else None
193
+ scope = parsed.scope
194
+ identity = f"pr-{pr_number}" if pr_number is not None else f"sha-{parsed.raw.hash}"
195
+
196
+ return ChangelogEntry(
197
+ id=identity,
198
+ summary=summary,
199
+ original_summary=(pr.title if pr is not None else parsed.raw.subject).strip(),
200
+ pr_body=pr.body if pr is not None else None,
201
+ section=section,
202
+ scope=scope,
203
+ pr_number=pr_number,
204
+ pr_url=pr_url,
205
+ authors=authors,
206
+ hashes=hashes,
207
+ short_hashes=short_hashes,
208
+ breaking=breaking,
209
+ breaking_description=breaking_description,
210
+ confidence=confidence,
211
+ source=source,
212
+ labels=labels,
213
+ linked_issues=list(pr.linked_issues) if pr is not None else [],
214
+ )
215
+
216
+
217
+ def collapse_bot_entries(
218
+ bots: list[ParsedCommit],
219
+ config: Config,
220
+ ) -> ChangelogEntry | None:
221
+ if not bots or not config.bots.collapse:
222
+ return None
223
+ count = len(bots)
224
+ noun = "update" if count == 1 else "updates"
225
+ summary = clean_summary(f"{count} dependency {noun}", config)
226
+ return ChangelogEntry(
227
+ id="bots-collapsed",
228
+ summary=summary,
229
+ original_summary=summary,
230
+ section=config.bots.section,
231
+ authors=[],
232
+ hashes=[item.raw.hash for item in bots],
233
+ short_hashes=[item.raw.short_hash for item in bots],
234
+ confidence=Confidence.MEDIUM,
235
+ source=EntrySource.BOT,
236
+ )
237
+
238
+
239
+ def collect_contributors(
240
+ entries: list[ChangelogEntry],
241
+ parsed: list[ParsedCommit],
242
+ first_time: set[str] | None = None,
243
+ ) -> list[Contributor]:
244
+ contribs: dict[str, Contributor] = {}
245
+ for entry in entries:
246
+ for author in entry.authors:
247
+ key = author.lower()
248
+ if key not in contribs:
249
+ login = author[1:] if author.startswith("@") else None
250
+ contribs[key] = Contributor(
251
+ login=login,
252
+ name=author,
253
+ first_time=bool(first_time and (author.lstrip("@").lower() in first_time)),
254
+ pr_count=1 if entry.pr_number else 0,
255
+ )
256
+ else:
257
+ if entry.pr_number:
258
+ contribs[key].pr_count += 1
259
+ for item in parsed:
260
+ key = item.raw.author_name.strip().lower()
261
+ if key and key not in contribs:
262
+ contribs[key] = Contributor(
263
+ name=item.raw.author_name.strip(),
264
+ email=item.raw.author_email,
265
+ )
266
+ return sorted(contribs.values(), key=lambda c: c.name.lower())
267
+
268
+
269
+ def build_changelog(
270
+ commits: list[Commit],
271
+ config: Config,
272
+ *,
273
+ from_ref: str,
274
+ to_ref: str,
275
+ version: str | None,
276
+ today: date | None = None,
277
+ pr_by_sha: dict[str, list[PullRequest]] | None = None,
278
+ include_all: bool = False,
279
+ repository_url: str | None = None,
280
+ compare_url: str | None = None,
281
+ previous_tag: str | None = None,
282
+ first_time_logins: set[str] | None = None,
283
+ tag_name: str | None = None,
284
+ ) -> Changelog:
285
+ """Pure assembly of a changelog from already-fetched commits and PRs."""
286
+ kept, bots = parse_and_filter(commits, config, include_all=include_all)
287
+ kept = collapse_reverts(kept)
288
+ pr_by_sha = pr_by_sha or {}
289
+ prs_for: dict[str, list[PullRequest]] = {}
290
+ for item in kept:
291
+ sha = item.raw.hash
292
+ prs_for[sha] = list(pr_by_sha.get(sha, []) or pr_by_sha.get(sha.lower(), []))
293
+
294
+ grouped = dedupe_by_pr(kept, prs_for)
295
+ entries: list[ChangelogEntry] = []
296
+ for representative, pr, group in grouped:
297
+ entry = classify_one(representative, config, pr, group, include_all=include_all)
298
+ if entry is not None:
299
+ if first_time_logins:
300
+ entry.first_time_contributors = [
301
+ author
302
+ for author in entry.authors
303
+ if author.lstrip("@").lower() in first_time_logins
304
+ ]
305
+ entries.append(entry)
306
+
307
+ bot_entry = collapse_bot_entries(bots, config) if not include_all else None
308
+ if bot_entry is not None:
309
+ entries.append(bot_entry)
310
+ elif bots and (include_all or not config.bots.collapse):
311
+ existing_hashes = {sha for entry in entries for sha in entry.hashes}
312
+ for bot in bots:
313
+ if bot.raw.hash in existing_hashes:
314
+ continue
315
+ entry = classify_one(bot, config, None, [bot], include_all=True)
316
+ if entry is not None:
317
+ entry.source = EntrySource.BOT
318
+ entries.append(entry)
319
+
320
+ breaking: list[ChangelogEntry] = []
321
+ sections: dict[str, list[ChangelogEntry]] = {name: [] for name in config.section_order()}
322
+ for entry in entries:
323
+ if entry.breaking or entry.section == config.breaking_section:
324
+ breaking.append(entry)
325
+ continue
326
+ sections.setdefault(entry.section, []).append(entry)
327
+
328
+ # Drop empty sections (keep order by using section_order at render time).
329
+ sections = {name: items for name, items in sections.items() if items}
330
+
331
+ contributors: list[Contributor] = []
332
+ if config.include_contributors:
333
+ contributors = collect_contributors(entries, kept, first_time_logins)
334
+
335
+ when = today or datetime.now(timezone.utc).date()
336
+ return Changelog(
337
+ version=version,
338
+ from_ref=from_ref,
339
+ to_ref=to_ref,
340
+ date=when,
341
+ sections=sections,
342
+ breaking=breaking,
343
+ contributors=contributors,
344
+ compare_url=compare_url,
345
+ previous_tag=previous_tag,
346
+ repository_url=repository_url,
347
+ style=config.style,
348
+ tag_name=tag_name,
349
+ )
350
+
351
+
352
+ def default_clock() -> date:
353
+ return datetime.now(timezone.utc).date()