sarj-python-lint 0.19.0__tar.gz → 0.20.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/PKG-INFO +17 -1
  2. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/README.md +16 -0
  3. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/pyproject.toml +1 -1
  4. sarj_python_lint-0.20.0/src/sarj_python_lint/rules/_comments.py +375 -0
  5. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_registry.py +6 -0
  6. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_comment_cruft.py +187 -41
  7. sarj_python_lint-0.20.0/src/sarj_python_lint/rules/no_restated_comment.py +318 -0
  8. sarj_python_lint-0.20.0/src/sarj_python_lint/rules/redundant_docstring.py +211 -0
  9. sarj_python_lint-0.20.0/src/sarj_python_lint/rules/trailing_value_narration.py +134 -0
  10. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/.gitignore +0 -0
  11. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/__init__.py +0 -0
  12. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/__main__.py +0 -0
  13. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/_secret_names.py +0 -0
  14. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/_version.py +0 -0
  15. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/py.typed +0 -0
  16. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rule_base.py +0 -0
  17. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/__init__.py +0 -0
  18. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_first_party.py +0 -0
  19. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_logging.py +0 -0
  20. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_paths.py +0 -0
  21. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_sql.py +0 -0
  22. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/fixture_returns_bare_tuple.py +0 -0
  23. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/inefficient_string_concat_in_loop.py +0 -0
  24. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/kwarg_heavy_construction_in_test.py +0 -0
  25. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/kwonly_same_type_params.py +0 -0
  26. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/mock_without_spec.py +0 -0
  27. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_aggregation_in_store_query.py +0 -0
  28. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_cors_wildcard_with_credentials.py +0 -0
  29. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_fat_try_blocks.py +0 -0
  30. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_file_level_suppression.py +0 -0
  31. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_first_party_private_import.py +0 -0
  32. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_fstring_in_log.py +0 -0
  33. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_isinstance_union_chain.py +0 -0
  34. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_offset_pagination.py +0 -0
  35. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_query_with_many_joins.py +0 -0
  36. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_raw_sql_in_tests.py +0 -0
  37. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_repeated_string_literal.py +0 -0
  38. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_secret_in_log.py +0 -0
  39. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_select_star.py +0 -0
  40. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_sentinel_return_on_except.py +0 -0
  41. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_sequential_await.py +0 -0
  42. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_sleep_in_test_body.py +0 -0
  43. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_unreachable_after_terminal.py +0 -0
  44. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/parametrize_case_needs_id.py +0 -0
  45. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_class_row.py +0 -0
  46. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_constant_time_secret_compare.py +0 -0
  47. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_match_assert_never.py +0 -0
  48. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_module_level_constant.py +0 -0
  49. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_namedtuple_over_tuple_return.py +0 -0
  50. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_str_enum.py +0 -0
  51. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_struct_over_namedtuple.py +0 -0
  52. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_timedelta_for_durations.py +0 -0
  53. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/pydantic_at_boundaries.py +0 -0
  54. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/single_public_export.py +0 -0
  55. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/sleep_with_computed_arg_in_test.py +0 -0
  56. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/stepdown.py +0 -0
  57. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/store_insert_requires_on_conflict.py +0 -0
  58. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/test_loops_over_literal_cases.py +0 -0
  59. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/xfail_requires_strict.py +0 -0
  60. {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/zero_assertion_test.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sarj-python-lint
3
- Version: 0.19.0
3
+ Version: 0.20.0
4
4
  Summary: Custom Python lint rules — AST-based, pre-commit-friendly, hypermodern defaults
5
5
  Project-URL: Homepage, https://github.com/sarj-ai/standards/tree/main/packages/python
6
6
  Project-URL: Repository, https://github.com/sarj-ai/standards
@@ -85,6 +85,22 @@ Attribute access (`session._stt`) is out of scope and stays with ruff's
85
85
  `SLF001`, which cannot make the distinction either — see the rationale in
86
86
  `ruff.strict.toml`.
87
87
 
88
+ ### Comment-hygiene rules (0.20.0)
89
+
90
+ From a 37,918-comment, nine-repo measurement study. All three are
91
+ deletion-class, so each was validated against pydantic / trio / attrs as well as
92
+ the maintained repos before shipping — the counts and the false-positive classes
93
+ each guard was built from are recorded in the rule module docstrings.
94
+
95
+ ```yaml
96
+ - id: sarj-no-restated-comment # SARJ049
97
+ - id: sarj-redundant-docstring # SARJ050
98
+ - id: sarj-trailing-value-narration # SARJ051
99
+ ```
100
+
101
+ `redundant-docstring` finds real volume on a codebase that has never had it
102
+ (105 in noura-be), so the same baseline ratchet applies.
103
+
88
104
  Adopting these against an existing suite is easier through the baseline ratchet
89
105
  than as a big-bang fix — snapshot the current counts, then let them only shrink:
90
106
 
@@ -67,6 +67,22 @@ Attribute access (`session._stt`) is out of scope and stays with ruff's
67
67
  `SLF001`, which cannot make the distinction either — see the rationale in
68
68
  `ruff.strict.toml`.
69
69
 
70
+ ### Comment-hygiene rules (0.20.0)
71
+
72
+ From a 37,918-comment, nine-repo measurement study. All three are
73
+ deletion-class, so each was validated against pydantic / trio / attrs as well as
74
+ the maintained repos before shipping — the counts and the false-positive classes
75
+ each guard was built from are recorded in the rule module docstrings.
76
+
77
+ ```yaml
78
+ - id: sarj-no-restated-comment # SARJ049
79
+ - id: sarj-redundant-docstring # SARJ050
80
+ - id: sarj-trailing-value-narration # SARJ051
81
+ ```
82
+
83
+ `redundant-docstring` finds real volume on a codebase that has never had it
84
+ (105 in noura-be), so the same baseline ratchet applies.
85
+
70
86
  Adopting these against an existing suite is easier through the baseline ratchet
71
87
  than as a big-bang fix — snapshot the current counts, then let them only shrink:
72
88
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "sarj-python-lint"
3
- version = "0.19.0"
3
+ version = "0.20.0"
4
4
  description = "Custom Python lint rules — AST-based, pre-commit-friendly, hypermodern defaults"
5
5
  readme = "README.md"
6
6
  authors = [{ name = "sarj-ai" }]
@@ -0,0 +1,375 @@
1
+ """Shared comment analysis for the comment-hygiene rules (SARJ016/049/050/051).
2
+
3
+ Two things live here, both needed by more than one rule.
4
+
5
+ **The protected class.** Nine deterministic signals that mark a comment as
6
+ carrying something the code cannot: an external reference, a version pin, a
7
+ number with a unit, a causal connective, a negation of the obvious, an
8
+ upstream-quirk word, a concurrency/invariant term, security reasoning, or a
9
+ vendor proper noun with *ascribed behaviour*. Measured over a 37,918-comment
10
+ corpus from nine repos: the nine signals protect **40/40** hand-picked best
11
+ comments and leak **~1%** of the hand-classified cruft list.
12
+
13
+ The class is an **EXEMPTION FLOOR, never a test**. `is_protected(body)` being
14
+ False says nothing at all about a comment — over pydantic / trio / attrs it
15
+ matches only 18-35% of comments a human called valuable. Every use here is of
16
+ the form "if protected, do not flag"; inverting it into "unprotected, so
17
+ delete" would flag two thirds of the best comments in Python's most carefully
18
+ commented libraries. If a future rule wants a *positive* test for value, it
19
+ needs its own measurement, not this.
20
+
21
+ **One tokenize pass per file.** `standalone_comments()` mirrors
22
+ `rule_base.parse_or_none`: a single-slot memo keyed on the source object, so
23
+ the four comment rules that all need "every comment that is alone on its line"
24
+ tokenize each file once between them rather than once each.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import io
30
+ import re
31
+ import tokenize
32
+ from typing import TYPE_CHECKING
33
+
34
+
35
+ if TYPE_CHECKING:
36
+ from collections.abc import Iterable, Sequence
37
+
38
+
39
+ # S1 — external reference: URL, issue/ticket key, RFC/PEP/CVE, bare GitHub issue
40
+ # number, or an email/handle domain. Ticket keys allow letters after the first
41
+ # digit (`PLATFORM-1YC`) and exclude the encoding/algorithm acronyms that share
42
+ # the shape (`UTF-8`, `SHA-256`, `ISO-8601`, `AES-256`).
43
+ _REF_RE = re.compile(
44
+ r"https?://|\bRFC[- ]?\d+|\bPEP[- ]?\d+|\bCVE-\d{4}|"
45
+ r"\b(?!UTF-|SHA-|ISO-|AES-|CRC-|MD-|PCM-|EOF-|API-|BASE-)[A-Z][A-Z0-9]{1,9}-\d[A-Z0-9]{0,5}\b|"
46
+ r"(?<![&\w])#\d{2,6}\b|"
47
+ r"@[a-z][\w.-]*\.(?:us|com|ai|io|net|org|dev)\b",
48
+ )
49
+
50
+ # S2 — version pin or comparison: ">= 0.137", "v5.0", "since Python 3.11".
51
+ _VERSION_RE = re.compile(
52
+ r"(?:>=|<=|==|<|>)\s*v?\d+\.\d+|\bv\d+\.\d+|\b(?:since|until|as of)\s+(?:v?\d+\.\d+|Python\s*\d)",
53
+ re.IGNORECASE,
54
+ )
55
+
56
+ # S3 — a number carrying a unit (time, size, rate, audio, percent) or an HTTP
57
+ # status code. `429` and `250 ms` are facts about the world, not about the code.
58
+ _UNITS_RE = re.compile(
59
+ r"[~<>]?\d+(?:\.\d+)?\s?(?:ms|s\b|sec\b|seconds?\b|min\b|minutes?\b|hours?\b|days?\b|"
60
+ r"KB|MB|MiB|GiB|kHz|Hz|bytes?\b|bit\b|-bit\b|%|px\b|rps\b|qps\b)|"
61
+ r"\b[1-5]xx\b|\b(?:301|302|304|307|308|400|401|403|404|405|409|410|412|422|425|429|500|501|502|503|504)\b",
62
+ )
63
+
64
+ # S4 — a causal connective tying behaviour to a consequence. This is the shape
65
+ # of a *why*: the comment says what breaks if the code changes.
66
+ _CAUSAL_RE = re.compile(
67
+ r"\b(?:because|otherwise|so that|or else|would (?:break|fail|race|deadlock|leak|clobber|loop|crash|page|stall)|"
68
+ r"breaks?\b|so we don'?t|to avoid\b|caused\b|causes\b|gets? clobbered|"
69
+ r"keeps? (?:us|it|them) from|doesn'?t\b.{0,24}\b(?:page|fire|break|leak|loop)|"
70
+ r"eat into|would otherwise|trade-?offs?\b)\b",
71
+ re.IGNORECASE,
72
+ )
73
+
74
+ # S5 — negation of the obvious, or a flagged deliberate deviation. "NOT a typo",
75
+ # "deliberately re-raises", "instead of the documented order".
76
+ _NEGATION_RE = re.compile(
77
+ r"\b(?:must not|must never|do(?:es)? not\b|don'?t\b.{0,30}\b(?:leak|log|cache|retry|block|steal|wipe)|"
78
+ r"never\b|deliberately|intentionally|counterintuitiv|NOT\b)|(?<!based )\bon purpose\b|"
79
+ r"\(not\s|\binstead of\b|\brather than\b",
80
+ )
81
+
82
+ # S6 — upstream/vendor quirk, workaround provenance, or an external contract.
83
+ _UPSTREAM_RE = re.compile(
84
+ r"\b(?:upstream|workaround|quirk|backport|vendored|regression|fixed upstream|"
85
+ r"requires?\b|convention\b|rate.?limit|deprecat|opts? in(?:to)?\b|"
86
+ r"raises?\b.{0,60}\b(?:when|if|unless)\b)",
87
+ re.IGNORECASE,
88
+ )
89
+
90
+ # S7 — concurrency, ordering, or invariant vocabulary. Nothing in the code text
91
+ # can state "this must run before the lock is taken".
92
+ _INVARIANT_RE = re.compile(
93
+ r"\b(?:invariant|idempotent|race\b|deadlock|re-?entran|atomic|thread-?safe|signal-?safe|"
94
+ r"lexicographic(?:al(?:ly)?)?|monotonic|must (?:run|be|happen|come|stay|hit|converge|configure)|"
95
+ r"before any\b|lost the (?:claim )?race)\b",
96
+ re.IGNORECASE,
97
+ )
98
+
99
+ # S8 — security reasoning.
100
+ _SECURITY_RE = re.compile(
101
+ r"\b(?:timing attack|constant-?time|replay|PII\b|redact|secret|injection|spoof|"
102
+ r"fail-?closed|fail-?open|auth bypass|early-?exit timing)\b",
103
+ re.IGNORECASE,
104
+ )
105
+
106
+ # S9 — a vendor proper noun with *ascribed behaviour* (possessive, or followed by
107
+ # a behavioural verb). A vendor name as the mere object of a narration verb
108
+ # ("Create the prompt for Gemini") carries nothing and is deliberately NOT
109
+ # protected — that distinction is what keeps the leak rate at ~1%.
110
+ _VENDOR_RE = re.compile(
111
+ r"\b(?:GitHub|Slack|Twilio|LiveKit|Kamailio|Groq|OpenAI|Anthropic|Cloudflare|FastAPI|"
112
+ r"Starlette|Sentry|Zoho|Salla|Ashby|Linear|BigQuery|Postgres|Neon|Drizzle|Vertex|Gemini|"
113
+ r"Firestore|Stripe|Next\.js|React Compiler|pydantic|ruff|loguru|Lexical|Farasa|Orpheus|"
114
+ r"Whisper|schemathesis)"
115
+ r"(?:'s\b|\s+(?:requires?|returns?|expects?|allows?|rejects?|accepts?|sends?|caps?|"
116
+ r"limits?|wraps?|silently|outputs?|stores?|treats?|doesn'?t|does not|won'?t|can'?t|"
117
+ r"only|models)\b)",
118
+ )
119
+
120
+ _SIGNALS: dict[str, re.Pattern[str]] = {
121
+ "ref": _REF_RE,
122
+ "version": _VERSION_RE,
123
+ "units": _UNITS_RE,
124
+ "causal": _CAUSAL_RE,
125
+ "negation": _NEGATION_RE,
126
+ "upstream": _UPSTREAM_RE,
127
+ "invariant": _INVARIANT_RE,
128
+ "security": _SECURITY_RE,
129
+ "vendor": _VENDOR_RE,
130
+ }
131
+
132
+
133
+ def protecting_signals(body: str) -> frozenset[str]:
134
+ """Name every protected-class signal that matches `body`.
135
+
136
+ Returns:
137
+ The set of signal names; empty when nothing protects the comment.
138
+
139
+ """
140
+ return frozenset(name for name, pattern in _SIGNALS.items() if pattern.search(body))
141
+
142
+
143
+ def is_protected(body: str) -> bool:
144
+ """Report whether a comment carries any protected-class signal.
145
+
146
+ EXEMPTION FLOOR ONLY — see the module docstring. A False result is not
147
+ evidence that the comment is worthless.
148
+
149
+ Returns:
150
+ True when at least one of the nine signals matches.
151
+
152
+ """
153
+ return any(pattern.search(body) for pattern in _SIGNALS.values())
154
+
155
+
156
+ def has_external_reference(body: str) -> bool:
157
+ """Report whether a comment cites a ticket, URL, RFC/PEP/CVE, or issue number.
158
+
159
+ Signal S1 on its own. A comment that names where the decision is recorded is
160
+ doing the one thing the code cannot, and it is the signal that separates a
161
+ scoping note with an owner ("EN-only for now — AR needs audio (PROD-249)")
162
+ from an unowned admission ("hacky, fix later").
163
+
164
+ Returns:
165
+ True when the comment carries an external reference.
166
+
167
+ """
168
+ return bool(_REF_RE.search(body))
169
+
170
+
171
+ # --- tokenisation shared by the restatement detectors ----------------------
172
+
173
+ # Below this length an inflection strip would eat the word itself.
174
+ _MIN_STEM_LENGTH = 3
175
+
176
+ _WORD_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*|\d+")
177
+ _CAMEL_RE = re.compile(r"[A-Z]+(?=[A-Z][a-z])|[A-Z]?[a-z]+|[A-Z]+|\d+")
178
+
179
+ # Words that say nothing about *which* code a comment describes. Kept close to
180
+ # the prototype's list: shrinking it costs recall, growing it costs precision by
181
+ # letting a genuinely novel word be discounted.
182
+ STOPWORDS: frozenset[str] = frozenset(
183
+ ["a", "an", "the", "this", "that", "these", "those", "it", "its", "their", "his", "her", "our", "your", "my", "is", "are", "was", "were", "be", "been", "being", "am", "do", "does", "did", "done", "doing", "has", "have", "had", "having", "will", "would", "shall", "should", "can", "could", "may", "might", "must", "and", "or", "but", "nor", "so", "yet", "not", "no", "none", "to", "of", "for", "in", "on", "at", "by", "with", "from", "into", "onto", "out", "up", "down", "over", "under", "about", "as", "if", "then", "than", "when", "where", "which", "who", "whom", "whose", "what", "how", "why", "while", "we", "you", "they", "i", "he", "she", "them", "him", "us", "me", "also", "just", "only", "even", "still", "already", "again", "there", "here", "all", "any", "each", "every", "some", "via", "per", "etc", "eg", "ie", "vs", "need", "needs", "needed", "want", "wants", "make", "makes", "making", "let", "lets", "please", "note", "see", "above", "below"]
184
+ )
185
+
186
+
187
+ def split_identifier(token: str) -> list[str]:
188
+ """Split `snake_case` / `camelCase` / `SCREAMING_CASE` into lowercase parts.
189
+
190
+ Returns:
191
+ The identifier's word parts, lowercased.
192
+
193
+ """
194
+ parts: list[str] = []
195
+ for chunk in token.split("_"):
196
+ parts.extend(match.group(0).lower() for match in _CAMEL_RE.finditer(chunk))
197
+ return [part for part in parts if part]
198
+
199
+
200
+ def stem(word: str) -> str:
201
+ """Fold the common English inflections so `updates`/`updating` match `update`.
202
+
203
+ The trailing-`e` strip is what makes the fold *symmetric*: without it
204
+ `creates`/`creating` reduce to `creat` while `create` stays `create`, and the
205
+ two never match — the shape that most often made a restatement look novel.
206
+
207
+ Deliberately crude otherwise. A real stemmer would conflate more pairs, and
208
+ every extra conflation is a chance to call a novel word a restatement.
209
+
210
+ Returns:
211
+ The stemmed word.
212
+
213
+ """
214
+ base = word
215
+ for suffix in ("ing", "ied", "ies", "ers", "er", "ed", "es", "s"):
216
+ if word.endswith(suffix) and len(word) - len(suffix) >= _MIN_STEM_LENGTH:
217
+ base = word[: len(word) - len(suffix)]
218
+ if suffix in {"ied", "ies"}:
219
+ return base + "y"
220
+ break
221
+ if base.endswith("e") and len(base) - 1 >= _MIN_STEM_LENGTH:
222
+ return base[:-1]
223
+ return base
224
+
225
+
226
+ def content_tokens(text: str) -> list[str]:
227
+ """Split prose into lowercase content words, dropping stopwords.
228
+
229
+ Returns:
230
+ The comment's content tokens, in order.
231
+
232
+ """
233
+ tokens: list[str] = []
234
+ for match in _WORD_RE.finditer(text):
235
+ tokens.extend(split_identifier(match.group(0)))
236
+ return [token for token in tokens if token not in STOPWORDS]
237
+
238
+
239
+ def code_tokens(text: str) -> set[str]:
240
+ """Collect every identifier part appearing in a slice of source.
241
+
242
+ Returns:
243
+ The lowercase identifier parts, as a set.
244
+
245
+ """
246
+ tokens: set[str] = set()
247
+ for match in _WORD_RE.finditer(text):
248
+ tokens.update(split_identifier(match.group(0)))
249
+ return tokens
250
+
251
+
252
+ def restates(comment_tokens: Sequence[str], code: Iterable[str]) -> bool:
253
+ """Report whether every content token of a comment already appears in the code.
254
+
255
+ Exact or stemmed match only. Prefix matching is deliberately absent: it is
256
+ what sank the first attempt at this shape (PR #98), where `service` matched
257
+ `locationService` and drove the false-positive rate to ~60%.
258
+
259
+ Returns:
260
+ True when the comment adds no token the code does not already carry.
261
+
262
+ """
263
+ present = set(code)
264
+ stems = {stem(token) for token in present}
265
+ return all(token in present or stem(token) in stems for token in comment_tokens)
266
+
267
+
268
+ # --- one tokenize pass per file --------------------------------------------
269
+
270
+ _LAYOUT_TOKENS = frozenset({tokenize.NL, tokenize.NEWLINE, tokenize.INDENT, tokenize.DEDENT})
271
+ _NON_CODE_TOKENS = _LAYOUT_TOKENS | frozenset({tokenize.COMMENT, tokenize.ENCODING, tokenize.ENDMARKER})
272
+
273
+ _Scan = tuple[list[tuple[int, int, str]], list[tuple[int, int, str]], set[int], int]
274
+
275
+ _last_scan: tuple[str, _Scan] | None = None
276
+
277
+
278
+ def _scan(source: str) -> _Scan:
279
+ standalone: list[tuple[int, int, str]] = []
280
+ trailing: list[tuple[int, int, str]] = []
281
+ nested: set[int] = set()
282
+ first_code_line = 1 << 30
283
+ prev_end_row = 0
284
+ depth = 0
285
+ readline = io.StringIO(source).readline
286
+ for tok in tokenize.generate_tokens(readline):
287
+ if tok.type == tokenize.COMMENT:
288
+ entry = (tok.start[0], tok.start[1], tok.string.lstrip("#").strip())
289
+ (trailing if tok.start[0] == prev_end_row else standalone).append(entry)
290
+ if depth > 0:
291
+ nested.add(tok.start[0])
292
+ elif tok.type == tokenize.OP:
293
+ if tok.string in {"(", "[", "{"}:
294
+ depth += 1
295
+ elif tok.string in {")", "]", "}"}:
296
+ depth = max(0, depth - 1)
297
+ if tok.type not in _LAYOUT_TOKENS:
298
+ prev_end_row = tok.end[0]
299
+ if tok.type not in _NON_CODE_TOKENS:
300
+ first_code_line = min(first_code_line, tok.start[0])
301
+ return standalone, trailing, nested, first_code_line
302
+
303
+
304
+ def _scan_memo(source: str) -> _Scan:
305
+ global _last_scan # ruff: ignore[global-statement] — single-slot memo; the CLI runs rules per file sequentially
306
+ if _last_scan is not None and _last_scan[0] is source:
307
+ return _last_scan[1]
308
+ result = _scan(source)
309
+ _last_scan = (source, result)
310
+ return result
311
+
312
+
313
+ def trailing_comments(source: str) -> list[tuple[int, int, str]]:
314
+ """Return every comment that shares its line with code, as `(line, col, body)`.
315
+
316
+ Raises out of here when `source` cannot be tokenized; see
317
+ `standalone_comments`.
318
+
319
+ Returns:
320
+ The trailing comments, in source order.
321
+
322
+ """
323
+ return _scan_memo(source)[1]
324
+
325
+
326
+ def nested_comment_lines(source: str) -> set[int]:
327
+ """Return the lines of comments sitting INSIDE a bracketed expression.
328
+
329
+ A comment at bracket depth > 0 is annotating an element of a list, dict or
330
+ call — `# config` inside pydantic's `__all__` groups the names beneath it —
331
+ rather than signposting the structure of the file. Both readings produce the
332
+ same one-word comment, and only the depth tells them apart.
333
+
334
+ Returns:
335
+ The line numbers of comments nested inside brackets.
336
+
337
+ """
338
+ return _scan_memo(source)[2]
339
+
340
+
341
+ def standalone_comments(source: str) -> tuple[list[tuple[int, int, str]], int]:
342
+ """Return every own-line comment as `(line, col, body)`, plus the first code line.
343
+
344
+ A comment is standalone when it is the only content on its line; `first code
345
+ line` is the row of the first real code token (a large sentinel when the file
346
+ has none). Memoized on the source *object* so the comment rules share one
347
+ tokenize pass per file, as `rule_base.parse_or_none` does for the AST.
348
+
349
+ A file the tokenizer rejects raises out of here rather than being silently
350
+ treated as comment-free; every caller catches that and returns no
351
+ diagnostics, because a rule has nothing useful to say about a file that does
352
+ not parse.
353
+
354
+ Returns:
355
+ The standalone comments and the first code line's row.
356
+
357
+ """
358
+ standalone, _, _, first_code_line = _scan_memo(source)
359
+ return standalone, first_code_line
360
+
361
+
362
+ def comment_runs(standalone: Sequence[tuple[int, int, str]]) -> list[list[tuple[int, int, str]]]:
363
+ """Group standalone comments into runs of consecutive lines.
364
+
365
+ Returns:
366
+ One list per contiguous `#` block, in source order.
367
+
368
+ """
369
+ runs: list[list[tuple[int, int, str]]] = []
370
+ for entry in sorted(standalone):
371
+ if runs and entry[0] == runs[-1][-1][0] + 1:
372
+ runs[-1].append(entry)
373
+ else:
374
+ runs.append([entry])
375
+ return runs
@@ -27,6 +27,7 @@ from sarj_python_lint.rules.no_offset_pagination import NoOffsetPagination
27
27
  from sarj_python_lint.rules.no_query_with_many_joins import NoQueryWithManyJoins
28
28
  from sarj_python_lint.rules.no_raw_sql_in_tests import NoRawSqlInTests
29
29
  from sarj_python_lint.rules.no_repeated_string_literal import NoRepeatedStringLiteral
30
+ from sarj_python_lint.rules.no_restated_comment import NoRestatedComment
30
31
  from sarj_python_lint.rules.no_secret_in_log import NoSecretInLog
31
32
  from sarj_python_lint.rules.no_select_star import NoSelectStar
32
33
  from sarj_python_lint.rules.no_sentinel_return_on_except import NoSentinelReturnOnExcept
@@ -55,6 +56,7 @@ from sarj_python_lint.rules.prefer_timedelta_for_durations import (
55
56
  PreferTimedeltaForDurations,
56
57
  )
57
58
  from sarj_python_lint.rules.pydantic_at_boundaries import PydanticAtBoundaries
59
+ from sarj_python_lint.rules.redundant_docstring import RedundantDocstring
58
60
  from sarj_python_lint.rules.single_public_export import SinglePublicExport
59
61
  from sarj_python_lint.rules.sleep_with_computed_arg_in_test import SleepWithComputedArgInTest
60
62
  from sarj_python_lint.rules.stepdown import Stepdown
@@ -64,6 +66,7 @@ from sarj_python_lint.rules.store_insert_requires_on_conflict import (
64
66
  from sarj_python_lint.rules.test_loops_over_literal_cases import (
65
67
  TestLoopsOverLiteralCases,
66
68
  )
69
+ from sarj_python_lint.rules.trailing_value_narration import TrailingValueNarration
67
70
  from sarj_python_lint.rules.xfail_requires_strict import XfailRequiresStrict
68
71
  from sarj_python_lint.rules.zero_assertion_test import ZeroAssertionTest
69
72
 
@@ -120,6 +123,9 @@ REGISTRY: dict[str, type[Rule]] = {
120
123
  SleepWithComputedArgInTest.id: SleepWithComputedArgInTest,
121
124
  ZeroAssertionTest.id: ZeroAssertionTest,
122
125
  NoFirstPartyPrivateImport.id: NoFirstPartyPrivateImport,
126
+ NoRestatedComment.id: NoRestatedComment,
127
+ RedundantDocstring.id: RedundantDocstring,
128
+ TrailingValueNarration.id: TrailingValueNarration,
123
129
  }
124
130
 
125
131
  __all__ = ["REGISTRY"]