sarj-python-lint 0.19.0__tar.gz → 0.20.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/PKG-INFO +17 -1
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/README.md +16 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/pyproject.toml +1 -1
- sarj_python_lint-0.20.0/src/sarj_python_lint/rules/_comments.py +375 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_registry.py +6 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_comment_cruft.py +187 -41
- sarj_python_lint-0.20.0/src/sarj_python_lint/rules/no_restated_comment.py +318 -0
- sarj_python_lint-0.20.0/src/sarj_python_lint/rules/redundant_docstring.py +211 -0
- sarj_python_lint-0.20.0/src/sarj_python_lint/rules/trailing_value_narration.py +134 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/.gitignore +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/__init__.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/__main__.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/_secret_names.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/_version.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/py.typed +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rule_base.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/__init__.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_first_party.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_logging.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_paths.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_sql.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/fixture_returns_bare_tuple.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/inefficient_string_concat_in_loop.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/kwarg_heavy_construction_in_test.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/kwonly_same_type_params.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/mock_without_spec.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_aggregation_in_store_query.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_cors_wildcard_with_credentials.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_fat_try_blocks.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_file_level_suppression.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_first_party_private_import.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_fstring_in_log.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_isinstance_union_chain.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_offset_pagination.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_query_with_many_joins.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_raw_sql_in_tests.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_repeated_string_literal.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_secret_in_log.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_select_star.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_sentinel_return_on_except.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_sequential_await.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_sleep_in_test_body.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_unreachable_after_terminal.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/parametrize_case_needs_id.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_class_row.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_constant_time_secret_compare.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_match_assert_never.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_module_level_constant.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_namedtuple_over_tuple_return.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_str_enum.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_struct_over_namedtuple.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_timedelta_for_durations.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/pydantic_at_boundaries.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/single_public_export.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/sleep_with_computed_arg_in_test.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/stepdown.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/store_insert_requires_on_conflict.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/test_loops_over_literal_cases.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/xfail_requires_strict.py +0 -0
- {sarj_python_lint-0.19.0 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/zero_assertion_test.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: sarj-python-lint
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.20.0
|
|
4
4
|
Summary: Custom Python lint rules — AST-based, pre-commit-friendly, hypermodern defaults
|
|
5
5
|
Project-URL: Homepage, https://github.com/sarj-ai/standards/tree/main/packages/python
|
|
6
6
|
Project-URL: Repository, https://github.com/sarj-ai/standards
|
|
@@ -85,6 +85,22 @@ Attribute access (`session._stt`) is out of scope and stays with ruff's
|
|
|
85
85
|
`SLF001`, which cannot make the distinction either — see the rationale in
|
|
86
86
|
`ruff.strict.toml`.
|
|
87
87
|
|
|
88
|
+
### Comment-hygiene rules (0.20.0)
|
|
89
|
+
|
|
90
|
+
From a 37,918-comment, nine-repo measurement study. All three are
|
|
91
|
+
deletion-class, so each was validated against pydantic / trio / attrs as well as
|
|
92
|
+
the maintained repos before shipping — the counts and the false-positive classes
|
|
93
|
+
each guard was built from are recorded in the rule module docstrings.
|
|
94
|
+
|
|
95
|
+
```yaml
|
|
96
|
+
- id: sarj-no-restated-comment # SARJ049
|
|
97
|
+
- id: sarj-redundant-docstring # SARJ050
|
|
98
|
+
- id: sarj-trailing-value-narration # SARJ051
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
`redundant-docstring` finds real volume on a codebase that has never had it
|
|
102
|
+
(105 in noura-be), so the same baseline ratchet applies.
|
|
103
|
+
|
|
88
104
|
Adopting these against an existing suite is easier through the baseline ratchet
|
|
89
105
|
than as a big-bang fix — snapshot the current counts, then let them only shrink:
|
|
90
106
|
|
|
@@ -67,6 +67,22 @@ Attribute access (`session._stt`) is out of scope and stays with ruff's
|
|
|
67
67
|
`SLF001`, which cannot make the distinction either — see the rationale in
|
|
68
68
|
`ruff.strict.toml`.
|
|
69
69
|
|
|
70
|
+
### Comment-hygiene rules (0.20.0)
|
|
71
|
+
|
|
72
|
+
From a 37,918-comment, nine-repo measurement study. All three are
|
|
73
|
+
deletion-class, so each was validated against pydantic / trio / attrs as well as
|
|
74
|
+
the maintained repos before shipping — the counts and the false-positive classes
|
|
75
|
+
each guard was built from are recorded in the rule module docstrings.
|
|
76
|
+
|
|
77
|
+
```yaml
|
|
78
|
+
- id: sarj-no-restated-comment # SARJ049
|
|
79
|
+
- id: sarj-redundant-docstring # SARJ050
|
|
80
|
+
- id: sarj-trailing-value-narration # SARJ051
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
`redundant-docstring` finds real volume on a codebase that has never had it
|
|
84
|
+
(105 in noura-be), so the same baseline ratchet applies.
|
|
85
|
+
|
|
70
86
|
Adopting these against an existing suite is easier through the baseline ratchet
|
|
71
87
|
than as a big-bang fix — snapshot the current counts, then let them only shrink:
|
|
72
88
|
|
|
@@ -0,0 +1,375 @@
|
|
|
1
|
+
"""Shared comment analysis for the comment-hygiene rules (SARJ016/049/050/051).
|
|
2
|
+
|
|
3
|
+
Two things live here, both needed by more than one rule.
|
|
4
|
+
|
|
5
|
+
**The protected class.** Nine deterministic signals that mark a comment as
|
|
6
|
+
carrying something the code cannot: an external reference, a version pin, a
|
|
7
|
+
number with a unit, a causal connective, a negation of the obvious, an
|
|
8
|
+
upstream-quirk word, a concurrency/invariant term, security reasoning, or a
|
|
9
|
+
vendor proper noun with *ascribed behaviour*. Measured over a 37,918-comment
|
|
10
|
+
corpus from nine repos: the nine signals protect **40/40** hand-picked best
|
|
11
|
+
comments and leak **~1%** of the hand-classified cruft list.
|
|
12
|
+
|
|
13
|
+
The class is an **EXEMPTION FLOOR, never a test**. `is_protected(body)` being
|
|
14
|
+
False says nothing at all about a comment — over pydantic / trio / attrs it
|
|
15
|
+
matches only 18-35% of comments a human called valuable. Every use here is of
|
|
16
|
+
the form "if protected, do not flag"; inverting it into "unprotected, so
|
|
17
|
+
delete" would flag two thirds of the best comments in Python's most carefully
|
|
18
|
+
commented libraries. If a future rule wants a *positive* test for value, it
|
|
19
|
+
needs its own measurement, not this.
|
|
20
|
+
|
|
21
|
+
**One tokenize pass per file.** `standalone_comments()` mirrors
|
|
22
|
+
`rule_base.parse_or_none`: a single-slot memo keyed on the source object, so
|
|
23
|
+
the four comment rules that all need "every comment that is alone on its line"
|
|
24
|
+
tokenize each file once between them rather than once each.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import io
|
|
30
|
+
import re
|
|
31
|
+
import tokenize
|
|
32
|
+
from typing import TYPE_CHECKING
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
if TYPE_CHECKING:
|
|
36
|
+
from collections.abc import Iterable, Sequence
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
# S1 — external reference: URL, issue/ticket key, RFC/PEP/CVE, bare GitHub issue
|
|
40
|
+
# number, or an email/handle domain. Ticket keys allow letters after the first
|
|
41
|
+
# digit (`PLATFORM-1YC`) and exclude the encoding/algorithm acronyms that share
|
|
42
|
+
# the shape (`UTF-8`, `SHA-256`, `ISO-8601`, `AES-256`).
|
|
43
|
+
_REF_RE = re.compile(
|
|
44
|
+
r"https?://|\bRFC[- ]?\d+|\bPEP[- ]?\d+|\bCVE-\d{4}|"
|
|
45
|
+
r"\b(?!UTF-|SHA-|ISO-|AES-|CRC-|MD-|PCM-|EOF-|API-|BASE-)[A-Z][A-Z0-9]{1,9}-\d[A-Z0-9]{0,5}\b|"
|
|
46
|
+
r"(?<![&\w])#\d{2,6}\b|"
|
|
47
|
+
r"@[a-z][\w.-]*\.(?:us|com|ai|io|net|org|dev)\b",
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
# S2 — version pin or comparison: ">= 0.137", "v5.0", "since Python 3.11".
|
|
51
|
+
_VERSION_RE = re.compile(
|
|
52
|
+
r"(?:>=|<=|==|<|>)\s*v?\d+\.\d+|\bv\d+\.\d+|\b(?:since|until|as of)\s+(?:v?\d+\.\d+|Python\s*\d)",
|
|
53
|
+
re.IGNORECASE,
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
# S3 — a number carrying a unit (time, size, rate, audio, percent) or an HTTP
|
|
57
|
+
# status code. `429` and `250 ms` are facts about the world, not about the code.
|
|
58
|
+
_UNITS_RE = re.compile(
|
|
59
|
+
r"[~<>]?\d+(?:\.\d+)?\s?(?:ms|s\b|sec\b|seconds?\b|min\b|minutes?\b|hours?\b|days?\b|"
|
|
60
|
+
r"KB|MB|MiB|GiB|kHz|Hz|bytes?\b|bit\b|-bit\b|%|px\b|rps\b|qps\b)|"
|
|
61
|
+
r"\b[1-5]xx\b|\b(?:301|302|304|307|308|400|401|403|404|405|409|410|412|422|425|429|500|501|502|503|504)\b",
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
# S4 — a causal connective tying behaviour to a consequence. This is the shape
|
|
65
|
+
# of a *why*: the comment says what breaks if the code changes.
|
|
66
|
+
_CAUSAL_RE = re.compile(
|
|
67
|
+
r"\b(?:because|otherwise|so that|or else|would (?:break|fail|race|deadlock|leak|clobber|loop|crash|page|stall)|"
|
|
68
|
+
r"breaks?\b|so we don'?t|to avoid\b|caused\b|causes\b|gets? clobbered|"
|
|
69
|
+
r"keeps? (?:us|it|them) from|doesn'?t\b.{0,24}\b(?:page|fire|break|leak|loop)|"
|
|
70
|
+
r"eat into|would otherwise|trade-?offs?\b)\b",
|
|
71
|
+
re.IGNORECASE,
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
# S5 — negation of the obvious, or a flagged deliberate deviation. "NOT a typo",
|
|
75
|
+
# "deliberately re-raises", "instead of the documented order".
|
|
76
|
+
_NEGATION_RE = re.compile(
|
|
77
|
+
r"\b(?:must not|must never|do(?:es)? not\b|don'?t\b.{0,30}\b(?:leak|log|cache|retry|block|steal|wipe)|"
|
|
78
|
+
r"never\b|deliberately|intentionally|counterintuitiv|NOT\b)|(?<!based )\bon purpose\b|"
|
|
79
|
+
r"\(not\s|\binstead of\b|\brather than\b",
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
# S6 — upstream/vendor quirk, workaround provenance, or an external contract.
|
|
83
|
+
_UPSTREAM_RE = re.compile(
|
|
84
|
+
r"\b(?:upstream|workaround|quirk|backport|vendored|regression|fixed upstream|"
|
|
85
|
+
r"requires?\b|convention\b|rate.?limit|deprecat|opts? in(?:to)?\b|"
|
|
86
|
+
r"raises?\b.{0,60}\b(?:when|if|unless)\b)",
|
|
87
|
+
re.IGNORECASE,
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
# S7 — concurrency, ordering, or invariant vocabulary. Nothing in the code text
|
|
91
|
+
# can state "this must run before the lock is taken".
|
|
92
|
+
_INVARIANT_RE = re.compile(
|
|
93
|
+
r"\b(?:invariant|idempotent|race\b|deadlock|re-?entran|atomic|thread-?safe|signal-?safe|"
|
|
94
|
+
r"lexicographic(?:al(?:ly)?)?|monotonic|must (?:run|be|happen|come|stay|hit|converge|configure)|"
|
|
95
|
+
r"before any\b|lost the (?:claim )?race)\b",
|
|
96
|
+
re.IGNORECASE,
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
# S8 — security reasoning.
|
|
100
|
+
_SECURITY_RE = re.compile(
|
|
101
|
+
r"\b(?:timing attack|constant-?time|replay|PII\b|redact|secret|injection|spoof|"
|
|
102
|
+
r"fail-?closed|fail-?open|auth bypass|early-?exit timing)\b",
|
|
103
|
+
re.IGNORECASE,
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
# S9 — a vendor proper noun with *ascribed behaviour* (possessive, or followed by
|
|
107
|
+
# a behavioural verb). A vendor name as the mere object of a narration verb
|
|
108
|
+
# ("Create the prompt for Gemini") carries nothing and is deliberately NOT
|
|
109
|
+
# protected — that distinction is what keeps the leak rate at ~1%.
|
|
110
|
+
_VENDOR_RE = re.compile(
|
|
111
|
+
r"\b(?:GitHub|Slack|Twilio|LiveKit|Kamailio|Groq|OpenAI|Anthropic|Cloudflare|FastAPI|"
|
|
112
|
+
r"Starlette|Sentry|Zoho|Salla|Ashby|Linear|BigQuery|Postgres|Neon|Drizzle|Vertex|Gemini|"
|
|
113
|
+
r"Firestore|Stripe|Next\.js|React Compiler|pydantic|ruff|loguru|Lexical|Farasa|Orpheus|"
|
|
114
|
+
r"Whisper|schemathesis)"
|
|
115
|
+
r"(?:'s\b|\s+(?:requires?|returns?|expects?|allows?|rejects?|accepts?|sends?|caps?|"
|
|
116
|
+
r"limits?|wraps?|silently|outputs?|stores?|treats?|doesn'?t|does not|won'?t|can'?t|"
|
|
117
|
+
r"only|models)\b)",
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
_SIGNALS: dict[str, re.Pattern[str]] = {
|
|
121
|
+
"ref": _REF_RE,
|
|
122
|
+
"version": _VERSION_RE,
|
|
123
|
+
"units": _UNITS_RE,
|
|
124
|
+
"causal": _CAUSAL_RE,
|
|
125
|
+
"negation": _NEGATION_RE,
|
|
126
|
+
"upstream": _UPSTREAM_RE,
|
|
127
|
+
"invariant": _INVARIANT_RE,
|
|
128
|
+
"security": _SECURITY_RE,
|
|
129
|
+
"vendor": _VENDOR_RE,
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def protecting_signals(body: str) -> frozenset[str]:
|
|
134
|
+
"""Name every protected-class signal that matches `body`.
|
|
135
|
+
|
|
136
|
+
Returns:
|
|
137
|
+
The set of signal names; empty when nothing protects the comment.
|
|
138
|
+
|
|
139
|
+
"""
|
|
140
|
+
return frozenset(name for name, pattern in _SIGNALS.items() if pattern.search(body))
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def is_protected(body: str) -> bool:
|
|
144
|
+
"""Report whether a comment carries any protected-class signal.
|
|
145
|
+
|
|
146
|
+
EXEMPTION FLOOR ONLY — see the module docstring. A False result is not
|
|
147
|
+
evidence that the comment is worthless.
|
|
148
|
+
|
|
149
|
+
Returns:
|
|
150
|
+
True when at least one of the nine signals matches.
|
|
151
|
+
|
|
152
|
+
"""
|
|
153
|
+
return any(pattern.search(body) for pattern in _SIGNALS.values())
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def has_external_reference(body: str) -> bool:
|
|
157
|
+
"""Report whether a comment cites a ticket, URL, RFC/PEP/CVE, or issue number.
|
|
158
|
+
|
|
159
|
+
Signal S1 on its own. A comment that names where the decision is recorded is
|
|
160
|
+
doing the one thing the code cannot, and it is the signal that separates a
|
|
161
|
+
scoping note with an owner ("EN-only for now — AR needs audio (PROD-249)")
|
|
162
|
+
from an unowned admission ("hacky, fix later").
|
|
163
|
+
|
|
164
|
+
Returns:
|
|
165
|
+
True when the comment carries an external reference.
|
|
166
|
+
|
|
167
|
+
"""
|
|
168
|
+
return bool(_REF_RE.search(body))
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
# --- tokenisation shared by the restatement detectors ----------------------
|
|
172
|
+
|
|
173
|
+
# Below this length an inflection strip would eat the word itself.
|
|
174
|
+
_MIN_STEM_LENGTH = 3
|
|
175
|
+
|
|
176
|
+
_WORD_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*|\d+")
|
|
177
|
+
_CAMEL_RE = re.compile(r"[A-Z]+(?=[A-Z][a-z])|[A-Z]?[a-z]+|[A-Z]+|\d+")
|
|
178
|
+
|
|
179
|
+
# Words that say nothing about *which* code a comment describes. Kept close to
|
|
180
|
+
# the prototype's list: shrinking it costs recall, growing it costs precision by
|
|
181
|
+
# letting a genuinely novel word be discounted.
|
|
182
|
+
STOPWORDS: frozenset[str] = frozenset(
|
|
183
|
+
["a", "an", "the", "this", "that", "these", "those", "it", "its", "their", "his", "her", "our", "your", "my", "is", "are", "was", "were", "be", "been", "being", "am", "do", "does", "did", "done", "doing", "has", "have", "had", "having", "will", "would", "shall", "should", "can", "could", "may", "might", "must", "and", "or", "but", "nor", "so", "yet", "not", "no", "none", "to", "of", "for", "in", "on", "at", "by", "with", "from", "into", "onto", "out", "up", "down", "over", "under", "about", "as", "if", "then", "than", "when", "where", "which", "who", "whom", "whose", "what", "how", "why", "while", "we", "you", "they", "i", "he", "she", "them", "him", "us", "me", "also", "just", "only", "even", "still", "already", "again", "there", "here", "all", "any", "each", "every", "some", "via", "per", "etc", "eg", "ie", "vs", "need", "needs", "needed", "want", "wants", "make", "makes", "making", "let", "lets", "please", "note", "see", "above", "below"]
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def split_identifier(token: str) -> list[str]:
|
|
188
|
+
"""Split `snake_case` / `camelCase` / `SCREAMING_CASE` into lowercase parts.
|
|
189
|
+
|
|
190
|
+
Returns:
|
|
191
|
+
The identifier's word parts, lowercased.
|
|
192
|
+
|
|
193
|
+
"""
|
|
194
|
+
parts: list[str] = []
|
|
195
|
+
for chunk in token.split("_"):
|
|
196
|
+
parts.extend(match.group(0).lower() for match in _CAMEL_RE.finditer(chunk))
|
|
197
|
+
return [part for part in parts if part]
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def stem(word: str) -> str:
|
|
201
|
+
"""Fold the common English inflections so `updates`/`updating` match `update`.
|
|
202
|
+
|
|
203
|
+
The trailing-`e` strip is what makes the fold *symmetric*: without it
|
|
204
|
+
`creates`/`creating` reduce to `creat` while `create` stays `create`, and the
|
|
205
|
+
two never match — the shape that most often made a restatement look novel.
|
|
206
|
+
|
|
207
|
+
Deliberately crude otherwise. A real stemmer would conflate more pairs, and
|
|
208
|
+
every extra conflation is a chance to call a novel word a restatement.
|
|
209
|
+
|
|
210
|
+
Returns:
|
|
211
|
+
The stemmed word.
|
|
212
|
+
|
|
213
|
+
"""
|
|
214
|
+
base = word
|
|
215
|
+
for suffix in ("ing", "ied", "ies", "ers", "er", "ed", "es", "s"):
|
|
216
|
+
if word.endswith(suffix) and len(word) - len(suffix) >= _MIN_STEM_LENGTH:
|
|
217
|
+
base = word[: len(word) - len(suffix)]
|
|
218
|
+
if suffix in {"ied", "ies"}:
|
|
219
|
+
return base + "y"
|
|
220
|
+
break
|
|
221
|
+
if base.endswith("e") and len(base) - 1 >= _MIN_STEM_LENGTH:
|
|
222
|
+
return base[:-1]
|
|
223
|
+
return base
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def content_tokens(text: str) -> list[str]:
|
|
227
|
+
"""Split prose into lowercase content words, dropping stopwords.
|
|
228
|
+
|
|
229
|
+
Returns:
|
|
230
|
+
The comment's content tokens, in order.
|
|
231
|
+
|
|
232
|
+
"""
|
|
233
|
+
tokens: list[str] = []
|
|
234
|
+
for match in _WORD_RE.finditer(text):
|
|
235
|
+
tokens.extend(split_identifier(match.group(0)))
|
|
236
|
+
return [token for token in tokens if token not in STOPWORDS]
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def code_tokens(text: str) -> set[str]:
|
|
240
|
+
"""Collect every identifier part appearing in a slice of source.
|
|
241
|
+
|
|
242
|
+
Returns:
|
|
243
|
+
The lowercase identifier parts, as a set.
|
|
244
|
+
|
|
245
|
+
"""
|
|
246
|
+
tokens: set[str] = set()
|
|
247
|
+
for match in _WORD_RE.finditer(text):
|
|
248
|
+
tokens.update(split_identifier(match.group(0)))
|
|
249
|
+
return tokens
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def restates(comment_tokens: Sequence[str], code: Iterable[str]) -> bool:
|
|
253
|
+
"""Report whether every content token of a comment already appears in the code.
|
|
254
|
+
|
|
255
|
+
Exact or stemmed match only. Prefix matching is deliberately absent: it is
|
|
256
|
+
what sank the first attempt at this shape (PR #98), where `service` matched
|
|
257
|
+
`locationService` and drove the false-positive rate to ~60%.
|
|
258
|
+
|
|
259
|
+
Returns:
|
|
260
|
+
True when the comment adds no token the code does not already carry.
|
|
261
|
+
|
|
262
|
+
"""
|
|
263
|
+
present = set(code)
|
|
264
|
+
stems = {stem(token) for token in present}
|
|
265
|
+
return all(token in present or stem(token) in stems for token in comment_tokens)
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
# --- one tokenize pass per file --------------------------------------------
|
|
269
|
+
|
|
270
|
+
_LAYOUT_TOKENS = frozenset({tokenize.NL, tokenize.NEWLINE, tokenize.INDENT, tokenize.DEDENT})
|
|
271
|
+
_NON_CODE_TOKENS = _LAYOUT_TOKENS | frozenset({tokenize.COMMENT, tokenize.ENCODING, tokenize.ENDMARKER})
|
|
272
|
+
|
|
273
|
+
_Scan = tuple[list[tuple[int, int, str]], list[tuple[int, int, str]], set[int], int]
|
|
274
|
+
|
|
275
|
+
_last_scan: tuple[str, _Scan] | None = None
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def _scan(source: str) -> _Scan:
|
|
279
|
+
standalone: list[tuple[int, int, str]] = []
|
|
280
|
+
trailing: list[tuple[int, int, str]] = []
|
|
281
|
+
nested: set[int] = set()
|
|
282
|
+
first_code_line = 1 << 30
|
|
283
|
+
prev_end_row = 0
|
|
284
|
+
depth = 0
|
|
285
|
+
readline = io.StringIO(source).readline
|
|
286
|
+
for tok in tokenize.generate_tokens(readline):
|
|
287
|
+
if tok.type == tokenize.COMMENT:
|
|
288
|
+
entry = (tok.start[0], tok.start[1], tok.string.lstrip("#").strip())
|
|
289
|
+
(trailing if tok.start[0] == prev_end_row else standalone).append(entry)
|
|
290
|
+
if depth > 0:
|
|
291
|
+
nested.add(tok.start[0])
|
|
292
|
+
elif tok.type == tokenize.OP:
|
|
293
|
+
if tok.string in {"(", "[", "{"}:
|
|
294
|
+
depth += 1
|
|
295
|
+
elif tok.string in {")", "]", "}"}:
|
|
296
|
+
depth = max(0, depth - 1)
|
|
297
|
+
if tok.type not in _LAYOUT_TOKENS:
|
|
298
|
+
prev_end_row = tok.end[0]
|
|
299
|
+
if tok.type not in _NON_CODE_TOKENS:
|
|
300
|
+
first_code_line = min(first_code_line, tok.start[0])
|
|
301
|
+
return standalone, trailing, nested, first_code_line
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def _scan_memo(source: str) -> _Scan:
|
|
305
|
+
global _last_scan # ruff: ignore[global-statement] — single-slot memo; the CLI runs rules per file sequentially
|
|
306
|
+
if _last_scan is not None and _last_scan[0] is source:
|
|
307
|
+
return _last_scan[1]
|
|
308
|
+
result = _scan(source)
|
|
309
|
+
_last_scan = (source, result)
|
|
310
|
+
return result
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def trailing_comments(source: str) -> list[tuple[int, int, str]]:
|
|
314
|
+
"""Return every comment that shares its line with code, as `(line, col, body)`.
|
|
315
|
+
|
|
316
|
+
Raises out of here when `source` cannot be tokenized; see
|
|
317
|
+
`standalone_comments`.
|
|
318
|
+
|
|
319
|
+
Returns:
|
|
320
|
+
The trailing comments, in source order.
|
|
321
|
+
|
|
322
|
+
"""
|
|
323
|
+
return _scan_memo(source)[1]
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def nested_comment_lines(source: str) -> set[int]:
|
|
327
|
+
"""Return the lines of comments sitting INSIDE a bracketed expression.
|
|
328
|
+
|
|
329
|
+
A comment at bracket depth > 0 is annotating an element of a list, dict or
|
|
330
|
+
call — `# config` inside pydantic's `__all__` groups the names beneath it —
|
|
331
|
+
rather than signposting the structure of the file. Both readings produce the
|
|
332
|
+
same one-word comment, and only the depth tells them apart.
|
|
333
|
+
|
|
334
|
+
Returns:
|
|
335
|
+
The line numbers of comments nested inside brackets.
|
|
336
|
+
|
|
337
|
+
"""
|
|
338
|
+
return _scan_memo(source)[2]
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def standalone_comments(source: str) -> tuple[list[tuple[int, int, str]], int]:
|
|
342
|
+
"""Return every own-line comment as `(line, col, body)`, plus the first code line.
|
|
343
|
+
|
|
344
|
+
A comment is standalone when it is the only content on its line; `first code
|
|
345
|
+
line` is the row of the first real code token (a large sentinel when the file
|
|
346
|
+
has none). Memoized on the source *object* so the comment rules share one
|
|
347
|
+
tokenize pass per file, as `rule_base.parse_or_none` does for the AST.
|
|
348
|
+
|
|
349
|
+
A file the tokenizer rejects raises out of here rather than being silently
|
|
350
|
+
treated as comment-free; every caller catches that and returns no
|
|
351
|
+
diagnostics, because a rule has nothing useful to say about a file that does
|
|
352
|
+
not parse.
|
|
353
|
+
|
|
354
|
+
Returns:
|
|
355
|
+
The standalone comments and the first code line's row.
|
|
356
|
+
|
|
357
|
+
"""
|
|
358
|
+
standalone, _, _, first_code_line = _scan_memo(source)
|
|
359
|
+
return standalone, first_code_line
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def comment_runs(standalone: Sequence[tuple[int, int, str]]) -> list[list[tuple[int, int, str]]]:
|
|
363
|
+
"""Group standalone comments into runs of consecutive lines.
|
|
364
|
+
|
|
365
|
+
Returns:
|
|
366
|
+
One list per contiguous `#` block, in source order.
|
|
367
|
+
|
|
368
|
+
"""
|
|
369
|
+
runs: list[list[tuple[int, int, str]]] = []
|
|
370
|
+
for entry in sorted(standalone):
|
|
371
|
+
if runs and entry[0] == runs[-1][-1][0] + 1:
|
|
372
|
+
runs[-1].append(entry)
|
|
373
|
+
else:
|
|
374
|
+
runs.append([entry])
|
|
375
|
+
return runs
|
|
@@ -27,6 +27,7 @@ from sarj_python_lint.rules.no_offset_pagination import NoOffsetPagination
|
|
|
27
27
|
from sarj_python_lint.rules.no_query_with_many_joins import NoQueryWithManyJoins
|
|
28
28
|
from sarj_python_lint.rules.no_raw_sql_in_tests import NoRawSqlInTests
|
|
29
29
|
from sarj_python_lint.rules.no_repeated_string_literal import NoRepeatedStringLiteral
|
|
30
|
+
from sarj_python_lint.rules.no_restated_comment import NoRestatedComment
|
|
30
31
|
from sarj_python_lint.rules.no_secret_in_log import NoSecretInLog
|
|
31
32
|
from sarj_python_lint.rules.no_select_star import NoSelectStar
|
|
32
33
|
from sarj_python_lint.rules.no_sentinel_return_on_except import NoSentinelReturnOnExcept
|
|
@@ -55,6 +56,7 @@ from sarj_python_lint.rules.prefer_timedelta_for_durations import (
|
|
|
55
56
|
PreferTimedeltaForDurations,
|
|
56
57
|
)
|
|
57
58
|
from sarj_python_lint.rules.pydantic_at_boundaries import PydanticAtBoundaries
|
|
59
|
+
from sarj_python_lint.rules.redundant_docstring import RedundantDocstring
|
|
58
60
|
from sarj_python_lint.rules.single_public_export import SinglePublicExport
|
|
59
61
|
from sarj_python_lint.rules.sleep_with_computed_arg_in_test import SleepWithComputedArgInTest
|
|
60
62
|
from sarj_python_lint.rules.stepdown import Stepdown
|
|
@@ -64,6 +66,7 @@ from sarj_python_lint.rules.store_insert_requires_on_conflict import (
|
|
|
64
66
|
from sarj_python_lint.rules.test_loops_over_literal_cases import (
|
|
65
67
|
TestLoopsOverLiteralCases,
|
|
66
68
|
)
|
|
69
|
+
from sarj_python_lint.rules.trailing_value_narration import TrailingValueNarration
|
|
67
70
|
from sarj_python_lint.rules.xfail_requires_strict import XfailRequiresStrict
|
|
68
71
|
from sarj_python_lint.rules.zero_assertion_test import ZeroAssertionTest
|
|
69
72
|
|
|
@@ -120,6 +123,9 @@ REGISTRY: dict[str, type[Rule]] = {
|
|
|
120
123
|
SleepWithComputedArgInTest.id: SleepWithComputedArgInTest,
|
|
121
124
|
ZeroAssertionTest.id: ZeroAssertionTest,
|
|
122
125
|
NoFirstPartyPrivateImport.id: NoFirstPartyPrivateImport,
|
|
126
|
+
NoRestatedComment.id: NoRestatedComment,
|
|
127
|
+
RedundantDocstring.id: RedundantDocstring,
|
|
128
|
+
TrailingValueNarration.id: TrailingValueNarration,
|
|
123
129
|
}
|
|
124
130
|
|
|
125
131
|
__all__ = ["REGISTRY"]
|