sarj-python-lint 0.18.1__tar.gz → 0.20.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/PKG-INFO +42 -1
  2. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/README.md +41 -0
  3. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/pyproject.toml +1 -1
  4. sarj_python_lint-0.20.0/src/sarj_python_lint/rules/_comments.py +375 -0
  5. sarj_python_lint-0.20.0/src/sarj_python_lint/rules/_first_party.py +192 -0
  6. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_registry.py +10 -0
  7. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_comment_cruft.py +187 -41
  8. sarj_python_lint-0.20.0/src/sarj_python_lint/rules/no_first_party_private_import.py +184 -0
  9. sarj_python_lint-0.20.0/src/sarj_python_lint/rules/no_restated_comment.py +318 -0
  10. sarj_python_lint-0.20.0/src/sarj_python_lint/rules/redundant_docstring.py +211 -0
  11. sarj_python_lint-0.20.0/src/sarj_python_lint/rules/trailing_value_narration.py +134 -0
  12. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/.gitignore +0 -0
  13. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/__init__.py +0 -0
  14. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/__main__.py +0 -0
  15. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/_secret_names.py +0 -0
  16. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/_version.py +0 -0
  17. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/py.typed +0 -0
  18. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rule_base.py +0 -0
  19. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/__init__.py +0 -0
  20. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_logging.py +0 -0
  21. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_paths.py +0 -0
  22. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_sql.py +0 -0
  23. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/fixture_returns_bare_tuple.py +0 -0
  24. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/inefficient_string_concat_in_loop.py +0 -0
  25. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/kwarg_heavy_construction_in_test.py +0 -0
  26. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/kwonly_same_type_params.py +0 -0
  27. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/mock_without_spec.py +0 -0
  28. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_aggregation_in_store_query.py +0 -0
  29. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_cors_wildcard_with_credentials.py +0 -0
  30. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_fat_try_blocks.py +0 -0
  31. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_file_level_suppression.py +0 -0
  32. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_fstring_in_log.py +0 -0
  33. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_isinstance_union_chain.py +0 -0
  34. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_offset_pagination.py +0 -0
  35. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_query_with_many_joins.py +0 -0
  36. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_raw_sql_in_tests.py +0 -0
  37. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_repeated_string_literal.py +0 -0
  38. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_secret_in_log.py +0 -0
  39. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_select_star.py +0 -0
  40. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_sentinel_return_on_except.py +0 -0
  41. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_sequential_await.py +0 -0
  42. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_sleep_in_test_body.py +0 -0
  43. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_unreachable_after_terminal.py +0 -0
  44. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/parametrize_case_needs_id.py +0 -0
  45. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_class_row.py +0 -0
  46. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_constant_time_secret_compare.py +0 -0
  47. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_match_assert_never.py +0 -0
  48. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_module_level_constant.py +0 -0
  49. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_namedtuple_over_tuple_return.py +0 -0
  50. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_str_enum.py +0 -0
  51. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_struct_over_namedtuple.py +0 -0
  52. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_timedelta_for_durations.py +0 -0
  53. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/pydantic_at_boundaries.py +0 -0
  54. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/single_public_export.py +0 -0
  55. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/sleep_with_computed_arg_in_test.py +0 -0
  56. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/stepdown.py +0 -0
  57. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/store_insert_requires_on_conflict.py +0 -0
  58. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/test_loops_over_literal_cases.py +0 -0
  59. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/xfail_requires_strict.py +0 -0
  60. {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/zero_assertion_test.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sarj-python-lint
3
- Version: 0.18.1
3
+ Version: 0.20.0
4
4
  Summary: Custom Python lint rules — AST-based, pre-commit-friendly, hypermodern defaults
5
5
  Project-URL: Homepage, https://github.com/sarj-ai/standards/tree/main/packages/python
6
6
  Project-URL: Repository, https://github.com/sarj-ai/standards
@@ -60,6 +60,47 @@ measured against.
60
60
  - id: sarj-sleep-with-computed-arg-in-test # SARJ047
61
61
  ```
62
62
 
63
+ ### Private access, first-party only (0.19.0)
64
+
65
+ ```yaml
66
+ - id: sarj-no-first-party-private-import # SARJ048
67
+ ```
68
+
69
+ Reaching past a module's public surface is a design finding when the module is
70
+ ours and an unavoidable fact of life when it is not: a dependency that moves an
71
+ API private in a minor release leaves no edit that satisfies the lint.
72
+
73
+ `SARJ048` fires only when the module declaring the private name resolves to a
74
+ package inside your own project. Third-party privates are never flagged.
75
+
76
+ **It replaces ruff's `PLC2701 import-private-name`,** whose only exemption is
77
+ *same top-level package* — a different question, and one that cannot separate
78
+ `from bulbul.stores.task_store import _row_to_task` (real; export it) from
79
+ `from livekit.agents.inference_runner import _InferenceRunner` (no fix exists).
80
+ `sarj-lint-configs` ≥ 0.8.0 ships `PLC2701` in its ignore list for exactly this
81
+ reason; if you take that config, turn this hook on, or you lose the check
82
+ entirely.
83
+
84
+ Attribute access (`session._stt`) is out of scope and stays with ruff's
85
+ `SLF001`, which cannot make the distinction either — see the rationale in
86
+ `ruff.strict.toml`.
87
+
88
+ ### Comment-hygiene rules (0.20.0)
89
+
90
+ From a 37,918-comment, nine-repo measurement study. All three are
91
+ deletion-class, so each was validated against pydantic / trio / attrs as well as
92
+ the maintained repos before shipping — the counts and the false-positive classes
93
+ each guard was built from are recorded in the rule module docstrings.
94
+
95
+ ```yaml
96
+ - id: sarj-no-restated-comment # SARJ049
97
+ - id: sarj-redundant-docstring # SARJ050
98
+ - id: sarj-trailing-value-narration # SARJ051
99
+ ```
100
+
101
+ `redundant-docstring` finds real volume on a codebase that has never had it
102
+ (105 in noura-be), so the same baseline ratchet applies.
103
+
63
104
  Adopting these against an existing suite is easier through the baseline ratchet
64
105
  than as a big-bang fix — snapshot the current counts, then let them only shrink:
65
106
 
@@ -42,6 +42,47 @@ measured against.
42
42
  - id: sarj-sleep-with-computed-arg-in-test # SARJ047
43
43
  ```
44
44
 
45
+ ### Private access, first-party only (0.19.0)
46
+
47
+ ```yaml
48
+ - id: sarj-no-first-party-private-import # SARJ048
49
+ ```
50
+
51
+ Reaching past a module's public surface is a design finding when the module is
52
+ ours and an unavoidable fact of life when it is not: a dependency that moves an
53
+ API private in a minor release leaves no edit that satisfies the lint.
54
+
55
+ `SARJ048` fires only when the module declaring the private name resolves to a
56
+ package inside your own project. Third-party privates are never flagged.
57
+
58
+ **It replaces ruff's `PLC2701 import-private-name`,** whose only exemption is
59
+ *same top-level package* — a different question, and one that cannot separate
60
+ `from bulbul.stores.task_store import _row_to_task` (real; export it) from
61
+ `from livekit.agents.inference_runner import _InferenceRunner` (no fix exists).
62
+ `sarj-lint-configs` ≥ 0.8.0 ships `PLC2701` in its ignore list for exactly this
63
+ reason; if you take that config, turn this hook on, or you lose the check
64
+ entirely.
65
+
66
+ Attribute access (`session._stt`) is out of scope and stays with ruff's
67
+ `SLF001`, which cannot make the distinction either — see the rationale in
68
+ `ruff.strict.toml`.
69
+
70
+ ### Comment-hygiene rules (0.20.0)
71
+
72
+ From a 37,918-comment, nine-repo measurement study. All three are
73
+ deletion-class, so each was validated against pydantic / trio / attrs as well as
74
+ the maintained repos before shipping — the counts and the false-positive classes
75
+ each guard was built from are recorded in the rule module docstrings.
76
+
77
+ ```yaml
78
+ - id: sarj-no-restated-comment # SARJ049
79
+ - id: sarj-redundant-docstring # SARJ050
80
+ - id: sarj-trailing-value-narration # SARJ051
81
+ ```
82
+
83
+ `redundant-docstring` finds real volume on a codebase that has never had it
84
+ (105 in noura-be), so the same baseline ratchet applies.
85
+
45
86
  Adopting these against an existing suite is easier through the baseline ratchet
46
87
  than as a big-bang fix — snapshot the current counts, then let them only shrink:
47
88
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "sarj-python-lint"
3
- version = "0.18.1"
3
+ version = "0.20.0"
4
4
  description = "Custom Python lint rules — AST-based, pre-commit-friendly, hypermodern defaults"
5
5
  readme = "README.md"
6
6
  authors = [{ name = "sarj-ai" }]
@@ -0,0 +1,375 @@
1
+ """Shared comment analysis for the comment-hygiene rules (SARJ016/049/050/051).
2
+
3
+ Two things live here, both needed by more than one rule.
4
+
5
+ **The protected class.** Nine deterministic signals that mark a comment as
6
+ carrying something the code cannot: an external reference, a version pin, a
7
+ number with a unit, a causal connective, a negation of the obvious, an
8
+ upstream-quirk word, a concurrency/invariant term, security reasoning, or a
9
+ vendor proper noun with *ascribed behaviour*. Measured over a 37,918-comment
10
+ corpus from nine repos: the nine signals protect **40/40** hand-picked best
11
+ comments and leak **~1%** of the hand-classified cruft list.
12
+
13
+ The class is an **EXEMPTION FLOOR, never a test**. `is_protected(body)` being
14
+ False says nothing at all about a comment — over pydantic / trio / attrs it
15
+ matches only 18-35% of comments a human called valuable. Every use here is of
16
+ the form "if protected, do not flag"; inverting it into "unprotected, so
17
+ delete" would flag two thirds of the best comments in Python's most carefully
18
+ commented libraries. If a future rule wants a *positive* test for value, it
19
+ needs its own measurement, not this.
20
+
21
+ **One tokenize pass per file.** `standalone_comments()` mirrors
22
+ `rule_base.parse_or_none`: a single-slot memo keyed on the source object, so
23
+ the four comment rules that all need "every comment that is alone on its line"
24
+ tokenize each file once between them rather than once each.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import io
30
+ import re
31
+ import tokenize
32
+ from typing import TYPE_CHECKING
33
+
34
+
35
+ if TYPE_CHECKING:
36
+ from collections.abc import Iterable, Sequence
37
+
38
+
39
+ # S1 — external reference: URL, issue/ticket key, RFC/PEP/CVE, bare GitHub issue
40
+ # number, or an email/handle domain. Ticket keys allow letters after the first
41
+ # digit (`PLATFORM-1YC`) and exclude the encoding/algorithm acronyms that share
42
+ # the shape (`UTF-8`, `SHA-256`, `ISO-8601`, `AES-256`).
43
+ _REF_RE = re.compile(
44
+ r"https?://|\bRFC[- ]?\d+|\bPEP[- ]?\d+|\bCVE-\d{4}|"
45
+ r"\b(?!UTF-|SHA-|ISO-|AES-|CRC-|MD-|PCM-|EOF-|API-|BASE-)[A-Z][A-Z0-9]{1,9}-\d[A-Z0-9]{0,5}\b|"
46
+ r"(?<![&\w])#\d{2,6}\b|"
47
+ r"@[a-z][\w.-]*\.(?:us|com|ai|io|net|org|dev)\b",
48
+ )
49
+
50
+ # S2 — version pin or comparison: ">= 0.137", "v5.0", "since Python 3.11".
51
+ _VERSION_RE = re.compile(
52
+ r"(?:>=|<=|==|<|>)\s*v?\d+\.\d+|\bv\d+\.\d+|\b(?:since|until|as of)\s+(?:v?\d+\.\d+|Python\s*\d)",
53
+ re.IGNORECASE,
54
+ )
55
+
56
+ # S3 — a number carrying a unit (time, size, rate, audio, percent) or an HTTP
57
+ # status code. `429` and `250 ms` are facts about the world, not about the code.
58
+ _UNITS_RE = re.compile(
59
+ r"[~<>]?\d+(?:\.\d+)?\s?(?:ms|s\b|sec\b|seconds?\b|min\b|minutes?\b|hours?\b|days?\b|"
60
+ r"KB|MB|MiB|GiB|kHz|Hz|bytes?\b|bit\b|-bit\b|%|px\b|rps\b|qps\b)|"
61
+ r"\b[1-5]xx\b|\b(?:301|302|304|307|308|400|401|403|404|405|409|410|412|422|425|429|500|501|502|503|504)\b",
62
+ )
63
+
64
+ # S4 — a causal connective tying behaviour to a consequence. This is the shape
65
+ # of a *why*: the comment says what breaks if the code changes.
66
+ _CAUSAL_RE = re.compile(
67
+ r"\b(?:because|otherwise|so that|or else|would (?:break|fail|race|deadlock|leak|clobber|loop|crash|page|stall)|"
68
+ r"breaks?\b|so we don'?t|to avoid\b|caused\b|causes\b|gets? clobbered|"
69
+ r"keeps? (?:us|it|them) from|doesn'?t\b.{0,24}\b(?:page|fire|break|leak|loop)|"
70
+ r"eat into|would otherwise|trade-?offs?\b)\b",
71
+ re.IGNORECASE,
72
+ )
73
+
74
+ # S5 — negation of the obvious, or a flagged deliberate deviation. "NOT a typo",
75
+ # "deliberately re-raises", "instead of the documented order".
76
+ _NEGATION_RE = re.compile(
77
+ r"\b(?:must not|must never|do(?:es)? not\b|don'?t\b.{0,30}\b(?:leak|log|cache|retry|block|steal|wipe)|"
78
+ r"never\b|deliberately|intentionally|counterintuitiv|NOT\b)|(?<!based )\bon purpose\b|"
79
+ r"\(not\s|\binstead of\b|\brather than\b",
80
+ )
81
+
82
+ # S6 — upstream/vendor quirk, workaround provenance, or an external contract.
83
+ _UPSTREAM_RE = re.compile(
84
+ r"\b(?:upstream|workaround|quirk|backport|vendored|regression|fixed upstream|"
85
+ r"requires?\b|convention\b|rate.?limit|deprecat|opts? in(?:to)?\b|"
86
+ r"raises?\b.{0,60}\b(?:when|if|unless)\b)",
87
+ re.IGNORECASE,
88
+ )
89
+
90
+ # S7 — concurrency, ordering, or invariant vocabulary. Nothing in the code text
91
+ # can state "this must run before the lock is taken".
92
+ _INVARIANT_RE = re.compile(
93
+ r"\b(?:invariant|idempotent|race\b|deadlock|re-?entran|atomic|thread-?safe|signal-?safe|"
94
+ r"lexicographic(?:al(?:ly)?)?|monotonic|must (?:run|be|happen|come|stay|hit|converge|configure)|"
95
+ r"before any\b|lost the (?:claim )?race)\b",
96
+ re.IGNORECASE,
97
+ )
98
+
99
+ # S8 — security reasoning.
100
+ _SECURITY_RE = re.compile(
101
+ r"\b(?:timing attack|constant-?time|replay|PII\b|redact|secret|injection|spoof|"
102
+ r"fail-?closed|fail-?open|auth bypass|early-?exit timing)\b",
103
+ re.IGNORECASE,
104
+ )
105
+
106
+ # S9 — a vendor proper noun with *ascribed behaviour* (possessive, or followed by
107
+ # a behavioural verb). A vendor name as the mere object of a narration verb
108
+ # ("Create the prompt for Gemini") carries nothing and is deliberately NOT
109
+ # protected — that distinction is what keeps the leak rate at ~1%.
110
+ _VENDOR_RE = re.compile(
111
+ r"\b(?:GitHub|Slack|Twilio|LiveKit|Kamailio|Groq|OpenAI|Anthropic|Cloudflare|FastAPI|"
112
+ r"Starlette|Sentry|Zoho|Salla|Ashby|Linear|BigQuery|Postgres|Neon|Drizzle|Vertex|Gemini|"
113
+ r"Firestore|Stripe|Next\.js|React Compiler|pydantic|ruff|loguru|Lexical|Farasa|Orpheus|"
114
+ r"Whisper|schemathesis)"
115
+ r"(?:'s\b|\s+(?:requires?|returns?|expects?|allows?|rejects?|accepts?|sends?|caps?|"
116
+ r"limits?|wraps?|silently|outputs?|stores?|treats?|doesn'?t|does not|won'?t|can'?t|"
117
+ r"only|models)\b)",
118
+ )
119
+
120
+ _SIGNALS: dict[str, re.Pattern[str]] = {
121
+ "ref": _REF_RE,
122
+ "version": _VERSION_RE,
123
+ "units": _UNITS_RE,
124
+ "causal": _CAUSAL_RE,
125
+ "negation": _NEGATION_RE,
126
+ "upstream": _UPSTREAM_RE,
127
+ "invariant": _INVARIANT_RE,
128
+ "security": _SECURITY_RE,
129
+ "vendor": _VENDOR_RE,
130
+ }
131
+
132
+
133
+ def protecting_signals(body: str) -> frozenset[str]:
134
+ """Name every protected-class signal that matches `body`.
135
+
136
+ Returns:
137
+ The set of signal names; empty when nothing protects the comment.
138
+
139
+ """
140
+ return frozenset(name for name, pattern in _SIGNALS.items() if pattern.search(body))
141
+
142
+
143
+ def is_protected(body: str) -> bool:
144
+ """Report whether a comment carries any protected-class signal.
145
+
146
+ EXEMPTION FLOOR ONLY — see the module docstring. A False result is not
147
+ evidence that the comment is worthless.
148
+
149
+ Returns:
150
+ True when at least one of the nine signals matches.
151
+
152
+ """
153
+ return any(pattern.search(body) for pattern in _SIGNALS.values())
154
+
155
+
156
+ def has_external_reference(body: str) -> bool:
157
+ """Report whether a comment cites a ticket, URL, RFC/PEP/CVE, or issue number.
158
+
159
+ Signal S1 on its own. A comment that names where the decision is recorded is
160
+ doing the one thing the code cannot, and it is the signal that separates a
161
+ scoping note with an owner ("EN-only for now — AR needs audio (PROD-249)")
162
+ from an unowned admission ("hacky, fix later").
163
+
164
+ Returns:
165
+ True when the comment carries an external reference.
166
+
167
+ """
168
+ return bool(_REF_RE.search(body))
169
+
170
+
171
+ # --- tokenisation shared by the restatement detectors ----------------------
172
+
173
+ # Below this length an inflection strip would eat the word itself.
174
+ _MIN_STEM_LENGTH = 3
175
+
176
+ _WORD_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*|\d+")
177
+ _CAMEL_RE = re.compile(r"[A-Z]+(?=[A-Z][a-z])|[A-Z]?[a-z]+|[A-Z]+|\d+")
178
+
179
+ # Words that say nothing about *which* code a comment describes. Kept close to
180
+ # the prototype's list: shrinking it costs recall, growing it costs precision by
181
+ # letting a genuinely novel word be discounted.
182
+ STOPWORDS: frozenset[str] = frozenset(
183
+ ["a", "an", "the", "this", "that", "these", "those", "it", "its", "their", "his", "her", "our", "your", "my", "is", "are", "was", "were", "be", "been", "being", "am", "do", "does", "did", "done", "doing", "has", "have", "had", "having", "will", "would", "shall", "should", "can", "could", "may", "might", "must", "and", "or", "but", "nor", "so", "yet", "not", "no", "none", "to", "of", "for", "in", "on", "at", "by", "with", "from", "into", "onto", "out", "up", "down", "over", "under", "about", "as", "if", "then", "than", "when", "where", "which", "who", "whom", "whose", "what", "how", "why", "while", "we", "you", "they", "i", "he", "she", "them", "him", "us", "me", "also", "just", "only", "even", "still", "already", "again", "there", "here", "all", "any", "each", "every", "some", "via", "per", "etc", "eg", "ie", "vs", "need", "needs", "needed", "want", "wants", "make", "makes", "making", "let", "lets", "please", "note", "see", "above", "below"]
184
+ )
185
+
186
+
187
+ def split_identifier(token: str) -> list[str]:
188
+ """Split `snake_case` / `camelCase` / `SCREAMING_CASE` into lowercase parts.
189
+
190
+ Returns:
191
+ The identifier's word parts, lowercased.
192
+
193
+ """
194
+ parts: list[str] = []
195
+ for chunk in token.split("_"):
196
+ parts.extend(match.group(0).lower() for match in _CAMEL_RE.finditer(chunk))
197
+ return [part for part in parts if part]
198
+
199
+
200
+ def stem(word: str) -> str:
201
+ """Fold the common English inflections so `updates`/`updating` match `update`.
202
+
203
+ The trailing-`e` strip is what makes the fold *symmetric*: without it
204
+ `creates`/`creating` reduce to `creat` while `create` stays `create`, and the
205
+ two never match — the shape that most often made a restatement look novel.
206
+
207
+ Deliberately crude otherwise. A real stemmer would conflate more pairs, and
208
+ every extra conflation is a chance to call a novel word a restatement.
209
+
210
+ Returns:
211
+ The stemmed word.
212
+
213
+ """
214
+ base = word
215
+ for suffix in ("ing", "ied", "ies", "ers", "er", "ed", "es", "s"):
216
+ if word.endswith(suffix) and len(word) - len(suffix) >= _MIN_STEM_LENGTH:
217
+ base = word[: len(word) - len(suffix)]
218
+ if suffix in {"ied", "ies"}:
219
+ return base + "y"
220
+ break
221
+ if base.endswith("e") and len(base) - 1 >= _MIN_STEM_LENGTH:
222
+ return base[:-1]
223
+ return base
224
+
225
+
226
+ def content_tokens(text: str) -> list[str]:
227
+ """Split prose into lowercase content words, dropping stopwords.
228
+
229
+ Returns:
230
+ The comment's content tokens, in order.
231
+
232
+ """
233
+ tokens: list[str] = []
234
+ for match in _WORD_RE.finditer(text):
235
+ tokens.extend(split_identifier(match.group(0)))
236
+ return [token for token in tokens if token not in STOPWORDS]
237
+
238
+
239
+ def code_tokens(text: str) -> set[str]:
240
+ """Collect every identifier part appearing in a slice of source.
241
+
242
+ Returns:
243
+ The lowercase identifier parts, as a set.
244
+
245
+ """
246
+ tokens: set[str] = set()
247
+ for match in _WORD_RE.finditer(text):
248
+ tokens.update(split_identifier(match.group(0)))
249
+ return tokens
250
+
251
+
252
+ def restates(comment_tokens: Sequence[str], code: Iterable[str]) -> bool:
253
+ """Report whether every content token of a comment already appears in the code.
254
+
255
+ Exact or stemmed match only. Prefix matching is deliberately absent: it is
256
+ what sank the first attempt at this shape (PR #98), where `service` matched
257
+ `locationService` and drove the false-positive rate to ~60%.
258
+
259
+ Returns:
260
+ True when the comment adds no token the code does not already carry.
261
+
262
+ """
263
+ present = set(code)
264
+ stems = {stem(token) for token in present}
265
+ return all(token in present or stem(token) in stems for token in comment_tokens)
266
+
267
+
268
+ # --- one tokenize pass per file --------------------------------------------
269
+
270
+ _LAYOUT_TOKENS = frozenset({tokenize.NL, tokenize.NEWLINE, tokenize.INDENT, tokenize.DEDENT})
271
+ _NON_CODE_TOKENS = _LAYOUT_TOKENS | frozenset({tokenize.COMMENT, tokenize.ENCODING, tokenize.ENDMARKER})
272
+
273
+ _Scan = tuple[list[tuple[int, int, str]], list[tuple[int, int, str]], set[int], int]
274
+
275
+ _last_scan: tuple[str, _Scan] | None = None
276
+
277
+
278
+ def _scan(source: str) -> _Scan:
279
+ standalone: list[tuple[int, int, str]] = []
280
+ trailing: list[tuple[int, int, str]] = []
281
+ nested: set[int] = set()
282
+ first_code_line = 1 << 30
283
+ prev_end_row = 0
284
+ depth = 0
285
+ readline = io.StringIO(source).readline
286
+ for tok in tokenize.generate_tokens(readline):
287
+ if tok.type == tokenize.COMMENT:
288
+ entry = (tok.start[0], tok.start[1], tok.string.lstrip("#").strip())
289
+ (trailing if tok.start[0] == prev_end_row else standalone).append(entry)
290
+ if depth > 0:
291
+ nested.add(tok.start[0])
292
+ elif tok.type == tokenize.OP:
293
+ if tok.string in {"(", "[", "{"}:
294
+ depth += 1
295
+ elif tok.string in {")", "]", "}"}:
296
+ depth = max(0, depth - 1)
297
+ if tok.type not in _LAYOUT_TOKENS:
298
+ prev_end_row = tok.end[0]
299
+ if tok.type not in _NON_CODE_TOKENS:
300
+ first_code_line = min(first_code_line, tok.start[0])
301
+ return standalone, trailing, nested, first_code_line
302
+
303
+
304
+ def _scan_memo(source: str) -> _Scan:
305
+ global _last_scan # ruff: ignore[global-statement] — single-slot memo; the CLI runs rules per file sequentially
306
+ if _last_scan is not None and _last_scan[0] is source:
307
+ return _last_scan[1]
308
+ result = _scan(source)
309
+ _last_scan = (source, result)
310
+ return result
311
+
312
+
313
+ def trailing_comments(source: str) -> list[tuple[int, int, str]]:
314
+ """Return every comment that shares its line with code, as `(line, col, body)`.
315
+
316
+ Raises out of here when `source` cannot be tokenized; see
317
+ `standalone_comments`.
318
+
319
+ Returns:
320
+ The trailing comments, in source order.
321
+
322
+ """
323
+ return _scan_memo(source)[1]
324
+
325
+
326
+ def nested_comment_lines(source: str) -> set[int]:
327
+ """Return the lines of comments sitting INSIDE a bracketed expression.
328
+
329
+ A comment at bracket depth > 0 is annotating an element of a list, dict or
330
+ call — `# config` inside pydantic's `__all__` groups the names beneath it —
331
+ rather than signposting the structure of the file. Both readings produce the
332
+ same one-word comment, and only the depth tells them apart.
333
+
334
+ Returns:
335
+ The line numbers of comments nested inside brackets.
336
+
337
+ """
338
+ return _scan_memo(source)[2]
339
+
340
+
341
+ def standalone_comments(source: str) -> tuple[list[tuple[int, int, str]], int]:
342
+ """Return every own-line comment as `(line, col, body)`, plus the first code line.
343
+
344
+ A comment is standalone when it is the only content on its line; `first code
345
+ line` is the row of the first real code token (a large sentinel when the file
346
+ has none). Memoized on the source *object* so the comment rules share one
347
+ tokenize pass per file, as `rule_base.parse_or_none` does for the AST.
348
+
349
+ A file the tokenizer rejects raises out of here rather than being silently
350
+ treated as comment-free; every caller catches that and returns no
351
+ diagnostics, because a rule has nothing useful to say about a file that does
352
+ not parse.
353
+
354
+ Returns:
355
+ The standalone comments and the first code line's row.
356
+
357
+ """
358
+ standalone, _, _, first_code_line = _scan_memo(source)
359
+ return standalone, first_code_line
360
+
361
+
362
+ def comment_runs(standalone: Sequence[tuple[int, int, str]]) -> list[list[tuple[int, int, str]]]:
363
+ """Group standalone comments into runs of consecutive lines.
364
+
365
+ Returns:
366
+ One list per contiguous `#` block, in source order.
367
+
368
+ """
369
+ runs: list[list[tuple[int, int, str]]] = []
370
+ for entry in sorted(standalone):
371
+ if runs and entry[0] == runs[-1][-1][0] + 1:
372
+ runs[-1].append(entry)
373
+ else:
374
+ runs.append([entry])
375
+ return runs
@@ -0,0 +1,192 @@
1
+ """Shared first-party / third-party module resolution.
2
+
3
+ Every "don't reach into privates" rule needs one thing no purely syntactic
4
+ checker has: whether the module that *declares* the private name is ours or
5
+ somebody else's. Reaching into our own module's underscore names is a design
6
+ problem we can fix by exporting a public surface. Reaching into a dependency's
7
+ underscore names is often the only option available — when a library moves an
8
+ API private in a minor release, the "just use the public API" advice names an
9
+ API that no longer exists, and the lint finding becomes an instruction to
10
+ perform an impossible edit.
11
+
12
+ Resolution is filesystem-based and deliberately conservative:
13
+
14
+ * a module is FIRST-PARTY when its top-level name is a package directory (one
15
+ containing `__init__.py`) found inside the enclosing project;
16
+ * everything else — stdlib, site-packages, anything unresolvable — is treated
17
+ as THIRD-PARTY, because the failure mode of guessing "third-party" is a
18
+ missed finding, while the failure mode of guessing "first-party" is exactly
19
+ the impossible-edit demand these rules exist to avoid.
20
+
21
+ The `__init__.py` requirement is load-bearing, not incidental: bulbul carries a
22
+ `python/bulbul/livekit/` directory of SIP trunk JSON, and a name-only match
23
+ would have classified the `livekit` dependency as first-party and re-flagged
24
+ the very imports this distinction exists to exempt. Requiring an importable
25
+ package makes a top-level name collide only when a real first-party package
26
+ shadows the distribution — at which point flagging it is correct.
27
+
28
+ The project root is the nearest ancestor holding `.git`, falling back to the
29
+ topmost contiguous run of ancestors holding `pyproject.toml` (worktrees,
30
+ sdist checkouts, and vendored trees all resolve). Scanning stops at the first
31
+ package directory on each branch, so only *top-level* package names are
32
+ collected — `agent.lk.custom_models` contributes `agent`, never `lk`.
33
+ """
34
+
35
+ from __future__ import annotations
36
+
37
+ from functools import lru_cache
38
+ import sys
39
+ from typing import TYPE_CHECKING
40
+
41
+
42
+ if TYPE_CHECKING:
43
+ from pathlib import Path
44
+
45
+
46
+ # `.git` is a file, not a directory, inside a linked worktree — test existence,
47
+ # never `is_dir()`.
48
+ _GIT_MARKER = ".git"
49
+ _PROJECT_MARKER = "pyproject.toml"
50
+
51
+ # Never descend into these: a virtualenv or vendored tree holds *third-party*
52
+ # packages, and collecting their names would classify every dependency as ours.
53
+ _SKIP_DIR_NAMES = frozenset({
54
+ ".git",
55
+ ".mypy_cache",
56
+ ".next",
57
+ ".pytest_cache",
58
+ ".ruff_cache",
59
+ ".tox",
60
+ ".turbo",
61
+ ".venv",
62
+ "__pycache__",
63
+ "build",
64
+ "coverage",
65
+ "dist",
66
+ "node_modules",
67
+ "site-packages",
68
+ "venv",
69
+ })
70
+
71
+ # Deep enough for a uv/pnpm-style monorepo (`<root>/packages/python/src/<pkg>`),
72
+ # shallow enough that the scan stays a few hundred `iterdir()` calls.
73
+ _MAX_SCAN_DEPTH = 5
74
+
75
+ # Hard stop so a pathological tree (a huge monorepo, a symlink loop) degrades to
76
+ # "nothing is first-party" — under-flagging — rather than hanging the linter.
77
+ _MAX_DIRS_SCANNED = 3000
78
+
79
+ # Walking up forever from a relative path is how a linter ends up scanning $HOME.
80
+ _MAX_ANCESTORS = 24
81
+
82
+
83
+ def is_first_party_module(module: str, path: Path) -> bool:
84
+ """Report whether dotted `module` is declared inside `path`'s own project.
85
+
86
+ Returns:
87
+ True when the module's top-level name resolves to a first-party package.
88
+
89
+ """
90
+ top = module.partition(".")[0]
91
+ if not top or top in sys.stdlib_module_names:
92
+ return False
93
+ root = _project_root(path)
94
+ if root is None:
95
+ return False
96
+ return top in _first_party_roots(root)
97
+
98
+
99
+ def own_top_package(path: Path) -> str | None:
100
+ """Return the name of the top-level package `path` itself belongs to.
101
+
102
+ The OUTERMOST importable ancestor wins rather than the outermost of an
103
+ unbroken `__init__.py` run, because PEP 420 namespace subpackages are
104
+ routine — bulbul's `agent/agent/lk/custom_models/` carries no `__init__.py`
105
+ while `agent/agent/` does, and a break-on-first-gap walk would report the
106
+ file as belonging to no package at all.
107
+
108
+ Returns:
109
+ The top-level package name, or None when `path` is not inside a package.
110
+
111
+ """
112
+ resolved = _resolved(path)
113
+ if resolved is None:
114
+ return None
115
+ root = _project_root(resolved)
116
+ top: str | None = None
117
+ for ancestor in list(resolved.parents)[:_MAX_ANCESTORS]:
118
+ if (ancestor / "__init__.py").exists():
119
+ top = ancestor.name
120
+ if ancestor == root:
121
+ break
122
+ return top
123
+
124
+
125
+ def _resolved(path: Path) -> Path | None:
126
+ try:
127
+ return path.resolve()
128
+ except OSError:
129
+ return None
130
+
131
+
132
+ @lru_cache(maxsize=256)
133
+ def _project_root(path: Path) -> Path | None:
134
+ """Locate the project boundary above `path`, memoized per file path.
135
+
136
+ Returns:
137
+ The project root directory, or None when no marker is found.
138
+
139
+ """
140
+ resolved = _resolved(path)
141
+ if resolved is None:
142
+ return None
143
+ ancestors = list(resolved.parents)[:_MAX_ANCESTORS]
144
+ for ancestor in ancestors:
145
+ if (ancestor / _GIT_MARKER).exists():
146
+ return ancestor
147
+ # No VCS boundary: take the OUTERMOST directory of the unbroken run of
148
+ # pyproject.toml ancestors, which is the workspace root in a uv workspace.
149
+ outermost: Path | None = None
150
+ for ancestor in ancestors:
151
+ if not (ancestor / _PROJECT_MARKER).exists():
152
+ if outermost is not None:
153
+ break
154
+ continue
155
+ outermost = ancestor
156
+ return outermost
157
+
158
+
159
+ @lru_cache(maxsize=32)
160
+ def _first_party_roots(root: Path) -> frozenset[str]:
161
+ """Collect the top-level package names declared anywhere under `root`.
162
+
163
+ Returns:
164
+ Every directory name under `root` that is an importable package.
165
+
166
+ """
167
+ names: set[str] = set()
168
+ queue: list[tuple[Path, int]] = [(root, 0)]
169
+ scanned = 0
170
+ while queue:
171
+ directory, depth = queue.pop()
172
+ scanned += 1
173
+ if scanned > _MAX_DIRS_SCANNED:
174
+ break
175
+ try:
176
+ entries = sorted(directory.iterdir())
177
+ except OSError:
178
+ continue
179
+ for entry in entries:
180
+ if entry.name.startswith(".") or entry.name in _SKIP_DIR_NAMES:
181
+ continue
182
+ try:
183
+ if not entry.is_dir():
184
+ continue
185
+ is_package = (entry / "__init__.py").exists()
186
+ except OSError:
187
+ continue
188
+ if is_package:
189
+ names.add(entry.name)
190
+ elif depth + 1 < _MAX_SCAN_DEPTH:
191
+ queue.append((entry, depth + 1))
192
+ return frozenset(names)