sarj-python-lint 0.11.1__tar.gz → 0.12.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/PKG-INFO +1 -1
  2. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/pyproject.toml +1 -1
  3. sarj_python_lint-0.12.1/src/sarj_python_lint/__init__.py +6 -0
  4. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/__main__.py +7 -2
  5. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/_secret_names.py +19 -5
  6. sarj_python_lint-0.12.1/src/sarj_python_lint/_version.py +20 -0
  7. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rule_base.py +18 -4
  8. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/_logging.py +5 -1
  9. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/_sql.py +31 -1
  10. sarj_python_lint-0.12.1/src/sarj_python_lint/rules/inefficient_string_concat_in_loop.py +294 -0
  11. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/no_aggregation_in_store_query.py +3 -1
  12. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/no_comment_cruft.py +43 -3
  13. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/no_cors_wildcard_with_credentials.py +12 -2
  14. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/no_fat_try_blocks.py +53 -24
  15. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/no_fstring_in_log.py +61 -4
  16. sarj_python_lint-0.12.1/src/sarj_python_lint/rules/no_isinstance_union_chain.py +258 -0
  17. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/no_offset_pagination.py +3 -1
  18. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/no_query_with_many_joins.py +3 -1
  19. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/no_secret_in_log.py +14 -4
  20. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/no_select_star.py +8 -2
  21. sarj_python_lint-0.12.1/src/sarj_python_lint/rules/no_sentinel_return_on_except.py +681 -0
  22. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/no_sequential_await.py +21 -3
  23. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/no_unreachable_after_terminal.py +9 -4
  24. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/prefer_class_row.py +7 -1
  25. sarj_python_lint-0.12.1/src/sarj_python_lint/rules/prefer_constant_time_secret_compare.py +213 -0
  26. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/prefer_namedtuple_over_tuple_return.py +32 -15
  27. sarj_python_lint-0.12.1/src/sarj_python_lint/rules/prefer_str_enum.py +537 -0
  28. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/prefer_struct_over_namedtuple.py +1 -0
  29. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/prefer_timedelta_for_durations.py +9 -0
  30. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/pydantic_at_boundaries.py +39 -6
  31. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/single_public_export.py +37 -5
  32. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/stepdown.py +187 -33
  33. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/store_insert_requires_on_conflict.py +3 -1
  34. sarj_python_lint-0.11.1/src/sarj_python_lint/__init__.py +0 -9
  35. sarj_python_lint-0.11.1/src/sarj_python_lint/rules/inefficient_string_concat_in_loop.py +0 -117
  36. sarj_python_lint-0.11.1/src/sarj_python_lint/rules/no_isinstance_union_chain.py +0 -201
  37. sarj_python_lint-0.11.1/src/sarj_python_lint/rules/no_sentinel_return_on_except.py +0 -304
  38. sarj_python_lint-0.11.1/src/sarj_python_lint/rules/prefer_constant_time_secret_compare.py +0 -105
  39. sarj_python_lint-0.11.1/src/sarj_python_lint/rules/prefer_str_enum.py +0 -320
  40. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/.gitignore +0 -0
  41. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/README.md +0 -0
  42. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/py.typed +0 -0
  43. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/__init__.py +0 -0
  44. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/_registry.py +0 -0
  45. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/no_repeated_string_literal.py +0 -0
  46. {sarj_python_lint-0.11.1 → sarj_python_lint-0.12.1}/src/sarj_python_lint/rules/no_sleep_in_test_body.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sarj-python-lint
3
- Version: 0.11.1
3
+ Version: 0.12.1
4
4
  Summary: Custom Python lint rules — AST-based, pre-commit-friendly, hypermodern defaults
5
5
  Project-URL: Homepage, https://github.com/sarj-ai/standards/tree/main/packages/python
6
6
  Project-URL: Repository, https://github.com/sarj-ai/standards
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "sarj-python-lint"
3
- version = "0.11.1"
3
+ version = "0.12.1"
4
4
  description = "Custom Python lint rules — AST-based, pre-commit-friendly, hypermodern defaults"
5
5
  readme = "README.md"
6
6
  authors = [{ name = "sarj-ai" }]
@@ -0,0 +1,6 @@
1
+ """sarj-python-lint — custom Python + SQL lint rules."""
2
+
3
+ from sarj_python_lint._version import __version__
4
+
5
+
6
+ __all__ = ["__version__"]
@@ -1,4 +1,4 @@
1
- """CLI: sarj-python-lint check --rule <id> [--rule <id2>] [--baseline <json>] <files>"""
1
+ """CLI: sarj-python-lint check --rule <id> [--rule <id2>] [--baseline <json>] <files>."""
2
2
 
3
3
  from __future__ import annotations
4
4
 
@@ -106,7 +106,12 @@ def _baseline_counts(diags: list[Diagnostic]) -> dict[str, dict[str, int]]:
106
106
 
107
107
 
108
108
  def _apply_baseline(diags: list[Diagnostic], baseline: dict[str, dict[str, int]]) -> list[Diagnostic]:
109
- """Suppress up to the baselined count per (path, code); excess diags survive."""
109
+ """Suppress up to the baselined count per (path, code); excess diags survive.
110
+
111
+ Returns:
112
+ The diagnostics that exceed the baselined count for their (path, code).
113
+
114
+ """
110
115
  seen: Counter[tuple[str, str]] = Counter()
111
116
  out: list[Diagnostic] = []
112
117
  for d in diags:
@@ -95,12 +95,16 @@ _CAMEL_RE = re.compile(r"[A-Z]+(?=[A-Z][a-z])|[A-Z]?[a-z]+|[A-Z]+|\d+")
95
95
  _SEGMENT_RE = re.compile(r"[^A-Za-z0-9]+")
96
96
 
97
97
 
98
- def _tokens(identifier: str) -> list[str]:
99
- """Ordered lowercase tokens from snake_case + camelCase decomposition.
98
+ def identifier_tokens(identifier: str) -> list[str]:
99
+ """Return ordered lowercase tokens from snake_case + camelCase decomposition.
100
100
 
101
101
  Also yields each whole snake/kebab segment lowercased, so a pathological
102
102
  mixed-case single word like `ToKeN` (which camel-splitting shreds into
103
103
  `to`/`ke`/`n`) still surfaces its intended `token` form.
104
+
105
+ Returns:
106
+ The ordered lowercase tokens decomposed from `identifier`.
107
+
104
108
  """
105
109
  tokens: list[str] = []
106
110
  for segment in _SEGMENT_RE.split(identifier):
@@ -112,8 +116,13 @@ def _tokens(identifier: str) -> list[str]:
112
116
 
113
117
 
114
118
  def is_secret_name(identifier: str) -> bool:
115
- """True if `identifier` names raw secret material (a credential, not metadata)."""
116
- tokens = _tokens(identifier)
119
+ """Report whether `identifier` names raw secret material (a credential, not metadata).
120
+
121
+ Returns:
122
+ True when `identifier` denotes a credential rather than metadata.
123
+
124
+ """
125
+ tokens = identifier_tokens(identifier)
117
126
  if tokens and tokens[-1] in _INNOCUOUS_WORDS:
118
127
  return False
119
128
  if any(tok in _SECRET_WORDS for tok in tokens):
@@ -122,5 +131,10 @@ def is_secret_name(identifier: str) -> bool:
122
131
 
123
132
 
124
133
  def _has_api_key(tokens: list[str]) -> bool:
125
- """True if `api` is immediately followed by `key` (the split form of `api_key`)."""
134
+ """Report whether `api` is immediately followed by `key` (the split form of `api_key`).
135
+
136
+ Returns:
137
+ True when an `api` token is directly followed by a `key` token.
138
+
139
+ """
126
140
  return any(a == "api" and b == "key" for a, b in pairwise(tokens))
@@ -0,0 +1,20 @@
1
+ """Resolve the installed package version, with a source-tree fallback."""
2
+
3
+ from importlib.metadata import PackageNotFoundError, version
4
+
5
+
6
+ def _resolve_version() -> str:
7
+ """Return the installed distribution version, or a dev sentinel from source.
8
+
9
+ Returns:
10
+ The distribution version string, or ``"0.0.0.dev0"`` when the package is
11
+ not installed (running straight from a source checkout).
12
+
13
+ """
14
+ try:
15
+ return version("sarj-python-lint")
16
+ except PackageNotFoundError:
17
+ return "0.0.0.dev0"
18
+
19
+
20
+ __version__ = _resolve_version()
@@ -5,13 +5,13 @@ from __future__ import annotations
5
5
  from abc import ABC, abstractmethod
6
6
  import ast
7
7
  from dataclasses import dataclass
8
+ from pathlib import Path
8
9
  import re
9
10
  from typing import TYPE_CHECKING
10
11
 
11
12
 
12
13
  if TYPE_CHECKING:
13
14
  from collections.abc import Sequence
14
- from pathlib import Path
15
15
 
16
16
 
17
17
  # Suppression syntax. Two forms supported:
@@ -28,9 +28,13 @@ _SARJ_NOQA_RE = re.compile(
28
28
 
29
29
 
30
30
  def is_suppressed(source_lines: Sequence[str], line: int, code: str) -> bool:
31
- """Return True if the diagnostic's line carries a `# sarj-noqa[: CODE]` comment.
31
+ """Report whether the diagnostic's line carries a `# sarj-noqa[: CODE]` comment.
32
32
 
33
33
  `line` is 1-based to match Diagnostic.line.
34
+
35
+ Returns:
36
+ True when the line is suppressed for `code`.
37
+
34
38
  """
35
39
  if line < 1 or line > len(source_lines):
36
40
  return False
@@ -57,7 +61,12 @@ class Diagnostic:
57
61
  message: str
58
62
 
59
63
  def format(self) -> str:
60
- """Ruff-compatible: `path:line:col: CODE message`."""
64
+ """Render the finding ruff-compatibly as `path:line:col: CODE message`.
65
+
66
+ Returns:
67
+ The formatted single-line diagnostic string.
68
+
69
+ """
61
70
  return f"{self.path}:{self.line}:{self.col}: {self.code} {self.message}"
62
71
 
63
72
 
@@ -82,7 +91,12 @@ _last_parse: tuple[tuple[str, int, int], ast.Module | None] | None = None
82
91
 
83
92
 
84
93
  def parse_or_none(path: Path, source: str) -> ast.Module | None:
85
- """Parse `source`, memoizing the most recent file so N rules share one parse."""
94
+ """Parse `source`, memoizing the most recent file so N rules share one parse.
95
+
96
+ Returns:
97
+ The parsed module, or None when `source` has a syntax error.
98
+
99
+ """
86
100
  global _last_parse # ruff:ignore[global-statement] — single-slot memo; the CLI runs rules per file sequentially
87
101
  key = (str(path), len(source), hash(source))
88
102
  if _last_parse is not None and _last_parse[0] == key:
@@ -16,11 +16,15 @@ _LOGGER_FACTORIES = frozenset({"getlogger", "get_logger"})
16
16
 
17
17
 
18
18
  def is_logger_expr(expr: ast.expr) -> bool:
19
- """True if `expr` evaluates to a logger.
19
+ """Report whether `expr` evaluates to a logger.
20
20
 
21
21
  Resolves the whole receiver chain so adapter/builder/factory calls are
22
22
  caught: `logger.bind(...).info(...)`, `logger.opt(lazy=True).debug(...)`,
23
23
  `logging.getLogger(__name__).info(...)`, `self.logger.error(...)`.
24
+
25
+ Returns:
26
+ True when `expr` resolves to a logger receiver.
27
+
24
28
  """
25
29
  if isinstance(expr, ast.Name):
26
30
  return expr.id.lower() in _LOGGER_NAMES
@@ -11,10 +11,36 @@ before any keyword or comment scan.
11
11
  from __future__ import annotations
12
12
 
13
13
  import ast
14
+ from typing import TYPE_CHECKING
15
+
16
+
17
+ if TYPE_CHECKING:
18
+ from pathlib import Path
19
+
20
+
21
+ def is_store_module(path: Path) -> bool:
22
+ """Report whether `path` is a store-layer module: basename ends `_store.py`, or lives under a `stores/` directory.
23
+
24
+ The SQL store-lint rules (SARJ018/020/021) encode store-write semantics —
25
+ column-naming, ON-CONFLICT upserts, no Postgres-side aggregation — that only
26
+ apply to the store layer. Non-store SQL (Flask view handlers, a Django ORM
27
+ SQL generator) legitimately writes `SELECT *`, bare `INSERT`, and `COUNT()`,
28
+ so those files are out of scope.
29
+
30
+ Returns:
31
+ True when `path` belongs to the store layer.
32
+
33
+ """
34
+ return path.name.endswith("_store.py") or "stores" in path.parts
14
35
 
15
36
 
16
37
  def sql_string_value(node: ast.expr) -> str | None:
17
- """Reconstruct a (possibly `+`-concatenated) string literal, else None."""
38
+ """Reconstruct a (possibly `+`-concatenated) string literal, else None.
39
+
40
+ Returns:
41
+ The reconstructed string, or None when `node` is not a string literal.
42
+
43
+ """
18
44
  if isinstance(node, ast.Constant) and isinstance(node.value, str):
19
45
  return node.value
20
46
  if isinstance(node, ast.BinOp) and isinstance(node.op, ast.Add):
@@ -35,6 +61,10 @@ def strip_sql_noise(text: str) -> str:
35
61
  preserved so line offsets — and therefore diagnostic positions — do not
36
62
  shift. Doubled quotes (`''` / `""`) are SQL's in-string escape and keep the
37
63
  scanner inside the literal.
64
+
65
+ Returns:
66
+ `text` with string-literal contents and comment bodies blanked out.
67
+
38
68
  """
39
69
  out = list(text)
40
70
  n = len(text)
@@ -0,0 +1,294 @@
1
+ """SARJ002: detect O(n²) single-accumulator string growth inside loops.
2
+
3
+ Growing a string with `s += <str>` or `s = s + <str>` inside a loop is O(n²)
4
+ in CPython because strings are immutable — each step allocates a new string and
5
+ copies the previous one. Append to a list and `"".join(parts)` at the end for
6
+ O(n).
7
+
8
+ The rule fires only on genuine single-string accumulation. It deliberately does
9
+ NOT treat a `str()/repr()/format()` coercion, a `.join()` / `.format()` /
10
+ `.strftime()` call, or an `os.path.join(...)`-style call as accumulation — those
11
+ are either the prescribed remedy or a bounded per-iteration transform, not the
12
+ O(n²) defect. Per-slot writes (`parts[i] = ...`) and idempotent rebinding
13
+ (`x = f(x)`) are likewise excluded.
14
+
15
+ A target that is freshly (re)bound earlier in the same loop body — `desc = ...`
16
+ then `desc += suffix`, or a tuple unpack `obj, path = q.popleft()` then
17
+ `path += ...` — is loop-local: it starts empty each iteration, so the growth is
18
+ bounded, not cross-iteration accumulation. Only a target initialised BEFORE the
19
+ loop is a true O(n²) accumulator, so a preceding non-accumulating rebind of the
20
+ target inside the loop suppresses the diagnostic.
21
+
22
+ References:
23
+ - https://docs.python.org/3/library/stdtypes.html#str.join
24
+ - https://wiki.python.org/moin/PythonSpeed/PerformanceTips
25
+
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ import ast
31
+ from typing import TYPE_CHECKING, TypeGuard, override
32
+
33
+ from sarj_python_lint.rule_base import Diagnostic, Rule, parse_or_none
34
+
35
+
36
+ if TYPE_CHECKING:
37
+ from collections.abc import Iterator
38
+ from pathlib import Path
39
+
40
+
41
+ class InefficientStringConcatInLoop(Rule):
42
+ """O(n²) string concatenation in a loop."""
43
+
44
+ id: str = "inefficient-string-concat-in-loop"
45
+ code: str = "SARJ002"
46
+ description: str = "`s += '...'` / `s = s + '...'` in a loop is O(n²); append to a list and join."
47
+
48
+ @override
49
+ def check(self, path: Path, source: str) -> list[Diagnostic]:
50
+ tree = parse_or_none(path, source)
51
+ if tree is None:
52
+ return []
53
+ visitor = _ConcatVisitor()
54
+ visitor.visit(tree)
55
+ return [
56
+ Diagnostic(
57
+ path=path,
58
+ line=node.lineno,
59
+ col=node.col_offset + 1,
60
+ code=self.code,
61
+ message="String concat in a loop is O(n²). Append to a list and `''.join(...)`.",
62
+ )
63
+ for node in visitor.hits
64
+ ]
65
+
66
+
67
+ class _ConcatVisitor(ast.NodeVisitor):
68
+ """Single O(n) pass flagging each in-loop string accumulation exactly once."""
69
+
70
+ def __init__(self) -> None:
71
+ self._loop_depth: int = 0
72
+ self._string_vars: list[frozenset[str]] = [frozenset()]
73
+ self._loop_reassigns: list[dict[str, list[int]]] = []
74
+ self.hits: list[ast.AugAssign | ast.Assign] = []
75
+
76
+ @override
77
+ def generic_visit(self, node: ast.AST) -> None:
78
+ if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.Lambda)):
79
+ saved_depth = self._loop_depth
80
+ self._loop_depth = 0
81
+ self._string_vars.append(_string_typed_locals(node))
82
+ super().generic_visit(node)
83
+ self._string_vars.pop()
84
+ self._loop_depth = saved_depth
85
+ return
86
+ if isinstance(node, (ast.For, ast.AsyncFor, ast.While)):
87
+ self._loop_depth += 1
88
+ self._loop_reassigns.append(_loop_local_reassignments(node))
89
+ super().generic_visit(node)
90
+ self._loop_reassigns.pop()
91
+ self._loop_depth -= 1
92
+ return
93
+ if self._loop_depth and self._is_in_loop_concat(node) and not self._is_loop_local_target(node):
94
+ self.hits.append(node)
95
+ super().generic_visit(node)
96
+
97
+ def _is_loop_local_target(self, node: ast.AugAssign | ast.Assign) -> bool:
98
+ """Report whether the concat target is freshly rebound earlier this iteration.
99
+
100
+ A target rebound (not self-accumulated) before the concat inside the same
101
+ innermost loop body starts empty each pass, so its growth is bounded.
102
+
103
+ Returns:
104
+ True when the target is loop-local rather than a cross-iteration accumulator.
105
+
106
+ """
107
+ target = self._accumulation_target(node)
108
+ rebinds = self._loop_reassigns[-1].get(ast.unparse(target), ())
109
+ return any(line < node.lineno for line in rebinds)
110
+
111
+ def _accumulation_target(self, node: ast.AugAssign | ast.Assign) -> ast.expr:
112
+ if isinstance(node, ast.AugAssign):
113
+ return node.target
114
+ for target in node.targets:
115
+ if self._is_self_add_growth(target, node.value):
116
+ return target
117
+ return node.targets[0]
118
+
119
+ def _is_in_loop_concat(self, node: ast.AST) -> TypeGuard[ast.AugAssign | ast.Assign]:
120
+ if isinstance(node, ast.AugAssign):
121
+ return isinstance(node.op, ast.Add) and self._is_string_growth(node.target, node.value)
122
+ if isinstance(node, ast.Assign):
123
+ return any(self._is_self_add_growth(target, node.value) for target in node.targets)
124
+ return False
125
+
126
+ def _is_self_add_growth(self, target: ast.expr, value: ast.expr) -> bool:
127
+ """Report whether `s = s + <str>` rebinds the target to itself-plus-more.
128
+
129
+ Returns:
130
+ True when the assignment is a BinOp(Add) accumulation onto the target.
131
+
132
+ """
133
+ if not isinstance(value, ast.BinOp) or not isinstance(value.op, ast.Add):
134
+ return False
135
+ other = _other_add_operand(target, value)
136
+ if other is None:
137
+ return False
138
+ return self._is_string_growth(target, other)
139
+
140
+ def _is_string_growth(self, target: ast.expr, rhs: ast.expr) -> bool:
141
+ """Report whether appending `rhs` to `target` is single-string accumulation.
142
+
143
+ Returns:
144
+ True when the append grows a string-typed target.
145
+
146
+ """
147
+ if isinstance(target, ast.Subscript):
148
+ return False
149
+ if _looks_like_string(rhs):
150
+ return True
151
+ if isinstance(rhs, ast.Name) and isinstance(target, ast.Name):
152
+ return target.id in self._string_vars[-1]
153
+ return False
154
+
155
+
156
+ def _loop_local_reassignments(loop: ast.For | ast.AsyncFor | ast.While) -> dict[str, list[int]]:
157
+ """Map each target rebound inside this loop's own body to the lines that rebind it.
158
+
159
+ Only rebinds that are NOT self-accumulation (`s = s + x`) count — those are the
160
+ defect itself, not a fresh reset. Nested loops / functions / classes are their
161
+ own scope and are excluded.
162
+
163
+ Returns:
164
+ Target source string → line numbers where it is freshly (re)bound.
165
+
166
+ """
167
+ reassigns: dict[str, list[int]] = {}
168
+ for stmt in loop.body:
169
+ _collect_reassignments(stmt, reassigns)
170
+ return reassigns
171
+
172
+
173
+ def _collect_reassignments(node: ast.AST, reassigns: dict[str, list[int]]) -> None:
174
+ if isinstance(
175
+ node,
176
+ (ast.For, ast.AsyncFor, ast.While, ast.FunctionDef, ast.AsyncFunctionDef, ast.Lambda, ast.ClassDef),
177
+ ):
178
+ return
179
+ if isinstance(node, ast.Assign):
180
+ for target in node.targets:
181
+ for bound in _iter_binding_targets(target):
182
+ if not _is_accumulation_assign(bound, node.value):
183
+ reassigns.setdefault(ast.unparse(bound), []).append(bound.lineno)
184
+ elif (
185
+ isinstance(node, ast.AnnAssign)
186
+ and node.value is not None
187
+ and not _is_accumulation_assign(node.target, node.value)
188
+ ):
189
+ reassigns.setdefault(ast.unparse(node.target), []).append(node.target.lineno)
190
+ for child in ast.iter_child_nodes(node):
191
+ _collect_reassignments(child, reassigns)
192
+
193
+
194
+ def _iter_binding_targets(target: ast.expr) -> Iterator[ast.Name | ast.Attribute]:
195
+ """Yield the Name / Attribute leaves a binding target rebinds.
196
+
197
+ Subscript leaves (`acc[i] = ...`) are per-slot writes, not a rebind of the
198
+ accumulator itself, so they are skipped.
199
+
200
+ Yields:
201
+ Each Name / Attribute node the target binds.
202
+
203
+ """
204
+ if isinstance(target, (ast.Tuple, ast.List)):
205
+ for elt in target.elts:
206
+ yield from _iter_binding_targets(elt)
207
+ elif isinstance(target, ast.Starred):
208
+ yield from _iter_binding_targets(target.value)
209
+ elif isinstance(target, (ast.Name, ast.Attribute)):
210
+ yield target
211
+
212
+
213
+ def _is_accumulation_assign(target: ast.expr, value: ast.expr) -> bool:
214
+ if isinstance(value, ast.BinOp) and isinstance(value.op, ast.Add):
215
+ return _other_add_operand(target, value) is not None
216
+ return False
217
+
218
+
219
+ def _other_add_operand(target: ast.expr, binop: ast.BinOp) -> ast.expr | None:
220
+ """Return the non-target operand of `target + x` / `x + target`.
221
+
222
+ Returns:
223
+ The other operand, or None if neither side matches the target.
224
+
225
+ """
226
+ target_src = ast.unparse(target)
227
+ if ast.unparse(binop.left) == target_src:
228
+ return binop.right
229
+ if ast.unparse(binop.right) == target_src:
230
+ return binop.left
231
+ return None
232
+
233
+
234
+ def _string_typed_locals(func: ast.FunctionDef | ast.AsyncFunctionDef | ast.Lambda) -> frozenset[str]:
235
+ """Collect names assigned a string-literal-ish value in this function's own body.
236
+
237
+ Used as the string-typed signal for bare-`Name` accumulation (`buf += line`):
238
+ a numeric accumulator (`total = 0`) is absent, so `total += x` stays clean.
239
+
240
+ Returns:
241
+ The frozenset of locally string-typed names.
242
+
243
+ """
244
+ if isinstance(func, ast.Lambda):
245
+ return frozenset()
246
+ names: set[str] = set()
247
+ for stmt in func.body:
248
+ _collect_string_targets(stmt, names)
249
+ return frozenset(names)
250
+
251
+
252
+ def _collect_string_targets(node: ast.AST, names: set[str]) -> None:
253
+ if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.Lambda, ast.ClassDef)):
254
+ return
255
+ if isinstance(node, ast.Assign) and _looks_like_string(node.value):
256
+ for target in node.targets:
257
+ if isinstance(target, ast.Name):
258
+ names.add(target.id)
259
+ if (
260
+ isinstance(node, ast.AnnAssign)
261
+ and node.value is not None
262
+ and isinstance(node.target, ast.Name)
263
+ and _looks_like_string(node.value)
264
+ ):
265
+ names.add(node.target.id)
266
+ for child in ast.iter_child_nodes(node):
267
+ _collect_string_targets(child, names)
268
+
269
+
270
+ def _looks_like_string(node: ast.AST) -> bool:
271
+ """Report whether this expression is obviously a string at runtime.
272
+
273
+ Deliberately conservative: a bare call (`str(x)`, `",".join(...)`,
274
+ `os.path.join(...)`) is NOT treated as a string — those shapes also appear in
275
+ benign one-shot reassignment and are not the accumulation defect.
276
+
277
+ Returns:
278
+ True when the expression is heuristically string-typed.
279
+
280
+ """
281
+ if isinstance(node, ast.Constant) and isinstance(node.value, str):
282
+ return True
283
+ if isinstance(node, ast.JoinedStr): # f-string
284
+ return True
285
+ if isinstance(node, ast.NamedExpr): # walrus `(y := <str>)`
286
+ return _looks_like_string(node.value)
287
+ if isinstance(node, ast.IfExp): # ternary — string only if both branches are
288
+ return _looks_like_string(node.body) and _looks_like_string(node.orelse)
289
+ if isinstance(node, ast.BinOp):
290
+ if isinstance(node.op, ast.Add):
291
+ return _looks_like_string(node.left) or _looks_like_string(node.right)
292
+ if isinstance(node.op, ast.Mod): # `"row %s" % x` — left operand decides
293
+ return _looks_like_string(node.left)
294
+ return False
@@ -38,7 +38,7 @@ import re
38
38
  from typing import TYPE_CHECKING, override
39
39
 
40
40
  from sarj_python_lint.rule_base import Diagnostic, Rule, parse_or_none
41
- from sarj_python_lint.rules._sql import sql_string_value, strip_sql_noise
41
+ from sarj_python_lint.rules._sql import is_store_module, sql_string_value, strip_sql_noise
42
42
 
43
43
 
44
44
  if TYPE_CHECKING:
@@ -121,6 +121,8 @@ class NoAggregationInStoreQuery(Rule):
121
121
 
122
122
  @override
123
123
  def check(self, path: Path, source: str) -> list[Diagnostic]:
124
+ if not is_store_module(path):
125
+ return []
124
126
  # A diagnostic needs some string literal that carries both a query verb
125
127
  # and an aggregation keyword; `source` is a strict superset of every
126
128
  # literal, so if either class is absent from the whole file no diagnostic
@@ -20,7 +20,10 @@ readability audit, not this rule):
20
20
  block of `#` lines.
21
21
 
22
22
  Deliberately NOT flagged: trailing/standalone *prose* comments (the legitimate
23
- "why"), and directive comments — `# type:`, `# noqa`, `# sarj-noqa`,
23
+ "why"); code-shaped *illustrations* — a line that parses as Python but sits
24
+ under a prose lead-in (`# For example:`, a wrapped sentence) or carries
25
+ pseudo-code markers (`%sent%`, `[opt]`, `<FunctionBody>`, `...`); and directive
26
+ comments — `# type:`, `# noqa`, `# sarj-noqa`,
24
27
  `# pragma:`, `# pyright:`, `# mypy:`, `# fmt:`, `# isort:`, `# ruff:`,
25
28
  `# nosec`, `# TODO`, `# FIXME`, shebangs, and coding declarations.
26
29
 
@@ -43,6 +46,7 @@ if TYPE_CHECKING:
43
46
 
44
47
 
45
48
  _LEADING_PREAMBLE_MIN = 4
49
+ _PROSE_MIN_WORDS = 3
46
50
 
47
51
  _DIRECTIVE_PREFIXES = (
48
52
  "type:",
@@ -90,6 +94,11 @@ _CODE_HEADER_RE = re.compile(
90
94
  )
91
95
  _ASSIGN_OR_CALL_RE = re.compile(r"^[A-Za-z_][\w.\[\]]*\s*(?:=|:=|\+=|-=|\*=|/=)\s*\S|^[A-Za-z_][\w.]*\(")
92
96
 
97
+ # Pseudo-code / grammar-example markers (`%sent%`, `[opt]`, `<FunctionBody>`,
98
+ # `...`). Real commented-out code doesn't carry these — they mark an
99
+ # illustration inside a doc comment, not a line that was once executed.
100
+ _PSEUDOCODE_RE = re.compile(r"%[^%\s]+%|\[opt\]|<[^<>]+>|\.\.\.")
101
+
93
102
 
94
103
  def _comment_body(raw: str) -> str:
95
104
  return raw.lstrip("#").strip()
@@ -125,6 +134,8 @@ def _looks_like_code(body: str) -> bool:
125
134
  c = body.strip()
126
135
  if not c:
127
136
  return False
137
+ if _PSEUDOCODE_RE.search(c):
138
+ return False
128
139
  if _CODE_STMT_RE.match(c):
129
140
  return _compiles(c)
130
141
  if _RISKY_STMT_RE.match(c):
@@ -138,6 +149,27 @@ def _looks_like_code(body: str) -> bool:
138
149
  return False
139
150
 
140
151
 
152
+ def _is_prose_line(body: str) -> bool:
153
+ """Report whether `body` reads as a natural-language sentence, not code.
154
+
155
+ Used to spot a doc/prose comment that immediately precedes a code-shaped
156
+ line: `# For example:` above `# result = {**a, **b}`, or a wrapped sentence
157
+ whose second line happens to parse as an expression. Such a line is an
158
+ illustration / prose continuation, not commented-out code.
159
+
160
+ Returns:
161
+ True when `body` reads as prose.
162
+
163
+ """
164
+ c = body.strip()
165
+ if not c or _is_banner(c) or _is_directive(c) or _looks_like_code(c):
166
+ return False
167
+ if c.endswith(":"):
168
+ return True
169
+ words = [w for w in re.split(r"\s+", c) if any(ch.isalpha() for ch in w)]
170
+ return len(words) >= _PROSE_MIN_WORDS
171
+
172
+
141
173
  def _compiles(snippet: str) -> bool:
142
174
  try:
143
175
  ast.parse(snippet)
@@ -176,20 +208,24 @@ class NoCommentCruft(Rule):
176
208
  except tokenize.TokenError, IndentationError, SyntaxError:
177
209
  return []
178
210
  diags: dict[int, Diagnostic] = {}
211
+ by_line = {line: body for line, _, body in standalone}
179
212
  for line, col, body in standalone:
180
213
  if _is_directive(body):
181
214
  continue
182
- msg = self._classify(body)
215
+ prev_body = by_line.get(line - 1)
216
+ msg = self._classify(body, prev_body)
183
217
  if msg is not None:
184
218
  diags[line] = Diagnostic(path=path, line=line, col=col + 1, code=self.code, message=msg)
185
219
  self._flag_leading_preamble(standalone, first_code_line, path, diags)
186
220
  return [diags[k] for k in sorted(diags)]
187
221
 
188
222
  @staticmethod
189
- def _classify(body: str) -> str | None:
223
+ def _classify(body: str, prev_body: str | None) -> str | None:
190
224
  if _is_banner(body):
191
225
  return "Section-banner / region comment — structure code with functions, not ASCII rules."
192
226
  if _looks_like_code(body):
227
+ if prev_body is not None and _is_prose_line(prev_body):
228
+ return None
193
229
  return "Commented-out code — delete it; git history remembers."
194
230
  return None
195
231
 
@@ -240,6 +276,10 @@ def _standalone_comments(source: str) -> tuple[list[tuple[int, int, str]], int]:
240
276
 
241
277
  A comment is standalone when it is the only content on its line. `first code
242
278
  line` is the row of the first real code token (a large sentinel if none).
279
+
280
+ Returns:
281
+ The standalone comments and the first code line's row.
282
+
243
283
  """
244
284
  out: list[tuple[int, int, str]] = []
245
285
  first_code_line = 1 << 30