sarj-python-lint 0.18.1__tar.gz → 0.20.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/PKG-INFO +42 -1
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/README.md +41 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/pyproject.toml +1 -1
- sarj_python_lint-0.20.0/src/sarj_python_lint/rules/_comments.py +375 -0
- sarj_python_lint-0.20.0/src/sarj_python_lint/rules/_first_party.py +192 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_registry.py +10 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_comment_cruft.py +187 -41
- sarj_python_lint-0.20.0/src/sarj_python_lint/rules/no_first_party_private_import.py +184 -0
- sarj_python_lint-0.20.0/src/sarj_python_lint/rules/no_restated_comment.py +318 -0
- sarj_python_lint-0.20.0/src/sarj_python_lint/rules/redundant_docstring.py +211 -0
- sarj_python_lint-0.20.0/src/sarj_python_lint/rules/trailing_value_narration.py +134 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/.gitignore +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/__init__.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/__main__.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/_secret_names.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/_version.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/py.typed +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rule_base.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/__init__.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_logging.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_paths.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/_sql.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/fixture_returns_bare_tuple.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/inefficient_string_concat_in_loop.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/kwarg_heavy_construction_in_test.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/kwonly_same_type_params.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/mock_without_spec.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_aggregation_in_store_query.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_cors_wildcard_with_credentials.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_fat_try_blocks.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_file_level_suppression.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_fstring_in_log.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_isinstance_union_chain.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_offset_pagination.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_query_with_many_joins.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_raw_sql_in_tests.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_repeated_string_literal.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_secret_in_log.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_select_star.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_sentinel_return_on_except.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_sequential_await.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_sleep_in_test_body.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/no_unreachable_after_terminal.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/parametrize_case_needs_id.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_class_row.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_constant_time_secret_compare.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_match_assert_never.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_module_level_constant.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_namedtuple_over_tuple_return.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_str_enum.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_struct_over_namedtuple.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/prefer_timedelta_for_durations.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/pydantic_at_boundaries.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/single_public_export.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/sleep_with_computed_arg_in_test.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/stepdown.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/store_insert_requires_on_conflict.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/test_loops_over_literal_cases.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/xfail_requires_strict.py +0 -0
- {sarj_python_lint-0.18.1 → sarj_python_lint-0.20.0}/src/sarj_python_lint/rules/zero_assertion_test.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: sarj-python-lint
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.20.0
|
|
4
4
|
Summary: Custom Python lint rules — AST-based, pre-commit-friendly, hypermodern defaults
|
|
5
5
|
Project-URL: Homepage, https://github.com/sarj-ai/standards/tree/main/packages/python
|
|
6
6
|
Project-URL: Repository, https://github.com/sarj-ai/standards
|
|
@@ -60,6 +60,47 @@ measured against.
|
|
|
60
60
|
- id: sarj-sleep-with-computed-arg-in-test # SARJ047
|
|
61
61
|
```
|
|
62
62
|
|
|
63
|
+
### Private access, first-party only (0.19.0)
|
|
64
|
+
|
|
65
|
+
```yaml
|
|
66
|
+
- id: sarj-no-first-party-private-import # SARJ048
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Reaching past a module's public surface is a design finding when the module is
|
|
70
|
+
ours and an unavoidable fact of life when it is not: a dependency that moves an
|
|
71
|
+
API private in a minor release leaves no edit that satisfies the lint.
|
|
72
|
+
|
|
73
|
+
`SARJ048` fires only when the module declaring the private name resolves to a
|
|
74
|
+
package inside your own project. Third-party privates are never flagged.
|
|
75
|
+
|
|
76
|
+
**It replaces ruff's `PLC2701 import-private-name`,** whose only exemption is
|
|
77
|
+
*same top-level package* — a different question, and one that cannot separate
|
|
78
|
+
`from bulbul.stores.task_store import _row_to_task` (real; export it) from
|
|
79
|
+
`from livekit.agents.inference_runner import _InferenceRunner` (no fix exists).
|
|
80
|
+
`sarj-lint-configs` ≥ 0.8.0 ships `PLC2701` in its ignore list for exactly this
|
|
81
|
+
reason; if you take that config, turn this hook on, or you lose the check
|
|
82
|
+
entirely.
|
|
83
|
+
|
|
84
|
+
Attribute access (`session._stt`) is out of scope and stays with ruff's
|
|
85
|
+
`SLF001`, which cannot make the distinction either — see the rationale in
|
|
86
|
+
`ruff.strict.toml`.
|
|
87
|
+
|
|
88
|
+
### Comment-hygiene rules (0.20.0)
|
|
89
|
+
|
|
90
|
+
From a 37,918-comment, nine-repo measurement study. All three are
|
|
91
|
+
deletion-class, so each was validated against pydantic / trio / attrs as well as
|
|
92
|
+
the maintained repos before shipping — the counts and the false-positive classes
|
|
93
|
+
each guard was built from are recorded in the rule module docstrings.
|
|
94
|
+
|
|
95
|
+
```yaml
|
|
96
|
+
- id: sarj-no-restated-comment # SARJ049
|
|
97
|
+
- id: sarj-redundant-docstring # SARJ050
|
|
98
|
+
- id: sarj-trailing-value-narration # SARJ051
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
`redundant-docstring` finds real volume on a codebase that has never had it
|
|
102
|
+
(105 in noura-be), so the same baseline ratchet applies.
|
|
103
|
+
|
|
63
104
|
Adopting these against an existing suite is easier through the baseline ratchet
|
|
64
105
|
than as a big-bang fix — snapshot the current counts, then let them only shrink:
|
|
65
106
|
|
|
@@ -42,6 +42,47 @@ measured against.
|
|
|
42
42
|
- id: sarj-sleep-with-computed-arg-in-test # SARJ047
|
|
43
43
|
```
|
|
44
44
|
|
|
45
|
+
### Private access, first-party only (0.19.0)
|
|
46
|
+
|
|
47
|
+
```yaml
|
|
48
|
+
- id: sarj-no-first-party-private-import # SARJ048
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
Reaching past a module's public surface is a design finding when the module is
|
|
52
|
+
ours and an unavoidable fact of life when it is not: a dependency that moves an
|
|
53
|
+
API private in a minor release leaves no edit that satisfies the lint.
|
|
54
|
+
|
|
55
|
+
`SARJ048` fires only when the module declaring the private name resolves to a
|
|
56
|
+
package inside your own project. Third-party privates are never flagged.
|
|
57
|
+
|
|
58
|
+
**It replaces ruff's `PLC2701 import-private-name`,** whose only exemption is
|
|
59
|
+
*same top-level package* — a different question, and one that cannot separate
|
|
60
|
+
`from bulbul.stores.task_store import _row_to_task` (real; export it) from
|
|
61
|
+
`from livekit.agents.inference_runner import _InferenceRunner` (no fix exists).
|
|
62
|
+
`sarj-lint-configs` ≥ 0.8.0 ships `PLC2701` in its ignore list for exactly this
|
|
63
|
+
reason; if you take that config, turn this hook on, or you lose the check
|
|
64
|
+
entirely.
|
|
65
|
+
|
|
66
|
+
Attribute access (`session._stt`) is out of scope and stays with ruff's
|
|
67
|
+
`SLF001`, which cannot make the distinction either — see the rationale in
|
|
68
|
+
`ruff.strict.toml`.
|
|
69
|
+
|
|
70
|
+
### Comment-hygiene rules (0.20.0)
|
|
71
|
+
|
|
72
|
+
From a 37,918-comment, nine-repo measurement study. All three are
|
|
73
|
+
deletion-class, so each was validated against pydantic / trio / attrs as well as
|
|
74
|
+
the maintained repos before shipping — the counts and the false-positive classes
|
|
75
|
+
each guard was built from are recorded in the rule module docstrings.
|
|
76
|
+
|
|
77
|
+
```yaml
|
|
78
|
+
- id: sarj-no-restated-comment # SARJ049
|
|
79
|
+
- id: sarj-redundant-docstring # SARJ050
|
|
80
|
+
- id: sarj-trailing-value-narration # SARJ051
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
`redundant-docstring` finds real volume on a codebase that has never had it
|
|
84
|
+
(105 in noura-be), so the same baseline ratchet applies.
|
|
85
|
+
|
|
45
86
|
Adopting these against an existing suite is easier through the baseline ratchet
|
|
46
87
|
than as a big-bang fix — snapshot the current counts, then let them only shrink:
|
|
47
88
|
|
|
@@ -0,0 +1,375 @@
|
|
|
1
|
+
"""Shared comment analysis for the comment-hygiene rules (SARJ016/049/050/051).
|
|
2
|
+
|
|
3
|
+
Two things live here, both needed by more than one rule.
|
|
4
|
+
|
|
5
|
+
**The protected class.** Nine deterministic signals that mark a comment as
|
|
6
|
+
carrying something the code cannot: an external reference, a version pin, a
|
|
7
|
+
number with a unit, a causal connective, a negation of the obvious, an
|
|
8
|
+
upstream-quirk word, a concurrency/invariant term, security reasoning, or a
|
|
9
|
+
vendor proper noun with *ascribed behaviour*. Measured over a 37,918-comment
|
|
10
|
+
corpus from nine repos: the nine signals protect **40/40** hand-picked best
|
|
11
|
+
comments and leak **~1%** of the hand-classified cruft list.
|
|
12
|
+
|
|
13
|
+
The class is an **EXEMPTION FLOOR, never a test**. `is_protected(body)` being
|
|
14
|
+
False says nothing at all about a comment — over pydantic / trio / attrs it
|
|
15
|
+
matches only 18-35% of comments a human called valuable. Every use here is of
|
|
16
|
+
the form "if protected, do not flag"; inverting it into "unprotected, so
|
|
17
|
+
delete" would flag two thirds of the best comments in Python's most carefully
|
|
18
|
+
commented libraries. If a future rule wants a *positive* test for value, it
|
|
19
|
+
needs its own measurement, not this.
|
|
20
|
+
|
|
21
|
+
**One tokenize pass per file.** `standalone_comments()` mirrors
|
|
22
|
+
`rule_base.parse_or_none`: a single-slot memo keyed on the source object, so
|
|
23
|
+
the four comment rules that all need "every comment that is alone on its line"
|
|
24
|
+
tokenize each file once between them rather than once each.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import io
|
|
30
|
+
import re
|
|
31
|
+
import tokenize
|
|
32
|
+
from typing import TYPE_CHECKING
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
if TYPE_CHECKING:
|
|
36
|
+
from collections.abc import Iterable, Sequence
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
# S1 — external reference: URL, issue/ticket key, RFC/PEP/CVE, bare GitHub issue
|
|
40
|
+
# number, or an email/handle domain. Ticket keys allow letters after the first
|
|
41
|
+
# digit (`PLATFORM-1YC`) and exclude the encoding/algorithm acronyms that share
|
|
42
|
+
# the shape (`UTF-8`, `SHA-256`, `ISO-8601`, `AES-256`).
|
|
43
|
+
_REF_RE = re.compile(
|
|
44
|
+
r"https?://|\bRFC[- ]?\d+|\bPEP[- ]?\d+|\bCVE-\d{4}|"
|
|
45
|
+
r"\b(?!UTF-|SHA-|ISO-|AES-|CRC-|MD-|PCM-|EOF-|API-|BASE-)[A-Z][A-Z0-9]{1,9}-\d[A-Z0-9]{0,5}\b|"
|
|
46
|
+
r"(?<![&\w])#\d{2,6}\b|"
|
|
47
|
+
r"@[a-z][\w.-]*\.(?:us|com|ai|io|net|org|dev)\b",
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
# S2 — version pin or comparison: ">= 0.137", "v5.0", "since Python 3.11".
|
|
51
|
+
_VERSION_RE = re.compile(
|
|
52
|
+
r"(?:>=|<=|==|<|>)\s*v?\d+\.\d+|\bv\d+\.\d+|\b(?:since|until|as of)\s+(?:v?\d+\.\d+|Python\s*\d)",
|
|
53
|
+
re.IGNORECASE,
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
# S3 — a number carrying a unit (time, size, rate, audio, percent) or an HTTP
|
|
57
|
+
# status code. `429` and `250 ms` are facts about the world, not about the code.
|
|
58
|
+
_UNITS_RE = re.compile(
|
|
59
|
+
r"[~<>]?\d+(?:\.\d+)?\s?(?:ms|s\b|sec\b|seconds?\b|min\b|minutes?\b|hours?\b|days?\b|"
|
|
60
|
+
r"KB|MB|MiB|GiB|kHz|Hz|bytes?\b|bit\b|-bit\b|%|px\b|rps\b|qps\b)|"
|
|
61
|
+
r"\b[1-5]xx\b|\b(?:301|302|304|307|308|400|401|403|404|405|409|410|412|422|425|429|500|501|502|503|504)\b",
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
# S4 — a causal connective tying behaviour to a consequence. This is the shape
|
|
65
|
+
# of a *why*: the comment says what breaks if the code changes.
|
|
66
|
+
_CAUSAL_RE = re.compile(
|
|
67
|
+
r"\b(?:because|otherwise|so that|or else|would (?:break|fail|race|deadlock|leak|clobber|loop|crash|page|stall)|"
|
|
68
|
+
r"breaks?\b|so we don'?t|to avoid\b|caused\b|causes\b|gets? clobbered|"
|
|
69
|
+
r"keeps? (?:us|it|them) from|doesn'?t\b.{0,24}\b(?:page|fire|break|leak|loop)|"
|
|
70
|
+
r"eat into|would otherwise|trade-?offs?\b)\b",
|
|
71
|
+
re.IGNORECASE,
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
# S5 — negation of the obvious, or a flagged deliberate deviation. "NOT a typo",
|
|
75
|
+
# "deliberately re-raises", "instead of the documented order".
|
|
76
|
+
_NEGATION_RE = re.compile(
|
|
77
|
+
r"\b(?:must not|must never|do(?:es)? not\b|don'?t\b.{0,30}\b(?:leak|log|cache|retry|block|steal|wipe)|"
|
|
78
|
+
r"never\b|deliberately|intentionally|counterintuitiv|NOT\b)|(?<!based )\bon purpose\b|"
|
|
79
|
+
r"\(not\s|\binstead of\b|\brather than\b",
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
# S6 — upstream/vendor quirk, workaround provenance, or an external contract.
|
|
83
|
+
_UPSTREAM_RE = re.compile(
|
|
84
|
+
r"\b(?:upstream|workaround|quirk|backport|vendored|regression|fixed upstream|"
|
|
85
|
+
r"requires?\b|convention\b|rate.?limit|deprecat|opts? in(?:to)?\b|"
|
|
86
|
+
r"raises?\b.{0,60}\b(?:when|if|unless)\b)",
|
|
87
|
+
re.IGNORECASE,
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
# S7 — concurrency, ordering, or invariant vocabulary. Nothing in the code text
|
|
91
|
+
# can state "this must run before the lock is taken".
|
|
92
|
+
_INVARIANT_RE = re.compile(
|
|
93
|
+
r"\b(?:invariant|idempotent|race\b|deadlock|re-?entran|atomic|thread-?safe|signal-?safe|"
|
|
94
|
+
r"lexicographic(?:al(?:ly)?)?|monotonic|must (?:run|be|happen|come|stay|hit|converge|configure)|"
|
|
95
|
+
r"before any\b|lost the (?:claim )?race)\b",
|
|
96
|
+
re.IGNORECASE,
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
# S8 — security reasoning.
|
|
100
|
+
_SECURITY_RE = re.compile(
|
|
101
|
+
r"\b(?:timing attack|constant-?time|replay|PII\b|redact|secret|injection|spoof|"
|
|
102
|
+
r"fail-?closed|fail-?open|auth bypass|early-?exit timing)\b",
|
|
103
|
+
re.IGNORECASE,
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
# S9 — a vendor proper noun with *ascribed behaviour* (possessive, or followed by
|
|
107
|
+
# a behavioural verb). A vendor name as the mere object of a narration verb
|
|
108
|
+
# ("Create the prompt for Gemini") carries nothing and is deliberately NOT
|
|
109
|
+
# protected — that distinction is what keeps the leak rate at ~1%.
|
|
110
|
+
_VENDOR_RE = re.compile(
|
|
111
|
+
r"\b(?:GitHub|Slack|Twilio|LiveKit|Kamailio|Groq|OpenAI|Anthropic|Cloudflare|FastAPI|"
|
|
112
|
+
r"Starlette|Sentry|Zoho|Salla|Ashby|Linear|BigQuery|Postgres|Neon|Drizzle|Vertex|Gemini|"
|
|
113
|
+
r"Firestore|Stripe|Next\.js|React Compiler|pydantic|ruff|loguru|Lexical|Farasa|Orpheus|"
|
|
114
|
+
r"Whisper|schemathesis)"
|
|
115
|
+
r"(?:'s\b|\s+(?:requires?|returns?|expects?|allows?|rejects?|accepts?|sends?|caps?|"
|
|
116
|
+
r"limits?|wraps?|silently|outputs?|stores?|treats?|doesn'?t|does not|won'?t|can'?t|"
|
|
117
|
+
r"only|models)\b)",
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
_SIGNALS: dict[str, re.Pattern[str]] = {
|
|
121
|
+
"ref": _REF_RE,
|
|
122
|
+
"version": _VERSION_RE,
|
|
123
|
+
"units": _UNITS_RE,
|
|
124
|
+
"causal": _CAUSAL_RE,
|
|
125
|
+
"negation": _NEGATION_RE,
|
|
126
|
+
"upstream": _UPSTREAM_RE,
|
|
127
|
+
"invariant": _INVARIANT_RE,
|
|
128
|
+
"security": _SECURITY_RE,
|
|
129
|
+
"vendor": _VENDOR_RE,
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def protecting_signals(body: str) -> frozenset[str]:
|
|
134
|
+
"""Name every protected-class signal that matches `body`.
|
|
135
|
+
|
|
136
|
+
Returns:
|
|
137
|
+
The set of signal names; empty when nothing protects the comment.
|
|
138
|
+
|
|
139
|
+
"""
|
|
140
|
+
return frozenset(name for name, pattern in _SIGNALS.items() if pattern.search(body))
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def is_protected(body: str) -> bool:
|
|
144
|
+
"""Report whether a comment carries any protected-class signal.
|
|
145
|
+
|
|
146
|
+
EXEMPTION FLOOR ONLY — see the module docstring. A False result is not
|
|
147
|
+
evidence that the comment is worthless.
|
|
148
|
+
|
|
149
|
+
Returns:
|
|
150
|
+
True when at least one of the nine signals matches.
|
|
151
|
+
|
|
152
|
+
"""
|
|
153
|
+
return any(pattern.search(body) for pattern in _SIGNALS.values())
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def has_external_reference(body: str) -> bool:
|
|
157
|
+
"""Report whether a comment cites a ticket, URL, RFC/PEP/CVE, or issue number.
|
|
158
|
+
|
|
159
|
+
Signal S1 on its own. A comment that names where the decision is recorded is
|
|
160
|
+
doing the one thing the code cannot, and it is the signal that separates a
|
|
161
|
+
scoping note with an owner ("EN-only for now — AR needs audio (PROD-249)")
|
|
162
|
+
from an unowned admission ("hacky, fix later").
|
|
163
|
+
|
|
164
|
+
Returns:
|
|
165
|
+
True when the comment carries an external reference.
|
|
166
|
+
|
|
167
|
+
"""
|
|
168
|
+
return bool(_REF_RE.search(body))
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
# --- tokenisation shared by the restatement detectors ----------------------
|
|
172
|
+
|
|
173
|
+
# Below this length an inflection strip would eat the word itself.
|
|
174
|
+
_MIN_STEM_LENGTH = 3
|
|
175
|
+
|
|
176
|
+
_WORD_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*|\d+")
|
|
177
|
+
_CAMEL_RE = re.compile(r"[A-Z]+(?=[A-Z][a-z])|[A-Z]?[a-z]+|[A-Z]+|\d+")
|
|
178
|
+
|
|
179
|
+
# Words that say nothing about *which* code a comment describes. Kept close to
|
|
180
|
+
# the prototype's list: shrinking it costs recall, growing it costs precision by
|
|
181
|
+
# letting a genuinely novel word be discounted.
|
|
182
|
+
STOPWORDS: frozenset[str] = frozenset(
|
|
183
|
+
["a", "an", "the", "this", "that", "these", "those", "it", "its", "their", "his", "her", "our", "your", "my", "is", "are", "was", "were", "be", "been", "being", "am", "do", "does", "did", "done", "doing", "has", "have", "had", "having", "will", "would", "shall", "should", "can", "could", "may", "might", "must", "and", "or", "but", "nor", "so", "yet", "not", "no", "none", "to", "of", "for", "in", "on", "at", "by", "with", "from", "into", "onto", "out", "up", "down", "over", "under", "about", "as", "if", "then", "than", "when", "where", "which", "who", "whom", "whose", "what", "how", "why", "while", "we", "you", "they", "i", "he", "she", "them", "him", "us", "me", "also", "just", "only", "even", "still", "already", "again", "there", "here", "all", "any", "each", "every", "some", "via", "per", "etc", "eg", "ie", "vs", "need", "needs", "needed", "want", "wants", "make", "makes", "making", "let", "lets", "please", "note", "see", "above", "below"]
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def split_identifier(token: str) -> list[str]:
|
|
188
|
+
"""Split `snake_case` / `camelCase` / `SCREAMING_CASE` into lowercase parts.
|
|
189
|
+
|
|
190
|
+
Returns:
|
|
191
|
+
The identifier's word parts, lowercased.
|
|
192
|
+
|
|
193
|
+
"""
|
|
194
|
+
parts: list[str] = []
|
|
195
|
+
for chunk in token.split("_"):
|
|
196
|
+
parts.extend(match.group(0).lower() for match in _CAMEL_RE.finditer(chunk))
|
|
197
|
+
return [part for part in parts if part]
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def stem(word: str) -> str:
|
|
201
|
+
"""Fold the common English inflections so `updates`/`updating` match `update`.
|
|
202
|
+
|
|
203
|
+
The trailing-`e` strip is what makes the fold *symmetric*: without it
|
|
204
|
+
`creates`/`creating` reduce to `creat` while `create` stays `create`, and the
|
|
205
|
+
two never match — the shape that most often made a restatement look novel.
|
|
206
|
+
|
|
207
|
+
Deliberately crude otherwise. A real stemmer would conflate more pairs, and
|
|
208
|
+
every extra conflation is a chance to call a novel word a restatement.
|
|
209
|
+
|
|
210
|
+
Returns:
|
|
211
|
+
The stemmed word.
|
|
212
|
+
|
|
213
|
+
"""
|
|
214
|
+
base = word
|
|
215
|
+
for suffix in ("ing", "ied", "ies", "ers", "er", "ed", "es", "s"):
|
|
216
|
+
if word.endswith(suffix) and len(word) - len(suffix) >= _MIN_STEM_LENGTH:
|
|
217
|
+
base = word[: len(word) - len(suffix)]
|
|
218
|
+
if suffix in {"ied", "ies"}:
|
|
219
|
+
return base + "y"
|
|
220
|
+
break
|
|
221
|
+
if base.endswith("e") and len(base) - 1 >= _MIN_STEM_LENGTH:
|
|
222
|
+
return base[:-1]
|
|
223
|
+
return base
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def content_tokens(text: str) -> list[str]:
|
|
227
|
+
"""Split prose into lowercase content words, dropping stopwords.
|
|
228
|
+
|
|
229
|
+
Returns:
|
|
230
|
+
The comment's content tokens, in order.
|
|
231
|
+
|
|
232
|
+
"""
|
|
233
|
+
tokens: list[str] = []
|
|
234
|
+
for match in _WORD_RE.finditer(text):
|
|
235
|
+
tokens.extend(split_identifier(match.group(0)))
|
|
236
|
+
return [token for token in tokens if token not in STOPWORDS]
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def code_tokens(text: str) -> set[str]:
|
|
240
|
+
"""Collect every identifier part appearing in a slice of source.
|
|
241
|
+
|
|
242
|
+
Returns:
|
|
243
|
+
The lowercase identifier parts, as a set.
|
|
244
|
+
|
|
245
|
+
"""
|
|
246
|
+
tokens: set[str] = set()
|
|
247
|
+
for match in _WORD_RE.finditer(text):
|
|
248
|
+
tokens.update(split_identifier(match.group(0)))
|
|
249
|
+
return tokens
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def restates(comment_tokens: Sequence[str], code: Iterable[str]) -> bool:
|
|
253
|
+
"""Report whether every content token of a comment already appears in the code.
|
|
254
|
+
|
|
255
|
+
Exact or stemmed match only. Prefix matching is deliberately absent: it is
|
|
256
|
+
what sank the first attempt at this shape (PR #98), where `service` matched
|
|
257
|
+
`locationService` and drove the false-positive rate to ~60%.
|
|
258
|
+
|
|
259
|
+
Returns:
|
|
260
|
+
True when the comment adds no token the code does not already carry.
|
|
261
|
+
|
|
262
|
+
"""
|
|
263
|
+
present = set(code)
|
|
264
|
+
stems = {stem(token) for token in present}
|
|
265
|
+
return all(token in present or stem(token) in stems for token in comment_tokens)
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
# --- one tokenize pass per file --------------------------------------------
|
|
269
|
+
|
|
270
|
+
_LAYOUT_TOKENS = frozenset({tokenize.NL, tokenize.NEWLINE, tokenize.INDENT, tokenize.DEDENT})
|
|
271
|
+
_NON_CODE_TOKENS = _LAYOUT_TOKENS | frozenset({tokenize.COMMENT, tokenize.ENCODING, tokenize.ENDMARKER})
|
|
272
|
+
|
|
273
|
+
_Scan = tuple[list[tuple[int, int, str]], list[tuple[int, int, str]], set[int], int]
|
|
274
|
+
|
|
275
|
+
_last_scan: tuple[str, _Scan] | None = None
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def _scan(source: str) -> _Scan:
|
|
279
|
+
standalone: list[tuple[int, int, str]] = []
|
|
280
|
+
trailing: list[tuple[int, int, str]] = []
|
|
281
|
+
nested: set[int] = set()
|
|
282
|
+
first_code_line = 1 << 30
|
|
283
|
+
prev_end_row = 0
|
|
284
|
+
depth = 0
|
|
285
|
+
readline = io.StringIO(source).readline
|
|
286
|
+
for tok in tokenize.generate_tokens(readline):
|
|
287
|
+
if tok.type == tokenize.COMMENT:
|
|
288
|
+
entry = (tok.start[0], tok.start[1], tok.string.lstrip("#").strip())
|
|
289
|
+
(trailing if tok.start[0] == prev_end_row else standalone).append(entry)
|
|
290
|
+
if depth > 0:
|
|
291
|
+
nested.add(tok.start[0])
|
|
292
|
+
elif tok.type == tokenize.OP:
|
|
293
|
+
if tok.string in {"(", "[", "{"}:
|
|
294
|
+
depth += 1
|
|
295
|
+
elif tok.string in {")", "]", "}"}:
|
|
296
|
+
depth = max(0, depth - 1)
|
|
297
|
+
if tok.type not in _LAYOUT_TOKENS:
|
|
298
|
+
prev_end_row = tok.end[0]
|
|
299
|
+
if tok.type not in _NON_CODE_TOKENS:
|
|
300
|
+
first_code_line = min(first_code_line, tok.start[0])
|
|
301
|
+
return standalone, trailing, nested, first_code_line
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def _scan_memo(source: str) -> _Scan:
|
|
305
|
+
global _last_scan # ruff: ignore[global-statement] — single-slot memo; the CLI runs rules per file sequentially
|
|
306
|
+
if _last_scan is not None and _last_scan[0] is source:
|
|
307
|
+
return _last_scan[1]
|
|
308
|
+
result = _scan(source)
|
|
309
|
+
_last_scan = (source, result)
|
|
310
|
+
return result
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def trailing_comments(source: str) -> list[tuple[int, int, str]]:
|
|
314
|
+
"""Return every comment that shares its line with code, as `(line, col, body)`.
|
|
315
|
+
|
|
316
|
+
Raises out of here when `source` cannot be tokenized; see
|
|
317
|
+
`standalone_comments`.
|
|
318
|
+
|
|
319
|
+
Returns:
|
|
320
|
+
The trailing comments, in source order.
|
|
321
|
+
|
|
322
|
+
"""
|
|
323
|
+
return _scan_memo(source)[1]
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def nested_comment_lines(source: str) -> set[int]:
|
|
327
|
+
"""Return the lines of comments sitting INSIDE a bracketed expression.
|
|
328
|
+
|
|
329
|
+
A comment at bracket depth > 0 is annotating an element of a list, dict or
|
|
330
|
+
call — `# config` inside pydantic's `__all__` groups the names beneath it —
|
|
331
|
+
rather than signposting the structure of the file. Both readings produce the
|
|
332
|
+
same one-word comment, and only the depth tells them apart.
|
|
333
|
+
|
|
334
|
+
Returns:
|
|
335
|
+
The line numbers of comments nested inside brackets.
|
|
336
|
+
|
|
337
|
+
"""
|
|
338
|
+
return _scan_memo(source)[2]
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def standalone_comments(source: str) -> tuple[list[tuple[int, int, str]], int]:
|
|
342
|
+
"""Return every own-line comment as `(line, col, body)`, plus the first code line.
|
|
343
|
+
|
|
344
|
+
A comment is standalone when it is the only content on its line; `first code
|
|
345
|
+
line` is the row of the first real code token (a large sentinel when the file
|
|
346
|
+
has none). Memoized on the source *object* so the comment rules share one
|
|
347
|
+
tokenize pass per file, as `rule_base.parse_or_none` does for the AST.
|
|
348
|
+
|
|
349
|
+
A file the tokenizer rejects raises out of here rather than being silently
|
|
350
|
+
treated as comment-free; every caller catches that and returns no
|
|
351
|
+
diagnostics, because a rule has nothing useful to say about a file that does
|
|
352
|
+
not parse.
|
|
353
|
+
|
|
354
|
+
Returns:
|
|
355
|
+
The standalone comments and the first code line's row.
|
|
356
|
+
|
|
357
|
+
"""
|
|
358
|
+
standalone, _, _, first_code_line = _scan_memo(source)
|
|
359
|
+
return standalone, first_code_line
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def comment_runs(standalone: Sequence[tuple[int, int, str]]) -> list[list[tuple[int, int, str]]]:
|
|
363
|
+
"""Group standalone comments into runs of consecutive lines.
|
|
364
|
+
|
|
365
|
+
Returns:
|
|
366
|
+
One list per contiguous `#` block, in source order.
|
|
367
|
+
|
|
368
|
+
"""
|
|
369
|
+
runs: list[list[tuple[int, int, str]]] = []
|
|
370
|
+
for entry in sorted(standalone):
|
|
371
|
+
if runs and entry[0] == runs[-1][-1][0] + 1:
|
|
372
|
+
runs[-1].append(entry)
|
|
373
|
+
else:
|
|
374
|
+
runs.append([entry])
|
|
375
|
+
return runs
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
"""Shared first-party / third-party module resolution.
|
|
2
|
+
|
|
3
|
+
Every "don't reach into privates" rule needs one thing no purely syntactic
|
|
4
|
+
checker has: whether the module that *declares* the private name is ours or
|
|
5
|
+
somebody else's. Reaching into our own module's underscore names is a design
|
|
6
|
+
problem we can fix by exporting a public surface. Reaching into a dependency's
|
|
7
|
+
underscore names is often the only option available — when a library moves an
|
|
8
|
+
API private in a minor release, the "just use the public API" advice names an
|
|
9
|
+
API that no longer exists, and the lint finding becomes an instruction to
|
|
10
|
+
perform an impossible edit.
|
|
11
|
+
|
|
12
|
+
Resolution is filesystem-based and deliberately conservative:
|
|
13
|
+
|
|
14
|
+
* a module is FIRST-PARTY when its top-level name is a package directory (one
|
|
15
|
+
containing `__init__.py`) found inside the enclosing project;
|
|
16
|
+
* everything else — stdlib, site-packages, anything unresolvable — is treated
|
|
17
|
+
as THIRD-PARTY, because the failure mode of guessing "third-party" is a
|
|
18
|
+
missed finding, while the failure mode of guessing "first-party" is exactly
|
|
19
|
+
the impossible-edit demand these rules exist to avoid.
|
|
20
|
+
|
|
21
|
+
The `__init__.py` requirement is load-bearing, not incidental: bulbul carries a
|
|
22
|
+
`python/bulbul/livekit/` directory of SIP trunk JSON, and a name-only match
|
|
23
|
+
would have classified the `livekit` dependency as first-party and re-flagged
|
|
24
|
+
the very imports this distinction exists to exempt. Requiring an importable
|
|
25
|
+
package makes a top-level name collide only when a real first-party package
|
|
26
|
+
shadows the distribution — at which point flagging it is correct.
|
|
27
|
+
|
|
28
|
+
The project root is the nearest ancestor holding `.git`, falling back to the
|
|
29
|
+
topmost contiguous run of ancestors holding `pyproject.toml` (worktrees,
|
|
30
|
+
sdist checkouts, and vendored trees all resolve). Scanning stops at the first
|
|
31
|
+
package directory on each branch, so only *top-level* package names are
|
|
32
|
+
collected — `agent.lk.custom_models` contributes `agent`, never `lk`.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
from __future__ import annotations
|
|
36
|
+
|
|
37
|
+
from functools import lru_cache
|
|
38
|
+
import sys
|
|
39
|
+
from typing import TYPE_CHECKING
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
if TYPE_CHECKING:
|
|
43
|
+
from pathlib import Path
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
# `.git` is a file, not a directory, inside a linked worktree — test existence,
|
|
47
|
+
# never `is_dir()`.
|
|
48
|
+
_GIT_MARKER = ".git"
|
|
49
|
+
_PROJECT_MARKER = "pyproject.toml"
|
|
50
|
+
|
|
51
|
+
# Never descend into these: a virtualenv or vendored tree holds *third-party*
|
|
52
|
+
# packages, and collecting their names would classify every dependency as ours.
|
|
53
|
+
_SKIP_DIR_NAMES = frozenset({
|
|
54
|
+
".git",
|
|
55
|
+
".mypy_cache",
|
|
56
|
+
".next",
|
|
57
|
+
".pytest_cache",
|
|
58
|
+
".ruff_cache",
|
|
59
|
+
".tox",
|
|
60
|
+
".turbo",
|
|
61
|
+
".venv",
|
|
62
|
+
"__pycache__",
|
|
63
|
+
"build",
|
|
64
|
+
"coverage",
|
|
65
|
+
"dist",
|
|
66
|
+
"node_modules",
|
|
67
|
+
"site-packages",
|
|
68
|
+
"venv",
|
|
69
|
+
})
|
|
70
|
+
|
|
71
|
+
# Deep enough for a uv/pnpm-style monorepo (`<root>/packages/python/src/<pkg>`),
|
|
72
|
+
# shallow enough that the scan stays a few hundred `iterdir()` calls.
|
|
73
|
+
_MAX_SCAN_DEPTH = 5
|
|
74
|
+
|
|
75
|
+
# Hard stop so a pathological tree (a huge monorepo, a symlink loop) degrades to
|
|
76
|
+
# "nothing is first-party" — under-flagging — rather than hanging the linter.
|
|
77
|
+
_MAX_DIRS_SCANNED = 3000
|
|
78
|
+
|
|
79
|
+
# Walking up forever from a relative path is how a linter ends up scanning $HOME.
|
|
80
|
+
_MAX_ANCESTORS = 24
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def is_first_party_module(module: str, path: Path) -> bool:
|
|
84
|
+
"""Report whether dotted `module` is declared inside `path`'s own project.
|
|
85
|
+
|
|
86
|
+
Returns:
|
|
87
|
+
True when the module's top-level name resolves to a first-party package.
|
|
88
|
+
|
|
89
|
+
"""
|
|
90
|
+
top = module.partition(".")[0]
|
|
91
|
+
if not top or top in sys.stdlib_module_names:
|
|
92
|
+
return False
|
|
93
|
+
root = _project_root(path)
|
|
94
|
+
if root is None:
|
|
95
|
+
return False
|
|
96
|
+
return top in _first_party_roots(root)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def own_top_package(path: Path) -> str | None:
|
|
100
|
+
"""Return the name of the top-level package `path` itself belongs to.
|
|
101
|
+
|
|
102
|
+
The OUTERMOST importable ancestor wins rather than the outermost of an
|
|
103
|
+
unbroken `__init__.py` run, because PEP 420 namespace subpackages are
|
|
104
|
+
routine — bulbul's `agent/agent/lk/custom_models/` carries no `__init__.py`
|
|
105
|
+
while `agent/agent/` does, and a break-on-first-gap walk would report the
|
|
106
|
+
file as belonging to no package at all.
|
|
107
|
+
|
|
108
|
+
Returns:
|
|
109
|
+
The top-level package name, or None when `path` is not inside a package.
|
|
110
|
+
|
|
111
|
+
"""
|
|
112
|
+
resolved = _resolved(path)
|
|
113
|
+
if resolved is None:
|
|
114
|
+
return None
|
|
115
|
+
root = _project_root(resolved)
|
|
116
|
+
top: str | None = None
|
|
117
|
+
for ancestor in list(resolved.parents)[:_MAX_ANCESTORS]:
|
|
118
|
+
if (ancestor / "__init__.py").exists():
|
|
119
|
+
top = ancestor.name
|
|
120
|
+
if ancestor == root:
|
|
121
|
+
break
|
|
122
|
+
return top
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _resolved(path: Path) -> Path | None:
|
|
126
|
+
try:
|
|
127
|
+
return path.resolve()
|
|
128
|
+
except OSError:
|
|
129
|
+
return None
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
@lru_cache(maxsize=256)
|
|
133
|
+
def _project_root(path: Path) -> Path | None:
|
|
134
|
+
"""Locate the project boundary above `path`, memoized per file path.
|
|
135
|
+
|
|
136
|
+
Returns:
|
|
137
|
+
The project root directory, or None when no marker is found.
|
|
138
|
+
|
|
139
|
+
"""
|
|
140
|
+
resolved = _resolved(path)
|
|
141
|
+
if resolved is None:
|
|
142
|
+
return None
|
|
143
|
+
ancestors = list(resolved.parents)[:_MAX_ANCESTORS]
|
|
144
|
+
for ancestor in ancestors:
|
|
145
|
+
if (ancestor / _GIT_MARKER).exists():
|
|
146
|
+
return ancestor
|
|
147
|
+
# No VCS boundary: take the OUTERMOST directory of the unbroken run of
|
|
148
|
+
# pyproject.toml ancestors, which is the workspace root in a uv workspace.
|
|
149
|
+
outermost: Path | None = None
|
|
150
|
+
for ancestor in ancestors:
|
|
151
|
+
if not (ancestor / _PROJECT_MARKER).exists():
|
|
152
|
+
if outermost is not None:
|
|
153
|
+
break
|
|
154
|
+
continue
|
|
155
|
+
outermost = ancestor
|
|
156
|
+
return outermost
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
@lru_cache(maxsize=32)
|
|
160
|
+
def _first_party_roots(root: Path) -> frozenset[str]:
|
|
161
|
+
"""Collect the top-level package names declared anywhere under `root`.
|
|
162
|
+
|
|
163
|
+
Returns:
|
|
164
|
+
Every directory name under `root` that is an importable package.
|
|
165
|
+
|
|
166
|
+
"""
|
|
167
|
+
names: set[str] = set()
|
|
168
|
+
queue: list[tuple[Path, int]] = [(root, 0)]
|
|
169
|
+
scanned = 0
|
|
170
|
+
while queue:
|
|
171
|
+
directory, depth = queue.pop()
|
|
172
|
+
scanned += 1
|
|
173
|
+
if scanned > _MAX_DIRS_SCANNED:
|
|
174
|
+
break
|
|
175
|
+
try:
|
|
176
|
+
entries = sorted(directory.iterdir())
|
|
177
|
+
except OSError:
|
|
178
|
+
continue
|
|
179
|
+
for entry in entries:
|
|
180
|
+
if entry.name.startswith(".") or entry.name in _SKIP_DIR_NAMES:
|
|
181
|
+
continue
|
|
182
|
+
try:
|
|
183
|
+
if not entry.is_dir():
|
|
184
|
+
continue
|
|
185
|
+
is_package = (entry / "__init__.py").exists()
|
|
186
|
+
except OSError:
|
|
187
|
+
continue
|
|
188
|
+
if is_package:
|
|
189
|
+
names.add(entry.name)
|
|
190
|
+
elif depth + 1 < _MAX_SCAN_DEPTH:
|
|
191
|
+
queue.append((entry, depth + 1))
|
|
192
|
+
return frozenset(names)
|