py-harness-cli 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- finetune/__init__.py +1 -0
- finetune/agent_system.py +41 -0
- finetune/agent_traces.py +157 -0
- finetune/everyday.py +30 -0
- finetune/hf_ollama.py +158 -0
- finetune/huggingface_store.py +144 -0
- finetune/models.py +74 -0
- finetune/paths.py +11 -0
- finetune/python_vibe.py +788 -0
- finetune/splits.py +54 -0
- finetune/systems.py +9 -0
- harness/__init__.py +42 -0
- harness/__main__.py +8 -0
- harness/act/__init__.py +6 -0
- harness/act/autofix/__init__.py +110 -0
- harness/act/autofix/additions.py +217 -0
- harness/act/autofix/conflicts.py +124 -0
- harness/act/autofix/cover.py +419 -0
- harness/act/autofix/mechanical.py +151 -0
- harness/act/autofix/missing_imports.py +50 -0
- harness/act/autofix/moves.py +439 -0
- harness/act/autofix/names.py +339 -0
- harness/act/autofix/scaffold.py +224 -0
- harness/act/code.py +157 -0
- harness/act/gate.py +229 -0
- harness/act/parse.py +247 -0
- harness/act/patch_fix.py +138 -0
- harness/act/tools.py +244 -0
- harness/agent/__init__.py +11 -0
- harness/agent/dispatch.py +235 -0
- harness/agent/loop.py +699 -0
- harness/agent/options.py +144 -0
- harness/agent/policy.py +856 -0
- harness/agent/prompt.py +170 -0
- harness/cli.py +393 -0
- harness/editor_kit.py +265 -0
- harness/guard/__init__.py +6 -0
- harness/guard/fallbacks.py +6 -0
- harness/guard/loop_guard.py +57 -0
- harness/guard/python_vibe.py +68 -0
- harness/guard/run.py +41 -0
- harness/guard/types.py +19 -0
- harness/locate.py +767 -0
- harness/mcp_stdio.py +306 -0
- harness/memory/__init__.py +5 -0
- harness/memory/conversation.py +104 -0
- harness/model/__init__.py +6 -0
- harness/model/chat_backend.py +100 -0
- harness/model/engine.py +165 -0
- harness/model/ollama_generate.py +60 -0
- harness/model/openai_generate.py +156 -0
- harness/model/outbound.py +83 -0
- harness/model/route.py +90 -0
- harness/observe/__init__.py +6 -0
- harness/observe/eval_gate.py +80 -0
- harness/observe/eval_loop.py +185 -0
- harness/observe/eval_tasks.py +399 -0
- harness/observe/report_md.py +102 -0
- harness/observe/trace_record.py +79 -0
- harness/openai_api.py +81 -0
- harness/paths.py +88 -0
- harness/py.typed +0 -0
- harness/scan/__init__.py +6 -0
- harness/scan/app_spec.py +338 -0
- harness/scan/design.py +112 -0
- harness/scan/existing.py +131 -0
- harness/scan/layout.py +254 -0
- harness/scan/names.py +308 -0
- harness/scan/project_brief.py +287 -0
- harness/scan/project_docs.py +42 -0
- harness/scan/project_scan.py +49 -0
- harness/scan/repo_map.py +101 -0
- harness/secrets.py +39 -0
- harness/server.py +199 -0
- harness/ship/__init__.py +1 -0
- harness/ship/bot_pr.py +221 -0
- harness/ship/git_ship.py +262 -0
- harness/ship/identity.py +62 -0
- harness/ship/ticket.py +251 -0
- harness/skillkit/__init__.py +6 -0
- harness/skillkit/catalog.py +241 -0
- harness/skillkit/refuse_change.py +640 -0
- harness/skillkit/refuse_finish.py +295 -0
- harness/skillkit/target.py +238 -0
- harness/task.py +717 -0
- py_harness_cli-0.3.0.dist-info/METADATA +177 -0
- py_harness_cli-0.3.0.dist-info/RECORD +92 -0
- py_harness_cli-0.3.0.dist-info/WHEEL +5 -0
- py_harness_cli-0.3.0.dist-info/entry_points.txt +3 -0
- py_harness_cli-0.3.0.dist-info/licenses/LICENSE +202 -0
- py_harness_cli-0.3.0.dist-info/licenses/NOTICE +6 -0
- py_harness_cli-0.3.0.dist-info/top_level.txt +2 -0
harness/scan/layout.py
ADDED
|
@@ -0,0 +1,254 @@
|
|
|
1
|
+
"""Report why a project is difficult to read, and what to change first.
|
|
2
|
+
|
|
3
|
+
Four problems are detected, listed here in the order they are worth
|
|
4
|
+
fixing:
|
|
5
|
+
|
|
6
|
+
* `cycle` - two modules import each other, so neither can be read alone.
|
|
7
|
+
* `flat` - one directory holds many modules with no grouping.
|
|
8
|
+
* `god` - one module is much larger than the others around it.
|
|
9
|
+
* `no-tests` - the project contains no test files.
|
|
10
|
+
|
|
11
|
+
Only the first problem is turned into an instruction. A model given four
|
|
12
|
+
instructions at once tends to change four things at once; a model given one
|
|
13
|
+
instruction changes one thing, which can then be checked.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import ast
|
|
19
|
+
from dataclasses import dataclass
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
from harness.paths import rel_posix
|
|
23
|
+
from harness.scan.project_scan import SKIP_DIR
|
|
24
|
+
|
|
25
|
+
FLAT_MAX_MODULES = 12
|
|
26
|
+
# A module far bigger than the ones beside it. This is not the same
|
|
27
|
+
# thing as a god module and no longer says it is: `scan.design` calls a
|
|
28
|
+
# file with too many top-level functions a god module, and the two
|
|
29
|
+
# disagreed in both directions. A 300-byte file with four functions is
|
|
30
|
+
# one by that rule and not by this; a 7 KB file holding two long
|
|
31
|
+
# functions is one by this and not by that, and since the design review
|
|
32
|
+
# started reporting long functions, that case has a better answer.
|
|
33
|
+
OUTSIZED_RATIO = 3
|
|
34
|
+
OUTSIZED_MIN_BYTES = 6000
|
|
35
|
+
MAX_FINDINGS = 4
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True)
|
|
39
|
+
class Finding:
|
|
40
|
+
"""One structural problem found in a project.
|
|
41
|
+
|
|
42
|
+
Fields:
|
|
43
|
+
kind: "cycle", "flat", "outsized" or "no-tests".
|
|
44
|
+
detail: what was found, naming the files involved.
|
|
45
|
+
move: the change to make, written as an instruction.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
kind: str
|
|
49
|
+
detail: str
|
|
50
|
+
move: str
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _modules(project: Path) -> list[Path]:
|
|
54
|
+
root = project.resolve()
|
|
55
|
+
return [
|
|
56
|
+
path
|
|
57
|
+
for path in sorted(root.rglob("*.py"))
|
|
58
|
+
if not any(part in SKIP_DIR for part in path.parts)
|
|
59
|
+
]
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _module_name(path: Path, root: Path) -> str:
|
|
63
|
+
"""Dotted name for a file, as an import inside this project spells it."""
|
|
64
|
+
rel = path.resolve().relative_to(root).with_suffix("")
|
|
65
|
+
parts = list(rel.parts)
|
|
66
|
+
if parts and parts[-1] == "__init__":
|
|
67
|
+
parts.pop()
|
|
68
|
+
return ".".join(parts)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _local_imports(path: Path, root: Path) -> set[str]:
|
|
72
|
+
"""Dotted modules this file imports, relative imports resolved.
|
|
73
|
+
|
|
74
|
+
The first version kept only the last component of each import, and
|
|
75
|
+
the graph was keyed on the file name. Two things followed on a real
|
|
76
|
+
repository, where the same file name appears many times: distinct
|
|
77
|
+
modules merged into one node, and `rich.console` counted as an
|
|
78
|
+
import of a local `console.py`. Every cycle reported on a 4,580-file
|
|
79
|
+
project was one of those, four out of four.
|
|
80
|
+
"""
|
|
81
|
+
try:
|
|
82
|
+
tree = ast.parse(path.read_text(encoding="utf-8"))
|
|
83
|
+
except (OSError, SyntaxError, ValueError):
|
|
84
|
+
return set()
|
|
85
|
+
package = _module_name(path, root).rsplit(".", 1)
|
|
86
|
+
package = package[0] if len(package) == 2 else ""
|
|
87
|
+
names: set[str] = set()
|
|
88
|
+
for node in ast.walk(tree):
|
|
89
|
+
if isinstance(node, ast.ImportFrom):
|
|
90
|
+
if node.level:
|
|
91
|
+
# `from . import x` and `from .mod import y` are relative
|
|
92
|
+
# to the package this file sits in.
|
|
93
|
+
base = package.split(".") if package else []
|
|
94
|
+
climb = node.level - 1
|
|
95
|
+
base = base[: len(base) - climb] if climb else base
|
|
96
|
+
stem = ".".join(part for part in (*base, node.module or "") if part)
|
|
97
|
+
else:
|
|
98
|
+
stem = node.module or ""
|
|
99
|
+
if not stem:
|
|
100
|
+
continue
|
|
101
|
+
names.add(stem)
|
|
102
|
+
for alias in node.names:
|
|
103
|
+
names.add(f"{stem}.{alias.name}")
|
|
104
|
+
elif isinstance(node, ast.Import):
|
|
105
|
+
for alias in node.names:
|
|
106
|
+
names.add(alias.name)
|
|
107
|
+
return names
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def find_cycles(project: Path) -> list[tuple[str, str]]:
|
|
111
|
+
root = project.resolve()
|
|
112
|
+
modules = _modules(project)
|
|
113
|
+
by_name = {_module_name(path, root): path for path in modules}
|
|
114
|
+
|
|
115
|
+
# A sub-project keeps its own root: demo/orders/src/report.py is
|
|
116
|
+
# imported as `src.report` from inside demo/orders. Accept a dotted
|
|
117
|
+
# path that is the tail of exactly one module, so that still counts,
|
|
118
|
+
# while an ambiguous tail counts for nothing.
|
|
119
|
+
tails: dict[str, list[str]] = {}
|
|
120
|
+
for name in by_name:
|
|
121
|
+
parts = name.split(".")
|
|
122
|
+
for start in range(1, len(parts)):
|
|
123
|
+
tails.setdefault(".".join(parts[start:]), []).append(name)
|
|
124
|
+
|
|
125
|
+
def resolve(imported: set[str], self_name: str) -> set[str]:
|
|
126
|
+
found = set()
|
|
127
|
+
for item in imported:
|
|
128
|
+
if item in by_name:
|
|
129
|
+
found.add(item)
|
|
130
|
+
continue
|
|
131
|
+
# Only a dotted path. A bare top-level name competes with
|
|
132
|
+
# every package on the machine: `import logging` beside a
|
|
133
|
+
# local `infrastructure/logging/` package resolved to it and
|
|
134
|
+
# invented two cycles on a real repository.
|
|
135
|
+
if "." not in item:
|
|
136
|
+
continue
|
|
137
|
+
unique = tails.get(item, ())
|
|
138
|
+
if len(unique) == 1:
|
|
139
|
+
found.add(unique[0])
|
|
140
|
+
return found - {self_name}
|
|
141
|
+
|
|
142
|
+
graph = {
|
|
143
|
+
name: resolve(_local_imports(path, root), name)
|
|
144
|
+
for name, path in by_name.items()
|
|
145
|
+
}
|
|
146
|
+
pairs = {
|
|
147
|
+
tuple(sorted((name, other)))
|
|
148
|
+
for name, deps in graph.items()
|
|
149
|
+
for other in deps
|
|
150
|
+
if name in graph.get(other, set())
|
|
151
|
+
}
|
|
152
|
+
return sorted(
|
|
153
|
+
(rel_posix(by_name[left], root), rel_posix(by_name[right], root))
|
|
154
|
+
for left, right in pairs
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def find_flat_packages(project: Path) -> list[tuple[str, int]]:
|
|
159
|
+
root = project.resolve()
|
|
160
|
+
counts: dict[str, int] = {}
|
|
161
|
+
for path in _modules(project):
|
|
162
|
+
parent = path.parent
|
|
163
|
+
key = rel_posix(parent, root) if parent != root else "."
|
|
164
|
+
counts[key] = counts.get(key, 0) + 1
|
|
165
|
+
return sorted(
|
|
166
|
+
((name, n) for name, n in counts.items() if n > FLAT_MAX_MODULES),
|
|
167
|
+
key=lambda item: (-item[1], item[0]),
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def find_outsized_modules(project: Path) -> list[tuple[str, int]]:
|
|
172
|
+
root = project.resolve()
|
|
173
|
+
sizes = []
|
|
174
|
+
for path in _modules(project):
|
|
175
|
+
try:
|
|
176
|
+
sizes.append((rel_posix(path, root), path.stat().st_size))
|
|
177
|
+
except OSError:
|
|
178
|
+
continue
|
|
179
|
+
if len(sizes) < 3:
|
|
180
|
+
return []
|
|
181
|
+
median = sorted(size for _rel, size in sizes)[len(sizes) // 2]
|
|
182
|
+
return sorted(
|
|
183
|
+
(
|
|
184
|
+
(rel, size)
|
|
185
|
+
for rel, size in sizes
|
|
186
|
+
if size >= OUTSIZED_MIN_BYTES and size > median * OUTSIZED_RATIO
|
|
187
|
+
),
|
|
188
|
+
key=lambda item: -item[1],
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def has_tests(project: Path) -> bool:
|
|
193
|
+
root = project.resolve()
|
|
194
|
+
return any(
|
|
195
|
+
path.name.startswith("test_")
|
|
196
|
+
for path in root.rglob("test_*.py")
|
|
197
|
+
if not any(part in SKIP_DIR for part in path.parts)
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def review_layout(project: Path) -> list[Finding]:
|
|
202
|
+
out: list[Finding] = []
|
|
203
|
+
for left, right in find_cycles(project):
|
|
204
|
+
out.append(
|
|
205
|
+
Finding(
|
|
206
|
+
"cycle",
|
|
207
|
+
f"{left} and {right} import each other",
|
|
208
|
+
f"Move what they share into a new module both import. "
|
|
209
|
+
f"Action: grep Query: def .* Path: {left}",
|
|
210
|
+
)
|
|
211
|
+
)
|
|
212
|
+
for name, count in find_flat_packages(project):
|
|
213
|
+
out.append(
|
|
214
|
+
Finding(
|
|
215
|
+
"flat",
|
|
216
|
+
f"{name}/ holds {count} modules with no grouping",
|
|
217
|
+
f"Group {name}/ by what each module is for, one folder per "
|
|
218
|
+
"job, and give each folder an __init__.py that says so.",
|
|
219
|
+
)
|
|
220
|
+
)
|
|
221
|
+
for rel, size in find_outsized_modules(project):
|
|
222
|
+
out.append(
|
|
223
|
+
Finding(
|
|
224
|
+
"outsized",
|
|
225
|
+
f"{rel} is {size // 1024} KB — far larger than its neighbours",
|
|
226
|
+
f"Action: read Path: {rel} and split the one group of "
|
|
227
|
+
"functions that does not belong with the rest.",
|
|
228
|
+
)
|
|
229
|
+
)
|
|
230
|
+
if not has_tests(project):
|
|
231
|
+
out.append(
|
|
232
|
+
Finding(
|
|
233
|
+
"no-tests",
|
|
234
|
+
"no test_*.py anywhere in this project",
|
|
235
|
+
"Action: patch Path: tests/test_smoke.py with one unittest "
|
|
236
|
+
"for the function you touch next.",
|
|
237
|
+
)
|
|
238
|
+
)
|
|
239
|
+
return out[:MAX_FINDINGS]
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def render_layout(project: Path) -> str:
|
|
243
|
+
findings = review_layout(project)
|
|
244
|
+
if not findings:
|
|
245
|
+
return (
|
|
246
|
+
"layout: no cycles, no oversized package, no outsized module, tests "
|
|
247
|
+
"present. Nothing to restructure — do the task."
|
|
248
|
+
)
|
|
249
|
+
lines = [f"layout: {len(findings)} finding(s), worst first."]
|
|
250
|
+
for finding in findings:
|
|
251
|
+
lines.append(f" [{finding.kind}] {finding.detail}")
|
|
252
|
+
lines.append("")
|
|
253
|
+
lines.append(f"Next move (do only this one): {findings[0].move}")
|
|
254
|
+
return "\n".join(lines)
|
harness/scan/names.py
ADDED
|
@@ -0,0 +1,308 @@
|
|
|
1
|
+
"""Undefined-name scan. Deterministic compiler-style oracle. No model.
|
|
2
|
+
|
|
3
|
+
A small model will write `subtotl` next to `subtotal = ...` and then say
|
|
4
|
+
done. The existing test suite often does not call the broken function, so
|
|
5
|
+
`run` exits 0. This scan is the extra oracle: names that are loaded but
|
|
6
|
+
never bound, the way a compiler would complain.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import ast
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
def _builtin_names() -> set[str]:
|
|
15
|
+
raw = __builtins__
|
|
16
|
+
names = set(raw) if isinstance(raw, dict) else set(dir(raw))
|
|
17
|
+
# Module dunders are bound by the interpreter, not by the source.
|
|
18
|
+
return names | {
|
|
19
|
+
"Ellipsis", "NotImplemented", "False", "None", "True",
|
|
20
|
+
"__file__", "__name__", "__doc__", "__package__", "__spec__",
|
|
21
|
+
"__loader__", "__builtins__", "__debug__",
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
_BUILTINS = _builtin_names()
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _store_names(node: ast.AST) -> set[str]:
|
|
29
|
+
names: set[str] = set()
|
|
30
|
+
if isinstance(node, ast.Name):
|
|
31
|
+
names.add(node.id)
|
|
32
|
+
elif isinstance(node, (ast.Tuple, ast.List)):
|
|
33
|
+
for child in node.elts:
|
|
34
|
+
names.update(_store_names(child))
|
|
35
|
+
elif isinstance(node, ast.Starred):
|
|
36
|
+
names.update(_store_names(node.value))
|
|
37
|
+
return names
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
# 3.12 added `type X = ...`. Absent on 3.11, which this project supports.
|
|
41
|
+
_TYPE_ALIAS = getattr(ast, "TypeAlias", None)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _assign_names(node: ast.AST) -> set[str]:
|
|
45
|
+
if isinstance(node, ast.Assign):
|
|
46
|
+
names: set[str] = set()
|
|
47
|
+
for target in node.targets:
|
|
48
|
+
names.update(_store_names(target))
|
|
49
|
+
return names
|
|
50
|
+
if isinstance(node, ast.AnnAssign) and node.target:
|
|
51
|
+
return _store_names(node.target)
|
|
52
|
+
if isinstance(node, ast.AugAssign):
|
|
53
|
+
return _store_names(node.target)
|
|
54
|
+
if isinstance(node, ast.NamedExpr):
|
|
55
|
+
return _store_names(node.target)
|
|
56
|
+
if isinstance(node, (ast.For, ast.AsyncFor)):
|
|
57
|
+
return _store_names(node.target)
|
|
58
|
+
if isinstance(node, ast.withitem) and node.optional_vars:
|
|
59
|
+
return _store_names(node.optional_vars)
|
|
60
|
+
if isinstance(node, ast.ExceptHandler) and node.name:
|
|
61
|
+
return {node.name}
|
|
62
|
+
if isinstance(node, (ast.Import, ast.ImportFrom)):
|
|
63
|
+
found: set[str] = set()
|
|
64
|
+
for alias in node.names:
|
|
65
|
+
found.add(alias.asname or alias.name.split(".")[0])
|
|
66
|
+
return found
|
|
67
|
+
if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)):
|
|
68
|
+
return {node.name}
|
|
69
|
+
# `type Properties = dict[str, JsonValue]`, the 3.12 alias spelling.
|
|
70
|
+
# The node type does not exist before 3.12, and this project runs on
|
|
71
|
+
# 3.11 as well, so it is looked up rather than named.
|
|
72
|
+
if _TYPE_ALIAS is not None and isinstance(node, _TYPE_ALIAS):
|
|
73
|
+
return _store_names(node.name)
|
|
74
|
+
# `case InputSubmitted(text):` binds `text` for the branch body.
|
|
75
|
+
if isinstance(node, (ast.MatchAs, ast.MatchStar)) and node.name:
|
|
76
|
+
return {node.name}
|
|
77
|
+
if isinstance(node, ast.MatchMapping) and node.rest:
|
|
78
|
+
return {node.rest}
|
|
79
|
+
return set()
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _argument_names(args: ast.arguments) -> set[str]:
|
|
83
|
+
"""Every parameter name a signature binds."""
|
|
84
|
+
names = {
|
|
85
|
+
arg.arg
|
|
86
|
+
for arg in (*args.posonlyargs, *args.args, *args.kwonlyargs)
|
|
87
|
+
}
|
|
88
|
+
if args.vararg:
|
|
89
|
+
names.add(args.vararg.arg)
|
|
90
|
+
if args.kwarg:
|
|
91
|
+
names.add(args.kwarg.arg)
|
|
92
|
+
return names
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _type_param_names(node: ast.AST) -> set[str]:
|
|
96
|
+
"""Names bound by PEP 695 type parameters: `def tool[F](...)`.
|
|
97
|
+
|
|
98
|
+
Python 3.12 spelling, and the only place `F` or `T` is declared in a
|
|
99
|
+
file that uses it. Without this they read as undefined, which was
|
|
100
|
+
the largest group left in a real project after module scope was
|
|
101
|
+
handled properly.
|
|
102
|
+
"""
|
|
103
|
+
return {param.name for param in getattr(node, "type_params", []) or []}
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _function_bound(function: ast.FunctionDef | ast.AsyncFunctionDef) -> set[str]:
|
|
107
|
+
bound = {function.name} | _argument_names(function.args) | _type_param_names(function)
|
|
108
|
+
for node in ast.walk(function):
|
|
109
|
+
if node is function:
|
|
110
|
+
continue
|
|
111
|
+
if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)):
|
|
112
|
+
bound.add(node.name)
|
|
113
|
+
bound.update(_type_param_names(node))
|
|
114
|
+
# A nested function or lambda binds its own parameters. Only their
|
|
115
|
+
# names were collected, so every such parameter read looked
|
|
116
|
+
# undefined: `item`, `text`, `prompt` across this project's own code.
|
|
117
|
+
if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.Lambda)):
|
|
118
|
+
bound.update(_argument_names(node.args))
|
|
119
|
+
bound.update(_assign_names(node))
|
|
120
|
+
if isinstance(node, ast.comprehension):
|
|
121
|
+
bound.update(_store_names(node.target))
|
|
122
|
+
return bound
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _module_scope_bound(tree: ast.Module) -> set[str]:
|
|
126
|
+
"""Every name module scope binds, including inside `if` and `try`.
|
|
127
|
+
|
|
128
|
+
Only the top level of `tree.body` used to count, so the two most
|
|
129
|
+
common shapes in typed and cross-platform code read as undefined:
|
|
130
|
+
|
|
131
|
+
if TYPE_CHECKING:
|
|
132
|
+
from rich.markdown import Markdown # used in an annotation
|
|
133
|
+
|
|
134
|
+
try:
|
|
135
|
+
import termios # POSIX only
|
|
136
|
+
except ImportError:
|
|
137
|
+
termios = None
|
|
138
|
+
|
|
139
|
+
Both run fine. On a sample of 600 files from a real project, 18 were
|
|
140
|
+
reported as having an undefined name and these two shapes accounted
|
|
141
|
+
for them. Function and class bodies are separate scopes and are not
|
|
142
|
+
descended into.
|
|
143
|
+
"""
|
|
144
|
+
bound: set[str] = set()
|
|
145
|
+
|
|
146
|
+
def walk(body: list[ast.stmt]) -> None:
|
|
147
|
+
for node in body:
|
|
148
|
+
bound.update(_assign_names(node))
|
|
149
|
+
bound.update(_type_param_names(node))
|
|
150
|
+
if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)):
|
|
151
|
+
continue
|
|
152
|
+
for field in ("body", "orelse", "finalbody"):
|
|
153
|
+
inner = getattr(node, field, None)
|
|
154
|
+
if isinstance(inner, list):
|
|
155
|
+
walk([item for item in inner if isinstance(item, ast.stmt)])
|
|
156
|
+
for handler in getattr(node, "handlers", []) or []:
|
|
157
|
+
bound.update(_assign_names(handler))
|
|
158
|
+
walk(handler.body)
|
|
159
|
+
for item in getattr(node, "items", []) or []:
|
|
160
|
+
bound.update(_assign_names(item))
|
|
161
|
+
|
|
162
|
+
walk(tree.body)
|
|
163
|
+
return bound
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def undefined_names(source: str) -> list[str]:
|
|
167
|
+
"""Load-names in functions that are not bound in the module or the function."""
|
|
168
|
+
try:
|
|
169
|
+
tree = ast.parse(source)
|
|
170
|
+
except (SyntaxError, ValueError):
|
|
171
|
+
return []
|
|
172
|
+
module_bound = set(_BUILTINS) | _module_scope_bound(tree)
|
|
173
|
+
found: list[str] = []
|
|
174
|
+
seen: set[str] = set()
|
|
175
|
+
|
|
176
|
+
def scan(function: ast.AST, bound: set[str]) -> None:
|
|
177
|
+
inner = bound | _function_bound(function)
|
|
178
|
+
for child in ast.walk(function):
|
|
179
|
+
if not isinstance(child, ast.Name) or not isinstance(child.ctx, ast.Load):
|
|
180
|
+
continue
|
|
181
|
+
if child.id in inner or child.id in seen:
|
|
182
|
+
continue
|
|
183
|
+
seen.add(child.id)
|
|
184
|
+
found.append(child.id)
|
|
185
|
+
|
|
186
|
+
for node in tree.body:
|
|
187
|
+
if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
188
|
+
scan(node, module_bound)
|
|
189
|
+
elif isinstance(node, ast.ClassDef):
|
|
190
|
+
# Methods were never scanned, and every unittest test is one.
|
|
191
|
+
# A test calling a function it forgot to import looked clean.
|
|
192
|
+
class_bound = set(module_bound)
|
|
193
|
+
for member in node.body:
|
|
194
|
+
class_bound.update(_assign_names(member))
|
|
195
|
+
if isinstance(member, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)):
|
|
196
|
+
class_bound.add(member.name)
|
|
197
|
+
for member in node.body:
|
|
198
|
+
if isinstance(member, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
199
|
+
scan(member, class_bound)
|
|
200
|
+
return found
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def new_undefined(original: str, draft: str) -> list[str]:
|
|
204
|
+
"""Undefined names the draft added. Existing planted bugs are ignored."""
|
|
205
|
+
before = set(undefined_names(original))
|
|
206
|
+
return [name for name in undefined_names(draft) if name not in before]
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def undefined_in_file(path: Path) -> list[str]:
|
|
210
|
+
try:
|
|
211
|
+
source = path.read_text(encoding="utf-8")
|
|
212
|
+
except OSError:
|
|
213
|
+
return []
|
|
214
|
+
return undefined_names(source)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _module_file(project: Path, dotted: str) -> Path | None:
|
|
218
|
+
"""The file a `from a.b import c` refers to, if it is in this project."""
|
|
219
|
+
parts = dotted.split(".")
|
|
220
|
+
for candidate in (
|
|
221
|
+
project.joinpath(*parts).with_suffix(".py"),
|
|
222
|
+
project.joinpath(*parts, "__init__.py"),
|
|
223
|
+
project.joinpath(*parts[1:]).with_suffix(".py") if len(parts) > 1 else None,
|
|
224
|
+
):
|
|
225
|
+
if candidate is not None and candidate.is_file():
|
|
226
|
+
return candidate
|
|
227
|
+
return None
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _defined_in(source: str) -> set[str]:
|
|
231
|
+
try:
|
|
232
|
+
tree = ast.parse(source)
|
|
233
|
+
except (SyntaxError, ValueError):
|
|
234
|
+
return set()
|
|
235
|
+
names: set[str] = set()
|
|
236
|
+
for node in tree.body:
|
|
237
|
+
if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)):
|
|
238
|
+
names.add(node.name)
|
|
239
|
+
names.update(_assign_names(node))
|
|
240
|
+
if isinstance(node, (ast.Import, ast.ImportFrom)):
|
|
241
|
+
for alias in node.names:
|
|
242
|
+
names.add(alias.asname or alias.name.split(".")[0])
|
|
243
|
+
return names
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def missing_import_targets(project: Path, source: str) -> list[tuple[str, str]]:
|
|
247
|
+
"""Imports of names this project's own modules do not define.
|
|
248
|
+
|
|
249
|
+
A test that imports a function nobody has written yet reads as valid
|
|
250
|
+
Python — the import binds the name, so the undefined-name scan sees
|
|
251
|
+
nothing — and fails only when the suite runs.
|
|
252
|
+
"""
|
|
253
|
+
try:
|
|
254
|
+
tree = ast.parse(source)
|
|
255
|
+
except (SyntaxError, ValueError):
|
|
256
|
+
return []
|
|
257
|
+
missing: list[tuple[str, str]] = []
|
|
258
|
+
for node in ast.walk(tree):
|
|
259
|
+
if not isinstance(node, ast.ImportFrom) or node.level or not node.module:
|
|
260
|
+
continue
|
|
261
|
+
target = _module_file(Path(project), node.module)
|
|
262
|
+
if target is None:
|
|
263
|
+
continue
|
|
264
|
+
try:
|
|
265
|
+
defined = _defined_in(target.read_text(encoding="utf-8"))
|
|
266
|
+
except OSError:
|
|
267
|
+
continue
|
|
268
|
+
for alias in node.names:
|
|
269
|
+
if alias.name != "*" and alias.name not in defined:
|
|
270
|
+
missing.append((node.module, alias.name))
|
|
271
|
+
return missing
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
# Names a small model reaches for without importing them. The fix for
|
|
275
|
+
# these is an import line, never a rename.
|
|
276
|
+
_IMPORTABLE = {
|
|
277
|
+
"Path": "from pathlib import Path",
|
|
278
|
+
"PurePath": "from pathlib import PurePath",
|
|
279
|
+
"dataclass": "from dataclasses import dataclass",
|
|
280
|
+
"field": "from dataclasses import field",
|
|
281
|
+
"Counter": "from collections import Counter",
|
|
282
|
+
"defaultdict": "from collections import defaultdict",
|
|
283
|
+
"Any": "from typing import Any",
|
|
284
|
+
"Iterable": "from collections.abc import Iterable",
|
|
285
|
+
"Sequence": "from collections.abc import Sequence",
|
|
286
|
+
"Callable": "from collections.abc import Callable",
|
|
287
|
+
"datetime": "from datetime import datetime",
|
|
288
|
+
"date": "from datetime import date",
|
|
289
|
+
"timedelta": "from datetime import timedelta",
|
|
290
|
+
"os": "import os",
|
|
291
|
+
"sys": "import sys",
|
|
292
|
+
"re": "import re",
|
|
293
|
+
"json": "import json",
|
|
294
|
+
"csv": "import csv",
|
|
295
|
+
"math": "import math",
|
|
296
|
+
"shutil": "import shutil",
|
|
297
|
+
"subprocess": "import subprocess",
|
|
298
|
+
"tempfile": "import tempfile",
|
|
299
|
+
"zipfile": "import zipfile",
|
|
300
|
+
"tarfile": "import tarfile",
|
|
301
|
+
"urllib": "import urllib.request",
|
|
302
|
+
"unittest": "import unittest",
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def import_for(name: str) -> str:
|
|
307
|
+
"""The import line that binds `name`, if it is one of the usual ones."""
|
|
308
|
+
return _IMPORTABLE.get(name, "")
|