@allansantos-dev/smart-tool 0.9.2 → 0.9.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +62 -0
- package/README.md +27 -2
- package/affected_tests.py +292 -0
- package/client_hooks.py +2 -2
- package/code_graph.py +75 -14
- package/code_graph_java.mjs +4 -1
- package/code_graph_js.cjs +15 -1
- package/code_impact.py +132 -0
- package/config.py +22 -0
- package/doc_check.py +256 -0
- package/duplicates.py +148 -13
- package/edit_preview.py +92 -0
- package/hook_decision.py +96 -2
- package/index_profile.py +1 -1
- package/package.json +1 -1
- package/router.py +4 -2
- package/setup_ui.py +53 -5
- package/smart_tool_daemon.py +108 -21
- package/version.py +1 -1
- package/web_adapters/browser/fetch_camoufox.py +16 -8
- package/web_adapters/browser/ranged_download.py +78 -0
- package/web_fetch.py +60 -3
- package/web_search_health.py +12 -4
package/code_impact.py
ADDED
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"""What changing a function touches, from the indexed code graph: where it is defined, who calls it, what it calls,
|
|
2
|
+
which tests reach it through static calls and which files import its module, each with the first line of its
|
|
3
|
+
docstring. Answers the question an agent asks before an edit without a chain of searches and file reads."""
|
|
4
|
+
import os
|
|
5
|
+
from collections import defaultdict
|
|
6
|
+
|
|
7
|
+
import code_graph
|
|
8
|
+
import doc_check
|
|
9
|
+
import index_profile
|
|
10
|
+
import index_scope
|
|
11
|
+
|
|
12
|
+
MAX_DEPTH = 4
|
|
13
|
+
NOTE = ("Static analysis of the indexed snapshot: calls made through callbacks, dynamic dispatch or names built at "
|
|
14
|
+
"runtime are not seen, so treat an empty list as 'none found', not as 'none exist'.")
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _matches(symbols, name):
|
|
18
|
+
path, _sep, wanted = name.strip().rpartition("::")
|
|
19
|
+
path = path.replace("\\", "/").removeprefix("./")
|
|
20
|
+
pool = [s for s in symbols if s["path"] == path] if path else symbols
|
|
21
|
+
exact = [s for s in pool if s["id"] == wanted or s["name"] == wanted]
|
|
22
|
+
return exact or [s for s in pool if s["name"].split(".")[-1] == wanted]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _site(by_id, symbol_id, line, of=None):
|
|
26
|
+
symbol = by_id.get(symbol_id) or {}
|
|
27
|
+
site = {"id": symbol_id, "function": symbol.get("name", symbol_id), "path": symbol.get("path"), "line": line}
|
|
28
|
+
if of:
|
|
29
|
+
site["of"] = of
|
|
30
|
+
return site
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def summaries(root, symbols, wanted_ids):
|
|
34
|
+
"""First docstring line of the wanted symbols, read from the working tree once per file."""
|
|
35
|
+
by_path = defaultdict(list)
|
|
36
|
+
for symbol in symbols:
|
|
37
|
+
by_path[symbol["path"]].append(symbol)
|
|
38
|
+
docs = {}
|
|
39
|
+
for path in {s["path"] for s in symbols if s["id"] in wanted_ids}:
|
|
40
|
+
try:
|
|
41
|
+
with open(os.path.join(root, path), encoding="utf-8") as stream:
|
|
42
|
+
source = stream.read()
|
|
43
|
+
except (OSError, UnicodeDecodeError):
|
|
44
|
+
continue
|
|
45
|
+
for function in doc_check.functions(path, source, by_path[path]):
|
|
46
|
+
if function["doc"]:
|
|
47
|
+
docs[(path, function["start"])] = function["doc"]
|
|
48
|
+
return {s["id"]: docs[(s["path"], s["start_line"])] for s in symbols
|
|
49
|
+
if s["id"] in wanted_ids and (s["path"], s["start_line"]) in docs}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def impact(root, symbol, view_id=None, depth=3, limit=30):
|
|
53
|
+
if not isinstance(symbol, str) or not symbol.strip():
|
|
54
|
+
raise ValueError("Pass symbol: a function, method (Class.method) or class name.")
|
|
55
|
+
if type(depth) is not int or not 1 <= depth <= MAX_DEPTH:
|
|
56
|
+
raise ValueError(f"depth must be an integer from 1 to {MAX_DEPTH}.")
|
|
57
|
+
data = code_graph.build(root, view_id)
|
|
58
|
+
found = _matches(data.get("symbols") or [], symbol)
|
|
59
|
+
diagnostics = [d.get("reason") for d in data.get("diagnostics") or [] if d.get("reason")]
|
|
60
|
+
if not found:
|
|
61
|
+
cause = f" Analysis problems: {'; '.join(diagnostics)}" if diagnostics else ""
|
|
62
|
+
raise ValueError(f"No function or class named {symbol!r} in the indexed view; check the name or reindex.{cause}")
|
|
63
|
+
profile = index_profile.current((index_scope.load_scope(root) or {}).get("profile"))
|
|
64
|
+
by_id = {s["id"]: s for s in data["symbols"]}
|
|
65
|
+
callers, callees = defaultdict(list), defaultdict(list)
|
|
66
|
+
for call in data["calls"]:
|
|
67
|
+
callers[call["target"]].append(call)
|
|
68
|
+
callees[call["source"]].append(call)
|
|
69
|
+
targets = {s["id"] for s in found}
|
|
70
|
+
label = {s["id"]: f"{s['path']}::{s['name']}" for s in found} if len(found) > 1 else {}
|
|
71
|
+
tests = {}
|
|
72
|
+
for origin in targets:
|
|
73
|
+
seen, frontier = set(targets), {origin}
|
|
74
|
+
for hops in range(1, depth + 1):
|
|
75
|
+
reached = set()
|
|
76
|
+
for target in frontier:
|
|
77
|
+
for call in callers[target]:
|
|
78
|
+
source = call["source"]
|
|
79
|
+
if source in seen:
|
|
80
|
+
continue
|
|
81
|
+
seen.add(source)
|
|
82
|
+
reached.add(source)
|
|
83
|
+
caller = by_id.get(source)
|
|
84
|
+
if caller and index_profile.kind(caller["path"], profile) == "test" and source not in tests:
|
|
85
|
+
tests[source] = {**_site(by_id, source, caller["start_line"], label.get(origin)), "hops": hops}
|
|
86
|
+
frontier = reached
|
|
87
|
+
direct = [_site(by_id, c["source"], c["line"], label.get(t)) for t in targets for c in callers[t]
|
|
88
|
+
if c["source"] not in targets]
|
|
89
|
+
outgoing = [_site(by_id, c["target"], c["line"], label.get(t)) for t in targets for c in callees[t]
|
|
90
|
+
if c["target"] not in targets]
|
|
91
|
+
files = {s["path"] for s in found}
|
|
92
|
+
importers = sorted({d["source"] for d in data["dependencies"] if d.get("target") in files and d["source"] not in files})
|
|
93
|
+
ordered_tests = sorted(tests.values(), key=lambda t: (t["hops"], t["path"], t["line"]))
|
|
94
|
+
shown = direct[:limit] + outgoing[:limit] + ordered_tests[:limit]
|
|
95
|
+
docs = summaries(root, data["symbols"], targets | {site["id"] for site in shown})
|
|
96
|
+
for site in shown:
|
|
97
|
+
doc = docs.get(site.pop("id"))
|
|
98
|
+
if doc:
|
|
99
|
+
site["doc"] = doc
|
|
100
|
+
definitions = []
|
|
101
|
+
for s in found:
|
|
102
|
+
entry = {"name": s["name"], "kind": s.get("kind"), "path": s["path"], "lines": f"{s['start_line']}-{s['end_line']}"}
|
|
103
|
+
if docs.get(s["id"]):
|
|
104
|
+
entry["doc"] = docs[s["id"]]
|
|
105
|
+
definitions.append(entry)
|
|
106
|
+
return {
|
|
107
|
+
"symbol": symbol.strip(),
|
|
108
|
+
"definitions": definitions,
|
|
109
|
+
"callers": direct[:limit],
|
|
110
|
+
"calls": outgoing[:limit],
|
|
111
|
+
"tests": ordered_tests[:limit],
|
|
112
|
+
"imported_by": importers[:limit],
|
|
113
|
+
"counts": {"callers": len(direct), "calls": len(outgoing), "tests": len(ordered_tests),
|
|
114
|
+
"imported_by": len(importers)},
|
|
115
|
+
"test_depth": depth,
|
|
116
|
+
"diagnostics": diagnostics,
|
|
117
|
+
"view": (data.get("selected") or {}).get("label"),
|
|
118
|
+
"note": NOTE + (f" {len(found)} definitions share this name; each entry says which one ('of'); pass "
|
|
119
|
+
"path::name to keep one." if label else ""),
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def for_agent(data, limit):
|
|
124
|
+
"""The file-focused graph without what an agent never reads (view list, display limits) and with lists capped."""
|
|
125
|
+
for key in ("symbols", "dependencies", "calls", "unresolved", "diagnostics"):
|
|
126
|
+
items = data.get(key) or []
|
|
127
|
+
if len(items) > limit:
|
|
128
|
+
data[key] = items[:limit]
|
|
129
|
+
data.setdefault("omitted", {})[key] = len(items) - limit
|
|
130
|
+
for key in ("views", "limits", "languages", "current_view", "generated_at", "cache_hit", "read_only", "source"):
|
|
131
|
+
data.pop(key, None)
|
|
132
|
+
return data
|
package/config.py
CHANGED
|
@@ -31,6 +31,8 @@ DEFAULT_CONFIG = {
|
|
|
31
31
|
"site_memory_similarity_threshold": 0.35,
|
|
32
32
|
"local_models": "fallback",
|
|
33
33
|
"hook_mode": "redirect",
|
|
34
|
+
"doc_mode": "remind",
|
|
35
|
+
"duplicate_mode": "warn",
|
|
34
36
|
"contact": "",
|
|
35
37
|
"model_adapter": "",
|
|
36
38
|
"model_adapter_options": {},
|
|
@@ -42,6 +44,26 @@ LOCAL_MODEL_MODES = ("fallback", "off", "prefer")
|
|
|
42
44
|
# PreToolUse hook: redirect = deny the native tool and point to Smart Tool; advise = let it run and add the tip to the
|
|
43
45
|
# agent's context; off = no routing at all.
|
|
44
46
|
HOOK_MODES = ("redirect", "advise", "off")
|
|
47
|
+
# Edit hook on functions without a docstring: require = deny the edit until it adds one; remind = let it run and tell
|
|
48
|
+
# the agent; off = say nothing. Independent of hook_mode, which only routes searches and reads.
|
|
49
|
+
DOC_MODES = ("require", "remind", "off")
|
|
50
|
+
# Edit hook on functions that copy one already indexed (exact or near-identical body): warn = let the edit run and
|
|
51
|
+
# point to the existing function; off = say nothing. Never blocks: near matches are leads, not defects.
|
|
52
|
+
DUPLICATE_MODES = ("warn", "off")
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def doc_mode(cfg=None):
|
|
56
|
+
value = (cfg if cfg is not None else load_config()).get("doc_mode") or DEFAULT_CONFIG["doc_mode"]
|
|
57
|
+
if value not in DOC_MODES:
|
|
58
|
+
raise ValueError(f"Invalid doc_mode ({value!r}) in {CONFIG_PATH}: use require, remind or off.")
|
|
59
|
+
return value
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def duplicate_mode(cfg=None):
|
|
63
|
+
value = (cfg if cfg is not None else load_config()).get("duplicate_mode") or DEFAULT_CONFIG["duplicate_mode"]
|
|
64
|
+
if value not in DUPLICATE_MODES:
|
|
65
|
+
raise ValueError(f"Invalid duplicate_mode ({value!r}) in {CONFIG_PATH}: use warn or off.")
|
|
66
|
+
return value
|
|
45
67
|
|
|
46
68
|
|
|
47
69
|
def hook_mode(cfg=None):
|
package/doc_check.py
ADDED
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
"""Function docstrings: which functions a file has, whether each one is documented, and which ones an edit touched.
|
|
2
|
+
Python is read with ast; TypeScript/JavaScript and Java with the code graph analyzers, where the documentation is the
|
|
3
|
+
/** */ block right above the function. Used by the edit hook (missing or possibly stale docstrings), the docstring
|
|
4
|
+
coverage listing and the one-line summaries in graph and impact results."""
|
|
5
|
+
import ast
|
|
6
|
+
import difflib
|
|
7
|
+
import hashlib
|
|
8
|
+
import os
|
|
9
|
+
import re
|
|
10
|
+
import threading
|
|
11
|
+
from collections import Counter, OrderedDict, defaultdict
|
|
12
|
+
|
|
13
|
+
import code_graph
|
|
14
|
+
import index_profile
|
|
15
|
+
import index_scope
|
|
16
|
+
|
|
17
|
+
PYTHON = (".py",)
|
|
18
|
+
SCRIPT = (".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".mts", ".cts")
|
|
19
|
+
JAVA = (".java",)
|
|
20
|
+
SUPPORTED = PYTHON + SCRIPT + JAVA
|
|
21
|
+
MAX_PARSED = 64
|
|
22
|
+
_SPACE = re.compile(r"\s+")
|
|
23
|
+
_PARSED = OrderedDict()
|
|
24
|
+
_PARSED_LOCK = threading.Lock()
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def supported(path):
|
|
28
|
+
return str(path).lower().endswith(SUPPORTED)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _first_line(text):
|
|
32
|
+
return next((line.strip(" *") for line in text.strip().splitlines() if line.strip(" *")), "")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _public_name(name):
|
|
36
|
+
return not name.startswith("_") or name.startswith("__") and name.endswith("__")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _python_functions(source):
|
|
40
|
+
"""Python functions with "required" following pydocstyle publicity (public name, public parents, not nested in a
|
|
41
|
+
function) and PEP 698: an @override method inherits its documentation."""
|
|
42
|
+
tree = ast.parse(source)
|
|
43
|
+
found = []
|
|
44
|
+
|
|
45
|
+
def visit(node, owner, public, nested):
|
|
46
|
+
for child in ast.iter_child_nodes(node):
|
|
47
|
+
if isinstance(child, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
48
|
+
doc = ast.get_docstring(child)
|
|
49
|
+
returns = f" -> {ast.unparse(child.returns)}" if child.returns else ""
|
|
50
|
+
found.append({"name": f"{owner}{child.name}", "start": child.lineno, "end": child.end_lineno,
|
|
51
|
+
"documented": doc is not None, "doc": _first_line(doc or ""),
|
|
52
|
+
"signature": f"{child.name}({ast.unparse(child.args)}){returns}",
|
|
53
|
+
"required": public and not nested and _public_name(child.name)
|
|
54
|
+
and not any(ast.unparse(d).split(".")[-1] == "override"
|
|
55
|
+
for d in child.decorator_list)})
|
|
56
|
+
visit(child, f"{owner}{child.name}.", False, True)
|
|
57
|
+
elif isinstance(child, ast.ClassDef):
|
|
58
|
+
visit(child, f"{owner}{child.name}.", public and _public_name(child.name), nested)
|
|
59
|
+
visit(tree, "", True, False)
|
|
60
|
+
return found
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _overload_head(lines, index, name):
|
|
64
|
+
"""Line where the overload signature ending at index starts, or None when the lines above are not an overload of
|
|
65
|
+
name (a declaration ending in ';')."""
|
|
66
|
+
head = re.compile(rf"^\s*(export\s+)?(default\s+)?(declare\s+)?(public\s+|protected\s+)?(static\s+)?(async\s+)?"
|
|
67
|
+
rf"(function\s+)?{re.escape(name)}\s*[<(]")
|
|
68
|
+
for k in range(index, max(index - 30, -1), -1):
|
|
69
|
+
line = lines[k].rstrip()
|
|
70
|
+
if k != index and (line.endswith(("*/", "}", ";")) or not line):
|
|
71
|
+
return None
|
|
72
|
+
if head.match(lines[k]):
|
|
73
|
+
return k
|
|
74
|
+
return None
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _block_above(lines, start, name=None):
|
|
78
|
+
"""Text of the /** */ block above the function starting at line start. With name (TypeScript/JavaScript), uncommented
|
|
79
|
+
overload signatures in between are skipped, as eslint-plugin-jsdoc does by default."""
|
|
80
|
+
index = start - 2
|
|
81
|
+
while index >= 0:
|
|
82
|
+
if not lines[index].strip() or lines[index].lstrip().startswith("@"):
|
|
83
|
+
index -= 1
|
|
84
|
+
continue
|
|
85
|
+
head = _overload_head(lines, index, name) if name and lines[index].rstrip().endswith(";") else None
|
|
86
|
+
if head is None:
|
|
87
|
+
break
|
|
88
|
+
index = head - 1
|
|
89
|
+
if index < 0 or not lines[index].rstrip().endswith("*/"):
|
|
90
|
+
return None
|
|
91
|
+
end = index
|
|
92
|
+
while index >= 0 and "/*" not in lines[index]:
|
|
93
|
+
index -= 1
|
|
94
|
+
if index < 0 or "/**" not in lines[index]:
|
|
95
|
+
return None
|
|
96
|
+
return "\n".join(line.strip().removeprefix("/**").removesuffix("*/").strip().removeprefix("*").strip()
|
|
97
|
+
for line in lines[index:end + 1])
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _annotations(lines, start, end):
|
|
101
|
+
"""Annotation lines right above the function and at its first lines (the Java analyzer starts a method at its
|
|
102
|
+
annotations), and the first line that is not an annotation."""
|
|
103
|
+
index, found = start - 2, []
|
|
104
|
+
while index >= 0 and (not lines[index].strip() or lines[index].lstrip().startswith("@")):
|
|
105
|
+
found.append(lines[index].strip())
|
|
106
|
+
index -= 1
|
|
107
|
+
index = start - 1
|
|
108
|
+
while index < min(end, len(lines)) and lines[index].lstrip().startswith("@"):
|
|
109
|
+
found.append(lines[index].strip())
|
|
110
|
+
index += 1
|
|
111
|
+
return found, lines[index] if index < len(lines) else ""
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _required(path, symbol, parent, first_line, annotations):
|
|
115
|
+
"""Whether a braced-language function needs documentation by the linter defaults: Checkstyle MissingJavadocMethod
|
|
116
|
+
(no @Override, not private) and eslint-plugin-jsdoc require-jsdoc (named declarations and class methods, never
|
|
117
|
+
callbacks or functions nested in another function; private members and unexported arrow functions left out)."""
|
|
118
|
+
short = symbol["name"].split("(")[0].split(".")[-1]
|
|
119
|
+
if short in ("callback", "anonymous") or not _public_name(short) or short.startswith("#"):
|
|
120
|
+
return False
|
|
121
|
+
if re.search(r"\bprivate\b", first_line):
|
|
122
|
+
return False
|
|
123
|
+
if path.lower().endswith(JAVA):
|
|
124
|
+
return not any(a.startswith("@Override") for a in annotations)
|
|
125
|
+
if parent and parent.get("kind") in ("function", "method"):
|
|
126
|
+
return False
|
|
127
|
+
if symbol.get("kind") == "method" or parent and parent.get("kind") in ("class", "interface"):
|
|
128
|
+
return True
|
|
129
|
+
return bool(re.match(r"\s*(export\s+)?(default\s+)?(declare\s+)?(async\s+)?function\b", first_line)
|
|
130
|
+
or re.match(r"\s*export\b", first_line))
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _braced_functions(path, source, symbols):
|
|
134
|
+
lines = source.splitlines()
|
|
135
|
+
by_id = {s.get("id"): s for s in symbols}
|
|
136
|
+
script = path.lower().endswith(SCRIPT)
|
|
137
|
+
found = []
|
|
138
|
+
for symbol in symbols:
|
|
139
|
+
if symbol.get("kind") not in ("function", "method") or "@" in symbol["name"]:
|
|
140
|
+
continue
|
|
141
|
+
start, end = symbol["start_line"], symbol["end_line"]
|
|
142
|
+
short = symbol["name"].split("(")[0].split(".")[-1]
|
|
143
|
+
block = _block_above(lines, start, short if script else None)
|
|
144
|
+
header = " ".join(lines[start - 1:min(end, start + 8)])
|
|
145
|
+
annotations, first_line = _annotations(lines, start, end)
|
|
146
|
+
required = _required(path, symbol, by_id.get(symbol.get("parent")), first_line, annotations)
|
|
147
|
+
found.append({"name": symbol["name"], "start": start, "end": end, "documented": block is not None,
|
|
148
|
+
"doc": _first_line(block or ""), "signature": _SPACE.sub(" ", header.split("{")[0]).strip(),
|
|
149
|
+
"required": required})
|
|
150
|
+
return found
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def functions(path, source, symbols=None):
|
|
154
|
+
"""Functions of one file with start/end lines, documented flag, first doc line and normalized signature. symbols
|
|
155
|
+
are the code graph symbols of that file when already known; otherwise the analyzer runs on source."""
|
|
156
|
+
lower = path.lower()
|
|
157
|
+
if lower.endswith(PYTHON):
|
|
158
|
+
try:
|
|
159
|
+
return _python_functions(source)
|
|
160
|
+
except SyntaxError:
|
|
161
|
+
return []
|
|
162
|
+
if symbols is None:
|
|
163
|
+
symbols = analyze(path, source)[0]
|
|
164
|
+
return _braced_functions(path, source, symbols)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def analyze(path, *sources):
|
|
168
|
+
"""Analyzer symbols of several versions of one TypeScript/JavaScript/Java file, from one analyzer run for the
|
|
169
|
+
versions not seen yet (an edit hook needs the file before and after; the next edit's 'before' is this 'after')."""
|
|
170
|
+
lower, name = path.lower(), os.path.basename(path)
|
|
171
|
+
keys = [hashlib.sha1(f"{name}\0{source}".encode("utf-8")).hexdigest() for source in sources]
|
|
172
|
+
with _PARSED_LOCK:
|
|
173
|
+
missing = {key: source for key, source in zip(keys, sources) if key not in _PARSED}
|
|
174
|
+
if missing:
|
|
175
|
+
batch = {f"v{i}/{name}": source for i, source in enumerate(missing.values())}
|
|
176
|
+
analysis = (code_graph._javascript_graph(batch) if lower.endswith(SCRIPT)
|
|
177
|
+
else code_graph._java_graph(batch) if lower.endswith(JAVA) else {})
|
|
178
|
+
found = {key: [dict(s, path=name) for s in analysis.get("symbols") or [] if s.get("path") == f"v{i}/{name}"]
|
|
179
|
+
for i, key in enumerate(missing)}
|
|
180
|
+
with _PARSED_LOCK:
|
|
181
|
+
_PARSED.update(found)
|
|
182
|
+
while len(_PARSED) > MAX_PARSED:
|
|
183
|
+
_PARSED.popitem(last=False)
|
|
184
|
+
with _PARSED_LOCK:
|
|
185
|
+
return [_PARSED[key] if key in _PARSED else [] for key in keys]
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _touched_lines(before, after):
|
|
189
|
+
old, new = (before or "").splitlines(), after.splitlines()
|
|
190
|
+
touched = set()
|
|
191
|
+
for tag, _i1, _i2, j1, j2 in difflib.SequenceMatcher(None, old, new, autojunk=False).get_opcodes():
|
|
192
|
+
if tag != "equal":
|
|
193
|
+
touched.update(range(j1 + 1, max(j1 + 1, j2) + 1))
|
|
194
|
+
return touched
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def written(path, before, after):
|
|
198
|
+
"""Functions of the edited file after the edit ('all') and the ones the edit touched ('touched')."""
|
|
199
|
+
if path.lower().endswith(SCRIPT + JAVA):
|
|
200
|
+
analyze(path, *(source for source in (before, after) if source is not None))
|
|
201
|
+
found = functions(path, after) if supported(path) else []
|
|
202
|
+
lines = _touched_lines(before, after)
|
|
203
|
+
return {"all": found, "touched": [f for f in found if lines.intersection(range(f["start"], f["end"] + 1))]}
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def review(path, before, after, edited=None):
|
|
207
|
+
"""Functions the edit from before to after touched that are left without a docstring although the language
|
|
208
|
+
convention expects one ('missing': public, not nested, not an override) or keep one while their signature changed
|
|
209
|
+
('signature_changed'). Untouched functions and body-only edits are not reported. edited is written() of the same
|
|
210
|
+
edit when the caller already has it."""
|
|
211
|
+
if not supported(path):
|
|
212
|
+
return []
|
|
213
|
+
edited = edited or written(path, before, after)
|
|
214
|
+
if not edited["touched"]:
|
|
215
|
+
return []
|
|
216
|
+
old = {f["name"]: f for f in functions(path, before)} if before else {}
|
|
217
|
+
findings = []
|
|
218
|
+
for function in edited["touched"]:
|
|
219
|
+
previous = old.get(function["name"])
|
|
220
|
+
if not function["documented"]:
|
|
221
|
+
if function["required"]:
|
|
222
|
+
findings.append({**function, "issue": "missing"})
|
|
223
|
+
elif previous and _SPACE.sub("", previous["signature"]) != _SPACE.sub("", function["signature"]):
|
|
224
|
+
findings.append({**function, "issue": "signature_changed"})
|
|
225
|
+
return findings
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def coverage(root, include_tests=False, limit=30, view_id=None):
|
|
229
|
+
"""Docstring coverage of the indexed code: functions the language convention expects documented (plus any already
|
|
230
|
+
documented) and those still missing, file by file, so an agent can document a project that started without it.
|
|
231
|
+
Tests are left out unless include_tests."""
|
|
232
|
+
data = code_graph.build(root, view_id)
|
|
233
|
+
profile = index_profile.current((index_scope.load_scope(root) or {}).get("profile"))
|
|
234
|
+
kinds = ("code", "test") if include_tests else ("code",)
|
|
235
|
+
by_path = defaultdict(list)
|
|
236
|
+
for symbol in data.get("symbols") or []:
|
|
237
|
+
if supported(symbol["path"]) and index_profile.kind(symbol["path"], profile) in kinds:
|
|
238
|
+
by_path[symbol["path"]].append(symbol)
|
|
239
|
+
total, missing = 0, []
|
|
240
|
+
for path in sorted(by_path):
|
|
241
|
+
try:
|
|
242
|
+
with open(os.path.join(root, path), encoding="utf-8") as stream:
|
|
243
|
+
source = stream.read()
|
|
244
|
+
except (OSError, UnicodeDecodeError):
|
|
245
|
+
continue
|
|
246
|
+
found = [f for f in functions(path, source, by_path[path]) if f["required"] or f["documented"]]
|
|
247
|
+
total += len(found)
|
|
248
|
+
missing += [{"path": path, "function": f["name"], "line": f["start"]} for f in found if not f["documented"]]
|
|
249
|
+
per_file = Counter(item["path"] for item in missing)
|
|
250
|
+
return {"functions": total, "documented": total - len(missing),
|
|
251
|
+
"coverage_percent": round(100 * (total - len(missing)) / total, 1) if total else None,
|
|
252
|
+
"missing_total": len(missing), "missing": missing[:limit],
|
|
253
|
+
"files_with_most_missing": [{"path": p, "missing": n} for p, n in per_file.most_common(10)],
|
|
254
|
+
"include_tests": include_tests,
|
|
255
|
+
"diagnostics": [d.get("reason") for d in data.get("diagnostics") or [] if d.get("reason")],
|
|
256
|
+
"view": (data.get("selected") or {}).get("label")}
|
package/duplicates.py
CHANGED
|
@@ -25,8 +25,35 @@ EMBED_SLICE = 64
|
|
|
25
25
|
CUTOFF_NOTE = ("Default semantic cutoff 0.90: in the 2026-10-02 measurement (claude-code, independent judge) pairs above "
|
|
26
26
|
"0.95 were 10/10 actionable and those from 0.80 to 0.94 only 7/20; below 0.90 most are false positives "
|
|
27
27
|
"and reviewing them costs time without leading to a refactor.")
|
|
28
|
+
MIN_TOKENS = 50
|
|
29
|
+
NEAR_MIN_TOKENS = 70
|
|
30
|
+
SHINGLE = 5
|
|
31
|
+
NEAR_NOTE = ("Near = at least 0.80 of 5-token sequences shared with keywords, called names, attributes and types kept "
|
|
32
|
+
"and only local names, strings and numbers abstracted; in the 2026-10-08 measurement on Python, "
|
|
33
|
+
"JavaScript, TypeScript and Java projects it caught every exact and locally renamed copy and 95-100% of "
|
|
34
|
+
"copies with one added statement, and fired on 0-1.6% of real functions.")
|
|
28
35
|
_HASH_COMMENT = {"py", "rb", "sh"}
|
|
29
|
-
_TOKEN = re.compile(r"[A-Za-z_]\w*|\d
|
|
36
|
+
_TOKEN = re.compile(r"\"(?:\\.|[^\"\\\n])*\"|'(?:\\.|[^'\\\n])*'|`(?:\\.|[^`\\])*`|[A-Za-z_$][\w$]*|\d[\w.]*|[^\s\w]")
|
|
37
|
+
_KEYWORDS = {
|
|
38
|
+
"py": "False None True and as assert async await break class continue def del elif else except finally for from "
|
|
39
|
+
"global if import in is lambda nonlocal not or pass raise return try while with yield self cls",
|
|
40
|
+
"js": "await break case catch class const continue debugger default delete do else export extends false finally "
|
|
41
|
+
"for function if import in instanceof let new null of return super switch this throw true try typeof "
|
|
42
|
+
"undefined var void while with yield async",
|
|
43
|
+
"java": "abstract assert boolean break byte case catch char class const continue default do double else enum "
|
|
44
|
+
"extends final finally float for if implements import instanceof int interface long native new null "
|
|
45
|
+
"package private protected public return short static super switch synchronized this throw throws "
|
|
46
|
+
"transient try void volatile while true false var",
|
|
47
|
+
}
|
|
48
|
+
_KEYWORDS["ts"] = _KEYWORDS["js"] + " as interface type enum implements private public protected readonly keyof never unknown any"
|
|
49
|
+
_KEYWORDS = {lang: frozenset(words.split()) for lang, words in _KEYWORDS.items()}
|
|
50
|
+
_LANG = {"py": "py", "js": "js", "jsx": "js", "mjs": "js", "cjs": "js", "ts": "ts", "tsx": "ts", "mts": "ts",
|
|
51
|
+
"cts": "ts", "java": "java"}
|
|
52
|
+
_CONTRACT = re.compile(r"^(__\w+__|equals|hashCode|toString|compareTo|clone|close|run|call|apply|get|set|render|"
|
|
53
|
+
r"constructor|ng[A-Z]\w*|componentDid\w+|main)$")
|
|
54
|
+
_DOCSTRING = re.compile(r"\A\s*[rRuUbB]{0,2}(\"\"\"|''')[\s\S]*?\1")
|
|
55
|
+
_GENERATED = re.compile(r"@generated|DO NOT EDIT|Code generated by|auto-generated", re.I)
|
|
56
|
+
_WRITE_POOLS = {}
|
|
30
57
|
_CONTAINER_KINDS = {"module", "template", "class", "interface"}
|
|
31
58
|
_GENERIC_NAMES = {"callback", "anonymous", "constructor"}
|
|
32
59
|
_WARMING = set()
|
|
@@ -64,18 +91,48 @@ def _strip_comments(text, ext):
|
|
|
64
91
|
|
|
65
92
|
|
|
66
93
|
def _body(text, ext):
|
|
94
|
+
"""Function body without the signature and, in Python, without the leading docstring, so two copies that differ
|
|
95
|
+
only in documentation compare as identical."""
|
|
67
96
|
if ext in _HASH_COMMENT:
|
|
68
97
|
lines = text.splitlines()
|
|
69
98
|
start = next((k for k, line in enumerate(lines) if re.match(r"\s*(async\s+)?def\s", line)), 0)
|
|
70
99
|
end = next((k for k in range(start, len(lines)) if lines[k].rstrip().endswith(":")), start)
|
|
71
|
-
return "\n".join(lines[end + 1:])
|
|
100
|
+
return _DOCSTRING.sub("", "\n".join(lines[end + 1:]), count=1)
|
|
72
101
|
brace = text.find("{")
|
|
73
102
|
return text[brace + 1:] if brace >= 0 else text
|
|
74
103
|
|
|
75
104
|
|
|
76
|
-
def
|
|
77
|
-
|
|
78
|
-
|
|
105
|
+
def _features(body, ext, name):
|
|
106
|
+
"""Near-duplicate features of a function body: 5-token shingles where keywords, called names, attribute names and
|
|
107
|
+
type names stay and local names, strings and numbers become placeholders, so parallel functions calling different
|
|
108
|
+
APIs (get_stdin/get_stdout, rotateLeft/rotateRight) do not match while a copy with renamed locals does."""
|
|
109
|
+
words = _KEYWORDS.get(_LANG.get(ext), frozenset())
|
|
110
|
+
raw = _TOKEN.findall(body)
|
|
111
|
+
tokens = []
|
|
112
|
+
for k, t in enumerate(raw):
|
|
113
|
+
if t[0] in "\"'`":
|
|
114
|
+
tokens.append("STR")
|
|
115
|
+
elif t[0].isdigit():
|
|
116
|
+
tokens.append("NUM")
|
|
117
|
+
elif re.match(r"[A-Za-z_$]", t):
|
|
118
|
+
kept = (t in words or t[0].isupper() or (k and raw[k - 1] == ".")
|
|
119
|
+
or (k + 1 < len(raw) and raw[k + 1] == "("))
|
|
120
|
+
tokens.append(t if kept else "ID")
|
|
121
|
+
else:
|
|
122
|
+
tokens.append(t)
|
|
123
|
+
return {"shingles": frozenset(tuple(tokens[i:i + SHINGLE]) for i in range(max(len(tokens) - SHINGLE + 1, 1))),
|
|
124
|
+
"tokens": len(tokens), "lang": _LANG.get(ext, ext),
|
|
125
|
+
"near_ok": len(tokens) >= NEAR_MIN_TOKENS and not _CONTRACT.match(name)}
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _idioms(functions):
|
|
129
|
+
"""Shingles shared by so many functions that they are language idioms (if err != nil, try/except), not copies."""
|
|
130
|
+
counts = {}
|
|
131
|
+
for f in functions:
|
|
132
|
+
for shingle in f["shingles"]:
|
|
133
|
+
counts[shingle] = counts.get(shingle, 0) + 1
|
|
134
|
+
cut = max(100, int(0.005 * len(functions)))
|
|
135
|
+
return frozenset(s for s, c in counts.items() if c > cut)
|
|
79
136
|
|
|
80
137
|
|
|
81
138
|
def _index_state(path):
|
|
@@ -115,10 +172,12 @@ def _functions(root, include_tests):
|
|
|
115
172
|
text = "\n".join(file_lines[s["start_line"] - 1:s["end_line"]])
|
|
116
173
|
if not text.strip():
|
|
117
174
|
continue
|
|
118
|
-
|
|
119
|
-
|
|
175
|
+
body = _strip_comments(_body(text, ext), ext)
|
|
176
|
+
normalized = re.sub(r"\s+", " ", body).strip()
|
|
177
|
+
name = s["name"].split("(")[0].split(".")[-1]
|
|
178
|
+
functions.append({"name": name, "path": path, "line": s["start_line"],
|
|
120
179
|
"end": s["end_line"], "lines": s["end_line"] - s["start_line"] + 1, "text": text[:MAX_TEXT],
|
|
121
|
-
"hash": hashlib.sha1(normalized.encode("utf-8")).hexdigest(),
|
|
180
|
+
"hash": hashlib.sha1(normalized.encode("utf-8")).hexdigest(), **_features(body, ext, name)})
|
|
122
181
|
occurrence = occurrences[functions[-1]["hash"]] = occurrences.get(functions[-1]["hash"], 0) + 1
|
|
123
182
|
functions[-1]["kind"] = kind
|
|
124
183
|
functions[-1]["fp"] = hashlib.sha1(f"{path}\0{functions[-1]['hash']}\0{occurrence}".encode("utf-8")).hexdigest()
|
|
@@ -183,16 +242,24 @@ def _warm(root, view, functions, embed):
|
|
|
183
242
|
_WARMING.discard(root)
|
|
184
243
|
|
|
185
244
|
|
|
245
|
+
def _similar(fa, fb):
|
|
246
|
+
"""Whether two functions may be compared as near copies: same language, not overloads of one name in one file."""
|
|
247
|
+
return fa["lang"] == fb["lang"] and fa["hash"] != fb["hash"] and not (fa["path"] == fb["path"]
|
|
248
|
+
and fa["name"] == fb["name"])
|
|
249
|
+
|
|
250
|
+
|
|
186
251
|
def _near(functions):
|
|
187
|
-
|
|
252
|
+
common = _idioms(functions)
|
|
253
|
+
cut = [f["shingles"] - common if f["near_ok"] else None for f in functions]
|
|
254
|
+
order = sorted((i for i in range(len(functions)) if cut[i]), key=lambda i: len(cut[i]))
|
|
188
255
|
found = []
|
|
189
256
|
for position, a in enumerate(order):
|
|
190
|
-
sa =
|
|
257
|
+
sa = cut[a]
|
|
191
258
|
for b in order[position + 1:]:
|
|
192
|
-
sb =
|
|
259
|
+
sb = cut[b]
|
|
193
260
|
if len(sb) * NEAR_JACCARD > len(sa):
|
|
194
261
|
break
|
|
195
|
-
if functions[a]
|
|
262
|
+
if not _similar(functions[a], functions[b]):
|
|
196
263
|
continue
|
|
197
264
|
inter = len(sa & sb)
|
|
198
265
|
score = inter / (len(sa) + len(sb) - inter)
|
|
@@ -296,7 +363,7 @@ def find(root, embed, configured_model, min_similarity=DEFAULT_MIN_SIMILARITY, i
|
|
|
296
363
|
remembered.pop(next(iter(remembered)))
|
|
297
364
|
result = {
|
|
298
365
|
"view": view["label"], "functions_analyzed": len(functions), "include_tests": include_tests,
|
|
299
|
-
"min_similarity": min_similarity, "cutoff_note": CUTOFF_NOTE, "semantic_status": status, "notes": notes,
|
|
366
|
+
"min_similarity": min_similarity, "cutoff_note": CUTOFF_NOTE, "near_note": NEAR_NOTE, "semantic_status": status, "notes": notes,
|
|
300
367
|
"counts": {"exact_groups": len(exact), "near_pairs": len(near), "semantic_pairs": semantic_total,
|
|
301
368
|
"dismissed_hidden": len(hidden) + len(hidden_semantic)},
|
|
302
369
|
"exact": [{"id": fid, "lines": g[0]["lines"], "copies": [_ref(f) for f in g]} for fid, g in exact[:limit]],
|
|
@@ -311,3 +378,71 @@ def find(root, embed, configured_model, min_similarity=DEFAULT_MIN_SIMILARITY, i
|
|
|
311
378
|
"members": [_ref(skip[fid][1]), _ref(skip[fid][2])], **dismissed[fid]}
|
|
312
379
|
for fid in hidden_semantic])
|
|
313
380
|
return result
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def _write_pool(root):
|
|
384
|
+
"""Indexed production functions with their idiom shingles, rebuilt only when the index changes; None while the
|
|
385
|
+
code graph is not cached yet, so an edit hook never waits for an analysis."""
|
|
386
|
+
view_path = indexer.existing_db_path(root)
|
|
387
|
+
if not view_path or code_graph.cached_symbols(root)[0] is None:
|
|
388
|
+
return None
|
|
389
|
+
state = (view_path, *_index_state(view_path))
|
|
390
|
+
with _LOCK:
|
|
391
|
+
cached = _WRITE_POOLS.get(root)
|
|
392
|
+
if cached and cached[0] == state:
|
|
393
|
+
return cached[1]
|
|
394
|
+
functions, _notes, _view = _functions(root, False)
|
|
395
|
+
production = [f for f in functions if f["kind"] == "code"]
|
|
396
|
+
pool = (production, _idioms(production))
|
|
397
|
+
with _LOCK:
|
|
398
|
+
_WRITE_POOLS[root] = (state, pool)
|
|
399
|
+
return pool
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
def on_write(root, path, before, after, written):
|
|
403
|
+
"""Functions an edit writes whose body copies one already indexed: exact (same tokens, comments and spacing
|
|
404
|
+
ignored) or near (see NEAR_NOTE). written lists the touched functions as doc_check.functions returns them. The
|
|
405
|
+
function being edited in place, overloads of one name and functions the edit removes from the file are skipped."""
|
|
406
|
+
ext = path.rsplit(".", 1)[-1].lower() if "." in path else ""
|
|
407
|
+
rel = os.path.relpath(path, root).replace(os.sep, "/")
|
|
408
|
+
profile = index_profile.current((index_scope.load_scope(root) or {}).get("profile"))
|
|
409
|
+
if ext not in _LANG or index_profile.kind(rel, profile) != "code" or _GENERATED.search(after[:4000]):
|
|
410
|
+
return []
|
|
411
|
+
pool = _write_pool(root)
|
|
412
|
+
if pool is None:
|
|
413
|
+
return []
|
|
414
|
+
production, common = pool
|
|
415
|
+
lines = after.splitlines()
|
|
416
|
+
remaining = {f["name"].split("(")[0].split(".")[-1] for f in written["all"]}
|
|
417
|
+
matches = []
|
|
418
|
+
for function in written["touched"]:
|
|
419
|
+
name = function["name"].split("(")[0].split(".")[-1]
|
|
420
|
+
text = "\n".join(lines[function["start"] - 1:function["end"]])
|
|
421
|
+
body = _strip_comments(_body(text, ext), ext)
|
|
422
|
+
features = _features(body, ext, name)
|
|
423
|
+
if function["end"] - function["start"] + 1 < MIN_LINES or features["tokens"] < MIN_TOKENS \
|
|
424
|
+
or _CONTRACT.match(name):
|
|
425
|
+
continue
|
|
426
|
+
digest = hashlib.sha1(re.sub(r"\s+", " ", body).strip().encode("utf-8")).hexdigest()
|
|
427
|
+
mine = features["shingles"] - common
|
|
428
|
+
best = None
|
|
429
|
+
for other in production:
|
|
430
|
+
if other["lang"] != features["lang"] or other["path"] == rel and (other["name"] == name
|
|
431
|
+
or other["name"] not in remaining):
|
|
432
|
+
continue
|
|
433
|
+
if other["hash"] == digest:
|
|
434
|
+
best = ("exact", 1.0, other)
|
|
435
|
+
break
|
|
436
|
+
if not (features["near_ok"] and other["near_ok"]):
|
|
437
|
+
continue
|
|
438
|
+
theirs = other["shingles"] - common
|
|
439
|
+
if not mine or min(len(mine), len(theirs)) < NEAR_JACCARD * max(len(mine), len(theirs)):
|
|
440
|
+
continue
|
|
441
|
+
inter = len(mine & theirs)
|
|
442
|
+
score = inter / (len(mine) + len(theirs) - inter)
|
|
443
|
+
if score >= NEAR_JACCARD and (best is None or score > best[1]):
|
|
444
|
+
best = ("near", score, other)
|
|
445
|
+
if best:
|
|
446
|
+
matches.append({"name": name, "lines": f"{function['start']}-{function['end']}", "type": best[0],
|
|
447
|
+
"score": round(best[1], 2), "existing": _ref(best[2])})
|
|
448
|
+
return matches
|