@allansantos-dev/smart-tool 0.9.2 → 0.9.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/code_impact.py ADDED
@@ -0,0 +1,132 @@
1
+ """What changing a function touches, from the indexed code graph: where it is defined, who calls it, what it calls,
2
+ which tests reach it through static calls and which files import its module, each with the first line of its
3
+ docstring. Answers the question an agent asks before an edit without a chain of searches and file reads."""
4
+ import os
5
+ from collections import defaultdict
6
+
7
+ import code_graph
8
+ import doc_check
9
+ import index_profile
10
+ import index_scope
11
+
12
+ MAX_DEPTH = 4
13
+ NOTE = ("Static analysis of the indexed snapshot: calls made through callbacks, dynamic dispatch or names built at "
14
+ "runtime are not seen, so treat an empty list as 'none found', not as 'none exist'.")
15
+
16
+
17
+ def _matches(symbols, name):
18
+ path, _sep, wanted = name.strip().rpartition("::")
19
+ path = path.replace("\\", "/").removeprefix("./")
20
+ pool = [s for s in symbols if s["path"] == path] if path else symbols
21
+ exact = [s for s in pool if s["id"] == wanted or s["name"] == wanted]
22
+ return exact or [s for s in pool if s["name"].split(".")[-1] == wanted]
23
+
24
+
25
+ def _site(by_id, symbol_id, line, of=None):
26
+ symbol = by_id.get(symbol_id) or {}
27
+ site = {"id": symbol_id, "function": symbol.get("name", symbol_id), "path": symbol.get("path"), "line": line}
28
+ if of:
29
+ site["of"] = of
30
+ return site
31
+
32
+
33
+ def summaries(root, symbols, wanted_ids):
34
+ """First docstring line of the wanted symbols, read from the working tree once per file."""
35
+ by_path = defaultdict(list)
36
+ for symbol in symbols:
37
+ by_path[symbol["path"]].append(symbol)
38
+ docs = {}
39
+ for path in {s["path"] for s in symbols if s["id"] in wanted_ids}:
40
+ try:
41
+ with open(os.path.join(root, path), encoding="utf-8") as stream:
42
+ source = stream.read()
43
+ except (OSError, UnicodeDecodeError):
44
+ continue
45
+ for function in doc_check.functions(path, source, by_path[path]):
46
+ if function["doc"]:
47
+ docs[(path, function["start"])] = function["doc"]
48
+ return {s["id"]: docs[(s["path"], s["start_line"])] for s in symbols
49
+ if s["id"] in wanted_ids and (s["path"], s["start_line"]) in docs}
50
+
51
+
52
+ def impact(root, symbol, view_id=None, depth=3, limit=30):
53
+ if not isinstance(symbol, str) or not symbol.strip():
54
+ raise ValueError("Pass symbol: a function, method (Class.method) or class name.")
55
+ if type(depth) is not int or not 1 <= depth <= MAX_DEPTH:
56
+ raise ValueError(f"depth must be an integer from 1 to {MAX_DEPTH}.")
57
+ data = code_graph.build(root, view_id)
58
+ found = _matches(data.get("symbols") or [], symbol)
59
+ diagnostics = [d.get("reason") for d in data.get("diagnostics") or [] if d.get("reason")]
60
+ if not found:
61
+ cause = f" Analysis problems: {'; '.join(diagnostics)}" if diagnostics else ""
62
+ raise ValueError(f"No function or class named {symbol!r} in the indexed view; check the name or reindex.{cause}")
63
+ profile = index_profile.current((index_scope.load_scope(root) or {}).get("profile"))
64
+ by_id = {s["id"]: s for s in data["symbols"]}
65
+ callers, callees = defaultdict(list), defaultdict(list)
66
+ for call in data["calls"]:
67
+ callers[call["target"]].append(call)
68
+ callees[call["source"]].append(call)
69
+ targets = {s["id"] for s in found}
70
+ label = {s["id"]: f"{s['path']}::{s['name']}" for s in found} if len(found) > 1 else {}
71
+ tests = {}
72
+ for origin in targets:
73
+ seen, frontier = set(targets), {origin}
74
+ for hops in range(1, depth + 1):
75
+ reached = set()
76
+ for target in frontier:
77
+ for call in callers[target]:
78
+ source = call["source"]
79
+ if source in seen:
80
+ continue
81
+ seen.add(source)
82
+ reached.add(source)
83
+ caller = by_id.get(source)
84
+ if caller and index_profile.kind(caller["path"], profile) == "test" and source not in tests:
85
+ tests[source] = {**_site(by_id, source, caller["start_line"], label.get(origin)), "hops": hops}
86
+ frontier = reached
87
+ direct = [_site(by_id, c["source"], c["line"], label.get(t)) for t in targets for c in callers[t]
88
+ if c["source"] not in targets]
89
+ outgoing = [_site(by_id, c["target"], c["line"], label.get(t)) for t in targets for c in callees[t]
90
+ if c["target"] not in targets]
91
+ files = {s["path"] for s in found}
92
+ importers = sorted({d["source"] for d in data["dependencies"] if d.get("target") in files and d["source"] not in files})
93
+ ordered_tests = sorted(tests.values(), key=lambda t: (t["hops"], t["path"], t["line"]))
94
+ shown = direct[:limit] + outgoing[:limit] + ordered_tests[:limit]
95
+ docs = summaries(root, data["symbols"], targets | {site["id"] for site in shown})
96
+ for site in shown:
97
+ doc = docs.get(site.pop("id"))
98
+ if doc:
99
+ site["doc"] = doc
100
+ definitions = []
101
+ for s in found:
102
+ entry = {"name": s["name"], "kind": s.get("kind"), "path": s["path"], "lines": f"{s['start_line']}-{s['end_line']}"}
103
+ if docs.get(s["id"]):
104
+ entry["doc"] = docs[s["id"]]
105
+ definitions.append(entry)
106
+ return {
107
+ "symbol": symbol.strip(),
108
+ "definitions": definitions,
109
+ "callers": direct[:limit],
110
+ "calls": outgoing[:limit],
111
+ "tests": ordered_tests[:limit],
112
+ "imported_by": importers[:limit],
113
+ "counts": {"callers": len(direct), "calls": len(outgoing), "tests": len(ordered_tests),
114
+ "imported_by": len(importers)},
115
+ "test_depth": depth,
116
+ "diagnostics": diagnostics,
117
+ "view": (data.get("selected") or {}).get("label"),
118
+ "note": NOTE + (f" {len(found)} definitions share this name; each entry says which one ('of'); pass "
119
+ "path::name to keep one." if label else ""),
120
+ }
121
+
122
+
123
+ def for_agent(data, limit):
124
+ """The file-focused graph without what an agent never reads (view list, display limits) and with lists capped."""
125
+ for key in ("symbols", "dependencies", "calls", "unresolved", "diagnostics"):
126
+ items = data.get(key) or []
127
+ if len(items) > limit:
128
+ data[key] = items[:limit]
129
+ data.setdefault("omitted", {})[key] = len(items) - limit
130
+ for key in ("views", "limits", "languages", "current_view", "generated_at", "cache_hit", "read_only", "source"):
131
+ data.pop(key, None)
132
+ return data
package/config.py CHANGED
@@ -31,6 +31,8 @@ DEFAULT_CONFIG = {
31
31
  "site_memory_similarity_threshold": 0.35,
32
32
  "local_models": "fallback",
33
33
  "hook_mode": "redirect",
34
+ "doc_mode": "remind",
35
+ "duplicate_mode": "warn",
34
36
  "contact": "",
35
37
  "model_adapter": "",
36
38
  "model_adapter_options": {},
@@ -42,6 +44,26 @@ LOCAL_MODEL_MODES = ("fallback", "off", "prefer")
42
44
  # PreToolUse hook: redirect = deny the native tool and point to Smart Tool; advise = let it run and add the tip to the
43
45
  # agent's context; off = no routing at all.
44
46
  HOOK_MODES = ("redirect", "advise", "off")
47
+ # Edit hook on functions without a docstring: require = deny the edit until it adds one; remind = let it run and tell
48
+ # the agent; off = say nothing. Independent of hook_mode, which only routes searches and reads.
49
+ DOC_MODES = ("require", "remind", "off")
50
+ # Edit hook on functions that copy one already indexed (exact or near-identical body): warn = let the edit run and
51
+ # point to the existing function; off = say nothing. Never blocks: near matches are leads, not defects.
52
+ DUPLICATE_MODES = ("warn", "off")
53
+
54
+
55
+ def doc_mode(cfg=None):
56
+ value = (cfg if cfg is not None else load_config()).get("doc_mode") or DEFAULT_CONFIG["doc_mode"]
57
+ if value not in DOC_MODES:
58
+ raise ValueError(f"Invalid doc_mode ({value!r}) in {CONFIG_PATH}: use require, remind or off.")
59
+ return value
60
+
61
+
62
+ def duplicate_mode(cfg=None):
63
+ value = (cfg if cfg is not None else load_config()).get("duplicate_mode") or DEFAULT_CONFIG["duplicate_mode"]
64
+ if value not in DUPLICATE_MODES:
65
+ raise ValueError(f"Invalid duplicate_mode ({value!r}) in {CONFIG_PATH}: use warn or off.")
66
+ return value
45
67
 
46
68
 
47
69
  def hook_mode(cfg=None):
package/doc_check.py ADDED
@@ -0,0 +1,256 @@
1
+ """Function docstrings: which functions a file has, whether each one is documented, and which ones an edit touched.
2
+ Python is read with ast; TypeScript/JavaScript and Java with the code graph analyzers, where the documentation is the
3
+ /** */ block right above the function. Used by the edit hook (missing or possibly stale docstrings), the docstring
4
+ coverage listing and the one-line summaries in graph and impact results."""
5
+ import ast
6
+ import difflib
7
+ import hashlib
8
+ import os
9
+ import re
10
+ import threading
11
+ from collections import Counter, OrderedDict, defaultdict
12
+
13
+ import code_graph
14
+ import index_profile
15
+ import index_scope
16
+
17
+ PYTHON = (".py",)
18
+ SCRIPT = (".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".mts", ".cts")
19
+ JAVA = (".java",)
20
+ SUPPORTED = PYTHON + SCRIPT + JAVA
21
+ MAX_PARSED = 64
22
+ _SPACE = re.compile(r"\s+")
23
+ _PARSED = OrderedDict()
24
+ _PARSED_LOCK = threading.Lock()
25
+
26
+
27
+ def supported(path):
28
+ return str(path).lower().endswith(SUPPORTED)
29
+
30
+
31
+ def _first_line(text):
32
+ return next((line.strip(" *") for line in text.strip().splitlines() if line.strip(" *")), "")
33
+
34
+
35
+ def _public_name(name):
36
+ return not name.startswith("_") or name.startswith("__") and name.endswith("__")
37
+
38
+
39
+ def _python_functions(source):
40
+ """Python functions with "required" following pydocstyle publicity (public name, public parents, not nested in a
41
+ function) and PEP 698: an @override method inherits its documentation."""
42
+ tree = ast.parse(source)
43
+ found = []
44
+
45
+ def visit(node, owner, public, nested):
46
+ for child in ast.iter_child_nodes(node):
47
+ if isinstance(child, (ast.FunctionDef, ast.AsyncFunctionDef)):
48
+ doc = ast.get_docstring(child)
49
+ returns = f" -> {ast.unparse(child.returns)}" if child.returns else ""
50
+ found.append({"name": f"{owner}{child.name}", "start": child.lineno, "end": child.end_lineno,
51
+ "documented": doc is not None, "doc": _first_line(doc or ""),
52
+ "signature": f"{child.name}({ast.unparse(child.args)}){returns}",
53
+ "required": public and not nested and _public_name(child.name)
54
+ and not any(ast.unparse(d).split(".")[-1] == "override"
55
+ for d in child.decorator_list)})
56
+ visit(child, f"{owner}{child.name}.", False, True)
57
+ elif isinstance(child, ast.ClassDef):
58
+ visit(child, f"{owner}{child.name}.", public and _public_name(child.name), nested)
59
+ visit(tree, "", True, False)
60
+ return found
61
+
62
+
63
+ def _overload_head(lines, index, name):
64
+ """Line where the overload signature ending at index starts, or None when the lines above are not an overload of
65
+ name (a declaration ending in ';')."""
66
+ head = re.compile(rf"^\s*(export\s+)?(default\s+)?(declare\s+)?(public\s+|protected\s+)?(static\s+)?(async\s+)?"
67
+ rf"(function\s+)?{re.escape(name)}\s*[<(]")
68
+ for k in range(index, max(index - 30, -1), -1):
69
+ line = lines[k].rstrip()
70
+ if k != index and (line.endswith(("*/", "}", ";")) or not line):
71
+ return None
72
+ if head.match(lines[k]):
73
+ return k
74
+ return None
75
+
76
+
77
+ def _block_above(lines, start, name=None):
78
+ """Text of the /** */ block above the function starting at line start. With name (TypeScript/JavaScript), uncommented
79
+ overload signatures in between are skipped, as eslint-plugin-jsdoc does by default."""
80
+ index = start - 2
81
+ while index >= 0:
82
+ if not lines[index].strip() or lines[index].lstrip().startswith("@"):
83
+ index -= 1
84
+ continue
85
+ head = _overload_head(lines, index, name) if name and lines[index].rstrip().endswith(";") else None
86
+ if head is None:
87
+ break
88
+ index = head - 1
89
+ if index < 0 or not lines[index].rstrip().endswith("*/"):
90
+ return None
91
+ end = index
92
+ while index >= 0 and "/*" not in lines[index]:
93
+ index -= 1
94
+ if index < 0 or "/**" not in lines[index]:
95
+ return None
96
+ return "\n".join(line.strip().removeprefix("/**").removesuffix("*/").strip().removeprefix("*").strip()
97
+ for line in lines[index:end + 1])
98
+
99
+
100
+ def _annotations(lines, start, end):
101
+ """Annotation lines right above the function and at its first lines (the Java analyzer starts a method at its
102
+ annotations), and the first line that is not an annotation."""
103
+ index, found = start - 2, []
104
+ while index >= 0 and (not lines[index].strip() or lines[index].lstrip().startswith("@")):
105
+ found.append(lines[index].strip())
106
+ index -= 1
107
+ index = start - 1
108
+ while index < min(end, len(lines)) and lines[index].lstrip().startswith("@"):
109
+ found.append(lines[index].strip())
110
+ index += 1
111
+ return found, lines[index] if index < len(lines) else ""
112
+
113
+
114
+ def _required(path, symbol, parent, first_line, annotations):
115
+ """Whether a braced-language function needs documentation by the linter defaults: Checkstyle MissingJavadocMethod
116
+ (no @Override, not private) and eslint-plugin-jsdoc require-jsdoc (named declarations and class methods, never
117
+ callbacks or functions nested in another function; private members and unexported arrow functions left out)."""
118
+ short = symbol["name"].split("(")[0].split(".")[-1]
119
+ if short in ("callback", "anonymous") or not _public_name(short) or short.startswith("#"):
120
+ return False
121
+ if re.search(r"\bprivate\b", first_line):
122
+ return False
123
+ if path.lower().endswith(JAVA):
124
+ return not any(a.startswith("@Override") for a in annotations)
125
+ if parent and parent.get("kind") in ("function", "method"):
126
+ return False
127
+ if symbol.get("kind") == "method" or parent and parent.get("kind") in ("class", "interface"):
128
+ return True
129
+ return bool(re.match(r"\s*(export\s+)?(default\s+)?(declare\s+)?(async\s+)?function\b", first_line)
130
+ or re.match(r"\s*export\b", first_line))
131
+
132
+
133
+ def _braced_functions(path, source, symbols):
134
+ lines = source.splitlines()
135
+ by_id = {s.get("id"): s for s in symbols}
136
+ script = path.lower().endswith(SCRIPT)
137
+ found = []
138
+ for symbol in symbols:
139
+ if symbol.get("kind") not in ("function", "method") or "@" in symbol["name"]:
140
+ continue
141
+ start, end = symbol["start_line"], symbol["end_line"]
142
+ short = symbol["name"].split("(")[0].split(".")[-1]
143
+ block = _block_above(lines, start, short if script else None)
144
+ header = " ".join(lines[start - 1:min(end, start + 8)])
145
+ annotations, first_line = _annotations(lines, start, end)
146
+ required = _required(path, symbol, by_id.get(symbol.get("parent")), first_line, annotations)
147
+ found.append({"name": symbol["name"], "start": start, "end": end, "documented": block is not None,
148
+ "doc": _first_line(block or ""), "signature": _SPACE.sub(" ", header.split("{")[0]).strip(),
149
+ "required": required})
150
+ return found
151
+
152
+
153
+ def functions(path, source, symbols=None):
154
+ """Functions of one file with start/end lines, documented flag, first doc line and normalized signature. symbols
155
+ are the code graph symbols of that file when already known; otherwise the analyzer runs on source."""
156
+ lower = path.lower()
157
+ if lower.endswith(PYTHON):
158
+ try:
159
+ return _python_functions(source)
160
+ except SyntaxError:
161
+ return []
162
+ if symbols is None:
163
+ symbols = analyze(path, source)[0]
164
+ return _braced_functions(path, source, symbols)
165
+
166
+
167
+ def analyze(path, *sources):
168
+ """Analyzer symbols of several versions of one TypeScript/JavaScript/Java file, from one analyzer run for the
169
+ versions not seen yet (an edit hook needs the file before and after; the next edit's 'before' is this 'after')."""
170
+ lower, name = path.lower(), os.path.basename(path)
171
+ keys = [hashlib.sha1(f"{name}\0{source}".encode("utf-8")).hexdigest() for source in sources]
172
+ with _PARSED_LOCK:
173
+ missing = {key: source for key, source in zip(keys, sources) if key not in _PARSED}
174
+ if missing:
175
+ batch = {f"v{i}/{name}": source for i, source in enumerate(missing.values())}
176
+ analysis = (code_graph._javascript_graph(batch) if lower.endswith(SCRIPT)
177
+ else code_graph._java_graph(batch) if lower.endswith(JAVA) else {})
178
+ found = {key: [dict(s, path=name) for s in analysis.get("symbols") or [] if s.get("path") == f"v{i}/{name}"]
179
+ for i, key in enumerate(missing)}
180
+ with _PARSED_LOCK:
181
+ _PARSED.update(found)
182
+ while len(_PARSED) > MAX_PARSED:
183
+ _PARSED.popitem(last=False)
184
+ with _PARSED_LOCK:
185
+ return [_PARSED[key] if key in _PARSED else [] for key in keys]
186
+
187
+
188
+ def _touched_lines(before, after):
189
+ old, new = (before or "").splitlines(), after.splitlines()
190
+ touched = set()
191
+ for tag, _i1, _i2, j1, j2 in difflib.SequenceMatcher(None, old, new, autojunk=False).get_opcodes():
192
+ if tag != "equal":
193
+ touched.update(range(j1 + 1, max(j1 + 1, j2) + 1))
194
+ return touched
195
+
196
+
197
+ def written(path, before, after):
198
+ """Functions of the edited file after the edit ('all') and the ones the edit touched ('touched')."""
199
+ if path.lower().endswith(SCRIPT + JAVA):
200
+ analyze(path, *(source for source in (before, after) if source is not None))
201
+ found = functions(path, after) if supported(path) else []
202
+ lines = _touched_lines(before, after)
203
+ return {"all": found, "touched": [f for f in found if lines.intersection(range(f["start"], f["end"] + 1))]}
204
+
205
+
206
+ def review(path, before, after, edited=None):
207
+ """Functions the edit from before to after touched that are left without a docstring although the language
208
+ convention expects one ('missing': public, not nested, not an override) or keep one while their signature changed
209
+ ('signature_changed'). Untouched functions and body-only edits are not reported. edited is written() of the same
210
+ edit when the caller already has it."""
211
+ if not supported(path):
212
+ return []
213
+ edited = edited or written(path, before, after)
214
+ if not edited["touched"]:
215
+ return []
216
+ old = {f["name"]: f for f in functions(path, before)} if before else {}
217
+ findings = []
218
+ for function in edited["touched"]:
219
+ previous = old.get(function["name"])
220
+ if not function["documented"]:
221
+ if function["required"]:
222
+ findings.append({**function, "issue": "missing"})
223
+ elif previous and _SPACE.sub("", previous["signature"]) != _SPACE.sub("", function["signature"]):
224
+ findings.append({**function, "issue": "signature_changed"})
225
+ return findings
226
+
227
+
228
+ def coverage(root, include_tests=False, limit=30, view_id=None):
229
+ """Docstring coverage of the indexed code: functions the language convention expects documented (plus any already
230
+ documented) and those still missing, file by file, so an agent can document a project that started without it.
231
+ Tests are left out unless include_tests."""
232
+ data = code_graph.build(root, view_id)
233
+ profile = index_profile.current((index_scope.load_scope(root) or {}).get("profile"))
234
+ kinds = ("code", "test") if include_tests else ("code",)
235
+ by_path = defaultdict(list)
236
+ for symbol in data.get("symbols") or []:
237
+ if supported(symbol["path"]) and index_profile.kind(symbol["path"], profile) in kinds:
238
+ by_path[symbol["path"]].append(symbol)
239
+ total, missing = 0, []
240
+ for path in sorted(by_path):
241
+ try:
242
+ with open(os.path.join(root, path), encoding="utf-8") as stream:
243
+ source = stream.read()
244
+ except (OSError, UnicodeDecodeError):
245
+ continue
246
+ found = [f for f in functions(path, source, by_path[path]) if f["required"] or f["documented"]]
247
+ total += len(found)
248
+ missing += [{"path": path, "function": f["name"], "line": f["start"]} for f in found if not f["documented"]]
249
+ per_file = Counter(item["path"] for item in missing)
250
+ return {"functions": total, "documented": total - len(missing),
251
+ "coverage_percent": round(100 * (total - len(missing)) / total, 1) if total else None,
252
+ "missing_total": len(missing), "missing": missing[:limit],
253
+ "files_with_most_missing": [{"path": p, "missing": n} for p, n in per_file.most_common(10)],
254
+ "include_tests": include_tests,
255
+ "diagnostics": [d.get("reason") for d in data.get("diagnostics") or [] if d.get("reason")],
256
+ "view": (data.get("selected") or {}).get("label")}
package/duplicates.py CHANGED
@@ -25,8 +25,35 @@ EMBED_SLICE = 64
25
25
  CUTOFF_NOTE = ("Default semantic cutoff 0.90: in the 2026-10-02 measurement (claude-code, independent judge) pairs above "
26
26
  "0.95 were 10/10 actionable and those from 0.80 to 0.94 only 7/20; below 0.90 most are false positives "
27
27
  "and reviewing them costs time without leading to a refactor.")
28
+ MIN_TOKENS = 50
29
+ NEAR_MIN_TOKENS = 70
30
+ SHINGLE = 5
31
+ NEAR_NOTE = ("Near = at least 0.80 of 5-token sequences shared with keywords, called names, attributes and types kept "
32
+ "and only local names, strings and numbers abstracted; in the 2026-10-08 measurement on Python, "
33
+ "JavaScript, TypeScript and Java projects it caught every exact and locally renamed copy and 95-100% of "
34
+ "copies with one added statement, and fired on 0-1.6% of real functions.")
28
35
  _HASH_COMMENT = {"py", "rb", "sh"}
29
- _TOKEN = re.compile(r"[A-Za-z_]\w*|\d+|[^\s\w]")
36
+ _TOKEN = re.compile(r"\"(?:\\.|[^\"\\\n])*\"|'(?:\\.|[^'\\\n])*'|`(?:\\.|[^`\\])*`|[A-Za-z_$][\w$]*|\d[\w.]*|[^\s\w]")
37
+ _KEYWORDS = {
38
+ "py": "False None True and as assert async await break class continue def del elif else except finally for from "
39
+ "global if import in is lambda nonlocal not or pass raise return try while with yield self cls",
40
+ "js": "await break case catch class const continue debugger default delete do else export extends false finally "
41
+ "for function if import in instanceof let new null of return super switch this throw true try typeof "
42
+ "undefined var void while with yield async",
43
+ "java": "abstract assert boolean break byte case catch char class const continue default do double else enum "
44
+ "extends final finally float for if implements import instanceof int interface long native new null "
45
+ "package private protected public return short static super switch synchronized this throw throws "
46
+ "transient try void volatile while true false var",
47
+ }
48
+ _KEYWORDS["ts"] = _KEYWORDS["js"] + " as interface type enum implements private public protected readonly keyof never unknown any"
49
+ _KEYWORDS = {lang: frozenset(words.split()) for lang, words in _KEYWORDS.items()}
50
+ _LANG = {"py": "py", "js": "js", "jsx": "js", "mjs": "js", "cjs": "js", "ts": "ts", "tsx": "ts", "mts": "ts",
51
+ "cts": "ts", "java": "java"}
52
+ _CONTRACT = re.compile(r"^(__\w+__|equals|hashCode|toString|compareTo|clone|close|run|call|apply|get|set|render|"
53
+ r"constructor|ng[A-Z]\w*|componentDid\w+|main)$")
54
+ _DOCSTRING = re.compile(r"\A\s*[rRuUbB]{0,2}(\"\"\"|''')[\s\S]*?\1")
55
+ _GENERATED = re.compile(r"@generated|DO NOT EDIT|Code generated by|auto-generated", re.I)
56
+ _WRITE_POOLS = {}
30
57
  _CONTAINER_KINDS = {"module", "template", "class", "interface"}
31
58
  _GENERIC_NAMES = {"callback", "anonymous", "constructor"}
32
59
  _WARMING = set()
@@ -64,18 +91,48 @@ def _strip_comments(text, ext):
64
91
 
65
92
 
66
93
  def _body(text, ext):
94
+ """Function body without the signature and, in Python, without the leading docstring, so two copies that differ
95
+ only in documentation compare as identical."""
67
96
  if ext in _HASH_COMMENT:
68
97
  lines = text.splitlines()
69
98
  start = next((k for k, line in enumerate(lines) if re.match(r"\s*(async\s+)?def\s", line)), 0)
70
99
  end = next((k for k in range(start, len(lines)) if lines[k].rstrip().endswith(":")), start)
71
- return "\n".join(lines[end + 1:])
100
+ return _DOCSTRING.sub("", "\n".join(lines[end + 1:]), count=1)
72
101
  brace = text.find("{")
73
102
  return text[brace + 1:] if brace >= 0 else text
74
103
 
75
104
 
76
- def _shingles(text):
77
- tokens = ["ID" if re.match(r"[A-Za-z_]", t) else t for t in _TOKEN.findall(text)]
78
- return frozenset(tuple(tokens[i:i + 5]) for i in range(max(len(tokens) - 4, 1)))
105
+ def _features(body, ext, name):
106
+ """Near-duplicate features of a function body: 5-token shingles where keywords, called names, attribute names and
107
+ type names stay and local names, strings and numbers become placeholders, so parallel functions calling different
108
+ APIs (get_stdin/get_stdout, rotateLeft/rotateRight) do not match while a copy with renamed locals does."""
109
+ words = _KEYWORDS.get(_LANG.get(ext), frozenset())
110
+ raw = _TOKEN.findall(body)
111
+ tokens = []
112
+ for k, t in enumerate(raw):
113
+ if t[0] in "\"'`":
114
+ tokens.append("STR")
115
+ elif t[0].isdigit():
116
+ tokens.append("NUM")
117
+ elif re.match(r"[A-Za-z_$]", t):
118
+ kept = (t in words or t[0].isupper() or (k and raw[k - 1] == ".")
119
+ or (k + 1 < len(raw) and raw[k + 1] == "("))
120
+ tokens.append(t if kept else "ID")
121
+ else:
122
+ tokens.append(t)
123
+ return {"shingles": frozenset(tuple(tokens[i:i + SHINGLE]) for i in range(max(len(tokens) - SHINGLE + 1, 1))),
124
+ "tokens": len(tokens), "lang": _LANG.get(ext, ext),
125
+ "near_ok": len(tokens) >= NEAR_MIN_TOKENS and not _CONTRACT.match(name)}
126
+
127
+
128
+ def _idioms(functions):
129
+ """Shingles shared by so many functions that they are language idioms (if err != nil, try/except), not copies."""
130
+ counts = {}
131
+ for f in functions:
132
+ for shingle in f["shingles"]:
133
+ counts[shingle] = counts.get(shingle, 0) + 1
134
+ cut = max(100, int(0.005 * len(functions)))
135
+ return frozenset(s for s, c in counts.items() if c > cut)
79
136
 
80
137
 
81
138
  def _index_state(path):
@@ -115,10 +172,12 @@ def _functions(root, include_tests):
115
172
  text = "\n".join(file_lines[s["start_line"] - 1:s["end_line"]])
116
173
  if not text.strip():
117
174
  continue
118
- normalized = re.sub(r"\s+", " ", _strip_comments(_body(text, ext), ext)).strip()
119
- functions.append({"name": s["name"].split("(")[0].split(".")[-1], "path": path, "line": s["start_line"],
175
+ body = _strip_comments(_body(text, ext), ext)
176
+ normalized = re.sub(r"\s+", " ", body).strip()
177
+ name = s["name"].split("(")[0].split(".")[-1]
178
+ functions.append({"name": name, "path": path, "line": s["start_line"],
120
179
  "end": s["end_line"], "lines": s["end_line"] - s["start_line"] + 1, "text": text[:MAX_TEXT],
121
- "hash": hashlib.sha1(normalized.encode("utf-8")).hexdigest(), "shingles": _shingles(normalized)})
180
+ "hash": hashlib.sha1(normalized.encode("utf-8")).hexdigest(), **_features(body, ext, name)})
122
181
  occurrence = occurrences[functions[-1]["hash"]] = occurrences.get(functions[-1]["hash"], 0) + 1
123
182
  functions[-1]["kind"] = kind
124
183
  functions[-1]["fp"] = hashlib.sha1(f"{path}\0{functions[-1]['hash']}\0{occurrence}".encode("utf-8")).hexdigest()
@@ -183,16 +242,24 @@ def _warm(root, view, functions, embed):
183
242
  _WARMING.discard(root)
184
243
 
185
244
 
245
+ def _similar(fa, fb):
246
+ """Whether two functions may be compared as near copies: same language, not overloads of one name in one file."""
247
+ return fa["lang"] == fb["lang"] and fa["hash"] != fb["hash"] and not (fa["path"] == fb["path"]
248
+ and fa["name"] == fb["name"])
249
+
250
+
186
251
  def _near(functions):
187
- order = sorted(range(len(functions)), key=lambda i: len(functions[i]["shingles"]))
252
+ common = _idioms(functions)
253
+ cut = [f["shingles"] - common if f["near_ok"] else None for f in functions]
254
+ order = sorted((i for i in range(len(functions)) if cut[i]), key=lambda i: len(cut[i]))
188
255
  found = []
189
256
  for position, a in enumerate(order):
190
- sa = functions[a]["shingles"]
257
+ sa = cut[a]
191
258
  for b in order[position + 1:]:
192
- sb = functions[b]["shingles"]
259
+ sb = cut[b]
193
260
  if len(sb) * NEAR_JACCARD > len(sa):
194
261
  break
195
- if functions[a]["hash"] == functions[b]["hash"]:
262
+ if not _similar(functions[a], functions[b]):
196
263
  continue
197
264
  inter = len(sa & sb)
198
265
  score = inter / (len(sa) + len(sb) - inter)
@@ -296,7 +363,7 @@ def find(root, embed, configured_model, min_similarity=DEFAULT_MIN_SIMILARITY, i
296
363
  remembered.pop(next(iter(remembered)))
297
364
  result = {
298
365
  "view": view["label"], "functions_analyzed": len(functions), "include_tests": include_tests,
299
- "min_similarity": min_similarity, "cutoff_note": CUTOFF_NOTE, "semantic_status": status, "notes": notes,
366
+ "min_similarity": min_similarity, "cutoff_note": CUTOFF_NOTE, "near_note": NEAR_NOTE, "semantic_status": status, "notes": notes,
300
367
  "counts": {"exact_groups": len(exact), "near_pairs": len(near), "semantic_pairs": semantic_total,
301
368
  "dismissed_hidden": len(hidden) + len(hidden_semantic)},
302
369
  "exact": [{"id": fid, "lines": g[0]["lines"], "copies": [_ref(f) for f in g]} for fid, g in exact[:limit]],
@@ -311,3 +378,71 @@ def find(root, embed, configured_model, min_similarity=DEFAULT_MIN_SIMILARITY, i
311
378
  "members": [_ref(skip[fid][1]), _ref(skip[fid][2])], **dismissed[fid]}
312
379
  for fid in hidden_semantic])
313
380
  return result
381
+
382
+
383
+ def _write_pool(root):
384
+ """Indexed production functions with their idiom shingles, rebuilt only when the index changes; None while the
385
+ code graph is not cached yet, so an edit hook never waits for an analysis."""
386
+ view_path = indexer.existing_db_path(root)
387
+ if not view_path or code_graph.cached_symbols(root)[0] is None:
388
+ return None
389
+ state = (view_path, *_index_state(view_path))
390
+ with _LOCK:
391
+ cached = _WRITE_POOLS.get(root)
392
+ if cached and cached[0] == state:
393
+ return cached[1]
394
+ functions, _notes, _view = _functions(root, False)
395
+ production = [f for f in functions if f["kind"] == "code"]
396
+ pool = (production, _idioms(production))
397
+ with _LOCK:
398
+ _WRITE_POOLS[root] = (state, pool)
399
+ return pool
400
+
401
+
402
+ def on_write(root, path, before, after, written):
403
+ """Functions an edit writes whose body copies one already indexed: exact (same tokens, comments and spacing
404
+ ignored) or near (see NEAR_NOTE). written lists the touched functions as doc_check.functions returns them. The
405
+ function being edited in place, overloads of one name and functions the edit removes from the file are skipped."""
406
+ ext = path.rsplit(".", 1)[-1].lower() if "." in path else ""
407
+ rel = os.path.relpath(path, root).replace(os.sep, "/")
408
+ profile = index_profile.current((index_scope.load_scope(root) or {}).get("profile"))
409
+ if ext not in _LANG or index_profile.kind(rel, profile) != "code" or _GENERATED.search(after[:4000]):
410
+ return []
411
+ pool = _write_pool(root)
412
+ if pool is None:
413
+ return []
414
+ production, common = pool
415
+ lines = after.splitlines()
416
+ remaining = {f["name"].split("(")[0].split(".")[-1] for f in written["all"]}
417
+ matches = []
418
+ for function in written["touched"]:
419
+ name = function["name"].split("(")[0].split(".")[-1]
420
+ text = "\n".join(lines[function["start"] - 1:function["end"]])
421
+ body = _strip_comments(_body(text, ext), ext)
422
+ features = _features(body, ext, name)
423
+ if function["end"] - function["start"] + 1 < MIN_LINES or features["tokens"] < MIN_TOKENS \
424
+ or _CONTRACT.match(name):
425
+ continue
426
+ digest = hashlib.sha1(re.sub(r"\s+", " ", body).strip().encode("utf-8")).hexdigest()
427
+ mine = features["shingles"] - common
428
+ best = None
429
+ for other in production:
430
+ if other["lang"] != features["lang"] or other["path"] == rel and (other["name"] == name
431
+ or other["name"] not in remaining):
432
+ continue
433
+ if other["hash"] == digest:
434
+ best = ("exact", 1.0, other)
435
+ break
436
+ if not (features["near_ok"] and other["near_ok"]):
437
+ continue
438
+ theirs = other["shingles"] - common
439
+ if not mine or min(len(mine), len(theirs)) < NEAR_JACCARD * max(len(mine), len(theirs)):
440
+ continue
441
+ inter = len(mine & theirs)
442
+ score = inter / (len(mine) + len(theirs) - inter)
443
+ if score >= NEAR_JACCARD and (best is None or score > best[1]):
444
+ best = ("near", score, other)
445
+ if best:
446
+ matches.append({"name": name, "lines": f"{function['start']}-{function['end']}", "type": best[0],
447
+ "score": round(best[1], 2), "existing": _ref(best[2])})
448
+ return matches