@allansantos-dev/smart-tool 0.9.3 → 0.9.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,29 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.9.4 - beta
4
+
5
+ Found by using 0.9.3 on its own repository right after installing it.
6
+
7
+ - Impact (`graph` with `symbol`), docstring coverage (`docs`), `affected_tests` and the duplicate warning on edit now see
8
+ every file the index does not reflect yet: working-tree changes, indexed files changed on disk and files committed
9
+ after the last indexing. In the default `on_search` mode the index only updates on a search, so a function added
10
+ earlier in the session was "not found" by `graph`, and a copy of it was not flagged by the duplicate warning. Only
11
+ the current view gets these files; a pinned or older view is read as indexed.
12
+ - The code graph analysis is saved next to the index (`graph-cache/`, about 0.3 MB per view) and reloaded after a
13
+ daemon restart or when a project left the in-memory cache: 0.03-0.12 s instead of 7-11 s of analysis on remeda
14
+ and gson. An indexing job warms the graph when it ends. The duplicate warning, which never waits for an analysis,
15
+ used to skip the first edit after a restart or a reindex; it now checks it.
16
+ - Hook routing decides more Bash commands without the router model: a search verb after a single `|` filters the
17
+ output of the command before it (`python x.py | tail -5`, `npm test | grep FAIL`) and no longer sends the call to
18
+ the model, unless that command reads code or pages (`git grep`, `git ls-files`, `curl`, `gh api`, ...). Replayed on
19
+ 2,489 real Bash decisions that went to the model (median 1.5 s, 7.8% redirected): 438 become mechanical, 13 minutes
20
+ of waiting saved in 58 hours, 1 of 155 redirects lost. Search verbs after `do`/`then`/`else` and inside a quoted
21
+ `bash -c "..."` / `powershell -Command "..."` are now recognized.
22
+ - `affected_tests` no longer asks for the whole suite when only CI configuration or repository metadata changed
23
+ (`.github/`, `.gitlab-ci.yml`, `Jenkinsfile`, `.gitignore`, `.gitattributes`, `LICENSE`, ...).
24
+ - The docstring reminder on edit only applies to code files of registered projects: test files (whose test functions
25
+ normally have no docstring) and files outside any project are left out, as in the `docs` coverage.
26
+
3
27
  ## 0.9.3 - beta
4
28
 
5
29
  - Camoufox download (1.3 GB) on networks that cut long transfers (seen stopping at 500-530 MB on every attempt): it is
package/affected_tests.py CHANGED
@@ -12,6 +12,7 @@ import subprocess
12
12
  from collections import defaultdict, deque
13
13
 
14
14
  import code_graph
15
+ import index_inventory
15
16
  import index_profile
16
17
  import index_scope
17
18
 
@@ -21,11 +22,14 @@ RUN_ALL = re.compile(
21
22
  r"uv\.lock|Pipfile(\.lock)?|tsconfig[^/]*\.json|jsconfig\.json|(jest|vitest|vite|babel)\.config\.[^/]+|"
22
23
  r"\.babelrc|\.mocharc[^/]*|pom\.xml|build\.gradle(\.kts)?|settings\.gradle(\.kts)?|gradle\.properties|"
23
24
  r"\.env[^/]*)$")
25
+ CI_CONFIG = re.compile(r"(^|/)(\.github|\.circleci|\.buildkite)/|(^|/)(\.gitlab-ci\.ya?ml|azure-pipelines\.ya?ml|"
26
+ r"Jenkinsfile|\.travis\.ya?ml|bitbucket-pipelines\.ya?ml|\.gitignore|\.gitattributes|"
27
+ r"\.editorconfig|\.mailmap|CODEOWNERS|LICENSE(\.[^/]*)?|\.npmignore)$")
24
28
  NOTE = ("Selection from the indexed import graph: tests that load code through strings (mock.patch targets, "
25
29
  "importlib, require(variable), reflection, Spring scanning) or read data files are not seen. Run the full "
26
30
  "suite before committing; run_all is true when a changed file can change every test (config, lockfile, "
27
- "unanalyzed code). Changed files are read from the working tree, so new files and imports count before "
28
- "reindexing; pytest fixtures link a test to the conftest.py that defines them.")
31
+ "unanalyzed code). Changed files and files committed after the last indexing are read from disk, so new "
32
+ "files and imports count before reindexing; pytest fixtures link a test to the conftest.py that defines them.")
29
33
  _HUNK = re.compile(r"^@@ -\d+(?:,\d+)? \+(\d+)(?:,(\d+))? @@", re.M)
30
34
 
31
35
 
@@ -195,7 +199,10 @@ def affected(root, base="HEAD", limit=30, view_id=None):
195
199
  if not isinstance(base, str) or not re.fullmatch(r"[\w./@^~{}-]{1,200}", base) or base.startswith("-"):
196
200
  raise ValueError("base must be a git revision such as HEAD, main or origin/main.")
197
201
  changes = _changes(root, base)
198
- data = code_graph.build_overlay(root, {path: change["status"] for path, change in changes.items()}, view_id)
202
+ selected = index_inventory.inspect(root, view_id)["selected"] or {}
203
+ pending = code_graph.pending_changes(root, selected["path"]) if selected.get("current") else {}
204
+ data = code_graph.build_overlay(root, {**pending, **{path: change["status"] for path, change in changes.items()}},
205
+ view_id)
199
206
  profile = index_profile.current((index_scope.load_scope(root) or {}).get("profile"))
200
207
  kind = {}
201
208
  for path in set(changes) | {f["path"] for f in data.get("files") or []}:
@@ -213,6 +220,8 @@ def affected(root, base="HEAD", limit=30, view_id=None):
213
220
  by_id = {s["id"]: s for s in data.get("symbols") or []}
214
221
  run_all, not_analyzed, changed_code, changed_tests, conftests = [], [], [], [], []
215
222
  for path, change in sorted(changes.items()):
223
+ if CI_CONFIG.search(path):
224
+ continue
216
225
  if RUN_ALL.search(path):
217
226
  run_all.append(f"{path} changed (configuration or dependencies)")
218
227
  elif os.path.basename(path) == "conftest.py":
package/code_graph.py CHANGED
@@ -2,6 +2,7 @@
2
2
  import ast
3
3
  import collections
4
4
  import copy
5
+ import gzip
5
6
  import contextvars
6
7
  import hashlib
7
8
  import json
@@ -14,6 +15,7 @@ from pathlib import Path
14
15
 
15
16
  import indexer
16
17
  import index_inventory
18
+ import index_scope
17
19
  import project_identity
18
20
  import web_search_adapters
19
21
  import web_document_graph
@@ -339,6 +341,41 @@ def _by_file(data):
339
341
  return {'symbols':by_file,'parsed':parsed,'note':note}
340
342
 
341
343
 
344
+ def _disk_path(view_path):
345
+ path=indexer.graph_cache_path(view_path)
346
+ return os.path.dirname(path),path
347
+
348
+
349
+ def _disk_load(key):
350
+ """Analysis saved for exactly this index state and analyzer version, or None. A damaged file is a miss: the
351
+ caller analyzes again and overwrites it."""
352
+ _folder,path=_disk_path(key[0])
353
+ try:
354
+ with gzip.open(path,'rt',encoding='utf-8') as stream:
355
+ saved=json.load(stream)
356
+ except (OSError,EOFError,ValueError):
357
+ return None
358
+ return saved.get('graph') if saved.get('key')==[key[1],key[2],key[3]] else None
359
+
360
+
361
+ def _disk_save(key,data):
362
+ """Writes the analysis next to the index (atomic replace) and removes caches of views that no longer exist.
363
+ Returns the error text when it could not be written, so the caller can report it."""
364
+ folder,path=_disk_path(key[0])
365
+ try:
366
+ os.makedirs(folder,exist_ok=True)
367
+ temporary=f'{path}.{os.getpid()}.{threading.get_ident()}.tmp'
368
+ with gzip.open(temporary,'wt',encoding='utf-8',compresslevel=5) as stream:
369
+ json.dump({'key':[key[1],key[2],key[3]],'graph':data},stream,ensure_ascii=False)
370
+ os.replace(temporary,path)
371
+ for name in os.listdir(folder):
372
+ if name.endswith('.json.gz') and not os.path.isfile(os.path.join(os.path.dirname(folder),name[:-8]+'.sqlite3')):
373
+ os.remove(os.path.join(folder,name))
374
+ except OSError as exc:
375
+ return f'Graph cache not saved ({type(exc).__name__}: {exc}); the next restart analyzes again.'[:300]
376
+ return None
377
+
378
+
342
379
  def cached_symbols(root):
343
380
  path=indexer.existing_db_path(root)
344
381
  if not path:
@@ -348,6 +385,13 @@ def cached_symbols(root):
348
385
  data=_CACHE.get(key)
349
386
  if data is not None:
350
387
  _CACHE.move_to_end(key)
388
+ if data is None:
389
+ data=_disk_load(key)
390
+ if data is not None:
391
+ with _LOCK:
392
+ _CACHE[key]=data
393
+ while len(_CACHE)>3:_CACHE.popitem(last=False)
394
+ with _LOCK:
351
395
  state=_WARM_STATE.get(root)
352
396
  if data is None and state and state['key']==key and (state['result'] or time.time()-state['at']<WARM_RETRY_S):
353
397
  return state['result'],state['note']
@@ -361,6 +405,12 @@ def cached_symbols(root):
361
405
  return None,None
362
406
 
363
407
 
408
+ def warm(root):
409
+ """Starts the analysis of the project's current index in the background unless memory or disk already has it, so
410
+ the next edit hook or search finds the graph ready (called when an indexing job ends)."""
411
+ cached_symbols(root)
412
+
413
+
364
414
  def _warm(root,key):
365
415
  result,note=None,None
366
416
  try:
@@ -388,6 +438,13 @@ def build(root,view_id=None,storage_id=None,file_path=None,background=False):
388
438
  cached=_CACHE.get(key)
389
439
  if cached:_CACHE.move_to_end(key)
390
440
  was_cached=cached is not None
441
+ save_note=None
442
+ if cached is None:
443
+ cached=_disk_load(key)
444
+ if cached is not None:
445
+ with _LOCK:
446
+ _CACHE[key]=cached
447
+ while len(_CACHE)>3:_CACHE.popitem(last=False)
391
448
  if cached is None:
392
449
  slots=_WARM_SLOT if background else _ANALYSIS_SLOTS
393
450
  if not slots.acquire(timeout=None if background else 3):
@@ -401,7 +458,9 @@ def build(root,view_id=None,storage_id=None,file_path=None,background=False):
401
458
  with _LOCK:
402
459
  _CACHE[key]=cached
403
460
  while len(_CACHE)>3:_CACHE.popitem(last=False)
461
+ save_note=_disk_save(key,cached)
404
462
  data=copy.deepcopy(cached)
463
+ if save_note:data['diagnostics']=[*data.get('diagnostics',[]),{'path':'','reason':save_note}]
405
464
  # Só compara metadados do diretório de trabalho quando a visão é a atual.
406
465
  for file in data['files']:
407
466
  file['status']='stored'
@@ -420,6 +479,72 @@ def build(root,view_id=None,storage_id=None,file_path=None,background=False):
420
479
  return _focus(data,file_path) if file_path else data
421
480
 
422
481
 
482
+ def working_tree_changes(root):
483
+ """Project-relative paths changed in the git working tree (staged, unstaged, untracked) -> 'deleted' or
484
+ 'changed'; empty when the folder is not in a git repository."""
485
+ def git(*args):
486
+ return subprocess.run(['git',*args],cwd=root,capture_output=True,timeout=30,
487
+ creationflags=getattr(subprocess,'CREATE_NO_WINDOW',0))
488
+ prefix=git('rev-parse','--show-prefix')
489
+ if prefix.returncode:
490
+ return {}
491
+ prefix=prefix.stdout.decode('utf-8','replace').strip()
492
+ status=git('status','--porcelain=v1','-z','--untracked-files=all','--','.')
493
+ if status.returncode:
494
+ raise RuntimeError('git status failed: '+status.stderr.decode('utf-8','replace').strip()[:200])
495
+ entries=status.stdout.decode('utf-8','replace').split('\0');changes={};i=0
496
+ while i<len(entries):
497
+ entry=entries[i];i+=1
498
+ if len(entry)<4:continue
499
+ code,path=entry[:2],entry[3:]
500
+ if 'R' in code or 'C' in code:i+=1
501
+ if path.startswith(prefix):changes[path[len(prefix):]]='deleted' if 'D' in code else 'changed'
502
+ return changes
503
+
504
+
505
+ def pending_changes(root,view_path):
506
+ """Files the indexed view does not reflect yet: working-tree changes, indexed files whose size or modification time
507
+ differs on disk, and files tracked by git inside the index scope that were never indexed (commits made after the
508
+ last indexing). Path -> 'deleted' or 'changed'."""
509
+ changes=working_tree_changes(root)
510
+ conn=indexer._readonly(view_path)
511
+ try:
512
+ columns={row[1] for row in conn.execute('PRAGMA table_info(manifest)')}
513
+ if not {'size','mtime_ns'} <= columns:
514
+ raise RuntimeError('Index without file sizes and times; reindex the project.')
515
+ rows=conn.execute('SELECT path,size,mtime_ns FROM manifest').fetchall()
516
+ finally:
517
+ conn.close()
518
+ indexed=set()
519
+ for original,size,stamp in rows:
520
+ rel=original.replace('\\','/');indexed.add(rel)
521
+ if rel in changes:continue
522
+ try:
523
+ stat=os.stat(os.path.join(root,rel))
524
+ except OSError:
525
+ changes[rel]='deleted';continue
526
+ if (stat.st_size,stat.st_mtime_ns)!=(size,stamp):changes[rel]='changed'
527
+ scope=index_scope.load_scope(root)
528
+ listed=subprocess.run(['git','ls-files','-z'],cwd=root,capture_output=True,timeout=30,
529
+ creationflags=getattr(subprocess,'CREATE_NO_WINDOW',0))
530
+ if scope and listed.returncode==0:
531
+ for rel in listed.stdout.decode('utf-8','replace').split('\0'):
532
+ if rel and rel not in indexed and rel not in changes and rel.lower().endswith(SOURCE_EXTENSIONS) \
533
+ and index_scope.in_scope(scope,rel) and os.path.isfile(os.path.join(root,rel)):
534
+ changes[rel]='changed'
535
+ return changes
536
+
537
+
538
+ def build_current(root,view_id=None):
539
+ """The graph an agent should see while editing: the indexed view plus every file it does not reflect yet (see
540
+ pending_changes and build_overlay) when that view is the current one; a pinned or older view is returned as
541
+ indexed."""
542
+ inventory=index_inventory.inspect(root,view_id)
543
+ selected=inventory.get('selected') or {}
544
+ changes=pending_changes(root,selected['path']) if selected.get('current') else {}
545
+ return build_overlay(root,changes,view_id) if changes else build(root,view_id)
546
+
547
+
423
548
  def build_overlay(root,changes,view_id=None):
424
549
  """Graph of the indexed view with the given changed files read from the working tree instead of the snapshot
425
550
  (changes: project-relative path -> 'deleted' or any other status), so a caller right after an edit sees new files,
package/code_impact.py CHANGED
@@ -10,7 +10,7 @@ import index_profile
10
10
  import index_scope
11
11
 
12
12
  MAX_DEPTH = 4
13
- NOTE = ("Static analysis of the indexed snapshot: calls made through callbacks, dynamic dispatch or names built at "
13
+ NOTE = ("Static analysis of the indexed snapshot plus the files changed in the working tree: calls made through callbacks, dynamic dispatch or names built at "
14
14
  "runtime are not seen, so treat an empty list as 'none found', not as 'none exist'.")
15
15
 
16
16
 
@@ -54,7 +54,7 @@ def impact(root, symbol, view_id=None, depth=3, limit=30):
54
54
  raise ValueError("Pass symbol: a function, method (Class.method) or class name.")
55
55
  if type(depth) is not int or not 1 <= depth <= MAX_DEPTH:
56
56
  raise ValueError(f"depth must be an integer from 1 to {MAX_DEPTH}.")
57
- data = code_graph.build(root, view_id)
57
+ data = code_graph.build_current(root, view_id)
58
58
  found = _matches(data.get("symbols") or [], symbol)
59
59
  diagnostics = [d.get("reason") for d in data.get("diagnostics") or [] if d.get("reason")]
60
60
  if not found:
package/doc_check.py CHANGED
@@ -229,7 +229,7 @@ def coverage(root, include_tests=False, limit=30, view_id=None):
229
229
  """Docstring coverage of the indexed code: functions the language convention expects documented (plus any already
230
230
  documented) and those still missing, file by file, so an agent can document a project that started without it.
231
231
  Tests are left out unless include_tests."""
232
- data = code_graph.build(root, view_id)
232
+ data = code_graph.build_current(root, view_id)
233
233
  profile = index_profile.current((index_scope.load_scope(root) or {}).get("profile"))
234
234
  kinds = ("code", "test") if include_tests else ("code",)
235
235
  by_path = defaultdict(list)
package/duplicates.py CHANGED
@@ -9,6 +9,7 @@ import threading
9
9
  import time
10
10
 
11
11
  import code_graph
12
+ import doc_check
12
13
  import embedding_cache
13
14
  import index_inventory
14
15
  import index_profile
@@ -169,15 +170,10 @@ def _functions(root, include_tests):
169
170
  if any(o is not s and o["start_line"] <= s["start_line"] and s["end_line"] <= o["end_line"]
170
171
  and (o["start_line"], o["end_line"]) != (s["start_line"], s["end_line"]) for o in symbols):
171
172
  continue
172
- text = "\n".join(file_lines[s["start_line"] - 1:s["end_line"]])
173
- if not text.strip():
173
+ entry = _entry(path, s["name"], s["start_line"], s["end_line"], file_lines, ext)
174
+ if entry is None:
174
175
  continue
175
- body = _strip_comments(_body(text, ext), ext)
176
- normalized = re.sub(r"\s+", " ", body).strip()
177
- name = s["name"].split("(")[0].split(".")[-1]
178
- functions.append({"name": name, "path": path, "line": s["start_line"],
179
- "end": s["end_line"], "lines": s["end_line"] - s["start_line"] + 1, "text": text[:MAX_TEXT],
180
- "hash": hashlib.sha1(normalized.encode("utf-8")).hexdigest(), **_features(body, ext, name)})
176
+ functions.append(entry)
181
177
  occurrence = occurrences[functions[-1]["hash"]] = occurrences.get(functions[-1]["hash"], 0) + 1
182
178
  functions[-1]["kind"] = kind
183
179
  functions[-1]["fp"] = hashlib.sha1(f"{path}\0{functions[-1]['hash']}\0{occurrence}".encode("utf-8")).hexdigest()
@@ -191,6 +187,43 @@ def _functions(root, include_tests):
191
187
  return functions, notes, view
192
188
 
193
189
 
190
+ def _entry(path, name, start, end, file_lines, ext):
191
+ """One function as the duplicate checks compare it: normalized body hash and near-duplicate features."""
192
+ text = "\n".join(file_lines[start - 1:end])
193
+ if not text.strip():
194
+ return None
195
+ body = _strip_comments(_body(text, ext), ext)
196
+ normalized = re.sub(r"\s+", " ", body).strip()
197
+ short = name.split("(")[0].split(".")[-1]
198
+ return {"name": short, "path": path, "line": start, "end": end, "lines": end - start + 1, "text": text[:MAX_TEXT],
199
+ "hash": hashlib.sha1(normalized.encode("utf-8")).hexdigest(), **_features(body, ext, short)}
200
+
201
+
202
+ def _working_tree_functions(root, changes, profile):
203
+ """Production functions of the files changed in the working tree, read from disk, so a copy of a function written
204
+ earlier in the same session is caught before the index catches up."""
205
+ found = []
206
+ for rel, status in changes.items():
207
+ ext = rel.rsplit(".", 1)[-1].lower() if "." in rel else ""
208
+ if status == "deleted" or ext not in _LANG or index_profile.kind(rel, profile) != "code":
209
+ continue
210
+ try:
211
+ with open(os.path.join(root, rel), encoding="utf-8") as stream:
212
+ source = stream.read()
213
+ except (OSError, UnicodeDecodeError):
214
+ continue
215
+ if _GENERATED.search(source[:4000]):
216
+ continue
217
+ lines = source.splitlines()
218
+ for function in doc_check.functions(rel, source):
219
+ if function["end"] - function["start"] + 1 < MIN_LINES:
220
+ continue
221
+ entry = _entry(rel, function["name"], function["start"], function["end"], lines, ext)
222
+ if entry:
223
+ found.append({**entry, "kind": "code"})
224
+ return found
225
+
226
+
194
227
  def _ref(f):
195
228
  return {"name": f["name"], "path": f["path"], "lines": f"{f['line']}-{f['end']}"}
196
229
 
@@ -381,8 +414,9 @@ def find(root, embed, configured_model, min_similarity=DEFAULT_MIN_SIMILARITY, i
381
414
 
382
415
 
383
416
  def _write_pool(root):
384
- """Indexed production functions with their idiom shingles, rebuilt only when the index changes; None while the
385
- code graph is not cached yet, so an edit hook never waits for an analysis."""
417
+ """Indexed production functions with their idiom shingles, with the files changed in the working tree read from
418
+ disk (see code_graph.pending_changes). The indexed part is rebuilt only when the index changes and the changed
419
+ files on every call (an edit hook sees what earlier edits and commits wrote). None while the code graph is not cached yet, so a hook never waits for it."""
386
420
  view_path = indexer.existing_db_path(root)
387
421
  if not view_path or code_graph.cached_symbols(root)[0] is None:
388
422
  return None
@@ -390,13 +424,16 @@ def _write_pool(root):
390
424
  with _LOCK:
391
425
  cached = _WRITE_POOLS.get(root)
392
426
  if cached and cached[0] == state:
393
- return cached[1]
394
- functions, _notes, _view = _functions(root, False)
395
- production = [f for f in functions if f["kind"] == "code"]
396
- pool = (production, _idioms(production))
397
- with _LOCK:
398
- _WRITE_POOLS[root] = (state, pool)
399
- return pool
427
+ indexed = cached[1]
428
+ else:
429
+ functions, _notes, _view = _functions(root, False)
430
+ indexed = [f for f in functions if f["kind"] == "code"]
431
+ with _LOCK:
432
+ _WRITE_POOLS[root] = (state, indexed)
433
+ changes = code_graph.pending_changes(root, view_path)
434
+ profile = index_profile.current((index_scope.load_scope(root) or {}).get("profile"))
435
+ production = [f for f in indexed if f["path"] not in changes] + _working_tree_functions(root, changes, profile)
436
+ return production, _idioms(production)
400
437
 
401
438
 
402
439
  def on_write(root, path, before, after, written):
package/hook_decision.py CHANGED
@@ -16,6 +16,8 @@ import config
16
16
  import doc_check
17
17
  import duplicates
18
18
  import edit_preview
19
+ import index_profile
20
+ import index_scope
19
21
  import endpoint_sync
20
22
  import project_store
21
23
  import router
@@ -116,9 +118,10 @@ def _duplicate_lines(findings):
116
118
 
117
119
 
118
120
  def _edit_review(payload, doc_mode, duplicate_mode):
119
- """Checks of an edit before it runs. Docstrings follow doc_mode: missing ones deny the edit in require mode and
120
- become a reminder in remind mode; a documented function whose signature changed always gets a reminder. Copies of
121
- indexed functions follow duplicate_mode and only add context, never deny."""
121
+ """Checks of an edit before it runs. Docstrings follow doc_mode, in code files of registered projects only (tests,
122
+ docs and files outside a project are left out, as in the docs coverage): missing ones deny the edit in require mode and become a reminder in remind mode; a
123
+ documented function whose signature changed always gets a reminder. Copies of indexed functions follow
124
+ duplicate_mode and only add context, never deny."""
122
125
  doc_findings, duplicate_findings, failures = [], [], []
123
126
  for path, before, after in edit_preview.preview(payload.get("tool_name"), payload.get("tool_input"), payload.get("cwd")):
124
127
  if not doc_check.supported(path):
@@ -126,7 +129,8 @@ def _edit_review(payload, doc_mode, duplicate_mode):
126
129
  root = _project_of(path)
127
130
  rel = os.path.relpath(path, root).replace(os.sep, "/") if root else os.path.basename(path)
128
131
  edited = doc_check.written(path, before, after)
129
- if doc_mode != "off":
132
+ profile = index_profile.current((index_scope.load_scope(root) or {}).get("profile")) if root else None
133
+ if doc_mode != "off" and root and index_profile.kind(rel, profile) == "code":
130
134
  doc_findings += [(root, rel, item) for item in doc_check.review(path, before, after, edited)]
131
135
  if duplicate_mode != "off" and root and edited["touched"]:
132
136
  try:
package/index_scope.py CHANGED
@@ -415,6 +415,22 @@ def _exclude_matcher(exclude):
415
415
  return matches
416
416
 
417
417
 
418
+ def in_scope(scope, rel_path):
419
+ """Whether resolve_included_files would list rel_path: under an include base, outside the exclusions and outside
420
+ hidden folders."""
421
+ rel_path = rel_path.replace("\\", "/")
422
+ include = scope.get("include") or [""]
423
+ matcher = _exclude_matcher(set(scope.get("exclude") or []) | set(scope.get("user_exclude") or []) |
424
+ _ALWAYS_EXCLUDE | _ALWAYS_EXCLUDE_FILES)
425
+ parts = rel_path.split("/")
426
+ if matcher(rel_path) or any(part.startswith(".") for part in parts[:-1]):
427
+ return False
428
+ if any(base in ("", ".") for base in include):
429
+ return True
430
+ return len(parts) == 1 or any(rel_path == base.strip("/") or rel_path.startswith(base.strip("/") + "/")
431
+ for base in include)
432
+
433
+
418
434
  def resolve_included_files(root, scope, cancel_check=None):
419
435
  if not valid_scope(scope):
420
436
  raise ValueError("Invalid scope; request a new project analysis.")
package/indexer.py CHANGED
@@ -841,6 +841,12 @@ def search(root, query_vector=None, top_k=20, query=None, model_id=None, include
841
841
  return [(scores[chunk_id], *by_id[chunk_id][1:5]) for chunk_id in ids[:top_k]]
842
842
 
843
843
 
844
+ def graph_cache_path(view_path):
845
+ """Where code_graph keeps the saved analysis of one index view (graph-cache/<view>.json.gz next to it)."""
846
+ return os.path.join(os.path.dirname(view_path), "graph-cache",
847
+ os.path.basename(view_path).removesuffix(".sqlite3") + ".json.gz")
848
+
849
+
844
850
  def remove_index(root):
845
851
  """Remove somente arquivos de índice da raiz conhecida, nunca arquivos do projeto."""
846
852
  import glob
@@ -855,6 +861,10 @@ def remove_index(root):
855
861
  except PermissionError as exc:
856
862
  raise RuntimeError("The index is open in another process. Close the external reader and try removing it again.") from exc
857
863
  removed.append(safe)
864
+ try:
865
+ os.remove(graph_cache_path(safe))
866
+ except FileNotFoundError:
867
+ pass
858
868
  invalidate_cache(root)
859
869
  return removed
860
870
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@allansantos-dev/smart-tool",
3
- "version": "0.9.3",
3
+ "version": "0.9.4",
4
4
  "description": "Local MCP server that gives coding agents (Claude Code, Codex) cheaper, sharper tools than their built-in search. Windows.",
5
5
  "license": "Apache-2.0",
6
6
  "author": "Allan Santos",
package/router.py CHANGED
@@ -292,18 +292,21 @@ def log_unavailable(tool_name, tool_input, exc):
292
292
  )
293
293
 
294
294
 
295
- # Verbo ancorado no início do comando ou de um segmento (`|`, `;`, `&&`, `$(`): casar por
296
- # substring solta faria qualquer palavra inglesa dentro de um heredoc Python contar como
297
- # busca. A ausência de verbo é prova mecânica de que o `command` não busca código.
295
+ # Search/read verbs count only where they start a command: at the beginning, after `;`, `&&`, `||`, `$(`, a backtick,
296
+ # a loop/if keyword or a shell invoker (`bash -c`, `powershell -Command`, `cmd /c`). After a single `|` they filter the
297
+ # output of the command before them (`python x.py | tail -5`), which reads no source code.
298
298
  _SEARCH_VERB_RE = re.compile(
299
- # Início, separador de segmento, ou logo depois de um invocador de shell
300
- # (`powershell -Command`, `bash -c`, `cmd /c`), que é como o verbo real costuma
301
- # aparecer no meio de um comando em máquina Windows.
302
- r"(?:^|[|;&]|\$\(|`|-c\s|-Command\s|/c\s|/k\s)\s*(?:sudo\s+|command\s+)?"
299
+ r"(?:^|[;&]|\|\||\$\(|`|-c\s|-Command\s|/c\s|/k\s)\s*[\"']?\s*(?:(?:do|then|else)\s+)?(?:sudo\s+|command\s+)?"
303
300
  r"(grep|egrep|fgrep|rg|ripgrep|find|findstr|fd|cat|type|head|tail|less|more|"
304
301
  r"ls|dir|awk|sed|ack|ag|select-string|sls|get-content|gc|get-childitem|gci)\b",
305
302
  re.IGNORECASE,
306
303
  )
304
+ # Commands that read code or pages without a search verb at their start: their pipe filters still search content.
305
+ _READ_PRODUCER_RE = re.compile(
306
+ r"\bgit\s+(grep|ls-files|show|log\s+-p|diff)\b|\b(curl|wget|iwr|invoke-webrequest)\b|\bgh\s+api\b|"
307
+ r"readFileSync|read_text\(|open\(",
308
+ re.IGNORECASE,
309
+ )
307
310
  _READ_TOOLS = frozenset({"Read", "NotebookRead"})
308
311
 
309
312
  BREAKER_PATH = os.path.join(os.path.dirname(METRICS_PATH), "router-breaker.json")
@@ -318,14 +321,17 @@ def _mechanical_decision(tool_name, tool_input):
318
321
  Medido em 1973 decisões reais: 97,8% eram `allow`, e 1512 delas (77%) caíam nestes
319
322
  casos — cada uma custando uma ida ao gateway antes de toda tool call do agente. Os
320
323
  dois únicos `Read` redirecionados nesse histórico eram decisões erradas (um deles com
321
- `offset`/`limit` explícitos, onde só as linhas pedidas servem)."""
324
+ `offset`/`limit` explícitos, onde só as linhas pedidas servem). Remedido em 2026-10-08
325
+ sobre 2488 Bash que foram ao modelo (mediana 1,5 s, 7,8% redirecionados): ignorar os
326
+ filtros depois de `|` (salvo quando o produtor lê código ou páginas) decide 439 deles
327
+ sem modelo (13,3 min em 58 h) e perde 1 dos 155 redirecionamentos."""
322
328
  if tool_name in _READ_TOOLS:
323
329
  return "allow", "Reading a specific file is never replaced by semantic search."
324
330
  if tool_name == "Bash":
325
331
  command = tool_input.get("command")
326
332
  if not isinstance(command, str) or not command.strip():
327
333
  return "allow", "Bash without a readable command."
328
- if not _SEARCH_VERB_RE.search(command):
334
+ if not _SEARCH_VERB_RE.search(command) and not _READ_PRODUCER_RE.search(command):
329
335
  return "allow", "The command neither searches nor reads source code."
330
336
  return None
331
337
 
@@ -860,6 +860,8 @@ def _prepare_project_index(arguments, job_id, cfg, token, query_vector=None, pre
860
860
  if not stats.get("ready", True):
861
861
  raise RuntimeError(f"{stats.get('failed', 0)} file(s) could not be read consistently. Resume indexing.")
862
862
  project_store.confirm_changes(project_identity.project_id(root), project.get('dirty_seq', 0))
863
+ import code_graph
864
+ code_graph.warm(root)
863
865
  if lexical:
864
866
  stats["warning"] = LEXICAL_WARNING
865
867
  return stats
package/version.py CHANGED
@@ -1,7 +1,7 @@
1
1
  """Single source of the Smart Tool version and of the User-Agent sent to public APIs."""
2
2
  import re
3
3
 
4
- VERSION = "0.9.3"
4
+ VERSION = "0.9.4"
5
5
  _CONTACT_RE = re.compile(r"^(?:[^\s@()<>;]+@[^\s@()<>;]+\.[^\s@()<>;]+|https?://[^\s()<>;]+)$")
6
6
 
7
7