hubble-cli 4.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
hubble/tools.py ADDED
@@ -0,0 +1,763 @@
1
+ """Workspace tools exposed to the model through native function calling.
2
+
3
+ Every path is confined to the workspace root (plus configured additional_dirs),
4
+ secret files are refused, and file changes are snapshotted for /undo.
5
+ """
6
+
7
+ import difflib
8
+ import fnmatch
9
+ import os
10
+ import re
11
+ import shutil
12
+ import subprocess
13
+ import sys
14
+ from dataclasses import dataclass, field
15
+ from pathlib import Path
16
+ from typing import Any, Dict, List, Optional
17
+
18
+ MAX_OUTPUT_CHARS = 30000
19
+ MAX_READ_LINES = 2000
20
+ MAX_LINE_CHARS = 2000
21
+ IGNORED_DIRS = {".git", "__pycache__", ".venv", "venv", "node_modules", "dist", "build",
22
+ ".mypy_cache", ".pytest_cache", ".idea", ".tox", ".next", "target"}
23
+ SECRET_NAMES = {"id_rsa", "id_dsa", "id_ecdsa", "id_ed25519", "credentials", ".netrc", ".pgpass",
24
+ ".npmrc", ".pypirc"}
25
+ SECRET_SUFFIXES = {".pem", ".key", ".p12", ".pfx", ".keystore", ".jks"}
26
+ SAFE_ENV_SUFFIXES = (".example", ".sample", ".template", ".dist")
27
+
28
+
29
+ class ToolError(Exception):
30
+ pass
31
+
32
+
33
+ def is_secret_path(path: Path) -> bool:
34
+ name = path.name.lower()
35
+ if name == ".env" or name.endswith(".env"):
36
+ return True
37
+ if name.startswith(".env.") and not name.endswith(SAFE_ENV_SUFFIXES):
38
+ return True
39
+ return name in SECRET_NAMES or path.suffix.lower() in SECRET_SUFFIXES
40
+
41
+
42
+ def truncate(text: str, limit: int = MAX_OUTPUT_CHARS) -> str:
43
+ if len(text) <= limit:
44
+ return text
45
+ head = text[: limit * 2 // 3]
46
+ tail = text[-limit // 3:]
47
+ return f"{head}\n\n... [{len(text) - len(head) - len(tail)} chars truncated] ...\n\n{tail}"
48
+
49
+
50
+ def detect_shell(preference: str = "auto") -> List[str]:
51
+ """argv prefix for running a command string."""
52
+ pref = (preference or "auto").lower()
53
+ if sys.platform == "win32":
54
+ if pref in ("bash", "git-bash") and shutil.which("bash"):
55
+ return [shutil.which("bash"), "-lc"]
56
+ if pref == "cmd":
57
+ return ["cmd", "/d", "/s", "/c"]
58
+ exe = shutil.which("pwsh") if pref in ("auto", "pwsh") else None
59
+ exe = exe or shutil.which("powershell") or "powershell"
60
+ return [exe, "-NoProfile", "-NonInteractive", "-Command"]
61
+ return [shutil.which("bash") or "/bin/sh", "-c"] if pref in ("auto", "bash") else [pref, "-c"]
62
+
63
+
64
+ def shell_name(argv: List[str]) -> str:
65
+ exe = Path(argv[0]).stem.lower()
66
+ return {"pwsh": "PowerShell 7", "powershell": "Windows PowerShell 5.1", "cmd": "cmd.exe"}.get(exe, exe)
67
+
68
+
69
+ @dataclass
70
+ class ToolContext:
71
+ root: Path
72
+ extra_dirs: List[Path] = field(default_factory=list)
73
+ allow_secrets: bool = False
74
+ shell_argv: List[str] = field(default_factory=lambda: detect_shell())
75
+ shell_timeout: int = 120
76
+ sandbox: str = "off" # "off" | "docker": run shell commands in an isolated container
77
+ sandbox_image: str = "python:3.12-slim"
78
+ sandbox_memory: str = "1g"
79
+ sandbox_cpus: str = "2"
80
+ sandbox_network: bool = True
81
+ read_mtimes: Dict[str, float] = field(default_factory=dict)
82
+ todos: List[Dict[str, str]] = field(default_factory=list)
83
+ # One dict per agent turn: absolute path -> original bytes (None if the file did not exist).
84
+ checkpoints: List[Dict[str, Optional[bytes]]] = field(default_factory=list)
85
+
86
+ def resolve(self, raw: str) -> Path:
87
+ if not raw or not str(raw).strip():
88
+ raise ToolError("path is required")
89
+ p = Path(str(raw).strip()).expanduser()
90
+ if not p.is_absolute():
91
+ p = self.root / p
92
+ p = p.resolve()
93
+ for base in [self.root, *self.extra_dirs]:
94
+ if p == base or p.is_relative_to(base):
95
+ return p
96
+ raise ToolError(f"'{raw}' is outside the workspace ({self.root}). "
97
+ "Add the directory to additional_dirs in .hubble/settings.json to allow it.")
98
+
99
+ def rel(self, p: Path) -> str:
100
+ try:
101
+ return p.relative_to(self.root).as_posix() or "."
102
+ except ValueError:
103
+ return str(p)
104
+
105
+ def check_secret(self, p: Path):
106
+ if is_secret_path(p) and not self.allow_secrets:
107
+ raise ToolError(f"Access to '{self.rel(p)}' is blocked: it looks like a secrets file.")
108
+
109
+ def begin_turn(self):
110
+ self.checkpoints.append({})
111
+
112
+ def reset_state(self):
113
+ self.checkpoints.clear()
114
+ self.read_mtimes.clear()
115
+ self.todos = []
116
+
117
+ def snapshot(self, p: Path):
118
+ if not self.checkpoints:
119
+ self.begin_turn()
120
+ key = str(p)
121
+ if key not in self.checkpoints[-1]:
122
+ self.checkpoints[-1][key] = p.read_bytes() if p.exists() else None
123
+
124
+ def undo(self) -> List[str]:
125
+ """Revert the most recent turn that changed files. Returns changed relative paths."""
126
+ while self.checkpoints and not self.checkpoints[-1]:
127
+ self.checkpoints.pop()
128
+ if not self.checkpoints:
129
+ return []
130
+ restored = []
131
+ for key, original in self.checkpoints.pop().items():
132
+ p = Path(key)
133
+ if original is None:
134
+ if p.exists():
135
+ p.unlink()
136
+ else:
137
+ p.parent.mkdir(parents=True, exist_ok=True)
138
+ p.write_bytes(original)
139
+ self.read_mtimes.pop(key, None)
140
+ restored.append(self.rel(p))
141
+ return restored
142
+
143
+
144
+ def _read_text(p: Path) -> str:
145
+ with open(p, "r", encoding="utf-8", errors="replace", newline="") as f:
146
+ return f.read()
147
+
148
+
149
+ def _read_text_strict(p: Path) -> str:
150
+ """For files about to be rewritten: lossy decoding would corrupt non-UTF-8 bytes."""
151
+ try:
152
+ with open(p, "r", encoding="utf-8", newline="") as f:
153
+ return f.read()
154
+ except UnicodeDecodeError:
155
+ raise ToolError(f"'{p.name}' is not valid UTF-8; refusing to rewrite it (it would be corrupted). "
156
+ "Edit it with a shell command that preserves its encoding, or ask the user.")
157
+
158
+
159
+ def _write_text(p: Path, text: str):
160
+ p.parent.mkdir(parents=True, exist_ok=True)
161
+ with open(p, "w", encoding="utf-8", newline="") as f:
162
+ f.write(text)
163
+
164
+
165
+ def _is_binary(p: Path) -> bool:
166
+ try:
167
+ with open(p, "rb") as f:
168
+ return b"\0" in f.read(8192)
169
+ except OSError:
170
+ return False
171
+
172
+
173
+ def unified_diff(old: str, new: str, name: str) -> str:
174
+ diff = difflib.unified_diff(old.replace("\r\n", "\n").splitlines(keepends=True),
175
+ new.replace("\r\n", "\n").splitlines(keepends=True),
176
+ fromfile=f"a/{name}", tofile=f"b/{name}", n=3)
177
+ return "".join(diff)
178
+
179
+
180
+ class Tool:
181
+ name = ""
182
+ description = ""
183
+ parameters: Dict[str, Any] = {}
184
+ kind = "read" # read | edit | exec
185
+
186
+ def schema(self) -> Dict[str, Any]:
187
+ return {"type": "function", "function": {
188
+ "name": self.name, "description": self.description, "parameters": self.parameters}}
189
+
190
+ def target(self, args: Dict[str, Any]) -> str:
191
+ """String matched against permission rules, e.g. a path or a command."""
192
+ return str(args.get("path") or args.get("command") or "")
193
+
194
+ def preview(self, args: Dict[str, Any], ctx: ToolContext) -> Optional[str]:
195
+ """Diff or description shown in the approval prompt."""
196
+ return None
197
+
198
+ def precheck(self, args: Dict[str, Any], ctx: ToolContext):
199
+ """Raise ToolError for calls certain to fail, so the user is not asked to approve them."""
200
+
201
+ def validate(self, args: Dict[str, Any]):
202
+ props = self.parameters.get("properties", {})
203
+ for req in self.parameters.get("required", []):
204
+ if req not in args or args[req] is None:
205
+ raise ToolError(f"missing required argument '{req}'")
206
+ for key, val in args.items():
207
+ expected = props.get(key, {}).get("type")
208
+ if expected == "integer" and not isinstance(val, int):
209
+ try:
210
+ args[key] = int(val)
211
+ except (TypeError, ValueError):
212
+ raise ToolError(f"argument '{key}' must be an integer")
213
+ elif expected == "boolean" and isinstance(val, str):
214
+ args[key] = val.lower() in ("true", "1", "yes")
215
+ elif expected == "string" and not isinstance(val, str):
216
+ args[key] = str(val)
217
+
218
+ def run(self, args: Dict[str, Any], ctx: ToolContext) -> str:
219
+ raise NotImplementedError
220
+
221
+
222
+ class ReadFile(Tool):
223
+ name = "read_file"
224
+ description = ("Read a text file from the workspace. Returns lines prefixed with line numbers "
225
+ "(`N | text`; the prefix is not part of the file). Use offset/limit for large files. "
226
+ "Always read a file before editing it.")
227
+ parameters = {"type": "object", "properties": {
228
+ "path": {"type": "string", "description": "File path relative to the workspace root"},
229
+ "offset": {"type": "integer", "description": "1-based line to start from (default 1)"},
230
+ "limit": {"type": "integer", "description": f"Max lines to return (default {MAX_READ_LINES})"},
231
+ }, "required": ["path"]}
232
+
233
+ def run(self, args, ctx):
234
+ p = ctx.resolve(args["path"])
235
+ ctx.check_secret(p)
236
+ if not p.exists():
237
+ raise ToolError(f"File not found: {args['path']}")
238
+ if p.is_dir():
239
+ raise ToolError(f"'{args['path']}' is a directory; use list_dir")
240
+ if _is_binary(p):
241
+ raise ToolError(f"'{args['path']}' is a binary file")
242
+ lines = _read_text(p).splitlines()
243
+ offset = max(1, int(args.get("offset") or 1))
244
+ limit = max(1, min(int(args.get("limit") or MAX_READ_LINES), MAX_READ_LINES))
245
+ chunk = lines[offset - 1: offset - 1 + limit]
246
+ ctx.read_mtimes[str(p)] = p.stat().st_mtime
247
+ if not lines:
248
+ return f"[{ctx.rel(p)} is empty]"
249
+ width = len(str(offset + len(chunk)))
250
+ body = "\n".join(f"{i:>{width}} | {line[:MAX_LINE_CHARS]}"
251
+ for i, line in enumerate(chunk, offset))
252
+ end = offset + len(chunk) - 1
253
+ header = f"[{ctx.rel(p)} lines {offset}-{end} of {len(lines)}]"
254
+ more = f"\n[... {len(lines) - end} more lines; continue with offset={end + 1}]" if end < len(lines) else ""
255
+ return truncate(f"{header}\n{body}{more}")
256
+
257
+
258
+ def _loose_line_match(text: str, old: str) -> Optional[tuple]:
259
+ """Find old as whole lines, ignoring trailing whitespace and surrounding blank lines.
260
+
261
+ Returns (start, end) character offsets of the unique match, excluding the final line break.
262
+ """
263
+ old_lines = [l.rstrip() for l in old.strip("\r\n").replace("\r\n", "\n").split("\n")]
264
+ if not old_lines or not any(old_lines):
265
+ return None
266
+ lines = text.splitlines(keepends=True)
267
+ offsets, pos = [], 0
268
+ for line in lines:
269
+ offsets.append(pos)
270
+ pos += len(line)
271
+ hits = [i for i in range(len(lines) - len(old_lines) + 1)
272
+ if all(lines[i + j].rstrip() == old_lines[j] for j in range(len(old_lines)))]
273
+ if len(hits) != 1:
274
+ return None
275
+ i = hits[0]
276
+ last = lines[i + len(old_lines) - 1]
277
+ end = offsets[i + len(old_lines) - 1] + len(last.rstrip("\r\n"))
278
+ return offsets[i], end
279
+
280
+
281
+ def _require_fresh_read(p: Path, ctx: ToolContext, verb: str):
282
+ if not p.exists():
283
+ return
284
+ seen = ctx.read_mtimes.get(str(p))
285
+ if seen is None:
286
+ raise ToolError(f"You must read_file '{ctx.rel(p)}' before you {verb} it.")
287
+ if p.stat().st_mtime > seen + 1e-6:
288
+ raise ToolError(f"'{ctx.rel(p)}' changed since you last read it. Read it again first.")
289
+
290
+
291
+ class WriteFile(Tool):
292
+ name = "write_file"
293
+ description = ("Create a new file or completely overwrite an existing one. Prefer edit_file for "
294
+ "changes to existing files. Existing files must be read first.")
295
+ parameters = {"type": "object", "properties": {
296
+ "path": {"type": "string", "description": "File path relative to the workspace root"},
297
+ "content": {"type": "string", "description": "Full file content"},
298
+ }, "required": ["path", "content"]}
299
+ kind = "edit"
300
+
301
+ def precheck(self, args, ctx):
302
+ p = ctx.resolve(args["path"])
303
+ ctx.check_secret(p)
304
+ if p.is_dir():
305
+ raise ToolError(f"'{args['path']}' is a directory")
306
+ _require_fresh_read(p, ctx, "overwrite")
307
+
308
+ def preview(self, args, ctx):
309
+ p = ctx.resolve(args["path"])
310
+ old = _read_text(p) if p.is_file() else ""
311
+ return unified_diff(old, args.get("content", ""), ctx.rel(p)) or "(no changes)"
312
+
313
+ def run(self, args, ctx):
314
+ p = ctx.resolve(args["path"])
315
+ ctx.check_secret(p)
316
+ if p.is_dir():
317
+ raise ToolError(f"'{args['path']}' is a directory")
318
+ _require_fresh_read(p, ctx, "overwrite")
319
+ content = args["content"]
320
+ existed = p.exists()
321
+ if existed and "\r\n" in _read_text_strict(p) and "\r\n" not in content:
322
+ content = content.replace("\n", "\r\n")
323
+ ctx.snapshot(p)
324
+ _write_text(p, content)
325
+ ctx.read_mtimes[str(p)] = p.stat().st_mtime
326
+ verb = "Updated" if existed else "Created"
327
+ return f"{verb} {ctx.rel(p)} ({len(content.splitlines())} lines)"
328
+
329
+
330
+ class EditFile(Tool):
331
+ name = "edit_file"
332
+ description = ("Replace an exact string in a file. old_string must match the file exactly "
333
+ "(including indentation, without line-number prefixes) and be unique unless "
334
+ "replace_all is true. Include surrounding lines to make it unique.")
335
+ parameters = {"type": "object", "properties": {
336
+ "path": {"type": "string", "description": "File path relative to the workspace root"},
337
+ "old_string": {"type": "string", "description": "Exact text to replace"},
338
+ "new_string": {"type": "string", "description": "Replacement text"},
339
+ "replace_all": {"type": "boolean", "description": "Replace every occurrence (default false)"},
340
+ }, "required": ["path", "old_string", "new_string"]}
341
+ kind = "edit"
342
+
343
+ def _apply(self, args, ctx):
344
+ p = ctx.resolve(args["path"])
345
+ ctx.check_secret(p)
346
+ if not p.is_file():
347
+ raise ToolError(f"File not found: {args['path']}. Use write_file to create it.")
348
+ text = _read_text_strict(p)
349
+ old, new = args["old_string"], args["new_string"]
350
+ if not old:
351
+ raise ToolError("old_string is empty. Use write_file to create or replace a whole file.")
352
+ if old == new:
353
+ raise ToolError("old_string and new_string are identical")
354
+ crlf = "\r\n" in text
355
+ count = text.count(old)
356
+ if count == 0 and crlf and "\r\n" not in old:
357
+ old, new = old.replace("\n", "\r\n"), new.replace("\n", "\r\n")
358
+ count = text.count(old)
359
+ if count == 0:
360
+ span = _loose_line_match(text, old)
361
+ if span:
362
+ start, end = span
363
+ replacement = new.strip("\r\n").replace("\r\n", "\n")
364
+ if crlf:
365
+ replacement = replacement.replace("\n", "\r\n")
366
+ return p, text, text[:start] + replacement + text[end:], 1
367
+ if count == 0:
368
+ hint = ""
369
+ first = old.strip().splitlines()[0].strip() if old.strip() else ""
370
+ if first and first in text:
371
+ line_no = text[: text.index(first)].count("\n") + 1
372
+ hint = f" The first line appears near line {line_no}; check whitespace and re-read the file."
373
+ raise ToolError(f"old_string not found in {ctx.rel(p)}.{hint}")
374
+ if count > 1 and not args.get("replace_all"):
375
+ raise ToolError(f"old_string matches {count} places in {ctx.rel(p)}. "
376
+ "Add surrounding context to make it unique, or set replace_all=true.")
377
+ updated = text.replace(old, new) if args.get("replace_all") else text.replace(old, new, 1)
378
+ return p, text, updated, count
379
+
380
+ def precheck(self, args, ctx):
381
+ p, _, _, _ = self._apply(args, ctx)
382
+ _require_fresh_read(p, ctx, "edit")
383
+
384
+ def preview(self, args, ctx):
385
+ try:
386
+ p, text, updated, _ = self._apply(args, ctx)
387
+ except ToolError as e:
388
+ return f"(edit will fail: {e})"
389
+ return unified_diff(text, updated, ctx.rel(p))
390
+
391
+ def run(self, args, ctx):
392
+ p, text, updated, count = self._apply(args, ctx)
393
+ _require_fresh_read(p, ctx, "edit")
394
+ ctx.snapshot(p)
395
+ _write_text(p, updated)
396
+ ctx.read_mtimes[str(p)] = p.stat().st_mtime
397
+ n = count if args.get("replace_all") else 1
398
+ return f"Edited {ctx.rel(p)} ({n} replacement{'s' if n != 1 else ''})"
399
+
400
+
401
+ class Shell(Tool):
402
+ name = "shell"
403
+ description = ("Run a shell command in the workspace root and return stdout, stderr and exit code. "
404
+ "Use it for builds, tests, git and package managers. Commands are non-interactive "
405
+ "(stdin is closed) and time out. Do not use it to read or edit files; use the file tools.")
406
+ parameters = {"type": "object", "properties": {
407
+ "command": {"type": "string", "description": "Command to run"},
408
+ "timeout": {"type": "integer", "description": "Timeout in seconds (default 120, max 600)"},
409
+ "description": {"type": "string", "description": "5-10 word summary of what the command does"},
410
+ }, "required": ["command"]}
411
+ kind = "exec"
412
+
413
+ def preview(self, args, ctx):
414
+ return args.get("command", "")
415
+
416
+ def run(self, args, ctx):
417
+ cmd = args["command"]
418
+ timeout = max(1, min(int(args.get("timeout") or ctx.shell_timeout), 600))
419
+ if ctx.sandbox == "docker":
420
+ return self._run_docker(cmd, timeout, ctx)
421
+ return self._run_host(cmd, timeout, ctx)
422
+
423
+ def _run_host(self, cmd: str, timeout: int, ctx: ToolContext) -> str:
424
+ argv = list(ctx.shell_argv)
425
+ if "powershell" in argv[0].lower() or "pwsh" in argv[0].lower():
426
+ cmd = "[Console]::OutputEncoding=[System.Text.Encoding]::UTF8; " + cmd
427
+ env = dict(os.environ, PYTHONIOENCODING="utf-8", PYTHONUTF8="1")
428
+ # Own process group, so timeouts and Ctrl+C kill the whole tree. Otherwise grandchildren
429
+ # (dev servers, watchers) keep the pipes open and the CLI hangs.
430
+ group = ({"creationflags": subprocess.CREATE_NEW_PROCESS_GROUP} if sys.platform == "win32"
431
+ else {"start_new_session": True})
432
+ try:
433
+ proc = subprocess.Popen(argv + [cmd], cwd=ctx.root, stdin=subprocess.DEVNULL,
434
+ stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True,
435
+ encoding="utf-8", errors="replace", env=env, **group)
436
+ except OSError as e:
437
+ raise ToolError(f"Could not start shell: {e}")
438
+ try:
439
+ stdout, stderr = proc.communicate(timeout=timeout)
440
+ except subprocess.TimeoutExpired:
441
+ _kill_tree(proc)
442
+ stdout, stderr = _drain(proc)
443
+ raise ToolError(f"Command timed out after {timeout}s (process tree killed). "
444
+ "Long-running servers are not supported.\n" + truncate(stdout or "", 5000))
445
+ except KeyboardInterrupt:
446
+ _kill_tree(proc)
447
+ _drain(proc)
448
+ raise
449
+ return _format_result(stdout, stderr, proc.returncode)
450
+
451
+ def _run_docker(self, cmd: str, timeout: int, ctx: ToolContext) -> str:
452
+ docker = shutil.which("docker")
453
+ if not docker:
454
+ raise ToolError("sandbox is set to 'docker' but the docker CLI was not found on PATH. "
455
+ "Install Docker Desktop, or set shell_sandbox to \"off\".")
456
+ import uuid
457
+ name = f"hubble-sandbox-{uuid.uuid4().hex[:12]}"
458
+ argv = [docker, "run", "--rm", "--name", name, "-i",
459
+ "--memory", ctx.sandbox_memory, "--cpus", ctx.sandbox_cpus,
460
+ "-v", f"{ctx.root}:/workspace", "-w", "/workspace"]
461
+ if not ctx.sandbox_network:
462
+ argv += ["--network", "none"]
463
+ argv += [ctx.sandbox_image, "sh", "-lc", cmd]
464
+ try:
465
+ proc = subprocess.Popen(argv, stdin=subprocess.DEVNULL, stdout=subprocess.PIPE,
466
+ stderr=subprocess.PIPE, text=True, encoding="utf-8", errors="replace")
467
+ except OSError as e:
468
+ raise ToolError(f"Could not start docker: {e}")
469
+ try:
470
+ stdout, stderr = proc.communicate(timeout=timeout)
471
+ except subprocess.TimeoutExpired:
472
+ self._docker_kill(docker, name)
473
+ stdout, stderr = _drain(proc)
474
+ raise ToolError(f"Command timed out after {timeout}s (sandbox container killed). "
475
+ "Long-running servers are not supported.\n" + truncate(stdout or "", 5000))
476
+ except KeyboardInterrupt:
477
+ self._docker_kill(docker, name)
478
+ _drain(proc)
479
+ raise
480
+ if proc.returncode == 125 and "Unable to find image" in (stderr or ""):
481
+ raise ToolError(f"Sandbox image '{ctx.sandbox_image}' could not be pulled.\n{stderr.strip()}")
482
+ if proc.returncode == 125 and "docker daemon" in (stderr or "").lower():
483
+ raise ToolError("Docker is installed but the daemon isn't running. Start Docker Desktop, "
484
+ f"or set shell_sandbox to \"off\".\n{stderr.strip()}")
485
+ return _format_result(stdout, stderr, proc.returncode)
486
+
487
+ @staticmethod
488
+ def _docker_kill(docker: str, name: str):
489
+ # Kill by container name directly: killing the `docker run` client process does not
490
+ # reliably stop the container itself, so the client kill alone is not enough.
491
+ try:
492
+ subprocess.run([docker, "kill", name], capture_output=True, stdin=subprocess.DEVNULL, timeout=10)
493
+ except (OSError, subprocess.TimeoutExpired):
494
+ pass
495
+
496
+
497
+ def _format_result(stdout: str, stderr: str, returncode: int) -> str:
498
+ parts = []
499
+ if stdout.strip():
500
+ parts.append(stdout.rstrip())
501
+ if stderr.strip():
502
+ parts.append(f"[stderr]\n{stderr.rstrip()}")
503
+ parts.append(f"[exit code {returncode}]")
504
+ return truncate("\n".join(parts))
505
+
506
+
507
+ def _kill_tree(proc: subprocess.Popen):
508
+ try:
509
+ if sys.platform == "win32":
510
+ subprocess.run(["taskkill", "/T", "/F", "/PID", str(proc.pid)], capture_output=True,
511
+ stdin=subprocess.DEVNULL, timeout=10)
512
+ else:
513
+ import signal
514
+ os.killpg(proc.pid, signal.SIGKILL)
515
+ except (OSError, subprocess.TimeoutExpired):
516
+ pass
517
+ try:
518
+ proc.kill()
519
+ except OSError:
520
+ pass
521
+
522
+
523
+ def _drain(proc: subprocess.Popen):
524
+ try:
525
+ return proc.communicate(timeout=5)
526
+ except (subprocess.TimeoutExpired, ValueError):
527
+ return "", ""
528
+
529
+
530
+ def _is_link(p: Path) -> bool:
531
+ try:
532
+ if p.is_symlink():
533
+ return True
534
+ # Path.is_junction() only exists from Python 3.12; fall back to the win32 reparse-point
535
+ # bit on older versions so a Windows junction is still refused pre-3.12, not skipped.
536
+ is_junction = getattr(p, "is_junction", None)
537
+ if is_junction is not None:
538
+ return is_junction()
539
+ if sys.platform == "win32":
540
+ attrs = getattr(p.stat(), "st_file_attributes", 0)
541
+ return bool(attrs & 0x400) # FILE_ATTRIBUTE_REPARSE_POINT
542
+ return False
543
+ except OSError:
544
+ return True
545
+
546
+
547
+ def _walk_files(base: Path):
548
+ for dirpath, dirs, files in os.walk(base):
549
+ # Links and junctions can point outside the workspace; do not descend into them.
550
+ dirs[:] = sorted(d for d in dirs if d not in IGNORED_DIRS and not d.endswith(".egg-info")
551
+ and not _is_link(Path(dirpath) / d))
552
+ for name in sorted(files):
553
+ yield Path(dirpath) / name
554
+
555
+
556
+ class Grep(Tool):
557
+ name = "grep"
558
+ description = ("Search file contents with a regular expression (ripgrep syntax). Returns "
559
+ "`path:line: text` matches, or only file paths with files_only=true.")
560
+ parameters = {"type": "object", "properties": {
561
+ "pattern": {"type": "string", "description": "Regular expression"},
562
+ "path": {"type": "string", "description": "File or directory to search (default workspace root)"},
563
+ "glob": {"type": "string", "description": "Only search files matching this glob, e.g. *.py"},
564
+ "ignore_case": {"type": "boolean", "description": "Case-insensitive search"},
565
+ "files_only": {"type": "boolean", "description": "Return only matching file paths"},
566
+ }, "required": ["pattern"]}
567
+ max_results = 200
568
+
569
+ def target(self, args):
570
+ return str(args.get("path") or ".")
571
+
572
+ def run(self, args, ctx):
573
+ base = ctx.resolve(args.get("path") or ".")
574
+ if base.is_file():
575
+ ctx.check_secret(base)
576
+ pattern = args["pattern"]
577
+ rg = shutil.which("rg")
578
+ if rg:
579
+ out = self._ripgrep(rg, pattern, base, args, ctx)
580
+ else:
581
+ out = self._python(pattern, base, args, ctx)
582
+ if not out:
583
+ return (f"No matches for /{pattern}/ in file contents. "
584
+ "(grep searches inside files; use glob to find files by name.)")
585
+ extra = f"\n[... truncated at {self.max_results} results]" if len(out) >= self.max_results else ""
586
+ return truncate("\n".join(out[: self.max_results]) + extra)
587
+
588
+ def _ripgrep(self, rg, pattern, base, args, ctx) -> List[str]:
589
+ cmd = [rg, "--no-heading", "--line-number", "--color", "never", "--max-columns", "300",
590
+ "--max-count", "50"]
591
+ if args.get("ignore_case"):
592
+ cmd.append("-i")
593
+ if args.get("files_only"):
594
+ cmd.append("-l")
595
+ if args.get("glob"):
596
+ cmd += ["--glob", args["glob"]]
597
+ if not ctx.allow_secrets:
598
+ # Speeds things up; the is_secret_path filter below is what enforces the policy.
599
+ cmd.append("--glob-case-insensitive")
600
+ for g in (".env", "*.env", ".env.*", "id_*", *SECRET_NAMES, *(f"*{s}" for s in SECRET_SUFFIXES)):
601
+ cmd += ["--glob", f"!{g}"]
602
+ cmd += ["-e", pattern, "--", ctx.rel(base)]
603
+ try:
604
+ res = subprocess.run(cmd, cwd=ctx.root, capture_output=True, text=True, encoding="utf-8",
605
+ errors="replace", timeout=60, stdin=subprocess.DEVNULL)
606
+ except (OSError, subprocess.TimeoutExpired) as e:
607
+ raise ToolError(f"ripgrep failed: {e}")
608
+ if res.returncode == 2 and not res.stdout:
609
+ raise ToolError(res.stderr.strip()[:500] or "ripgrep error")
610
+ lines = []
611
+ for line in res.stdout.splitlines():
612
+ m = re.match(r"^((?:[A-Za-z]:)?[^:]*)(:.*)?$", line)
613
+ path, rest = (m.group(1), m.group(2) or "") if m else (line, "")
614
+ path = path.replace("\\", "/").removeprefix("./")
615
+ if not ctx.allow_secrets and is_secret_path(Path(path)):
616
+ continue
617
+ lines.append(path + rest)
618
+ if len(lines) >= self.max_results:
619
+ break
620
+ return lines
621
+
622
+ def _python(self, pattern, base, args, ctx) -> List[str]:
623
+ try:
624
+ rx = re.compile(pattern, re.IGNORECASE if args.get("ignore_case") else 0)
625
+ except re.error as e:
626
+ raise ToolError(f"Invalid regex: {e}")
627
+ files = [base] if base.is_file() else _walk_files(base)
628
+ out = []
629
+ for f in files:
630
+ if args.get("glob") and not fnmatch.fnmatch(f.name, args["glob"]):
631
+ continue
632
+ if (is_secret_path(f) and not ctx.allow_secrets) or _is_binary(f):
633
+ continue
634
+ try:
635
+ with open(f, "r", encoding="utf-8", errors="ignore") as fh:
636
+ for i, line in enumerate(fh, 1):
637
+ if rx.search(line):
638
+ if args.get("files_only"):
639
+ out.append(ctx.rel(f))
640
+ break
641
+ out.append(f"{ctx.rel(f)}:{i}:{line.rstrip()[:300]}")
642
+ if len(out) >= self.max_results:
643
+ return out
644
+ except OSError:
645
+ continue
646
+ return out
647
+
648
+
649
+ class Glob(Tool):
650
+ name = "glob"
651
+ description = ("Find files by glob pattern, e.g. **/*.py or src/**/test_*.js. "
652
+ "Returns paths sorted by most recently modified.")
653
+ parameters = {"type": "object", "properties": {
654
+ "pattern": {"type": "string", "description": "Glob pattern relative to path"},
655
+ "path": {"type": "string", "description": "Directory to search in (default workspace root)"},
656
+ }, "required": ["pattern"]}
657
+ max_results = 200
658
+
659
+ def target(self, args):
660
+ return str(args.get("path") or ".")
661
+
662
+ def run(self, args, ctx):
663
+ base = ctx.resolve(args.get("path") or ".")
664
+ pattern = args["pattern"].replace("\\", "/")
665
+ if not pattern.startswith("**/") and "/" not in pattern:
666
+ pattern = "**/" + pattern
667
+ # fnmatch's * also matches "/", so "**/" only needs extra variants for zero directories.
668
+ variants = {pattern, pattern.replace("/**/", "/")}
669
+ if pattern.startswith("**/"):
670
+ variants.add(pattern[3:])
671
+ matches = []
672
+ for f in _walk_files(base):
673
+ rel = f.relative_to(base).as_posix()
674
+ if any(fnmatch.fnmatch(rel, v) for v in variants):
675
+ matches.append(f)
676
+ if not matches:
677
+ return f"No files match '{args['pattern']}'"
678
+
679
+ def mtime(f: Path) -> float:
680
+ try:
681
+ return f.stat().st_mtime
682
+ except OSError:
683
+ return 0.0
684
+ matches.sort(key=mtime, reverse=True)
685
+ more = f"\n[... {len(matches) - self.max_results} more]" if len(matches) > self.max_results else ""
686
+ return "\n".join(ctx.rel(f) for f in matches[: self.max_results]) + more
687
+
688
+
689
+ class ListDir(Tool):
690
+ name = "list_dir"
691
+ description = "List the files and subdirectories of a directory."
692
+ parameters = {"type": "object", "properties": {
693
+ "path": {"type": "string", "description": "Directory (default workspace root)"},
694
+ }}
695
+
696
+ def run(self, args, ctx):
697
+ p = ctx.resolve(args.get("path") or ".")
698
+ if not p.is_dir():
699
+ raise ToolError(f"Not a directory: {args.get('path')}")
700
+ entries = sorted((e for e in p.iterdir() if e.name not in IGNORED_DIRS),
701
+ key=lambda e: (not e.is_dir(), e.name.lower()))
702
+ lines = [f"[{ctx.rel(p)}]"]
703
+ for e in entries[:300]:
704
+ if e.is_dir():
705
+ lines.append(f" {e.name}/")
706
+ else:
707
+ try:
708
+ size = e.stat().st_size
709
+ except OSError:
710
+ size = 0
711
+ lines.append(f" {e.name} ({size} bytes)")
712
+ if len(entries) > 300:
713
+ lines.append(f" ... {len(entries) - 300} more")
714
+ return "\n".join(lines)
715
+
716
+
717
+ class TodoWrite(Tool):
718
+ name = "todo_write"
719
+ description = ("Create or update the task list for multi-step work. Send the full list each time. "
720
+ "Keep exactly one item in_progress while working; mark items completed as soon as done.")
721
+ parameters = {"type": "object", "properties": {
722
+ "todos": {"type": "array", "items": {"type": "object", "properties": {
723
+ "content": {"type": "string"},
724
+ "status": {"type": "string", "enum": ["pending", "in_progress", "completed"]},
725
+ }, "required": ["content", "status"]}},
726
+ }, "required": ["todos"]}
727
+
728
+ def target(self, args):
729
+ return ""
730
+
731
+ def validate(self, args):
732
+ super().validate(args)
733
+ if not isinstance(args["todos"], list):
734
+ raise ToolError("todos must be an array")
735
+
736
+ def run(self, args, ctx):
737
+ todos = []
738
+ for t in args["todos"]:
739
+ if isinstance(t, dict) and t.get("content"):
740
+ status = t.get("status", "pending")
741
+ todos.append({"content": str(t["content"]),
742
+ "status": status if status in ("pending", "in_progress", "completed") else "pending"})
743
+ ctx.todos = todos
744
+ done = sum(t["status"] == "completed" for t in todos)
745
+ return f"Todo list updated ({done}/{len(todos)} completed)"
746
+
747
+
748
+ def default_tools() -> List[Tool]:
749
+ return [ReadFile(), WriteFile(), EditFile(), Shell(), Grep(), Glob(), ListDir(), TodoWrite()]
750
+
751
+
752
+ READ_ONLY_TOOL_NAMES = {"read_file", "grep", "glob", "list_dir"}
753
+
754
+
755
+ def run_tool(tool: Tool, args: Dict[str, Any], ctx: ToolContext) -> tuple:
756
+ """Returns (output, is_error). Never raises for tool-level failures."""
757
+ try:
758
+ tool.validate(args)
759
+ return tool.run(args, ctx), False
760
+ except ToolError as e:
761
+ return f"Error: {e}", True
762
+ except OSError as e:
763
+ return f"Error: {e}", True