sanityops-cli 0.1.3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sanityops_cli/__init__.py +16 -0
- sanityops_cli/agents/__init__.py +14 -0
- sanityops_cli/agents/repair_agent/__init__.py +19 -0
- sanityops_cli/agents/repair_agent/agent.py +230 -0
- sanityops_cli/agents/repair_agent/prompts.py +100 -0
- sanityops_cli/agents/repair_agent/tools/__init__.py +19 -0
- sanityops_cli/agents/repair_agent/tools/store_repairs_tool.py +144 -0
- sanityops_cli/agents/scanner_agent/agent.py +332 -0
- sanityops_cli/agents/scanner_agent/hooks/progress_hook.py +152 -0
- sanityops_cli/agents/scanner_agent/models/finding.py +120 -0
- sanityops_cli/agents/scanner_agent/prompts.py +316 -0
- sanityops_cli/agents/scanner_agent/tools/grep_tool.py +667 -0
- sanityops_cli/agents/scanner_agent/tools/listfiles_tool.py +88 -0
- sanityops_cli/agents/scanner_agent/tools/readfile_tool.py +804 -0
- sanityops_cli/agents/scanner_agent/tools/storefindings_tool.py +178 -0
- sanityops_cli/api/__init__.py +16 -0
- sanityops_cli/api/client.py +403 -0
- sanityops_cli/commands/__init__.py +14 -0
- sanityops_cli/commands/config.py +312 -0
- sanityops_cli/commands/init.py +132 -0
- sanityops_cli/commands/inspect.py +646 -0
- sanityops_cli/constants/__init__.py +14 -0
- sanityops_cli/constants/config_defaults.py +22 -0
- sanityops_cli/constants/exit_codes.py +19 -0
- sanityops_cli/defect_checker/__init__.py +16 -0
- sanityops_cli/defect_checker/checker.py +100 -0
- sanityops_cli/defect_checker/llm_config.py +112 -0
- sanityops_cli/defect_checker/markdown_reporter.py +249 -0
- sanityops_cli/defect_checker/renderer.py +203 -0
- sanityops_cli/exceptions/__init__.py +14 -0
- sanityops_cli/exceptions/api_exceptions.py +60 -0
- sanityops_cli/exceptions/base_exceptions.py +25 -0
- sanityops_cli/help_panel.py +49 -0
- sanityops_cli/logging/__init__.py +18 -0
- sanityops_cli/logging/logger.py +108 -0
- sanityops_cli/main.py +123 -0
- sanityops_cli/progress/__init__.py +18 -0
- sanityops_cli/progress/tracker.py +159 -0
- sanityops_cli/renderers/__init__.py +14 -0
- sanityops_cli/renderers/command_renderer/inspect_command_renderer.py +87 -0
- sanityops_cli/templates/__init__.py +14 -0
- sanityops_cli/templates/inspect_config.yaml +55 -0
- sanityops_cli/utils/__init__.py +14 -0
- sanityops_cli/utils/artifact_packer.py +407 -0
- sanityops_cli/utils/config_loader.py +296 -0
- sanityops_cli/utils/config_resolver.py +358 -0
- sanityops_cli/utils/validators.py +117 -0
- sanityops_cli-0.1.3.dist-info/METADATA +213 -0
- sanityops_cli-0.1.3.dist-info/RECORD +53 -0
- sanityops_cli-0.1.3.dist-info/WHEEL +4 -0
- sanityops_cli-0.1.3.dist-info/entry_points.txt +2 -0
- sanityops_cli-0.1.3.dist-info/licenses/LICENSE +201 -0
- sanityops_cli-0.1.3.dist-info/licenses/NOTICE +5 -0
|
@@ -0,0 +1,804 @@
|
|
|
1
|
+
# Copyright 2026 zipsonken
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
#
|
|
15
|
+
|
|
16
|
+
"""
|
|
17
|
+
FileReadTool - A comprehensive file reading tool.
|
|
18
|
+
|
|
19
|
+
Adapted for the deeplogic-cli agent framework.
|
|
20
|
+
|
|
21
|
+
This module provides functionality to read various file types including:
|
|
22
|
+
- Text files (with line numbers and offset/limit support)
|
|
23
|
+
- Image files (PNG, JPG, JPEG, GIF, WEBP) with base64 encoding
|
|
24
|
+
- PDF files (with optional page range extraction)
|
|
25
|
+
- Jupyter notebooks (.ipynb)
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
import base64
|
|
29
|
+
import json
|
|
30
|
+
import os
|
|
31
|
+
from pathlib import Path
|
|
32
|
+
|
|
33
|
+
from sanityops_agent.tools.base import Tool, ToolResult
|
|
34
|
+
|
|
35
|
+
# =============================================================================
|
|
36
|
+
# Constants
|
|
37
|
+
# =============================================================================
|
|
38
|
+
|
|
39
|
+
# Supported image extensions
|
|
40
|
+
IMAGE_EXTENSIONS = {'png', 'jpg', 'jpeg', 'gif', 'webp'}
|
|
41
|
+
|
|
42
|
+
# Binary file extensions that cannot be read as text
|
|
43
|
+
BINARY_EXTENSIONS = {
|
|
44
|
+
'exe', 'dll', 'so', 'dylib', 'bin', 'dat',
|
|
45
|
+
'pyc', 'pyo', 'pyd', 'class', 'jar', 'war',
|
|
46
|
+
'zip', 'tar', 'gz', 'bz2', 'xz', '7z', 'rar',
|
|
47
|
+
'mp3', 'mp4', 'avi', 'mov', 'mkv', 'flv', 'wmv',
|
|
48
|
+
'wav', 'flac', 'aac', 'ogg',
|
|
49
|
+
'doc', 'docx', 'xls', 'xlsx', 'ppt', 'pptx',
|
|
50
|
+
'sqlite', 'db', 'mdb',
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
# Device files that would hang the process
|
|
54
|
+
BLOCKED_DEVICE_PATHS = {
|
|
55
|
+
'/dev/zero',
|
|
56
|
+
'/dev/random',
|
|
57
|
+
'/dev/urandom',
|
|
58
|
+
'/dev/full',
|
|
59
|
+
'/dev/stdin',
|
|
60
|
+
'/dev/tty',
|
|
61
|
+
'/dev/console',
|
|
62
|
+
'/dev/stdout',
|
|
63
|
+
'/dev/stderr',
|
|
64
|
+
'/dev/fd/0',
|
|
65
|
+
'/dev/fd/1',
|
|
66
|
+
'/dev/fd/2',
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
# Default limits
|
|
70
|
+
DEFAULT_MAX_SIZE_BYTES = 2 * 1024 * 1024 # 2MB
|
|
71
|
+
DEFAULT_MAX_TOKENS = 20000
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
# =============================================================================
|
|
75
|
+
# Helper Functions
|
|
76
|
+
# =============================================================================
|
|
77
|
+
|
|
78
|
+
def get_cwd() -> str:
|
|
79
|
+
"""Get current working directory."""
|
|
80
|
+
return os.getcwd()
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def expand_path(path: str, base_dir: str | None = None) -> str:
|
|
84
|
+
"""
|
|
85
|
+
Expand a path that may contain tilde notation (~) to an absolute path.
|
|
86
|
+
|
|
87
|
+
Args:
|
|
88
|
+
path: The path to expand
|
|
89
|
+
base_dir: Base directory for relative paths (defaults to cwd)
|
|
90
|
+
|
|
91
|
+
Returns:
|
|
92
|
+
The expanded absolute path
|
|
93
|
+
"""
|
|
94
|
+
actual_base_dir = base_dir or get_cwd()
|
|
95
|
+
|
|
96
|
+
# Handle empty path
|
|
97
|
+
if not path or not path.strip():
|
|
98
|
+
return os.path.normpath(actual_base_dir)
|
|
99
|
+
|
|
100
|
+
path = path.strip()
|
|
101
|
+
|
|
102
|
+
# Handle home directory notation
|
|
103
|
+
if path == '~':
|
|
104
|
+
return os.path.expanduser('~')
|
|
105
|
+
if path.startswith('~/'):
|
|
106
|
+
return os.path.normpath(os.path.join(os.path.expanduser('~'), path[2:]))
|
|
107
|
+
|
|
108
|
+
# Handle absolute paths
|
|
109
|
+
if os.path.isabs(path):
|
|
110
|
+
return os.path.normpath(path)
|
|
111
|
+
|
|
112
|
+
# Handle relative paths
|
|
113
|
+
return os.path.normpath(os.path.abspath(os.path.join(actual_base_dir, path)))
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def get_file_extension(file_path: str) -> str:
|
|
117
|
+
"""Get the file extension without the leading dot."""
|
|
118
|
+
return Path(file_path).suffix.lstrip('.').lower()
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def is_image_file(file_path: str) -> bool:
|
|
122
|
+
"""Check if file is an image based on extension."""
|
|
123
|
+
ext = get_file_extension(file_path)
|
|
124
|
+
return ext in IMAGE_EXTENSIONS
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def is_binary_file(file_path: str) -> bool:
|
|
128
|
+
"""Check if file is a binary file based on extension."""
|
|
129
|
+
ext = get_file_extension(file_path)
|
|
130
|
+
return ext in BINARY_EXTENSIONS
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def is_blocked_device(file_path: str) -> bool:
|
|
134
|
+
"""Check if path is a blocked device file."""
|
|
135
|
+
return file_path in BLOCKED_DEVICE_PATHS
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def is_notebook_file(file_path: str) -> bool:
|
|
139
|
+
"""Check if file is a Jupyter notebook."""
|
|
140
|
+
return get_file_extension(file_path) == 'ipynb'
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def is_pdf_file(file_path: str) -> bool:
|
|
144
|
+
"""Check if file is a PDF."""
|
|
145
|
+
return get_file_extension(file_path) == 'pdf'
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def add_line_numbers(content: str, start_line: int = 1) -> str:
|
|
149
|
+
"""
|
|
150
|
+
Add line numbers to content.
|
|
151
|
+
|
|
152
|
+
Args:
|
|
153
|
+
content: The text content
|
|
154
|
+
start_line: Starting line number (1-indexed)
|
|
155
|
+
|
|
156
|
+
Returns:
|
|
157
|
+
Content with line numbers prefixed
|
|
158
|
+
"""
|
|
159
|
+
lines = content.split('\n')
|
|
160
|
+
max_line_num = start_line + len(lines) - 1
|
|
161
|
+
width = len(str(max_line_num))
|
|
162
|
+
|
|
163
|
+
numbered_lines = []
|
|
164
|
+
for i, line in enumerate(lines):
|
|
165
|
+
line_num = start_line + i
|
|
166
|
+
numbered_lines.append(f"{line_num:{width}d}\t{line}")
|
|
167
|
+
|
|
168
|
+
return '\n'.join(numbered_lines)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def detect_image_format(data: bytes) -> tuple[str, str]:
|
|
172
|
+
"""
|
|
173
|
+
Detect image format from binary data.
|
|
174
|
+
|
|
175
|
+
Args:
|
|
176
|
+
data: Binary image data
|
|
177
|
+
|
|
178
|
+
Returns:
|
|
179
|
+
Tuple of (format_name, mime_type)
|
|
180
|
+
|
|
181
|
+
Raises:
|
|
182
|
+
ValueError: If format cannot be detected
|
|
183
|
+
"""
|
|
184
|
+
# PNG
|
|
185
|
+
if data[:8] == b'\x89PNG\r\n\x1a\n':
|
|
186
|
+
return 'png', 'image/png'
|
|
187
|
+
|
|
188
|
+
# JPEG
|
|
189
|
+
if data[:2] == b'\xff\xd8':
|
|
190
|
+
return 'jpeg', 'image/jpeg'
|
|
191
|
+
|
|
192
|
+
# GIF
|
|
193
|
+
if data[:6] in (b'GIF87a', b'GIF89a'):
|
|
194
|
+
return 'gif', 'image/gif'
|
|
195
|
+
|
|
196
|
+
# WebP
|
|
197
|
+
if data[:4] == b'RIFF' and data[8:12] == b'WEBP':
|
|
198
|
+
return 'webp', 'image/webp'
|
|
199
|
+
|
|
200
|
+
# Default to PNG
|
|
201
|
+
return 'png', 'image/png'
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def parse_pdf_page_range(pages: str) -> tuple[int, int] | None:
|
|
205
|
+
"""
|
|
206
|
+
Parse a PDF page range string.
|
|
207
|
+
|
|
208
|
+
Args:
|
|
209
|
+
pages: Page range string (e.g., "1-5", "3", "10-20")
|
|
210
|
+
|
|
211
|
+
Returns:
|
|
212
|
+
Tuple of (start_page, end_page) or None if invalid
|
|
213
|
+
"""
|
|
214
|
+
import re
|
|
215
|
+
|
|
216
|
+
# Single page
|
|
217
|
+
if re.match(r'^\d+$', pages):
|
|
218
|
+
page = int(pages)
|
|
219
|
+
if page >= 1:
|
|
220
|
+
return (page, page)
|
|
221
|
+
|
|
222
|
+
# Page range
|
|
223
|
+
match = re.match(r'^(\d+)-(\d+)$', pages)
|
|
224
|
+
if match:
|
|
225
|
+
start = int(match.group(1))
|
|
226
|
+
end = int(match.group(2))
|
|
227
|
+
if start >= 1 and end >= start:
|
|
228
|
+
return (start, end)
|
|
229
|
+
|
|
230
|
+
return None
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def rough_token_estimate(content: str) -> int:
|
|
234
|
+
"""
|
|
235
|
+
Estimate token count for content.
|
|
236
|
+
|
|
237
|
+
Uses a simple heuristic: ~4 characters per token for most text.
|
|
238
|
+
"""
|
|
239
|
+
return len(content) // 4
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def find_similar_file(file_path: str) -> str | None:
|
|
243
|
+
"""
|
|
244
|
+
Find a similar file if the given file doesn't exist.
|
|
245
|
+
|
|
246
|
+
Args:
|
|
247
|
+
file_path: Path to check
|
|
248
|
+
|
|
249
|
+
Returns:
|
|
250
|
+
Path to similar file or None
|
|
251
|
+
"""
|
|
252
|
+
path = Path(file_path)
|
|
253
|
+
|
|
254
|
+
if not path.parent.exists():
|
|
255
|
+
return None
|
|
256
|
+
|
|
257
|
+
# Get filename without extension
|
|
258
|
+
stem = path.stem.lower()
|
|
259
|
+
suffix = path.suffix.lower()
|
|
260
|
+
|
|
261
|
+
# Search for similar files in the same directory
|
|
262
|
+
for f in path.parent.iterdir():
|
|
263
|
+
if f.is_file():
|
|
264
|
+
f_stem = f.stem.lower()
|
|
265
|
+
f_suffix = f.suffix.lower()
|
|
266
|
+
|
|
267
|
+
# Check if filename is similar (typo or case difference)
|
|
268
|
+
if f_stem == stem or f_suffix == suffix:
|
|
269
|
+
return str(f)
|
|
270
|
+
|
|
271
|
+
# Check for common typo patterns
|
|
272
|
+
if stem in f_stem or f_stem in stem:
|
|
273
|
+
return str(f)
|
|
274
|
+
|
|
275
|
+
return None
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
# =============================================================================
|
|
279
|
+
# Read Functions
|
|
280
|
+
# =============================================================================
|
|
281
|
+
|
|
282
|
+
def read_text_file(
|
|
283
|
+
file_path: str,
|
|
284
|
+
offset: int = 1,
|
|
285
|
+
limit: int | None = None,
|
|
286
|
+
max_size_bytes: int = DEFAULT_MAX_SIZE_BYTES,
|
|
287
|
+
add_line_nums: bool = True
|
|
288
|
+
) -> tuple[str, dict]:
|
|
289
|
+
"""
|
|
290
|
+
Read a text file with optional offset and limit.
|
|
291
|
+
|
|
292
|
+
Args:
|
|
293
|
+
file_path: Path to the file
|
|
294
|
+
offset: Starting line number (1-indexed)
|
|
295
|
+
limit: Maximum number of lines to read
|
|
296
|
+
max_size_bytes: Maximum file size in bytes
|
|
297
|
+
add_line_nums: Whether to add line numbers
|
|
298
|
+
|
|
299
|
+
Returns:
|
|
300
|
+
Tuple of (content, metadata)
|
|
301
|
+
|
|
302
|
+
Raises:
|
|
303
|
+
FileNotFoundError: If file doesn't exist
|
|
304
|
+
FileTooLargeError: If file exceeds size limit
|
|
305
|
+
BinaryFileError: If file appears to be binary
|
|
306
|
+
"""
|
|
307
|
+
path = Path(file_path)
|
|
308
|
+
|
|
309
|
+
if not path.exists():
|
|
310
|
+
similar = find_similar_file(file_path)
|
|
311
|
+
if similar:
|
|
312
|
+
raise FileNotFoundError(
|
|
313
|
+
f"File does not exist: {file_path}. Did you mean {similar}?"
|
|
314
|
+
)
|
|
315
|
+
raise FileNotFoundError(f"File does not exist: {file_path}")
|
|
316
|
+
|
|
317
|
+
if not path.is_file():
|
|
318
|
+
raise ValueError(f"Path is not a file: {file_path}")
|
|
319
|
+
|
|
320
|
+
# Check file size
|
|
321
|
+
file_size = path.stat().st_size
|
|
322
|
+
if file_size > max_size_bytes:
|
|
323
|
+
raise ValueError(
|
|
324
|
+
f"File size ({file_size} bytes) exceeds maximum allowed size "
|
|
325
|
+
f"({max_size_bytes} bytes). Use offset and limit to read specific portions."
|
|
326
|
+
)
|
|
327
|
+
|
|
328
|
+
# Try to read as text
|
|
329
|
+
try:
|
|
330
|
+
with open(path, encoding='utf-8') as f:
|
|
331
|
+
all_lines = f.readlines()
|
|
332
|
+
except UnicodeDecodeError:
|
|
333
|
+
# Try other common encodings
|
|
334
|
+
encodings = ['latin-1', 'cp1252', 'iso-8859-1']
|
|
335
|
+
for enc in encodings:
|
|
336
|
+
try:
|
|
337
|
+
with open(path, encoding=enc) as f:
|
|
338
|
+
all_lines = f.readlines()
|
|
339
|
+
break
|
|
340
|
+
except UnicodeDecodeError:
|
|
341
|
+
continue
|
|
342
|
+
else:
|
|
343
|
+
raise ValueError(
|
|
344
|
+
f"Cannot read file as text. The file appears to be binary: {file_path}"
|
|
345
|
+
)
|
|
346
|
+
|
|
347
|
+
total_lines = len(all_lines)
|
|
348
|
+
|
|
349
|
+
# Convert 1-indexed offset to 0-indexed
|
|
350
|
+
start_idx = max(0, offset - 1)
|
|
351
|
+
|
|
352
|
+
# Apply limit
|
|
353
|
+
if limit is not None:
|
|
354
|
+
end_idx = min(start_idx + limit, total_lines)
|
|
355
|
+
else:
|
|
356
|
+
end_idx = total_lines
|
|
357
|
+
|
|
358
|
+
# Extract requested lines
|
|
359
|
+
selected_lines = all_lines[start_idx:end_idx]
|
|
360
|
+
content = ''.join(selected_lines)
|
|
361
|
+
|
|
362
|
+
# Add line numbers if requested
|
|
363
|
+
if add_line_nums and content:
|
|
364
|
+
content = add_line_numbers(content.rstrip('\n'), offset)
|
|
365
|
+
|
|
366
|
+
metadata = {
|
|
367
|
+
"file_path": file_path,
|
|
368
|
+
"num_lines": len(selected_lines),
|
|
369
|
+
"start_line": offset,
|
|
370
|
+
"total_lines": total_lines,
|
|
371
|
+
"file_size": file_size,
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
return content, metadata
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
def read_image_file(
|
|
378
|
+
file_path: str,
|
|
379
|
+
max_size_bytes: int | None = None
|
|
380
|
+
) -> tuple[str, dict]:
|
|
381
|
+
"""
|
|
382
|
+
Read an image file and encode as base64.
|
|
383
|
+
|
|
384
|
+
Args:
|
|
385
|
+
file_path: Path to the image file
|
|
386
|
+
max_size_bytes: Maximum file size (optional)
|
|
387
|
+
|
|
388
|
+
Returns:
|
|
389
|
+
Tuple of (content, metadata)
|
|
390
|
+
"""
|
|
391
|
+
path = Path(file_path)
|
|
392
|
+
|
|
393
|
+
if not path.exists():
|
|
394
|
+
raise FileNotFoundError(f"Image file does not exist: {file_path}")
|
|
395
|
+
|
|
396
|
+
if not path.is_file():
|
|
397
|
+
raise ValueError(f"Path is not a file: {file_path}")
|
|
398
|
+
|
|
399
|
+
file_size = path.stat().st_size
|
|
400
|
+
|
|
401
|
+
if max_size_bytes and file_size > max_size_bytes:
|
|
402
|
+
raise ValueError(
|
|
403
|
+
f"Image file size ({file_size} bytes) exceeds maximum allowed size "
|
|
404
|
+
f"({max_size_bytes} bytes)."
|
|
405
|
+
)
|
|
406
|
+
|
|
407
|
+
# Read and encode image
|
|
408
|
+
with open(path, 'rb') as f:
|
|
409
|
+
data = f.read()
|
|
410
|
+
|
|
411
|
+
if len(data) == 0:
|
|
412
|
+
raise ValueError(f"Image file is empty: {file_path}")
|
|
413
|
+
|
|
414
|
+
# Detect format
|
|
415
|
+
format_name, mime_type = detect_image_format(data)
|
|
416
|
+
|
|
417
|
+
# Try to get dimensions using PIL if available
|
|
418
|
+
width, height = None, None
|
|
419
|
+
try:
|
|
420
|
+
import io
|
|
421
|
+
|
|
422
|
+
from PIL import Image
|
|
423
|
+
with Image.open(io.BytesIO(data)) as img:
|
|
424
|
+
width, height = img.size
|
|
425
|
+
except ImportError:
|
|
426
|
+
pass # PIL not available, skip dimensions
|
|
427
|
+
except Exception:
|
|
428
|
+
pass # Could not read image dimensions
|
|
429
|
+
|
|
430
|
+
base64_data = base64.b64encode(data).decode('ascii')
|
|
431
|
+
|
|
432
|
+
# For images, return a description string as content
|
|
433
|
+
dims = f" ({width}x{height})" if width and height else ""
|
|
434
|
+
content = f"Image file read: {file_path}{dims} ({file_size} bytes, {mime_type})"
|
|
435
|
+
|
|
436
|
+
metadata = {
|
|
437
|
+
"file_path": file_path,
|
|
438
|
+
"base64": base64_data,
|
|
439
|
+
"media_type": mime_type,
|
|
440
|
+
"original_size": file_size,
|
|
441
|
+
"width": width,
|
|
442
|
+
"height": height,
|
|
443
|
+
}
|
|
444
|
+
|
|
445
|
+
return content, metadata
|
|
446
|
+
|
|
447
|
+
|
|
448
|
+
def read_notebook_file(file_path: str) -> tuple[str, dict]:
|
|
449
|
+
"""
|
|
450
|
+
Read a Jupyter notebook file.
|
|
451
|
+
|
|
452
|
+
Args:
|
|
453
|
+
file_path: Path to the notebook file
|
|
454
|
+
|
|
455
|
+
Returns:
|
|
456
|
+
Tuple of (content, metadata)
|
|
457
|
+
"""
|
|
458
|
+
path = Path(file_path)
|
|
459
|
+
|
|
460
|
+
if not path.exists():
|
|
461
|
+
raise FileNotFoundError(f"Notebook file does not exist: {file_path}")
|
|
462
|
+
|
|
463
|
+
try:
|
|
464
|
+
with open(path, encoding='utf-8') as f:
|
|
465
|
+
notebook = json.load(f)
|
|
466
|
+
except json.JSONDecodeError as e:
|
|
467
|
+
raise ValueError(f"Invalid notebook format: {e}") from e
|
|
468
|
+
|
|
469
|
+
# Extract cells
|
|
470
|
+
cells = []
|
|
471
|
+
for cell_data in notebook.get('cells', []):
|
|
472
|
+
cell_type = cell_data.get('cell_type', 'code')
|
|
473
|
+
|
|
474
|
+
# Handle source which can be string or list
|
|
475
|
+
source = cell_data.get('source', '')
|
|
476
|
+
if isinstance(source, list):
|
|
477
|
+
source = ''.join(source)
|
|
478
|
+
|
|
479
|
+
cells.append({
|
|
480
|
+
"cell_type": cell_type,
|
|
481
|
+
"source": source,
|
|
482
|
+
"outputs": cell_data.get('outputs'),
|
|
483
|
+
"execution_count": cell_data.get('execution_count'),
|
|
484
|
+
})
|
|
485
|
+
|
|
486
|
+
# Get notebook language
|
|
487
|
+
language = None
|
|
488
|
+
if 'metadata' in notebook:
|
|
489
|
+
kernelspec = notebook['metadata'].get('kernelspec', {})
|
|
490
|
+
language = kernelspec.get('language')
|
|
491
|
+
|
|
492
|
+
# Format content as readable text
|
|
493
|
+
content_lines = []
|
|
494
|
+
for i, cell in enumerate(cells, 1):
|
|
495
|
+
cell_type = cell['cell_type']
|
|
496
|
+
source = cell['source']
|
|
497
|
+
content_lines.append(f"### Cell {i} ({cell_type})")
|
|
498
|
+
if source:
|
|
499
|
+
content_lines.append(source)
|
|
500
|
+
content_lines.append("")
|
|
501
|
+
|
|
502
|
+
content = '\n'.join(content_lines)
|
|
503
|
+
|
|
504
|
+
metadata = {
|
|
505
|
+
"file_path": file_path,
|
|
506
|
+
"cell_count": len(cells),
|
|
507
|
+
"language": language,
|
|
508
|
+
}
|
|
509
|
+
|
|
510
|
+
return content, metadata
|
|
511
|
+
|
|
512
|
+
|
|
513
|
+
def read_pdf_file(
|
|
514
|
+
file_path: str,
|
|
515
|
+
pages: str | None = None,
|
|
516
|
+
max_size_bytes: int = DEFAULT_MAX_SIZE_BYTES
|
|
517
|
+
) -> tuple[str, dict]:
|
|
518
|
+
"""
|
|
519
|
+
Read a PDF file.
|
|
520
|
+
|
|
521
|
+
Args:
|
|
522
|
+
file_path: Path to the PDF file
|
|
523
|
+
pages: Optional page range (e.g., "1-5")
|
|
524
|
+
max_size_bytes: Maximum file size in bytes
|
|
525
|
+
|
|
526
|
+
Returns:
|
|
527
|
+
Tuple of (content, metadata)
|
|
528
|
+
"""
|
|
529
|
+
path = Path(file_path)
|
|
530
|
+
|
|
531
|
+
if not path.exists():
|
|
532
|
+
raise FileNotFoundError(f"PDF file does not exist: {file_path}")
|
|
533
|
+
|
|
534
|
+
file_size = path.stat().st_size
|
|
535
|
+
|
|
536
|
+
if file_size > max_size_bytes:
|
|
537
|
+
raise ValueError(
|
|
538
|
+
f"PDF file size ({file_size} bytes) exceeds maximum allowed size "
|
|
539
|
+
f"({max_size_bytes} bytes)."
|
|
540
|
+
)
|
|
541
|
+
|
|
542
|
+
# Read and encode PDF
|
|
543
|
+
with open(path, 'rb') as f:
|
|
544
|
+
data = f.read()
|
|
545
|
+
|
|
546
|
+
base64_data = base64.b64encode(data).decode('ascii')
|
|
547
|
+
|
|
548
|
+
# Try to get page count
|
|
549
|
+
page_count = None
|
|
550
|
+
try:
|
|
551
|
+
import pypdf
|
|
552
|
+
reader = pypdf.PdfReader(path)
|
|
553
|
+
page_count = len(reader.pages)
|
|
554
|
+
except ImportError:
|
|
555
|
+
pass # pypdf not available
|
|
556
|
+
except Exception:
|
|
557
|
+
pass
|
|
558
|
+
|
|
559
|
+
pages_info = f", {page_count} pages" if page_count else ""
|
|
560
|
+
content = f"PDF file read: {file_path} ({file_size} bytes{pages_info})"
|
|
561
|
+
|
|
562
|
+
metadata = {
|
|
563
|
+
"file_path": file_path,
|
|
564
|
+
"base64": base64_data,
|
|
565
|
+
"original_size": file_size,
|
|
566
|
+
"page_count": page_count,
|
|
567
|
+
}
|
|
568
|
+
|
|
569
|
+
return content, metadata
|
|
570
|
+
|
|
571
|
+
|
|
572
|
+
# =============================================================================
|
|
573
|
+
# Main Tool Class
|
|
574
|
+
# =============================================================================
|
|
575
|
+
|
|
576
|
+
class FileReadTool(Tool):
|
|
577
|
+
"""
|
|
578
|
+
A comprehensive file reading tool.
|
|
579
|
+
|
|
580
|
+
This tool provides file reading functionality for various file types:
|
|
581
|
+
- Text files (with line numbers, offset, and limit support)
|
|
582
|
+
- Image files (PNG, JPG, JPEG, GIF, WEBP) with base64 encoding
|
|
583
|
+
- PDF files (with optional page range)
|
|
584
|
+
- Jupyter notebooks (.ipynb)
|
|
585
|
+
"""
|
|
586
|
+
|
|
587
|
+
name = "read"
|
|
588
|
+
description = """Reads a file from the local filesystem.
|
|
589
|
+
|
|
590
|
+
Usage:
|
|
591
|
+
- The file_path must be an absolute path (not a relative path)
|
|
592
|
+
- By default, it reads up to 2000 lines starting from line 1
|
|
593
|
+
- Use offset and limit to read specific portions of a large file
|
|
594
|
+
- For images (PNG, JPG, GIF, WEBP), the image content is encoded as base64
|
|
595
|
+
- For PDFs, use the pages parameter to read specific page ranges (e.g., pages: "1-5")
|
|
596
|
+
- For Jupyter notebooks (.ipynb), returns the notebook cells with their outputs
|
|
597
|
+
|
|
598
|
+
This tool is read-only and does not modify the filesystem.
|
|
599
|
+
"""
|
|
600
|
+
parameters = {
|
|
601
|
+
"type": "object",
|
|
602
|
+
"properties": {
|
|
603
|
+
"file_path": {
|
|
604
|
+
"type": "string",
|
|
605
|
+
"description": "The absolute path to the file to read"
|
|
606
|
+
},
|
|
607
|
+
"offset": {
|
|
608
|
+
"type": "integer",
|
|
609
|
+
"description": "The line number to start reading from (1-indexed)",
|
|
610
|
+
"default": 1
|
|
611
|
+
},
|
|
612
|
+
"limit": {
|
|
613
|
+
"type": "integer",
|
|
614
|
+
"description": "The number of lines to read"
|
|
615
|
+
},
|
|
616
|
+
"pages": {
|
|
617
|
+
"type": "string",
|
|
618
|
+
"description": "Page range for PDF files (e.g., '1-5', '3', '10-20')"
|
|
619
|
+
}
|
|
620
|
+
},
|
|
621
|
+
"required": ["file_path"]
|
|
622
|
+
}
|
|
623
|
+
tags = ["scanner", "reader"]
|
|
624
|
+
|
|
625
|
+
def __init__(
|
|
626
|
+
self,
|
|
627
|
+
max_size_bytes: int = DEFAULT_MAX_SIZE_BYTES,
|
|
628
|
+
max_tokens: int = DEFAULT_MAX_TOKENS
|
|
629
|
+
):
|
|
630
|
+
"""
|
|
631
|
+
Initialize the FileReadTool.
|
|
632
|
+
|
|
633
|
+
Args:
|
|
634
|
+
max_size_bytes: Maximum file size in bytes
|
|
635
|
+
max_tokens: Maximum estimated tokens for text content
|
|
636
|
+
"""
|
|
637
|
+
from sanityops_agent.tools.context import ToolContext
|
|
638
|
+
super().__init__(context=ToolContext())
|
|
639
|
+
self.max_size_bytes = max_size_bytes
|
|
640
|
+
self.max_tokens = max_tokens
|
|
641
|
+
|
|
642
|
+
def _validate_input(self, file_path: str) -> None:
|
|
643
|
+
"""
|
|
644
|
+
Validate input before reading.
|
|
645
|
+
|
|
646
|
+
Args:
|
|
647
|
+
file_path: Path to validate
|
|
648
|
+
|
|
649
|
+
Raises:
|
|
650
|
+
ValueError: If path is a blocked device or binary file
|
|
651
|
+
"""
|
|
652
|
+
# Check for blocked device paths
|
|
653
|
+
if is_blocked_device(file_path):
|
|
654
|
+
raise ValueError(
|
|
655
|
+
f"Cannot read '{file_path}': this device file would block or produce infinite output."
|
|
656
|
+
)
|
|
657
|
+
|
|
658
|
+
# Check for binary files
|
|
659
|
+
ext = get_file_extension(file_path)
|
|
660
|
+
if is_binary_file(file_path) and not is_image_file(file_path) and ext != 'pdf':
|
|
661
|
+
raise ValueError(
|
|
662
|
+
f"This tool cannot read binary files. "
|
|
663
|
+
f"The file appears to be a binary .{ext} file."
|
|
664
|
+
)
|
|
665
|
+
|
|
666
|
+
async def execute(
|
|
667
|
+
self,
|
|
668
|
+
file_path: str,
|
|
669
|
+
offset: int | None = 1,
|
|
670
|
+
limit: int | None = None,
|
|
671
|
+
pages: str | None = None,
|
|
672
|
+
**kwargs
|
|
673
|
+
) -> ToolResult:
|
|
674
|
+
"""
|
|
675
|
+
Read a file.
|
|
676
|
+
|
|
677
|
+
Args:
|
|
678
|
+
file_path: Absolute path to the file to read
|
|
679
|
+
offset: The line number to start reading from (1-indexed)
|
|
680
|
+
limit: The number of lines to read
|
|
681
|
+
pages: Page range for PDF files (e.g., "1-5", "3", "10-20")
|
|
682
|
+
|
|
683
|
+
Returns:
|
|
684
|
+
ToolResult containing the file content
|
|
685
|
+
"""
|
|
686
|
+
try:
|
|
687
|
+
# Expand and normalize path
|
|
688
|
+
full_path = expand_path(file_path)
|
|
689
|
+
|
|
690
|
+
# Validate input
|
|
691
|
+
self._validate_input(full_path)
|
|
692
|
+
|
|
693
|
+
# Determine file type and read accordingly
|
|
694
|
+
if is_notebook_file(full_path):
|
|
695
|
+
content, metadata = read_notebook_file(full_path)
|
|
696
|
+
return ToolResult(
|
|
697
|
+
content=content,
|
|
698
|
+
success=True,
|
|
699
|
+
metadata={"file_type": "notebook", **metadata}
|
|
700
|
+
)
|
|
701
|
+
|
|
702
|
+
if is_image_file(full_path):
|
|
703
|
+
content, metadata = read_image_file(full_path, self.max_size_bytes)
|
|
704
|
+
return ToolResult(
|
|
705
|
+
content=content,
|
|
706
|
+
success=True,
|
|
707
|
+
metadata={"file_type": "image", **metadata}
|
|
708
|
+
)
|
|
709
|
+
|
|
710
|
+
if is_pdf_file(full_path):
|
|
711
|
+
content, metadata = read_pdf_file(full_path, pages, self.max_size_bytes)
|
|
712
|
+
return ToolResult(
|
|
713
|
+
content=content,
|
|
714
|
+
success=True,
|
|
715
|
+
metadata={"file_type": "pdf", **metadata}
|
|
716
|
+
)
|
|
717
|
+
|
|
718
|
+
# Default: read as text file
|
|
719
|
+
content, metadata = read_text_file(full_path, offset or 1, limit, self.max_size_bytes)
|
|
720
|
+
|
|
721
|
+
# Check token limit for text content
|
|
722
|
+
if content:
|
|
723
|
+
token_estimate = rough_token_estimate(content)
|
|
724
|
+
if token_estimate > self.max_tokens:
|
|
725
|
+
return ToolResult(
|
|
726
|
+
content=f"File content ({token_estimate} estimated tokens) exceeds maximum allowed tokens ({self.max_tokens}). Use offset and limit parameters to read specific portions of the file.",
|
|
727
|
+
success=False,
|
|
728
|
+
error="Token limit exceeded"
|
|
729
|
+
)
|
|
730
|
+
|
|
731
|
+
# Handle empty file
|
|
732
|
+
if not content:
|
|
733
|
+
if metadata.get("total_lines", 0) == 0:
|
|
734
|
+
content = "<system-reminder>Warning: the file exists but the contents are empty.</system-reminder>"
|
|
735
|
+
else:
|
|
736
|
+
content = f"<system-reminder>Warning: the file exists but is shorter than the provided offset ({metadata.get('start_line')}). The file has {metadata.get('total_lines')} lines.</system-reminder>"
|
|
737
|
+
|
|
738
|
+
return ToolResult(
|
|
739
|
+
content=content,
|
|
740
|
+
success=True,
|
|
741
|
+
metadata={"file_type": "text", **metadata}
|
|
742
|
+
)
|
|
743
|
+
|
|
744
|
+
except FileNotFoundError as e:
|
|
745
|
+
return ToolResult(
|
|
746
|
+
content=f"Error: {e}",
|
|
747
|
+
success=False,
|
|
748
|
+
error=str(e)
|
|
749
|
+
)
|
|
750
|
+
except ValueError as e:
|
|
751
|
+
return ToolResult(
|
|
752
|
+
content=f"Error: {e}",
|
|
753
|
+
success=False,
|
|
754
|
+
error=str(e)
|
|
755
|
+
)
|
|
756
|
+
except Exception as e:
|
|
757
|
+
return ToolResult(
|
|
758
|
+
content=f"Error reading file: {e}",
|
|
759
|
+
success=False,
|
|
760
|
+
error=str(e)
|
|
761
|
+
)
|
|
762
|
+
|
|
763
|
+
|
|
764
|
+
# =============================================================================
|
|
765
|
+
# Convenience Functions
|
|
766
|
+
# =============================================================================
|
|
767
|
+
|
|
768
|
+
def read_file(
|
|
769
|
+
file_path: str,
|
|
770
|
+
offset: int | None = 1,
|
|
771
|
+
limit: int | None = None,
|
|
772
|
+
pages: str | None = None
|
|
773
|
+
) -> tuple[str, dict]:
|
|
774
|
+
"""
|
|
775
|
+
Quick file read function.
|
|
776
|
+
|
|
777
|
+
Args:
|
|
778
|
+
file_path: Absolute path to the file to read
|
|
779
|
+
offset: The line number to start reading from (1-indexed)
|
|
780
|
+
limit: The number of lines to read
|
|
781
|
+
pages: Page range for PDF files
|
|
782
|
+
|
|
783
|
+
Returns:
|
|
784
|
+
Tuple of (content, metadata)
|
|
785
|
+
"""
|
|
786
|
+
tool = FileReadTool()
|
|
787
|
+
result = tool.read(file_path=file_path, offset=offset, limit=limit, pages=pages)
|
|
788
|
+
return result.content, result.metadata
|
|
789
|
+
|
|
790
|
+
|
|
791
|
+
# =============================================================================
|
|
792
|
+
# Exports
|
|
793
|
+
# =============================================================================
|
|
794
|
+
|
|
795
|
+
__all__ = [
|
|
796
|
+
'FileReadTool',
|
|
797
|
+
'read_file',
|
|
798
|
+
'expand_path',
|
|
799
|
+
'add_line_numbers',
|
|
800
|
+
'IMAGE_EXTENSIONS',
|
|
801
|
+
'BINARY_EXTENSIONS',
|
|
802
|
+
'DEFAULT_MAX_SIZE_BYTES',
|
|
803
|
+
'DEFAULT_MAX_TOKENS',
|
|
804
|
+
]
|