sanityops-cli 0.1.3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. sanityops_cli/__init__.py +16 -0
  2. sanityops_cli/agents/__init__.py +14 -0
  3. sanityops_cli/agents/repair_agent/__init__.py +19 -0
  4. sanityops_cli/agents/repair_agent/agent.py +230 -0
  5. sanityops_cli/agents/repair_agent/prompts.py +100 -0
  6. sanityops_cli/agents/repair_agent/tools/__init__.py +19 -0
  7. sanityops_cli/agents/repair_agent/tools/store_repairs_tool.py +144 -0
  8. sanityops_cli/agents/scanner_agent/agent.py +332 -0
  9. sanityops_cli/agents/scanner_agent/hooks/progress_hook.py +152 -0
  10. sanityops_cli/agents/scanner_agent/models/finding.py +120 -0
  11. sanityops_cli/agents/scanner_agent/prompts.py +316 -0
  12. sanityops_cli/agents/scanner_agent/tools/grep_tool.py +667 -0
  13. sanityops_cli/agents/scanner_agent/tools/listfiles_tool.py +88 -0
  14. sanityops_cli/agents/scanner_agent/tools/readfile_tool.py +804 -0
  15. sanityops_cli/agents/scanner_agent/tools/storefindings_tool.py +178 -0
  16. sanityops_cli/api/__init__.py +16 -0
  17. sanityops_cli/api/client.py +403 -0
  18. sanityops_cli/commands/__init__.py +14 -0
  19. sanityops_cli/commands/config.py +312 -0
  20. sanityops_cli/commands/init.py +132 -0
  21. sanityops_cli/commands/inspect.py +646 -0
  22. sanityops_cli/constants/__init__.py +14 -0
  23. sanityops_cli/constants/config_defaults.py +22 -0
  24. sanityops_cli/constants/exit_codes.py +19 -0
  25. sanityops_cli/defect_checker/__init__.py +16 -0
  26. sanityops_cli/defect_checker/checker.py +100 -0
  27. sanityops_cli/defect_checker/llm_config.py +112 -0
  28. sanityops_cli/defect_checker/markdown_reporter.py +249 -0
  29. sanityops_cli/defect_checker/renderer.py +203 -0
  30. sanityops_cli/exceptions/__init__.py +14 -0
  31. sanityops_cli/exceptions/api_exceptions.py +60 -0
  32. sanityops_cli/exceptions/base_exceptions.py +25 -0
  33. sanityops_cli/help_panel.py +49 -0
  34. sanityops_cli/logging/__init__.py +18 -0
  35. sanityops_cli/logging/logger.py +108 -0
  36. sanityops_cli/main.py +123 -0
  37. sanityops_cli/progress/__init__.py +18 -0
  38. sanityops_cli/progress/tracker.py +159 -0
  39. sanityops_cli/renderers/__init__.py +14 -0
  40. sanityops_cli/renderers/command_renderer/inspect_command_renderer.py +87 -0
  41. sanityops_cli/templates/__init__.py +14 -0
  42. sanityops_cli/templates/inspect_config.yaml +55 -0
  43. sanityops_cli/utils/__init__.py +14 -0
  44. sanityops_cli/utils/artifact_packer.py +407 -0
  45. sanityops_cli/utils/config_loader.py +296 -0
  46. sanityops_cli/utils/config_resolver.py +358 -0
  47. sanityops_cli/utils/validators.py +117 -0
  48. sanityops_cli-0.1.3.dist-info/METADATA +213 -0
  49. sanityops_cli-0.1.3.dist-info/RECORD +53 -0
  50. sanityops_cli-0.1.3.dist-info/WHEEL +4 -0
  51. sanityops_cli-0.1.3.dist-info/entry_points.txt +2 -0
  52. sanityops_cli-0.1.3.dist-info/licenses/LICENSE +201 -0
  53. sanityops_cli-0.1.3.dist-info/licenses/NOTICE +5 -0
@@ -0,0 +1,804 @@
1
+ # Copyright 2026 zipsonken
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+ #
15
+
16
+ """
17
+ FileReadTool - A comprehensive file reading tool.
18
+
19
+ Adapted for the deeplogic-cli agent framework.
20
+
21
+ This module provides functionality to read various file types including:
22
+ - Text files (with line numbers and offset/limit support)
23
+ - Image files (PNG, JPG, JPEG, GIF, WEBP) with base64 encoding
24
+ - PDF files (with optional page range extraction)
25
+ - Jupyter notebooks (.ipynb)
26
+ """
27
+
28
+ import base64
29
+ import json
30
+ import os
31
+ from pathlib import Path
32
+
33
+ from sanityops_agent.tools.base import Tool, ToolResult
34
+
35
+ # =============================================================================
36
+ # Constants
37
+ # =============================================================================
38
+
39
+ # Supported image extensions
40
+ IMAGE_EXTENSIONS = {'png', 'jpg', 'jpeg', 'gif', 'webp'}
41
+
42
+ # Binary file extensions that cannot be read as text
43
+ BINARY_EXTENSIONS = {
44
+ 'exe', 'dll', 'so', 'dylib', 'bin', 'dat',
45
+ 'pyc', 'pyo', 'pyd', 'class', 'jar', 'war',
46
+ 'zip', 'tar', 'gz', 'bz2', 'xz', '7z', 'rar',
47
+ 'mp3', 'mp4', 'avi', 'mov', 'mkv', 'flv', 'wmv',
48
+ 'wav', 'flac', 'aac', 'ogg',
49
+ 'doc', 'docx', 'xls', 'xlsx', 'ppt', 'pptx',
50
+ 'sqlite', 'db', 'mdb',
51
+ }
52
+
53
+ # Device files that would hang the process
54
+ BLOCKED_DEVICE_PATHS = {
55
+ '/dev/zero',
56
+ '/dev/random',
57
+ '/dev/urandom',
58
+ '/dev/full',
59
+ '/dev/stdin',
60
+ '/dev/tty',
61
+ '/dev/console',
62
+ '/dev/stdout',
63
+ '/dev/stderr',
64
+ '/dev/fd/0',
65
+ '/dev/fd/1',
66
+ '/dev/fd/2',
67
+ }
68
+
69
+ # Default limits
70
+ DEFAULT_MAX_SIZE_BYTES = 2 * 1024 * 1024 # 2MB
71
+ DEFAULT_MAX_TOKENS = 20000
72
+
73
+
74
+ # =============================================================================
75
+ # Helper Functions
76
+ # =============================================================================
77
+
78
+ def get_cwd() -> str:
79
+ """Get current working directory."""
80
+ return os.getcwd()
81
+
82
+
83
+ def expand_path(path: str, base_dir: str | None = None) -> str:
84
+ """
85
+ Expand a path that may contain tilde notation (~) to an absolute path.
86
+
87
+ Args:
88
+ path: The path to expand
89
+ base_dir: Base directory for relative paths (defaults to cwd)
90
+
91
+ Returns:
92
+ The expanded absolute path
93
+ """
94
+ actual_base_dir = base_dir or get_cwd()
95
+
96
+ # Handle empty path
97
+ if not path or not path.strip():
98
+ return os.path.normpath(actual_base_dir)
99
+
100
+ path = path.strip()
101
+
102
+ # Handle home directory notation
103
+ if path == '~':
104
+ return os.path.expanduser('~')
105
+ if path.startswith('~/'):
106
+ return os.path.normpath(os.path.join(os.path.expanduser('~'), path[2:]))
107
+
108
+ # Handle absolute paths
109
+ if os.path.isabs(path):
110
+ return os.path.normpath(path)
111
+
112
+ # Handle relative paths
113
+ return os.path.normpath(os.path.abspath(os.path.join(actual_base_dir, path)))
114
+
115
+
116
+ def get_file_extension(file_path: str) -> str:
117
+ """Get the file extension without the leading dot."""
118
+ return Path(file_path).suffix.lstrip('.').lower()
119
+
120
+
121
+ def is_image_file(file_path: str) -> bool:
122
+ """Check if file is an image based on extension."""
123
+ ext = get_file_extension(file_path)
124
+ return ext in IMAGE_EXTENSIONS
125
+
126
+
127
+ def is_binary_file(file_path: str) -> bool:
128
+ """Check if file is a binary file based on extension."""
129
+ ext = get_file_extension(file_path)
130
+ return ext in BINARY_EXTENSIONS
131
+
132
+
133
+ def is_blocked_device(file_path: str) -> bool:
134
+ """Check if path is a blocked device file."""
135
+ return file_path in BLOCKED_DEVICE_PATHS
136
+
137
+
138
+ def is_notebook_file(file_path: str) -> bool:
139
+ """Check if file is a Jupyter notebook."""
140
+ return get_file_extension(file_path) == 'ipynb'
141
+
142
+
143
+ def is_pdf_file(file_path: str) -> bool:
144
+ """Check if file is a PDF."""
145
+ return get_file_extension(file_path) == 'pdf'
146
+
147
+
148
+ def add_line_numbers(content: str, start_line: int = 1) -> str:
149
+ """
150
+ Add line numbers to content.
151
+
152
+ Args:
153
+ content: The text content
154
+ start_line: Starting line number (1-indexed)
155
+
156
+ Returns:
157
+ Content with line numbers prefixed
158
+ """
159
+ lines = content.split('\n')
160
+ max_line_num = start_line + len(lines) - 1
161
+ width = len(str(max_line_num))
162
+
163
+ numbered_lines = []
164
+ for i, line in enumerate(lines):
165
+ line_num = start_line + i
166
+ numbered_lines.append(f"{line_num:{width}d}\t{line}")
167
+
168
+ return '\n'.join(numbered_lines)
169
+
170
+
171
+ def detect_image_format(data: bytes) -> tuple[str, str]:
172
+ """
173
+ Detect image format from binary data.
174
+
175
+ Args:
176
+ data: Binary image data
177
+
178
+ Returns:
179
+ Tuple of (format_name, mime_type)
180
+
181
+ Raises:
182
+ ValueError: If format cannot be detected
183
+ """
184
+ # PNG
185
+ if data[:8] == b'\x89PNG\r\n\x1a\n':
186
+ return 'png', 'image/png'
187
+
188
+ # JPEG
189
+ if data[:2] == b'\xff\xd8':
190
+ return 'jpeg', 'image/jpeg'
191
+
192
+ # GIF
193
+ if data[:6] in (b'GIF87a', b'GIF89a'):
194
+ return 'gif', 'image/gif'
195
+
196
+ # WebP
197
+ if data[:4] == b'RIFF' and data[8:12] == b'WEBP':
198
+ return 'webp', 'image/webp'
199
+
200
+ # Default to PNG
201
+ return 'png', 'image/png'
202
+
203
+
204
+ def parse_pdf_page_range(pages: str) -> tuple[int, int] | None:
205
+ """
206
+ Parse a PDF page range string.
207
+
208
+ Args:
209
+ pages: Page range string (e.g., "1-5", "3", "10-20")
210
+
211
+ Returns:
212
+ Tuple of (start_page, end_page) or None if invalid
213
+ """
214
+ import re
215
+
216
+ # Single page
217
+ if re.match(r'^\d+$', pages):
218
+ page = int(pages)
219
+ if page >= 1:
220
+ return (page, page)
221
+
222
+ # Page range
223
+ match = re.match(r'^(\d+)-(\d+)$', pages)
224
+ if match:
225
+ start = int(match.group(1))
226
+ end = int(match.group(2))
227
+ if start >= 1 and end >= start:
228
+ return (start, end)
229
+
230
+ return None
231
+
232
+
233
+ def rough_token_estimate(content: str) -> int:
234
+ """
235
+ Estimate token count for content.
236
+
237
+ Uses a simple heuristic: ~4 characters per token for most text.
238
+ """
239
+ return len(content) // 4
240
+
241
+
242
+ def find_similar_file(file_path: str) -> str | None:
243
+ """
244
+ Find a similar file if the given file doesn't exist.
245
+
246
+ Args:
247
+ file_path: Path to check
248
+
249
+ Returns:
250
+ Path to similar file or None
251
+ """
252
+ path = Path(file_path)
253
+
254
+ if not path.parent.exists():
255
+ return None
256
+
257
+ # Get filename without extension
258
+ stem = path.stem.lower()
259
+ suffix = path.suffix.lower()
260
+
261
+ # Search for similar files in the same directory
262
+ for f in path.parent.iterdir():
263
+ if f.is_file():
264
+ f_stem = f.stem.lower()
265
+ f_suffix = f.suffix.lower()
266
+
267
+ # Check if filename is similar (typo or case difference)
268
+ if f_stem == stem or f_suffix == suffix:
269
+ return str(f)
270
+
271
+ # Check for common typo patterns
272
+ if stem in f_stem or f_stem in stem:
273
+ return str(f)
274
+
275
+ return None
276
+
277
+
278
+ # =============================================================================
279
+ # Read Functions
280
+ # =============================================================================
281
+
282
+ def read_text_file(
283
+ file_path: str,
284
+ offset: int = 1,
285
+ limit: int | None = None,
286
+ max_size_bytes: int = DEFAULT_MAX_SIZE_BYTES,
287
+ add_line_nums: bool = True
288
+ ) -> tuple[str, dict]:
289
+ """
290
+ Read a text file with optional offset and limit.
291
+
292
+ Args:
293
+ file_path: Path to the file
294
+ offset: Starting line number (1-indexed)
295
+ limit: Maximum number of lines to read
296
+ max_size_bytes: Maximum file size in bytes
297
+ add_line_nums: Whether to add line numbers
298
+
299
+ Returns:
300
+ Tuple of (content, metadata)
301
+
302
+ Raises:
303
+ FileNotFoundError: If file doesn't exist
304
+ FileTooLargeError: If file exceeds size limit
305
+ BinaryFileError: If file appears to be binary
306
+ """
307
+ path = Path(file_path)
308
+
309
+ if not path.exists():
310
+ similar = find_similar_file(file_path)
311
+ if similar:
312
+ raise FileNotFoundError(
313
+ f"File does not exist: {file_path}. Did you mean {similar}?"
314
+ )
315
+ raise FileNotFoundError(f"File does not exist: {file_path}")
316
+
317
+ if not path.is_file():
318
+ raise ValueError(f"Path is not a file: {file_path}")
319
+
320
+ # Check file size
321
+ file_size = path.stat().st_size
322
+ if file_size > max_size_bytes:
323
+ raise ValueError(
324
+ f"File size ({file_size} bytes) exceeds maximum allowed size "
325
+ f"({max_size_bytes} bytes). Use offset and limit to read specific portions."
326
+ )
327
+
328
+ # Try to read as text
329
+ try:
330
+ with open(path, encoding='utf-8') as f:
331
+ all_lines = f.readlines()
332
+ except UnicodeDecodeError:
333
+ # Try other common encodings
334
+ encodings = ['latin-1', 'cp1252', 'iso-8859-1']
335
+ for enc in encodings:
336
+ try:
337
+ with open(path, encoding=enc) as f:
338
+ all_lines = f.readlines()
339
+ break
340
+ except UnicodeDecodeError:
341
+ continue
342
+ else:
343
+ raise ValueError(
344
+ f"Cannot read file as text. The file appears to be binary: {file_path}"
345
+ )
346
+
347
+ total_lines = len(all_lines)
348
+
349
+ # Convert 1-indexed offset to 0-indexed
350
+ start_idx = max(0, offset - 1)
351
+
352
+ # Apply limit
353
+ if limit is not None:
354
+ end_idx = min(start_idx + limit, total_lines)
355
+ else:
356
+ end_idx = total_lines
357
+
358
+ # Extract requested lines
359
+ selected_lines = all_lines[start_idx:end_idx]
360
+ content = ''.join(selected_lines)
361
+
362
+ # Add line numbers if requested
363
+ if add_line_nums and content:
364
+ content = add_line_numbers(content.rstrip('\n'), offset)
365
+
366
+ metadata = {
367
+ "file_path": file_path,
368
+ "num_lines": len(selected_lines),
369
+ "start_line": offset,
370
+ "total_lines": total_lines,
371
+ "file_size": file_size,
372
+ }
373
+
374
+ return content, metadata
375
+
376
+
377
+ def read_image_file(
378
+ file_path: str,
379
+ max_size_bytes: int | None = None
380
+ ) -> tuple[str, dict]:
381
+ """
382
+ Read an image file and encode as base64.
383
+
384
+ Args:
385
+ file_path: Path to the image file
386
+ max_size_bytes: Maximum file size (optional)
387
+
388
+ Returns:
389
+ Tuple of (content, metadata)
390
+ """
391
+ path = Path(file_path)
392
+
393
+ if not path.exists():
394
+ raise FileNotFoundError(f"Image file does not exist: {file_path}")
395
+
396
+ if not path.is_file():
397
+ raise ValueError(f"Path is not a file: {file_path}")
398
+
399
+ file_size = path.stat().st_size
400
+
401
+ if max_size_bytes and file_size > max_size_bytes:
402
+ raise ValueError(
403
+ f"Image file size ({file_size} bytes) exceeds maximum allowed size "
404
+ f"({max_size_bytes} bytes)."
405
+ )
406
+
407
+ # Read and encode image
408
+ with open(path, 'rb') as f:
409
+ data = f.read()
410
+
411
+ if len(data) == 0:
412
+ raise ValueError(f"Image file is empty: {file_path}")
413
+
414
+ # Detect format
415
+ format_name, mime_type = detect_image_format(data)
416
+
417
+ # Try to get dimensions using PIL if available
418
+ width, height = None, None
419
+ try:
420
+ import io
421
+
422
+ from PIL import Image
423
+ with Image.open(io.BytesIO(data)) as img:
424
+ width, height = img.size
425
+ except ImportError:
426
+ pass # PIL not available, skip dimensions
427
+ except Exception:
428
+ pass # Could not read image dimensions
429
+
430
+ base64_data = base64.b64encode(data).decode('ascii')
431
+
432
+ # For images, return a description string as content
433
+ dims = f" ({width}x{height})" if width and height else ""
434
+ content = f"Image file read: {file_path}{dims} ({file_size} bytes, {mime_type})"
435
+
436
+ metadata = {
437
+ "file_path": file_path,
438
+ "base64": base64_data,
439
+ "media_type": mime_type,
440
+ "original_size": file_size,
441
+ "width": width,
442
+ "height": height,
443
+ }
444
+
445
+ return content, metadata
446
+
447
+
448
+ def read_notebook_file(file_path: str) -> tuple[str, dict]:
449
+ """
450
+ Read a Jupyter notebook file.
451
+
452
+ Args:
453
+ file_path: Path to the notebook file
454
+
455
+ Returns:
456
+ Tuple of (content, metadata)
457
+ """
458
+ path = Path(file_path)
459
+
460
+ if not path.exists():
461
+ raise FileNotFoundError(f"Notebook file does not exist: {file_path}")
462
+
463
+ try:
464
+ with open(path, encoding='utf-8') as f:
465
+ notebook = json.load(f)
466
+ except json.JSONDecodeError as e:
467
+ raise ValueError(f"Invalid notebook format: {e}") from e
468
+
469
+ # Extract cells
470
+ cells = []
471
+ for cell_data in notebook.get('cells', []):
472
+ cell_type = cell_data.get('cell_type', 'code')
473
+
474
+ # Handle source which can be string or list
475
+ source = cell_data.get('source', '')
476
+ if isinstance(source, list):
477
+ source = ''.join(source)
478
+
479
+ cells.append({
480
+ "cell_type": cell_type,
481
+ "source": source,
482
+ "outputs": cell_data.get('outputs'),
483
+ "execution_count": cell_data.get('execution_count'),
484
+ })
485
+
486
+ # Get notebook language
487
+ language = None
488
+ if 'metadata' in notebook:
489
+ kernelspec = notebook['metadata'].get('kernelspec', {})
490
+ language = kernelspec.get('language')
491
+
492
+ # Format content as readable text
493
+ content_lines = []
494
+ for i, cell in enumerate(cells, 1):
495
+ cell_type = cell['cell_type']
496
+ source = cell['source']
497
+ content_lines.append(f"### Cell {i} ({cell_type})")
498
+ if source:
499
+ content_lines.append(source)
500
+ content_lines.append("")
501
+
502
+ content = '\n'.join(content_lines)
503
+
504
+ metadata = {
505
+ "file_path": file_path,
506
+ "cell_count": len(cells),
507
+ "language": language,
508
+ }
509
+
510
+ return content, metadata
511
+
512
+
513
+ def read_pdf_file(
514
+ file_path: str,
515
+ pages: str | None = None,
516
+ max_size_bytes: int = DEFAULT_MAX_SIZE_BYTES
517
+ ) -> tuple[str, dict]:
518
+ """
519
+ Read a PDF file.
520
+
521
+ Args:
522
+ file_path: Path to the PDF file
523
+ pages: Optional page range (e.g., "1-5")
524
+ max_size_bytes: Maximum file size in bytes
525
+
526
+ Returns:
527
+ Tuple of (content, metadata)
528
+ """
529
+ path = Path(file_path)
530
+
531
+ if not path.exists():
532
+ raise FileNotFoundError(f"PDF file does not exist: {file_path}")
533
+
534
+ file_size = path.stat().st_size
535
+
536
+ if file_size > max_size_bytes:
537
+ raise ValueError(
538
+ f"PDF file size ({file_size} bytes) exceeds maximum allowed size "
539
+ f"({max_size_bytes} bytes)."
540
+ )
541
+
542
+ # Read and encode PDF
543
+ with open(path, 'rb') as f:
544
+ data = f.read()
545
+
546
+ base64_data = base64.b64encode(data).decode('ascii')
547
+
548
+ # Try to get page count
549
+ page_count = None
550
+ try:
551
+ import pypdf
552
+ reader = pypdf.PdfReader(path)
553
+ page_count = len(reader.pages)
554
+ except ImportError:
555
+ pass # pypdf not available
556
+ except Exception:
557
+ pass
558
+
559
+ pages_info = f", {page_count} pages" if page_count else ""
560
+ content = f"PDF file read: {file_path} ({file_size} bytes{pages_info})"
561
+
562
+ metadata = {
563
+ "file_path": file_path,
564
+ "base64": base64_data,
565
+ "original_size": file_size,
566
+ "page_count": page_count,
567
+ }
568
+
569
+ return content, metadata
570
+
571
+
572
+ # =============================================================================
573
+ # Main Tool Class
574
+ # =============================================================================
575
+
576
+ class FileReadTool(Tool):
577
+ """
578
+ A comprehensive file reading tool.
579
+
580
+ This tool provides file reading functionality for various file types:
581
+ - Text files (with line numbers, offset, and limit support)
582
+ - Image files (PNG, JPG, JPEG, GIF, WEBP) with base64 encoding
583
+ - PDF files (with optional page range)
584
+ - Jupyter notebooks (.ipynb)
585
+ """
586
+
587
+ name = "read"
588
+ description = """Reads a file from the local filesystem.
589
+
590
+ Usage:
591
+ - The file_path must be an absolute path (not a relative path)
592
+ - By default, it reads up to 2000 lines starting from line 1
593
+ - Use offset and limit to read specific portions of a large file
594
+ - For images (PNG, JPG, GIF, WEBP), the image content is encoded as base64
595
+ - For PDFs, use the pages parameter to read specific page ranges (e.g., pages: "1-5")
596
+ - For Jupyter notebooks (.ipynb), returns the notebook cells with their outputs
597
+
598
+ This tool is read-only and does not modify the filesystem.
599
+ """
600
+ parameters = {
601
+ "type": "object",
602
+ "properties": {
603
+ "file_path": {
604
+ "type": "string",
605
+ "description": "The absolute path to the file to read"
606
+ },
607
+ "offset": {
608
+ "type": "integer",
609
+ "description": "The line number to start reading from (1-indexed)",
610
+ "default": 1
611
+ },
612
+ "limit": {
613
+ "type": "integer",
614
+ "description": "The number of lines to read"
615
+ },
616
+ "pages": {
617
+ "type": "string",
618
+ "description": "Page range for PDF files (e.g., '1-5', '3', '10-20')"
619
+ }
620
+ },
621
+ "required": ["file_path"]
622
+ }
623
+ tags = ["scanner", "reader"]
624
+
625
+ def __init__(
626
+ self,
627
+ max_size_bytes: int = DEFAULT_MAX_SIZE_BYTES,
628
+ max_tokens: int = DEFAULT_MAX_TOKENS
629
+ ):
630
+ """
631
+ Initialize the FileReadTool.
632
+
633
+ Args:
634
+ max_size_bytes: Maximum file size in bytes
635
+ max_tokens: Maximum estimated tokens for text content
636
+ """
637
+ from sanityops_agent.tools.context import ToolContext
638
+ super().__init__(context=ToolContext())
639
+ self.max_size_bytes = max_size_bytes
640
+ self.max_tokens = max_tokens
641
+
642
+ def _validate_input(self, file_path: str) -> None:
643
+ """
644
+ Validate input before reading.
645
+
646
+ Args:
647
+ file_path: Path to validate
648
+
649
+ Raises:
650
+ ValueError: If path is a blocked device or binary file
651
+ """
652
+ # Check for blocked device paths
653
+ if is_blocked_device(file_path):
654
+ raise ValueError(
655
+ f"Cannot read '{file_path}': this device file would block or produce infinite output."
656
+ )
657
+
658
+ # Check for binary files
659
+ ext = get_file_extension(file_path)
660
+ if is_binary_file(file_path) and not is_image_file(file_path) and ext != 'pdf':
661
+ raise ValueError(
662
+ f"This tool cannot read binary files. "
663
+ f"The file appears to be a binary .{ext} file."
664
+ )
665
+
666
+ async def execute(
667
+ self,
668
+ file_path: str,
669
+ offset: int | None = 1,
670
+ limit: int | None = None,
671
+ pages: str | None = None,
672
+ **kwargs
673
+ ) -> ToolResult:
674
+ """
675
+ Read a file.
676
+
677
+ Args:
678
+ file_path: Absolute path to the file to read
679
+ offset: The line number to start reading from (1-indexed)
680
+ limit: The number of lines to read
681
+ pages: Page range for PDF files (e.g., "1-5", "3", "10-20")
682
+
683
+ Returns:
684
+ ToolResult containing the file content
685
+ """
686
+ try:
687
+ # Expand and normalize path
688
+ full_path = expand_path(file_path)
689
+
690
+ # Validate input
691
+ self._validate_input(full_path)
692
+
693
+ # Determine file type and read accordingly
694
+ if is_notebook_file(full_path):
695
+ content, metadata = read_notebook_file(full_path)
696
+ return ToolResult(
697
+ content=content,
698
+ success=True,
699
+ metadata={"file_type": "notebook", **metadata}
700
+ )
701
+
702
+ if is_image_file(full_path):
703
+ content, metadata = read_image_file(full_path, self.max_size_bytes)
704
+ return ToolResult(
705
+ content=content,
706
+ success=True,
707
+ metadata={"file_type": "image", **metadata}
708
+ )
709
+
710
+ if is_pdf_file(full_path):
711
+ content, metadata = read_pdf_file(full_path, pages, self.max_size_bytes)
712
+ return ToolResult(
713
+ content=content,
714
+ success=True,
715
+ metadata={"file_type": "pdf", **metadata}
716
+ )
717
+
718
+ # Default: read as text file
719
+ content, metadata = read_text_file(full_path, offset or 1, limit, self.max_size_bytes)
720
+
721
+ # Check token limit for text content
722
+ if content:
723
+ token_estimate = rough_token_estimate(content)
724
+ if token_estimate > self.max_tokens:
725
+ return ToolResult(
726
+ content=f"File content ({token_estimate} estimated tokens) exceeds maximum allowed tokens ({self.max_tokens}). Use offset and limit parameters to read specific portions of the file.",
727
+ success=False,
728
+ error="Token limit exceeded"
729
+ )
730
+
731
+ # Handle empty file
732
+ if not content:
733
+ if metadata.get("total_lines", 0) == 0:
734
+ content = "<system-reminder>Warning: the file exists but the contents are empty.</system-reminder>"
735
+ else:
736
+ content = f"<system-reminder>Warning: the file exists but is shorter than the provided offset ({metadata.get('start_line')}). The file has {metadata.get('total_lines')} lines.</system-reminder>"
737
+
738
+ return ToolResult(
739
+ content=content,
740
+ success=True,
741
+ metadata={"file_type": "text", **metadata}
742
+ )
743
+
744
+ except FileNotFoundError as e:
745
+ return ToolResult(
746
+ content=f"Error: {e}",
747
+ success=False,
748
+ error=str(e)
749
+ )
750
+ except ValueError as e:
751
+ return ToolResult(
752
+ content=f"Error: {e}",
753
+ success=False,
754
+ error=str(e)
755
+ )
756
+ except Exception as e:
757
+ return ToolResult(
758
+ content=f"Error reading file: {e}",
759
+ success=False,
760
+ error=str(e)
761
+ )
762
+
763
+
764
+ # =============================================================================
765
+ # Convenience Functions
766
+ # =============================================================================
767
+
768
+ def read_file(
769
+ file_path: str,
770
+ offset: int | None = 1,
771
+ limit: int | None = None,
772
+ pages: str | None = None
773
+ ) -> tuple[str, dict]:
774
+ """
775
+ Quick file read function.
776
+
777
+ Args:
778
+ file_path: Absolute path to the file to read
779
+ offset: The line number to start reading from (1-indexed)
780
+ limit: The number of lines to read
781
+ pages: Page range for PDF files
782
+
783
+ Returns:
784
+ Tuple of (content, metadata)
785
+ """
786
+ tool = FileReadTool()
787
+ result = tool.read(file_path=file_path, offset=offset, limit=limit, pages=pages)
788
+ return result.content, result.metadata
789
+
790
+
791
+ # =============================================================================
792
+ # Exports
793
+ # =============================================================================
794
+
795
+ __all__ = [
796
+ 'FileReadTool',
797
+ 'read_file',
798
+ 'expand_path',
799
+ 'add_line_numbers',
800
+ 'IMAGE_EXTENSIONS',
801
+ 'BINARY_EXTENSIONS',
802
+ 'DEFAULT_MAX_SIZE_BYTES',
803
+ 'DEFAULT_MAX_TOKENS',
804
+ ]