opsmith-cli 0.1.2a0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- opsmith/__init__.py +0 -0
- opsmith/agent.py +243 -0
- opsmith/cloud_providers/__init__.py +10 -0
- opsmith/cloud_providers/aws.py +46 -0
- opsmith/cloud_providers/base.py +57 -0
- opsmith/cloud_providers/gcp.py +67 -0
- opsmith/constants.py +175 -0
- opsmith/deployer.py +236 -0
- opsmith/git_repo.py +66 -0
- opsmith/main.py +269 -0
- opsmith/prompts.py +76 -0
- opsmith/queries/tree-sitter-languages/README.md +7 -0
- opsmith/queries/tree-sitter-languages/arduino-tags.scm +5 -0
- opsmith/queries/tree-sitter-languages/c-tags.scm +9 -0
- opsmith/queries/tree-sitter-languages/c_sharp-tags.scm +46 -0
- opsmith/queries/tree-sitter-languages/chatito-tags.scm +16 -0
- opsmith/queries/tree-sitter-languages/commonlisp-tags.scm +122 -0
- opsmith/queries/tree-sitter-languages/cpp-tags.scm +15 -0
- opsmith/queries/tree-sitter-languages/csharp-tags.scm +26 -0
- opsmith/queries/tree-sitter-languages/d-tags.scm +26 -0
- opsmith/queries/tree-sitter-languages/dart-tags.scm +92 -0
- opsmith/queries/tree-sitter-languages/elisp-tags.scm +5 -0
- opsmith/queries/tree-sitter-languages/elixir-tags.scm +54 -0
- opsmith/queries/tree-sitter-languages/elm-tags.scm +19 -0
- opsmith/queries/tree-sitter-languages/gleam-tags.scm +41 -0
- opsmith/queries/tree-sitter-languages/go-tags.scm +42 -0
- opsmith/queries/tree-sitter-languages/hcl-tags.scm +77 -0
- opsmith/queries/tree-sitter-languages/java-tags.scm +20 -0
- opsmith/queries/tree-sitter-languages/javascript-tags.scm +88 -0
- opsmith/queries/tree-sitter-languages/kotlin-tags.scm +27 -0
- opsmith/queries/tree-sitter-languages/lua-tags.scm +34 -0
- opsmith/queries/tree-sitter-languages/ocaml-tags.scm +115 -0
- opsmith/queries/tree-sitter-languages/ocaml_interface-tags.scm +98 -0
- opsmith/queries/tree-sitter-languages/php-tags.scm +26 -0
- opsmith/queries/tree-sitter-languages/pony-tags.scm +39 -0
- opsmith/queries/tree-sitter-languages/properties-tags.scm +5 -0
- opsmith/queries/tree-sitter-languages/python-tags.scm +14 -0
- opsmith/queries/tree-sitter-languages/ql-tags.scm +26 -0
- opsmith/queries/tree-sitter-languages/r-tags.scm +21 -0
- opsmith/queries/tree-sitter-languages/racket-tags.scm +12 -0
- opsmith/queries/tree-sitter-languages/ruby-tags.scm +64 -0
- opsmith/queries/tree-sitter-languages/rust-tags.scm +60 -0
- opsmith/queries/tree-sitter-languages/scala-tags.scm +65 -0
- opsmith/queries/tree-sitter-languages/solidity-tags.scm +43 -0
- opsmith/queries/tree-sitter-languages/swift-tags.scm +51 -0
- opsmith/queries/tree-sitter-languages/typescript-tags.scm +41 -0
- opsmith/queries/tree-sitter-languages/udev-tags.scm +20 -0
- opsmith/repo_map.py +572 -0
- opsmith/settings.py +26 -0
- opsmith/spinner.py +221 -0
- opsmith_cli-0.1.2a0.dist-info/METADATA +31 -0
- opsmith_cli-0.1.2a0.dist-info/RECORD +55 -0
- opsmith_cli-0.1.2a0.dist-info/WHEEL +4 -0
- opsmith_cli-0.1.2a0.dist-info/entry_points.txt +2 -0
- opsmith_cli-0.1.2a0.dist-info/licenses/LICENSE.txt +674 -0
opsmith/repo_map.py
ADDED
|
@@ -0,0 +1,572 @@
|
|
|
1
|
+
import os
|
|
2
|
+
from collections import defaultdict
|
|
3
|
+
from importlib import resources
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Dict, List, NamedTuple, Optional, Set, Tuple
|
|
6
|
+
|
|
7
|
+
import typer
|
|
8
|
+
from grep_ast import TreeContext, filename_to_lang
|
|
9
|
+
from grep_ast.tsl import get_language, get_parser
|
|
10
|
+
from tqdm import tqdm
|
|
11
|
+
|
|
12
|
+
from opsmith.constants import ROOT_IMPORTANT_FILES
|
|
13
|
+
from opsmith.git_repo import GitRepo
|
|
14
|
+
from opsmith.spinner import WaitingSpinner
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class Tag(NamedTuple):
|
|
18
|
+
rel_filename: str
|
|
19
|
+
filename: str
|
|
20
|
+
line: int
|
|
21
|
+
name: str
|
|
22
|
+
kind: str
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
REPO_MAP_MESSAGE = "Generating repo map"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def get_scm_filename(lang: str) -> Optional[Path]:
|
|
29
|
+
"""
|
|
30
|
+
Retrieve the filename of the `.scm` (S-expression-based queries) file for the
|
|
31
|
+
specified programming language from the package's resource directory.
|
|
32
|
+
|
|
33
|
+
This function attempts to locate a file corresponding to the given language's
|
|
34
|
+
tags, stored in the "tree-sitter-languages" subdirectory under the "queries"
|
|
35
|
+
directory within the package. If the file exists, its corresponding `Path`
|
|
36
|
+
object is returned. If the file cannot be found or an error occurs while
|
|
37
|
+
accessing resources, the function returns `None`.
|
|
38
|
+
|
|
39
|
+
:param lang: The name of the programming language for which the `.scm` file
|
|
40
|
+
is being queried.
|
|
41
|
+
:type lang: str
|
|
42
|
+
:return: A `Path` object pointing to the `.scm` file if it exists, or `None`
|
|
43
|
+
if the file is not found or an error occurs.
|
|
44
|
+
:rtype: Optional[Path]
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
# Try tree-sitter-language-pack subdir first
|
|
48
|
+
try:
|
|
49
|
+
path = resources.files(__package__).joinpath(
|
|
50
|
+
"queries", "tree-sitter-languages", f"{lang}-tags.scm"
|
|
51
|
+
)
|
|
52
|
+
if path.is_file():
|
|
53
|
+
return path
|
|
54
|
+
except KeyError:
|
|
55
|
+
pass # Silently continue if path doesn't exist or package structure is not as expected
|
|
56
|
+
|
|
57
|
+
return None
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class RepoMap:
|
|
61
|
+
warned_files: Set[str]
|
|
62
|
+
|
|
63
|
+
def __init__(
|
|
64
|
+
self,
|
|
65
|
+
src_dir: str,
|
|
66
|
+
map_tokens: int = 5120,
|
|
67
|
+
max_tags_depth: int = 2,
|
|
68
|
+
repo_content_prefix: Optional[
|
|
69
|
+
str
|
|
70
|
+
] = "This repo map contains a list of files and important symbols.\n\n",
|
|
71
|
+
verbose: bool = False,
|
|
72
|
+
):
|
|
73
|
+
"""
|
|
74
|
+
Initializes an instance of the class, which sets up configuration and parameters
|
|
75
|
+
required for processing a repository. It validates paths, initializes relevant
|
|
76
|
+
attributes for managing warnings and tag depths, and optionally logs verbose
|
|
77
|
+
messages if enabled.
|
|
78
|
+
|
|
79
|
+
:param src_dir: The directory path of the repository to be processed.
|
|
80
|
+
:param map_tokens: The maximum number of tokens allowed in the mapping
|
|
81
|
+
process.
|
|
82
|
+
:param max_tags_depth: The maximum depth level for tagging content within
|
|
83
|
+
the repository.
|
|
84
|
+
:param repo_content_prefix: An optional prefix string that specifies initial
|
|
85
|
+
details or descriptions before mapping repository content.
|
|
86
|
+
:param verbose: A flag indicating if detailed logging should be enabled.
|
|
87
|
+
"""
|
|
88
|
+
self.src_dir = Path(src_dir).resolve()
|
|
89
|
+
self.git_repo = GitRepo(self.src_dir)
|
|
90
|
+
self.verbose = verbose
|
|
91
|
+
|
|
92
|
+
# Initialize tracking variables
|
|
93
|
+
self.warned_files = set()
|
|
94
|
+
|
|
95
|
+
self.max_map_tokens = map_tokens
|
|
96
|
+
self.max_tags_depth = max_tags_depth
|
|
97
|
+
self.repo_content_prefix = repo_content_prefix if repo_content_prefix is not None else ""
|
|
98
|
+
|
|
99
|
+
self._warned_missing_scm = set()
|
|
100
|
+
if self.verbose:
|
|
101
|
+
typer.echo(f"RepoMap initialized for {self.src_dir}")
|
|
102
|
+
|
|
103
|
+
@staticmethod
|
|
104
|
+
def _simple_token_count(text: str) -> int:
|
|
105
|
+
"""A very rough estimate of token count."""
|
|
106
|
+
return len(text) // 4 # Common heuristic: 1 token ~ 4 chars
|
|
107
|
+
|
|
108
|
+
def _token_count(self, text: str) -> int:
|
|
109
|
+
"""
|
|
110
|
+
Estimates token count. For large texts, it samples to speed up.
|
|
111
|
+
Uses a simple character-based heuristic.
|
|
112
|
+
"""
|
|
113
|
+
len_text = len(text)
|
|
114
|
+
if len_text == 0:
|
|
115
|
+
return 0
|
|
116
|
+
if len_text < 200: # For small texts, count directly
|
|
117
|
+
return self._simple_token_count(text)
|
|
118
|
+
|
|
119
|
+
# For larger texts, sample to estimate
|
|
120
|
+
lines = text.splitlines(keepends=True)
|
|
121
|
+
num_lines = len(lines)
|
|
122
|
+
if num_lines == 0:
|
|
123
|
+
return 0
|
|
124
|
+
|
|
125
|
+
step = num_lines // 100 or 1 # Sample ~100 lines
|
|
126
|
+
sampled_lines = lines[::step]
|
|
127
|
+
sample_text = "".join(sampled_lines)
|
|
128
|
+
|
|
129
|
+
if not sample_text: # handle case where sampling results in empty text
|
|
130
|
+
return self._simple_token_count(text) # fallback to full count
|
|
131
|
+
|
|
132
|
+
sample_tokens = self._simple_token_count(sample_text)
|
|
133
|
+
|
|
134
|
+
# Extrapolate from sample to full text
|
|
135
|
+
# Ensure len(sample_text) is not zero to avoid division by zero
|
|
136
|
+
if len(sample_text) > 0:
|
|
137
|
+
est_tokens = (sample_tokens / len(sample_text)) * len_text
|
|
138
|
+
else: # if sample_text is empty (e.g. all sampled lines were empty)
|
|
139
|
+
est_tokens = self._simple_token_count(text) # fallback to full text estimate
|
|
140
|
+
|
|
141
|
+
return int(est_tokens)
|
|
142
|
+
|
|
143
|
+
def get_repo_map(
|
|
144
|
+
self,
|
|
145
|
+
) -> Optional[str]:
|
|
146
|
+
if self.max_map_tokens <= 0:
|
|
147
|
+
return None # Repo map is disabled
|
|
148
|
+
|
|
149
|
+
all_tracked_files_paths = self.git_repo.get_git_tracked_files(
|
|
150
|
+
[str(self.src_dir), ":!**/*test*"]
|
|
151
|
+
)
|
|
152
|
+
if not all_tracked_files_paths:
|
|
153
|
+
if self.verbose:
|
|
154
|
+
typer.echo("RepoMap: No git-tracked files found.", err=True)
|
|
155
|
+
return None # No files to map
|
|
156
|
+
|
|
157
|
+
all_tracked_files_abs_str = [str(p.resolve()) for p in all_tracked_files_paths]
|
|
158
|
+
|
|
159
|
+
current_max_map_tokens = self.max_map_tokens
|
|
160
|
+
|
|
161
|
+
try:
|
|
162
|
+
with WaitingSpinner(text=REPO_MAP_MESSAGE, delay=0.1) as spinner_obj:
|
|
163
|
+
# Progress within get_tags_map will use spinner.step or tqdm
|
|
164
|
+
files_listing = self.get_tags_map(
|
|
165
|
+
all_filenames_abs=all_tracked_files_abs_str,
|
|
166
|
+
max_tokens=current_max_map_tokens,
|
|
167
|
+
progress_callback=spinner_obj.spinner.step,
|
|
168
|
+
)
|
|
169
|
+
except RecursionError:
|
|
170
|
+
typer.echo(
|
|
171
|
+
(
|
|
172
|
+
"Error: Disabling repo map, recursion depth exceeded. "
|
|
173
|
+
"Git repo might be too large or complex."
|
|
174
|
+
),
|
|
175
|
+
err=True,
|
|
176
|
+
)
|
|
177
|
+
self.max_map_tokens = 0 # Disable for future calls
|
|
178
|
+
return None
|
|
179
|
+
except Exception as e:
|
|
180
|
+
typer.echo(f"Error generating repo map: {e}", err=True)
|
|
181
|
+
if self.verbose:
|
|
182
|
+
import traceback
|
|
183
|
+
|
|
184
|
+
traceback.print_exc()
|
|
185
|
+
return None
|
|
186
|
+
|
|
187
|
+
if not files_listing:
|
|
188
|
+
return None
|
|
189
|
+
|
|
190
|
+
if self.verbose:
|
|
191
|
+
num_tokens = self._token_count(files_listing)
|
|
192
|
+
typer.echo(
|
|
193
|
+
f"RepoMap: Final map size {num_tokens / 1024:.1f} k-tokens (estimated)",
|
|
194
|
+
err=True,
|
|
195
|
+
)
|
|
196
|
+
|
|
197
|
+
repo_content = self.repo_content_prefix + files_listing
|
|
198
|
+
return repo_content
|
|
199
|
+
|
|
200
|
+
def _get_rel_filename(self, filename: str) -> str:
|
|
201
|
+
try:
|
|
202
|
+
return os.path.relpath(filename, str(self.src_dir))
|
|
203
|
+
except ValueError: # Handles cross-drive issues on Windows
|
|
204
|
+
return filename # Return absolute path if relpath fails
|
|
205
|
+
|
|
206
|
+
@staticmethod
|
|
207
|
+
def _filter_important_files(filenames: List[str]) -> List[str]:
|
|
208
|
+
"""
|
|
209
|
+
Filters out the important files from a given list of filenames by checking their names
|
|
210
|
+
against a predefined set of important files.
|
|
211
|
+
|
|
212
|
+
Important files are determined by their presence in the `ROOT_IMPORTANT_FILES`.
|
|
213
|
+
|
|
214
|
+
:param filenames: List of file paths as strings.
|
|
215
|
+
:type filenames: List[str]
|
|
216
|
+
|
|
217
|
+
:return: A filtered list containing only the important file paths that match
|
|
218
|
+
the names in `ROOT_IMPORTANT_FILES`.
|
|
219
|
+
:rtype: List[str]
|
|
220
|
+
"""
|
|
221
|
+
priority_files = []
|
|
222
|
+
for filename_str in filenames:
|
|
223
|
+
p_filename = Path(filename_str)
|
|
224
|
+
if p_filename.name in ROOT_IMPORTANT_FILES:
|
|
225
|
+
priority_files.append(filename_str)
|
|
226
|
+
return priority_files
|
|
227
|
+
|
|
228
|
+
def _get_tags(self, filename_abs_str: str, rel_filename_str: str) -> List[Tag]:
|
|
229
|
+
# Directly get raw tags without caching
|
|
230
|
+
return list(self._get_tags_raw(filename_abs_str, rel_filename_str))
|
|
231
|
+
|
|
232
|
+
def _get_tags_raw(self, filename_abs_str: str, rel_filename_str: str) -> List[Tag]:
|
|
233
|
+
tags = []
|
|
234
|
+
lang = filename_to_lang(filename_abs_str)
|
|
235
|
+
if not lang:
|
|
236
|
+
return tags
|
|
237
|
+
|
|
238
|
+
try:
|
|
239
|
+
# These are needed by language.query. grep-ast might handle their loading.
|
|
240
|
+
language_module = get_language(lang) # from grep_ast.tsl
|
|
241
|
+
parser_module = get_parser(lang) # from grep_ast.tsl
|
|
242
|
+
except Exception as err:
|
|
243
|
+
# This can happen if tree-sitter binaries/parsers for the lang are not found
|
|
244
|
+
if self.verbose:
|
|
245
|
+
typer.echo(
|
|
246
|
+
(
|
|
247
|
+
f"RepoMap: Skipping file {filename_abs_str} for tags (parser/lang init"
|
|
248
|
+
f" error): {err}"
|
|
249
|
+
),
|
|
250
|
+
err=True,
|
|
251
|
+
)
|
|
252
|
+
return tags
|
|
253
|
+
|
|
254
|
+
query_scm_path = get_scm_filename(lang)
|
|
255
|
+
if not query_scm_path or not query_scm_path.exists():
|
|
256
|
+
if self.verbose and lang not in getattr(self, "_warned_missing_scm", set()):
|
|
257
|
+
self._warned_missing_scm.add(lang)
|
|
258
|
+
return tags
|
|
259
|
+
query_scm_content = query_scm_path.read_text(encoding="utf-8")
|
|
260
|
+
|
|
261
|
+
try:
|
|
262
|
+
code = Path(filename_abs_str).read_text(encoding="utf-8", errors="ignore")
|
|
263
|
+
except Exception as e:
|
|
264
|
+
if self.verbose:
|
|
265
|
+
typer.echo(
|
|
266
|
+
f"RepoMap: Could not read file {filename_abs_str} for tagging: {e}",
|
|
267
|
+
err=True,
|
|
268
|
+
)
|
|
269
|
+
return tags
|
|
270
|
+
|
|
271
|
+
if not code:
|
|
272
|
+
return tags
|
|
273
|
+
|
|
274
|
+
tree = parser_module.parse(bytes(code, "utf-8"))
|
|
275
|
+
query = language_module.query(query_scm_content)
|
|
276
|
+
captures = query.captures(tree.root_node)
|
|
277
|
+
|
|
278
|
+
processed_tags = set() # To avoid duplicate tags from different patterns
|
|
279
|
+
|
|
280
|
+
saw_kinds = set()
|
|
281
|
+
all_nodes = []
|
|
282
|
+
for tag, nodes in captures.items():
|
|
283
|
+
all_nodes += [(node, tag) for node in nodes]
|
|
284
|
+
|
|
285
|
+
for node, tag_name_str in all_nodes:
|
|
286
|
+
# Example tag_name_str: "name.definition.function", "name.reference.class"
|
|
287
|
+
if tag_name_str.startswith("name.definition."):
|
|
288
|
+
kind = "def"
|
|
289
|
+
elif tag_name_str.startswith("name.reference."):
|
|
290
|
+
kind = "ref"
|
|
291
|
+
else: # Other patterns not directly used for defs/refs
|
|
292
|
+
continue
|
|
293
|
+
|
|
294
|
+
saw_kinds.add(kind)
|
|
295
|
+
node_text = node.text.decode("utf-8", "ignore")
|
|
296
|
+
line_no = node.start_point[0]
|
|
297
|
+
|
|
298
|
+
tag_tuple = (rel_filename_str, filename_abs_str, line_no, node_text, kind)
|
|
299
|
+
if tag_tuple not in processed_tags:
|
|
300
|
+
tags.append(Tag(*tag_tuple))
|
|
301
|
+
processed_tags.add(tag_tuple)
|
|
302
|
+
|
|
303
|
+
return tags
|
|
304
|
+
|
|
305
|
+
def _get_all_tags(
|
|
306
|
+
self,
|
|
307
|
+
all_filenames_abs: List[str], # All git-tracked files, absolute paths
|
|
308
|
+
progress_callback: Optional[callable] = None,
|
|
309
|
+
) -> List[Tag | Tuple[str]]:
|
|
310
|
+
""" """
|
|
311
|
+
all_tags: List[Tag | Tuple[str]] = []
|
|
312
|
+
if not all_filenames_abs:
|
|
313
|
+
return []
|
|
314
|
+
|
|
315
|
+
# Use tqdm for progress if no callback or if verbose
|
|
316
|
+
filenames_iterable = all_filenames_abs
|
|
317
|
+
if (
|
|
318
|
+
self.verbose and not progress_callback
|
|
319
|
+
): # Only use tqdm if no specific callback and verbose
|
|
320
|
+
filenames_iterable = tqdm(
|
|
321
|
+
all_filenames_abs,
|
|
322
|
+
desc="Scanning repo files for tags",
|
|
323
|
+
unit="file",
|
|
324
|
+
disable=not self.verbose,
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
for i, filename_abs in enumerate(filenames_iterable):
|
|
328
|
+
if progress_callback:
|
|
329
|
+
progress_callback(f"{REPO_MAP_MESSAGE}: Scanning file {Path(filename_abs).name}")
|
|
330
|
+
|
|
331
|
+
try:
|
|
332
|
+
if not Path(filename_abs).is_file():
|
|
333
|
+
if filename_abs not in self.warned_files:
|
|
334
|
+
typer.echo(
|
|
335
|
+
f"RepoMap: File not found or not a file: {filename_abs}",
|
|
336
|
+
err=True,
|
|
337
|
+
)
|
|
338
|
+
self.warned_files.add(filename_abs)
|
|
339
|
+
continue
|
|
340
|
+
except OSError as e: # Permissions, etc.
|
|
341
|
+
if filename_abs not in self.warned_files:
|
|
342
|
+
typer.echo(f"RepoMap: OSError for file {filename_abs}: {e}", err=True)
|
|
343
|
+
self.warned_files.add(filename_abs)
|
|
344
|
+
continue
|
|
345
|
+
|
|
346
|
+
rel_filename = self._get_rel_filename(filename_abs)
|
|
347
|
+
if len(Path(rel_filename).parts) <= self.max_tags_depth:
|
|
348
|
+
file_tags = self._get_tags(filename_abs, rel_filename)
|
|
349
|
+
file_def_tags = [tag for tag in file_tags if tag.kind == "def"]
|
|
350
|
+
if file_def_tags:
|
|
351
|
+
all_tags.extend(file_def_tags)
|
|
352
|
+
else:
|
|
353
|
+
all_tags.append((rel_filename,))
|
|
354
|
+
else:
|
|
355
|
+
all_tags.append((rel_filename,))
|
|
356
|
+
|
|
357
|
+
return all_tags
|
|
358
|
+
|
|
359
|
+
def get_tags_map(
|
|
360
|
+
self,
|
|
361
|
+
all_filenames_abs: List[str], # All git-tracked files
|
|
362
|
+
max_tokens: int,
|
|
363
|
+
progress_callback: Optional[callable] = None,
|
|
364
|
+
) -> Optional[str]:
|
|
365
|
+
if progress_callback:
|
|
366
|
+
progress_callback(f"{REPO_MAP_MESSAGE}: Ranking tags and files...")
|
|
367
|
+
tags = self._get_all_tags(
|
|
368
|
+
all_filenames_abs=all_filenames_abs, progress_callback=progress_callback
|
|
369
|
+
)
|
|
370
|
+
|
|
371
|
+
# Prioritize "special" files (e.g. README) from all_filenames_abs
|
|
372
|
+
# These are added at the beginning of the ranked list if not already effectively there.
|
|
373
|
+
all_rel_filenames = sorted(
|
|
374
|
+
list(set(self._get_rel_filename(filename) for filename in all_filenames_abs))
|
|
375
|
+
)
|
|
376
|
+
|
|
377
|
+
# Get relative paths of files already represented by ranked tags
|
|
378
|
+
tags_rel_filenames = set()
|
|
379
|
+
for tag in tags:
|
|
380
|
+
if isinstance(tag, Tag):
|
|
381
|
+
tags_rel_filenames.add(tag.rel_filename)
|
|
382
|
+
else:
|
|
383
|
+
tags_rel_filenames.add(tag[0])
|
|
384
|
+
|
|
385
|
+
# Identify special files not already covered by high-ranking tags
|
|
386
|
+
# These are added as (rel_filename,) tuples to ensure their presence.
|
|
387
|
+
special_file_tuples_to_prepend = []
|
|
388
|
+
# Note: filter_important_files returns list of rel_filenames
|
|
389
|
+
for special_rel_filename in self._filter_important_files(
|
|
390
|
+
all_rel_filenames
|
|
391
|
+
): # Use all_rel_filenames
|
|
392
|
+
if special_rel_filename not in tags_rel_filenames and special_rel_filename:
|
|
393
|
+
special_file_tuples_to_prepend.append((special_rel_filename,))
|
|
394
|
+
|
|
395
|
+
# Prepend special files, then the ranked list
|
|
396
|
+
# This ensures special files are considered early in the truncation process.
|
|
397
|
+
final_item_list = special_file_tuples_to_prepend + tags
|
|
398
|
+
|
|
399
|
+
# Deduplicate while preserving order (important for prepended special files)
|
|
400
|
+
# An item can be Tag or Tuple[str]. Need a consistent way to check uniqueness.
|
|
401
|
+
# Uniqueness check: For Tags, by (rel_filename, name, line, kind). For Tuples,
|
|
402
|
+
# by (rel_filename,).
|
|
403
|
+
seen_items_repr = set()
|
|
404
|
+
deduplicated_final_item_list = []
|
|
405
|
+
for item in final_item_list:
|
|
406
|
+
if isinstance(item, Tag):
|
|
407
|
+
repr_key = (
|
|
408
|
+
"Tag",
|
|
409
|
+
item.rel_filename,
|
|
410
|
+
item.name,
|
|
411
|
+
) # Simpler key for deduplication
|
|
412
|
+
else: # Tuple (rel_filename,)
|
|
413
|
+
repr_key = ("File", item[0])
|
|
414
|
+
|
|
415
|
+
if repr_key not in seen_items_repr:
|
|
416
|
+
deduplicated_final_item_list.append(item)
|
|
417
|
+
seen_items_repr.add(repr_key)
|
|
418
|
+
|
|
419
|
+
final_item_list = deduplicated_final_item_list
|
|
420
|
+
|
|
421
|
+
# Binary search to find the number of items (tags or files) that fit token limit
|
|
422
|
+
num_total_items = len(final_item_list)
|
|
423
|
+
if num_total_items == 0:
|
|
424
|
+
return "" # Empty map if no items
|
|
425
|
+
|
|
426
|
+
lower_bound = 0
|
|
427
|
+
upper_bound = num_total_items
|
|
428
|
+
best_map_text = ""
|
|
429
|
+
best_map_tokens = 0 # Tokens of the best map found so far that is <= max_tokens
|
|
430
|
+
|
|
431
|
+
# Iterative refinement (binary search like) to select items fitting token budget
|
|
432
|
+
# Max iterations to prevent infinite loops with tricky token counts
|
|
433
|
+
max_iterations = 20 # Should be enough for a wide range of num_total_items
|
|
434
|
+
current_iter = 0
|
|
435
|
+
|
|
436
|
+
# Heuristic for initial guess of items (middle)
|
|
437
|
+
# Average tokens per item is unknown. Start with a fraction of total items.
|
|
438
|
+
# Aider used `min(int(max_tokens // 25), num_tags)`. 25 is empirical avg tokens/tag line.
|
|
439
|
+
middle = min(
|
|
440
|
+
num_total_items, max(1, int(max_tokens / 25))
|
|
441
|
+
) # Ensure middle >= 1 if num_total_items > 0
|
|
442
|
+
|
|
443
|
+
while lower_bound <= upper_bound and current_iter < max_iterations:
|
|
444
|
+
current_iter += 1
|
|
445
|
+
if progress_callback:
|
|
446
|
+
progress_callback(
|
|
447
|
+
f"Formatting map, trying {middle} items (bounds: {lower_bound}-{upper_bound})"
|
|
448
|
+
)
|
|
449
|
+
|
|
450
|
+
current_selection = final_item_list[:middle]
|
|
451
|
+
map_text = self.to_tree(current_selection)
|
|
452
|
+
num_tokens = self._token_count(map_text)
|
|
453
|
+
|
|
454
|
+
# Percentage error from target token count
|
|
455
|
+
# Accept if within a certain tolerance (e.g., 15% of max_tokens)
|
|
456
|
+
# This helps converge faster if an exact match is hard.
|
|
457
|
+
token_err_pct = abs(num_tokens - max_tokens) / max_tokens if max_tokens > 0 else 0.0
|
|
458
|
+
err_tolerance = 0.15
|
|
459
|
+
|
|
460
|
+
if num_tokens <= max_tokens: # Current map fits
|
|
461
|
+
if num_tokens > best_map_tokens: # And it's better than previous best
|
|
462
|
+
best_map_text = map_text
|
|
463
|
+
best_map_tokens = num_tokens
|
|
464
|
+
|
|
465
|
+
if token_err_pct < err_tolerance: # Good enough, stop
|
|
466
|
+
break
|
|
467
|
+
lower_bound = middle + 1 # Try to include more items
|
|
468
|
+
else: # Current map is too large
|
|
469
|
+
upper_bound = middle - 1 # Need to include fewer items
|
|
470
|
+
|
|
471
|
+
if lower_bound > upper_bound: # Bounds crossed
|
|
472
|
+
break
|
|
473
|
+
|
|
474
|
+
middle = (lower_bound + upper_bound) // 2
|
|
475
|
+
if middle == 0 and lower_bound == 0 and upper_bound == 0 and num_total_items > 0:
|
|
476
|
+
# If stuck at 0 but there are items, try at least 1 if map was too large
|
|
477
|
+
if num_tokens > max_tokens:
|
|
478
|
+
middle = 0 # Stay at 0 if even 0 items (empty map) is too large (prefix issue?)
|
|
479
|
+
else:
|
|
480
|
+
middle = 1 # Try 1 item if 0 items fit (empty map)
|
|
481
|
+
|
|
482
|
+
if progress_callback:
|
|
483
|
+
progress_callback("Map formatting complete.")
|
|
484
|
+
return best_map_text
|
|
485
|
+
|
|
486
|
+
@staticmethod
|
|
487
|
+
def render_tree(
|
|
488
|
+
abs_filename_str: str, rel_filename_str: str, lines_of_interest: List[int]
|
|
489
|
+
) -> str:
|
|
490
|
+
"""
|
|
491
|
+
Renders a summarized view of a file, focusing on lines_of_interest.
|
|
492
|
+
Uses grep_ast.TreeContext for structured formatting.
|
|
493
|
+
"""
|
|
494
|
+
code = Path(abs_filename_str).read_text(encoding="utf-8", errors="ignore")
|
|
495
|
+
if not code.endswith("\n"):
|
|
496
|
+
code += "\n" # Ensure trailing newline for TreeContext
|
|
497
|
+
|
|
498
|
+
# TreeContext parameters from aider, adjusted for opsmith
|
|
499
|
+
context = TreeContext(
|
|
500
|
+
filename=rel_filename_str,
|
|
501
|
+
code=code,
|
|
502
|
+
color=False,
|
|
503
|
+
line_number=False,
|
|
504
|
+
child_context=False,
|
|
505
|
+
last_line=False,
|
|
506
|
+
margin=0,
|
|
507
|
+
mark_lois=False,
|
|
508
|
+
loi_pad=0,
|
|
509
|
+
show_top_of_file_parent_scope=False,
|
|
510
|
+
)
|
|
511
|
+
|
|
512
|
+
# Set lines of interest for this specific rendering
|
|
513
|
+
context.lines_of_interest = set(lines_of_interest)
|
|
514
|
+
context.add_context() # Compute context around LOIs
|
|
515
|
+
|
|
516
|
+
formatted_text = context.format()
|
|
517
|
+
return formatted_text
|
|
518
|
+
|
|
519
|
+
def to_tree(self, items: List[Tag | Tuple[str]]) -> str:
|
|
520
|
+
"""
|
|
521
|
+
Converts a list of Tags into a string tree representation.
|
|
522
|
+
Tags are grouped by file.
|
|
523
|
+
"""
|
|
524
|
+
if not items:
|
|
525
|
+
return ""
|
|
526
|
+
output_lines = []
|
|
527
|
+
|
|
528
|
+
# Group tags by file to render each file's summary once
|
|
529
|
+
file_to_tags: Dict[str, List[Tag]] = defaultdict(list)
|
|
530
|
+
# Standalone files (rel_filename,) are processed separately
|
|
531
|
+
standalone_files: List[str] = []
|
|
532
|
+
|
|
533
|
+
for item in items:
|
|
534
|
+
if isinstance(item, Tag):
|
|
535
|
+
# item.filename is absolute path, item.rel_filename is relative
|
|
536
|
+
file_to_tags[item.rel_filename].append(item)
|
|
537
|
+
else:
|
|
538
|
+
file_to_tags[item[0]] = []
|
|
539
|
+
|
|
540
|
+
# Process files with tags
|
|
541
|
+
# Sort by rel_filename for consistent output order
|
|
542
|
+
for rel_filename_sorted, tags_in_file in sorted(file_to_tags.items(), key=lambda x: x[0]):
|
|
543
|
+
if tags_in_file:
|
|
544
|
+
output_lines.append(f"\n{rel_filename_sorted}:")
|
|
545
|
+
# Need absolute path for render_tree to read file content
|
|
546
|
+
# Assuming all tags in tags_in_file have the same abs_filename for this rel_filename
|
|
547
|
+
abs_filename_for_render = tags_in_file[0].filename # Get abs path from first tag
|
|
548
|
+
|
|
549
|
+
lines_of_interest = [
|
|
550
|
+
tag.line for tag in tags_in_file if tag.line >= 0
|
|
551
|
+
] # Valid line numbers
|
|
552
|
+
rendered_content = self.render_tree(
|
|
553
|
+
abs_filename_for_render, rel_filename_sorted, lines_of_interest
|
|
554
|
+
)
|
|
555
|
+
output_lines.append(rendered_content)
|
|
556
|
+
else:
|
|
557
|
+
output_lines.append(f"\n{rel_filename_sorted}\n")
|
|
558
|
+
|
|
559
|
+
# Process standalone files (those without specific tags to show, just list the filename)
|
|
560
|
+
# These are files that were ranked but either had no tags or their tags weren't high enough.
|
|
561
|
+
# Ensure they are not already covered by file_to_tags processing.
|
|
562
|
+
processed_standalone_files = set(file_to_tags.keys())
|
|
563
|
+
for rel_fname_standalone in sorted(list(set(standalone_files))): # Sort and unique
|
|
564
|
+
if rel_fname_standalone not in processed_standalone_files:
|
|
565
|
+
output_lines.append(f"\n{rel_fname_standalone}") # Just list filename
|
|
566
|
+
|
|
567
|
+
full_output = "".join(output_lines)
|
|
568
|
+
|
|
569
|
+
# Truncate very long lines (e.g., minified JS) as a final safety measure
|
|
570
|
+
# This was in aider's original to_tree.
|
|
571
|
+
truncated_lines = [line[:100] for line in full_output.splitlines()]
|
|
572
|
+
return "\n".join(truncated_lines) + "\n"
|
opsmith/settings.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
from pydantic_settings import (
|
|
2
|
+
BaseSettings,
|
|
3
|
+
PydanticBaseSettingsSource,
|
|
4
|
+
SettingsConfigDict,
|
|
5
|
+
YamlConfigSettingsSource,
|
|
6
|
+
)
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class OpsmithSettings(BaseSettings):
|
|
10
|
+
model_config = SettingsConfigDict(env_prefix="OPSMITH_", yaml_file=".opsmith.conf.yml")
|
|
11
|
+
|
|
12
|
+
deployments_dir: str = ".opsmith.deployments"
|
|
13
|
+
|
|
14
|
+
@classmethod
|
|
15
|
+
def settings_customise_sources(
|
|
16
|
+
cls,
|
|
17
|
+
settings_cls: type[BaseSettings],
|
|
18
|
+
init_settings: PydanticBaseSettingsSource,
|
|
19
|
+
env_settings: PydanticBaseSettingsSource,
|
|
20
|
+
dotenv_settings: PydanticBaseSettingsSource,
|
|
21
|
+
file_secret_settings: PydanticBaseSettingsSource,
|
|
22
|
+
) -> tuple[PydanticBaseSettingsSource, ...]:
|
|
23
|
+
return (YamlConfigSettingsSource(settings_cls),)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
settings = OpsmithSettings()
|