agentramen 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
agentramen/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """Local-first repository memory and context tools."""
2
+
3
+ __version__ = "0.1.0"
@@ -0,0 +1,229 @@
1
+ """Language-specific extraction kept independent from graph storage."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import ast
6
+ import importlib
7
+ import posixpath
8
+ import re
9
+ from dataclasses import dataclass
10
+ from pathlib import Path
11
+
12
+
13
+ @dataclass(frozen=True)
14
+ class Analysis:
15
+ symbols: tuple[str, ...] = ()
16
+ imports: tuple[str, ...] = ()
17
+ calls: tuple[str, ...] = ()
18
+
19
+
20
+ class LanguageAnalyzer:
21
+ language: str
22
+
23
+ def analyze(self, path: str, source: str) -> Analysis:
24
+ raise NotImplementedError
25
+
26
+
27
+ class PythonAnalyzer(LanguageAnalyzer):
28
+ language = "python"
29
+
30
+ def analyze(self, path: str, source: str) -> Analysis:
31
+ try:
32
+ tree = ast.parse(source, filename=path)
33
+ except (SyntaxError, ValueError):
34
+ return Analysis()
35
+ symbols: set[str] = set()
36
+ imports: set[str] = set()
37
+ calls: set[str] = set()
38
+
39
+ class Visitor(ast.NodeVisitor):
40
+ def __init__(self) -> None:
41
+ self.scope: list[str] = []
42
+
43
+ def visit_ClassDef(self, node: ast.ClassDef) -> None:
44
+ symbols.add(".".join([*self.scope, node.name]))
45
+ self.scope.append(node.name)
46
+ self.generic_visit(node)
47
+ self.scope.pop()
48
+
49
+ def visit_FunctionDef(self, node: ast.FunctionDef | ast.AsyncFunctionDef) -> None:
50
+ symbols.add(".".join([*self.scope, node.name]))
51
+ self.scope.append(node.name)
52
+ self.generic_visit(node)
53
+ self.scope.pop()
54
+
55
+ visit_AsyncFunctionDef = visit_FunctionDef
56
+
57
+ def visit_Import(self, node: ast.Import) -> None:
58
+ imports.update(alias.name for alias in node.names)
59
+
60
+ def visit_ImportFrom(self, node: ast.ImportFrom) -> None:
61
+ module = "." * node.level + (node.module or "")
62
+ imports.add(module)
63
+ imports.update(f"{module}.{alias.name}" for alias in node.names)
64
+
65
+ def visit_Call(self, node: ast.Call) -> None:
66
+ current = node.func
67
+ parts: list[str] = []
68
+ while isinstance(current, ast.Attribute):
69
+ parts.append(current.attr)
70
+ current = current.value
71
+ if isinstance(current, ast.Name):
72
+ parts.append(current.id)
73
+ if parts:
74
+ calls.add(".".join(reversed(parts)))
75
+ self.generic_visit(node)
76
+
77
+ Visitor().visit(tree)
78
+ return Analysis(tuple(sorted(symbols)), tuple(sorted(imports)), tuple(sorted(calls)))
79
+
80
+
81
+ class RegexAnalyzer(LanguageAnalyzer):
82
+ """Conservative symbol/import extraction for languages without a built-in parser."""
83
+
84
+ def __init__(self, language: str):
85
+ self.language = language
86
+ self.symbol_pattern = re.compile(
87
+ r"^\s*(?:export\s+)?(?:async\s+)?(?:class|interface|type|enum|function|def|fn|func|struct|trait)\s+([A-Za-z_$][\w$]*)",
88
+ re.MULTILINE,
89
+ )
90
+ self.import_pattern = re.compile(
91
+ r"""^\s*(?:import\s+(?:.*?\s+from\s+)?|from\s+[\w.]+\s+import\s+|(?:const|let|var)\s+\w+\s*=\s*require\()\s*['"]([^'"]+)['"]""",
92
+ re.MULTILINE,
93
+ )
94
+ self.additional_import_patterns = {
95
+ "java": r"^\s*import\s+(?:static\s+)?([\w.*]+)",
96
+ "kotlin": r"^\s*import\s+([\w.*]+)",
97
+ "csharp": r"^\s*using\s+([\w.]+)",
98
+ "go": r'^\s*import\s+(?:\w+\s+)?["\']([^"\']+)["\']',
99
+ "rust": r"^\s*use\s+([\w:{},*]+)\s*;",
100
+ "c": r'^\s*#\s*include\s+[<"]([^>"]+)[>"]',
101
+ "cpp": r'^\s*#\s*include\s+[<"]([^>"]+)[>"]',
102
+ "ruby": r"""^\s*require(?:_relative)?\s*\(?\s*['"]([^'"]+)['"]""",
103
+ "php": r"^\s*use\s+([\w\\]+)",
104
+ "swift": r"^\s*import\s+([\w.]+)",
105
+ }
106
+
107
+ def analyze(self, path: str, source: str) -> Analysis:
108
+ symbols = tuple(sorted(set(self.symbol_pattern.findall(source))))
109
+ imports_found = set(self.import_pattern.findall(source))
110
+ extra_pattern = self.additional_import_patterns.get(self.language)
111
+ if extra_pattern:
112
+ imports_found.update(re.findall(extra_pattern, source, re.MULTILINE))
113
+ imports = tuple(sorted(imports_found))
114
+ return Analysis(symbols=symbols, imports=imports)
115
+
116
+
117
+ class TreeSitterAnalyzer(LanguageAnalyzer):
118
+ """Use bundled Tree-sitter grammars when the optional parser extra is installed."""
119
+
120
+ _definitions = {
121
+ "class_declaration", "class_definition", "class_specifier",
122
+ "function_declaration", "function_definition", "function_item",
123
+ "method_declaration", "method_definition", "method_item",
124
+ "interface_declaration", "trait_item", "struct_item", "enum_item",
125
+ "type_declaration", "type_alias_declaration", "record_declaration",
126
+ }
127
+ _calls = {"call_expression", "call", "method_invocation"}
128
+
129
+ def __init__(self, language: str):
130
+ self.language = language
131
+
132
+ def analyze(self, path: str, source: str) -> Analysis:
133
+ try:
134
+ get_parser = importlib.import_module("tree_sitter_language_pack").get_parser
135
+ tree = get_parser(self.language).parse(source.encode("utf-8"))
136
+ except (ImportError, LookupError, RuntimeError, TypeError, ValueError):
137
+ return RegexAnalyzer(self.language).analyze(path, source)
138
+
139
+ symbols: set[str] = set()
140
+ calls: set[str] = set()
141
+ source_bytes = source.encode("utf-8")
142
+ pending = [tree.root_node]
143
+ while pending:
144
+ node = pending.pop()
145
+ pending.extend(reversed(node.children))
146
+ if node.type in self._definitions:
147
+ name = node.child_by_field_name("name")
148
+ if name is None:
149
+ name = next(
150
+ (
151
+ child for child in node.children
152
+ if child.type in {
153
+ "identifier", "type_identifier", "field_identifier",
154
+ }
155
+ ),
156
+ None,
157
+ )
158
+ if name is not None:
159
+ symbols.add(source_bytes[name.start_byte:name.end_byte].decode("utf-8"))
160
+ if node.type in self._calls:
161
+ function = node.child_by_field_name("function") or node.child_by_field_name("name")
162
+ if function is None and node.children:
163
+ function = node.children[0]
164
+ if function is not None:
165
+ call = source_bytes[function.start_byte:function.end_byte].decode("utf-8")
166
+ if call and len(call) <= 200:
167
+ calls.add(call)
168
+
169
+ regex_analysis = RegexAnalyzer(self.language).analyze(path, source)
170
+ return Analysis(
171
+ symbols=tuple(sorted(symbols)) or regex_analysis.symbols,
172
+ imports=regex_analysis.imports,
173
+ calls=tuple(sorted(calls)),
174
+ )
175
+
176
+
177
+ ANALYZERS: dict[str, LanguageAnalyzer] = {
178
+ "python": PythonAnalyzer(),
179
+ }
180
+
181
+
182
+ def analyze(path: str, source: str, language: str | None = None) -> Analysis:
183
+ language = language or "unknown"
184
+ analyzer = ANALYZERS.get(
185
+ language,
186
+ TreeSitterAnalyzer(language) if language != "unknown" else RegexAnalyzer(language),
187
+ )
188
+ return analyzer.analyze(path, source)
189
+
190
+
191
+ def module_candidates(
192
+ import_name: str, source_path: str, paths: set[str] | None = None
193
+ ) -> set[str]:
194
+ """Resolve an import to known source files without assuming one package layout."""
195
+ if import_name.startswith(("./", "../")):
196
+ module = posixpath.normpath(
197
+ posixpath.join(posixpath.dirname(source_path), import_name)
198
+ )
199
+ candidates = {
200
+ f"{module}{suffix}"
201
+ for suffix in (
202
+ ".py", ".js", ".jsx", ".ts", ".tsx", ".java", ".go", ".rs",
203
+ ".c", ".h", ".cpp", ".hpp",
204
+ )
205
+ }
206
+ candidates.update(f"{module}/index{suffix}" for suffix in (".js", ".ts", ".tsx"))
207
+ return candidates & paths if paths is not None else candidates
208
+ relative_level = len(import_name) - len(import_name.lstrip("."))
209
+ module = import_name.lstrip(".").replace(".", "/")
210
+ if import_name.startswith("."):
211
+ parent = Path(source_path).parent
212
+ if relative_level > 1 and len(parent.parents) >= relative_level - 1:
213
+ parent = parent.parents[relative_level - 2]
214
+ module = (parent / module).as_posix()
215
+ candidates = {
216
+ f"{module}{suffix}"
217
+ for suffix in (".py", ".js", ".jsx", ".ts", ".tsx", ".java", ".go", ".rs")
218
+ }
219
+ candidates.update(f"{module}/__init__.py" for _ in (0,))
220
+ if paths is None:
221
+ return candidates
222
+ candidates.update(
223
+ path
224
+ for path in paths
225
+ if any(path.endswith("/" + module + suffix) for suffix in (
226
+ ".py", ".js", ".jsx", ".ts", ".tsx", ".java", ".go", ".rs"
227
+ ))
228
+ )
229
+ return candidates & paths
Binary file
@@ -0,0 +1,227 @@
1
+ """Reproducible synthetic and temporary-copy repository indexing benchmarks."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import shutil
8
+ import subprocess
9
+ import tempfile
10
+ import time
11
+ from pathlib import Path
12
+
13
+ from .core import (
14
+ DEFAULT_EMBEDDING_MODEL,
15
+ AgentRamenError,
16
+ LANGUAGES,
17
+ _terms,
18
+ context_for,
19
+ connect,
20
+ index_repository,
21
+ )
22
+
23
+
24
+ def _git(root: Path, *args: str) -> str:
25
+ return subprocess.run(
26
+ ["git", "-C", str(root), *args],
27
+ check=True,
28
+ stdout=subprocess.PIPE,
29
+ stderr=subprocess.PIPE,
30
+ text=True,
31
+ encoding="utf-8",
32
+ errors="replace",
33
+ ).stdout
34
+
35
+
36
+ def _setup_git(root: Path) -> None:
37
+ subprocess.run(["git", "init", "-q", str(root)], check=True)
38
+ _git(root, "config", "user.email", "benchmark@example.invalid")
39
+ _git(root, "config", "user.name", "agentRamen Benchmark")
40
+
41
+
42
+ def _measure(
43
+ root: Path, files: int, task: str, target: Path, semantic: bool = False
44
+ ) -> dict[str, object]:
45
+ if semantic:
46
+ (root / ".agentramen.yml").write_text(
47
+ "semantic:\n enabled: true\n"
48
+ f" model: {DEFAULT_EMBEDDING_MODEL}\n",
49
+ encoding="utf-8",
50
+ )
51
+ started = time.perf_counter()
52
+ initial = index_repository(root)
53
+ initial_seconds = time.perf_counter() - started
54
+ target.write_text(
55
+ target.read_text(encoding="utf-8") + "\n# benchmark incremental edit\n",
56
+ encoding="utf-8",
57
+ )
58
+ started = time.perf_counter()
59
+ incremental = index_repository(root)
60
+ incremental_seconds = time.perf_counter() - started
61
+ started = time.perf_counter()
62
+ context = context_for(root, task, 500)
63
+ retrieval_seconds = time.perf_counter() - started
64
+ terms = _terms(task)
65
+ conn = connect(root)
66
+ placeholders = ",".join("?" for _ in terms)
67
+ candidates = (
68
+ conn.execute(
69
+ f"SELECT COUNT(DISTINCT path) FROM file_search_terms "
70
+ f"WHERE term IN ({placeholders})",
71
+ sorted(terms),
72
+ ).fetchone()[0]
73
+ if terms
74
+ else 0
75
+ )
76
+ database_bytes = sum(
77
+ path.stat().st_size
78
+ for path in (root / ".agentramen").glob("graph.db*")
79
+ if path.is_file()
80
+ )
81
+ conn.close()
82
+ return {
83
+ "files": files,
84
+ "task": task,
85
+ "semantic_enabled": semantic,
86
+ "initial_index_seconds": round(initial_seconds, 6),
87
+ "incremental_index_seconds": round(incremental_seconds, 6),
88
+ "context_retrieval_seconds": round(retrieval_seconds, 6),
89
+ "initial": initial,
90
+ "incremental": incremental,
91
+ "lexical_candidates": candidates,
92
+ "returned_files": len(context["files"]),
93
+ "context_estimated_tokens": context["estimated_tokens"],
94
+ "database_bytes": database_bytes,
95
+ "note": "Measured locally for this run; not a cross-repository performance claim.",
96
+ }
97
+
98
+
99
+ def run_benchmark(file_count: int, semantic: bool = False) -> dict[str, object]:
100
+ if file_count < 1:
101
+ raise ValueError("File count must be positive.")
102
+ with tempfile.TemporaryDirectory(prefix="agentramen-benchmark-") as directory:
103
+ root = Path(directory)
104
+ _setup_git(root)
105
+ source = root / "src"
106
+ source.mkdir()
107
+ for index in range(file_count):
108
+ identifier = 1000 + index
109
+ (source / f"service_{identifier}.py").write_text(
110
+ f"class Service{identifier}:\n"
111
+ f" def run(self):\n return {index}\n",
112
+ encoding="utf-8",
113
+ )
114
+ _git(root, "add", "src")
115
+ _git(root, "commit", "-qm", "benchmark fixture")
116
+ identifier = 1000 + file_count - 1
117
+ target = source / f"service_{identifier}.py"
118
+ return _measure(root, file_count, str(identifier), target, semantic)
119
+
120
+
121
+ def run_repository_benchmark(source_root: Path, semantic: bool = False) -> dict[str, object]:
122
+ source_root = source_root.resolve()
123
+ tracked = [path for path in _git(source_root, "ls-files", "-z").split("\0") if path]
124
+ if not tracked:
125
+ raise ValueError(f"No tracked files found in {source_root}.")
126
+ with tempfile.TemporaryDirectory(prefix="agentramen-repository-benchmark-") as directory:
127
+ root = Path(directory) / "repository"
128
+ root.mkdir()
129
+ copied = []
130
+ for relative in tracked:
131
+ source = source_root / relative
132
+ if not source.is_file() or source.is_symlink():
133
+ continue
134
+ destination = root / relative
135
+ destination.parent.mkdir(parents=True, exist_ok=True)
136
+ shutil.copy2(source, destination)
137
+ copied.append(relative)
138
+ eligible = [
139
+ relative for relative in copied
140
+ if Path(relative).suffix.lower() in LANGUAGES
141
+ and (root / relative).stat().st_size <= 2_000_000
142
+ ]
143
+ if not eligible:
144
+ raise ValueError(f"No supported source files found in {source_root}.")
145
+ target_relative = eligible[0]
146
+ target = root / target_relative
147
+ task = Path(target_relative).stem.strip("_").replace("_", " ")
148
+ _setup_git(root)
149
+ _git(root, "add", "-A")
150
+ _git(root, "commit", "-qm", "repository benchmark fixture")
151
+ return _measure(root, len(eligible), task, target, semantic)
152
+
153
+
154
+ def run_comparisons(
155
+ file_counts: list[int],
156
+ repository: Path | None,
157
+ semantic_modes: tuple[bool, ...],
158
+ ) -> list[dict[str, object]]:
159
+ results = []
160
+ semantic_error: str | None = None
161
+ for semantic in semantic_modes:
162
+ scenarios = [
163
+ (f"synthetic-{count}", lambda count=count: run_benchmark(count, semantic))
164
+ for count in file_counts
165
+ ]
166
+ if repository:
167
+ scenarios.append(
168
+ (
169
+ f"repository-{repository}",
170
+ lambda: run_repository_benchmark(repository, semantic),
171
+ )
172
+ )
173
+ for name, run in scenarios:
174
+ if semantic and semantic_error:
175
+ results.append(
176
+ {
177
+ "scenario": name,
178
+ "semantic_enabled": True,
179
+ "error": f"Skipped after semantic setup failed: {semantic_error}",
180
+ }
181
+ )
182
+ continue
183
+ try:
184
+ results.append(run())
185
+ except (AgentRamenError, OSError, ValueError, subprocess.CalledProcessError) as exc:
186
+ if len(semantic_modes) == 1:
187
+ raise
188
+ if semantic:
189
+ semantic_error = str(exc)
190
+ results.append(
191
+ {
192
+ "scenario": name,
193
+ "semantic_enabled": semantic,
194
+ "error": str(exc),
195
+ }
196
+ )
197
+ return results
198
+
199
+
200
+ def main(argv: list[str] | None = None) -> int:
201
+ parser = argparse.ArgumentParser(description="Measure agentRamen indexing and context retrieval")
202
+ parser.add_argument("--files", nargs="+", type=int, default=[10, 1000])
203
+ parser.add_argument(
204
+ "--repository",
205
+ type=Path,
206
+ help="Also benchmark a copy of a Git repository's tracked working-tree files",
207
+ )
208
+ parser.add_argument(
209
+ "--semantic-mode",
210
+ choices=("off", "on", "both"),
211
+ default="off",
212
+ help="Benchmark semantic retrieval off, on, or in both modes",
213
+ )
214
+ args = parser.parse_args(argv)
215
+ try:
216
+ modes = (False, True) if args.semantic_mode == "both" else (
217
+ args.semantic_mode == "on",
218
+ )
219
+ results = run_comparisons(args.files, args.repository, modes)
220
+ except (OSError, ValueError, subprocess.CalledProcessError) as exc:
221
+ parser.exit(1, f"agentramen-benchmark: {exc}\n")
222
+ print(json.dumps(results, indent=2))
223
+ return 0
224
+
225
+
226
+ if __name__ == "__main__":
227
+ raise SystemExit(main())