agent-coderag 1.4.1__tar.gz → 1.5.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agent_coderag-1.4.1/agent_coderag.egg-info → agent_coderag-1.5.1}/PKG-INFO +2 -1
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/README.md +1 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1/agent_coderag.egg-info}/PKG-INFO +2 -1
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/agent_coderag.egg-info/SOURCES.txt +7 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/api/client.py +12 -2
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/core/interfaces.py +7 -1
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/entry/args.py +11 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/entry/cli.py +9 -2
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/intelligence/distiller.py +1 -0
- agent_coderag-1.5.1/code_rag/parsers/multi_parser.py +46 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/parsers/tree_sitter.py +36 -19
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/paths.py +5 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/indexing.py +19 -2
- agent_coderag-1.5.1/code_rag/services/path_migration.py +121 -0
- agent_coderag-1.5.1/code_rag/services/search.py +20 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/sync.py +65 -11
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/storage/duckdb_impl.py +106 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/pyproject.toml +1 -1
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_cli_startup.py +7 -0
- agent_coderag-1.5.1/tests/test_gitignore_root.py +67 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_interfaces.py +7 -1
- agent_coderag-1.5.1/tests/test_migration_cli_e2e.py +141 -0
- agent_coderag-1.5.1/tests/test_migration_coderag.py +246 -0
- agent_coderag-1.5.1/tests/test_path_migration.py +565 -0
- agent_coderag-1.5.1/tests/test_relative_sync_paths.py +42 -0
- agent_coderag-1.5.1/tests/test_search_relative_paths.py +111 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_tree_sitter_parser.py +52 -0
- agent_coderag-1.4.1/code_rag/parsers/multi_parser.py +0 -31
- agent_coderag-1.4.1/code_rag/services/search.py +0 -9
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/LICENSE +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/NOTICE +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/agent_coderag.egg-info/dependency_links.txt +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/agent_coderag.egg-info/entry_points.txt +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/agent_coderag.egg-info/requires.txt +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/agent_coderag.egg-info/top_level.txt +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/__init__.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/api/__init__.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/api/models.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/core/__init__.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/core/constants.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/core/error_codes.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/core/exceptions.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/core/models.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/core/utils.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/__init__.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/dependency.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/java_discovery.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/manager.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/__init__.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/base.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/csharp.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/go.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/java.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/javascript.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/python.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/rust.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/entry/__init__.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/intelligence/__init__.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/intelligence/embedder.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/intelligence/factory.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/intelligence/openai_embedder.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/parsers/__init__.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/parsers/languages.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/__init__.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/config.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/dependencies.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/discovery_api.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/factory.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/setup.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/storage/__init__.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/storage/db_connection.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/setup.cfg +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_api_client.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_api_lazy_db.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_api_models.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_api_rebuild_wipe.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_cli.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_cli_detailed.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_cli_embeddings.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_cli_json_parity.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_code_rag_simple.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_coderag_lifetime.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_csharp_discovery_detailed.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_db_connection.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_db_path.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_dependency_discovery.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_discovery_manager_detailed.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_discovery_providers_extra.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_discovery_python.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_distiller.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_distiller_config_embedding.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_distiller_extra.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_embedder.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_embedder_detailed.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_embedder_factory.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_embedder_interface.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_embedder_stubs.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_error_codes.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_factory_async.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_indexing_detailed.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_indexing_embeddings.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_indexing_extra.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_js_discovery_detailed.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_languages.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_local_onnx_embedder.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_models.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_multi_parser.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_openai_embedder.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_readme_library_usage.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_readme_lifetime_storage.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_readme_offline_embeddings.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_rust_discovery_detailed.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_storage_detailed.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_storage_embeddings.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_sync_worker_errors.py +0 -0
- {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_utils_detailed.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: agent-coderag
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.5.1
|
|
4
4
|
Summary: Lightweight semantic code search and distillation utility for AI coding agents. It solves the API knowledge gap via real-time local signature extraction and intent analysis without PyTorch. Optimized for token efficiency, it compresses codebase context into compact semantic summaries stored in a local DuckDB vector similarity index.
|
|
5
5
|
Author-email: Igor Boloban <naranor@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -190,6 +190,7 @@ async def main():
|
|
|
190
190
|
- **Sidecars:** DuckDB may write WAL sidecars (e.g. `.coderag.db.wal`) beside the index during writes; locks should not persist after an operation finishes.
|
|
191
191
|
- **Connect timeout:** `connect_timeout_seconds=5` (CLI `--connect-timeout`) waits on file locks, then raises `StorageBusyError` (`ErrorCode.STORAGE_BUSY`). Pass `0` for a single attempt.
|
|
192
192
|
- **Read-only search:** `search` (and `api` when storage is needed) opens read-only. A missing **index file** is an error — use `sync`/`rebuild` to create it. An index file that exists but has no embeddings table (e.g. opened/written without a completed vector sync) raises `StorageError` with `ErrorCode.EMBEDDINGS_MISSING` — run `sync` (library: `CodeRAG.sync`) before search. With `--json`, success is a hit array; errors are `{"status":"error","message":...}` and include `"code"` when the exception carries an `ErrorCode`.
|
|
193
|
+
- **Paths:** `sync` stores paths relative to `root` (`src/a.py`). The first sync rewrites an older absolute index when the file is still under `root` or its unit hashes match a file in the tree. Search returns an absolute path under the current root. `search --relative-paths`, config `relative_paths: true`, or `CodeRAG(relative_paths=True)` returns the stored relative path. Until that sync runs, search returns the absolute path stored in the index. Ignore rules use the same project-relative path, so a `.worktrees/<name>` checkout is indexed when that directory is the root.
|
|
193
194
|
- **Lifetime:** embedder/parser/distiller stay warm; DuckDB opens per operation and closes afterward. One `CodeRAG` instance serializes overlapping ops. `config()` with embedding flags / `--clear-embedding` closes the process embedder so the next op rebuilds it; distill-only `config` refreshes Distiller and keeps the embedder.
|
|
194
195
|
- **`api()` without DB:** providers that do not need the index (e.g. Python) skip DuckDB entirely; Java uses a short read-only open for JAR cache lookup.
|
|
195
196
|
- **Errors:** catch `CodeRAGError` and inspect `.code` — `STORAGE_BUSY`, `STORAGE_CORRUPT`, `EMBEDDING_MISMATCH`, `EMBEDDINGS_MISSING` (`from code_rag import ErrorCode`).
|
|
@@ -144,6 +144,7 @@ async def main():
|
|
|
144
144
|
- **Sidecars:** DuckDB may write WAL sidecars (e.g. `.coderag.db.wal`) beside the index during writes; locks should not persist after an operation finishes.
|
|
145
145
|
- **Connect timeout:** `connect_timeout_seconds=5` (CLI `--connect-timeout`) waits on file locks, then raises `StorageBusyError` (`ErrorCode.STORAGE_BUSY`). Pass `0` for a single attempt.
|
|
146
146
|
- **Read-only search:** `search` (and `api` when storage is needed) opens read-only. A missing **index file** is an error — use `sync`/`rebuild` to create it. An index file that exists but has no embeddings table (e.g. opened/written without a completed vector sync) raises `StorageError` with `ErrorCode.EMBEDDINGS_MISSING` — run `sync` (library: `CodeRAG.sync`) before search. With `--json`, success is a hit array; errors are `{"status":"error","message":...}` and include `"code"` when the exception carries an `ErrorCode`.
|
|
147
|
+
- **Paths:** `sync` stores paths relative to `root` (`src/a.py`). The first sync rewrites an older absolute index when the file is still under `root` or its unit hashes match a file in the tree. Search returns an absolute path under the current root. `search --relative-paths`, config `relative_paths: true`, or `CodeRAG(relative_paths=True)` returns the stored relative path. Until that sync runs, search returns the absolute path stored in the index. Ignore rules use the same project-relative path, so a `.worktrees/<name>` checkout is indexed when that directory is the root.
|
|
147
148
|
- **Lifetime:** embedder/parser/distiller stay warm; DuckDB opens per operation and closes afterward. One `CodeRAG` instance serializes overlapping ops. `config()` with embedding flags / `--clear-embedding` closes the process embedder so the next op rebuilds it; distill-only `config` refreshes Distiller and keeps the embedder.
|
|
148
149
|
- **`api()` without DB:** providers that do not need the index (e.g. Python) skip DuckDB entirely; Java uses a short read-only open for JAR cache lookup.
|
|
149
150
|
- **Errors:** catch `CodeRAGError` and inspect `.code` — `STORAGE_BUSY`, `STORAGE_CORRUPT`, `EMBEDDING_MISMATCH`, `EMBEDDINGS_MISSING` (`from code_rag import ErrorCode`).
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: agent-coderag
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.5.1
|
|
4
4
|
Summary: Lightweight semantic code search and distillation utility for AI coding agents. It solves the API knowledge gap via real-time local signature extraction and intent analysis without PyTorch. Optimized for token efficiency, it compresses codebase context into compact semantic summaries stored in a local DuckDB vector similarity index.
|
|
5
5
|
Author-email: Igor Boloban <naranor@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -190,6 +190,7 @@ async def main():
|
|
|
190
190
|
- **Sidecars:** DuckDB may write WAL sidecars (e.g. `.coderag.db.wal`) beside the index during writes; locks should not persist after an operation finishes.
|
|
191
191
|
- **Connect timeout:** `connect_timeout_seconds=5` (CLI `--connect-timeout`) waits on file locks, then raises `StorageBusyError` (`ErrorCode.STORAGE_BUSY`). Pass `0` for a single attempt.
|
|
192
192
|
- **Read-only search:** `search` (and `api` when storage is needed) opens read-only. A missing **index file** is an error — use `sync`/`rebuild` to create it. An index file that exists but has no embeddings table (e.g. opened/written without a completed vector sync) raises `StorageError` with `ErrorCode.EMBEDDINGS_MISSING` — run `sync` (library: `CodeRAG.sync`) before search. With `--json`, success is a hit array; errors are `{"status":"error","message":...}` and include `"code"` when the exception carries an `ErrorCode`.
|
|
193
|
+
- **Paths:** `sync` stores paths relative to `root` (`src/a.py`). The first sync rewrites an older absolute index when the file is still under `root` or its unit hashes match a file in the tree. Search returns an absolute path under the current root. `search --relative-paths`, config `relative_paths: true`, or `CodeRAG(relative_paths=True)` returns the stored relative path. Until that sync runs, search returns the absolute path stored in the index. Ignore rules use the same project-relative path, so a `.worktrees/<name>` checkout is indexed when that directory is the root.
|
|
193
194
|
- **Lifetime:** embedder/parser/distiller stay warm; DuckDB opens per operation and closes afterward. One `CodeRAG` instance serializes overlapping ops. `config()` with embedding flags / `--clear-embedding` closes the process embedder so the next op rebuilds it; distill-only `config` refreshes Distiller and keeps the embedder.
|
|
194
195
|
- **`api()` without DB:** providers that do not need the index (e.g. Python) skip DuckDB entirely; Java uses a short read-only open for JAR cache lookup.
|
|
195
196
|
- **Errors:** catch `CodeRAGError` and inspect `.code` — `STORAGE_BUSY`, `STORAGE_CORRUPT`, `EMBEDDING_MISMATCH`, `EMBEDDINGS_MISSING` (`from code_rag import ErrorCode`).
|
|
@@ -50,6 +50,7 @@ code_rag/services/dependencies.py
|
|
|
50
50
|
code_rag/services/discovery_api.py
|
|
51
51
|
code_rag/services/factory.py
|
|
52
52
|
code_rag/services/indexing.py
|
|
53
|
+
code_rag/services/path_migration.py
|
|
53
54
|
code_rag/services/search.py
|
|
54
55
|
code_rag/services/setup.py
|
|
55
56
|
code_rag/services/sync.py
|
|
@@ -84,6 +85,7 @@ tests/test_embedder_interface.py
|
|
|
84
85
|
tests/test_embedder_stubs.py
|
|
85
86
|
tests/test_error_codes.py
|
|
86
87
|
tests/test_factory_async.py
|
|
88
|
+
tests/test_gitignore_root.py
|
|
87
89
|
tests/test_indexing_detailed.py
|
|
88
90
|
tests/test_indexing_embeddings.py
|
|
89
91
|
tests/test_indexing_extra.py
|
|
@@ -91,13 +93,18 @@ tests/test_interfaces.py
|
|
|
91
93
|
tests/test_js_discovery_detailed.py
|
|
92
94
|
tests/test_languages.py
|
|
93
95
|
tests/test_local_onnx_embedder.py
|
|
96
|
+
tests/test_migration_cli_e2e.py
|
|
97
|
+
tests/test_migration_coderag.py
|
|
94
98
|
tests/test_models.py
|
|
95
99
|
tests/test_multi_parser.py
|
|
96
100
|
tests/test_openai_embedder.py
|
|
101
|
+
tests/test_path_migration.py
|
|
97
102
|
tests/test_readme_library_usage.py
|
|
98
103
|
tests/test_readme_lifetime_storage.py
|
|
99
104
|
tests/test_readme_offline_embeddings.py
|
|
105
|
+
tests/test_relative_sync_paths.py
|
|
100
106
|
tests/test_rust_discovery_detailed.py
|
|
107
|
+
tests/test_search_relative_paths.py
|
|
101
108
|
tests/test_storage_detailed.py
|
|
102
109
|
tests/test_storage_embeddings.py
|
|
103
110
|
tests/test_sync_worker_errors.py
|
|
@@ -19,7 +19,7 @@ from code_rag.services.config import (
|
|
|
19
19
|
)
|
|
20
20
|
from code_rag.services.discovery_api import run_api
|
|
21
21
|
from code_rag.services.factory import create_stack
|
|
22
|
-
from code_rag.services.search import run_search
|
|
22
|
+
from code_rag.services.search import format_unit_path, run_search
|
|
23
23
|
from code_rag.services.setup import run_setup
|
|
24
24
|
from code_rag.services.indexing import IndexStack
|
|
25
25
|
from code_rag.services.sync import SyncOptions, run_rebuild, run_sync
|
|
@@ -40,12 +40,17 @@ class CodeRAG: # pylint: disable=too-many-instance-attributes
|
|
|
40
40
|
root: Optional[Union[str, Path]] = None,
|
|
41
41
|
allow_build_execution: bool = False,
|
|
42
42
|
connect_timeout_seconds: float = DEFAULT_CONNECT_TIMEOUT_SECONDS,
|
|
43
|
+
relative_paths: bool | None = None,
|
|
43
44
|
):
|
|
44
45
|
self._onnx = onnx
|
|
45
46
|
self._root = Path(root) if root is not None else Path.cwd()
|
|
46
47
|
self._allow_build_execution = allow_build_execution
|
|
47
48
|
self._connect_timeout_seconds = connect_timeout_seconds
|
|
48
49
|
self._db_path = resolve_db_path(db, root=self._root)
|
|
50
|
+
if relative_paths is None:
|
|
51
|
+
self._relative_paths = bool(DistillerConfig.load().relative_paths)
|
|
52
|
+
else:
|
|
53
|
+
self._relative_paths = relative_paths
|
|
49
54
|
self._embedder: Optional[IEmbedder] = None
|
|
50
55
|
self._parser: Optional[IParser] = None
|
|
51
56
|
self._distiller: Optional[IIntelligence] = None
|
|
@@ -199,7 +204,12 @@ class CodeRAG: # pylint: disable=too-many-instance-attributes
|
|
|
199
204
|
|
|
200
205
|
async def search(self, query: str, *, limit: int = 5) -> list[KnowledgeUnit]:
|
|
201
206
|
async with self._with_storage(AccessMode.READ_ONLY) as storage:
|
|
202
|
-
|
|
207
|
+
units = await run_search(storage, query, limit=limit)
|
|
208
|
+
for unit in units:
|
|
209
|
+
unit.path = format_unit_path(
|
|
210
|
+
unit.path, root=self._root, relative_paths=self._relative_paths
|
|
211
|
+
)
|
|
212
|
+
return units
|
|
203
213
|
|
|
204
214
|
async def _api_java(
|
|
205
215
|
self,
|
|
@@ -7,7 +7,13 @@ class IParser(ABC):
|
|
|
7
7
|
"""Interface for extracting structure from code."""
|
|
8
8
|
|
|
9
9
|
@abstractmethod
|
|
10
|
-
async def distill_file(
|
|
10
|
+
async def distill_file(
|
|
11
|
+
self,
|
|
12
|
+
file_path: str,
|
|
13
|
+
*,
|
|
14
|
+
stored_path: str | None = None,
|
|
15
|
+
raise_on_failure: bool = False,
|
|
16
|
+
) -> List[KnowledgeUnit]:
|
|
11
17
|
pass # pragma: no cover
|
|
12
18
|
|
|
13
19
|
|
|
@@ -50,6 +50,13 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
50
50
|
search = subparsers.add_parser("search", help="Semantic search.")
|
|
51
51
|
search.add_argument("query", help="Natural language query.")
|
|
52
52
|
search.add_argument("--limit", type=int, default=5, help="Result limit.")
|
|
53
|
+
search.add_argument(
|
|
54
|
+
"--relative-paths",
|
|
55
|
+
dest="relative_paths",
|
|
56
|
+
action="store_true",
|
|
57
|
+
default=None,
|
|
58
|
+
help="Print paths relative to the project root. Default: absolute paths under the current root.",
|
|
59
|
+
)
|
|
53
60
|
|
|
54
61
|
api = subparsers.add_parser("api", help="Discover library API.")
|
|
55
62
|
api.add_argument("library", help="Library name (e.g., pydantic).")
|
|
@@ -111,3 +118,7 @@ def main() -> None:
|
|
|
111
118
|
except Exception as exc:
|
|
112
119
|
logging.getLogger(__name__).error("Unexpected error: %s", exc)
|
|
113
120
|
sys.exit(1)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
if __name__ == "__main__":
|
|
124
|
+
main()
|
|
@@ -46,6 +46,7 @@ __all__ = [
|
|
|
46
46
|
|
|
47
47
|
|
|
48
48
|
def _coderag_from_args(args) -> CodeRAG:
|
|
49
|
+
relative_paths = getattr(args, "relative_paths", None)
|
|
49
50
|
return CodeRAG(
|
|
50
51
|
db=args.db,
|
|
51
52
|
onnx=getattr(args, "onnx", None),
|
|
@@ -54,6 +55,7 @@ def _coderag_from_args(args) -> CodeRAG:
|
|
|
54
55
|
getattr(args, "connect_timeout", DEFAULT_CONNECT_TIMEOUT_SECONDS)
|
|
55
56
|
),
|
|
56
57
|
allow_build_execution=bool(getattr(args, "allow_build_execution", False)),
|
|
58
|
+
relative_paths=True if relative_paths else None,
|
|
57
59
|
)
|
|
58
60
|
|
|
59
61
|
|
|
@@ -70,9 +72,14 @@ def load_ignore_patterns() -> pathspec.PathSpec:
|
|
|
70
72
|
return sync_service.load_ignore_patterns(Path.cwd())
|
|
71
73
|
|
|
72
74
|
|
|
73
|
-
def should_index(
|
|
75
|
+
def should_index(
|
|
76
|
+
path: Path,
|
|
77
|
+
ignore_spec: Optional[pathspec.PathSpec] = None,
|
|
78
|
+
*,
|
|
79
|
+
root: Optional[Path] = None,
|
|
80
|
+
) -> bool:
|
|
74
81
|
"""Filters files that should NOT be indexed."""
|
|
75
|
-
return sync_service.should_index(path, ignore_spec)
|
|
82
|
+
return sync_service.should_index(path, ignore_spec, root=root)
|
|
76
83
|
|
|
77
84
|
|
|
78
85
|
def _emit_sync_outcome(result: SyncResult, *, json_mode: bool, label: str) -> None:
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import logging
|
|
3
|
+
from typing import List
|
|
4
|
+
from ..core.interfaces import IParser
|
|
5
|
+
from ..core.models import KnowledgeUnit
|
|
6
|
+
from .tree_sitter import TreeSitterParser
|
|
7
|
+
from .languages import EXTENSION_TO_LANGUAGE
|
|
8
|
+
|
|
9
|
+
logger = logging.getLogger(__name__)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class MultiParser(IParser):
|
|
13
|
+
"""
|
|
14
|
+
Delegates parsing to TreeSitterParser for all supported languages.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
def __init__(self) -> None:
|
|
18
|
+
self.tree_sitter_parser = TreeSitterParser()
|
|
19
|
+
|
|
20
|
+
async def distill_file(
|
|
21
|
+
self,
|
|
22
|
+
file_path: str,
|
|
23
|
+
*,
|
|
24
|
+
stored_path: str | None = None,
|
|
25
|
+
raise_on_failure: bool = False,
|
|
26
|
+
) -> List[KnowledgeUnit]:
|
|
27
|
+
_, ext = os.path.splitext(file_path)
|
|
28
|
+
ext = ext.lower()
|
|
29
|
+
|
|
30
|
+
if ext in EXTENSION_TO_LANGUAGE:
|
|
31
|
+
if stored_path is None and not raise_on_failure:
|
|
32
|
+
return await self.tree_sitter_parser.distill_file(file_path)
|
|
33
|
+
if stored_path is not None and raise_on_failure:
|
|
34
|
+
return await self.tree_sitter_parser.distill_file(
|
|
35
|
+
file_path, stored_path=stored_path, raise_on_failure=True
|
|
36
|
+
)
|
|
37
|
+
if stored_path is not None:
|
|
38
|
+
return await self.tree_sitter_parser.distill_file(
|
|
39
|
+
file_path, stored_path=stored_path
|
|
40
|
+
)
|
|
41
|
+
return await self.tree_sitter_parser.distill_file(
|
|
42
|
+
file_path, raise_on_failure=True
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
logger.debug("No parser for extension %s", ext)
|
|
46
|
+
return []
|
|
@@ -45,10 +45,14 @@ class TreeSitterParser(IParser):
|
|
|
45
45
|
f"[MISSING DEPENDENCY] Please install: pip install {lang_config.package}"
|
|
46
46
|
) from e
|
|
47
47
|
|
|
48
|
-
async def distill_file(
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
48
|
+
async def distill_file(
|
|
49
|
+
self,
|
|
50
|
+
file_path: str,
|
|
51
|
+
*,
|
|
52
|
+
stored_path: str | None = None,
|
|
53
|
+
raise_on_failure: bool = False,
|
|
54
|
+
) -> List[KnowledgeUnit]:
|
|
55
|
+
identity = file_path if stored_path is None else stored_path
|
|
52
56
|
ext = Path(file_path).suffix.lower()
|
|
53
57
|
lang_config = next((cfg for cfg in LANGUAGES if ext in cfg.extensions), None)
|
|
54
58
|
|
|
@@ -67,36 +71,52 @@ class TreeSitterParser(IParser):
|
|
|
67
71
|
ctx: Dict[str, Any] = {
|
|
68
72
|
"source": source,
|
|
69
73
|
"config": lang_config,
|
|
70
|
-
"file_path":
|
|
74
|
+
"file_path": identity,
|
|
71
75
|
"units": units,
|
|
72
76
|
"scope": None,
|
|
77
|
+
"name_counts": {},
|
|
73
78
|
}
|
|
74
79
|
self._recursive_distill(tree.root_node, ctx)
|
|
75
80
|
return units
|
|
76
81
|
except GrammarNotFoundError:
|
|
77
|
-
# Re-raise grammar errors so callers can handle them (e.g. CLI setup suggestion)
|
|
78
82
|
raise
|
|
79
83
|
except Exception as e:
|
|
80
84
|
logger.error("Failed to parse %s: %s", file_path, e)
|
|
85
|
+
if raise_on_failure:
|
|
86
|
+
raise
|
|
81
87
|
return []
|
|
82
88
|
|
|
83
89
|
def _recursive_distill(self, node: Node, ctx: Dict[str, Any]) -> None:
|
|
84
90
|
config = ctx["config"]
|
|
85
|
-
|
|
91
|
+
child_scope = ctx["scope"]
|
|
92
|
+
child_counts = ctx["name_counts"]
|
|
86
93
|
|
|
87
94
|
if node.type in config.canonical_map:
|
|
88
|
-
self.
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
child_ctx = {**ctx, "scope": current_scope}
|
|
95
|
+
qname = self._qualified_name(node, ctx)
|
|
96
|
+
self._process_node(node, ctx, qname)
|
|
97
|
+
if config.canonical_map.get(node.type) in ("CLASS", "FUNCTION", "METHOD"):
|
|
98
|
+
child_scope = qname
|
|
99
|
+
child_counts = {}
|
|
100
|
+
|
|
101
|
+
child_ctx = {**ctx, "scope": child_scope, "name_counts": child_counts}
|
|
96
102
|
for child in node.children:
|
|
97
103
|
self._recursive_distill(child, child_ctx)
|
|
98
104
|
|
|
99
|
-
def
|
|
105
|
+
def _qualified_name(self, node: Node, ctx: Dict[str, Any]) -> str:
|
|
106
|
+
"""Qualified id segment. Enclosing functions are part of the scope.
|
|
107
|
+
|
|
108
|
+
A repeated name in the same scope gets ``#2``, ``#3``, … so two
|
|
109
|
+
different bodies cannot share one primary key.
|
|
110
|
+
"""
|
|
111
|
+
node_name = self._resolve_name(node)
|
|
112
|
+
counts: Dict[str, int] = ctx["name_counts"]
|
|
113
|
+
counts[node_name] = counts.get(node_name, 0) + 1
|
|
114
|
+
seen = counts[node_name]
|
|
115
|
+
label = node_name if seen == 1 else f"{node_name}#{seen}"
|
|
116
|
+
scope = ctx["scope"]
|
|
117
|
+
return f"{scope}.{label}" if scope else label
|
|
118
|
+
|
|
119
|
+
def _process_node(self, node: Node, ctx: Dict[str, Any], qname: str) -> None:
|
|
100
120
|
"""Processes a single node and adds it to the units list."""
|
|
101
121
|
node_name = self._resolve_name(node)
|
|
102
122
|
docstring = self._extract_docstring(node)
|
|
@@ -108,9 +128,6 @@ class TreeSitterParser(IParser):
|
|
|
108
128
|
|
|
109
129
|
kind = self._determine_kind(node, ctx["config"])
|
|
110
130
|
|
|
111
|
-
# Stable ID based on qualified name
|
|
112
|
-
scope = ctx["scope"]
|
|
113
|
-
qname = f"{scope}.{node_name}" if scope else node_name
|
|
114
131
|
unit_id = f"{ctx['file_path']}:{qname}"
|
|
115
132
|
|
|
116
133
|
unit = KnowledgeUnit(
|
|
@@ -13,6 +13,11 @@ def default_db_path(root: Path | None = None) -> Path:
|
|
|
13
13
|
return root_path / ".coderag.db"
|
|
14
14
|
|
|
15
15
|
|
|
16
|
+
def project_relative_posix(path: Path, root: Path) -> str:
|
|
17
|
+
"""Posix path of ``path`` relative to ``root``. Raises ValueError when outside ``root``."""
|
|
18
|
+
return path.resolve().relative_to(root.resolve()).as_posix()
|
|
19
|
+
|
|
20
|
+
|
|
16
21
|
def resolve_db_path(db: str | Path | None, *, root: Path | None = None) -> Path:
|
|
17
22
|
if db is not None:
|
|
18
23
|
path = Path(db)
|
|
@@ -7,12 +7,14 @@ import inspect
|
|
|
7
7
|
import logging
|
|
8
8
|
import sys
|
|
9
9
|
from dataclasses import dataclass
|
|
10
|
+
from pathlib import Path
|
|
10
11
|
from typing import List, Optional
|
|
11
12
|
|
|
12
13
|
from code_rag.core.constants import EMBEDDING_BATCH_SIZE, MAX_CONCURRENT_TASKS
|
|
13
14
|
from code_rag.core.exceptions import IntelligenceError, StorageError
|
|
14
15
|
from code_rag.core.interfaces import IIntelligence, IParser, IStorage
|
|
15
16
|
from code_rag.core.models import KnowledgeUnit
|
|
17
|
+
from code_rag.paths import project_relative_posix
|
|
16
18
|
|
|
17
19
|
logger = logging.getLogger(__name__)
|
|
18
20
|
|
|
@@ -24,6 +26,7 @@ class IndexStack:
|
|
|
24
26
|
storage: IStorage
|
|
25
27
|
parser: IParser
|
|
26
28
|
intelligence: IIntelligence
|
|
29
|
+
root: Optional[Path] = None
|
|
27
30
|
|
|
28
31
|
|
|
29
32
|
def _report_worker_failure(path: str, exc: BaseException) -> tuple[str, str]:
|
|
@@ -146,11 +149,19 @@ async def sync_file(
|
|
|
146
149
|
*,
|
|
147
150
|
force_distill: bool = False,
|
|
148
151
|
max_concurrency: int = MAX_CONCURRENT_TASKS,
|
|
152
|
+
stored_path: str | None = None,
|
|
149
153
|
) -> None:
|
|
150
154
|
"""Parse, distill, embed, and upsert one file into ``stack.storage``."""
|
|
151
155
|
await _reject_dirty_incremental(stack.storage)
|
|
152
156
|
semaphore = asyncio.Semaphore(max_concurrency)
|
|
153
|
-
|
|
157
|
+
if stored_path is None:
|
|
158
|
+
current_units = await stack.parser.distill_file(file_path)
|
|
159
|
+
identity = file_path
|
|
160
|
+
else:
|
|
161
|
+
current_units = await stack.parser.distill_file(
|
|
162
|
+
file_path, stored_path=stored_path
|
|
163
|
+
)
|
|
164
|
+
identity = stored_path
|
|
154
165
|
pending: list[KnowledgeUnit] = []
|
|
155
166
|
for unit in current_units:
|
|
156
167
|
await _process_unit(
|
|
@@ -162,7 +173,7 @@ async def sync_file(
|
|
|
162
173
|
)
|
|
163
174
|
await _embed_and_upsert(stack.storage, pending)
|
|
164
175
|
await stack.storage.delete_stale_units(
|
|
165
|
-
|
|
176
|
+
identity, [unit.id for unit in current_units]
|
|
166
177
|
)
|
|
167
178
|
|
|
168
179
|
|
|
@@ -194,11 +205,17 @@ async def sync_project(
|
|
|
194
205
|
except asyncio.QueueEmpty:
|
|
195
206
|
return failures
|
|
196
207
|
try:
|
|
208
|
+
stored_path = (
|
|
209
|
+
None
|
|
210
|
+
if stack.root is None
|
|
211
|
+
else project_relative_posix(Path(path), stack.root)
|
|
212
|
+
)
|
|
197
213
|
await sync_file(
|
|
198
214
|
stack,
|
|
199
215
|
path,
|
|
200
216
|
force_distill=force_distill,
|
|
201
217
|
max_concurrency=max_concurrency,
|
|
218
|
+
stored_path=stored_path,
|
|
202
219
|
)
|
|
203
220
|
except Exception as exc:
|
|
204
221
|
failures.append(_report_worker_failure(path, exc))
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
"""One-time rewrite of absolute index paths onto project-relative posix paths."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from collections import Counter
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from code_rag.core.models import KnowledgeUnit
|
|
10
|
+
from code_rag.parsers.tree_sitter import GrammarNotFoundError
|
|
11
|
+
from code_rag.paths import project_relative_posix
|
|
12
|
+
from code_rag.storage.duckdb_impl import DuckDBStorage
|
|
13
|
+
|
|
14
|
+
logger = logging.getLogger(__name__)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _groups(units: list[KnowledgeUnit]) -> dict[str, Counter[str]]:
|
|
18
|
+
groups: dict[str, Counter[str]] = {}
|
|
19
|
+
for unit in units:
|
|
20
|
+
if not Path(unit.path).is_absolute():
|
|
21
|
+
continue
|
|
22
|
+
groups.setdefault(unit.path, Counter())[unit.code_hash] += 1
|
|
23
|
+
return groups
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def claim_relative_paths(
|
|
27
|
+
groups: dict[str, Counter[str]],
|
|
28
|
+
root: Path,
|
|
29
|
+
files: list[tuple[Path, Counter[str]]],
|
|
30
|
+
) -> dict[str, str]:
|
|
31
|
+
mapping: dict[str, str] = {}
|
|
32
|
+
taken: set[str] = set()
|
|
33
|
+
pending: list[tuple[str, Counter[str]]] = []
|
|
34
|
+
for old, hashes in groups.items():
|
|
35
|
+
path = Path(old)
|
|
36
|
+
if path.is_absolute():
|
|
37
|
+
try:
|
|
38
|
+
rel = project_relative_posix(path, root)
|
|
39
|
+
except ValueError:
|
|
40
|
+
pending.append((old, hashes))
|
|
41
|
+
continue
|
|
42
|
+
if rel in taken:
|
|
43
|
+
continue
|
|
44
|
+
mapping[old] = rel
|
|
45
|
+
taken.add(rel)
|
|
46
|
+
continue
|
|
47
|
+
pending.append((old, hashes))
|
|
48
|
+
|
|
49
|
+
claimed: set[str] = set()
|
|
50
|
+
for file_path, hashes in files:
|
|
51
|
+
try:
|
|
52
|
+
rel = project_relative_posix(file_path, root)
|
|
53
|
+
except ValueError:
|
|
54
|
+
continue
|
|
55
|
+
if rel in taken:
|
|
56
|
+
continue
|
|
57
|
+
for old, group_hashes in pending:
|
|
58
|
+
if old in claimed or old in mapping:
|
|
59
|
+
continue
|
|
60
|
+
if group_hashes != hashes:
|
|
61
|
+
continue
|
|
62
|
+
mapping[old] = rel
|
|
63
|
+
taken.add(rel)
|
|
64
|
+
claimed.add(old)
|
|
65
|
+
break
|
|
66
|
+
return mapping
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _suffix(path: str) -> str:
|
|
70
|
+
return Path(path).suffix.lower()
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
async def migrate_absolute_paths(
|
|
74
|
+
storage, parser, root: Path, files: list[Path]
|
|
75
|
+
) -> bool:
|
|
76
|
+
if not isinstance(storage, DuckDBStorage):
|
|
77
|
+
return False
|
|
78
|
+
if await storage.paths_migration_done():
|
|
79
|
+
return False
|
|
80
|
+
groups = _groups(await storage.list_units())
|
|
81
|
+
if not groups:
|
|
82
|
+
await storage.commit_path_migration({})
|
|
83
|
+
return False
|
|
84
|
+
incomplete = False
|
|
85
|
+
skipped_suffixes: set[str] = set()
|
|
86
|
+
parsed: list[tuple[Path, Counter[str]]] = []
|
|
87
|
+
for path in files:
|
|
88
|
+
try:
|
|
89
|
+
units = await parser.distill_file(str(path), raise_on_failure=True)
|
|
90
|
+
except GrammarNotFoundError:
|
|
91
|
+
logger.warning("Skipping unparsed file during path migration: %s", path)
|
|
92
|
+
skipped_suffixes.add(_suffix(str(path)))
|
|
93
|
+
incomplete = True
|
|
94
|
+
continue
|
|
95
|
+
except Exception:
|
|
96
|
+
logger.exception("Path migration could not parse %s", path)
|
|
97
|
+
skipped_suffixes.add(_suffix(str(path)))
|
|
98
|
+
incomplete = True
|
|
99
|
+
continue
|
|
100
|
+
parsed.append((path, Counter(unit.code_hash for unit in units)))
|
|
101
|
+
mapping = claim_relative_paths(groups, root, parsed)
|
|
102
|
+
failed_commits: set[str] = set()
|
|
103
|
+
for old, new in mapping.items():
|
|
104
|
+
try:
|
|
105
|
+
await storage.commit_rewritten_path(old, new)
|
|
106
|
+
except Exception:
|
|
107
|
+
logger.exception("Path migration failed for %s", old)
|
|
108
|
+
failed_commits.add(old)
|
|
109
|
+
incomplete = True
|
|
110
|
+
orphans = [
|
|
111
|
+
old
|
|
112
|
+
for old in groups
|
|
113
|
+
if old not in mapping
|
|
114
|
+
and old not in failed_commits
|
|
115
|
+
and _suffix(old) not in skipped_suffixes
|
|
116
|
+
]
|
|
117
|
+
if orphans:
|
|
118
|
+
await storage.delete_absolute_paths(orphans)
|
|
119
|
+
if not incomplete:
|
|
120
|
+
await storage.mark_paths_migrated()
|
|
121
|
+
return True
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
|
|
3
|
+
from code_rag.core.interfaces import IStorage
|
|
4
|
+
from code_rag.core.models import KnowledgeUnit
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def format_unit_path(stored: str, *, root: Path, relative_paths: bool) -> str:
|
|
8
|
+
"""Format a stored index path for search output. Does not read or write storage."""
|
|
9
|
+
if Path(stored).is_absolute():
|
|
10
|
+
return stored
|
|
11
|
+
if relative_paths:
|
|
12
|
+
return stored
|
|
13
|
+
return str((root / stored).resolve())
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
async def run_search(
|
|
17
|
+
storage: IStorage, query: str, *, limit: int = 5
|
|
18
|
+
) -> list[KnowledgeUnit]:
|
|
19
|
+
"""Performs semantic search across indexed units."""
|
|
20
|
+
return await storage.search_units(query, limit=limit)
|