agent-coderag 1.4.1__tar.gz → 1.5.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. {agent_coderag-1.4.1/agent_coderag.egg-info → agent_coderag-1.5.1}/PKG-INFO +2 -1
  2. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/README.md +1 -0
  3. {agent_coderag-1.4.1 → agent_coderag-1.5.1/agent_coderag.egg-info}/PKG-INFO +2 -1
  4. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/agent_coderag.egg-info/SOURCES.txt +7 -0
  5. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/api/client.py +12 -2
  6. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/core/interfaces.py +7 -1
  7. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/entry/args.py +11 -0
  8. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/entry/cli.py +9 -2
  9. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/intelligence/distiller.py +1 -0
  10. agent_coderag-1.5.1/code_rag/parsers/multi_parser.py +46 -0
  11. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/parsers/tree_sitter.py +36 -19
  12. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/paths.py +5 -0
  13. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/indexing.py +19 -2
  14. agent_coderag-1.5.1/code_rag/services/path_migration.py +121 -0
  15. agent_coderag-1.5.1/code_rag/services/search.py +20 -0
  16. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/sync.py +65 -11
  17. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/storage/duckdb_impl.py +106 -0
  18. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/pyproject.toml +1 -1
  19. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_cli_startup.py +7 -0
  20. agent_coderag-1.5.1/tests/test_gitignore_root.py +67 -0
  21. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_interfaces.py +7 -1
  22. agent_coderag-1.5.1/tests/test_migration_cli_e2e.py +141 -0
  23. agent_coderag-1.5.1/tests/test_migration_coderag.py +246 -0
  24. agent_coderag-1.5.1/tests/test_path_migration.py +565 -0
  25. agent_coderag-1.5.1/tests/test_relative_sync_paths.py +42 -0
  26. agent_coderag-1.5.1/tests/test_search_relative_paths.py +111 -0
  27. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_tree_sitter_parser.py +52 -0
  28. agent_coderag-1.4.1/code_rag/parsers/multi_parser.py +0 -31
  29. agent_coderag-1.4.1/code_rag/services/search.py +0 -9
  30. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/LICENSE +0 -0
  31. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/NOTICE +0 -0
  32. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/agent_coderag.egg-info/dependency_links.txt +0 -0
  33. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/agent_coderag.egg-info/entry_points.txt +0 -0
  34. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/agent_coderag.egg-info/requires.txt +0 -0
  35. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/agent_coderag.egg-info/top_level.txt +0 -0
  36. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/__init__.py +0 -0
  37. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/api/__init__.py +0 -0
  38. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/api/models.py +0 -0
  39. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/core/__init__.py +0 -0
  40. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/core/constants.py +0 -0
  41. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/core/error_codes.py +0 -0
  42. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/core/exceptions.py +0 -0
  43. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/core/models.py +0 -0
  44. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/core/utils.py +0 -0
  45. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/__init__.py +0 -0
  46. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/dependency.py +0 -0
  47. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/java_discovery.py +0 -0
  48. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/manager.py +0 -0
  49. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/__init__.py +0 -0
  50. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/base.py +0 -0
  51. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/csharp.py +0 -0
  52. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/go.py +0 -0
  53. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/java.py +0 -0
  54. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/javascript.py +0 -0
  55. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/python.py +0 -0
  56. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/discovery/providers/rust.py +0 -0
  57. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/entry/__init__.py +0 -0
  58. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/intelligence/__init__.py +0 -0
  59. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/intelligence/embedder.py +0 -0
  60. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/intelligence/factory.py +0 -0
  61. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/intelligence/openai_embedder.py +0 -0
  62. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/parsers/__init__.py +0 -0
  63. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/parsers/languages.py +0 -0
  64. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/__init__.py +0 -0
  65. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/config.py +0 -0
  66. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/dependencies.py +0 -0
  67. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/discovery_api.py +0 -0
  68. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/factory.py +0 -0
  69. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/services/setup.py +0 -0
  70. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/storage/__init__.py +0 -0
  71. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/code_rag/storage/db_connection.py +0 -0
  72. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/setup.cfg +0 -0
  73. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_api_client.py +0 -0
  74. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_api_lazy_db.py +0 -0
  75. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_api_models.py +0 -0
  76. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_api_rebuild_wipe.py +0 -0
  77. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_cli.py +0 -0
  78. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_cli_detailed.py +0 -0
  79. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_cli_embeddings.py +0 -0
  80. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_cli_json_parity.py +0 -0
  81. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_code_rag_simple.py +0 -0
  82. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_coderag_lifetime.py +0 -0
  83. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_csharp_discovery_detailed.py +0 -0
  84. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_db_connection.py +0 -0
  85. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_db_path.py +0 -0
  86. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_dependency_discovery.py +0 -0
  87. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_discovery_manager_detailed.py +0 -0
  88. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_discovery_providers_extra.py +0 -0
  89. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_discovery_python.py +0 -0
  90. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_distiller.py +0 -0
  91. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_distiller_config_embedding.py +0 -0
  92. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_distiller_extra.py +0 -0
  93. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_embedder.py +0 -0
  94. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_embedder_detailed.py +0 -0
  95. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_embedder_factory.py +0 -0
  96. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_embedder_interface.py +0 -0
  97. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_embedder_stubs.py +0 -0
  98. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_error_codes.py +0 -0
  99. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_factory_async.py +0 -0
  100. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_indexing_detailed.py +0 -0
  101. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_indexing_embeddings.py +0 -0
  102. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_indexing_extra.py +0 -0
  103. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_js_discovery_detailed.py +0 -0
  104. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_languages.py +0 -0
  105. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_local_onnx_embedder.py +0 -0
  106. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_models.py +0 -0
  107. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_multi_parser.py +0 -0
  108. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_openai_embedder.py +0 -0
  109. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_readme_library_usage.py +0 -0
  110. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_readme_lifetime_storage.py +0 -0
  111. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_readme_offline_embeddings.py +0 -0
  112. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_rust_discovery_detailed.py +0 -0
  113. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_storage_detailed.py +0 -0
  114. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_storage_embeddings.py +0 -0
  115. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_sync_worker_errors.py +0 -0
  116. {agent_coderag-1.4.1 → agent_coderag-1.5.1}/tests/test_utils_detailed.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agent-coderag
3
- Version: 1.4.1
3
+ Version: 1.5.1
4
4
  Summary: Lightweight semantic code search and distillation utility for AI coding agents. It solves the API knowledge gap via real-time local signature extraction and intent analysis without PyTorch. Optimized for token efficiency, it compresses codebase context into compact semantic summaries stored in a local DuckDB vector similarity index.
5
5
  Author-email: Igor Boloban <naranor@gmail.com>
6
6
  License: MIT
@@ -190,6 +190,7 @@ async def main():
190
190
  - **Sidecars:** DuckDB may write WAL sidecars (e.g. `.coderag.db.wal`) beside the index during writes; locks should not persist after an operation finishes.
191
191
  - **Connect timeout:** `connect_timeout_seconds=5` (CLI `--connect-timeout`) waits on file locks, then raises `StorageBusyError` (`ErrorCode.STORAGE_BUSY`). Pass `0` for a single attempt.
192
192
  - **Read-only search:** `search` (and `api` when storage is needed) opens read-only. A missing **index file** is an error — use `sync`/`rebuild` to create it. An index file that exists but has no embeddings table (e.g. opened/written without a completed vector sync) raises `StorageError` with `ErrorCode.EMBEDDINGS_MISSING` — run `sync` (library: `CodeRAG.sync`) before search. With `--json`, success is a hit array; errors are `{"status":"error","message":...}` and include `"code"` when the exception carries an `ErrorCode`.
193
+ - **Paths:** `sync` stores paths relative to `root` (`src/a.py`). The first sync rewrites an older absolute index when the file is still under `root` or its unit hashes match a file in the tree. Search returns an absolute path under the current root. `search --relative-paths`, config `relative_paths: true`, or `CodeRAG(relative_paths=True)` returns the stored relative path. Until that sync runs, search returns the absolute path stored in the index. Ignore rules use the same project-relative path, so a `.worktrees/<name>` checkout is indexed when that directory is the root.
193
194
  - **Lifetime:** embedder/parser/distiller stay warm; DuckDB opens per operation and closes afterward. One `CodeRAG` instance serializes overlapping ops. `config()` with embedding flags / `--clear-embedding` closes the process embedder so the next op rebuilds it; distill-only `config` refreshes Distiller and keeps the embedder.
194
195
  - **`api()` without DB:** providers that do not need the index (e.g. Python) skip DuckDB entirely; Java uses a short read-only open for JAR cache lookup.
195
196
  - **Errors:** catch `CodeRAGError` and inspect `.code` — `STORAGE_BUSY`, `STORAGE_CORRUPT`, `EMBEDDING_MISMATCH`, `EMBEDDINGS_MISSING` (`from code_rag import ErrorCode`).
@@ -144,6 +144,7 @@ async def main():
144
144
  - **Sidecars:** DuckDB may write WAL sidecars (e.g. `.coderag.db.wal`) beside the index during writes; locks should not persist after an operation finishes.
145
145
  - **Connect timeout:** `connect_timeout_seconds=5` (CLI `--connect-timeout`) waits on file locks, then raises `StorageBusyError` (`ErrorCode.STORAGE_BUSY`). Pass `0` for a single attempt.
146
146
  - **Read-only search:** `search` (and `api` when storage is needed) opens read-only. A missing **index file** is an error — use `sync`/`rebuild` to create it. An index file that exists but has no embeddings table (e.g. opened/written without a completed vector sync) raises `StorageError` with `ErrorCode.EMBEDDINGS_MISSING` — run `sync` (library: `CodeRAG.sync`) before search. With `--json`, success is a hit array; errors are `{"status":"error","message":...}` and include `"code"` when the exception carries an `ErrorCode`.
147
+ - **Paths:** `sync` stores paths relative to `root` (`src/a.py`). The first sync rewrites an older absolute index when the file is still under `root` or its unit hashes match a file in the tree. Search returns an absolute path under the current root. `search --relative-paths`, config `relative_paths: true`, or `CodeRAG(relative_paths=True)` returns the stored relative path. Until that sync runs, search returns the absolute path stored in the index. Ignore rules use the same project-relative path, so a `.worktrees/<name>` checkout is indexed when that directory is the root.
147
148
  - **Lifetime:** embedder/parser/distiller stay warm; DuckDB opens per operation and closes afterward. One `CodeRAG` instance serializes overlapping ops. `config()` with embedding flags / `--clear-embedding` closes the process embedder so the next op rebuilds it; distill-only `config` refreshes Distiller and keeps the embedder.
148
149
  - **`api()` without DB:** providers that do not need the index (e.g. Python) skip DuckDB entirely; Java uses a short read-only open for JAR cache lookup.
149
150
  - **Errors:** catch `CodeRAGError` and inspect `.code` — `STORAGE_BUSY`, `STORAGE_CORRUPT`, `EMBEDDING_MISMATCH`, `EMBEDDINGS_MISSING` (`from code_rag import ErrorCode`).
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agent-coderag
3
- Version: 1.4.1
3
+ Version: 1.5.1
4
4
  Summary: Lightweight semantic code search and distillation utility for AI coding agents. It solves the API knowledge gap via real-time local signature extraction and intent analysis without PyTorch. Optimized for token efficiency, it compresses codebase context into compact semantic summaries stored in a local DuckDB vector similarity index.
5
5
  Author-email: Igor Boloban <naranor@gmail.com>
6
6
  License: MIT
@@ -190,6 +190,7 @@ async def main():
190
190
  - **Sidecars:** DuckDB may write WAL sidecars (e.g. `.coderag.db.wal`) beside the index during writes; locks should not persist after an operation finishes.
191
191
  - **Connect timeout:** `connect_timeout_seconds=5` (CLI `--connect-timeout`) waits on file locks, then raises `StorageBusyError` (`ErrorCode.STORAGE_BUSY`). Pass `0` for a single attempt.
192
192
  - **Read-only search:** `search` (and `api` when storage is needed) opens read-only. A missing **index file** is an error — use `sync`/`rebuild` to create it. An index file that exists but has no embeddings table (e.g. opened/written without a completed vector sync) raises `StorageError` with `ErrorCode.EMBEDDINGS_MISSING` — run `sync` (library: `CodeRAG.sync`) before search. With `--json`, success is a hit array; errors are `{"status":"error","message":...}` and include `"code"` when the exception carries an `ErrorCode`.
193
+ - **Paths:** `sync` stores paths relative to `root` (`src/a.py`). The first sync rewrites an older absolute index when the file is still under `root` or its unit hashes match a file in the tree. Search returns an absolute path under the current root. `search --relative-paths`, config `relative_paths: true`, or `CodeRAG(relative_paths=True)` returns the stored relative path. Until that sync runs, search returns the absolute path stored in the index. Ignore rules use the same project-relative path, so a `.worktrees/<name>` checkout is indexed when that directory is the root.
193
194
  - **Lifetime:** embedder/parser/distiller stay warm; DuckDB opens per operation and closes afterward. One `CodeRAG` instance serializes overlapping ops. `config()` with embedding flags / `--clear-embedding` closes the process embedder so the next op rebuilds it; distill-only `config` refreshes Distiller and keeps the embedder.
194
195
  - **`api()` without DB:** providers that do not need the index (e.g. Python) skip DuckDB entirely; Java uses a short read-only open for JAR cache lookup.
195
196
  - **Errors:** catch `CodeRAGError` and inspect `.code` — `STORAGE_BUSY`, `STORAGE_CORRUPT`, `EMBEDDING_MISMATCH`, `EMBEDDINGS_MISSING` (`from code_rag import ErrorCode`).
@@ -50,6 +50,7 @@ code_rag/services/dependencies.py
50
50
  code_rag/services/discovery_api.py
51
51
  code_rag/services/factory.py
52
52
  code_rag/services/indexing.py
53
+ code_rag/services/path_migration.py
53
54
  code_rag/services/search.py
54
55
  code_rag/services/setup.py
55
56
  code_rag/services/sync.py
@@ -84,6 +85,7 @@ tests/test_embedder_interface.py
84
85
  tests/test_embedder_stubs.py
85
86
  tests/test_error_codes.py
86
87
  tests/test_factory_async.py
88
+ tests/test_gitignore_root.py
87
89
  tests/test_indexing_detailed.py
88
90
  tests/test_indexing_embeddings.py
89
91
  tests/test_indexing_extra.py
@@ -91,13 +93,18 @@ tests/test_interfaces.py
91
93
  tests/test_js_discovery_detailed.py
92
94
  tests/test_languages.py
93
95
  tests/test_local_onnx_embedder.py
96
+ tests/test_migration_cli_e2e.py
97
+ tests/test_migration_coderag.py
94
98
  tests/test_models.py
95
99
  tests/test_multi_parser.py
96
100
  tests/test_openai_embedder.py
101
+ tests/test_path_migration.py
97
102
  tests/test_readme_library_usage.py
98
103
  tests/test_readme_lifetime_storage.py
99
104
  tests/test_readme_offline_embeddings.py
105
+ tests/test_relative_sync_paths.py
100
106
  tests/test_rust_discovery_detailed.py
107
+ tests/test_search_relative_paths.py
101
108
  tests/test_storage_detailed.py
102
109
  tests/test_storage_embeddings.py
103
110
  tests/test_sync_worker_errors.py
@@ -19,7 +19,7 @@ from code_rag.services.config import (
19
19
  )
20
20
  from code_rag.services.discovery_api import run_api
21
21
  from code_rag.services.factory import create_stack
22
- from code_rag.services.search import run_search
22
+ from code_rag.services.search import format_unit_path, run_search
23
23
  from code_rag.services.setup import run_setup
24
24
  from code_rag.services.indexing import IndexStack
25
25
  from code_rag.services.sync import SyncOptions, run_rebuild, run_sync
@@ -40,12 +40,17 @@ class CodeRAG: # pylint: disable=too-many-instance-attributes
40
40
  root: Optional[Union[str, Path]] = None,
41
41
  allow_build_execution: bool = False,
42
42
  connect_timeout_seconds: float = DEFAULT_CONNECT_TIMEOUT_SECONDS,
43
+ relative_paths: bool | None = None,
43
44
  ):
44
45
  self._onnx = onnx
45
46
  self._root = Path(root) if root is not None else Path.cwd()
46
47
  self._allow_build_execution = allow_build_execution
47
48
  self._connect_timeout_seconds = connect_timeout_seconds
48
49
  self._db_path = resolve_db_path(db, root=self._root)
50
+ if relative_paths is None:
51
+ self._relative_paths = bool(DistillerConfig.load().relative_paths)
52
+ else:
53
+ self._relative_paths = relative_paths
49
54
  self._embedder: Optional[IEmbedder] = None
50
55
  self._parser: Optional[IParser] = None
51
56
  self._distiller: Optional[IIntelligence] = None
@@ -199,7 +204,12 @@ class CodeRAG: # pylint: disable=too-many-instance-attributes
199
204
 
200
205
  async def search(self, query: str, *, limit: int = 5) -> list[KnowledgeUnit]:
201
206
  async with self._with_storage(AccessMode.READ_ONLY) as storage:
202
- return await run_search(storage, query, limit=limit)
207
+ units = await run_search(storage, query, limit=limit)
208
+ for unit in units:
209
+ unit.path = format_unit_path(
210
+ unit.path, root=self._root, relative_paths=self._relative_paths
211
+ )
212
+ return units
203
213
 
204
214
  async def _api_java(
205
215
  self,
@@ -7,7 +7,13 @@ class IParser(ABC):
7
7
  """Interface for extracting structure from code."""
8
8
 
9
9
  @abstractmethod
10
- async def distill_file(self, file_path: str) -> List[KnowledgeUnit]:
10
+ async def distill_file(
11
+ self,
12
+ file_path: str,
13
+ *,
14
+ stored_path: str | None = None,
15
+ raise_on_failure: bool = False,
16
+ ) -> List[KnowledgeUnit]:
11
17
  pass # pragma: no cover
12
18
 
13
19
 
@@ -50,6 +50,13 @@ def build_parser() -> argparse.ArgumentParser:
50
50
  search = subparsers.add_parser("search", help="Semantic search.")
51
51
  search.add_argument("query", help="Natural language query.")
52
52
  search.add_argument("--limit", type=int, default=5, help="Result limit.")
53
+ search.add_argument(
54
+ "--relative-paths",
55
+ dest="relative_paths",
56
+ action="store_true",
57
+ default=None,
58
+ help="Print paths relative to the project root. Default: absolute paths under the current root.",
59
+ )
53
60
 
54
61
  api = subparsers.add_parser("api", help="Discover library API.")
55
62
  api.add_argument("library", help="Library name (e.g., pydantic).")
@@ -111,3 +118,7 @@ def main() -> None:
111
118
  except Exception as exc:
112
119
  logging.getLogger(__name__).error("Unexpected error: %s", exc)
113
120
  sys.exit(1)
121
+
122
+
123
+ if __name__ == "__main__":
124
+ main()
@@ -46,6 +46,7 @@ __all__ = [
46
46
 
47
47
 
48
48
  def _coderag_from_args(args) -> CodeRAG:
49
+ relative_paths = getattr(args, "relative_paths", None)
49
50
  return CodeRAG(
50
51
  db=args.db,
51
52
  onnx=getattr(args, "onnx", None),
@@ -54,6 +55,7 @@ def _coderag_from_args(args) -> CodeRAG:
54
55
  getattr(args, "connect_timeout", DEFAULT_CONNECT_TIMEOUT_SECONDS)
55
56
  ),
56
57
  allow_build_execution=bool(getattr(args, "allow_build_execution", False)),
58
+ relative_paths=True if relative_paths else None,
57
59
  )
58
60
 
59
61
 
@@ -70,9 +72,14 @@ def load_ignore_patterns() -> pathspec.PathSpec:
70
72
  return sync_service.load_ignore_patterns(Path.cwd())
71
73
 
72
74
 
73
- def should_index(path: Path, ignore_spec: Optional[pathspec.PathSpec] = None) -> bool:
75
+ def should_index(
76
+ path: Path,
77
+ ignore_spec: Optional[pathspec.PathSpec] = None,
78
+ *,
79
+ root: Optional[Path] = None,
80
+ ) -> bool:
74
81
  """Filters files that should NOT be indexed."""
75
- return sync_service.should_index(path, ignore_spec)
82
+ return sync_service.should_index(path, ignore_spec, root=root)
76
83
 
77
84
 
78
85
  def _emit_sync_outcome(result: SyncResult, *, json_mode: bool, label: str) -> None:
@@ -23,6 +23,7 @@ class DistillerConfig(BaseModel):
23
23
  embedding_key: Optional[str] = None
24
24
  embedding_model: Optional[str] = None
25
25
  embedding_provider: Optional[str] = None
26
+ relative_paths: bool = False
26
27
 
27
28
  @field_validator(
28
29
  "model",
@@ -0,0 +1,46 @@
1
+ import os
2
+ import logging
3
+ from typing import List
4
+ from ..core.interfaces import IParser
5
+ from ..core.models import KnowledgeUnit
6
+ from .tree_sitter import TreeSitterParser
7
+ from .languages import EXTENSION_TO_LANGUAGE
8
+
9
+ logger = logging.getLogger(__name__)
10
+
11
+
12
+ class MultiParser(IParser):
13
+ """
14
+ Delegates parsing to TreeSitterParser for all supported languages.
15
+ """
16
+
17
+ def __init__(self) -> None:
18
+ self.tree_sitter_parser = TreeSitterParser()
19
+
20
+ async def distill_file(
21
+ self,
22
+ file_path: str,
23
+ *,
24
+ stored_path: str | None = None,
25
+ raise_on_failure: bool = False,
26
+ ) -> List[KnowledgeUnit]:
27
+ _, ext = os.path.splitext(file_path)
28
+ ext = ext.lower()
29
+
30
+ if ext in EXTENSION_TO_LANGUAGE:
31
+ if stored_path is None and not raise_on_failure:
32
+ return await self.tree_sitter_parser.distill_file(file_path)
33
+ if stored_path is not None and raise_on_failure:
34
+ return await self.tree_sitter_parser.distill_file(
35
+ file_path, stored_path=stored_path, raise_on_failure=True
36
+ )
37
+ if stored_path is not None:
38
+ return await self.tree_sitter_parser.distill_file(
39
+ file_path, stored_path=stored_path
40
+ )
41
+ return await self.tree_sitter_parser.distill_file(
42
+ file_path, raise_on_failure=True
43
+ )
44
+
45
+ logger.debug("No parser for extension %s", ext)
46
+ return []
@@ -45,10 +45,14 @@ class TreeSitterParser(IParser):
45
45
  f"[MISSING DEPENDENCY] Please install: pip install {lang_config.package}"
46
46
  ) from e
47
47
 
48
- async def distill_file(self, file_path: str) -> List[KnowledgeUnit]:
49
- """
50
- Parses a file and extracts high-level units (classes, functions).
51
- """
48
+ async def distill_file(
49
+ self,
50
+ file_path: str,
51
+ *,
52
+ stored_path: str | None = None,
53
+ raise_on_failure: bool = False,
54
+ ) -> List[KnowledgeUnit]:
55
+ identity = file_path if stored_path is None else stored_path
52
56
  ext = Path(file_path).suffix.lower()
53
57
  lang_config = next((cfg for cfg in LANGUAGES if ext in cfg.extensions), None)
54
58
 
@@ -67,36 +71,52 @@ class TreeSitterParser(IParser):
67
71
  ctx: Dict[str, Any] = {
68
72
  "source": source,
69
73
  "config": lang_config,
70
- "file_path": file_path,
74
+ "file_path": identity,
71
75
  "units": units,
72
76
  "scope": None,
77
+ "name_counts": {},
73
78
  }
74
79
  self._recursive_distill(tree.root_node, ctx)
75
80
  return units
76
81
  except GrammarNotFoundError:
77
- # Re-raise grammar errors so callers can handle them (e.g. CLI setup suggestion)
78
82
  raise
79
83
  except Exception as e:
80
84
  logger.error("Failed to parse %s: %s", file_path, e)
85
+ if raise_on_failure:
86
+ raise
81
87
  return []
82
88
 
83
89
  def _recursive_distill(self, node: Node, ctx: Dict[str, Any]) -> None:
84
90
  config = ctx["config"]
85
- current_scope = ctx["scope"]
91
+ child_scope = ctx["scope"]
92
+ child_counts = ctx["name_counts"]
86
93
 
87
94
  if node.type in config.canonical_map:
88
- self._process_node(node, ctx)
89
- if config.canonical_map.get(node.type) == "CLASS":
90
- node_name = self._resolve_name(node)
91
- scope = ctx["scope"]
92
- current_scope = f"{scope}.{node_name}" if scope else node_name
93
-
94
- # Recurse into children with updated scope
95
- child_ctx = {**ctx, "scope": current_scope}
95
+ qname = self._qualified_name(node, ctx)
96
+ self._process_node(node, ctx, qname)
97
+ if config.canonical_map.get(node.type) in ("CLASS", "FUNCTION", "METHOD"):
98
+ child_scope = qname
99
+ child_counts = {}
100
+
101
+ child_ctx = {**ctx, "scope": child_scope, "name_counts": child_counts}
96
102
  for child in node.children:
97
103
  self._recursive_distill(child, child_ctx)
98
104
 
99
- def _process_node(self, node: Node, ctx: Dict[str, Any]) -> None:
105
+ def _qualified_name(self, node: Node, ctx: Dict[str, Any]) -> str:
106
+ """Qualified id segment. Enclosing functions are part of the scope.
107
+
108
+ A repeated name in the same scope gets ``#2``, ``#3``, … so two
109
+ different bodies cannot share one primary key.
110
+ """
111
+ node_name = self._resolve_name(node)
112
+ counts: Dict[str, int] = ctx["name_counts"]
113
+ counts[node_name] = counts.get(node_name, 0) + 1
114
+ seen = counts[node_name]
115
+ label = node_name if seen == 1 else f"{node_name}#{seen}"
116
+ scope = ctx["scope"]
117
+ return f"{scope}.{label}" if scope else label
118
+
119
+ def _process_node(self, node: Node, ctx: Dict[str, Any], qname: str) -> None:
100
120
  """Processes a single node and adds it to the units list."""
101
121
  node_name = self._resolve_name(node)
102
122
  docstring = self._extract_docstring(node)
@@ -108,9 +128,6 @@ class TreeSitterParser(IParser):
108
128
 
109
129
  kind = self._determine_kind(node, ctx["config"])
110
130
 
111
- # Stable ID based on qualified name
112
- scope = ctx["scope"]
113
- qname = f"{scope}.{node_name}" if scope else node_name
114
131
  unit_id = f"{ctx['file_path']}:{qname}"
115
132
 
116
133
  unit = KnowledgeUnit(
@@ -13,6 +13,11 @@ def default_db_path(root: Path | None = None) -> Path:
13
13
  return root_path / ".coderag.db"
14
14
 
15
15
 
16
+ def project_relative_posix(path: Path, root: Path) -> str:
17
+ """Posix path of ``path`` relative to ``root``. Raises ValueError when outside ``root``."""
18
+ return path.resolve().relative_to(root.resolve()).as_posix()
19
+
20
+
16
21
  def resolve_db_path(db: str | Path | None, *, root: Path | None = None) -> Path:
17
22
  if db is not None:
18
23
  path = Path(db)
@@ -7,12 +7,14 @@ import inspect
7
7
  import logging
8
8
  import sys
9
9
  from dataclasses import dataclass
10
+ from pathlib import Path
10
11
  from typing import List, Optional
11
12
 
12
13
  from code_rag.core.constants import EMBEDDING_BATCH_SIZE, MAX_CONCURRENT_TASKS
13
14
  from code_rag.core.exceptions import IntelligenceError, StorageError
14
15
  from code_rag.core.interfaces import IIntelligence, IParser, IStorage
15
16
  from code_rag.core.models import KnowledgeUnit
17
+ from code_rag.paths import project_relative_posix
16
18
 
17
19
  logger = logging.getLogger(__name__)
18
20
 
@@ -24,6 +26,7 @@ class IndexStack:
24
26
  storage: IStorage
25
27
  parser: IParser
26
28
  intelligence: IIntelligence
29
+ root: Optional[Path] = None
27
30
 
28
31
 
29
32
  def _report_worker_failure(path: str, exc: BaseException) -> tuple[str, str]:
@@ -146,11 +149,19 @@ async def sync_file(
146
149
  *,
147
150
  force_distill: bool = False,
148
151
  max_concurrency: int = MAX_CONCURRENT_TASKS,
152
+ stored_path: str | None = None,
149
153
  ) -> None:
150
154
  """Parse, distill, embed, and upsert one file into ``stack.storage``."""
151
155
  await _reject_dirty_incremental(stack.storage)
152
156
  semaphore = asyncio.Semaphore(max_concurrency)
153
- current_units = await stack.parser.distill_file(file_path)
157
+ if stored_path is None:
158
+ current_units = await stack.parser.distill_file(file_path)
159
+ identity = file_path
160
+ else:
161
+ current_units = await stack.parser.distill_file(
162
+ file_path, stored_path=stored_path
163
+ )
164
+ identity = stored_path
154
165
  pending: list[KnowledgeUnit] = []
155
166
  for unit in current_units:
156
167
  await _process_unit(
@@ -162,7 +173,7 @@ async def sync_file(
162
173
  )
163
174
  await _embed_and_upsert(stack.storage, pending)
164
175
  await stack.storage.delete_stale_units(
165
- file_path, [unit.id for unit in current_units]
176
+ identity, [unit.id for unit in current_units]
166
177
  )
167
178
 
168
179
 
@@ -194,11 +205,17 @@ async def sync_project(
194
205
  except asyncio.QueueEmpty:
195
206
  return failures
196
207
  try:
208
+ stored_path = (
209
+ None
210
+ if stack.root is None
211
+ else project_relative_posix(Path(path), stack.root)
212
+ )
197
213
  await sync_file(
198
214
  stack,
199
215
  path,
200
216
  force_distill=force_distill,
201
217
  max_concurrency=max_concurrency,
218
+ stored_path=stored_path,
202
219
  )
203
220
  except Exception as exc:
204
221
  failures.append(_report_worker_failure(path, exc))
@@ -0,0 +1,121 @@
1
+ """One-time rewrite of absolute index paths onto project-relative posix paths."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from collections import Counter
7
+ from pathlib import Path
8
+
9
+ from code_rag.core.models import KnowledgeUnit
10
+ from code_rag.parsers.tree_sitter import GrammarNotFoundError
11
+ from code_rag.paths import project_relative_posix
12
+ from code_rag.storage.duckdb_impl import DuckDBStorage
13
+
14
+ logger = logging.getLogger(__name__)
15
+
16
+
17
+ def _groups(units: list[KnowledgeUnit]) -> dict[str, Counter[str]]:
18
+ groups: dict[str, Counter[str]] = {}
19
+ for unit in units:
20
+ if not Path(unit.path).is_absolute():
21
+ continue
22
+ groups.setdefault(unit.path, Counter())[unit.code_hash] += 1
23
+ return groups
24
+
25
+
26
+ def claim_relative_paths(
27
+ groups: dict[str, Counter[str]],
28
+ root: Path,
29
+ files: list[tuple[Path, Counter[str]]],
30
+ ) -> dict[str, str]:
31
+ mapping: dict[str, str] = {}
32
+ taken: set[str] = set()
33
+ pending: list[tuple[str, Counter[str]]] = []
34
+ for old, hashes in groups.items():
35
+ path = Path(old)
36
+ if path.is_absolute():
37
+ try:
38
+ rel = project_relative_posix(path, root)
39
+ except ValueError:
40
+ pending.append((old, hashes))
41
+ continue
42
+ if rel in taken:
43
+ continue
44
+ mapping[old] = rel
45
+ taken.add(rel)
46
+ continue
47
+ pending.append((old, hashes))
48
+
49
+ claimed: set[str] = set()
50
+ for file_path, hashes in files:
51
+ try:
52
+ rel = project_relative_posix(file_path, root)
53
+ except ValueError:
54
+ continue
55
+ if rel in taken:
56
+ continue
57
+ for old, group_hashes in pending:
58
+ if old in claimed or old in mapping:
59
+ continue
60
+ if group_hashes != hashes:
61
+ continue
62
+ mapping[old] = rel
63
+ taken.add(rel)
64
+ claimed.add(old)
65
+ break
66
+ return mapping
67
+
68
+
69
+ def _suffix(path: str) -> str:
70
+ return Path(path).suffix.lower()
71
+
72
+
73
+ async def migrate_absolute_paths(
74
+ storage, parser, root: Path, files: list[Path]
75
+ ) -> bool:
76
+ if not isinstance(storage, DuckDBStorage):
77
+ return False
78
+ if await storage.paths_migration_done():
79
+ return False
80
+ groups = _groups(await storage.list_units())
81
+ if not groups:
82
+ await storage.commit_path_migration({})
83
+ return False
84
+ incomplete = False
85
+ skipped_suffixes: set[str] = set()
86
+ parsed: list[tuple[Path, Counter[str]]] = []
87
+ for path in files:
88
+ try:
89
+ units = await parser.distill_file(str(path), raise_on_failure=True)
90
+ except GrammarNotFoundError:
91
+ logger.warning("Skipping unparsed file during path migration: %s", path)
92
+ skipped_suffixes.add(_suffix(str(path)))
93
+ incomplete = True
94
+ continue
95
+ except Exception:
96
+ logger.exception("Path migration could not parse %s", path)
97
+ skipped_suffixes.add(_suffix(str(path)))
98
+ incomplete = True
99
+ continue
100
+ parsed.append((path, Counter(unit.code_hash for unit in units)))
101
+ mapping = claim_relative_paths(groups, root, parsed)
102
+ failed_commits: set[str] = set()
103
+ for old, new in mapping.items():
104
+ try:
105
+ await storage.commit_rewritten_path(old, new)
106
+ except Exception:
107
+ logger.exception("Path migration failed for %s", old)
108
+ failed_commits.add(old)
109
+ incomplete = True
110
+ orphans = [
111
+ old
112
+ for old in groups
113
+ if old not in mapping
114
+ and old not in failed_commits
115
+ and _suffix(old) not in skipped_suffixes
116
+ ]
117
+ if orphans:
118
+ await storage.delete_absolute_paths(orphans)
119
+ if not incomplete:
120
+ await storage.mark_paths_migrated()
121
+ return True
@@ -0,0 +1,20 @@
1
+ from pathlib import Path
2
+
3
+ from code_rag.core.interfaces import IStorage
4
+ from code_rag.core.models import KnowledgeUnit
5
+
6
+
7
+ def format_unit_path(stored: str, *, root: Path, relative_paths: bool) -> str:
8
+ """Format a stored index path for search output. Does not read or write storage."""
9
+ if Path(stored).is_absolute():
10
+ return stored
11
+ if relative_paths:
12
+ return stored
13
+ return str((root / stored).resolve())
14
+
15
+
16
+ async def run_search(
17
+ storage: IStorage, query: str, *, limit: int = 5
18
+ ) -> list[KnowledgeUnit]:
19
+ """Performs semantic search across indexed units."""
20
+ return await storage.search_units(query, limit=limit)