capyidx 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- capyidx/__init__.py +62 -0
- capyidx/api.py +397 -0
- capyidx/base/__init__.py +1 -0
- capyidx/base/index_d.py +194 -0
- capyidx/base/index_types.py +85 -0
- capyidx/base/refresh_index.py +472 -0
- capyidx/chunker/__init__.py +1 -0
- capyidx/chunker/basic.py +50 -0
- capyidx/chunker/chunk.py +161 -0
- capyidx/chunker/chunk_codebase_index.py +351 -0
- capyidx/chunker/code.py +715 -0
- capyidx/codesnippet/__init__.py +1 -0
- capyidx/codesnippet/code_snippets_index.py +634 -0
- capyidx/db/__init__.py +1 -0
- capyidx/db/db.py +107 -0
- capyidx/db/paths.py +28 -0
- capyidx/db/schema.py +119 -0
- capyidx/embeddings/__init__.py +1 -0
- capyidx/embeddings/base.py +8 -0
- capyidx/embeddings/local.py +57 -0
- capyidx/embeddings/ollama.py +24 -0
- capyidx/embeddings/openai.py +42 -0
- capyidx/fts/__init__.py +1 -0
- capyidx/fts/fulltextsearch_codebase_index.py +282 -0
- capyidx/indexer/__init__.py +1 -0
- capyidx/indexer/codebase_indexer.py +876 -0
- capyidx/lance_db/__init__.py +1 -0
- capyidx/lance_db/lancedb_index.py +566 -0
- capyidx/retrieval/__init__.py +1 -0
- capyidx/retrieval/models.py +122 -0
- capyidx/retrieval/retrieval_pipeline.py +653 -0
- capyidx/utils/__init__.py +1 -0
- capyidx/utils/chunk_utils.py +32 -0
- capyidx/utils/count_tokens.py +475 -0
- capyidx/utils/disk_operations.py +144 -0
- capyidx/utils/ignore.py +100 -0
- capyidx/utils/parameters.py +28 -0
- capyidx/utils/paths.py +74 -0
- capyidx/utils/retrieval_utils.py +54 -0
- capyidx/utils/tree_sitter.py +324 -0
- capyidx/utils/uri.py +152 -0
- capyidx/utils/uri2.py +215 -0
- capyidx/walker/__init__.py +1 -0
- capyidx/walker/walk_dir.py +359 -0
- capyidx/watcher/__init__.py +1 -0
- capyidx/watcher/file_watcher.py +272 -0
- capyidx-0.1.0.dist-info/METADATA +163 -0
- capyidx-0.1.0.dist-info/RECORD +51 -0
- capyidx-0.1.0.dist-info/WHEEL +5 -0
- capyidx-0.1.0.dist-info/licenses/LICENSE +21 -0
- capyidx-0.1.0.dist-info/top_level.txt +1 -0
capyidx/__init__.py
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""
|
|
2
|
+
CoreIndexer — index a codebase, look up symbols by name.
|
|
3
|
+
|
|
4
|
+
The public surface is re-exported here so consumers can write::
|
|
5
|
+
|
|
6
|
+
from coreindexer import index_repo, lookup_symbol, resolve_lookup
|
|
7
|
+
|
|
8
|
+
rather than reaching into submodules. Everything documented in
|
|
9
|
+
:mod:`coreindexer.api` is available from the package root.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
from capyidx.api import (
|
|
15
|
+
index_repo,
|
|
16
|
+
index_repo_iter,
|
|
17
|
+
lookup_symbol,
|
|
18
|
+
open_lookup,
|
|
19
|
+
resolve_lookup,
|
|
20
|
+
reconstruct_symbol,
|
|
21
|
+
PathLike,
|
|
22
|
+
ResolvedLookup,
|
|
23
|
+
)
|
|
24
|
+
from capyidx.base.index_d import (
|
|
25
|
+
Chunk,
|
|
26
|
+
Chonk,
|
|
27
|
+
ChunkingResult,
|
|
28
|
+
Symbol,
|
|
29
|
+
BranchAndDir
|
|
30
|
+
)
|
|
31
|
+
from capyidx.retrieval.models import (
|
|
32
|
+
LookupResult,
|
|
33
|
+
SymbolCode,
|
|
34
|
+
SymbolMatch,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
__version__ = "0.1.0"
|
|
38
|
+
|
|
39
|
+
__all__ = [
|
|
40
|
+
# Types
|
|
41
|
+
"PathLike",
|
|
42
|
+
"__version__",
|
|
43
|
+
# Indexing
|
|
44
|
+
"index_repo",
|
|
45
|
+
"index_repo_iter",
|
|
46
|
+
# Querying
|
|
47
|
+
"open_lookup",
|
|
48
|
+
"lookup_symbol",
|
|
49
|
+
"resolve_lookup",
|
|
50
|
+
"reconstruct_symbol",
|
|
51
|
+
"ResolvedLookup",
|
|
52
|
+
# Result types
|
|
53
|
+
"LookupResult",
|
|
54
|
+
"SymbolCode",
|
|
55
|
+
"SymbolMatch",
|
|
56
|
+
# Core data types
|
|
57
|
+
"Chunk",
|
|
58
|
+
"Chonk",
|
|
59
|
+
"ChunkingResult",
|
|
60
|
+
"Symbol",
|
|
61
|
+
"BranchAndDir"
|
|
62
|
+
]
|
capyidx/api.py
ADDED
|
@@ -0,0 +1,397 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Public API for CoreIndexer.
|
|
3
|
+
|
|
4
|
+
Two workflows:
|
|
5
|
+
|
|
6
|
+
* **Index** — walk a repo, chunk it, persist to SQLite.
|
|
7
|
+
|
|
8
|
+
- ``index_repo`` — build the index once, discard progress, return.
|
|
9
|
+
- ``index_repo_iter`` — same, but yield progress updates. Pass
|
|
10
|
+
``watch=True`` to keep yielding after the initial pass as files
|
|
11
|
+
change on disk.
|
|
12
|
+
|
|
13
|
+
* **Query** — look up symbols against an existing index. Four entry
|
|
14
|
+
points, cheapest to most convenient:
|
|
15
|
+
|
|
16
|
+
- ``open_lookup`` — context manager yielding a raw ``SymbolLookup``
|
|
17
|
+
bound to one SQLite connection. Use it when you're making several
|
|
18
|
+
calls against the same repo, or when you need behaviour the
|
|
19
|
+
one-shot helpers don't expose. Connection is opened once and reused
|
|
20
|
+
for the duration of the ``async with`` block.
|
|
21
|
+
|
|
22
|
+
- ``lookup_symbol`` — one name in, one ``LookupResult`` out.
|
|
23
|
+
Metadata only: ``matches`` (id, name, type, path, line range,
|
|
24
|
+
match kind), plus ``selected`` populated *only* when there is a
|
|
25
|
+
unique exact match. Nothing is reconstructed unless the lookup
|
|
26
|
+
pipeline already had to. Use this for pick-lists, autocomplete,
|
|
27
|
+
"does this symbol exist?" probes.
|
|
28
|
+
|
|
29
|
+
- ``reconstruct_symbol`` — one ``symbol_id`` in, one ``SymbolCode``
|
|
30
|
+
out. Use this on the second half of a pick-list flow: the user
|
|
31
|
+
chose a match, now you fetch its body. Do *not* call this in a
|
|
32
|
+
loop — use ``resolve_lookup`` or ``open_lookup`` instead.
|
|
33
|
+
|
|
34
|
+
- ``resolve_lookup`` — one name in, ``ResolvedLookup`` out, which
|
|
35
|
+
carries both the ``LookupResult`` (for metadata) and a ``codes``
|
|
36
|
+
list containing a reconstructed ``SymbolCode`` for *every* match.
|
|
37
|
+
This is the answer to "``selected`` is ``None`` because there are
|
|
38
|
+
multiple matches" — instead of looping yourself, you get all the
|
|
39
|
+
bodies in one round-trip. Order of ``codes`` matches order of
|
|
40
|
+
``matches``.
|
|
41
|
+
|
|
42
|
+
Typical flows:
|
|
43
|
+
|
|
44
|
+
* *Just show me what exists* — ``lookup_symbol``.
|
|
45
|
+
* *User picked one, give me its body* — ``lookup_symbol`` then
|
|
46
|
+
``reconstruct_symbol``.
|
|
47
|
+
* *Dump every definition of this name* — ``resolve_lookup``.
|
|
48
|
+
* *I'm a long-lived tool doing many lookups* — ``open_lookup``.
|
|
49
|
+
|
|
50
|
+
All query entry points accept ``filter_paths`` to restrict work to
|
|
51
|
+
symbols under given path prefixes; the name-based ones also accept
|
|
52
|
+
``detail`` (``"body"`` or ``"signature"``), ``include_children``, and
|
|
53
|
+
``max_lines``.
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
from __future__ import annotations
|
|
57
|
+
|
|
58
|
+
import asyncio
|
|
59
|
+
from collections.abc import AsyncGenerator, AsyncIterator
|
|
60
|
+
from contextlib import asynccontextmanager
|
|
61
|
+
from pathlib import Path
|
|
62
|
+
from typing import Literal, Union, List, Tuple, Optional, Sequence
|
|
63
|
+
|
|
64
|
+
from capyidx.base.index_d import BranchAndDir
|
|
65
|
+
from capyidx.base.index_types import IndexingProgressUpdate
|
|
66
|
+
from capyidx.chunker.chunk_codebase_index import ChunkCodebaseIndex
|
|
67
|
+
from capyidx.db.db import open_index
|
|
68
|
+
from capyidx.indexer.codebase_indexer import CodeIndexer
|
|
69
|
+
from capyidx.retrieval.models import LookupResult, SymbolCode, ResolvedLookup
|
|
70
|
+
from capyidx.retrieval.retrieval_pipeline import SymbolLookup
|
|
71
|
+
from capyidx.utils.disk_operations import DiskOperations
|
|
72
|
+
from capyidx.utils.retrieval_utils import get_current_tags
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
PathLike = Union[str, Path]
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
# ---------------------------------------------------------------------------
|
|
79
|
+
# Internal setup
|
|
80
|
+
# ---------------------------------------------------------------------------
|
|
81
|
+
|
|
82
|
+
async def _setup(
|
|
83
|
+
repo: PathLike,
|
|
84
|
+
) -> Tuple[DiskOperations, List[str], BranchAndDir]:
|
|
85
|
+
"""Resolve filesystem, workspace roots, and tags for a repository.
|
|
86
|
+
"""
|
|
87
|
+
repo_path = Path(repo).resolve()
|
|
88
|
+
|
|
89
|
+
if not repo_path.exists():
|
|
90
|
+
raise FileNotFoundError(f"repo path does not exist: {repo_path}")
|
|
91
|
+
if not repo_path.is_dir():
|
|
92
|
+
raise NotADirectoryError(f"repo path is not a directory: {repo_path}")
|
|
93
|
+
|
|
94
|
+
fs = DiskOperations(roots=[str(repo_path)])
|
|
95
|
+
roots = await fs.get_workspace_dirs()
|
|
96
|
+
|
|
97
|
+
tags_list = get_current_tags(roots)
|
|
98
|
+
if not tags_list:
|
|
99
|
+
raise RuntimeError(
|
|
100
|
+
f"could not determine tags for repo: {repo_path}. "
|
|
101
|
+
f"Is it a git repository?"
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
return fs, roots, tags_list[0]
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
# ---------------------------------------------------------------------------
|
|
108
|
+
# Indexing
|
|
109
|
+
# ---------------------------------------------------------------------------
|
|
110
|
+
|
|
111
|
+
async def index_repo(
|
|
112
|
+
repo: PathLike,
|
|
113
|
+
*,
|
|
114
|
+
max_chunk_size: int = 512,
|
|
115
|
+
) -> None:
|
|
116
|
+
"""Index a repository once, waiting for completion.
|
|
117
|
+
|
|
118
|
+
Convenience wrapper around :func:`index_repo_iter` that discards
|
|
119
|
+
progress updates. Use this in scripts and CLIs where you just want
|
|
120
|
+
the index built. For progress reporting, use ``index_repo_iter``.
|
|
121
|
+
|
|
122
|
+
Args:
|
|
123
|
+
repo: Path to the repository root.
|
|
124
|
+
max_chunk_size: Approximate token budget per chunk.
|
|
125
|
+
|
|
126
|
+
Raises:
|
|
127
|
+
FileNotFoundError: If ``repo`` does not exist.
|
|
128
|
+
NotADirectoryError: If ``repo`` is not a directory.
|
|
129
|
+
RuntimeError: If ``repo`` is not a git repository.
|
|
130
|
+
"""
|
|
131
|
+
async for _ in index_repo_iter(repo, max_chunk_size=max_chunk_size):
|
|
132
|
+
pass
|
|
133
|
+
print("Indexing Completed.")
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
async def index_repo_iter(
|
|
137
|
+
repo: PathLike,
|
|
138
|
+
*,
|
|
139
|
+
watch: bool = False,
|
|
140
|
+
max_chunk_size: int = 512,
|
|
141
|
+
flush_interval: float = 5.0,
|
|
142
|
+
) -> AsyncIterator[IndexingProgressUpdate]:
|
|
143
|
+
"""Index a repository, yielding progress updates as it works.
|
|
144
|
+
|
|
145
|
+
With ``watch=False`` (default), the iterator ends when the initial
|
|
146
|
+
indexing pass completes. With ``watch=True``, after the initial pass
|
|
147
|
+
the iterator keeps yielding updates as files change. The caller stops
|
|
148
|
+
watching by breaking out of the loop or cancelling the consuming task.
|
|
149
|
+
|
|
150
|
+
Args:
|
|
151
|
+
repo: Path to the repository root.
|
|
152
|
+
watch: If True, continue watching for file changes after the
|
|
153
|
+
initial index pass.
|
|
154
|
+
max_chunk_size: Approximate token budget per chunk.
|
|
155
|
+
flush_interval: Watch-mode debounce in seconds. Ignored when
|
|
156
|
+
``watch`` is False.
|
|
157
|
+
|
|
158
|
+
Yields:
|
|
159
|
+
:class:`IndexingProgressUpdate` for each phase of work.
|
|
160
|
+
|
|
161
|
+
Example:
|
|
162
|
+
>>> async for update in index_repo_iter("/path/to/repo"):
|
|
163
|
+
... print(f"[{update.status}] {update.progress:.1%} {update.desc}")
|
|
164
|
+
|
|
165
|
+
>>> async for update in index_repo_iter("/path/to/repo", watch=True):
|
|
166
|
+
... print(f"[watch] {update.desc}")
|
|
167
|
+
"""
|
|
168
|
+
fs, roots, tags = await _setup(repo)
|
|
169
|
+
conn = open_index(tags)
|
|
170
|
+
|
|
171
|
+
try:
|
|
172
|
+
chunk_index = ChunkCodebaseIndex(
|
|
173
|
+
db=conn,
|
|
174
|
+
filesystem=fs,
|
|
175
|
+
max_chunk_size=max_chunk_size,
|
|
176
|
+
)
|
|
177
|
+
indexer = CodeIndexer(fs=fs, indexes=[chunk_index])
|
|
178
|
+
|
|
179
|
+
async for update in indexer.refresh_codebase_index(roots, conn):
|
|
180
|
+
yield update
|
|
181
|
+
print("Indexing Completed.")
|
|
182
|
+
|
|
183
|
+
if watch:
|
|
184
|
+
async for update in indexer.start_watch(
|
|
185
|
+
db=conn,
|
|
186
|
+
workspace_dirs=roots,
|
|
187
|
+
flush_interval=flush_interval
|
|
188
|
+
):
|
|
189
|
+
yield update
|
|
190
|
+
finally:
|
|
191
|
+
conn.close()
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
# ---------------------------------------------------------------------------
|
|
195
|
+
# Querying
|
|
196
|
+
# ---------------------------------------------------------------------------
|
|
197
|
+
|
|
198
|
+
@asynccontextmanager
|
|
199
|
+
async def open_lookup(
|
|
200
|
+
repo: PathLike,
|
|
201
|
+
) -> AsyncGenerator[SymbolLookup, None]:
|
|
202
|
+
"""Open a symbol lookup session for a repository.
|
|
203
|
+
|
|
204
|
+
The returned :class:`SymbolLookup` holds an open SQLite connection
|
|
205
|
+
for the duration of the ``async with`` block. Use this when you plan
|
|
206
|
+
to make multiple lookups against the same index — the connection is
|
|
207
|
+
opened once and reused.
|
|
208
|
+
|
|
209
|
+
Args:
|
|
210
|
+
repo: Path to the repository root.
|
|
211
|
+
|
|
212
|
+
Yields:
|
|
213
|
+
A :class:`SymbolLookup` bound to the repo's index.
|
|
214
|
+
|
|
215
|
+
Example:
|
|
216
|
+
>>> async with open_lookup("/path/to/repo") as lk:
|
|
217
|
+
... a = lk.lookup("SymbolLookup", detail="signature")
|
|
218
|
+
... b = lk.lookup("ChunkCodebaseIndex", detail="body")
|
|
219
|
+
"""
|
|
220
|
+
fs, roots, tags = await _setup(repo)
|
|
221
|
+
conn = open_index(tags)
|
|
222
|
+
|
|
223
|
+
try:
|
|
224
|
+
yield SymbolLookup(conn, roots)
|
|
225
|
+
finally:
|
|
226
|
+
conn.close()
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
async def lookup_symbol(
|
|
230
|
+
repo: PathLike,
|
|
231
|
+
name: str,
|
|
232
|
+
*,
|
|
233
|
+
detail: Literal["signature", "body"] = "body",
|
|
234
|
+
symbol_id: str | None = None,
|
|
235
|
+
filter_paths: Optional[Sequence[str]] = None,
|
|
236
|
+
include_children: bool = True,
|
|
237
|
+
max_lines: int = 0,
|
|
238
|
+
) -> LookupResult:
|
|
239
|
+
"""One-shot symbol lookup.
|
|
240
|
+
|
|
241
|
+
Opens the index, runs a single query, closes the index, returns the
|
|
242
|
+
result. For a batch of lookups, use :func:`open_lookup` to reuse the
|
|
243
|
+
connection across calls.
|
|
244
|
+
|
|
245
|
+
Args:
|
|
246
|
+
repo: Path to the repository root.
|
|
247
|
+
name: Symbol name to search for. Case-sensitive exact, then
|
|
248
|
+
case-insensitive exact, then case-insensitive substring.
|
|
249
|
+
detail: ``"body"`` returns the reconstructed code;
|
|
250
|
+
``"signature"`` returns only declaration headers.
|
|
251
|
+
symbol_id: If the query has multiple matches, pass the id of the
|
|
252
|
+
one to reconstruct. Ignored when there is a unique match.
|
|
253
|
+
filter_paths: Only consider symbols whose ``path`` starts with one of
|
|
254
|
+
these prefixes. ``None`` means no filtering.
|
|
255
|
+
include_children: If True, a class/file symbol also includes its
|
|
256
|
+
descendants in the reconstruction tree.
|
|
257
|
+
max_lines: Truncate the reconstructed body to this many lines.
|
|
258
|
+
``0`` means no limit.
|
|
259
|
+
|
|
260
|
+
Returns:
|
|
261
|
+
:class:`LookupResult` with ``matches``, optionally ``selected``,
|
|
262
|
+
and ``limitations``.
|
|
263
|
+
|
|
264
|
+
Example:
|
|
265
|
+
>>> result = await lookup_symbol(
|
|
266
|
+
... "/path/to/repo", "SymbolLookup", detail="signature"
|
|
267
|
+
... )
|
|
268
|
+
>>> for m in result.matches:
|
|
269
|
+
... print(f"{m.type:8} {m.name:30} {m.path}:{m.start_line}")
|
|
270
|
+
"""
|
|
271
|
+
async with open_lookup(repo) as lk:
|
|
272
|
+
return lk.lookup(name, symbol_id=symbol_id, detail=detail, filter_paths=filter_paths, include_children=include_children, max_lines=max_lines)
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
async def reconstruct_symbol(
|
|
278
|
+
repo: PathLike,
|
|
279
|
+
symbol_id: str,
|
|
280
|
+
*,
|
|
281
|
+
detail: Literal["signature", "body"] = "body",
|
|
282
|
+
filter_paths: Optional[Sequence[str]] = None,
|
|
283
|
+
) -> SymbolCode:
|
|
284
|
+
"""One-shot reconstruction of a single symbol by id.
|
|
285
|
+
|
|
286
|
+
Companion to :func:`lookup_symbol`. Use this when you already have a
|
|
287
|
+
``SymbolMatch.id`` (e.g. from a previous ``LookupResult.matches``)
|
|
288
|
+
and want just that symbol's code.
|
|
289
|
+
|
|
290
|
+
Args:
|
|
291
|
+
repo: Path to the repository root.
|
|
292
|
+
symbol_id: The ``SymbolMatch.id`` to reconstruct.
|
|
293
|
+
detail: ``"body"`` for full code, ``"signature"`` for headers.
|
|
294
|
+
filter_paths: Only consider symbols whose ``path`` starts with one
|
|
295
|
+
of these prefixes. ``None`` means no filtering.
|
|
296
|
+
|
|
297
|
+
Returns:
|
|
298
|
+
The reconstructed :class:`SymbolCode`.
|
|
299
|
+
"""
|
|
300
|
+
async with open_lookup(repo) as lk:
|
|
301
|
+
return lk.reconstruct(symbol_id, detail=detail, filter_paths=filter_paths)
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
async def resolve_lookup(
|
|
307
|
+
repo: PathLike,
|
|
308
|
+
name: str,
|
|
309
|
+
*,
|
|
310
|
+
detail: Literal["signature", "body"] = "body",
|
|
311
|
+
symbol_id: str | None = None,
|
|
312
|
+
filter_paths: Optional[Sequence[str]] = None,
|
|
313
|
+
include_children: bool = True,
|
|
314
|
+
max_lines: int = 0,
|
|
315
|
+
) -> ResolvedLookup:
|
|
316
|
+
"""Look up a symbol and reconstruct code for *every* match.
|
|
317
|
+
|
|
318
|
+
This is the fix for the "multiple matches → ``selected`` is None"
|
|
319
|
+
problem: instead of making the caller loop over ``matches`` and call
|
|
320
|
+
:meth:`SymbolLookup.reconstruct` themselves, this returns the bodies
|
|
321
|
+
in one shot.
|
|
322
|
+
|
|
323
|
+
Args:
|
|
324
|
+
repo: Path to the repository root.
|
|
325
|
+
name: Symbol name to search for. Case-sensitive exact, then
|
|
326
|
+
case-insensitive exact, then case-insensitive substring.
|
|
327
|
+
detail: ``"body"`` for full code, ``"signature"`` for headers.
|
|
328
|
+
symbol_id: The ``SymbolMatch.id`` to reconstruct.
|
|
329
|
+
filter_paths: Only consider symbols whose ``path`` starts with one
|
|
330
|
+
of these prefixes. ``None`` means no filtering.
|
|
331
|
+
include_children: If True, a class/file symbol also includes its
|
|
332
|
+
descendants in the reconstruction tree.
|
|
333
|
+
max_lines: Truncate the reconstructed body to this many lines.
|
|
334
|
+
``0`` means no limit.
|
|
335
|
+
|
|
336
|
+
Returns:
|
|
337
|
+
The reconstructed :class:`SymbolCode`.
|
|
338
|
+
|
|
339
|
+
Behaviour:
|
|
340
|
+
|
|
341
|
+
* Unique exact match → ``codes`` contains just that symbol.
|
|
342
|
+
* No unique match (multiple matches, or only case-insensitive /
|
|
343
|
+
substring hits) → ``codes`` contains the reconstruction of every
|
|
344
|
+
entry in ``result.matches``, in the same order.
|
|
345
|
+
* No matches at all → ``codes`` is empty; ``result.limitations``
|
|
346
|
+
explains why.
|
|
347
|
+
|
|
348
|
+
Individual reconstruct failures are swallowed so one broken symbol
|
|
349
|
+
doesn't nuke the whole batch — the corresponding ``SymbolMatch`` is
|
|
350
|
+
still present in ``result.matches`` for diagnostics.
|
|
351
|
+
|
|
352
|
+
Example:
|
|
353
|
+
>>> r = await resolve_lookup("/path/to/repo", "SymbolLookup")
|
|
354
|
+
>>> for code in r.codes:
|
|
355
|
+
... print(code.path, code.start_line, "→", len(code.body), "lines")
|
|
356
|
+
"""
|
|
357
|
+
async with open_lookup(repo) as lk:
|
|
358
|
+
result = lk.lookup(name=name, detail=detail, filter_paths=filter_paths, symbol_id=symbol_id, include_children=include_children, max_lines=max_lines)
|
|
359
|
+
|
|
360
|
+
codes: list[SymbolCode] = []
|
|
361
|
+
|
|
362
|
+
if result.selected is not None:
|
|
363
|
+
codes.append(result.selected)
|
|
364
|
+
|
|
365
|
+
elif symbol_id is not None:
|
|
366
|
+
codes.append(
|
|
367
|
+
await asyncio.to_thread(
|
|
368
|
+
lk.reconstruct, symbol_id=symbol_id, detail=detail, filter_paths=filter_paths
|
|
369
|
+
)
|
|
370
|
+
)
|
|
371
|
+
|
|
372
|
+
elif result.matches:
|
|
373
|
+
for match in result.matches:
|
|
374
|
+
try:
|
|
375
|
+
codes.append(
|
|
376
|
+
await asyncio.to_thread(
|
|
377
|
+
lk.reconstruct, symbol_id=match.id, detail=detail, filter_paths=filter_paths
|
|
378
|
+
)
|
|
379
|
+
)
|
|
380
|
+
except (KeyError, ValueError) as exc:
|
|
381
|
+
# Leave the match in result.matches for the caller to
|
|
382
|
+
# inspect; skip just the reconstruction.
|
|
383
|
+
continue
|
|
384
|
+
|
|
385
|
+
return ResolvedLookup(result=result, codes=codes)
|
|
386
|
+
|
|
387
|
+
|
|
388
|
+
__all__ = [
|
|
389
|
+
"index_repo",
|
|
390
|
+
"index_repo_iter",
|
|
391
|
+
"open_lookup",
|
|
392
|
+
"lookup_symbol",
|
|
393
|
+
"reconstruct_symbol",
|
|
394
|
+
"resolve_lookup",
|
|
395
|
+
"ResolvedLookup",
|
|
396
|
+
"PathLike"
|
|
397
|
+
]
|
capyidx/base/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Internal layer. Not part of the public API."""
|
capyidx/base/index_d.py
ADDED
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from enum import IntEnum
|
|
5
|
+
from typing import Any, Literal, Optional, Protocol, Tuple, List, Dict
|
|
6
|
+
from uuid import UUID
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
# --- chunks ---
|
|
10
|
+
@dataclass
|
|
11
|
+
class Symbol:
|
|
12
|
+
id: UUID
|
|
13
|
+
type: Literal["file", "class", "method"]
|
|
14
|
+
name: str
|
|
15
|
+
|
|
16
|
+
parent_id: UUID | None
|
|
17
|
+
|
|
18
|
+
children: list[UUID] = field(default_factory=list)
|
|
19
|
+
chunk_ids: list[UUID] = field(default_factory=list)
|
|
20
|
+
|
|
21
|
+
start_line: int = 0
|
|
22
|
+
end_line: int = 0
|
|
23
|
+
|
|
24
|
+
filepath: Optional[str] = None
|
|
25
|
+
cache_key: Optional[str] = None
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class Chonk:
|
|
30
|
+
content: str
|
|
31
|
+
|
|
32
|
+
start_line: int
|
|
33
|
+
end_line: int
|
|
34
|
+
id: Optional[UUID | None] = None
|
|
35
|
+
symbol_id: Optional[UUID | None] = None
|
|
36
|
+
|
|
37
|
+
piece_index: Optional[int | None] = None
|
|
38
|
+
piece_count: Optional[int | None] = None
|
|
39
|
+
|
|
40
|
+
prev_chunk: Optional[UUID | None] = None
|
|
41
|
+
next_chunk: Optional[UUID | None] = None
|
|
42
|
+
|
|
43
|
+
signature: Optional[str] = None
|
|
44
|
+
|
|
45
|
+
@dataclass
|
|
46
|
+
class Chunk(Chonk):
|
|
47
|
+
digest: str = ""
|
|
48
|
+
filepath: str = ""
|
|
49
|
+
index: int = 0
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass
|
|
53
|
+
class ChunkingResult:
|
|
54
|
+
symbols: list[Symbol] = field(default_factory=list)
|
|
55
|
+
chunks: list[Chonk] = field(default_factory=list)
|
|
56
|
+
symbol_map: dict[UUID, Symbol] = field(default_factory=dict)
|
|
57
|
+
|
|
58
|
+
@dataclass
|
|
59
|
+
class ChunkWithoutID:
|
|
60
|
+
content: str
|
|
61
|
+
start_line: int
|
|
62
|
+
end_line: int
|
|
63
|
+
signature: Optional[str] = None
|
|
64
|
+
other_metadata: Optional[dict[str, Any]] = None
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
# --- indexing progress ---
|
|
70
|
+
|
|
71
|
+
IndexingStatus = Literal[
|
|
72
|
+
"loading",
|
|
73
|
+
"waiting",
|
|
74
|
+
"indexing",
|
|
75
|
+
"done",
|
|
76
|
+
"failed",
|
|
77
|
+
"paused",
|
|
78
|
+
"disabled",
|
|
79
|
+
"cancelled",
|
|
80
|
+
]
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@dataclass
|
|
84
|
+
class IndexingProgressUpdate:
|
|
85
|
+
progress: float
|
|
86
|
+
desc: str
|
|
87
|
+
status: IndexingStatus
|
|
88
|
+
should_clear_indexes: Optional[bool] = None
|
|
89
|
+
debug_info: Optional[str] = None
|
|
90
|
+
warnings: Optional[list[str]] = None
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
# --- which indexes to run ---
|
|
94
|
+
|
|
95
|
+
ContextIndexingType = Literal[
|
|
96
|
+
"chunk",
|
|
97
|
+
"embeddings",
|
|
98
|
+
"full_text_search",
|
|
99
|
+
"code_snippets",
|
|
100
|
+
]
|
|
101
|
+
|
|
102
|
+
# -- fts --
|
|
103
|
+
|
|
104
|
+
@dataclass(frozen=True)
|
|
105
|
+
class BranchAndDir:
|
|
106
|
+
directory: str
|
|
107
|
+
branch: str
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
@dataclass
|
|
111
|
+
class RetrieveConfig:
|
|
112
|
+
tags: list[BranchAndDir]
|
|
113
|
+
text: str
|
|
114
|
+
n: int
|
|
115
|
+
directory: Optional[str] = None
|
|
116
|
+
filter_paths: Optional[list[str]] = None
|
|
117
|
+
bm25_threshold: Optional[float] = None
|
|
118
|
+
|
|
119
|
+
# --- positions (snippets / ranges) ---
|
|
120
|
+
|
|
121
|
+
@dataclass
|
|
122
|
+
class Position:
|
|
123
|
+
line: int
|
|
124
|
+
character: int
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
@dataclass
|
|
128
|
+
class Range:
|
|
129
|
+
start: RangePosition
|
|
130
|
+
end: RangePosition
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
@dataclass
|
|
134
|
+
class RangeInFile:
|
|
135
|
+
filepath: str
|
|
136
|
+
range: Range
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
# --- tag = (workspace, branch, artifact) ---
|
|
140
|
+
|
|
141
|
+
@dataclass(frozen=True)
|
|
142
|
+
class IndexTag:
|
|
143
|
+
directory: str
|
|
144
|
+
branch: str
|
|
145
|
+
artifact_id: str
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
class FileType(IntEnum):
|
|
149
|
+
UNKNOWN = 0
|
|
150
|
+
FILE = 1
|
|
151
|
+
DIRECTORY = 2
|
|
152
|
+
SYMBOLIC_LINK = 64
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
@dataclass(frozen=True)
|
|
156
|
+
class FileStats:
|
|
157
|
+
size: int
|
|
158
|
+
last_modified: int
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
@dataclass
|
|
162
|
+
class RangePosition:
|
|
163
|
+
line: int
|
|
164
|
+
character: int
|
|
165
|
+
|
|
166
|
+
@dataclass
|
|
167
|
+
class SymbolWithRange:
|
|
168
|
+
filepath: str
|
|
169
|
+
type: str
|
|
170
|
+
name: str
|
|
171
|
+
range: Range
|
|
172
|
+
content: str
|
|
173
|
+
|
|
174
|
+
FileSymbolMap = Dict[str, List[SymbolWithRange]]
|
|
175
|
+
|
|
176
|
+
FileStatsMap = dict[str, FileStats]
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
# --- filesystem the indexer actually calls (replaces IDE) ---
|
|
180
|
+
|
|
181
|
+
class FileSystem(Protocol):
|
|
182
|
+
async def read_file(self, uri: str) -> str: ...
|
|
183
|
+
|
|
184
|
+
async def file_exists(self, uri: str) -> bool: ...
|
|
185
|
+
|
|
186
|
+
async def list_dir(self, uri: str) -> List[Tuple[str, FileType]]: ...
|
|
187
|
+
|
|
188
|
+
async def get_file_stats(self, uris: list[str]) -> FileStatsMap: ...
|
|
189
|
+
|
|
190
|
+
async def get_workspace_dirs(self) -> list[str]: ...
|
|
191
|
+
|
|
192
|
+
async def get_branch(self, directory_uri: str) -> str: ...
|
|
193
|
+
|
|
194
|
+
async def get_repo_name(self, directory_uri: str) -> Optional[str]: ...
|