java-codebase-rag 0.8.0__py3-none-any.whl → 0.9.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- java_codebase_rag/cli.py +14 -1
- java_codebase_rag/config.py +27 -3
- java_codebase_rag/install_data/agents/explorer-rag-cli.md +65 -208
- java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +78 -232
- java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +44 -83
- java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +67 -135
- java_codebase_rag/installer.py +310 -32
- java_codebase_rag/jrag.py +112 -7
- java_codebase_rag/jrag_envelope.py +1 -1
- java_codebase_rag/jrag_render.py +12 -3
- java_codebase_rag/lance_optimize.py +43 -0
- java_codebase_rag/pipeline.py +34 -0
- {java_codebase_rag-0.8.0.dist-info → java_codebase_rag-0.9.1.dist-info}/METADATA +2 -2
- {java_codebase_rag-0.8.0.dist-info → java_codebase_rag-0.9.1.dist-info}/RECORD +22 -22
- java_index_flow_lancedb.py +66 -6
- mcp_v2.py +82 -26
- search_lancedb.py +149 -3
- server.py +36 -0
- {java_codebase_rag-0.8.0.dist-info → java_codebase_rag-0.9.1.dist-info}/WHEEL +0 -0
- {java_codebase_rag-0.8.0.dist-info → java_codebase_rag-0.9.1.dist-info}/entry_points.txt +0 -0
- {java_codebase_rag-0.8.0.dist-info → java_codebase_rag-0.9.1.dist-info}/licenses/LICENSE +0 -0
- {java_codebase_rag-0.8.0.dist-info → java_codebase_rag-0.9.1.dist-info}/top_level.txt +0 -0
|
@@ -184,6 +184,49 @@ async def optimize_lance_tables(
|
|
|
184
184
|
|
|
185
185
|
if last_exc is None:
|
|
186
186
|
results[name] = "ok"
|
|
187
|
+
# Best-effort BTREE scalar index on the primary key ("id").
|
|
188
|
+
# cocoindex's merge_insert defaults to use_index=True but
|
|
189
|
+
# never creates a scalar PK index itself (declaring
|
|
190
|
+
# primary_key in the schema does NOT auto-build a lance
|
|
191
|
+
# index), so without this every merge_insert — increment
|
|
192
|
+
# included — is a forced full scan of the PK column,
|
|
193
|
+
# O(existing rows). On a large repo that scan dominates
|
|
194
|
+
# increment wall-clock; with the index present the join does
|
|
195
|
+
# lookups (~O(batch*log N)). Failure is non-fatal (the table
|
|
196
|
+
# is still correct, just un-indexed) and never alters the
|
|
197
|
+
# "ok" status, mirroring the FTS block below. ``replace=True``
|
|
198
|
+
# keeps it idempotent across runs; table.optimize() above
|
|
199
|
+
# maintains it on subsequent runs.
|
|
200
|
+
try:
|
|
201
|
+
from lancedb.index import BTree
|
|
202
|
+
await table.create_index("id", config=BTree(), replace=True)
|
|
203
|
+
except Exception as exc:
|
|
204
|
+
low = str(exc).lower()
|
|
205
|
+
if not any(
|
|
206
|
+
w in low for w in ("exist", "duplicate", "already", "same name")
|
|
207
|
+
) and not quiet:
|
|
208
|
+
print(
|
|
209
|
+
f"java-codebase-rag: optimize: {name} id-index skipped: {exc}",
|
|
210
|
+
file=sys.stderr,
|
|
211
|
+
)
|
|
212
|
+
# Best-effort FTS index at index time (PR-SEARCH-3) so hybrid
|
|
213
|
+
# search works on all tables (java/sql/yaml) without a
|
|
214
|
+
# first-query race. Failure is non-fatal — the lazy
|
|
215
|
+
# ensure_text_fts_index in search_lancedb.py is the runtime
|
|
216
|
+
# fallback — so it never alters the "ok" optimize status; we
|
|
217
|
+
# only log the skip when verbose.
|
|
218
|
+
try:
|
|
219
|
+
from lancedb.index import FTS
|
|
220
|
+
await table.create_index("text", config=FTS(), replace=True)
|
|
221
|
+
except Exception as exc:
|
|
222
|
+
low = str(exc).lower()
|
|
223
|
+
if not any(
|
|
224
|
+
w in low for w in ("exist", "duplicate", "already", "same name")
|
|
225
|
+
) and not quiet:
|
|
226
|
+
print(
|
|
227
|
+
f"java-codebase-rag: optimize: {name} fts skipped: {exc}",
|
|
228
|
+
file=sys.stderr,
|
|
229
|
+
)
|
|
187
230
|
if not quiet:
|
|
188
231
|
print(
|
|
189
232
|
f"java-codebase-rag: optimize: {name} ok",
|
java_codebase_rag/pipeline.py
CHANGED
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
from __future__ import annotations
|
|
3
3
|
|
|
4
4
|
import asyncio
|
|
5
|
+
import importlib.util
|
|
5
6
|
import os
|
|
6
7
|
import shutil
|
|
7
8
|
import subprocess
|
|
@@ -131,6 +132,26 @@ def run_cocoindex_update(
|
|
|
131
132
|
on_progress: Callable[[ProgressEvent], None] | None = None,
|
|
132
133
|
on_progress_console: object | None = None,
|
|
133
134
|
) -> subprocess.CompletedProcess[str]:
|
|
135
|
+
if full_reprocess:
|
|
136
|
+
# A full reprocess rebuilds every row, so DROP the Lance target tables
|
|
137
|
+
# first and let cocoindex recreate them via the fast INSERT path. The
|
|
138
|
+
# in-place alternative (cocoindex's bulk-update merge_insert) emits
|
|
139
|
+
# ~one deletion-vector + version commit PER matched row — O(rows) of
|
|
140
|
+
# tiny file IO that scales to multi-minute hangs on large repos
|
|
141
|
+
# (measured ~83s sys time / 3474 deletion files for 3475 chunks on
|
|
142
|
+
# Shopizer; drop+recreate is ~3.6s sys / 0 deletions, ~3.7x faster and
|
|
143
|
+
# hang-free). Output is identical either way (full recompute); only the
|
|
144
|
+
# write path differs. Drop failure is non-fatal — if it somehow fails,
|
|
145
|
+
# the update falls back to the slow in-place path. The same fix is
|
|
146
|
+
# applied on the async server path (``server.run_refresh_pipeline``).
|
|
147
|
+
drop = run_cocoindex_drop(env, quiet=quiet)
|
|
148
|
+
if drop.returncode != 0 and not is_cocoindex_preflight_blocker(drop):
|
|
149
|
+
print(
|
|
150
|
+
"java-codebase-rag: drop-before-reprocess failed "
|
|
151
|
+
f"(exit {drop.returncode}); falling back to in-place update: "
|
|
152
|
+
f"{(drop.stderr or '').strip()[:200]}",
|
|
153
|
+
file=sys.stderr,
|
|
154
|
+
)
|
|
134
155
|
result = _run_cocoindex_update_impl(
|
|
135
156
|
env,
|
|
136
157
|
full_reprocess=full_reprocess,
|
|
@@ -200,6 +221,19 @@ def is_graph_preflight_blocker(proc: subprocess.CompletedProcess[str]) -> bool:
|
|
|
200
221
|
return bool(proc.returncode in (126, 127) and len(getattr(proc, "args", ()) or ()) <= 1)
|
|
201
222
|
|
|
202
223
|
|
|
224
|
+
def vector_stack_installed() -> bool:
|
|
225
|
+
"""True when the optional vector stack (cocoindex/lancedb/sentence-transformers) is importable.
|
|
226
|
+
|
|
227
|
+
False on graph-only installs (macOS Intel), where PEP 508 markers exclude the trio.
|
|
228
|
+
Used to skip vector-only wizard steps (e.g. embedding-model selection) and to preflight
|
|
229
|
+
branching without spawning cocoindex. Probes all three since they are gated together.
|
|
230
|
+
"""
|
|
231
|
+
return all(
|
|
232
|
+
importlib.util.find_spec(m) is not None
|
|
233
|
+
for m in ("cocoindex", "lancedb", "sentence_transformers")
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
|
|
203
237
|
def _run_cocoindex_update_impl(
|
|
204
238
|
env: dict[str, str],
|
|
205
239
|
*,
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: java-codebase-rag
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.9.1
|
|
4
4
|
Summary: MCP server for semantic + structural search over Java codebases
|
|
5
5
|
Author: HumanBean17
|
|
6
6
|
License-Expression: MIT
|
|
@@ -122,7 +122,7 @@ If you prefer manual configuration, see [`docs/JAVA-CODEBASE-RAG-CLI.md`](./docs
|
|
|
122
122
|
|
|
123
123
|
## Tools & commands at a glance
|
|
124
124
|
|
|
125
|
-
Pick a surface
|
|
125
|
+
Pick a surface at install time — `java-codebase-rag install --surface mcp|cli` (default `cli`, recommended). Both surfaces walk the same LanceDB vectors + LadybugDB graph. Switch an existing install later with `java-codebase-rag update --surface mcp|cli`.
|
|
126
126
|
|
|
127
127
|
**MCP surface — five tools over stdio**
|
|
128
128
|
|
|
@@ -5,39 +5,39 @@ chunk_heuristics.py,sha256=aQk2NOKxzUdqoUAJUO3G3LE0MN_bYZWNLQ0tkmj5uts,1813
|
|
|
5
5
|
graph_enrich.py,sha256=Fxp7JurH9wWTOTX1nErjH4H6Jw5bB68YCqR17rsX6gw,62829
|
|
6
6
|
graph_types.py,sha256=P6RVdEiqZlaSgTyXepnGbtMoZq0m_MaP2q6M6i7Uxlc,4539
|
|
7
7
|
index_common.py,sha256=HT6FKHFJ084eFvd3fR1j8z8gf4eWoPHVW8GXLpw464I,285
|
|
8
|
-
java_index_flow_lancedb.py,sha256=
|
|
8
|
+
java_index_flow_lancedb.py,sha256=kDkhnRxl-Kz0YNMY27DTk3gwa8H2nWyGDCQC8hxx-jI,28970
|
|
9
9
|
java_index_v1_common.py,sha256=nF1KrSqboF_RRvWerG9knRRFmWwsrG_CvhgnsoZ8KqA,1154
|
|
10
10
|
java_ontology.py,sha256=ooqr8GucOINpzhdEQ3QzVe5A9GfiR0nUTlySDehn9GA,17129
|
|
11
11
|
ladybug_queries.py,sha256=rVnVEHwWwE4USeX7tICYEl1SSiSxJGnBNe1VX66R3Xk,100531
|
|
12
12
|
mcp_hints.py,sha256=zp-4cnOmbYD0YovmZiLS2oGcvWcWE7n8jKVz4_xifno,42512
|
|
13
|
-
mcp_v2.py,sha256=
|
|
13
|
+
mcp_v2.py,sha256=ABNHiZEEwQ76uPF9E5fQWncJ4JmmgtG3vmt5TDxBGVo,68901
|
|
14
14
|
path_filtering.py,sha256=R--XzI51LXBu5IBKMCnJWbkNr6I5d-SDmltyQQnWco0,17674
|
|
15
15
|
pr_analysis.py,sha256=zrmZZD5yotJtM02Kif6_jgI_oeformOao793akp0N6Y,18394
|
|
16
16
|
resolve_service.py,sha256=tC5FQsGmqhqn0EOexVDlRq5egnzKDTrI7CMzk0nPpG8,25135
|
|
17
|
-
search_lancedb.py,sha256=
|
|
18
|
-
server.py,sha256=
|
|
17
|
+
search_lancedb.py,sha256=1sGSZ6H8J9hGcKAwHpzB7F0_in3A_sEtOI5LvlZYIRI,43864
|
|
18
|
+
server.py,sha256=nOK3DOr-i3PJnVmE7neye_tptJumgXfIXPlTIn01MPQ,37065
|
|
19
19
|
java_codebase_rag/__init__.py,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
|
|
20
20
|
java_codebase_rag/_fdlimit.py,sha256=vkwjsPbZfxzZ2DZTPWO5DxtuNlLzOADzIq07iYX7GCU,2465
|
|
21
21
|
java_codebase_rag/_stdio.py,sha256=TDNbpt2EP0_Zd622ihdlKwlMfxkKHWOfgLVcU6TcNbo,1458
|
|
22
|
-
java_codebase_rag/cli.py,sha256=
|
|
22
|
+
java_codebase_rag/cli.py,sha256=8WKk-_Zl1Sp7wHX4Y5f3bkIy8hHRH_QWrNuX_3WYTm4,45269
|
|
23
23
|
java_codebase_rag/cli_format.py,sha256=CT7-xdwZ0bMCdP68_UOwkvm-mnLluU3LutlM-mDNk60,1839
|
|
24
24
|
java_codebase_rag/cli_progress.py,sha256=q6Wh97yzLGs1B8UFk_WAKivfQu7Y5RnUUE-T2YHWkIs,3237
|
|
25
|
-
java_codebase_rag/config.py,sha256=
|
|
26
|
-
java_codebase_rag/installer.py,sha256=
|
|
27
|
-
java_codebase_rag/jrag.py,sha256=
|
|
28
|
-
java_codebase_rag/jrag_envelope.py,sha256=
|
|
25
|
+
java_codebase_rag/config.py,sha256=1EAlKtQx7LUo-gPoE6BOVLz0YEl1OpWFRABkF06zjgk,26464
|
|
26
|
+
java_codebase_rag/installer.py,sha256=c-_tR1Ct_O3yhmruFRnDoM1Wilz-igD9qdkDPsNqLMI,79934
|
|
27
|
+
java_codebase_rag/jrag.py,sha256=cVUWrKOtkwe2r4u1nknzmrMQMy16bdt1n-dZQpiSFWs,191811
|
|
28
|
+
java_codebase_rag/jrag_envelope.py,sha256=5jD3p2O-p9acAKHqoif5FFoqgP7Q2ziCuKSiPCHoywc,47505
|
|
29
29
|
java_codebase_rag/jrag_hints.py,sha256=k2PFE4s3lZgBYHMdZcTjx1-w28nfQcBtQEVsSxI_DvE,9262
|
|
30
|
-
java_codebase_rag/jrag_render.py,sha256=
|
|
31
|
-
java_codebase_rag/lance_optimize.py,sha256=
|
|
32
|
-
java_codebase_rag/pipeline.py,sha256=
|
|
30
|
+
java_codebase_rag/jrag_render.py,sha256=1nUyamL-MOOXDlKvssp-LsBgEtznUnGj_cPteSwuS90,31953
|
|
31
|
+
java_codebase_rag/lance_optimize.py,sha256=HI3aFebP1fenLL6Cav1jMG5kLXSrHLndO_MIJY6qQVo,11977
|
|
32
|
+
java_codebase_rag/pipeline.py,sha256=L65mjK-IxkVWazUNVyIvzfQNi24behQu-Kdc7o_HwEk,17754
|
|
33
33
|
java_codebase_rag/progress.py,sha256=2IxdMALDM0wAQCyJrrfZ975zM_85C-4BfHxf4AtYifE,23212
|
|
34
|
-
java_codebase_rag/install_data/agents/explorer-rag-cli.md,sha256=
|
|
35
|
-
java_codebase_rag/install_data/agents/explorer-rag-enhanced.md,sha256=
|
|
36
|
-
java_codebase_rag/install_data/skills/explore-codebase/SKILL.md,sha256=
|
|
37
|
-
java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md,sha256=
|
|
38
|
-
java_codebase_rag-0.
|
|
39
|
-
java_codebase_rag-0.
|
|
40
|
-
java_codebase_rag-0.
|
|
41
|
-
java_codebase_rag-0.
|
|
42
|
-
java_codebase_rag-0.
|
|
43
|
-
java_codebase_rag-0.
|
|
34
|
+
java_codebase_rag/install_data/agents/explorer-rag-cli.md,sha256=mMij_BIQM4agaYhGVYjC-fQSe3We1HFeBvc5JBjJj6A,10071
|
|
35
|
+
java_codebase_rag/install_data/agents/explorer-rag-enhanced.md,sha256=gZsNFbuK0lSnOIlplbbS_muz2ozokJqvFUv65QM0NDM,10406
|
|
36
|
+
java_codebase_rag/install_data/skills/explore-codebase/SKILL.md,sha256=A-v2dueVnxwBzBlxoRjZ2zOJk8DranLQ1TElwn94h0s,11529
|
|
37
|
+
java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md,sha256=V5gIKKGkgk2KlAFf9JqPareQDMt1iQOI7cwhfOqJL1c,11348
|
|
38
|
+
java_codebase_rag-0.9.1.dist-info/licenses/LICENSE,sha256=gxvtiHtuviR_q8ZAjWw-QTcF3DyPzg6ZY-lQrr8OPpw,1068
|
|
39
|
+
java_codebase_rag-0.9.1.dist-info/METADATA,sha256=v3Zl2nJ80PhrygxQD-aojw6z1CSZ_uRrOdkIWuA65No,20088
|
|
40
|
+
java_codebase_rag-0.9.1.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
|
|
41
|
+
java_codebase_rag-0.9.1.dist-info/entry_points.txt,sha256=cj3QTc11UYVQnj9T3orc4daiIGaCYrXP149vKbH2R4U,168
|
|
42
|
+
java_codebase_rag-0.9.1.dist-info/top_level.txt,sha256=8vC-VN3cMwz5vhkSTaeJ1a1bDeqLWEfrTks1CvEvIg0,273
|
|
43
|
+
java_codebase_rag-0.9.1.dist-info/RECORD,,
|
java_index_flow_lancedb.py
CHANGED
|
@@ -112,6 +112,38 @@ _NUM_TXN_BEFORE_OPTIMIZE = 10**12
|
|
|
112
112
|
# parent clamps to total on the terminal event anyway).
|
|
113
113
|
_VECTORS_TICK_EVERY = 25
|
|
114
114
|
|
|
115
|
+
# Bounded concurrency for the per-file drain in app_main. cocoindex's embedder
|
|
116
|
+
# is ``@coco.fn.as_async(batching=True, runner=GPU, max_batch_size=64)`` — but
|
|
117
|
+
# its batching layer only coalesces calls that are in flight SIMULTANEOUSLY. A
|
|
118
|
+
# serial ``async for … await`` loop keeps just one file's chunks (avg 1–3) in
|
|
119
|
+
# flight, so real batches stay tiny and MPS idles between them (measured ~138
|
|
120
|
+
# chunks/s vs the ~235 chunks/s ceiling at batch=64 for all-MiniLM-L6-v2).
|
|
121
|
+
# Draining many files at once with a semaphore puts their chunks in flight
|
|
122
|
+
# together → the embedder coalesces them into full batches → MPS climbs toward
|
|
123
|
+
# the ceiling. Measured on Shopizer (1167 files / 3475 chunks): full init drops
|
|
124
|
+
# from ~46.7s (serial) to ~36.0s (32) / ~34.3s (64), with identical row output.
|
|
125
|
+
#
|
|
126
|
+
# This stays inside ONE component, so the earlier mount_each→app_main win is
|
|
127
|
+
# preserved: still exactly ONE merge_insert per table at commit. Memoization
|
|
128
|
+
# (``@coco.fn(memo=True)``) and the lock-guarded tick counter are safe under
|
|
129
|
+
# concurrency; ``parse_java`` uses a per-thread tree-sitter Parser (already
|
|
130
|
+
# routed via ``asyncio.to_thread``) and ``splitter.split`` is synchronous so the
|
|
131
|
+
# event loop cannot reenter it.
|
|
132
|
+
#
|
|
133
|
+
# Default 64 is sized to MATCH the embedder's hardcoded max_batch_size=64 (the
|
|
134
|
+
# decorator above; not a constructor arg, so not raisable from the flow): ~64
|
|
135
|
+
# files in flight reliably fills a 64-chunk batch and saturates MPS. Going higher
|
|
136
|
+
# buys nothing — the batch is already capped — and lower underfills it. Memory
|
|
137
|
+
# is NOT the limiting factor here: cocoindex buffers ALL staged rows until the
|
|
138
|
+
# single final merge_insert regardless of concurrency, so peak RSS is set by
|
|
139
|
+
# total chunk count (the commit buffer), not by how many files process at once.
|
|
140
|
+
# Set to ``1`` for the old serial behavior; raise/lower only if you have also
|
|
141
|
+
# changed the effective batch size or are constraining the commit buffer itself.
|
|
142
|
+
_FILE_CONCURRENCY = max(
|
|
143
|
+
1,
|
|
144
|
+
int(os.environ.get("JAVA_CODEBASE_RAG_FILE_CONCURRENCY", "64") or "64"),
|
|
145
|
+
)
|
|
146
|
+
|
|
115
147
|
# Thread-safe counter: cocoindex may call process_*_file concurrently
|
|
116
148
|
# (mount_each parallelism is implementation-defined). A module-level lock guards
|
|
117
149
|
# both the counter and the emission so two threads never interleave a tick.
|
|
@@ -533,6 +565,29 @@ async def process_yaml_file(
|
|
|
533
565
|
)
|
|
534
566
|
|
|
535
567
|
|
|
568
|
+
async def _drain_files_concurrently(
|
|
569
|
+
files: Any, process_fn: Any, table: Any, sem: asyncio.Semaphore
|
|
570
|
+
) -> None:
|
|
571
|
+
"""Run ``process_fn(file, table)`` over every file with bounded concurrency.
|
|
572
|
+
|
|
573
|
+
Replaces the serial ``async for … await process_*_file`` loop so the
|
|
574
|
+
embedder's batching layer sees many files' chunks in flight at once (see
|
|
575
|
+
``_FILE_CONCURRENCY``). Materializes the async iterable up front — file
|
|
576
|
+
handles are lightweight and cocoindex already realized the collection when
|
|
577
|
+
the walker mounted, so this is not a second walk. An empty collection is a
|
|
578
|
+
no-op (e.g. SQL/YAML tables on a repo with none).
|
|
579
|
+
"""
|
|
580
|
+
items = [f async for _, f in files.items()]
|
|
581
|
+
if not items:
|
|
582
|
+
return
|
|
583
|
+
|
|
584
|
+
async def _one(_file: Any) -> None:
|
|
585
|
+
async with sem:
|
|
586
|
+
await process_fn(_file, table)
|
|
587
|
+
|
|
588
|
+
await asyncio.gather(*(_one(f) for f in items))
|
|
589
|
+
|
|
590
|
+
|
|
536
591
|
@coco.fn
|
|
537
592
|
async def app_main() -> None:
|
|
538
593
|
java_schema = await lancedb.TableSchema.from_class(
|
|
@@ -646,12 +701,17 @@ async def app_main() -> None:
|
|
|
646
701
|
# unchanged files still skip re-embedding on incremental; _RowHandler.
|
|
647
702
|
# reconcile skips rows whose fingerprint is unchanged → increment carries
|
|
648
703
|
# only changed rows in its single merge_insert.
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
async for
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
704
|
+
#
|
|
705
|
+
# PERF (concurrency): drain files with a bounded semaphore instead of a
|
|
706
|
+
# serial ``async for … await``. See ``_FILE_CONCURRENCY`` — this is what
|
|
707
|
+
# lets the embedder's batching layer fill real batches (embedding dominates
|
|
708
|
+
# init cost, and serial files starve it). One shared semaphore bounds total
|
|
709
|
+
# in-flight work; tables are drained in order (java dominates, sql/yaml are
|
|
710
|
+
# usually near-empty).
|
|
711
|
+
_sem = asyncio.Semaphore(_FILE_CONCURRENCY)
|
|
712
|
+
await _drain_files_concurrently(java_files, process_java_file, java_table, _sem)
|
|
713
|
+
await _drain_files_concurrently(sql_files, process_sql_file, sql_table, _sem)
|
|
714
|
+
await _drain_files_concurrently(yaml_files, process_yaml_file, yaml_table, _sem)
|
|
655
715
|
|
|
656
716
|
|
|
657
717
|
app = coco.App(
|
mcp_v2.py
CHANGED
|
@@ -466,6 +466,8 @@ class SearchHit(BaseModel):
|
|
|
466
466
|
role: str | None = None
|
|
467
467
|
filename: str | None = None
|
|
468
468
|
start_line: int | None = None
|
|
469
|
+
score_components: dict[str, float] | None = None
|
|
470
|
+
chunks: int | None = None
|
|
469
471
|
|
|
470
472
|
|
|
471
473
|
# NodeRef is now defined in graph_types.py and imported above
|
|
@@ -583,7 +585,7 @@ def _chunk_id_from_row(row: dict[str, Any]) -> str:
|
|
|
583
585
|
return f"{filename}:{sb}:{eb}"
|
|
584
586
|
|
|
585
587
|
|
|
586
|
-
def _row_to_search_hit(row: dict[str, Any]) -> SearchHit:
|
|
588
|
+
def _row_to_search_hit(row: dict[str, Any], explain: bool = False) -> SearchHit:
|
|
587
589
|
score = float(row.get("_rrf_score") or row.get("_score") or 0.0)
|
|
588
590
|
filename = str(row.get("filename") or "") or None
|
|
589
591
|
start_line: int | None = None
|
|
@@ -595,6 +597,8 @@ def _row_to_search_hit(row: dict[str, Any]) -> SearchHit:
|
|
|
595
597
|
start_line = int(ln)
|
|
596
598
|
except (TypeError, ValueError):
|
|
597
599
|
start_line = None
|
|
600
|
+
chunks = row.get("_chunks_collapsed")
|
|
601
|
+
chunks_int = int(chunks) if chunks is not None and int(chunks) >= 2 else None
|
|
598
602
|
return SearchHit(
|
|
599
603
|
chunk_id=_chunk_id_from_row(row),
|
|
600
604
|
symbol_id=_chunk_to_symbol_id(row),
|
|
@@ -606,6 +610,8 @@ def _row_to_search_hit(row: dict[str, Any]) -> SearchHit:
|
|
|
606
610
|
role=str(row.get("role")) if row.get("role") else None,
|
|
607
611
|
filename=filename,
|
|
608
612
|
start_line=start_line,
|
|
613
|
+
score_components=row.get("_score_components") if explain else None,
|
|
614
|
+
chunks=chunks_int,
|
|
609
615
|
)
|
|
610
616
|
|
|
611
617
|
|
|
@@ -820,7 +826,9 @@ def search_v2(
|
|
|
820
826
|
offset: int = 0,
|
|
821
827
|
path_contains: str | None = None,
|
|
822
828
|
filter: NodeFilter | dict[str, Any] | str | None = None,
|
|
829
|
+
explain: bool = False,
|
|
823
830
|
graph: LadybugGraph | None = None,
|
|
831
|
+
dedup: bool = True,
|
|
824
832
|
) -> SearchOutput:
|
|
825
833
|
try:
|
|
826
834
|
raw_filter = _coerce_filter(filter)
|
|
@@ -852,6 +860,17 @@ def search_v2(
|
|
|
852
860
|
limit=None,
|
|
853
861
|
offset=None,
|
|
854
862
|
)
|
|
863
|
+
# hybrid + table='all' is unsupported (hybrid fuses vector+FTS on ONE
|
|
864
|
+
# table); fail fast with a clean envelope BEFORE loading the embedding
|
|
865
|
+
# model. run_search also guards this — this is the user-facing fast path.
|
|
866
|
+
if hybrid and table == "all":
|
|
867
|
+
return SearchOutput(
|
|
868
|
+
success=False,
|
|
869
|
+
message="hybrid search requires a single table; use java, sql, or yaml (not all)",
|
|
870
|
+
advisories=[],
|
|
871
|
+
limit=None,
|
|
872
|
+
offset=None,
|
|
873
|
+
)
|
|
855
874
|
model_name = resolved_sbert_model_for_process_env(SBERT_MODEL)
|
|
856
875
|
device = os.environ.get("SBERT_DEVICE") or None
|
|
857
876
|
model = _get_sentence_transformer(model_name, device)
|
|
@@ -862,29 +881,66 @@ def search_v2(
|
|
|
862
881
|
if not uri.startswith(("s3://", "gs://", "az://")) and uri_path.exists():
|
|
863
882
|
uri = str(uri_path.resolve())
|
|
864
883
|
table_keys = list(TABLES) if table == "all" else [table]
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
|
|
875
|
-
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
|
|
881
|
-
|
|
882
|
-
|
|
883
|
-
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
884
|
+
|
|
885
|
+
# Graceful fallback: if hybrid=True and FTS index is missing (old index),
|
|
886
|
+
# retry with hybrid=False and return vector-only results with an advisory.
|
|
887
|
+
advisories: list[str] = []
|
|
888
|
+
try:
|
|
889
|
+
rows = run_search(
|
|
890
|
+
query,
|
|
891
|
+
uri=uri,
|
|
892
|
+
table_keys=table_keys,
|
|
893
|
+
hybrid=hybrid,
|
|
894
|
+
limit=limit,
|
|
895
|
+
offset=offset,
|
|
896
|
+
path_substring=path_contains,
|
|
897
|
+
model_name=model_name,
|
|
898
|
+
device=device,
|
|
899
|
+
model=model,
|
|
900
|
+
# Push the NodeFilter structural predicates into the LanceDB query so
|
|
901
|
+
# they apply BEFORE pagination (issue #353) — previously they were only
|
|
902
|
+
# a post-filter on the already-paginated page, which could shrink or
|
|
903
|
+
# empty filtered pages even when many matches existed deeper in the
|
|
904
|
+
# ranking. _node_matches_filter below still re-checks every row (it
|
|
905
|
+
# covers the non-pushdownable fields and is the contract guarantee).
|
|
906
|
+
role=nf.role if nf else None,
|
|
907
|
+
module=nf.module if nf else None,
|
|
908
|
+
microservice=nf.microservice if nf else None,
|
|
909
|
+
capability=nf.capability if nf else None,
|
|
910
|
+
exclude_roles=nf.exclude_roles if nf else None,
|
|
911
|
+
dedup_by_fqn=dedup,
|
|
912
|
+
)
|
|
913
|
+
except Exception as exc:
|
|
914
|
+
# Check if this is a missing-FTS error (old index built before PR-SEARCH-3)
|
|
915
|
+
exc_text = str(exc).lower()
|
|
916
|
+
is_fts_missing = "full text search" in exc_text or "inverted index" in exc_text
|
|
917
|
+
if hybrid and is_fts_missing:
|
|
918
|
+
# Retry with vector-only search
|
|
919
|
+
rows = run_search(
|
|
920
|
+
query,
|
|
921
|
+
uri=uri,
|
|
922
|
+
table_keys=table_keys,
|
|
923
|
+
hybrid=False, # Fallback to vector-only
|
|
924
|
+
limit=limit,
|
|
925
|
+
offset=offset,
|
|
926
|
+
path_substring=path_contains,
|
|
927
|
+
model_name=model_name,
|
|
928
|
+
device=device,
|
|
929
|
+
model=model,
|
|
930
|
+
role=nf.role if nf else None,
|
|
931
|
+
module=nf.module if nf else None,
|
|
932
|
+
microservice=nf.microservice if nf else None,
|
|
933
|
+
capability=nf.capability if nf else None,
|
|
934
|
+
exclude_roles=nf.exclude_roles if nf else None,
|
|
935
|
+
dedup_by_fqn=dedup,
|
|
936
|
+
)
|
|
937
|
+
advisories.append(
|
|
938
|
+
f"hybrid unavailable on table '{table}' (FTS index missing on this index built before "
|
|
939
|
+
f"PR-SEARCH-3); fell back to vector-only — reindex to enable hybrid"
|
|
940
|
+
)
|
|
941
|
+
else:
|
|
942
|
+
# Non-FTS error: surface as structured failure
|
|
943
|
+
raise
|
|
888
944
|
hits: list[SearchHit] = []
|
|
889
945
|
for row in rows:
|
|
890
946
|
if path_contains and path_contains not in str(row.get("filename") or ""):
|
|
@@ -893,7 +949,7 @@ def search_v2(
|
|
|
893
949
|
row_kind = "symbol"
|
|
894
950
|
if not _node_matches_filter(row_kind, row, nf):
|
|
895
951
|
continue
|
|
896
|
-
hits.append(_row_to_search_hit(row))
|
|
952
|
+
hits.append(_row_to_search_hit(row, explain=explain))
|
|
897
953
|
hint_payload = {
|
|
898
954
|
"success": True,
|
|
899
955
|
"results": [h.model_dump() for h in hits],
|
|
@@ -906,7 +962,7 @@ def search_v2(
|
|
|
906
962
|
results=hits,
|
|
907
963
|
limit=limit,
|
|
908
964
|
offset=offset,
|
|
909
|
-
advisories=raw_advisories,
|
|
965
|
+
advisories=advisories + raw_advisories, # Merge fallback + hints advisories
|
|
910
966
|
hints_structured=_to_structured_hints(raw_struct),
|
|
911
967
|
)
|
|
912
968
|
except Exception as exc:
|