java-codebase-rag 0.8.0__py3-none-any.whl → 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- java_codebase_rag/cli.py +14 -1
- java_codebase_rag/install_data/agents/explorer-rag-cli.md +65 -208
- java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +78 -232
- java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +44 -83
- java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +67 -135
- java_codebase_rag/installer.py +310 -32
- java_codebase_rag/jrag.py +112 -7
- java_codebase_rag/jrag_envelope.py +1 -1
- java_codebase_rag/jrag_render.py +12 -3
- java_codebase_rag/lance_optimize.py +18 -0
- java_codebase_rag/pipeline.py +14 -0
- {java_codebase_rag-0.8.0.dist-info → java_codebase_rag-0.9.0.dist-info}/METADATA +2 -2
- {java_codebase_rag-0.8.0.dist-info → java_codebase_rag-0.9.0.dist-info}/RECORD +20 -20
- mcp_v2.py +82 -26
- search_lancedb.py +149 -3
- server.py +11 -0
- {java_codebase_rag-0.8.0.dist-info → java_codebase_rag-0.9.0.dist-info}/WHEEL +0 -0
- {java_codebase_rag-0.8.0.dist-info → java_codebase_rag-0.9.0.dist-info}/entry_points.txt +0 -0
- {java_codebase_rag-0.8.0.dist-info → java_codebase_rag-0.9.0.dist-info}/licenses/LICENSE +0 -0
- {java_codebase_rag-0.8.0.dist-info → java_codebase_rag-0.9.0.dist-info}/top_level.txt +0 -0
|
@@ -184,6 +184,24 @@ async def optimize_lance_tables(
|
|
|
184
184
|
|
|
185
185
|
if last_exc is None:
|
|
186
186
|
results[name] = "ok"
|
|
187
|
+
# Best-effort FTS index at index time (PR-SEARCH-3) so hybrid
|
|
188
|
+
# search works on all tables (java/sql/yaml) without a
|
|
189
|
+
# first-query race. Failure is non-fatal — the lazy
|
|
190
|
+
# ensure_text_fts_index in search_lancedb.py is the runtime
|
|
191
|
+
# fallback — so it never alters the "ok" optimize status; we
|
|
192
|
+
# only log the skip when verbose.
|
|
193
|
+
try:
|
|
194
|
+
from lancedb.index import FTS
|
|
195
|
+
await table.create_index("text", config=FTS(), replace=True)
|
|
196
|
+
except Exception as exc:
|
|
197
|
+
low = str(exc).lower()
|
|
198
|
+
if not any(
|
|
199
|
+
w in low for w in ("exist", "duplicate", "already", "same name")
|
|
200
|
+
) and not quiet:
|
|
201
|
+
print(
|
|
202
|
+
f"java-codebase-rag: optimize: {name} fts skipped: {exc}",
|
|
203
|
+
file=sys.stderr,
|
|
204
|
+
)
|
|
187
205
|
if not quiet:
|
|
188
206
|
print(
|
|
189
207
|
f"java-codebase-rag: optimize: {name} ok",
|
java_codebase_rag/pipeline.py
CHANGED
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
from __future__ import annotations
|
|
3
3
|
|
|
4
4
|
import asyncio
|
|
5
|
+
import importlib.util
|
|
5
6
|
import os
|
|
6
7
|
import shutil
|
|
7
8
|
import subprocess
|
|
@@ -200,6 +201,19 @@ def is_graph_preflight_blocker(proc: subprocess.CompletedProcess[str]) -> bool:
|
|
|
200
201
|
return bool(proc.returncode in (126, 127) and len(getattr(proc, "args", ()) or ()) <= 1)
|
|
201
202
|
|
|
202
203
|
|
|
204
|
+
def vector_stack_installed() -> bool:
|
|
205
|
+
"""True when the optional vector stack (cocoindex/lancedb/sentence-transformers) is importable.
|
|
206
|
+
|
|
207
|
+
False on graph-only installs (macOS Intel), where PEP 508 markers exclude the trio.
|
|
208
|
+
Used to skip vector-only wizard steps (e.g. embedding-model selection) and to preflight
|
|
209
|
+
branching without spawning cocoindex. Probes all three since they are gated together.
|
|
210
|
+
"""
|
|
211
|
+
return all(
|
|
212
|
+
importlib.util.find_spec(m) is not None
|
|
213
|
+
for m in ("cocoindex", "lancedb", "sentence_transformers")
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
|
|
203
217
|
def _run_cocoindex_update_impl(
|
|
204
218
|
env: dict[str, str],
|
|
205
219
|
*,
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: java-codebase-rag
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.9.0
|
|
4
4
|
Summary: MCP server for semantic + structural search over Java codebases
|
|
5
5
|
Author: HumanBean17
|
|
6
6
|
License-Expression: MIT
|
|
@@ -122,7 +122,7 @@ If you prefer manual configuration, see [`docs/JAVA-CODEBASE-RAG-CLI.md`](./docs
|
|
|
122
122
|
|
|
123
123
|
## Tools & commands at a glance
|
|
124
124
|
|
|
125
|
-
Pick a surface
|
|
125
|
+
Pick a surface at install time — `java-codebase-rag install --surface mcp|cli` (default `cli`, recommended). Both surfaces walk the same LanceDB vectors + LadybugDB graph. Switch an existing install later with `java-codebase-rag update --surface mcp|cli`.
|
|
126
126
|
|
|
127
127
|
**MCP surface — five tools over stdio**
|
|
128
128
|
|
|
@@ -10,34 +10,34 @@ java_index_v1_common.py,sha256=nF1KrSqboF_RRvWerG9knRRFmWwsrG_CvhgnsoZ8KqA,1154
|
|
|
10
10
|
java_ontology.py,sha256=ooqr8GucOINpzhdEQ3QzVe5A9GfiR0nUTlySDehn9GA,17129
|
|
11
11
|
ladybug_queries.py,sha256=rVnVEHwWwE4USeX7tICYEl1SSiSxJGnBNe1VX66R3Xk,100531
|
|
12
12
|
mcp_hints.py,sha256=zp-4cnOmbYD0YovmZiLS2oGcvWcWE7n8jKVz4_xifno,42512
|
|
13
|
-
mcp_v2.py,sha256=
|
|
13
|
+
mcp_v2.py,sha256=ABNHiZEEwQ76uPF9E5fQWncJ4JmmgtG3vmt5TDxBGVo,68901
|
|
14
14
|
path_filtering.py,sha256=R--XzI51LXBu5IBKMCnJWbkNr6I5d-SDmltyQQnWco0,17674
|
|
15
15
|
pr_analysis.py,sha256=zrmZZD5yotJtM02Kif6_jgI_oeformOao793akp0N6Y,18394
|
|
16
16
|
resolve_service.py,sha256=tC5FQsGmqhqn0EOexVDlRq5egnzKDTrI7CMzk0nPpG8,25135
|
|
17
|
-
search_lancedb.py,sha256=
|
|
18
|
-
server.py,sha256=
|
|
17
|
+
search_lancedb.py,sha256=1sGSZ6H8J9hGcKAwHpzB7F0_in3A_sEtOI5LvlZYIRI,43864
|
|
18
|
+
server.py,sha256=yNpJX_0D1xXY1FHGGBrb4JMOg2THw4_2Ao8YCdijBp0,35944
|
|
19
19
|
java_codebase_rag/__init__.py,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
|
|
20
20
|
java_codebase_rag/_fdlimit.py,sha256=vkwjsPbZfxzZ2DZTPWO5DxtuNlLzOADzIq07iYX7GCU,2465
|
|
21
21
|
java_codebase_rag/_stdio.py,sha256=TDNbpt2EP0_Zd622ihdlKwlMfxkKHWOfgLVcU6TcNbo,1458
|
|
22
|
-
java_codebase_rag/cli.py,sha256=
|
|
22
|
+
java_codebase_rag/cli.py,sha256=8WKk-_Zl1Sp7wHX4Y5f3bkIy8hHRH_QWrNuX_3WYTm4,45269
|
|
23
23
|
java_codebase_rag/cli_format.py,sha256=CT7-xdwZ0bMCdP68_UOwkvm-mnLluU3LutlM-mDNk60,1839
|
|
24
24
|
java_codebase_rag/cli_progress.py,sha256=q6Wh97yzLGs1B8UFk_WAKivfQu7Y5RnUUE-T2YHWkIs,3237
|
|
25
25
|
java_codebase_rag/config.py,sha256=Yl7Nf0O_ZOZTPtyPMap83jRM7qNhWwfbVJTKSEdERi4,24916
|
|
26
|
-
java_codebase_rag/installer.py,sha256=
|
|
27
|
-
java_codebase_rag/jrag.py,sha256=
|
|
28
|
-
java_codebase_rag/jrag_envelope.py,sha256=
|
|
26
|
+
java_codebase_rag/installer.py,sha256=c-_tR1Ct_O3yhmruFRnDoM1Wilz-igD9qdkDPsNqLMI,79934
|
|
27
|
+
java_codebase_rag/jrag.py,sha256=cVUWrKOtkwe2r4u1nknzmrMQMy16bdt1n-dZQpiSFWs,191811
|
|
28
|
+
java_codebase_rag/jrag_envelope.py,sha256=5jD3p2O-p9acAKHqoif5FFoqgP7Q2ziCuKSiPCHoywc,47505
|
|
29
29
|
java_codebase_rag/jrag_hints.py,sha256=k2PFE4s3lZgBYHMdZcTjx1-w28nfQcBtQEVsSxI_DvE,9262
|
|
30
|
-
java_codebase_rag/jrag_render.py,sha256=
|
|
31
|
-
java_codebase_rag/lance_optimize.py,sha256=
|
|
32
|
-
java_codebase_rag/pipeline.py,sha256=
|
|
30
|
+
java_codebase_rag/jrag_render.py,sha256=1nUyamL-MOOXDlKvssp-LsBgEtznUnGj_cPteSwuS90,31953
|
|
31
|
+
java_codebase_rag/lance_optimize.py,sha256=_90eajcIpGUNm3GWtx6AYDERNHaMsaMZs7zu2KqNPEU,10347
|
|
32
|
+
java_codebase_rag/pipeline.py,sha256=TkHb7DybFlpHje30aYbuFt5jWFSJBdCQ-7hc4jFaNhI,16481
|
|
33
33
|
java_codebase_rag/progress.py,sha256=2IxdMALDM0wAQCyJrrfZ975zM_85C-4BfHxf4AtYifE,23212
|
|
34
|
-
java_codebase_rag/install_data/agents/explorer-rag-cli.md,sha256=
|
|
35
|
-
java_codebase_rag/install_data/agents/explorer-rag-enhanced.md,sha256=
|
|
36
|
-
java_codebase_rag/install_data/skills/explore-codebase/SKILL.md,sha256=
|
|
37
|
-
java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md,sha256=
|
|
38
|
-
java_codebase_rag-0.
|
|
39
|
-
java_codebase_rag-0.
|
|
40
|
-
java_codebase_rag-0.
|
|
41
|
-
java_codebase_rag-0.
|
|
42
|
-
java_codebase_rag-0.
|
|
43
|
-
java_codebase_rag-0.
|
|
34
|
+
java_codebase_rag/install_data/agents/explorer-rag-cli.md,sha256=mMij_BIQM4agaYhGVYjC-fQSe3We1HFeBvc5JBjJj6A,10071
|
|
35
|
+
java_codebase_rag/install_data/agents/explorer-rag-enhanced.md,sha256=gZsNFbuK0lSnOIlplbbS_muz2ozokJqvFUv65QM0NDM,10406
|
|
36
|
+
java_codebase_rag/install_data/skills/explore-codebase/SKILL.md,sha256=A-v2dueVnxwBzBlxoRjZ2zOJk8DranLQ1TElwn94h0s,11529
|
|
37
|
+
java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md,sha256=V5gIKKGkgk2KlAFf9JqPareQDMt1iQOI7cwhfOqJL1c,11348
|
|
38
|
+
java_codebase_rag-0.9.0.dist-info/licenses/LICENSE,sha256=gxvtiHtuviR_q8ZAjWw-QTcF3DyPzg6ZY-lQrr8OPpw,1068
|
|
39
|
+
java_codebase_rag-0.9.0.dist-info/METADATA,sha256=pIQfvYY6sWdmq-PIXm0vOUzNobTUtW14Z-Zp-53qaf8,20088
|
|
40
|
+
java_codebase_rag-0.9.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
|
|
41
|
+
java_codebase_rag-0.9.0.dist-info/entry_points.txt,sha256=cj3QTc11UYVQnj9T3orc4daiIGaCYrXP149vKbH2R4U,168
|
|
42
|
+
java_codebase_rag-0.9.0.dist-info/top_level.txt,sha256=8vC-VN3cMwz5vhkSTaeJ1a1bDeqLWEfrTks1CvEvIg0,273
|
|
43
|
+
java_codebase_rag-0.9.0.dist-info/RECORD,,
|
mcp_v2.py
CHANGED
|
@@ -466,6 +466,8 @@ class SearchHit(BaseModel):
|
|
|
466
466
|
role: str | None = None
|
|
467
467
|
filename: str | None = None
|
|
468
468
|
start_line: int | None = None
|
|
469
|
+
score_components: dict[str, float] | None = None
|
|
470
|
+
chunks: int | None = None
|
|
469
471
|
|
|
470
472
|
|
|
471
473
|
# NodeRef is now defined in graph_types.py and imported above
|
|
@@ -583,7 +585,7 @@ def _chunk_id_from_row(row: dict[str, Any]) -> str:
|
|
|
583
585
|
return f"{filename}:{sb}:{eb}"
|
|
584
586
|
|
|
585
587
|
|
|
586
|
-
def _row_to_search_hit(row: dict[str, Any]) -> SearchHit:
|
|
588
|
+
def _row_to_search_hit(row: dict[str, Any], explain: bool = False) -> SearchHit:
|
|
587
589
|
score = float(row.get("_rrf_score") or row.get("_score") or 0.0)
|
|
588
590
|
filename = str(row.get("filename") or "") or None
|
|
589
591
|
start_line: int | None = None
|
|
@@ -595,6 +597,8 @@ def _row_to_search_hit(row: dict[str, Any]) -> SearchHit:
|
|
|
595
597
|
start_line = int(ln)
|
|
596
598
|
except (TypeError, ValueError):
|
|
597
599
|
start_line = None
|
|
600
|
+
chunks = row.get("_chunks_collapsed")
|
|
601
|
+
chunks_int = int(chunks) if chunks is not None and int(chunks) >= 2 else None
|
|
598
602
|
return SearchHit(
|
|
599
603
|
chunk_id=_chunk_id_from_row(row),
|
|
600
604
|
symbol_id=_chunk_to_symbol_id(row),
|
|
@@ -606,6 +610,8 @@ def _row_to_search_hit(row: dict[str, Any]) -> SearchHit:
|
|
|
606
610
|
role=str(row.get("role")) if row.get("role") else None,
|
|
607
611
|
filename=filename,
|
|
608
612
|
start_line=start_line,
|
|
613
|
+
score_components=row.get("_score_components") if explain else None,
|
|
614
|
+
chunks=chunks_int,
|
|
609
615
|
)
|
|
610
616
|
|
|
611
617
|
|
|
@@ -820,7 +826,9 @@ def search_v2(
|
|
|
820
826
|
offset: int = 0,
|
|
821
827
|
path_contains: str | None = None,
|
|
822
828
|
filter: NodeFilter | dict[str, Any] | str | None = None,
|
|
829
|
+
explain: bool = False,
|
|
823
830
|
graph: LadybugGraph | None = None,
|
|
831
|
+
dedup: bool = True,
|
|
824
832
|
) -> SearchOutput:
|
|
825
833
|
try:
|
|
826
834
|
raw_filter = _coerce_filter(filter)
|
|
@@ -852,6 +860,17 @@ def search_v2(
|
|
|
852
860
|
limit=None,
|
|
853
861
|
offset=None,
|
|
854
862
|
)
|
|
863
|
+
# hybrid + table='all' is unsupported (hybrid fuses vector+FTS on ONE
|
|
864
|
+
# table); fail fast with a clean envelope BEFORE loading the embedding
|
|
865
|
+
# model. run_search also guards this — this is the user-facing fast path.
|
|
866
|
+
if hybrid and table == "all":
|
|
867
|
+
return SearchOutput(
|
|
868
|
+
success=False,
|
|
869
|
+
message="hybrid search requires a single table; use java, sql, or yaml (not all)",
|
|
870
|
+
advisories=[],
|
|
871
|
+
limit=None,
|
|
872
|
+
offset=None,
|
|
873
|
+
)
|
|
855
874
|
model_name = resolved_sbert_model_for_process_env(SBERT_MODEL)
|
|
856
875
|
device = os.environ.get("SBERT_DEVICE") or None
|
|
857
876
|
model = _get_sentence_transformer(model_name, device)
|
|
@@ -862,29 +881,66 @@ def search_v2(
|
|
|
862
881
|
if not uri.startswith(("s3://", "gs://", "az://")) and uri_path.exists():
|
|
863
882
|
uri = str(uri_path.resolve())
|
|
864
883
|
table_keys = list(TABLES) if table == "all" else [table]
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
|
|
875
|
-
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
|
|
881
|
-
|
|
882
|
-
|
|
883
|
-
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
884
|
+
|
|
885
|
+
# Graceful fallback: if hybrid=True and FTS index is missing (old index),
|
|
886
|
+
# retry with hybrid=False and return vector-only results with an advisory.
|
|
887
|
+
advisories: list[str] = []
|
|
888
|
+
try:
|
|
889
|
+
rows = run_search(
|
|
890
|
+
query,
|
|
891
|
+
uri=uri,
|
|
892
|
+
table_keys=table_keys,
|
|
893
|
+
hybrid=hybrid,
|
|
894
|
+
limit=limit,
|
|
895
|
+
offset=offset,
|
|
896
|
+
path_substring=path_contains,
|
|
897
|
+
model_name=model_name,
|
|
898
|
+
device=device,
|
|
899
|
+
model=model,
|
|
900
|
+
# Push the NodeFilter structural predicates into the LanceDB query so
|
|
901
|
+
# they apply BEFORE pagination (issue #353) — previously they were only
|
|
902
|
+
# a post-filter on the already-paginated page, which could shrink or
|
|
903
|
+
# empty filtered pages even when many matches existed deeper in the
|
|
904
|
+
# ranking. _node_matches_filter below still re-checks every row (it
|
|
905
|
+
# covers the non-pushdownable fields and is the contract guarantee).
|
|
906
|
+
role=nf.role if nf else None,
|
|
907
|
+
module=nf.module if nf else None,
|
|
908
|
+
microservice=nf.microservice if nf else None,
|
|
909
|
+
capability=nf.capability if nf else None,
|
|
910
|
+
exclude_roles=nf.exclude_roles if nf else None,
|
|
911
|
+
dedup_by_fqn=dedup,
|
|
912
|
+
)
|
|
913
|
+
except Exception as exc:
|
|
914
|
+
# Check if this is a missing-FTS error (old index built before PR-SEARCH-3)
|
|
915
|
+
exc_text = str(exc).lower()
|
|
916
|
+
is_fts_missing = "full text search" in exc_text or "inverted index" in exc_text
|
|
917
|
+
if hybrid and is_fts_missing:
|
|
918
|
+
# Retry with vector-only search
|
|
919
|
+
rows = run_search(
|
|
920
|
+
query,
|
|
921
|
+
uri=uri,
|
|
922
|
+
table_keys=table_keys,
|
|
923
|
+
hybrid=False, # Fallback to vector-only
|
|
924
|
+
limit=limit,
|
|
925
|
+
offset=offset,
|
|
926
|
+
path_substring=path_contains,
|
|
927
|
+
model_name=model_name,
|
|
928
|
+
device=device,
|
|
929
|
+
model=model,
|
|
930
|
+
role=nf.role if nf else None,
|
|
931
|
+
module=nf.module if nf else None,
|
|
932
|
+
microservice=nf.microservice if nf else None,
|
|
933
|
+
capability=nf.capability if nf else None,
|
|
934
|
+
exclude_roles=nf.exclude_roles if nf else None,
|
|
935
|
+
dedup_by_fqn=dedup,
|
|
936
|
+
)
|
|
937
|
+
advisories.append(
|
|
938
|
+
f"hybrid unavailable on table '{table}' (FTS index missing on this index built before "
|
|
939
|
+
f"PR-SEARCH-3); fell back to vector-only — reindex to enable hybrid"
|
|
940
|
+
)
|
|
941
|
+
else:
|
|
942
|
+
# Non-FTS error: surface as structured failure
|
|
943
|
+
raise
|
|
888
944
|
hits: list[SearchHit] = []
|
|
889
945
|
for row in rows:
|
|
890
946
|
if path_contains and path_contains not in str(row.get("filename") or ""):
|
|
@@ -893,7 +949,7 @@ def search_v2(
|
|
|
893
949
|
row_kind = "symbol"
|
|
894
950
|
if not _node_matches_filter(row_kind, row, nf):
|
|
895
951
|
continue
|
|
896
|
-
hits.append(_row_to_search_hit(row))
|
|
952
|
+
hits.append(_row_to_search_hit(row, explain=explain))
|
|
897
953
|
hint_payload = {
|
|
898
954
|
"success": True,
|
|
899
955
|
"results": [h.model_dump() for h in hits],
|
|
@@ -906,7 +962,7 @@ def search_v2(
|
|
|
906
962
|
results=hits,
|
|
907
963
|
limit=limit,
|
|
908
964
|
offset=offset,
|
|
909
|
-
advisories=raw_advisories,
|
|
965
|
+
advisories=advisories + raw_advisories, # Merge fallback + hints advisories
|
|
910
966
|
hints_structured=_to_structured_hints(raw_struct),
|
|
911
967
|
)
|
|
912
968
|
except Exception as exc:
|
search_lancedb.py
CHANGED
|
@@ -41,6 +41,11 @@ JAVA_ENRICHED_COLUMNS: tuple[str, ...] = (
|
|
|
41
41
|
"capabilities",
|
|
42
42
|
)
|
|
43
43
|
|
|
44
|
+
# Over-fetch multiplier for dedup: fetch 4x to absorb per-FQN chunk multiplicity
|
|
45
|
+
# so that after collapsing by primary_type_fqn, a page stays full and the +1
|
|
46
|
+
# truncation sentinel survives. The formula: need = max((limit + offset) * 4, limit + offset + 1)
|
|
47
|
+
DEDUP_OVERFETCH = 4
|
|
48
|
+
|
|
44
49
|
VECTOR_COLUMN = "embedding"
|
|
45
50
|
_FTS_READY: set[tuple[str, str]] = set()
|
|
46
51
|
_FTS_LOCK = threading.Lock()
|
|
@@ -201,6 +206,14 @@ _ROLE_SCORE_WEIGHTS: dict[str, float] = {
|
|
|
201
206
|
"DTO": -0.08,
|
|
202
207
|
}
|
|
203
208
|
|
|
209
|
+
# Theoretical maximum for hybrid composite score (used for display normalization).
|
|
210
|
+
# Hybrid sort metric: raw_rrf * (import_factor if import_heavy else 1)
|
|
211
|
+
# + role_weight + symbol_bonus
|
|
212
|
+
# where raw_rrf ≤ 2/(k+1) for 2-list RRF, role_weight ≤ max(_ROLE_SCORE_WEIGHTS),
|
|
213
|
+
# and symbol_bonus ≤ _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS.
|
|
214
|
+
# The import factor is ≤ 1, so we use the raw max (2/61).
|
|
215
|
+
_HYBRID_SCORE_MAX = (2.0 / 61.0) + max(_ROLE_SCORE_WEIGHTS.values()) + _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS
|
|
216
|
+
|
|
204
217
|
|
|
205
218
|
def _query_tokens(query: str) -> set[str]:
|
|
206
219
|
"""Lowercased alpha-only tokens from the query, minus stopwords, len >= 3.
|
|
@@ -353,7 +366,7 @@ def _hybrid_sort_key(r: dict) -> float:
|
|
|
353
366
|
comps["hybrid_rrf"] = s
|
|
354
367
|
if r.get("_hints", {}).get("import_heavy"):
|
|
355
368
|
s *= _IMPORT_HYBRID_SCORE_FACTOR
|
|
356
|
-
comps["import_penalty"] = _IMPORT_HYBRID_SCORE_FACTOR
|
|
369
|
+
comps["import_penalty"] = 1.0 - _IMPORT_HYBRID_SCORE_FACTOR
|
|
357
370
|
s += _role_weight(r)
|
|
358
371
|
s += float(comps.get("symbol_bonus", 0.0))
|
|
359
372
|
return -s
|
|
@@ -376,7 +389,8 @@ def explain_score_components(
|
|
|
376
389
|
comps = {}
|
|
377
390
|
parts: list[str] = []
|
|
378
391
|
if hybrid:
|
|
379
|
-
|
|
392
|
+
# Prefer rrf_raw (added by PR-SEARCH-1a) for explanation
|
|
393
|
+
rrf = comps.get("rrf_raw") or comps.get("hybrid_rrf")
|
|
380
394
|
if rrf is not None:
|
|
381
395
|
parts.append(f"rrf={float(rrf):.3f}")
|
|
382
396
|
else:
|
|
@@ -403,6 +417,46 @@ def l2_distance_to_score(distance: float) -> float:
|
|
|
403
417
|
return 1.0 - distance * distance / 2.0
|
|
404
418
|
|
|
405
419
|
|
|
420
|
+
def _effective_distance(comps: dict[str, float]) -> float:
|
|
421
|
+
"""Compute the adjusted distance used for sorting.
|
|
422
|
+
|
|
423
|
+
Matches _vector_sort_key logic: distance + import_penalty - role_weight - symbol_bonus.
|
|
424
|
+
"""
|
|
425
|
+
d = comps.get("distance", 0.0)
|
|
426
|
+
d += comps.get("import_penalty", 0.0)
|
|
427
|
+
d -= comps.get("role_weight", 0.0)
|
|
428
|
+
d -= comps.get("symbol_bonus", 0.0)
|
|
429
|
+
return d
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
def _clamp01(x: float) -> float:
|
|
433
|
+
"""Clamp a value to the [0.0, 1.0] range."""
|
|
434
|
+
if x < 0.0:
|
|
435
|
+
return 0.0
|
|
436
|
+
if x > 1.0:
|
|
437
|
+
return 1.0
|
|
438
|
+
return x
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
def _hybrid_post_sort_normalization(rows: list[dict]) -> None:
|
|
442
|
+
"""Set honest displayed scores for hybrid search after sorting.
|
|
443
|
+
|
|
444
|
+
Reconstructs the composite score (raw_rrf * import_factor + role_weight + symbol_bonus)
|
|
445
|
+
and normalizes by _HYBRID_SCORE_MAX to ensure rank-monotonicity.
|
|
446
|
+
|
|
447
|
+
Mutates rows in-place, replacing _score with the normalized value.
|
|
448
|
+
"""
|
|
449
|
+
for r in rows:
|
|
450
|
+
comps = r.setdefault("_score_components", {})
|
|
451
|
+
raw = comps.get("hybrid_rrf", 0.0)
|
|
452
|
+
comps["rrf_raw"] = raw # preserve raw RRF for --explain. NOTE: when graph_expand + hybrid combine (Phase 2), _rrf_merge below overwrites this with graph-RRF, so --explain would show graph-RRF not hybrid-RRF.
|
|
453
|
+
s = raw
|
|
454
|
+
if r.get("_hints", {}).get("import_heavy"):
|
|
455
|
+
s *= _IMPORT_HYBRID_SCORE_FACTOR
|
|
456
|
+
s += comps.get("role_weight", 0.0) + comps.get("symbol_bonus", 0.0)
|
|
457
|
+
r["_score"] = _clamp01(s / _HYBRID_SCORE_MAX)
|
|
458
|
+
|
|
459
|
+
|
|
406
460
|
def _escape_like_fragment(s: str) -> str:
|
|
407
461
|
return s.replace("'", "''")
|
|
408
462
|
|
|
@@ -790,9 +844,72 @@ def _rrf_merge(
|
|
|
790
844
|
existing["_rrf_score"] = float(existing.get("_rrf_score", 0.0)) + contribution
|
|
791
845
|
merged = list(pool.values())
|
|
792
846
|
merged.sort(key=lambda r: -float(r.get("_rrf_score", 0.0)))
|
|
847
|
+
# Normalize displayed _rrf_score to [0,1] by theoretical max
|
|
848
|
+
# RRF max = Σ weight·1/(k+rank+1); theoretical max when all rows are rank 0
|
|
849
|
+
# with weight 1.0 = num_lists / (k + 1)
|
|
850
|
+
num_lists = len(lists)
|
|
851
|
+
max_rrf = num_lists / (k + 1)
|
|
852
|
+
for r in merged:
|
|
853
|
+
raw_score = float(r.get("_rrf_score", 0.0))
|
|
854
|
+
comps = r.setdefault("_score_components", {})
|
|
855
|
+
comps["rrf_raw"] = raw_score
|
|
856
|
+
r["_rrf_score"] = _clamp01(raw_score / max_rrf)
|
|
793
857
|
return merged
|
|
794
858
|
|
|
795
859
|
|
|
860
|
+
def _dedup_by_fqn(rows: list[dict], dedup_by_fqn: bool = True) -> list[dict]:
|
|
861
|
+
"""Deduplicate rows by primary_type_fqn (java table only).
|
|
862
|
+
|
|
863
|
+
When dedup_by_fqn is True, collapses multiple chunks of the same
|
|
864
|
+
primary_type_fqn into one row (first-seen-wins, since rows are pre-sorted
|
|
865
|
+
so the first is the best chunk). Each survivor gets a _chunks_collapsed
|
|
866
|
+
field (>=1) counting how many rows were collapsed into it.
|
|
867
|
+
|
|
868
|
+
Rows without primary_type_fqn (sql/yaml tables) get a unique __id:<id>
|
|
869
|
+
key so they pass through unchanged (each row is unique).
|
|
870
|
+
|
|
871
|
+
When dedup_by_fqn is False, returns rows unchanged (regression guard).
|
|
872
|
+
"""
|
|
873
|
+
if not dedup_by_fqn:
|
|
874
|
+
# Non-dedup path: return unchanged, byte-identical to prior behavior
|
|
875
|
+
return rows
|
|
876
|
+
|
|
877
|
+
deduped: list[dict] = []
|
|
878
|
+
seen_keys: dict[str, dict] = {}
|
|
879
|
+
collapsed_counts: dict[str, int] = {}
|
|
880
|
+
|
|
881
|
+
for row in rows:
|
|
882
|
+
# Build dedup key: primary_type_fqn for java rows, unique __id:<id> for sql/yaml
|
|
883
|
+
fqn = row.get("primary_type_fqn")
|
|
884
|
+
if fqn:
|
|
885
|
+
key = str(fqn)
|
|
886
|
+
else:
|
|
887
|
+
# sql/yaml rows have no primary_type_fqn → unique key per row
|
|
888
|
+
row_id = row.get("id") or id(row)
|
|
889
|
+
key = f"__id:{row_id}"
|
|
890
|
+
|
|
891
|
+
if key not in seen_keys:
|
|
892
|
+
# First occurrence: keep it
|
|
893
|
+
seen_keys[key] = row
|
|
894
|
+
collapsed_counts[key] = 1
|
|
895
|
+
deduped.append(row)
|
|
896
|
+
else:
|
|
897
|
+
# Duplicate: increment collapse count, discard this row
|
|
898
|
+
collapsed_counts[key] += 1
|
|
899
|
+
|
|
900
|
+
# Annotate each survivor with _chunks_collapsed
|
|
901
|
+
for row in deduped:
|
|
902
|
+
fqn = row.get("primary_type_fqn")
|
|
903
|
+
if fqn:
|
|
904
|
+
key = str(fqn)
|
|
905
|
+
else:
|
|
906
|
+
row_id = row.get("id") or id(row)
|
|
907
|
+
key = f"__id:{row_id}"
|
|
908
|
+
row["_chunks_collapsed"] = collapsed_counts[key]
|
|
909
|
+
|
|
910
|
+
return deduped
|
|
911
|
+
|
|
912
|
+
|
|
796
913
|
def run_search(
|
|
797
914
|
query: str,
|
|
798
915
|
*,
|
|
@@ -819,6 +936,7 @@ def run_search(
|
|
|
819
936
|
exclude_roles: list[str] | None = None,
|
|
820
937
|
capability: str | None = None,
|
|
821
938
|
capability_in: list[str] | None = None,
|
|
939
|
+
dedup_by_fqn: bool = False,
|
|
822
940
|
) -> list[dict]:
|
|
823
941
|
effective_hybrid = hybrid
|
|
824
942
|
effective_fts = fts_text
|
|
@@ -852,7 +970,16 @@ def run_search(
|
|
|
852
970
|
fts_for_hybrid = effective_fts if effective_fts is not None else query
|
|
853
971
|
|
|
854
972
|
db = lancedb.connect(uri)
|
|
855
|
-
|
|
973
|
+
if dedup_by_fqn:
|
|
974
|
+
# Over-fetch to absorb per-FQN chunk multiplicity: fetch 4x so that
|
|
975
|
+
# after collapsing, the page stays full and the +1 truncation sentinel survives.
|
|
976
|
+
# The 4× factor assumes typical per-FQN chunk multiplicity; a single type with
|
|
977
|
+
# many high-ranking chunks (e.g. generated/God classes) could starve the page or
|
|
978
|
+
# make the +1 truncation sentinel unreliable; Phase 1 may revisit adaptive over-fetch (plan risk #1).
|
|
979
|
+
need = max((limit + offset) * DEDUP_OVERFETCH, limit + offset + 1)
|
|
980
|
+
else:
|
|
981
|
+
# Non-dedup path: exact fetch as before
|
|
982
|
+
need = max(limit + offset, 1)
|
|
856
983
|
|
|
857
984
|
extra_java = _build_extra_predicates(
|
|
858
985
|
columns=_table_columns(uri, TABLES["java"], db),
|
|
@@ -887,8 +1014,15 @@ def run_search(
|
|
|
887
1014
|
_apply_symbol_bonus(rows, query_toks)
|
|
888
1015
|
if effective_hybrid:
|
|
889
1016
|
rows.sort(key=_hybrid_sort_key)
|
|
1017
|
+
# Hybrid: set honest displayed score from composite sort metric, clamped to [0,1]
|
|
1018
|
+
_hybrid_post_sort_normalization(rows)
|
|
890
1019
|
else:
|
|
891
1020
|
rows.sort(key=_vector_sort_key)
|
|
1021
|
+
# Vector: set honest displayed score from adjusted distance, clamped to [0,1]
|
|
1022
|
+
for r in rows:
|
|
1023
|
+
comps = r.setdefault("_score_components", {})
|
|
1024
|
+
effective_dist = _effective_distance(comps)
|
|
1025
|
+
r["_score"] = _clamp01(l2_distance_to_score(effective_dist))
|
|
892
1026
|
|
|
893
1027
|
if graph_expand and key == "java" and expand_depth > 0:
|
|
894
1028
|
rows = _graph_expand_merge(
|
|
@@ -902,6 +1036,9 @@ def run_search(
|
|
|
902
1036
|
ladybug_path=ladybug_path,
|
|
903
1037
|
)
|
|
904
1038
|
|
|
1039
|
+
# Dedup by primary_type_fqn after all sorting/merging, before windowing
|
|
1040
|
+
rows = _dedup_by_fqn(rows, dedup_by_fqn=dedup_by_fqn)
|
|
1041
|
+
|
|
905
1042
|
window = rows[offset : offset + limit]
|
|
906
1043
|
if context_neighbors > 0 and key == "java":
|
|
907
1044
|
_attach_neighbor_context(window, db=db, neighbors=context_neighbors, uri=uri)
|
|
@@ -931,6 +1068,15 @@ def run_search(
|
|
|
931
1068
|
r["_skip_role_weight"] = True
|
|
932
1069
|
_apply_symbol_bonus(merged, query_toks)
|
|
933
1070
|
merged.sort(key=_vector_sort_key)
|
|
1071
|
+
# Vector: set honest displayed score from adjusted distance, clamped to [0,1]
|
|
1072
|
+
for r in merged:
|
|
1073
|
+
comps = r.setdefault("_score_components", {})
|
|
1074
|
+
effective_dist = _effective_distance(comps)
|
|
1075
|
+
r["_score"] = _clamp01(l2_distance_to_score(effective_dist))
|
|
1076
|
+
|
|
1077
|
+
# Dedup by primary_type_fqn after all sorting/merging, before windowing
|
|
1078
|
+
merged = _dedup_by_fqn(merged, dedup_by_fqn=dedup_by_fqn)
|
|
1079
|
+
|
|
934
1080
|
window = merged[offset : offset + limit]
|
|
935
1081
|
if context_neighbors > 0:
|
|
936
1082
|
_attach_neighbor_context(window, db=db, neighbors=context_neighbors, uri=uri)
|
server.py
CHANGED
|
@@ -512,6 +512,7 @@ def create_mcp_server() -> FastMCP:
|
|
|
512
512
|
"structured DSL inside `query`; structured predicates belong in `find`. "
|
|
513
513
|
"For identifier-shaped lookups (FQN, id, route/client identifiers, …), use `resolve` first; "
|
|
514
514
|
"use `search` for natural-language or ranked fuzzy discovery. "
|
|
515
|
+
"Set `explain=true` to include score breakdown per hit. "
|
|
515
516
|
"Successful responses echo `limit`/`offset`."
|
|
516
517
|
),
|
|
517
518
|
)
|
|
@@ -538,6 +539,14 @@ def create_mcp_server() -> FastMCP:
|
|
|
538
539
|
"predicate. Unknown keys or populated fields not applicable to symbols return success=false."
|
|
539
540
|
),
|
|
540
541
|
),
|
|
542
|
+
explain: bool = Field(
|
|
543
|
+
default=False,
|
|
544
|
+
description="If true, include score_components in each SearchHit (breakdown of distance/rrf, role, symbol, import_penalty).",
|
|
545
|
+
),
|
|
546
|
+
chunks: bool = Field(
|
|
547
|
+
default=False,
|
|
548
|
+
description="If true, show every chunk (default collapses to one row per symbol/type).",
|
|
549
|
+
),
|
|
541
550
|
) -> mcp_v2.SearchOutput:
|
|
542
551
|
scoped_filter = _scope_manager.apply_auto_scope(filter) if _scope_manager else filter
|
|
543
552
|
return await asyncio.to_thread(
|
|
@@ -549,7 +558,9 @@ def create_mcp_server() -> FastMCP:
|
|
|
549
558
|
offset,
|
|
550
559
|
path_contains,
|
|
551
560
|
scoped_filter,
|
|
561
|
+
explain,
|
|
552
562
|
None,
|
|
563
|
+
not chunks, # dedup=True by default; chunks=True opts out
|
|
553
564
|
)
|
|
554
565
|
|
|
555
566
|
@mcp.tool(
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|