java-codebase-rag 0.8.0__py3-none-any.whl → 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -184,6 +184,24 @@ async def optimize_lance_tables(
184
184
 
185
185
  if last_exc is None:
186
186
  results[name] = "ok"
187
+ # Best-effort FTS index at index time (PR-SEARCH-3) so hybrid
188
+ # search works on all tables (java/sql/yaml) without a
189
+ # first-query race. Failure is non-fatal — the lazy
190
+ # ensure_text_fts_index in search_lancedb.py is the runtime
191
+ # fallback — so it never alters the "ok" optimize status; we
192
+ # only log the skip when verbose.
193
+ try:
194
+ from lancedb.index import FTS
195
+ await table.create_index("text", config=FTS(), replace=True)
196
+ except Exception as exc:
197
+ low = str(exc).lower()
198
+ if not any(
199
+ w in low for w in ("exist", "duplicate", "already", "same name")
200
+ ) and not quiet:
201
+ print(
202
+ f"java-codebase-rag: optimize: {name} fts skipped: {exc}",
203
+ file=sys.stderr,
204
+ )
187
205
  if not quiet:
188
206
  print(
189
207
  f"java-codebase-rag: optimize: {name} ok",
@@ -2,6 +2,7 @@
2
2
  from __future__ import annotations
3
3
 
4
4
  import asyncio
5
+ import importlib.util
5
6
  import os
6
7
  import shutil
7
8
  import subprocess
@@ -200,6 +201,19 @@ def is_graph_preflight_blocker(proc: subprocess.CompletedProcess[str]) -> bool:
200
201
  return bool(proc.returncode in (126, 127) and len(getattr(proc, "args", ()) or ()) <= 1)
201
202
 
202
203
 
204
+ def vector_stack_installed() -> bool:
205
+ """True when the optional vector stack (cocoindex/lancedb/sentence-transformers) is importable.
206
+
207
+ False on graph-only installs (macOS Intel), where PEP 508 markers exclude the trio.
208
+ Used to skip vector-only wizard steps (e.g. embedding-model selection) and to preflight
209
+ branching without spawning cocoindex. Probes all three since they are gated together.
210
+ """
211
+ return all(
212
+ importlib.util.find_spec(m) is not None
213
+ for m in ("cocoindex", "lancedb", "sentence_transformers")
214
+ )
215
+
216
+
203
217
  def _run_cocoindex_update_impl(
204
218
  env: dict[str, str],
205
219
  *,
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: java-codebase-rag
3
- Version: 0.8.0
3
+ Version: 0.9.0
4
4
  Summary: MCP server for semantic + structural search over Java codebases
5
5
  Author: HumanBean17
6
6
  License-Expression: MIT
@@ -122,7 +122,7 @@ If you prefer manual configuration, see [`docs/JAVA-CODEBASE-RAG-CLI.md`](./docs
122
122
 
123
123
  ## Tools & commands at a glance
124
124
 
125
- Pick a surface once at install time — `java-codebase-rag install --surface mcp|cli` (default `mcp`). Both surfaces walk the same LanceDB vectors + LadybugDB graph.
125
+ Pick a surface at install time — `java-codebase-rag install --surface mcp|cli` (default `cli`, recommended). Both surfaces walk the same LanceDB vectors + LadybugDB graph. Switch an existing install later with `java-codebase-rag update --surface mcp|cli`.
126
126
 
127
127
  **MCP surface — five tools over stdio**
128
128
 
@@ -10,34 +10,34 @@ java_index_v1_common.py,sha256=nF1KrSqboF_RRvWerG9knRRFmWwsrG_CvhgnsoZ8KqA,1154
10
10
  java_ontology.py,sha256=ooqr8GucOINpzhdEQ3QzVe5A9GfiR0nUTlySDehn9GA,17129
11
11
  ladybug_queries.py,sha256=rVnVEHwWwE4USeX7tICYEl1SSiSxJGnBNe1VX66R3Xk,100531
12
12
  mcp_hints.py,sha256=zp-4cnOmbYD0YovmZiLS2oGcvWcWE7n8jKVz4_xifno,42512
13
- mcp_v2.py,sha256=_pSZrbbImgLOrq8RxBhggoaKHP8oOroUiqh3jz86gG0,66084
13
+ mcp_v2.py,sha256=ABNHiZEEwQ76uPF9E5fQWncJ4JmmgtG3vmt5TDxBGVo,68901
14
14
  path_filtering.py,sha256=R--XzI51LXBu5IBKMCnJWbkNr6I5d-SDmltyQQnWco0,17674
15
15
  pr_analysis.py,sha256=zrmZZD5yotJtM02Kif6_jgI_oeformOao793akp0N6Y,18394
16
16
  resolve_service.py,sha256=tC5FQsGmqhqn0EOexVDlRq5egnzKDTrI7CMzk0nPpG8,25135
17
- search_lancedb.py,sha256=MKnrRpRQnZNbp5oAFHMBFILoKcQEMaJBouTXakXPSXM,37390
18
- server.py,sha256=o5TQTO76CEK0SngU5JwGfd8MPn1wyPQODf2zNfbfBTw,35406
17
+ search_lancedb.py,sha256=1sGSZ6H8J9hGcKAwHpzB7F0_in3A_sEtOI5LvlZYIRI,43864
18
+ server.py,sha256=yNpJX_0D1xXY1FHGGBrb4JMOg2THw4_2Ao8YCdijBp0,35944
19
19
  java_codebase_rag/__init__.py,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
20
20
  java_codebase_rag/_fdlimit.py,sha256=vkwjsPbZfxzZ2DZTPWO5DxtuNlLzOADzIq07iYX7GCU,2465
21
21
  java_codebase_rag/_stdio.py,sha256=TDNbpt2EP0_Zd622ihdlKwlMfxkKHWOfgLVcU6TcNbo,1458
22
- java_codebase_rag/cli.py,sha256=0W14fsod0ZJSfSD15BRE4A16V5f5hy_bZ_A1-jZPO4w,44738
22
+ java_codebase_rag/cli.py,sha256=8WKk-_Zl1Sp7wHX4Y5f3bkIy8hHRH_QWrNuX_3WYTm4,45269
23
23
  java_codebase_rag/cli_format.py,sha256=CT7-xdwZ0bMCdP68_UOwkvm-mnLluU3LutlM-mDNk60,1839
24
24
  java_codebase_rag/cli_progress.py,sha256=q6Wh97yzLGs1B8UFk_WAKivfQu7Y5RnUUE-T2YHWkIs,3237
25
25
  java_codebase_rag/config.py,sha256=Yl7Nf0O_ZOZTPtyPMap83jRM7qNhWwfbVJTKSEdERi4,24916
26
- java_codebase_rag/installer.py,sha256=NgOdsML1BOmqIoIXyoDDLg4r3JxQ9hqqYGCwVW6FHQs,67582
27
- java_codebase_rag/jrag.py,sha256=K1_QNMO2mRWSeatJIEJ0esVZztraiA8VB6Pkphg6ITU,187655
28
- java_codebase_rag/jrag_envelope.py,sha256=mvAgoZm5XIpmV-J1ntlBzwJW62Fs71RGcpxFQa-l0lc,47484
26
+ java_codebase_rag/installer.py,sha256=c-_tR1Ct_O3yhmruFRnDoM1Wilz-igD9qdkDPsNqLMI,79934
27
+ java_codebase_rag/jrag.py,sha256=cVUWrKOtkwe2r4u1nknzmrMQMy16bdt1n-dZQpiSFWs,191811
28
+ java_codebase_rag/jrag_envelope.py,sha256=5jD3p2O-p9acAKHqoif5FFoqgP7Q2ziCuKSiPCHoywc,47505
29
29
  java_codebase_rag/jrag_hints.py,sha256=k2PFE4s3lZgBYHMdZcTjx1-w28nfQcBtQEVsSxI_DvE,9262
30
- java_codebase_rag/jrag_render.py,sha256=tIKdOQOO1-uimS2MBxrLma5tjiid0nG0OcaPRgR7uEk,31611
31
- java_codebase_rag/lance_optimize.py,sha256=25Rwj7HNO8F-35MxhFK6naqgbjd3H-T0zKb3pXB4H0s,9268
32
- java_codebase_rag/pipeline.py,sha256=4o5SdvHC4z4PTtzkjnhmkolODXnlibndCorI6gDUyPI,15903
30
+ java_codebase_rag/jrag_render.py,sha256=1nUyamL-MOOXDlKvssp-LsBgEtznUnGj_cPteSwuS90,31953
31
+ java_codebase_rag/lance_optimize.py,sha256=_90eajcIpGUNm3GWtx6AYDERNHaMsaMZs7zu2KqNPEU,10347
32
+ java_codebase_rag/pipeline.py,sha256=TkHb7DybFlpHje30aYbuFt5jWFSJBdCQ-7hc4jFaNhI,16481
33
33
  java_codebase_rag/progress.py,sha256=2IxdMALDM0wAQCyJrrfZ975zM_85C-4BfHxf4AtYifE,23212
34
- java_codebase_rag/install_data/agents/explorer-rag-cli.md,sha256=Ahjq39Kgyydgr_Vno-gFj02DZQQX-a65AvzEl1K1nss,12923
35
- java_codebase_rag/install_data/agents/explorer-rag-enhanced.md,sha256=O8twSh5aOVe2s_8HdwFKBTl_VxgwiVPm2YlyQAPXhh0,14614
36
- java_codebase_rag/install_data/skills/explore-codebase/SKILL.md,sha256=7SQSC0sHD_2F64QdEiwgSyVWHQXLxsOBxWWqJpIGPbk,12485
37
- java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md,sha256=qSGsC90LpjTZSUV6v3YtgFSHVb_hd4S1QlOedKm2NnU,14319
38
- java_codebase_rag-0.8.0.dist-info/licenses/LICENSE,sha256=gxvtiHtuviR_q8ZAjWw-QTcF3DyPzg6ZY-lQrr8OPpw,1068
39
- java_codebase_rag-0.8.0.dist-info/METADATA,sha256=VnPaESLzcc1V_ltQw6U1Ehyk6HV1TbTZjgGphLO6tWQ,19996
40
- java_codebase_rag-0.8.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
41
- java_codebase_rag-0.8.0.dist-info/entry_points.txt,sha256=cj3QTc11UYVQnj9T3orc4daiIGaCYrXP149vKbH2R4U,168
42
- java_codebase_rag-0.8.0.dist-info/top_level.txt,sha256=8vC-VN3cMwz5vhkSTaeJ1a1bDeqLWEfrTks1CvEvIg0,273
43
- java_codebase_rag-0.8.0.dist-info/RECORD,,
34
+ java_codebase_rag/install_data/agents/explorer-rag-cli.md,sha256=mMij_BIQM4agaYhGVYjC-fQSe3We1HFeBvc5JBjJj6A,10071
35
+ java_codebase_rag/install_data/agents/explorer-rag-enhanced.md,sha256=gZsNFbuK0lSnOIlplbbS_muz2ozokJqvFUv65QM0NDM,10406
36
+ java_codebase_rag/install_data/skills/explore-codebase/SKILL.md,sha256=A-v2dueVnxwBzBlxoRjZ2zOJk8DranLQ1TElwn94h0s,11529
37
+ java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md,sha256=V5gIKKGkgk2KlAFf9JqPareQDMt1iQOI7cwhfOqJL1c,11348
38
+ java_codebase_rag-0.9.0.dist-info/licenses/LICENSE,sha256=gxvtiHtuviR_q8ZAjWw-QTcF3DyPzg6ZY-lQrr8OPpw,1068
39
+ java_codebase_rag-0.9.0.dist-info/METADATA,sha256=pIQfvYY6sWdmq-PIXm0vOUzNobTUtW14Z-Zp-53qaf8,20088
40
+ java_codebase_rag-0.9.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
41
+ java_codebase_rag-0.9.0.dist-info/entry_points.txt,sha256=cj3QTc11UYVQnj9T3orc4daiIGaCYrXP149vKbH2R4U,168
42
+ java_codebase_rag-0.9.0.dist-info/top_level.txt,sha256=8vC-VN3cMwz5vhkSTaeJ1a1bDeqLWEfrTks1CvEvIg0,273
43
+ java_codebase_rag-0.9.0.dist-info/RECORD,,
mcp_v2.py CHANGED
@@ -466,6 +466,8 @@ class SearchHit(BaseModel):
466
466
  role: str | None = None
467
467
  filename: str | None = None
468
468
  start_line: int | None = None
469
+ score_components: dict[str, float] | None = None
470
+ chunks: int | None = None
469
471
 
470
472
 
471
473
  # NodeRef is now defined in graph_types.py and imported above
@@ -583,7 +585,7 @@ def _chunk_id_from_row(row: dict[str, Any]) -> str:
583
585
  return f"{filename}:{sb}:{eb}"
584
586
 
585
587
 
586
- def _row_to_search_hit(row: dict[str, Any]) -> SearchHit:
588
+ def _row_to_search_hit(row: dict[str, Any], explain: bool = False) -> SearchHit:
587
589
  score = float(row.get("_rrf_score") or row.get("_score") or 0.0)
588
590
  filename = str(row.get("filename") or "") or None
589
591
  start_line: int | None = None
@@ -595,6 +597,8 @@ def _row_to_search_hit(row: dict[str, Any]) -> SearchHit:
595
597
  start_line = int(ln)
596
598
  except (TypeError, ValueError):
597
599
  start_line = None
600
+ chunks = row.get("_chunks_collapsed")
601
+ chunks_int = int(chunks) if chunks is not None and int(chunks) >= 2 else None
598
602
  return SearchHit(
599
603
  chunk_id=_chunk_id_from_row(row),
600
604
  symbol_id=_chunk_to_symbol_id(row),
@@ -606,6 +610,8 @@ def _row_to_search_hit(row: dict[str, Any]) -> SearchHit:
606
610
  role=str(row.get("role")) if row.get("role") else None,
607
611
  filename=filename,
608
612
  start_line=start_line,
613
+ score_components=row.get("_score_components") if explain else None,
614
+ chunks=chunks_int,
609
615
  )
610
616
 
611
617
 
@@ -820,7 +826,9 @@ def search_v2(
820
826
  offset: int = 0,
821
827
  path_contains: str | None = None,
822
828
  filter: NodeFilter | dict[str, Any] | str | None = None,
829
+ explain: bool = False,
823
830
  graph: LadybugGraph | None = None,
831
+ dedup: bool = True,
824
832
  ) -> SearchOutput:
825
833
  try:
826
834
  raw_filter = _coerce_filter(filter)
@@ -852,6 +860,17 @@ def search_v2(
852
860
  limit=None,
853
861
  offset=None,
854
862
  )
863
+ # hybrid + table='all' is unsupported (hybrid fuses vector+FTS on ONE
864
+ # table); fail fast with a clean envelope BEFORE loading the embedding
865
+ # model. run_search also guards this — this is the user-facing fast path.
866
+ if hybrid and table == "all":
867
+ return SearchOutput(
868
+ success=False,
869
+ message="hybrid search requires a single table; use java, sql, or yaml (not all)",
870
+ advisories=[],
871
+ limit=None,
872
+ offset=None,
873
+ )
855
874
  model_name = resolved_sbert_model_for_process_env(SBERT_MODEL)
856
875
  device = os.environ.get("SBERT_DEVICE") or None
857
876
  model = _get_sentence_transformer(model_name, device)
@@ -862,29 +881,66 @@ def search_v2(
862
881
  if not uri.startswith(("s3://", "gs://", "az://")) and uri_path.exists():
863
882
  uri = str(uri_path.resolve())
864
883
  table_keys = list(TABLES) if table == "all" else [table]
865
- rows = run_search(
866
- query,
867
- uri=uri,
868
- table_keys=table_keys,
869
- hybrid=hybrid,
870
- limit=limit,
871
- offset=offset,
872
- path_substring=path_contains,
873
- model_name=model_name,
874
- device=device,
875
- model=model,
876
- # Push the NodeFilter structural predicates into the LanceDB query so
877
- # they apply BEFORE pagination (issue #353) — previously they were only
878
- # a post-filter on the already-paginated page, which could shrink or
879
- # empty filtered pages even when many matches existed deeper in the
880
- # ranking. _node_matches_filter below still re-checks every row (it
881
- # covers the non-pushdownable fields and is the contract guarantee).
882
- role=nf.role if nf else None,
883
- module=nf.module if nf else None,
884
- microservice=nf.microservice if nf else None,
885
- capability=nf.capability if nf else None,
886
- exclude_roles=nf.exclude_roles if nf else None,
887
- )
884
+
885
+ # Graceful fallback: if hybrid=True and FTS index is missing (old index),
886
+ # retry with hybrid=False and return vector-only results with an advisory.
887
+ advisories: list[str] = []
888
+ try:
889
+ rows = run_search(
890
+ query,
891
+ uri=uri,
892
+ table_keys=table_keys,
893
+ hybrid=hybrid,
894
+ limit=limit,
895
+ offset=offset,
896
+ path_substring=path_contains,
897
+ model_name=model_name,
898
+ device=device,
899
+ model=model,
900
+ # Push the NodeFilter structural predicates into the LanceDB query so
901
+ # they apply BEFORE pagination (issue #353) — previously they were only
902
+ # a post-filter on the already-paginated page, which could shrink or
903
+ # empty filtered pages even when many matches existed deeper in the
904
+ # ranking. _node_matches_filter below still re-checks every row (it
905
+ # covers the non-pushdownable fields and is the contract guarantee).
906
+ role=nf.role if nf else None,
907
+ module=nf.module if nf else None,
908
+ microservice=nf.microservice if nf else None,
909
+ capability=nf.capability if nf else None,
910
+ exclude_roles=nf.exclude_roles if nf else None,
911
+ dedup_by_fqn=dedup,
912
+ )
913
+ except Exception as exc:
914
+ # Check if this is a missing-FTS error (old index built before PR-SEARCH-3)
915
+ exc_text = str(exc).lower()
916
+ is_fts_missing = "full text search" in exc_text or "inverted index" in exc_text
917
+ if hybrid and is_fts_missing:
918
+ # Retry with vector-only search
919
+ rows = run_search(
920
+ query,
921
+ uri=uri,
922
+ table_keys=table_keys,
923
+ hybrid=False, # Fallback to vector-only
924
+ limit=limit,
925
+ offset=offset,
926
+ path_substring=path_contains,
927
+ model_name=model_name,
928
+ device=device,
929
+ model=model,
930
+ role=nf.role if nf else None,
931
+ module=nf.module if nf else None,
932
+ microservice=nf.microservice if nf else None,
933
+ capability=nf.capability if nf else None,
934
+ exclude_roles=nf.exclude_roles if nf else None,
935
+ dedup_by_fqn=dedup,
936
+ )
937
+ advisories.append(
938
+ f"hybrid unavailable on table '{table}' (FTS index missing on this index built before "
939
+ f"PR-SEARCH-3); fell back to vector-only — reindex to enable hybrid"
940
+ )
941
+ else:
942
+ # Non-FTS error: surface as structured failure
943
+ raise
888
944
  hits: list[SearchHit] = []
889
945
  for row in rows:
890
946
  if path_contains and path_contains not in str(row.get("filename") or ""):
@@ -893,7 +949,7 @@ def search_v2(
893
949
  row_kind = "symbol"
894
950
  if not _node_matches_filter(row_kind, row, nf):
895
951
  continue
896
- hits.append(_row_to_search_hit(row))
952
+ hits.append(_row_to_search_hit(row, explain=explain))
897
953
  hint_payload = {
898
954
  "success": True,
899
955
  "results": [h.model_dump() for h in hits],
@@ -906,7 +962,7 @@ def search_v2(
906
962
  results=hits,
907
963
  limit=limit,
908
964
  offset=offset,
909
- advisories=raw_advisories,
965
+ advisories=advisories + raw_advisories, # Merge fallback + hints advisories
910
966
  hints_structured=_to_structured_hints(raw_struct),
911
967
  )
912
968
  except Exception as exc:
search_lancedb.py CHANGED
@@ -41,6 +41,11 @@ JAVA_ENRICHED_COLUMNS: tuple[str, ...] = (
41
41
  "capabilities",
42
42
  )
43
43
 
44
+ # Over-fetch multiplier for dedup: fetch 4x to absorb per-FQN chunk multiplicity
45
+ # so that after collapsing by primary_type_fqn, a page stays full and the +1
46
+ # truncation sentinel survives. The formula: need = max((limit + offset) * 4, limit + offset + 1)
47
+ DEDUP_OVERFETCH = 4
48
+
44
49
  VECTOR_COLUMN = "embedding"
45
50
  _FTS_READY: set[tuple[str, str]] = set()
46
51
  _FTS_LOCK = threading.Lock()
@@ -201,6 +206,14 @@ _ROLE_SCORE_WEIGHTS: dict[str, float] = {
201
206
  "DTO": -0.08,
202
207
  }
203
208
 
209
+ # Theoretical maximum for hybrid composite score (used for display normalization).
210
+ # Hybrid sort metric: raw_rrf * (import_factor if import_heavy else 1)
211
+ # + role_weight + symbol_bonus
212
+ # where raw_rrf ≤ 2/(k+1) for 2-list RRF, role_weight ≤ max(_ROLE_SCORE_WEIGHTS),
213
+ # and symbol_bonus ≤ _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS.
214
+ # The import factor is ≤ 1, so we use the raw max (2/61).
215
+ _HYBRID_SCORE_MAX = (2.0 / 61.0) + max(_ROLE_SCORE_WEIGHTS.values()) + _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS
216
+
204
217
 
205
218
  def _query_tokens(query: str) -> set[str]:
206
219
  """Lowercased alpha-only tokens from the query, minus stopwords, len >= 3.
@@ -353,7 +366,7 @@ def _hybrid_sort_key(r: dict) -> float:
353
366
  comps["hybrid_rrf"] = s
354
367
  if r.get("_hints", {}).get("import_heavy"):
355
368
  s *= _IMPORT_HYBRID_SCORE_FACTOR
356
- comps["import_penalty"] = _IMPORT_HYBRID_SCORE_FACTOR
369
+ comps["import_penalty"] = 1.0 - _IMPORT_HYBRID_SCORE_FACTOR
357
370
  s += _role_weight(r)
358
371
  s += float(comps.get("symbol_bonus", 0.0))
359
372
  return -s
@@ -376,7 +389,8 @@ def explain_score_components(
376
389
  comps = {}
377
390
  parts: list[str] = []
378
391
  if hybrid:
379
- rrf = comps.get("hybrid_rrf")
392
+ # Prefer rrf_raw (added by PR-SEARCH-1a) for explanation
393
+ rrf = comps.get("rrf_raw") or comps.get("hybrid_rrf")
380
394
  if rrf is not None:
381
395
  parts.append(f"rrf={float(rrf):.3f}")
382
396
  else:
@@ -403,6 +417,46 @@ def l2_distance_to_score(distance: float) -> float:
403
417
  return 1.0 - distance * distance / 2.0
404
418
 
405
419
 
420
+ def _effective_distance(comps: dict[str, float]) -> float:
421
+ """Compute the adjusted distance used for sorting.
422
+
423
+ Matches _vector_sort_key logic: distance + import_penalty - role_weight - symbol_bonus.
424
+ """
425
+ d = comps.get("distance", 0.0)
426
+ d += comps.get("import_penalty", 0.0)
427
+ d -= comps.get("role_weight", 0.0)
428
+ d -= comps.get("symbol_bonus", 0.0)
429
+ return d
430
+
431
+
432
+ def _clamp01(x: float) -> float:
433
+ """Clamp a value to the [0.0, 1.0] range."""
434
+ if x < 0.0:
435
+ return 0.0
436
+ if x > 1.0:
437
+ return 1.0
438
+ return x
439
+
440
+
441
+ def _hybrid_post_sort_normalization(rows: list[dict]) -> None:
442
+ """Set honest displayed scores for hybrid search after sorting.
443
+
444
+ Reconstructs the composite score (raw_rrf * import_factor + role_weight + symbol_bonus)
445
+ and normalizes by _HYBRID_SCORE_MAX to ensure rank-monotonicity.
446
+
447
+ Mutates rows in-place, replacing _score with the normalized value.
448
+ """
449
+ for r in rows:
450
+ comps = r.setdefault("_score_components", {})
451
+ raw = comps.get("hybrid_rrf", 0.0)
452
+ comps["rrf_raw"] = raw # preserve raw RRF for --explain. NOTE: when graph_expand + hybrid combine (Phase 2), _rrf_merge below overwrites this with graph-RRF, so --explain would show graph-RRF not hybrid-RRF.
453
+ s = raw
454
+ if r.get("_hints", {}).get("import_heavy"):
455
+ s *= _IMPORT_HYBRID_SCORE_FACTOR
456
+ s += comps.get("role_weight", 0.0) + comps.get("symbol_bonus", 0.0)
457
+ r["_score"] = _clamp01(s / _HYBRID_SCORE_MAX)
458
+
459
+
406
460
  def _escape_like_fragment(s: str) -> str:
407
461
  return s.replace("'", "''")
408
462
 
@@ -790,9 +844,72 @@ def _rrf_merge(
790
844
  existing["_rrf_score"] = float(existing.get("_rrf_score", 0.0)) + contribution
791
845
  merged = list(pool.values())
792
846
  merged.sort(key=lambda r: -float(r.get("_rrf_score", 0.0)))
847
+ # Normalize displayed _rrf_score to [0,1] by theoretical max
848
+ # RRF max = Σ weight·1/(k+rank+1); theoretical max when all rows are rank 0
849
+ # with weight 1.0 = num_lists / (k + 1)
850
+ num_lists = len(lists)
851
+ max_rrf = num_lists / (k + 1)
852
+ for r in merged:
853
+ raw_score = float(r.get("_rrf_score", 0.0))
854
+ comps = r.setdefault("_score_components", {})
855
+ comps["rrf_raw"] = raw_score
856
+ r["_rrf_score"] = _clamp01(raw_score / max_rrf)
793
857
  return merged
794
858
 
795
859
 
860
+ def _dedup_by_fqn(rows: list[dict], dedup_by_fqn: bool = True) -> list[dict]:
861
+ """Deduplicate rows by primary_type_fqn (java table only).
862
+
863
+ When dedup_by_fqn is True, collapses multiple chunks of the same
864
+ primary_type_fqn into one row (first-seen-wins, since rows are pre-sorted
865
+ so the first is the best chunk). Each survivor gets a _chunks_collapsed
866
+ field (>=1) counting how many rows were collapsed into it.
867
+
868
+ Rows without primary_type_fqn (sql/yaml tables) get a unique __id:<id>
869
+ key so they pass through unchanged (each row is unique).
870
+
871
+ When dedup_by_fqn is False, returns rows unchanged (regression guard).
872
+ """
873
+ if not dedup_by_fqn:
874
+ # Non-dedup path: return unchanged, byte-identical to prior behavior
875
+ return rows
876
+
877
+ deduped: list[dict] = []
878
+ seen_keys: dict[str, dict] = {}
879
+ collapsed_counts: dict[str, int] = {}
880
+
881
+ for row in rows:
882
+ # Build dedup key: primary_type_fqn for java rows, unique __id:<id> for sql/yaml
883
+ fqn = row.get("primary_type_fqn")
884
+ if fqn:
885
+ key = str(fqn)
886
+ else:
887
+ # sql/yaml rows have no primary_type_fqn → unique key per row
888
+ row_id = row.get("id") or id(row)
889
+ key = f"__id:{row_id}"
890
+
891
+ if key not in seen_keys:
892
+ # First occurrence: keep it
893
+ seen_keys[key] = row
894
+ collapsed_counts[key] = 1
895
+ deduped.append(row)
896
+ else:
897
+ # Duplicate: increment collapse count, discard this row
898
+ collapsed_counts[key] += 1
899
+
900
+ # Annotate each survivor with _chunks_collapsed
901
+ for row in deduped:
902
+ fqn = row.get("primary_type_fqn")
903
+ if fqn:
904
+ key = str(fqn)
905
+ else:
906
+ row_id = row.get("id") or id(row)
907
+ key = f"__id:{row_id}"
908
+ row["_chunks_collapsed"] = collapsed_counts[key]
909
+
910
+ return deduped
911
+
912
+
796
913
  def run_search(
797
914
  query: str,
798
915
  *,
@@ -819,6 +936,7 @@ def run_search(
819
936
  exclude_roles: list[str] | None = None,
820
937
  capability: str | None = None,
821
938
  capability_in: list[str] | None = None,
939
+ dedup_by_fqn: bool = False,
822
940
  ) -> list[dict]:
823
941
  effective_hybrid = hybrid
824
942
  effective_fts = fts_text
@@ -852,7 +970,16 @@ def run_search(
852
970
  fts_for_hybrid = effective_fts if effective_fts is not None else query
853
971
 
854
972
  db = lancedb.connect(uri)
855
- need = max(limit + offset, 1)
973
+ if dedup_by_fqn:
974
+ # Over-fetch to absorb per-FQN chunk multiplicity: fetch 4x so that
975
+ # after collapsing, the page stays full and the +1 truncation sentinel survives.
976
+ # The 4× factor assumes typical per-FQN chunk multiplicity; a single type with
977
+ # many high-ranking chunks (e.g. generated/God classes) could starve the page or
978
+ # make the +1 truncation sentinel unreliable; Phase 1 may revisit adaptive over-fetch (plan risk #1).
979
+ need = max((limit + offset) * DEDUP_OVERFETCH, limit + offset + 1)
980
+ else:
981
+ # Non-dedup path: exact fetch as before
982
+ need = max(limit + offset, 1)
856
983
 
857
984
  extra_java = _build_extra_predicates(
858
985
  columns=_table_columns(uri, TABLES["java"], db),
@@ -887,8 +1014,15 @@ def run_search(
887
1014
  _apply_symbol_bonus(rows, query_toks)
888
1015
  if effective_hybrid:
889
1016
  rows.sort(key=_hybrid_sort_key)
1017
+ # Hybrid: set honest displayed score from composite sort metric, clamped to [0,1]
1018
+ _hybrid_post_sort_normalization(rows)
890
1019
  else:
891
1020
  rows.sort(key=_vector_sort_key)
1021
+ # Vector: set honest displayed score from adjusted distance, clamped to [0,1]
1022
+ for r in rows:
1023
+ comps = r.setdefault("_score_components", {})
1024
+ effective_dist = _effective_distance(comps)
1025
+ r["_score"] = _clamp01(l2_distance_to_score(effective_dist))
892
1026
 
893
1027
  if graph_expand and key == "java" and expand_depth > 0:
894
1028
  rows = _graph_expand_merge(
@@ -902,6 +1036,9 @@ def run_search(
902
1036
  ladybug_path=ladybug_path,
903
1037
  )
904
1038
 
1039
+ # Dedup by primary_type_fqn after all sorting/merging, before windowing
1040
+ rows = _dedup_by_fqn(rows, dedup_by_fqn=dedup_by_fqn)
1041
+
905
1042
  window = rows[offset : offset + limit]
906
1043
  if context_neighbors > 0 and key == "java":
907
1044
  _attach_neighbor_context(window, db=db, neighbors=context_neighbors, uri=uri)
@@ -931,6 +1068,15 @@ def run_search(
931
1068
  r["_skip_role_weight"] = True
932
1069
  _apply_symbol_bonus(merged, query_toks)
933
1070
  merged.sort(key=_vector_sort_key)
1071
+ # Vector: set honest displayed score from adjusted distance, clamped to [0,1]
1072
+ for r in merged:
1073
+ comps = r.setdefault("_score_components", {})
1074
+ effective_dist = _effective_distance(comps)
1075
+ r["_score"] = _clamp01(l2_distance_to_score(effective_dist))
1076
+
1077
+ # Dedup by primary_type_fqn after all sorting/merging, before windowing
1078
+ merged = _dedup_by_fqn(merged, dedup_by_fqn=dedup_by_fqn)
1079
+
934
1080
  window = merged[offset : offset + limit]
935
1081
  if context_neighbors > 0:
936
1082
  _attach_neighbor_context(window, db=db, neighbors=context_neighbors, uri=uri)
server.py CHANGED
@@ -512,6 +512,7 @@ def create_mcp_server() -> FastMCP:
512
512
  "structured DSL inside `query`; structured predicates belong in `find`. "
513
513
  "For identifier-shaped lookups (FQN, id, route/client identifiers, …), use `resolve` first; "
514
514
  "use `search` for natural-language or ranked fuzzy discovery. "
515
+ "Set `explain=true` to include score breakdown per hit. "
515
516
  "Successful responses echo `limit`/`offset`."
516
517
  ),
517
518
  )
@@ -538,6 +539,14 @@ def create_mcp_server() -> FastMCP:
538
539
  "predicate. Unknown keys or populated fields not applicable to symbols return success=false."
539
540
  ),
540
541
  ),
542
+ explain: bool = Field(
543
+ default=False,
544
+ description="If true, include score_components in each SearchHit (breakdown of distance/rrf, role, symbol, import_penalty).",
545
+ ),
546
+ chunks: bool = Field(
547
+ default=False,
548
+ description="If true, show every chunk (default collapses to one row per symbol/type).",
549
+ ),
541
550
  ) -> mcp_v2.SearchOutput:
542
551
  scoped_filter = _scope_manager.apply_auto_scope(filter) if _scope_manager else filter
543
552
  return await asyncio.to_thread(
@@ -549,7 +558,9 @@ def create_mcp_server() -> FastMCP:
549
558
  offset,
550
559
  path_contains,
551
560
  scoped_filter,
561
+ explain,
552
562
  None,
563
+ not chunks, # dedup=True by default; chunks=True opts out
553
564
  )
554
565
 
555
566
  @mcp.tool(