java-codebase-rag 0.10.2__py3-none-any.whl → 0.11.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -20,6 +20,7 @@ from sentence_transformers import SentenceTransformer
20
20
 
21
21
  from java_codebase_rag.ast.chunk_heuristics import analyze_chunk, looks_like_code_identifier
22
22
  from java_codebase_rag.search.index_common import SBERT_MODEL
23
+ from java_codebase_rag.search import search_lexical
23
24
  from java_codebase_rag.config import maybe_expand_embedding_model_path, resolved_sbert_model_for_process_env
24
25
 
25
26
  # Scoring & dedup primitives live in `search_scoring` (dependency-free — no
@@ -27,7 +28,11 @@ from java_codebase_rag.config import maybe_expand_embedding_model_path, resolved
27
28
  # graph-only (macOS Intel) installs where this module is unimportable. Re-exported
28
29
  # here for backward compatibility (`from search_lancedb import _clamp01`, etc.).
29
30
  from java_codebase_rag.search.search_scoring import ( # noqa: F401
31
+ BASELINE_2LIST_CONFIG,
32
+ DEFAULT_RANK_CONFIG,
30
33
  DEDUP_OVERFETCH,
34
+ RankConfig,
35
+ build_fts_query,
31
36
  _ACTION_VERB_BONUS,
32
37
  _ACTION_VERB_PREFIXES,
33
38
  _HYBRID_SCORE_MAX,
@@ -632,9 +637,150 @@ def _attach_neighbor_context(
632
637
  _debug_ctx(f"attached context to {attached}/{len(java_rows)} java rows")
633
638
 
634
639
 
640
+ def _bm25_candidate_rows(
641
+ *,
642
+ g: object,
643
+ query: str,
644
+ uri: str,
645
+ db: object,
646
+ extra_predicates: list[str],
647
+ columns: set[str],
648
+ limit: int = 100,
649
+ ) -> list[dict]:
650
+ """Fetch BM25-ranked Symbol candidates from the FTS index and resolve them to
651
+ chunk rows in BM25 rank order. Returns ``[]`` on any failure (silent degradation).
652
+
653
+ Pipeline:
654
+ 1. ``search_lexical.fetch_fts_candidates(g, query)`` → BM25-ranked Symbols +
655
+ a ``{symbol_node_id: bm25_score}`` map. ``None`` / empty → return ``[]``.
656
+ 2. Map each Symbol fqn to its enclosing TYPE fqn (``primary_type_fqn`` has no
657
+ ``#``; a member ``Type#method`` maps to ``Type``). Dedupe by type fqn,
658
+ keeping the MAX BM25 score among same-type symbols.
659
+ 3. Order type fqns by BM25 desc (fqn asc tiebreak — deterministic).
660
+ 4. Fetch chunk rows from LanceDB with a FILTER-ONLY query (no vector ranking,
661
+ so BM25 order is preserved). Predicates = caller's ``extra_predicates`` +
662
+ the ``primary_type_fqn IN (...)`` built from the ordered types — preserving
663
+ filter parity with the vector path.
664
+ 5. Group fetched chunks by ``primary_type_fqn``; emit in BM25 rank order,
665
+ each chunk carrying ``_score_components["bm25"]``.
666
+ 6. Apply ``_apply_chunk_hints`` + ``_refine_java_start_lines`` for consistency
667
+ with graph_rows handling.
668
+
669
+ Any exception (FTS or LanceDB) → ``_debug_ctx`` log + return ``[]`` (silent
670
+ degradation; the vector path is unaffected).
671
+ """
672
+ # 1. BM25 candidate fetch via the FTS index.
673
+ # Pre-split the query with the same tokenizer the ``sym_fts`` index uses
674
+ # (``search_text`` stores ``_split_identifier`` tokens). LadybugDB FTS's own
675
+ # tokenizer does NOT split camelCase, so a raw ``DistributionChunkService``
676
+ # would match nothing — ``build_fts_query`` mirrors what the lexical backend
677
+ # does at search_lexical.py (run_lexical_search), keeping index/query token
678
+ # spaces aligned. An empty split (degenerate / stopword-only query) → no FTS
679
+ # candidates → degrade silently to the vector path.
680
+ fts_query = build_fts_query(query)
681
+ if not fts_query or not fts_query.strip():
682
+ return []
683
+ try:
684
+ fts = search_lexical.fetch_fts_candidates(g, fts_query, filter=None, path_contains=None)
685
+ except Exception as exc: # noqa: BLE001 — silent degradation
686
+ _debug_ctx(f"bm25 FTS fetch raised: {exc!r}")
687
+ return []
688
+ if not fts or not fts.get("rows"):
689
+ return []
690
+ sym_rows = fts["rows"]
691
+ scores = fts.get("scores") or {}
692
+
693
+ # 2. Map symbol fqns → enclosing type fqns; keep MAX bm25 per type.
694
+ type_fqn_to_bm25: dict[str, float] = {}
695
+ for r in sym_rows:
696
+ fqn = r.get("fqn")
697
+ if not fqn:
698
+ continue
699
+ type_fqn = search_lexical.enclosing_type_fqn(str(fqn))
700
+ if not type_fqn:
701
+ continue
702
+ score = float(scores.get(r.get("id"), 0.0))
703
+ prev = type_fqn_to_bm25.get(type_fqn)
704
+ if prev is None or score > prev:
705
+ type_fqn_to_bm25[type_fqn] = score
706
+ if not type_fqn_to_bm25:
707
+ return []
708
+
709
+ # 3. Deterministic ordering: BM25 desc, fqn asc.
710
+ ordered_types = sorted(
711
+ type_fqn_to_bm25.keys(),
712
+ key=lambda f: (-type_fqn_to_bm25[f], f),
713
+ )
714
+
715
+ # 4. Filter-only chunk fetch (NO vector ranking → BM25 order preserved). The
716
+ # ``primary_type_fqn IN (...)`` predicate must be buildable; if the index is so
717
+ # old that the column is absent, we can't restrict the fetch and degrade to [].
718
+ if "primary_type_fqn" not in columns:
719
+ _debug_ctx("bm25 fetch skipped: primary_type_fqn column absent from schema")
720
+ return []
721
+ preds = list(extra_predicates) + _build_extra_predicates(
722
+ columns=columns,
723
+ role=None, module=None, microservice=None,
724
+ package_prefix=None, fqn_in=ordered_types,
725
+ )
726
+ combined_pred = _combine_predicates(preds)
727
+ base_cols = ["filename", "text", "start", "end"]
728
+ for col in ("range_start", "range_end"):
729
+ if col in columns:
730
+ base_cols.append(col)
731
+ java_extra = [c for c in JAVA_ENRICHED_COLUMNS if c in columns]
732
+ select_cols = [*base_cols, "language", *java_extra]
733
+
734
+ try:
735
+ tbl = db.open_table(TABLES["java"])
736
+ # LanceDB 0.34 filter-only path: search() with no vector arg issues a
737
+ # non-vector scan; .where/.select/.limit/.to_list returns rows in table
738
+ # order without re-ranking by similarity. (tbl.query() is NOT available in
739
+ # 0.34; to_lance().scanner() requires pylance, which isn't installed on the
740
+ # PEP 508 graph-only profile — search() with no vector is the supported API.)
741
+ q = tbl.search().select(select_cols).limit(
742
+ max(limit, len(ordered_types) * 4)
743
+ )
744
+ if combined_pred:
745
+ q = q.where(combined_pred, prefilter=True)
746
+ with _silence_lance_autoproj_warnings():
747
+ fetched = q.to_list()
748
+ except Exception as exc: # noqa: BLE001 — silent degradation
749
+ _debug_ctx(f"bm25 chunk fetch failed: {exc!r}")
750
+ return []
751
+
752
+ # 5. Group by primary_type_fqn; emit in BM25 rank order.
753
+ by_type: dict[str, list[dict]] = {}
754
+ for ch in fetched:
755
+ tf = ch.get("primary_type_fqn")
756
+ if tf is None:
757
+ continue
758
+ by_type.setdefault(str(tf), []).append(ch)
759
+
760
+ out: list[dict] = []
761
+ for type_fqn in ordered_types:
762
+ chunks = by_type.get(type_fqn)
763
+ if not chunks:
764
+ continue # filtered out by extra_predicates / absent from index
765
+ bm25_val = round(float(type_fqn_to_bm25[type_fqn]), 4)
766
+ for ch in chunks:
767
+ ch["_kind"] = "java"
768
+ ch["_hybrid"] = False
769
+ ch.setdefault("_score_components", {})["bm25"] = bm25_val
770
+ ch["start"] = coerce_position_field(ch.get("start"))
771
+ ch["end"] = coerce_position_field(ch.get("end"))
772
+ out.append(ch)
773
+
774
+ # 6. Consistency with graph_rows handling.
775
+ _apply_chunk_hints(out)
776
+ _refine_java_start_lines(out)
777
+ return out
778
+
779
+
635
780
  def _graph_expand_merge(
636
781
  vector_rows: list[dict],
637
782
  *,
783
+ query: str,
638
784
  query_vec: np.ndarray,
639
785
  db: object,
640
786
  uri: str,
@@ -642,8 +788,24 @@ def _graph_expand_merge(
642
788
  extra_predicates: list[str],
643
789
  expand_depth: int,
644
790
  ladybug_path: str | None,
791
+ rank_config: RankConfig = DEFAULT_RANK_CONFIG,
645
792
  ) -> list[dict]:
646
- """Expand vector top-k through the LadybugDB graph and fuse (RRF) with the original list."""
793
+ """Expand vector top-k through the graph and/or fuse BM25, then RRF-merge.
794
+
795
+ Which lists contribute is controlled by ``rank_config.lists``:
796
+ - ``"vector"`` — always present (the backbone; validated by RankConfig).
797
+ - ``"graph"`` — graph expand + fetch (skipped entirely when absent).
798
+ - ``"bm25"`` — LadybugDB FTS candidate fetch fused as a third list.
799
+
800
+ Silent degradation: any failure in the graph or BM25 path drops just that list;
801
+ the vector list is never lost. Returns ``vector_rows`` unchanged when no
802
+ auxiliary list yields rows.
803
+ """
804
+ want_graph = "graph" in rank_config.lists
805
+ want_bm25 = "bm25" in rank_config.lists
806
+ if not want_graph and not want_bm25:
807
+ return vector_rows
808
+
647
809
  # Lazy import so the module works without ladybug installed when graph_expand=False.
648
810
  try:
649
811
  from java_codebase_rag.graph.ladybug_queries import LadybugGraph
@@ -653,67 +815,97 @@ def _graph_expand_merge(
653
815
  if not LadybugGraph.exists(ladybug_path):
654
816
  return vector_rows
655
817
 
656
- seed_fqns = sorted({r.get("primary_type_fqn") for r in vector_rows if r.get("primary_type_fqn")})
657
- if not seed_fqns:
658
- return vector_rows
818
+ java_cols = _table_columns(uri, TABLES["java"], db)
659
819
 
660
- try:
661
- graph = LadybugGraph.get(ladybug_path)
662
- structural = graph.expand_fqns(seed_fqns, depth=expand_depth)
663
- method_pairs = graph.expand_methods(
664
- seed_fqns, depth=expand_depth, exclude_external=True,
665
- )
666
- expand_weight_by_fqn: dict[str, float] = {}
667
- for f in structural:
668
- if f:
669
- expand_weight_by_fqn[f] = max(expand_weight_by_fqn.get(f, 0.0), 1.0)
670
- for f, conf in method_pairs:
671
- if f:
672
- expand_weight_by_fqn[f] = max(expand_weight_by_fqn.get(f, 0.0), conf)
673
- neighbor_fqns = list(dict.fromkeys(
674
- list(structural) + [f for f, _ in method_pairs],
675
- ))
676
- except Exception:
677
- return vector_rows
820
+ # --- graph list ---
821
+ graph_rows: list[dict] = []
822
+ expand_weight_by_fqn: dict[str, float] = {}
823
+ if want_graph:
824
+ seed_fqns = sorted({r.get("primary_type_fqn") for r in vector_rows if r.get("primary_type_fqn")})
825
+ neighbor_fqns: list[str] = []
826
+ if seed_fqns:
827
+ try:
828
+ graph_obj = LadybugGraph.get(ladybug_path)
829
+ structural = graph_obj.expand_fqns(seed_fqns, depth=expand_depth)
830
+ method_pairs = graph_obj.expand_methods(
831
+ seed_fqns, depth=expand_depth, exclude_external=True,
832
+ )
833
+ for f in structural:
834
+ if f:
835
+ expand_weight_by_fqn[f] = max(expand_weight_by_fqn.get(f, 0.0), 1.0)
836
+ for f, conf in method_pairs:
837
+ if f:
838
+ expand_weight_by_fqn[f] = max(expand_weight_by_fqn.get(f, 0.0), conf)
839
+ neighbor_fqns = list(dict.fromkeys(
840
+ list(structural) + [f for f, _ in method_pairs],
841
+ ))
842
+ except Exception:
843
+ neighbor_fqns = []
844
+
845
+ novel = [fqn for fqn in neighbor_fqns if fqn and fqn not in set(seed_fqns)]
846
+ if novel:
847
+ extra = list(extra_predicates)
848
+ extra.extend(_build_extra_predicates(
849
+ columns=java_cols,
850
+ role=None, module=None, microservice=None,
851
+ package_prefix=None, fqn_in=novel,
852
+ ))
853
+ try:
854
+ graph_rows = _search_one_table(
855
+ TABLES["java"],
856
+ uri=uri, db=db, query_vec=query_vec,
857
+ limit=max(limit, 20),
858
+ path_predicate=None, kind="java",
859
+ hybrid=False, fts_text=None,
860
+ extra_predicates=extra,
861
+ )
862
+ except Exception:
863
+ graph_rows = []
864
+ _apply_chunk_hints(graph_rows)
865
+ _refine_java_start_lines(graph_rows)
866
+ graph_rows.sort(key=_vector_sort_key)
867
+ for r in graph_rows:
868
+ r["_graph_expanded"] = True
869
+ r["_graph_expand_weight"] = expand_weight_by_fqn.get(
870
+ r.get("primary_type_fqn"), 1.0,
871
+ )
872
+
873
+ # --- bm25 list ---
874
+ bm25_rows: list[dict] = []
875
+ if want_bm25:
876
+ try:
877
+ graph_obj = LadybugGraph.get(ladybug_path)
878
+ except Exception:
879
+ graph_obj = None
880
+ if graph_obj is not None:
881
+ bm25_rows = _bm25_candidate_rows(
882
+ g=graph_obj,
883
+ query=query,
884
+ uri=uri,
885
+ db=db,
886
+ extra_predicates=extra_predicates,
887
+ columns=java_cols,
888
+ limit=limit,
889
+ )
678
890
 
679
- novel = [fqn for fqn in neighbor_fqns if fqn and fqn not in set(seed_fqns)]
680
- if not novel:
891
+ # --- RRF fusion (only lists that yielded rows beyond vector) ---
892
+ lists: list[list[dict]] = [vector_rows]
893
+ row_weights: list[Callable[[dict], float] | None] = [None]
894
+ if want_graph and graph_rows:
895
+ lists.append(graph_rows)
896
+ row_weights.append(lambda row: float(row.get("_graph_expand_weight", 1.0)))
897
+ if want_bm25 and bm25_rows:
898
+ lists.append(bm25_rows)
899
+ row_weights.append(None)
900
+
901
+ if len(lists) == 1:
681
902
  return vector_rows
682
903
 
683
- extra = list(extra_predicates)
684
- extra.extend(_build_extra_predicates(
685
- columns=_table_columns(uri, TABLES["java"], db),
686
- role=None, module=None, microservice=None,
687
- package_prefix=None, fqn_in=novel,
688
- ))
689
-
690
- try:
691
- graph_rows = _search_one_table(
692
- TABLES["java"],
693
- uri=uri, db=db, query_vec=query_vec,
694
- limit=max(limit, 20),
695
- path_predicate=None, kind="java",
696
- hybrid=False, fts_text=None,
697
- extra_predicates=extra,
698
- )
699
- except Exception:
700
- return vector_rows
701
- _apply_chunk_hints(graph_rows)
702
- _refine_java_start_lines(graph_rows)
703
- graph_rows.sort(key=_vector_sort_key)
704
- for r in graph_rows:
705
- r["_graph_expanded"] = True
706
- r["_graph_expand_weight"] = expand_weight_by_fqn.get(
707
- r.get("primary_type_fqn"), 1.0,
708
- )
709
- fused = _rrf_merge(
710
- [vector_rows, graph_rows],
711
- row_weight_for_list_index=[
712
- None,
713
- lambda row: float(row.get("_graph_expand_weight", 1.0)),
714
- ],
904
+ return _rrf_merge(
905
+ lists,
906
+ k=rank_config.rrf_k,
907
+ row_weight_for_list_index=row_weights,
715
908
  )
716
- return fused
717
909
 
718
910
 
719
911
  def _rrf_merge(
@@ -790,6 +982,7 @@ def run_search(
790
982
  generated_only: bool = False,
791
983
  exclude_generated: bool = False,
792
984
  dedup_by_fqn: bool = False,
985
+ rank_config: RankConfig = DEFAULT_RANK_CONFIG,
793
986
  ) -> list[dict]:
794
987
  effective_hybrid = hybrid
795
988
  effective_fts = fts_text
@@ -887,6 +1080,7 @@ def run_search(
887
1080
  if graph_expand and key == "java" and expand_depth > 0:
888
1081
  rows = _graph_expand_merge(
889
1082
  rows,
1083
+ query=query,
890
1084
  query_vec=query_vec,
891
1085
  db=db,
892
1086
  uri=uri,
@@ -894,6 +1088,7 @@ def run_search(
894
1088
  extra_predicates=extra_java,
895
1089
  expand_depth=expand_depth,
896
1090
  ladybug_path=ladybug_path,
1091
+ rank_config=rank_config,
897
1092
  )
898
1093
 
899
1094
  # Dedup by primary_type_fqn after all sorting/merging, before windowing
@@ -127,6 +127,11 @@ def _enclosing_type_fqn(fqn: str) -> str:
127
127
  return fqn.split("#", 1)[0] if fqn else fqn
128
128
 
129
129
 
130
+ # Non-underscore aliases for cross-module callers (search_lancedb's BM25 fusion).
131
+ # Behavior is identical; the leading-underscore originals stay module-private.
132
+ enclosing_type_fqn = _enclosing_type_fqn
133
+
134
+
130
135
  def _resolve_source_root(graph: LadybugGraph) -> str:
131
136
  """Authoritative source root is the one cached on the graph at index time."""
132
137
  try:
@@ -255,6 +260,11 @@ def _try_fts_candidates(
255
260
  return {"rows": rows, "scores": scores}
256
261
 
257
262
 
263
+ # Non-underscore alias for cross-module callers (search_lancedb's BM25 fusion on the
264
+ # vector path). Behavior is identical to the leading-underscore original.
265
+ fetch_fts_candidates = _try_fts_candidates
266
+
267
+
258
268
  def run_lexical_search(
259
269
  query: str,
260
270
  *,
@@ -13,6 +13,7 @@ from __future__ import annotations
13
13
 
14
14
  import json
15
15
  import re
16
+ from dataclasses import dataclass
16
17
 
17
18
  # Name of the LadybugDB FTS (Okapi BM25) index over Symbol.search_text (fork A).
18
19
  # Shared by the build path (build_ast_graph._ensure_symbol_fts_index) and the
@@ -84,13 +85,86 @@ _ROLE_SCORE_WEIGHTS: dict[str, float] = {
84
85
  "DTO": -0.08,
85
86
  }
86
87
 
88
+
89
+ def _rrf_max(num_lists: int, k: int = 60) -> float:
90
+ """Return the theoretical maximum RRF score for N-list fusion.
91
+
92
+ Reciprocal Rank Fusion (RRF) bounds each contribution to ≤ 1/(rank + k).
93
+ For N fused lists, the maximum possible sum is N/(k + 1) (achieved when
94
+ an item ranks #1 across all lists).
95
+
96
+ Args:
97
+ num_lists: Number of ranked lists being fused (e.g., 2 for vector+lexical).
98
+ k: The RRF constant (default 60 per the original paper).
99
+
100
+ Returns:
101
+ The maximum RRF contribution: num_lists / (k + 1).
102
+ """
103
+ return num_lists / (k + 1)
104
+
105
+
106
+ # Allowed list names in a RankConfig: vector is always required (it is the
107
+ # backbone retrieval signal); graph and bm25 are optional fusion participants.
108
+ # The "bm25" list is wired in ``search_lancedb._graph_expand_merge`` (LadybugDB
109
+ # FTS candidate fetch), which fuses BM25-ranked Symbol candidates as a third
110
+ # RRF list alongside vector and graph.
111
+ _RANK_LIST_NAMES: frozenset[str] = frozenset({"vector", "graph", "bm25"})
112
+
113
+
114
+ @dataclass(frozen=True)
115
+ class RankConfig:
116
+ """Which ranked lists to fuse and the RRF constant to fuse them with.
117
+
118
+ This is a dep-free value object (no lancedb/torch) so it can be constructed
119
+ on every install flavor, including graph-only (macOS Intel). It is plumbed
120
+ through ``run_search`` → ``_graph_expand_merge`` to control (a) which lists
121
+ contribute to the final RRF fusion and (b) the ``k`` constant passed into
122
+ ``_rrf_merge``.
123
+
124
+ Attributes:
125
+ lists: Subset of ``{"vector", "graph", "bm25"}``. Must contain
126
+ ``"vector"`` (the backbone retrieval signal) and be non-empty.
127
+ rrf_k: The RRF constant (default 60 per the original paper). Must be ≥ 1.
128
+ """
129
+
130
+ lists: frozenset[str]
131
+ rrf_k: int = 60
132
+
133
+ def __post_init__(self) -> None:
134
+ if not isinstance(self.lists, frozenset) or not self.lists:
135
+ raise ValueError("RankConfig.lists must be a non-empty frozenset")
136
+ if "vector" not in self.lists:
137
+ raise ValueError(
138
+ "RankConfig.lists must contain 'vector' (the backbone signal)"
139
+ )
140
+ unknown = self.lists - _RANK_LIST_NAMES
141
+ if unknown:
142
+ raise ValueError(
143
+ f"RankConfig.lists has unknown names {sorted(unknown)!r}; "
144
+ f"allowed: {sorted(_RANK_LIST_NAMES)!r}"
145
+ )
146
+ if not isinstance(self.rrf_k, int) or self.rrf_k < 1:
147
+ raise ValueError(f"RankConfig.rrf_k must be an int >= 1, got {self.rrf_k!r}")
148
+
149
+
150
+ # Production default: ship the 3-list config (vector+graph+bm25). The BM25 list
151
+ # is wired in ``search_lancedb._graph_expand_merge``; on installs without the
152
+ # vector stack, or when the FTS index is unavailable, it degrades silently to
153
+ # the 2-list (vector+graph) fusion.
154
+ DEFAULT_RANK_CONFIG = RankConfig(lists=frozenset({"vector", "graph", "bm25"}), rrf_k=60)
155
+
156
+ # Eval convenience: the historical 2-list (vector+graph) fusion, used by
157
+ # evaluation harnesses that isolate the vector+graph baseline from the bm25 list.
158
+ BASELINE_2LIST_CONFIG = RankConfig(lists=frozenset({"vector", "graph"}), rrf_k=60)
159
+
160
+
87
161
  # Theoretical maximum for hybrid composite score (used for display normalization).
88
162
  # Hybrid sort metric: raw_rrf * (import_factor if import_heavy else 1)
89
163
  # + role_weight + symbol_bonus
90
164
  # where raw_rrf ≤ 2/(k+1) for 2-list RRF, role_weight ≤ max(_ROLE_SCORE_WEIGHTS),
91
165
  # and symbol_bonus ≤ _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS.
92
- # The import factor is ≤ 1, so we use the raw max (2/61).
93
- _HYBRID_SCORE_MAX = (2.0 / 61.0) + max(_ROLE_SCORE_WEIGHTS.values()) + _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS
166
+ # The import factor is ≤ 1, so we use the raw max (derived via _rrf_max(2)).
167
+ _HYBRID_SCORE_MAX = _rrf_max(2) + max(_ROLE_SCORE_WEIGHTS.values()) + _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS
94
168
 
95
169
 
96
170
  def _query_tokens(query: str) -> set[str]:
@@ -374,6 +448,9 @@ def explain_score_components(
374
448
  rrf = comps.get("rrf_raw") or comps.get("hybrid_rrf")
375
449
  if rrf is not None:
376
450
  parts.append(f"rrf={float(rrf):.3f}")
451
+ bm25 = comps.get("bm25")
452
+ if bm25:
453
+ parts.append(f"bm25={float(bm25):.3f}")
377
454
  else:
378
455
  d = comps.get("distance")
379
456
  if d is not None:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: java-codebase-rag
3
- Version: 0.10.2
3
+ Version: 0.11.0
4
4
  Summary: MCP server for semantic + structural search over Java codebases
5
5
  Author: HumanBean17
6
6
  License-Expression: MIT
@@ -2,7 +2,7 @@ java_codebase_rag/__init__.py,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs
2
2
  java_codebase_rag/_fdlimit.py,sha256=vkwjsPbZfxzZ2DZTPWO5DxtuNlLzOADzIq07iYX7GCU,2465
3
3
  java_codebase_rag/_stdio.py,sha256=TDNbpt2EP0_Zd622ihdlKwlMfxkKHWOfgLVcU6TcNbo,1458
4
4
  java_codebase_rag/_version.py,sha256=Dnoh_a-c13fqPQ-HV9CsSPbFUkihUJl_z5CACW4qDyA,1365
5
- java_codebase_rag/cli.py,sha256=5gDLwKM7EPjdoTQMB0-WM2b2FVWIyGbtRJxyhIjhzq8,46404
5
+ java_codebase_rag/cli.py,sha256=Sf5HWQ607cg8LoZ-xlNpKKK-RslOjj5DDDsZDpHnUj8,47721
6
6
  java_codebase_rag/cli_format.py,sha256=CT7-xdwZ0bMCdP68_UOwkvm-mnLluU3LutlM-mDNk60,1839
7
7
  java_codebase_rag/cli_progress.py,sha256=q6Wh97yzLGs1B8UFk_WAKivfQu7Y5RnUUE-T2YHWkIs,3237
8
8
  java_codebase_rag/config.py,sha256=4rlRQUj0gFM-StUvhMD61uB2hZd2zqVHeNNBRXJMfN4,33283
@@ -26,6 +26,10 @@ java_codebase_rag/ast/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3h
26
26
  java_codebase_rag/ast/ast_java.py,sha256=ikMDBZGMYe0E2Doe6nf-Zw3k4MnwoNYvLqg3xsnfxmI,99614
27
27
  java_codebase_rag/ast/brownfield_events.py,sha256=yxXkKDgMb3VPtaiakGzncHM_EGnda8xIue6w90yYp8s,2055
28
28
  java_codebase_rag/ast/chunk_heuristics.py,sha256=aQk2NOKxzUdqoUAJUO3G3LE0MN_bYZWNLQ0tkmj5uts,1813
29
+ java_codebase_rag/eval/__init__.py,sha256=V49SsVzkfHmVhRGMl9KbzMlxDSI8aHZA1YmxzgzNtYo,30
30
+ java_codebase_rag/eval/ground_truth.py,sha256=q0gzsCF6hqR-NcMRNDDMUlvOA13RCcGcVLWWqlA4syY,3246
31
+ java_codebase_rag/eval/metrics.py,sha256=nEDbN3VjYLUwyAvNcd8w4Kmjm0PlgX_Mt6cfOmH7f9k,3118
32
+ java_codebase_rag/eval/runner.py,sha256=k_A5TLi16qeqT8WQ2lGqtioDXBlSzjsj0mLv-7Tml1o,20193
29
33
  java_codebase_rag/graph/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
30
34
  java_codebase_rag/graph/build_ast_graph.py,sha256=-MitMiz23HjiRHTt9MDA39Q9gLTCE07QoFxGa2Xl4uI,181704
31
35
  java_codebase_rag/graph/graph_enrich.py,sha256=51Q8aE0Mw09SVYBESd3H57IftsOpfkwgJV2ibt0RzwE,72598
@@ -47,9 +51,9 @@ java_codebase_rag/mcp/mcp_v2.py,sha256=w9kqz0AcWu9SqC1GuXZbLjRe0j9XhM3uONI4RB3ro
47
51
  java_codebase_rag/mcp/server.py,sha256=DJGlx8U5SD4Pe6Ro71OeFHi6HsYZ4r4uINn7R7uSKk4,40935
48
52
  java_codebase_rag/search/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
49
53
  java_codebase_rag/search/index_common.py,sha256=HT6FKHFJ084eFvd3fR1j8z8gf4eWoPHVW8GXLpw464I,285
50
- java_codebase_rag/search/search_lancedb.py,sha256=wag0xrtmZZxdjVCMmINm3faSf0FCc9AHwEZk2kFZE24,40855
51
- java_codebase_rag/search/search_lexical.py,sha256=DsOYJiLpR9YhzCoW_IPRKIa7FWi_gBVvcUoRHImkDoI,19553
52
- java_codebase_rag/search/search_scoring.py,sha256=-qKU4LWf-meobSdhaKBNdukEeQsDAwPELXfr3cvwtkM,17528
54
+ java_codebase_rag/search/search_lancedb.py,sha256=MxfTivzeMbScmYIf9s8nKSGCmNr5gMGDXldZotAfW-8,49368
55
+ java_codebase_rag/search/search_lexical.py,sha256=iVui6KneEiVA1TRkA5UhO8XI9toayMDOCU-tXYjwRe8,19961
56
+ java_codebase_rag/search/search_scoring.py,sha256=OsaX6zWyl4BIcqB7nRJRjp3SBEatmGAQ5EZEtQDaww0,20804
53
57
  java_codebase_rag/watch/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
54
58
  java_codebase_rag/watch/client.py,sha256=PO1uyllGZWoYgYYgm6vToCvaILkp0coXfXCEitzWXMU,9258
55
59
  java_codebase_rag/watch/daemon.py,sha256=ZgQrhUkfAJaExCaMqP8QfnNWxLQaoJyxes-FUo-qr78,15698
@@ -59,9 +63,9 @@ java_codebase_rag/watch/protocol.py,sha256=Vj16d0d0eA7XiLFX6o7JMY-I-mGva9VRxC41Z
59
63
  java_codebase_rag/watch/server.py,sha256=GUyCDmbl30wm9GhWTBh-K_BEhFSCHiM2kahmncMGf5E,11168
60
64
  java_codebase_rag/watch/warm.py,sha256=EN_39jVV1Kohsq8OI2LqGdTz_X-OCy4s4GWykiDAK7k,5113
61
65
  java_codebase_rag/watch/watcher.py,sha256=flD17vjPYuUjae6eWzSYcN-lsZnJUWVoXXvywDD1J24,15272
62
- java_codebase_rag-0.10.2.dist-info/licenses/LICENSE,sha256=gxvtiHtuviR_q8ZAjWw-QTcF3DyPzg6ZY-lQrr8OPpw,1068
63
- java_codebase_rag-0.10.2.dist-info/METADATA,sha256=GA0Kkn7JPyGGP4gn_aw6V4gOHBs2MsK25qh9UokFxWA,20573
64
- java_codebase_rag-0.10.2.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
65
- java_codebase_rag-0.10.2.dist-info/entry_points.txt,sha256=pa6yYHbHG7ZZ7G9ZJgYK1Fz6eR9okldPLo7qLY2THuo,190
66
- java_codebase_rag-0.10.2.dist-info/top_level.txt,sha256=u2i_IKLLOkYyNE32m4TIED0vtKtQDdPEVqRtNuhC1Hk,18
67
- java_codebase_rag-0.10.2.dist-info/RECORD,,
66
+ java_codebase_rag-0.11.0.dist-info/licenses/LICENSE,sha256=gxvtiHtuviR_q8ZAjWw-QTcF3DyPzg6ZY-lQrr8OPpw,1068
67
+ java_codebase_rag-0.11.0.dist-info/METADATA,sha256=mVrkmI3Fdj9YcFkj-vSJmoP-LrWgLDjQN4kmY-cmUEE,20573
68
+ java_codebase_rag-0.11.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
69
+ java_codebase_rag-0.11.0.dist-info/entry_points.txt,sha256=pa6yYHbHG7ZZ7G9ZJgYK1Fz6eR9okldPLo7qLY2THuo,190
70
+ java_codebase_rag-0.11.0.dist-info/top_level.txt,sha256=u2i_IKLLOkYyNE32m4TIED0vtKtQDdPEVqRtNuhC1Hk,18
71
+ java_codebase_rag-0.11.0.dist-info/RECORD,,