java-codebase-rag 0.10.2__py3-none-any.whl → 0.11.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- java_codebase_rag/cli.py +29 -2
- java_codebase_rag/eval/__init__.py +1 -0
- java_codebase_rag/eval/ground_truth.py +100 -0
- java_codebase_rag/eval/metrics.py +107 -0
- java_codebase_rag/eval/runner.py +556 -0
- java_codebase_rag/search/search_lancedb.py +252 -57
- java_codebase_rag/search/search_lexical.py +10 -0
- java_codebase_rag/search/search_scoring.py +79 -2
- {java_codebase_rag-0.10.2.dist-info → java_codebase_rag-0.11.0.dist-info}/METADATA +1 -1
- {java_codebase_rag-0.10.2.dist-info → java_codebase_rag-0.11.0.dist-info}/RECORD +14 -10
- {java_codebase_rag-0.10.2.dist-info → java_codebase_rag-0.11.0.dist-info}/WHEEL +0 -0
- {java_codebase_rag-0.10.2.dist-info → java_codebase_rag-0.11.0.dist-info}/entry_points.txt +0 -0
- {java_codebase_rag-0.10.2.dist-info → java_codebase_rag-0.11.0.dist-info}/licenses/LICENSE +0 -0
- {java_codebase_rag-0.10.2.dist-info → java_codebase_rag-0.11.0.dist-info}/top_level.txt +0 -0
|
@@ -20,6 +20,7 @@ from sentence_transformers import SentenceTransformer
|
|
|
20
20
|
|
|
21
21
|
from java_codebase_rag.ast.chunk_heuristics import analyze_chunk, looks_like_code_identifier
|
|
22
22
|
from java_codebase_rag.search.index_common import SBERT_MODEL
|
|
23
|
+
from java_codebase_rag.search import search_lexical
|
|
23
24
|
from java_codebase_rag.config import maybe_expand_embedding_model_path, resolved_sbert_model_for_process_env
|
|
24
25
|
|
|
25
26
|
# Scoring & dedup primitives live in `search_scoring` (dependency-free — no
|
|
@@ -27,7 +28,11 @@ from java_codebase_rag.config import maybe_expand_embedding_model_path, resolved
|
|
|
27
28
|
# graph-only (macOS Intel) installs where this module is unimportable. Re-exported
|
|
28
29
|
# here for backward compatibility (`from search_lancedb import _clamp01`, etc.).
|
|
29
30
|
from java_codebase_rag.search.search_scoring import ( # noqa: F401
|
|
31
|
+
BASELINE_2LIST_CONFIG,
|
|
32
|
+
DEFAULT_RANK_CONFIG,
|
|
30
33
|
DEDUP_OVERFETCH,
|
|
34
|
+
RankConfig,
|
|
35
|
+
build_fts_query,
|
|
31
36
|
_ACTION_VERB_BONUS,
|
|
32
37
|
_ACTION_VERB_PREFIXES,
|
|
33
38
|
_HYBRID_SCORE_MAX,
|
|
@@ -632,9 +637,150 @@ def _attach_neighbor_context(
|
|
|
632
637
|
_debug_ctx(f"attached context to {attached}/{len(java_rows)} java rows")
|
|
633
638
|
|
|
634
639
|
|
|
640
|
+
def _bm25_candidate_rows(
|
|
641
|
+
*,
|
|
642
|
+
g: object,
|
|
643
|
+
query: str,
|
|
644
|
+
uri: str,
|
|
645
|
+
db: object,
|
|
646
|
+
extra_predicates: list[str],
|
|
647
|
+
columns: set[str],
|
|
648
|
+
limit: int = 100,
|
|
649
|
+
) -> list[dict]:
|
|
650
|
+
"""Fetch BM25-ranked Symbol candidates from the FTS index and resolve them to
|
|
651
|
+
chunk rows in BM25 rank order. Returns ``[]`` on any failure (silent degradation).
|
|
652
|
+
|
|
653
|
+
Pipeline:
|
|
654
|
+
1. ``search_lexical.fetch_fts_candidates(g, query)`` → BM25-ranked Symbols +
|
|
655
|
+
a ``{symbol_node_id: bm25_score}`` map. ``None`` / empty → return ``[]``.
|
|
656
|
+
2. Map each Symbol fqn to its enclosing TYPE fqn (``primary_type_fqn`` has no
|
|
657
|
+
``#``; a member ``Type#method`` maps to ``Type``). Dedupe by type fqn,
|
|
658
|
+
keeping the MAX BM25 score among same-type symbols.
|
|
659
|
+
3. Order type fqns by BM25 desc (fqn asc tiebreak — deterministic).
|
|
660
|
+
4. Fetch chunk rows from LanceDB with a FILTER-ONLY query (no vector ranking,
|
|
661
|
+
so BM25 order is preserved). Predicates = caller's ``extra_predicates`` +
|
|
662
|
+
the ``primary_type_fqn IN (...)`` built from the ordered types — preserving
|
|
663
|
+
filter parity with the vector path.
|
|
664
|
+
5. Group fetched chunks by ``primary_type_fqn``; emit in BM25 rank order,
|
|
665
|
+
each chunk carrying ``_score_components["bm25"]``.
|
|
666
|
+
6. Apply ``_apply_chunk_hints`` + ``_refine_java_start_lines`` for consistency
|
|
667
|
+
with graph_rows handling.
|
|
668
|
+
|
|
669
|
+
Any exception (FTS or LanceDB) → ``_debug_ctx`` log + return ``[]`` (silent
|
|
670
|
+
degradation; the vector path is unaffected).
|
|
671
|
+
"""
|
|
672
|
+
# 1. BM25 candidate fetch via the FTS index.
|
|
673
|
+
# Pre-split the query with the same tokenizer the ``sym_fts`` index uses
|
|
674
|
+
# (``search_text`` stores ``_split_identifier`` tokens). LadybugDB FTS's own
|
|
675
|
+
# tokenizer does NOT split camelCase, so a raw ``DistributionChunkService``
|
|
676
|
+
# would match nothing — ``build_fts_query`` mirrors what the lexical backend
|
|
677
|
+
# does at search_lexical.py (run_lexical_search), keeping index/query token
|
|
678
|
+
# spaces aligned. An empty split (degenerate / stopword-only query) → no FTS
|
|
679
|
+
# candidates → degrade silently to the vector path.
|
|
680
|
+
fts_query = build_fts_query(query)
|
|
681
|
+
if not fts_query or not fts_query.strip():
|
|
682
|
+
return []
|
|
683
|
+
try:
|
|
684
|
+
fts = search_lexical.fetch_fts_candidates(g, fts_query, filter=None, path_contains=None)
|
|
685
|
+
except Exception as exc: # noqa: BLE001 — silent degradation
|
|
686
|
+
_debug_ctx(f"bm25 FTS fetch raised: {exc!r}")
|
|
687
|
+
return []
|
|
688
|
+
if not fts or not fts.get("rows"):
|
|
689
|
+
return []
|
|
690
|
+
sym_rows = fts["rows"]
|
|
691
|
+
scores = fts.get("scores") or {}
|
|
692
|
+
|
|
693
|
+
# 2. Map symbol fqns → enclosing type fqns; keep MAX bm25 per type.
|
|
694
|
+
type_fqn_to_bm25: dict[str, float] = {}
|
|
695
|
+
for r in sym_rows:
|
|
696
|
+
fqn = r.get("fqn")
|
|
697
|
+
if not fqn:
|
|
698
|
+
continue
|
|
699
|
+
type_fqn = search_lexical.enclosing_type_fqn(str(fqn))
|
|
700
|
+
if not type_fqn:
|
|
701
|
+
continue
|
|
702
|
+
score = float(scores.get(r.get("id"), 0.0))
|
|
703
|
+
prev = type_fqn_to_bm25.get(type_fqn)
|
|
704
|
+
if prev is None or score > prev:
|
|
705
|
+
type_fqn_to_bm25[type_fqn] = score
|
|
706
|
+
if not type_fqn_to_bm25:
|
|
707
|
+
return []
|
|
708
|
+
|
|
709
|
+
# 3. Deterministic ordering: BM25 desc, fqn asc.
|
|
710
|
+
ordered_types = sorted(
|
|
711
|
+
type_fqn_to_bm25.keys(),
|
|
712
|
+
key=lambda f: (-type_fqn_to_bm25[f], f),
|
|
713
|
+
)
|
|
714
|
+
|
|
715
|
+
# 4. Filter-only chunk fetch (NO vector ranking → BM25 order preserved). The
|
|
716
|
+
# ``primary_type_fqn IN (...)`` predicate must be buildable; if the index is so
|
|
717
|
+
# old that the column is absent, we can't restrict the fetch and degrade to [].
|
|
718
|
+
if "primary_type_fqn" not in columns:
|
|
719
|
+
_debug_ctx("bm25 fetch skipped: primary_type_fqn column absent from schema")
|
|
720
|
+
return []
|
|
721
|
+
preds = list(extra_predicates) + _build_extra_predicates(
|
|
722
|
+
columns=columns,
|
|
723
|
+
role=None, module=None, microservice=None,
|
|
724
|
+
package_prefix=None, fqn_in=ordered_types,
|
|
725
|
+
)
|
|
726
|
+
combined_pred = _combine_predicates(preds)
|
|
727
|
+
base_cols = ["filename", "text", "start", "end"]
|
|
728
|
+
for col in ("range_start", "range_end"):
|
|
729
|
+
if col in columns:
|
|
730
|
+
base_cols.append(col)
|
|
731
|
+
java_extra = [c for c in JAVA_ENRICHED_COLUMNS if c in columns]
|
|
732
|
+
select_cols = [*base_cols, "language", *java_extra]
|
|
733
|
+
|
|
734
|
+
try:
|
|
735
|
+
tbl = db.open_table(TABLES["java"])
|
|
736
|
+
# LanceDB 0.34 filter-only path: search() with no vector arg issues a
|
|
737
|
+
# non-vector scan; .where/.select/.limit/.to_list returns rows in table
|
|
738
|
+
# order without re-ranking by similarity. (tbl.query() is NOT available in
|
|
739
|
+
# 0.34; to_lance().scanner() requires pylance, which isn't installed on the
|
|
740
|
+
# PEP 508 graph-only profile — search() with no vector is the supported API.)
|
|
741
|
+
q = tbl.search().select(select_cols).limit(
|
|
742
|
+
max(limit, len(ordered_types) * 4)
|
|
743
|
+
)
|
|
744
|
+
if combined_pred:
|
|
745
|
+
q = q.where(combined_pred, prefilter=True)
|
|
746
|
+
with _silence_lance_autoproj_warnings():
|
|
747
|
+
fetched = q.to_list()
|
|
748
|
+
except Exception as exc: # noqa: BLE001 — silent degradation
|
|
749
|
+
_debug_ctx(f"bm25 chunk fetch failed: {exc!r}")
|
|
750
|
+
return []
|
|
751
|
+
|
|
752
|
+
# 5. Group by primary_type_fqn; emit in BM25 rank order.
|
|
753
|
+
by_type: dict[str, list[dict]] = {}
|
|
754
|
+
for ch in fetched:
|
|
755
|
+
tf = ch.get("primary_type_fqn")
|
|
756
|
+
if tf is None:
|
|
757
|
+
continue
|
|
758
|
+
by_type.setdefault(str(tf), []).append(ch)
|
|
759
|
+
|
|
760
|
+
out: list[dict] = []
|
|
761
|
+
for type_fqn in ordered_types:
|
|
762
|
+
chunks = by_type.get(type_fqn)
|
|
763
|
+
if not chunks:
|
|
764
|
+
continue # filtered out by extra_predicates / absent from index
|
|
765
|
+
bm25_val = round(float(type_fqn_to_bm25[type_fqn]), 4)
|
|
766
|
+
for ch in chunks:
|
|
767
|
+
ch["_kind"] = "java"
|
|
768
|
+
ch["_hybrid"] = False
|
|
769
|
+
ch.setdefault("_score_components", {})["bm25"] = bm25_val
|
|
770
|
+
ch["start"] = coerce_position_field(ch.get("start"))
|
|
771
|
+
ch["end"] = coerce_position_field(ch.get("end"))
|
|
772
|
+
out.append(ch)
|
|
773
|
+
|
|
774
|
+
# 6. Consistency with graph_rows handling.
|
|
775
|
+
_apply_chunk_hints(out)
|
|
776
|
+
_refine_java_start_lines(out)
|
|
777
|
+
return out
|
|
778
|
+
|
|
779
|
+
|
|
635
780
|
def _graph_expand_merge(
|
|
636
781
|
vector_rows: list[dict],
|
|
637
782
|
*,
|
|
783
|
+
query: str,
|
|
638
784
|
query_vec: np.ndarray,
|
|
639
785
|
db: object,
|
|
640
786
|
uri: str,
|
|
@@ -642,8 +788,24 @@ def _graph_expand_merge(
|
|
|
642
788
|
extra_predicates: list[str],
|
|
643
789
|
expand_depth: int,
|
|
644
790
|
ladybug_path: str | None,
|
|
791
|
+
rank_config: RankConfig = DEFAULT_RANK_CONFIG,
|
|
645
792
|
) -> list[dict]:
|
|
646
|
-
"""Expand vector top-k through the
|
|
793
|
+
"""Expand vector top-k through the graph and/or fuse BM25, then RRF-merge.
|
|
794
|
+
|
|
795
|
+
Which lists contribute is controlled by ``rank_config.lists``:
|
|
796
|
+
- ``"vector"`` — always present (the backbone; validated by RankConfig).
|
|
797
|
+
- ``"graph"`` — graph expand + fetch (skipped entirely when absent).
|
|
798
|
+
- ``"bm25"`` — LadybugDB FTS candidate fetch fused as a third list.
|
|
799
|
+
|
|
800
|
+
Silent degradation: any failure in the graph or BM25 path drops just that list;
|
|
801
|
+
the vector list is never lost. Returns ``vector_rows`` unchanged when no
|
|
802
|
+
auxiliary list yields rows.
|
|
803
|
+
"""
|
|
804
|
+
want_graph = "graph" in rank_config.lists
|
|
805
|
+
want_bm25 = "bm25" in rank_config.lists
|
|
806
|
+
if not want_graph and not want_bm25:
|
|
807
|
+
return vector_rows
|
|
808
|
+
|
|
647
809
|
# Lazy import so the module works without ladybug installed when graph_expand=False.
|
|
648
810
|
try:
|
|
649
811
|
from java_codebase_rag.graph.ladybug_queries import LadybugGraph
|
|
@@ -653,67 +815,97 @@ def _graph_expand_merge(
|
|
|
653
815
|
if not LadybugGraph.exists(ladybug_path):
|
|
654
816
|
return vector_rows
|
|
655
817
|
|
|
656
|
-
|
|
657
|
-
if not seed_fqns:
|
|
658
|
-
return vector_rows
|
|
818
|
+
java_cols = _table_columns(uri, TABLES["java"], db)
|
|
659
819
|
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
|
|
668
|
-
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
|
|
820
|
+
# --- graph list ---
|
|
821
|
+
graph_rows: list[dict] = []
|
|
822
|
+
expand_weight_by_fqn: dict[str, float] = {}
|
|
823
|
+
if want_graph:
|
|
824
|
+
seed_fqns = sorted({r.get("primary_type_fqn") for r in vector_rows if r.get("primary_type_fqn")})
|
|
825
|
+
neighbor_fqns: list[str] = []
|
|
826
|
+
if seed_fqns:
|
|
827
|
+
try:
|
|
828
|
+
graph_obj = LadybugGraph.get(ladybug_path)
|
|
829
|
+
structural = graph_obj.expand_fqns(seed_fqns, depth=expand_depth)
|
|
830
|
+
method_pairs = graph_obj.expand_methods(
|
|
831
|
+
seed_fqns, depth=expand_depth, exclude_external=True,
|
|
832
|
+
)
|
|
833
|
+
for f in structural:
|
|
834
|
+
if f:
|
|
835
|
+
expand_weight_by_fqn[f] = max(expand_weight_by_fqn.get(f, 0.0), 1.0)
|
|
836
|
+
for f, conf in method_pairs:
|
|
837
|
+
if f:
|
|
838
|
+
expand_weight_by_fqn[f] = max(expand_weight_by_fqn.get(f, 0.0), conf)
|
|
839
|
+
neighbor_fqns = list(dict.fromkeys(
|
|
840
|
+
list(structural) + [f for f, _ in method_pairs],
|
|
841
|
+
))
|
|
842
|
+
except Exception:
|
|
843
|
+
neighbor_fqns = []
|
|
844
|
+
|
|
845
|
+
novel = [fqn for fqn in neighbor_fqns if fqn and fqn not in set(seed_fqns)]
|
|
846
|
+
if novel:
|
|
847
|
+
extra = list(extra_predicates)
|
|
848
|
+
extra.extend(_build_extra_predicates(
|
|
849
|
+
columns=java_cols,
|
|
850
|
+
role=None, module=None, microservice=None,
|
|
851
|
+
package_prefix=None, fqn_in=novel,
|
|
852
|
+
))
|
|
853
|
+
try:
|
|
854
|
+
graph_rows = _search_one_table(
|
|
855
|
+
TABLES["java"],
|
|
856
|
+
uri=uri, db=db, query_vec=query_vec,
|
|
857
|
+
limit=max(limit, 20),
|
|
858
|
+
path_predicate=None, kind="java",
|
|
859
|
+
hybrid=False, fts_text=None,
|
|
860
|
+
extra_predicates=extra,
|
|
861
|
+
)
|
|
862
|
+
except Exception:
|
|
863
|
+
graph_rows = []
|
|
864
|
+
_apply_chunk_hints(graph_rows)
|
|
865
|
+
_refine_java_start_lines(graph_rows)
|
|
866
|
+
graph_rows.sort(key=_vector_sort_key)
|
|
867
|
+
for r in graph_rows:
|
|
868
|
+
r["_graph_expanded"] = True
|
|
869
|
+
r["_graph_expand_weight"] = expand_weight_by_fqn.get(
|
|
870
|
+
r.get("primary_type_fqn"), 1.0,
|
|
871
|
+
)
|
|
872
|
+
|
|
873
|
+
# --- bm25 list ---
|
|
874
|
+
bm25_rows: list[dict] = []
|
|
875
|
+
if want_bm25:
|
|
876
|
+
try:
|
|
877
|
+
graph_obj = LadybugGraph.get(ladybug_path)
|
|
878
|
+
except Exception:
|
|
879
|
+
graph_obj = None
|
|
880
|
+
if graph_obj is not None:
|
|
881
|
+
bm25_rows = _bm25_candidate_rows(
|
|
882
|
+
g=graph_obj,
|
|
883
|
+
query=query,
|
|
884
|
+
uri=uri,
|
|
885
|
+
db=db,
|
|
886
|
+
extra_predicates=extra_predicates,
|
|
887
|
+
columns=java_cols,
|
|
888
|
+
limit=limit,
|
|
889
|
+
)
|
|
678
890
|
|
|
679
|
-
|
|
680
|
-
|
|
891
|
+
# --- RRF fusion (only lists that yielded rows beyond vector) ---
|
|
892
|
+
lists: list[list[dict]] = [vector_rows]
|
|
893
|
+
row_weights: list[Callable[[dict], float] | None] = [None]
|
|
894
|
+
if want_graph and graph_rows:
|
|
895
|
+
lists.append(graph_rows)
|
|
896
|
+
row_weights.append(lambda row: float(row.get("_graph_expand_weight", 1.0)))
|
|
897
|
+
if want_bm25 and bm25_rows:
|
|
898
|
+
lists.append(bm25_rows)
|
|
899
|
+
row_weights.append(None)
|
|
900
|
+
|
|
901
|
+
if len(lists) == 1:
|
|
681
902
|
return vector_rows
|
|
682
903
|
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
|
|
686
|
-
|
|
687
|
-
package_prefix=None, fqn_in=novel,
|
|
688
|
-
))
|
|
689
|
-
|
|
690
|
-
try:
|
|
691
|
-
graph_rows = _search_one_table(
|
|
692
|
-
TABLES["java"],
|
|
693
|
-
uri=uri, db=db, query_vec=query_vec,
|
|
694
|
-
limit=max(limit, 20),
|
|
695
|
-
path_predicate=None, kind="java",
|
|
696
|
-
hybrid=False, fts_text=None,
|
|
697
|
-
extra_predicates=extra,
|
|
698
|
-
)
|
|
699
|
-
except Exception:
|
|
700
|
-
return vector_rows
|
|
701
|
-
_apply_chunk_hints(graph_rows)
|
|
702
|
-
_refine_java_start_lines(graph_rows)
|
|
703
|
-
graph_rows.sort(key=_vector_sort_key)
|
|
704
|
-
for r in graph_rows:
|
|
705
|
-
r["_graph_expanded"] = True
|
|
706
|
-
r["_graph_expand_weight"] = expand_weight_by_fqn.get(
|
|
707
|
-
r.get("primary_type_fqn"), 1.0,
|
|
708
|
-
)
|
|
709
|
-
fused = _rrf_merge(
|
|
710
|
-
[vector_rows, graph_rows],
|
|
711
|
-
row_weight_for_list_index=[
|
|
712
|
-
None,
|
|
713
|
-
lambda row: float(row.get("_graph_expand_weight", 1.0)),
|
|
714
|
-
],
|
|
904
|
+
return _rrf_merge(
|
|
905
|
+
lists,
|
|
906
|
+
k=rank_config.rrf_k,
|
|
907
|
+
row_weight_for_list_index=row_weights,
|
|
715
908
|
)
|
|
716
|
-
return fused
|
|
717
909
|
|
|
718
910
|
|
|
719
911
|
def _rrf_merge(
|
|
@@ -790,6 +982,7 @@ def run_search(
|
|
|
790
982
|
generated_only: bool = False,
|
|
791
983
|
exclude_generated: bool = False,
|
|
792
984
|
dedup_by_fqn: bool = False,
|
|
985
|
+
rank_config: RankConfig = DEFAULT_RANK_CONFIG,
|
|
793
986
|
) -> list[dict]:
|
|
794
987
|
effective_hybrid = hybrid
|
|
795
988
|
effective_fts = fts_text
|
|
@@ -887,6 +1080,7 @@ def run_search(
|
|
|
887
1080
|
if graph_expand and key == "java" and expand_depth > 0:
|
|
888
1081
|
rows = _graph_expand_merge(
|
|
889
1082
|
rows,
|
|
1083
|
+
query=query,
|
|
890
1084
|
query_vec=query_vec,
|
|
891
1085
|
db=db,
|
|
892
1086
|
uri=uri,
|
|
@@ -894,6 +1088,7 @@ def run_search(
|
|
|
894
1088
|
extra_predicates=extra_java,
|
|
895
1089
|
expand_depth=expand_depth,
|
|
896
1090
|
ladybug_path=ladybug_path,
|
|
1091
|
+
rank_config=rank_config,
|
|
897
1092
|
)
|
|
898
1093
|
|
|
899
1094
|
# Dedup by primary_type_fqn after all sorting/merging, before windowing
|
|
@@ -127,6 +127,11 @@ def _enclosing_type_fqn(fqn: str) -> str:
|
|
|
127
127
|
return fqn.split("#", 1)[0] if fqn else fqn
|
|
128
128
|
|
|
129
129
|
|
|
130
|
+
# Non-underscore aliases for cross-module callers (search_lancedb's BM25 fusion).
|
|
131
|
+
# Behavior is identical; the leading-underscore originals stay module-private.
|
|
132
|
+
enclosing_type_fqn = _enclosing_type_fqn
|
|
133
|
+
|
|
134
|
+
|
|
130
135
|
def _resolve_source_root(graph: LadybugGraph) -> str:
|
|
131
136
|
"""Authoritative source root is the one cached on the graph at index time."""
|
|
132
137
|
try:
|
|
@@ -255,6 +260,11 @@ def _try_fts_candidates(
|
|
|
255
260
|
return {"rows": rows, "scores": scores}
|
|
256
261
|
|
|
257
262
|
|
|
263
|
+
# Non-underscore alias for cross-module callers (search_lancedb's BM25 fusion on the
|
|
264
|
+
# vector path). Behavior is identical to the leading-underscore original.
|
|
265
|
+
fetch_fts_candidates = _try_fts_candidates
|
|
266
|
+
|
|
267
|
+
|
|
258
268
|
def run_lexical_search(
|
|
259
269
|
query: str,
|
|
260
270
|
*,
|
|
@@ -13,6 +13,7 @@ from __future__ import annotations
|
|
|
13
13
|
|
|
14
14
|
import json
|
|
15
15
|
import re
|
|
16
|
+
from dataclasses import dataclass
|
|
16
17
|
|
|
17
18
|
# Name of the LadybugDB FTS (Okapi BM25) index over Symbol.search_text (fork A).
|
|
18
19
|
# Shared by the build path (build_ast_graph._ensure_symbol_fts_index) and the
|
|
@@ -84,13 +85,86 @@ _ROLE_SCORE_WEIGHTS: dict[str, float] = {
|
|
|
84
85
|
"DTO": -0.08,
|
|
85
86
|
}
|
|
86
87
|
|
|
88
|
+
|
|
89
|
+
def _rrf_max(num_lists: int, k: int = 60) -> float:
|
|
90
|
+
"""Return the theoretical maximum RRF score for N-list fusion.
|
|
91
|
+
|
|
92
|
+
Reciprocal Rank Fusion (RRF) bounds each contribution to ≤ 1/(rank + k).
|
|
93
|
+
For N fused lists, the maximum possible sum is N/(k + 1) (achieved when
|
|
94
|
+
an item ranks #1 across all lists).
|
|
95
|
+
|
|
96
|
+
Args:
|
|
97
|
+
num_lists: Number of ranked lists being fused (e.g., 2 for vector+lexical).
|
|
98
|
+
k: The RRF constant (default 60 per the original paper).
|
|
99
|
+
|
|
100
|
+
Returns:
|
|
101
|
+
The maximum RRF contribution: num_lists / (k + 1).
|
|
102
|
+
"""
|
|
103
|
+
return num_lists / (k + 1)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
# Allowed list names in a RankConfig: vector is always required (it is the
|
|
107
|
+
# backbone retrieval signal); graph and bm25 are optional fusion participants.
|
|
108
|
+
# The "bm25" list is wired in ``search_lancedb._graph_expand_merge`` (LadybugDB
|
|
109
|
+
# FTS candidate fetch), which fuses BM25-ranked Symbol candidates as a third
|
|
110
|
+
# RRF list alongside vector and graph.
|
|
111
|
+
_RANK_LIST_NAMES: frozenset[str] = frozenset({"vector", "graph", "bm25"})
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
@dataclass(frozen=True)
|
|
115
|
+
class RankConfig:
|
|
116
|
+
"""Which ranked lists to fuse and the RRF constant to fuse them with.
|
|
117
|
+
|
|
118
|
+
This is a dep-free value object (no lancedb/torch) so it can be constructed
|
|
119
|
+
on every install flavor, including graph-only (macOS Intel). It is plumbed
|
|
120
|
+
through ``run_search`` → ``_graph_expand_merge`` to control (a) which lists
|
|
121
|
+
contribute to the final RRF fusion and (b) the ``k`` constant passed into
|
|
122
|
+
``_rrf_merge``.
|
|
123
|
+
|
|
124
|
+
Attributes:
|
|
125
|
+
lists: Subset of ``{"vector", "graph", "bm25"}``. Must contain
|
|
126
|
+
``"vector"`` (the backbone retrieval signal) and be non-empty.
|
|
127
|
+
rrf_k: The RRF constant (default 60 per the original paper). Must be ≥ 1.
|
|
128
|
+
"""
|
|
129
|
+
|
|
130
|
+
lists: frozenset[str]
|
|
131
|
+
rrf_k: int = 60
|
|
132
|
+
|
|
133
|
+
def __post_init__(self) -> None:
|
|
134
|
+
if not isinstance(self.lists, frozenset) or not self.lists:
|
|
135
|
+
raise ValueError("RankConfig.lists must be a non-empty frozenset")
|
|
136
|
+
if "vector" not in self.lists:
|
|
137
|
+
raise ValueError(
|
|
138
|
+
"RankConfig.lists must contain 'vector' (the backbone signal)"
|
|
139
|
+
)
|
|
140
|
+
unknown = self.lists - _RANK_LIST_NAMES
|
|
141
|
+
if unknown:
|
|
142
|
+
raise ValueError(
|
|
143
|
+
f"RankConfig.lists has unknown names {sorted(unknown)!r}; "
|
|
144
|
+
f"allowed: {sorted(_RANK_LIST_NAMES)!r}"
|
|
145
|
+
)
|
|
146
|
+
if not isinstance(self.rrf_k, int) or self.rrf_k < 1:
|
|
147
|
+
raise ValueError(f"RankConfig.rrf_k must be an int >= 1, got {self.rrf_k!r}")
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
# Production default: ship the 3-list config (vector+graph+bm25). The BM25 list
|
|
151
|
+
# is wired in ``search_lancedb._graph_expand_merge``; on installs without the
|
|
152
|
+
# vector stack, or when the FTS index is unavailable, it degrades silently to
|
|
153
|
+
# the 2-list (vector+graph) fusion.
|
|
154
|
+
DEFAULT_RANK_CONFIG = RankConfig(lists=frozenset({"vector", "graph", "bm25"}), rrf_k=60)
|
|
155
|
+
|
|
156
|
+
# Eval convenience: the historical 2-list (vector+graph) fusion, used by
|
|
157
|
+
# evaluation harnesses that isolate the vector+graph baseline from the bm25 list.
|
|
158
|
+
BASELINE_2LIST_CONFIG = RankConfig(lists=frozenset({"vector", "graph"}), rrf_k=60)
|
|
159
|
+
|
|
160
|
+
|
|
87
161
|
# Theoretical maximum for hybrid composite score (used for display normalization).
|
|
88
162
|
# Hybrid sort metric: raw_rrf * (import_factor if import_heavy else 1)
|
|
89
163
|
# + role_weight + symbol_bonus
|
|
90
164
|
# where raw_rrf ≤ 2/(k+1) for 2-list RRF, role_weight ≤ max(_ROLE_SCORE_WEIGHTS),
|
|
91
165
|
# and symbol_bonus ≤ _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS.
|
|
92
|
-
# The import factor is ≤ 1, so we use the raw max (2
|
|
93
|
-
_HYBRID_SCORE_MAX = (2
|
|
166
|
+
# The import factor is ≤ 1, so we use the raw max (derived via _rrf_max(2)).
|
|
167
|
+
_HYBRID_SCORE_MAX = _rrf_max(2) + max(_ROLE_SCORE_WEIGHTS.values()) + _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS
|
|
94
168
|
|
|
95
169
|
|
|
96
170
|
def _query_tokens(query: str) -> set[str]:
|
|
@@ -374,6 +448,9 @@ def explain_score_components(
|
|
|
374
448
|
rrf = comps.get("rrf_raw") or comps.get("hybrid_rrf")
|
|
375
449
|
if rrf is not None:
|
|
376
450
|
parts.append(f"rrf={float(rrf):.3f}")
|
|
451
|
+
bm25 = comps.get("bm25")
|
|
452
|
+
if bm25:
|
|
453
|
+
parts.append(f"bm25={float(bm25):.3f}")
|
|
377
454
|
else:
|
|
378
455
|
d = comps.get("distance")
|
|
379
456
|
if d is not None:
|
|
@@ -2,7 +2,7 @@ java_codebase_rag/__init__.py,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs
|
|
|
2
2
|
java_codebase_rag/_fdlimit.py,sha256=vkwjsPbZfxzZ2DZTPWO5DxtuNlLzOADzIq07iYX7GCU,2465
|
|
3
3
|
java_codebase_rag/_stdio.py,sha256=TDNbpt2EP0_Zd622ihdlKwlMfxkKHWOfgLVcU6TcNbo,1458
|
|
4
4
|
java_codebase_rag/_version.py,sha256=Dnoh_a-c13fqPQ-HV9CsSPbFUkihUJl_z5CACW4qDyA,1365
|
|
5
|
-
java_codebase_rag/cli.py,sha256=
|
|
5
|
+
java_codebase_rag/cli.py,sha256=Sf5HWQ607cg8LoZ-xlNpKKK-RslOjj5DDDsZDpHnUj8,47721
|
|
6
6
|
java_codebase_rag/cli_format.py,sha256=CT7-xdwZ0bMCdP68_UOwkvm-mnLluU3LutlM-mDNk60,1839
|
|
7
7
|
java_codebase_rag/cli_progress.py,sha256=q6Wh97yzLGs1B8UFk_WAKivfQu7Y5RnUUE-T2YHWkIs,3237
|
|
8
8
|
java_codebase_rag/config.py,sha256=4rlRQUj0gFM-StUvhMD61uB2hZd2zqVHeNNBRXJMfN4,33283
|
|
@@ -26,6 +26,10 @@ java_codebase_rag/ast/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3h
|
|
|
26
26
|
java_codebase_rag/ast/ast_java.py,sha256=ikMDBZGMYe0E2Doe6nf-Zw3k4MnwoNYvLqg3xsnfxmI,99614
|
|
27
27
|
java_codebase_rag/ast/brownfield_events.py,sha256=yxXkKDgMb3VPtaiakGzncHM_EGnda8xIue6w90yYp8s,2055
|
|
28
28
|
java_codebase_rag/ast/chunk_heuristics.py,sha256=aQk2NOKxzUdqoUAJUO3G3LE0MN_bYZWNLQ0tkmj5uts,1813
|
|
29
|
+
java_codebase_rag/eval/__init__.py,sha256=V49SsVzkfHmVhRGMl9KbzMlxDSI8aHZA1YmxzgzNtYo,30
|
|
30
|
+
java_codebase_rag/eval/ground_truth.py,sha256=q0gzsCF6hqR-NcMRNDDMUlvOA13RCcGcVLWWqlA4syY,3246
|
|
31
|
+
java_codebase_rag/eval/metrics.py,sha256=nEDbN3VjYLUwyAvNcd8w4Kmjm0PlgX_Mt6cfOmH7f9k,3118
|
|
32
|
+
java_codebase_rag/eval/runner.py,sha256=k_A5TLi16qeqT8WQ2lGqtioDXBlSzjsj0mLv-7Tml1o,20193
|
|
29
33
|
java_codebase_rag/graph/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
30
34
|
java_codebase_rag/graph/build_ast_graph.py,sha256=-MitMiz23HjiRHTt9MDA39Q9gLTCE07QoFxGa2Xl4uI,181704
|
|
31
35
|
java_codebase_rag/graph/graph_enrich.py,sha256=51Q8aE0Mw09SVYBESd3H57IftsOpfkwgJV2ibt0RzwE,72598
|
|
@@ -47,9 +51,9 @@ java_codebase_rag/mcp/mcp_v2.py,sha256=w9kqz0AcWu9SqC1GuXZbLjRe0j9XhM3uONI4RB3ro
|
|
|
47
51
|
java_codebase_rag/mcp/server.py,sha256=DJGlx8U5SD4Pe6Ro71OeFHi6HsYZ4r4uINn7R7uSKk4,40935
|
|
48
52
|
java_codebase_rag/search/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
49
53
|
java_codebase_rag/search/index_common.py,sha256=HT6FKHFJ084eFvd3fR1j8z8gf4eWoPHVW8GXLpw464I,285
|
|
50
|
-
java_codebase_rag/search/search_lancedb.py,sha256=
|
|
51
|
-
java_codebase_rag/search/search_lexical.py,sha256=
|
|
52
|
-
java_codebase_rag/search/search_scoring.py,sha256
|
|
54
|
+
java_codebase_rag/search/search_lancedb.py,sha256=MxfTivzeMbScmYIf9s8nKSGCmNr5gMGDXldZotAfW-8,49368
|
|
55
|
+
java_codebase_rag/search/search_lexical.py,sha256=iVui6KneEiVA1TRkA5UhO8XI9toayMDOCU-tXYjwRe8,19961
|
|
56
|
+
java_codebase_rag/search/search_scoring.py,sha256=OsaX6zWyl4BIcqB7nRJRjp3SBEatmGAQ5EZEtQDaww0,20804
|
|
53
57
|
java_codebase_rag/watch/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
54
58
|
java_codebase_rag/watch/client.py,sha256=PO1uyllGZWoYgYYgm6vToCvaILkp0coXfXCEitzWXMU,9258
|
|
55
59
|
java_codebase_rag/watch/daemon.py,sha256=ZgQrhUkfAJaExCaMqP8QfnNWxLQaoJyxes-FUo-qr78,15698
|
|
@@ -59,9 +63,9 @@ java_codebase_rag/watch/protocol.py,sha256=Vj16d0d0eA7XiLFX6o7JMY-I-mGva9VRxC41Z
|
|
|
59
63
|
java_codebase_rag/watch/server.py,sha256=GUyCDmbl30wm9GhWTBh-K_BEhFSCHiM2kahmncMGf5E,11168
|
|
60
64
|
java_codebase_rag/watch/warm.py,sha256=EN_39jVV1Kohsq8OI2LqGdTz_X-OCy4s4GWykiDAK7k,5113
|
|
61
65
|
java_codebase_rag/watch/watcher.py,sha256=flD17vjPYuUjae6eWzSYcN-lsZnJUWVoXXvywDD1J24,15272
|
|
62
|
-
java_codebase_rag-0.
|
|
63
|
-
java_codebase_rag-0.
|
|
64
|
-
java_codebase_rag-0.
|
|
65
|
-
java_codebase_rag-0.
|
|
66
|
-
java_codebase_rag-0.
|
|
67
|
-
java_codebase_rag-0.
|
|
66
|
+
java_codebase_rag-0.11.0.dist-info/licenses/LICENSE,sha256=gxvtiHtuviR_q8ZAjWw-QTcF3DyPzg6ZY-lQrr8OPpw,1068
|
|
67
|
+
java_codebase_rag-0.11.0.dist-info/METADATA,sha256=mVrkmI3Fdj9YcFkj-vSJmoP-LrWgLDjQN4kmY-cmUEE,20573
|
|
68
|
+
java_codebase_rag-0.11.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
|
|
69
|
+
java_codebase_rag-0.11.0.dist-info/entry_points.txt,sha256=pa6yYHbHG7ZZ7G9ZJgYK1Fz6eR9okldPLo7qLY2THuo,190
|
|
70
|
+
java_codebase_rag-0.11.0.dist-info/top_level.txt,sha256=u2i_IKLLOkYyNE32m4TIED0vtKtQDdPEVqRtNuhC1Hk,18
|
|
71
|
+
java_codebase_rag-0.11.0.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|