java-codebase-rag 0.9.7__py3-none-any.whl → 0.10.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- java_codebase_rag/analysis/pr_analysis.py +33 -3
- java_codebase_rag/ast/ast_java.py +2 -1
- java_codebase_rag/cli.py +23 -8
- java_codebase_rag/config.py +68 -1
- java_codebase_rag/graph/build_ast_graph.py +123 -4
- java_codebase_rag/graph/graph_types.py +109 -22
- java_codebase_rag/graph/ladybug_queries.py +45 -2
- java_codebase_rag/index/java_index_flow_lancedb.py +10 -16
- java_codebase_rag/install_data/agents/explorer-rag-cli.md +3 -1
- java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +3 -1
- java_codebase_rag/jrag.py +627 -661
- java_codebase_rag/jrag_render.py +160 -3
- java_codebase_rag/lance_optimize.py +11 -12
- java_codebase_rag/mcp/mcp_v2.py +2 -1
- java_codebase_rag/pipeline.py +47 -1
- java_codebase_rag/read_payloads.py +781 -0
- java_codebase_rag/search/search_lancedb.py +138 -6
- java_codebase_rag/search/search_lexical.py +128 -30
- java_codebase_rag/search/search_scoring.py +82 -0
- java_codebase_rag/watch/__init__.py +0 -0
- java_codebase_rag/watch/client.py +230 -0
- java_codebase_rag/watch/daemon.py +368 -0
- java_codebase_rag/watch/lock.py +201 -0
- java_codebase_rag/watch/paths.py +76 -0
- java_codebase_rag/watch/protocol.py +122 -0
- java_codebase_rag/watch/server.py +273 -0
- java_codebase_rag/watch/warm.py +105 -0
- java_codebase_rag/watch/watcher.py +352 -0
- {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.0.dist-info}/METADATA +30 -31
- {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.0.dist-info}/RECORD +34 -24
- {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.0.dist-info}/WHEEL +0 -0
- {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.0.dist-info}/entry_points.txt +0 -0
- {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.0.dist-info}/licenses/LICENSE +0 -0
- {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.0.dist-info}/top_level.txt +0 -0
|
@@ -7,8 +7,11 @@ import argparse
|
|
|
7
7
|
import json
|
|
8
8
|
import os
|
|
9
9
|
import sys
|
|
10
|
+
import tempfile
|
|
10
11
|
import threading
|
|
12
|
+
import warnings
|
|
11
13
|
from collections.abc import Callable
|
|
14
|
+
from contextlib import contextmanager
|
|
12
15
|
from pathlib import Path
|
|
13
16
|
|
|
14
17
|
import lancedb
|
|
@@ -44,8 +47,10 @@ from java_codebase_rag.search.search_scoring import ( # noqa: F401
|
|
|
44
47
|
_role_weight,
|
|
45
48
|
_split_identifier,
|
|
46
49
|
_symbol_bonus,
|
|
50
|
+
declaration_line_number,
|
|
47
51
|
explain_score_components,
|
|
48
52
|
l2_distance_to_score,
|
|
53
|
+
vector_display_score,
|
|
49
54
|
)
|
|
50
55
|
|
|
51
56
|
TABLES: dict[str, str] = {
|
|
@@ -296,6 +301,117 @@ def _combine_predicates(parts: list[str | None]) -> str | None:
|
|
|
296
301
|
return " AND ".join(f"({p})" for p in clean)
|
|
297
302
|
|
|
298
303
|
|
|
304
|
+
# LanceDB (0.30.x) emits two Rust `tracing` WARN lines per hybrid query to stderr
|
|
305
|
+
# — "specified output columns but did not include `_score`/`_distance` ... Call
|
|
306
|
+
# `disable_scoring_autoprojection`". They are noise on the agent's stderr, not
|
|
307
|
+
# Python warnings (so `warnings.filterwarnings` can't catch them), and the fluent
|
|
308
|
+
# query builder exposes no `disable_scoring_autoprojection()` (the lower-level
|
|
309
|
+
# `to_lance().scanner(...)` path needs `pylance`, which isn't installed on the
|
|
310
|
+
# PEP 508 graph-only profile). We match them by stable substring so anything that
|
|
311
|
+
# is a REAL error still reaches stderr.
|
|
312
|
+
_LANCE_AUTOPROJ_MARKERS: tuple[str, ...] = (
|
|
313
|
+
"disable_scoring_autoprojection",
|
|
314
|
+
"did not include `_distance`",
|
|
315
|
+
"did not include `_score`",
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
# The fd-2 redirect below mutates the PROCESS-GLOBAL fd 2 (and
|
|
319
|
+
# ``warnings.catch_warnings`` mutates global warning state). The MCP server
|
|
320
|
+
# dispatches every tool call through ``asyncio.to_thread`` on a thread pool
|
|
321
|
+
# (server.py), so two concurrent hybrid/auto-hybrid searches would race on the
|
|
322
|
+
# dup2 bookkeeping — corrupting the saved fd and crashing the whole server with
|
|
323
|
+
# ``Bad file descriptor``. Serialize the redirect so only one thread mutates fd
|
|
324
|
+
# 2 / warning state at a time. Concurrent hybrid queries therefore serialize
|
|
325
|
+
# their ``to_list()`` (correctness over throughput); a Rust-tracing-level
|
|
326
|
+
# suppression would remove the fd hijack entirely (follow-up).
|
|
327
|
+
_LANCE_WARN_REDIRECT_LOCK = threading.Lock()
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def _is_autoproj_noise(line: str) -> bool:
|
|
331
|
+
"""True for a LanceDB autoprojection-deprecation line (to drop).
|
|
332
|
+
|
|
333
|
+
Preserves genuine errors/tracebacks even if they happen to reference the API
|
|
334
|
+
name — only the bare deprecation log lines (no Error/Traceback/Exception) are
|
|
335
|
+
treated as noise.
|
|
336
|
+
"""
|
|
337
|
+
if not any(marker in line for marker in _LANCE_AUTOPROJ_MARKERS):
|
|
338
|
+
return False
|
|
339
|
+
return not any(seg in line for seg in ("Traceback", "Error:", "error:", "Exception"))
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
@contextmanager
|
|
343
|
+
def _silence_lance_autoproj_warnings():
|
|
344
|
+
"""Swallow LanceDB's `_score`/`_distance` autoprojection deprecation warnings.
|
|
345
|
+
|
|
346
|
+
Redirects fd 2 to a temp buffer for the duration of the wrapped call, drops
|
|
347
|
+
only the autoprojection deprecation lines, and re-emits everything else to
|
|
348
|
+
the real stderr so genuine errors stay visible. No-op if the caller opted
|
|
349
|
+
back in via ``JAVA_CODEBASE_RAG_KEEP_LANCE_WARNINGS`` (debugging).
|
|
350
|
+
|
|
351
|
+
Thread-safety: the redirect is serialized under ``_LANCE_WARN_REDIRECT_LOCK``
|
|
352
|
+
because it mutates process-global fd 2 and warning state — the MCP server
|
|
353
|
+
runs tool calls concurrently on a thread pool.
|
|
354
|
+
"""
|
|
355
|
+
if os.environ.get("JAVA_CODEBASE_RAG_KEEP_LANCE_WARNINGS"):
|
|
356
|
+
yield
|
|
357
|
+
return
|
|
358
|
+
# Also catch the (unlikely) Python-warning form defensively.
|
|
359
|
+
with _LANCE_WARN_REDIRECT_LOCK, warnings.catch_warnings():
|
|
360
|
+
warnings.filterwarnings(
|
|
361
|
+
"ignore",
|
|
362
|
+
message=r".*(disable_scoring_autoprojection|did not include `(_distance|_score)`).*",
|
|
363
|
+
)
|
|
364
|
+
with tempfile.TemporaryFile(mode="w+", encoding="utf-8", errors="replace") as captured:
|
|
365
|
+
saved = os.dup(2)
|
|
366
|
+
try:
|
|
367
|
+
os.dup2(captured.fileno(), 2)
|
|
368
|
+
yield
|
|
369
|
+
finally:
|
|
370
|
+
# Restore fd 2 FIRST so the re-emit below reaches real stderr.
|
|
371
|
+
os.dup2(saved, 2)
|
|
372
|
+
os.close(saved)
|
|
373
|
+
captured.seek(0)
|
|
374
|
+
kept = "".join(line for line in captured if not _is_autoproj_noise(line))
|
|
375
|
+
if kept:
|
|
376
|
+
sys.stderr.write(kept)
|
|
377
|
+
sys.stderr.flush()
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def _simple_type_name(fqn: str | None) -> str | None:
|
|
381
|
+
"""``com.foo.Bar`` -> ``Bar``; None/empty -> None."""
|
|
382
|
+
if not fqn:
|
|
383
|
+
return None
|
|
384
|
+
return str(fqn).rsplit(".", 1)[-1] or None
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
def _refine_java_start_lines(rows: list[dict]) -> None:
|
|
388
|
+
"""Point each java row's ``start.line`` at the type declaration, not the chunk anchor.
|
|
389
|
+
|
|
390
|
+
LanceDB chunks are anchored at the chunk's first source line — for a
|
|
391
|
+
file-spanning chunk that's the package/import line (``start.line`` = 1)
|
|
392
|
+
while the ``class``/``interface`` declaration sits several lines down. The
|
|
393
|
+
chunk anchor is a poor display line for a symbol hit (renders as
|
|
394
|
+
``File.java:1``); derive the real declaration line from the chunk text
|
|
395
|
+
(pinned to the primary type) so a hit shows ``File.java:<decl>`` instead
|
|
396
|
+
(F8). Method-only chunks whose range doesn't include a type declaration
|
|
397
|
+
keep their chunk anchor unchanged.
|
|
398
|
+
"""
|
|
399
|
+
for r in rows:
|
|
400
|
+
if str(r.get("_kind", "")) != "java":
|
|
401
|
+
continue
|
|
402
|
+
start = r.get("start")
|
|
403
|
+
if not isinstance(start, dict):
|
|
404
|
+
continue
|
|
405
|
+
anchor = start.get("line")
|
|
406
|
+
if anchor is None:
|
|
407
|
+
continue
|
|
408
|
+
hints = r.get("_hints") or {}
|
|
409
|
+
type_name = hints.get("primary_type_hint") or _simple_type_name(r.get("primary_type_fqn"))
|
|
410
|
+
decl = declaration_line_number(r.get("text"), int(anchor), type_name)
|
|
411
|
+
if decl is not None:
|
|
412
|
+
start["line"] = decl
|
|
413
|
+
|
|
414
|
+
|
|
299
415
|
def _search_one_table(
|
|
300
416
|
table_name: str,
|
|
301
417
|
*,
|
|
@@ -342,7 +458,11 @@ def _search_one_table(
|
|
|
342
458
|
)
|
|
343
459
|
if combined_pred:
|
|
344
460
|
q = q.where(combined_pred, prefilter=True)
|
|
345
|
-
|
|
461
|
+
# Hybrid selects explicit output columns without `_score`/`_distance`, so
|
|
462
|
+
# LanceDB (0.30.x) emits two Rust autoprojection deprecation WARNs to
|
|
463
|
+
# stderr per query. Silence just those lines; real errors still surface.
|
|
464
|
+
with _silence_lance_autoproj_warnings():
|
|
465
|
+
rows = q.to_list()
|
|
346
466
|
for r in rows:
|
|
347
467
|
r["_kind"] = kind
|
|
348
468
|
rs = r.pop("_relevance_score", None)
|
|
@@ -375,7 +495,11 @@ def _search_one_table(
|
|
|
375
495
|
# exposed score was always 0.0, making results look unranked.
|
|
376
496
|
d = r.get("_distance")
|
|
377
497
|
if d is not None:
|
|
378
|
-
|
|
498
|
+
# Use the same non-clamping map as the display sites so graph-expand
|
|
499
|
+
# rows (which run_search does NOT overwrite) never carry the old
|
|
500
|
+
# 1-d²/2 value that collapses to 0 past √2. (The main single/multi
|
|
501
|
+
# paths overwrite this with the bonus-adjusted effective distance.)
|
|
502
|
+
r["_score"] = vector_display_score(float(d))
|
|
379
503
|
r["start"] = coerce_position_field(r.get("start"))
|
|
380
504
|
r["end"] = coerce_position_field(r.get("end"))
|
|
381
505
|
return rows
|
|
@@ -575,6 +699,7 @@ def _graph_expand_merge(
|
|
|
575
699
|
except Exception:
|
|
576
700
|
return vector_rows
|
|
577
701
|
_apply_chunk_hints(graph_rows)
|
|
702
|
+
_refine_java_start_lines(graph_rows)
|
|
578
703
|
graph_rows.sort(key=_vector_sort_key)
|
|
579
704
|
for r in graph_rows:
|
|
580
705
|
r["_graph_expanded"] = True
|
|
@@ -737,6 +862,9 @@ def run_search(
|
|
|
737
862
|
extra_predicates=preds,
|
|
738
863
|
)
|
|
739
864
|
_apply_chunk_hints(rows)
|
|
865
|
+
# Anchor each java row's start.line on the type declaration instead of
|
|
866
|
+
# the chunk's first source line (often the package/import line = 1).
|
|
867
|
+
_refine_java_start_lines(rows)
|
|
740
868
|
if skip_role_weight:
|
|
741
869
|
for r in rows:
|
|
742
870
|
r["_skip_role_weight"] = True
|
|
@@ -747,11 +875,14 @@ def run_search(
|
|
|
747
875
|
_hybrid_post_sort_normalization(rows)
|
|
748
876
|
else:
|
|
749
877
|
rows.sort(key=_vector_sort_key)
|
|
750
|
-
# Vector:
|
|
878
|
+
# Vector: displayed score from the effective (bonus-adjusted) distance,
|
|
879
|
+
# normalized over the unit-embedding range so a correctly-ranked top
|
|
880
|
+
# hit never collapses to 0.000 (the cosine map 1 - d²/2 clamps to 0
|
|
881
|
+
# past √2; weak-but-best matches commonly sit at d ≈ 1.5).
|
|
751
882
|
for r in rows:
|
|
752
883
|
comps = r.setdefault("_score_components", {})
|
|
753
884
|
effective_dist = _effective_distance(comps)
|
|
754
|
-
r["_score"] =
|
|
885
|
+
r["_score"] = vector_display_score(effective_dist)
|
|
755
886
|
|
|
756
887
|
if graph_expand and key == "java" and expand_depth > 0:
|
|
757
888
|
rows = _graph_expand_merge(
|
|
@@ -792,16 +923,17 @@ def run_search(
|
|
|
792
923
|
)
|
|
793
924
|
)
|
|
794
925
|
_apply_chunk_hints(merged)
|
|
926
|
+
_refine_java_start_lines(merged)
|
|
795
927
|
if skip_role_weight:
|
|
796
928
|
for r in merged:
|
|
797
929
|
r["_skip_role_weight"] = True
|
|
798
930
|
_apply_symbol_bonus(merged, query_toks)
|
|
799
931
|
merged.sort(key=_vector_sort_key)
|
|
800
|
-
# Vector:
|
|
932
|
+
# Vector: displayed score from the effective (bonus-adjusted) distance.
|
|
801
933
|
for r in merged:
|
|
802
934
|
comps = r.setdefault("_score_components", {})
|
|
803
935
|
effective_dist = _effective_distance(comps)
|
|
804
|
-
r["_score"] =
|
|
936
|
+
r["_score"] = vector_display_score(effective_dist)
|
|
805
937
|
|
|
806
938
|
# Dedup by primary_type_fqn after all sorting/merging, before windowing
|
|
807
939
|
merged = _dedup_by_fqn(merged, dedup_by_fqn=dedup_by_fqn)
|
|
@@ -18,11 +18,13 @@ guarded by a parity unit test.
|
|
|
18
18
|
from __future__ import annotations
|
|
19
19
|
|
|
20
20
|
import os
|
|
21
|
+
import weakref
|
|
21
22
|
from pathlib import Path
|
|
22
23
|
from typing import TYPE_CHECKING, Any
|
|
23
24
|
|
|
24
25
|
from java_codebase_rag.graph.ladybug_queries import LadybugGraph
|
|
25
26
|
from java_codebase_rag.search.search_scoring import (
|
|
27
|
+
SYMBOL_FTS_INDEX,
|
|
26
28
|
_ROLE_SCORE_WEIGHTS,
|
|
27
29
|
_TYPE_MATCH_BONUS_CAP,
|
|
28
30
|
_TYPE_MATCH_BONUS_PER_HIT,
|
|
@@ -170,6 +172,88 @@ def _token_overlap(haystack_toks: set[str], needle_toks: set[str]) -> float:
|
|
|
170
172
|
return len(needle_toks & haystack_toks) / len(needle_toks)
|
|
171
173
|
|
|
172
174
|
|
|
175
|
+
# BM25 candidate fetch via the LadybugDB FTS index (fork A). DB-side indexed ranking
|
|
176
|
+
# replaces the heuristic's bounded Python scan; the heuristic below still scores the
|
|
177
|
+
# fetched candidates (name/type/fqn/role) and is the fallback when the FTS index or
|
|
178
|
+
# extension is unavailable (older graph, offline first run).
|
|
179
|
+
_FTS_CANDIDATE_K = 200 # top-K BM25 candidates; re-filtered by NodeFilter before ranking
|
|
180
|
+
# Connections that have run LOAD EXTENSION FTS. Keyed by the connection OBJECT (WeakSet),
|
|
181
|
+
# NOT id() — id() is reused after GC, which would let a fresh connection skip LOAD and then
|
|
182
|
+
# fail at QUERY_FTS_INDEX under test batching. Entries die with the connection.
|
|
183
|
+
_FTS_LOADED_CONNS: "weakref.WeakSet[object]" = weakref.WeakSet()
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _ensure_fts_loaded(g: LadybugGraph) -> bool:
|
|
187
|
+
"""LOAD EXTENSION FTS on the graph's (read-only) connection, once per connection.
|
|
188
|
+
|
|
189
|
+
Returns False if the extension can't be loaded (absent / offline) so the caller
|
|
190
|
+
falls back to the heuristic scan.
|
|
191
|
+
"""
|
|
192
|
+
conn = g._conn # noqa: SLF001
|
|
193
|
+
try:
|
|
194
|
+
if conn in _FTS_LOADED_CONNS:
|
|
195
|
+
return True
|
|
196
|
+
except Exception: # connection not weakref-able → LOAD every call (correct, slow)
|
|
197
|
+
pass
|
|
198
|
+
try:
|
|
199
|
+
g._rows("LOAD EXTENSION FTS") # noqa: SLF001
|
|
200
|
+
try:
|
|
201
|
+
_FTS_LOADED_CONNS.add(conn)
|
|
202
|
+
except Exception:
|
|
203
|
+
pass
|
|
204
|
+
return True
|
|
205
|
+
except Exception:
|
|
206
|
+
return False
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _try_fts_candidates(
|
|
210
|
+
g: LadybugGraph,
|
|
211
|
+
query: str,
|
|
212
|
+
filter: NodeFilter | None,
|
|
213
|
+
path_contains: str | None,
|
|
214
|
+
) -> dict | None:
|
|
215
|
+
"""Fetch BM25-ranked Symbol candidates via the FTS index; re-apply NodeFilter.
|
|
216
|
+
|
|
217
|
+
Returns ``{"rows": [...], "scores": {id: bm25}}`` (rows are the same shape the
|
|
218
|
+
heuristic scan yields), or ``None`` when FTS is unavailable (extension won't load,
|
|
219
|
+
or the index isn't present on this graph) so the caller falls back.
|
|
220
|
+
|
|
221
|
+
Two-step: (1) ``QUERY_FTS_INDEX`` returns the top-K node ids by Okapi BM25 over
|
|
222
|
+
``Symbol.search_text``; (2) re-MATCH those ids with the full ``_lexical_where``
|
|
223
|
+
predicates (role / module / path / kind≠file,package) so the filter logic stays
|
|
224
|
+
defined in one place. ``search_text`` is built at index time by ``build_ast_graph``
|
|
225
|
+
from the same ``_split_identifier`` the re-rank below uses, so index- and query-time
|
|
226
|
+
tokenization agree.
|
|
227
|
+
"""
|
|
228
|
+
if not _ensure_fts_loaded(g):
|
|
229
|
+
return None
|
|
230
|
+
idx_rows = g._rows("CALL SHOW_INDEXES() RETURN index_name") # noqa: SLF001
|
|
231
|
+
names = {row.get("index_name") for row in idx_rows}
|
|
232
|
+
if SYMBOL_FTS_INDEX not in names:
|
|
233
|
+
return None
|
|
234
|
+
fts = g._rows( # noqa: SLF001
|
|
235
|
+
f"CALL QUERY_FTS_INDEX('Symbol', '{SYMBOL_FTS_INDEX}', $q, top := $k) "
|
|
236
|
+
"RETURN node.id AS id, score",
|
|
237
|
+
{"q": query, "k": _FTS_CANDIDATE_K},
|
|
238
|
+
)
|
|
239
|
+
if not fts:
|
|
240
|
+
return {"rows": [], "scores": {}}
|
|
241
|
+
scores = {row["id"]: float(row.get("score") or 0.0) for row in fts}
|
|
242
|
+
ids = list(scores.keys())
|
|
243
|
+
|
|
244
|
+
# Re-MATCH the K ids with the SAME predicates the heuristic pushes down, so
|
|
245
|
+
# NodeFilter / path / structural-kind filtering is defined exactly once.
|
|
246
|
+
where, params = _lexical_where(filter, path_contains=path_contains)
|
|
247
|
+
struct_pred = "(s.kind <> 'file' AND s.kind <> 'package')"
|
|
248
|
+
if not where:
|
|
249
|
+
where = f"WHERE s.id IN $ids AND {struct_pred}"
|
|
250
|
+
else:
|
|
251
|
+
where = where.replace("WHERE ", f"WHERE s.id IN $ids AND {struct_pred} AND ", 1)
|
|
252
|
+
params["ids"] = ids
|
|
253
|
+
rows = g._rows(f"MATCH (s:Symbol) {where} RETURN {_SYMBOL_RETURN}", params) # noqa: SLF001
|
|
254
|
+
return {"rows": rows, "scores": scores}
|
|
255
|
+
|
|
256
|
+
|
|
173
257
|
def run_lexical_search(
|
|
174
258
|
query: str,
|
|
175
259
|
*,
|
|
@@ -185,6 +269,12 @@ def run_lexical_search(
|
|
|
185
269
|
) -> list[dict]:
|
|
186
270
|
"""Keyword search over Symbol nodes; returns ``run_search``-shaped row-dicts.
|
|
187
271
|
|
|
272
|
+
BM25-first (fork A): when the LadybugDB ``sym_fts`` index exists, candidates are
|
|
273
|
+
fetched DB-side via Okapi BM25 over ``Symbol.search_text`` (killing the bounded
|
|
274
|
+
Python scan that silently missed matches past the cap on large repos) and then
|
|
275
|
+
re-ranked here by the name/type/fqn/role heuristic. Falls back to that heuristic
|
|
276
|
+
scan when the FTS index or extension is unavailable (older graph, offline first run).
|
|
277
|
+
|
|
188
278
|
Raises ``RuntimeError`` (message contains "lexical search unavailable") if no
|
|
189
279
|
symbol graph exists — the caller maps that to a clean failure envelope. Returns
|
|
190
280
|
``[]`` for ``table in ("sql", "yaml")`` (those LanceDB tables aren't built in
|
|
@@ -201,33 +291,38 @@ def run_lexical_search(
|
|
|
201
291
|
)
|
|
202
292
|
g = graph or LadybugGraph.get()
|
|
203
293
|
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
294
|
+
# --- candidate fetch: BM25 (FTS) preferred, heuristic scan fallback ---
|
|
295
|
+
bm25_scores: dict[str, float] = {}
|
|
296
|
+
use_fts = False
|
|
297
|
+
fts = _try_fts_candidates(g, query, filter, path_contains)
|
|
298
|
+
if fts is not None:
|
|
299
|
+
rows = fts["rows"]
|
|
300
|
+
bm25_scores = fts["scores"]
|
|
301
|
+
use_fts = True
|
|
302
|
+
else:
|
|
303
|
+
where, params = _lexical_where(filter, path_contains=path_contains)
|
|
304
|
+
# Always exclude structural Symbol nodes. Files and packages are :Symbol-labeled
|
|
305
|
+
# (kind='file'/'package') but aren't searchable code declarations — without this
|
|
306
|
+
# a token that appears in a filename (e.g. 'distribution' in
|
|
307
|
+
# 'DistributionChunkService.java') would surface the file node as a hit.
|
|
308
|
+
struct_pred = "(s.kind <> 'file' AND s.kind <> 'package')"
|
|
309
|
+
where = f"WHERE {struct_pred}" if not where else where.replace("WHERE ", f"WHERE {struct_pred} AND ", 1)
|
|
310
|
+
# The heuristic scan returns rows in storage order — there is NO DB-side relevance
|
|
311
|
+
# ORDER BY without the FTS index — so fetch the FULL candidate pool up to the safety
|
|
312
|
+
# cap and rank here. A pagination-derived LIMIT (4x the page) is correct on the
|
|
313
|
+
# vector path where LanceDB returns rows pre-ranked, but on this unordered scan it
|
|
314
|
+
# would return only the first ~N symbols in arbitrary storage order and silently
|
|
315
|
+
# miss the best match on any non-trivial repo. The BM25 (FTS) path above has no cap.
|
|
316
|
+
params["lim"] = _CANDIDATE_LIMIT_CAP
|
|
317
|
+
cypher = f"MATCH (s:Symbol) {where} RETURN {_SYMBOL_RETURN} LIMIT $lim"
|
|
318
|
+
rows = g._rows(cypher, params) # noqa: SLF001 — de facto public read API (see find_v2)
|
|
319
|
+
# If the fetch hit the safety cap, deeper matches were never ranked. Surface it so
|
|
320
|
+
# a user on a large repo isn't silently shown an incomplete result set.
|
|
321
|
+
if advisories is not None and len(rows) >= _CANDIDATE_LIMIT_CAP:
|
|
322
|
+
advisories.append(
|
|
323
|
+
f"lexical search scanned the first {_CANDIDATE_LIMIT_CAP} matching symbols "
|
|
324
|
+
"(repo cap); deeper matches were not ranked — refine the query or add a filter"
|
|
325
|
+
)
|
|
231
326
|
|
|
232
327
|
query_toks = _query_tokens(query)
|
|
233
328
|
source_root = _resolve_source_root(g)
|
|
@@ -265,9 +360,10 @@ def run_lexical_search(
|
|
|
265
360
|
text_match = text_overlap * _TEXT_MATCH_WEIGHT
|
|
266
361
|
|
|
267
362
|
# A keyword search must require at least one lexical hit — role alone never
|
|
268
|
-
# qualifies a row (it only boosts/reorders matches).
|
|
269
|
-
#
|
|
270
|
-
|
|
363
|
+
# qualifies a row (it only boosts/reorders matches). On the BM25 path the FTS
|
|
364
|
+
# index already established textual relevance, so the qualifier is heuristic-only.
|
|
365
|
+
# Degenerate queries with no usable tokens fall through to role-ranked listing.
|
|
366
|
+
if query_toks and not use_fts and not (name_overlap or type_hits or fqn_match or text_overlap):
|
|
271
367
|
continue
|
|
272
368
|
|
|
273
369
|
role_w = 0.0 if role_locked else _ROLE_SCORE_WEIGHTS.get(role_raw.upper(), 0.0)
|
|
@@ -291,6 +387,8 @@ def run_lexical_search(
|
|
|
291
387
|
"lexical_relevance": round(raw, 4),
|
|
292
388
|
"role_weight": role_w,
|
|
293
389
|
}
|
|
390
|
+
if use_fts:
|
|
391
|
+
comps["bm25"] = round(float(bm25_scores.get(r.get("id"), 0.0)), 4)
|
|
294
392
|
|
|
295
393
|
sl, el, sb, eb = r.get("start_line"), r.get("end_line"), r.get("start_byte"), r.get("end_byte")
|
|
296
394
|
out.append(
|
|
@@ -12,6 +12,12 @@ Everything here is pure-Python dict/list math with no third-party deps.
|
|
|
12
12
|
from __future__ import annotations
|
|
13
13
|
|
|
14
14
|
import json
|
|
15
|
+
import re
|
|
16
|
+
|
|
17
|
+
# Name of the LadybugDB FTS (Okapi BM25) index over Symbol.search_text (fork A).
|
|
18
|
+
# Shared by the build path (build_ast_graph._ensure_symbol_fts_index) and the
|
|
19
|
+
# query path (search_lexical.run_lexical_search) so the two never drift.
|
|
20
|
+
SYMBOL_FTS_INDEX = "sym_fts"
|
|
15
21
|
|
|
16
22
|
# Over-fetch multiplier for dedup: fetch 4x to absorb per-FQN chunk multiplicity
|
|
17
23
|
# so that after collapsing by primary_type_fqn, a page stays full and the +1
|
|
@@ -210,6 +216,29 @@ def l2_distance_to_score(distance: float) -> float:
|
|
|
210
216
|
return 1.0 - distance * distance / 2.0
|
|
211
217
|
|
|
212
218
|
|
|
219
|
+
# Display-score denominator for the vector backend. Unit-normalized embeddings
|
|
220
|
+
# have L2 distance in [0, 2]; the cosine map ``l2_distance_to_score`` (1 - d²/2)
|
|
221
|
+
# goes NEGATIVE past √2 ≈ 1.414 and clamps to 0. Weak-but-best semantic matches
|
|
222
|
+
# (e.g. a lone keyword like "controller") commonly sit at d ≈ 1.5, so EVERY hit
|
|
223
|
+
# clamps to score=0.000 even though the ranking is correct. ``vector_display_score``
|
|
224
|
+
# instead normalizes the effective (bonus-adjusted) distance over the full
|
|
225
|
+
# unit-embedding range, so a top-ranked hit stays visibly non-zero. Role/symbol
|
|
226
|
+
# bonuses reduce the effective distance and so raise the displayed score,
|
|
227
|
+
# keeping it rank-monotonic with the distance-based sort key.
|
|
228
|
+
_VECTOR_DISTANCE_REF = 2.0
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def vector_display_score(effective_distance: float) -> float:
|
|
232
|
+
"""Displayed vector score in [0, 1] from the effective (bonus-adjusted) distance.
|
|
233
|
+
|
|
234
|
+
Bounded linear normalization over the unit-embedding L2 range [0, 2]: lower
|
|
235
|
+
distance → higher score. Unlike ``l2_distance_to_score`` (which goes
|
|
236
|
+
negative past √2 and clamps a correctly-ranked top hit to 0.000), this keeps
|
|
237
|
+
a top result visibly non-zero while staying rank-monotonic with the sort key.
|
|
238
|
+
"""
|
|
239
|
+
return _clamp01(1.0 - effective_distance / _VECTOR_DISTANCE_REF)
|
|
240
|
+
|
|
241
|
+
|
|
213
242
|
def _effective_distance(comps: dict[str, float]) -> float:
|
|
214
243
|
"""Compute the adjusted distance used for sorting.
|
|
215
244
|
|
|
@@ -231,6 +260,59 @@ def _clamp01(x: float) -> float:
|
|
|
231
260
|
return x
|
|
232
261
|
|
|
233
262
|
|
|
263
|
+
# Matches a Java top-level type declaration and captures its simple name. Mirrors
|
|
264
|
+
# the heuristic in ``ast.chunk_heuristics._JAVA_TYPE`` but is duplicated here so
|
|
265
|
+
# this module stays dependency-free (importable on graph-only Intel installs).
|
|
266
|
+
_JAVA_TYPE_DECL_RE = re.compile(
|
|
267
|
+
r"\b(?:public\s+|private\s+|protected\s+|sealed\s+|non-sealed\s+|final\s+|"
|
|
268
|
+
r"abstract\s+|static\s+)*"
|
|
269
|
+
r"(?:class|interface|enum|record)\s+([A-Za-z_][A-Za-z0-9_]*)"
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def declaration_line_number(
|
|
274
|
+
text: str | None, anchor_line: int | None, type_name: str | None = None
|
|
275
|
+
) -> int | None:
|
|
276
|
+
"""Absolute 1-based line of the Java type declaration within chunk ``text``.
|
|
277
|
+
|
|
278
|
+
LanceDB chunks are anchored at the chunk's first source line, which for a
|
|
279
|
+
file-spanning chunk is the package/import line (``anchor_line`` = 1) while
|
|
280
|
+
the ``class``/``interface`` declaration sits several lines down. Without
|
|
281
|
+
this, hits render as ``File.java:1`` even though the symbol is declared
|
|
282
|
+
later (F8). Returns ``anchor_line + i`` for the first matching declaration
|
|
283
|
+
(pinned to ``type_name`` when given, so a nested type doesn't win), or
|
|
284
|
+
``anchor_line`` unchanged when no declaration is found in the chunk.
|
|
285
|
+
|
|
286
|
+
Comment-aware: Javadoc/line/block-comment lines that merely MENTION the type
|
|
287
|
+
name (e.g. ``* This class Bar handles...``) are skipped so the returned line
|
|
288
|
+
is the real declaration, not a comment above it.
|
|
289
|
+
"""
|
|
290
|
+
if not text or anchor_line is None:
|
|
291
|
+
return anchor_line
|
|
292
|
+
in_block = False
|
|
293
|
+
for i, raw in enumerate(text.splitlines()):
|
|
294
|
+
# Drop a trailing ``// ...`` line comment before any matching (a ``//``
|
|
295
|
+
# inside a string literal is unrealistic for a declaration line).
|
|
296
|
+
code = raw.split("//", 1)[0]
|
|
297
|
+
stripped = code.strip()
|
|
298
|
+
if in_block:
|
|
299
|
+
if "*/" in stripped:
|
|
300
|
+
in_block = False
|
|
301
|
+
continue
|
|
302
|
+
if stripped.startswith("/*"):
|
|
303
|
+
# Single-line ``/* ... */`` -> skip without entering block state.
|
|
304
|
+
if "*/" not in stripped[2:]:
|
|
305
|
+
in_block = True
|
|
306
|
+
continue
|
|
307
|
+
if not stripped or stripped.startswith("*"):
|
|
308
|
+
# Blank or a Javadoc continuation line (`` * ...``).
|
|
309
|
+
continue
|
|
310
|
+
m = _JAVA_TYPE_DECL_RE.search(code)
|
|
311
|
+
if m and (not type_name or m.group(1) == type_name):
|
|
312
|
+
return anchor_line + i
|
|
313
|
+
return anchor_line
|
|
314
|
+
|
|
315
|
+
|
|
234
316
|
def explain_score_components(
|
|
235
317
|
comps: dict[str, float] | None,
|
|
236
318
|
*,
|
|
File without changes
|