java-codebase-rag 0.9.7__py3-none-any.whl → 0.10.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. java_codebase_rag/analysis/pr_analysis.py +33 -3
  2. java_codebase_rag/ast/ast_java.py +2 -1
  3. java_codebase_rag/cli.py +23 -8
  4. java_codebase_rag/config.py +68 -1
  5. java_codebase_rag/graph/build_ast_graph.py +123 -4
  6. java_codebase_rag/graph/graph_types.py +109 -22
  7. java_codebase_rag/graph/ladybug_queries.py +45 -2
  8. java_codebase_rag/index/java_index_flow_lancedb.py +10 -16
  9. java_codebase_rag/install_data/agents/explorer-rag-cli.md +3 -1
  10. java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +3 -1
  11. java_codebase_rag/jrag.py +627 -661
  12. java_codebase_rag/jrag_render.py +160 -3
  13. java_codebase_rag/lance_optimize.py +11 -12
  14. java_codebase_rag/mcp/mcp_v2.py +2 -1
  15. java_codebase_rag/pipeline.py +47 -1
  16. java_codebase_rag/read_payloads.py +781 -0
  17. java_codebase_rag/search/search_lancedb.py +138 -6
  18. java_codebase_rag/search/search_lexical.py +128 -30
  19. java_codebase_rag/search/search_scoring.py +82 -0
  20. java_codebase_rag/watch/__init__.py +0 -0
  21. java_codebase_rag/watch/client.py +230 -0
  22. java_codebase_rag/watch/daemon.py +368 -0
  23. java_codebase_rag/watch/lock.py +201 -0
  24. java_codebase_rag/watch/paths.py +76 -0
  25. java_codebase_rag/watch/protocol.py +122 -0
  26. java_codebase_rag/watch/server.py +273 -0
  27. java_codebase_rag/watch/warm.py +105 -0
  28. java_codebase_rag/watch/watcher.py +352 -0
  29. {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.0.dist-info}/METADATA +30 -31
  30. {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.0.dist-info}/RECORD +34 -24
  31. {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.0.dist-info}/WHEEL +0 -0
  32. {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.0.dist-info}/entry_points.txt +0 -0
  33. {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.0.dist-info}/licenses/LICENSE +0 -0
  34. {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.0.dist-info}/top_level.txt +0 -0
@@ -7,8 +7,11 @@ import argparse
7
7
  import json
8
8
  import os
9
9
  import sys
10
+ import tempfile
10
11
  import threading
12
+ import warnings
11
13
  from collections.abc import Callable
14
+ from contextlib import contextmanager
12
15
  from pathlib import Path
13
16
 
14
17
  import lancedb
@@ -44,8 +47,10 @@ from java_codebase_rag.search.search_scoring import ( # noqa: F401
44
47
  _role_weight,
45
48
  _split_identifier,
46
49
  _symbol_bonus,
50
+ declaration_line_number,
47
51
  explain_score_components,
48
52
  l2_distance_to_score,
53
+ vector_display_score,
49
54
  )
50
55
 
51
56
  TABLES: dict[str, str] = {
@@ -296,6 +301,117 @@ def _combine_predicates(parts: list[str | None]) -> str | None:
296
301
  return " AND ".join(f"({p})" for p in clean)
297
302
 
298
303
 
304
+ # LanceDB (0.30.x) emits two Rust `tracing` WARN lines per hybrid query to stderr
305
+ # — "specified output columns but did not include `_score`/`_distance` ... Call
306
+ # `disable_scoring_autoprojection`". They are noise on the agent's stderr, not
307
+ # Python warnings (so `warnings.filterwarnings` can't catch them), and the fluent
308
+ # query builder exposes no `disable_scoring_autoprojection()` (the lower-level
309
+ # `to_lance().scanner(...)` path needs `pylance`, which isn't installed on the
310
+ # PEP 508 graph-only profile). We match them by stable substring so anything that
311
+ # is a REAL error still reaches stderr.
312
+ _LANCE_AUTOPROJ_MARKERS: tuple[str, ...] = (
313
+ "disable_scoring_autoprojection",
314
+ "did not include `_distance`",
315
+ "did not include `_score`",
316
+ )
317
+
318
+ # The fd-2 redirect below mutates the PROCESS-GLOBAL fd 2 (and
319
+ # ``warnings.catch_warnings`` mutates global warning state). The MCP server
320
+ # dispatches every tool call through ``asyncio.to_thread`` on a thread pool
321
+ # (server.py), so two concurrent hybrid/auto-hybrid searches would race on the
322
+ # dup2 bookkeeping — corrupting the saved fd and crashing the whole server with
323
+ # ``Bad file descriptor``. Serialize the redirect so only one thread mutates fd
324
+ # 2 / warning state at a time. Concurrent hybrid queries therefore serialize
325
+ # their ``to_list()`` (correctness over throughput); a Rust-tracing-level
326
+ # suppression would remove the fd hijack entirely (follow-up).
327
+ _LANCE_WARN_REDIRECT_LOCK = threading.Lock()
328
+
329
+
330
+ def _is_autoproj_noise(line: str) -> bool:
331
+ """True for a LanceDB autoprojection-deprecation line (to drop).
332
+
333
+ Preserves genuine errors/tracebacks even if they happen to reference the API
334
+ name — only the bare deprecation log lines (no Error/Traceback/Exception) are
335
+ treated as noise.
336
+ """
337
+ if not any(marker in line for marker in _LANCE_AUTOPROJ_MARKERS):
338
+ return False
339
+ return not any(seg in line for seg in ("Traceback", "Error:", "error:", "Exception"))
340
+
341
+
342
+ @contextmanager
343
+ def _silence_lance_autoproj_warnings():
344
+ """Swallow LanceDB's `_score`/`_distance` autoprojection deprecation warnings.
345
+
346
+ Redirects fd 2 to a temp buffer for the duration of the wrapped call, drops
347
+ only the autoprojection deprecation lines, and re-emits everything else to
348
+ the real stderr so genuine errors stay visible. No-op if the caller opted
349
+ back in via ``JAVA_CODEBASE_RAG_KEEP_LANCE_WARNINGS`` (debugging).
350
+
351
+ Thread-safety: the redirect is serialized under ``_LANCE_WARN_REDIRECT_LOCK``
352
+ because it mutates process-global fd 2 and warning state — the MCP server
353
+ runs tool calls concurrently on a thread pool.
354
+ """
355
+ if os.environ.get("JAVA_CODEBASE_RAG_KEEP_LANCE_WARNINGS"):
356
+ yield
357
+ return
358
+ # Also catch the (unlikely) Python-warning form defensively.
359
+ with _LANCE_WARN_REDIRECT_LOCK, warnings.catch_warnings():
360
+ warnings.filterwarnings(
361
+ "ignore",
362
+ message=r".*(disable_scoring_autoprojection|did not include `(_distance|_score)`).*",
363
+ )
364
+ with tempfile.TemporaryFile(mode="w+", encoding="utf-8", errors="replace") as captured:
365
+ saved = os.dup(2)
366
+ try:
367
+ os.dup2(captured.fileno(), 2)
368
+ yield
369
+ finally:
370
+ # Restore fd 2 FIRST so the re-emit below reaches real stderr.
371
+ os.dup2(saved, 2)
372
+ os.close(saved)
373
+ captured.seek(0)
374
+ kept = "".join(line for line in captured if not _is_autoproj_noise(line))
375
+ if kept:
376
+ sys.stderr.write(kept)
377
+ sys.stderr.flush()
378
+
379
+
380
+ def _simple_type_name(fqn: str | None) -> str | None:
381
+ """``com.foo.Bar`` -> ``Bar``; None/empty -> None."""
382
+ if not fqn:
383
+ return None
384
+ return str(fqn).rsplit(".", 1)[-1] or None
385
+
386
+
387
+ def _refine_java_start_lines(rows: list[dict]) -> None:
388
+ """Point each java row's ``start.line`` at the type declaration, not the chunk anchor.
389
+
390
+ LanceDB chunks are anchored at the chunk's first source line — for a
391
+ file-spanning chunk that's the package/import line (``start.line`` = 1)
392
+ while the ``class``/``interface`` declaration sits several lines down. The
393
+ chunk anchor is a poor display line for a symbol hit (renders as
394
+ ``File.java:1``); derive the real declaration line from the chunk text
395
+ (pinned to the primary type) so a hit shows ``File.java:<decl>`` instead
396
+ (F8). Method-only chunks whose range doesn't include a type declaration
397
+ keep their chunk anchor unchanged.
398
+ """
399
+ for r in rows:
400
+ if str(r.get("_kind", "")) != "java":
401
+ continue
402
+ start = r.get("start")
403
+ if not isinstance(start, dict):
404
+ continue
405
+ anchor = start.get("line")
406
+ if anchor is None:
407
+ continue
408
+ hints = r.get("_hints") or {}
409
+ type_name = hints.get("primary_type_hint") or _simple_type_name(r.get("primary_type_fqn"))
410
+ decl = declaration_line_number(r.get("text"), int(anchor), type_name)
411
+ if decl is not None:
412
+ start["line"] = decl
413
+
414
+
299
415
  def _search_one_table(
300
416
  table_name: str,
301
417
  *,
@@ -342,7 +458,11 @@ def _search_one_table(
342
458
  )
343
459
  if combined_pred:
344
460
  q = q.where(combined_pred, prefilter=True)
345
- rows = q.to_list()
461
+ # Hybrid selects explicit output columns without `_score`/`_distance`, so
462
+ # LanceDB (0.30.x) emits two Rust autoprojection deprecation WARNs to
463
+ # stderr per query. Silence just those lines; real errors still surface.
464
+ with _silence_lance_autoproj_warnings():
465
+ rows = q.to_list()
346
466
  for r in rows:
347
467
  r["_kind"] = kind
348
468
  rs = r.pop("_relevance_score", None)
@@ -375,7 +495,11 @@ def _search_one_table(
375
495
  # exposed score was always 0.0, making results look unranked.
376
496
  d = r.get("_distance")
377
497
  if d is not None:
378
- r["_score"] = l2_distance_to_score(float(d))
498
+ # Use the same non-clamping map as the display sites so graph-expand
499
+ # rows (which run_search does NOT overwrite) never carry the old
500
+ # 1-d²/2 value that collapses to 0 past √2. (The main single/multi
501
+ # paths overwrite this with the bonus-adjusted effective distance.)
502
+ r["_score"] = vector_display_score(float(d))
379
503
  r["start"] = coerce_position_field(r.get("start"))
380
504
  r["end"] = coerce_position_field(r.get("end"))
381
505
  return rows
@@ -575,6 +699,7 @@ def _graph_expand_merge(
575
699
  except Exception:
576
700
  return vector_rows
577
701
  _apply_chunk_hints(graph_rows)
702
+ _refine_java_start_lines(graph_rows)
578
703
  graph_rows.sort(key=_vector_sort_key)
579
704
  for r in graph_rows:
580
705
  r["_graph_expanded"] = True
@@ -737,6 +862,9 @@ def run_search(
737
862
  extra_predicates=preds,
738
863
  )
739
864
  _apply_chunk_hints(rows)
865
+ # Anchor each java row's start.line on the type declaration instead of
866
+ # the chunk's first source line (often the package/import line = 1).
867
+ _refine_java_start_lines(rows)
740
868
  if skip_role_weight:
741
869
  for r in rows:
742
870
  r["_skip_role_weight"] = True
@@ -747,11 +875,14 @@ def run_search(
747
875
  _hybrid_post_sort_normalization(rows)
748
876
  else:
749
877
  rows.sort(key=_vector_sort_key)
750
- # Vector: set honest displayed score from adjusted distance, clamped to [0,1]
878
+ # Vector: displayed score from the effective (bonus-adjusted) distance,
879
+ # normalized over the unit-embedding range so a correctly-ranked top
880
+ # hit never collapses to 0.000 (the cosine map 1 - d²/2 clamps to 0
881
+ # past √2; weak-but-best matches commonly sit at d ≈ 1.5).
751
882
  for r in rows:
752
883
  comps = r.setdefault("_score_components", {})
753
884
  effective_dist = _effective_distance(comps)
754
- r["_score"] = _clamp01(l2_distance_to_score(effective_dist))
885
+ r["_score"] = vector_display_score(effective_dist)
755
886
 
756
887
  if graph_expand and key == "java" and expand_depth > 0:
757
888
  rows = _graph_expand_merge(
@@ -792,16 +923,17 @@ def run_search(
792
923
  )
793
924
  )
794
925
  _apply_chunk_hints(merged)
926
+ _refine_java_start_lines(merged)
795
927
  if skip_role_weight:
796
928
  for r in merged:
797
929
  r["_skip_role_weight"] = True
798
930
  _apply_symbol_bonus(merged, query_toks)
799
931
  merged.sort(key=_vector_sort_key)
800
- # Vector: set honest displayed score from adjusted distance, clamped to [0,1]
932
+ # Vector: displayed score from the effective (bonus-adjusted) distance.
801
933
  for r in merged:
802
934
  comps = r.setdefault("_score_components", {})
803
935
  effective_dist = _effective_distance(comps)
804
- r["_score"] = _clamp01(l2_distance_to_score(effective_dist))
936
+ r["_score"] = vector_display_score(effective_dist)
805
937
 
806
938
  # Dedup by primary_type_fqn after all sorting/merging, before windowing
807
939
  merged = _dedup_by_fqn(merged, dedup_by_fqn=dedup_by_fqn)
@@ -18,11 +18,13 @@ guarded by a parity unit test.
18
18
  from __future__ import annotations
19
19
 
20
20
  import os
21
+ import weakref
21
22
  from pathlib import Path
22
23
  from typing import TYPE_CHECKING, Any
23
24
 
24
25
  from java_codebase_rag.graph.ladybug_queries import LadybugGraph
25
26
  from java_codebase_rag.search.search_scoring import (
27
+ SYMBOL_FTS_INDEX,
26
28
  _ROLE_SCORE_WEIGHTS,
27
29
  _TYPE_MATCH_BONUS_CAP,
28
30
  _TYPE_MATCH_BONUS_PER_HIT,
@@ -170,6 +172,88 @@ def _token_overlap(haystack_toks: set[str], needle_toks: set[str]) -> float:
170
172
  return len(needle_toks & haystack_toks) / len(needle_toks)
171
173
 
172
174
 
175
+ # BM25 candidate fetch via the LadybugDB FTS index (fork A). DB-side indexed ranking
176
+ # replaces the heuristic's bounded Python scan; the heuristic below still scores the
177
+ # fetched candidates (name/type/fqn/role) and is the fallback when the FTS index or
178
+ # extension is unavailable (older graph, offline first run).
179
+ _FTS_CANDIDATE_K = 200 # top-K BM25 candidates; re-filtered by NodeFilter before ranking
180
+ # Connections that have run LOAD EXTENSION FTS. Keyed by the connection OBJECT (WeakSet),
181
+ # NOT id() — id() is reused after GC, which would let a fresh connection skip LOAD and then
182
+ # fail at QUERY_FTS_INDEX under test batching. Entries die with the connection.
183
+ _FTS_LOADED_CONNS: "weakref.WeakSet[object]" = weakref.WeakSet()
184
+
185
+
186
+ def _ensure_fts_loaded(g: LadybugGraph) -> bool:
187
+ """LOAD EXTENSION FTS on the graph's (read-only) connection, once per connection.
188
+
189
+ Returns False if the extension can't be loaded (absent / offline) so the caller
190
+ falls back to the heuristic scan.
191
+ """
192
+ conn = g._conn # noqa: SLF001
193
+ try:
194
+ if conn in _FTS_LOADED_CONNS:
195
+ return True
196
+ except Exception: # connection not weakref-able → LOAD every call (correct, slow)
197
+ pass
198
+ try:
199
+ g._rows("LOAD EXTENSION FTS") # noqa: SLF001
200
+ try:
201
+ _FTS_LOADED_CONNS.add(conn)
202
+ except Exception:
203
+ pass
204
+ return True
205
+ except Exception:
206
+ return False
207
+
208
+
209
+ def _try_fts_candidates(
210
+ g: LadybugGraph,
211
+ query: str,
212
+ filter: NodeFilter | None,
213
+ path_contains: str | None,
214
+ ) -> dict | None:
215
+ """Fetch BM25-ranked Symbol candidates via the FTS index; re-apply NodeFilter.
216
+
217
+ Returns ``{"rows": [...], "scores": {id: bm25}}`` (rows are the same shape the
218
+ heuristic scan yields), or ``None`` when FTS is unavailable (extension won't load,
219
+ or the index isn't present on this graph) so the caller falls back.
220
+
221
+ Two-step: (1) ``QUERY_FTS_INDEX`` returns the top-K node ids by Okapi BM25 over
222
+ ``Symbol.search_text``; (2) re-MATCH those ids with the full ``_lexical_where``
223
+ predicates (role / module / path / kind≠file,package) so the filter logic stays
224
+ defined in one place. ``search_text`` is built at index time by ``build_ast_graph``
225
+ from the same ``_split_identifier`` the re-rank below uses, so index- and query-time
226
+ tokenization agree.
227
+ """
228
+ if not _ensure_fts_loaded(g):
229
+ return None
230
+ idx_rows = g._rows("CALL SHOW_INDEXES() RETURN index_name") # noqa: SLF001
231
+ names = {row.get("index_name") for row in idx_rows}
232
+ if SYMBOL_FTS_INDEX not in names:
233
+ return None
234
+ fts = g._rows( # noqa: SLF001
235
+ f"CALL QUERY_FTS_INDEX('Symbol', '{SYMBOL_FTS_INDEX}', $q, top := $k) "
236
+ "RETURN node.id AS id, score",
237
+ {"q": query, "k": _FTS_CANDIDATE_K},
238
+ )
239
+ if not fts:
240
+ return {"rows": [], "scores": {}}
241
+ scores = {row["id"]: float(row.get("score") or 0.0) for row in fts}
242
+ ids = list(scores.keys())
243
+
244
+ # Re-MATCH the K ids with the SAME predicates the heuristic pushes down, so
245
+ # NodeFilter / path / structural-kind filtering is defined exactly once.
246
+ where, params = _lexical_where(filter, path_contains=path_contains)
247
+ struct_pred = "(s.kind <> 'file' AND s.kind <> 'package')"
248
+ if not where:
249
+ where = f"WHERE s.id IN $ids AND {struct_pred}"
250
+ else:
251
+ where = where.replace("WHERE ", f"WHERE s.id IN $ids AND {struct_pred} AND ", 1)
252
+ params["ids"] = ids
253
+ rows = g._rows(f"MATCH (s:Symbol) {where} RETURN {_SYMBOL_RETURN}", params) # noqa: SLF001
254
+ return {"rows": rows, "scores": scores}
255
+
256
+
173
257
  def run_lexical_search(
174
258
  query: str,
175
259
  *,
@@ -185,6 +269,12 @@ def run_lexical_search(
185
269
  ) -> list[dict]:
186
270
  """Keyword search over Symbol nodes; returns ``run_search``-shaped row-dicts.
187
271
 
272
+ BM25-first (fork A): when the LadybugDB ``sym_fts`` index exists, candidates are
273
+ fetched DB-side via Okapi BM25 over ``Symbol.search_text`` (killing the bounded
274
+ Python scan that silently missed matches past the cap on large repos) and then
275
+ re-ranked here by the name/type/fqn/role heuristic. Falls back to that heuristic
276
+ scan when the FTS index or extension is unavailable (older graph, offline first run).
277
+
188
278
  Raises ``RuntimeError`` (message contains "lexical search unavailable") if no
189
279
  symbol graph exists — the caller maps that to a clean failure envelope. Returns
190
280
  ``[]`` for ``table in ("sql", "yaml")`` (those LanceDB tables aren't built in
@@ -201,33 +291,38 @@ def run_lexical_search(
201
291
  )
202
292
  g = graph or LadybugGraph.get()
203
293
 
204
- where, params = _lexical_where(filter, path_contains=path_contains)
205
- # Always exclude structural Symbol nodes. Files and packages are :Symbol-labeled
206
- # (kind='file'/'package') but aren't searchable code declarations — without this
207
- # a token that appears in a filename (e.g. 'distribution' in
208
- # 'DistributionChunkService.java') would surface the file node as a hit.
209
- struct_pred = "(s.kind <> 'file' AND s.kind <> 'package')"
210
- where = f"WHERE {struct_pred}" if not where else where.replace("WHERE ", f"WHERE {struct_pred} AND ", 1)
211
- # Lexical ranking is done in Python (LadybugDB/kuzu has no keyword ranking without
212
- # FTS5, which is deferred), and the MATCH scan returns rows in storage order — there
213
- # is NO DB-side relevance ORDER BY. So fetch the FULL candidate pool up to the safety
214
- # cap and rank here. A pagination-derived LIMIT (4x the page) is correct on the vector
215
- # path where LanceDB returns rows pre-ranked by similarity, but on this unordered scan
216
- # it would return only the first ~N symbols in arbitrary storage order and silently
217
- # miss the best match on any non-trivial repo.
218
- params["lim"] = _CANDIDATE_LIMIT_CAP
219
- cypher = f"MATCH (s:Symbol) {where} RETURN {_SYMBOL_RETURN} LIMIT $lim"
220
- rows = g._rows(cypher, params) # noqa: SLF001 — de facto public read API (see find_v2)
221
- # If the fetch hit the safety cap, deeper matches were never ranked (the scan has no
222
- # ORDER BY — kuzu returns an arbitrary, storage-order-dependent subset). Surface it so
223
- # a user on a large repo isn't silently shown an incomplete result set; refining the
224
- # query or adding a filter narrows the pool below the cap. Raising the cap / FTS5 is
225
- # the deferred long-term fix (see the plan's "Out of scope" note).
226
- if advisories is not None and len(rows) >= _CANDIDATE_LIMIT_CAP:
227
- advisories.append(
228
- f"lexical search scanned the first {_CANDIDATE_LIMIT_CAP} matching symbols "
229
- "(repo cap); deeper matches were not ranked — refine the query or add a filter"
230
- )
294
+ # --- candidate fetch: BM25 (FTS) preferred, heuristic scan fallback ---
295
+ bm25_scores: dict[str, float] = {}
296
+ use_fts = False
297
+ fts = _try_fts_candidates(g, query, filter, path_contains)
298
+ if fts is not None:
299
+ rows = fts["rows"]
300
+ bm25_scores = fts["scores"]
301
+ use_fts = True
302
+ else:
303
+ where, params = _lexical_where(filter, path_contains=path_contains)
304
+ # Always exclude structural Symbol nodes. Files and packages are :Symbol-labeled
305
+ # (kind='file'/'package') but aren't searchable code declarations — without this
306
+ # a token that appears in a filename (e.g. 'distribution' in
307
+ # 'DistributionChunkService.java') would surface the file node as a hit.
308
+ struct_pred = "(s.kind <> 'file' AND s.kind <> 'package')"
309
+ where = f"WHERE {struct_pred}" if not where else where.replace("WHERE ", f"WHERE {struct_pred} AND ", 1)
310
+ # The heuristic scan returns rows in storage order — there is NO DB-side relevance
311
+ # ORDER BY without the FTS index — so fetch the FULL candidate pool up to the safety
312
+ # cap and rank here. A pagination-derived LIMIT (4x the page) is correct on the
313
+ # vector path where LanceDB returns rows pre-ranked, but on this unordered scan it
314
+ # would return only the first ~N symbols in arbitrary storage order and silently
315
+ # miss the best match on any non-trivial repo. The BM25 (FTS) path above has no cap.
316
+ params["lim"] = _CANDIDATE_LIMIT_CAP
317
+ cypher = f"MATCH (s:Symbol) {where} RETURN {_SYMBOL_RETURN} LIMIT $lim"
318
+ rows = g._rows(cypher, params) # noqa: SLF001 — de facto public read API (see find_v2)
319
+ # If the fetch hit the safety cap, deeper matches were never ranked. Surface it so
320
+ # a user on a large repo isn't silently shown an incomplete result set.
321
+ if advisories is not None and len(rows) >= _CANDIDATE_LIMIT_CAP:
322
+ advisories.append(
323
+ f"lexical search scanned the first {_CANDIDATE_LIMIT_CAP} matching symbols "
324
+ "(repo cap); deeper matches were not ranked — refine the query or add a filter"
325
+ )
231
326
 
232
327
  query_toks = _query_tokens(query)
233
328
  source_root = _resolve_source_root(g)
@@ -265,9 +360,10 @@ def run_lexical_search(
265
360
  text_match = text_overlap * _TEXT_MATCH_WEIGHT
266
361
 
267
362
  # A keyword search must require at least one lexical hit — role alone never
268
- # qualifies a row (it only boosts/reorders matches). Degenerate queries with
269
- # no usable tokens fall through to role-ranked listing.
270
- if query_toks and not (name_overlap or type_hits or fqn_match or text_overlap):
363
+ # qualifies a row (it only boosts/reorders matches). On the BM25 path the FTS
364
+ # index already established textual relevance, so the qualifier is heuristic-only.
365
+ # Degenerate queries with no usable tokens fall through to role-ranked listing.
366
+ if query_toks and not use_fts and not (name_overlap or type_hits or fqn_match or text_overlap):
271
367
  continue
272
368
 
273
369
  role_w = 0.0 if role_locked else _ROLE_SCORE_WEIGHTS.get(role_raw.upper(), 0.0)
@@ -291,6 +387,8 @@ def run_lexical_search(
291
387
  "lexical_relevance": round(raw, 4),
292
388
  "role_weight": role_w,
293
389
  }
390
+ if use_fts:
391
+ comps["bm25"] = round(float(bm25_scores.get(r.get("id"), 0.0)), 4)
294
392
 
295
393
  sl, el, sb, eb = r.get("start_line"), r.get("end_line"), r.get("start_byte"), r.get("end_byte")
296
394
  out.append(
@@ -12,6 +12,12 @@ Everything here is pure-Python dict/list math with no third-party deps.
12
12
  from __future__ import annotations
13
13
 
14
14
  import json
15
+ import re
16
+
17
+ # Name of the LadybugDB FTS (Okapi BM25) index over Symbol.search_text (fork A).
18
+ # Shared by the build path (build_ast_graph._ensure_symbol_fts_index) and the
19
+ # query path (search_lexical.run_lexical_search) so the two never drift.
20
+ SYMBOL_FTS_INDEX = "sym_fts"
15
21
 
16
22
  # Over-fetch multiplier for dedup: fetch 4x to absorb per-FQN chunk multiplicity
17
23
  # so that after collapsing by primary_type_fqn, a page stays full and the +1
@@ -210,6 +216,29 @@ def l2_distance_to_score(distance: float) -> float:
210
216
  return 1.0 - distance * distance / 2.0
211
217
 
212
218
 
219
+ # Display-score denominator for the vector backend. Unit-normalized embeddings
220
+ # have L2 distance in [0, 2]; the cosine map ``l2_distance_to_score`` (1 - d²/2)
221
+ # goes NEGATIVE past √2 ≈ 1.414 and clamps to 0. Weak-but-best semantic matches
222
+ # (e.g. a lone keyword like "controller") commonly sit at d ≈ 1.5, so EVERY hit
223
+ # clamps to score=0.000 even though the ranking is correct. ``vector_display_score``
224
+ # instead normalizes the effective (bonus-adjusted) distance over the full
225
+ # unit-embedding range, so a top-ranked hit stays visibly non-zero. Role/symbol
226
+ # bonuses reduce the effective distance and so raise the displayed score,
227
+ # keeping it rank-monotonic with the distance-based sort key.
228
+ _VECTOR_DISTANCE_REF = 2.0
229
+
230
+
231
+ def vector_display_score(effective_distance: float) -> float:
232
+ """Displayed vector score in [0, 1] from the effective (bonus-adjusted) distance.
233
+
234
+ Bounded linear normalization over the unit-embedding L2 range [0, 2]: lower
235
+ distance → higher score. Unlike ``l2_distance_to_score`` (which goes
236
+ negative past √2 and clamps a correctly-ranked top hit to 0.000), this keeps
237
+ a top result visibly non-zero while staying rank-monotonic with the sort key.
238
+ """
239
+ return _clamp01(1.0 - effective_distance / _VECTOR_DISTANCE_REF)
240
+
241
+
213
242
  def _effective_distance(comps: dict[str, float]) -> float:
214
243
  """Compute the adjusted distance used for sorting.
215
244
 
@@ -231,6 +260,59 @@ def _clamp01(x: float) -> float:
231
260
  return x
232
261
 
233
262
 
263
+ # Matches a Java top-level type declaration and captures its simple name. Mirrors
264
+ # the heuristic in ``ast.chunk_heuristics._JAVA_TYPE`` but is duplicated here so
265
+ # this module stays dependency-free (importable on graph-only Intel installs).
266
+ _JAVA_TYPE_DECL_RE = re.compile(
267
+ r"\b(?:public\s+|private\s+|protected\s+|sealed\s+|non-sealed\s+|final\s+|"
268
+ r"abstract\s+|static\s+)*"
269
+ r"(?:class|interface|enum|record)\s+([A-Za-z_][A-Za-z0-9_]*)"
270
+ )
271
+
272
+
273
+ def declaration_line_number(
274
+ text: str | None, anchor_line: int | None, type_name: str | None = None
275
+ ) -> int | None:
276
+ """Absolute 1-based line of the Java type declaration within chunk ``text``.
277
+
278
+ LanceDB chunks are anchored at the chunk's first source line, which for a
279
+ file-spanning chunk is the package/import line (``anchor_line`` = 1) while
280
+ the ``class``/``interface`` declaration sits several lines down. Without
281
+ this, hits render as ``File.java:1`` even though the symbol is declared
282
+ later (F8). Returns ``anchor_line + i`` for the first matching declaration
283
+ (pinned to ``type_name`` when given, so a nested type doesn't win), or
284
+ ``anchor_line`` unchanged when no declaration is found in the chunk.
285
+
286
+ Comment-aware: Javadoc/line/block-comment lines that merely MENTION the type
287
+ name (e.g. ``* This class Bar handles...``) are skipped so the returned line
288
+ is the real declaration, not a comment above it.
289
+ """
290
+ if not text or anchor_line is None:
291
+ return anchor_line
292
+ in_block = False
293
+ for i, raw in enumerate(text.splitlines()):
294
+ # Drop a trailing ``// ...`` line comment before any matching (a ``//``
295
+ # inside a string literal is unrealistic for a declaration line).
296
+ code = raw.split("//", 1)[0]
297
+ stripped = code.strip()
298
+ if in_block:
299
+ if "*/" in stripped:
300
+ in_block = False
301
+ continue
302
+ if stripped.startswith("/*"):
303
+ # Single-line ``/* ... */`` -> skip without entering block state.
304
+ if "*/" not in stripped[2:]:
305
+ in_block = True
306
+ continue
307
+ if not stripped or stripped.startswith("*"):
308
+ # Blank or a Javadoc continuation line (`` * ...``).
309
+ continue
310
+ m = _JAVA_TYPE_DECL_RE.search(code)
311
+ if m and (not type_name or m.group(1) == type_name):
312
+ return anchor_line + i
313
+ return anchor_line
314
+
315
+
234
316
  def explain_score_components(
235
317
  comps: dict[str, float] | None,
236
318
  *,
File without changes