java-codebase-rag 0.9.7__py3-none-any.whl → 0.10.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. java_codebase_rag/analysis/pr_analysis.py +33 -3
  2. java_codebase_rag/ast/ast_java.py +2 -1
  3. java_codebase_rag/cli.py +23 -8
  4. java_codebase_rag/config.py +68 -1
  5. java_codebase_rag/graph/build_ast_graph.py +123 -4
  6. java_codebase_rag/graph/graph_types.py +109 -22
  7. java_codebase_rag/graph/ladybug_queries.py +45 -2
  8. java_codebase_rag/index/java_index_flow_lancedb.py +10 -16
  9. java_codebase_rag/install_data/agents/explorer-rag-cli.md +3 -1
  10. java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +3 -1
  11. java_codebase_rag/jrag.py +627 -661
  12. java_codebase_rag/jrag_render.py +160 -3
  13. java_codebase_rag/lance_optimize.py +11 -12
  14. java_codebase_rag/mcp/mcp_v2.py +2 -1
  15. java_codebase_rag/pipeline.py +47 -1
  16. java_codebase_rag/read_payloads.py +781 -0
  17. java_codebase_rag/search/search_lancedb.py +138 -6
  18. java_codebase_rag/search/search_lexical.py +140 -30
  19. java_codebase_rag/search/search_scoring.py +108 -0
  20. java_codebase_rag/watch/__init__.py +0 -0
  21. java_codebase_rag/watch/client.py +230 -0
  22. java_codebase_rag/watch/daemon.py +368 -0
  23. java_codebase_rag/watch/lock.py +201 -0
  24. java_codebase_rag/watch/paths.py +76 -0
  25. java_codebase_rag/watch/protocol.py +122 -0
  26. java_codebase_rag/watch/server.py +273 -0
  27. java_codebase_rag/watch/warm.py +105 -0
  28. java_codebase_rag/watch/watcher.py +352 -0
  29. {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.1.dist-info}/METADATA +30 -31
  30. {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.1.dist-info}/RECORD +34 -24
  31. {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.1.dist-info}/WHEEL +0 -0
  32. {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.1.dist-info}/entry_points.txt +0 -0
  33. {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.1.dist-info}/licenses/LICENSE +0 -0
  34. {java_codebase_rag-0.9.7.dist-info → java_codebase_rag-0.10.1.dist-info}/top_level.txt +0 -0
@@ -7,8 +7,11 @@ import argparse
7
7
  import json
8
8
  import os
9
9
  import sys
10
+ import tempfile
10
11
  import threading
12
+ import warnings
11
13
  from collections.abc import Callable
14
+ from contextlib import contextmanager
12
15
  from pathlib import Path
13
16
 
14
17
  import lancedb
@@ -44,8 +47,10 @@ from java_codebase_rag.search.search_scoring import ( # noqa: F401
44
47
  _role_weight,
45
48
  _split_identifier,
46
49
  _symbol_bonus,
50
+ declaration_line_number,
47
51
  explain_score_components,
48
52
  l2_distance_to_score,
53
+ vector_display_score,
49
54
  )
50
55
 
51
56
  TABLES: dict[str, str] = {
@@ -296,6 +301,117 @@ def _combine_predicates(parts: list[str | None]) -> str | None:
296
301
  return " AND ".join(f"({p})" for p in clean)
297
302
 
298
303
 
304
+ # LanceDB (0.30.x) emits two Rust `tracing` WARN lines per hybrid query to stderr
305
+ # — "specified output columns but did not include `_score`/`_distance` ... Call
306
+ # `disable_scoring_autoprojection`". They are noise on the agent's stderr, not
307
+ # Python warnings (so `warnings.filterwarnings` can't catch them), and the fluent
308
+ # query builder exposes no `disable_scoring_autoprojection()` (the lower-level
309
+ # `to_lance().scanner(...)` path needs `pylance`, which isn't installed on the
310
+ # PEP 508 graph-only profile). We match them by stable substring so anything that
311
+ # is a REAL error still reaches stderr.
312
+ _LANCE_AUTOPROJ_MARKERS: tuple[str, ...] = (
313
+ "disable_scoring_autoprojection",
314
+ "did not include `_distance`",
315
+ "did not include `_score`",
316
+ )
317
+
318
+ # The fd-2 redirect below mutates the PROCESS-GLOBAL fd 2 (and
319
+ # ``warnings.catch_warnings`` mutates global warning state). The MCP server
320
+ # dispatches every tool call through ``asyncio.to_thread`` on a thread pool
321
+ # (server.py), so two concurrent hybrid/auto-hybrid searches would race on the
322
+ # dup2 bookkeeping — corrupting the saved fd and crashing the whole server with
323
+ # ``Bad file descriptor``. Serialize the redirect so only one thread mutates fd
324
+ # 2 / warning state at a time. Concurrent hybrid queries therefore serialize
325
+ # their ``to_list()`` (correctness over throughput); a Rust-tracing-level
326
+ # suppression would remove the fd hijack entirely (follow-up).
327
+ _LANCE_WARN_REDIRECT_LOCK = threading.Lock()
328
+
329
+
330
+ def _is_autoproj_noise(line: str) -> bool:
331
+ """True for a LanceDB autoprojection-deprecation line (to drop).
332
+
333
+ Preserves genuine errors/tracebacks even if they happen to reference the API
334
+ name — only the bare deprecation log lines (no Error/Traceback/Exception) are
335
+ treated as noise.
336
+ """
337
+ if not any(marker in line for marker in _LANCE_AUTOPROJ_MARKERS):
338
+ return False
339
+ return not any(seg in line for seg in ("Traceback", "Error:", "error:", "Exception"))
340
+
341
+
342
+ @contextmanager
343
+ def _silence_lance_autoproj_warnings():
344
+ """Swallow LanceDB's `_score`/`_distance` autoprojection deprecation warnings.
345
+
346
+ Redirects fd 2 to a temp buffer for the duration of the wrapped call, drops
347
+ only the autoprojection deprecation lines, and re-emits everything else to
348
+ the real stderr so genuine errors stay visible. No-op if the caller opted
349
+ back in via ``JAVA_CODEBASE_RAG_KEEP_LANCE_WARNINGS`` (debugging).
350
+
351
+ Thread-safety: the redirect is serialized under ``_LANCE_WARN_REDIRECT_LOCK``
352
+ because it mutates process-global fd 2 and warning state — the MCP server
353
+ runs tool calls concurrently on a thread pool.
354
+ """
355
+ if os.environ.get("JAVA_CODEBASE_RAG_KEEP_LANCE_WARNINGS"):
356
+ yield
357
+ return
358
+ # Also catch the (unlikely) Python-warning form defensively.
359
+ with _LANCE_WARN_REDIRECT_LOCK, warnings.catch_warnings():
360
+ warnings.filterwarnings(
361
+ "ignore",
362
+ message=r".*(disable_scoring_autoprojection|did not include `(_distance|_score)`).*",
363
+ )
364
+ with tempfile.TemporaryFile(mode="w+", encoding="utf-8", errors="replace") as captured:
365
+ saved = os.dup(2)
366
+ try:
367
+ os.dup2(captured.fileno(), 2)
368
+ yield
369
+ finally:
370
+ # Restore fd 2 FIRST so the re-emit below reaches real stderr.
371
+ os.dup2(saved, 2)
372
+ os.close(saved)
373
+ captured.seek(0)
374
+ kept = "".join(line for line in captured if not _is_autoproj_noise(line))
375
+ if kept:
376
+ sys.stderr.write(kept)
377
+ sys.stderr.flush()
378
+
379
+
380
+ def _simple_type_name(fqn: str | None) -> str | None:
381
+ """``com.foo.Bar`` -> ``Bar``; None/empty -> None."""
382
+ if not fqn:
383
+ return None
384
+ return str(fqn).rsplit(".", 1)[-1] or None
385
+
386
+
387
+ def _refine_java_start_lines(rows: list[dict]) -> None:
388
+ """Point each java row's ``start.line`` at the type declaration, not the chunk anchor.
389
+
390
+ LanceDB chunks are anchored at the chunk's first source line — for a
391
+ file-spanning chunk that's the package/import line (``start.line`` = 1)
392
+ while the ``class``/``interface`` declaration sits several lines down. The
393
+ chunk anchor is a poor display line for a symbol hit (renders as
394
+ ``File.java:1``); derive the real declaration line from the chunk text
395
+ (pinned to the primary type) so a hit shows ``File.java:<decl>`` instead
396
+ (F8). Method-only chunks whose range doesn't include a type declaration
397
+ keep their chunk anchor unchanged.
398
+ """
399
+ for r in rows:
400
+ if str(r.get("_kind", "")) != "java":
401
+ continue
402
+ start = r.get("start")
403
+ if not isinstance(start, dict):
404
+ continue
405
+ anchor = start.get("line")
406
+ if anchor is None:
407
+ continue
408
+ hints = r.get("_hints") or {}
409
+ type_name = hints.get("primary_type_hint") or _simple_type_name(r.get("primary_type_fqn"))
410
+ decl = declaration_line_number(r.get("text"), int(anchor), type_name)
411
+ if decl is not None:
412
+ start["line"] = decl
413
+
414
+
299
415
  def _search_one_table(
300
416
  table_name: str,
301
417
  *,
@@ -342,7 +458,11 @@ def _search_one_table(
342
458
  )
343
459
  if combined_pred:
344
460
  q = q.where(combined_pred, prefilter=True)
345
- rows = q.to_list()
461
+ # Hybrid selects explicit output columns without `_score`/`_distance`, so
462
+ # LanceDB (0.30.x) emits two Rust autoprojection deprecation WARNs to
463
+ # stderr per query. Silence just those lines; real errors still surface.
464
+ with _silence_lance_autoproj_warnings():
465
+ rows = q.to_list()
346
466
  for r in rows:
347
467
  r["_kind"] = kind
348
468
  rs = r.pop("_relevance_score", None)
@@ -375,7 +495,11 @@ def _search_one_table(
375
495
  # exposed score was always 0.0, making results look unranked.
376
496
  d = r.get("_distance")
377
497
  if d is not None:
378
- r["_score"] = l2_distance_to_score(float(d))
498
+ # Use the same non-clamping map as the display sites so graph-expand
499
+ # rows (which run_search does NOT overwrite) never carry the old
500
+ # 1-d²/2 value that collapses to 0 past √2. (The main single/multi
501
+ # paths overwrite this with the bonus-adjusted effective distance.)
502
+ r["_score"] = vector_display_score(float(d))
379
503
  r["start"] = coerce_position_field(r.get("start"))
380
504
  r["end"] = coerce_position_field(r.get("end"))
381
505
  return rows
@@ -575,6 +699,7 @@ def _graph_expand_merge(
575
699
  except Exception:
576
700
  return vector_rows
577
701
  _apply_chunk_hints(graph_rows)
702
+ _refine_java_start_lines(graph_rows)
578
703
  graph_rows.sort(key=_vector_sort_key)
579
704
  for r in graph_rows:
580
705
  r["_graph_expanded"] = True
@@ -737,6 +862,9 @@ def run_search(
737
862
  extra_predicates=preds,
738
863
  )
739
864
  _apply_chunk_hints(rows)
865
+ # Anchor each java row's start.line on the type declaration instead of
866
+ # the chunk's first source line (often the package/import line = 1).
867
+ _refine_java_start_lines(rows)
740
868
  if skip_role_weight:
741
869
  for r in rows:
742
870
  r["_skip_role_weight"] = True
@@ -747,11 +875,14 @@ def run_search(
747
875
  _hybrid_post_sort_normalization(rows)
748
876
  else:
749
877
  rows.sort(key=_vector_sort_key)
750
- # Vector: set honest displayed score from adjusted distance, clamped to [0,1]
878
+ # Vector: displayed score from the effective (bonus-adjusted) distance,
879
+ # normalized over the unit-embedding range so a correctly-ranked top
880
+ # hit never collapses to 0.000 (the cosine map 1 - d²/2 clamps to 0
881
+ # past √2; weak-but-best matches commonly sit at d ≈ 1.5).
751
882
  for r in rows:
752
883
  comps = r.setdefault("_score_components", {})
753
884
  effective_dist = _effective_distance(comps)
754
- r["_score"] = _clamp01(l2_distance_to_score(effective_dist))
885
+ r["_score"] = vector_display_score(effective_dist)
755
886
 
756
887
  if graph_expand and key == "java" and expand_depth > 0:
757
888
  rows = _graph_expand_merge(
@@ -792,16 +923,17 @@ def run_search(
792
923
  )
793
924
  )
794
925
  _apply_chunk_hints(merged)
926
+ _refine_java_start_lines(merged)
795
927
  if skip_role_weight:
796
928
  for r in merged:
797
929
  r["_skip_role_weight"] = True
798
930
  _apply_symbol_bonus(merged, query_toks)
799
931
  merged.sort(key=_vector_sort_key)
800
- # Vector: set honest displayed score from adjusted distance, clamped to [0,1]
932
+ # Vector: displayed score from the effective (bonus-adjusted) distance.
801
933
  for r in merged:
802
934
  comps = r.setdefault("_score_components", {})
803
935
  effective_dist = _effective_distance(comps)
804
- r["_score"] = _clamp01(l2_distance_to_score(effective_dist))
936
+ r["_score"] = vector_display_score(effective_dist)
805
937
 
806
938
  # Dedup by primary_type_fqn after all sorting/merging, before windowing
807
939
  merged = _dedup_by_fqn(merged, dedup_by_fqn=dedup_by_fqn)
@@ -18,11 +18,13 @@ guarded by a parity unit test.
18
18
  from __future__ import annotations
19
19
 
20
20
  import os
21
+ import weakref
21
22
  from pathlib import Path
22
23
  from typing import TYPE_CHECKING, Any
23
24
 
24
25
  from java_codebase_rag.graph.ladybug_queries import LadybugGraph
25
26
  from java_codebase_rag.search.search_scoring import (
27
+ SYMBOL_FTS_INDEX,
26
28
  _ROLE_SCORE_WEIGHTS,
27
29
  _TYPE_MATCH_BONUS_CAP,
28
30
  _TYPE_MATCH_BONUS_PER_HIT,
@@ -30,6 +32,7 @@ from java_codebase_rag.search.search_scoring import (
30
32
  _dedup_by_fqn,
31
33
  _query_tokens,
32
34
  _split_identifier,
35
+ build_fts_query,
33
36
  )
34
37
 
35
38
  if TYPE_CHECKING:
@@ -170,6 +173,88 @@ def _token_overlap(haystack_toks: set[str], needle_toks: set[str]) -> float:
170
173
  return len(needle_toks & haystack_toks) / len(needle_toks)
171
174
 
172
175
 
176
+ # BM25 candidate fetch via the LadybugDB FTS index (fork A). DB-side indexed ranking
177
+ # replaces the heuristic's bounded Python scan; the heuristic below still scores the
178
+ # fetched candidates (name/type/fqn/role) and is the fallback when the FTS index or
179
+ # extension is unavailable (older graph, offline first run).
180
+ _FTS_CANDIDATE_K = 200 # top-K BM25 candidates; re-filtered by NodeFilter before ranking
181
+ # Connections that have run LOAD EXTENSION FTS. Keyed by the connection OBJECT (WeakSet),
182
+ # NOT id() — id() is reused after GC, which would let a fresh connection skip LOAD and then
183
+ # fail at QUERY_FTS_INDEX under test batching. Entries die with the connection.
184
+ _FTS_LOADED_CONNS: "weakref.WeakSet[object]" = weakref.WeakSet()
185
+
186
+
187
+ def _ensure_fts_loaded(g: LadybugGraph) -> bool:
188
+ """LOAD EXTENSION FTS on the graph's (read-only) connection, once per connection.
189
+
190
+ Returns False if the extension can't be loaded (absent / offline) so the caller
191
+ falls back to the heuristic scan.
192
+ """
193
+ conn = g._conn # noqa: SLF001
194
+ try:
195
+ if conn in _FTS_LOADED_CONNS:
196
+ return True
197
+ except Exception: # connection not weakref-able → LOAD every call (correct, slow)
198
+ pass
199
+ try:
200
+ g._rows("LOAD EXTENSION FTS") # noqa: SLF001
201
+ try:
202
+ _FTS_LOADED_CONNS.add(conn)
203
+ except Exception:
204
+ pass
205
+ return True
206
+ except Exception:
207
+ return False
208
+
209
+
210
+ def _try_fts_candidates(
211
+ g: LadybugGraph,
212
+ query: str,
213
+ filter: NodeFilter | None,
214
+ path_contains: str | None,
215
+ ) -> dict | None:
216
+ """Fetch BM25-ranked Symbol candidates via the FTS index; re-apply NodeFilter.
217
+
218
+ Returns ``{"rows": [...], "scores": {id: bm25}}`` (rows are the same shape the
219
+ heuristic scan yields), or ``None`` when FTS is unavailable (extension won't load,
220
+ or the index isn't present on this graph) so the caller falls back.
221
+
222
+ Two-step: (1) ``QUERY_FTS_INDEX`` returns the top-K node ids by Okapi BM25 over
223
+ ``Symbol.search_text``; (2) re-MATCH those ids with the full ``_lexical_where``
224
+ predicates (role / module / path / kind≠file,package) so the filter logic stays
225
+ defined in one place. ``search_text`` is built at index time by ``build_ast_graph``
226
+ from the same ``_split_identifier`` the re-rank below uses, so index- and query-time
227
+ tokenization agree.
228
+ """
229
+ if not _ensure_fts_loaded(g):
230
+ return None
231
+ idx_rows = g._rows("CALL SHOW_INDEXES() RETURN index_name") # noqa: SLF001
232
+ names = {row.get("index_name") for row in idx_rows}
233
+ if SYMBOL_FTS_INDEX not in names:
234
+ return None
235
+ fts = g._rows( # noqa: SLF001
236
+ f"CALL QUERY_FTS_INDEX('Symbol', '{SYMBOL_FTS_INDEX}', $q, top := $k) "
237
+ "RETURN node.id AS id, score",
238
+ {"q": query, "k": _FTS_CANDIDATE_K},
239
+ )
240
+ if not fts:
241
+ return {"rows": [], "scores": {}}
242
+ scores = {row["id"]: float(row.get("score") or 0.0) for row in fts}
243
+ ids = list(scores.keys())
244
+
245
+ # Re-MATCH the K ids with the SAME predicates the heuristic pushes down, so
246
+ # NodeFilter / path / structural-kind filtering is defined exactly once.
247
+ where, params = _lexical_where(filter, path_contains=path_contains)
248
+ struct_pred = "(s.kind <> 'file' AND s.kind <> 'package')"
249
+ if not where:
250
+ where = f"WHERE s.id IN $ids AND {struct_pred}"
251
+ else:
252
+ where = where.replace("WHERE ", f"WHERE s.id IN $ids AND {struct_pred} AND ", 1)
253
+ params["ids"] = ids
254
+ rows = g._rows(f"MATCH (s:Symbol) {where} RETURN {_SYMBOL_RETURN}", params) # noqa: SLF001
255
+ return {"rows": rows, "scores": scores}
256
+
257
+
173
258
  def run_lexical_search(
174
259
  query: str,
175
260
  *,
@@ -185,6 +270,14 @@ def run_lexical_search(
185
270
  ) -> list[dict]:
186
271
  """Keyword search over Symbol nodes; returns ``run_search``-shaped row-dicts.
187
272
 
273
+ BM25-first (fork A): when the LadybugDB ``sym_fts`` index exists, candidates are
274
+ fetched DB-side via Okapi BM25 over ``Symbol.search_text`` (killing the bounded
275
+ Python scan that silently missed matches past the cap on large repos) and then
276
+ re-ranked here by the name/type/fqn/role heuristic. The query is pre-split with the
277
+ same tokenizer as ``search_text`` so pasted camelCase identifiers match. Falls back
278
+ to the heuristic scan when the FTS index or extension is unavailable (older graph,
279
+ offline first run), or when the query is degenerate / the BM25 result is empty.
280
+
188
281
  Raises ``RuntimeError`` (message contains "lexical search unavailable") if no
189
282
  symbol graph exists — the caller maps that to a clean failure envelope. Returns
190
283
  ``[]`` for ``table in ("sql", "yaml")`` (those LanceDB tables aren't built in
@@ -201,33 +294,47 @@ def run_lexical_search(
201
294
  )
202
295
  g = graph or LadybugGraph.get()
203
296
 
204
- where, params = _lexical_where(filter, path_contains=path_contains)
205
- # Always exclude structural Symbol nodes. Files and packages are :Symbol-labeled
206
- # (kind='file'/'package') but aren't searchable code declarations — without this
207
- # a token that appears in a filename (e.g. 'distribution' in
208
- # 'DistributionChunkService.java') would surface the file node as a hit.
209
- struct_pred = "(s.kind <> 'file' AND s.kind <> 'package')"
210
- where = f"WHERE {struct_pred}" if not where else where.replace("WHERE ", f"WHERE {struct_pred} AND ", 1)
211
- # Lexical ranking is done in Python (LadybugDB/kuzu has no keyword ranking without
212
- # FTS5, which is deferred), and the MATCH scan returns rows in storage order — there
213
- # is NO DB-side relevance ORDER BY. So fetch the FULL candidate pool up to the safety
214
- # cap and rank here. A pagination-derived LIMIT (4x the page) is correct on the vector
215
- # path where LanceDB returns rows pre-ranked by similarity, but on this unordered scan
216
- # it would return only the first ~N symbols in arbitrary storage order and silently
217
- # miss the best match on any non-trivial repo.
218
- params["lim"] = _CANDIDATE_LIMIT_CAP
219
- cypher = f"MATCH (s:Symbol) {where} RETURN {_SYMBOL_RETURN} LIMIT $lim"
220
- rows = g._rows(cypher, params) # noqa: SLF001 — de facto public read API (see find_v2)
221
- # If the fetch hit the safety cap, deeper matches were never ranked (the scan has no
222
- # ORDER BY — kuzu returns an arbitrary, storage-order-dependent subset). Surface it so
223
- # a user on a large repo isn't silently shown an incomplete result set; refining the
224
- # query or adding a filter narrows the pool below the cap. Raising the cap / FTS5 is
225
- # the deferred long-term fix (see the plan's "Out of scope" note).
226
- if advisories is not None and len(rows) >= _CANDIDATE_LIMIT_CAP:
227
- advisories.append(
228
- f"lexical search scanned the first {_CANDIDATE_LIMIT_CAP} matching symbols "
229
- "(repo cap); deeper matches were not ranked — refine the query or add a filter"
230
- )
297
+ # --- candidate fetch: BM25 (FTS) preferred, heuristic scan fallback ---
298
+ # FTS indexes Symbol.search_text (camelCase-split tokens); pre-split the query the
299
+ # same way (build_fts_query) so a pasted identifier like "DistributionChunkService"
300
+ # matches — LadybugDB FTS's own tokenizer does not split camelCase. An empty split
301
+ # (degenerate / stopword-only query), an unavailable FTS index, OR an empty BM25
302
+ # result all fall back to the heuristic scan, which yields role-ranked output for
303
+ # degenerate queries and covers selective filters where BM25's top-K thins to nil.
304
+ bm25_scores: dict[str, float] = {}
305
+ use_fts = False
306
+ rows: list[dict] | None = None
307
+ q_fts = build_fts_query(query)
308
+ if q_fts:
309
+ fts = _try_fts_candidates(g, q_fts, filter, path_contains)
310
+ if fts is not None and fts["rows"]:
311
+ rows = fts["rows"]
312
+ bm25_scores = fts["scores"]
313
+ use_fts = True
314
+ if rows is None:
315
+ where, params = _lexical_where(filter, path_contains=path_contains)
316
+ # Always exclude structural Symbol nodes. Files and packages are :Symbol-labeled
317
+ # (kind='file'/'package') but aren't searchable code declarations — without this
318
+ # a token that appears in a filename (e.g. 'distribution' in
319
+ # 'DistributionChunkService.java') would surface the file node as a hit.
320
+ struct_pred = "(s.kind <> 'file' AND s.kind <> 'package')"
321
+ where = f"WHERE {struct_pred}" if not where else where.replace("WHERE ", f"WHERE {struct_pred} AND ", 1)
322
+ # The heuristic scan returns rows in storage order — there is NO DB-side relevance
323
+ # ORDER BY without the FTS index — so fetch the FULL candidate pool up to the safety
324
+ # cap and rank here. A pagination-derived LIMIT (4x the page) is correct on the
325
+ # vector path where LanceDB returns rows pre-ranked, but on this unordered scan it
326
+ # would return only the first ~N symbols in arbitrary storage order and silently
327
+ # miss the best match on any non-trivial repo. The BM25 (FTS) path above has no cap.
328
+ params["lim"] = _CANDIDATE_LIMIT_CAP
329
+ cypher = f"MATCH (s:Symbol) {where} RETURN {_SYMBOL_RETURN} LIMIT $lim"
330
+ rows = g._rows(cypher, params) # noqa: SLF001 — de facto public read API (see find_v2)
331
+ # If the fetch hit the safety cap, deeper matches were never ranked. Surface it so
332
+ # a user on a large repo isn't silently shown an incomplete result set.
333
+ if advisories is not None and len(rows) >= _CANDIDATE_LIMIT_CAP:
334
+ advisories.append(
335
+ f"lexical search scanned the first {_CANDIDATE_LIMIT_CAP} matching symbols "
336
+ "(repo cap); deeper matches were not ranked — refine the query or add a filter"
337
+ )
231
338
 
232
339
  query_toks = _query_tokens(query)
233
340
  source_root = _resolve_source_root(g)
@@ -265,9 +372,10 @@ def run_lexical_search(
265
372
  text_match = text_overlap * _TEXT_MATCH_WEIGHT
266
373
 
267
374
  # A keyword search must require at least one lexical hit — role alone never
268
- # qualifies a row (it only boosts/reorders matches). Degenerate queries with
269
- # no usable tokens fall through to role-ranked listing.
270
- if query_toks and not (name_overlap or type_hits or fqn_match or text_overlap):
375
+ # qualifies a row (it only boosts/reorders matches). On the BM25 path the FTS
376
+ # index already established textual relevance, so the qualifier is heuristic-only.
377
+ # Degenerate queries with no usable tokens fall through to role-ranked listing.
378
+ if query_toks and not use_fts and not (name_overlap or type_hits or fqn_match or text_overlap):
271
379
  continue
272
380
 
273
381
  role_w = 0.0 if role_locked else _ROLE_SCORE_WEIGHTS.get(role_raw.upper(), 0.0)
@@ -291,6 +399,8 @@ def run_lexical_search(
291
399
  "lexical_relevance": round(raw, 4),
292
400
  "role_weight": role_w,
293
401
  }
402
+ if use_fts:
403
+ comps["bm25"] = round(float(bm25_scores.get(r.get("id"), 0.0)), 4)
294
404
 
295
405
  sl, el, sb, eb = r.get("start_line"), r.get("end_line"), r.get("start_byte"), r.get("end_byte")
296
406
  out.append(
@@ -12,6 +12,12 @@ Everything here is pure-Python dict/list math with no third-party deps.
12
12
  from __future__ import annotations
13
13
 
14
14
  import json
15
+ import re
16
+
17
+ # Name of the LadybugDB FTS (Okapi BM25) index over Symbol.search_text (fork A).
18
+ # Shared by the build path (build_ast_graph._ensure_symbol_fts_index) and the
19
+ # query path (search_lexical.run_lexical_search) so the two never drift.
20
+ SYMBOL_FTS_INDEX = "sym_fts"
15
21
 
16
22
  # Over-fetch multiplier for dedup: fetch 4x to absorb per-FQN chunk multiplicity
17
23
  # so that after collapsing by primary_type_fqn, a page stays full and the +1
@@ -130,6 +136,32 @@ def _split_identifier(name: str) -> list[str]:
130
136
  return [p for p in parts if p]
131
137
 
132
138
 
139
+ # Alphanumeric-word extractor for FTS query building (mirrors the index side, which
140
+ # runs the same regex over name/fqn/signature/annotations/capabilities/package fields).
141
+ _FTS_WORD_RE = re.compile(r"[A-Za-z][A-Za-z0-9]*")
142
+
143
+
144
+ def build_fts_query(text: str) -> str:
145
+ """Tokenize a search query into the ``Symbol.search_text`` token space (fork A).
146
+
147
+ ``search_text`` is indexed from ``_split_identifier`` tokens (camelCase / snake_case
148
+ split, lowercased). LadybugDB FTS's own tokenizer does NOT split camelCase, so a raw
149
+ pasted identifier like ``DistributionChunkService`` would match nothing — this
150
+ extracts alphanumeric words from the query and splits each via ``_split_identifier``
151
+ so the query lands in the index's token space. Tokens shorter than 2 chars are
152
+ dropped (mirrors the index side); duplicates collapse. Returns ``""`` for a query
153
+ with no usable tokens — the caller then falls back to the heuristic / role listing.
154
+ """
155
+ out: list[str] = []
156
+ seen: set[str] = set()
157
+ for word in _FTS_WORD_RE.findall(text or ""):
158
+ for tok in _split_identifier(word):
159
+ if len(tok) >= 2 and tok not in seen:
160
+ seen.add(tok)
161
+ out.append(tok)
162
+ return " ".join(out)
163
+
164
+
133
165
  def _symbol_bonus(r: dict, query_toks: set[str]) -> float:
134
166
  """Symbol-name overlap + action-verb bump for java chunks.
135
167
 
@@ -210,6 +242,29 @@ def l2_distance_to_score(distance: float) -> float:
210
242
  return 1.0 - distance * distance / 2.0
211
243
 
212
244
 
245
+ # Display-score denominator for the vector backend. Unit-normalized embeddings
246
+ # have L2 distance in [0, 2]; the cosine map ``l2_distance_to_score`` (1 - d²/2)
247
+ # goes NEGATIVE past √2 ≈ 1.414 and clamps to 0. Weak-but-best semantic matches
248
+ # (e.g. a lone keyword like "controller") commonly sit at d ≈ 1.5, so EVERY hit
249
+ # clamps to score=0.000 even though the ranking is correct. ``vector_display_score``
250
+ # instead normalizes the effective (bonus-adjusted) distance over the full
251
+ # unit-embedding range, so a top-ranked hit stays visibly non-zero. Role/symbol
252
+ # bonuses reduce the effective distance and so raise the displayed score,
253
+ # keeping it rank-monotonic with the distance-based sort key.
254
+ _VECTOR_DISTANCE_REF = 2.0
255
+
256
+
257
+ def vector_display_score(effective_distance: float) -> float:
258
+ """Displayed vector score in [0, 1] from the effective (bonus-adjusted) distance.
259
+
260
+ Bounded linear normalization over the unit-embedding L2 range [0, 2]: lower
261
+ distance → higher score. Unlike ``l2_distance_to_score`` (which goes
262
+ negative past √2 and clamps a correctly-ranked top hit to 0.000), this keeps
263
+ a top result visibly non-zero while staying rank-monotonic with the sort key.
264
+ """
265
+ return _clamp01(1.0 - effective_distance / _VECTOR_DISTANCE_REF)
266
+
267
+
213
268
  def _effective_distance(comps: dict[str, float]) -> float:
214
269
  """Compute the adjusted distance used for sorting.
215
270
 
@@ -231,6 +286,59 @@ def _clamp01(x: float) -> float:
231
286
  return x
232
287
 
233
288
 
289
+ # Matches a Java top-level type declaration and captures its simple name. Mirrors
290
+ # the heuristic in ``ast.chunk_heuristics._JAVA_TYPE`` but is duplicated here so
291
+ # this module stays dependency-free (importable on graph-only Intel installs).
292
+ _JAVA_TYPE_DECL_RE = re.compile(
293
+ r"\b(?:public\s+|private\s+|protected\s+|sealed\s+|non-sealed\s+|final\s+|"
294
+ r"abstract\s+|static\s+)*"
295
+ r"(?:class|interface|enum|record)\s+([A-Za-z_][A-Za-z0-9_]*)"
296
+ )
297
+
298
+
299
+ def declaration_line_number(
300
+ text: str | None, anchor_line: int | None, type_name: str | None = None
301
+ ) -> int | None:
302
+ """Absolute 1-based line of the Java type declaration within chunk ``text``.
303
+
304
+ LanceDB chunks are anchored at the chunk's first source line, which for a
305
+ file-spanning chunk is the package/import line (``anchor_line`` = 1) while
306
+ the ``class``/``interface`` declaration sits several lines down. Without
307
+ this, hits render as ``File.java:1`` even though the symbol is declared
308
+ later (F8). Returns ``anchor_line + i`` for the first matching declaration
309
+ (pinned to ``type_name`` when given, so a nested type doesn't win), or
310
+ ``anchor_line`` unchanged when no declaration is found in the chunk.
311
+
312
+ Comment-aware: Javadoc/line/block-comment lines that merely MENTION the type
313
+ name (e.g. ``* This class Bar handles...``) are skipped so the returned line
314
+ is the real declaration, not a comment above it.
315
+ """
316
+ if not text or anchor_line is None:
317
+ return anchor_line
318
+ in_block = False
319
+ for i, raw in enumerate(text.splitlines()):
320
+ # Drop a trailing ``// ...`` line comment before any matching (a ``//``
321
+ # inside a string literal is unrealistic for a declaration line).
322
+ code = raw.split("//", 1)[0]
323
+ stripped = code.strip()
324
+ if in_block:
325
+ if "*/" in stripped:
326
+ in_block = False
327
+ continue
328
+ if stripped.startswith("/*"):
329
+ # Single-line ``/* ... */`` -> skip without entering block state.
330
+ if "*/" not in stripped[2:]:
331
+ in_block = True
332
+ continue
333
+ if not stripped or stripped.startswith("*"):
334
+ # Blank or a Javadoc continuation line (`` * ...``).
335
+ continue
336
+ m = _JAVA_TYPE_DECL_RE.search(code)
337
+ if m and (not type_name or m.group(1) == type_name):
338
+ return anchor_line + i
339
+ return anchor_line
340
+
341
+
234
342
  def explain_score_components(
235
343
  comps: dict[str, float] | None,
236
344
  *,
File without changes