java-codebase-rag 0.12.0__py3-none-any.whl → 0.12.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. java_codebase_rag-0.12.2.dist-info/METADATA +35 -0
  2. java_codebase_rag-0.12.2.dist-info/RECORD +4 -0
  3. {java_codebase_rag-0.12.0.dist-info → java_codebase_rag-0.12.2.dist-info}/WHEEL +1 -1
  4. java_codebase_rag/_deprecation.py +0 -103
  5. java_codebase_rag/_fdlimit.py +0 -56
  6. java_codebase_rag/_stdio.py +0 -32
  7. java_codebase_rag/_version.py +0 -35
  8. java_codebase_rag/absence/__init__.py +0 -0
  9. java_codebase_rag/absence/absence_diagnosis.py +0 -700
  10. java_codebase_rag/absence/absence_types.py +0 -124
  11. java_codebase_rag/absence/absence_vocab.py +0 -460
  12. java_codebase_rag/analysis/__init__.py +0 -0
  13. java_codebase_rag/analysis/pr_analysis.py +0 -563
  14. java_codebase_rag/analysis/resolve_service.py +0 -740
  15. java_codebase_rag/ast/__init__.py +0 -0
  16. java_codebase_rag/ast/ast_java.py +0 -2847
  17. java_codebase_rag/ast/ast_kotlin.py +0 -1794
  18. java_codebase_rag/ast/brownfield_events.py +0 -58
  19. java_codebase_rag/ast/chunk_heuristics.py +0 -83
  20. java_codebase_rag/ast/language.py +0 -117
  21. java_codebase_rag/cli.py +0 -1215
  22. java_codebase_rag/cli_dispatch.py +0 -251
  23. java_codebase_rag/cli_format.py +0 -85
  24. java_codebase_rag/cli_progress.py +0 -94
  25. java_codebase_rag/config.py +0 -833
  26. java_codebase_rag/eval/__init__.py +0 -1
  27. java_codebase_rag/eval/ground_truth.py +0 -100
  28. java_codebase_rag/eval/metrics.py +0 -107
  29. java_codebase_rag/eval/runner.py +0 -556
  30. java_codebase_rag/graph/__init__.py +0 -0
  31. java_codebase_rag/graph/build_ast_graph.py +0 -4593
  32. java_codebase_rag/graph/graph_enrich.py +0 -1940
  33. java_codebase_rag/graph/graph_types.py +0 -224
  34. java_codebase_rag/graph/java_ontology.py +0 -465
  35. java_codebase_rag/graph/ladybug_queries.py +0 -2213
  36. java_codebase_rag/graph/path_filtering.py +0 -509
  37. java_codebase_rag/index/__init__.py +0 -0
  38. java_codebase_rag/index/java_index_flow_lancedb.py +0 -879
  39. java_codebase_rag/index/java_index_v1_common.py +0 -33
  40. java_codebase_rag/install_data/__init__.py +0 -0
  41. java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -110
  42. java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
  43. java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
  44. java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
  45. java_codebase_rag/installer.py +0 -2188
  46. java_codebase_rag/jrag.py +0 -4545
  47. java_codebase_rag/jrag_envelope.py +0 -1107
  48. java_codebase_rag/jrag_hints.py +0 -204
  49. java_codebase_rag/jrag_render.py +0 -926
  50. java_codebase_rag/lance_optimize.py +0 -264
  51. java_codebase_rag/mcp/__init__.py +0 -0
  52. java_codebase_rag/mcp/mcp_hints.py +0 -932
  53. java_codebase_rag/mcp/mcp_v2.py +0 -1916
  54. java_codebase_rag/mcp/server.py +0 -886
  55. java_codebase_rag/pipeline.py +0 -531
  56. java_codebase_rag/progress.py +0 -570
  57. java_codebase_rag/read_payloads.py +0 -781
  58. java_codebase_rag/search/__init__.py +0 -0
  59. java_codebase_rag/search/index_common.py +0 -10
  60. java_codebase_rag/search/search_lancedb.py +0 -1296
  61. java_codebase_rag/search/search_lexical.py +0 -449
  62. java_codebase_rag/search/search_scoring.py +0 -537
  63. java_codebase_rag/watch/__init__.py +0 -0
  64. java_codebase_rag/watch/client.py +0 -230
  65. java_codebase_rag/watch/daemon.py +0 -396
  66. java_codebase_rag/watch/lock.py +0 -201
  67. java_codebase_rag/watch/paths.py +0 -76
  68. java_codebase_rag/watch/protocol.py +0 -122
  69. java_codebase_rag/watch/server.py +0 -273
  70. java_codebase_rag/watch/warm.py +0 -105
  71. java_codebase_rag/watch/watcher.py +0 -394
  72. java_codebase_rag-0.12.0.dist-info/METADATA +0 -340
  73. java_codebase_rag-0.12.0.dist-info/RECORD +0 -75
  74. java_codebase_rag-0.12.0.dist-info/entry_points.txt +0 -5
  75. java_codebase_rag-0.12.0.dist-info/licenses/LICENSE +0 -21
  76. java_codebase_rag-0.12.0.dist-info/top_level.txt +0 -1
  77. /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.2.dist-info/top_level.txt +0 -0
@@ -1,1296 +0,0 @@
1
- #!/usr/bin/env python3
2
- """Semantic search over LanceDB tables built by CocoIndex (java_index_flow_lancedb)."""
3
-
4
- from __future__ import annotations
5
-
6
- import argparse
7
- import json
8
- import os
9
- import sys
10
- import tempfile
11
- import threading
12
- import warnings
13
- from collections.abc import Callable
14
- from contextlib import contextmanager
15
- from pathlib import Path
16
-
17
- import lancedb
18
- import numpy as np
19
- from sentence_transformers import SentenceTransformer
20
-
21
- from java_codebase_rag.ast.chunk_heuristics import analyze_chunk, looks_like_code_identifier
22
- from java_codebase_rag.search.index_common import SBERT_MODEL
23
- from java_codebase_rag.search import search_lexical
24
- from java_codebase_rag.config import maybe_expand_embedding_model_path, resolved_sbert_model_for_process_env
25
-
26
- # Scoring & dedup primitives live in `search_scoring` (dependency-free — no
27
- # lancedb/torch) so the lexical backend `search_lexical` can share them on
28
- # graph-only (macOS Intel) installs where this module is unimportable. Re-exported
29
- # here for backward compatibility (`from search_lancedb import _clamp01`, etc.).
30
- from java_codebase_rag.search.search_scoring import ( # noqa: F401
31
- BASELINE_2LIST_CONFIG,
32
- DEFAULT_RANK_CONFIG,
33
- DEDUP_OVERFETCH,
34
- RankConfig,
35
- build_fts_query,
36
- _ACTION_VERB_BONUS,
37
- _ACTION_VERB_PREFIXES,
38
- _HYBRID_SCORE_MAX,
39
- _IMPORT_DISTANCE_PENALTY,
40
- _IMPORT_HYBRID_SCORE_FACTOR,
41
- _ROLE_SCORE_WEIGHTS,
42
- _STOPWORDS,
43
- _SYMBOL_MATCH_BONUS_CAP,
44
- _SYMBOL_MATCH_BONUS_PER_HIT,
45
- _TYPE_MATCH_BONUS_CAP,
46
- _TYPE_MATCH_BONUS_PER_HIT,
47
- _apply_symbol_bonus,
48
- _clamp01,
49
- _dedup_by_fqn,
50
- _effective_distance,
51
- _query_tokens,
52
- _role_weight,
53
- _split_identifier,
54
- _symbol_bonus,
55
- declaration_line_number,
56
- explain_score_components,
57
- l2_distance_to_score,
58
- vector_display_score,
59
- )
60
-
61
- TABLES: dict[str, str] = {
62
- "java": "javacodeindex_java_code",
63
- "sql": "sqlschemaindex_sql_schema",
64
- "yaml": "yamlconfigindex_yaml_config",
65
- }
66
-
67
- # Optional enrichment columns on the java chunk table (absent on older indexes).
68
- JAVA_ENRICHED_COLUMNS: tuple[str, ...] = (
69
- "package",
70
- "module",
71
- "microservice",
72
- "primary_type_fqn",
73
- "primary_type_kind",
74
- "role",
75
- "annotations_on_type",
76
- "symbols",
77
- "symbol_id",
78
- "metadata",
79
- "ontology_version",
80
- "capabilities",
81
- "generated",
82
- "generated_by",
83
- )
84
-
85
- VECTOR_COLUMN = "embedding"
86
- _FTS_READY: set[tuple[str, str]] = set()
87
- _FTS_LOCK = threading.Lock()
88
- _SCHEMA_CACHE: dict[tuple[str, str], set[str]] = {}
89
- _SCHEMA_LOCK = threading.Lock()
90
-
91
-
92
- def _table_columns(uri: str, lance_table_name: str, db_obj: object | None = None) -> set[str]:
93
- key = (uri, lance_table_name)
94
- with _SCHEMA_LOCK:
95
- cached = _SCHEMA_CACHE.get(key)
96
- if cached is not None:
97
- return cached
98
- db = db_obj if db_obj is not None else lancedb.connect(uri)
99
- tbl = db.open_table(lance_table_name)
100
- cols = {f.name for f in tbl.schema}
101
- with _SCHEMA_LOCK:
102
- _SCHEMA_CACHE[key] = cols
103
- return cols
104
-
105
-
106
- def _escape_sql_str(s: str) -> str:
107
- return s.replace("'", "''")
108
-
109
-
110
- def _build_extra_predicates(
111
- *,
112
- columns: set[str],
113
- role: str | None,
114
- module: str | None,
115
- microservice: str | None,
116
- package_prefix: str | None,
117
- fqn_in: list[str] | None,
118
- role_in: list[str] | None = None,
119
- exclude_roles: list[str] | None = None,
120
- capability: str | None = None,
121
- capability_in: list[str] | None = None,
122
- generated_only: bool = False,
123
- exclude_generated: bool = False,
124
- ) -> list[str]:
125
- preds: list[str] = []
126
- if role and "role" in columns:
127
- preds.append(f"role = '{_escape_sql_str(role)}'")
128
-
129
- # When both role_in and capability_in are set, combine as OR so that
130
- # capability-only entrypoints (e.g. role=OTHER with MESSAGE_LISTENER)
131
- # are not silently excluded by the role filter.
132
- role_pred: str | None = None
133
- if role_in and "role" in columns:
134
- vals = ", ".join(f"'{_escape_sql_str(v)}'" for v in role_in)
135
- role_pred = f"role IN ({vals})"
136
-
137
- cap_in_pred: str | None = None
138
- if capability_in and "capabilities" in columns:
139
- # array_has is the preferred form in LanceDB >= 0.10 (verified against 0.30.2).
140
- parts = [
141
- f"array_has(capabilities, '{_escape_sql_str(c)}')"
142
- for c in capability_in
143
- ]
144
- cap_in_pred = "(" + " OR ".join(parts) + ")"
145
-
146
- if role_pred and cap_in_pred:
147
- preds.append(f"({role_pred} OR {cap_in_pred})")
148
- elif role_pred:
149
- preds.append(role_pred)
150
- elif cap_in_pred:
151
- preds.append(cap_in_pred)
152
-
153
- if exclude_roles and "role" in columns:
154
- vals = ", ".join(f"'{_escape_sql_str(v)}'" for v in exclude_roles)
155
- preds.append(f"(role IS NULL OR role NOT IN ({vals}))")
156
- if generated_only and "generated" in columns:
157
- preds.append("generated = true")
158
- if exclude_generated and "generated" in columns:
159
- preds.append("(generated IS NULL OR generated = false)")
160
- if module and "module" in columns:
161
- preds.append(f"module = '{_escape_sql_str(module)}'")
162
- if microservice and "microservice" in columns:
163
- preds.append(f"microservice = '{_escape_sql_str(microservice)}'")
164
- if package_prefix and "package" in columns:
165
- esc = _escape_sql_str(package_prefix)
166
- preds.append(f"(package = '{esc}' OR package LIKE '{esc}.%')")
167
- if fqn_in and "primary_type_fqn" in columns:
168
- # LanceDB/Arrow SQL supports IN; quote each.
169
- vals = ", ".join(f"'{_escape_sql_str(v)}'" for v in fqn_in)
170
- preds.append(f"primary_type_fqn IN ({vals})")
171
- if capability and "capabilities" in columns:
172
- preds.append(f"array_has(capabilities, '{_escape_sql_str(capability)}')")
173
- return preds
174
-
175
-
176
- def coerce_position_field(val: object) -> dict[str, object]:
177
- """LanceDB may return struct columns as JSON strings; normalize to a dict."""
178
- if val is None:
179
- return {}
180
- if isinstance(val, dict):
181
- return val
182
- if isinstance(val, str):
183
- try:
184
- parsed = json.loads(val)
185
- except json.JSONDecodeError:
186
- return {}
187
- return parsed if isinstance(parsed, dict) else {}
188
- return {}
189
-
190
-
191
- def _apply_chunk_hints(rows: list[dict]) -> None:
192
- for r in rows:
193
- lang = r.get("language") or ""
194
- kind = str(r.get("_kind", ""))
195
- if kind == "sql" and not lang:
196
- lang = "sql"
197
- if kind == "yaml" and not lang:
198
- lang = "yaml"
199
- h = analyze_chunk(r.get("text"), language=str(lang), kind=kind)
200
- r["_hints"] = {
201
- "primary_type_hint": h.primary_type_hint,
202
- "import_heavy": h.import_heavy,
203
- }
204
-
205
-
206
- def _vector_sort_key(r: dict) -> float:
207
- d = float(r["_distance"])
208
- comps = r.setdefault("_score_components", {})
209
- comps["distance"] = d
210
- if r.get("_hints", {}).get("import_heavy"):
211
- d += _IMPORT_DISTANCE_PENALTY
212
- comps["import_penalty"] = _IMPORT_DISTANCE_PENALTY
213
- d -= _role_weight(r)
214
- d -= float(comps.get("symbol_bonus", 0.0))
215
- return d
216
-
217
-
218
- def _hybrid_sort_key(r: dict) -> float:
219
- s = float(r.get("_score", 0.0))
220
- comps = r.setdefault("_score_components", {})
221
- comps["hybrid_rrf"] = s
222
- if r.get("_hints", {}).get("import_heavy"):
223
- s *= _IMPORT_HYBRID_SCORE_FACTOR
224
- comps["import_penalty"] = 1.0 - _IMPORT_HYBRID_SCORE_FACTOR
225
- s += _role_weight(r)
226
- s += float(comps.get("symbol_bonus", 0.0))
227
- return -s
228
-
229
-
230
- def _hybrid_post_sort_normalization(rows: list[dict]) -> None:
231
- """Set honest displayed scores for hybrid search after sorting.
232
-
233
- Reconstructs the composite score (raw_rrf * import_factor + role_weight + symbol_bonus)
234
- and normalizes by _HYBRID_SCORE_MAX to ensure rank-monotonicity.
235
-
236
- Mutates rows in-place, replacing _score with the normalized value.
237
- """
238
- for r in rows:
239
- comps = r.setdefault("_score_components", {})
240
- raw = comps.get("hybrid_rrf", 0.0)
241
- comps["rrf_raw"] = raw # preserve raw RRF for --explain. NOTE: when graph_expand + hybrid combine (Phase 2), _rrf_merge below overwrites this with graph-RRF, so --explain would show graph-RRF not hybrid-RRF.
242
- s = raw
243
- if r.get("_hints", {}).get("import_heavy"):
244
- s *= _IMPORT_HYBRID_SCORE_FACTOR
245
- s += comps.get("role_weight", 0.0) + comps.get("symbol_bonus", 0.0)
246
- r["_score"] = _clamp01(s / _HYBRID_SCORE_MAX)
247
-
248
-
249
- def _escape_like_fragment(s: str) -> str:
250
- return s.replace("'", "''")
251
-
252
-
253
- def _escape_sql_like_pattern(s: str) -> str:
254
- out: list[str] = []
255
- for c in s:
256
- if c in ("\\", "%", "_"):
257
- out.append("\\" + c)
258
- else:
259
- out.append(c)
260
- return "".join(out)
261
-
262
-
263
- def _build_path_predicate(path_substring: str) -> str:
264
- pat = _escape_sql_like_pattern(path_substring)
265
- pat = _escape_like_fragment(pat)
266
- return f"filename LIKE '%{pat}%' ESCAPE '\\'"
267
-
268
-
269
- def ensure_text_fts_index(uri: str, lance_table_name: str) -> None:
270
- key = (uri, lance_table_name)
271
- with _FTS_LOCK:
272
- if key in _FTS_READY:
273
- return
274
- db = lancedb.connect(uri)
275
- tbl = db.open_table(lance_table_name)
276
- try:
277
- tbl.create_fts_index("text", replace=False)
278
- except Exception as e:
279
- low = str(e).lower()
280
- if any(
281
- w in low
282
- for w in ("exist", "duplicate", "already", "same name")
283
- ):
284
- pass
285
- else:
286
- raise
287
- _FTS_READY.add(key)
288
-
289
-
290
- def _query_vector(model: SentenceTransformer, text: str) -> np.ndarray:
291
- v = model.encode(
292
- text,
293
- convert_to_numpy=True,
294
- normalize_embeddings=True,
295
- show_progress_bar=False,
296
- )
297
- return np.asarray(v, dtype=np.float32)
298
-
299
-
300
- def _combine_predicates(parts: list[str | None]) -> str | None:
301
- clean = [p for p in parts if p]
302
- if not clean:
303
- return None
304
- if len(clean) == 1:
305
- return clean[0]
306
- return " AND ".join(f"({p})" for p in clean)
307
-
308
-
309
- # LanceDB (0.30.x) emits two Rust `tracing` WARN lines per hybrid query to stderr
310
- # — "specified output columns but did not include `_score`/`_distance` ... Call
311
- # `disable_scoring_autoprojection`". They are noise on the agent's stderr, not
312
- # Python warnings (so `warnings.filterwarnings` can't catch them), and the fluent
313
- # query builder exposes no `disable_scoring_autoprojection()` (the lower-level
314
- # `to_lance().scanner(...)` path needs `pylance`, which isn't installed on the
315
- # PEP 508 graph-only profile). We match them by stable substring so anything that
316
- # is a REAL error still reaches stderr.
317
- _LANCE_AUTOPROJ_MARKERS: tuple[str, ...] = (
318
- "disable_scoring_autoprojection",
319
- "did not include `_distance`",
320
- "did not include `_score`",
321
- )
322
-
323
- # The fd-2 redirect below mutates the PROCESS-GLOBAL fd 2 (and
324
- # ``warnings.catch_warnings`` mutates global warning state). The MCP server
325
- # dispatches every tool call through ``asyncio.to_thread`` on a thread pool
326
- # (server.py), so two concurrent hybrid/auto-hybrid searches would race on the
327
- # dup2 bookkeeping — corrupting the saved fd and crashing the whole server with
328
- # ``Bad file descriptor``. Serialize the redirect so only one thread mutates fd
329
- # 2 / warning state at a time. Concurrent hybrid queries therefore serialize
330
- # their ``to_list()`` (correctness over throughput); a Rust-tracing-level
331
- # suppression would remove the fd hijack entirely (follow-up).
332
- _LANCE_WARN_REDIRECT_LOCK = threading.Lock()
333
-
334
-
335
- def _is_autoproj_noise(line: str) -> bool:
336
- """True for a LanceDB autoprojection-deprecation line (to drop).
337
-
338
- Preserves genuine errors/tracebacks even if they happen to reference the API
339
- name — only the bare deprecation log lines (no Error/Traceback/Exception) are
340
- treated as noise.
341
- """
342
- if not any(marker in line for marker in _LANCE_AUTOPROJ_MARKERS):
343
- return False
344
- return not any(seg in line for seg in ("Traceback", "Error:", "error:", "Exception"))
345
-
346
-
347
- @contextmanager
348
- def _silence_lance_autoproj_warnings():
349
- """Swallow LanceDB's `_score`/`_distance` autoprojection deprecation warnings.
350
-
351
- Redirects fd 2 to a temp buffer for the duration of the wrapped call, drops
352
- only the autoprojection deprecation lines, and re-emits everything else to
353
- the real stderr so genuine errors stay visible. No-op if the caller opted
354
- back in via ``JAVA_CODEBASE_RAG_KEEP_LANCE_WARNINGS`` (debugging).
355
-
356
- Thread-safety: the redirect is serialized under ``_LANCE_WARN_REDIRECT_LOCK``
357
- because it mutates process-global fd 2 and warning state — the MCP server
358
- runs tool calls concurrently on a thread pool.
359
- """
360
- if os.environ.get("JAVA_CODEBASE_RAG_KEEP_LANCE_WARNINGS"):
361
- yield
362
- return
363
- # Also catch the (unlikely) Python-warning form defensively.
364
- with _LANCE_WARN_REDIRECT_LOCK, warnings.catch_warnings():
365
- warnings.filterwarnings(
366
- "ignore",
367
- message=r".*(disable_scoring_autoprojection|did not include `(_distance|_score)`).*",
368
- )
369
- with tempfile.TemporaryFile(mode="w+", encoding="utf-8", errors="replace") as captured:
370
- saved = os.dup(2)
371
- try:
372
- os.dup2(captured.fileno(), 2)
373
- yield
374
- finally:
375
- # Restore fd 2 FIRST so the re-emit below reaches real stderr.
376
- os.dup2(saved, 2)
377
- os.close(saved)
378
- captured.seek(0)
379
- kept = "".join(line for line in captured if not _is_autoproj_noise(line))
380
- if kept:
381
- sys.stderr.write(kept)
382
- sys.stderr.flush()
383
-
384
-
385
- def _simple_type_name(fqn: str | None) -> str | None:
386
- """``com.foo.Bar`` -> ``Bar``; None/empty -> None."""
387
- if not fqn:
388
- return None
389
- return str(fqn).rsplit(".", 1)[-1] or None
390
-
391
-
392
- def _refine_java_start_lines(rows: list[dict]) -> None:
393
- """Point each java row's ``start.line`` at the type declaration, not the chunk anchor.
394
-
395
- LanceDB chunks are anchored at the chunk's first source line — for a
396
- file-spanning chunk that's the package/import line (``start.line`` = 1)
397
- while the ``class``/``interface`` declaration sits several lines down. The
398
- chunk anchor is a poor display line for a symbol hit (renders as
399
- ``File.java:1``); derive the real declaration line from the chunk text
400
- (pinned to the primary type) so a hit shows ``File.java:<decl>`` instead
401
- (F8). Method-only chunks whose range doesn't include a type declaration
402
- keep their chunk anchor unchanged.
403
- """
404
- for r in rows:
405
- if str(r.get("_kind", "")) != "java":
406
- continue
407
- start = r.get("start")
408
- if not isinstance(start, dict):
409
- continue
410
- anchor = start.get("line")
411
- if anchor is None:
412
- continue
413
- hints = r.get("_hints") or {}
414
- type_name = hints.get("primary_type_hint") or _simple_type_name(r.get("primary_type_fqn"))
415
- decl = declaration_line_number(r.get("text"), int(anchor), type_name)
416
- if decl is not None:
417
- start["line"] = decl
418
-
419
-
420
- def _search_one_table(
421
- table_name: str,
422
- *,
423
- uri: str,
424
- db: object,
425
- query_vec: np.ndarray,
426
- limit: int,
427
- path_predicate: str | None,
428
- kind: str,
429
- hybrid: bool,
430
- fts_text: str | None,
431
- extra_predicates: list[str] | None = None,
432
- ) -> list[dict]:
433
- tbl = db.open_table(table_name)
434
- has_lang = kind == "java"
435
- table_cols = _table_columns(uri, table_name, db)
436
- enriched_cols = table_cols if has_lang else set()
437
- # `range_start` / `range_end` are needed downstream by `_attach_neighbor_context`
438
- # to locate the chunk inside its file; select them whenever the schema has them.
439
- base_cols = ["filename", "text", "start", "end"]
440
- for col in ("range_start", "range_end"):
441
- if col in table_cols:
442
- base_cols.append(col)
443
- java_extra = [c for c in JAVA_ENRICHED_COLUMNS if c in enriched_cols] if has_lang else []
444
- combined_pred = _combine_predicates([path_predicate, *(extra_predicates or [])])
445
-
446
- if hybrid:
447
- ensure_text_fts_index(uri, table_name)
448
- text_for_fts = fts_text if fts_text is not None else ""
449
- columns = (
450
- [*base_cols, "language", *java_extra]
451
- if has_lang
452
- else [*base_cols]
453
- )
454
- q = (
455
- tbl.search(
456
- query_type="hybrid",
457
- vector_column_name=VECTOR_COLUMN,
458
- )
459
- .vector(query_vec)
460
- .text(text_for_fts)
461
- .select(columns)
462
- .limit(limit)
463
- )
464
- if combined_pred:
465
- q = q.where(combined_pred, prefilter=True)
466
- # Hybrid selects explicit output columns without `_score`/`_distance`, so
467
- # LanceDB (0.30.x) emits two Rust autoprojection deprecation WARNs to
468
- # stderr per query. Silence just those lines; real errors still surface.
469
- with _silence_lance_autoproj_warnings():
470
- rows = q.to_list()
471
- for r in rows:
472
- r["_kind"] = kind
473
- rs = r.pop("_relevance_score", None)
474
- r["_hybrid"] = True
475
- if rs is not None:
476
- r["_score"] = float(rs)
477
- r["start"] = coerce_position_field(r.get("start"))
478
- r["end"] = coerce_position_field(r.get("end"))
479
- return rows
480
-
481
- columns = (
482
- [*base_cols, "language", *java_extra, "_distance"]
483
- if has_lang
484
- else [*base_cols, "_distance"]
485
- )
486
- q = tbl.search(query_vec, vector_column_name=VECTOR_COLUMN).select(
487
- columns
488
- ).limit(limit)
489
- if combined_pred:
490
- q = q.where(combined_pred, prefilter=True)
491
- rows = q.to_list()
492
- for r in rows:
493
- r["_kind"] = kind
494
- r["_hybrid"] = False
495
- # Populate `_score` from `_distance` so the SearchHit.score reflects
496
- # relevance. The hybrid branch sets `_score` from `_relevance_score`
497
- # above; without this, non-hybrid (default) search left `_score` unset
498
- # and mcp_v2._row_to_search_hit fell back to 0.0 for EVERY hit —
499
- # ranking still worked (the sort key uses `_distance` directly) but the
500
- # exposed score was always 0.0, making results look unranked.
501
- d = r.get("_distance")
502
- if d is not None:
503
- # Use the same non-clamping map as the display sites so graph-expand
504
- # rows (which run_search does NOT overwrite) never carry the old
505
- # 1-d²/2 value that collapses to 0 past √2. (The main single/multi
506
- # paths overwrite this with the bonus-adjusted effective distance.)
507
- r["_score"] = vector_display_score(float(d))
508
- r["start"] = coerce_position_field(r.get("start"))
509
- r["end"] = coerce_position_field(r.get("end"))
510
- return rows
511
-
512
-
513
- def _debug_ctx(msg: str) -> None:
514
- """Emit context-expansion diagnostics when JAVA_CODEBASE_RAG_DEBUG_CONTEXT is set.
515
-
516
- Writes to stderr so it doesn't pollute MCP stdout. Cheap no-op otherwise.
517
- """
518
- if os.environ.get("JAVA_CODEBASE_RAG_DEBUG_CONTEXT"):
519
- print(f"[context_neighbors] {msg}", file=sys.stderr)
520
-
521
-
522
- def _attach_neighbor_context(
523
- rows: list[dict], *, db: object, neighbors: int, uri: str | None = None,
524
- ) -> None:
525
- """Populate `_context_before` / `_context_after` with adjacent Java chunk text.
526
-
527
- Strategy (in order):
528
- 1. Schema-aware scan of the java table, selecting only columns that exist
529
- (`filename` + `text` always; `range_start`/`range_end` when present).
530
- 2. Sort the per-file bucket by `range_start` if available; otherwise keep
531
- the table's natural order (good enough because chunks are produced in
532
- file order by CocoIndex).
533
- 3. Locate each row's index via (a) range tuple match, (b) exact text match
534
- as fallback. Missing both -> log and skip.
535
- 4. Any exception is logged (behind env flag) and the field stays empty; we
536
- never break search because of context expansion.
537
- """
538
- if neighbors <= 0:
539
- return
540
- java_rows = [r for r in rows if str(r.get("_kind", "")) == "java"]
541
- if not java_rows:
542
- _debug_ctx("no java rows in window; nothing to expand")
543
- return
544
- filenames = {str(r.get("filename", "")) for r in java_rows if r.get("filename")}
545
- if not filenames:
546
- _debug_ctx("java rows had no filename field; skipping")
547
- return
548
-
549
- java_table = TABLES["java"]
550
- try:
551
- tbl = db.open_table(java_table)
552
- except Exception as exc:
553
- _debug_ctx(f"open_table({java_table}) failed: {exc!r}")
554
- return
555
-
556
- # Discover which positional columns the index actually carries. Older
557
- # indexes may predate `range_start`/`range_end`; newer ones always have
558
- # them. Asking for a missing column makes the whole scan fail.
559
- try:
560
- schema_cols = _table_columns(uri, java_table, db) if uri else {f.name for f in tbl.schema}
561
- except Exception as exc:
562
- _debug_ctx(f"schema lookup failed: {exc!r}")
563
- schema_cols = set()
564
-
565
- has_range = {"range_start", "range_end"}.issubset(schema_cols)
566
- scan_cols = ["filename", "text"]
567
- if has_range:
568
- scan_cols.extend(("range_start", "range_end"))
569
-
570
- try:
571
- in_list = ", ".join(f"'{_escape_sql_str(f)}'" for f in filenames)
572
- scanner = tbl.to_lance().scanner(
573
- filter=f"filename IN ({in_list})",
574
- columns=scan_cols,
575
- )
576
- all_chunks = scanner.to_table().to_pylist()
577
- except Exception as exc:
578
- _debug_ctx(f"bucket scan failed (cols={scan_cols}): {exc!r}")
579
- return
580
-
581
- if not all_chunks:
582
- _debug_ctx(f"bucket scan returned 0 chunks for {len(filenames)} filenames")
583
- return
584
-
585
- by_file: dict[str, list[dict]] = {}
586
- for ch in all_chunks:
587
- by_file.setdefault(str(ch.get("filename", "")), []).append(ch)
588
- if has_range:
589
- for lst in by_file.values():
590
- lst.sort(
591
- key=lambda c: (int(c.get("range_start") or 0), int(c.get("range_end") or 0))
592
- )
593
-
594
- attached = 0
595
- for r in java_rows:
596
- fn = str(r.get("filename", ""))
597
- bucket = by_file.get(fn, [])
598
- if not bucket:
599
- _debug_ctx(f"no bucket for filename={fn!r}")
600
- continue
601
-
602
- idx: int | None = None
603
- if has_range:
604
- start = int(r.get("range_start") or 0)
605
- end = int(r.get("range_end") or 0)
606
- if start or end:
607
- idx = next(
608
- (
609
- i for i, c in enumerate(bucket)
610
- if int(c.get("range_start") or -1) == start
611
- and int(c.get("range_end") or -1) == end
612
- ),
613
- None,
614
- )
615
-
616
- if idx is None:
617
- r_text = str(r.get("text") or "")
618
- if r_text:
619
- idx = next(
620
- (i for i, c in enumerate(bucket) if str(c.get("text") or "") == r_text),
621
- None,
622
- )
623
-
624
- if idx is None:
625
- _debug_ctx(
626
- f"could not locate chunk in bucket (file={fn!r}, "
627
- f"has_range={has_range}, bucket_size={len(bucket)})"
628
- )
629
- continue
630
-
631
- before_parts = [str(c.get("text") or "") for c in bucket[max(0, idx - neighbors):idx]]
632
- after_parts = [str(c.get("text") or "") for c in bucket[idx + 1 : idx + 1 + neighbors]]
633
- r["_context_before"] = "\n".join(before_parts)
634
- r["_context_after"] = "\n".join(after_parts)
635
- attached += 1
636
-
637
- _debug_ctx(f"attached context to {attached}/{len(java_rows)} java rows")
638
-
639
-
640
- def _bm25_candidate_rows(
641
- *,
642
- g: object,
643
- query: str,
644
- uri: str,
645
- db: object,
646
- extra_predicates: list[str],
647
- columns: set[str],
648
- limit: int = 100,
649
- ) -> list[dict]:
650
- """Fetch BM25-ranked Symbol candidates from the FTS index and resolve them to
651
- chunk rows in BM25 rank order. Returns ``[]`` on any failure (silent degradation).
652
-
653
- Pipeline:
654
- 1. ``search_lexical.fetch_fts_candidates(g, query)`` → BM25-ranked Symbols +
655
- a ``{symbol_node_id: bm25_score}`` map. ``None`` / empty → return ``[]``.
656
- 2. Map each Symbol fqn to its enclosing TYPE fqn (``primary_type_fqn`` has no
657
- ``#``; a member ``Type#method`` maps to ``Type``). Dedupe by type fqn,
658
- keeping the MAX BM25 score among same-type symbols.
659
- 3. Order type fqns by BM25 desc (fqn asc tiebreak — deterministic).
660
- 4. Fetch chunk rows from LanceDB with a FILTER-ONLY query (no vector ranking,
661
- so BM25 order is preserved). Predicates = caller's ``extra_predicates`` +
662
- the ``primary_type_fqn IN (...)`` built from the ordered types — preserving
663
- filter parity with the vector path.
664
- 5. Group fetched chunks by ``primary_type_fqn``; emit in BM25 rank order,
665
- each chunk carrying ``_score_components["bm25"]``.
666
- 6. Apply ``_apply_chunk_hints`` + ``_refine_java_start_lines`` for consistency
667
- with graph_rows handling.
668
-
669
- Any exception (FTS or LanceDB) → ``_debug_ctx`` log + return ``[]`` (silent
670
- degradation; the vector path is unaffected).
671
- """
672
- # 1. BM25 candidate fetch via the FTS index.
673
- # Pre-split the query with the same tokenizer the ``sym_fts`` index uses
674
- # (``search_text`` stores ``_split_identifier`` tokens). LadybugDB FTS's own
675
- # tokenizer does NOT split camelCase, so a raw ``DistributionChunkService``
676
- # would match nothing — ``build_fts_query`` mirrors what the lexical backend
677
- # does at search_lexical.py (run_lexical_search), keeping index/query token
678
- # spaces aligned. An empty split (degenerate / stopword-only query) → no FTS
679
- # candidates → degrade silently to the vector path.
680
- fts_query = build_fts_query(query)
681
- if not fts_query or not fts_query.strip():
682
- return []
683
- try:
684
- fts = search_lexical.fetch_fts_candidates(g, fts_query, filter=None, path_contains=None)
685
- except Exception as exc: # noqa: BLE001 — silent degradation
686
- _debug_ctx(f"bm25 FTS fetch raised: {exc!r}")
687
- return []
688
- if not fts or not fts.get("rows"):
689
- return []
690
- sym_rows = fts["rows"]
691
- scores = fts.get("scores") or {}
692
-
693
- # 2. Map symbol fqns → enclosing type fqns; keep MAX bm25 per type.
694
- type_fqn_to_bm25: dict[str, float] = {}
695
- for r in sym_rows:
696
- fqn = r.get("fqn")
697
- if not fqn:
698
- continue
699
- type_fqn = search_lexical.enclosing_type_fqn(str(fqn))
700
- if not type_fqn:
701
- continue
702
- score = float(scores.get(r.get("id"), 0.0))
703
- prev = type_fqn_to_bm25.get(type_fqn)
704
- if prev is None or score > prev:
705
- type_fqn_to_bm25[type_fqn] = score
706
- if not type_fqn_to_bm25:
707
- return []
708
-
709
- # 3. Deterministic ordering: BM25 desc, fqn asc.
710
- ordered_types = sorted(
711
- type_fqn_to_bm25.keys(),
712
- key=lambda f: (-type_fqn_to_bm25[f], f),
713
- )
714
-
715
- # 4. Filter-only chunk fetch (NO vector ranking → BM25 order preserved). The
716
- # ``primary_type_fqn IN (...)`` predicate must be buildable; if the index is so
717
- # old that the column is absent, we can't restrict the fetch and degrade to [].
718
- if "primary_type_fqn" not in columns:
719
- _debug_ctx("bm25 fetch skipped: primary_type_fqn column absent from schema")
720
- return []
721
- preds = list(extra_predicates) + _build_extra_predicates(
722
- columns=columns,
723
- role=None, module=None, microservice=None,
724
- package_prefix=None, fqn_in=ordered_types,
725
- )
726
- combined_pred = _combine_predicates(preds)
727
- base_cols = ["filename", "text", "start", "end"]
728
- for col in ("range_start", "range_end"):
729
- if col in columns:
730
- base_cols.append(col)
731
- java_extra = [c for c in JAVA_ENRICHED_COLUMNS if c in columns]
732
- select_cols = [*base_cols, "language", *java_extra]
733
-
734
- try:
735
- tbl = db.open_table(TABLES["java"])
736
- # LanceDB 0.34 filter-only path: search() with no vector arg issues a
737
- # non-vector scan; .where/.select/.limit/.to_list returns rows in table
738
- # order without re-ranking by similarity. (tbl.query() is NOT available in
739
- # 0.34; to_lance().scanner() requires pylance, which isn't installed on the
740
- # PEP 508 graph-only profile — search() with no vector is the supported API.)
741
- q = tbl.search().select(select_cols).limit(
742
- max(limit, len(ordered_types) * 4)
743
- )
744
- if combined_pred:
745
- q = q.where(combined_pred, prefilter=True)
746
- with _silence_lance_autoproj_warnings():
747
- fetched = q.to_list()
748
- except Exception as exc: # noqa: BLE001 — silent degradation
749
- _debug_ctx(f"bm25 chunk fetch failed: {exc!r}")
750
- return []
751
-
752
- # 5. Group by primary_type_fqn; emit in BM25 rank order.
753
- by_type: dict[str, list[dict]] = {}
754
- for ch in fetched:
755
- tf = ch.get("primary_type_fqn")
756
- if tf is None:
757
- continue
758
- by_type.setdefault(str(tf), []).append(ch)
759
-
760
- out: list[dict] = []
761
- for type_fqn in ordered_types:
762
- chunks = by_type.get(type_fqn)
763
- if not chunks:
764
- continue # filtered out by extra_predicates / absent from index
765
- bm25_val = round(float(type_fqn_to_bm25[type_fqn]), 4)
766
- for ch in chunks:
767
- ch["_kind"] = "java"
768
- ch["_hybrid"] = False
769
- ch.setdefault("_score_components", {})["bm25"] = bm25_val
770
- ch["start"] = coerce_position_field(ch.get("start"))
771
- ch["end"] = coerce_position_field(ch.get("end"))
772
- out.append(ch)
773
-
774
- # 6. Consistency with graph_rows handling.
775
- _apply_chunk_hints(out)
776
- _refine_java_start_lines(out)
777
- return out
778
-
779
-
780
- def _graph_expand_merge(
781
- vector_rows: list[dict],
782
- *,
783
- query: str,
784
- query_vec: np.ndarray,
785
- db: object,
786
- uri: str,
787
- limit: int,
788
- extra_predicates: list[str],
789
- expand_depth: int,
790
- ladybug_path: str | None,
791
- rank_config: RankConfig = DEFAULT_RANK_CONFIG,
792
- ) -> list[dict]:
793
- """Expand vector top-k through the graph and/or fuse BM25, then RRF-merge.
794
-
795
- Which lists contribute is controlled by ``rank_config.lists``:
796
- - ``"vector"`` — always present (the backbone; validated by RankConfig).
797
- - ``"graph"`` — graph expand + fetch (skipped entirely when absent).
798
- - ``"bm25"`` — LadybugDB FTS candidate fetch fused as a third list.
799
-
800
- Silent degradation: any failure in the graph or BM25 path drops just that list;
801
- the vector list is never lost. Returns ``vector_rows`` unchanged when no
802
- auxiliary list yields rows.
803
- """
804
- want_graph = "graph" in rank_config.lists
805
- want_bm25 = "bm25" in rank_config.lists
806
- if not want_graph and not want_bm25:
807
- return vector_rows
808
-
809
- # Lazy import so the module works without ladybug installed when graph_expand=False.
810
- try:
811
- from java_codebase_rag.graph.ladybug_queries import LadybugGraph
812
- except Exception:
813
- return vector_rows
814
-
815
- if not LadybugGraph.exists(ladybug_path):
816
- return vector_rows
817
-
818
- java_cols = _table_columns(uri, TABLES["java"], db)
819
-
820
- # --- graph list ---
821
- graph_rows: list[dict] = []
822
- expand_weight_by_fqn: dict[str, float] = {}
823
- if want_graph:
824
- seed_fqns = sorted({r.get("primary_type_fqn") for r in vector_rows if r.get("primary_type_fqn")})
825
- neighbor_fqns: list[str] = []
826
- if seed_fqns:
827
- try:
828
- graph_obj = LadybugGraph.get(ladybug_path)
829
- structural = graph_obj.expand_fqns(seed_fqns, depth=expand_depth)
830
- method_pairs = graph_obj.expand_methods(
831
- seed_fqns, depth=expand_depth, exclude_external=True,
832
- )
833
- for f in structural:
834
- if f:
835
- expand_weight_by_fqn[f] = max(expand_weight_by_fqn.get(f, 0.0), 1.0)
836
- for f, conf in method_pairs:
837
- if f:
838
- expand_weight_by_fqn[f] = max(expand_weight_by_fqn.get(f, 0.0), conf)
839
- neighbor_fqns = list(dict.fromkeys(
840
- list(structural) + [f for f, _ in method_pairs],
841
- ))
842
- except Exception:
843
- neighbor_fqns = []
844
-
845
- novel = [fqn for fqn in neighbor_fqns if fqn and fqn not in set(seed_fqns)]
846
- if novel:
847
- extra = list(extra_predicates)
848
- extra.extend(_build_extra_predicates(
849
- columns=java_cols,
850
- role=None, module=None, microservice=None,
851
- package_prefix=None, fqn_in=novel,
852
- ))
853
- try:
854
- graph_rows = _search_one_table(
855
- TABLES["java"],
856
- uri=uri, db=db, query_vec=query_vec,
857
- limit=max(limit, 20),
858
- path_predicate=None, kind="java",
859
- hybrid=False, fts_text=None,
860
- extra_predicates=extra,
861
- )
862
- except Exception:
863
- graph_rows = []
864
- _apply_chunk_hints(graph_rows)
865
- _refine_java_start_lines(graph_rows)
866
- graph_rows.sort(key=_vector_sort_key)
867
- for r in graph_rows:
868
- r["_graph_expanded"] = True
869
- r["_graph_expand_weight"] = expand_weight_by_fqn.get(
870
- r.get("primary_type_fqn"), 1.0,
871
- )
872
-
873
- # --- bm25 list ---
874
- bm25_rows: list[dict] = []
875
- if want_bm25:
876
- try:
877
- graph_obj = LadybugGraph.get(ladybug_path)
878
- except Exception:
879
- graph_obj = None
880
- if graph_obj is not None:
881
- bm25_rows = _bm25_candidate_rows(
882
- g=graph_obj,
883
- query=query,
884
- uri=uri,
885
- db=db,
886
- extra_predicates=extra_predicates,
887
- columns=java_cols,
888
- limit=limit,
889
- )
890
-
891
- # --- RRF fusion (only lists that yielded rows beyond vector) ---
892
- lists: list[list[dict]] = [vector_rows]
893
- row_weights: list[Callable[[dict], float] | None] = [None]
894
- if want_graph and graph_rows:
895
- lists.append(graph_rows)
896
- row_weights.append(lambda row: float(row.get("_graph_expand_weight", 1.0)))
897
- if want_bm25 and bm25_rows:
898
- lists.append(bm25_rows)
899
- row_weights.append(None)
900
-
901
- if len(lists) == 1:
902
- return vector_rows
903
-
904
- return _rrf_merge(
905
- lists,
906
- k=rank_config.rrf_k,
907
- row_weight_for_list_index=row_weights,
908
- )
909
-
910
-
911
- def _rrf_merge(
912
- lists: list[list[dict]],
913
- *,
914
- k: int = 60,
915
- row_weight_for_list_index: list[Callable[[dict], float] | None] | None = None,
916
- ) -> list[dict]:
917
- """Reciprocal-rank-fuse several ranked lists of chunk rows.
918
-
919
- Rows are deduplicated by (filename, range_start, range_end). The merged
920
- rows get a `_rrf_score` field so callers can inspect or re-sort.
921
-
922
- When ``row_weight_for_list_index`` is set, its length must match ``lists``;
923
- a non-None entry is a callable ``row -> weight`` multiplied into that list's
924
- rank contribution (``None`` means weight ``1.0`` for every row).
925
- """
926
- pool: dict[tuple, dict] = {}
927
- for li, ranked in enumerate(lists):
928
- wfn: Callable[[dict], float] | None = None
929
- if row_weight_for_list_index is not None and li < len(row_weight_for_list_index):
930
- wfn = row_weight_for_list_index[li]
931
- for rank, row in enumerate(ranked):
932
- key = (row.get("filename"), row.get("range_start"), row.get("range_end"))
933
- existing = pool.get(key)
934
- weight = 1.0 if wfn is None else float(wfn(row))
935
- contribution = weight * (1.0 / (k + rank + 1))
936
- if existing is None:
937
- row["_rrf_score"] = contribution
938
- pool[key] = row
939
- else:
940
- existing["_rrf_score"] = float(existing.get("_rrf_score", 0.0)) + contribution
941
- merged = list(pool.values())
942
- merged.sort(key=lambda r: -float(r.get("_rrf_score", 0.0)))
943
- # Normalize displayed _rrf_score to [0,1] by theoretical max
944
- # RRF max = Σ weight·1/(k+rank+1); theoretical max when all rows are rank 0
945
- # with weight 1.0 = num_lists / (k + 1)
946
- num_lists = len(lists)
947
- max_rrf = num_lists / (k + 1)
948
- for r in merged:
949
- raw_score = float(r.get("_rrf_score", 0.0))
950
- comps = r.setdefault("_score_components", {})
951
- comps["rrf_raw"] = raw_score
952
- r["_rrf_score"] = _clamp01(raw_score / max_rrf)
953
- return merged
954
-
955
-
956
- def run_search(
957
- query: str,
958
- *,
959
- uri: str,
960
- table_keys: list[str],
961
- limit: int,
962
- path_substring: str | None,
963
- model_name: str,
964
- device: str | None,
965
- offset: int = 0,
966
- model: SentenceTransformer | None = None,
967
- hybrid: bool = False,
968
- fts_text: str | None = None,
969
- auto_hybrid: bool = False,
970
- role: str | None = None,
971
- module: str | None = None,
972
- microservice: str | None = None,
973
- package_prefix: str | None = None,
974
- graph_expand: bool = False,
975
- expand_depth: int = 1,
976
- ladybug_path: str | None = None,
977
- context_neighbors: int = 0,
978
- role_in: list[str] | None = None,
979
- exclude_roles: list[str] | None = None,
980
- capability: str | None = None,
981
- capability_in: list[str] | None = None,
982
- generated_only: bool = False,
983
- exclude_generated: bool = False,
984
- dedup_by_fqn: bool = False,
985
- rank_config: RankConfig = DEFAULT_RANK_CONFIG,
986
- ) -> list[dict]:
987
- effective_hybrid = hybrid
988
- effective_fts = fts_text
989
- if (
990
- auto_hybrid
991
- and not hybrid
992
- and len(table_keys) == 1
993
- and looks_like_code_identifier(query)
994
- ):
995
- effective_hybrid = True
996
- if effective_fts is None:
997
- effective_fts = query.strip()
998
-
999
- if effective_hybrid and len(table_keys) != 1:
1000
- raise ValueError(
1001
- "hybrid search requires exactly one table; "
1002
- "use table java, sql, or yaml (not all)."
1003
- )
1004
-
1005
- path_predicate = (
1006
- _build_path_predicate(path_substring) if path_substring else None
1007
- )
1008
-
1009
- if model is None:
1010
- model = SentenceTransformer(
1011
- model_name,
1012
- device=device,
1013
- trust_remote_code=True,
1014
- )
1015
- query_vec = _query_vector(model, query)
1016
- fts_for_hybrid = effective_fts if effective_fts is not None else query
1017
-
1018
- db = lancedb.connect(uri)
1019
- if dedup_by_fqn:
1020
- # Over-fetch to absorb per-FQN chunk multiplicity: fetch 4x so that
1021
- # after collapsing, the page stays full and the +1 truncation sentinel survives.
1022
- # The 4× factor assumes typical per-FQN chunk multiplicity; a single type with
1023
- # many high-ranking chunks (e.g. generated/God classes) could starve the page or
1024
- # make the +1 truncation sentinel unreliable; Phase 1 may revisit adaptive over-fetch (plan risk #1).
1025
- need = max((limit + offset) * DEDUP_OVERFETCH, limit + offset + 1)
1026
- else:
1027
- # Non-dedup path: exact fetch as before
1028
- need = max(limit + offset, 1)
1029
-
1030
- extra_java = _build_extra_predicates(
1031
- columns=_table_columns(uri, TABLES["java"], db),
1032
- role=role, module=module, microservice=microservice,
1033
- package_prefix=package_prefix, fqn_in=None,
1034
- role_in=role_in, exclude_roles=exclude_roles,
1035
- capability=capability, capability_in=capability_in,
1036
- generated_only=generated_only, exclude_generated=exclude_generated,
1037
- ) if "java" in table_keys else []
1038
-
1039
- skip_role_weight = bool(role or role_in or exclude_roles)
1040
- query_toks = _query_tokens(query)
1041
-
1042
- if len(table_keys) == 1:
1043
- key = table_keys[0]
1044
- preds = extra_java if key == "java" else []
1045
- rows = _search_one_table(
1046
- TABLES[key],
1047
- uri=uri,
1048
- db=db,
1049
- query_vec=query_vec,
1050
- limit=need,
1051
- path_predicate=path_predicate,
1052
- kind=key,
1053
- hybrid=effective_hybrid,
1054
- fts_text=fts_for_hybrid,
1055
- extra_predicates=preds,
1056
- )
1057
- _apply_chunk_hints(rows)
1058
- # Anchor each java row's start.line on the type declaration instead of
1059
- # the chunk's first source line (often the package/import line = 1).
1060
- _refine_java_start_lines(rows)
1061
- if skip_role_weight:
1062
- for r in rows:
1063
- r["_skip_role_weight"] = True
1064
- _apply_symbol_bonus(rows, query_toks)
1065
- if effective_hybrid:
1066
- rows.sort(key=_hybrid_sort_key)
1067
- # Hybrid: set honest displayed score from composite sort metric, clamped to [0,1]
1068
- _hybrid_post_sort_normalization(rows)
1069
- else:
1070
- rows.sort(key=_vector_sort_key)
1071
- # Vector: displayed score from the effective (bonus-adjusted) distance,
1072
- # normalized over the unit-embedding range so a correctly-ranked top
1073
- # hit never collapses to 0.000 (the cosine map 1 - d²/2 clamps to 0
1074
- # past √2; weak-but-best matches commonly sit at d ≈ 1.5).
1075
- for r in rows:
1076
- comps = r.setdefault("_score_components", {})
1077
- effective_dist = _effective_distance(comps)
1078
- r["_score"] = vector_display_score(effective_dist)
1079
-
1080
- if graph_expand and key == "java" and expand_depth > 0:
1081
- rows = _graph_expand_merge(
1082
- rows,
1083
- query=query,
1084
- query_vec=query_vec,
1085
- db=db,
1086
- uri=uri,
1087
- limit=need,
1088
- extra_predicates=extra_java,
1089
- expand_depth=expand_depth,
1090
- ladybug_path=ladybug_path,
1091
- rank_config=rank_config,
1092
- )
1093
-
1094
- # Dedup by primary_type_fqn after all sorting/merging, before windowing
1095
- rows = _dedup_by_fqn(rows, dedup_by_fqn=dedup_by_fqn)
1096
-
1097
- window = rows[offset : offset + limit]
1098
- if context_neighbors > 0 and key == "java":
1099
- _attach_neighbor_context(window, db=db, neighbors=context_neighbors, uri=uri)
1100
- return window
1101
-
1102
- merged: list[dict] = []
1103
- per_table = max(need * 3, need)
1104
- for key in table_keys:
1105
- preds = extra_java if key == "java" else []
1106
- merged.extend(
1107
- _search_one_table(
1108
- TABLES[key],
1109
- uri=uri,
1110
- db=db,
1111
- query_vec=query_vec,
1112
- limit=per_table,
1113
- path_predicate=path_predicate,
1114
- kind=key,
1115
- hybrid=False,
1116
- fts_text=None,
1117
- extra_predicates=preds,
1118
- )
1119
- )
1120
- _apply_chunk_hints(merged)
1121
- _refine_java_start_lines(merged)
1122
- if skip_role_weight:
1123
- for r in merged:
1124
- r["_skip_role_weight"] = True
1125
- _apply_symbol_bonus(merged, query_toks)
1126
- merged.sort(key=_vector_sort_key)
1127
- # Vector: displayed score from the effective (bonus-adjusted) distance.
1128
- for r in merged:
1129
- comps = r.setdefault("_score_components", {})
1130
- effective_dist = _effective_distance(comps)
1131
- r["_score"] = vector_display_score(effective_dist)
1132
-
1133
- # Dedup by primary_type_fqn after all sorting/merging, before windowing
1134
- merged = _dedup_by_fqn(merged, dedup_by_fqn=dedup_by_fqn)
1135
-
1136
- window = merged[offset : offset + limit]
1137
- if context_neighbors > 0:
1138
- _attach_neighbor_context(window, db=db, neighbors=context_neighbors, uri=uri)
1139
- return window
1140
-
1141
-
1142
- def main() -> None:
1143
- parser = argparse.ArgumentParser(
1144
- description="Vector search in LanceDB index.",
1145
- )
1146
- parser.add_argument("query", help="Natural-language search query")
1147
- parser.add_argument(
1148
- "--table",
1149
- choices=["java", "sql", "yaml", "all"],
1150
- default="java",
1151
- )
1152
- parser.add_argument("--limit", type=int, default=10)
1153
- parser.add_argument(
1154
- "--lancedb-uri",
1155
- default=os.environ.get("JAVA_CODEBASE_RAG_INDEX_DIR", "")
1156
- or str((Path.cwd() / ".java-codebase-rag").resolve()),
1157
- )
1158
- parser.add_argument("--path-contains", metavar="SUBSTR", default=None)
1159
- parser.add_argument(
1160
- "--model",
1161
- default=None,
1162
- help=(
1163
- "sentence-transformers hub id or local model directory "
1164
- f"(default: SBERT_MODEL env or {SBERT_MODEL!r})"
1165
- ),
1166
- )
1167
- parser.add_argument("--device", default=None)
1168
- parser.add_argument("--text-width", type=int, default=320)
1169
- parser.add_argument("--hybrid", action="store_true")
1170
- parser.add_argument("--fts-text", metavar="TEXT", default=None)
1171
- parser.add_argument("--auto-hybrid", action="store_true")
1172
- parser.add_argument("--role", default=None)
1173
- parser.add_argument("--exclude-generated", action="store_true",
1174
- help="Exclude generated sources from results.")
1175
- parser.add_argument("--generated-only", action="store_true",
1176
- help="Return only generated sources in results.")
1177
- parser.add_argument("--module", default=None,
1178
- help="Filter to a single Maven/Gradle module name.")
1179
- parser.add_argument("--microservice", default=None,
1180
- help="Filter to a single deployable microservice (top-level dir under project root).")
1181
- parser.add_argument("--package-prefix", default=None)
1182
- parser.add_argument("--graph-expand", action="store_true")
1183
- parser.add_argument("--expand-depth", type=int, default=1)
1184
- parser.add_argument("--ladybug-path", default=None)
1185
- parser.add_argument(
1186
- "--context-neighbors", type=int, default=0,
1187
- help="Attach N adjacent chunks per hit as surrounding context (Java only).",
1188
- )
1189
- args = parser.parse_args()
1190
-
1191
- uri_path = Path(args.lancedb_uri)
1192
- if not uri_path.exists():
1193
- print(f"Error: LanceDB path missing: {uri_path.resolve()}", file=sys.stderr)
1194
- sys.exit(1)
1195
-
1196
- keys = list(TABLES) if args.table == "all" else [args.table]
1197
- if args.hybrid and args.table == "all":
1198
- print("Error: --hybrid needs a single --table.", file=sys.stderr)
1199
- sys.exit(2)
1200
- if args.auto_hybrid and args.table == "all":
1201
- print("Error: --auto-hybrid needs a single --table.", file=sys.stderr)
1202
- sys.exit(2)
1203
-
1204
- raw_model = args.model
1205
- if raw_model is None or not str(raw_model).strip():
1206
- model_name = resolved_sbert_model_for_process_env(SBERT_MODEL)
1207
- else:
1208
- model_name = maybe_expand_embedding_model_path(str(raw_model).strip())
1209
-
1210
- try:
1211
- results = run_search(
1212
- args.query,
1213
- uri=str(uri_path),
1214
- table_keys=keys,
1215
- limit=args.limit,
1216
- path_substring=args.path_contains,
1217
- model_name=model_name,
1218
- device=args.device,
1219
- hybrid=args.hybrid,
1220
- fts_text=args.fts_text,
1221
- auto_hybrid=args.auto_hybrid,
1222
- role=args.role,
1223
- module=args.module,
1224
- microservice=args.microservice,
1225
- package_prefix=args.package_prefix,
1226
- graph_expand=args.graph_expand,
1227
- expand_depth=args.expand_depth,
1228
- ladybug_path=args.ladybug_path,
1229
- context_neighbors=args.context_neighbors,
1230
- exclude_generated=args.exclude_generated,
1231
- generated_only=args.generated_only,
1232
- )
1233
- except Exception as e:
1234
- print(f"Search failed: {e}", file=sys.stderr)
1235
- sys.exit(1)
1236
-
1237
- if not results:
1238
- print("No results.")
1239
- return
1240
-
1241
- w = args.text_width
1242
- for i, row in enumerate(results, start=1):
1243
- kind = row["_kind"]
1244
- fn = row["filename"]
1245
- lang = row.get("language", "—")
1246
- start = row.get("start") or {}
1247
- end = row.get("end") or {}
1248
- line_hint = ""
1249
- if isinstance(start, dict) and "line" in start:
1250
- el = (
1251
- end["line"]
1252
- if isinstance(end, dict) and "line" in end
1253
- else start["line"]
1254
- )
1255
- line_hint = f" L{start['line']}-{el}"
1256
- text = (row.get("text") or "").replace("\n", " ")
1257
- preview = text if len(text) <= w else text[: w - 3] + "..."
1258
- if row.get("_hybrid"):
1259
- rank_s = f"hybrid RRF={float(row.get('_score', 0.0)):.4f}"
1260
- else:
1261
- rank_s = f"L2 distance={float(row['_distance']):.4f}"
1262
- hints = row.get("_hints") or {}
1263
- hint_s = ""
1264
- if hints.get("primary_type_hint"):
1265
- hint_s += f" | type:{hints['primary_type_hint']}"
1266
- if hints.get("import_heavy"):
1267
- hint_s += " | mostly-imports"
1268
- role = row.get("role") or ""
1269
- if role:
1270
- hint_s += f" | role:{role}"
1271
- ms = row.get("microservice") or ""
1272
- if ms:
1273
- hint_s += f" | microservice:{ms}"
1274
- mod = row.get("module") or ""
1275
- if mod and mod != ms:
1276
- hint_s += f" | module:{mod}"
1277
- gen = row.get("generated")
1278
- gen_by = row.get("generated_by") or ""
1279
- if gen:
1280
- hint_s += f" | generated:{gen_by}" if gen_by else " | generated"
1281
- comps = row.get("_score_components") or {}
1282
- rw = comps.get("role_weight")
1283
- if rw:
1284
- hint_s += f" | role_weight:{rw:+.2f}"
1285
- sb = comps.get("symbol_bonus")
1286
- if sb:
1287
- hint_s += f" | symbol_bonus:{sb:+.2f}"
1288
- if row.get("_graph_expanded"):
1289
- hint_s += " | graph"
1290
- print(f"--- {i}. [{kind}] {rank_s} | {fn}{line_hint} | lang={lang}{hint_s}")
1291
- print(preview)
1292
- print()
1293
-
1294
-
1295
- if __name__ == "__main__":
1296
- main()