java-codebase-rag 0.12.0__py3-none-any.whl → 0.12.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- java_codebase_rag-0.12.2.dist-info/METADATA +35 -0
- java_codebase_rag-0.12.2.dist-info/RECORD +4 -0
- {java_codebase_rag-0.12.0.dist-info → java_codebase_rag-0.12.2.dist-info}/WHEEL +1 -1
- java_codebase_rag/_deprecation.py +0 -103
- java_codebase_rag/_fdlimit.py +0 -56
- java_codebase_rag/_stdio.py +0 -32
- java_codebase_rag/_version.py +0 -35
- java_codebase_rag/absence/__init__.py +0 -0
- java_codebase_rag/absence/absence_diagnosis.py +0 -700
- java_codebase_rag/absence/absence_types.py +0 -124
- java_codebase_rag/absence/absence_vocab.py +0 -460
- java_codebase_rag/analysis/__init__.py +0 -0
- java_codebase_rag/analysis/pr_analysis.py +0 -563
- java_codebase_rag/analysis/resolve_service.py +0 -740
- java_codebase_rag/ast/__init__.py +0 -0
- java_codebase_rag/ast/ast_java.py +0 -2847
- java_codebase_rag/ast/ast_kotlin.py +0 -1794
- java_codebase_rag/ast/brownfield_events.py +0 -58
- java_codebase_rag/ast/chunk_heuristics.py +0 -83
- java_codebase_rag/ast/language.py +0 -117
- java_codebase_rag/cli.py +0 -1215
- java_codebase_rag/cli_dispatch.py +0 -251
- java_codebase_rag/cli_format.py +0 -85
- java_codebase_rag/cli_progress.py +0 -94
- java_codebase_rag/config.py +0 -833
- java_codebase_rag/eval/__init__.py +0 -1
- java_codebase_rag/eval/ground_truth.py +0 -100
- java_codebase_rag/eval/metrics.py +0 -107
- java_codebase_rag/eval/runner.py +0 -556
- java_codebase_rag/graph/__init__.py +0 -0
- java_codebase_rag/graph/build_ast_graph.py +0 -4593
- java_codebase_rag/graph/graph_enrich.py +0 -1940
- java_codebase_rag/graph/graph_types.py +0 -224
- java_codebase_rag/graph/java_ontology.py +0 -465
- java_codebase_rag/graph/ladybug_queries.py +0 -2213
- java_codebase_rag/graph/path_filtering.py +0 -509
- java_codebase_rag/index/__init__.py +0 -0
- java_codebase_rag/index/java_index_flow_lancedb.py +0 -879
- java_codebase_rag/index/java_index_v1_common.py +0 -33
- java_codebase_rag/install_data/__init__.py +0 -0
- java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -110
- java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
- java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
- java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
- java_codebase_rag/installer.py +0 -2188
- java_codebase_rag/jrag.py +0 -4545
- java_codebase_rag/jrag_envelope.py +0 -1107
- java_codebase_rag/jrag_hints.py +0 -204
- java_codebase_rag/jrag_render.py +0 -926
- java_codebase_rag/lance_optimize.py +0 -264
- java_codebase_rag/mcp/__init__.py +0 -0
- java_codebase_rag/mcp/mcp_hints.py +0 -932
- java_codebase_rag/mcp/mcp_v2.py +0 -1916
- java_codebase_rag/mcp/server.py +0 -886
- java_codebase_rag/pipeline.py +0 -531
- java_codebase_rag/progress.py +0 -570
- java_codebase_rag/read_payloads.py +0 -781
- java_codebase_rag/search/__init__.py +0 -0
- java_codebase_rag/search/index_common.py +0 -10
- java_codebase_rag/search/search_lancedb.py +0 -1296
- java_codebase_rag/search/search_lexical.py +0 -449
- java_codebase_rag/search/search_scoring.py +0 -537
- java_codebase_rag/watch/__init__.py +0 -0
- java_codebase_rag/watch/client.py +0 -230
- java_codebase_rag/watch/daemon.py +0 -396
- java_codebase_rag/watch/lock.py +0 -201
- java_codebase_rag/watch/paths.py +0 -76
- java_codebase_rag/watch/protocol.py +0 -122
- java_codebase_rag/watch/server.py +0 -273
- java_codebase_rag/watch/warm.py +0 -105
- java_codebase_rag/watch/watcher.py +0 -394
- java_codebase_rag-0.12.0.dist-info/METADATA +0 -340
- java_codebase_rag-0.12.0.dist-info/RECORD +0 -75
- java_codebase_rag-0.12.0.dist-info/entry_points.txt +0 -5
- java_codebase_rag-0.12.0.dist-info/licenses/LICENSE +0 -21
- java_codebase_rag-0.12.0.dist-info/top_level.txt +0 -1
- /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.2.dist-info/top_level.txt +0 -0
|
@@ -1,1296 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env python3
|
|
2
|
-
"""Semantic search over LanceDB tables built by CocoIndex (java_index_flow_lancedb)."""
|
|
3
|
-
|
|
4
|
-
from __future__ import annotations
|
|
5
|
-
|
|
6
|
-
import argparse
|
|
7
|
-
import json
|
|
8
|
-
import os
|
|
9
|
-
import sys
|
|
10
|
-
import tempfile
|
|
11
|
-
import threading
|
|
12
|
-
import warnings
|
|
13
|
-
from collections.abc import Callable
|
|
14
|
-
from contextlib import contextmanager
|
|
15
|
-
from pathlib import Path
|
|
16
|
-
|
|
17
|
-
import lancedb
|
|
18
|
-
import numpy as np
|
|
19
|
-
from sentence_transformers import SentenceTransformer
|
|
20
|
-
|
|
21
|
-
from java_codebase_rag.ast.chunk_heuristics import analyze_chunk, looks_like_code_identifier
|
|
22
|
-
from java_codebase_rag.search.index_common import SBERT_MODEL
|
|
23
|
-
from java_codebase_rag.search import search_lexical
|
|
24
|
-
from java_codebase_rag.config import maybe_expand_embedding_model_path, resolved_sbert_model_for_process_env
|
|
25
|
-
|
|
26
|
-
# Scoring & dedup primitives live in `search_scoring` (dependency-free — no
|
|
27
|
-
# lancedb/torch) so the lexical backend `search_lexical` can share them on
|
|
28
|
-
# graph-only (macOS Intel) installs where this module is unimportable. Re-exported
|
|
29
|
-
# here for backward compatibility (`from search_lancedb import _clamp01`, etc.).
|
|
30
|
-
from java_codebase_rag.search.search_scoring import ( # noqa: F401
|
|
31
|
-
BASELINE_2LIST_CONFIG,
|
|
32
|
-
DEFAULT_RANK_CONFIG,
|
|
33
|
-
DEDUP_OVERFETCH,
|
|
34
|
-
RankConfig,
|
|
35
|
-
build_fts_query,
|
|
36
|
-
_ACTION_VERB_BONUS,
|
|
37
|
-
_ACTION_VERB_PREFIXES,
|
|
38
|
-
_HYBRID_SCORE_MAX,
|
|
39
|
-
_IMPORT_DISTANCE_PENALTY,
|
|
40
|
-
_IMPORT_HYBRID_SCORE_FACTOR,
|
|
41
|
-
_ROLE_SCORE_WEIGHTS,
|
|
42
|
-
_STOPWORDS,
|
|
43
|
-
_SYMBOL_MATCH_BONUS_CAP,
|
|
44
|
-
_SYMBOL_MATCH_BONUS_PER_HIT,
|
|
45
|
-
_TYPE_MATCH_BONUS_CAP,
|
|
46
|
-
_TYPE_MATCH_BONUS_PER_HIT,
|
|
47
|
-
_apply_symbol_bonus,
|
|
48
|
-
_clamp01,
|
|
49
|
-
_dedup_by_fqn,
|
|
50
|
-
_effective_distance,
|
|
51
|
-
_query_tokens,
|
|
52
|
-
_role_weight,
|
|
53
|
-
_split_identifier,
|
|
54
|
-
_symbol_bonus,
|
|
55
|
-
declaration_line_number,
|
|
56
|
-
explain_score_components,
|
|
57
|
-
l2_distance_to_score,
|
|
58
|
-
vector_display_score,
|
|
59
|
-
)
|
|
60
|
-
|
|
61
|
-
TABLES: dict[str, str] = {
|
|
62
|
-
"java": "javacodeindex_java_code",
|
|
63
|
-
"sql": "sqlschemaindex_sql_schema",
|
|
64
|
-
"yaml": "yamlconfigindex_yaml_config",
|
|
65
|
-
}
|
|
66
|
-
|
|
67
|
-
# Optional enrichment columns on the java chunk table (absent on older indexes).
|
|
68
|
-
JAVA_ENRICHED_COLUMNS: tuple[str, ...] = (
|
|
69
|
-
"package",
|
|
70
|
-
"module",
|
|
71
|
-
"microservice",
|
|
72
|
-
"primary_type_fqn",
|
|
73
|
-
"primary_type_kind",
|
|
74
|
-
"role",
|
|
75
|
-
"annotations_on_type",
|
|
76
|
-
"symbols",
|
|
77
|
-
"symbol_id",
|
|
78
|
-
"metadata",
|
|
79
|
-
"ontology_version",
|
|
80
|
-
"capabilities",
|
|
81
|
-
"generated",
|
|
82
|
-
"generated_by",
|
|
83
|
-
)
|
|
84
|
-
|
|
85
|
-
VECTOR_COLUMN = "embedding"
|
|
86
|
-
_FTS_READY: set[tuple[str, str]] = set()
|
|
87
|
-
_FTS_LOCK = threading.Lock()
|
|
88
|
-
_SCHEMA_CACHE: dict[tuple[str, str], set[str]] = {}
|
|
89
|
-
_SCHEMA_LOCK = threading.Lock()
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
def _table_columns(uri: str, lance_table_name: str, db_obj: object | None = None) -> set[str]:
|
|
93
|
-
key = (uri, lance_table_name)
|
|
94
|
-
with _SCHEMA_LOCK:
|
|
95
|
-
cached = _SCHEMA_CACHE.get(key)
|
|
96
|
-
if cached is not None:
|
|
97
|
-
return cached
|
|
98
|
-
db = db_obj if db_obj is not None else lancedb.connect(uri)
|
|
99
|
-
tbl = db.open_table(lance_table_name)
|
|
100
|
-
cols = {f.name for f in tbl.schema}
|
|
101
|
-
with _SCHEMA_LOCK:
|
|
102
|
-
_SCHEMA_CACHE[key] = cols
|
|
103
|
-
return cols
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
def _escape_sql_str(s: str) -> str:
|
|
107
|
-
return s.replace("'", "''")
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
def _build_extra_predicates(
|
|
111
|
-
*,
|
|
112
|
-
columns: set[str],
|
|
113
|
-
role: str | None,
|
|
114
|
-
module: str | None,
|
|
115
|
-
microservice: str | None,
|
|
116
|
-
package_prefix: str | None,
|
|
117
|
-
fqn_in: list[str] | None,
|
|
118
|
-
role_in: list[str] | None = None,
|
|
119
|
-
exclude_roles: list[str] | None = None,
|
|
120
|
-
capability: str | None = None,
|
|
121
|
-
capability_in: list[str] | None = None,
|
|
122
|
-
generated_only: bool = False,
|
|
123
|
-
exclude_generated: bool = False,
|
|
124
|
-
) -> list[str]:
|
|
125
|
-
preds: list[str] = []
|
|
126
|
-
if role and "role" in columns:
|
|
127
|
-
preds.append(f"role = '{_escape_sql_str(role)}'")
|
|
128
|
-
|
|
129
|
-
# When both role_in and capability_in are set, combine as OR so that
|
|
130
|
-
# capability-only entrypoints (e.g. role=OTHER with MESSAGE_LISTENER)
|
|
131
|
-
# are not silently excluded by the role filter.
|
|
132
|
-
role_pred: str | None = None
|
|
133
|
-
if role_in and "role" in columns:
|
|
134
|
-
vals = ", ".join(f"'{_escape_sql_str(v)}'" for v in role_in)
|
|
135
|
-
role_pred = f"role IN ({vals})"
|
|
136
|
-
|
|
137
|
-
cap_in_pred: str | None = None
|
|
138
|
-
if capability_in and "capabilities" in columns:
|
|
139
|
-
# array_has is the preferred form in LanceDB >= 0.10 (verified against 0.30.2).
|
|
140
|
-
parts = [
|
|
141
|
-
f"array_has(capabilities, '{_escape_sql_str(c)}')"
|
|
142
|
-
for c in capability_in
|
|
143
|
-
]
|
|
144
|
-
cap_in_pred = "(" + " OR ".join(parts) + ")"
|
|
145
|
-
|
|
146
|
-
if role_pred and cap_in_pred:
|
|
147
|
-
preds.append(f"({role_pred} OR {cap_in_pred})")
|
|
148
|
-
elif role_pred:
|
|
149
|
-
preds.append(role_pred)
|
|
150
|
-
elif cap_in_pred:
|
|
151
|
-
preds.append(cap_in_pred)
|
|
152
|
-
|
|
153
|
-
if exclude_roles and "role" in columns:
|
|
154
|
-
vals = ", ".join(f"'{_escape_sql_str(v)}'" for v in exclude_roles)
|
|
155
|
-
preds.append(f"(role IS NULL OR role NOT IN ({vals}))")
|
|
156
|
-
if generated_only and "generated" in columns:
|
|
157
|
-
preds.append("generated = true")
|
|
158
|
-
if exclude_generated and "generated" in columns:
|
|
159
|
-
preds.append("(generated IS NULL OR generated = false)")
|
|
160
|
-
if module and "module" in columns:
|
|
161
|
-
preds.append(f"module = '{_escape_sql_str(module)}'")
|
|
162
|
-
if microservice and "microservice" in columns:
|
|
163
|
-
preds.append(f"microservice = '{_escape_sql_str(microservice)}'")
|
|
164
|
-
if package_prefix and "package" in columns:
|
|
165
|
-
esc = _escape_sql_str(package_prefix)
|
|
166
|
-
preds.append(f"(package = '{esc}' OR package LIKE '{esc}.%')")
|
|
167
|
-
if fqn_in and "primary_type_fqn" in columns:
|
|
168
|
-
# LanceDB/Arrow SQL supports IN; quote each.
|
|
169
|
-
vals = ", ".join(f"'{_escape_sql_str(v)}'" for v in fqn_in)
|
|
170
|
-
preds.append(f"primary_type_fqn IN ({vals})")
|
|
171
|
-
if capability and "capabilities" in columns:
|
|
172
|
-
preds.append(f"array_has(capabilities, '{_escape_sql_str(capability)}')")
|
|
173
|
-
return preds
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
def coerce_position_field(val: object) -> dict[str, object]:
|
|
177
|
-
"""LanceDB may return struct columns as JSON strings; normalize to a dict."""
|
|
178
|
-
if val is None:
|
|
179
|
-
return {}
|
|
180
|
-
if isinstance(val, dict):
|
|
181
|
-
return val
|
|
182
|
-
if isinstance(val, str):
|
|
183
|
-
try:
|
|
184
|
-
parsed = json.loads(val)
|
|
185
|
-
except json.JSONDecodeError:
|
|
186
|
-
return {}
|
|
187
|
-
return parsed if isinstance(parsed, dict) else {}
|
|
188
|
-
return {}
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
def _apply_chunk_hints(rows: list[dict]) -> None:
|
|
192
|
-
for r in rows:
|
|
193
|
-
lang = r.get("language") or ""
|
|
194
|
-
kind = str(r.get("_kind", ""))
|
|
195
|
-
if kind == "sql" and not lang:
|
|
196
|
-
lang = "sql"
|
|
197
|
-
if kind == "yaml" and not lang:
|
|
198
|
-
lang = "yaml"
|
|
199
|
-
h = analyze_chunk(r.get("text"), language=str(lang), kind=kind)
|
|
200
|
-
r["_hints"] = {
|
|
201
|
-
"primary_type_hint": h.primary_type_hint,
|
|
202
|
-
"import_heavy": h.import_heavy,
|
|
203
|
-
}
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
def _vector_sort_key(r: dict) -> float:
|
|
207
|
-
d = float(r["_distance"])
|
|
208
|
-
comps = r.setdefault("_score_components", {})
|
|
209
|
-
comps["distance"] = d
|
|
210
|
-
if r.get("_hints", {}).get("import_heavy"):
|
|
211
|
-
d += _IMPORT_DISTANCE_PENALTY
|
|
212
|
-
comps["import_penalty"] = _IMPORT_DISTANCE_PENALTY
|
|
213
|
-
d -= _role_weight(r)
|
|
214
|
-
d -= float(comps.get("symbol_bonus", 0.0))
|
|
215
|
-
return d
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
def _hybrid_sort_key(r: dict) -> float:
|
|
219
|
-
s = float(r.get("_score", 0.0))
|
|
220
|
-
comps = r.setdefault("_score_components", {})
|
|
221
|
-
comps["hybrid_rrf"] = s
|
|
222
|
-
if r.get("_hints", {}).get("import_heavy"):
|
|
223
|
-
s *= _IMPORT_HYBRID_SCORE_FACTOR
|
|
224
|
-
comps["import_penalty"] = 1.0 - _IMPORT_HYBRID_SCORE_FACTOR
|
|
225
|
-
s += _role_weight(r)
|
|
226
|
-
s += float(comps.get("symbol_bonus", 0.0))
|
|
227
|
-
return -s
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
def _hybrid_post_sort_normalization(rows: list[dict]) -> None:
|
|
231
|
-
"""Set honest displayed scores for hybrid search after sorting.
|
|
232
|
-
|
|
233
|
-
Reconstructs the composite score (raw_rrf * import_factor + role_weight + symbol_bonus)
|
|
234
|
-
and normalizes by _HYBRID_SCORE_MAX to ensure rank-monotonicity.
|
|
235
|
-
|
|
236
|
-
Mutates rows in-place, replacing _score with the normalized value.
|
|
237
|
-
"""
|
|
238
|
-
for r in rows:
|
|
239
|
-
comps = r.setdefault("_score_components", {})
|
|
240
|
-
raw = comps.get("hybrid_rrf", 0.0)
|
|
241
|
-
comps["rrf_raw"] = raw # preserve raw RRF for --explain. NOTE: when graph_expand + hybrid combine (Phase 2), _rrf_merge below overwrites this with graph-RRF, so --explain would show graph-RRF not hybrid-RRF.
|
|
242
|
-
s = raw
|
|
243
|
-
if r.get("_hints", {}).get("import_heavy"):
|
|
244
|
-
s *= _IMPORT_HYBRID_SCORE_FACTOR
|
|
245
|
-
s += comps.get("role_weight", 0.0) + comps.get("symbol_bonus", 0.0)
|
|
246
|
-
r["_score"] = _clamp01(s / _HYBRID_SCORE_MAX)
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
def _escape_like_fragment(s: str) -> str:
|
|
250
|
-
return s.replace("'", "''")
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
def _escape_sql_like_pattern(s: str) -> str:
|
|
254
|
-
out: list[str] = []
|
|
255
|
-
for c in s:
|
|
256
|
-
if c in ("\\", "%", "_"):
|
|
257
|
-
out.append("\\" + c)
|
|
258
|
-
else:
|
|
259
|
-
out.append(c)
|
|
260
|
-
return "".join(out)
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
def _build_path_predicate(path_substring: str) -> str:
|
|
264
|
-
pat = _escape_sql_like_pattern(path_substring)
|
|
265
|
-
pat = _escape_like_fragment(pat)
|
|
266
|
-
return f"filename LIKE '%{pat}%' ESCAPE '\\'"
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
def ensure_text_fts_index(uri: str, lance_table_name: str) -> None:
|
|
270
|
-
key = (uri, lance_table_name)
|
|
271
|
-
with _FTS_LOCK:
|
|
272
|
-
if key in _FTS_READY:
|
|
273
|
-
return
|
|
274
|
-
db = lancedb.connect(uri)
|
|
275
|
-
tbl = db.open_table(lance_table_name)
|
|
276
|
-
try:
|
|
277
|
-
tbl.create_fts_index("text", replace=False)
|
|
278
|
-
except Exception as e:
|
|
279
|
-
low = str(e).lower()
|
|
280
|
-
if any(
|
|
281
|
-
w in low
|
|
282
|
-
for w in ("exist", "duplicate", "already", "same name")
|
|
283
|
-
):
|
|
284
|
-
pass
|
|
285
|
-
else:
|
|
286
|
-
raise
|
|
287
|
-
_FTS_READY.add(key)
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
def _query_vector(model: SentenceTransformer, text: str) -> np.ndarray:
|
|
291
|
-
v = model.encode(
|
|
292
|
-
text,
|
|
293
|
-
convert_to_numpy=True,
|
|
294
|
-
normalize_embeddings=True,
|
|
295
|
-
show_progress_bar=False,
|
|
296
|
-
)
|
|
297
|
-
return np.asarray(v, dtype=np.float32)
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
def _combine_predicates(parts: list[str | None]) -> str | None:
|
|
301
|
-
clean = [p for p in parts if p]
|
|
302
|
-
if not clean:
|
|
303
|
-
return None
|
|
304
|
-
if len(clean) == 1:
|
|
305
|
-
return clean[0]
|
|
306
|
-
return " AND ".join(f"({p})" for p in clean)
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
# LanceDB (0.30.x) emits two Rust `tracing` WARN lines per hybrid query to stderr
|
|
310
|
-
# — "specified output columns but did not include `_score`/`_distance` ... Call
|
|
311
|
-
# `disable_scoring_autoprojection`". They are noise on the agent's stderr, not
|
|
312
|
-
# Python warnings (so `warnings.filterwarnings` can't catch them), and the fluent
|
|
313
|
-
# query builder exposes no `disable_scoring_autoprojection()` (the lower-level
|
|
314
|
-
# `to_lance().scanner(...)` path needs `pylance`, which isn't installed on the
|
|
315
|
-
# PEP 508 graph-only profile). We match them by stable substring so anything that
|
|
316
|
-
# is a REAL error still reaches stderr.
|
|
317
|
-
_LANCE_AUTOPROJ_MARKERS: tuple[str, ...] = (
|
|
318
|
-
"disable_scoring_autoprojection",
|
|
319
|
-
"did not include `_distance`",
|
|
320
|
-
"did not include `_score`",
|
|
321
|
-
)
|
|
322
|
-
|
|
323
|
-
# The fd-2 redirect below mutates the PROCESS-GLOBAL fd 2 (and
|
|
324
|
-
# ``warnings.catch_warnings`` mutates global warning state). The MCP server
|
|
325
|
-
# dispatches every tool call through ``asyncio.to_thread`` on a thread pool
|
|
326
|
-
# (server.py), so two concurrent hybrid/auto-hybrid searches would race on the
|
|
327
|
-
# dup2 bookkeeping — corrupting the saved fd and crashing the whole server with
|
|
328
|
-
# ``Bad file descriptor``. Serialize the redirect so only one thread mutates fd
|
|
329
|
-
# 2 / warning state at a time. Concurrent hybrid queries therefore serialize
|
|
330
|
-
# their ``to_list()`` (correctness over throughput); a Rust-tracing-level
|
|
331
|
-
# suppression would remove the fd hijack entirely (follow-up).
|
|
332
|
-
_LANCE_WARN_REDIRECT_LOCK = threading.Lock()
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
def _is_autoproj_noise(line: str) -> bool:
|
|
336
|
-
"""True for a LanceDB autoprojection-deprecation line (to drop).
|
|
337
|
-
|
|
338
|
-
Preserves genuine errors/tracebacks even if they happen to reference the API
|
|
339
|
-
name — only the bare deprecation log lines (no Error/Traceback/Exception) are
|
|
340
|
-
treated as noise.
|
|
341
|
-
"""
|
|
342
|
-
if not any(marker in line for marker in _LANCE_AUTOPROJ_MARKERS):
|
|
343
|
-
return False
|
|
344
|
-
return not any(seg in line for seg in ("Traceback", "Error:", "error:", "Exception"))
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
@contextmanager
|
|
348
|
-
def _silence_lance_autoproj_warnings():
|
|
349
|
-
"""Swallow LanceDB's `_score`/`_distance` autoprojection deprecation warnings.
|
|
350
|
-
|
|
351
|
-
Redirects fd 2 to a temp buffer for the duration of the wrapped call, drops
|
|
352
|
-
only the autoprojection deprecation lines, and re-emits everything else to
|
|
353
|
-
the real stderr so genuine errors stay visible. No-op if the caller opted
|
|
354
|
-
back in via ``JAVA_CODEBASE_RAG_KEEP_LANCE_WARNINGS`` (debugging).
|
|
355
|
-
|
|
356
|
-
Thread-safety: the redirect is serialized under ``_LANCE_WARN_REDIRECT_LOCK``
|
|
357
|
-
because it mutates process-global fd 2 and warning state — the MCP server
|
|
358
|
-
runs tool calls concurrently on a thread pool.
|
|
359
|
-
"""
|
|
360
|
-
if os.environ.get("JAVA_CODEBASE_RAG_KEEP_LANCE_WARNINGS"):
|
|
361
|
-
yield
|
|
362
|
-
return
|
|
363
|
-
# Also catch the (unlikely) Python-warning form defensively.
|
|
364
|
-
with _LANCE_WARN_REDIRECT_LOCK, warnings.catch_warnings():
|
|
365
|
-
warnings.filterwarnings(
|
|
366
|
-
"ignore",
|
|
367
|
-
message=r".*(disable_scoring_autoprojection|did not include `(_distance|_score)`).*",
|
|
368
|
-
)
|
|
369
|
-
with tempfile.TemporaryFile(mode="w+", encoding="utf-8", errors="replace") as captured:
|
|
370
|
-
saved = os.dup(2)
|
|
371
|
-
try:
|
|
372
|
-
os.dup2(captured.fileno(), 2)
|
|
373
|
-
yield
|
|
374
|
-
finally:
|
|
375
|
-
# Restore fd 2 FIRST so the re-emit below reaches real stderr.
|
|
376
|
-
os.dup2(saved, 2)
|
|
377
|
-
os.close(saved)
|
|
378
|
-
captured.seek(0)
|
|
379
|
-
kept = "".join(line for line in captured if not _is_autoproj_noise(line))
|
|
380
|
-
if kept:
|
|
381
|
-
sys.stderr.write(kept)
|
|
382
|
-
sys.stderr.flush()
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
def _simple_type_name(fqn: str | None) -> str | None:
|
|
386
|
-
"""``com.foo.Bar`` -> ``Bar``; None/empty -> None."""
|
|
387
|
-
if not fqn:
|
|
388
|
-
return None
|
|
389
|
-
return str(fqn).rsplit(".", 1)[-1] or None
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
def _refine_java_start_lines(rows: list[dict]) -> None:
|
|
393
|
-
"""Point each java row's ``start.line`` at the type declaration, not the chunk anchor.
|
|
394
|
-
|
|
395
|
-
LanceDB chunks are anchored at the chunk's first source line — for a
|
|
396
|
-
file-spanning chunk that's the package/import line (``start.line`` = 1)
|
|
397
|
-
while the ``class``/``interface`` declaration sits several lines down. The
|
|
398
|
-
chunk anchor is a poor display line for a symbol hit (renders as
|
|
399
|
-
``File.java:1``); derive the real declaration line from the chunk text
|
|
400
|
-
(pinned to the primary type) so a hit shows ``File.java:<decl>`` instead
|
|
401
|
-
(F8). Method-only chunks whose range doesn't include a type declaration
|
|
402
|
-
keep their chunk anchor unchanged.
|
|
403
|
-
"""
|
|
404
|
-
for r in rows:
|
|
405
|
-
if str(r.get("_kind", "")) != "java":
|
|
406
|
-
continue
|
|
407
|
-
start = r.get("start")
|
|
408
|
-
if not isinstance(start, dict):
|
|
409
|
-
continue
|
|
410
|
-
anchor = start.get("line")
|
|
411
|
-
if anchor is None:
|
|
412
|
-
continue
|
|
413
|
-
hints = r.get("_hints") or {}
|
|
414
|
-
type_name = hints.get("primary_type_hint") or _simple_type_name(r.get("primary_type_fqn"))
|
|
415
|
-
decl = declaration_line_number(r.get("text"), int(anchor), type_name)
|
|
416
|
-
if decl is not None:
|
|
417
|
-
start["line"] = decl
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
def _search_one_table(
|
|
421
|
-
table_name: str,
|
|
422
|
-
*,
|
|
423
|
-
uri: str,
|
|
424
|
-
db: object,
|
|
425
|
-
query_vec: np.ndarray,
|
|
426
|
-
limit: int,
|
|
427
|
-
path_predicate: str | None,
|
|
428
|
-
kind: str,
|
|
429
|
-
hybrid: bool,
|
|
430
|
-
fts_text: str | None,
|
|
431
|
-
extra_predicates: list[str] | None = None,
|
|
432
|
-
) -> list[dict]:
|
|
433
|
-
tbl = db.open_table(table_name)
|
|
434
|
-
has_lang = kind == "java"
|
|
435
|
-
table_cols = _table_columns(uri, table_name, db)
|
|
436
|
-
enriched_cols = table_cols if has_lang else set()
|
|
437
|
-
# `range_start` / `range_end` are needed downstream by `_attach_neighbor_context`
|
|
438
|
-
# to locate the chunk inside its file; select them whenever the schema has them.
|
|
439
|
-
base_cols = ["filename", "text", "start", "end"]
|
|
440
|
-
for col in ("range_start", "range_end"):
|
|
441
|
-
if col in table_cols:
|
|
442
|
-
base_cols.append(col)
|
|
443
|
-
java_extra = [c for c in JAVA_ENRICHED_COLUMNS if c in enriched_cols] if has_lang else []
|
|
444
|
-
combined_pred = _combine_predicates([path_predicate, *(extra_predicates or [])])
|
|
445
|
-
|
|
446
|
-
if hybrid:
|
|
447
|
-
ensure_text_fts_index(uri, table_name)
|
|
448
|
-
text_for_fts = fts_text if fts_text is not None else ""
|
|
449
|
-
columns = (
|
|
450
|
-
[*base_cols, "language", *java_extra]
|
|
451
|
-
if has_lang
|
|
452
|
-
else [*base_cols]
|
|
453
|
-
)
|
|
454
|
-
q = (
|
|
455
|
-
tbl.search(
|
|
456
|
-
query_type="hybrid",
|
|
457
|
-
vector_column_name=VECTOR_COLUMN,
|
|
458
|
-
)
|
|
459
|
-
.vector(query_vec)
|
|
460
|
-
.text(text_for_fts)
|
|
461
|
-
.select(columns)
|
|
462
|
-
.limit(limit)
|
|
463
|
-
)
|
|
464
|
-
if combined_pred:
|
|
465
|
-
q = q.where(combined_pred, prefilter=True)
|
|
466
|
-
# Hybrid selects explicit output columns without `_score`/`_distance`, so
|
|
467
|
-
# LanceDB (0.30.x) emits two Rust autoprojection deprecation WARNs to
|
|
468
|
-
# stderr per query. Silence just those lines; real errors still surface.
|
|
469
|
-
with _silence_lance_autoproj_warnings():
|
|
470
|
-
rows = q.to_list()
|
|
471
|
-
for r in rows:
|
|
472
|
-
r["_kind"] = kind
|
|
473
|
-
rs = r.pop("_relevance_score", None)
|
|
474
|
-
r["_hybrid"] = True
|
|
475
|
-
if rs is not None:
|
|
476
|
-
r["_score"] = float(rs)
|
|
477
|
-
r["start"] = coerce_position_field(r.get("start"))
|
|
478
|
-
r["end"] = coerce_position_field(r.get("end"))
|
|
479
|
-
return rows
|
|
480
|
-
|
|
481
|
-
columns = (
|
|
482
|
-
[*base_cols, "language", *java_extra, "_distance"]
|
|
483
|
-
if has_lang
|
|
484
|
-
else [*base_cols, "_distance"]
|
|
485
|
-
)
|
|
486
|
-
q = tbl.search(query_vec, vector_column_name=VECTOR_COLUMN).select(
|
|
487
|
-
columns
|
|
488
|
-
).limit(limit)
|
|
489
|
-
if combined_pred:
|
|
490
|
-
q = q.where(combined_pred, prefilter=True)
|
|
491
|
-
rows = q.to_list()
|
|
492
|
-
for r in rows:
|
|
493
|
-
r["_kind"] = kind
|
|
494
|
-
r["_hybrid"] = False
|
|
495
|
-
# Populate `_score` from `_distance` so the SearchHit.score reflects
|
|
496
|
-
# relevance. The hybrid branch sets `_score` from `_relevance_score`
|
|
497
|
-
# above; without this, non-hybrid (default) search left `_score` unset
|
|
498
|
-
# and mcp_v2._row_to_search_hit fell back to 0.0 for EVERY hit —
|
|
499
|
-
# ranking still worked (the sort key uses `_distance` directly) but the
|
|
500
|
-
# exposed score was always 0.0, making results look unranked.
|
|
501
|
-
d = r.get("_distance")
|
|
502
|
-
if d is not None:
|
|
503
|
-
# Use the same non-clamping map as the display sites so graph-expand
|
|
504
|
-
# rows (which run_search does NOT overwrite) never carry the old
|
|
505
|
-
# 1-d²/2 value that collapses to 0 past √2. (The main single/multi
|
|
506
|
-
# paths overwrite this with the bonus-adjusted effective distance.)
|
|
507
|
-
r["_score"] = vector_display_score(float(d))
|
|
508
|
-
r["start"] = coerce_position_field(r.get("start"))
|
|
509
|
-
r["end"] = coerce_position_field(r.get("end"))
|
|
510
|
-
return rows
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
def _debug_ctx(msg: str) -> None:
|
|
514
|
-
"""Emit context-expansion diagnostics when JAVA_CODEBASE_RAG_DEBUG_CONTEXT is set.
|
|
515
|
-
|
|
516
|
-
Writes to stderr so it doesn't pollute MCP stdout. Cheap no-op otherwise.
|
|
517
|
-
"""
|
|
518
|
-
if os.environ.get("JAVA_CODEBASE_RAG_DEBUG_CONTEXT"):
|
|
519
|
-
print(f"[context_neighbors] {msg}", file=sys.stderr)
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
def _attach_neighbor_context(
|
|
523
|
-
rows: list[dict], *, db: object, neighbors: int, uri: str | None = None,
|
|
524
|
-
) -> None:
|
|
525
|
-
"""Populate `_context_before` / `_context_after` with adjacent Java chunk text.
|
|
526
|
-
|
|
527
|
-
Strategy (in order):
|
|
528
|
-
1. Schema-aware scan of the java table, selecting only columns that exist
|
|
529
|
-
(`filename` + `text` always; `range_start`/`range_end` when present).
|
|
530
|
-
2. Sort the per-file bucket by `range_start` if available; otherwise keep
|
|
531
|
-
the table's natural order (good enough because chunks are produced in
|
|
532
|
-
file order by CocoIndex).
|
|
533
|
-
3. Locate each row's index via (a) range tuple match, (b) exact text match
|
|
534
|
-
as fallback. Missing both -> log and skip.
|
|
535
|
-
4. Any exception is logged (behind env flag) and the field stays empty; we
|
|
536
|
-
never break search because of context expansion.
|
|
537
|
-
"""
|
|
538
|
-
if neighbors <= 0:
|
|
539
|
-
return
|
|
540
|
-
java_rows = [r for r in rows if str(r.get("_kind", "")) == "java"]
|
|
541
|
-
if not java_rows:
|
|
542
|
-
_debug_ctx("no java rows in window; nothing to expand")
|
|
543
|
-
return
|
|
544
|
-
filenames = {str(r.get("filename", "")) for r in java_rows if r.get("filename")}
|
|
545
|
-
if not filenames:
|
|
546
|
-
_debug_ctx("java rows had no filename field; skipping")
|
|
547
|
-
return
|
|
548
|
-
|
|
549
|
-
java_table = TABLES["java"]
|
|
550
|
-
try:
|
|
551
|
-
tbl = db.open_table(java_table)
|
|
552
|
-
except Exception as exc:
|
|
553
|
-
_debug_ctx(f"open_table({java_table}) failed: {exc!r}")
|
|
554
|
-
return
|
|
555
|
-
|
|
556
|
-
# Discover which positional columns the index actually carries. Older
|
|
557
|
-
# indexes may predate `range_start`/`range_end`; newer ones always have
|
|
558
|
-
# them. Asking for a missing column makes the whole scan fail.
|
|
559
|
-
try:
|
|
560
|
-
schema_cols = _table_columns(uri, java_table, db) if uri else {f.name for f in tbl.schema}
|
|
561
|
-
except Exception as exc:
|
|
562
|
-
_debug_ctx(f"schema lookup failed: {exc!r}")
|
|
563
|
-
schema_cols = set()
|
|
564
|
-
|
|
565
|
-
has_range = {"range_start", "range_end"}.issubset(schema_cols)
|
|
566
|
-
scan_cols = ["filename", "text"]
|
|
567
|
-
if has_range:
|
|
568
|
-
scan_cols.extend(("range_start", "range_end"))
|
|
569
|
-
|
|
570
|
-
try:
|
|
571
|
-
in_list = ", ".join(f"'{_escape_sql_str(f)}'" for f in filenames)
|
|
572
|
-
scanner = tbl.to_lance().scanner(
|
|
573
|
-
filter=f"filename IN ({in_list})",
|
|
574
|
-
columns=scan_cols,
|
|
575
|
-
)
|
|
576
|
-
all_chunks = scanner.to_table().to_pylist()
|
|
577
|
-
except Exception as exc:
|
|
578
|
-
_debug_ctx(f"bucket scan failed (cols={scan_cols}): {exc!r}")
|
|
579
|
-
return
|
|
580
|
-
|
|
581
|
-
if not all_chunks:
|
|
582
|
-
_debug_ctx(f"bucket scan returned 0 chunks for {len(filenames)} filenames")
|
|
583
|
-
return
|
|
584
|
-
|
|
585
|
-
by_file: dict[str, list[dict]] = {}
|
|
586
|
-
for ch in all_chunks:
|
|
587
|
-
by_file.setdefault(str(ch.get("filename", "")), []).append(ch)
|
|
588
|
-
if has_range:
|
|
589
|
-
for lst in by_file.values():
|
|
590
|
-
lst.sort(
|
|
591
|
-
key=lambda c: (int(c.get("range_start") or 0), int(c.get("range_end") or 0))
|
|
592
|
-
)
|
|
593
|
-
|
|
594
|
-
attached = 0
|
|
595
|
-
for r in java_rows:
|
|
596
|
-
fn = str(r.get("filename", ""))
|
|
597
|
-
bucket = by_file.get(fn, [])
|
|
598
|
-
if not bucket:
|
|
599
|
-
_debug_ctx(f"no bucket for filename={fn!r}")
|
|
600
|
-
continue
|
|
601
|
-
|
|
602
|
-
idx: int | None = None
|
|
603
|
-
if has_range:
|
|
604
|
-
start = int(r.get("range_start") or 0)
|
|
605
|
-
end = int(r.get("range_end") or 0)
|
|
606
|
-
if start or end:
|
|
607
|
-
idx = next(
|
|
608
|
-
(
|
|
609
|
-
i for i, c in enumerate(bucket)
|
|
610
|
-
if int(c.get("range_start") or -1) == start
|
|
611
|
-
and int(c.get("range_end") or -1) == end
|
|
612
|
-
),
|
|
613
|
-
None,
|
|
614
|
-
)
|
|
615
|
-
|
|
616
|
-
if idx is None:
|
|
617
|
-
r_text = str(r.get("text") or "")
|
|
618
|
-
if r_text:
|
|
619
|
-
idx = next(
|
|
620
|
-
(i for i, c in enumerate(bucket) if str(c.get("text") or "") == r_text),
|
|
621
|
-
None,
|
|
622
|
-
)
|
|
623
|
-
|
|
624
|
-
if idx is None:
|
|
625
|
-
_debug_ctx(
|
|
626
|
-
f"could not locate chunk in bucket (file={fn!r}, "
|
|
627
|
-
f"has_range={has_range}, bucket_size={len(bucket)})"
|
|
628
|
-
)
|
|
629
|
-
continue
|
|
630
|
-
|
|
631
|
-
before_parts = [str(c.get("text") or "") for c in bucket[max(0, idx - neighbors):idx]]
|
|
632
|
-
after_parts = [str(c.get("text") or "") for c in bucket[idx + 1 : idx + 1 + neighbors]]
|
|
633
|
-
r["_context_before"] = "\n".join(before_parts)
|
|
634
|
-
r["_context_after"] = "\n".join(after_parts)
|
|
635
|
-
attached += 1
|
|
636
|
-
|
|
637
|
-
_debug_ctx(f"attached context to {attached}/{len(java_rows)} java rows")
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
def _bm25_candidate_rows(
|
|
641
|
-
*,
|
|
642
|
-
g: object,
|
|
643
|
-
query: str,
|
|
644
|
-
uri: str,
|
|
645
|
-
db: object,
|
|
646
|
-
extra_predicates: list[str],
|
|
647
|
-
columns: set[str],
|
|
648
|
-
limit: int = 100,
|
|
649
|
-
) -> list[dict]:
|
|
650
|
-
"""Fetch BM25-ranked Symbol candidates from the FTS index and resolve them to
|
|
651
|
-
chunk rows in BM25 rank order. Returns ``[]`` on any failure (silent degradation).
|
|
652
|
-
|
|
653
|
-
Pipeline:
|
|
654
|
-
1. ``search_lexical.fetch_fts_candidates(g, query)`` → BM25-ranked Symbols +
|
|
655
|
-
a ``{symbol_node_id: bm25_score}`` map. ``None`` / empty → return ``[]``.
|
|
656
|
-
2. Map each Symbol fqn to its enclosing TYPE fqn (``primary_type_fqn`` has no
|
|
657
|
-
``#``; a member ``Type#method`` maps to ``Type``). Dedupe by type fqn,
|
|
658
|
-
keeping the MAX BM25 score among same-type symbols.
|
|
659
|
-
3. Order type fqns by BM25 desc (fqn asc tiebreak — deterministic).
|
|
660
|
-
4. Fetch chunk rows from LanceDB with a FILTER-ONLY query (no vector ranking,
|
|
661
|
-
so BM25 order is preserved). Predicates = caller's ``extra_predicates`` +
|
|
662
|
-
the ``primary_type_fqn IN (...)`` built from the ordered types — preserving
|
|
663
|
-
filter parity with the vector path.
|
|
664
|
-
5. Group fetched chunks by ``primary_type_fqn``; emit in BM25 rank order,
|
|
665
|
-
each chunk carrying ``_score_components["bm25"]``.
|
|
666
|
-
6. Apply ``_apply_chunk_hints`` + ``_refine_java_start_lines`` for consistency
|
|
667
|
-
with graph_rows handling.
|
|
668
|
-
|
|
669
|
-
Any exception (FTS or LanceDB) → ``_debug_ctx`` log + return ``[]`` (silent
|
|
670
|
-
degradation; the vector path is unaffected).
|
|
671
|
-
"""
|
|
672
|
-
# 1. BM25 candidate fetch via the FTS index.
|
|
673
|
-
# Pre-split the query with the same tokenizer the ``sym_fts`` index uses
|
|
674
|
-
# (``search_text`` stores ``_split_identifier`` tokens). LadybugDB FTS's own
|
|
675
|
-
# tokenizer does NOT split camelCase, so a raw ``DistributionChunkService``
|
|
676
|
-
# would match nothing — ``build_fts_query`` mirrors what the lexical backend
|
|
677
|
-
# does at search_lexical.py (run_lexical_search), keeping index/query token
|
|
678
|
-
# spaces aligned. An empty split (degenerate / stopword-only query) → no FTS
|
|
679
|
-
# candidates → degrade silently to the vector path.
|
|
680
|
-
fts_query = build_fts_query(query)
|
|
681
|
-
if not fts_query or not fts_query.strip():
|
|
682
|
-
return []
|
|
683
|
-
try:
|
|
684
|
-
fts = search_lexical.fetch_fts_candidates(g, fts_query, filter=None, path_contains=None)
|
|
685
|
-
except Exception as exc: # noqa: BLE001 — silent degradation
|
|
686
|
-
_debug_ctx(f"bm25 FTS fetch raised: {exc!r}")
|
|
687
|
-
return []
|
|
688
|
-
if not fts or not fts.get("rows"):
|
|
689
|
-
return []
|
|
690
|
-
sym_rows = fts["rows"]
|
|
691
|
-
scores = fts.get("scores") or {}
|
|
692
|
-
|
|
693
|
-
# 2. Map symbol fqns → enclosing type fqns; keep MAX bm25 per type.
|
|
694
|
-
type_fqn_to_bm25: dict[str, float] = {}
|
|
695
|
-
for r in sym_rows:
|
|
696
|
-
fqn = r.get("fqn")
|
|
697
|
-
if not fqn:
|
|
698
|
-
continue
|
|
699
|
-
type_fqn = search_lexical.enclosing_type_fqn(str(fqn))
|
|
700
|
-
if not type_fqn:
|
|
701
|
-
continue
|
|
702
|
-
score = float(scores.get(r.get("id"), 0.0))
|
|
703
|
-
prev = type_fqn_to_bm25.get(type_fqn)
|
|
704
|
-
if prev is None or score > prev:
|
|
705
|
-
type_fqn_to_bm25[type_fqn] = score
|
|
706
|
-
if not type_fqn_to_bm25:
|
|
707
|
-
return []
|
|
708
|
-
|
|
709
|
-
# 3. Deterministic ordering: BM25 desc, fqn asc.
|
|
710
|
-
ordered_types = sorted(
|
|
711
|
-
type_fqn_to_bm25.keys(),
|
|
712
|
-
key=lambda f: (-type_fqn_to_bm25[f], f),
|
|
713
|
-
)
|
|
714
|
-
|
|
715
|
-
# 4. Filter-only chunk fetch (NO vector ranking → BM25 order preserved). The
|
|
716
|
-
# ``primary_type_fqn IN (...)`` predicate must be buildable; if the index is so
|
|
717
|
-
# old that the column is absent, we can't restrict the fetch and degrade to [].
|
|
718
|
-
if "primary_type_fqn" not in columns:
|
|
719
|
-
_debug_ctx("bm25 fetch skipped: primary_type_fqn column absent from schema")
|
|
720
|
-
return []
|
|
721
|
-
preds = list(extra_predicates) + _build_extra_predicates(
|
|
722
|
-
columns=columns,
|
|
723
|
-
role=None, module=None, microservice=None,
|
|
724
|
-
package_prefix=None, fqn_in=ordered_types,
|
|
725
|
-
)
|
|
726
|
-
combined_pred = _combine_predicates(preds)
|
|
727
|
-
base_cols = ["filename", "text", "start", "end"]
|
|
728
|
-
for col in ("range_start", "range_end"):
|
|
729
|
-
if col in columns:
|
|
730
|
-
base_cols.append(col)
|
|
731
|
-
java_extra = [c for c in JAVA_ENRICHED_COLUMNS if c in columns]
|
|
732
|
-
select_cols = [*base_cols, "language", *java_extra]
|
|
733
|
-
|
|
734
|
-
try:
|
|
735
|
-
tbl = db.open_table(TABLES["java"])
|
|
736
|
-
# LanceDB 0.34 filter-only path: search() with no vector arg issues a
|
|
737
|
-
# non-vector scan; .where/.select/.limit/.to_list returns rows in table
|
|
738
|
-
# order without re-ranking by similarity. (tbl.query() is NOT available in
|
|
739
|
-
# 0.34; to_lance().scanner() requires pylance, which isn't installed on the
|
|
740
|
-
# PEP 508 graph-only profile — search() with no vector is the supported API.)
|
|
741
|
-
q = tbl.search().select(select_cols).limit(
|
|
742
|
-
max(limit, len(ordered_types) * 4)
|
|
743
|
-
)
|
|
744
|
-
if combined_pred:
|
|
745
|
-
q = q.where(combined_pred, prefilter=True)
|
|
746
|
-
with _silence_lance_autoproj_warnings():
|
|
747
|
-
fetched = q.to_list()
|
|
748
|
-
except Exception as exc: # noqa: BLE001 — silent degradation
|
|
749
|
-
_debug_ctx(f"bm25 chunk fetch failed: {exc!r}")
|
|
750
|
-
return []
|
|
751
|
-
|
|
752
|
-
# 5. Group by primary_type_fqn; emit in BM25 rank order.
|
|
753
|
-
by_type: dict[str, list[dict]] = {}
|
|
754
|
-
for ch in fetched:
|
|
755
|
-
tf = ch.get("primary_type_fqn")
|
|
756
|
-
if tf is None:
|
|
757
|
-
continue
|
|
758
|
-
by_type.setdefault(str(tf), []).append(ch)
|
|
759
|
-
|
|
760
|
-
out: list[dict] = []
|
|
761
|
-
for type_fqn in ordered_types:
|
|
762
|
-
chunks = by_type.get(type_fqn)
|
|
763
|
-
if not chunks:
|
|
764
|
-
continue # filtered out by extra_predicates / absent from index
|
|
765
|
-
bm25_val = round(float(type_fqn_to_bm25[type_fqn]), 4)
|
|
766
|
-
for ch in chunks:
|
|
767
|
-
ch["_kind"] = "java"
|
|
768
|
-
ch["_hybrid"] = False
|
|
769
|
-
ch.setdefault("_score_components", {})["bm25"] = bm25_val
|
|
770
|
-
ch["start"] = coerce_position_field(ch.get("start"))
|
|
771
|
-
ch["end"] = coerce_position_field(ch.get("end"))
|
|
772
|
-
out.append(ch)
|
|
773
|
-
|
|
774
|
-
# 6. Consistency with graph_rows handling.
|
|
775
|
-
_apply_chunk_hints(out)
|
|
776
|
-
_refine_java_start_lines(out)
|
|
777
|
-
return out
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
def _graph_expand_merge(
|
|
781
|
-
vector_rows: list[dict],
|
|
782
|
-
*,
|
|
783
|
-
query: str,
|
|
784
|
-
query_vec: np.ndarray,
|
|
785
|
-
db: object,
|
|
786
|
-
uri: str,
|
|
787
|
-
limit: int,
|
|
788
|
-
extra_predicates: list[str],
|
|
789
|
-
expand_depth: int,
|
|
790
|
-
ladybug_path: str | None,
|
|
791
|
-
rank_config: RankConfig = DEFAULT_RANK_CONFIG,
|
|
792
|
-
) -> list[dict]:
|
|
793
|
-
"""Expand vector top-k through the graph and/or fuse BM25, then RRF-merge.
|
|
794
|
-
|
|
795
|
-
Which lists contribute is controlled by ``rank_config.lists``:
|
|
796
|
-
- ``"vector"`` — always present (the backbone; validated by RankConfig).
|
|
797
|
-
- ``"graph"`` — graph expand + fetch (skipped entirely when absent).
|
|
798
|
-
- ``"bm25"`` — LadybugDB FTS candidate fetch fused as a third list.
|
|
799
|
-
|
|
800
|
-
Silent degradation: any failure in the graph or BM25 path drops just that list;
|
|
801
|
-
the vector list is never lost. Returns ``vector_rows`` unchanged when no
|
|
802
|
-
auxiliary list yields rows.
|
|
803
|
-
"""
|
|
804
|
-
want_graph = "graph" in rank_config.lists
|
|
805
|
-
want_bm25 = "bm25" in rank_config.lists
|
|
806
|
-
if not want_graph and not want_bm25:
|
|
807
|
-
return vector_rows
|
|
808
|
-
|
|
809
|
-
# Lazy import so the module works without ladybug installed when graph_expand=False.
|
|
810
|
-
try:
|
|
811
|
-
from java_codebase_rag.graph.ladybug_queries import LadybugGraph
|
|
812
|
-
except Exception:
|
|
813
|
-
return vector_rows
|
|
814
|
-
|
|
815
|
-
if not LadybugGraph.exists(ladybug_path):
|
|
816
|
-
return vector_rows
|
|
817
|
-
|
|
818
|
-
java_cols = _table_columns(uri, TABLES["java"], db)
|
|
819
|
-
|
|
820
|
-
# --- graph list ---
|
|
821
|
-
graph_rows: list[dict] = []
|
|
822
|
-
expand_weight_by_fqn: dict[str, float] = {}
|
|
823
|
-
if want_graph:
|
|
824
|
-
seed_fqns = sorted({r.get("primary_type_fqn") for r in vector_rows if r.get("primary_type_fqn")})
|
|
825
|
-
neighbor_fqns: list[str] = []
|
|
826
|
-
if seed_fqns:
|
|
827
|
-
try:
|
|
828
|
-
graph_obj = LadybugGraph.get(ladybug_path)
|
|
829
|
-
structural = graph_obj.expand_fqns(seed_fqns, depth=expand_depth)
|
|
830
|
-
method_pairs = graph_obj.expand_methods(
|
|
831
|
-
seed_fqns, depth=expand_depth, exclude_external=True,
|
|
832
|
-
)
|
|
833
|
-
for f in structural:
|
|
834
|
-
if f:
|
|
835
|
-
expand_weight_by_fqn[f] = max(expand_weight_by_fqn.get(f, 0.0), 1.0)
|
|
836
|
-
for f, conf in method_pairs:
|
|
837
|
-
if f:
|
|
838
|
-
expand_weight_by_fqn[f] = max(expand_weight_by_fqn.get(f, 0.0), conf)
|
|
839
|
-
neighbor_fqns = list(dict.fromkeys(
|
|
840
|
-
list(structural) + [f for f, _ in method_pairs],
|
|
841
|
-
))
|
|
842
|
-
except Exception:
|
|
843
|
-
neighbor_fqns = []
|
|
844
|
-
|
|
845
|
-
novel = [fqn for fqn in neighbor_fqns if fqn and fqn not in set(seed_fqns)]
|
|
846
|
-
if novel:
|
|
847
|
-
extra = list(extra_predicates)
|
|
848
|
-
extra.extend(_build_extra_predicates(
|
|
849
|
-
columns=java_cols,
|
|
850
|
-
role=None, module=None, microservice=None,
|
|
851
|
-
package_prefix=None, fqn_in=novel,
|
|
852
|
-
))
|
|
853
|
-
try:
|
|
854
|
-
graph_rows = _search_one_table(
|
|
855
|
-
TABLES["java"],
|
|
856
|
-
uri=uri, db=db, query_vec=query_vec,
|
|
857
|
-
limit=max(limit, 20),
|
|
858
|
-
path_predicate=None, kind="java",
|
|
859
|
-
hybrid=False, fts_text=None,
|
|
860
|
-
extra_predicates=extra,
|
|
861
|
-
)
|
|
862
|
-
except Exception:
|
|
863
|
-
graph_rows = []
|
|
864
|
-
_apply_chunk_hints(graph_rows)
|
|
865
|
-
_refine_java_start_lines(graph_rows)
|
|
866
|
-
graph_rows.sort(key=_vector_sort_key)
|
|
867
|
-
for r in graph_rows:
|
|
868
|
-
r["_graph_expanded"] = True
|
|
869
|
-
r["_graph_expand_weight"] = expand_weight_by_fqn.get(
|
|
870
|
-
r.get("primary_type_fqn"), 1.0,
|
|
871
|
-
)
|
|
872
|
-
|
|
873
|
-
# --- bm25 list ---
|
|
874
|
-
bm25_rows: list[dict] = []
|
|
875
|
-
if want_bm25:
|
|
876
|
-
try:
|
|
877
|
-
graph_obj = LadybugGraph.get(ladybug_path)
|
|
878
|
-
except Exception:
|
|
879
|
-
graph_obj = None
|
|
880
|
-
if graph_obj is not None:
|
|
881
|
-
bm25_rows = _bm25_candidate_rows(
|
|
882
|
-
g=graph_obj,
|
|
883
|
-
query=query,
|
|
884
|
-
uri=uri,
|
|
885
|
-
db=db,
|
|
886
|
-
extra_predicates=extra_predicates,
|
|
887
|
-
columns=java_cols,
|
|
888
|
-
limit=limit,
|
|
889
|
-
)
|
|
890
|
-
|
|
891
|
-
# --- RRF fusion (only lists that yielded rows beyond vector) ---
|
|
892
|
-
lists: list[list[dict]] = [vector_rows]
|
|
893
|
-
row_weights: list[Callable[[dict], float] | None] = [None]
|
|
894
|
-
if want_graph and graph_rows:
|
|
895
|
-
lists.append(graph_rows)
|
|
896
|
-
row_weights.append(lambda row: float(row.get("_graph_expand_weight", 1.0)))
|
|
897
|
-
if want_bm25 and bm25_rows:
|
|
898
|
-
lists.append(bm25_rows)
|
|
899
|
-
row_weights.append(None)
|
|
900
|
-
|
|
901
|
-
if len(lists) == 1:
|
|
902
|
-
return vector_rows
|
|
903
|
-
|
|
904
|
-
return _rrf_merge(
|
|
905
|
-
lists,
|
|
906
|
-
k=rank_config.rrf_k,
|
|
907
|
-
row_weight_for_list_index=row_weights,
|
|
908
|
-
)
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
def _rrf_merge(
|
|
912
|
-
lists: list[list[dict]],
|
|
913
|
-
*,
|
|
914
|
-
k: int = 60,
|
|
915
|
-
row_weight_for_list_index: list[Callable[[dict], float] | None] | None = None,
|
|
916
|
-
) -> list[dict]:
|
|
917
|
-
"""Reciprocal-rank-fuse several ranked lists of chunk rows.
|
|
918
|
-
|
|
919
|
-
Rows are deduplicated by (filename, range_start, range_end). The merged
|
|
920
|
-
rows get a `_rrf_score` field so callers can inspect or re-sort.
|
|
921
|
-
|
|
922
|
-
When ``row_weight_for_list_index`` is set, its length must match ``lists``;
|
|
923
|
-
a non-None entry is a callable ``row -> weight`` multiplied into that list's
|
|
924
|
-
rank contribution (``None`` means weight ``1.0`` for every row).
|
|
925
|
-
"""
|
|
926
|
-
pool: dict[tuple, dict] = {}
|
|
927
|
-
for li, ranked in enumerate(lists):
|
|
928
|
-
wfn: Callable[[dict], float] | None = None
|
|
929
|
-
if row_weight_for_list_index is not None and li < len(row_weight_for_list_index):
|
|
930
|
-
wfn = row_weight_for_list_index[li]
|
|
931
|
-
for rank, row in enumerate(ranked):
|
|
932
|
-
key = (row.get("filename"), row.get("range_start"), row.get("range_end"))
|
|
933
|
-
existing = pool.get(key)
|
|
934
|
-
weight = 1.0 if wfn is None else float(wfn(row))
|
|
935
|
-
contribution = weight * (1.0 / (k + rank + 1))
|
|
936
|
-
if existing is None:
|
|
937
|
-
row["_rrf_score"] = contribution
|
|
938
|
-
pool[key] = row
|
|
939
|
-
else:
|
|
940
|
-
existing["_rrf_score"] = float(existing.get("_rrf_score", 0.0)) + contribution
|
|
941
|
-
merged = list(pool.values())
|
|
942
|
-
merged.sort(key=lambda r: -float(r.get("_rrf_score", 0.0)))
|
|
943
|
-
# Normalize displayed _rrf_score to [0,1] by theoretical max
|
|
944
|
-
# RRF max = Σ weight·1/(k+rank+1); theoretical max when all rows are rank 0
|
|
945
|
-
# with weight 1.0 = num_lists / (k + 1)
|
|
946
|
-
num_lists = len(lists)
|
|
947
|
-
max_rrf = num_lists / (k + 1)
|
|
948
|
-
for r in merged:
|
|
949
|
-
raw_score = float(r.get("_rrf_score", 0.0))
|
|
950
|
-
comps = r.setdefault("_score_components", {})
|
|
951
|
-
comps["rrf_raw"] = raw_score
|
|
952
|
-
r["_rrf_score"] = _clamp01(raw_score / max_rrf)
|
|
953
|
-
return merged
|
|
954
|
-
|
|
955
|
-
|
|
956
|
-
def run_search(
|
|
957
|
-
query: str,
|
|
958
|
-
*,
|
|
959
|
-
uri: str,
|
|
960
|
-
table_keys: list[str],
|
|
961
|
-
limit: int,
|
|
962
|
-
path_substring: str | None,
|
|
963
|
-
model_name: str,
|
|
964
|
-
device: str | None,
|
|
965
|
-
offset: int = 0,
|
|
966
|
-
model: SentenceTransformer | None = None,
|
|
967
|
-
hybrid: bool = False,
|
|
968
|
-
fts_text: str | None = None,
|
|
969
|
-
auto_hybrid: bool = False,
|
|
970
|
-
role: str | None = None,
|
|
971
|
-
module: str | None = None,
|
|
972
|
-
microservice: str | None = None,
|
|
973
|
-
package_prefix: str | None = None,
|
|
974
|
-
graph_expand: bool = False,
|
|
975
|
-
expand_depth: int = 1,
|
|
976
|
-
ladybug_path: str | None = None,
|
|
977
|
-
context_neighbors: int = 0,
|
|
978
|
-
role_in: list[str] | None = None,
|
|
979
|
-
exclude_roles: list[str] | None = None,
|
|
980
|
-
capability: str | None = None,
|
|
981
|
-
capability_in: list[str] | None = None,
|
|
982
|
-
generated_only: bool = False,
|
|
983
|
-
exclude_generated: bool = False,
|
|
984
|
-
dedup_by_fqn: bool = False,
|
|
985
|
-
rank_config: RankConfig = DEFAULT_RANK_CONFIG,
|
|
986
|
-
) -> list[dict]:
|
|
987
|
-
effective_hybrid = hybrid
|
|
988
|
-
effective_fts = fts_text
|
|
989
|
-
if (
|
|
990
|
-
auto_hybrid
|
|
991
|
-
and not hybrid
|
|
992
|
-
and len(table_keys) == 1
|
|
993
|
-
and looks_like_code_identifier(query)
|
|
994
|
-
):
|
|
995
|
-
effective_hybrid = True
|
|
996
|
-
if effective_fts is None:
|
|
997
|
-
effective_fts = query.strip()
|
|
998
|
-
|
|
999
|
-
if effective_hybrid and len(table_keys) != 1:
|
|
1000
|
-
raise ValueError(
|
|
1001
|
-
"hybrid search requires exactly one table; "
|
|
1002
|
-
"use table java, sql, or yaml (not all)."
|
|
1003
|
-
)
|
|
1004
|
-
|
|
1005
|
-
path_predicate = (
|
|
1006
|
-
_build_path_predicate(path_substring) if path_substring else None
|
|
1007
|
-
)
|
|
1008
|
-
|
|
1009
|
-
if model is None:
|
|
1010
|
-
model = SentenceTransformer(
|
|
1011
|
-
model_name,
|
|
1012
|
-
device=device,
|
|
1013
|
-
trust_remote_code=True,
|
|
1014
|
-
)
|
|
1015
|
-
query_vec = _query_vector(model, query)
|
|
1016
|
-
fts_for_hybrid = effective_fts if effective_fts is not None else query
|
|
1017
|
-
|
|
1018
|
-
db = lancedb.connect(uri)
|
|
1019
|
-
if dedup_by_fqn:
|
|
1020
|
-
# Over-fetch to absorb per-FQN chunk multiplicity: fetch 4x so that
|
|
1021
|
-
# after collapsing, the page stays full and the +1 truncation sentinel survives.
|
|
1022
|
-
# The 4× factor assumes typical per-FQN chunk multiplicity; a single type with
|
|
1023
|
-
# many high-ranking chunks (e.g. generated/God classes) could starve the page or
|
|
1024
|
-
# make the +1 truncation sentinel unreliable; Phase 1 may revisit adaptive over-fetch (plan risk #1).
|
|
1025
|
-
need = max((limit + offset) * DEDUP_OVERFETCH, limit + offset + 1)
|
|
1026
|
-
else:
|
|
1027
|
-
# Non-dedup path: exact fetch as before
|
|
1028
|
-
need = max(limit + offset, 1)
|
|
1029
|
-
|
|
1030
|
-
extra_java = _build_extra_predicates(
|
|
1031
|
-
columns=_table_columns(uri, TABLES["java"], db),
|
|
1032
|
-
role=role, module=module, microservice=microservice,
|
|
1033
|
-
package_prefix=package_prefix, fqn_in=None,
|
|
1034
|
-
role_in=role_in, exclude_roles=exclude_roles,
|
|
1035
|
-
capability=capability, capability_in=capability_in,
|
|
1036
|
-
generated_only=generated_only, exclude_generated=exclude_generated,
|
|
1037
|
-
) if "java" in table_keys else []
|
|
1038
|
-
|
|
1039
|
-
skip_role_weight = bool(role or role_in or exclude_roles)
|
|
1040
|
-
query_toks = _query_tokens(query)
|
|
1041
|
-
|
|
1042
|
-
if len(table_keys) == 1:
|
|
1043
|
-
key = table_keys[0]
|
|
1044
|
-
preds = extra_java if key == "java" else []
|
|
1045
|
-
rows = _search_one_table(
|
|
1046
|
-
TABLES[key],
|
|
1047
|
-
uri=uri,
|
|
1048
|
-
db=db,
|
|
1049
|
-
query_vec=query_vec,
|
|
1050
|
-
limit=need,
|
|
1051
|
-
path_predicate=path_predicate,
|
|
1052
|
-
kind=key,
|
|
1053
|
-
hybrid=effective_hybrid,
|
|
1054
|
-
fts_text=fts_for_hybrid,
|
|
1055
|
-
extra_predicates=preds,
|
|
1056
|
-
)
|
|
1057
|
-
_apply_chunk_hints(rows)
|
|
1058
|
-
# Anchor each java row's start.line on the type declaration instead of
|
|
1059
|
-
# the chunk's first source line (often the package/import line = 1).
|
|
1060
|
-
_refine_java_start_lines(rows)
|
|
1061
|
-
if skip_role_weight:
|
|
1062
|
-
for r in rows:
|
|
1063
|
-
r["_skip_role_weight"] = True
|
|
1064
|
-
_apply_symbol_bonus(rows, query_toks)
|
|
1065
|
-
if effective_hybrid:
|
|
1066
|
-
rows.sort(key=_hybrid_sort_key)
|
|
1067
|
-
# Hybrid: set honest displayed score from composite sort metric, clamped to [0,1]
|
|
1068
|
-
_hybrid_post_sort_normalization(rows)
|
|
1069
|
-
else:
|
|
1070
|
-
rows.sort(key=_vector_sort_key)
|
|
1071
|
-
# Vector: displayed score from the effective (bonus-adjusted) distance,
|
|
1072
|
-
# normalized over the unit-embedding range so a correctly-ranked top
|
|
1073
|
-
# hit never collapses to 0.000 (the cosine map 1 - d²/2 clamps to 0
|
|
1074
|
-
# past √2; weak-but-best matches commonly sit at d ≈ 1.5).
|
|
1075
|
-
for r in rows:
|
|
1076
|
-
comps = r.setdefault("_score_components", {})
|
|
1077
|
-
effective_dist = _effective_distance(comps)
|
|
1078
|
-
r["_score"] = vector_display_score(effective_dist)
|
|
1079
|
-
|
|
1080
|
-
if graph_expand and key == "java" and expand_depth > 0:
|
|
1081
|
-
rows = _graph_expand_merge(
|
|
1082
|
-
rows,
|
|
1083
|
-
query=query,
|
|
1084
|
-
query_vec=query_vec,
|
|
1085
|
-
db=db,
|
|
1086
|
-
uri=uri,
|
|
1087
|
-
limit=need,
|
|
1088
|
-
extra_predicates=extra_java,
|
|
1089
|
-
expand_depth=expand_depth,
|
|
1090
|
-
ladybug_path=ladybug_path,
|
|
1091
|
-
rank_config=rank_config,
|
|
1092
|
-
)
|
|
1093
|
-
|
|
1094
|
-
# Dedup by primary_type_fqn after all sorting/merging, before windowing
|
|
1095
|
-
rows = _dedup_by_fqn(rows, dedup_by_fqn=dedup_by_fqn)
|
|
1096
|
-
|
|
1097
|
-
window = rows[offset : offset + limit]
|
|
1098
|
-
if context_neighbors > 0 and key == "java":
|
|
1099
|
-
_attach_neighbor_context(window, db=db, neighbors=context_neighbors, uri=uri)
|
|
1100
|
-
return window
|
|
1101
|
-
|
|
1102
|
-
merged: list[dict] = []
|
|
1103
|
-
per_table = max(need * 3, need)
|
|
1104
|
-
for key in table_keys:
|
|
1105
|
-
preds = extra_java if key == "java" else []
|
|
1106
|
-
merged.extend(
|
|
1107
|
-
_search_one_table(
|
|
1108
|
-
TABLES[key],
|
|
1109
|
-
uri=uri,
|
|
1110
|
-
db=db,
|
|
1111
|
-
query_vec=query_vec,
|
|
1112
|
-
limit=per_table,
|
|
1113
|
-
path_predicate=path_predicate,
|
|
1114
|
-
kind=key,
|
|
1115
|
-
hybrid=False,
|
|
1116
|
-
fts_text=None,
|
|
1117
|
-
extra_predicates=preds,
|
|
1118
|
-
)
|
|
1119
|
-
)
|
|
1120
|
-
_apply_chunk_hints(merged)
|
|
1121
|
-
_refine_java_start_lines(merged)
|
|
1122
|
-
if skip_role_weight:
|
|
1123
|
-
for r in merged:
|
|
1124
|
-
r["_skip_role_weight"] = True
|
|
1125
|
-
_apply_symbol_bonus(merged, query_toks)
|
|
1126
|
-
merged.sort(key=_vector_sort_key)
|
|
1127
|
-
# Vector: displayed score from the effective (bonus-adjusted) distance.
|
|
1128
|
-
for r in merged:
|
|
1129
|
-
comps = r.setdefault("_score_components", {})
|
|
1130
|
-
effective_dist = _effective_distance(comps)
|
|
1131
|
-
r["_score"] = vector_display_score(effective_dist)
|
|
1132
|
-
|
|
1133
|
-
# Dedup by primary_type_fqn after all sorting/merging, before windowing
|
|
1134
|
-
merged = _dedup_by_fqn(merged, dedup_by_fqn=dedup_by_fqn)
|
|
1135
|
-
|
|
1136
|
-
window = merged[offset : offset + limit]
|
|
1137
|
-
if context_neighbors > 0:
|
|
1138
|
-
_attach_neighbor_context(window, db=db, neighbors=context_neighbors, uri=uri)
|
|
1139
|
-
return window
|
|
1140
|
-
|
|
1141
|
-
|
|
1142
|
-
def main() -> None:
|
|
1143
|
-
parser = argparse.ArgumentParser(
|
|
1144
|
-
description="Vector search in LanceDB index.",
|
|
1145
|
-
)
|
|
1146
|
-
parser.add_argument("query", help="Natural-language search query")
|
|
1147
|
-
parser.add_argument(
|
|
1148
|
-
"--table",
|
|
1149
|
-
choices=["java", "sql", "yaml", "all"],
|
|
1150
|
-
default="java",
|
|
1151
|
-
)
|
|
1152
|
-
parser.add_argument("--limit", type=int, default=10)
|
|
1153
|
-
parser.add_argument(
|
|
1154
|
-
"--lancedb-uri",
|
|
1155
|
-
default=os.environ.get("JAVA_CODEBASE_RAG_INDEX_DIR", "")
|
|
1156
|
-
or str((Path.cwd() / ".java-codebase-rag").resolve()),
|
|
1157
|
-
)
|
|
1158
|
-
parser.add_argument("--path-contains", metavar="SUBSTR", default=None)
|
|
1159
|
-
parser.add_argument(
|
|
1160
|
-
"--model",
|
|
1161
|
-
default=None,
|
|
1162
|
-
help=(
|
|
1163
|
-
"sentence-transformers hub id or local model directory "
|
|
1164
|
-
f"(default: SBERT_MODEL env or {SBERT_MODEL!r})"
|
|
1165
|
-
),
|
|
1166
|
-
)
|
|
1167
|
-
parser.add_argument("--device", default=None)
|
|
1168
|
-
parser.add_argument("--text-width", type=int, default=320)
|
|
1169
|
-
parser.add_argument("--hybrid", action="store_true")
|
|
1170
|
-
parser.add_argument("--fts-text", metavar="TEXT", default=None)
|
|
1171
|
-
parser.add_argument("--auto-hybrid", action="store_true")
|
|
1172
|
-
parser.add_argument("--role", default=None)
|
|
1173
|
-
parser.add_argument("--exclude-generated", action="store_true",
|
|
1174
|
-
help="Exclude generated sources from results.")
|
|
1175
|
-
parser.add_argument("--generated-only", action="store_true",
|
|
1176
|
-
help="Return only generated sources in results.")
|
|
1177
|
-
parser.add_argument("--module", default=None,
|
|
1178
|
-
help="Filter to a single Maven/Gradle module name.")
|
|
1179
|
-
parser.add_argument("--microservice", default=None,
|
|
1180
|
-
help="Filter to a single deployable microservice (top-level dir under project root).")
|
|
1181
|
-
parser.add_argument("--package-prefix", default=None)
|
|
1182
|
-
parser.add_argument("--graph-expand", action="store_true")
|
|
1183
|
-
parser.add_argument("--expand-depth", type=int, default=1)
|
|
1184
|
-
parser.add_argument("--ladybug-path", default=None)
|
|
1185
|
-
parser.add_argument(
|
|
1186
|
-
"--context-neighbors", type=int, default=0,
|
|
1187
|
-
help="Attach N adjacent chunks per hit as surrounding context (Java only).",
|
|
1188
|
-
)
|
|
1189
|
-
args = parser.parse_args()
|
|
1190
|
-
|
|
1191
|
-
uri_path = Path(args.lancedb_uri)
|
|
1192
|
-
if not uri_path.exists():
|
|
1193
|
-
print(f"Error: LanceDB path missing: {uri_path.resolve()}", file=sys.stderr)
|
|
1194
|
-
sys.exit(1)
|
|
1195
|
-
|
|
1196
|
-
keys = list(TABLES) if args.table == "all" else [args.table]
|
|
1197
|
-
if args.hybrid and args.table == "all":
|
|
1198
|
-
print("Error: --hybrid needs a single --table.", file=sys.stderr)
|
|
1199
|
-
sys.exit(2)
|
|
1200
|
-
if args.auto_hybrid and args.table == "all":
|
|
1201
|
-
print("Error: --auto-hybrid needs a single --table.", file=sys.stderr)
|
|
1202
|
-
sys.exit(2)
|
|
1203
|
-
|
|
1204
|
-
raw_model = args.model
|
|
1205
|
-
if raw_model is None or not str(raw_model).strip():
|
|
1206
|
-
model_name = resolved_sbert_model_for_process_env(SBERT_MODEL)
|
|
1207
|
-
else:
|
|
1208
|
-
model_name = maybe_expand_embedding_model_path(str(raw_model).strip())
|
|
1209
|
-
|
|
1210
|
-
try:
|
|
1211
|
-
results = run_search(
|
|
1212
|
-
args.query,
|
|
1213
|
-
uri=str(uri_path),
|
|
1214
|
-
table_keys=keys,
|
|
1215
|
-
limit=args.limit,
|
|
1216
|
-
path_substring=args.path_contains,
|
|
1217
|
-
model_name=model_name,
|
|
1218
|
-
device=args.device,
|
|
1219
|
-
hybrid=args.hybrid,
|
|
1220
|
-
fts_text=args.fts_text,
|
|
1221
|
-
auto_hybrid=args.auto_hybrid,
|
|
1222
|
-
role=args.role,
|
|
1223
|
-
module=args.module,
|
|
1224
|
-
microservice=args.microservice,
|
|
1225
|
-
package_prefix=args.package_prefix,
|
|
1226
|
-
graph_expand=args.graph_expand,
|
|
1227
|
-
expand_depth=args.expand_depth,
|
|
1228
|
-
ladybug_path=args.ladybug_path,
|
|
1229
|
-
context_neighbors=args.context_neighbors,
|
|
1230
|
-
exclude_generated=args.exclude_generated,
|
|
1231
|
-
generated_only=args.generated_only,
|
|
1232
|
-
)
|
|
1233
|
-
except Exception as e:
|
|
1234
|
-
print(f"Search failed: {e}", file=sys.stderr)
|
|
1235
|
-
sys.exit(1)
|
|
1236
|
-
|
|
1237
|
-
if not results:
|
|
1238
|
-
print("No results.")
|
|
1239
|
-
return
|
|
1240
|
-
|
|
1241
|
-
w = args.text_width
|
|
1242
|
-
for i, row in enumerate(results, start=1):
|
|
1243
|
-
kind = row["_kind"]
|
|
1244
|
-
fn = row["filename"]
|
|
1245
|
-
lang = row.get("language", "—")
|
|
1246
|
-
start = row.get("start") or {}
|
|
1247
|
-
end = row.get("end") or {}
|
|
1248
|
-
line_hint = ""
|
|
1249
|
-
if isinstance(start, dict) and "line" in start:
|
|
1250
|
-
el = (
|
|
1251
|
-
end["line"]
|
|
1252
|
-
if isinstance(end, dict) and "line" in end
|
|
1253
|
-
else start["line"]
|
|
1254
|
-
)
|
|
1255
|
-
line_hint = f" L{start['line']}-{el}"
|
|
1256
|
-
text = (row.get("text") or "").replace("\n", " ")
|
|
1257
|
-
preview = text if len(text) <= w else text[: w - 3] + "..."
|
|
1258
|
-
if row.get("_hybrid"):
|
|
1259
|
-
rank_s = f"hybrid RRF={float(row.get('_score', 0.0)):.4f}"
|
|
1260
|
-
else:
|
|
1261
|
-
rank_s = f"L2 distance={float(row['_distance']):.4f}"
|
|
1262
|
-
hints = row.get("_hints") or {}
|
|
1263
|
-
hint_s = ""
|
|
1264
|
-
if hints.get("primary_type_hint"):
|
|
1265
|
-
hint_s += f" | type:{hints['primary_type_hint']}"
|
|
1266
|
-
if hints.get("import_heavy"):
|
|
1267
|
-
hint_s += " | mostly-imports"
|
|
1268
|
-
role = row.get("role") or ""
|
|
1269
|
-
if role:
|
|
1270
|
-
hint_s += f" | role:{role}"
|
|
1271
|
-
ms = row.get("microservice") or ""
|
|
1272
|
-
if ms:
|
|
1273
|
-
hint_s += f" | microservice:{ms}"
|
|
1274
|
-
mod = row.get("module") or ""
|
|
1275
|
-
if mod and mod != ms:
|
|
1276
|
-
hint_s += f" | module:{mod}"
|
|
1277
|
-
gen = row.get("generated")
|
|
1278
|
-
gen_by = row.get("generated_by") or ""
|
|
1279
|
-
if gen:
|
|
1280
|
-
hint_s += f" | generated:{gen_by}" if gen_by else " | generated"
|
|
1281
|
-
comps = row.get("_score_components") or {}
|
|
1282
|
-
rw = comps.get("role_weight")
|
|
1283
|
-
if rw:
|
|
1284
|
-
hint_s += f" | role_weight:{rw:+.2f}"
|
|
1285
|
-
sb = comps.get("symbol_bonus")
|
|
1286
|
-
if sb:
|
|
1287
|
-
hint_s += f" | symbol_bonus:{sb:+.2f}"
|
|
1288
|
-
if row.get("_graph_expanded"):
|
|
1289
|
-
hint_s += " | graph"
|
|
1290
|
-
print(f"--- {i}. [{kind}] {rank_s} | {fn}{line_hint} | lang={lang}{hint_s}")
|
|
1291
|
-
print(preview)
|
|
1292
|
-
print()
|
|
1293
|
-
|
|
1294
|
-
|
|
1295
|
-
if __name__ == "__main__":
|
|
1296
|
-
main()
|