java-codebase-rag 0.11.2__py3-none-any.whl → 0.12.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- java_codebase_rag-0.12.1.dist-info/METADATA +35 -0
- java_codebase_rag-0.12.1.dist-info/RECORD +4 -0
- {java_codebase_rag-0.11.2.dist-info → java_codebase_rag-0.12.1.dist-info}/WHEEL +1 -1
- java_codebase_rag/_fdlimit.py +0 -56
- java_codebase_rag/_stdio.py +0 -32
- java_codebase_rag/_version.py +0 -35
- java_codebase_rag/absence/__init__.py +0 -0
- java_codebase_rag/absence/absence_diagnosis.py +0 -700
- java_codebase_rag/absence/absence_types.py +0 -124
- java_codebase_rag/absence/absence_vocab.py +0 -460
- java_codebase_rag/analysis/__init__.py +0 -0
- java_codebase_rag/analysis/pr_analysis.py +0 -563
- java_codebase_rag/analysis/resolve_service.py +0 -740
- java_codebase_rag/ast/__init__.py +0 -0
- java_codebase_rag/ast/ast_java.py +0 -2825
- java_codebase_rag/ast/brownfield_events.py +0 -58
- java_codebase_rag/ast/chunk_heuristics.py +0 -62
- java_codebase_rag/cli.py +0 -1215
- java_codebase_rag/cli_format.py +0 -85
- java_codebase_rag/cli_progress.py +0 -94
- java_codebase_rag/config.py +0 -833
- java_codebase_rag/eval/__init__.py +0 -1
- java_codebase_rag/eval/ground_truth.py +0 -100
- java_codebase_rag/eval/metrics.py +0 -107
- java_codebase_rag/eval/runner.py +0 -556
- java_codebase_rag/graph/__init__.py +0 -0
- java_codebase_rag/graph/build_ast_graph.py +0 -4471
- java_codebase_rag/graph/graph_enrich.py +0 -1937
- java_codebase_rag/graph/graph_types.py +0 -224
- java_codebase_rag/graph/java_ontology.py +0 -465
- java_codebase_rag/graph/ladybug_queries.py +0 -2213
- java_codebase_rag/graph/path_filtering.py +0 -477
- java_codebase_rag/index/__init__.py +0 -0
- java_codebase_rag/index/java_index_flow_lancedb.py +0 -734
- java_codebase_rag/index/java_index_v1_common.py +0 -33
- java_codebase_rag/install_data/__init__.py +0 -0
- java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -108
- java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
- java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
- java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
- java_codebase_rag/installer.py +0 -2188
- java_codebase_rag/jrag.py +0 -4531
- java_codebase_rag/jrag_envelope.py +0 -1107
- java_codebase_rag/jrag_hints.py +0 -204
- java_codebase_rag/jrag_render.py +0 -926
- java_codebase_rag/lance_optimize.py +0 -264
- java_codebase_rag/mcp/__init__.py +0 -0
- java_codebase_rag/mcp/mcp_hints.py +0 -932
- java_codebase_rag/mcp/mcp_v2.py +0 -1916
- java_codebase_rag/mcp/server.py +0 -884
- java_codebase_rag/pipeline.py +0 -531
- java_codebase_rag/progress.py +0 -570
- java_codebase_rag/read_payloads.py +0 -781
- java_codebase_rag/search/__init__.py +0 -0
- java_codebase_rag/search/index_common.py +0 -10
- java_codebase_rag/search/search_lancedb.py +0 -1296
- java_codebase_rag/search/search_lexical.py +0 -449
- java_codebase_rag/search/search_scoring.py +0 -523
- java_codebase_rag/watch/__init__.py +0 -0
- java_codebase_rag/watch/client.py +0 -230
- java_codebase_rag/watch/daemon.py +0 -396
- java_codebase_rag/watch/lock.py +0 -201
- java_codebase_rag/watch/paths.py +0 -76
- java_codebase_rag/watch/protocol.py +0 -122
- java_codebase_rag/watch/server.py +0 -273
- java_codebase_rag/watch/warm.py +0 -105
- java_codebase_rag/watch/watcher.py +0 -370
- java_codebase_rag-0.11.2.dist-info/METADATA +0 -331
- java_codebase_rag-0.11.2.dist-info/RECORD +0 -71
- java_codebase_rag-0.11.2.dist-info/entry_points.txt +0 -4
- java_codebase_rag-0.11.2.dist-info/licenses/LICENSE +0 -21
- java_codebase_rag-0.11.2.dist-info/top_level.txt +0 -1
- /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.1.dist-info/top_level.txt +0 -0
|
@@ -1,449 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env python3
|
|
2
|
-
"""Lexical (keyword) search over the LadybugDB symbol graph.
|
|
3
|
-
|
|
4
|
-
Graph-only fallback for the `search` tool on macOS Intel installs, where the
|
|
5
|
-
vector stack (lancedb / torch / sentence-transformers) is unavailable (see the
|
|
6
|
-
PEP 508 markers in pyproject.toml). Returns row-dicts in the SAME shape as
|
|
7
|
-
`search_lancedb.run_search`, so `mcp_v2._row_to_search_hit` and the rest of
|
|
8
|
-
`search_v2` work unchanged — `search` simply ranks by keyword relevance instead
|
|
9
|
-
of embeddings, with an advisory noting the mode.
|
|
10
|
-
|
|
11
|
-
This module imports only LadybugDB (always installed) and `search_scoring`
|
|
12
|
-
(dependency-free). It MUST NOT import lancedb/torch, and it MUST NOT import
|
|
13
|
-
`mcp_v2` (circular: mcp_v2 dispatches to this module). The NodeFilter is
|
|
14
|
-
duck-typed; `_lexical_where` mirrors `mcp_v2._symbol_where_from_filter` and is
|
|
15
|
-
guarded by a parity unit test.
|
|
16
|
-
"""
|
|
17
|
-
|
|
18
|
-
from __future__ import annotations
|
|
19
|
-
|
|
20
|
-
import os
|
|
21
|
-
import weakref
|
|
22
|
-
from pathlib import Path
|
|
23
|
-
from typing import TYPE_CHECKING, Any
|
|
24
|
-
|
|
25
|
-
from java_codebase_rag.graph.ladybug_queries import LadybugGraph
|
|
26
|
-
from java_codebase_rag.search.search_scoring import (
|
|
27
|
-
SYMBOL_FTS_INDEX,
|
|
28
|
-
_ROLE_SCORE_WEIGHTS,
|
|
29
|
-
_TYPE_MATCH_BONUS_CAP,
|
|
30
|
-
_TYPE_MATCH_BONUS_PER_HIT,
|
|
31
|
-
_clamp01,
|
|
32
|
-
_dedup_by_fqn,
|
|
33
|
-
_query_tokens,
|
|
34
|
-
_split_identifier,
|
|
35
|
-
build_fts_query,
|
|
36
|
-
)
|
|
37
|
-
|
|
38
|
-
if TYPE_CHECKING:
|
|
39
|
-
from java_codebase_rag.mcp.mcp_v2 import NodeFilter
|
|
40
|
-
|
|
41
|
-
# Lexical relevance weights. The class/file name is the strongest discovery
|
|
42
|
-
# signal (mirrors the type-name bonus rationale in search_scoring); fqn/package
|
|
43
|
-
# overlap is next; signature/annotation/capability text is a weaker corroborator.
|
|
44
|
-
_NAME_MATCH_WEIGHT = 0.45
|
|
45
|
-
_FQN_MATCH_WEIGHT = 0.20
|
|
46
|
-
_TEXT_MATCH_WEIGHT = 0.15
|
|
47
|
-
|
|
48
|
-
# Display-score normalization denominator. Mirrors search_scoring._HYBRID_SCORE_MAX
|
|
49
|
-
# discipline: sum of each additive component's maximum so the displayed score is
|
|
50
|
-
# rank-monotonic in [0, 1]. (= 0.45 + 0.10 + 0.20 + 0.15 + 0.10 = 1.00)
|
|
51
|
-
LEXICAL_SCORE_MAX = (
|
|
52
|
-
_NAME_MATCH_WEIGHT
|
|
53
|
-
+ _TYPE_MATCH_BONUS_CAP
|
|
54
|
-
+ _FQN_MATCH_WEIGHT
|
|
55
|
-
+ _TEXT_MATCH_WEIGHT
|
|
56
|
-
+ max(_ROLE_SCORE_WEIGHTS.values())
|
|
57
|
-
)
|
|
58
|
-
|
|
59
|
-
_SNIPPET_MAX_LINES = 20
|
|
60
|
-
_SNIPPET_MAX_CHARS = 800
|
|
61
|
-
# Safety bound on the candidate fetch (full scan over Symbols; bounded so huge
|
|
62
|
-
# repos don't pull unbounded rows into Python).
|
|
63
|
-
_CANDIDATE_LIMIT_CAP = 5000
|
|
64
|
-
|
|
65
|
-
_SYMBOL_RETURN = (
|
|
66
|
-
"s.id AS id, s.kind AS kind, s.name AS name, s.fqn AS fqn, "
|
|
67
|
-
"s.package AS package, s.module AS module, s.microservice AS microservice, "
|
|
68
|
-
"s.filename AS filename, s.start_line AS start_line, s.end_line AS end_line, "
|
|
69
|
-
"s.start_byte AS start_byte, s.end_byte AS end_byte, "
|
|
70
|
-
"s.annotations AS annotations, s.capabilities AS capabilities, "
|
|
71
|
-
"s.role AS role, s.signature AS signature, s.parent_id AS parent_id"
|
|
72
|
-
)
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
def _lexical_where(f: Any, *, path_contains: str | None) -> tuple[str, dict[str, Any]]:
|
|
76
|
-
"""Cypher WHERE for Symbol nodes from a NodeFilter (+ path_contains pushdown).
|
|
77
|
-
|
|
78
|
-
Mirrors ``mcp_v2._symbol_where_from_filter``; kept local so this module stays
|
|
79
|
-
import-isolated from mcp_v2 (and unit-testable standalone). ``path_contains``
|
|
80
|
-
is pushed down into Cypher because ``search_v2`` only re-filters the windowed
|
|
81
|
-
page post-fetch — without pushdown a path filter could empty the page even
|
|
82
|
-
when deeper-ranked rows match. A parity unit test guards drift.
|
|
83
|
-
"""
|
|
84
|
-
preds: list[str] = []
|
|
85
|
-
params: dict[str, Any] = {}
|
|
86
|
-
if f is not None:
|
|
87
|
-
if getattr(f, "microservice", None):
|
|
88
|
-
preds.append("s.microservice = $microservice")
|
|
89
|
-
params["microservice"] = f.microservice
|
|
90
|
-
if getattr(f, "module", None):
|
|
91
|
-
preds.append("s.module = $module")
|
|
92
|
-
params["module"] = f.module
|
|
93
|
-
if getattr(f, "role", None):
|
|
94
|
-
preds.append("s.role = $role")
|
|
95
|
-
params["role"] = f.role
|
|
96
|
-
if getattr(f, "exclude_roles", None):
|
|
97
|
-
preds.append("NOT s.role IN $exclude_roles")
|
|
98
|
-
params["exclude_roles"] = list(f.exclude_roles)
|
|
99
|
-
if getattr(f, "generated_only", False):
|
|
100
|
-
preds.append("s.generated = true")
|
|
101
|
-
if getattr(f, "exclude_generated", False):
|
|
102
|
-
preds.append("(s.generated IS NULL OR s.generated = false)")
|
|
103
|
-
if getattr(f, "annotation", None):
|
|
104
|
-
preds.append("list_contains(s.annotations, $annotation)")
|
|
105
|
-
params["annotation"] = f.annotation
|
|
106
|
-
if getattr(f, "capability", None):
|
|
107
|
-
preds.append("$capability IN s.capabilities")
|
|
108
|
-
params["capability"] = f.capability
|
|
109
|
-
if getattr(f, "fqn_contains", None):
|
|
110
|
-
preds.append("s.fqn CONTAINS $fqn_contains")
|
|
111
|
-
params["fqn_contains"] = f.fqn_contains
|
|
112
|
-
if getattr(f, "symbol_kind", None):
|
|
113
|
-
preds.append("s.kind = $symbol_kind")
|
|
114
|
-
params["symbol_kind"] = f.symbol_kind
|
|
115
|
-
if getattr(f, "symbol_kinds", None):
|
|
116
|
-
preds.append("s.kind IN $symbol_kinds")
|
|
117
|
-
params["symbol_kinds"] = list(f.symbol_kinds)
|
|
118
|
-
if path_contains:
|
|
119
|
-
preds.append("s.filename CONTAINS $path_contains")
|
|
120
|
-
params["path_contains"] = path_contains
|
|
121
|
-
where = f"WHERE {' AND '.join(preds)}" if preds else ""
|
|
122
|
-
return where, params
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
def _enclosing_type_fqn(fqn: str) -> str:
|
|
126
|
-
"""Member fqn ``{parent_fqn}#{signature}`` -> parent type fqn; type fqn (no '#') unchanged."""
|
|
127
|
-
return fqn.split("#", 1)[0] if fqn else fqn
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
# Non-underscore aliases for cross-module callers (search_lancedb's BM25 fusion).
|
|
131
|
-
# Behavior is identical; the leading-underscore originals stay module-private.
|
|
132
|
-
enclosing_type_fqn = _enclosing_type_fqn
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
def _resolve_source_root(graph: LadybugGraph) -> str:
|
|
136
|
-
"""Authoritative source root is the one cached on the graph at index time."""
|
|
137
|
-
try:
|
|
138
|
-
root = str(graph.meta().get("source_root") or "")
|
|
139
|
-
except Exception:
|
|
140
|
-
root = ""
|
|
141
|
-
return root or os.environ.get("JAVA_CODEBASE_RAG_SOURCE_ROOT", "").strip()
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
def _read_snippet(
|
|
145
|
-
source_root: str, filename: str, start_line: Any, end_line: Any, signature: str, fqn: str
|
|
146
|
-
) -> str:
|
|
147
|
-
"""Real source snippet for [start_line, end_line] from disk (capped); synthesized
|
|
148
|
-
from `signature` (fallback `fqn`) on any failure."""
|
|
149
|
-
synth = (signature or "").strip() or fqn
|
|
150
|
-
try:
|
|
151
|
-
sl = int(start_line) if start_line else 0
|
|
152
|
-
el = int(end_line) if end_line else sl
|
|
153
|
-
except (TypeError, ValueError):
|
|
154
|
-
return synth
|
|
155
|
-
if sl <= 0 or not filename:
|
|
156
|
-
return synth
|
|
157
|
-
try:
|
|
158
|
-
p = Path(filename)
|
|
159
|
-
full = p if p.is_absolute() else (Path(source_root) / filename)
|
|
160
|
-
if not p.is_absolute() and not source_root:
|
|
161
|
-
return synth
|
|
162
|
-
text = full.read_text(encoding="utf-8", errors="replace")
|
|
163
|
-
except OSError:
|
|
164
|
-
return synth
|
|
165
|
-
lines = text.splitlines()
|
|
166
|
-
lo = max(sl - 1, 0)
|
|
167
|
-
hi = min(el if el >= sl else sl, lo + _SNIPPET_MAX_LINES)
|
|
168
|
-
chunk = "\n".join(lines[lo:hi]).strip()
|
|
169
|
-
if len(chunk) > _SNIPPET_MAX_CHARS:
|
|
170
|
-
chunk = chunk[: _SNIPPET_MAX_CHARS - 1] + "…"
|
|
171
|
-
return chunk or synth
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
def _token_overlap(haystack_toks: set[str], needle_toks: set[str]) -> float:
|
|
175
|
-
"""Fraction of needle tokens present in haystack (0..1)."""
|
|
176
|
-
if not needle_toks:
|
|
177
|
-
return 0.0
|
|
178
|
-
return len(needle_toks & haystack_toks) / len(needle_toks)
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
# BM25 candidate fetch via the LadybugDB FTS index (fork A). DB-side indexed ranking
|
|
182
|
-
# replaces the heuristic's bounded Python scan; the heuristic below still scores the
|
|
183
|
-
# fetched candidates (name/type/fqn/role) and is the fallback when the FTS index or
|
|
184
|
-
# extension is unavailable (older graph, offline first run).
|
|
185
|
-
_FTS_CANDIDATE_K = 200 # top-K BM25 candidates; re-filtered by NodeFilter before ranking
|
|
186
|
-
# Connections that have run LOAD EXTENSION FTS. Keyed by the connection OBJECT (WeakSet),
|
|
187
|
-
# NOT id() — id() is reused after GC, which would let a fresh connection skip LOAD and then
|
|
188
|
-
# fail at QUERY_FTS_INDEX under test batching. Entries die with the connection.
|
|
189
|
-
_FTS_LOADED_CONNS: "weakref.WeakSet[object]" = weakref.WeakSet()
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
def _ensure_fts_loaded(g: LadybugGraph) -> bool:
|
|
193
|
-
"""LOAD EXTENSION FTS on the graph's (read-only) connection, once per connection.
|
|
194
|
-
|
|
195
|
-
Returns False if the extension can't be loaded (absent / offline) so the caller
|
|
196
|
-
falls back to the heuristic scan.
|
|
197
|
-
"""
|
|
198
|
-
conn = g._conn # noqa: SLF001
|
|
199
|
-
try:
|
|
200
|
-
if conn in _FTS_LOADED_CONNS:
|
|
201
|
-
return True
|
|
202
|
-
except Exception: # connection not weakref-able → LOAD every call (correct, slow)
|
|
203
|
-
pass
|
|
204
|
-
try:
|
|
205
|
-
g._rows("LOAD EXTENSION FTS") # noqa: SLF001
|
|
206
|
-
try:
|
|
207
|
-
_FTS_LOADED_CONNS.add(conn)
|
|
208
|
-
except Exception:
|
|
209
|
-
pass
|
|
210
|
-
return True
|
|
211
|
-
except Exception:
|
|
212
|
-
return False
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
def _try_fts_candidates(
|
|
216
|
-
g: LadybugGraph,
|
|
217
|
-
query: str,
|
|
218
|
-
filter: NodeFilter | None,
|
|
219
|
-
path_contains: str | None,
|
|
220
|
-
) -> dict | None:
|
|
221
|
-
"""Fetch BM25-ranked Symbol candidates via the FTS index; re-apply NodeFilter.
|
|
222
|
-
|
|
223
|
-
Returns ``{"rows": [...], "scores": {id: bm25}}`` (rows are the same shape the
|
|
224
|
-
heuristic scan yields), or ``None`` when FTS is unavailable (extension won't load,
|
|
225
|
-
or the index isn't present on this graph) so the caller falls back.
|
|
226
|
-
|
|
227
|
-
Two-step: (1) ``QUERY_FTS_INDEX`` returns the top-K node ids by Okapi BM25 over
|
|
228
|
-
``Symbol.search_text``; (2) re-MATCH those ids with the full ``_lexical_where``
|
|
229
|
-
predicates (role / module / path / kind≠file,package) so the filter logic stays
|
|
230
|
-
defined in one place. ``search_text`` is built at index time by ``build_ast_graph``
|
|
231
|
-
from the same ``_split_identifier`` the re-rank below uses, so index- and query-time
|
|
232
|
-
tokenization agree.
|
|
233
|
-
"""
|
|
234
|
-
if not _ensure_fts_loaded(g):
|
|
235
|
-
return None
|
|
236
|
-
idx_rows = g._rows("CALL SHOW_INDEXES() RETURN index_name") # noqa: SLF001
|
|
237
|
-
names = {row.get("index_name") for row in idx_rows}
|
|
238
|
-
if SYMBOL_FTS_INDEX not in names:
|
|
239
|
-
return None
|
|
240
|
-
fts = g._rows( # noqa: SLF001
|
|
241
|
-
f"CALL QUERY_FTS_INDEX('Symbol', '{SYMBOL_FTS_INDEX}', $q, top := $k) "
|
|
242
|
-
"RETURN node.id AS id, score",
|
|
243
|
-
{"q": query, "k": _FTS_CANDIDATE_K},
|
|
244
|
-
)
|
|
245
|
-
if not fts:
|
|
246
|
-
return {"rows": [], "scores": {}}
|
|
247
|
-
scores = {row["id"]: float(row.get("score") or 0.0) for row in fts}
|
|
248
|
-
ids = list(scores.keys())
|
|
249
|
-
|
|
250
|
-
# Re-MATCH the K ids with the SAME predicates the heuristic pushes down, so
|
|
251
|
-
# NodeFilter / path / structural-kind filtering is defined exactly once.
|
|
252
|
-
where, params = _lexical_where(filter, path_contains=path_contains)
|
|
253
|
-
struct_pred = "(s.kind <> 'file' AND s.kind <> 'package')"
|
|
254
|
-
if not where:
|
|
255
|
-
where = f"WHERE s.id IN $ids AND {struct_pred}"
|
|
256
|
-
else:
|
|
257
|
-
where = where.replace("WHERE ", f"WHERE s.id IN $ids AND {struct_pred} AND ", 1)
|
|
258
|
-
params["ids"] = ids
|
|
259
|
-
rows = g._rows(f"MATCH (s:Symbol) {where} RETURN {_SYMBOL_RETURN}", params) # noqa: SLF001
|
|
260
|
-
return {"rows": rows, "scores": scores}
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
# Non-underscore alias for cross-module callers (search_lancedb's BM25 fusion on the
|
|
264
|
-
# vector path). Behavior is identical to the leading-underscore original.
|
|
265
|
-
fetch_fts_candidates = _try_fts_candidates
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
def run_lexical_search(
|
|
269
|
-
query: str,
|
|
270
|
-
*,
|
|
271
|
-
table: str = "java",
|
|
272
|
-
limit: int = 5,
|
|
273
|
-
offset: int = 0,
|
|
274
|
-
path_contains: str | None = None,
|
|
275
|
-
filter: NodeFilter | None = None,
|
|
276
|
-
explain: bool = False,
|
|
277
|
-
dedup: bool = True,
|
|
278
|
-
advisories: list[str] | None = None,
|
|
279
|
-
graph: LadybugGraph | None = None,
|
|
280
|
-
) -> list[dict]:
|
|
281
|
-
"""Keyword search over Symbol nodes; returns ``run_search``-shaped row-dicts.
|
|
282
|
-
|
|
283
|
-
BM25-first (fork A): when the LadybugDB ``sym_fts`` index exists, candidates are
|
|
284
|
-
fetched DB-side via Okapi BM25 over ``Symbol.search_text`` (killing the bounded
|
|
285
|
-
Python scan that silently missed matches past the cap on large repos) and then
|
|
286
|
-
re-ranked here by the name/type/fqn/role heuristic. The query is pre-split with the
|
|
287
|
-
same tokenizer as ``search_text`` so pasted camelCase identifiers match. Falls back
|
|
288
|
-
to the heuristic scan when the FTS index or extension is unavailable (older graph,
|
|
289
|
-
offline first run), or when the query is degenerate / the BM25 result is empty.
|
|
290
|
-
|
|
291
|
-
Raises ``RuntimeError`` (message contains "lexical search unavailable") if no
|
|
292
|
-
symbol graph exists — the caller maps that to a clean failure envelope. Returns
|
|
293
|
-
``[]`` for ``table in ("sql", "yaml")`` (those LanceDB tables aren't built in
|
|
294
|
-
graph-only mode) and when the graph exists but nothing matches.
|
|
295
|
-
"""
|
|
296
|
-
# sql/yaml LanceDB tables don't exist in graph-only mode.
|
|
297
|
-
if table in ("sql", "yaml"):
|
|
298
|
-
return []
|
|
299
|
-
|
|
300
|
-
if graph is None and not LadybugGraph.exists():
|
|
301
|
-
raise RuntimeError(
|
|
302
|
-
"lexical search unavailable: no symbol graph found; "
|
|
303
|
-
"run `java-codebase-rag init` or `java-codebase-rag reprocess` to build one"
|
|
304
|
-
)
|
|
305
|
-
g = graph or LadybugGraph.get()
|
|
306
|
-
|
|
307
|
-
# --- candidate fetch: BM25 (FTS) preferred, heuristic scan fallback ---
|
|
308
|
-
# FTS indexes Symbol.search_text (camelCase-split tokens); pre-split the query the
|
|
309
|
-
# same way (build_fts_query) so a pasted identifier like "DistributionChunkService"
|
|
310
|
-
# matches — LadybugDB FTS's own tokenizer does not split camelCase. An empty split
|
|
311
|
-
# (degenerate / stopword-only query), an unavailable FTS index, OR an empty BM25
|
|
312
|
-
# result all fall back to the heuristic scan, which yields role-ranked output for
|
|
313
|
-
# degenerate queries and covers selective filters where BM25's top-K thins to nil.
|
|
314
|
-
bm25_scores: dict[str, float] = {}
|
|
315
|
-
use_fts = False
|
|
316
|
-
rows: list[dict] | None = None
|
|
317
|
-
q_fts = build_fts_query(query)
|
|
318
|
-
if q_fts:
|
|
319
|
-
fts = _try_fts_candidates(g, q_fts, filter, path_contains)
|
|
320
|
-
if fts is not None and fts["rows"]:
|
|
321
|
-
rows = fts["rows"]
|
|
322
|
-
bm25_scores = fts["scores"]
|
|
323
|
-
use_fts = True
|
|
324
|
-
if rows is None:
|
|
325
|
-
where, params = _lexical_where(filter, path_contains=path_contains)
|
|
326
|
-
# Always exclude structural Symbol nodes. Files and packages are :Symbol-labeled
|
|
327
|
-
# (kind='file'/'package') but aren't searchable code declarations — without this
|
|
328
|
-
# a token that appears in a filename (e.g. 'distribution' in
|
|
329
|
-
# 'DistributionChunkService.java') would surface the file node as a hit.
|
|
330
|
-
struct_pred = "(s.kind <> 'file' AND s.kind <> 'package')"
|
|
331
|
-
where = f"WHERE {struct_pred}" if not where else where.replace("WHERE ", f"WHERE {struct_pred} AND ", 1)
|
|
332
|
-
# The heuristic scan returns rows in storage order — there is NO DB-side relevance
|
|
333
|
-
# ORDER BY without the FTS index — so fetch the FULL candidate pool up to the safety
|
|
334
|
-
# cap and rank here. A pagination-derived LIMIT (4x the page) is correct on the
|
|
335
|
-
# vector path where LanceDB returns rows pre-ranked, but on this unordered scan it
|
|
336
|
-
# would return only the first ~N symbols in arbitrary storage order and silently
|
|
337
|
-
# miss the best match on any non-trivial repo. The BM25 (FTS) path above has no cap.
|
|
338
|
-
params["lim"] = _CANDIDATE_LIMIT_CAP
|
|
339
|
-
cypher = f"MATCH (s:Symbol) {where} RETURN {_SYMBOL_RETURN} LIMIT $lim"
|
|
340
|
-
rows = g._rows(cypher, params) # noqa: SLF001 — de facto public read API (see find_v2)
|
|
341
|
-
# If the fetch hit the safety cap, deeper matches were never ranked. Surface it so
|
|
342
|
-
# a user on a large repo isn't silently shown an incomplete result set.
|
|
343
|
-
if advisories is not None and len(rows) >= _CANDIDATE_LIMIT_CAP:
|
|
344
|
-
advisories.append(
|
|
345
|
-
f"lexical search scanned the first {_CANDIDATE_LIMIT_CAP} matching symbols "
|
|
346
|
-
"(repo cap); deeper matches were not ranked — refine the query or add a filter"
|
|
347
|
-
)
|
|
348
|
-
|
|
349
|
-
query_toks = _query_tokens(query)
|
|
350
|
-
source_root = _resolve_source_root(g)
|
|
351
|
-
role_locked = bool(
|
|
352
|
-
filter and (getattr(filter, "role", None) or getattr(filter, "exclude_roles", None))
|
|
353
|
-
)
|
|
354
|
-
|
|
355
|
-
out: list[dict] = []
|
|
356
|
-
for r in rows:
|
|
357
|
-
name = str(r.get("name") or "")
|
|
358
|
-
fqn = str(r.get("fqn") or "")
|
|
359
|
-
type_fqn = _enclosing_type_fqn(fqn)
|
|
360
|
-
sig = str(r.get("signature") or "")
|
|
361
|
-
anns = list(r.get("annotations") or [])
|
|
362
|
-
caps = list(r.get("capabilities") or [])
|
|
363
|
-
role_raw = str(r.get("role") or "")
|
|
364
|
-
|
|
365
|
-
name_toks = set(_split_identifier(name))
|
|
366
|
-
type_toks = set(_split_identifier(type_fqn.rsplit(".", 1)[-1]))
|
|
367
|
-
fqn_toks = set(_split_identifier(fqn))
|
|
368
|
-
sig_toks = _query_tokens(
|
|
369
|
-
" ".join([sig, " ".join(anns), " ".join(caps), str(r.get("package") or "")])
|
|
370
|
-
)
|
|
371
|
-
|
|
372
|
-
name_overlap = len(query_toks & name_toks)
|
|
373
|
-
name_match = (
|
|
374
|
-
1.0
|
|
375
|
-
if (query_toks and query_toks <= name_toks)
|
|
376
|
-
else (min(name_overlap / max(len(name_toks), 1), 1.0) if name_toks else 0.0)
|
|
377
|
-
)
|
|
378
|
-
type_hits = len(query_toks & type_toks)
|
|
379
|
-
type_match = min(type_hits * _TYPE_MATCH_BONUS_PER_HIT, _TYPE_MATCH_BONUS_CAP)
|
|
380
|
-
fqn_match = _token_overlap(fqn_toks, query_toks)
|
|
381
|
-
text_overlap = _token_overlap(sig_toks, query_toks)
|
|
382
|
-
text_match = text_overlap * _TEXT_MATCH_WEIGHT
|
|
383
|
-
|
|
384
|
-
# A keyword search must require at least one lexical hit — role alone never
|
|
385
|
-
# qualifies a row (it only boosts/reorders matches). On the BM25 path the FTS
|
|
386
|
-
# index already established textual relevance, so the qualifier is heuristic-only.
|
|
387
|
-
# Degenerate queries with no usable tokens fall through to role-ranked listing.
|
|
388
|
-
if query_toks and not use_fts and not (name_overlap or type_hits or fqn_match or text_overlap):
|
|
389
|
-
continue
|
|
390
|
-
|
|
391
|
-
role_w = 0.0 if role_locked else _ROLE_SCORE_WEIGHTS.get(role_raw.upper(), 0.0)
|
|
392
|
-
|
|
393
|
-
if query_toks:
|
|
394
|
-
raw = (
|
|
395
|
-
_NAME_MATCH_WEIGHT * name_match
|
|
396
|
-
+ type_match
|
|
397
|
-
+ _FQN_MATCH_WEIGHT * fqn_match
|
|
398
|
-
+ text_match
|
|
399
|
-
+ role_w
|
|
400
|
-
)
|
|
401
|
-
else:
|
|
402
|
-
raw = role_w # degenerate query: rank by role only
|
|
403
|
-
score = _clamp01(raw / LEXICAL_SCORE_MAX)
|
|
404
|
-
|
|
405
|
-
comps = {
|
|
406
|
-
"name_match": round(name_match, 4),
|
|
407
|
-
"type_match": round(type_match, 4),
|
|
408
|
-
"fqn_match": round(fqn_match, 4),
|
|
409
|
-
"lexical_relevance": round(raw, 4),
|
|
410
|
-
"role_weight": role_w,
|
|
411
|
-
}
|
|
412
|
-
if use_fts:
|
|
413
|
-
comps["bm25"] = round(float(bm25_scores.get(r.get("id"), 0.0)), 4)
|
|
414
|
-
|
|
415
|
-
sl, el, sb, eb = r.get("start_line"), r.get("end_line"), r.get("start_byte"), r.get("end_byte")
|
|
416
|
-
out.append(
|
|
417
|
-
{
|
|
418
|
-
"_score": score,
|
|
419
|
-
"_kind": "java",
|
|
420
|
-
"_score_components": comps,
|
|
421
|
-
"filename": str(r.get("filename") or ""),
|
|
422
|
-
"text": _read_snippet(source_root, str(r.get("filename") or ""), sl, el, sig, fqn),
|
|
423
|
-
"primary_type_fqn": type_fqn or None,
|
|
424
|
-
# Raw node fqn (members are 'Type#method(...)') feeds
|
|
425
|
-
# _node_matches_filter's fqn_contains re-check (mcp_v2.py). Without it the
|
|
426
|
-
# post-filter falls back to primary_type_fqn (the bare type) and drops
|
|
427
|
-
# member-level matches the Cypher pushdown already accepted.
|
|
428
|
-
"fqn": fqn,
|
|
429
|
-
"microservice": r.get("microservice"),
|
|
430
|
-
"module": r.get("module"),
|
|
431
|
-
"role": role_raw or None,
|
|
432
|
-
"kind": r.get("kind"),
|
|
433
|
-
"symbol_id": r.get("id"),
|
|
434
|
-
"annotations": anns,
|
|
435
|
-
"capabilities": caps,
|
|
436
|
-
"start": {
|
|
437
|
-
"line": int(sl) if sl is not None else None,
|
|
438
|
-
"byte_offset": int(sb) if sb is not None else 0,
|
|
439
|
-
},
|
|
440
|
-
"end": {
|
|
441
|
-
"line": int(el) if el is not None else None,
|
|
442
|
-
"byte_offset": int(eb) if eb is not None else 0,
|
|
443
|
-
},
|
|
444
|
-
}
|
|
445
|
-
)
|
|
446
|
-
|
|
447
|
-
out.sort(key=lambda d: float(d.get("_score", 0.0)), reverse=True)
|
|
448
|
-
out = _dedup_by_fqn(out, dedup_by_fqn=dedup)
|
|
449
|
-
return out[offset : offset + limit]
|