java-codebase-rag 0.11.2__py3-none-any.whl → 0.12.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. java_codebase_rag-0.12.1.dist-info/METADATA +35 -0
  2. java_codebase_rag-0.12.1.dist-info/RECORD +4 -0
  3. {java_codebase_rag-0.11.2.dist-info → java_codebase_rag-0.12.1.dist-info}/WHEEL +1 -1
  4. java_codebase_rag/_fdlimit.py +0 -56
  5. java_codebase_rag/_stdio.py +0 -32
  6. java_codebase_rag/_version.py +0 -35
  7. java_codebase_rag/absence/__init__.py +0 -0
  8. java_codebase_rag/absence/absence_diagnosis.py +0 -700
  9. java_codebase_rag/absence/absence_types.py +0 -124
  10. java_codebase_rag/absence/absence_vocab.py +0 -460
  11. java_codebase_rag/analysis/__init__.py +0 -0
  12. java_codebase_rag/analysis/pr_analysis.py +0 -563
  13. java_codebase_rag/analysis/resolve_service.py +0 -740
  14. java_codebase_rag/ast/__init__.py +0 -0
  15. java_codebase_rag/ast/ast_java.py +0 -2825
  16. java_codebase_rag/ast/brownfield_events.py +0 -58
  17. java_codebase_rag/ast/chunk_heuristics.py +0 -62
  18. java_codebase_rag/cli.py +0 -1215
  19. java_codebase_rag/cli_format.py +0 -85
  20. java_codebase_rag/cli_progress.py +0 -94
  21. java_codebase_rag/config.py +0 -833
  22. java_codebase_rag/eval/__init__.py +0 -1
  23. java_codebase_rag/eval/ground_truth.py +0 -100
  24. java_codebase_rag/eval/metrics.py +0 -107
  25. java_codebase_rag/eval/runner.py +0 -556
  26. java_codebase_rag/graph/__init__.py +0 -0
  27. java_codebase_rag/graph/build_ast_graph.py +0 -4471
  28. java_codebase_rag/graph/graph_enrich.py +0 -1937
  29. java_codebase_rag/graph/graph_types.py +0 -224
  30. java_codebase_rag/graph/java_ontology.py +0 -465
  31. java_codebase_rag/graph/ladybug_queries.py +0 -2213
  32. java_codebase_rag/graph/path_filtering.py +0 -477
  33. java_codebase_rag/index/__init__.py +0 -0
  34. java_codebase_rag/index/java_index_flow_lancedb.py +0 -734
  35. java_codebase_rag/index/java_index_v1_common.py +0 -33
  36. java_codebase_rag/install_data/__init__.py +0 -0
  37. java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -108
  38. java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
  39. java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
  40. java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
  41. java_codebase_rag/installer.py +0 -2188
  42. java_codebase_rag/jrag.py +0 -4531
  43. java_codebase_rag/jrag_envelope.py +0 -1107
  44. java_codebase_rag/jrag_hints.py +0 -204
  45. java_codebase_rag/jrag_render.py +0 -926
  46. java_codebase_rag/lance_optimize.py +0 -264
  47. java_codebase_rag/mcp/__init__.py +0 -0
  48. java_codebase_rag/mcp/mcp_hints.py +0 -932
  49. java_codebase_rag/mcp/mcp_v2.py +0 -1916
  50. java_codebase_rag/mcp/server.py +0 -884
  51. java_codebase_rag/pipeline.py +0 -531
  52. java_codebase_rag/progress.py +0 -570
  53. java_codebase_rag/read_payloads.py +0 -781
  54. java_codebase_rag/search/__init__.py +0 -0
  55. java_codebase_rag/search/index_common.py +0 -10
  56. java_codebase_rag/search/search_lancedb.py +0 -1296
  57. java_codebase_rag/search/search_lexical.py +0 -449
  58. java_codebase_rag/search/search_scoring.py +0 -523
  59. java_codebase_rag/watch/__init__.py +0 -0
  60. java_codebase_rag/watch/client.py +0 -230
  61. java_codebase_rag/watch/daemon.py +0 -396
  62. java_codebase_rag/watch/lock.py +0 -201
  63. java_codebase_rag/watch/paths.py +0 -76
  64. java_codebase_rag/watch/protocol.py +0 -122
  65. java_codebase_rag/watch/server.py +0 -273
  66. java_codebase_rag/watch/warm.py +0 -105
  67. java_codebase_rag/watch/watcher.py +0 -370
  68. java_codebase_rag-0.11.2.dist-info/METADATA +0 -331
  69. java_codebase_rag-0.11.2.dist-info/RECORD +0 -71
  70. java_codebase_rag-0.11.2.dist-info/entry_points.txt +0 -4
  71. java_codebase_rag-0.11.2.dist-info/licenses/LICENSE +0 -21
  72. java_codebase_rag-0.11.2.dist-info/top_level.txt +0 -1
  73. /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.1.dist-info/top_level.txt +0 -0
@@ -1,449 +0,0 @@
1
- #!/usr/bin/env python3
2
- """Lexical (keyword) search over the LadybugDB symbol graph.
3
-
4
- Graph-only fallback for the `search` tool on macOS Intel installs, where the
5
- vector stack (lancedb / torch / sentence-transformers) is unavailable (see the
6
- PEP 508 markers in pyproject.toml). Returns row-dicts in the SAME shape as
7
- `search_lancedb.run_search`, so `mcp_v2._row_to_search_hit` and the rest of
8
- `search_v2` work unchanged — `search` simply ranks by keyword relevance instead
9
- of embeddings, with an advisory noting the mode.
10
-
11
- This module imports only LadybugDB (always installed) and `search_scoring`
12
- (dependency-free). It MUST NOT import lancedb/torch, and it MUST NOT import
13
- `mcp_v2` (circular: mcp_v2 dispatches to this module). The NodeFilter is
14
- duck-typed; `_lexical_where` mirrors `mcp_v2._symbol_where_from_filter` and is
15
- guarded by a parity unit test.
16
- """
17
-
18
- from __future__ import annotations
19
-
20
- import os
21
- import weakref
22
- from pathlib import Path
23
- from typing import TYPE_CHECKING, Any
24
-
25
- from java_codebase_rag.graph.ladybug_queries import LadybugGraph
26
- from java_codebase_rag.search.search_scoring import (
27
- SYMBOL_FTS_INDEX,
28
- _ROLE_SCORE_WEIGHTS,
29
- _TYPE_MATCH_BONUS_CAP,
30
- _TYPE_MATCH_BONUS_PER_HIT,
31
- _clamp01,
32
- _dedup_by_fqn,
33
- _query_tokens,
34
- _split_identifier,
35
- build_fts_query,
36
- )
37
-
38
- if TYPE_CHECKING:
39
- from java_codebase_rag.mcp.mcp_v2 import NodeFilter
40
-
41
- # Lexical relevance weights. The class/file name is the strongest discovery
42
- # signal (mirrors the type-name bonus rationale in search_scoring); fqn/package
43
- # overlap is next; signature/annotation/capability text is a weaker corroborator.
44
- _NAME_MATCH_WEIGHT = 0.45
45
- _FQN_MATCH_WEIGHT = 0.20
46
- _TEXT_MATCH_WEIGHT = 0.15
47
-
48
- # Display-score normalization denominator. Mirrors search_scoring._HYBRID_SCORE_MAX
49
- # discipline: sum of each additive component's maximum so the displayed score is
50
- # rank-monotonic in [0, 1]. (= 0.45 + 0.10 + 0.20 + 0.15 + 0.10 = 1.00)
51
- LEXICAL_SCORE_MAX = (
52
- _NAME_MATCH_WEIGHT
53
- + _TYPE_MATCH_BONUS_CAP
54
- + _FQN_MATCH_WEIGHT
55
- + _TEXT_MATCH_WEIGHT
56
- + max(_ROLE_SCORE_WEIGHTS.values())
57
- )
58
-
59
- _SNIPPET_MAX_LINES = 20
60
- _SNIPPET_MAX_CHARS = 800
61
- # Safety bound on the candidate fetch (full scan over Symbols; bounded so huge
62
- # repos don't pull unbounded rows into Python).
63
- _CANDIDATE_LIMIT_CAP = 5000
64
-
65
- _SYMBOL_RETURN = (
66
- "s.id AS id, s.kind AS kind, s.name AS name, s.fqn AS fqn, "
67
- "s.package AS package, s.module AS module, s.microservice AS microservice, "
68
- "s.filename AS filename, s.start_line AS start_line, s.end_line AS end_line, "
69
- "s.start_byte AS start_byte, s.end_byte AS end_byte, "
70
- "s.annotations AS annotations, s.capabilities AS capabilities, "
71
- "s.role AS role, s.signature AS signature, s.parent_id AS parent_id"
72
- )
73
-
74
-
75
- def _lexical_where(f: Any, *, path_contains: str | None) -> tuple[str, dict[str, Any]]:
76
- """Cypher WHERE for Symbol nodes from a NodeFilter (+ path_contains pushdown).
77
-
78
- Mirrors ``mcp_v2._symbol_where_from_filter``; kept local so this module stays
79
- import-isolated from mcp_v2 (and unit-testable standalone). ``path_contains``
80
- is pushed down into Cypher because ``search_v2`` only re-filters the windowed
81
- page post-fetch — without pushdown a path filter could empty the page even
82
- when deeper-ranked rows match. A parity unit test guards drift.
83
- """
84
- preds: list[str] = []
85
- params: dict[str, Any] = {}
86
- if f is not None:
87
- if getattr(f, "microservice", None):
88
- preds.append("s.microservice = $microservice")
89
- params["microservice"] = f.microservice
90
- if getattr(f, "module", None):
91
- preds.append("s.module = $module")
92
- params["module"] = f.module
93
- if getattr(f, "role", None):
94
- preds.append("s.role = $role")
95
- params["role"] = f.role
96
- if getattr(f, "exclude_roles", None):
97
- preds.append("NOT s.role IN $exclude_roles")
98
- params["exclude_roles"] = list(f.exclude_roles)
99
- if getattr(f, "generated_only", False):
100
- preds.append("s.generated = true")
101
- if getattr(f, "exclude_generated", False):
102
- preds.append("(s.generated IS NULL OR s.generated = false)")
103
- if getattr(f, "annotation", None):
104
- preds.append("list_contains(s.annotations, $annotation)")
105
- params["annotation"] = f.annotation
106
- if getattr(f, "capability", None):
107
- preds.append("$capability IN s.capabilities")
108
- params["capability"] = f.capability
109
- if getattr(f, "fqn_contains", None):
110
- preds.append("s.fqn CONTAINS $fqn_contains")
111
- params["fqn_contains"] = f.fqn_contains
112
- if getattr(f, "symbol_kind", None):
113
- preds.append("s.kind = $symbol_kind")
114
- params["symbol_kind"] = f.symbol_kind
115
- if getattr(f, "symbol_kinds", None):
116
- preds.append("s.kind IN $symbol_kinds")
117
- params["symbol_kinds"] = list(f.symbol_kinds)
118
- if path_contains:
119
- preds.append("s.filename CONTAINS $path_contains")
120
- params["path_contains"] = path_contains
121
- where = f"WHERE {' AND '.join(preds)}" if preds else ""
122
- return where, params
123
-
124
-
125
- def _enclosing_type_fqn(fqn: str) -> str:
126
- """Member fqn ``{parent_fqn}#{signature}`` -> parent type fqn; type fqn (no '#') unchanged."""
127
- return fqn.split("#", 1)[0] if fqn else fqn
128
-
129
-
130
- # Non-underscore aliases for cross-module callers (search_lancedb's BM25 fusion).
131
- # Behavior is identical; the leading-underscore originals stay module-private.
132
- enclosing_type_fqn = _enclosing_type_fqn
133
-
134
-
135
- def _resolve_source_root(graph: LadybugGraph) -> str:
136
- """Authoritative source root is the one cached on the graph at index time."""
137
- try:
138
- root = str(graph.meta().get("source_root") or "")
139
- except Exception:
140
- root = ""
141
- return root or os.environ.get("JAVA_CODEBASE_RAG_SOURCE_ROOT", "").strip()
142
-
143
-
144
- def _read_snippet(
145
- source_root: str, filename: str, start_line: Any, end_line: Any, signature: str, fqn: str
146
- ) -> str:
147
- """Real source snippet for [start_line, end_line] from disk (capped); synthesized
148
- from `signature` (fallback `fqn`) on any failure."""
149
- synth = (signature or "").strip() or fqn
150
- try:
151
- sl = int(start_line) if start_line else 0
152
- el = int(end_line) if end_line else sl
153
- except (TypeError, ValueError):
154
- return synth
155
- if sl <= 0 or not filename:
156
- return synth
157
- try:
158
- p = Path(filename)
159
- full = p if p.is_absolute() else (Path(source_root) / filename)
160
- if not p.is_absolute() and not source_root:
161
- return synth
162
- text = full.read_text(encoding="utf-8", errors="replace")
163
- except OSError:
164
- return synth
165
- lines = text.splitlines()
166
- lo = max(sl - 1, 0)
167
- hi = min(el if el >= sl else sl, lo + _SNIPPET_MAX_LINES)
168
- chunk = "\n".join(lines[lo:hi]).strip()
169
- if len(chunk) > _SNIPPET_MAX_CHARS:
170
- chunk = chunk[: _SNIPPET_MAX_CHARS - 1] + "…"
171
- return chunk or synth
172
-
173
-
174
- def _token_overlap(haystack_toks: set[str], needle_toks: set[str]) -> float:
175
- """Fraction of needle tokens present in haystack (0..1)."""
176
- if not needle_toks:
177
- return 0.0
178
- return len(needle_toks & haystack_toks) / len(needle_toks)
179
-
180
-
181
- # BM25 candidate fetch via the LadybugDB FTS index (fork A). DB-side indexed ranking
182
- # replaces the heuristic's bounded Python scan; the heuristic below still scores the
183
- # fetched candidates (name/type/fqn/role) and is the fallback when the FTS index or
184
- # extension is unavailable (older graph, offline first run).
185
- _FTS_CANDIDATE_K = 200 # top-K BM25 candidates; re-filtered by NodeFilter before ranking
186
- # Connections that have run LOAD EXTENSION FTS. Keyed by the connection OBJECT (WeakSet),
187
- # NOT id() — id() is reused after GC, which would let a fresh connection skip LOAD and then
188
- # fail at QUERY_FTS_INDEX under test batching. Entries die with the connection.
189
- _FTS_LOADED_CONNS: "weakref.WeakSet[object]" = weakref.WeakSet()
190
-
191
-
192
- def _ensure_fts_loaded(g: LadybugGraph) -> bool:
193
- """LOAD EXTENSION FTS on the graph's (read-only) connection, once per connection.
194
-
195
- Returns False if the extension can't be loaded (absent / offline) so the caller
196
- falls back to the heuristic scan.
197
- """
198
- conn = g._conn # noqa: SLF001
199
- try:
200
- if conn in _FTS_LOADED_CONNS:
201
- return True
202
- except Exception: # connection not weakref-able → LOAD every call (correct, slow)
203
- pass
204
- try:
205
- g._rows("LOAD EXTENSION FTS") # noqa: SLF001
206
- try:
207
- _FTS_LOADED_CONNS.add(conn)
208
- except Exception:
209
- pass
210
- return True
211
- except Exception:
212
- return False
213
-
214
-
215
- def _try_fts_candidates(
216
- g: LadybugGraph,
217
- query: str,
218
- filter: NodeFilter | None,
219
- path_contains: str | None,
220
- ) -> dict | None:
221
- """Fetch BM25-ranked Symbol candidates via the FTS index; re-apply NodeFilter.
222
-
223
- Returns ``{"rows": [...], "scores": {id: bm25}}`` (rows are the same shape the
224
- heuristic scan yields), or ``None`` when FTS is unavailable (extension won't load,
225
- or the index isn't present on this graph) so the caller falls back.
226
-
227
- Two-step: (1) ``QUERY_FTS_INDEX`` returns the top-K node ids by Okapi BM25 over
228
- ``Symbol.search_text``; (2) re-MATCH those ids with the full ``_lexical_where``
229
- predicates (role / module / path / kind≠file,package) so the filter logic stays
230
- defined in one place. ``search_text`` is built at index time by ``build_ast_graph``
231
- from the same ``_split_identifier`` the re-rank below uses, so index- and query-time
232
- tokenization agree.
233
- """
234
- if not _ensure_fts_loaded(g):
235
- return None
236
- idx_rows = g._rows("CALL SHOW_INDEXES() RETURN index_name") # noqa: SLF001
237
- names = {row.get("index_name") for row in idx_rows}
238
- if SYMBOL_FTS_INDEX not in names:
239
- return None
240
- fts = g._rows( # noqa: SLF001
241
- f"CALL QUERY_FTS_INDEX('Symbol', '{SYMBOL_FTS_INDEX}', $q, top := $k) "
242
- "RETURN node.id AS id, score",
243
- {"q": query, "k": _FTS_CANDIDATE_K},
244
- )
245
- if not fts:
246
- return {"rows": [], "scores": {}}
247
- scores = {row["id"]: float(row.get("score") or 0.0) for row in fts}
248
- ids = list(scores.keys())
249
-
250
- # Re-MATCH the K ids with the SAME predicates the heuristic pushes down, so
251
- # NodeFilter / path / structural-kind filtering is defined exactly once.
252
- where, params = _lexical_where(filter, path_contains=path_contains)
253
- struct_pred = "(s.kind <> 'file' AND s.kind <> 'package')"
254
- if not where:
255
- where = f"WHERE s.id IN $ids AND {struct_pred}"
256
- else:
257
- where = where.replace("WHERE ", f"WHERE s.id IN $ids AND {struct_pred} AND ", 1)
258
- params["ids"] = ids
259
- rows = g._rows(f"MATCH (s:Symbol) {where} RETURN {_SYMBOL_RETURN}", params) # noqa: SLF001
260
- return {"rows": rows, "scores": scores}
261
-
262
-
263
- # Non-underscore alias for cross-module callers (search_lancedb's BM25 fusion on the
264
- # vector path). Behavior is identical to the leading-underscore original.
265
- fetch_fts_candidates = _try_fts_candidates
266
-
267
-
268
- def run_lexical_search(
269
- query: str,
270
- *,
271
- table: str = "java",
272
- limit: int = 5,
273
- offset: int = 0,
274
- path_contains: str | None = None,
275
- filter: NodeFilter | None = None,
276
- explain: bool = False,
277
- dedup: bool = True,
278
- advisories: list[str] | None = None,
279
- graph: LadybugGraph | None = None,
280
- ) -> list[dict]:
281
- """Keyword search over Symbol nodes; returns ``run_search``-shaped row-dicts.
282
-
283
- BM25-first (fork A): when the LadybugDB ``sym_fts`` index exists, candidates are
284
- fetched DB-side via Okapi BM25 over ``Symbol.search_text`` (killing the bounded
285
- Python scan that silently missed matches past the cap on large repos) and then
286
- re-ranked here by the name/type/fqn/role heuristic. The query is pre-split with the
287
- same tokenizer as ``search_text`` so pasted camelCase identifiers match. Falls back
288
- to the heuristic scan when the FTS index or extension is unavailable (older graph,
289
- offline first run), or when the query is degenerate / the BM25 result is empty.
290
-
291
- Raises ``RuntimeError`` (message contains "lexical search unavailable") if no
292
- symbol graph exists — the caller maps that to a clean failure envelope. Returns
293
- ``[]`` for ``table in ("sql", "yaml")`` (those LanceDB tables aren't built in
294
- graph-only mode) and when the graph exists but nothing matches.
295
- """
296
- # sql/yaml LanceDB tables don't exist in graph-only mode.
297
- if table in ("sql", "yaml"):
298
- return []
299
-
300
- if graph is None and not LadybugGraph.exists():
301
- raise RuntimeError(
302
- "lexical search unavailable: no symbol graph found; "
303
- "run `java-codebase-rag init` or `java-codebase-rag reprocess` to build one"
304
- )
305
- g = graph or LadybugGraph.get()
306
-
307
- # --- candidate fetch: BM25 (FTS) preferred, heuristic scan fallback ---
308
- # FTS indexes Symbol.search_text (camelCase-split tokens); pre-split the query the
309
- # same way (build_fts_query) so a pasted identifier like "DistributionChunkService"
310
- # matches — LadybugDB FTS's own tokenizer does not split camelCase. An empty split
311
- # (degenerate / stopword-only query), an unavailable FTS index, OR an empty BM25
312
- # result all fall back to the heuristic scan, which yields role-ranked output for
313
- # degenerate queries and covers selective filters where BM25's top-K thins to nil.
314
- bm25_scores: dict[str, float] = {}
315
- use_fts = False
316
- rows: list[dict] | None = None
317
- q_fts = build_fts_query(query)
318
- if q_fts:
319
- fts = _try_fts_candidates(g, q_fts, filter, path_contains)
320
- if fts is not None and fts["rows"]:
321
- rows = fts["rows"]
322
- bm25_scores = fts["scores"]
323
- use_fts = True
324
- if rows is None:
325
- where, params = _lexical_where(filter, path_contains=path_contains)
326
- # Always exclude structural Symbol nodes. Files and packages are :Symbol-labeled
327
- # (kind='file'/'package') but aren't searchable code declarations — without this
328
- # a token that appears in a filename (e.g. 'distribution' in
329
- # 'DistributionChunkService.java') would surface the file node as a hit.
330
- struct_pred = "(s.kind <> 'file' AND s.kind <> 'package')"
331
- where = f"WHERE {struct_pred}" if not where else where.replace("WHERE ", f"WHERE {struct_pred} AND ", 1)
332
- # The heuristic scan returns rows in storage order — there is NO DB-side relevance
333
- # ORDER BY without the FTS index — so fetch the FULL candidate pool up to the safety
334
- # cap and rank here. A pagination-derived LIMIT (4x the page) is correct on the
335
- # vector path where LanceDB returns rows pre-ranked, but on this unordered scan it
336
- # would return only the first ~N symbols in arbitrary storage order and silently
337
- # miss the best match on any non-trivial repo. The BM25 (FTS) path above has no cap.
338
- params["lim"] = _CANDIDATE_LIMIT_CAP
339
- cypher = f"MATCH (s:Symbol) {where} RETURN {_SYMBOL_RETURN} LIMIT $lim"
340
- rows = g._rows(cypher, params) # noqa: SLF001 — de facto public read API (see find_v2)
341
- # If the fetch hit the safety cap, deeper matches were never ranked. Surface it so
342
- # a user on a large repo isn't silently shown an incomplete result set.
343
- if advisories is not None and len(rows) >= _CANDIDATE_LIMIT_CAP:
344
- advisories.append(
345
- f"lexical search scanned the first {_CANDIDATE_LIMIT_CAP} matching symbols "
346
- "(repo cap); deeper matches were not ranked — refine the query or add a filter"
347
- )
348
-
349
- query_toks = _query_tokens(query)
350
- source_root = _resolve_source_root(g)
351
- role_locked = bool(
352
- filter and (getattr(filter, "role", None) or getattr(filter, "exclude_roles", None))
353
- )
354
-
355
- out: list[dict] = []
356
- for r in rows:
357
- name = str(r.get("name") or "")
358
- fqn = str(r.get("fqn") or "")
359
- type_fqn = _enclosing_type_fqn(fqn)
360
- sig = str(r.get("signature") or "")
361
- anns = list(r.get("annotations") or [])
362
- caps = list(r.get("capabilities") or [])
363
- role_raw = str(r.get("role") or "")
364
-
365
- name_toks = set(_split_identifier(name))
366
- type_toks = set(_split_identifier(type_fqn.rsplit(".", 1)[-1]))
367
- fqn_toks = set(_split_identifier(fqn))
368
- sig_toks = _query_tokens(
369
- " ".join([sig, " ".join(anns), " ".join(caps), str(r.get("package") or "")])
370
- )
371
-
372
- name_overlap = len(query_toks & name_toks)
373
- name_match = (
374
- 1.0
375
- if (query_toks and query_toks <= name_toks)
376
- else (min(name_overlap / max(len(name_toks), 1), 1.0) if name_toks else 0.0)
377
- )
378
- type_hits = len(query_toks & type_toks)
379
- type_match = min(type_hits * _TYPE_MATCH_BONUS_PER_HIT, _TYPE_MATCH_BONUS_CAP)
380
- fqn_match = _token_overlap(fqn_toks, query_toks)
381
- text_overlap = _token_overlap(sig_toks, query_toks)
382
- text_match = text_overlap * _TEXT_MATCH_WEIGHT
383
-
384
- # A keyword search must require at least one lexical hit — role alone never
385
- # qualifies a row (it only boosts/reorders matches). On the BM25 path the FTS
386
- # index already established textual relevance, so the qualifier is heuristic-only.
387
- # Degenerate queries with no usable tokens fall through to role-ranked listing.
388
- if query_toks and not use_fts and not (name_overlap or type_hits or fqn_match or text_overlap):
389
- continue
390
-
391
- role_w = 0.0 if role_locked else _ROLE_SCORE_WEIGHTS.get(role_raw.upper(), 0.0)
392
-
393
- if query_toks:
394
- raw = (
395
- _NAME_MATCH_WEIGHT * name_match
396
- + type_match
397
- + _FQN_MATCH_WEIGHT * fqn_match
398
- + text_match
399
- + role_w
400
- )
401
- else:
402
- raw = role_w # degenerate query: rank by role only
403
- score = _clamp01(raw / LEXICAL_SCORE_MAX)
404
-
405
- comps = {
406
- "name_match": round(name_match, 4),
407
- "type_match": round(type_match, 4),
408
- "fqn_match": round(fqn_match, 4),
409
- "lexical_relevance": round(raw, 4),
410
- "role_weight": role_w,
411
- }
412
- if use_fts:
413
- comps["bm25"] = round(float(bm25_scores.get(r.get("id"), 0.0)), 4)
414
-
415
- sl, el, sb, eb = r.get("start_line"), r.get("end_line"), r.get("start_byte"), r.get("end_byte")
416
- out.append(
417
- {
418
- "_score": score,
419
- "_kind": "java",
420
- "_score_components": comps,
421
- "filename": str(r.get("filename") or ""),
422
- "text": _read_snippet(source_root, str(r.get("filename") or ""), sl, el, sig, fqn),
423
- "primary_type_fqn": type_fqn or None,
424
- # Raw node fqn (members are 'Type#method(...)') feeds
425
- # _node_matches_filter's fqn_contains re-check (mcp_v2.py). Without it the
426
- # post-filter falls back to primary_type_fqn (the bare type) and drops
427
- # member-level matches the Cypher pushdown already accepted.
428
- "fqn": fqn,
429
- "microservice": r.get("microservice"),
430
- "module": r.get("module"),
431
- "role": role_raw or None,
432
- "kind": r.get("kind"),
433
- "symbol_id": r.get("id"),
434
- "annotations": anns,
435
- "capabilities": caps,
436
- "start": {
437
- "line": int(sl) if sl is not None else None,
438
- "byte_offset": int(sb) if sb is not None else 0,
439
- },
440
- "end": {
441
- "line": int(el) if el is not None else None,
442
- "byte_offset": int(eb) if eb is not None else 0,
443
- },
444
- }
445
- )
446
-
447
- out.sort(key=lambda d: float(d.get("_score", 0.0)), reverse=True)
448
- out = _dedup_by_fqn(out, dedup_by_fqn=dedup)
449
- return out[offset : offset + limit]