java-codebase-rag 0.9.4__py3-none-any.whl → 0.9.5__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
search_lexical.py ADDED
@@ -0,0 +1,329 @@
1
+ #!/usr/bin/env python3
2
+ """Lexical (keyword) search over the LadybugDB symbol graph.
3
+
4
+ Graph-only fallback for the `search` tool on macOS Intel installs, where the
5
+ vector stack (lancedb / torch / sentence-transformers) is unavailable (see the
6
+ PEP 508 markers in pyproject.toml). Returns row-dicts in the SAME shape as
7
+ `search_lancedb.run_search`, so `mcp_v2._row_to_search_hit` and the rest of
8
+ `search_v2` work unchanged — `search` simply ranks by keyword relevance instead
9
+ of embeddings, with an advisory noting the mode.
10
+
11
+ This module imports only LadybugDB (always installed) and `search_scoring`
12
+ (dependency-free). It MUST NOT import lancedb/torch, and it MUST NOT import
13
+ `mcp_v2` (circular: mcp_v2 dispatches to this module). The NodeFilter is
14
+ duck-typed; `_lexical_where` mirrors `mcp_v2._symbol_where_from_filter` and is
15
+ guarded by a parity unit test.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import os
21
+ from pathlib import Path
22
+ from typing import TYPE_CHECKING, Any
23
+
24
+ from ladybug_queries import LadybugGraph
25
+ from search_scoring import (
26
+ _ROLE_SCORE_WEIGHTS,
27
+ _TYPE_MATCH_BONUS_CAP,
28
+ _TYPE_MATCH_BONUS_PER_HIT,
29
+ _clamp01,
30
+ _dedup_by_fqn,
31
+ _query_tokens,
32
+ _split_identifier,
33
+ )
34
+
35
+ if TYPE_CHECKING:
36
+ from mcp_v2 import NodeFilter
37
+
38
+ # Lexical relevance weights. The class/file name is the strongest discovery
39
+ # signal (mirrors the type-name bonus rationale in search_scoring); fqn/package
40
+ # overlap is next; signature/annotation/capability text is a weaker corroborator.
41
+ _NAME_MATCH_WEIGHT = 0.45
42
+ _FQN_MATCH_WEIGHT = 0.20
43
+ _TEXT_MATCH_WEIGHT = 0.15
44
+
45
+ # Display-score normalization denominator. Mirrors search_scoring._HYBRID_SCORE_MAX
46
+ # discipline: sum of each additive component's maximum so the displayed score is
47
+ # rank-monotonic in [0, 1]. (= 0.45 + 0.10 + 0.20 + 0.15 + 0.10 = 1.00)
48
+ LEXICAL_SCORE_MAX = (
49
+ _NAME_MATCH_WEIGHT
50
+ + _TYPE_MATCH_BONUS_CAP
51
+ + _FQN_MATCH_WEIGHT
52
+ + _TEXT_MATCH_WEIGHT
53
+ + max(_ROLE_SCORE_WEIGHTS.values())
54
+ )
55
+
56
+ _SNIPPET_MAX_LINES = 20
57
+ _SNIPPET_MAX_CHARS = 800
58
+ # Safety bound on the candidate fetch (full scan over Symbols; bounded so huge
59
+ # repos don't pull unbounded rows into Python).
60
+ _CANDIDATE_LIMIT_CAP = 5000
61
+
62
+ _SYMBOL_RETURN = (
63
+ "s.id AS id, s.kind AS kind, s.name AS name, s.fqn AS fqn, "
64
+ "s.package AS package, s.module AS module, s.microservice AS microservice, "
65
+ "s.filename AS filename, s.start_line AS start_line, s.end_line AS end_line, "
66
+ "s.start_byte AS start_byte, s.end_byte AS end_byte, "
67
+ "s.annotations AS annotations, s.capabilities AS capabilities, "
68
+ "s.role AS role, s.signature AS signature, s.parent_id AS parent_id"
69
+ )
70
+
71
+
72
+ def _lexical_where(f: Any, *, path_contains: str | None) -> tuple[str, dict[str, Any]]:
73
+ """Cypher WHERE for Symbol nodes from a NodeFilter (+ path_contains pushdown).
74
+
75
+ Mirrors ``mcp_v2._symbol_where_from_filter``; kept local so this module stays
76
+ import-isolated from mcp_v2 (and unit-testable standalone). ``path_contains``
77
+ is pushed down into Cypher because ``search_v2`` only re-filters the windowed
78
+ page post-fetch — without pushdown a path filter could empty the page even
79
+ when deeper-ranked rows match. A parity unit test guards drift.
80
+ """
81
+ preds: list[str] = []
82
+ params: dict[str, Any] = {}
83
+ if f is not None:
84
+ if getattr(f, "microservice", None):
85
+ preds.append("s.microservice = $microservice")
86
+ params["microservice"] = f.microservice
87
+ if getattr(f, "module", None):
88
+ preds.append("s.module = $module")
89
+ params["module"] = f.module
90
+ if getattr(f, "role", None):
91
+ preds.append("s.role = $role")
92
+ params["role"] = f.role
93
+ if getattr(f, "exclude_roles", None):
94
+ preds.append("NOT s.role IN $exclude_roles")
95
+ params["exclude_roles"] = list(f.exclude_roles)
96
+ if getattr(f, "generated_only", False):
97
+ preds.append("s.generated = true")
98
+ if getattr(f, "exclude_generated", False):
99
+ preds.append("(s.generated IS NULL OR s.generated = false)")
100
+ if getattr(f, "annotation", None):
101
+ preds.append("list_contains(s.annotations, $annotation)")
102
+ params["annotation"] = f.annotation
103
+ if getattr(f, "capability", None):
104
+ preds.append("$capability IN s.capabilities")
105
+ params["capability"] = f.capability
106
+ if getattr(f, "fqn_contains", None):
107
+ preds.append("s.fqn CONTAINS $fqn_contains")
108
+ params["fqn_contains"] = f.fqn_contains
109
+ if getattr(f, "symbol_kind", None):
110
+ preds.append("s.kind = $symbol_kind")
111
+ params["symbol_kind"] = f.symbol_kind
112
+ if getattr(f, "symbol_kinds", None):
113
+ preds.append("s.kind IN $symbol_kinds")
114
+ params["symbol_kinds"] = list(f.symbol_kinds)
115
+ if path_contains:
116
+ preds.append("s.filename CONTAINS $path_contains")
117
+ params["path_contains"] = path_contains
118
+ where = f"WHERE {' AND '.join(preds)}" if preds else ""
119
+ return where, params
120
+
121
+
122
+ def _enclosing_type_fqn(fqn: str) -> str:
123
+ """Member fqn ``{parent_fqn}#{signature}`` -> parent type fqn; type fqn (no '#') unchanged."""
124
+ return fqn.split("#", 1)[0] if fqn else fqn
125
+
126
+
127
+ def _resolve_source_root(graph: LadybugGraph) -> str:
128
+ """Authoritative source root is the one cached on the graph at index time."""
129
+ try:
130
+ root = str(graph.meta().get("source_root") or "")
131
+ except Exception:
132
+ root = ""
133
+ return root or os.environ.get("JAVA_CODEBASE_RAG_SOURCE_ROOT", "").strip()
134
+
135
+
136
+ def _read_snippet(
137
+ source_root: str, filename: str, start_line: Any, end_line: Any, signature: str, fqn: str
138
+ ) -> str:
139
+ """Real source snippet for [start_line, end_line] from disk (capped); synthesized
140
+ from `signature` (fallback `fqn`) on any failure."""
141
+ synth = (signature or "").strip() or fqn
142
+ try:
143
+ sl = int(start_line) if start_line else 0
144
+ el = int(end_line) if end_line else sl
145
+ except (TypeError, ValueError):
146
+ return synth
147
+ if sl <= 0 or not filename:
148
+ return synth
149
+ try:
150
+ p = Path(filename)
151
+ full = p if p.is_absolute() else (Path(source_root) / filename)
152
+ if not p.is_absolute() and not source_root:
153
+ return synth
154
+ text = full.read_text(encoding="utf-8", errors="replace")
155
+ except OSError:
156
+ return synth
157
+ lines = text.splitlines()
158
+ lo = max(sl - 1, 0)
159
+ hi = min(el if el >= sl else sl, lo + _SNIPPET_MAX_LINES)
160
+ chunk = "\n".join(lines[lo:hi]).strip()
161
+ if len(chunk) > _SNIPPET_MAX_CHARS:
162
+ chunk = chunk[: _SNIPPET_MAX_CHARS - 1] + "…"
163
+ return chunk or synth
164
+
165
+
166
+ def _token_overlap(haystack_toks: set[str], needle_toks: set[str]) -> float:
167
+ """Fraction of needle tokens present in haystack (0..1)."""
168
+ if not needle_toks:
169
+ return 0.0
170
+ return len(needle_toks & haystack_toks) / len(needle_toks)
171
+
172
+
173
+ def run_lexical_search(
174
+ query: str,
175
+ *,
176
+ table: str = "java",
177
+ limit: int = 5,
178
+ offset: int = 0,
179
+ path_contains: str | None = None,
180
+ filter: NodeFilter | None = None,
181
+ explain: bool = False,
182
+ dedup: bool = True,
183
+ advisories: list[str] | None = None,
184
+ graph: LadybugGraph | None = None,
185
+ ) -> list[dict]:
186
+ """Keyword search over Symbol nodes; returns ``run_search``-shaped row-dicts.
187
+
188
+ Raises ``RuntimeError`` (message contains "lexical search unavailable") if no
189
+ symbol graph exists — the caller maps that to a clean failure envelope. Returns
190
+ ``[]`` for ``table in ("sql", "yaml")`` (those LanceDB tables aren't built in
191
+ graph-only mode) and when the graph exists but nothing matches.
192
+ """
193
+ # sql/yaml LanceDB tables don't exist in graph-only mode.
194
+ if table in ("sql", "yaml"):
195
+ return []
196
+
197
+ if graph is None and not LadybugGraph.exists():
198
+ raise RuntimeError(
199
+ "lexical search unavailable: no symbol graph found; "
200
+ "run `java-codebase-rag init` or `java-codebase-rag reprocess` to build one"
201
+ )
202
+ g = graph or LadybugGraph.get()
203
+
204
+ where, params = _lexical_where(filter, path_contains=path_contains)
205
+ # Always exclude structural Symbol nodes. Files and packages are :Symbol-labeled
206
+ # (kind='file'/'package') but aren't searchable code declarations — without this
207
+ # a token that appears in a filename (e.g. 'distribution' in
208
+ # 'DistributionChunkService.java') would surface the file node as a hit.
209
+ struct_pred = "(s.kind <> 'file' AND s.kind <> 'package')"
210
+ where = f"WHERE {struct_pred}" if not where else where.replace("WHERE ", f"WHERE {struct_pred} AND ", 1)
211
+ # Lexical ranking is done in Python (LadybugDB/kuzu has no keyword ranking without
212
+ # FTS5, which is deferred), and the MATCH scan returns rows in storage order — there
213
+ # is NO DB-side relevance ORDER BY. So fetch the FULL candidate pool up to the safety
214
+ # cap and rank here. A pagination-derived LIMIT (4x the page) is correct on the vector
215
+ # path where LanceDB returns rows pre-ranked by similarity, but on this unordered scan
216
+ # it would return only the first ~N symbols in arbitrary storage order and silently
217
+ # miss the best match on any non-trivial repo.
218
+ params["lim"] = _CANDIDATE_LIMIT_CAP
219
+ cypher = f"MATCH (s:Symbol) {where} RETURN {_SYMBOL_RETURN} LIMIT $lim"
220
+ rows = g._rows(cypher, params) # noqa: SLF001 — de facto public read API (see find_v2)
221
+ # If the fetch hit the safety cap, deeper matches were never ranked (the scan has no
222
+ # ORDER BY — kuzu returns an arbitrary, storage-order-dependent subset). Surface it so
223
+ # a user on a large repo isn't silently shown an incomplete result set; refining the
224
+ # query or adding a filter narrows the pool below the cap. Raising the cap / FTS5 is
225
+ # the deferred long-term fix (see the plan's "Out of scope" note).
226
+ if advisories is not None and len(rows) >= _CANDIDATE_LIMIT_CAP:
227
+ advisories.append(
228
+ f"lexical search scanned the first {_CANDIDATE_LIMIT_CAP} matching symbols "
229
+ "(repo cap); deeper matches were not ranked — refine the query or add a filter"
230
+ )
231
+
232
+ query_toks = _query_tokens(query)
233
+ source_root = _resolve_source_root(g)
234
+ role_locked = bool(
235
+ filter and (getattr(filter, "role", None) or getattr(filter, "exclude_roles", None))
236
+ )
237
+
238
+ out: list[dict] = []
239
+ for r in rows:
240
+ name = str(r.get("name") or "")
241
+ fqn = str(r.get("fqn") or "")
242
+ type_fqn = _enclosing_type_fqn(fqn)
243
+ sig = str(r.get("signature") or "")
244
+ anns = list(r.get("annotations") or [])
245
+ caps = list(r.get("capabilities") or [])
246
+ role_raw = str(r.get("role") or "")
247
+
248
+ name_toks = set(_split_identifier(name))
249
+ type_toks = set(_split_identifier(type_fqn.rsplit(".", 1)[-1]))
250
+ fqn_toks = set(_split_identifier(fqn))
251
+ sig_toks = _query_tokens(
252
+ " ".join([sig, " ".join(anns), " ".join(caps), str(r.get("package") or "")])
253
+ )
254
+
255
+ name_overlap = len(query_toks & name_toks)
256
+ name_match = (
257
+ 1.0
258
+ if (query_toks and query_toks <= name_toks)
259
+ else (min(name_overlap / max(len(name_toks), 1), 1.0) if name_toks else 0.0)
260
+ )
261
+ type_hits = len(query_toks & type_toks)
262
+ type_match = min(type_hits * _TYPE_MATCH_BONUS_PER_HIT, _TYPE_MATCH_BONUS_CAP)
263
+ fqn_match = _token_overlap(fqn_toks, query_toks)
264
+ text_overlap = _token_overlap(sig_toks, query_toks)
265
+ text_match = text_overlap * _TEXT_MATCH_WEIGHT
266
+
267
+ # A keyword search must require at least one lexical hit — role alone never
268
+ # qualifies a row (it only boosts/reorders matches). Degenerate queries with
269
+ # no usable tokens fall through to role-ranked listing.
270
+ if query_toks and not (name_overlap or type_hits or fqn_match or text_overlap):
271
+ continue
272
+
273
+ role_w = 0.0 if role_locked else _ROLE_SCORE_WEIGHTS.get(role_raw.upper(), 0.0)
274
+
275
+ if query_toks:
276
+ raw = (
277
+ _NAME_MATCH_WEIGHT * name_match
278
+ + type_match
279
+ + _FQN_MATCH_WEIGHT * fqn_match
280
+ + text_match
281
+ + role_w
282
+ )
283
+ else:
284
+ raw = role_w # degenerate query: rank by role only
285
+ score = _clamp01(raw / LEXICAL_SCORE_MAX)
286
+
287
+ comps = {
288
+ "name_match": round(name_match, 4),
289
+ "type_match": round(type_match, 4),
290
+ "fqn_match": round(fqn_match, 4),
291
+ "lexical_relevance": round(raw, 4),
292
+ "role_weight": role_w,
293
+ }
294
+
295
+ sl, el, sb, eb = r.get("start_line"), r.get("end_line"), r.get("start_byte"), r.get("end_byte")
296
+ out.append(
297
+ {
298
+ "_score": score,
299
+ "_kind": "java",
300
+ "_score_components": comps,
301
+ "filename": str(r.get("filename") or ""),
302
+ "text": _read_snippet(source_root, str(r.get("filename") or ""), sl, el, sig, fqn),
303
+ "primary_type_fqn": type_fqn or None,
304
+ # Raw node fqn (members are 'Type#method(...)') feeds
305
+ # _node_matches_filter's fqn_contains re-check (mcp_v2.py). Without it the
306
+ # post-filter falls back to primary_type_fqn (the bare type) and drops
307
+ # member-level matches the Cypher pushdown already accepted.
308
+ "fqn": fqn,
309
+ "microservice": r.get("microservice"),
310
+ "module": r.get("module"),
311
+ "role": role_raw or None,
312
+ "kind": r.get("kind"),
313
+ "symbol_id": r.get("id"),
314
+ "annotations": anns,
315
+ "capabilities": caps,
316
+ "start": {
317
+ "line": int(sl) if sl is not None else None,
318
+ "byte_offset": int(sb) if sb is not None else 0,
319
+ },
320
+ "end": {
321
+ "line": int(el) if el is not None else None,
322
+ "byte_offset": int(eb) if eb is not None else 0,
323
+ },
324
+ }
325
+ )
326
+
327
+ out.sort(key=lambda d: float(d.get("_score", 0.0)), reverse=True)
328
+ out = _dedup_by_fqn(out, dedup_by_fqn=dedup)
329
+ return out[offset : offset + limit]
search_scoring.py ADDED
@@ -0,0 +1,338 @@
1
+ #!/usr/bin/env python3
2
+ """Dependency-free scoring & dedup primitives shared by the search backends.
3
+
4
+ Imported by both the vector backend (`search_lancedb`) and the lexical backend
5
+ (`search_lexical`). This module MUST NOT import lancedb / torch /
6
+ sentence_transformers / cocoindex — it is imported on graph-only (macOS Intel)
7
+ installs where those packages are absent (see pyproject.toml PEP 508 markers).
8
+
9
+ Everything here is pure-Python dict/list math with no third-party deps.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import json
15
+
16
+ # Over-fetch multiplier for dedup: fetch 4x to absorb per-FQN chunk multiplicity
17
+ # so that after collapsing by primary_type_fqn, a page stays full and the +1
18
+ # truncation sentinel survives. The formula: need = max((limit + offset) * 4, limit + offset + 1)
19
+ DEDUP_OVERFETCH = 4
20
+
21
+ _IMPORT_DISTANCE_PENALTY = 0.08
22
+ _IMPORT_HYBRID_SCORE_FACTOR = 0.88
23
+
24
+ # Bonus for chunks whose declared symbols (method / field names) share tokens with
25
+ # the query. Behavioural queries like "what happens when a client message arrives"
26
+ # should float chunks containing `processClientMessage` above ones that only
27
+ # enqueue; this is a cheap, query-dependent signal computed at rank time.
28
+ _SYMBOL_MATCH_BONUS_PER_HIT = 0.03
29
+ _SYMBOL_MATCH_BONUS_CAP = 0.06
30
+
31
+ # Action verbs that typically mark behavioural entry points in this codebase.
32
+ # A chunk whose symbols begin with one of these verbs earns a small flat bump
33
+ # — again only for java chunks and only when role-filtering is off.
34
+ _ACTION_VERB_PREFIXES: tuple[str, ...] = (
35
+ "process", "handle", "on", "pick", "select", "assign",
36
+ "notify", "dispatch", "publish", "consume", "route",
37
+ "trigger", "enqueue", "distribute", "update", "create",
38
+ "apply", "resolve", "reassign", "close", "open",
39
+ )
40
+ _ACTION_VERB_BONUS = 0.02
41
+
42
+ # Type-name overlap bonus. The class name is a much stronger discovery signal
43
+ # than any individual method, because class naming in this codebase encodes
44
+ # the domain concept (`DistributionChunkService`, `OperatorSessionService`,
45
+ # `JoinOperatorController`). So we reward overlap between query tokens and the
46
+ # simple name of `primary_type_fqn` more heavily than per-method overlap, and
47
+ # we stack it on top of the existing `_symbol_bonus`.
48
+ _TYPE_MATCH_BONUS_PER_HIT = 0.05
49
+ _TYPE_MATCH_BONUS_CAP = 0.10
50
+
51
+ _STOPWORDS: frozenset[str] = frozenset({
52
+ "a", "an", "the", "is", "are", "was", "were", "be", "been", "being",
53
+ "to", "of", "in", "on", "at", "by", "for", "with", "from", "as", "or",
54
+ "and", "but", "if", "then", "else", "when", "what", "how", "why", "does",
55
+ "do", "did", "has", "have", "had", "this", "that", "these", "those", "it",
56
+ "its", "new", "no", "not", "will", "would", "should", "can", "could",
57
+ "may", "might", "happens", "happen", "happened", "get", "gets", "got",
58
+ })
59
+
60
+ # Role-aware reweighting for Java chunks. Positive values favour actionable
61
+ # behavioural code (entrypoints, orchestrators, integrations) over configuration,
62
+ # schema, and persistence stubs for "what happens when..."-style queries.
63
+ # Applied to the similarity score (higher = better); distance-based sort subtracts
64
+ # the weight. Skipped when caller filters explicitly by role.
65
+ _ROLE_SCORE_WEIGHTS: dict[str, float] = {
66
+ "CONTROLLER": 0.10,
67
+ "SERVICE": 0.08,
68
+ "CLIENT": 0.06,
69
+ "COMPONENT": 0.03,
70
+ "REPOSITORY": 0.02,
71
+ "MAPPER": 0.00,
72
+ "OTHER": 0.00,
73
+ "ENTITY": -0.06,
74
+ "CONFIG": -0.10,
75
+ # DTOs are passive data carriers; they almost never answer "how/what
76
+ # happens" queries. Penalty is slightly stronger than ENTITY so a DTO
77
+ # with a great embedding match still loses to a mediocre SERVICE hit.
78
+ "DTO": -0.08,
79
+ }
80
+
81
+ # Theoretical maximum for hybrid composite score (used for display normalization).
82
+ # Hybrid sort metric: raw_rrf * (import_factor if import_heavy else 1)
83
+ # + role_weight + symbol_bonus
84
+ # where raw_rrf ≤ 2/(k+1) for 2-list RRF, role_weight ≤ max(_ROLE_SCORE_WEIGHTS),
85
+ # and symbol_bonus ≤ _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS.
86
+ # The import factor is ≤ 1, so we use the raw max (2/61).
87
+ _HYBRID_SCORE_MAX = (2.0 / 61.0) + max(_ROLE_SCORE_WEIGHTS.values()) + _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS
88
+
89
+
90
+ def _query_tokens(query: str) -> set[str]:
91
+ """Lowercased alpha-only tokens from the query, minus stopwords, len >= 3.
92
+
93
+ Used to score symbol-name overlap; we keep it simple and locale-free.
94
+ """
95
+ out: set[str] = set()
96
+ cur: list[str] = []
97
+
98
+ def _flush() -> None:
99
+ if cur:
100
+ tok = "".join(cur).lower()
101
+ cur.clear()
102
+ if len(tok) >= 3 and tok not in _STOPWORDS:
103
+ out.add(tok)
104
+
105
+ for c in query:
106
+ if c.isalpha():
107
+ cur.append(c)
108
+ else:
109
+ _flush()
110
+ _flush()
111
+ return out
112
+
113
+
114
+ def _split_identifier(name: str) -> list[str]:
115
+ """camelCase / snake_case -> lowercase token list."""
116
+ parts: list[str] = []
117
+ cur: list[str] = []
118
+ for c in name:
119
+ if c == "_":
120
+ if cur:
121
+ parts.append("".join(cur).lower())
122
+ cur = []
123
+ elif c.isupper() and cur:
124
+ parts.append("".join(cur).lower())
125
+ cur = [c]
126
+ else:
127
+ cur.append(c)
128
+ if cur:
129
+ parts.append("".join(cur).lower())
130
+ return [p for p in parts if p]
131
+
132
+
133
+ def _symbol_bonus(r: dict, query_toks: set[str]) -> float:
134
+ """Symbol-name overlap + action-verb bump for java chunks.
135
+
136
+ Caps at `_SYMBOL_MATCH_BONUS_CAP + _ACTION_VERB_BONUS` to avoid runaway
137
+ ranks on chunks declaring many symbols.
138
+ """
139
+ if str(r.get("_kind", "")) != "java":
140
+ return 0.0
141
+ raw = r.get("symbols") or []
142
+ if isinstance(raw, str):
143
+ # Legacy JSON-encoded list column; parse defensively.
144
+ try:
145
+ parsed = json.loads(raw)
146
+ raw = parsed if isinstance(parsed, list) else []
147
+ except Exception:
148
+ raw = []
149
+ symbols = [str(s) for s in raw if s]
150
+
151
+ overlap_hits = 0
152
+ has_action = False
153
+ for s in symbols:
154
+ bare = s.split("(", 1)[0].strip()
155
+ if not bare:
156
+ continue
157
+ toks = _split_identifier(bare)
158
+ if toks:
159
+ if toks[0] in _ACTION_VERB_PREFIXES:
160
+ has_action = True
161
+ if query_toks & set(toks):
162
+ overlap_hits += 1
163
+
164
+ bonus = min(overlap_hits * _SYMBOL_MATCH_BONUS_PER_HIT, _SYMBOL_MATCH_BONUS_CAP)
165
+ if has_action:
166
+ bonus += _ACTION_VERB_BONUS
167
+
168
+ # Type-name overlap: strongest single lexical signal for "which class is
169
+ # the answer?" queries. Uses the simple name of primary_type_fqn.
170
+ fqn = str(r.get("primary_type_fqn") or "")
171
+ if fqn:
172
+ simple = fqn.rsplit(".", 1)[-1]
173
+ type_toks = set(_split_identifier(simple))
174
+ type_hits = len(query_toks & type_toks)
175
+ if type_hits:
176
+ bonus += min(type_hits * _TYPE_MATCH_BONUS_PER_HIT, _TYPE_MATCH_BONUS_CAP)
177
+ return bonus
178
+
179
+
180
+ def _role_weight(r: dict) -> float:
181
+ """Effective role weight for a row, captured into `_score_components.role_weight`."""
182
+ comps = r.setdefault("_score_components", {})
183
+ cached = comps.get("role_weight")
184
+ if cached is not None:
185
+ return float(cached)
186
+ if r.get("_skip_role_weight") or str(r.get("_kind", "")) != "java":
187
+ comps["role_weight"] = 0.0
188
+ return 0.0
189
+ role = (r.get("role") or "").upper()
190
+ w = _ROLE_SCORE_WEIGHTS.get(role, 0.0)
191
+ comps["role_weight"] = w
192
+ return w
193
+
194
+
195
+ def _apply_symbol_bonus(rows: list[dict], query_toks: set[str]) -> None:
196
+ """Pre-compute symbol-match bonus into `_score_components.symbol_bonus`."""
197
+ if not query_toks:
198
+ return
199
+ for r in rows:
200
+ if r.get("_skip_role_weight"):
201
+ # When the caller locked role, respect their intent everywhere.
202
+ continue
203
+ b = _symbol_bonus(r, query_toks)
204
+ if b:
205
+ r.setdefault("_score_components", {})["symbol_bonus"] = b
206
+
207
+
208
+ def l2_distance_to_score(distance: float) -> float:
209
+ """Map L2 distance to a similarity score for unit-normalized embeddings."""
210
+ return 1.0 - distance * distance / 2.0
211
+
212
+
213
+ def _effective_distance(comps: dict[str, float]) -> float:
214
+ """Compute the adjusted distance used for sorting.
215
+
216
+ Matches _vector_sort_key logic: distance + import_penalty - role_weight - symbol_bonus.
217
+ """
218
+ d = comps.get("distance", 0.0)
219
+ d += comps.get("import_penalty", 0.0)
220
+ d -= comps.get("role_weight", 0.0)
221
+ d -= comps.get("symbol_bonus", 0.0)
222
+ return d
223
+
224
+
225
+ def _clamp01(x: float) -> float:
226
+ """Clamp a value to the [0.0, 1.0] range."""
227
+ if x < 0.0:
228
+ return 0.0
229
+ if x > 1.0:
230
+ return 1.0
231
+ return x
232
+
233
+
234
+ def explain_score_components(
235
+ comps: dict[str, float] | None,
236
+ *,
237
+ role: str | None = None,
238
+ hybrid: bool = False,
239
+ graph_expanded: bool = False,
240
+ lexical: bool = False,
241
+ ) -> str:
242
+ """Compact human-readable 'why' string for a ranked hit.
243
+
244
+ Joins the interesting components of `_score_components` in a stable order
245
+ so agents can reason about rankings without chasing raw floats. Returns
246
+ "" if there's nothing worth mentioning.
247
+ """
248
+ if not comps:
249
+ comps = {}
250
+ parts: list[str] = []
251
+ if lexical:
252
+ rel = comps.get("lexical_relevance")
253
+ if rel is not None:
254
+ parts.append(f"relevance={float(rel):.2f}")
255
+ nm = comps.get("name_match")
256
+ if nm is not None:
257
+ parts.append(f"name={float(nm):.2f}")
258
+ ty = comps.get("type_match")
259
+ if ty:
260
+ parts.append(f"type:{float(ty):+.2f}")
261
+ fq = comps.get("fqn_match")
262
+ if fq:
263
+ parts.append(f"fqn:{float(fq):+.2f}")
264
+ elif hybrid:
265
+ # Prefer rrf_raw (added by PR-SEARCH-1a) for explanation
266
+ rrf = comps.get("rrf_raw") or comps.get("hybrid_rrf")
267
+ if rrf is not None:
268
+ parts.append(f"rrf={float(rrf):.3f}")
269
+ else:
270
+ d = comps.get("distance")
271
+ if d is not None:
272
+ parts.append(f"dist={float(d):.2f}")
273
+ rw = comps.get("role_weight")
274
+ if rw:
275
+ label = f"role:{role}" if role else "role"
276
+ parts.append(f"{label}:{float(rw):+.02f}")
277
+ sb = comps.get("symbol_bonus")
278
+ if sb:
279
+ parts.append(f"symbol:{float(sb):+.02f}")
280
+ ip = comps.get("import_penalty")
281
+ if ip:
282
+ parts.append(f"import_penalty:{float(ip):+.02f}")
283
+ if graph_expanded:
284
+ parts.append("graph")
285
+ return " ".join(parts)
286
+
287
+
288
+ def _dedup_by_fqn(rows: list[dict], dedup_by_fqn: bool = True) -> list[dict]:
289
+ """Deduplicate rows by primary_type_fqn (java table only).
290
+
291
+ When dedup_by_fqn is True, collapses multiple chunks of the same
292
+ primary_type_fqn into one row (first-seen-wins, since rows are pre-sorted
293
+ so the first is the best chunk). Each survivor gets a _chunks_collapsed
294
+ field (>=1) counting how many rows were collapsed into it.
295
+
296
+ Rows without primary_type_fqn (sql/yaml tables) get a unique __id:<id>
297
+ key so they pass through unchanged (each row is unique).
298
+
299
+ When dedup_by_fqn is False, returns rows unchanged (regression guard).
300
+ """
301
+ if not dedup_by_fqn:
302
+ # Non-dedup path: return unchanged, byte-identical to prior behavior
303
+ return rows
304
+
305
+ deduped: list[dict] = []
306
+ seen_keys: dict[str, dict] = {}
307
+ collapsed_counts: dict[str, int] = {}
308
+
309
+ for row in rows:
310
+ # Build dedup key: primary_type_fqn for java rows, unique __id:<id> for sql/yaml
311
+ fqn = row.get("primary_type_fqn")
312
+ if fqn:
313
+ key = str(fqn)
314
+ else:
315
+ # sql/yaml rows have no primary_type_fqn → unique key per row
316
+ row_id = row.get("id") or id(row)
317
+ key = f"__id:{row_id}"
318
+
319
+ if key not in seen_keys:
320
+ # First occurrence: keep it
321
+ seen_keys[key] = row
322
+ collapsed_counts[key] = 1
323
+ deduped.append(row)
324
+ else:
325
+ # Duplicate: increment collapse count, discard this row
326
+ collapsed_counts[key] += 1
327
+
328
+ # Annotate each survivor with _chunks_collapsed
329
+ for row in deduped:
330
+ fqn = row.get("primary_type_fqn")
331
+ if fqn:
332
+ key = str(fqn)
333
+ else:
334
+ row_id = row.get("id") or id(row)
335
+ key = f"__id:{row_id}"
336
+ row["_chunks_collapsed"] = collapsed_counts[key]
337
+
338
+ return deduped