java-codebase-rag 0.9.4__py3-none-any.whl → 0.9.5__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- absence_diagnosis.py +700 -0
- absence_types.py +124 -0
- absence_vocab.py +455 -0
- ast_java.py +1 -1
- build_ast_graph.py +85 -7
- graph_enrich.py +245 -0
- graph_types.py +4 -0
- java_codebase_rag/cli.py +1 -6
- java_codebase_rag/config.py +116 -0
- java_codebase_rag/jrag.py +56 -1
- java_codebase_rag/jrag_envelope.py +10 -1
- java_codebase_rag/jrag_render.py +67 -3
- java_codebase_rag/pipeline.py +10 -0
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.5.dist-info}/METADATA +2 -2
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.5.dist-info}/RECORD +27 -22
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.5.dist-info}/top_level.txt +5 -0
- java_index_flow_lancedb.py +27 -7
- ladybug_queries.py +4 -4
- mcp_v2.py +272 -73
- resolve_service.py +69 -2
- search_lancedb.py +50 -311
- search_lexical.py +329 -0
- search_scoring.py +338 -0
- server.py +101 -34
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.5.dist-info}/WHEEL +0 -0
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.5.dist-info}/entry_points.txt +0 -0
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.5.dist-info}/licenses/LICENSE +0 -0
search_lexical.py
ADDED
|
@@ -0,0 +1,329 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Lexical (keyword) search over the LadybugDB symbol graph.
|
|
3
|
+
|
|
4
|
+
Graph-only fallback for the `search` tool on macOS Intel installs, where the
|
|
5
|
+
vector stack (lancedb / torch / sentence-transformers) is unavailable (see the
|
|
6
|
+
PEP 508 markers in pyproject.toml). Returns row-dicts in the SAME shape as
|
|
7
|
+
`search_lancedb.run_search`, so `mcp_v2._row_to_search_hit` and the rest of
|
|
8
|
+
`search_v2` work unchanged — `search` simply ranks by keyword relevance instead
|
|
9
|
+
of embeddings, with an advisory noting the mode.
|
|
10
|
+
|
|
11
|
+
This module imports only LadybugDB (always installed) and `search_scoring`
|
|
12
|
+
(dependency-free). It MUST NOT import lancedb/torch, and it MUST NOT import
|
|
13
|
+
`mcp_v2` (circular: mcp_v2 dispatches to this module). The NodeFilter is
|
|
14
|
+
duck-typed; `_lexical_where` mirrors `mcp_v2._symbol_where_from_filter` and is
|
|
15
|
+
guarded by a parity unit test.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import os
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import TYPE_CHECKING, Any
|
|
23
|
+
|
|
24
|
+
from ladybug_queries import LadybugGraph
|
|
25
|
+
from search_scoring import (
|
|
26
|
+
_ROLE_SCORE_WEIGHTS,
|
|
27
|
+
_TYPE_MATCH_BONUS_CAP,
|
|
28
|
+
_TYPE_MATCH_BONUS_PER_HIT,
|
|
29
|
+
_clamp01,
|
|
30
|
+
_dedup_by_fqn,
|
|
31
|
+
_query_tokens,
|
|
32
|
+
_split_identifier,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
if TYPE_CHECKING:
|
|
36
|
+
from mcp_v2 import NodeFilter
|
|
37
|
+
|
|
38
|
+
# Lexical relevance weights. The class/file name is the strongest discovery
|
|
39
|
+
# signal (mirrors the type-name bonus rationale in search_scoring); fqn/package
|
|
40
|
+
# overlap is next; signature/annotation/capability text is a weaker corroborator.
|
|
41
|
+
_NAME_MATCH_WEIGHT = 0.45
|
|
42
|
+
_FQN_MATCH_WEIGHT = 0.20
|
|
43
|
+
_TEXT_MATCH_WEIGHT = 0.15
|
|
44
|
+
|
|
45
|
+
# Display-score normalization denominator. Mirrors search_scoring._HYBRID_SCORE_MAX
|
|
46
|
+
# discipline: sum of each additive component's maximum so the displayed score is
|
|
47
|
+
# rank-monotonic in [0, 1]. (= 0.45 + 0.10 + 0.20 + 0.15 + 0.10 = 1.00)
|
|
48
|
+
LEXICAL_SCORE_MAX = (
|
|
49
|
+
_NAME_MATCH_WEIGHT
|
|
50
|
+
+ _TYPE_MATCH_BONUS_CAP
|
|
51
|
+
+ _FQN_MATCH_WEIGHT
|
|
52
|
+
+ _TEXT_MATCH_WEIGHT
|
|
53
|
+
+ max(_ROLE_SCORE_WEIGHTS.values())
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
_SNIPPET_MAX_LINES = 20
|
|
57
|
+
_SNIPPET_MAX_CHARS = 800
|
|
58
|
+
# Safety bound on the candidate fetch (full scan over Symbols; bounded so huge
|
|
59
|
+
# repos don't pull unbounded rows into Python).
|
|
60
|
+
_CANDIDATE_LIMIT_CAP = 5000
|
|
61
|
+
|
|
62
|
+
_SYMBOL_RETURN = (
|
|
63
|
+
"s.id AS id, s.kind AS kind, s.name AS name, s.fqn AS fqn, "
|
|
64
|
+
"s.package AS package, s.module AS module, s.microservice AS microservice, "
|
|
65
|
+
"s.filename AS filename, s.start_line AS start_line, s.end_line AS end_line, "
|
|
66
|
+
"s.start_byte AS start_byte, s.end_byte AS end_byte, "
|
|
67
|
+
"s.annotations AS annotations, s.capabilities AS capabilities, "
|
|
68
|
+
"s.role AS role, s.signature AS signature, s.parent_id AS parent_id"
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _lexical_where(f: Any, *, path_contains: str | None) -> tuple[str, dict[str, Any]]:
|
|
73
|
+
"""Cypher WHERE for Symbol nodes from a NodeFilter (+ path_contains pushdown).
|
|
74
|
+
|
|
75
|
+
Mirrors ``mcp_v2._symbol_where_from_filter``; kept local so this module stays
|
|
76
|
+
import-isolated from mcp_v2 (and unit-testable standalone). ``path_contains``
|
|
77
|
+
is pushed down into Cypher because ``search_v2`` only re-filters the windowed
|
|
78
|
+
page post-fetch — without pushdown a path filter could empty the page even
|
|
79
|
+
when deeper-ranked rows match. A parity unit test guards drift.
|
|
80
|
+
"""
|
|
81
|
+
preds: list[str] = []
|
|
82
|
+
params: dict[str, Any] = {}
|
|
83
|
+
if f is not None:
|
|
84
|
+
if getattr(f, "microservice", None):
|
|
85
|
+
preds.append("s.microservice = $microservice")
|
|
86
|
+
params["microservice"] = f.microservice
|
|
87
|
+
if getattr(f, "module", None):
|
|
88
|
+
preds.append("s.module = $module")
|
|
89
|
+
params["module"] = f.module
|
|
90
|
+
if getattr(f, "role", None):
|
|
91
|
+
preds.append("s.role = $role")
|
|
92
|
+
params["role"] = f.role
|
|
93
|
+
if getattr(f, "exclude_roles", None):
|
|
94
|
+
preds.append("NOT s.role IN $exclude_roles")
|
|
95
|
+
params["exclude_roles"] = list(f.exclude_roles)
|
|
96
|
+
if getattr(f, "generated_only", False):
|
|
97
|
+
preds.append("s.generated = true")
|
|
98
|
+
if getattr(f, "exclude_generated", False):
|
|
99
|
+
preds.append("(s.generated IS NULL OR s.generated = false)")
|
|
100
|
+
if getattr(f, "annotation", None):
|
|
101
|
+
preds.append("list_contains(s.annotations, $annotation)")
|
|
102
|
+
params["annotation"] = f.annotation
|
|
103
|
+
if getattr(f, "capability", None):
|
|
104
|
+
preds.append("$capability IN s.capabilities")
|
|
105
|
+
params["capability"] = f.capability
|
|
106
|
+
if getattr(f, "fqn_contains", None):
|
|
107
|
+
preds.append("s.fqn CONTAINS $fqn_contains")
|
|
108
|
+
params["fqn_contains"] = f.fqn_contains
|
|
109
|
+
if getattr(f, "symbol_kind", None):
|
|
110
|
+
preds.append("s.kind = $symbol_kind")
|
|
111
|
+
params["symbol_kind"] = f.symbol_kind
|
|
112
|
+
if getattr(f, "symbol_kinds", None):
|
|
113
|
+
preds.append("s.kind IN $symbol_kinds")
|
|
114
|
+
params["symbol_kinds"] = list(f.symbol_kinds)
|
|
115
|
+
if path_contains:
|
|
116
|
+
preds.append("s.filename CONTAINS $path_contains")
|
|
117
|
+
params["path_contains"] = path_contains
|
|
118
|
+
where = f"WHERE {' AND '.join(preds)}" if preds else ""
|
|
119
|
+
return where, params
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _enclosing_type_fqn(fqn: str) -> str:
|
|
123
|
+
"""Member fqn ``{parent_fqn}#{signature}`` -> parent type fqn; type fqn (no '#') unchanged."""
|
|
124
|
+
return fqn.split("#", 1)[0] if fqn else fqn
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _resolve_source_root(graph: LadybugGraph) -> str:
|
|
128
|
+
"""Authoritative source root is the one cached on the graph at index time."""
|
|
129
|
+
try:
|
|
130
|
+
root = str(graph.meta().get("source_root") or "")
|
|
131
|
+
except Exception:
|
|
132
|
+
root = ""
|
|
133
|
+
return root or os.environ.get("JAVA_CODEBASE_RAG_SOURCE_ROOT", "").strip()
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _read_snippet(
|
|
137
|
+
source_root: str, filename: str, start_line: Any, end_line: Any, signature: str, fqn: str
|
|
138
|
+
) -> str:
|
|
139
|
+
"""Real source snippet for [start_line, end_line] from disk (capped); synthesized
|
|
140
|
+
from `signature` (fallback `fqn`) on any failure."""
|
|
141
|
+
synth = (signature or "").strip() or fqn
|
|
142
|
+
try:
|
|
143
|
+
sl = int(start_line) if start_line else 0
|
|
144
|
+
el = int(end_line) if end_line else sl
|
|
145
|
+
except (TypeError, ValueError):
|
|
146
|
+
return synth
|
|
147
|
+
if sl <= 0 or not filename:
|
|
148
|
+
return synth
|
|
149
|
+
try:
|
|
150
|
+
p = Path(filename)
|
|
151
|
+
full = p if p.is_absolute() else (Path(source_root) / filename)
|
|
152
|
+
if not p.is_absolute() and not source_root:
|
|
153
|
+
return synth
|
|
154
|
+
text = full.read_text(encoding="utf-8", errors="replace")
|
|
155
|
+
except OSError:
|
|
156
|
+
return synth
|
|
157
|
+
lines = text.splitlines()
|
|
158
|
+
lo = max(sl - 1, 0)
|
|
159
|
+
hi = min(el if el >= sl else sl, lo + _SNIPPET_MAX_LINES)
|
|
160
|
+
chunk = "\n".join(lines[lo:hi]).strip()
|
|
161
|
+
if len(chunk) > _SNIPPET_MAX_CHARS:
|
|
162
|
+
chunk = chunk[: _SNIPPET_MAX_CHARS - 1] + "…"
|
|
163
|
+
return chunk or synth
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _token_overlap(haystack_toks: set[str], needle_toks: set[str]) -> float:
|
|
167
|
+
"""Fraction of needle tokens present in haystack (0..1)."""
|
|
168
|
+
if not needle_toks:
|
|
169
|
+
return 0.0
|
|
170
|
+
return len(needle_toks & haystack_toks) / len(needle_toks)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def run_lexical_search(
|
|
174
|
+
query: str,
|
|
175
|
+
*,
|
|
176
|
+
table: str = "java",
|
|
177
|
+
limit: int = 5,
|
|
178
|
+
offset: int = 0,
|
|
179
|
+
path_contains: str | None = None,
|
|
180
|
+
filter: NodeFilter | None = None,
|
|
181
|
+
explain: bool = False,
|
|
182
|
+
dedup: bool = True,
|
|
183
|
+
advisories: list[str] | None = None,
|
|
184
|
+
graph: LadybugGraph | None = None,
|
|
185
|
+
) -> list[dict]:
|
|
186
|
+
"""Keyword search over Symbol nodes; returns ``run_search``-shaped row-dicts.
|
|
187
|
+
|
|
188
|
+
Raises ``RuntimeError`` (message contains "lexical search unavailable") if no
|
|
189
|
+
symbol graph exists — the caller maps that to a clean failure envelope. Returns
|
|
190
|
+
``[]`` for ``table in ("sql", "yaml")`` (those LanceDB tables aren't built in
|
|
191
|
+
graph-only mode) and when the graph exists but nothing matches.
|
|
192
|
+
"""
|
|
193
|
+
# sql/yaml LanceDB tables don't exist in graph-only mode.
|
|
194
|
+
if table in ("sql", "yaml"):
|
|
195
|
+
return []
|
|
196
|
+
|
|
197
|
+
if graph is None and not LadybugGraph.exists():
|
|
198
|
+
raise RuntimeError(
|
|
199
|
+
"lexical search unavailable: no symbol graph found; "
|
|
200
|
+
"run `java-codebase-rag init` or `java-codebase-rag reprocess` to build one"
|
|
201
|
+
)
|
|
202
|
+
g = graph or LadybugGraph.get()
|
|
203
|
+
|
|
204
|
+
where, params = _lexical_where(filter, path_contains=path_contains)
|
|
205
|
+
# Always exclude structural Symbol nodes. Files and packages are :Symbol-labeled
|
|
206
|
+
# (kind='file'/'package') but aren't searchable code declarations — without this
|
|
207
|
+
# a token that appears in a filename (e.g. 'distribution' in
|
|
208
|
+
# 'DistributionChunkService.java') would surface the file node as a hit.
|
|
209
|
+
struct_pred = "(s.kind <> 'file' AND s.kind <> 'package')"
|
|
210
|
+
where = f"WHERE {struct_pred}" if not where else where.replace("WHERE ", f"WHERE {struct_pred} AND ", 1)
|
|
211
|
+
# Lexical ranking is done in Python (LadybugDB/kuzu has no keyword ranking without
|
|
212
|
+
# FTS5, which is deferred), and the MATCH scan returns rows in storage order — there
|
|
213
|
+
# is NO DB-side relevance ORDER BY. So fetch the FULL candidate pool up to the safety
|
|
214
|
+
# cap and rank here. A pagination-derived LIMIT (4x the page) is correct on the vector
|
|
215
|
+
# path where LanceDB returns rows pre-ranked by similarity, but on this unordered scan
|
|
216
|
+
# it would return only the first ~N symbols in arbitrary storage order and silently
|
|
217
|
+
# miss the best match on any non-trivial repo.
|
|
218
|
+
params["lim"] = _CANDIDATE_LIMIT_CAP
|
|
219
|
+
cypher = f"MATCH (s:Symbol) {where} RETURN {_SYMBOL_RETURN} LIMIT $lim"
|
|
220
|
+
rows = g._rows(cypher, params) # noqa: SLF001 — de facto public read API (see find_v2)
|
|
221
|
+
# If the fetch hit the safety cap, deeper matches were never ranked (the scan has no
|
|
222
|
+
# ORDER BY — kuzu returns an arbitrary, storage-order-dependent subset). Surface it so
|
|
223
|
+
# a user on a large repo isn't silently shown an incomplete result set; refining the
|
|
224
|
+
# query or adding a filter narrows the pool below the cap. Raising the cap / FTS5 is
|
|
225
|
+
# the deferred long-term fix (see the plan's "Out of scope" note).
|
|
226
|
+
if advisories is not None and len(rows) >= _CANDIDATE_LIMIT_CAP:
|
|
227
|
+
advisories.append(
|
|
228
|
+
f"lexical search scanned the first {_CANDIDATE_LIMIT_CAP} matching symbols "
|
|
229
|
+
"(repo cap); deeper matches were not ranked — refine the query or add a filter"
|
|
230
|
+
)
|
|
231
|
+
|
|
232
|
+
query_toks = _query_tokens(query)
|
|
233
|
+
source_root = _resolve_source_root(g)
|
|
234
|
+
role_locked = bool(
|
|
235
|
+
filter and (getattr(filter, "role", None) or getattr(filter, "exclude_roles", None))
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
out: list[dict] = []
|
|
239
|
+
for r in rows:
|
|
240
|
+
name = str(r.get("name") or "")
|
|
241
|
+
fqn = str(r.get("fqn") or "")
|
|
242
|
+
type_fqn = _enclosing_type_fqn(fqn)
|
|
243
|
+
sig = str(r.get("signature") or "")
|
|
244
|
+
anns = list(r.get("annotations") or [])
|
|
245
|
+
caps = list(r.get("capabilities") or [])
|
|
246
|
+
role_raw = str(r.get("role") or "")
|
|
247
|
+
|
|
248
|
+
name_toks = set(_split_identifier(name))
|
|
249
|
+
type_toks = set(_split_identifier(type_fqn.rsplit(".", 1)[-1]))
|
|
250
|
+
fqn_toks = set(_split_identifier(fqn))
|
|
251
|
+
sig_toks = _query_tokens(
|
|
252
|
+
" ".join([sig, " ".join(anns), " ".join(caps), str(r.get("package") or "")])
|
|
253
|
+
)
|
|
254
|
+
|
|
255
|
+
name_overlap = len(query_toks & name_toks)
|
|
256
|
+
name_match = (
|
|
257
|
+
1.0
|
|
258
|
+
if (query_toks and query_toks <= name_toks)
|
|
259
|
+
else (min(name_overlap / max(len(name_toks), 1), 1.0) if name_toks else 0.0)
|
|
260
|
+
)
|
|
261
|
+
type_hits = len(query_toks & type_toks)
|
|
262
|
+
type_match = min(type_hits * _TYPE_MATCH_BONUS_PER_HIT, _TYPE_MATCH_BONUS_CAP)
|
|
263
|
+
fqn_match = _token_overlap(fqn_toks, query_toks)
|
|
264
|
+
text_overlap = _token_overlap(sig_toks, query_toks)
|
|
265
|
+
text_match = text_overlap * _TEXT_MATCH_WEIGHT
|
|
266
|
+
|
|
267
|
+
# A keyword search must require at least one lexical hit — role alone never
|
|
268
|
+
# qualifies a row (it only boosts/reorders matches). Degenerate queries with
|
|
269
|
+
# no usable tokens fall through to role-ranked listing.
|
|
270
|
+
if query_toks and not (name_overlap or type_hits or fqn_match or text_overlap):
|
|
271
|
+
continue
|
|
272
|
+
|
|
273
|
+
role_w = 0.0 if role_locked else _ROLE_SCORE_WEIGHTS.get(role_raw.upper(), 0.0)
|
|
274
|
+
|
|
275
|
+
if query_toks:
|
|
276
|
+
raw = (
|
|
277
|
+
_NAME_MATCH_WEIGHT * name_match
|
|
278
|
+
+ type_match
|
|
279
|
+
+ _FQN_MATCH_WEIGHT * fqn_match
|
|
280
|
+
+ text_match
|
|
281
|
+
+ role_w
|
|
282
|
+
)
|
|
283
|
+
else:
|
|
284
|
+
raw = role_w # degenerate query: rank by role only
|
|
285
|
+
score = _clamp01(raw / LEXICAL_SCORE_MAX)
|
|
286
|
+
|
|
287
|
+
comps = {
|
|
288
|
+
"name_match": round(name_match, 4),
|
|
289
|
+
"type_match": round(type_match, 4),
|
|
290
|
+
"fqn_match": round(fqn_match, 4),
|
|
291
|
+
"lexical_relevance": round(raw, 4),
|
|
292
|
+
"role_weight": role_w,
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
sl, el, sb, eb = r.get("start_line"), r.get("end_line"), r.get("start_byte"), r.get("end_byte")
|
|
296
|
+
out.append(
|
|
297
|
+
{
|
|
298
|
+
"_score": score,
|
|
299
|
+
"_kind": "java",
|
|
300
|
+
"_score_components": comps,
|
|
301
|
+
"filename": str(r.get("filename") or ""),
|
|
302
|
+
"text": _read_snippet(source_root, str(r.get("filename") or ""), sl, el, sig, fqn),
|
|
303
|
+
"primary_type_fqn": type_fqn or None,
|
|
304
|
+
# Raw node fqn (members are 'Type#method(...)') feeds
|
|
305
|
+
# _node_matches_filter's fqn_contains re-check (mcp_v2.py). Without it the
|
|
306
|
+
# post-filter falls back to primary_type_fqn (the bare type) and drops
|
|
307
|
+
# member-level matches the Cypher pushdown already accepted.
|
|
308
|
+
"fqn": fqn,
|
|
309
|
+
"microservice": r.get("microservice"),
|
|
310
|
+
"module": r.get("module"),
|
|
311
|
+
"role": role_raw or None,
|
|
312
|
+
"kind": r.get("kind"),
|
|
313
|
+
"symbol_id": r.get("id"),
|
|
314
|
+
"annotations": anns,
|
|
315
|
+
"capabilities": caps,
|
|
316
|
+
"start": {
|
|
317
|
+
"line": int(sl) if sl is not None else None,
|
|
318
|
+
"byte_offset": int(sb) if sb is not None else 0,
|
|
319
|
+
},
|
|
320
|
+
"end": {
|
|
321
|
+
"line": int(el) if el is not None else None,
|
|
322
|
+
"byte_offset": int(eb) if eb is not None else 0,
|
|
323
|
+
},
|
|
324
|
+
}
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
out.sort(key=lambda d: float(d.get("_score", 0.0)), reverse=True)
|
|
328
|
+
out = _dedup_by_fqn(out, dedup_by_fqn=dedup)
|
|
329
|
+
return out[offset : offset + limit]
|
search_scoring.py
ADDED
|
@@ -0,0 +1,338 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Dependency-free scoring & dedup primitives shared by the search backends.
|
|
3
|
+
|
|
4
|
+
Imported by both the vector backend (`search_lancedb`) and the lexical backend
|
|
5
|
+
(`search_lexical`). This module MUST NOT import lancedb / torch /
|
|
6
|
+
sentence_transformers / cocoindex — it is imported on graph-only (macOS Intel)
|
|
7
|
+
installs where those packages are absent (see pyproject.toml PEP 508 markers).
|
|
8
|
+
|
|
9
|
+
Everything here is pure-Python dict/list math with no third-party deps.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
|
|
16
|
+
# Over-fetch multiplier for dedup: fetch 4x to absorb per-FQN chunk multiplicity
|
|
17
|
+
# so that after collapsing by primary_type_fqn, a page stays full and the +1
|
|
18
|
+
# truncation sentinel survives. The formula: need = max((limit + offset) * 4, limit + offset + 1)
|
|
19
|
+
DEDUP_OVERFETCH = 4
|
|
20
|
+
|
|
21
|
+
_IMPORT_DISTANCE_PENALTY = 0.08
|
|
22
|
+
_IMPORT_HYBRID_SCORE_FACTOR = 0.88
|
|
23
|
+
|
|
24
|
+
# Bonus for chunks whose declared symbols (method / field names) share tokens with
|
|
25
|
+
# the query. Behavioural queries like "what happens when a client message arrives"
|
|
26
|
+
# should float chunks containing `processClientMessage` above ones that only
|
|
27
|
+
# enqueue; this is a cheap, query-dependent signal computed at rank time.
|
|
28
|
+
_SYMBOL_MATCH_BONUS_PER_HIT = 0.03
|
|
29
|
+
_SYMBOL_MATCH_BONUS_CAP = 0.06
|
|
30
|
+
|
|
31
|
+
# Action verbs that typically mark behavioural entry points in this codebase.
|
|
32
|
+
# A chunk whose symbols begin with one of these verbs earns a small flat bump
|
|
33
|
+
# — again only for java chunks and only when role-filtering is off.
|
|
34
|
+
_ACTION_VERB_PREFIXES: tuple[str, ...] = (
|
|
35
|
+
"process", "handle", "on", "pick", "select", "assign",
|
|
36
|
+
"notify", "dispatch", "publish", "consume", "route",
|
|
37
|
+
"trigger", "enqueue", "distribute", "update", "create",
|
|
38
|
+
"apply", "resolve", "reassign", "close", "open",
|
|
39
|
+
)
|
|
40
|
+
_ACTION_VERB_BONUS = 0.02
|
|
41
|
+
|
|
42
|
+
# Type-name overlap bonus. The class name is a much stronger discovery signal
|
|
43
|
+
# than any individual method, because class naming in this codebase encodes
|
|
44
|
+
# the domain concept (`DistributionChunkService`, `OperatorSessionService`,
|
|
45
|
+
# `JoinOperatorController`). So we reward overlap between query tokens and the
|
|
46
|
+
# simple name of `primary_type_fqn` more heavily than per-method overlap, and
|
|
47
|
+
# we stack it on top of the existing `_symbol_bonus`.
|
|
48
|
+
_TYPE_MATCH_BONUS_PER_HIT = 0.05
|
|
49
|
+
_TYPE_MATCH_BONUS_CAP = 0.10
|
|
50
|
+
|
|
51
|
+
_STOPWORDS: frozenset[str] = frozenset({
|
|
52
|
+
"a", "an", "the", "is", "are", "was", "were", "be", "been", "being",
|
|
53
|
+
"to", "of", "in", "on", "at", "by", "for", "with", "from", "as", "or",
|
|
54
|
+
"and", "but", "if", "then", "else", "when", "what", "how", "why", "does",
|
|
55
|
+
"do", "did", "has", "have", "had", "this", "that", "these", "those", "it",
|
|
56
|
+
"its", "new", "no", "not", "will", "would", "should", "can", "could",
|
|
57
|
+
"may", "might", "happens", "happen", "happened", "get", "gets", "got",
|
|
58
|
+
})
|
|
59
|
+
|
|
60
|
+
# Role-aware reweighting for Java chunks. Positive values favour actionable
|
|
61
|
+
# behavioural code (entrypoints, orchestrators, integrations) over configuration,
|
|
62
|
+
# schema, and persistence stubs for "what happens when..."-style queries.
|
|
63
|
+
# Applied to the similarity score (higher = better); distance-based sort subtracts
|
|
64
|
+
# the weight. Skipped when caller filters explicitly by role.
|
|
65
|
+
_ROLE_SCORE_WEIGHTS: dict[str, float] = {
|
|
66
|
+
"CONTROLLER": 0.10,
|
|
67
|
+
"SERVICE": 0.08,
|
|
68
|
+
"CLIENT": 0.06,
|
|
69
|
+
"COMPONENT": 0.03,
|
|
70
|
+
"REPOSITORY": 0.02,
|
|
71
|
+
"MAPPER": 0.00,
|
|
72
|
+
"OTHER": 0.00,
|
|
73
|
+
"ENTITY": -0.06,
|
|
74
|
+
"CONFIG": -0.10,
|
|
75
|
+
# DTOs are passive data carriers; they almost never answer "how/what
|
|
76
|
+
# happens" queries. Penalty is slightly stronger than ENTITY so a DTO
|
|
77
|
+
# with a great embedding match still loses to a mediocre SERVICE hit.
|
|
78
|
+
"DTO": -0.08,
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
# Theoretical maximum for hybrid composite score (used for display normalization).
|
|
82
|
+
# Hybrid sort metric: raw_rrf * (import_factor if import_heavy else 1)
|
|
83
|
+
# + role_weight + symbol_bonus
|
|
84
|
+
# where raw_rrf ≤ 2/(k+1) for 2-list RRF, role_weight ≤ max(_ROLE_SCORE_WEIGHTS),
|
|
85
|
+
# and symbol_bonus ≤ _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS.
|
|
86
|
+
# The import factor is ≤ 1, so we use the raw max (2/61).
|
|
87
|
+
_HYBRID_SCORE_MAX = (2.0 / 61.0) + max(_ROLE_SCORE_WEIGHTS.values()) + _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _query_tokens(query: str) -> set[str]:
|
|
91
|
+
"""Lowercased alpha-only tokens from the query, minus stopwords, len >= 3.
|
|
92
|
+
|
|
93
|
+
Used to score symbol-name overlap; we keep it simple and locale-free.
|
|
94
|
+
"""
|
|
95
|
+
out: set[str] = set()
|
|
96
|
+
cur: list[str] = []
|
|
97
|
+
|
|
98
|
+
def _flush() -> None:
|
|
99
|
+
if cur:
|
|
100
|
+
tok = "".join(cur).lower()
|
|
101
|
+
cur.clear()
|
|
102
|
+
if len(tok) >= 3 and tok not in _STOPWORDS:
|
|
103
|
+
out.add(tok)
|
|
104
|
+
|
|
105
|
+
for c in query:
|
|
106
|
+
if c.isalpha():
|
|
107
|
+
cur.append(c)
|
|
108
|
+
else:
|
|
109
|
+
_flush()
|
|
110
|
+
_flush()
|
|
111
|
+
return out
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _split_identifier(name: str) -> list[str]:
|
|
115
|
+
"""camelCase / snake_case -> lowercase token list."""
|
|
116
|
+
parts: list[str] = []
|
|
117
|
+
cur: list[str] = []
|
|
118
|
+
for c in name:
|
|
119
|
+
if c == "_":
|
|
120
|
+
if cur:
|
|
121
|
+
parts.append("".join(cur).lower())
|
|
122
|
+
cur = []
|
|
123
|
+
elif c.isupper() and cur:
|
|
124
|
+
parts.append("".join(cur).lower())
|
|
125
|
+
cur = [c]
|
|
126
|
+
else:
|
|
127
|
+
cur.append(c)
|
|
128
|
+
if cur:
|
|
129
|
+
parts.append("".join(cur).lower())
|
|
130
|
+
return [p for p in parts if p]
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _symbol_bonus(r: dict, query_toks: set[str]) -> float:
|
|
134
|
+
"""Symbol-name overlap + action-verb bump for java chunks.
|
|
135
|
+
|
|
136
|
+
Caps at `_SYMBOL_MATCH_BONUS_CAP + _ACTION_VERB_BONUS` to avoid runaway
|
|
137
|
+
ranks on chunks declaring many symbols.
|
|
138
|
+
"""
|
|
139
|
+
if str(r.get("_kind", "")) != "java":
|
|
140
|
+
return 0.0
|
|
141
|
+
raw = r.get("symbols") or []
|
|
142
|
+
if isinstance(raw, str):
|
|
143
|
+
# Legacy JSON-encoded list column; parse defensively.
|
|
144
|
+
try:
|
|
145
|
+
parsed = json.loads(raw)
|
|
146
|
+
raw = parsed if isinstance(parsed, list) else []
|
|
147
|
+
except Exception:
|
|
148
|
+
raw = []
|
|
149
|
+
symbols = [str(s) for s in raw if s]
|
|
150
|
+
|
|
151
|
+
overlap_hits = 0
|
|
152
|
+
has_action = False
|
|
153
|
+
for s in symbols:
|
|
154
|
+
bare = s.split("(", 1)[0].strip()
|
|
155
|
+
if not bare:
|
|
156
|
+
continue
|
|
157
|
+
toks = _split_identifier(bare)
|
|
158
|
+
if toks:
|
|
159
|
+
if toks[0] in _ACTION_VERB_PREFIXES:
|
|
160
|
+
has_action = True
|
|
161
|
+
if query_toks & set(toks):
|
|
162
|
+
overlap_hits += 1
|
|
163
|
+
|
|
164
|
+
bonus = min(overlap_hits * _SYMBOL_MATCH_BONUS_PER_HIT, _SYMBOL_MATCH_BONUS_CAP)
|
|
165
|
+
if has_action:
|
|
166
|
+
bonus += _ACTION_VERB_BONUS
|
|
167
|
+
|
|
168
|
+
# Type-name overlap: strongest single lexical signal for "which class is
|
|
169
|
+
# the answer?" queries. Uses the simple name of primary_type_fqn.
|
|
170
|
+
fqn = str(r.get("primary_type_fqn") or "")
|
|
171
|
+
if fqn:
|
|
172
|
+
simple = fqn.rsplit(".", 1)[-1]
|
|
173
|
+
type_toks = set(_split_identifier(simple))
|
|
174
|
+
type_hits = len(query_toks & type_toks)
|
|
175
|
+
if type_hits:
|
|
176
|
+
bonus += min(type_hits * _TYPE_MATCH_BONUS_PER_HIT, _TYPE_MATCH_BONUS_CAP)
|
|
177
|
+
return bonus
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _role_weight(r: dict) -> float:
|
|
181
|
+
"""Effective role weight for a row, captured into `_score_components.role_weight`."""
|
|
182
|
+
comps = r.setdefault("_score_components", {})
|
|
183
|
+
cached = comps.get("role_weight")
|
|
184
|
+
if cached is not None:
|
|
185
|
+
return float(cached)
|
|
186
|
+
if r.get("_skip_role_weight") or str(r.get("_kind", "")) != "java":
|
|
187
|
+
comps["role_weight"] = 0.0
|
|
188
|
+
return 0.0
|
|
189
|
+
role = (r.get("role") or "").upper()
|
|
190
|
+
w = _ROLE_SCORE_WEIGHTS.get(role, 0.0)
|
|
191
|
+
comps["role_weight"] = w
|
|
192
|
+
return w
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _apply_symbol_bonus(rows: list[dict], query_toks: set[str]) -> None:
|
|
196
|
+
"""Pre-compute symbol-match bonus into `_score_components.symbol_bonus`."""
|
|
197
|
+
if not query_toks:
|
|
198
|
+
return
|
|
199
|
+
for r in rows:
|
|
200
|
+
if r.get("_skip_role_weight"):
|
|
201
|
+
# When the caller locked role, respect their intent everywhere.
|
|
202
|
+
continue
|
|
203
|
+
b = _symbol_bonus(r, query_toks)
|
|
204
|
+
if b:
|
|
205
|
+
r.setdefault("_score_components", {})["symbol_bonus"] = b
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def l2_distance_to_score(distance: float) -> float:
|
|
209
|
+
"""Map L2 distance to a similarity score for unit-normalized embeddings."""
|
|
210
|
+
return 1.0 - distance * distance / 2.0
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _effective_distance(comps: dict[str, float]) -> float:
|
|
214
|
+
"""Compute the adjusted distance used for sorting.
|
|
215
|
+
|
|
216
|
+
Matches _vector_sort_key logic: distance + import_penalty - role_weight - symbol_bonus.
|
|
217
|
+
"""
|
|
218
|
+
d = comps.get("distance", 0.0)
|
|
219
|
+
d += comps.get("import_penalty", 0.0)
|
|
220
|
+
d -= comps.get("role_weight", 0.0)
|
|
221
|
+
d -= comps.get("symbol_bonus", 0.0)
|
|
222
|
+
return d
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _clamp01(x: float) -> float:
|
|
226
|
+
"""Clamp a value to the [0.0, 1.0] range."""
|
|
227
|
+
if x < 0.0:
|
|
228
|
+
return 0.0
|
|
229
|
+
if x > 1.0:
|
|
230
|
+
return 1.0
|
|
231
|
+
return x
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def explain_score_components(
|
|
235
|
+
comps: dict[str, float] | None,
|
|
236
|
+
*,
|
|
237
|
+
role: str | None = None,
|
|
238
|
+
hybrid: bool = False,
|
|
239
|
+
graph_expanded: bool = False,
|
|
240
|
+
lexical: bool = False,
|
|
241
|
+
) -> str:
|
|
242
|
+
"""Compact human-readable 'why' string for a ranked hit.
|
|
243
|
+
|
|
244
|
+
Joins the interesting components of `_score_components` in a stable order
|
|
245
|
+
so agents can reason about rankings without chasing raw floats. Returns
|
|
246
|
+
"" if there's nothing worth mentioning.
|
|
247
|
+
"""
|
|
248
|
+
if not comps:
|
|
249
|
+
comps = {}
|
|
250
|
+
parts: list[str] = []
|
|
251
|
+
if lexical:
|
|
252
|
+
rel = comps.get("lexical_relevance")
|
|
253
|
+
if rel is not None:
|
|
254
|
+
parts.append(f"relevance={float(rel):.2f}")
|
|
255
|
+
nm = comps.get("name_match")
|
|
256
|
+
if nm is not None:
|
|
257
|
+
parts.append(f"name={float(nm):.2f}")
|
|
258
|
+
ty = comps.get("type_match")
|
|
259
|
+
if ty:
|
|
260
|
+
parts.append(f"type:{float(ty):+.2f}")
|
|
261
|
+
fq = comps.get("fqn_match")
|
|
262
|
+
if fq:
|
|
263
|
+
parts.append(f"fqn:{float(fq):+.2f}")
|
|
264
|
+
elif hybrid:
|
|
265
|
+
# Prefer rrf_raw (added by PR-SEARCH-1a) for explanation
|
|
266
|
+
rrf = comps.get("rrf_raw") or comps.get("hybrid_rrf")
|
|
267
|
+
if rrf is not None:
|
|
268
|
+
parts.append(f"rrf={float(rrf):.3f}")
|
|
269
|
+
else:
|
|
270
|
+
d = comps.get("distance")
|
|
271
|
+
if d is not None:
|
|
272
|
+
parts.append(f"dist={float(d):.2f}")
|
|
273
|
+
rw = comps.get("role_weight")
|
|
274
|
+
if rw:
|
|
275
|
+
label = f"role:{role}" if role else "role"
|
|
276
|
+
parts.append(f"{label}:{float(rw):+.02f}")
|
|
277
|
+
sb = comps.get("symbol_bonus")
|
|
278
|
+
if sb:
|
|
279
|
+
parts.append(f"symbol:{float(sb):+.02f}")
|
|
280
|
+
ip = comps.get("import_penalty")
|
|
281
|
+
if ip:
|
|
282
|
+
parts.append(f"import_penalty:{float(ip):+.02f}")
|
|
283
|
+
if graph_expanded:
|
|
284
|
+
parts.append("graph")
|
|
285
|
+
return " ".join(parts)
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def _dedup_by_fqn(rows: list[dict], dedup_by_fqn: bool = True) -> list[dict]:
|
|
289
|
+
"""Deduplicate rows by primary_type_fqn (java table only).
|
|
290
|
+
|
|
291
|
+
When dedup_by_fqn is True, collapses multiple chunks of the same
|
|
292
|
+
primary_type_fqn into one row (first-seen-wins, since rows are pre-sorted
|
|
293
|
+
so the first is the best chunk). Each survivor gets a _chunks_collapsed
|
|
294
|
+
field (>=1) counting how many rows were collapsed into it.
|
|
295
|
+
|
|
296
|
+
Rows without primary_type_fqn (sql/yaml tables) get a unique __id:<id>
|
|
297
|
+
key so they pass through unchanged (each row is unique).
|
|
298
|
+
|
|
299
|
+
When dedup_by_fqn is False, returns rows unchanged (regression guard).
|
|
300
|
+
"""
|
|
301
|
+
if not dedup_by_fqn:
|
|
302
|
+
# Non-dedup path: return unchanged, byte-identical to prior behavior
|
|
303
|
+
return rows
|
|
304
|
+
|
|
305
|
+
deduped: list[dict] = []
|
|
306
|
+
seen_keys: dict[str, dict] = {}
|
|
307
|
+
collapsed_counts: dict[str, int] = {}
|
|
308
|
+
|
|
309
|
+
for row in rows:
|
|
310
|
+
# Build dedup key: primary_type_fqn for java rows, unique __id:<id> for sql/yaml
|
|
311
|
+
fqn = row.get("primary_type_fqn")
|
|
312
|
+
if fqn:
|
|
313
|
+
key = str(fqn)
|
|
314
|
+
else:
|
|
315
|
+
# sql/yaml rows have no primary_type_fqn → unique key per row
|
|
316
|
+
row_id = row.get("id") or id(row)
|
|
317
|
+
key = f"__id:{row_id}"
|
|
318
|
+
|
|
319
|
+
if key not in seen_keys:
|
|
320
|
+
# First occurrence: keep it
|
|
321
|
+
seen_keys[key] = row
|
|
322
|
+
collapsed_counts[key] = 1
|
|
323
|
+
deduped.append(row)
|
|
324
|
+
else:
|
|
325
|
+
# Duplicate: increment collapse count, discard this row
|
|
326
|
+
collapsed_counts[key] += 1
|
|
327
|
+
|
|
328
|
+
# Annotate each survivor with _chunks_collapsed
|
|
329
|
+
for row in deduped:
|
|
330
|
+
fqn = row.get("primary_type_fqn")
|
|
331
|
+
if fqn:
|
|
332
|
+
key = str(fqn)
|
|
333
|
+
else:
|
|
334
|
+
row_id = row.get("id") or id(row)
|
|
335
|
+
key = f"__id:{row_id}"
|
|
336
|
+
row["_chunks_collapsed"] = collapsed_counts[key]
|
|
337
|
+
|
|
338
|
+
return deduped
|