java-codebase-rag 0.9.4__py3-none-any.whl → 0.9.6__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- java_codebase_rag/absence/__init__.py +0 -0
- java_codebase_rag/absence/absence_diagnosis.py +700 -0
- java_codebase_rag/absence/absence_types.py +124 -0
- java_codebase_rag/absence/absence_vocab.py +455 -0
- java_codebase_rag/analysis/__init__.py +0 -0
- pr_analysis.py → java_codebase_rag/analysis/pr_analysis.py +1 -1
- resolve_service.py → java_codebase_rag/analysis/resolve_service.py +73 -6
- java_codebase_rag/ast/__init__.py +0 -0
- ast_java.py → java_codebase_rag/ast/ast_java.py +5 -5
- java_codebase_rag/cli.py +13 -18
- java_codebase_rag/config.py +116 -0
- java_codebase_rag/graph/__init__.py +0 -0
- build_ast_graph.py → java_codebase_rag/graph/build_ast_graph.py +89 -11
- graph_enrich.py → java_codebase_rag/graph/graph_enrich.py +248 -3
- graph_types.py → java_codebase_rag/graph/graph_types.py +6 -2
- java_ontology.py → java_codebase_rag/graph/java_ontology.py +1 -1
- ladybug_queries.py → java_codebase_rag/graph/ladybug_queries.py +6 -6
- java_codebase_rag/index/__init__.py +0 -0
- java_index_flow_lancedb.py → java_codebase_rag/index/java_index_flow_lancedb.py +30 -10
- java_codebase_rag/install_data/__init__.py +0 -0
- java_codebase_rag/jrag.py +71 -16
- java_codebase_rag/jrag_envelope.py +13 -4
- java_codebase_rag/jrag_hints.py +1 -1
- java_codebase_rag/jrag_render.py +67 -3
- java_codebase_rag/mcp/__init__.py +0 -0
- mcp_hints.py → java_codebase_rag/mcp/mcp_hints.py +1 -1
- mcp_v2.py → java_codebase_rag/mcp/mcp_v2.py +280 -81
- server.py → java_codebase_rag/mcp/server.py +138 -54
- java_codebase_rag/pipeline.py +26 -7
- java_codebase_rag/search/__init__.py +0 -0
- search_lancedb.py → java_codebase_rag/search/search_lancedb.py +53 -314
- java_codebase_rag/search/search_lexical.py +329 -0
- java_codebase_rag/search/search_scoring.py +338 -0
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/METADATA +2 -2
- java_codebase_rag-0.9.6.dist-info/RECORD +57 -0
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/entry_points.txt +1 -1
- java_codebase_rag-0.9.6.dist-info/top_level.txt +1 -0
- java_codebase_rag-0.9.4.dist-info/RECORD +0 -44
- java_codebase_rag-0.9.4.dist-info/top_level.txt +0 -19
- /brownfield_events.py → /java_codebase_rag/ast/brownfield_events.py +0 -0
- /chunk_heuristics.py → /java_codebase_rag/ast/chunk_heuristics.py +0 -0
- /path_filtering.py → /java_codebase_rag/graph/path_filtering.py +0 -0
- /java_index_v1_common.py → /java_codebase_rag/index/java_index_v1_common.py +0 -0
- /index_common.py → /java_codebase_rag/search/index_common.py +0 -0
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/WHEEL +0 -0
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/licenses/LICENSE +0 -0
|
@@ -15,10 +15,39 @@ import lancedb
|
|
|
15
15
|
import numpy as np
|
|
16
16
|
from sentence_transformers import SentenceTransformer
|
|
17
17
|
|
|
18
|
-
from chunk_heuristics import analyze_chunk, looks_like_code_identifier
|
|
19
|
-
from index_common import SBERT_MODEL
|
|
18
|
+
from java_codebase_rag.ast.chunk_heuristics import analyze_chunk, looks_like_code_identifier
|
|
19
|
+
from java_codebase_rag.search.index_common import SBERT_MODEL
|
|
20
20
|
from java_codebase_rag.config import maybe_expand_embedding_model_path, resolved_sbert_model_for_process_env
|
|
21
21
|
|
|
22
|
+
# Scoring & dedup primitives live in `search_scoring` (dependency-free — no
|
|
23
|
+
# lancedb/torch) so the lexical backend `search_lexical` can share them on
|
|
24
|
+
# graph-only (macOS Intel) installs where this module is unimportable. Re-exported
|
|
25
|
+
# here for backward compatibility (`from search_lancedb import _clamp01`, etc.).
|
|
26
|
+
from java_codebase_rag.search.search_scoring import ( # noqa: F401
|
|
27
|
+
DEDUP_OVERFETCH,
|
|
28
|
+
_ACTION_VERB_BONUS,
|
|
29
|
+
_ACTION_VERB_PREFIXES,
|
|
30
|
+
_HYBRID_SCORE_MAX,
|
|
31
|
+
_IMPORT_DISTANCE_PENALTY,
|
|
32
|
+
_IMPORT_HYBRID_SCORE_FACTOR,
|
|
33
|
+
_ROLE_SCORE_WEIGHTS,
|
|
34
|
+
_STOPWORDS,
|
|
35
|
+
_SYMBOL_MATCH_BONUS_CAP,
|
|
36
|
+
_SYMBOL_MATCH_BONUS_PER_HIT,
|
|
37
|
+
_TYPE_MATCH_BONUS_CAP,
|
|
38
|
+
_TYPE_MATCH_BONUS_PER_HIT,
|
|
39
|
+
_apply_symbol_bonus,
|
|
40
|
+
_clamp01,
|
|
41
|
+
_dedup_by_fqn,
|
|
42
|
+
_effective_distance,
|
|
43
|
+
_query_tokens,
|
|
44
|
+
_role_weight,
|
|
45
|
+
_split_identifier,
|
|
46
|
+
_symbol_bonus,
|
|
47
|
+
explain_score_components,
|
|
48
|
+
l2_distance_to_score,
|
|
49
|
+
)
|
|
50
|
+
|
|
22
51
|
TABLES: dict[str, str] = {
|
|
23
52
|
"java": "javacodeindex_java_code",
|
|
24
53
|
"sql": "sqlschemaindex_sql_schema",
|
|
@@ -39,13 +68,10 @@ JAVA_ENRICHED_COLUMNS: tuple[str, ...] = (
|
|
|
39
68
|
"metadata",
|
|
40
69
|
"ontology_version",
|
|
41
70
|
"capabilities",
|
|
71
|
+
"generated",
|
|
72
|
+
"generated_by",
|
|
42
73
|
)
|
|
43
74
|
|
|
44
|
-
# Over-fetch multiplier for dedup: fetch 4x to absorb per-FQN chunk multiplicity
|
|
45
|
-
# so that after collapsing by primary_type_fqn, a page stays full and the +1
|
|
46
|
-
# truncation sentinel survives. The formula: need = max((limit + offset) * 4, limit + offset + 1)
|
|
47
|
-
DEDUP_OVERFETCH = 4
|
|
48
|
-
|
|
49
75
|
VECTOR_COLUMN = "embedding"
|
|
50
76
|
_FTS_READY: set[tuple[str, str]] = set()
|
|
51
77
|
_FTS_LOCK = threading.Lock()
|
|
@@ -83,6 +109,8 @@ def _build_extra_predicates(
|
|
|
83
109
|
exclude_roles: list[str] | None = None,
|
|
84
110
|
capability: str | None = None,
|
|
85
111
|
capability_in: list[str] | None = None,
|
|
112
|
+
generated_only: bool = False,
|
|
113
|
+
exclude_generated: bool = False,
|
|
86
114
|
) -> list[str]:
|
|
87
115
|
preds: list[str] = []
|
|
88
116
|
if role and "role" in columns:
|
|
@@ -115,6 +143,10 @@ def _build_extra_predicates(
|
|
|
115
143
|
if exclude_roles and "role" in columns:
|
|
116
144
|
vals = ", ".join(f"'{_escape_sql_str(v)}'" for v in exclude_roles)
|
|
117
145
|
preds.append(f"(role IS NULL OR role NOT IN ({vals}))")
|
|
146
|
+
if generated_only and "generated" in columns:
|
|
147
|
+
preds.append("generated = true")
|
|
148
|
+
if exclude_generated and "generated" in columns:
|
|
149
|
+
preds.append("(generated IS NULL OR generated = false)")
|
|
118
150
|
if module and "module" in columns:
|
|
119
151
|
preds.append(f"module = '{_escape_sql_str(module)}'")
|
|
120
152
|
if microservice and "microservice" in columns:
|
|
@@ -146,165 +178,6 @@ def coerce_position_field(val: object) -> dict[str, object]:
|
|
|
146
178
|
return {}
|
|
147
179
|
|
|
148
180
|
|
|
149
|
-
_IMPORT_DISTANCE_PENALTY = 0.08
|
|
150
|
-
_IMPORT_HYBRID_SCORE_FACTOR = 0.88
|
|
151
|
-
|
|
152
|
-
# Bonus for chunks whose declared symbols (method / field names) share tokens with
|
|
153
|
-
# the query. Behavioural queries like "what happens when a client message arrives"
|
|
154
|
-
# should float chunks containing `processClientMessage` above ones that only
|
|
155
|
-
# enqueue; this is a cheap, query-dependent signal computed at rank time.
|
|
156
|
-
_SYMBOL_MATCH_BONUS_PER_HIT = 0.03
|
|
157
|
-
_SYMBOL_MATCH_BONUS_CAP = 0.06
|
|
158
|
-
|
|
159
|
-
# Action verbs that typically mark behavioural entry points in this codebase.
|
|
160
|
-
# A chunk whose symbols begin with one of these verbs earns a small flat bump
|
|
161
|
-
# — again only for java chunks and only when role-filtering is off.
|
|
162
|
-
_ACTION_VERB_PREFIXES: tuple[str, ...] = (
|
|
163
|
-
"process", "handle", "on", "pick", "select", "assign",
|
|
164
|
-
"notify", "dispatch", "publish", "consume", "route",
|
|
165
|
-
"trigger", "enqueue", "distribute", "update", "create",
|
|
166
|
-
"apply", "resolve", "reassign", "close", "open",
|
|
167
|
-
)
|
|
168
|
-
_ACTION_VERB_BONUS = 0.02
|
|
169
|
-
|
|
170
|
-
# Type-name overlap bonus. The class name is a much stronger discovery signal
|
|
171
|
-
# than any individual method, because class naming in this codebase encodes
|
|
172
|
-
# the domain concept (`DistributionChunkService`, `OperatorSessionService`,
|
|
173
|
-
# `JoinOperatorController`). So we reward overlap between query tokens and the
|
|
174
|
-
# simple name of `primary_type_fqn` more heavily than per-method overlap, and
|
|
175
|
-
# we stack it on top of the existing `_symbol_bonus`.
|
|
176
|
-
_TYPE_MATCH_BONUS_PER_HIT = 0.05
|
|
177
|
-
_TYPE_MATCH_BONUS_CAP = 0.10
|
|
178
|
-
|
|
179
|
-
_STOPWORDS: frozenset[str] = frozenset({
|
|
180
|
-
"a", "an", "the", "is", "are", "was", "were", "be", "been", "being",
|
|
181
|
-
"to", "of", "in", "on", "at", "by", "for", "with", "from", "as", "or",
|
|
182
|
-
"and", "but", "if", "then", "else", "when", "what", "how", "why", "does",
|
|
183
|
-
"do", "did", "has", "have", "had", "this", "that", "these", "those", "it",
|
|
184
|
-
"its", "new", "no", "not", "will", "would", "should", "can", "could",
|
|
185
|
-
"may", "might", "happens", "happen", "happened", "get", "gets", "got",
|
|
186
|
-
})
|
|
187
|
-
|
|
188
|
-
# Role-aware reweighting for Java chunks. Positive values favour actionable
|
|
189
|
-
# behavioural code (entrypoints, orchestrators, integrations) over configuration,
|
|
190
|
-
# schema, and persistence stubs for "what happens when..."-style queries.
|
|
191
|
-
# Applied to the similarity score (higher = better); distance-based sort subtracts
|
|
192
|
-
# the weight. Skipped when caller filters explicitly by role.
|
|
193
|
-
_ROLE_SCORE_WEIGHTS: dict[str, float] = {
|
|
194
|
-
"CONTROLLER": 0.10,
|
|
195
|
-
"SERVICE": 0.08,
|
|
196
|
-
"CLIENT": 0.06,
|
|
197
|
-
"COMPONENT": 0.03,
|
|
198
|
-
"REPOSITORY": 0.02,
|
|
199
|
-
"MAPPER": 0.00,
|
|
200
|
-
"OTHER": 0.00,
|
|
201
|
-
"ENTITY": -0.06,
|
|
202
|
-
"CONFIG": -0.10,
|
|
203
|
-
# DTOs are passive data carriers; they almost never answer "how/what
|
|
204
|
-
# happens" queries. Penalty is slightly stronger than ENTITY so a DTO
|
|
205
|
-
# with a great embedding match still loses to a mediocre SERVICE hit.
|
|
206
|
-
"DTO": -0.08,
|
|
207
|
-
}
|
|
208
|
-
|
|
209
|
-
# Theoretical maximum for hybrid composite score (used for display normalization).
|
|
210
|
-
# Hybrid sort metric: raw_rrf * (import_factor if import_heavy else 1)
|
|
211
|
-
# + role_weight + symbol_bonus
|
|
212
|
-
# where raw_rrf ≤ 2/(k+1) for 2-list RRF, role_weight ≤ max(_ROLE_SCORE_WEIGHTS),
|
|
213
|
-
# and symbol_bonus ≤ _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS.
|
|
214
|
-
# The import factor is ≤ 1, so we use the raw max (2/61).
|
|
215
|
-
_HYBRID_SCORE_MAX = (2.0 / 61.0) + max(_ROLE_SCORE_WEIGHTS.values()) + _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
def _query_tokens(query: str) -> set[str]:
|
|
219
|
-
"""Lowercased alpha-only tokens from the query, minus stopwords, len >= 3.
|
|
220
|
-
|
|
221
|
-
Used to score symbol-name overlap; we keep it simple and locale-free.
|
|
222
|
-
"""
|
|
223
|
-
out: set[str] = set()
|
|
224
|
-
cur: list[str] = []
|
|
225
|
-
|
|
226
|
-
def _flush() -> None:
|
|
227
|
-
if cur:
|
|
228
|
-
tok = "".join(cur).lower()
|
|
229
|
-
cur.clear()
|
|
230
|
-
if len(tok) >= 3 and tok not in _STOPWORDS:
|
|
231
|
-
out.add(tok)
|
|
232
|
-
|
|
233
|
-
for c in query:
|
|
234
|
-
if c.isalpha():
|
|
235
|
-
cur.append(c)
|
|
236
|
-
else:
|
|
237
|
-
_flush()
|
|
238
|
-
_flush()
|
|
239
|
-
return out
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
def _split_identifier(name: str) -> list[str]:
|
|
243
|
-
"""camelCase / snake_case -> lowercase token list."""
|
|
244
|
-
parts: list[str] = []
|
|
245
|
-
cur: list[str] = []
|
|
246
|
-
for c in name:
|
|
247
|
-
if c == "_":
|
|
248
|
-
if cur:
|
|
249
|
-
parts.append("".join(cur).lower())
|
|
250
|
-
cur = []
|
|
251
|
-
elif c.isupper() and cur:
|
|
252
|
-
parts.append("".join(cur).lower())
|
|
253
|
-
cur = [c]
|
|
254
|
-
else:
|
|
255
|
-
cur.append(c)
|
|
256
|
-
if cur:
|
|
257
|
-
parts.append("".join(cur).lower())
|
|
258
|
-
return [p for p in parts if p]
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
def _symbol_bonus(r: dict, query_toks: set[str]) -> float:
|
|
262
|
-
"""Symbol-name overlap + action-verb bump for java chunks.
|
|
263
|
-
|
|
264
|
-
Caps at `_SYMBOL_MATCH_BONUS_CAP + _ACTION_VERB_BONUS` to avoid runaway
|
|
265
|
-
ranks on chunks declaring many symbols.
|
|
266
|
-
"""
|
|
267
|
-
if str(r.get("_kind", "")) != "java":
|
|
268
|
-
return 0.0
|
|
269
|
-
raw = r.get("symbols") or []
|
|
270
|
-
if isinstance(raw, str):
|
|
271
|
-
# Legacy JSON-encoded list column; parse defensively.
|
|
272
|
-
try:
|
|
273
|
-
parsed = json.loads(raw)
|
|
274
|
-
raw = parsed if isinstance(parsed, list) else []
|
|
275
|
-
except Exception:
|
|
276
|
-
raw = []
|
|
277
|
-
symbols = [str(s) for s in raw if s]
|
|
278
|
-
|
|
279
|
-
overlap_hits = 0
|
|
280
|
-
has_action = False
|
|
281
|
-
for s in symbols:
|
|
282
|
-
bare = s.split("(", 1)[0].strip()
|
|
283
|
-
if not bare:
|
|
284
|
-
continue
|
|
285
|
-
toks = _split_identifier(bare)
|
|
286
|
-
if toks:
|
|
287
|
-
if toks[0] in _ACTION_VERB_PREFIXES:
|
|
288
|
-
has_action = True
|
|
289
|
-
if query_toks & set(toks):
|
|
290
|
-
overlap_hits += 1
|
|
291
|
-
|
|
292
|
-
bonus = min(overlap_hits * _SYMBOL_MATCH_BONUS_PER_HIT, _SYMBOL_MATCH_BONUS_CAP)
|
|
293
|
-
if has_action:
|
|
294
|
-
bonus += _ACTION_VERB_BONUS
|
|
295
|
-
|
|
296
|
-
# Type-name overlap: strongest single lexical signal for "which class is
|
|
297
|
-
# the answer?" queries. Uses the simple name of primary_type_fqn.
|
|
298
|
-
fqn = str(r.get("primary_type_fqn") or "")
|
|
299
|
-
if fqn:
|
|
300
|
-
simple = fqn.rsplit(".", 1)[-1]
|
|
301
|
-
type_toks = set(_split_identifier(simple))
|
|
302
|
-
type_hits = len(query_toks & type_toks)
|
|
303
|
-
if type_hits:
|
|
304
|
-
bonus += min(type_hits * _TYPE_MATCH_BONUS_PER_HIT, _TYPE_MATCH_BONUS_CAP)
|
|
305
|
-
return bonus
|
|
306
|
-
|
|
307
|
-
|
|
308
181
|
def _apply_chunk_hints(rows: list[dict]) -> None:
|
|
309
182
|
for r in rows:
|
|
310
183
|
lang = r.get("language") or ""
|
|
@@ -320,34 +193,6 @@ def _apply_chunk_hints(rows: list[dict]) -> None:
|
|
|
320
193
|
}
|
|
321
194
|
|
|
322
195
|
|
|
323
|
-
def _role_weight(r: dict) -> float:
|
|
324
|
-
"""Effective role weight for a row, captured into `_score_components.role_weight`."""
|
|
325
|
-
comps = r.setdefault("_score_components", {})
|
|
326
|
-
cached = comps.get("role_weight")
|
|
327
|
-
if cached is not None:
|
|
328
|
-
return float(cached)
|
|
329
|
-
if r.get("_skip_role_weight") or str(r.get("_kind", "")) != "java":
|
|
330
|
-
comps["role_weight"] = 0.0
|
|
331
|
-
return 0.0
|
|
332
|
-
role = (r.get("role") or "").upper()
|
|
333
|
-
w = _ROLE_SCORE_WEIGHTS.get(role, 0.0)
|
|
334
|
-
comps["role_weight"] = w
|
|
335
|
-
return w
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
def _apply_symbol_bonus(rows: list[dict], query_toks: set[str]) -> None:
|
|
339
|
-
"""Pre-compute symbol-match bonus into `_score_components.symbol_bonus`."""
|
|
340
|
-
if not query_toks:
|
|
341
|
-
return
|
|
342
|
-
for r in rows:
|
|
343
|
-
if r.get("_skip_role_weight"):
|
|
344
|
-
# When the caller locked role, respect their intent everywhere.
|
|
345
|
-
continue
|
|
346
|
-
b = _symbol_bonus(r, query_toks)
|
|
347
|
-
if b:
|
|
348
|
-
r.setdefault("_score_components", {})["symbol_bonus"] = b
|
|
349
|
-
|
|
350
|
-
|
|
351
196
|
def _vector_sort_key(r: dict) -> float:
|
|
352
197
|
d = float(r["_distance"])
|
|
353
198
|
comps = r.setdefault("_score_components", {})
|
|
@@ -372,72 +217,6 @@ def _hybrid_sort_key(r: dict) -> float:
|
|
|
372
217
|
return -s
|
|
373
218
|
|
|
374
219
|
|
|
375
|
-
def explain_score_components(
|
|
376
|
-
comps: dict[str, float] | None,
|
|
377
|
-
*,
|
|
378
|
-
role: str | None = None,
|
|
379
|
-
hybrid: bool = False,
|
|
380
|
-
graph_expanded: bool = False,
|
|
381
|
-
) -> str:
|
|
382
|
-
"""Compact human-readable 'why' string for a ranked hit.
|
|
383
|
-
|
|
384
|
-
Joins the interesting components of `_score_components` in a stable order
|
|
385
|
-
so agents can reason about rankings without chasing raw floats. Returns
|
|
386
|
-
"" if there's nothing worth mentioning.
|
|
387
|
-
"""
|
|
388
|
-
if not comps:
|
|
389
|
-
comps = {}
|
|
390
|
-
parts: list[str] = []
|
|
391
|
-
if hybrid:
|
|
392
|
-
# Prefer rrf_raw (added by PR-SEARCH-1a) for explanation
|
|
393
|
-
rrf = comps.get("rrf_raw") or comps.get("hybrid_rrf")
|
|
394
|
-
if rrf is not None:
|
|
395
|
-
parts.append(f"rrf={float(rrf):.3f}")
|
|
396
|
-
else:
|
|
397
|
-
d = comps.get("distance")
|
|
398
|
-
if d is not None:
|
|
399
|
-
parts.append(f"dist={float(d):.2f}")
|
|
400
|
-
rw = comps.get("role_weight")
|
|
401
|
-
if rw:
|
|
402
|
-
label = f"role:{role}" if role else "role"
|
|
403
|
-
parts.append(f"{label}:{float(rw):+.02f}")
|
|
404
|
-
sb = comps.get("symbol_bonus")
|
|
405
|
-
if sb:
|
|
406
|
-
parts.append(f"symbol:{float(sb):+.02f}")
|
|
407
|
-
ip = comps.get("import_penalty")
|
|
408
|
-
if ip:
|
|
409
|
-
parts.append(f"import_penalty:{float(ip):+.02f}")
|
|
410
|
-
if graph_expanded:
|
|
411
|
-
parts.append("graph")
|
|
412
|
-
return " ".join(parts)
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
def l2_distance_to_score(distance: float) -> float:
|
|
416
|
-
"""Map L2 distance to a similarity score for unit-normalized embeddings."""
|
|
417
|
-
return 1.0 - distance * distance / 2.0
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
def _effective_distance(comps: dict[str, float]) -> float:
|
|
421
|
-
"""Compute the adjusted distance used for sorting.
|
|
422
|
-
|
|
423
|
-
Matches _vector_sort_key logic: distance + import_penalty - role_weight - symbol_bonus.
|
|
424
|
-
"""
|
|
425
|
-
d = comps.get("distance", 0.0)
|
|
426
|
-
d += comps.get("import_penalty", 0.0)
|
|
427
|
-
d -= comps.get("role_weight", 0.0)
|
|
428
|
-
d -= comps.get("symbol_bonus", 0.0)
|
|
429
|
-
return d
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
def _clamp01(x: float) -> float:
|
|
433
|
-
"""Clamp a value to the [0.0, 1.0] range."""
|
|
434
|
-
if x < 0.0:
|
|
435
|
-
return 0.0
|
|
436
|
-
if x > 1.0:
|
|
437
|
-
return 1.0
|
|
438
|
-
return x
|
|
439
|
-
|
|
440
|
-
|
|
441
220
|
def _hybrid_post_sort_normalization(rows: list[dict]) -> None:
|
|
442
221
|
"""Set honest displayed scores for hybrid search after sorting.
|
|
443
222
|
|
|
@@ -743,7 +522,7 @@ def _graph_expand_merge(
|
|
|
743
522
|
"""Expand vector top-k through the LadybugDB graph and fuse (RRF) with the original list."""
|
|
744
523
|
# Lazy import so the module works without ladybug installed when graph_expand=False.
|
|
745
524
|
try:
|
|
746
|
-
from ladybug_queries import LadybugGraph
|
|
525
|
+
from java_codebase_rag.graph.ladybug_queries import LadybugGraph
|
|
747
526
|
except Exception:
|
|
748
527
|
return vector_rows
|
|
749
528
|
|
|
@@ -857,59 +636,6 @@ def _rrf_merge(
|
|
|
857
636
|
return merged
|
|
858
637
|
|
|
859
638
|
|
|
860
|
-
def _dedup_by_fqn(rows: list[dict], dedup_by_fqn: bool = True) -> list[dict]:
|
|
861
|
-
"""Deduplicate rows by primary_type_fqn (java table only).
|
|
862
|
-
|
|
863
|
-
When dedup_by_fqn is True, collapses multiple chunks of the same
|
|
864
|
-
primary_type_fqn into one row (first-seen-wins, since rows are pre-sorted
|
|
865
|
-
so the first is the best chunk). Each survivor gets a _chunks_collapsed
|
|
866
|
-
field (>=1) counting how many rows were collapsed into it.
|
|
867
|
-
|
|
868
|
-
Rows without primary_type_fqn (sql/yaml tables) get a unique __id:<id>
|
|
869
|
-
key so they pass through unchanged (each row is unique).
|
|
870
|
-
|
|
871
|
-
When dedup_by_fqn is False, returns rows unchanged (regression guard).
|
|
872
|
-
"""
|
|
873
|
-
if not dedup_by_fqn:
|
|
874
|
-
# Non-dedup path: return unchanged, byte-identical to prior behavior
|
|
875
|
-
return rows
|
|
876
|
-
|
|
877
|
-
deduped: list[dict] = []
|
|
878
|
-
seen_keys: dict[str, dict] = {}
|
|
879
|
-
collapsed_counts: dict[str, int] = {}
|
|
880
|
-
|
|
881
|
-
for row in rows:
|
|
882
|
-
# Build dedup key: primary_type_fqn for java rows, unique __id:<id> for sql/yaml
|
|
883
|
-
fqn = row.get("primary_type_fqn")
|
|
884
|
-
if fqn:
|
|
885
|
-
key = str(fqn)
|
|
886
|
-
else:
|
|
887
|
-
# sql/yaml rows have no primary_type_fqn → unique key per row
|
|
888
|
-
row_id = row.get("id") or id(row)
|
|
889
|
-
key = f"__id:{row_id}"
|
|
890
|
-
|
|
891
|
-
if key not in seen_keys:
|
|
892
|
-
# First occurrence: keep it
|
|
893
|
-
seen_keys[key] = row
|
|
894
|
-
collapsed_counts[key] = 1
|
|
895
|
-
deduped.append(row)
|
|
896
|
-
else:
|
|
897
|
-
# Duplicate: increment collapse count, discard this row
|
|
898
|
-
collapsed_counts[key] += 1
|
|
899
|
-
|
|
900
|
-
# Annotate each survivor with _chunks_collapsed
|
|
901
|
-
for row in deduped:
|
|
902
|
-
fqn = row.get("primary_type_fqn")
|
|
903
|
-
if fqn:
|
|
904
|
-
key = str(fqn)
|
|
905
|
-
else:
|
|
906
|
-
row_id = row.get("id") or id(row)
|
|
907
|
-
key = f"__id:{row_id}"
|
|
908
|
-
row["_chunks_collapsed"] = collapsed_counts[key]
|
|
909
|
-
|
|
910
|
-
return deduped
|
|
911
|
-
|
|
912
|
-
|
|
913
639
|
def run_search(
|
|
914
640
|
query: str,
|
|
915
641
|
*,
|
|
@@ -936,6 +662,8 @@ def run_search(
|
|
|
936
662
|
exclude_roles: list[str] | None = None,
|
|
937
663
|
capability: str | None = None,
|
|
938
664
|
capability_in: list[str] | None = None,
|
|
665
|
+
generated_only: bool = False,
|
|
666
|
+
exclude_generated: bool = False,
|
|
939
667
|
dedup_by_fqn: bool = False,
|
|
940
668
|
) -> list[dict]:
|
|
941
669
|
effective_hybrid = hybrid
|
|
@@ -987,6 +715,7 @@ def run_search(
|
|
|
987
715
|
package_prefix=package_prefix, fqn_in=None,
|
|
988
716
|
role_in=role_in, exclude_roles=exclude_roles,
|
|
989
717
|
capability=capability, capability_in=capability_in,
|
|
718
|
+
generated_only=generated_only, exclude_generated=exclude_generated,
|
|
990
719
|
) if "java" in table_keys else []
|
|
991
720
|
|
|
992
721
|
skip_role_weight = bool(role or role_in or exclude_roles)
|
|
@@ -1114,6 +843,10 @@ def main() -> None:
|
|
|
1114
843
|
parser.add_argument("--fts-text", metavar="TEXT", default=None)
|
|
1115
844
|
parser.add_argument("--auto-hybrid", action="store_true")
|
|
1116
845
|
parser.add_argument("--role", default=None)
|
|
846
|
+
parser.add_argument("--exclude-generated", action="store_true",
|
|
847
|
+
help="Exclude generated sources from results.")
|
|
848
|
+
parser.add_argument("--generated-only", action="store_true",
|
|
849
|
+
help="Return only generated sources in results.")
|
|
1117
850
|
parser.add_argument("--module", default=None,
|
|
1118
851
|
help="Filter to a single Maven/Gradle module name.")
|
|
1119
852
|
parser.add_argument("--microservice", default=None,
|
|
@@ -1167,6 +900,8 @@ def main() -> None:
|
|
|
1167
900
|
expand_depth=args.expand_depth,
|
|
1168
901
|
ladybug_path=args.ladybug_path,
|
|
1169
902
|
context_neighbors=args.context_neighbors,
|
|
903
|
+
exclude_generated=args.exclude_generated,
|
|
904
|
+
generated_only=args.generated_only,
|
|
1170
905
|
)
|
|
1171
906
|
except Exception as e:
|
|
1172
907
|
print(f"Search failed: {e}", file=sys.stderr)
|
|
@@ -1212,6 +947,10 @@ def main() -> None:
|
|
|
1212
947
|
mod = row.get("module") or ""
|
|
1213
948
|
if mod and mod != ms:
|
|
1214
949
|
hint_s += f" | module:{mod}"
|
|
950
|
+
gen = row.get("generated")
|
|
951
|
+
gen_by = row.get("generated_by") or ""
|
|
952
|
+
if gen:
|
|
953
|
+
hint_s += f" | generated:{gen_by}" if gen_by else " | generated"
|
|
1215
954
|
comps = row.get("_score_components") or {}
|
|
1216
955
|
rw = comps.get("role_weight")
|
|
1217
956
|
if rw:
|