java-codebase-rag 0.12.0__py3-none-any.whl → 0.12.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- java_codebase_rag-0.12.2.dist-info/METADATA +35 -0
- java_codebase_rag-0.12.2.dist-info/RECORD +4 -0
- {java_codebase_rag-0.12.0.dist-info → java_codebase_rag-0.12.2.dist-info}/WHEEL +1 -1
- java_codebase_rag/_deprecation.py +0 -103
- java_codebase_rag/_fdlimit.py +0 -56
- java_codebase_rag/_stdio.py +0 -32
- java_codebase_rag/_version.py +0 -35
- java_codebase_rag/absence/__init__.py +0 -0
- java_codebase_rag/absence/absence_diagnosis.py +0 -700
- java_codebase_rag/absence/absence_types.py +0 -124
- java_codebase_rag/absence/absence_vocab.py +0 -460
- java_codebase_rag/analysis/__init__.py +0 -0
- java_codebase_rag/analysis/pr_analysis.py +0 -563
- java_codebase_rag/analysis/resolve_service.py +0 -740
- java_codebase_rag/ast/__init__.py +0 -0
- java_codebase_rag/ast/ast_java.py +0 -2847
- java_codebase_rag/ast/ast_kotlin.py +0 -1794
- java_codebase_rag/ast/brownfield_events.py +0 -58
- java_codebase_rag/ast/chunk_heuristics.py +0 -83
- java_codebase_rag/ast/language.py +0 -117
- java_codebase_rag/cli.py +0 -1215
- java_codebase_rag/cli_dispatch.py +0 -251
- java_codebase_rag/cli_format.py +0 -85
- java_codebase_rag/cli_progress.py +0 -94
- java_codebase_rag/config.py +0 -833
- java_codebase_rag/eval/__init__.py +0 -1
- java_codebase_rag/eval/ground_truth.py +0 -100
- java_codebase_rag/eval/metrics.py +0 -107
- java_codebase_rag/eval/runner.py +0 -556
- java_codebase_rag/graph/__init__.py +0 -0
- java_codebase_rag/graph/build_ast_graph.py +0 -4593
- java_codebase_rag/graph/graph_enrich.py +0 -1940
- java_codebase_rag/graph/graph_types.py +0 -224
- java_codebase_rag/graph/java_ontology.py +0 -465
- java_codebase_rag/graph/ladybug_queries.py +0 -2213
- java_codebase_rag/graph/path_filtering.py +0 -509
- java_codebase_rag/index/__init__.py +0 -0
- java_codebase_rag/index/java_index_flow_lancedb.py +0 -879
- java_codebase_rag/index/java_index_v1_common.py +0 -33
- java_codebase_rag/install_data/__init__.py +0 -0
- java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -110
- java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
- java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
- java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
- java_codebase_rag/installer.py +0 -2188
- java_codebase_rag/jrag.py +0 -4545
- java_codebase_rag/jrag_envelope.py +0 -1107
- java_codebase_rag/jrag_hints.py +0 -204
- java_codebase_rag/jrag_render.py +0 -926
- java_codebase_rag/lance_optimize.py +0 -264
- java_codebase_rag/mcp/__init__.py +0 -0
- java_codebase_rag/mcp/mcp_hints.py +0 -932
- java_codebase_rag/mcp/mcp_v2.py +0 -1916
- java_codebase_rag/mcp/server.py +0 -886
- java_codebase_rag/pipeline.py +0 -531
- java_codebase_rag/progress.py +0 -570
- java_codebase_rag/read_payloads.py +0 -781
- java_codebase_rag/search/__init__.py +0 -0
- java_codebase_rag/search/index_common.py +0 -10
- java_codebase_rag/search/search_lancedb.py +0 -1296
- java_codebase_rag/search/search_lexical.py +0 -449
- java_codebase_rag/search/search_scoring.py +0 -537
- java_codebase_rag/watch/__init__.py +0 -0
- java_codebase_rag/watch/client.py +0 -230
- java_codebase_rag/watch/daemon.py +0 -396
- java_codebase_rag/watch/lock.py +0 -201
- java_codebase_rag/watch/paths.py +0 -76
- java_codebase_rag/watch/protocol.py +0 -122
- java_codebase_rag/watch/server.py +0 -273
- java_codebase_rag/watch/warm.py +0 -105
- java_codebase_rag/watch/watcher.py +0 -394
- java_codebase_rag-0.12.0.dist-info/METADATA +0 -340
- java_codebase_rag-0.12.0.dist-info/RECORD +0 -75
- java_codebase_rag-0.12.0.dist-info/entry_points.txt +0 -5
- java_codebase_rag-0.12.0.dist-info/licenses/LICENSE +0 -21
- java_codebase_rag-0.12.0.dist-info/top_level.txt +0 -1
- /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.2.dist-info/top_level.txt +0 -0
|
@@ -1,537 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env python3
|
|
2
|
-
"""Dependency-free scoring & dedup primitives shared by the search backends.
|
|
3
|
-
|
|
4
|
-
Imported by both the vector backend (`search_lancedb`) and the lexical backend
|
|
5
|
-
(`search_lexical`). This module MUST NOT import lancedb / torch /
|
|
6
|
-
sentence_transformers / cocoindex — it is imported on graph-only (macOS Intel)
|
|
7
|
-
installs where those packages are absent (see pyproject.toml PEP 508 markers).
|
|
8
|
-
|
|
9
|
-
Everything here is pure-Python dict/list math with no third-party deps.
|
|
10
|
-
"""
|
|
11
|
-
|
|
12
|
-
from __future__ import annotations
|
|
13
|
-
|
|
14
|
-
import json
|
|
15
|
-
import re
|
|
16
|
-
from dataclasses import dataclass
|
|
17
|
-
|
|
18
|
-
# Name of the LadybugDB FTS (Okapi BM25) index over Symbol.search_text (fork A).
|
|
19
|
-
# Shared by the build path (build_ast_graph._ensure_symbol_fts_index) and the
|
|
20
|
-
# query path (search_lexical.run_lexical_search) so the two never drift.
|
|
21
|
-
SYMBOL_FTS_INDEX = "sym_fts"
|
|
22
|
-
|
|
23
|
-
# Over-fetch multiplier for dedup: fetch 4x to absorb per-FQN chunk multiplicity
|
|
24
|
-
# so that after collapsing by primary_type_fqn, a page stays full and the +1
|
|
25
|
-
# truncation sentinel survives. The formula: need = max((limit + offset) * 4, limit + offset + 1)
|
|
26
|
-
DEDUP_OVERFETCH = 4
|
|
27
|
-
|
|
28
|
-
_IMPORT_DISTANCE_PENALTY = 0.08
|
|
29
|
-
_IMPORT_HYBRID_SCORE_FACTOR = 0.88
|
|
30
|
-
|
|
31
|
-
# Bonus for chunks whose declared symbols (method / field names) share tokens with
|
|
32
|
-
# the query. Behavioural queries like "what happens when a client message arrives"
|
|
33
|
-
# should float chunks containing `processClientMessage` above ones that only
|
|
34
|
-
# enqueue; this is a cheap, query-dependent signal computed at rank time.
|
|
35
|
-
_SYMBOL_MATCH_BONUS_PER_HIT = 0.03
|
|
36
|
-
_SYMBOL_MATCH_BONUS_CAP = 0.06
|
|
37
|
-
|
|
38
|
-
# Action verbs that typically mark behavioural entry points in this codebase.
|
|
39
|
-
# A chunk whose symbols begin with one of these verbs earns a small flat bump
|
|
40
|
-
# — again only for JVM (Java/Kotlin) chunks and only when role-filtering is off.
|
|
41
|
-
_ACTION_VERB_PREFIXES: tuple[str, ...] = (
|
|
42
|
-
"process", "handle", "on", "pick", "select", "assign",
|
|
43
|
-
"notify", "dispatch", "publish", "consume", "route",
|
|
44
|
-
"trigger", "enqueue", "distribute", "update", "create",
|
|
45
|
-
"apply", "resolve", "reassign", "close", "open",
|
|
46
|
-
)
|
|
47
|
-
_ACTION_VERB_BONUS = 0.02
|
|
48
|
-
|
|
49
|
-
# Type-name overlap bonus. The class name is a much stronger discovery signal
|
|
50
|
-
# than any individual method, because class naming in this codebase encodes
|
|
51
|
-
# the domain concept (`DistributionChunkService`, `OperatorSessionService`,
|
|
52
|
-
# `JoinOperatorController`). So we reward overlap between query tokens and the
|
|
53
|
-
# simple name of `primary_type_fqn` more heavily than per-method overlap, and
|
|
54
|
-
# we stack it on top of the existing `_symbol_bonus`.
|
|
55
|
-
_TYPE_MATCH_BONUS_PER_HIT = 0.05
|
|
56
|
-
_TYPE_MATCH_BONUS_CAP = 0.10
|
|
57
|
-
|
|
58
|
-
_STOPWORDS: frozenset[str] = frozenset({
|
|
59
|
-
"a", "an", "the", "is", "are", "was", "were", "be", "been", "being",
|
|
60
|
-
"to", "of", "in", "on", "at", "by", "for", "with", "from", "as", "or",
|
|
61
|
-
"and", "but", "if", "then", "else", "when", "what", "how", "why", "does",
|
|
62
|
-
"do", "did", "has", "have", "had", "this", "that", "these", "those", "it",
|
|
63
|
-
"its", "new", "no", "not", "will", "would", "should", "can", "could",
|
|
64
|
-
"may", "might", "happens", "happen", "happened", "get", "gets", "got",
|
|
65
|
-
})
|
|
66
|
-
|
|
67
|
-
# Role-aware reweighting for Java/Kotlin chunks. Positive values favour actionable
|
|
68
|
-
# behavioural code (entrypoints, orchestrators, integrations) over configuration,
|
|
69
|
-
# schema, and persistence stubs for "what happens when..."-style queries.
|
|
70
|
-
# Applied to the similarity score (higher = better); distance-based sort subtracts
|
|
71
|
-
# the weight. Skipped when caller filters explicitly by role.
|
|
72
|
-
_ROLE_SCORE_WEIGHTS: dict[str, float] = {
|
|
73
|
-
"CONTROLLER": 0.10,
|
|
74
|
-
"SERVICE": 0.08,
|
|
75
|
-
"CLIENT": 0.06,
|
|
76
|
-
"COMPONENT": 0.03,
|
|
77
|
-
"REPOSITORY": 0.02,
|
|
78
|
-
"MAPPER": 0.00,
|
|
79
|
-
"OTHER": 0.00,
|
|
80
|
-
"ENTITY": -0.06,
|
|
81
|
-
"CONFIG": -0.10,
|
|
82
|
-
# DTOs are passive data carriers; they almost never answer "how/what
|
|
83
|
-
# happens" queries. Penalty is slightly stronger than ENTITY so a DTO
|
|
84
|
-
# with a great embedding match still loses to a mediocre SERVICE hit.
|
|
85
|
-
"DTO": -0.08,
|
|
86
|
-
}
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
def _rrf_max(num_lists: int, k: int = 60) -> float:
|
|
90
|
-
"""Return the theoretical maximum RRF score for N-list fusion.
|
|
91
|
-
|
|
92
|
-
Reciprocal Rank Fusion (RRF) bounds each contribution to ≤ 1/(rank + k).
|
|
93
|
-
For N fused lists, the maximum possible sum is N/(k + 1) (achieved when
|
|
94
|
-
an item ranks #1 across all lists).
|
|
95
|
-
|
|
96
|
-
Args:
|
|
97
|
-
num_lists: Number of ranked lists being fused (e.g., 2 for vector+lexical).
|
|
98
|
-
k: The RRF constant (default 60 per the original paper).
|
|
99
|
-
|
|
100
|
-
Returns:
|
|
101
|
-
The maximum RRF contribution: num_lists / (k + 1).
|
|
102
|
-
"""
|
|
103
|
-
return num_lists / (k + 1)
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
# Allowed list names in a RankConfig: vector is always required (it is the
|
|
107
|
-
# backbone retrieval signal); graph and bm25 are optional fusion participants.
|
|
108
|
-
# The "bm25" list is wired in ``search_lancedb._graph_expand_merge`` (LadybugDB
|
|
109
|
-
# FTS candidate fetch), which fuses BM25-ranked Symbol candidates as a third
|
|
110
|
-
# RRF list alongside vector and graph.
|
|
111
|
-
_RANK_LIST_NAMES: frozenset[str] = frozenset({"vector", "graph", "bm25"})
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
@dataclass(frozen=True)
|
|
115
|
-
class RankConfig:
|
|
116
|
-
"""Which ranked lists to fuse and the RRF constant to fuse them with.
|
|
117
|
-
|
|
118
|
-
This is a dep-free value object (no lancedb/torch) so it can be constructed
|
|
119
|
-
on every install flavor, including graph-only (macOS Intel). It is plumbed
|
|
120
|
-
through ``run_search`` → ``_graph_expand_merge`` to control (a) which lists
|
|
121
|
-
contribute to the final RRF fusion and (b) the ``k`` constant passed into
|
|
122
|
-
``_rrf_merge``.
|
|
123
|
-
|
|
124
|
-
Attributes:
|
|
125
|
-
lists: Subset of ``{"vector", "graph", "bm25"}``. Must contain
|
|
126
|
-
``"vector"`` (the backbone retrieval signal) and be non-empty.
|
|
127
|
-
rrf_k: The RRF constant (default 60 per the original paper). Must be ≥ 1.
|
|
128
|
-
"""
|
|
129
|
-
|
|
130
|
-
lists: frozenset[str]
|
|
131
|
-
rrf_k: int = 60
|
|
132
|
-
|
|
133
|
-
def __post_init__(self) -> None:
|
|
134
|
-
if not isinstance(self.lists, frozenset) or not self.lists:
|
|
135
|
-
raise ValueError("RankConfig.lists must be a non-empty frozenset")
|
|
136
|
-
if "vector" not in self.lists:
|
|
137
|
-
raise ValueError(
|
|
138
|
-
"RankConfig.lists must contain 'vector' (the backbone signal)"
|
|
139
|
-
)
|
|
140
|
-
unknown = self.lists - _RANK_LIST_NAMES
|
|
141
|
-
if unknown:
|
|
142
|
-
raise ValueError(
|
|
143
|
-
f"RankConfig.lists has unknown names {sorted(unknown)!r}; "
|
|
144
|
-
f"allowed: {sorted(_RANK_LIST_NAMES)!r}"
|
|
145
|
-
)
|
|
146
|
-
if not isinstance(self.rrf_k, int) or self.rrf_k < 1:
|
|
147
|
-
raise ValueError(f"RankConfig.rrf_k must be an int >= 1, got {self.rrf_k!r}")
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
# Production default: ship the 3-list config (vector+graph+bm25). The BM25 list
|
|
151
|
-
# is wired in ``search_lancedb._graph_expand_merge``; on installs without the
|
|
152
|
-
# vector stack, or when the FTS index is unavailable, it degrades silently to
|
|
153
|
-
# the 2-list (vector+graph) fusion.
|
|
154
|
-
DEFAULT_RANK_CONFIG = RankConfig(lists=frozenset({"vector", "graph", "bm25"}), rrf_k=60)
|
|
155
|
-
|
|
156
|
-
# Eval convenience: the historical 2-list (vector+graph) fusion, used by
|
|
157
|
-
# evaluation harnesses that isolate the vector+graph baseline from the bm25 list.
|
|
158
|
-
BASELINE_2LIST_CONFIG = RankConfig(lists=frozenset({"vector", "graph"}), rrf_k=60)
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
# Theoretical maximum for hybrid composite score (used for display normalization).
|
|
162
|
-
# Hybrid sort metric: raw_rrf * (import_factor if import_heavy else 1)
|
|
163
|
-
# + role_weight + symbol_bonus
|
|
164
|
-
# where raw_rrf ≤ 2/(k+1) for 2-list RRF, role_weight ≤ max(_ROLE_SCORE_WEIGHTS),
|
|
165
|
-
# and symbol_bonus ≤ _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS.
|
|
166
|
-
# The import factor is ≤ 1, so we use the raw max (derived via _rrf_max(2)).
|
|
167
|
-
_HYBRID_SCORE_MAX = _rrf_max(2) + max(_ROLE_SCORE_WEIGHTS.values()) + _SYMBOL_MATCH_BONUS_CAP + _TYPE_MATCH_BONUS_CAP + _ACTION_VERB_BONUS
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
def _query_tokens(query: str) -> set[str]:
|
|
171
|
-
"""Lowercased alpha-only tokens from the query, minus stopwords, len >= 3.
|
|
172
|
-
|
|
173
|
-
Used to score symbol-name overlap; we keep it simple and locale-free.
|
|
174
|
-
"""
|
|
175
|
-
out: set[str] = set()
|
|
176
|
-
cur: list[str] = []
|
|
177
|
-
|
|
178
|
-
def _flush() -> None:
|
|
179
|
-
if cur:
|
|
180
|
-
tok = "".join(cur).lower()
|
|
181
|
-
cur.clear()
|
|
182
|
-
if len(tok) >= 3 and tok not in _STOPWORDS:
|
|
183
|
-
out.add(tok)
|
|
184
|
-
|
|
185
|
-
for c in query:
|
|
186
|
-
if c.isalpha():
|
|
187
|
-
cur.append(c)
|
|
188
|
-
else:
|
|
189
|
-
_flush()
|
|
190
|
-
_flush()
|
|
191
|
-
return out
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
def _split_identifier(name: str) -> list[str]:
|
|
195
|
-
"""camelCase / snake_case -> lowercase token list."""
|
|
196
|
-
parts: list[str] = []
|
|
197
|
-
cur: list[str] = []
|
|
198
|
-
for c in name:
|
|
199
|
-
if c == "_":
|
|
200
|
-
if cur:
|
|
201
|
-
parts.append("".join(cur).lower())
|
|
202
|
-
cur = []
|
|
203
|
-
elif c.isupper() and cur:
|
|
204
|
-
parts.append("".join(cur).lower())
|
|
205
|
-
cur = [c]
|
|
206
|
-
else:
|
|
207
|
-
cur.append(c)
|
|
208
|
-
if cur:
|
|
209
|
-
parts.append("".join(cur).lower())
|
|
210
|
-
return [p for p in parts if p]
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
# Alphanumeric-word extractor for FTS query building (mirrors the index side, which
|
|
214
|
-
# runs the same regex over name/fqn/signature/annotations/capabilities/package fields).
|
|
215
|
-
_FTS_WORD_RE = re.compile(r"[A-Za-z][A-Za-z0-9]*")
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
def build_fts_query(text: str) -> str:
|
|
219
|
-
"""Tokenize a search query into the ``Symbol.search_text`` token space (fork A).
|
|
220
|
-
|
|
221
|
-
``search_text`` is indexed from ``_split_identifier`` tokens (camelCase / snake_case
|
|
222
|
-
split, lowercased). LadybugDB FTS's own tokenizer does NOT split camelCase, so a raw
|
|
223
|
-
pasted identifier like ``DistributionChunkService`` would match nothing — this
|
|
224
|
-
extracts alphanumeric words from the query and splits each via ``_split_identifier``
|
|
225
|
-
so the query lands in the index's token space. Tokens shorter than 2 chars are
|
|
226
|
-
dropped (mirrors the index side); duplicates collapse. Returns ``""`` for a query
|
|
227
|
-
with no usable tokens — the caller then falls back to the heuristic / role listing.
|
|
228
|
-
"""
|
|
229
|
-
out: list[str] = []
|
|
230
|
-
seen: set[str] = set()
|
|
231
|
-
for word in _FTS_WORD_RE.findall(text or ""):
|
|
232
|
-
for tok in _split_identifier(word):
|
|
233
|
-
if len(tok) >= 2 and tok not in seen:
|
|
234
|
-
seen.add(tok)
|
|
235
|
-
out.append(tok)
|
|
236
|
-
return " ".join(out)
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
def _is_jvm_row(r: dict) -> bool:
|
|
240
|
-
"""True when a row is Java or Kotlin source (both indexed in the java table).
|
|
241
|
-
|
|
242
|
-
The additive role/bonus weighting applies to both JVM languages. Kotlin
|
|
243
|
-
chunks live in the java LanceDB table (``_kind == "java"``), but we key off
|
|
244
|
-
the semantic ``language`` field when present so detection is robust to the
|
|
245
|
-
row's table-key and any future language-keyed table split.
|
|
246
|
-
"""
|
|
247
|
-
lang = str(r.get("language") or "").lower()
|
|
248
|
-
if lang in ("java", "kotlin"):
|
|
249
|
-
return True
|
|
250
|
-
return str(r.get("_kind", "")) == "java"
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
def _symbol_bonus(r: dict, query_toks: set[str]) -> float:
|
|
254
|
-
"""Symbol-name overlap + action-verb bump for Java/Kotlin chunks.
|
|
255
|
-
|
|
256
|
-
Caps at `_SYMBOL_MATCH_BONUS_CAP + _ACTION_VERB_BONUS` to avoid runaway
|
|
257
|
-
ranks on chunks declaring many symbols.
|
|
258
|
-
"""
|
|
259
|
-
if not _is_jvm_row(r):
|
|
260
|
-
return 0.0
|
|
261
|
-
raw = r.get("symbols") or []
|
|
262
|
-
if isinstance(raw, str):
|
|
263
|
-
# Legacy JSON-encoded list column; parse defensively.
|
|
264
|
-
try:
|
|
265
|
-
parsed = json.loads(raw)
|
|
266
|
-
raw = parsed if isinstance(parsed, list) else []
|
|
267
|
-
except Exception:
|
|
268
|
-
raw = []
|
|
269
|
-
symbols = [str(s) for s in raw if s]
|
|
270
|
-
|
|
271
|
-
overlap_hits = 0
|
|
272
|
-
has_action = False
|
|
273
|
-
for s in symbols:
|
|
274
|
-
bare = s.split("(", 1)[0].strip()
|
|
275
|
-
if not bare:
|
|
276
|
-
continue
|
|
277
|
-
toks = _split_identifier(bare)
|
|
278
|
-
if toks:
|
|
279
|
-
if toks[0] in _ACTION_VERB_PREFIXES:
|
|
280
|
-
has_action = True
|
|
281
|
-
if query_toks & set(toks):
|
|
282
|
-
overlap_hits += 1
|
|
283
|
-
|
|
284
|
-
bonus = min(overlap_hits * _SYMBOL_MATCH_BONUS_PER_HIT, _SYMBOL_MATCH_BONUS_CAP)
|
|
285
|
-
if has_action:
|
|
286
|
-
bonus += _ACTION_VERB_BONUS
|
|
287
|
-
|
|
288
|
-
# Type-name overlap: strongest single lexical signal for "which class is
|
|
289
|
-
# the answer?" queries. Uses the simple name of primary_type_fqn.
|
|
290
|
-
fqn = str(r.get("primary_type_fqn") or "")
|
|
291
|
-
if fqn:
|
|
292
|
-
simple = fqn.rsplit(".", 1)[-1]
|
|
293
|
-
type_toks = set(_split_identifier(simple))
|
|
294
|
-
type_hits = len(query_toks & type_toks)
|
|
295
|
-
if type_hits:
|
|
296
|
-
bonus += min(type_hits * _TYPE_MATCH_BONUS_PER_HIT, _TYPE_MATCH_BONUS_CAP)
|
|
297
|
-
return bonus
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
def _role_weight(r: dict) -> float:
|
|
301
|
-
"""Effective role weight for a row, captured into `_score_components.role_weight`."""
|
|
302
|
-
comps = r.setdefault("_score_components", {})
|
|
303
|
-
cached = comps.get("role_weight")
|
|
304
|
-
if cached is not None:
|
|
305
|
-
return float(cached)
|
|
306
|
-
if r.get("_skip_role_weight") or not _is_jvm_row(r):
|
|
307
|
-
comps["role_weight"] = 0.0
|
|
308
|
-
return 0.0
|
|
309
|
-
role = (r.get("role") or "").upper()
|
|
310
|
-
w = _ROLE_SCORE_WEIGHTS.get(role, 0.0)
|
|
311
|
-
comps["role_weight"] = w
|
|
312
|
-
return w
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
def _apply_symbol_bonus(rows: list[dict], query_toks: set[str]) -> None:
|
|
316
|
-
"""Pre-compute symbol-match bonus into `_score_components.symbol_bonus`."""
|
|
317
|
-
if not query_toks:
|
|
318
|
-
return
|
|
319
|
-
for r in rows:
|
|
320
|
-
if r.get("_skip_role_weight"):
|
|
321
|
-
# When the caller locked role, respect their intent everywhere.
|
|
322
|
-
continue
|
|
323
|
-
b = _symbol_bonus(r, query_toks)
|
|
324
|
-
if b:
|
|
325
|
-
r.setdefault("_score_components", {})["symbol_bonus"] = b
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
def l2_distance_to_score(distance: float) -> float:
|
|
329
|
-
"""Map L2 distance to a similarity score for unit-normalized embeddings."""
|
|
330
|
-
return 1.0 - distance * distance / 2.0
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
# Display-score denominator for the vector backend. Unit-normalized embeddings
|
|
334
|
-
# have L2 distance in [0, 2]; the cosine map ``l2_distance_to_score`` (1 - d²/2)
|
|
335
|
-
# goes NEGATIVE past √2 ≈ 1.414 and clamps to 0. Weak-but-best semantic matches
|
|
336
|
-
# (e.g. a lone keyword like "controller") commonly sit at d ≈ 1.5, so EVERY hit
|
|
337
|
-
# clamps to score=0.000 even though the ranking is correct. ``vector_display_score``
|
|
338
|
-
# instead normalizes the effective (bonus-adjusted) distance over the full
|
|
339
|
-
# unit-embedding range, so a top-ranked hit stays visibly non-zero. Role/symbol
|
|
340
|
-
# bonuses reduce the effective distance and so raise the displayed score,
|
|
341
|
-
# keeping it rank-monotonic with the distance-based sort key.
|
|
342
|
-
_VECTOR_DISTANCE_REF = 2.0
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
def vector_display_score(effective_distance: float) -> float:
|
|
346
|
-
"""Displayed vector score in [0, 1] from the effective (bonus-adjusted) distance.
|
|
347
|
-
|
|
348
|
-
Bounded linear normalization over the unit-embedding L2 range [0, 2]: lower
|
|
349
|
-
distance → higher score. Unlike ``l2_distance_to_score`` (which goes
|
|
350
|
-
negative past √2 and clamps a correctly-ranked top hit to 0.000), this keeps
|
|
351
|
-
a top result visibly non-zero while staying rank-monotonic with the sort key.
|
|
352
|
-
"""
|
|
353
|
-
return _clamp01(1.0 - effective_distance / _VECTOR_DISTANCE_REF)
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
def _effective_distance(comps: dict[str, float]) -> float:
|
|
357
|
-
"""Compute the adjusted distance used for sorting.
|
|
358
|
-
|
|
359
|
-
Matches _vector_sort_key logic: distance + import_penalty - role_weight - symbol_bonus.
|
|
360
|
-
"""
|
|
361
|
-
d = comps.get("distance", 0.0)
|
|
362
|
-
d += comps.get("import_penalty", 0.0)
|
|
363
|
-
d -= comps.get("role_weight", 0.0)
|
|
364
|
-
d -= comps.get("symbol_bonus", 0.0)
|
|
365
|
-
return d
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
def _clamp01(x: float) -> float:
|
|
369
|
-
"""Clamp a value to the [0.0, 1.0] range."""
|
|
370
|
-
if x < 0.0:
|
|
371
|
-
return 0.0
|
|
372
|
-
if x > 1.0:
|
|
373
|
-
return 1.0
|
|
374
|
-
return x
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
# Matches a Java top-level type declaration and captures its simple name. Mirrors
|
|
378
|
-
# the heuristic in ``ast.chunk_heuristics._JAVA_TYPE`` but is duplicated here so
|
|
379
|
-
# this module stays dependency-free (importable on graph-only Intel installs).
|
|
380
|
-
_JAVA_TYPE_DECL_RE = re.compile(
|
|
381
|
-
r"\b(?:public\s+|private\s+|protected\s+|sealed\s+|non-sealed\s+|final\s+|"
|
|
382
|
-
r"abstract\s+|static\s+)*"
|
|
383
|
-
r"(?:class|interface|enum|record)\s+([A-Za-z_][A-Za-z0-9_]*)"
|
|
384
|
-
)
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
def declaration_line_number(
|
|
388
|
-
text: str | None, anchor_line: int | None, type_name: str | None = None
|
|
389
|
-
) -> int | None:
|
|
390
|
-
"""Absolute 1-based line of the Java type declaration within chunk ``text``.
|
|
391
|
-
|
|
392
|
-
LanceDB chunks are anchored at the chunk's first source line, which for a
|
|
393
|
-
file-spanning chunk is the package/import line (``anchor_line`` = 1) while
|
|
394
|
-
the ``class``/``interface`` declaration sits several lines down. Without
|
|
395
|
-
this, hits render as ``File.java:1`` even though the symbol is declared
|
|
396
|
-
later (F8). Returns ``anchor_line + i`` for the first matching declaration
|
|
397
|
-
(pinned to ``type_name`` when given, so a nested type doesn't win), or
|
|
398
|
-
``anchor_line`` unchanged when no declaration is found in the chunk.
|
|
399
|
-
|
|
400
|
-
Comment-aware: Javadoc/line/block-comment lines that merely MENTION the type
|
|
401
|
-
name (e.g. ``* This class Bar handles...``) are skipped so the returned line
|
|
402
|
-
is the real declaration, not a comment above it.
|
|
403
|
-
"""
|
|
404
|
-
if not text or anchor_line is None:
|
|
405
|
-
return anchor_line
|
|
406
|
-
in_block = False
|
|
407
|
-
for i, raw in enumerate(text.splitlines()):
|
|
408
|
-
# Drop a trailing ``// ...`` line comment before any matching (a ``//``
|
|
409
|
-
# inside a string literal is unrealistic for a declaration line).
|
|
410
|
-
code = raw.split("//", 1)[0]
|
|
411
|
-
stripped = code.strip()
|
|
412
|
-
if in_block:
|
|
413
|
-
if "*/" in stripped:
|
|
414
|
-
in_block = False
|
|
415
|
-
continue
|
|
416
|
-
if stripped.startswith("/*"):
|
|
417
|
-
# Single-line ``/* ... */`` -> skip without entering block state.
|
|
418
|
-
if "*/" not in stripped[2:]:
|
|
419
|
-
in_block = True
|
|
420
|
-
continue
|
|
421
|
-
if not stripped or stripped.startswith("*"):
|
|
422
|
-
# Blank or a Javadoc continuation line (`` * ...``).
|
|
423
|
-
continue
|
|
424
|
-
m = _JAVA_TYPE_DECL_RE.search(code)
|
|
425
|
-
if m and (not type_name or m.group(1) == type_name):
|
|
426
|
-
return anchor_line + i
|
|
427
|
-
return anchor_line
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
def explain_score_components(
|
|
431
|
-
comps: dict[str, float] | None,
|
|
432
|
-
*,
|
|
433
|
-
role: str | None = None,
|
|
434
|
-
hybrid: bool = False,
|
|
435
|
-
graph_expanded: bool = False,
|
|
436
|
-
lexical: bool = False,
|
|
437
|
-
) -> str:
|
|
438
|
-
"""Compact human-readable 'why' string for a ranked hit.
|
|
439
|
-
|
|
440
|
-
Joins the interesting components of `_score_components` in a stable order
|
|
441
|
-
so agents can reason about rankings without chasing raw floats. Returns
|
|
442
|
-
"" if there's nothing worth mentioning.
|
|
443
|
-
"""
|
|
444
|
-
if not comps:
|
|
445
|
-
comps = {}
|
|
446
|
-
parts: list[str] = []
|
|
447
|
-
if lexical:
|
|
448
|
-
rel = comps.get("lexical_relevance")
|
|
449
|
-
if rel is not None:
|
|
450
|
-
parts.append(f"relevance={float(rel):.2f}")
|
|
451
|
-
nm = comps.get("name_match")
|
|
452
|
-
if nm is not None:
|
|
453
|
-
parts.append(f"name={float(nm):.2f}")
|
|
454
|
-
ty = comps.get("type_match")
|
|
455
|
-
if ty:
|
|
456
|
-
parts.append(f"type:{float(ty):+.2f}")
|
|
457
|
-
fq = comps.get("fqn_match")
|
|
458
|
-
if fq:
|
|
459
|
-
parts.append(f"fqn:{float(fq):+.2f}")
|
|
460
|
-
elif hybrid:
|
|
461
|
-
# Prefer rrf_raw (added by PR-SEARCH-1a) for explanation
|
|
462
|
-
rrf = comps.get("rrf_raw") or comps.get("hybrid_rrf")
|
|
463
|
-
if rrf is not None:
|
|
464
|
-
parts.append(f"rrf={float(rrf):.3f}")
|
|
465
|
-
bm25 = comps.get("bm25")
|
|
466
|
-
if bm25:
|
|
467
|
-
parts.append(f"bm25={float(bm25):.3f}")
|
|
468
|
-
else:
|
|
469
|
-
d = comps.get("distance")
|
|
470
|
-
if d is not None:
|
|
471
|
-
parts.append(f"dist={float(d):.2f}")
|
|
472
|
-
rw = comps.get("role_weight")
|
|
473
|
-
if rw:
|
|
474
|
-
label = f"role:{role}" if role else "role"
|
|
475
|
-
parts.append(f"{label}:{float(rw):+.02f}")
|
|
476
|
-
sb = comps.get("symbol_bonus")
|
|
477
|
-
if sb:
|
|
478
|
-
parts.append(f"symbol:{float(sb):+.02f}")
|
|
479
|
-
ip = comps.get("import_penalty")
|
|
480
|
-
if ip:
|
|
481
|
-
parts.append(f"import_penalty:{float(ip):+.02f}")
|
|
482
|
-
if graph_expanded:
|
|
483
|
-
parts.append("graph")
|
|
484
|
-
return " ".join(parts)
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
def _dedup_by_fqn(rows: list[dict], dedup_by_fqn: bool = True) -> list[dict]:
|
|
488
|
-
"""Deduplicate rows by primary_type_fqn (java table only).
|
|
489
|
-
|
|
490
|
-
When dedup_by_fqn is True, collapses multiple chunks of the same
|
|
491
|
-
primary_type_fqn into one row (first-seen-wins, since rows are pre-sorted
|
|
492
|
-
so the first is the best chunk). Each survivor gets a _chunks_collapsed
|
|
493
|
-
field (>=1) counting how many rows were collapsed into it.
|
|
494
|
-
|
|
495
|
-
Rows without primary_type_fqn (sql/yaml tables) get a unique __id:<id>
|
|
496
|
-
key so they pass through unchanged (each row is unique).
|
|
497
|
-
|
|
498
|
-
When dedup_by_fqn is False, returns rows unchanged (regression guard).
|
|
499
|
-
"""
|
|
500
|
-
if not dedup_by_fqn:
|
|
501
|
-
# Non-dedup path: return unchanged, byte-identical to prior behavior
|
|
502
|
-
return rows
|
|
503
|
-
|
|
504
|
-
deduped: list[dict] = []
|
|
505
|
-
seen_keys: dict[str, dict] = {}
|
|
506
|
-
collapsed_counts: dict[str, int] = {}
|
|
507
|
-
|
|
508
|
-
for row in rows:
|
|
509
|
-
# Build dedup key: primary_type_fqn for java rows, unique __id:<id> for sql/yaml
|
|
510
|
-
fqn = row.get("primary_type_fqn")
|
|
511
|
-
if fqn:
|
|
512
|
-
key = str(fqn)
|
|
513
|
-
else:
|
|
514
|
-
# sql/yaml rows have no primary_type_fqn → unique key per row
|
|
515
|
-
row_id = row.get("id") or id(row)
|
|
516
|
-
key = f"__id:{row_id}"
|
|
517
|
-
|
|
518
|
-
if key not in seen_keys:
|
|
519
|
-
# First occurrence: keep it
|
|
520
|
-
seen_keys[key] = row
|
|
521
|
-
collapsed_counts[key] = 1
|
|
522
|
-
deduped.append(row)
|
|
523
|
-
else:
|
|
524
|
-
# Duplicate: increment collapse count, discard this row
|
|
525
|
-
collapsed_counts[key] += 1
|
|
526
|
-
|
|
527
|
-
# Annotate each survivor with _chunks_collapsed
|
|
528
|
-
for row in deduped:
|
|
529
|
-
fqn = row.get("primary_type_fqn")
|
|
530
|
-
if fqn:
|
|
531
|
-
key = str(fqn)
|
|
532
|
-
else:
|
|
533
|
-
row_id = row.get("id") or id(row)
|
|
534
|
-
key = f"__id:{row_id}"
|
|
535
|
-
row["_chunks_collapsed"] = collapsed_counts[key]
|
|
536
|
-
|
|
537
|
-
return deduped
|
|
File without changes
|