java-codebase-rag 0.11.2__py3-none-any.whl → 0.12.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. java_codebase_rag-0.12.1.dist-info/METADATA +35 -0
  2. java_codebase_rag-0.12.1.dist-info/RECORD +4 -0
  3. {java_codebase_rag-0.11.2.dist-info → java_codebase_rag-0.12.1.dist-info}/WHEEL +1 -1
  4. java_codebase_rag/_fdlimit.py +0 -56
  5. java_codebase_rag/_stdio.py +0 -32
  6. java_codebase_rag/_version.py +0 -35
  7. java_codebase_rag/absence/__init__.py +0 -0
  8. java_codebase_rag/absence/absence_diagnosis.py +0 -700
  9. java_codebase_rag/absence/absence_types.py +0 -124
  10. java_codebase_rag/absence/absence_vocab.py +0 -460
  11. java_codebase_rag/analysis/__init__.py +0 -0
  12. java_codebase_rag/analysis/pr_analysis.py +0 -563
  13. java_codebase_rag/analysis/resolve_service.py +0 -740
  14. java_codebase_rag/ast/__init__.py +0 -0
  15. java_codebase_rag/ast/ast_java.py +0 -2825
  16. java_codebase_rag/ast/brownfield_events.py +0 -58
  17. java_codebase_rag/ast/chunk_heuristics.py +0 -62
  18. java_codebase_rag/cli.py +0 -1215
  19. java_codebase_rag/cli_format.py +0 -85
  20. java_codebase_rag/cli_progress.py +0 -94
  21. java_codebase_rag/config.py +0 -833
  22. java_codebase_rag/eval/__init__.py +0 -1
  23. java_codebase_rag/eval/ground_truth.py +0 -100
  24. java_codebase_rag/eval/metrics.py +0 -107
  25. java_codebase_rag/eval/runner.py +0 -556
  26. java_codebase_rag/graph/__init__.py +0 -0
  27. java_codebase_rag/graph/build_ast_graph.py +0 -4471
  28. java_codebase_rag/graph/graph_enrich.py +0 -1937
  29. java_codebase_rag/graph/graph_types.py +0 -224
  30. java_codebase_rag/graph/java_ontology.py +0 -465
  31. java_codebase_rag/graph/ladybug_queries.py +0 -2213
  32. java_codebase_rag/graph/path_filtering.py +0 -477
  33. java_codebase_rag/index/__init__.py +0 -0
  34. java_codebase_rag/index/java_index_flow_lancedb.py +0 -734
  35. java_codebase_rag/index/java_index_v1_common.py +0 -33
  36. java_codebase_rag/install_data/__init__.py +0 -0
  37. java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -108
  38. java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
  39. java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
  40. java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
  41. java_codebase_rag/installer.py +0 -2188
  42. java_codebase_rag/jrag.py +0 -4531
  43. java_codebase_rag/jrag_envelope.py +0 -1107
  44. java_codebase_rag/jrag_hints.py +0 -204
  45. java_codebase_rag/jrag_render.py +0 -926
  46. java_codebase_rag/lance_optimize.py +0 -264
  47. java_codebase_rag/mcp/__init__.py +0 -0
  48. java_codebase_rag/mcp/mcp_hints.py +0 -932
  49. java_codebase_rag/mcp/mcp_v2.py +0 -1916
  50. java_codebase_rag/mcp/server.py +0 -884
  51. java_codebase_rag/pipeline.py +0 -531
  52. java_codebase_rag/progress.py +0 -570
  53. java_codebase_rag/read_payloads.py +0 -781
  54. java_codebase_rag/search/__init__.py +0 -0
  55. java_codebase_rag/search/index_common.py +0 -10
  56. java_codebase_rag/search/search_lancedb.py +0 -1296
  57. java_codebase_rag/search/search_lexical.py +0 -449
  58. java_codebase_rag/search/search_scoring.py +0 -523
  59. java_codebase_rag/watch/__init__.py +0 -0
  60. java_codebase_rag/watch/client.py +0 -230
  61. java_codebase_rag/watch/daemon.py +0 -396
  62. java_codebase_rag/watch/lock.py +0 -201
  63. java_codebase_rag/watch/paths.py +0 -76
  64. java_codebase_rag/watch/protocol.py +0 -122
  65. java_codebase_rag/watch/server.py +0 -273
  66. java_codebase_rag/watch/warm.py +0 -105
  67. java_codebase_rag/watch/watcher.py +0 -370
  68. java_codebase_rag-0.11.2.dist-info/METADATA +0 -331
  69. java_codebase_rag-0.11.2.dist-info/RECORD +0 -71
  70. java_codebase_rag-0.11.2.dist-info/entry_points.txt +0 -4
  71. java_codebase_rag-0.11.2.dist-info/licenses/LICENSE +0 -21
  72. java_codebase_rag-0.11.2.dist-info/top_level.txt +0 -1
  73. /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.1.dist-info/top_level.txt +0 -0
@@ -1,1916 +0,0 @@
1
- """MCP V2 graph query surface (``search`` / ``find`` / ``describe`` / ``neighbors`` / ``resolve``).
2
-
3
- Strict frame contract
4
- ---------------------
5
- NodeFilter is a typed predicate bag: each populated field maps to one stored graph
6
- attribute for the selected kind; inapplicable fields fail loud with a teaching message.
7
- The ``search`` tool's ``query`` parameter is the ranked-text carve-out; the substring
8
- fields (``fqn_contains``, ``path_contains``, ``target_path_contains``, ``topic_contains``)
9
- match literally (Cypher ``CONTAINS``) — no wildcard/metacharacter handling.
10
-
11
- Revisit trigger (``propose/completed/MCP-FILTER-FRAME-PROPOSE.md`` section 3.4.6)
12
- --------------------------------------------------------------
13
- If **three** legitimate issue-tracker workflows appear within **six months** of frame
14
- lock where the strict frame has no clean analog under ``search``, deferred
15
- ``resolve``, or documented multi-call patterns, reopen the frame for revision.
16
- """
17
-
18
- from __future__ import annotations
19
-
20
- import json
21
- import os
22
- import sys
23
- from pathlib import Path
24
- import threading
25
- from typing import Annotated, Any, Literal, TYPE_CHECKING, get_args
26
-
27
- from pydantic import BaseModel, ConfigDict, Field, TypeAdapter, ValidationError, model_validator, validate_call
28
-
29
- if TYPE_CHECKING:
30
- # Eager import would pull torch at module load. The vector stack is optional (graph-only
31
- # installs ship without torch/lancedb); it is imported lazily in _get_sentence_transformer.
32
- from sentence_transformers import SentenceTransformer
33
-
34
- from java_codebase_rag.absence.absence_types import AbsenceDiagnosis
35
- from java_codebase_rag.absence.absence_diagnosis import diagnose
36
- from java_codebase_rag.absence.absence_vocab import get_vocabulary_index
37
- from java_codebase_rag.graph.graph_types import (
38
- NodeRef,
39
- StructuredHint,
40
- _hints_or_skip,
41
- _node_ref_from_row,
42
- _resolve_node_kind,
43
- _to_structured_hints,
44
- set_hints_enabled,
45
- )
46
- from java_codebase_rag.search.index_common import SBERT_MODEL
47
- from java_codebase_rag.config import resolved_sbert_model_for_process_env
48
- from java_codebase_rag.graph.java_ontology import EDGE_SCHEMA
49
- from java_codebase_rag.graph.ladybug_queries import LadybugGraph, OVERRIDE_AXIS_COMPOSED_EDGE_TYPES
50
- from java_codebase_rag.mcp.mcp_hints import MCP_HINTS_STRUCTURED_FIELD_DESCRIPTION
51
-
52
- # The vector stack (lancedb/torch, reached via search_lancedb) is optional — it is absent on
53
- # graph-only installs (macOS Intel). It is imported LAZILY on the first ``search_v2`` call via
54
- # ``_ensure_vector_backend``, NOT at module load. Reason: ``search_lancedb`` imports
55
- # ``sentence_transformers``/``torch`` at its own top level (~2.8s), and the warm ``jrag watch``
56
- # client reconstructs ``SearchOutput`` from this module (see ``watch/client._reconstruct``)
57
- # without ever calling the vector backend — an eager import here would force every warm CLI
58
- # read to pay that ~2.8s, masking the daemon's win.
59
- #
60
- # ``run_search`` uses a distinct ``_NOT_LOADED`` sentinel (not ``None``) for the unloaded state,
61
- # so ``None`` unambiguously means "lexical/disabled". This lets tests force the lexical path by
62
- # monkeypatching ``run_search=None`` (``_ensure_vector_backend`` treats that as authoritative and
63
- # will not overwrite it), while a non-None monkeypatch drives the vector path. ``TABLES`` is
64
- # ``{}`` until first successful load (only read in the vector path, where it is guaranteed set).
65
- _NOT_LOADED = object()
66
- run_search: Any = _NOT_LOADED
67
- TABLES: dict = {}
68
- _vector_backend_lock = threading.Lock()
69
-
70
-
71
- def _ensure_vector_backend() -> None:
72
- """Populate ``run_search``/``TABLES`` from ``search_lancedb`` on first use.
73
-
74
- No-op once ``run_search`` has left the ``_NOT_LOADED`` state — real callable (vector path),
75
- ``None`` (graph-only install or a test-forced lexical mode), or a test monkeypatch.
76
- Thread-safe (double-checked locking): the MCP server dispatches ``search_v2`` via
77
- ``asyncio.to_thread``, so two concurrent first-batch searches could otherwise race the
78
- sentinel mutation. On ImportError (graph-only) ``run_search`` becomes ``None`` and
79
- ``search_v2`` takes the lexical fallback path. A non-ImportError propagates (caught by
80
- ``search_v2``'s outer handler) and leaves ``run_search`` as ``_NOT_LOADED`` so the next
81
- call retries rather than poisoning the process.
82
- """
83
- global run_search, TABLES
84
- if run_search is not _NOT_LOADED:
85
- return
86
- with _vector_backend_lock:
87
- if run_search is not _NOT_LOADED:
88
- return
89
- try:
90
- from java_codebase_rag.search.search_lancedb import TABLES as _TABLES, run_search as _run_search
91
- except ImportError: # graph-only install: no torch/lancedb — lexical fallback
92
- run_search = None
93
- else:
94
- run_search = _run_search
95
- TABLES = _TABLES
96
- __all__ = [
97
- "search_v2",
98
- "find_v2",
99
- "describe_v2",
100
- "neighbors_v2",
101
- "resolve_v2",
102
- "SearchOutput",
103
- "FindOutput",
104
- "DescribeOutput",
105
- "NeighborsOutput",
106
- "ResolveOutput",
107
- "ResolveCandidate",
108
- "ResolveStatus",
109
- "NodeRef",
110
- "NodeFilter",
111
- "EdgeFilter",
112
- "StructuredHint",
113
- "set_hints_enabled",
114
- "set_absence_config",
115
- ]
116
-
117
- DeclarationSymbolKind = Literal["class", "interface", "enum", "record", "annotation", "method", "constructor"]
118
-
119
- # Closed value taxonomies surfaced to MCP consumers as enums. Sources of truth:
120
- # Role — VALID_ROLES in java_ontology.py + the "OTHER" inference fallback (ast_java.infer_role)
121
- # Framework — hardcoded literals across ast_java.py / build_ast_graph.py
122
- # SourceLayer — exhaustive classifier build_ast_graph._client_source_layer / _producer_source_layer
123
- # ClientKind — VALID_CLIENT_KINDS in java_ontology.py (every producer validated at index time)
124
- # ProducerKind — VALID_PRODUCER_KINDS in java_ontology.py (every producer validated at index time)
125
- # Keep these in sync with the indexing-side taxonomies if they change.
126
- Role = Literal[
127
- "CONTROLLER", "SERVICE", "REPOSITORY", "COMPONENT", "CONFIG",
128
- "ENTITY", "CLIENT", "MAPPER", "DTO", "OTHER",
129
- ]
130
- Framework = Literal["spring_mvc", "webflux", "kafka", "rabbitmq", "jms", "stream", "feign", ""]
131
- SourceLayer = Literal["builtin", "layer_a_meta", "layer_b_ann", "layer_b_fqn", "layer_c_source"]
132
- ClientKind = Literal["feign_method", "rest_template", "web_client"]
133
- ProducerKind = Literal["kafka_send", "stream_bridge_send"]
134
-
135
- # Stored graph edge labels for one-hop neighbors. Composed DECLARES.* and OVERRIDDEN_BY.*
136
- # dot-keys are separate ComposedEdgeType literals (2-hop traversal). Stored OVERRIDES is an EdgeType.
137
- EdgeType = Literal[
138
- "EXTENDS",
139
- "IMPLEMENTS",
140
- "INJECTS",
141
- "OVERRIDES",
142
- "DECLARES",
143
- "DECLARES_CLIENT",
144
- "DECLARES_PRODUCER",
145
- "CALLS",
146
- "EXPOSES",
147
- "HTTP_CALLS",
148
- "ASYNC_CALLS",
149
- ]
150
-
151
- ComposedEdgeType = Literal[
152
- "DECLARES.DECLARES_CLIENT",
153
- "DECLARES.DECLARES_PRODUCER",
154
- "DECLARES.EXPOSES",
155
- "OVERRIDDEN_BY",
156
- "OVERRIDDEN_BY.DECLARES_CLIENT",
157
- "OVERRIDDEN_BY.DECLARES_PRODUCER",
158
- "OVERRIDDEN_BY.EXPOSES",
159
- ]
160
-
161
- NeighborEdgeType = EdgeType | ComposedEdgeType
162
-
163
- _COMPOSED_EDGE_TYPES = frozenset(get_args(ComposedEdgeType))
164
- _MEMBER_COMPOSED_EDGE_TYPES = frozenset(
165
- k for k in _COMPOSED_EDGE_TYPES if k.startswith("DECLARES.")
166
- )
167
- _OVERRIDE_COMPOSED_EDGE_TYPES = OVERRIDE_AXIS_COMPOSED_EDGE_TYPES
168
-
169
- _NEIGHBOR_EDGE_TYPES_ADAPTER = TypeAdapter(
170
- Annotated[
171
- list[NeighborEdgeType],
172
- Field(min_length=1, description="At least one graph edge label or DECLARES.* dot-key"),
173
- ]
174
- )
175
-
176
- _st_lock = threading.Lock()
177
- _st_model: SentenceTransformer | None = None
178
-
179
- _TYPE_SYMBOL_KINDS_FOR_EDGE_ROLLUP = frozenset(
180
- {"class", "interface", "enum", "record", "annotation"}
181
- )
182
-
183
- _METHOD_SYMBOL_KINDS_FOR_OVERRIDE_ROLLUP = frozenset({"method"})
184
-
185
- _fail_loud_counts: dict[str, int] = {}
186
- _fail_loud_lock = threading.Lock()
187
-
188
- # Module-level holder for absence diagnosis config (set by server.py)
189
- _absence_config: Any = None
190
-
191
-
192
- def set_absence_config(cfg: Any) -> None:
193
- """Set the global absence diagnosis config for MCP tools.
194
-
195
- Mirrors set_hints_enabled: called from server.py to make cfg reachable
196
- in tool functions without threading it through every signature.
197
- """
198
- global _absence_config
199
- _absence_config = cfg
200
-
201
-
202
- def _get_absence_config() -> Any:
203
- """Get the absence diagnosis config, with safe fallback.
204
-
205
- Returns the module-level config if set; otherwise falls back to a
206
- default config (for direct tool calls in tests without server init).
207
- """
208
- if _absence_config is not None:
209
- return _absence_config
210
-
211
- # Fallback: resolve from cwd for direct test calls
212
- from java_codebase_rag.config import resolve_operator_config
213
- from pathlib import Path
214
-
215
- return resolve_operator_config(source_root=Path.cwd())
216
-
217
-
218
- def _log_fail_loud(category: str) -> None:
219
- """Increment process-local fail-loud counter and emit one stderr line (PR-FRAME-3).
220
-
221
- The stderr line is gated on ``JAVA_CODEBASE_RAG_FAIL_LOUD`` (default ``"1"`` =
222
- emit) so the MCP server keeps its operator diagnostic while the agent-facing
223
- ``jrag`` CLI (which surfaces the same failure as a clean status:error
224
- envelope) can run it with the diagnostic silenced.
225
- """
226
- with _fail_loud_lock:
227
- _fail_loud_counts[category] = _fail_loud_counts.get(category, 0) + 1
228
- n = _fail_loud_counts[category]
229
- if os.environ.get("JAVA_CODEBASE_RAG_FAIL_LOUD", "1") != "0":
230
- print(f"[filter-frame] fail-loud category={category} count={n}", file=sys.stderr, flush=True)
231
-
232
-
233
- def filter_frame_counters() -> dict[str, int]:
234
- """Snapshot of fail-loud counts (tests / local diagnostics; not an MCP tool)."""
235
- with _fail_loud_lock:
236
- return dict(_fail_loud_counts)
237
-
238
-
239
- def _get_sentence_transformer(model_name: str, device: str | None) -> SentenceTransformer:
240
- global _st_model
241
- from sentence_transformers import SentenceTransformer
242
-
243
- with _st_lock:
244
- if _st_model is None:
245
- _st_model = SentenceTransformer(
246
- model_name,
247
- device=device,
248
- trust_remote_code=True,
249
- )
250
- return _st_model
251
-
252
-
253
- class NodeFilter(BaseModel):
254
- model_config = ConfigDict(extra="forbid")
255
-
256
- microservice: str | None = None
257
- module: str | None = None
258
- source_layer: SourceLayer | None = None
259
- role: Role | None = None
260
- exclude_roles: list[Role] | None = None
261
- generated_only: bool = False
262
- exclude_generated: bool = False
263
- annotation: str | None = None
264
- capability: str | None = None
265
- fqn_contains: str | None = None
266
- symbol_kind: DeclarationSymbolKind | None = None
267
- symbol_kinds: list[DeclarationSymbolKind] | None = None
268
- http_method: str | None = Field(
269
- default=None,
270
- description="HTTP verb (commonly GET/POST/PUT/DELETE/PATCH; user route annotations may yield others).",
271
- )
272
- path_contains: str | None = None
273
- framework: Framework | None = None
274
- client_kind: ClientKind | None = Field(
275
- default=None,
276
- description="Outbound HTTP client kind: feign_method, rest_template, or web_client.",
277
- )
278
- target_service: str | None = None
279
- target_path_contains: str | None = None
280
- producer_kind: ProducerKind | None = Field(
281
- default=None,
282
- description="Outbound async producer kind: kafka_send or stream_bridge_send.",
283
- )
284
- topic_contains: str | None = None
285
-
286
-
287
- class EdgeFilter(BaseModel):
288
- model_config = ConfigDict(extra="forbid")
289
-
290
- min_confidence: float | None = None
291
- exclude_strategies: list[str] | None = None
292
- include_strategies: list[str] | None = None
293
- callee_declaring_role: Role | None = None
294
- callee_declaring_roles: list[Role] | None = None
295
- exclude_callee_declaring_roles: list[Role] | None = None
296
-
297
- @model_validator(mode="after")
298
- def _strategy_axes_mutually_exclusive(self) -> EdgeFilter:
299
- has_include = bool(self.include_strategies)
300
- has_exclude = bool(self.exclude_strategies)
301
- if has_include and has_exclude:
302
- raise ValueError("include_strategies and exclude_strategies are mutually exclusive")
303
- return self
304
-
305
- @model_validator(mode="after")
306
- def _role_axes_mutually_exclusive(self) -> EdgeFilter:
307
- role_axes = (
308
- self.callee_declaring_role is not None,
309
- bool(self.callee_declaring_roles),
310
- bool(self.exclude_callee_declaring_roles),
311
- )
312
- if sum(role_axes) > 1:
313
- raise ValueError(
314
- "callee_declaring_role, callee_declaring_roles, and "
315
- "exclude_callee_declaring_roles are mutually exclusive"
316
- )
317
- return self
318
-
319
-
320
- _NODEFILTER_FIELD_ORDER: tuple[str, ...] = tuple(NodeFilter.model_fields.keys())
321
- _EDGEFILTER_FIELD_ORDER: tuple[str, ...] = tuple(EdgeFilter.model_fields.keys())
322
-
323
-
324
- # StructuredHint is now defined in graph_types.py and imported above
325
-
326
-
327
- # Populated EdgeFilter field -> EDGE_SCHEMA attribute name used in Cypher pushdown.
328
- _EDGEFILTER_FIELD_TO_ATTR: dict[str, str] = {
329
- "min_confidence": "confidence",
330
- "exclude_strategies": "strategy",
331
- "include_strategies": "strategy",
332
- "callee_declaring_role": "callee_declaring_role",
333
- "callee_declaring_roles": "callee_declaring_role",
334
- "exclude_callee_declaring_roles": "callee_declaring_role",
335
- }
336
-
337
- _ROLE_FILTER_OTHER_FALLBACK_VALUES = frozenset({"SERVICE", "REPOSITORY"})
338
-
339
- _NODEFILTER_APPLICABLE_FIELDS: dict[Literal["symbol", "route", "client", "producer"], tuple[str, ...]] = {
340
- "symbol": (
341
- "microservice",
342
- "module",
343
- "role",
344
- "exclude_roles",
345
- "generated_only",
346
- "exclude_generated",
347
- "annotation",
348
- "capability",
349
- "fqn_contains",
350
- "symbol_kind",
351
- "symbol_kinds",
352
- ),
353
- "route": (
354
- "microservice",
355
- "module",
356
- "http_method",
357
- "path_contains",
358
- "framework",
359
- ),
360
- "client": (
361
- "microservice",
362
- "module",
363
- "source_layer",
364
- "client_kind",
365
- "target_service",
366
- "target_path_contains",
367
- "http_method",
368
- ),
369
- "producer": (
370
- "microservice",
371
- "module",
372
- "source_layer",
373
- "producer_kind",
374
- "topic_contains",
375
- ),
376
- }
377
-
378
-
379
- def _ordered_nodefilter_fields(field_names: set[str]) -> list[str]:
380
- return [name for name in _NODEFILTER_FIELD_ORDER if name in field_names]
381
-
382
-
383
- def _populated_nodefilter_fields(nf: NodeFilter) -> set[str]:
384
- populated: set[str] = set()
385
- for field_name in _NODEFILTER_FIELD_ORDER:
386
- value = getattr(nf, field_name)
387
- if value is None:
388
- continue
389
- if isinstance(value, list) and not value:
390
- continue
391
- if isinstance(value, bool) and not value:
392
- # default-False NodeFilter fields (generated_only, exclude_generated) must not count as "populated"
393
- continue
394
- populated.add(field_name)
395
- return populated
396
-
397
-
398
- def _nodefilter_inapplicable_fields(
399
- kind: Literal["symbol", "route", "client", "producer"], nf: NodeFilter,
400
- ) -> list[str]:
401
- populated = _populated_nodefilter_fields(nf)
402
- applicable = set(_NODEFILTER_APPLICABLE_FIELDS[kind])
403
- return _ordered_nodefilter_fields(populated - applicable)
404
-
405
-
406
- def _nodefilter_applicability_error(
407
- kind: Literal["symbol", "route", "client", "producer"], nf: NodeFilter,
408
- ) -> str | None:
409
- inapplicable = _nodefilter_inapplicable_fields(kind, nf)
410
- if not inapplicable:
411
- return None
412
- applicable = ", ".join(_NODEFILTER_APPLICABLE_FIELDS[kind])
413
- bad = ", ".join(inapplicable)
414
- return (
415
- f"Invalid filter for kind='{kind}': populated field(s) not applicable: [{bad}]. "
416
- f"Applicable field(s): [{applicable}]"
417
- )
418
-
419
-
420
- def _filter_validation_error_message(exc: ValidationError) -> str:
421
- items: list[str] = []
422
- for err in exc.errors():
423
- loc = ".".join(str(part) for part in err.get("loc", ()))
424
- msg = str(err.get("msg") or "invalid value")
425
- if loc:
426
- items.append(f"{loc}: {msg}")
427
- else:
428
- items.append(msg)
429
- details = "; ".join(items) if items else str(exc)
430
- return f"Invalid filter: {details}"
431
-
432
-
433
- def _populated_edgefilter_fields(ef: EdgeFilter) -> set[str]:
434
- populated: set[str] = set()
435
- for field_name in _EDGEFILTER_FIELD_ORDER:
436
- value = getattr(ef, field_name)
437
- if value is None:
438
- continue
439
- if isinstance(value, list) and not value:
440
- continue
441
- if isinstance(value, bool) and not value:
442
- continue
443
- populated.add(field_name)
444
- return populated
445
-
446
-
447
- def _edge_schema_attr_names(edge_type: str) -> set[str]:
448
- spec = EDGE_SCHEMA.get(edge_type)
449
- if spec is None:
450
- return set()
451
- return {attr.name for attr in spec.attrs}
452
-
453
-
454
- def _edgefilter_applicability_error(edge_types: list[str], ef: EdgeFilter) -> str | None:
455
- populated = _populated_edgefilter_fields(ef)
456
- if not populated:
457
- return None
458
- flat_types = [et for et in edge_types if et not in _COMPOSED_EDGE_TYPES]
459
- composed = [et for et in edge_types if et in _COMPOSED_EDGE_TYPES]
460
- if composed or flat_types != ["CALLS"]:
461
- parts: list[str] = []
462
- if flat_types != ["CALLS"]:
463
- parts.append(f"stored labels {flat_types!r}")
464
- if composed:
465
- parts.append(f"composed keys {composed!r}")
466
- detail = " and ".join(parts) if parts else "requested edge_types"
467
- return (
468
- f"edge_filter requires edge_types=['CALLS'] only; {detail} is not supported — "
469
- "split into separate neighbors calls"
470
- )
471
- for edge_type in flat_types:
472
- available = _edge_schema_attr_names(edge_type)
473
- for field_name in _EDGEFILTER_FIELD_ORDER:
474
- if field_name not in populated:
475
- continue
476
- attr = _EDGEFILTER_FIELD_TO_ATTR[field_name]
477
- if attr not in available:
478
- return (
479
- f"{attr} is not on {edge_type}; restrict edge_types to ['CALLS'] "
480
- "or split into two neighbors_v2 calls"
481
- )
482
- return None
483
-
484
-
485
- # _to_structured_hints is now defined in graph_types.py and imported above
486
-
487
-
488
- def _coerce_edge_filter(
489
- value: EdgeFilter | dict[str, Any] | str | None,
490
- ) -> EdgeFilter | dict[str, Any] | None:
491
- """Normalize MCP tool input: weak clients sometimes pass JSON-encoded strings."""
492
- if value is None or isinstance(value, EdgeFilter):
493
- return value
494
- if isinstance(value, str):
495
- s = value.strip()
496
- if not s:
497
- return None
498
- try:
499
- decoded = json.loads(s)
500
- except json.JSONDecodeError as exc:
501
- raise ValueError(f"edge_filter must be a JSON object; invalid JSON: {exc.msg}") from exc
502
- if decoded is None:
503
- return None
504
- if not isinstance(decoded, dict):
505
- raise ValueError(
506
- f"edge_filter must decode to a JSON object, got {type(decoded).__name__}"
507
- )
508
- return decoded
509
- return value
510
-
511
-
512
- def _coerce_filter(
513
- value: NodeFilter | dict[str, Any] | str | None,
514
- ) -> NodeFilter | dict[str, Any] | None:
515
- """Normalize MCP tool input: weak clients sometimes pass JSON-encoded strings."""
516
- if value is None or isinstance(value, NodeFilter):
517
- return value
518
- if isinstance(value, str):
519
- s = value.strip()
520
- if not s:
521
- return None
522
- try:
523
- decoded = json.loads(s)
524
- except json.JSONDecodeError as exc:
525
- raise ValueError(f"filter must be a JSON object; invalid JSON: {exc.msg}") from exc
526
- if decoded is None:
527
- return None
528
- if not isinstance(decoded, dict):
529
- raise ValueError(f"filter must decode to a JSON object, got {type(decoded).__name__}")
530
- return decoded
531
- return value
532
-
533
-
534
- class SearchHit(BaseModel):
535
- chunk_id: str
536
- symbol_id: str | None = None
537
- fqn: str | None = None
538
- score: float
539
- snippet: str
540
- microservice: str | None = None
541
- module: str | None = None
542
- role: str | None = None
543
- generated: bool | None = None
544
- generated_by: str | None = None
545
- filename: str | None = None
546
- start_line: int | None = None
547
- score_components: dict[str, float] | None = None
548
- chunks: int | None = None
549
-
550
-
551
- # NodeRef is now defined in graph_types.py and imported above
552
-
553
-
554
- class NodeRecord(BaseModel):
555
- id: str
556
- kind: Literal["symbol", "route", "client", "producer"]
557
- fqn: str
558
- data: dict[str, Any] = Field(default_factory=dict)
559
- edge_summary: dict[str, dict[str, int]] | None = Field(
560
- default=None,
561
- description=(
562
- "Per graph edge label, in/out incident counts. For type Symbols (class, interface, "
563
- "enum, record, annotation), may also include composed dot-keys "
564
- "`DECLARES.DECLARES_CLIENT`, `DECLARES.DECLARES_PRODUCER`, and `DECLARES.EXPOSES`: 2-hop summaries "
565
- "(DECLARES to member, then that edge) — edge-row counts; navigable via neighbors for type "
566
- "Symbol origins (`direction=\"out\"` only). For non-static method Symbols, may include "
567
- "override-axis virtual keys `OVERRIDDEN_BY`, `OVERRIDDEN_BY.DECLARES_CLIENT`, "
568
- "`OVERRIDDEN_BY.DECLARES_PRODUCER`, `OVERRIDDEN_BY.EXPOSES` (stored `[:OVERRIDES]` "
569
- "dispatch hop, then terminal edges; navigable via neighbors for method Symbol origins, "
570
- "`direction=\"out\"` only; composed results include `via_id` in attrs). Plus an "
571
- "`OVERRIDES` map entry that **merges** stored `[:OVERRIDES]` in/out counts with the "
572
- "describe-time dispatch-up rollup (per direction `max`, so inbound stored overrides "
573
- "are not dropped). The stored relationship label `OVERRIDES` **is** also a valid "
574
- "EdgeType for one-hop neighbors (`direction=\"in\"` from declaration toward overriders)."
575
- ),
576
- )
577
-
578
-
579
- class Edge(BaseModel):
580
- origin_id: str
581
- edge_type: str
582
- direction: Literal["in", "out"]
583
- other: NodeRef
584
- attrs: dict[str, Any] = Field(default_factory=dict)
585
-
586
-
587
- class SearchOutput(BaseModel):
588
- success: bool
589
- results: list[SearchHit] = Field(default_factory=list)
590
- message: str | None = None
591
- limit: int | None = Field(
592
- default=None,
593
- description="Echoed from the request — the page size the server applied. None on success=False.",
594
- )
595
- offset: int | None = Field(
596
- default=None,
597
- description="Echoed from the request — the page offset the server applied. None on success=False.",
598
- )
599
- advisories: list[str] = Field(default_factory=list, description="Pure informational text with no tool call suggestion")
600
- hints_structured: list[StructuredHint] = Field(default_factory=list, description=MCP_HINTS_STRUCTURED_FIELD_DESCRIPTION)
601
- lexical_mode: bool = Field(
602
- default=False,
603
- description="True when results come from the graph-only lexical (keyword) backend instead of semantic/vector search.",
604
- )
605
- absence: AbsenceDiagnosis | None = None
606
-
607
-
608
- class FindOutput(BaseModel):
609
- success: bool
610
- results: list[NodeRef] = Field(default_factory=list)
611
- message: str | None = None
612
- limit: int | None = Field(
613
- default=None,
614
- description="Echoed from the request — the page size the server applied. None on success=False.",
615
- )
616
- offset: int | None = Field(
617
- default=None,
618
- description="Echoed from the request — the page offset the server applied. None on success=False.",
619
- )
620
- has_more_results: bool | None = Field(
621
- default=None,
622
- description="True when additional pages remain beyond offset+limit (more matches exist). "
623
- "None when unset (e.g. success=False).",
624
- )
625
- advisories: list[str] = Field(default_factory=list, description="Pure informational text with no tool call suggestion")
626
- hints_structured: list[StructuredHint] = Field(default_factory=list, description=MCP_HINTS_STRUCTURED_FIELD_DESCRIPTION)
627
- absence: AbsenceDiagnosis | None = None
628
-
629
-
630
- class DescribeOutput(BaseModel):
631
- success: bool
632
- record: NodeRecord | None = None
633
- message: str | None = None
634
- advisories: list[str] = Field(default_factory=list, description="Pure informational text with no tool call suggestion")
635
- hints_structured: list[StructuredHint] = Field(default_factory=list, description=MCP_HINTS_STRUCTURED_FIELD_DESCRIPTION)
636
- absence: AbsenceDiagnosis | None = None
637
-
638
-
639
- class NeighborsOutput(BaseModel):
640
- success: bool
641
- results: list[Edge] = Field(default_factory=list)
642
- message: str | None = None
643
- requested_edge_types: list[str] = Field(
644
- default_factory=list,
645
- description="Echo of neighbors(edge_types=...) from the request; empty when success=False.",
646
- )
647
- has_more_results: bool | None = Field(
648
- default=None,
649
- description="True when additional pages remain beyond offset+limit. None when unset or "
650
- "when the single-origin CALLS path paginated in SQL (use unfiltered_calls_count / "
651
- "calls_row_count there).",
652
- )
653
- advisories: list[str] = Field(default_factory=list, description="Pure informational text with no tool call suggestion")
654
- hints_structured: list[StructuredHint] = Field(default_factory=list, description=MCP_HINTS_STRUCTURED_FIELD_DESCRIPTION)
655
- absence: AbsenceDiagnosis | None = None
656
-
657
-
658
- # Re-exported from resolve_service.py (imported at end of module to avoid circular import)
659
- # resolve_v2, ResolveOutput, ResolveCandidate, ResolveStatus are imported below
660
-
661
-
662
- # _node_kind_from_id and _resolve_node_kind are now defined in graph_types.py and imported above
663
-
664
-
665
- def _chunk_id_from_row(row: dict[str, Any]) -> str:
666
- filename = str(row.get("filename") or "")
667
- start = row.get("start") or {}
668
- end = row.get("end") or {}
669
- sb = int(start.get("byte_offset") or 0) if isinstance(start, dict) else 0
670
- eb = int(end.get("byte_offset") or 0) if isinstance(end, dict) else 0
671
- return f"{filename}:{sb}:{eb}"
672
-
673
-
674
- def _row_to_search_hit(row: dict[str, Any], explain: bool = False) -> SearchHit:
675
- score = float(row.get("_rrf_score") or row.get("_score") or 0.0)
676
- filename = str(row.get("filename") or "") or None
677
- start_line: int | None = None
678
- start = row.get("start")
679
- if isinstance(start, dict):
680
- ln = start.get("line")
681
- if ln is not None:
682
- try:
683
- start_line = int(ln)
684
- except (TypeError, ValueError):
685
- start_line = None
686
- chunks = row.get("_chunks_collapsed")
687
- chunks_int = int(chunks) if chunks is not None and int(chunks) >= 2 else None
688
- return SearchHit(
689
- chunk_id=_chunk_id_from_row(row),
690
- symbol_id=_chunk_to_symbol_id(row),
691
- fqn=str(row.get("primary_type_fqn")) if row.get("primary_type_fqn") else None,
692
- score=score,
693
- snippet=str(row.get("text") or ""),
694
- microservice=str(row.get("microservice")) if row.get("microservice") else None,
695
- module=str(row.get("module")) if row.get("module") else None,
696
- role=str(row.get("role")) if row.get("role") else None,
697
- generated=bool(row.get("generated")) if row.get("generated") is not None else None,
698
- generated_by=str(row.get("generated_by")) if row.get("generated_by") else None,
699
- filename=filename,
700
- start_line=start_line,
701
- score_components=row.get("_score_components") if explain else None,
702
- chunks=chunks_int,
703
- )
704
-
705
-
706
- def _chunk_to_symbol_id(chunk_row: dict[str, Any]) -> str | None:
707
- symbol_id = chunk_row.get("symbol_id")
708
- if symbol_id:
709
- return str(symbol_id)
710
- meta = chunk_row.get("metadata")
711
- if isinstance(meta, str):
712
- try:
713
- parsed = json.loads(meta)
714
- if isinstance(parsed, dict):
715
- meta = parsed
716
- except Exception:
717
- meta = None
718
- if isinstance(meta, dict):
719
- nested = meta.get("symbol_id")
720
- if nested:
721
- return str(nested)
722
- return None
723
-
724
-
725
- def _symbol_where_from_filter(f: NodeFilter) -> tuple[str, dict[str, Any]]:
726
- preds: list[str] = []
727
- params: dict[str, Any] = {}
728
- if f.microservice:
729
- preds.append("s.microservice = $microservice")
730
- params["microservice"] = f.microservice
731
- if f.module:
732
- preds.append("s.module = $module")
733
- params["module"] = f.module
734
- if f.role:
735
- preds.append("s.role = $role")
736
- params["role"] = f.role
737
- if f.exclude_roles:
738
- preds.append("NOT s.role IN $exclude_roles")
739
- params["exclude_roles"] = list(f.exclude_roles)
740
- if f.generated_only:
741
- preds.append("s.generated = true")
742
- if f.exclude_generated:
743
- preds.append("(s.generated IS NULL OR s.generated = false)")
744
- if f.annotation:
745
- preds.append("list_contains(s.annotations, $annotation)")
746
- params["annotation"] = f.annotation
747
- if f.capability:
748
- preds.append("$capability IN s.capabilities")
749
- params["capability"] = f.capability
750
- if f.fqn_contains:
751
- preds.append("s.fqn CONTAINS $fqn_contains")
752
- params["fqn_contains"] = f.fqn_contains
753
- if f.symbol_kind:
754
- preds.append("s.kind = $symbol_kind")
755
- params["symbol_kind"] = f.symbol_kind
756
- if f.symbol_kinds:
757
- preds.append("s.kind IN $symbol_kinds")
758
- params["symbol_kinds"] = list(f.symbol_kinds)
759
- where = f"WHERE {' AND '.join(preds)}" if preds else ""
760
- return where, params
761
-
762
-
763
- # _node_ref_from_row is now defined in graph_types.py and imported above
764
-
765
-
766
- def _load_node_record(
767
- graph: LadybugGraph, node_id: str, kind: Literal["symbol", "route", "client", "producer"],
768
- ) -> dict[str, Any] | None:
769
- if kind == "symbol":
770
- projection = (
771
- "n.id AS id, n.kind AS kind, n.name AS name, n.fqn AS fqn, n.package AS package, "
772
- "n.module AS module, n.microservice AS microservice, n.filename AS filename, "
773
- "n.start_line AS start_line, n.end_line AS end_line, n.start_byte AS start_byte, "
774
- "n.end_byte AS end_byte, n.modifiers AS modifiers, n.annotations AS annotations, "
775
- "n.capabilities AS capabilities, n.role AS role, n.signature AS signature, "
776
- "n.parent_id AS parent_id, n.resolved AS resolved, n.generated AS generated, n.generated_by AS generated_by"
777
- )
778
- label = "Symbol"
779
- elif kind == "route":
780
- projection = (
781
- "n.id AS id, n.kind AS kind, n.framework AS framework, n.method AS method, n.path AS path, "
782
- "n.path_template AS path_template, n.path_regex AS path_regex, n.topic AS topic, "
783
- "n.broker AS broker, n.feign_name AS feign_name, n.feign_url AS feign_url, "
784
- "n.microservice AS microservice, n.module AS module, n.filename AS filename, "
785
- "n.start_line AS start_line, n.end_line AS end_line, n.resolved AS resolved"
786
- )
787
- label = "Route"
788
- elif kind == "client":
789
- projection = (
790
- "n.id AS id, n.client_kind AS client_kind, n.target_service AS target_service, "
791
- "n.method AS method, n.path AS path, n.path_template AS path_template, "
792
- "n.path_regex AS path_regex, n.member_fqn AS member_fqn, n.member_id AS member_id, "
793
- "n.microservice AS microservice, n.module AS module, n.filename AS filename, "
794
- "n.start_line AS start_line, n.end_line AS end_line, n.resolved AS resolved, "
795
- "n.source_layer AS source_layer"
796
- )
797
- label = "Client"
798
- else:
799
- projection = (
800
- "n.id AS id, n.producer_kind AS producer_kind, n.topic AS topic, n.broker AS broker, "
801
- "n.direction AS direction, n.member_fqn AS member_fqn, n.member_id AS member_id, "
802
- "n.microservice AS microservice, n.module AS module, n.filename AS filename, "
803
- "n.start_line AS start_line, n.end_line AS end_line, n.resolved AS resolved, "
804
- "n.source_layer AS source_layer"
805
- )
806
- label = "Producer"
807
- rows = graph._rows(f"MATCH (n:{label}) WHERE n.id = $id RETURN {projection}", {"id": node_id}) # noqa: SLF001
808
- if not rows:
809
- return None
810
- return rows[0]
811
-
812
-
813
- def _incident_counts(cell: dict[str, int] | None) -> dict[str, int]:
814
- if not cell:
815
- return {"in": 0, "out": 0}
816
- return {"in": int(cell.get("in", 0)), "out": int(cell.get("out", 0))}
817
-
818
-
819
- def _merge_overrides_edge_summary(
820
- stored_before_rollups: dict[str, int],
821
- summary_after_rollups: dict[str, dict[str, int]],
822
- ) -> None:
823
- """Reconcile `OVERRIDES` with `override_axis_rollup_for` without clobbering stored `in`.
824
-
825
- Rollup rows reuse the ``OVERRIDES`` key for dispatch-up counts only (``in`` is always
826
- zero there). Stored ``[:OVERRIDES]`` edges contribute real ``in``/``out`` from LadybugDB;
827
- merge per direction with ``max`` so inbound override edges stay visible.
828
- """
829
- roll = _incident_counts(summary_after_rollups.get("OVERRIDES"))
830
- if "OVERRIDES" not in summary_after_rollups and not any(stored_before_rollups.values()):
831
- return
832
- merged_in = max(stored_before_rollups["in"], roll["in"])
833
- merged_out = max(stored_before_rollups["out"], roll["out"])
834
- if merged_in == 0 and merged_out == 0:
835
- summary_after_rollups.pop("OVERRIDES", None)
836
- else:
837
- summary_after_rollups["OVERRIDES"] = {"in": merged_in, "out": merged_out}
838
-
839
-
840
- def _edge_summary_for_node(
841
- graph: LadybugGraph, node_id: str, *, kind: str, row: dict[str, Any]
842
- ) -> dict[str, dict[str, int]]:
843
- summary = dict(graph.edge_counts_for(node_id))
844
- sym_kind = str(row.get("kind") or "")
845
- if kind == "symbol" and sym_kind in _TYPE_SYMBOL_KINDS_FOR_EDGE_ROLLUP:
846
- summary.update(graph.member_edge_rollup_for(node_id))
847
- elif kind == "symbol" and sym_kind in _METHOD_SYMBOL_KINDS_FOR_OVERRIDE_ROLLUP:
848
- stored_overrides = _incident_counts(summary.get("OVERRIDES"))
849
- summary.update(graph.override_axis_rollup_for(node_id))
850
- _merge_overrides_edge_summary(stored_overrides, summary)
851
- return summary
852
-
853
-
854
- def _node_matches_filter(
855
- kind: Literal["symbol", "route", "client", "producer"], row: dict[str, Any], f: NodeFilter | None,
856
- ) -> bool:
857
- if f is None:
858
- return True
859
- if f.microservice and str(row.get("microservice") or "") != f.microservice:
860
- return False
861
- if f.module and str(row.get("module") or "") != f.module:
862
- return False
863
- if kind in ("client", "producer") and f.source_layer and str(row.get("source_layer") or "") != f.source_layer:
864
- return False
865
- if kind == "symbol":
866
- role = str(row.get("role") or "")
867
- fqn_val = str(row.get("fqn") or row.get("primary_type_fqn") or "")
868
- symbol_kind_val = str(row.get("kind") or row.get("symbol_kind") or "")
869
- if f.role and role != f.role:
870
- return False
871
- if f.exclude_roles and role in set(f.exclude_roles):
872
- return False
873
- generated = row.get("generated")
874
- if f.generated_only and not generated:
875
- return False
876
- if f.exclude_generated and generated:
877
- return False
878
- if f.annotation and f.annotation not in list(row.get("annotations") or []):
879
- return False
880
- if f.capability and f.capability not in list(row.get("capabilities") or []):
881
- return False
882
- if f.fqn_contains and f.fqn_contains not in fqn_val:
883
- return False
884
- if f.symbol_kind and symbol_kind_val != f.symbol_kind:
885
- return False
886
- if f.symbol_kinds and symbol_kind_val not in set(f.symbol_kinds):
887
- return False
888
- elif kind == "route":
889
- if f.http_method and str(row.get("method") or "") != f.http_method:
890
- return False
891
- if f.path_contains:
892
- path = str(row.get("path") or "")
893
- if f.path_contains not in path:
894
- return False
895
- if f.framework and str(row.get("framework") or "") != f.framework:
896
- return False
897
- elif kind == "client":
898
- if f.client_kind and str(row.get("client_kind") or "") != f.client_kind:
899
- return False
900
- if f.target_service and str(row.get("target_service") or "") != f.target_service:
901
- return False
902
- if f.target_path_contains:
903
- path = str(row.get("path") or "")
904
- if f.target_path_contains not in path:
905
- return False
906
- if f.http_method and str(row.get("method") or "") != f.http_method:
907
- return False
908
- else:
909
- if f.producer_kind and str(row.get("producer_kind") or "") != f.producer_kind:
910
- return False
911
- if f.topic_contains:
912
- topic = str(row.get("topic") or "")
913
- if f.topic_contains not in topic:
914
- return False
915
- return True
916
-
917
-
918
- def search_v2(
919
- query: str,
920
- table: str = "java",
921
- hybrid: bool = False,
922
- limit: int = 5,
923
- offset: int = 0,
924
- path_contains: str | None = None,
925
- filter: NodeFilter | dict[str, Any] | str | None = None,
926
- explain: bool = False,
927
- graph: LadybugGraph | None = None,
928
- dedup: bool = True,
929
- ) -> SearchOutput:
930
- try:
931
- raw_filter = _coerce_filter(filter)
932
- try:
933
- nf = (
934
- NodeFilter.model_validate(raw_filter)
935
- if raw_filter is not None and not isinstance(raw_filter, NodeFilter)
936
- else raw_filter
937
- )
938
- except ValidationError as exc:
939
- _log_fail_loud("unknown_key")
940
- return SearchOutput(
941
- success=False,
942
- message=_filter_validation_error_message(exc),
943
- advisories=[],
944
- limit=None,
945
- offset=None,
946
- )
947
- if nf and (err := _nodefilter_applicability_error("symbol", nf)):
948
- _log_fail_loud("applicability")
949
- return SearchOutput(success=False, message=err, advisories=[], limit=None, offset=None)
950
- _ensure_vector_backend()
951
- advisories: list[str] = []
952
- lexical_mode = run_search is None
953
- if lexical_mode:
954
- # Graph-only install (macOS Intel: no torch/lancedb). Fall back to lexical
955
- # (keyword) search over the symbol graph that graph-only mode already builds.
956
- # run_lexical_search returns rows in the same shape as run_search, so the
957
- # shared row->hit loop below works unchanged. It raises (message contains
958
- # "lexical search unavailable") when no graph exists — caught by the outer
959
- # try -> success=False. It returns [] for sql/yaml (advisory below) and for
960
- # empty-but-valid results.
961
- try:
962
- from java_codebase_rag.search.search_lexical import run_lexical_search
963
- except ImportError: # pragma: no cover - search_lexical has no heavy deps
964
- run_lexical_search = None # type: ignore[assignment]
965
- if run_lexical_search is None:
966
- return SearchOutput(
967
- success=False,
968
- message="search unavailable: graph-only mode and lexical backend not importable.",
969
- advisories=[],
970
- limit=None,
971
- offset=None,
972
- )
973
- advisories.append(
974
- "lexical (graph-only) mode — keyword ranking only; "
975
- "semantic/vector search requires Apple Silicon, Linux, or Windows"
976
- )
977
- if table in ("sql", "yaml", "all"):
978
- advisories.append(
979
- "sql/yaml tables are not indexed in graph-only mode; only Java symbols were searched"
980
- )
981
- if hybrid:
982
- advisories.append("hybrid is ignored in graph-only lexical mode")
983
- rows = run_lexical_search(
984
- query,
985
- table=table,
986
- limit=limit,
987
- offset=offset,
988
- path_contains=path_contains,
989
- filter=nf,
990
- explain=explain,
991
- dedup=dedup,
992
- advisories=advisories,
993
- graph=graph,
994
- )
995
- else:
996
- # hybrid + table='all' is unsupported (hybrid fuses vector+FTS on ONE
997
- # table); fail fast with a clean envelope BEFORE loading the embedding
998
- # model. run_search also guards this — this is the user-facing fast path.
999
- if hybrid and table == "all":
1000
- return SearchOutput(
1001
- success=False,
1002
- message="hybrid search requires a single table; use java, sql, or yaml (not all)",
1003
- advisories=[],
1004
- limit=None,
1005
- offset=None,
1006
- )
1007
- model_name = resolved_sbert_model_for_process_env(SBERT_MODEL)
1008
- device = os.environ.get("SBERT_DEVICE") or None
1009
- model = _get_sentence_transformer(model_name, device)
1010
- uri = os.environ.get("JAVA_CODEBASE_RAG_INDEX_DIR", "").strip() or str(
1011
- (Path.cwd() / ".java-codebase-rag").resolve()
1012
- )
1013
- uri_path = Path(uri)
1014
- if not uri.startswith(("s3://", "gs://", "az://")) and uri_path.exists():
1015
- uri = str(uri_path.resolve())
1016
- table_keys = list(TABLES) if table == "all" else [table]
1017
-
1018
- # Graceful fallback: if hybrid=True and FTS index is missing (old index),
1019
- # retry with hybrid=False and return vector-only results with an advisory.
1020
- try:
1021
- rows = run_search(
1022
- query,
1023
- uri=uri,
1024
- table_keys=table_keys,
1025
- hybrid=hybrid,
1026
- limit=limit,
1027
- offset=offset,
1028
- path_substring=path_contains,
1029
- model_name=model_name,
1030
- device=device,
1031
- model=model,
1032
- # Push the NodeFilter structural predicates into the LanceDB query so
1033
- # they apply BEFORE pagination (issue #353) — previously they were only
1034
- # a post-filter on the already-paginated page, which could shrink or
1035
- # empty filtered pages even when many matches existed deeper in the
1036
- # ranking. _node_matches_filter below still re-checks every row (it
1037
- # covers the non-pushdownable fields and is the contract guarantee).
1038
- role=nf.role if nf else None,
1039
- module=nf.module if nf else None,
1040
- microservice=nf.microservice if nf else None,
1041
- capability=nf.capability if nf else None,
1042
- exclude_roles=nf.exclude_roles if nf else None,
1043
- exclude_generated=nf.exclude_generated if nf else None,
1044
- generated_only=nf.generated_only if nf else None,
1045
- dedup_by_fqn=dedup,
1046
- # Always-on 3-list RRF fusion (vector + graph + BM25) on the java
1047
- # path — design spec, issue #431. run_search guards this to the
1048
- # java single-table path and degrades silently to pure-vector when
1049
- # the graph/FTS index is unavailable. Omitting this kwarg left the
1050
- # fusion dormant for every user-facing search (jrag search / MCP
1051
- # search); see the search_v2 graph_expand integration test.
1052
- graph_expand=(table == "java"),
1053
- )
1054
- except Exception as exc:
1055
- # Check if this is a missing-FTS error (old index built before PR-SEARCH-3)
1056
- exc_text = str(exc).lower()
1057
- is_fts_missing = "full text search" in exc_text or "inverted index" in exc_text
1058
- if hybrid and is_fts_missing:
1059
- # Retry with vector-only search
1060
- rows = run_search(
1061
- query,
1062
- uri=uri,
1063
- table_keys=table_keys,
1064
- hybrid=False, # Fallback to vector-only
1065
- limit=limit,
1066
- offset=offset,
1067
- path_substring=path_contains,
1068
- model_name=model_name,
1069
- device=device,
1070
- model=model,
1071
- role=nf.role if nf else None,
1072
- module=nf.module if nf else None,
1073
- microservice=nf.microservice if nf else None,
1074
- capability=nf.capability if nf else None,
1075
- exclude_roles=nf.exclude_roles if nf else None,
1076
- exclude_generated=nf.exclude_generated if nf else None,
1077
- generated_only=nf.generated_only if nf else None,
1078
- dedup_by_fqn=dedup,
1079
- graph_expand=(table == "java"), # 3-list fusion survives the FTS fallback
1080
- )
1081
- advisories.append(
1082
- f"hybrid unavailable on table '{table}' (FTS index missing on this index built before "
1083
- f"PR-SEARCH-3); fell back to vector-only — reindex to enable hybrid"
1084
- )
1085
- else:
1086
- # Non-FTS error: surface as structured failure
1087
- raise
1088
- hits: list[SearchHit] = []
1089
- for row in rows:
1090
- if path_contains and path_contains not in str(row.get("filename") or ""):
1091
- continue
1092
- if nf:
1093
- row_kind = "symbol"
1094
- if not _node_matches_filter(row_kind, row, nf):
1095
- continue
1096
- hits.append(_row_to_search_hit(row, explain=explain))
1097
-
1098
- # Absence diagnosis for empty results
1099
- cfg = _get_absence_config()
1100
- diag: AbsenceDiagnosis | None = None
1101
- if not hits:
1102
- g = graph or LadybugGraph.get()
1103
- vocab = get_vocabulary_index(g, cfg)
1104
- diag = diagnose(
1105
- tool="search",
1106
- query=query,
1107
- filt=None,
1108
- filter_kind=None,
1109
- root_node=None,
1110
- scope={},
1111
- vocab=vocab,
1112
- graph=g,
1113
- cfg=cfg,
1114
- )
1115
-
1116
- hint_payload = {
1117
- "success": True,
1118
- "results": [h.model_dump() for h in hits],
1119
- "limit": limit,
1120
- "offset": offset,
1121
- }
1122
- raw_struct, raw_advisories = _hints_or_skip("search", hint_payload)
1123
- return SearchOutput(
1124
- success=True,
1125
- results=hits,
1126
- limit=limit,
1127
- offset=offset,
1128
- advisories=advisories + raw_advisories, # Merge fallback + hints advisories
1129
- hints_structured=_to_structured_hints(raw_struct),
1130
- lexical_mode=lexical_mode,
1131
- absence=diag,
1132
- )
1133
- except Exception as exc:
1134
- return SearchOutput(success=False, message=str(exc), advisories=[], limit=None, offset=None)
1135
-
1136
-
1137
- def find_v2(
1138
- kind: Literal["symbol", "route", "client", "producer"],
1139
- filter: NodeFilter | dict[str, Any] | str,
1140
- limit: int = 25,
1141
- offset: int = 0,
1142
- graph: LadybugGraph | None = None,
1143
- ) -> FindOutput:
1144
- try:
1145
- g = graph or LadybugGraph.get()
1146
- raw_filter = _coerce_filter(filter)
1147
- if raw_filter is None:
1148
- raw_filter = {}
1149
- try:
1150
- nf = NodeFilter.model_validate(raw_filter) if not isinstance(raw_filter, NodeFilter) else raw_filter
1151
- except ValidationError as exc:
1152
- _log_fail_loud("unknown_key")
1153
- return FindOutput(
1154
- success=False,
1155
- message=_filter_validation_error_message(exc),
1156
- advisories=[],
1157
- limit=None,
1158
- offset=None,
1159
- )
1160
- if err := _nodefilter_applicability_error(kind, nf):
1161
- _log_fail_loud("applicability")
1162
- return FindOutput(success=False, message=err, advisories=[], limit=None, offset=None)
1163
- fetch_cap = int(limit) + int(offset) + 1
1164
- if kind == "symbol":
1165
- where, params = _symbol_where_from_filter(nf)
1166
- # Exclude structural Symbol nodes. Files and packages are :Symbol-
1167
- # labeled (kind='file'/'package') but aren't code declarations —
1168
- # without this, `fqn_contains` matches their filesystem-path fqn
1169
- # (e.g. 'Assign' in '.../DevAssignmentController.java') and surfaces
1170
- # them as hits. Mirrors search_lexical.py. Safe to apply
1171
- # unconditionally: DeclarationSymbolKind (the only values
1172
- # symbol_kind/symbol_kinds can take) excludes 'file'/'package', so no
1173
- # filter ever requests them.
1174
- struct_pred = "(s.kind <> 'file' AND s.kind <> 'package')"
1175
- where = (
1176
- f"WHERE {struct_pred}"
1177
- if not where
1178
- else where.replace("WHERE ", f"WHERE {struct_pred} AND ", 1)
1179
- )
1180
- params["lim"] = fetch_cap
1181
- rows = g._rows( # noqa: SLF001
1182
- f"MATCH (s:Symbol) {where} RETURN s.id AS id, s.fqn AS fqn, s.name AS name, "
1183
- "s.filename AS filename, s.start_line AS start_line, s.microservice AS microservice, "
1184
- "s.module AS module, s.role AS role, s.kind AS symbol_kind, s.generated AS generated, s.generated_by AS generated_by ORDER BY s.fqn LIMIT $lim",
1185
- params,
1186
- )
1187
- elif kind == "route":
1188
- rows = g.list_routes(
1189
- microservice=nf.microservice,
1190
- framework=nf.framework,
1191
- path_contains=nf.path_contains,
1192
- method=nf.http_method,
1193
- limit=max(500, fetch_cap),
1194
- )
1195
- rows = [r for r in rows if _node_matches_filter("route", r, nf)]
1196
- elif kind == "client":
1197
- rows = g.list_clients(
1198
- microservice=nf.microservice,
1199
- client_kind=nf.client_kind,
1200
- target_service=nf.target_service,
1201
- path_contains=nf.target_path_contains,
1202
- method=nf.http_method,
1203
- limit=max(500, fetch_cap),
1204
- )
1205
- rows = [r for r in rows if _node_matches_filter("client", r, nf)]
1206
- else:
1207
- rows = g.list_producers(
1208
- microservice=nf.microservice,
1209
- producer_kind=nf.producer_kind,
1210
- topic_contains=nf.topic_contains,
1211
- limit=max(500, fetch_cap),
1212
- )
1213
- rows = [r for r in rows if _node_matches_filter("producer", r, nf)]
1214
- has_more_results = len(rows) > int(offset) + int(limit)
1215
- rows = rows[offset : offset + limit]
1216
- refs = [_node_ref_from_row(kind, r) for r in rows]
1217
- filter_dump = nf.model_dump(exclude_none=True)
1218
-
1219
- # Absence diagnosis for empty results
1220
- cfg = _get_absence_config()
1221
- diag: AbsenceDiagnosis | None = None
1222
- if not refs:
1223
- vocab = get_vocabulary_index(g, cfg)
1224
- diag = diagnose(
1225
- tool="find",
1226
- query=None,
1227
- filt=filter_dump,
1228
- filter_kind=kind,
1229
- root_node=None,
1230
- scope={},
1231
- vocab=vocab,
1232
- graph=g,
1233
- cfg=cfg,
1234
- )
1235
-
1236
- hint_payload: dict[str, Any] = {
1237
- "success": True,
1238
- "kind": kind,
1239
- # exclude_none: this dict feeds generate_hints (which reads fields
1240
- # defensively via .get), not the tool result (FindOutput below holds the
1241
- # pydantic objects). Drop null fields -- including the NodeRef.name field
1242
- # that is None for every structured ref -- to match filter_dump above and
1243
- # avoid spurious "name": null noise in the hint input.
1244
- "results": [r.model_dump(exclude_none=True) for r in refs],
1245
- "limit": limit,
1246
- "offset": offset,
1247
- "filter": filter_dump,
1248
- "has_more_results": has_more_results,
1249
- }
1250
- raw_struct, raw_advisories = _hints_or_skip("find", hint_payload)
1251
- return FindOutput(
1252
- success=True,
1253
- results=refs,
1254
- limit=limit,
1255
- offset=offset,
1256
- has_more_results=has_more_results,
1257
- advisories=raw_advisories,
1258
- hints_structured=_to_structured_hints(raw_struct),
1259
- absence=diag,
1260
- )
1261
- except Exception as exc:
1262
- return FindOutput(success=False, message=str(exc), advisories=[], limit=None, offset=None)
1263
-
1264
-
1265
- _DESCRIBE_UCS_ID_MESSAGE = (
1266
- "UnresolvedCallSite ids (ucs:…) are not describable — use describe(caller_method_id) "
1267
- "for record.data.unresolved_call_sites, neighbors(..., include_unresolved=True), "
1268
- "or java-codebase-rag unresolved-calls list --method-id <caller_id>"
1269
- )
1270
-
1271
-
1272
- def describe_v2(
1273
- id: str | None = None,
1274
- fqn: str | None = None,
1275
- graph: LadybugGraph | None = None,
1276
- ) -> DescribeOutput:
1277
- try:
1278
- g = graph or LadybugGraph.get()
1279
- has_id = bool(id and str(id).strip())
1280
- has_fqn = bool(fqn and str(fqn).strip())
1281
- if not has_id and not has_fqn:
1282
- return DescribeOutput(success=False, message="id or fqn required")
1283
- if has_id and str(id).strip().startswith("ucs:"):
1284
- return DescribeOutput(success=False, message=_DESCRIBE_UCS_ID_MESSAGE)
1285
- hint_message: str | None = None
1286
- node_id: str
1287
- if has_id:
1288
- node_id = str(id).strip()
1289
- else:
1290
- fqn_val = str(fqn).strip()
1291
- rows = g._rows( # noqa: SLF001
1292
- "MATCH (s:Symbol) WHERE s.fqn = $fqn RETURN s.id AS id LIMIT 2",
1293
- {"fqn": fqn_val},
1294
- )
1295
- if not rows:
1296
- # FQN not found: run diagnosis with query=fqn
1297
- cfg = _get_absence_config()
1298
- vocab = get_vocabulary_index(g, cfg)
1299
- diag = diagnose(
1300
- tool="describe",
1301
- query=fqn_val,
1302
- filt=None,
1303
- filter_kind=None,
1304
- root_node=None,
1305
- scope={},
1306
- vocab=vocab,
1307
- graph=g,
1308
- cfg=cfg,
1309
- )
1310
- return DescribeOutput(
1311
- success=False,
1312
- message=f"No Symbol found for fqn='{fqn_val}'",
1313
- absence=diag,
1314
- )
1315
- node_id = str(rows[0]["id"] or "")
1316
- if len(rows) > 1:
1317
- hint_message = (
1318
- "multiple symbols share this FQN; use "
1319
- f"resolve(identifier={fqn_val!r}, hint_kind='symbol') to list candidates with reasons, "
1320
- "then describe(id=...) on the chosen node"
1321
- )
1322
- kind = _resolve_node_kind(g, node_id)
1323
- if kind == "unresolved_call_site":
1324
- return DescribeOutput(success=False, message=_DESCRIBE_UCS_ID_MESSAGE, advisories=[])
1325
- row = _load_node_record(g, node_id, kind)
1326
- if row is None:
1327
- # Node ID not found: run diagnosis with query=None (minimal refine)
1328
- cfg = _get_absence_config()
1329
- vocab = get_vocabulary_index(g, cfg)
1330
- diag = diagnose(
1331
- tool="describe",
1332
- query=None,
1333
- filt=None,
1334
- filter_kind=None,
1335
- root_node=None,
1336
- scope={},
1337
- vocab=vocab,
1338
- graph=g,
1339
- cfg=cfg,
1340
- )
1341
- return DescribeOutput(
1342
- success=False,
1343
- message=f"No node found for `{node_id}`",
1344
- advisories=[],
1345
- absence=diag,
1346
- )
1347
- ref = _node_ref_from_row(kind, row)
1348
- edge_summary = _edge_summary_for_node(g, node_id, kind=kind, row=row)
1349
- data = dict(row)
1350
- if kind == "symbol" and str(row.get("kind") or "") in _METHOD_SYMBOL_KINDS_FOR_OVERRIDE_ROLLUP:
1351
- inline, total = g.unresolved_sites_for_describe(node_id)
1352
- if total > 0:
1353
- data["unresolved_call_sites_total"] = total
1354
- data["unresolved_call_sites"] = [
1355
- {
1356
- "line": int(r.get("line") or 0),
1357
- "reason": str(r.get("reason") or ""),
1358
- "callee_simple": str(r.get("callee_simple") or ""),
1359
- "receiver_expr": str(r.get("receiver_expr") or ""),
1360
- }
1361
- for r in inline
1362
- ]
1363
- if total > len(inline):
1364
- data["unresolved_call_sites_footer"] = (
1365
- f"{total} unresolved call sites — see "
1366
- f"java-codebase-rag unresolved-calls list --method-id {node_id} for the full list"
1367
- )
1368
- record = NodeRecord(id=ref.id, kind=kind, fqn=ref.fqn, data=data, edge_summary=edge_summary)
1369
- raw_struct, raw_advisories = _hints_or_skip("describe", {"success": True, "record": record.model_dump()})
1370
- return DescribeOutput(
1371
- success=True,
1372
- record=record,
1373
- message=hint_message,
1374
- advisories=raw_advisories,
1375
- hints_structured=_to_structured_hints(raw_struct),
1376
- )
1377
- except ValueError as exc:
1378
- return DescribeOutput(success=False, message=str(exc), advisories=[])
1379
- except Exception as exc:
1380
- return DescribeOutput(success=False, message=str(exc), advisories=[])
1381
-
1382
-
1383
-
1384
-
1385
- # Per-edge-type attribute columns selected by the generic (flat-label) neighbors
1386
- # query (issue #356). RETURNing a fixed superset of columns regardless of which
1387
- # edge type matched is the typed-union RETURN anti-pattern: a stricter binder
1388
- # (e.g. Kùzu) errors when a RETURNed column does not exist on the matched type.
1389
- # Selecting columns per edge type keeps the query portable; _neighbor_edge_attrs
1390
- # still drops None/"" so each edge exposes only the attrs that exist for its type.
1391
- # Aligned with the REL TABLE schemas in build_ast_graph.py.
1392
- _FLAT_EDGE_ATTR_COLUMNS: dict[str, tuple[str, ...]] = {
1393
- "CALLS": ("confidence", "strategy", "source", "call_site_line", "call_site_byte", "arg_count", "resolved"),
1394
- "HTTP_CALLS": ("confidence", "strategy", "match"),
1395
- "ASYNC_CALLS": ("confidence", "strategy", "match"),
1396
- "EXPOSES": ("confidence", "strategy"),
1397
- "DECLARES_CLIENT": ("confidence", "strategy"),
1398
- "DECLARES_PRODUCER": ("confidence", "strategy"),
1399
- "INJECTS": ("mechanism", "annotation", "field_or_param", "resolved"),
1400
- "EXTENDS": ("resolved",),
1401
- "IMPLEMENTS": ("resolved",),
1402
- "DECLARES": (),
1403
- "OVERRIDES": (),
1404
- }
1405
-
1406
-
1407
- def _neighbor_edge_attrs(row: dict[str, Any]) -> dict[str, Any]:
1408
- attrs = {
1409
- k: v
1410
- for k, v in row.items()
1411
- if k not in {"other_id", "edge_type", "stored_edge_type"}
1412
- and v not in (None, "")
1413
- }
1414
- attrs.setdefault("row_kind", "resolved")
1415
- return attrs
1416
-
1417
-
1418
- def _unresolved_site_to_edge(origin_id: str, row: dict[str, Any]) -> Edge:
1419
- ucs_id = str(row.get("id") or "")
1420
- callee = str(row.get("callee_simple") or "")
1421
- line = int(row.get("call_site_line") or 0)
1422
- byte = int(row.get("call_site_byte") or 0)
1423
- return Edge(
1424
- origin_id=origin_id,
1425
- edge_type="CALLS",
1426
- direction="out",
1427
- other=NodeRef(id=ucs_id, kind="unresolved_call_site", fqn="", name=callee),
1428
- attrs={
1429
- "row_kind": "unresolved_call_site",
1430
- "unresolved_call_site_id": ucs_id,
1431
- "reason": str(row.get("reason") or ""),
1432
- "call_site_line": line,
1433
- "call_site_byte": byte,
1434
- "arg_count": int(row.get("arg_count") or 0),
1435
- "callee_simple": callee,
1436
- "receiver_expr": str(row.get("receiver_expr") or ""),
1437
- },
1438
- )
1439
-
1440
-
1441
- def _calls_transcript_sort_key(edge: Edge) -> tuple[int, int, int]:
1442
- attrs = edge.attrs or {}
1443
- line = int(attrs.get("call_site_line") or 0)
1444
- byte = int(attrs.get("call_site_byte") or 0)
1445
- kind_rank = 0 if str(attrs.get("row_kind") or "resolved") == "resolved" else 1
1446
- return (line, byte, kind_rank)
1447
-
1448
-
1449
- def _dedup_call_edges(edges: list[Edge]) -> list[Edge]:
1450
- """Collapse resolved CALLS rows sharing (origin_id, other.id); unresolved rows pass through."""
1451
- resolved: list[Edge] = []
1452
- unresolved: list[Edge] = []
1453
- for e in edges:
1454
- if str((e.attrs or {}).get("row_kind") or "resolved") == "unresolved_call_site":
1455
- unresolved.append(e)
1456
- else:
1457
- resolved.append(e)
1458
- groups: dict[tuple[str, str], list[Edge]] = {}
1459
- for e in resolved:
1460
- key = (e.origin_id, e.other.id)
1461
- groups.setdefault(key, []).append(e)
1462
- collapsed: list[Edge] = []
1463
- for group in groups.values():
1464
- ordered = sorted(group, key=_calls_transcript_sort_key)
1465
- canonical = ordered[0]
1466
- lines = sorted(
1467
- {int((x.attrs or {}).get("call_site_line") or 0) for x in group},
1468
- )
1469
- attrs = dict(canonical.attrs or {})
1470
- attrs["call_site_count"] = len(group)
1471
- attrs["call_site_lines"] = lines
1472
- collapsed.append(canonical.model_copy(update={"attrs": attrs}))
1473
- merged = collapsed + unresolved
1474
- merged.sort(key=_calls_transcript_sort_key)
1475
- return merged
1476
-
1477
-
1478
- def _edgefilter_pushdown_kwargs(ef: EdgeFilter | None) -> dict[str, Any]:
1479
- if ef is None:
1480
- return {}
1481
- return {
1482
- "min_confidence": ef.min_confidence,
1483
- "include_strategies": ef.include_strategies,
1484
- "exclude_strategies": ef.exclude_strategies,
1485
- "callee_declaring_role": ef.callee_declaring_role,
1486
- "callee_declaring_roles": ef.callee_declaring_roles,
1487
- "exclude_callee_declaring_roles": ef.exclude_callee_declaring_roles,
1488
- }
1489
-
1490
-
1491
- def _rows_to_call_edges(
1492
- g: Any,
1493
- *,
1494
- origin_id: str,
1495
- direction: Literal["in", "out"],
1496
- rows: list[dict[str, Any]],
1497
- nf: NodeFilter | None,
1498
- ) -> list[Edge]:
1499
- edges: list[Edge] = []
1500
- for row in rows:
1501
- other_id = str(row.get("other_id") or "")
1502
- other_kind = _resolve_node_kind(g, other_id)
1503
- other_rec = _load_node_record(g, other_id, other_kind)
1504
- if other_rec is None:
1505
- continue
1506
- if nf and (err := _nodefilter_applicability_error(other_kind, nf)):
1507
- _log_fail_loud("applicability")
1508
- raise ValueError(err)
1509
- if not _node_matches_filter(other_kind, other_rec, nf):
1510
- continue
1511
- edges.append(
1512
- Edge(
1513
- origin_id=origin_id,
1514
- edge_type=str(row.get("edge_type") or "CALLS"),
1515
- direction=direction,
1516
- other=_node_ref_from_row(other_kind, other_rec),
1517
- attrs=_neighbor_edge_attrs(row),
1518
- )
1519
- )
1520
- return edges
1521
-
1522
-
1523
- def _neighbors_calls_for_origin(
1524
- g: Any,
1525
- origin_id: str,
1526
- *,
1527
- direction: Literal["in", "out"],
1528
- nf: NodeFilter | None,
1529
- ef: EdgeFilter | None,
1530
- offset: int,
1531
- limit: int | None,
1532
- include_unresolved: bool = False,
1533
- dedup_calls: bool = False,
1534
- ) -> list[Edge]:
1535
- pushdown = _edgefilter_pushdown_kwargs(ef)
1536
- needs_full_stream = (
1537
- nf is not None
1538
- or dedup_calls
1539
- or include_unresolved
1540
- or limit is None
1541
- )
1542
- sql_pagination = not needs_full_stream and limit is not None
1543
- if sql_pagination:
1544
- rows = g.neighbor_calls_for_symbol(
1545
- origin_id,
1546
- direction=direction,
1547
- offset=offset,
1548
- limit=limit,
1549
- sql_pagination=True,
1550
- **pushdown,
1551
- )
1552
- return _rows_to_call_edges(g, origin_id=origin_id, direction=direction, rows=rows, nf=nf)
1553
- rows = g.neighbor_calls_for_symbol(
1554
- origin_id,
1555
- direction=direction,
1556
- offset=0,
1557
- limit=None,
1558
- sql_pagination=False,
1559
- **pushdown,
1560
- )
1561
- edges = _rows_to_call_edges(g, origin_id=origin_id, direction=direction, rows=rows, nf=nf)
1562
- if include_unresolved and direction == "out":
1563
- ucs_rows = g.unresolved_sites_for_caller(origin_id, direction=direction)
1564
- edges.extend(_unresolved_site_to_edge(origin_id, r) for r in ucs_rows)
1565
- edges.sort(key=_calls_transcript_sort_key)
1566
- if dedup_calls:
1567
- edges = _dedup_call_edges(edges)
1568
- if limit is None:
1569
- return edges
1570
- return edges[offset : offset + limit]
1571
-
1572
-
1573
- def _composed_axis_origin_error(
1574
- *,
1575
- symbol_kind: str,
1576
- modifiers: list[str] | None,
1577
- declares_composed: list[str],
1578
- override_composed: list[str],
1579
- ) -> str | None:
1580
- """Fail-fast origin gate for composed DECLARES.* vs OVERRIDDEN_BY.* families."""
1581
- if declares_composed and symbol_kind not in _TYPE_SYMBOL_KINDS_FOR_EDGE_ROLLUP:
1582
- return f"Composed edge types ({declares_composed[0]}) require a type Symbol origin"
1583
- if override_composed:
1584
- key = override_composed[0]
1585
- mods = modifiers or []
1586
- if symbol_kind == "constructor":
1587
- return (
1588
- f"Composed edge types ({key}) require a non-static method Symbol origin "
1589
- "(constructors are not supported)"
1590
- )
1591
- if symbol_kind not in _METHOD_SYMBOL_KINDS_FOR_OVERRIDE_ROLLUP:
1592
- return f"Composed edge types ({key}) require a method Symbol origin"
1593
- if "static" in mods:
1594
- return (
1595
- f"Composed edge types ({key}) require a non-static method Symbol origin "
1596
- "(static methods are not supported)"
1597
- )
1598
- return None
1599
-
1600
-
1601
- @validate_call(config={"arbitrary_types_allowed": True})
1602
- def neighbors_v2(
1603
- ids: str | list[str],
1604
- # Required fields are intentional: direct Python calls and MCP-bound calls
1605
- # share the same validation contract through @validate_call.
1606
- direction: Literal["in", "out"] = Field(...),
1607
- edge_types: list[NeighborEdgeType] = Field(...),
1608
- limit: int = 25,
1609
- offset: int = 0,
1610
- filter: NodeFilter | dict[str, Any] | str | None = None,
1611
- edge_filter: EdgeFilter | dict[str, Any] | str | None = None,
1612
- include_unresolved: bool = False,
1613
- dedup_calls: bool = False,
1614
- graph: Any | None = None,
1615
- ) -> NeighborsOutput:
1616
- try:
1617
- validated_types = _NEIGHBOR_EDGE_TYPES_ADAPTER.validate_python(edge_types)
1618
- requested_edge_types = list(dict.fromkeys(validated_types))
1619
- flat_labels = [et for et in requested_edge_types if et not in _COMPOSED_EDGE_TYPES]
1620
- composed_keys = [et for et in requested_edge_types if et in _COMPOSED_EDGE_TYPES]
1621
- declares_composed = [k for k in composed_keys if k in _MEMBER_COMPOSED_EDGE_TYPES]
1622
- override_composed = [k for k in composed_keys if k in _OVERRIDE_COMPOSED_EDGE_TYPES]
1623
- ordered_composed = declares_composed + override_composed
1624
- g = graph or LadybugGraph.get()
1625
- try:
1626
- raw_filter = _coerce_filter(filter)
1627
- nf = (
1628
- NodeFilter.model_validate(raw_filter)
1629
- if raw_filter is not None and not isinstance(raw_filter, NodeFilter)
1630
- else raw_filter
1631
- )
1632
- except ValidationError as exc:
1633
- _log_fail_loud("unknown_key")
1634
- return NeighborsOutput(
1635
- success=False,
1636
- message=_filter_validation_error_message(exc),
1637
- advisories=[],
1638
- requested_edge_types=[],
1639
- )
1640
- try:
1641
- raw_edge_filter = _coerce_edge_filter(edge_filter)
1642
- ef = (
1643
- EdgeFilter.model_validate(raw_edge_filter)
1644
- if raw_edge_filter is not None and not isinstance(raw_edge_filter, EdgeFilter)
1645
- else raw_edge_filter
1646
- )
1647
- except ValidationError as exc:
1648
- _log_fail_loud("edge_filter")
1649
- return NeighborsOutput(
1650
- success=False,
1651
- message=_filter_validation_error_message(exc),
1652
- advisories=[],
1653
- requested_edge_types=[],
1654
- )
1655
- except ValueError as exc:
1656
- _log_fail_loud("edge_filter")
1657
- return NeighborsOutput(success=False, message=str(exc), requested_edge_types=[])
1658
- if include_unresolved and ef is not None:
1659
- return NeighborsOutput(
1660
- success=False,
1661
- message=(
1662
- "include_unresolved=True is incompatible with edge_filter; "
1663
- "UnresolvedCallSite rows have no edge attributes to filter on"
1664
- ),
1665
- requested_edge_types=requested_edge_types,
1666
- )
1667
- if include_unresolved and requested_edge_types != ["CALLS"]:
1668
- return NeighborsOutput(
1669
- success=False,
1670
- message="include_unresolved requires edge_types=['CALLS']",
1671
- requested_edge_types=requested_edge_types,
1672
- )
1673
- if include_unresolved and direction != "out":
1674
- return NeighborsOutput(
1675
- success=False,
1676
- message='include_unresolved requires direction="out"',
1677
- requested_edge_types=requested_edge_types,
1678
- )
1679
- if ef and (err := _edgefilter_applicability_error(requested_edge_types, ef)):
1680
- _log_fail_loud("edge_filter")
1681
- return NeighborsOutput(
1682
- success=False,
1683
- message=err,
1684
- requested_edge_types=requested_edge_types,
1685
- )
1686
- if composed_keys and direction != "out":
1687
- return NeighborsOutput(
1688
- success=False,
1689
- message='Composed edge types require direction="out"',
1690
- requested_edge_types=requested_edge_types,
1691
- )
1692
- use_calls_path = flat_labels == ["CALLS"] and not composed_keys
1693
- origins = [ids] if isinstance(ids, str) else list(ids)
1694
- results: list[Edge] = []
1695
- unfiltered_calls_count: int | None = None
1696
- unresolved_count: int | None = None
1697
- calls_row_count: int | None = None
1698
- if use_calls_path and len(origins) == 1 and direction == "out":
1699
- unresolved_count = g.count_unresolved_for_caller(origins[0])
1700
- calls_row_count = g.count_calls_for_symbol(origins[0], direction=direction)
1701
- for origin_id in origins:
1702
- origin_kind = _resolve_node_kind(g, origin_id)
1703
- if ordered_composed:
1704
- if origin_kind != "symbol":
1705
- first_key = ordered_composed[0]
1706
- axis_msg = (
1707
- f"Composed edge types ({first_key}) require a method Symbol origin"
1708
- if first_key in _OVERRIDE_COMPOSED_EDGE_TYPES
1709
- else f"Composed edge types ({first_key}) require a type Symbol origin"
1710
- )
1711
- return NeighborsOutput(
1712
- success=False,
1713
- message=axis_msg,
1714
- requested_edge_types=requested_edge_types,
1715
- )
1716
- origin_row = _load_node_record(g, origin_id, "symbol")
1717
- sym_kind = str((origin_row or {}).get("kind") or "")
1718
- mods_raw = (origin_row or {}).get("modifiers")
1719
- mods = mods_raw if isinstance(mods_raw, list) else None
1720
- if err := _composed_axis_origin_error(
1721
- symbol_kind=sym_kind,
1722
- modifiers=mods,
1723
- declares_composed=declares_composed,
1724
- override_composed=override_composed,
1725
- ):
1726
- return NeighborsOutput(
1727
- success=False,
1728
- message=err,
1729
- requested_edge_types=requested_edge_types,
1730
- )
1731
- if use_calls_path:
1732
- paginate_in_sql = (
1733
- len(origins) == 1
1734
- and nf is None
1735
- and not include_unresolved
1736
- and not dedup_calls
1737
- )
1738
- try:
1739
- origin_edges = _neighbors_calls_for_origin(
1740
- g,
1741
- origin_id,
1742
- direction=direction,
1743
- nf=nf,
1744
- ef=ef,
1745
- offset=offset if paginate_in_sql else 0,
1746
- limit=limit if paginate_in_sql else None,
1747
- include_unresolved=include_unresolved,
1748
- dedup_calls=dedup_calls,
1749
- )
1750
- except ValueError as exc:
1751
- return NeighborsOutput(
1752
- success=False,
1753
- message=str(exc),
1754
- requested_edge_types=requested_edge_types,
1755
- )
1756
- if (
1757
- ef is not None
1758
- and ef.callee_declaring_role in _ROLE_FILTER_OTHER_FALLBACK_VALUES
1759
- and not origin_edges
1760
- and unfiltered_calls_count is None
1761
- ):
1762
- unfiltered_calls_count = g.count_calls_for_symbol(origin_id, direction=direction)
1763
- results.extend(origin_edges)
1764
- continue
1765
- if flat_labels:
1766
- # Select attribute columns per edge type (issue #356). A single
1767
- # multi-label query RETURNing a fixed column superset references
1768
- # columns that don't exist on every matched type — the typed-union
1769
- # RETURN anti-pattern, which errors on stricter binders (e.g. Kùzu).
1770
- # Run one single-label query per type, RETURNing only that type's
1771
- # columns, and merge the rows. `label(e) = $label` scalar equality
1772
- # (not `label(e) IN [...]`) per the CLAUDE.md Cypher note.
1773
- rows: list[dict[str, Any]] = []
1774
- match_clause = "MATCH (a)-[e]->(b)" if direction == "out" else "MATCH (a)<-[e]-(b)"
1775
- for label in flat_labels:
1776
- cols = _FLAT_EDGE_ATTR_COLUMNS.get(label, ())
1777
- select = "b.id AS other_id, label(e) AS edge_type"
1778
- if cols:
1779
- select += ", " + ", ".join(f"e.{c} AS {c}" for c in cols)
1780
- rows.extend(
1781
- g._rows( # noqa: SLF001
1782
- f"{match_clause} WHERE a.id = $id AND label(e) = $label RETURN {select}",
1783
- {"id": origin_id, "label": label},
1784
- )
1785
- )
1786
- for row in rows:
1787
- other_id = str(row.get("other_id") or "")
1788
- other_kind = _resolve_node_kind(g, other_id)
1789
- other_rec = _load_node_record(g, other_id, other_kind)
1790
- if other_rec is None:
1791
- continue
1792
- if nf and (err := _nodefilter_applicability_error(other_kind, nf)):
1793
- _log_fail_loud("applicability")
1794
- return NeighborsOutput(
1795
- success=False, message=err, requested_edge_types=[]
1796
- )
1797
- if not _node_matches_filter(other_kind, other_rec, nf):
1798
- continue
1799
- results.append(
1800
- Edge(
1801
- origin_id=origin_id,
1802
- edge_type=str(row.get("edge_type") or ""),
1803
- direction=direction,
1804
- other=_node_ref_from_row(other_kind, other_rec),
1805
- attrs=_neighbor_edge_attrs(row),
1806
- )
1807
- )
1808
- for composed_key in ordered_composed:
1809
- if composed_key in _MEMBER_COMPOSED_EDGE_TYPES:
1810
- traversal_rows = g.member_edge_traversal_for(origin_id, composed_key)
1811
- else:
1812
- traversal_rows = g.override_axis_traversal_for(origin_id, composed_key)
1813
- for row in traversal_rows:
1814
- other_id = str(row.get("other_id") or "")
1815
- other_kind = _resolve_node_kind(g, other_id)
1816
- other_rec = _load_node_record(g, other_id, other_kind)
1817
- if other_rec is None:
1818
- continue
1819
- if nf and (err := _nodefilter_applicability_error(other_kind, nf)):
1820
- _log_fail_loud("applicability")
1821
- return NeighborsOutput(
1822
- success=False, message=err, requested_edge_types=[]
1823
- )
1824
- if not _node_matches_filter(other_kind, other_rec, nf):
1825
- continue
1826
- if composed_key == "OVERRIDDEN_BY":
1827
- edge_attrs: dict[str, Any] = {}
1828
- else:
1829
- edge_attrs = _neighbor_edge_attrs(row)
1830
- results.append(
1831
- Edge(
1832
- origin_id=origin_id,
1833
- edge_type=composed_key,
1834
- direction="out",
1835
- other=_node_ref_from_row(other_kind, other_rec),
1836
- attrs=edge_attrs,
1837
- )
1838
- )
1839
- if use_calls_path and len(origins) > 1:
1840
- sliced = results[offset : offset + limit]
1841
- neighbors_has_more = len(results) > offset + limit
1842
- elif use_calls_path:
1843
- # Single-origin CALLS path. When paginate_in_sql is True the SQL did
1844
- # the OFFSET/LIMIT and the row/unfiltered counts carry the has-more
1845
- # signal, so this field stays None (unknown). When paginate_in_sql is
1846
- # False (a node_filter is set, include_unresolved, or dedup_calls) we
1847
- # loaded the FULL matching set with no pushdown, so the client already
1848
- # has every edge -> False (not None), so a paging client need not probe.
1849
- sliced = results
1850
- neighbors_has_more = None if paginate_in_sql else False
1851
- else:
1852
- sliced = results[offset : offset + limit]
1853
- neighbors_has_more = len(results) > offset + limit
1854
- first_origin = origins[0]
1855
- origin_kind = _resolve_node_kind(g, first_origin)
1856
- subject_record = _load_node_record(g, first_origin, origin_kind)
1857
-
1858
- # Absence diagnosis for empty results
1859
- cfg = _get_absence_config()
1860
- diag: AbsenceDiagnosis | None = None
1861
- if not sliced:
1862
- # Build root_node from first_origin + subject_record
1863
- root_node = _node_ref_from_row(origin_kind, subject_record) if subject_record else None
1864
- vocab = get_vocabulary_index(g, cfg)
1865
- diag = diagnose(
1866
- tool="neighbors",
1867
- query=None,
1868
- filt=None,
1869
- filter_kind=None,
1870
- root_node=root_node,
1871
- scope={},
1872
- vocab=vocab,
1873
- graph=g,
1874
- cfg=cfg,
1875
- )
1876
-
1877
- neigh_payload = {
1878
- "success": True,
1879
- "results": [e.model_dump(exclude_none=True) for e in sliced],
1880
- "requested_edge_types": requested_edge_types,
1881
- "requested_direction": direction,
1882
- "offset": offset,
1883
- "origin_id": first_origin,
1884
- "subject_record": subject_record,
1885
- "node_filter": nf.model_dump(exclude_none=True) if nf else None,
1886
- "edge_filter": ef.model_dump(exclude_none=True) if ef else None,
1887
- "edge_filter_provided": ef is not None,
1888
- "include_unresolved": include_unresolved,
1889
- "dedup_calls": dedup_calls,
1890
- "unfiltered_calls_count": unfiltered_calls_count,
1891
- "unresolved_count": unresolved_count,
1892
- "calls_row_count": calls_row_count,
1893
- }
1894
- raw_struct, raw_advisories = _hints_or_skip("neighbors", neigh_payload)
1895
- return NeighborsOutput(
1896
- success=True,
1897
- results=sliced,
1898
- requested_edge_types=requested_edge_types,
1899
- has_more_results=neighbors_has_more,
1900
- advisories=raw_advisories,
1901
- hints_structured=_to_structured_hints(raw_struct),
1902
- absence=diag,
1903
- )
1904
- except ValidationError:
1905
- raise
1906
- except Exception as exc:
1907
- return NeighborsOutput(success=False, message=str(exc), advisories=[], requested_edge_types=[])
1908
-
1909
-
1910
- # Re-export resolve symbols from resolve_service.py (imported here to avoid circular import)
1911
- from java_codebase_rag.analysis.resolve_service import ( # noqa: E402
1912
- ResolveCandidate,
1913
- ResolveOutput,
1914
- ResolveStatus,
1915
- resolve_v2,
1916
- )