java-codebase-rag 0.11.2__py3-none-any.whl → 0.12.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. java_codebase_rag-0.12.1.dist-info/METADATA +35 -0
  2. java_codebase_rag-0.12.1.dist-info/RECORD +4 -0
  3. {java_codebase_rag-0.11.2.dist-info → java_codebase_rag-0.12.1.dist-info}/WHEEL +1 -1
  4. java_codebase_rag/_fdlimit.py +0 -56
  5. java_codebase_rag/_stdio.py +0 -32
  6. java_codebase_rag/_version.py +0 -35
  7. java_codebase_rag/absence/__init__.py +0 -0
  8. java_codebase_rag/absence/absence_diagnosis.py +0 -700
  9. java_codebase_rag/absence/absence_types.py +0 -124
  10. java_codebase_rag/absence/absence_vocab.py +0 -460
  11. java_codebase_rag/analysis/__init__.py +0 -0
  12. java_codebase_rag/analysis/pr_analysis.py +0 -563
  13. java_codebase_rag/analysis/resolve_service.py +0 -740
  14. java_codebase_rag/ast/__init__.py +0 -0
  15. java_codebase_rag/ast/ast_java.py +0 -2825
  16. java_codebase_rag/ast/brownfield_events.py +0 -58
  17. java_codebase_rag/ast/chunk_heuristics.py +0 -62
  18. java_codebase_rag/cli.py +0 -1215
  19. java_codebase_rag/cli_format.py +0 -85
  20. java_codebase_rag/cli_progress.py +0 -94
  21. java_codebase_rag/config.py +0 -833
  22. java_codebase_rag/eval/__init__.py +0 -1
  23. java_codebase_rag/eval/ground_truth.py +0 -100
  24. java_codebase_rag/eval/metrics.py +0 -107
  25. java_codebase_rag/eval/runner.py +0 -556
  26. java_codebase_rag/graph/__init__.py +0 -0
  27. java_codebase_rag/graph/build_ast_graph.py +0 -4471
  28. java_codebase_rag/graph/graph_enrich.py +0 -1937
  29. java_codebase_rag/graph/graph_types.py +0 -224
  30. java_codebase_rag/graph/java_ontology.py +0 -465
  31. java_codebase_rag/graph/ladybug_queries.py +0 -2213
  32. java_codebase_rag/graph/path_filtering.py +0 -477
  33. java_codebase_rag/index/__init__.py +0 -0
  34. java_codebase_rag/index/java_index_flow_lancedb.py +0 -734
  35. java_codebase_rag/index/java_index_v1_common.py +0 -33
  36. java_codebase_rag/install_data/__init__.py +0 -0
  37. java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -108
  38. java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
  39. java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
  40. java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
  41. java_codebase_rag/installer.py +0 -2188
  42. java_codebase_rag/jrag.py +0 -4531
  43. java_codebase_rag/jrag_envelope.py +0 -1107
  44. java_codebase_rag/jrag_hints.py +0 -204
  45. java_codebase_rag/jrag_render.py +0 -926
  46. java_codebase_rag/lance_optimize.py +0 -264
  47. java_codebase_rag/mcp/__init__.py +0 -0
  48. java_codebase_rag/mcp/mcp_hints.py +0 -932
  49. java_codebase_rag/mcp/mcp_v2.py +0 -1916
  50. java_codebase_rag/mcp/server.py +0 -884
  51. java_codebase_rag/pipeline.py +0 -531
  52. java_codebase_rag/progress.py +0 -570
  53. java_codebase_rag/read_payloads.py +0 -781
  54. java_codebase_rag/search/__init__.py +0 -0
  55. java_codebase_rag/search/index_common.py +0 -10
  56. java_codebase_rag/search/search_lancedb.py +0 -1296
  57. java_codebase_rag/search/search_lexical.py +0 -449
  58. java_codebase_rag/search/search_scoring.py +0 -523
  59. java_codebase_rag/watch/__init__.py +0 -0
  60. java_codebase_rag/watch/client.py +0 -230
  61. java_codebase_rag/watch/daemon.py +0 -396
  62. java_codebase_rag/watch/lock.py +0 -201
  63. java_codebase_rag/watch/paths.py +0 -76
  64. java_codebase_rag/watch/protocol.py +0 -122
  65. java_codebase_rag/watch/server.py +0 -273
  66. java_codebase_rag/watch/warm.py +0 -105
  67. java_codebase_rag/watch/watcher.py +0 -370
  68. java_codebase_rag-0.11.2.dist-info/METADATA +0 -331
  69. java_codebase_rag-0.11.2.dist-info/RECORD +0 -71
  70. java_codebase_rag-0.11.2.dist-info/entry_points.txt +0 -4
  71. java_codebase_rag-0.11.2.dist-info/licenses/LICENSE +0 -21
  72. java_codebase_rag-0.11.2.dist-info/top_level.txt +0 -1
  73. /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.1.dist-info/top_level.txt +0 -0
@@ -1,884 +0,0 @@
1
- #!/usr/bin/env python3
2
- """LanceDB code-search MCP (stdio)."""
3
- from __future__ import annotations
4
-
5
- import asyncio
6
- import os
7
- import sys
8
- import time
9
- from pathlib import Path
10
- from typing import Literal
11
-
12
- from java_codebase_rag.mcp import mcp_v2
13
- from java_codebase_rag.analysis import resolve_service
14
- from java_codebase_rag.search.index_common import SBERT_MODEL
15
- from java_codebase_rag.cli_progress import (
16
- accumulate_and_relay_subprocess_streams,
17
- )
18
- from java_codebase_rag.pipeline import (
19
- VECTORS_SKIPPED_GRAPH_ONLY,
20
- cocoindex_bin as resolve_cocoindex_bin,
21
- vector_stack_installed,
22
- )
23
- from java_codebase_rag.progress import ProgressEvent
24
- from java_codebase_rag._fdlimit import raise_fd_limit
25
- from java_codebase_rag.config import (
26
- cocoindex_subprocess_env_defaults,
27
- discover_project_root,
28
- emit_legacy_env_hints_if_present,
29
- resolved_sbert_model_for_process_env,
30
- resolve_operator_config,
31
- )
32
- from java_codebase_rag.graph.ladybug_queries import LadybugGraph, resolve_ladybug_path
33
- from mcp.server.fastmcp import FastMCP
34
- from pydantic import BaseModel, Field
35
- # NOTE: search_lancedb.TABLES is imported lazily in list_code_index_tables_payload() — it
36
- # pulls lancedb/torch and is unavailable on graph-only installs (macOS Intel).
37
-
38
- _COCOINDEX_TARGET = "java_index_flow_lancedb.py:JavaCodeIndexLance"
39
-
40
- # Package-internal locations of the cocoindex flow and the graph builder, both
41
- # executed by file path (see java_codebase_rag.pipeline). Derived from this
42
- # file's location so they resolve under editable and wheel installs alike.
43
- _PKG_DIR = Path(__file__).resolve().parent.parent
44
- _FLOW_FILE = _PKG_DIR / "index" / "java_index_flow_lancedb.py"
45
- _BUILDER_FILE = _PKG_DIR / "graph" / "build_ast_graph.py"
46
- _INSTRUCTIONS = (
47
- "Java codebase graph navigator over an indexed Java codebase. "
48
- "Tools: search (NL/code locate), find (structured NodeFilter), describe (one node + edge_summary: stored edge-label counts and optional composed keys for type Symbols and override-axis virtual keys for method Symbols), "
49
- "neighbors (one hop; you MUST pass direction in|out AND edge_types list — no defaults), "
50
- "resolve (identifier-shaped lookup for symbol/route/client/producer — three statuses: one | many | none). "
51
- "Unknown filter keys and populated fields not applicable to the effective node kind fail with success=false and message. "
52
- "Successful responses from any tool may include `hints_structured` (tool call suggestions with a `reason` field) and `advisories` (pure informational text) when hints are enabled. "
53
- "Edge labels: EXTENDS, IMPLEMENTS, INJECTS, OVERRIDES, DECLARES, DECLARES_CLIENT, DECLARES_PRODUCER, CALLS, EXPOSES, HTTP_CALLS, ASYNC_CALLS; "
54
- "type Symbols may also use composed neighbors edge_types DECLARES.DECLARES_CLIENT, DECLARES.DECLARES_PRODUCER, DECLARES.EXPOSES (out only, type Symbol origin). "
55
- "Reprocess/init, meta, tables, diagnose-ignore, analyze-pr: use java-codebase-rag CLI — not MCP."
56
- )
57
-
58
-
59
- class GraphMetaOutput(BaseModel):
60
- success: bool
61
- enabled: bool
62
- db_path: str
63
- ontology_version: int = 0
64
- built_at: int = 0
65
- source_root: str = ""
66
- parse_errors: int = 0
67
- counts: dict[str, int] = Field(default_factory=dict)
68
- module_counts: dict[str, int] = Field(default_factory=dict)
69
- microservice_counts: dict[str, int] = Field(default_factory=dict)
70
- routes_total: int = 0
71
- exposes_total: int = 0
72
- routes_by_framework: dict[str, int] = Field(default_factory=dict)
73
- routes_resolved_pct: float = 0.0
74
- routes_from_brownfield_pct: float = 0.0
75
- routes_by_layer: dict[str, int] = Field(default_factory=dict)
76
- edge_counts: dict[str, int] = Field(default_factory=dict)
77
- http_calls_match_breakdown: dict[str, int] = Field(default_factory=dict)
78
- async_calls_match_breakdown: dict[str, int] = Field(default_factory=dict)
79
- cross_service_calls_total: int = 0
80
- cross_service_resolution: str | None = None
81
- message: str | None = None
82
-
83
-
84
- class RefreshIndexOutput(BaseModel):
85
- """Structured result for ``run_refresh_pipeline`` / CLI ``reprocess`` JSON.
86
-
87
- ``phases_run`` records which phase subprocesses actually started; the CLI maps
88
- failures to exit **2** when it is empty (setup / nothing spawned) and exit **1**
89
- when it is non-empty (build failure). Callers constructing this model manually
90
- must set ``phases_run`` accordingly — omitting it leaves the default ``[]``,
91
- which the CLI treats like a preflight failure.
92
- """
93
-
94
- success: bool
95
- exit_code: int | None = None
96
- stdout: str = ""
97
- stderr: str = ""
98
- message: str | None = None
99
- graph_exit_code: int | None = None
100
- graph_stdout: str = ""
101
- graph_stderr: str = ""
102
- phases_run: list[Literal["vectors", "graph"]] = Field(default_factory=list)
103
- optimize_error: str | None = None
104
-
105
-
106
- class IndexInfoOutput(BaseModel):
107
- lancedb_uri: str
108
- embedding_model: str
109
- project_root: str
110
- cocoindex_target: str
111
- tables: dict[str, str]
112
- graph: GraphMetaOutput
113
-
114
-
115
- # Module-level scope manager, initialized in main()
116
- _scope_manager: ScopeManager | None = None
117
-
118
-
119
- class ScopeManager:
120
- """Manages automatic microservice scope detection and injection."""
121
-
122
- def __init__(self, source_root: Path):
123
- self.source_root = source_root
124
- self.default_scope: str | None = self._detect_scope()
125
- self._log_detection()
126
-
127
- def _detect_scope(self) -> str | None:
128
- from java_codebase_rag.graph.graph_enrich import detect_microservice_from_path
129
-
130
- candidate = detect_microservice_from_path(Path.cwd(), self.source_root)
131
- if candidate is None:
132
- return None
133
- # Only auto-scope to a microservice that actually has indexed code.
134
- # detect_microservice_from_path can mislabel a non-microservice
135
- # top-level child of source_root — most importantly the config/context
136
- # directory the MCP server is launched from (no build marker, no
137
- # source) — via its "first path segment under root" fallback. Scoping
138
- # every query to such a name yields zero matches, so all tools return
139
- # empty. A real microservice the operator is working in is, by
140
- # definition, present in the index, so validating against the indexed
141
- # set cannot suppress a legitimate scope. When the index is unreadable
142
- # (empty known set) we keep the detected candidate rather than silently
143
- # disabling auto-scope on a transient graph error.
144
- known = self._indexed_microservices()
145
- if known and candidate not in known:
146
- return None
147
- return candidate
148
-
149
- def _indexed_microservices(self) -> set[str]:
150
- """Microservice names that have indexed type symbols.
151
-
152
- Graph-only source of truth: the graph is always built alongside Lance,
153
- and a Lance-only index (no graph) is not a supported state. Any failure
154
- (graph missing, open error, empty index) returns an empty set, which
155
- ``_detect_scope`` treats as "cannot validate — keep detection".
156
- """
157
- try:
158
- if not LadybugGraph.exists():
159
- return set()
160
- # LadybugGraph.get() opens the DB and runs meta(); it can raise
161
- # (e.g. RuntimeError on ontology-version mismatch). Caught here ->
162
- # empty set -> _detect_scope keeps the detected scope.
163
- counts = LadybugGraph.get().microservice_counts()
164
- return {name for name in counts if name}
165
- except Exception:
166
- return set()
167
-
168
- def _log_detection(self) -> None:
169
- if self.default_scope:
170
- print(f"[scope] Detected microservice: {self.default_scope}", file=sys.stderr)
171
- print(f"[scope] Queries scoped to {self.default_scope}", file=sys.stderr)
172
- else:
173
- print("[scope] No microservice detected (at project root)", file=sys.stderr)
174
- print("[scope] Queries will span all microservices", file=sys.stderr)
175
-
176
- def apply_auto_scope(self, node_filter: mcp_v2.NodeFilter | None) -> mcp_v2.NodeFilter | None:
177
- """Apply auto-detected scope to filter if no explicit microservice is set."""
178
- if self.default_scope is None:
179
- return node_filter
180
- if node_filter is None:
181
- return mcp_v2.NodeFilter(microservice=self.default_scope)
182
- if node_filter.microservice is None:
183
- return node_filter.model_copy(update={"microservice": self.default_scope})
184
- return node_filter
185
-
186
-
187
- def _resolve_lancedb_uri() -> str:
188
- raw = os.environ.get("JAVA_CODEBASE_RAG_INDEX_DIR", "").strip()
189
- if not raw:
190
- raw = str((_project_root() / ".java-codebase-rag").resolve())
191
- p = Path(raw).expanduser()
192
- if not str(raw).startswith(("s3://", "gs://", "az://")):
193
- try:
194
- return str(p.resolve())
195
- except OSError:
196
- return str(p)
197
- return raw
198
-
199
-
200
- def _project_root() -> Path:
201
- env = os.environ.get("JAVA_CODEBASE_RAG_SOURCE_ROOT", "").strip()
202
- if env:
203
- return Path(env).expanduser().resolve()
204
- discovered = discover_project_root(Path.cwd())
205
- return discovered if discovered is not None else Path.cwd().resolve()
206
-
207
-
208
- def _source_root_for_operator_config() -> Path | None:
209
- """``source_root`` arg to hand ``resolve_operator_config`` from the MCP server.
210
-
211
- Returns ``JAVA_CODEBASE_RAG_SOURCE_ROOT`` when set (an explicit operator
212
- override that wins and suppresses the YAML ``source_root`` field, exactly
213
- like CLI ``--source-root``), otherwise ``None`` — so
214
- ``resolve_operator_config`` runs its OWN walk-up discovery and HONORS the
215
- YAML ``source_root`` field, matching the CLI (``init`` / ``increment`` /
216
- ``reprocess``) path.
217
-
218
- Do NOT pass ``_project_root()`` (the walk-up-discovered dir) here: a
219
- non-``None`` value routes into the "explicit source root" branch that
220
- skips the YAML ``source_root`` field, which made the MCP server and the
221
- CLI resolve different ``source_root`` / ``index_dir`` from the same config
222
- file (the init-vs-MCP index_dir divergence). ``_project_root()`` is kept
223
- only for the ``_resolve_lancedb_uri()`` fallback below.
224
- """
225
- env = os.environ.get("JAVA_CODEBASE_RAG_SOURCE_ROOT", "").strip()
226
- return Path(env).expanduser().resolve() if env else None
227
-
228
-
229
- def _cocoindex_subprocess_env(project_root: Path) -> dict[str, str]:
230
- sub_env = os.environ.copy()
231
- sub_env["JAVA_CODEBASE_RAG_SOURCE_ROOT"] = str(project_root)
232
- idx = os.environ.get("JAVA_CODEBASE_RAG_INDEX_DIR", "").strip()
233
- if idx:
234
- sub_env["JAVA_CODEBASE_RAG_INDEX_DIR"] = str(Path(idx).expanduser().resolve())
235
- # Cap CocoIndex concurrency to avoid EMFILE ("too many open files") under
236
- # default OS fd limits. See: https://github.com/HumanBean17/java-codebase-rag/issues/306
237
- for _k, _v in cocoindex_subprocess_env_defaults().items():
238
- sub_env.setdefault(_k, _v)
239
- return sub_env
240
-
241
-
242
- def _graph_enabled() -> bool:
243
- return LadybugGraph.exists()
244
-
245
-
246
- def _graph_meta_output() -> GraphMetaOutput:
247
- if not LadybugGraph.exists():
248
- return GraphMetaOutput(
249
- success=True,
250
- enabled=False,
251
- db_path=resolve_ladybug_path(),
252
- message="Ladybug graph not present; run java-codebase-rag reprocess or build_ast_graph.py",
253
- )
254
- try:
255
- graph = LadybugGraph.get()
256
- meta = graph.meta()
257
- except Exception as e:
258
- return GraphMetaOutput(
259
- success=False,
260
- enabled=_graph_enabled(),
261
- db_path=resolve_ladybug_path(),
262
- message=f"Ladybug open failed: {e}",
263
- )
264
- if "error" in meta:
265
- return GraphMetaOutput(
266
- success=False,
267
- enabled=_graph_enabled(),
268
- db_path=meta.get("db_path", resolve_ladybug_path()),
269
- message=str(meta["error"]),
270
- )
271
- try:
272
- mod_counts = graph.module_counts()
273
- except Exception:
274
- mod_counts = {}
275
- try:
276
- ms_counts = graph.microservice_counts()
277
- except Exception:
278
- ms_counts = {}
279
- rfw = meta.get("routes_by_framework") or {}
280
- routes_by_framework = {str(k): int(v) for k, v in rfw.items()} if isinstance(rfw, dict) else {}
281
- rbl = meta.get("routes_by_layer") or {}
282
- routes_by_layer = {str(k): int(v) for k, v in rbl.items()} if isinstance(rbl, dict) else {}
283
- return GraphMetaOutput(
284
- success=True,
285
- enabled=_graph_enabled(),
286
- db_path=meta.get("db_path", resolve_ladybug_path()),
287
- ontology_version=int(meta.get("ontology_version") or 0),
288
- built_at=int(meta.get("built_at") or 0),
289
- source_root=str(meta.get("source_root") or ""),
290
- parse_errors=int(meta.get("parse_errors") or 0),
291
- counts={k: int(v) for k, v in (meta.get("counts") or {}).items()},
292
- module_counts=mod_counts,
293
- microservice_counts=ms_counts,
294
- routes_total=int(meta.get("routes_total") or 0),
295
- exposes_total=int(meta.get("exposes_total") or 0),
296
- routes_by_framework=routes_by_framework,
297
- routes_resolved_pct=float(meta.get("routes_resolved_pct") or 0.0),
298
- routes_from_brownfield_pct=float(meta.get("routes_from_brownfield_pct") or 0.0),
299
- routes_by_layer=routes_by_layer,
300
- edge_counts={str(k): int(v) for k, v in (meta.get("edge_counts") or {}).items()},
301
- http_calls_match_breakdown={
302
- str(k): int(v) for k, v in (meta.get("http_calls_match_breakdown") or {}).items()
303
- },
304
- async_calls_match_breakdown={
305
- str(k): int(v) for k, v in (meta.get("async_calls_match_breakdown") or {}).items()
306
- },
307
- cross_service_calls_total=int(meta.get("cross_service_calls_total") or 0),
308
- cross_service_resolution=meta.get("cross_service_resolution"),
309
- )
310
-
311
-
312
- def list_code_index_tables_payload() -> IndexInfoOutput:
313
- try:
314
- from java_codebase_rag.search.search_lancedb import TABLES
315
-
316
- tables = dict(TABLES)
317
- except ImportError:
318
- # Graph-only install (no lancedb): no Lance vector tables exist.
319
- tables = {}
320
- return IndexInfoOutput(
321
- lancedb_uri=_resolve_lancedb_uri(),
322
- embedding_model=resolved_sbert_model_for_process_env(SBERT_MODEL),
323
- project_root=str(_project_root()),
324
- cocoindex_target=_COCOINDEX_TARGET,
325
- tables=tables,
326
- graph=_graph_meta_output(),
327
- )
328
-
329
-
330
- async def _run_graph_phase(
331
- root: Path,
332
- *,
333
- quiet: bool,
334
- verbose: bool,
335
- on_progress: object | None,
336
- on_progress_console: object | None,
337
- ) -> tuple[int | None, str, str, bool]:
338
- """Run ``build_ast_graph.py`` and return ``(code, stdout, stderr, started)``.
339
-
340
- Shared by the vectors→graph refresh path and the graph-only path (macOS Intel,
341
- where the vector stack is gated off). ``started`` is True only when the graph
342
- subprocess was actually created, so callers set ``phases_run`` accurately — the
343
- CLI maps an empty ``phases_run`` to a preflight exit code 2 (nothing spawned).
344
- A missing builder or a spawn failure returns ``started=False`` with the graph
345
- code carrying the reason (``None`` for missing builder, ``-1`` for spawn error).
346
- """
347
- builder = _BUILDER_FILE
348
- if not builder.is_file():
349
- return None, "", "", False
350
- try:
351
- graph_args = [
352
- sys.executable,
353
- str(builder),
354
- "--source-root",
355
- str(root),
356
- "--ladybug-path",
357
- resolve_ladybug_path(),
358
- ]
359
- if not quiet:
360
- graph_args.append("--verbose")
361
- gproc = await asyncio.create_subprocess_exec(
362
- *graph_args,
363
- cwd=str(root),
364
- env=_cocoindex_subprocess_env(root),
365
- stdout=asyncio.subprocess.PIPE,
366
- stderr=asyncio.subprocess.PIPE,
367
- )
368
- if quiet:
369
- gout_b, gerr_b = await gproc.communicate()
370
- else:
371
- gout_b, gerr_b = await accumulate_and_relay_subprocess_streams(
372
- gproc, relay=True, verbose=verbose,
373
- on_progress=on_progress, on_progress_console=on_progress_console,
374
- )
375
- return (
376
- gproc.returncode,
377
- gout_b.decode(errors="replace"),
378
- gerr_b.decode(errors="replace"),
379
- True,
380
- )
381
- except Exception as exc:
382
- return -1, "", f"graph builder spawn failed: {exc}", False
383
-
384
-
385
- async def run_refresh_pipeline(
386
- *,
387
- quiet: bool = False,
388
- verbose: bool = True,
389
- on_progress=None,
390
- on_progress_console: object | None = None,
391
- ) -> RefreshIndexOutput:
392
- root = _project_root()
393
- if not vector_stack_installed():
394
- # Graph-only install (macOS Intel): the vector stack (cocoindex/lancedb/
395
- # sentence-transformers) is gated off by PEP 508 markers and uninstallable,
396
- # so the cocoindex binary is absent. Skip the vectors phase and build the
397
- # graph only — mirroring init/increment, which treat cocoindex-absent as a
398
- # skip, not a failure (the graph layer is the supported surface there). No
399
- # vectors progress event is emitted, so the renderer's vectors task stays
400
- # invisible (its "never spawned" invariant) instead of hanging at running.
401
- print(VECTORS_SKIPPED_GRAPH_ONLY, file=sys.stderr, flush=True)
402
- if not quiet:
403
- print(file=sys.stderr, flush=True)
404
- graph_code, graph_out, graph_err, started = await _run_graph_phase(
405
- root, quiet=quiet, verbose=verbose,
406
- on_progress=on_progress, on_progress_console=on_progress_console,
407
- )
408
- ok = graph_code == 0
409
- if not ok:
410
- message = (
411
- f"graph builder exit {graph_code}"
412
- if graph_code is not None
413
- else (graph_err.strip() or "graph builder unavailable")
414
- )
415
- else:
416
- message = "reprocess completed (graph-only; vectors skipped — vector stack not installed)"
417
- return RefreshIndexOutput(
418
- success=ok,
419
- exit_code=None,
420
- stdout="",
421
- stderr="",
422
- message=message,
423
- graph_exit_code=graph_code,
424
- graph_stdout=graph_out[-4000:] if len(graph_out) > 4000 else graph_out,
425
- graph_stderr=graph_err[-4000:] if len(graph_err) > 4000 else graph_err,
426
- phases_run=["graph"] if started else [],
427
- optimize_error=None,
428
- )
429
- # Resolve cocoindex the same way the sync path does (pipeline.cocoindex_bin):
430
- # next to the interpreter first, then via PATH (``shutil.which``). A console
431
- # script legitimately lives away from the venv python — e.g. ``pip install
432
- # --user`` places it in ``~/.local/bin``. Honoring PATH keeps ``reprocess``
433
- # (no flags) consistent with init/increment/``reprocess --vectors-only``,
434
- # which all resolve through cocoindex_bin(); previously this path checked
435
- # only next-to-python and failed with "cocoindex not found next to Python"
436
- # even though cocoindex was reachable on PATH.
437
- cocoindex_bin = resolve_cocoindex_bin()
438
- if not cocoindex_bin.is_file():
439
- # 127 pre-spawn: emit a terminal failed vectors event so the renderer's
440
- # task doesn't hang at running (matches the sync pipeline path).
441
- if on_progress is not None:
442
- on_progress(ProgressEvent(kind="vectors", phase=None, pass_=None, done=None, total=None, status="failed", elapsed_s=None))
443
- return RefreshIndexOutput(
444
- success=False,
445
- message=(
446
- f"cocoindex not found next to Python ({cocoindex_bin}) or on PATH; "
447
- "install cocoindex[lancedb] into the same venv or add its bin/ to PATH."
448
- ),
449
- phases_run=[],
450
- )
451
- flow_path = _FLOW_FILE
452
- if not flow_path.is_file():
453
- if on_progress is not None:
454
- on_progress(ProgressEvent(kind="vectors", phase=None, pass_=None, done=None, total=None, status="failed", elapsed_s=None))
455
- return RefreshIndexOutput(
456
- success=False,
457
- message=f"java_index_flow_lancedb.py not found at {flow_path}",
458
- phases_run=[],
459
- )
460
- proc: asyncio.subprocess.Process | None = None
461
- out_b, err_b = b"", b""
462
- # DROP the Lance target tables so the update takes the fast INSERT path
463
- # instead of cocoindex's in-place bulk-update, which emits ~one deletion-
464
- # vector + version commit PER matched row — O(rows) of tiny file IO that
465
- # hangs for many minutes on large repos. Drop+recreate is identical output
466
- # for a full rebuild (the very thing --full-reprocess means). Same fix on
467
- # the sync path: pipeline.run_cocoindex_update. Drop failure is non-fatal:
468
- # the update falls back to the slow in-place path.
469
- try:
470
- drop_proc = await asyncio.create_subprocess_exec(
471
- str(cocoindex_bin),
472
- "drop",
473
- _COCOINDEX_TARGET,
474
- "-f",
475
- cwd=str(flow_path.parent),
476
- env=_cocoindex_subprocess_env(root),
477
- stdout=asyncio.subprocess.PIPE,
478
- stderr=asyncio.subprocess.PIPE,
479
- )
480
- await drop_proc.communicate()
481
- except Exception as exc:
482
- print(
483
- f"java-codebase-rag: drop-before-reprocess failed ({exc!s}); "
484
- "falling back to in-place update",
485
- file=sys.stderr,
486
- )
487
- if quiet:
488
- try:
489
- proc = await asyncio.create_subprocess_exec(
490
- str(cocoindex_bin),
491
- "update",
492
- _COCOINDEX_TARGET,
493
- "--full-reprocess",
494
- "-f",
495
- cwd=str(flow_path.parent),
496
- env=_cocoindex_subprocess_env(root),
497
- stdout=asyncio.subprocess.PIPE,
498
- stderr=asyncio.subprocess.PIPE,
499
- )
500
- out_b, err_b = await proc.communicate()
501
- except Exception as exc:
502
- if on_progress is not None:
503
- on_progress(ProgressEvent(kind="vectors", phase=None, pass_=None, done=None, total=None, status="failed", elapsed_s=None))
504
- return RefreshIndexOutput(
505
- success=False,
506
- message=f"spawn failed: {exc!s}",
507
- phases_run=[],
508
- )
509
- else:
510
- t0 = time.perf_counter()
511
- code_c = -1
512
- try:
513
- proc = await asyncio.create_subprocess_exec(
514
- str(cocoindex_bin),
515
- "update",
516
- _COCOINDEX_TARGET,
517
- "--full-reprocess",
518
- "-f",
519
- cwd=str(flow_path.parent),
520
- env=_cocoindex_subprocess_env(root),
521
- stdout=asyncio.subprocess.PIPE,
522
- stderr=asyncio.subprocess.PIPE,
523
- )
524
- # The vectors task is fed by the child's per-file ticks + the
525
- # approximate total line, parsed by the ProgressRelay inside the
526
- # async drain and routed to on_progress.
527
- out_b, err_b = await accumulate_and_relay_subprocess_streams(
528
- proc, relay=True, verbose=verbose,
529
- on_progress=on_progress, on_progress_console=on_progress_console,
530
- )
531
- code_c = proc.returncode if proc.returncode is not None else -1
532
- except Exception as exc:
533
- if on_progress is not None:
534
- on_progress(ProgressEvent(kind="vectors", phase=None, pass_=None, done=None, total=None, status="failed", elapsed_s=None))
535
- return RefreshIndexOutput(
536
- success=False,
537
- message=f"spawn failed: {exc!s}",
538
- phases_run=[],
539
- )
540
- finally:
541
- # The parent emits the terminal vectors event (the flow can't — no
542
- # "all files done" hook). Drives clamp-on-completion + phase
543
- # transition to Optimize.
544
- if on_progress is not None:
545
- elapsed = time.perf_counter() - t0
546
- status = "done" if code_c == 0 else "failed"
547
- on_progress(ProgressEvent(kind="vectors", phase=None, pass_=None, done=None, total=None, status=status, elapsed_s=elapsed))
548
- assert proc is not None
549
- out = out_b.decode(errors="replace")
550
- err = err_b.decode(errors="replace")
551
- ok = proc.returncode == 0
552
- phases_run: list[Literal["vectors", "graph"]] = ["vectors"]
553
- graph_code: int | None = None
554
- graph_out = ""
555
- graph_err = ""
556
- optimize_error: str | None = None
557
- if ok:
558
- if not quiet:
559
- print(file=sys.stderr, flush=True)
560
- # Serialized post-flow Lance optimize: the flow disabled its background
561
- # optimize, so with cocoindex returned exit 0 there are no concurrent
562
- # writers — this is the safe window to compact. An optimize failure is
563
- # surfaced via optimize_error / stderr and must NOT flip the success of
564
- # a vectors phase that succeeded; the index is still searchable.
565
- try:
566
- from java_codebase_rag.lance_optimize import optimize_lance_tables
567
-
568
- idx_raw = os.environ.get("JAVA_CODEBASE_RAG_INDEX_DIR", "").strip()
569
- if idx_raw and not idx_raw.startswith(("s3://", "gs://", "az://")):
570
- idx_dir = Path(idx_raw).expanduser().resolve()
571
- elif idx_raw:
572
- idx_dir = Path(idx_raw)
573
- else:
574
- idx_dir = (root / ".java-codebase-rag").resolve()
575
- await optimize_lance_tables(idx_dir, quiet=quiet, on_progress=on_progress)
576
- except Exception as exc:
577
- optimize_error = f"lance optimize failed: {exc}"
578
- print(f"java-codebase-rag: {optimize_error}", file=sys.stderr)
579
- graph_code, graph_out, graph_err, graph_started = await _run_graph_phase(
580
- root, quiet=quiet, verbose=verbose,
581
- on_progress=on_progress, on_progress_console=on_progress_console,
582
- )
583
- if graph_started:
584
- phases_run = ["vectors", "graph"]
585
- message: str | None = None
586
- if not ok:
587
- message = f"cocoindex exit {proc.returncode}"
588
- elif graph_code is not None and graph_code != 0:
589
- message = f"graph builder exit {graph_code}"
590
- # Surface a post-flow optimize failure in the message too (success is not
591
- # flipped — the vectors phase succeeded and the index is still usable).
592
- if optimize_error is not None:
593
- message = optimize_error if message is None else f"{message}; {optimize_error}"
594
- return RefreshIndexOutput(
595
- success=ok and (graph_code is None or graph_code == 0),
596
- exit_code=proc.returncode,
597
- stdout=out[-8000:] if len(out) > 8000 else out,
598
- stderr=err[-8000:] if len(err) > 8000 else err,
599
- message=message,
600
- graph_exit_code=graph_code,
601
- graph_stdout=graph_out[-4000:] if len(graph_out) > 4000 else graph_out,
602
- graph_stderr=graph_err[-4000:] if len(graph_err) > 4000 else graph_err,
603
- phases_run=phases_run,
604
- optimize_error=optimize_error,
605
- )
606
-
607
-
608
- def create_mcp_server() -> FastMCP:
609
- mcp = FastMCP("java-codebase-rag", instructions=_INSTRUCTIONS)
610
-
611
- @mcp.tool(
612
- name="search",
613
- description=(
614
- "Ranked chunk retrieval over content tables (java/sql/yaml); `query` is opaque text (natural language or code "
615
- "fragments) and results are score-ranked, not boolean-matched. For graph-structured listing "
616
- "(symbols/routes/clients/producers) use `find`, not `search`. Optional `filter` uses the same NodeFilter "
617
- "schema as `find` but only **symbol-applicable** fields apply — others return success=false. Substring "
618
- "fields match literally (no `*`/`?` metacharacters)—use ranked `query` text for fuzzy discovery. There is **no** "
619
- "structured DSL inside `query`; structured predicates belong in `find`. "
620
- "For identifier-shaped lookups (FQN, id, route/client identifiers, …), use `resolve` first; "
621
- "use `search` for natural-language or ranked fuzzy discovery. "
622
- "Set `explain=true` to include score breakdown per hit. "
623
- "Successful responses echo `limit`/`offset`."
624
- ),
625
- )
626
- async def search(
627
- query: str = Field(description="Search query"),
628
- table: Literal["java", "sql", "yaml", "all"] = Field(
629
- default="java",
630
- description="Which content table to search. 'all' fuses java/sql/yaml results.",
631
- ),
632
- hybrid: bool = Field(
633
- default=False,
634
- description="If true, fuse FTS + vector. Requires a single table (java/sql/yaml); hybrid with table='all' returns success=false.",
635
- ),
636
- limit: int = Field(default=5, ge=1, le=50, description="Max hits to return"),
637
- offset: int = Field(default=0, ge=0, le=500, description="Skip this many hits (pagination)"),
638
- path_contains: str | None = Field(
639
- default=None,
640
- description="Substring match on file path (pre-filter from index)",
641
- ),
642
- filter: mcp_v2.NodeFilter | None = Field(
643
- default=None,
644
- description=(
645
- "Optional NodeFilter post-filter on symbol-oriented hit rows. An empty object or omitted means no "
646
- "predicate. Unknown keys or populated fields not applicable to symbols return success=false."
647
- ),
648
- ),
649
- explain: bool = Field(
650
- default=False,
651
- description="If true, include score_components in each SearchHit (breakdown of distance/rrf, role, symbol, import_penalty).",
652
- ),
653
- chunks: bool = Field(
654
- default=False,
655
- description="If true, show every chunk (default collapses to one row per symbol/type).",
656
- ),
657
- ) -> mcp_v2.SearchOutput:
658
- scoped_filter = _scope_manager.apply_auto_scope(filter) if _scope_manager else filter
659
- return await asyncio.to_thread(
660
- mcp_v2.search_v2,
661
- query,
662
- table,
663
- hybrid,
664
- limit,
665
- offset,
666
- path_contains,
667
- scoped_filter,
668
- explain,
669
- None,
670
- not chunks, # dedup=True by default; chunks=True opts out
671
- )
672
-
673
- @mcp.tool(
674
- name="find",
675
- description=(
676
- "Exact structured listing for one node kind. Per-kind applicable fields: **symbol** — "
677
- "microservice, module, role, exclude_roles, annotation, capability, fqn_contains, symbol_kind, symbol_kinds; "
678
- "**route** — microservice, module, http_method, path_contains, framework; **client** — microservice, module, "
679
- "source_layer, client_kind, target_service, target_path_contains, http_method; **producer** — microservice, "
680
- "module, source_layer, producer_kind, topic_contains. "
681
- "`role` is singular and `exclude_roles` plural; `capability` is a functional tag assigned during indexing. "
682
- "`fqn_contains` is a substring predicate — for exact FQN or id lookup use `resolve`/`describe`. "
683
- "Substring fields match literally (Cypher `CONTAINS`); no wildcard metacharacters. An empty filter (`{}`) or `filter=None` means no predicate (all nodes of "
684
- "that kind; use pagination). Unknown keys or inapplicable populated fields return success=false. "
685
- "Successful responses echo `limit`/`offset`."
686
- ),
687
- )
688
- async def find(
689
- kind: Literal["symbol", "route", "client", "producer"] = Field(
690
- description=(
691
- "Which graph table to search. 'symbol' = declarations, "
692
- "'route' = endpoints, 'client' = outbound HTTP clients, "
693
- "'producer' = outbound async producers."
694
- )
695
- ),
696
- filter: mcp_v2.NodeFilter = Field(
697
- ...,
698
- description=(
699
- "Required NodeFilter object (extra keys forbidden). Fields must be applicable to `kind`."
700
- ),
701
- ),
702
- limit: int = Field(default=25, ge=1, le=500, description="Max nodes to return"),
703
- offset: int = Field(default=0, ge=0, le=499, description="Skip this many nodes (pagination)"),
704
- ) -> mcp_v2.FindOutput:
705
- scoped_filter = _scope_manager.apply_auto_scope(filter) if _scope_manager else filter
706
- return await asyncio.to_thread(mcp_v2.find_v2, kind, scoped_filter, limit, offset, None)
707
-
708
- @mcp.tool(
709
- name="describe",
710
- description=(
711
- "Full node record plus `edge_summary` (in/out counts per stored edge label). For type Symbols, `edge_summary` "
712
- "also exposes composed keys (DECLARES.DECLARES_CLIENT, DECLARES.DECLARES_PRODUCER, DECLARES.EXPOSES); for "
713
- "non-static method Symbols it adds override-axis virtual keys (OVERRIDDEN_BY and its composed forms, plus an "
714
- "`OVERRIDES` map merging stored `[:OVERRIDES]` counts with the dispatch-up rollup). These composed/override keys "
715
- "are out-only and navigable via `neighbors`; the stored `OVERRIDES` is also a normal edge label (in toward declaration). "
716
- "Pass `id` for any kind, or exact `fqn` for Symbol lookup (`id` wins when both are set). "
717
- "`describe(fqn=…)` keeps the first graph row when multiple symbols share that FQN; when an FQN may collide, "
718
- "prefer `resolve(identifier=…, hint_kind='symbol')` first, then `describe(id=…)` on the chosen node."
719
- ),
720
- )
721
- async def describe(
722
- id: str | None = Field(
723
- default=None,
724
- description=(
725
- "Graph node id: sym:, route:, client:, or producer: prefix "
726
- '(e.g. sym:com.bank.chat.core.api.ChatController#joinOperator(JoinOperatorRequest); '
727
- "producer: p:a1b2c3d4e5f67890 — the stored id from the graph, not a human-readable "
728
- "pipe key). For producers by topic, prefer resolve(identifier=<topic>, hint_kind='producer'). "
729
- "When set, takes precedence over fqn."
730
- ),
731
- ),
732
- fqn: str | None = Field(
733
- default=None,
734
- description="Exact FQN for Symbol lookup (alternative to id; Symbol kind only)",
735
- ),
736
- ) -> mcp_v2.DescribeOutput:
737
- return await asyncio.to_thread(mcp_v2.describe_v2, id, fqn, None)
738
-
739
- @mcp.tool(
740
- name="neighbors",
741
- description=(
742
- "Graph walk: **direction** (`in` | `out`) and non-empty **edge_types** are required (one hop over stored edge "
743
- "labels; type/method Symbol origins may also pass composed or override-axis keys — see `edge_types`). From a "
744
- "type Symbol, `direction='out'` with EXPOSES yields route nodes and HTTP_CALLS/ASYNC_CALLS yield client/producer "
745
- "nodes; `direction='in'` reverses each relationship. "
746
- "`direction` and `edge_types` have no defaults; an empty `edge_types` fails. The CALLS-only features — "
747
- "`edge_filter`, `include_unresolved`, `dedup_calls` — each require `edge_types=['CALLS']`; `edge_filter` and "
748
- "`include_unresolved` are mutually exclusive. Violating a precondition (wrong CALLS context, composed/override "
749
- "keys on an ineligible origin or with `direction='in'`, unknown filter keys) returns "
750
- "success=false with a message; `dedup_calls` with other edge_types is a silent no-op. "
751
- "Optional `filter` applies to each neighbor endpoint row; populated fields must be applicable to that "
752
- "neighbor's kind—mixed-kind result sets fail on the first inapplicable neighbor (per-neighbor strict frame). "
753
- "Each edge's `attrs.strategy` indicates resolution quality (brownfield/fallback vs primary paths). "
754
- "Successful responses echo `requested_edge_types`."
755
- ),
756
- )
757
- async def neighbors(
758
- ids: str | list[str] = Field(
759
- description="Origin symbol/route/client/producer id, or list for batch",
760
- ),
761
- direction: Literal["in", "out"] = Field(
762
- description="Required. 'in' = predecessors (callers), 'out' = successors (callees). No default.",
763
- ),
764
- edge_types: list[mcp_v2.NeighborEdgeType] = Field(
765
- description=(
766
- "Required non-empty list of stored edge labels (e.g. CALLS, EXPOSES, HTTP_CALLS, OVERRIDES) "
767
- "and/or composed DECLARES.DECLARES_* (type Symbol origin, out only) or OVERRIDDEN_BY* "
768
- "(non-static method Symbol origin, out only)"
769
- ),
770
- ),
771
- limit: int = Field(
772
- default=25,
773
- ge=1,
774
- le=500,
775
- description=(
776
- "Max edges after concatenating all origins (ids order; offset/limit on merged list)"
777
- ),
778
- ),
779
- offset: int = Field(
780
- default=0,
781
- ge=0,
782
- le=1000,
783
- description="Skip this many edges after merge (pagination)",
784
- ),
785
- filter: mcp_v2.NodeFilter | None = Field(
786
- default=None,
787
- description=(
788
- "Optional NodeFilter on the neighbor node. An empty object or omitted means no predicate. "
789
- "Same applicability rules as `find` for that node's kind."
790
- ),
791
- ),
792
- edge_filter: mcp_v2.EdgeFilter | None = Field(
793
- default=None,
794
- description=(
795
- "Optional EdgeFilter on CALLS edge attributes (edge_types=['CALLS'] only). Use "
796
- "callee_declaring_role for callee stereotype projection — not NodeFilter.role on method neighbors. "
797
- "Mutually exclusive with include_unresolved."
798
- ),
799
- ),
800
- include_unresolved: bool = Field(
801
- default=False,
802
- description=(
803
- "When true with edge_types=['CALLS'] and direction='out', interleave UnresolvedCallSite "
804
- "rows (row_kind='unresolved_call_site') with resolved CALLS in source order. "
805
- "Mutually exclusive with edge_filter."
806
- ),
807
- ),
808
- dedup_calls: bool = Field(
809
- default=False,
810
- description=(
811
- "When true with edge_types=['CALLS'], collapse identical (origin, callee) CALLS to one row "
812
- "with call_site_count and call_site_lines; unresolved sites are not deduped."
813
- ),
814
- ),
815
- ) -> mcp_v2.NeighborsOutput:
816
- scoped_filter = _scope_manager.apply_auto_scope(filter) if _scope_manager else filter
817
- return await asyncio.to_thread(
818
- mcp_v2.neighbors_v2,
819
- ids,
820
- direction,
821
- edge_types,
822
- limit,
823
- offset,
824
- scoped_filter,
825
- edge_filter,
826
- include_unresolved,
827
- dedup_calls,
828
- None,
829
- )
830
-
831
- @mcp.tool(
832
- name="resolve",
833
- description=(
834
- "Identifier-shaped node lookup (FQN, sym:/route:/client:/producer: id, HTTP method+path, "
835
- "route path template, client target_service, target+path pair, or producer topic). Returns "
836
- "status=one (single node), many (≥2 ranked candidates with reason), or none "
837
- "(no match — fall back to search(query=...) for natural language or fuzzy text). "
838
- "Optional hint_kind narrows to symbol, route, client, or producer. "
839
- "Malformed empty/whitespace identifier returns success=false. "
840
- "Examples: resolve('com.foo.Bar', hint_kind='symbol'); "
841
- "resolve('GET /api/v1/customers', hint_kind='route'); "
842
- "resolve('PaymentClient', hint_kind='client'); "
843
- "resolve('order.created', hint_kind='producer'); "
844
- "resolve('the client that handles assignments') → none (use search instead)."
845
- ),
846
- )
847
- async def resolve(
848
- identifier: str = Field(
849
- description=(
850
- "Identifier-shaped node lookup (FQN, id prefix, route path, client target, producer topic, …)"
851
- ),
852
- ),
853
- hint_kind: Literal["symbol", "route", "client", "producer"] | None = Field(
854
- default=None,
855
- description="Optional kind constraint. Omit to search symbol, route, client, and producer.",
856
- ),
857
- ) -> mcp_v2.ResolveOutput:
858
- return await asyncio.to_thread(mcp_v2.resolve_v2, identifier, hint_kind, None)
859
-
860
- return mcp
861
-
862
-
863
- def main() -> None:
864
- raise_fd_limit()
865
- emit_legacy_env_hints_if_present()
866
-
867
- # Load YAML config and apply embedding settings to environment
868
- # This ensures SBERT_MODEL and SBERT_DEVICE from .java-codebase-rag.yml are available
869
- # before any tool handler runs (same behavior as CLI path)
870
- cfg = resolve_operator_config(source_root=_source_root_for_operator_config())
871
- cfg.apply_to_os_environ()
872
- mcp_v2.set_hints_enabled(cfg.hints_enabled)
873
- mcp_v2.set_absence_config(cfg)
874
- resolve_service.set_absence_config(cfg)
875
-
876
- # Initialize scope manager for automatic microservice detection
877
- global _scope_manager
878
- _scope_manager = ScopeManager(cfg.source_root)
879
-
880
- asyncio.run(create_mcp_server().run_stdio_async())
881
-
882
-
883
- if __name__ == "__main__":
884
- main()