java-codebase-rag 0.12.0__py3-none-any.whl → 0.12.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. java_codebase_rag-0.12.2.dist-info/METADATA +35 -0
  2. java_codebase_rag-0.12.2.dist-info/RECORD +4 -0
  3. {java_codebase_rag-0.12.0.dist-info → java_codebase_rag-0.12.2.dist-info}/WHEEL +1 -1
  4. java_codebase_rag/_deprecation.py +0 -103
  5. java_codebase_rag/_fdlimit.py +0 -56
  6. java_codebase_rag/_stdio.py +0 -32
  7. java_codebase_rag/_version.py +0 -35
  8. java_codebase_rag/absence/__init__.py +0 -0
  9. java_codebase_rag/absence/absence_diagnosis.py +0 -700
  10. java_codebase_rag/absence/absence_types.py +0 -124
  11. java_codebase_rag/absence/absence_vocab.py +0 -460
  12. java_codebase_rag/analysis/__init__.py +0 -0
  13. java_codebase_rag/analysis/pr_analysis.py +0 -563
  14. java_codebase_rag/analysis/resolve_service.py +0 -740
  15. java_codebase_rag/ast/__init__.py +0 -0
  16. java_codebase_rag/ast/ast_java.py +0 -2847
  17. java_codebase_rag/ast/ast_kotlin.py +0 -1794
  18. java_codebase_rag/ast/brownfield_events.py +0 -58
  19. java_codebase_rag/ast/chunk_heuristics.py +0 -83
  20. java_codebase_rag/ast/language.py +0 -117
  21. java_codebase_rag/cli.py +0 -1215
  22. java_codebase_rag/cli_dispatch.py +0 -251
  23. java_codebase_rag/cli_format.py +0 -85
  24. java_codebase_rag/cli_progress.py +0 -94
  25. java_codebase_rag/config.py +0 -833
  26. java_codebase_rag/eval/__init__.py +0 -1
  27. java_codebase_rag/eval/ground_truth.py +0 -100
  28. java_codebase_rag/eval/metrics.py +0 -107
  29. java_codebase_rag/eval/runner.py +0 -556
  30. java_codebase_rag/graph/__init__.py +0 -0
  31. java_codebase_rag/graph/build_ast_graph.py +0 -4593
  32. java_codebase_rag/graph/graph_enrich.py +0 -1940
  33. java_codebase_rag/graph/graph_types.py +0 -224
  34. java_codebase_rag/graph/java_ontology.py +0 -465
  35. java_codebase_rag/graph/ladybug_queries.py +0 -2213
  36. java_codebase_rag/graph/path_filtering.py +0 -509
  37. java_codebase_rag/index/__init__.py +0 -0
  38. java_codebase_rag/index/java_index_flow_lancedb.py +0 -879
  39. java_codebase_rag/index/java_index_v1_common.py +0 -33
  40. java_codebase_rag/install_data/__init__.py +0 -0
  41. java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -110
  42. java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
  43. java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
  44. java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
  45. java_codebase_rag/installer.py +0 -2188
  46. java_codebase_rag/jrag.py +0 -4545
  47. java_codebase_rag/jrag_envelope.py +0 -1107
  48. java_codebase_rag/jrag_hints.py +0 -204
  49. java_codebase_rag/jrag_render.py +0 -926
  50. java_codebase_rag/lance_optimize.py +0 -264
  51. java_codebase_rag/mcp/__init__.py +0 -0
  52. java_codebase_rag/mcp/mcp_hints.py +0 -932
  53. java_codebase_rag/mcp/mcp_v2.py +0 -1916
  54. java_codebase_rag/mcp/server.py +0 -886
  55. java_codebase_rag/pipeline.py +0 -531
  56. java_codebase_rag/progress.py +0 -570
  57. java_codebase_rag/read_payloads.py +0 -781
  58. java_codebase_rag/search/__init__.py +0 -0
  59. java_codebase_rag/search/index_common.py +0 -10
  60. java_codebase_rag/search/search_lancedb.py +0 -1296
  61. java_codebase_rag/search/search_lexical.py +0 -449
  62. java_codebase_rag/search/search_scoring.py +0 -537
  63. java_codebase_rag/watch/__init__.py +0 -0
  64. java_codebase_rag/watch/client.py +0 -230
  65. java_codebase_rag/watch/daemon.py +0 -396
  66. java_codebase_rag/watch/lock.py +0 -201
  67. java_codebase_rag/watch/paths.py +0 -76
  68. java_codebase_rag/watch/protocol.py +0 -122
  69. java_codebase_rag/watch/server.py +0 -273
  70. java_codebase_rag/watch/warm.py +0 -105
  71. java_codebase_rag/watch/watcher.py +0 -394
  72. java_codebase_rag-0.12.0.dist-info/METADATA +0 -340
  73. java_codebase_rag-0.12.0.dist-info/RECORD +0 -75
  74. java_codebase_rag-0.12.0.dist-info/entry_points.txt +0 -5
  75. java_codebase_rag-0.12.0.dist-info/licenses/LICENSE +0 -21
  76. java_codebase_rag-0.12.0.dist-info/top_level.txt +0 -1
  77. /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.2.dist-info/top_level.txt +0 -0
@@ -1,886 +0,0 @@
1
- #!/usr/bin/env python3
2
- """LanceDB code-search MCP (stdio)."""
3
- from __future__ import annotations
4
-
5
- import asyncio
6
- import os
7
- import sys
8
- import time
9
- from pathlib import Path
10
- from typing import Literal
11
-
12
- from java_codebase_rag._deprecation import maybe_warn_legacy_alias
13
- from java_codebase_rag.mcp import mcp_v2
14
- from java_codebase_rag.analysis import resolve_service
15
- from java_codebase_rag.search.index_common import SBERT_MODEL
16
- from java_codebase_rag.cli_progress import (
17
- accumulate_and_relay_subprocess_streams,
18
- )
19
- from java_codebase_rag.pipeline import (
20
- VECTORS_SKIPPED_GRAPH_ONLY,
21
- cocoindex_bin as resolve_cocoindex_bin,
22
- vector_stack_installed,
23
- )
24
- from java_codebase_rag.progress import ProgressEvent
25
- from java_codebase_rag._fdlimit import raise_fd_limit
26
- from java_codebase_rag.config import (
27
- cocoindex_subprocess_env_defaults,
28
- discover_project_root,
29
- emit_legacy_env_hints_if_present,
30
- resolved_sbert_model_for_process_env,
31
- resolve_operator_config,
32
- )
33
- from java_codebase_rag.graph.ladybug_queries import LadybugGraph, resolve_ladybug_path
34
- from mcp.server.fastmcp import FastMCP
35
- from pydantic import BaseModel, Field
36
- # NOTE: search_lancedb.TABLES is imported lazily in list_code_index_tables_payload() — it
37
- # pulls lancedb/torch and is unavailable on graph-only installs (macOS Intel).
38
-
39
- _COCOINDEX_TARGET = "java_index_flow_lancedb.py:JavaCodeIndexLance"
40
-
41
- # Package-internal locations of the cocoindex flow and the graph builder, both
42
- # executed by file path (see java_codebase_rag.pipeline). Derived from this
43
- # file's location so they resolve under editable and wheel installs alike.
44
- _PKG_DIR = Path(__file__).resolve().parent.parent
45
- _FLOW_FILE = _PKG_DIR / "index" / "java_index_flow_lancedb.py"
46
- _BUILDER_FILE = _PKG_DIR / "graph" / "build_ast_graph.py"
47
- _INSTRUCTIONS = (
48
- "Java codebase graph navigator over an indexed Java codebase. "
49
- "Tools: search (NL/code locate), find (structured NodeFilter), describe (one node + edge_summary: stored edge-label counts and optional composed keys for type Symbols and override-axis virtual keys for method Symbols), "
50
- "neighbors (one hop; you MUST pass direction in|out AND edge_types list — no defaults), "
51
- "resolve (identifier-shaped lookup for symbol/route/client/producer — three statuses: one | many | none). "
52
- "Unknown filter keys and populated fields not applicable to the effective node kind fail with success=false and message. "
53
- "Successful responses from any tool may include `hints_structured` (tool call suggestions with a `reason` field) and `advisories` (pure informational text) when hints are enabled. "
54
- "Edge labels: EXTENDS, IMPLEMENTS, INJECTS, OVERRIDES, DECLARES, DECLARES_CLIENT, DECLARES_PRODUCER, CALLS, EXPOSES, HTTP_CALLS, ASYNC_CALLS; "
55
- "type Symbols may also use composed neighbors edge_types DECLARES.DECLARES_CLIENT, DECLARES.DECLARES_PRODUCER, DECLARES.EXPOSES (out only, type Symbol origin). "
56
- "Reprocess/init, meta, tables, diagnose-ignore, analyze-pr: use jrag CLI — not MCP."
57
- )
58
-
59
-
60
- class GraphMetaOutput(BaseModel):
61
- success: bool
62
- enabled: bool
63
- db_path: str
64
- ontology_version: int = 0
65
- built_at: int = 0
66
- source_root: str = ""
67
- parse_errors: int = 0
68
- counts: dict[str, int] = Field(default_factory=dict)
69
- module_counts: dict[str, int] = Field(default_factory=dict)
70
- microservice_counts: dict[str, int] = Field(default_factory=dict)
71
- routes_total: int = 0
72
- exposes_total: int = 0
73
- routes_by_framework: dict[str, int] = Field(default_factory=dict)
74
- routes_resolved_pct: float = 0.0
75
- routes_from_brownfield_pct: float = 0.0
76
- routes_by_layer: dict[str, int] = Field(default_factory=dict)
77
- edge_counts: dict[str, int] = Field(default_factory=dict)
78
- http_calls_match_breakdown: dict[str, int] = Field(default_factory=dict)
79
- async_calls_match_breakdown: dict[str, int] = Field(default_factory=dict)
80
- cross_service_calls_total: int = 0
81
- cross_service_resolution: str | None = None
82
- message: str | None = None
83
-
84
-
85
- class RefreshIndexOutput(BaseModel):
86
- """Structured result for ``run_refresh_pipeline`` / CLI ``reprocess`` JSON.
87
-
88
- ``phases_run`` records which phase subprocesses actually started; the CLI maps
89
- failures to exit **2** when it is empty (setup / nothing spawned) and exit **1**
90
- when it is non-empty (build failure). Callers constructing this model manually
91
- must set ``phases_run`` accordingly — omitting it leaves the default ``[]``,
92
- which the CLI treats like a preflight failure.
93
- """
94
-
95
- success: bool
96
- exit_code: int | None = None
97
- stdout: str = ""
98
- stderr: str = ""
99
- message: str | None = None
100
- graph_exit_code: int | None = None
101
- graph_stdout: str = ""
102
- graph_stderr: str = ""
103
- phases_run: list[Literal["vectors", "graph"]] = Field(default_factory=list)
104
- optimize_error: str | None = None
105
-
106
-
107
- class IndexInfoOutput(BaseModel):
108
- lancedb_uri: str
109
- embedding_model: str
110
- project_root: str
111
- cocoindex_target: str
112
- tables: dict[str, str]
113
- graph: GraphMetaOutput
114
-
115
-
116
- # Module-level scope manager, initialized in main()
117
- _scope_manager: ScopeManager | None = None
118
-
119
-
120
- class ScopeManager:
121
- """Manages automatic microservice scope detection and injection."""
122
-
123
- def __init__(self, source_root: Path):
124
- self.source_root = source_root
125
- self.default_scope: str | None = self._detect_scope()
126
- self._log_detection()
127
-
128
- def _detect_scope(self) -> str | None:
129
- from java_codebase_rag.graph.graph_enrich import detect_microservice_from_path
130
-
131
- candidate = detect_microservice_from_path(Path.cwd(), self.source_root)
132
- if candidate is None:
133
- return None
134
- # Only auto-scope to a microservice that actually has indexed code.
135
- # detect_microservice_from_path can mislabel a non-microservice
136
- # top-level child of source_root — most importantly the config/context
137
- # directory the MCP server is launched from (no build marker, no
138
- # source) — via its "first path segment under root" fallback. Scoping
139
- # every query to such a name yields zero matches, so all tools return
140
- # empty. A real microservice the operator is working in is, by
141
- # definition, present in the index, so validating against the indexed
142
- # set cannot suppress a legitimate scope. When the index is unreadable
143
- # (empty known set) we keep the detected candidate rather than silently
144
- # disabling auto-scope on a transient graph error.
145
- known = self._indexed_microservices()
146
- if known and candidate not in known:
147
- return None
148
- return candidate
149
-
150
- def _indexed_microservices(self) -> set[str]:
151
- """Microservice names that have indexed type symbols.
152
-
153
- Graph-only source of truth: the graph is always built alongside Lance,
154
- and a Lance-only index (no graph) is not a supported state. Any failure
155
- (graph missing, open error, empty index) returns an empty set, which
156
- ``_detect_scope`` treats as "cannot validate — keep detection".
157
- """
158
- try:
159
- if not LadybugGraph.exists():
160
- return set()
161
- # LadybugGraph.get() opens the DB and runs meta(); it can raise
162
- # (e.g. RuntimeError on ontology-version mismatch). Caught here ->
163
- # empty set -> _detect_scope keeps the detected scope.
164
- counts = LadybugGraph.get().microservice_counts()
165
- return {name for name in counts if name}
166
- except Exception:
167
- return set()
168
-
169
- def _log_detection(self) -> None:
170
- if self.default_scope:
171
- print(f"[scope] Detected microservice: {self.default_scope}", file=sys.stderr)
172
- print(f"[scope] Queries scoped to {self.default_scope}", file=sys.stderr)
173
- else:
174
- print("[scope] No microservice detected (at project root)", file=sys.stderr)
175
- print("[scope] Queries will span all microservices", file=sys.stderr)
176
-
177
- def apply_auto_scope(self, node_filter: mcp_v2.NodeFilter | None) -> mcp_v2.NodeFilter | None:
178
- """Apply auto-detected scope to filter if no explicit microservice is set."""
179
- if self.default_scope is None:
180
- return node_filter
181
- if node_filter is None:
182
- return mcp_v2.NodeFilter(microservice=self.default_scope)
183
- if node_filter.microservice is None:
184
- return node_filter.model_copy(update={"microservice": self.default_scope})
185
- return node_filter
186
-
187
-
188
- def _resolve_lancedb_uri() -> str:
189
- raw = os.environ.get("JAVA_CODEBASE_RAG_INDEX_DIR", "").strip()
190
- if not raw:
191
- raw = str((_project_root() / ".java-codebase-rag").resolve())
192
- p = Path(raw).expanduser()
193
- if not str(raw).startswith(("s3://", "gs://", "az://")):
194
- try:
195
- return str(p.resolve())
196
- except OSError:
197
- return str(p)
198
- return raw
199
-
200
-
201
- def _project_root() -> Path:
202
- env = os.environ.get("JAVA_CODEBASE_RAG_SOURCE_ROOT", "").strip()
203
- if env:
204
- return Path(env).expanduser().resolve()
205
- discovered = discover_project_root(Path.cwd())
206
- return discovered if discovered is not None else Path.cwd().resolve()
207
-
208
-
209
- def _source_root_for_operator_config() -> Path | None:
210
- """``source_root`` arg to hand ``resolve_operator_config`` from the MCP server.
211
-
212
- Returns ``JAVA_CODEBASE_RAG_SOURCE_ROOT`` when set (an explicit operator
213
- override that wins and suppresses the YAML ``source_root`` field, exactly
214
- like CLI ``--source-root``), otherwise ``None`` — so
215
- ``resolve_operator_config`` runs its OWN walk-up discovery and HONORS the
216
- YAML ``source_root`` field, matching the CLI (``init`` / ``increment`` /
217
- ``reprocess``) path.
218
-
219
- Do NOT pass ``_project_root()`` (the walk-up-discovered dir) here: a
220
- non-``None`` value routes into the "explicit source root" branch that
221
- skips the YAML ``source_root`` field, which made the MCP server and the
222
- CLI resolve different ``source_root`` / ``index_dir`` from the same config
223
- file (the init-vs-MCP index_dir divergence). ``_project_root()`` is kept
224
- only for the ``_resolve_lancedb_uri()`` fallback below.
225
- """
226
- env = os.environ.get("JAVA_CODEBASE_RAG_SOURCE_ROOT", "").strip()
227
- return Path(env).expanduser().resolve() if env else None
228
-
229
-
230
- def _cocoindex_subprocess_env(project_root: Path) -> dict[str, str]:
231
- sub_env = os.environ.copy()
232
- sub_env["JAVA_CODEBASE_RAG_SOURCE_ROOT"] = str(project_root)
233
- idx = os.environ.get("JAVA_CODEBASE_RAG_INDEX_DIR", "").strip()
234
- if idx:
235
- sub_env["JAVA_CODEBASE_RAG_INDEX_DIR"] = str(Path(idx).expanduser().resolve())
236
- # Cap CocoIndex concurrency to avoid EMFILE ("too many open files") under
237
- # default OS fd limits. See: https://github.com/HumanBean17/java-codebase-rag/issues/306
238
- for _k, _v in cocoindex_subprocess_env_defaults().items():
239
- sub_env.setdefault(_k, _v)
240
- return sub_env
241
-
242
-
243
- def _graph_enabled() -> bool:
244
- return LadybugGraph.exists()
245
-
246
-
247
- def _graph_meta_output() -> GraphMetaOutput:
248
- if not LadybugGraph.exists():
249
- return GraphMetaOutput(
250
- success=True,
251
- enabled=False,
252
- db_path=resolve_ladybug_path(),
253
- message="Ladybug graph not present; run jrag reprocess or build_ast_graph.py",
254
- )
255
- try:
256
- graph = LadybugGraph.get()
257
- meta = graph.meta()
258
- except Exception as e:
259
- return GraphMetaOutput(
260
- success=False,
261
- enabled=_graph_enabled(),
262
- db_path=resolve_ladybug_path(),
263
- message=f"Ladybug open failed: {e}",
264
- )
265
- if "error" in meta:
266
- return GraphMetaOutput(
267
- success=False,
268
- enabled=_graph_enabled(),
269
- db_path=meta.get("db_path", resolve_ladybug_path()),
270
- message=str(meta["error"]),
271
- )
272
- try:
273
- mod_counts = graph.module_counts()
274
- except Exception:
275
- mod_counts = {}
276
- try:
277
- ms_counts = graph.microservice_counts()
278
- except Exception:
279
- ms_counts = {}
280
- rfw = meta.get("routes_by_framework") or {}
281
- routes_by_framework = {str(k): int(v) for k, v in rfw.items()} if isinstance(rfw, dict) else {}
282
- rbl = meta.get("routes_by_layer") or {}
283
- routes_by_layer = {str(k): int(v) for k, v in rbl.items()} if isinstance(rbl, dict) else {}
284
- return GraphMetaOutput(
285
- success=True,
286
- enabled=_graph_enabled(),
287
- db_path=meta.get("db_path", resolve_ladybug_path()),
288
- ontology_version=int(meta.get("ontology_version") or 0),
289
- built_at=int(meta.get("built_at") or 0),
290
- source_root=str(meta.get("source_root") or ""),
291
- parse_errors=int(meta.get("parse_errors") or 0),
292
- counts={k: int(v) for k, v in (meta.get("counts") or {}).items()},
293
- module_counts=mod_counts,
294
- microservice_counts=ms_counts,
295
- routes_total=int(meta.get("routes_total") or 0),
296
- exposes_total=int(meta.get("exposes_total") or 0),
297
- routes_by_framework=routes_by_framework,
298
- routes_resolved_pct=float(meta.get("routes_resolved_pct") or 0.0),
299
- routes_from_brownfield_pct=float(meta.get("routes_from_brownfield_pct") or 0.0),
300
- routes_by_layer=routes_by_layer,
301
- edge_counts={str(k): int(v) for k, v in (meta.get("edge_counts") or {}).items()},
302
- http_calls_match_breakdown={
303
- str(k): int(v) for k, v in (meta.get("http_calls_match_breakdown") or {}).items()
304
- },
305
- async_calls_match_breakdown={
306
- str(k): int(v) for k, v in (meta.get("async_calls_match_breakdown") or {}).items()
307
- },
308
- cross_service_calls_total=int(meta.get("cross_service_calls_total") or 0),
309
- cross_service_resolution=meta.get("cross_service_resolution"),
310
- )
311
-
312
-
313
- def list_code_index_tables_payload() -> IndexInfoOutput:
314
- try:
315
- from java_codebase_rag.search.search_lancedb import TABLES
316
-
317
- tables = dict(TABLES)
318
- except ImportError:
319
- # Graph-only install (no lancedb): no Lance vector tables exist.
320
- tables = {}
321
- return IndexInfoOutput(
322
- lancedb_uri=_resolve_lancedb_uri(),
323
- embedding_model=resolved_sbert_model_for_process_env(SBERT_MODEL),
324
- project_root=str(_project_root()),
325
- cocoindex_target=_COCOINDEX_TARGET,
326
- tables=tables,
327
- graph=_graph_meta_output(),
328
- )
329
-
330
-
331
- async def _run_graph_phase(
332
- root: Path,
333
- *,
334
- quiet: bool,
335
- verbose: bool,
336
- on_progress: object | None,
337
- on_progress_console: object | None,
338
- ) -> tuple[int | None, str, str, bool]:
339
- """Run ``build_ast_graph.py`` and return ``(code, stdout, stderr, started)``.
340
-
341
- Shared by the vectors→graph refresh path and the graph-only path (macOS Intel,
342
- where the vector stack is gated off). ``started`` is True only when the graph
343
- subprocess was actually created, so callers set ``phases_run`` accurately — the
344
- CLI maps an empty ``phases_run`` to a preflight exit code 2 (nothing spawned).
345
- A missing builder or a spawn failure returns ``started=False`` with the graph
346
- code carrying the reason (``None`` for missing builder, ``-1`` for spawn error).
347
- """
348
- builder = _BUILDER_FILE
349
- if not builder.is_file():
350
- return None, "", "", False
351
- try:
352
- graph_args = [
353
- sys.executable,
354
- str(builder),
355
- "--source-root",
356
- str(root),
357
- "--ladybug-path",
358
- resolve_ladybug_path(),
359
- ]
360
- if not quiet:
361
- graph_args.append("--verbose")
362
- gproc = await asyncio.create_subprocess_exec(
363
- *graph_args,
364
- cwd=str(root),
365
- env=_cocoindex_subprocess_env(root),
366
- stdout=asyncio.subprocess.PIPE,
367
- stderr=asyncio.subprocess.PIPE,
368
- )
369
- if quiet:
370
- gout_b, gerr_b = await gproc.communicate()
371
- else:
372
- gout_b, gerr_b = await accumulate_and_relay_subprocess_streams(
373
- gproc, relay=True, verbose=verbose,
374
- on_progress=on_progress, on_progress_console=on_progress_console,
375
- )
376
- return (
377
- gproc.returncode,
378
- gout_b.decode(errors="replace"),
379
- gerr_b.decode(errors="replace"),
380
- True,
381
- )
382
- except Exception as exc:
383
- return -1, "", f"graph builder spawn failed: {exc}", False
384
-
385
-
386
- async def run_refresh_pipeline(
387
- *,
388
- quiet: bool = False,
389
- verbose: bool = True,
390
- on_progress=None,
391
- on_progress_console: object | None = None,
392
- ) -> RefreshIndexOutput:
393
- root = _project_root()
394
- if not vector_stack_installed():
395
- # Graph-only install (macOS Intel): the vector stack (cocoindex/lancedb/
396
- # sentence-transformers) is gated off by PEP 508 markers and uninstallable,
397
- # so the cocoindex binary is absent. Skip the vectors phase and build the
398
- # graph only — mirroring init/increment, which treat cocoindex-absent as a
399
- # skip, not a failure (the graph layer is the supported surface there). No
400
- # vectors progress event is emitted, so the renderer's vectors task stays
401
- # invisible (its "never spawned" invariant) instead of hanging at running.
402
- print(VECTORS_SKIPPED_GRAPH_ONLY, file=sys.stderr, flush=True)
403
- if not quiet:
404
- print(file=sys.stderr, flush=True)
405
- graph_code, graph_out, graph_err, started = await _run_graph_phase(
406
- root, quiet=quiet, verbose=verbose,
407
- on_progress=on_progress, on_progress_console=on_progress_console,
408
- )
409
- ok = graph_code == 0
410
- if not ok:
411
- message = (
412
- f"graph builder exit {graph_code}"
413
- if graph_code is not None
414
- else (graph_err.strip() or "graph builder unavailable")
415
- )
416
- else:
417
- message = "reprocess completed (graph-only; vectors skipped — vector stack not installed)"
418
- return RefreshIndexOutput(
419
- success=ok,
420
- exit_code=None,
421
- stdout="",
422
- stderr="",
423
- message=message,
424
- graph_exit_code=graph_code,
425
- graph_stdout=graph_out[-4000:] if len(graph_out) > 4000 else graph_out,
426
- graph_stderr=graph_err[-4000:] if len(graph_err) > 4000 else graph_err,
427
- phases_run=["graph"] if started else [],
428
- optimize_error=None,
429
- )
430
- # Resolve cocoindex the same way the sync path does (pipeline.cocoindex_bin):
431
- # next to the interpreter first, then via PATH (``shutil.which``). A console
432
- # script legitimately lives away from the venv python — e.g. ``pip install
433
- # --user`` places it in ``~/.local/bin``. Honoring PATH keeps ``reprocess``
434
- # (no flags) consistent with init/increment/``reprocess --vectors-only``,
435
- # which all resolve through cocoindex_bin(); previously this path checked
436
- # only next-to-python and failed with "cocoindex not found next to Python"
437
- # even though cocoindex was reachable on PATH.
438
- cocoindex_bin = resolve_cocoindex_bin()
439
- if not cocoindex_bin.is_file():
440
- # 127 pre-spawn: emit a terminal failed vectors event so the renderer's
441
- # task doesn't hang at running (matches the sync pipeline path).
442
- if on_progress is not None:
443
- on_progress(ProgressEvent(kind="vectors", phase=None, pass_=None, done=None, total=None, status="failed", elapsed_s=None))
444
- return RefreshIndexOutput(
445
- success=False,
446
- message=(
447
- f"cocoindex not found next to Python ({cocoindex_bin}) or on PATH; "
448
- "install cocoindex[lancedb] into the same venv or add its bin/ to PATH."
449
- ),
450
- phases_run=[],
451
- )
452
- flow_path = _FLOW_FILE
453
- if not flow_path.is_file():
454
- if on_progress is not None:
455
- on_progress(ProgressEvent(kind="vectors", phase=None, pass_=None, done=None, total=None, status="failed", elapsed_s=None))
456
- return RefreshIndexOutput(
457
- success=False,
458
- message=f"java_index_flow_lancedb.py not found at {flow_path}",
459
- phases_run=[],
460
- )
461
- proc: asyncio.subprocess.Process | None = None
462
- out_b, err_b = b"", b""
463
- # DROP the Lance target tables so the update takes the fast INSERT path
464
- # instead of cocoindex's in-place bulk-update, which emits ~one deletion-
465
- # vector + version commit PER matched row — O(rows) of tiny file IO that
466
- # hangs for many minutes on large repos. Drop+recreate is identical output
467
- # for a full rebuild (the very thing --full-reprocess means). Same fix on
468
- # the sync path: pipeline.run_cocoindex_update. Drop failure is non-fatal:
469
- # the update falls back to the slow in-place path.
470
- try:
471
- drop_proc = await asyncio.create_subprocess_exec(
472
- str(cocoindex_bin),
473
- "drop",
474
- _COCOINDEX_TARGET,
475
- "-f",
476
- cwd=str(flow_path.parent),
477
- env=_cocoindex_subprocess_env(root),
478
- stdout=asyncio.subprocess.PIPE,
479
- stderr=asyncio.subprocess.PIPE,
480
- )
481
- await drop_proc.communicate()
482
- except Exception as exc:
483
- print(
484
- f"jrag: drop-before-reprocess failed ({exc!s}); "
485
- "falling back to in-place update",
486
- file=sys.stderr,
487
- )
488
- if quiet:
489
- try:
490
- proc = await asyncio.create_subprocess_exec(
491
- str(cocoindex_bin),
492
- "update",
493
- _COCOINDEX_TARGET,
494
- "--full-reprocess",
495
- "-f",
496
- cwd=str(flow_path.parent),
497
- env=_cocoindex_subprocess_env(root),
498
- stdout=asyncio.subprocess.PIPE,
499
- stderr=asyncio.subprocess.PIPE,
500
- )
501
- out_b, err_b = await proc.communicate()
502
- except Exception as exc:
503
- if on_progress is not None:
504
- on_progress(ProgressEvent(kind="vectors", phase=None, pass_=None, done=None, total=None, status="failed", elapsed_s=None))
505
- return RefreshIndexOutput(
506
- success=False,
507
- message=f"spawn failed: {exc!s}",
508
- phases_run=[],
509
- )
510
- else:
511
- t0 = time.perf_counter()
512
- code_c = -1
513
- try:
514
- proc = await asyncio.create_subprocess_exec(
515
- str(cocoindex_bin),
516
- "update",
517
- _COCOINDEX_TARGET,
518
- "--full-reprocess",
519
- "-f",
520
- cwd=str(flow_path.parent),
521
- env=_cocoindex_subprocess_env(root),
522
- stdout=asyncio.subprocess.PIPE,
523
- stderr=asyncio.subprocess.PIPE,
524
- )
525
- # The vectors task is fed by the child's per-file ticks + the
526
- # approximate total line, parsed by the ProgressRelay inside the
527
- # async drain and routed to on_progress.
528
- out_b, err_b = await accumulate_and_relay_subprocess_streams(
529
- proc, relay=True, verbose=verbose,
530
- on_progress=on_progress, on_progress_console=on_progress_console,
531
- )
532
- code_c = proc.returncode if proc.returncode is not None else -1
533
- except Exception as exc:
534
- if on_progress is not None:
535
- on_progress(ProgressEvent(kind="vectors", phase=None, pass_=None, done=None, total=None, status="failed", elapsed_s=None))
536
- return RefreshIndexOutput(
537
- success=False,
538
- message=f"spawn failed: {exc!s}",
539
- phases_run=[],
540
- )
541
- finally:
542
- # The parent emits the terminal vectors event (the flow can't — no
543
- # "all files done" hook). Drives clamp-on-completion + phase
544
- # transition to Optimize.
545
- if on_progress is not None:
546
- elapsed = time.perf_counter() - t0
547
- status = "done" if code_c == 0 else "failed"
548
- on_progress(ProgressEvent(kind="vectors", phase=None, pass_=None, done=None, total=None, status=status, elapsed_s=elapsed))
549
- assert proc is not None
550
- out = out_b.decode(errors="replace")
551
- err = err_b.decode(errors="replace")
552
- ok = proc.returncode == 0
553
- phases_run: list[Literal["vectors", "graph"]] = ["vectors"]
554
- graph_code: int | None = None
555
- graph_out = ""
556
- graph_err = ""
557
- optimize_error: str | None = None
558
- if ok:
559
- if not quiet:
560
- print(file=sys.stderr, flush=True)
561
- # Serialized post-flow Lance optimize: the flow disabled its background
562
- # optimize, so with cocoindex returned exit 0 there are no concurrent
563
- # writers — this is the safe window to compact. An optimize failure is
564
- # surfaced via optimize_error / stderr and must NOT flip the success of
565
- # a vectors phase that succeeded; the index is still searchable.
566
- try:
567
- from java_codebase_rag.lance_optimize import optimize_lance_tables
568
-
569
- idx_raw = os.environ.get("JAVA_CODEBASE_RAG_INDEX_DIR", "").strip()
570
- if idx_raw and not idx_raw.startswith(("s3://", "gs://", "az://")):
571
- idx_dir = Path(idx_raw).expanduser().resolve()
572
- elif idx_raw:
573
- idx_dir = Path(idx_raw)
574
- else:
575
- idx_dir = (root / ".java-codebase-rag").resolve()
576
- await optimize_lance_tables(idx_dir, quiet=quiet, on_progress=on_progress)
577
- except Exception as exc:
578
- optimize_error = f"lance optimize failed: {exc}"
579
- print(f"jrag: {optimize_error}", file=sys.stderr)
580
- graph_code, graph_out, graph_err, graph_started = await _run_graph_phase(
581
- root, quiet=quiet, verbose=verbose,
582
- on_progress=on_progress, on_progress_console=on_progress_console,
583
- )
584
- if graph_started:
585
- phases_run = ["vectors", "graph"]
586
- message: str | None = None
587
- if not ok:
588
- message = f"cocoindex exit {proc.returncode}"
589
- elif graph_code is not None and graph_code != 0:
590
- message = f"graph builder exit {graph_code}"
591
- # Surface a post-flow optimize failure in the message too (success is not
592
- # flipped — the vectors phase succeeded and the index is still usable).
593
- if optimize_error is not None:
594
- message = optimize_error if message is None else f"{message}; {optimize_error}"
595
- return RefreshIndexOutput(
596
- success=ok and (graph_code is None or graph_code == 0),
597
- exit_code=proc.returncode,
598
- stdout=out[-8000:] if len(out) > 8000 else out,
599
- stderr=err[-8000:] if len(err) > 8000 else err,
600
- message=message,
601
- graph_exit_code=graph_code,
602
- graph_stdout=graph_out[-4000:] if len(graph_out) > 4000 else graph_out,
603
- graph_stderr=graph_err[-4000:] if len(graph_err) > 4000 else graph_err,
604
- phases_run=phases_run,
605
- optimize_error=optimize_error,
606
- )
607
-
608
-
609
- def create_mcp_server() -> FastMCP:
610
- mcp = FastMCP("java-codebase-rag", instructions=_INSTRUCTIONS)
611
-
612
- @mcp.tool(
613
- name="search",
614
- description=(
615
- "Ranked chunk retrieval over content tables (java/sql/yaml); `query` is opaque text (natural language or code "
616
- "fragments) and results are score-ranked, not boolean-matched. For graph-structured listing "
617
- "(symbols/routes/clients/producers) use `find`, not `search`. Optional `filter` uses the same NodeFilter "
618
- "schema as `find` but only **symbol-applicable** fields apply — others return success=false. Substring "
619
- "fields match literally (no `*`/`?` metacharacters)—use ranked `query` text for fuzzy discovery. There is **no** "
620
- "structured DSL inside `query`; structured predicates belong in `find`. "
621
- "For identifier-shaped lookups (FQN, id, route/client identifiers, …), use `resolve` first; "
622
- "use `search` for natural-language or ranked fuzzy discovery. "
623
- "Set `explain=true` to include score breakdown per hit. "
624
- "Successful responses echo `limit`/`offset`."
625
- ),
626
- )
627
- async def search(
628
- query: str = Field(description="Search query"),
629
- table: Literal["java", "sql", "yaml", "all"] = Field(
630
- default="java",
631
- description="Which content table to search. 'all' fuses java/sql/yaml results.",
632
- ),
633
- hybrid: bool = Field(
634
- default=False,
635
- description="If true, fuse FTS + vector. Requires a single table (java/sql/yaml); hybrid with table='all' returns success=false.",
636
- ),
637
- limit: int = Field(default=5, ge=1, le=50, description="Max hits to return"),
638
- offset: int = Field(default=0, ge=0, le=500, description="Skip this many hits (pagination)"),
639
- path_contains: str | None = Field(
640
- default=None,
641
- description="Substring match on file path (pre-filter from index)",
642
- ),
643
- filter: mcp_v2.NodeFilter | None = Field(
644
- default=None,
645
- description=(
646
- "Optional NodeFilter post-filter on symbol-oriented hit rows. An empty object or omitted means no "
647
- "predicate. Unknown keys or populated fields not applicable to symbols return success=false."
648
- ),
649
- ),
650
- explain: bool = Field(
651
- default=False,
652
- description="If true, include score_components in each SearchHit (breakdown of distance/rrf, role, symbol, import_penalty).",
653
- ),
654
- chunks: bool = Field(
655
- default=False,
656
- description="If true, show every chunk (default collapses to one row per symbol/type).",
657
- ),
658
- ) -> mcp_v2.SearchOutput:
659
- scoped_filter = _scope_manager.apply_auto_scope(filter) if _scope_manager else filter
660
- return await asyncio.to_thread(
661
- mcp_v2.search_v2,
662
- query,
663
- table,
664
- hybrid,
665
- limit,
666
- offset,
667
- path_contains,
668
- scoped_filter,
669
- explain,
670
- None,
671
- not chunks, # dedup=True by default; chunks=True opts out
672
- )
673
-
674
- @mcp.tool(
675
- name="find",
676
- description=(
677
- "Exact structured listing for one node kind. Per-kind applicable fields: **symbol** — "
678
- "microservice, module, role, exclude_roles, annotation, capability, fqn_contains, symbol_kind, symbol_kinds; "
679
- "**route** — microservice, module, http_method, path_contains, framework; **client** — microservice, module, "
680
- "source_layer, client_kind, target_service, target_path_contains, http_method; **producer** — microservice, "
681
- "module, source_layer, producer_kind, topic_contains. "
682
- "`role` is singular and `exclude_roles` plural; `capability` is a functional tag assigned during indexing. "
683
- "`fqn_contains` is a substring predicate — for exact FQN or id lookup use `resolve`/`describe`. "
684
- "Substring fields match literally (Cypher `CONTAINS`); no wildcard metacharacters. An empty filter (`{}`) or `filter=None` means no predicate (all nodes of "
685
- "that kind; use pagination). Unknown keys or inapplicable populated fields return success=false. "
686
- "Successful responses echo `limit`/`offset`."
687
- ),
688
- )
689
- async def find(
690
- kind: Literal["symbol", "route", "client", "producer"] = Field(
691
- description=(
692
- "Which graph table to search. 'symbol' = declarations, "
693
- "'route' = endpoints, 'client' = outbound HTTP clients, "
694
- "'producer' = outbound async producers."
695
- )
696
- ),
697
- filter: mcp_v2.NodeFilter = Field(
698
- ...,
699
- description=(
700
- "Required NodeFilter object (extra keys forbidden). Fields must be applicable to `kind`."
701
- ),
702
- ),
703
- limit: int = Field(default=25, ge=1, le=500, description="Max nodes to return"),
704
- offset: int = Field(default=0, ge=0, le=499, description="Skip this many nodes (pagination)"),
705
- ) -> mcp_v2.FindOutput:
706
- scoped_filter = _scope_manager.apply_auto_scope(filter) if _scope_manager else filter
707
- return await asyncio.to_thread(mcp_v2.find_v2, kind, scoped_filter, limit, offset, None)
708
-
709
- @mcp.tool(
710
- name="describe",
711
- description=(
712
- "Full node record plus `edge_summary` (in/out counts per stored edge label). For type Symbols, `edge_summary` "
713
- "also exposes composed keys (DECLARES.DECLARES_CLIENT, DECLARES.DECLARES_PRODUCER, DECLARES.EXPOSES); for "
714
- "non-static method Symbols it adds override-axis virtual keys (OVERRIDDEN_BY and its composed forms, plus an "
715
- "`OVERRIDES` map merging stored `[:OVERRIDES]` counts with the dispatch-up rollup). These composed/override keys "
716
- "are out-only and navigable via `neighbors`; the stored `OVERRIDES` is also a normal edge label (in toward declaration). "
717
- "Pass `id` for any kind, or exact `fqn` for Symbol lookup (`id` wins when both are set). "
718
- "`describe(fqn=…)` keeps the first graph row when multiple symbols share that FQN; when an FQN may collide, "
719
- "prefer `resolve(identifier=…, hint_kind='symbol')` first, then `describe(id=…)` on the chosen node."
720
- ),
721
- )
722
- async def describe(
723
- id: str | None = Field(
724
- default=None,
725
- description=(
726
- "Graph node id: sym:, route:, client:, or producer: prefix "
727
- '(e.g. sym:com.bank.chat.core.api.ChatController#joinOperator(JoinOperatorRequest); '
728
- "producer: p:a1b2c3d4e5f67890 — the stored id from the graph, not a human-readable "
729
- "pipe key). For producers by topic, prefer resolve(identifier=<topic>, hint_kind='producer'). "
730
- "When set, takes precedence over fqn."
731
- ),
732
- ),
733
- fqn: str | None = Field(
734
- default=None,
735
- description="Exact FQN for Symbol lookup (alternative to id; Symbol kind only)",
736
- ),
737
- ) -> mcp_v2.DescribeOutput:
738
- return await asyncio.to_thread(mcp_v2.describe_v2, id, fqn, None)
739
-
740
- @mcp.tool(
741
- name="neighbors",
742
- description=(
743
- "Graph walk: **direction** (`in` | `out`) and non-empty **edge_types** are required (one hop over stored edge "
744
- "labels; type/method Symbol origins may also pass composed or override-axis keys — see `edge_types`). From a "
745
- "type Symbol, `direction='out'` with EXPOSES yields route nodes and HTTP_CALLS/ASYNC_CALLS yield client/producer "
746
- "nodes; `direction='in'` reverses each relationship. "
747
- "`direction` and `edge_types` have no defaults; an empty `edge_types` fails. The CALLS-only features — "
748
- "`edge_filter`, `include_unresolved`, `dedup_calls` — each require `edge_types=['CALLS']`; `edge_filter` and "
749
- "`include_unresolved` are mutually exclusive. Violating a precondition (wrong CALLS context, composed/override "
750
- "keys on an ineligible origin or with `direction='in'`, unknown filter keys) returns "
751
- "success=false with a message; `dedup_calls` with other edge_types is a silent no-op. "
752
- "Optional `filter` applies to each neighbor endpoint row; populated fields must be applicable to that "
753
- "neighbor's kind—mixed-kind result sets fail on the first inapplicable neighbor (per-neighbor strict frame). "
754
- "Each edge's `attrs.strategy` indicates resolution quality (brownfield/fallback vs primary paths). "
755
- "Successful responses echo `requested_edge_types`."
756
- ),
757
- )
758
- async def neighbors(
759
- ids: str | list[str] = Field(
760
- description="Origin symbol/route/client/producer id, or list for batch",
761
- ),
762
- direction: Literal["in", "out"] = Field(
763
- description="Required. 'in' = predecessors (callers), 'out' = successors (callees). No default.",
764
- ),
765
- edge_types: list[mcp_v2.NeighborEdgeType] = Field(
766
- description=(
767
- "Required non-empty list of stored edge labels (e.g. CALLS, EXPOSES, HTTP_CALLS, OVERRIDES) "
768
- "and/or composed DECLARES.DECLARES_* (type Symbol origin, out only) or OVERRIDDEN_BY* "
769
- "(non-static method Symbol origin, out only)"
770
- ),
771
- ),
772
- limit: int = Field(
773
- default=25,
774
- ge=1,
775
- le=500,
776
- description=(
777
- "Max edges after concatenating all origins (ids order; offset/limit on merged list)"
778
- ),
779
- ),
780
- offset: int = Field(
781
- default=0,
782
- ge=0,
783
- le=1000,
784
- description="Skip this many edges after merge (pagination)",
785
- ),
786
- filter: mcp_v2.NodeFilter | None = Field(
787
- default=None,
788
- description=(
789
- "Optional NodeFilter on the neighbor node. An empty object or omitted means no predicate. "
790
- "Same applicability rules as `find` for that node's kind."
791
- ),
792
- ),
793
- edge_filter: mcp_v2.EdgeFilter | None = Field(
794
- default=None,
795
- description=(
796
- "Optional EdgeFilter on CALLS edge attributes (edge_types=['CALLS'] only). Use "
797
- "callee_declaring_role for callee stereotype projection — not NodeFilter.role on method neighbors. "
798
- "Mutually exclusive with include_unresolved."
799
- ),
800
- ),
801
- include_unresolved: bool = Field(
802
- default=False,
803
- description=(
804
- "When true with edge_types=['CALLS'] and direction='out', interleave UnresolvedCallSite "
805
- "rows (row_kind='unresolved_call_site') with resolved CALLS in source order. "
806
- "Mutually exclusive with edge_filter."
807
- ),
808
- ),
809
- dedup_calls: bool = Field(
810
- default=False,
811
- description=(
812
- "When true with edge_types=['CALLS'], collapse identical (origin, callee) CALLS to one row "
813
- "with call_site_count and call_site_lines; unresolved sites are not deduped."
814
- ),
815
- ),
816
- ) -> mcp_v2.NeighborsOutput:
817
- scoped_filter = _scope_manager.apply_auto_scope(filter) if _scope_manager else filter
818
- return await asyncio.to_thread(
819
- mcp_v2.neighbors_v2,
820
- ids,
821
- direction,
822
- edge_types,
823
- limit,
824
- offset,
825
- scoped_filter,
826
- edge_filter,
827
- include_unresolved,
828
- dedup_calls,
829
- None,
830
- )
831
-
832
- @mcp.tool(
833
- name="resolve",
834
- description=(
835
- "Identifier-shaped node lookup (FQN, sym:/route:/client:/producer: id, HTTP method+path, "
836
- "route path template, client target_service, target+path pair, or producer topic). Returns "
837
- "status=one (single node), many (≥2 ranked candidates with reason), or none "
838
- "(no match — fall back to search(query=...) for natural language or fuzzy text). "
839
- "Optional hint_kind narrows to symbol, route, client, or producer. "
840
- "Malformed empty/whitespace identifier returns success=false. "
841
- "Examples: resolve('com.foo.Bar', hint_kind='symbol'); "
842
- "resolve('GET /api/v1/customers', hint_kind='route'); "
843
- "resolve('PaymentClient', hint_kind='client'); "
844
- "resolve('order.created', hint_kind='producer'); "
845
- "resolve('the client that handles assignments') → none (use search instead)."
846
- ),
847
- )
848
- async def resolve(
849
- identifier: str = Field(
850
- description=(
851
- "Identifier-shaped node lookup (FQN, id prefix, route path, client target, producer topic, …)"
852
- ),
853
- ),
854
- hint_kind: Literal["symbol", "route", "client", "producer"] | None = Field(
855
- default=None,
856
- description="Optional kind constraint. Omit to search symbol, route, client, and producer.",
857
- ),
858
- ) -> mcp_v2.ResolveOutput:
859
- return await asyncio.to_thread(mcp_v2.resolve_v2, identifier, hint_kind, None)
860
-
861
- return mcp
862
-
863
-
864
- def main() -> None:
865
- maybe_warn_legacy_alias()
866
- raise_fd_limit()
867
- emit_legacy_env_hints_if_present()
868
-
869
- # Load YAML config and apply embedding settings to environment
870
- # This ensures SBERT_MODEL and SBERT_DEVICE from .java-codebase-rag.yml are available
871
- # before any tool handler runs (same behavior as CLI path)
872
- cfg = resolve_operator_config(source_root=_source_root_for_operator_config())
873
- cfg.apply_to_os_environ()
874
- mcp_v2.set_hints_enabled(cfg.hints_enabled)
875
- mcp_v2.set_absence_config(cfg)
876
- resolve_service.set_absence_config(cfg)
877
-
878
- # Initialize scope manager for automatic microservice detection
879
- global _scope_manager
880
- _scope_manager = ScopeManager(cfg.source_root)
881
-
882
- asyncio.run(create_mcp_server().run_stdio_async())
883
-
884
-
885
- if __name__ == "__main__":
886
- main()