java-codebase-rag 0.12.0__py3-none-any.whl → 0.12.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. java_codebase_rag-0.12.2.dist-info/METADATA +35 -0
  2. java_codebase_rag-0.12.2.dist-info/RECORD +4 -0
  3. {java_codebase_rag-0.12.0.dist-info → java_codebase_rag-0.12.2.dist-info}/WHEEL +1 -1
  4. java_codebase_rag/_deprecation.py +0 -103
  5. java_codebase_rag/_fdlimit.py +0 -56
  6. java_codebase_rag/_stdio.py +0 -32
  7. java_codebase_rag/_version.py +0 -35
  8. java_codebase_rag/absence/__init__.py +0 -0
  9. java_codebase_rag/absence/absence_diagnosis.py +0 -700
  10. java_codebase_rag/absence/absence_types.py +0 -124
  11. java_codebase_rag/absence/absence_vocab.py +0 -460
  12. java_codebase_rag/analysis/__init__.py +0 -0
  13. java_codebase_rag/analysis/pr_analysis.py +0 -563
  14. java_codebase_rag/analysis/resolve_service.py +0 -740
  15. java_codebase_rag/ast/__init__.py +0 -0
  16. java_codebase_rag/ast/ast_java.py +0 -2847
  17. java_codebase_rag/ast/ast_kotlin.py +0 -1794
  18. java_codebase_rag/ast/brownfield_events.py +0 -58
  19. java_codebase_rag/ast/chunk_heuristics.py +0 -83
  20. java_codebase_rag/ast/language.py +0 -117
  21. java_codebase_rag/cli.py +0 -1215
  22. java_codebase_rag/cli_dispatch.py +0 -251
  23. java_codebase_rag/cli_format.py +0 -85
  24. java_codebase_rag/cli_progress.py +0 -94
  25. java_codebase_rag/config.py +0 -833
  26. java_codebase_rag/eval/__init__.py +0 -1
  27. java_codebase_rag/eval/ground_truth.py +0 -100
  28. java_codebase_rag/eval/metrics.py +0 -107
  29. java_codebase_rag/eval/runner.py +0 -556
  30. java_codebase_rag/graph/__init__.py +0 -0
  31. java_codebase_rag/graph/build_ast_graph.py +0 -4593
  32. java_codebase_rag/graph/graph_enrich.py +0 -1940
  33. java_codebase_rag/graph/graph_types.py +0 -224
  34. java_codebase_rag/graph/java_ontology.py +0 -465
  35. java_codebase_rag/graph/ladybug_queries.py +0 -2213
  36. java_codebase_rag/graph/path_filtering.py +0 -509
  37. java_codebase_rag/index/__init__.py +0 -0
  38. java_codebase_rag/index/java_index_flow_lancedb.py +0 -879
  39. java_codebase_rag/index/java_index_v1_common.py +0 -33
  40. java_codebase_rag/install_data/__init__.py +0 -0
  41. java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -110
  42. java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
  43. java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
  44. java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
  45. java_codebase_rag/installer.py +0 -2188
  46. java_codebase_rag/jrag.py +0 -4545
  47. java_codebase_rag/jrag_envelope.py +0 -1107
  48. java_codebase_rag/jrag_hints.py +0 -204
  49. java_codebase_rag/jrag_render.py +0 -926
  50. java_codebase_rag/lance_optimize.py +0 -264
  51. java_codebase_rag/mcp/__init__.py +0 -0
  52. java_codebase_rag/mcp/mcp_hints.py +0 -932
  53. java_codebase_rag/mcp/mcp_v2.py +0 -1916
  54. java_codebase_rag/mcp/server.py +0 -886
  55. java_codebase_rag/pipeline.py +0 -531
  56. java_codebase_rag/progress.py +0 -570
  57. java_codebase_rag/read_payloads.py +0 -781
  58. java_codebase_rag/search/__init__.py +0 -0
  59. java_codebase_rag/search/index_common.py +0 -10
  60. java_codebase_rag/search/search_lancedb.py +0 -1296
  61. java_codebase_rag/search/search_lexical.py +0 -449
  62. java_codebase_rag/search/search_scoring.py +0 -537
  63. java_codebase_rag/watch/__init__.py +0 -0
  64. java_codebase_rag/watch/client.py +0 -230
  65. java_codebase_rag/watch/daemon.py +0 -396
  66. java_codebase_rag/watch/lock.py +0 -201
  67. java_codebase_rag/watch/paths.py +0 -76
  68. java_codebase_rag/watch/protocol.py +0 -122
  69. java_codebase_rag/watch/server.py +0 -273
  70. java_codebase_rag/watch/warm.py +0 -105
  71. java_codebase_rag/watch/watcher.py +0 -394
  72. java_codebase_rag-0.12.0.dist-info/METADATA +0 -340
  73. java_codebase_rag-0.12.0.dist-info/RECORD +0 -75
  74. java_codebase_rag-0.12.0.dist-info/entry_points.txt +0 -5
  75. java_codebase_rag-0.12.0.dist-info/licenses/LICENSE +0 -21
  76. java_codebase_rag-0.12.0.dist-info/top_level.txt +0 -1
  77. /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.2.dist-info/top_level.txt +0 -0
@@ -1,879 +0,0 @@
1
- """
2
- CocoIndex 1.0 app: index Java, Flyway SQL, and YAML into LanceDB.
3
-
4
- LanceDB requires a single primary key per table; each chunk gets a UUID `id`.
5
-
6
- Environment:
7
- JAVA_CODEBASE_RAG_INDEX_DIR — Lance tables + LadybugDB + cocoindex state (default: ./.java-codebase-rag)
8
- JAVA_CODEBASE_RAG_SOURCE_ROOT — Java repo root for indexing (optional; else cocoindex cwd)
9
- SBERT_MODEL / SBERT_DEVICE — embedding (optional; YAML also supported via jrag CLI)
10
-
11
- Dependencies:
12
- pip install "cocoindex[lancedb]" sentence-transformers
13
-
14
- Usage:
15
- cocoindex update java_index_flow_lancedb.py:JavaCodeIndexLance --full-reprocess
16
- """
17
- from __future__ import annotations
18
-
19
- import asyncio
20
- import inspect
21
- import os
22
- import sys
23
- import threading
24
- import uuid
25
- from collections.abc import AsyncIterator
26
- from contextlib import asynccontextmanager
27
- from dataclasses import dataclass
28
- from fnmatch import fnmatch
29
- from pathlib import Path
30
- from typing import Annotated, Any
31
-
32
- import cocoindex as coco
33
- import numpy as np
34
- import numpy.typing as npt
35
- import pyarrow as pa
36
- from cocoindex.connectors import lancedb, localfs
37
- from cocoindex.connectors.lancedb import LanceType
38
- from cocoindex.ops.sentence_transformers import SentenceTransformerEmbedder
39
- from cocoindex.ops.text import RecursiveSplitter, detect_code_language
40
- from cocoindex.resources.file import PatternFilePathMatcher
41
-
42
- from java_codebase_rag.config import resolved_sbert_model_for_process_env
43
- from java_codebase_rag.lance_optimize import LANCE_TABLE_NAMES
44
- from java_codebase_rag.index.java_index_v1_common import (
45
- JAVA_CHUNK,
46
- SBERT_MODEL,
47
- SQL_CHUNK,
48
- YAML_CHUNK,
49
- chunk_key_range,
50
- position_to_json,
51
- )
52
- from java_codebase_rag.graph.path_filtering import LayeredIgnore
53
- from java_codebase_rag.ast.ast_java import ONTOLOGY_VERSION
54
- from java_codebase_rag.ast.language import LANG_BACKENDS, backend_for
55
- from java_codebase_rag.graph.graph_enrich import (
56
- classify_java_file,
57
- collect_annotation_meta_chain,
58
- enrich_chunk,
59
- load_brownfield_overrides,
60
- load_generated_detection,
61
- )
62
-
63
- # Older cocoindex (e.g. 1.0.0a43) uses ``tracked=False``; newer releases renamed
64
- # the flag to ``detect_change`` (default False) and reject ``tracked``.
65
- _ck_params = inspect.signature(coco.ContextKey.__init__).parameters
66
- if "detect_change" in _ck_params:
67
- PROJECT_ROOT = coco.ContextKey[Path]("java_lance_project_root")
68
- LANCE_DB = coco.ContextKey("java_lance_async_conn")
69
- EMBEDDER = coco.ContextKey[SentenceTransformerEmbedder]("java_lance_embedder")
70
- IGNORE = coco.ContextKey[LayeredIgnore]("java_lance_layered_ignore")
71
- elif "tracked" in _ck_params:
72
- PROJECT_ROOT = coco.ContextKey[Path]("java_lance_project_root", tracked=False)
73
- LANCE_DB = coco.ContextKey("java_lance_async_conn", tracked=False)
74
- EMBEDDER = coco.ContextKey[SentenceTransformerEmbedder](
75
- "java_lance_embedder", tracked=False
76
- )
77
- IGNORE = coco.ContextKey[LayeredIgnore](
78
- "java_lance_layered_ignore", tracked=False
79
- )
80
- else:
81
- PROJECT_ROOT = coco.ContextKey[Path]("java_lance_project_root")
82
- LANCE_DB = coco.ContextKey("java_lance_async_conn")
83
- EMBEDDER = coco.ContextKey[SentenceTransformerEmbedder]("java_lance_embedder")
84
- IGNORE = coco.ContextKey[LayeredIgnore]("java_lance_layered_ignore")
85
-
86
- splitter = RecursiveSplitter()
87
-
88
- # LanceDB table optimization: cocoindex >=1.0.15 runs ``table.optimize()``
89
- # INLINE during merge_insert commits, gated by stats (only when small fragments
90
- # accumulate — see _RowHandler._maybe_optimize / _evaluate_optimize). That
91
- # replaces the old 1.0.7 *background* asyncio optimize that raced concurrent
92
- # Deletes (lancedb#1504 commit conflicts) and which we used to disable via
93
- # ``num_transactions_before_optimize`` (kwarg removed in 1.0.16). Being inline,
94
- # it no longer races anything. ``lance_optimize.optimize_lance_tables`` still
95
- # runs a final serialized compaction post-flow. ``optimize()`` is pure
96
- # maintenance (compact/prune/index); upsert/delete correctness via merge_insert
97
- # does not depend on it.
98
-
99
-
100
- # --- Vectors-phase progress emission (JCIRAG_PROGRESS kind=vectors) -----------
101
- #
102
- # The flow runs in a CHILD cocoindex process; it prints structured progress to
103
- # its stderr and the parent (pipeline._popen_capturing_stderr /
104
- # cli_progress.accumulate_and_relay_subprocess_streams) parses it via
105
- # ProgressRelay and feeds the renderer. The flow CANNOT know when all files are
106
- # done (cocoindex offers no "all files done" hook in the flow), so it emits:
107
- # - ONE ``total=N status=running`` line from ``app_main`` (approximate
108
- # pre-walk: matcher includes + LayeredIgnore), and
109
- # - per-file ``done=k status=running`` ticks (throttled every ~25 files) from
110
- # ``process_*_file`` (shared atomic counter).
111
- # The PARENT emits the terminal ``status=done``/``failed`` vectors event on
112
- # cocoindex exit (drives clamp-on-completion + phase transition to Optimize).
113
-
114
- # Per-file tick cadence: bound stderr volume on huge trees without making the
115
- # bar feel stale. Every 25th file (and the modulo boundary is enough — the
116
- # parent clamps to total on the terminal event anyway).
117
- _VECTORS_TICK_EVERY = 25
118
-
119
- # Suffixes that index into the ``JavaLanceChunk`` table — the registered
120
- # language backends (``.java`` always; ``.kt`` when the Kotlin grammar imports).
121
- # Derived from ``LANG_BACKENDS`` so this never drifts from what the graph builder
122
- # parses (mirrors the watcher's ``INDEXED_SUFFIXES``). On a grammar-absent
123
- # install this is just ``(".java",)`` and ``.kt`` files are skipped cleanly.
124
- _INDEXED_SOURCE_SUFFIXES: tuple[str, ...] = tuple(
125
- suffix for backend in LANG_BACKENDS.values() for suffix in backend.suffixes
126
- )
127
- # True iff some registered backend claims ``.kt`` (i.e. ``tree-sitter-kotlin``
128
- # imported). Gates the ``.kt`` cocoindex matcher + ``process_kotlin_file`` drain
129
- # in ``app_main`` so a grammar-absent install skips ``.kt`` by construction
130
- # instead of crashing inside ``_parse_and_enrich_java`` (backend-for-``.kt``
131
- # returns ``None`` → ``classify_java_file`` dereferences ``ast.all_types``).
132
- _KOTLIN_REGISTERED: bool = ".kt" in _INDEXED_SOURCE_SUFFIXES
133
-
134
- # Bounded concurrency for the per-file drain in app_main. cocoindex's embedder
135
- # is ``@coco.fn.as_async(batching=True, runner=GPU, max_batch_size=64)`` — but
136
- # its batching layer only coalesces calls that are in flight SIMULTANEOUSLY. A
137
- # serial ``async for … await`` loop keeps just one file's chunks (avg 1–3) in
138
- # flight, so real batches stay tiny and MPS idles between them (measured ~138
139
- # chunks/s vs the ~235 chunks/s ceiling at batch=64 for all-MiniLM-L6-v2).
140
- # Draining many files at once with a semaphore puts their chunks in flight
141
- # together → the embedder coalesces them into full batches → MPS climbs toward
142
- # the ceiling. Measured on Shopizer (1167 files / 3475 chunks): full init drops
143
- # from ~46.7s (serial) to ~36.0s (32) / ~34.3s (64), with identical row output.
144
- #
145
- # This stays inside ONE component, so the earlier mount_each→app_main win is
146
- # preserved: still exactly ONE merge_insert per table at commit. Memoization
147
- # (``@coco.fn(memo=True)``) and the lock-guarded tick counter are safe under
148
- # concurrency; the backend parse (``parse_java`` / ``parse_kotlin``) uses a
149
- # per-thread tree-sitter Parser (already routed via ``asyncio.to_thread``)
150
- # and ``splitter.split`` is synchronous so the event loop cannot reenter it.
151
- #
152
- # Default 64 is sized to MATCH the embedder's hardcoded max_batch_size=64 (the
153
- # decorator above; not a constructor arg, so not raisable from the flow): ~64
154
- # files in flight reliably fills a 64-chunk batch and saturates MPS. Going higher
155
- # buys nothing — the batch is already capped — and lower underfills it. Memory
156
- # is NOT the limiting factor here: cocoindex buffers ALL staged rows until the
157
- # single final merge_insert regardless of concurrency, so peak RSS is set by
158
- # total chunk count (the commit buffer), not by how many files process at once.
159
- # Set to ``1`` for the old serial behavior; raise/lower only if you have also
160
- # changed the effective batch size or are constraining the commit buffer itself.
161
- _FILE_CONCURRENCY = max(
162
- 1,
163
- int(os.environ.get("JAVA_CODEBASE_RAG_FILE_CONCURRENCY", "64") or "64"),
164
- )
165
-
166
- # Thread-safe counter: cocoindex may call process_*_file concurrently
167
- # (mount_each parallelism is implementation-defined). A module-level lock guards
168
- # both the counter and the emission so two threads never interleave a tick.
169
- _vectors_done_lock = threading.Lock()
170
- _vectors_done_count = 0
171
-
172
-
173
- def _emit_vectors_progress(
174
- *,
175
- done: int | None = None,
176
- total: int | None = None,
177
- status: str = "running",
178
- elapsed_s: float | None = None,
179
- ) -> None:
180
- """Emit one ``JCIRAG_PROGRESS kind=vectors …`` line to stderr (flushed).
181
-
182
- Field order is fixed (kind, done, total, status, elapsed_s) so the parser
183
- and tests can pin substrings. Omitted fields are simply absent.
184
- """
185
- fields = ["kind=vectors"]
186
- if done is not None:
187
- fields.append(f"done={done}")
188
- if total is not None:
189
- fields.append(f"total={total}")
190
- fields.append(f"status={status}")
191
- if elapsed_s is not None:
192
- fields.append(f"elapsed_s={elapsed_s:.2f}")
193
- print("JCIRAG_PROGRESS " + " ".join(fields), file=sys.stderr, flush=True)
194
-
195
-
196
- def _tick_vectors_done() -> None:
197
- """Increment the shared per-file counter and emit a throttled ``done=k`` tick.
198
-
199
- Called once per successfully-processed file (after the ignore / empty
200
- early-returns). The tick is emitted every ``_VECTORS_TICK_EVERY`` files so
201
- stderr volume stays bounded on huge trees; the parent clamps to total on
202
- the terminal event, so the exact tick cadence is not load-bearing.
203
- """
204
- global _vectors_done_count
205
- with _vectors_done_lock:
206
- _vectors_done_count += 1
207
- n = _vectors_done_count
208
- if n % _VECTORS_TICK_EVERY != 0:
209
- return
210
- # Emit under the lock: the docstring above promises the lock guards both
211
- # the counter AND the emission, so two concurrent ticks can't emit their
212
- # ``done=N`` lines out of order. Contention is negligible (fires every
213
- # ~25 files).
214
- _emit_vectors_progress(done=n, status="running")
215
-
216
-
217
- def _approximate_vectors_total(project_root: Path) -> int:
218
- """Reproduce the matchers' include globs + LayeredIgnore for an approximate total.
219
-
220
- The flow applies two filtering layers: (1) ``PatternFilePathMatcher``
221
- excludes at walk time via ``LayeredIgnore.cocoindex_excluded_patterns()``,
222
- then (2) ``LayeredIgnore.is_ignored()`` plus an early-return for empty /
223
- undecodable files inside each ``process_*_file``. Files that early-return
224
- never tick, so this pre-walk OVERSTATES the total by the ignored / empty
225
- count. The parent clamps the bar to 100% on the terminal ``status=done``
226
- event, so the over-count cannot stall the bar.
227
-
228
- Mirrors the ``localfs.walk_dir`` matchers in ``app_main``:
229
- - ``**/*.java`` and ``**/*.kt`` (registered language suffixes)
230
- - ``**/src/main/resources/db/migration/*.sql``
231
- - ``**/src/main/resources/application*.yml`` and ``.yaml``
232
- """
233
- ignore = LayeredIgnore(project_root)
234
- excluded = ignore.cocoindex_excluded_patterns()
235
-
236
- def _excluded(rel_posix: str) -> bool:
237
- return any(fnmatch(rel_posix, pat) for pat in excluded)
238
-
239
- total = 0
240
- for dirpath, dirnames, filenames in os.walk(project_root):
241
- # Prune the same universal nuisance dirs as iter_source_files /
242
- # cocoindex walk. (build-output pruning is matcher-dependent in the
243
- # real walk; for an APPROXIMATE total this cheap prune is sufficient
244
- # — the clamp absorbs any residual divergence.)
245
- dirnames[:] = [
246
- d for d in dirnames if d not in (".git", ".hg", ".svn", "node_modules", ".venv", "venv")
247
- ]
248
- for fn in filenames:
249
- full = Path(dirpath) / fn
250
- try:
251
- rel = full.resolve().relative_to(project_root).as_posix()
252
- except ValueError:
253
- continue
254
- if _excluded(rel):
255
- continue
256
- # Java + Kotlin: the registered source-language suffixes (see
257
- # ``_INDEXED_SOURCE_SUFFIXES`` / ``LANG_BACKENDS`` — ``.java`` always,
258
- # ``.kt`` when the Kotlin grammar imports). Both index into the same
259
- # ``JavaLanceChunk`` table via ``process_java_file`` /
260
- # ``process_kotlin_file``.
261
- if fn.endswith(_INDEXED_SOURCE_SUFFIXES):
262
- if not ignore.is_ignored(full):
263
- total += 1
264
- continue
265
- # SQL: **/src/main/resources/db/migration/*.sql
266
- if fn.endswith(".sql") and "/db/migration/" in rel:
267
- if not ignore.is_ignored(full):
268
- total += 1
269
- continue
270
- # YAML: **/src/main/resources/application*.yml / .yaml
271
- # NOTE: ``fn`` is the bare filename (e.g. ``application-cloud.yml``), so
272
- # the prefix predicate must be ``fn.startswith("application")`` —
273
- # ``"/application" in fn`` was always False (no leading slash in a bare
274
- # name) and under-counted every application YAML, driving the pre-walk
275
- # total below the actual done count. The ``rel``-based
276
- # ``"/src/main/resources/"`` gate stays (full path component).
277
- if fn.endswith((".yml", ".yaml")) and fn.startswith("application") and "/src/main/resources/" in rel:
278
- if not ignore.is_ignored(full):
279
- total += 1
280
- return total
281
-
282
-
283
- @dataclass
284
- class JavaLanceChunk:
285
- id: str
286
- filename: str
287
- language: str
288
- text: str
289
- range_start: int
290
- range_end: int
291
- start: dict[str, Any]
292
- end: dict[str, Any]
293
- embedding: Annotated[npt.NDArray[np.float32], EMBEDDER]
294
- package: str
295
- module: str
296
- microservice: str
297
- primary_type_fqn: str
298
- primary_type_kind: str
299
- role: str
300
- # Native PyArrow lists: without the LanceType override CocoIndex would JSON-encode
301
- # `list[str]` into a STRING column, which caller code then iterates character-by-character.
302
- capabilities: Annotated[list[str], LanceType(pa.list_(pa.string()))]
303
- annotations_on_type: Annotated[list[str], LanceType(pa.list_(pa.string()))]
304
- symbols: Annotated[list[str], LanceType(pa.list_(pa.string()))]
305
- ontology_version: int
306
- # Generated source detection: populated per-file, not per-chunk
307
- generated: bool
308
- generated_by: str | None
309
-
310
-
311
- @dataclass
312
- class SqlLanceChunk:
313
- id: str
314
- filename: str
315
- text: str
316
- range_start: int
317
- range_end: int
318
- start: dict[str, Any]
319
- end: dict[str, Any]
320
- embedding: Annotated[npt.NDArray[np.float32], EMBEDDER]
321
-
322
-
323
- @dataclass
324
- class YamlLanceChunk:
325
- id: str
326
- filename: str
327
- text: str
328
- range_start: int
329
- range_end: int
330
- start: dict[str, Any]
331
- end: dict[str, Any]
332
- embedding: Annotated[npt.NDArray[np.float32], EMBEDDER]
333
-
334
-
335
- @coco.lifespan
336
- async def coco_lifespan(builder: coco.EnvironmentBuilder) -> AsyncIterator[None]:
337
- idx_raw = os.environ.get("JAVA_CODEBASE_RAG_INDEX_DIR", "").strip()
338
- if idx_raw and not idx_raw.startswith(("s3://", "gs://", "az://")):
339
- index_dir = Path(idx_raw).expanduser().resolve()
340
- else:
341
- index_dir = (Path(".").resolve() / ".java-codebase-rag").resolve()
342
- index_dir.mkdir(parents=True, exist_ok=True)
343
- builder.settings.db_path = index_dir / "cocoindex.db"
344
-
345
- env_root = os.environ.get("JAVA_CODEBASE_RAG_SOURCE_ROOT", "").strip()
346
- if env_root:
347
- root = Path(env_root).expanduser().resolve()
348
- else:
349
- root = Path(".").resolve()
350
- builder.provide(PROJECT_ROOT, root)
351
-
352
- # Default to Apple Metal (MPS) when available: ~1.7x faster encode on
353
- # all-MiniLM-L6-v2 (measured), and the win grows with repo size since
354
- # embedding dominates on large trees. torch is already on the import path
355
- # here (sentence-transformers pulls it), so the availability check is free
356
- # in this child process — and it keeps the CLI parent (config.py) from ever
357
- # paying a torch import. Operators force CPU with SBERT_DEVICE=cpu.
358
- device = os.environ.get("SBERT_DEVICE") or None
359
- if device is None:
360
- try:
361
- import torch # noqa: WPS433 (local import: avoid parent-import cost)
362
- if torch.backends.mps.is_available():
363
- device = "mps"
364
- except Exception:
365
- pass
366
- embedder = SentenceTransformerEmbedder(
367
- resolved_sbert_model_for_process_env(SBERT_MODEL),
368
- device=device,
369
- trust_remote_code=True,
370
- )
371
- builder.provide(EMBEDDER, embedder)
372
- builder.provide(IGNORE, LayeredIgnore(root))
373
-
374
- uri = str(index_dir)
375
-
376
- @asynccontextmanager
377
- async def _lance_cm() -> AsyncIterator[Any]:
378
- conn = await lancedb.connect_async(uri)
379
- try:
380
- yield conn
381
- finally:
382
- conn.close()
383
-
384
- await builder.provide_async_with(LANCE_DB, _lance_cm())
385
- yield
386
-
387
-
388
- def _parse_and_enrich_java(
389
- content_bytes: bytes,
390
- chunks: list[Any],
391
- rel: str,
392
- project_root: Path,
393
- ) -> tuple[list[Any], Any]:
394
- """Parse one Java file and enrich every chunk, off the event loop.
395
-
396
- Returns a tuple of (enrichments, ast) where enrichments is a list of
397
- :class:`graph_enrich.ChunkEnrichment` aligned 1:1 with ``chunks``, and ast
398
- is the parsed :class:`JavaFileAst`. Intended to run via ``asyncio.to_thread``
399
- from ``process_java_file`` (vectors perf lever #2): while the worker thread
400
- parses + enriches, the event loop is free to drive other files and keep the
401
- embedder's batching queue fed.
402
-
403
- Thread-safety: the backend parse (``parse_java`` / ``parse_kotlin``) uses
404
- a per-thread tree-sitter ``Parser`` (see ``ast_java._parser`` /
405
- ``ast_kotlin._parser``), so it is safe to call concurrently from these
406
- worker threads — including the transitive re-parse that ``enrich_chunk``
407
- triggers via ``collect_annotation_meta_chain`` → ``_collect_annotation_decl_index``.
408
- ``enrich_chunk`` is otherwise pure-Python over the now-immutable AST; its
409
- ``lru_cache`` reads are thread-safe under the GIL.
410
- """
411
- backend = backend_for(rel)
412
- if backend is None:
413
- # Defensive: ``app_main`` registers the ``.kt`` matcher + kotlin drain
414
- # only when ``_KOTLIN_REGISTERED`` (registry-derived), and ``.java`` is
415
- # always registered — so by construction this is unreachable for every
416
- # suffix the flow yields. Kept to honor the dispatch contract (a
417
- # grammar-absent install never yields ``.kt`` here).
418
- return [], None
419
- ast = backend.parse(content_bytes, filename=rel)
420
- enrichments = [
421
- enrich_chunk(
422
- ast,
423
- chunk_start_byte=ch.start.byte_offset,
424
- chunk_end_byte=ch.end.byte_offset,
425
- file_path=rel,
426
- project_root=project_root,
427
- )
428
- for ch in chunks
429
- ]
430
- return enrichments, ast
431
-
432
-
433
- @coco.fn(memo=True)
434
- async def process_java_file(
435
- file: localfs.File,
436
- table: lancedb.TableTarget[JavaLanceChunk],
437
- ) -> None:
438
- embedder = coco.use_context(EMBEDDER)
439
- project_root = coco.use_context(PROJECT_ROOT)
440
- ignore = coco.use_context(IGNORE)
441
- if ignore.is_ignored((project_root / file.file_path.path).resolve()):
442
- return
443
- try:
444
- content = await file.read_text()
445
- except UnicodeDecodeError:
446
- return
447
- if not content.strip():
448
- return
449
-
450
- _tick_vectors_done()
451
-
452
- language = detect_code_language(filename=file.file_path.path.name) or "text"
453
- cs, mn, ov = JAVA_CHUNK
454
- # ``splitter.split`` stays inline: the module-level ``RecursiveSplitter``
455
- # shares one Rust object, so keeping split on the event loop preserves its
456
- # existing single-threaded access (no new cross-file concurrency hazard).
457
- chunks = splitter.split(
458
- content,
459
- cs,
460
- min_chunk_size=mn,
461
- chunk_overlap=ov,
462
- language=language,
463
- )
464
- rel = file.file_path.path.as_posix()
465
- content_bytes = content.encode("utf-8", errors="replace")
466
-
467
- # (vectors perf lever #2) parse + enrich off the event loop so the loop can
468
- # keep the embedder's batching queue fed while this file is being parsed.
469
- # The backend parse (parse_java / parse_kotlin) is thread-safe (per-thread
470
- # tree-sitter Parser in ast_java / ast_kotlin).
471
- enrichments, ast = await asyncio.to_thread(
472
- _parse_and_enrich_java, content_bytes, chunks, rel, project_root
473
- )
474
-
475
- # Compute generated source detection once per file (uses the AST and content_bytes)
476
- generated_config = load_generated_detection(project_root)
477
- generated, generated_by = classify_java_file(
478
- content_bytes, ast, config=generated_config, project_root=project_root
479
- )
480
-
481
- # (vectors perf lever #1) embed all chunks concurrently so the batched
482
- # embedder groups them into one ``model.encode(...)`` (max_batch_size=64)
483
- # instead of N serial batch-of-1 calls. Dominant win for ``increment``
484
- # (few changed files → little cross-file concurrency → otherwise no batching).
485
- embeddings = await asyncio.gather(*(embedder.embed(ch.text) for ch in chunks))
486
-
487
- for ch, enrich, emb in zip(chunks, enrichments, embeddings):
488
- rs, re = chunk_key_range(ch)
489
- table.declare_row(
490
- row=JavaLanceChunk(
491
- id=str(uuid.uuid4()),
492
- filename=rel,
493
- language=language,
494
- text=ch.text,
495
- range_start=rs,
496
- range_end=re,
497
- start=position_to_json(ch.start),
498
- end=position_to_json(ch.end),
499
- embedding=emb,
500
- package=enrich.package,
501
- module=enrich.module,
502
- microservice=enrich.microservice,
503
- primary_type_fqn=enrich.primary_type_fqn,
504
- primary_type_kind=enrich.primary_type_kind,
505
- role=enrich.role,
506
- capabilities=list(enrich.capabilities),
507
- annotations_on_type=enrich.annotations_on_type,
508
- symbols=enrich.symbols,
509
- ontology_version=ONTOLOGY_VERSION,
510
- generated=generated,
511
- generated_by=generated_by,
512
- )
513
- )
514
-
515
-
516
- @coco.fn(memo=True)
517
- async def process_kotlin_file(
518
- file: localfs.File,
519
- table: lancedb.TableTarget[JavaLanceChunk],
520
- ) -> None:
521
- """Index one ``.kt`` file into the SAME ``JavaLanceChunk`` table as Java.
522
-
523
- Mirrors ``process_java_file``'s enrichment path (``enrich_chunk`` /
524
- ``classify_java_file``) but parsing dispatches through ``backend_for(rel)``
525
- (= ``parse_kotlin``) inside the shared ``_parse_and_enrich_java`` helper,
526
- which is already language-agnostic. The chunk ``language`` field is set to
527
- ``"kotlin"`` (``detect_code_language`` returns ``"kotlin"`` for ``.kt``).
528
- The chunk schema (``primary_type_kind`` / ``role`` / ``capabilities``) is
529
- language-agnostic, so no new column is needed.
530
-
531
- Multifile-facade merge is NOT wired here: each ``process_*_file`` parses ONE
532
- file independently (concurrent per-file drain), so a cross-file pre-pass is
533
- awkward inside cocoindex's dataflow. The merge runs in ``build_ast_graph``
534
- pass1 instead — the only site that registers facade TypeDecls into
535
- ``tables.types`` (where unmerged facades would collide). Chunk enrichment
536
- (``enrich_chunk``) uses the per-file AST only, so it is merge-independent.
537
- """
538
- embedder = coco.use_context(EMBEDDER)
539
- project_root = coco.use_context(PROJECT_ROOT)
540
- ignore = coco.use_context(IGNORE)
541
- if ignore.is_ignored((project_root / file.file_path.path).resolve()):
542
- return
543
- try:
544
- content = await file.read_text()
545
- except UnicodeDecodeError:
546
- return
547
- if not content.strip():
548
- return
549
-
550
- _tick_vectors_done()
551
-
552
- language = detect_code_language(filename=file.file_path.path.name) or "text"
553
- cs, mn, ov = JAVA_CHUNK
554
- chunks = splitter.split(
555
- content,
556
- cs,
557
- min_chunk_size=mn,
558
- chunk_overlap=ov,
559
- language=language,
560
- )
561
- rel = file.file_path.path.as_posix()
562
- content_bytes = content.encode("utf-8", errors="replace")
563
-
564
- # ``_parse_and_enrich_java`` dispatches via ``backend_for(rel)`` → parse_kotlin
565
- # for ``.kt``; the helper and ``enrich_chunk`` are language-agnostic. Run off
566
- # the event loop so the embedder batching queue stays fed (vectors perf #2).
567
- enrichments, ast = await asyncio.to_thread(
568
- _parse_and_enrich_java, content_bytes, chunks, rel, project_root
569
- )
570
-
571
- generated_config = load_generated_detection(project_root)
572
- generated, generated_by = classify_java_file(
573
- content_bytes, ast, config=generated_config, project_root=project_root
574
- )
575
-
576
- # Embed all chunks concurrently → batched encode (vectors perf #1).
577
- embeddings = await asyncio.gather(*(embedder.embed(ch.text) for ch in chunks))
578
-
579
- for ch, enrich, emb in zip(chunks, enrichments, embeddings):
580
- rs, re = chunk_key_range(ch)
581
- table.declare_row(
582
- row=JavaLanceChunk(
583
- id=str(uuid.uuid4()),
584
- filename=rel,
585
- language=language,
586
- text=ch.text,
587
- range_start=rs,
588
- range_end=re,
589
- start=position_to_json(ch.start),
590
- end=position_to_json(ch.end),
591
- embedding=emb,
592
- package=enrich.package,
593
- module=enrich.module,
594
- microservice=enrich.microservice,
595
- primary_type_fqn=enrich.primary_type_fqn,
596
- primary_type_kind=enrich.primary_type_kind,
597
- role=enrich.role,
598
- capabilities=list(enrich.capabilities),
599
- annotations_on_type=enrich.annotations_on_type,
600
- symbols=enrich.symbols,
601
- ontology_version=ONTOLOGY_VERSION,
602
- generated=generated,
603
- generated_by=generated_by,
604
- )
605
- )
606
-
607
-
608
- @coco.fn(memo=True)
609
- async def process_sql_file(
610
- file: localfs.File,
611
- table: lancedb.TableTarget[SqlLanceChunk],
612
- ) -> None:
613
- embedder = coco.use_context(EMBEDDER)
614
- project_root = coco.use_context(PROJECT_ROOT)
615
- ignore = coco.use_context(IGNORE)
616
- if ignore.is_ignored((project_root / file.file_path.path).resolve()):
617
- return
618
- try:
619
- content = await file.read_text()
620
- except UnicodeDecodeError:
621
- return
622
- if not content.strip():
623
- return
624
-
625
- _tick_vectors_done()
626
-
627
- language = "sql"
628
- cs, mn, ov = SQL_CHUNK
629
- chunks = splitter.split(
630
- content,
631
- cs,
632
- min_chunk_size=mn,
633
- chunk_overlap=ov,
634
- language=language,
635
- )
636
- rel = file.file_path.path.as_posix()
637
-
638
- # (vectors perf lever #1) embed chunks concurrently → batched encode.
639
- embeddings = await asyncio.gather(*(embedder.embed(ch.text) for ch in chunks))
640
-
641
- for ch, emb in zip(chunks, embeddings):
642
- rs, re = chunk_key_range(ch)
643
- table.declare_row(
644
- row=SqlLanceChunk(
645
- id=str(uuid.uuid4()),
646
- filename=rel,
647
- text=ch.text,
648
- range_start=rs,
649
- range_end=re,
650
- start=position_to_json(ch.start),
651
- end=position_to_json(ch.end),
652
- embedding=emb,
653
- )
654
- )
655
-
656
-
657
- @coco.fn(memo=True)
658
- async def process_yaml_file(
659
- file: localfs.File,
660
- table: lancedb.TableTarget[YamlLanceChunk],
661
- ) -> None:
662
- embedder = coco.use_context(EMBEDDER)
663
- project_root = coco.use_context(PROJECT_ROOT)
664
- ignore = coco.use_context(IGNORE)
665
- if ignore.is_ignored((project_root / file.file_path.path).resolve()):
666
- return
667
- try:
668
- content = await file.read_text()
669
- except UnicodeDecodeError:
670
- return
671
- if not content.strip():
672
- return
673
-
674
- _tick_vectors_done()
675
-
676
- ext = file.file_path.path.suffix.lower()
677
- language = "yaml" if ext in (".yml", ".yaml") else "text"
678
- cs, mn, ov = YAML_CHUNK
679
- chunks = splitter.split(
680
- content,
681
- cs,
682
- min_chunk_size=mn,
683
- chunk_overlap=ov,
684
- language=language,
685
- )
686
- rel = file.file_path.path.as_posix()
687
-
688
- # (vectors perf lever #1) embed chunks concurrently → batched encode.
689
- embeddings = await asyncio.gather(*(embedder.embed(ch.text) for ch in chunks))
690
-
691
- for ch, emb in zip(chunks, embeddings):
692
- rs, re = chunk_key_range(ch)
693
- table.declare_row(
694
- row=YamlLanceChunk(
695
- id=str(uuid.uuid4()),
696
- filename=rel,
697
- text=ch.text,
698
- range_start=rs,
699
- range_end=re,
700
- start=position_to_json(ch.start),
701
- end=position_to_json(ch.end),
702
- embedding=emb,
703
- )
704
- )
705
-
706
-
707
- async def _drain_files_concurrently(
708
- files: Any, process_fn: Any, table: Any, sem: asyncio.Semaphore
709
- ) -> None:
710
- """Run ``process_fn(file, table)`` over every file with bounded concurrency.
711
-
712
- Replaces the serial ``async for … await process_*_file`` loop so the
713
- embedder's batching layer sees many files' chunks in flight at once (see
714
- ``_FILE_CONCURRENCY``). Materializes the async iterable up front — file
715
- handles are lightweight and cocoindex already realized the collection when
716
- the walker mounted, so this is not a second walk. An empty collection is a
717
- no-op (e.g. SQL/YAML tables on a repo with none).
718
- """
719
- items = [f async for _, f in files.items()]
720
- if not items:
721
- return
722
-
723
- async def _one(_file: Any) -> None:
724
- async with sem:
725
- await process_fn(_file, table)
726
-
727
- await asyncio.gather(*(_one(f) for f in items))
728
-
729
-
730
- @coco.fn
731
- async def app_main() -> None:
732
- java_schema = await lancedb.TableSchema.from_class(
733
- JavaLanceChunk,
734
- primary_key=["id"],
735
- )
736
- java_table = await lancedb.mount_table_target(
737
- LANCE_DB,
738
- LANCE_TABLE_NAMES[0],
739
- java_schema,
740
- )
741
-
742
- sql_schema = await lancedb.TableSchema.from_class(
743
- SqlLanceChunk,
744
- primary_key=["id"],
745
- )
746
- sql_table = await lancedb.mount_table_target(
747
- LANCE_DB,
748
- LANCE_TABLE_NAMES[1],
749
- sql_schema,
750
- )
751
-
752
- yaml_schema = await lancedb.TableSchema.from_class(
753
- YamlLanceChunk,
754
- primary_key=["id"],
755
- )
756
- yaml_table = await lancedb.mount_table_target(
757
- LANCE_DB,
758
- LANCE_TABLE_NAMES[2],
759
- yaml_schema,
760
- )
761
-
762
- project_root = coco.use_context(PROJECT_ROOT)
763
- # Warm per-project enrichment caches ONCE on the event-loop thread, BEFORE
764
- # coco.mount_each fans files into worker threads. collect_annotation_meta_chain
765
- # and load_brownfield_overrides are lru_cached per (resolved) project root;
766
- # without warming, the first wave of concurrent process_java_file worker
767
- # threads each cold-miss and redundantly walk+parse the ENTIRE project (a
768
- # thundering herd that would offset the embedding-batching win on large
769
- # repos — perf lever #2 made enrich concurrent). With warming, every worker
770
- # hits a populated cache (lru_cache reads are thread-safe). Key derivation
771
- # mirrors enrich_chunk exactly so the warmed entries are the ones workers hit.
772
- try:
773
- load_brownfield_overrides(project_root)
774
- try:
775
- prs = str(Path(project_root).resolve())
776
- except OSError:
777
- prs = str(project_root)
778
- collect_annotation_meta_chain(prs)
779
- except Exception:
780
- # Warm-up must never break indexing — a failure just means workers
781
- # cold-miss lazily (the pre-warming behavior). Swallow and continue.
782
- pass
783
- _ignore = LayeredIgnore(project_root)
784
- _walk_excludes = _ignore.cocoindex_excluded_patterns()
785
- # Emit ONE approximate total so the parent's renderer can show a determinate
786
- # bar (clamps to 100% on the terminal vectors event the parent emits on
787
- # cocoindex exit). Approximate — ignored / empty files over-state it; see
788
- # ``_approximate_vectors_total``. ``--full-reprocess`` only: on incremental
789
- # catch-up the @coco.fn(memo=True) cache skips unchanged files, so no total
790
- # is knowable up front → the parent renders indeterminate from the absence.
791
- try:
792
- total = _approximate_vectors_total(project_root)
793
- if total > 0:
794
- _emit_vectors_progress(total=total, status="running")
795
- except Exception:
796
- # The pre-walk must never break indexing — a failure here just means
797
- # the parent falls back to indeterminate. Swallow and continue.
798
- pass
799
- java_files = localfs.walk_dir(
800
- PROJECT_ROOT,
801
- recursive=True,
802
- path_matcher=PatternFilePathMatcher(
803
- included_patterns=["**/*.java"],
804
- excluded_patterns=_walk_excludes,
805
- ),
806
- )
807
- kotlin_files = (
808
- localfs.walk_dir(
809
- PROJECT_ROOT,
810
- recursive=True,
811
- path_matcher=PatternFilePathMatcher(
812
- included_patterns=["**/*.kt"],
813
- excluded_patterns=_walk_excludes,
814
- ),
815
- )
816
- if _KOTLIN_REGISTERED
817
- else None
818
- )
819
- sql_files = localfs.walk_dir(
820
- PROJECT_ROOT,
821
- recursive=True,
822
- path_matcher=PatternFilePathMatcher(
823
- included_patterns=["**/src/main/resources/db/migration/*.sql"],
824
- excluded_patterns=_walk_excludes,
825
- ),
826
- )
827
- yaml_files = localfs.walk_dir(
828
- PROJECT_ROOT,
829
- recursive=True,
830
- path_matcher=PatternFilePathMatcher(
831
- included_patterns=[
832
- "**/src/main/resources/application*.yml",
833
- "**/src/main/resources/application*.yaml",
834
- ],
835
- excluded_patterns=_walk_excludes,
836
- ),
837
- )
838
-
839
- # PERF: declare all rows in ONE component (app_main) instead of one
840
- # component per file via coco.mount_each. cocoindex flushes target writes
841
- # once per processing component, and all declare_row calls inside a
842
- # component batch into a single Lance merge_insert (see _RowHandler.
843
- # _apply_actions). mount_each created one component PER FILE → ~1167
844
- # merge_insert transactions (one fragment + manifest commit each) → ~91s
845
- # of kernel I/O on a 1167-file repo. The single-component loop collapses
846
- # that to ONE merge_insert per table. cocoindex does not yet batch across
847
- # mount_each components natively (open issue cocoindex#2219), so the loop
848
- # is the supported workaround. process_*_file stay @coco.fn(memo=True), so
849
- # unchanged files still skip re-embedding on incremental; _RowHandler.
850
- # reconcile skips rows whose fingerprint is unchanged → increment carries
851
- # only changed rows in its single merge_insert.
852
- #
853
- # PERF (concurrency): drain files with a bounded semaphore instead of a
854
- # serial ``async for … await``. See ``_FILE_CONCURRENCY`` — this is what
855
- # lets the embedder's batching layer fill real batches (embedding dominates
856
- # init cost, and serial files starve it). One shared semaphore bounds total
857
- # in-flight work; tables are drained in order (java dominates, sql/yaml are
858
- # usually near-empty).
859
- _sem = asyncio.Semaphore(_FILE_CONCURRENCY)
860
- await _drain_files_concurrently(java_files, process_java_file, java_table, _sem)
861
- # Kotlin drains into the SAME ``java_table`` (JavaLanceChunk) — the chunk
862
- # schema is language-agnostic and the ``language`` column distinguishes rows.
863
- # Gated on ``_KOTLIN_REGISTERED`` (registry-derived): on a grammar-absent
864
- # install ``.kt`` has no backend, so the matcher + drain are skipped entirely
865
- # — no wasted read/chunk/embed, and no crash in ``_parse_and_enrich_java``
866
- # (``backend_for(.kt)`` returns ``None`` → ``classify_java_file`` would
867
- # dereference ``ast.all_types`` on ``None``).
868
- if _KOTLIN_REGISTERED:
869
- await _drain_files_concurrently(
870
- kotlin_files, process_kotlin_file, java_table, _sem
871
- )
872
- await _drain_files_concurrently(sql_files, process_sql_file, sql_table, _sem)
873
- await _drain_files_concurrently(yaml_files, process_yaml_file, yaml_table, _sem)
874
-
875
-
876
- app = coco.App(
877
- coco.AppConfig(name="JavaCodeIndexLance"),
878
- app_main,
879
- )