java-codebase-rag 0.11.2__py3-none-any.whl → 0.12.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. java_codebase_rag-0.12.1.dist-info/METADATA +35 -0
  2. java_codebase_rag-0.12.1.dist-info/RECORD +4 -0
  3. {java_codebase_rag-0.11.2.dist-info → java_codebase_rag-0.12.1.dist-info}/WHEEL +1 -1
  4. java_codebase_rag/_fdlimit.py +0 -56
  5. java_codebase_rag/_stdio.py +0 -32
  6. java_codebase_rag/_version.py +0 -35
  7. java_codebase_rag/absence/__init__.py +0 -0
  8. java_codebase_rag/absence/absence_diagnosis.py +0 -700
  9. java_codebase_rag/absence/absence_types.py +0 -124
  10. java_codebase_rag/absence/absence_vocab.py +0 -460
  11. java_codebase_rag/analysis/__init__.py +0 -0
  12. java_codebase_rag/analysis/pr_analysis.py +0 -563
  13. java_codebase_rag/analysis/resolve_service.py +0 -740
  14. java_codebase_rag/ast/__init__.py +0 -0
  15. java_codebase_rag/ast/ast_java.py +0 -2825
  16. java_codebase_rag/ast/brownfield_events.py +0 -58
  17. java_codebase_rag/ast/chunk_heuristics.py +0 -62
  18. java_codebase_rag/cli.py +0 -1215
  19. java_codebase_rag/cli_format.py +0 -85
  20. java_codebase_rag/cli_progress.py +0 -94
  21. java_codebase_rag/config.py +0 -833
  22. java_codebase_rag/eval/__init__.py +0 -1
  23. java_codebase_rag/eval/ground_truth.py +0 -100
  24. java_codebase_rag/eval/metrics.py +0 -107
  25. java_codebase_rag/eval/runner.py +0 -556
  26. java_codebase_rag/graph/__init__.py +0 -0
  27. java_codebase_rag/graph/build_ast_graph.py +0 -4471
  28. java_codebase_rag/graph/graph_enrich.py +0 -1937
  29. java_codebase_rag/graph/graph_types.py +0 -224
  30. java_codebase_rag/graph/java_ontology.py +0 -465
  31. java_codebase_rag/graph/ladybug_queries.py +0 -2213
  32. java_codebase_rag/graph/path_filtering.py +0 -477
  33. java_codebase_rag/index/__init__.py +0 -0
  34. java_codebase_rag/index/java_index_flow_lancedb.py +0 -734
  35. java_codebase_rag/index/java_index_v1_common.py +0 -33
  36. java_codebase_rag/install_data/__init__.py +0 -0
  37. java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -108
  38. java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
  39. java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
  40. java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
  41. java_codebase_rag/installer.py +0 -2188
  42. java_codebase_rag/jrag.py +0 -4531
  43. java_codebase_rag/jrag_envelope.py +0 -1107
  44. java_codebase_rag/jrag_hints.py +0 -204
  45. java_codebase_rag/jrag_render.py +0 -926
  46. java_codebase_rag/lance_optimize.py +0 -264
  47. java_codebase_rag/mcp/__init__.py +0 -0
  48. java_codebase_rag/mcp/mcp_hints.py +0 -932
  49. java_codebase_rag/mcp/mcp_v2.py +0 -1916
  50. java_codebase_rag/mcp/server.py +0 -884
  51. java_codebase_rag/pipeline.py +0 -531
  52. java_codebase_rag/progress.py +0 -570
  53. java_codebase_rag/read_payloads.py +0 -781
  54. java_codebase_rag/search/__init__.py +0 -0
  55. java_codebase_rag/search/index_common.py +0 -10
  56. java_codebase_rag/search/search_lancedb.py +0 -1296
  57. java_codebase_rag/search/search_lexical.py +0 -449
  58. java_codebase_rag/search/search_scoring.py +0 -523
  59. java_codebase_rag/watch/__init__.py +0 -0
  60. java_codebase_rag/watch/client.py +0 -230
  61. java_codebase_rag/watch/daemon.py +0 -396
  62. java_codebase_rag/watch/lock.py +0 -201
  63. java_codebase_rag/watch/paths.py +0 -76
  64. java_codebase_rag/watch/protocol.py +0 -122
  65. java_codebase_rag/watch/server.py +0 -273
  66. java_codebase_rag/watch/warm.py +0 -105
  67. java_codebase_rag/watch/watcher.py +0 -370
  68. java_codebase_rag-0.11.2.dist-info/METADATA +0 -331
  69. java_codebase_rag-0.11.2.dist-info/RECORD +0 -71
  70. java_codebase_rag-0.11.2.dist-info/entry_points.txt +0 -4
  71. java_codebase_rag-0.11.2.dist-info/licenses/LICENSE +0 -21
  72. java_codebase_rag-0.11.2.dist-info/top_level.txt +0 -1
  73. /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.1.dist-info/top_level.txt +0 -0
@@ -1,734 +0,0 @@
1
- """
2
- CocoIndex 1.0 app: index Java, Flyway SQL, and YAML into LanceDB.
3
-
4
- LanceDB requires a single primary key per table; each chunk gets a UUID `id`.
5
-
6
- Environment:
7
- JAVA_CODEBASE_RAG_INDEX_DIR — Lance tables + LadybugDB + cocoindex state (default: ./.java-codebase-rag)
8
- JAVA_CODEBASE_RAG_SOURCE_ROOT — Java repo root for indexing (optional; else cocoindex cwd)
9
- SBERT_MODEL / SBERT_DEVICE — embedding (optional; YAML also supported via java-codebase-rag CLI)
10
-
11
- Dependencies:
12
- pip install "cocoindex[lancedb]" sentence-transformers
13
-
14
- Usage:
15
- cocoindex update java_index_flow_lancedb.py:JavaCodeIndexLance --full-reprocess
16
- """
17
- from __future__ import annotations
18
-
19
- import asyncio
20
- import inspect
21
- import os
22
- import sys
23
- import threading
24
- import uuid
25
- from collections.abc import AsyncIterator
26
- from contextlib import asynccontextmanager
27
- from dataclasses import dataclass
28
- from fnmatch import fnmatch
29
- from pathlib import Path
30
- from typing import Annotated, Any
31
-
32
- import cocoindex as coco
33
- import numpy as np
34
- import numpy.typing as npt
35
- import pyarrow as pa
36
- from cocoindex.connectors import lancedb, localfs
37
- from cocoindex.connectors.lancedb import LanceType
38
- from cocoindex.ops.sentence_transformers import SentenceTransformerEmbedder
39
- from cocoindex.ops.text import RecursiveSplitter, detect_code_language
40
- from cocoindex.resources.file import PatternFilePathMatcher
41
-
42
- from java_codebase_rag.config import resolved_sbert_model_for_process_env
43
- from java_codebase_rag.lance_optimize import LANCE_TABLE_NAMES
44
- from java_codebase_rag.index.java_index_v1_common import (
45
- JAVA_CHUNK,
46
- SBERT_MODEL,
47
- SQL_CHUNK,
48
- YAML_CHUNK,
49
- chunk_key_range,
50
- position_to_json,
51
- )
52
- from java_codebase_rag.graph.path_filtering import LayeredIgnore
53
- from java_codebase_rag.ast.ast_java import ONTOLOGY_VERSION, parse_java
54
- from java_codebase_rag.graph.graph_enrich import (
55
- classify_java_file,
56
- collect_annotation_meta_chain,
57
- enrich_chunk,
58
- load_brownfield_overrides,
59
- load_generated_detection,
60
- )
61
-
62
- # Older cocoindex (e.g. 1.0.0a43) uses ``tracked=False``; newer releases renamed
63
- # the flag to ``detect_change`` (default False) and reject ``tracked``.
64
- _ck_params = inspect.signature(coco.ContextKey.__init__).parameters
65
- if "detect_change" in _ck_params:
66
- PROJECT_ROOT = coco.ContextKey[Path]("java_lance_project_root")
67
- LANCE_DB = coco.ContextKey("java_lance_async_conn")
68
- EMBEDDER = coco.ContextKey[SentenceTransformerEmbedder]("java_lance_embedder")
69
- IGNORE = coco.ContextKey[LayeredIgnore]("java_lance_layered_ignore")
70
- elif "tracked" in _ck_params:
71
- PROJECT_ROOT = coco.ContextKey[Path]("java_lance_project_root", tracked=False)
72
- LANCE_DB = coco.ContextKey("java_lance_async_conn", tracked=False)
73
- EMBEDDER = coco.ContextKey[SentenceTransformerEmbedder](
74
- "java_lance_embedder", tracked=False
75
- )
76
- IGNORE = coco.ContextKey[LayeredIgnore](
77
- "java_lance_layered_ignore", tracked=False
78
- )
79
- else:
80
- PROJECT_ROOT = coco.ContextKey[Path]("java_lance_project_root")
81
- LANCE_DB = coco.ContextKey("java_lance_async_conn")
82
- EMBEDDER = coco.ContextKey[SentenceTransformerEmbedder]("java_lance_embedder")
83
- IGNORE = coco.ContextKey[LayeredIgnore]("java_lance_layered_ignore")
84
-
85
- splitter = RecursiveSplitter()
86
-
87
- # LanceDB table optimization: cocoindex >=1.0.15 runs ``table.optimize()``
88
- # INLINE during merge_insert commits, gated by stats (only when small fragments
89
- # accumulate — see _RowHandler._maybe_optimize / _evaluate_optimize). That
90
- # replaces the old 1.0.7 *background* asyncio optimize that raced concurrent
91
- # Deletes (lancedb#1504 commit conflicts) and which we used to disable via
92
- # ``num_transactions_before_optimize`` (kwarg removed in 1.0.16). Being inline,
93
- # it no longer races anything. ``lance_optimize.optimize_lance_tables`` still
94
- # runs a final serialized compaction post-flow. ``optimize()`` is pure
95
- # maintenance (compact/prune/index); upsert/delete correctness via merge_insert
96
- # does not depend on it.
97
-
98
-
99
- # --- Vectors-phase progress emission (JCIRAG_PROGRESS kind=vectors) -----------
100
- #
101
- # The flow runs in a CHILD cocoindex process; it prints structured progress to
102
- # its stderr and the parent (pipeline._popen_capturing_stderr /
103
- # cli_progress.accumulate_and_relay_subprocess_streams) parses it via
104
- # ProgressRelay and feeds the renderer. The flow CANNOT know when all files are
105
- # done (cocoindex offers no "all files done" hook in the flow), so it emits:
106
- # - ONE ``total=N status=running`` line from ``app_main`` (approximate
107
- # pre-walk: matcher includes + LayeredIgnore), and
108
- # - per-file ``done=k status=running`` ticks (throttled every ~25 files) from
109
- # ``process_*_file`` (shared atomic counter).
110
- # The PARENT emits the terminal ``status=done``/``failed`` vectors event on
111
- # cocoindex exit (drives clamp-on-completion + phase transition to Optimize).
112
-
113
- # Per-file tick cadence: bound stderr volume on huge trees without making the
114
- # bar feel stale. Every 25th file (and the modulo boundary is enough — the
115
- # parent clamps to total on the terminal event anyway).
116
- _VECTORS_TICK_EVERY = 25
117
-
118
- # Bounded concurrency for the per-file drain in app_main. cocoindex's embedder
119
- # is ``@coco.fn.as_async(batching=True, runner=GPU, max_batch_size=64)`` — but
120
- # its batching layer only coalesces calls that are in flight SIMULTANEOUSLY. A
121
- # serial ``async for … await`` loop keeps just one file's chunks (avg 1–3) in
122
- # flight, so real batches stay tiny and MPS idles between them (measured ~138
123
- # chunks/s vs the ~235 chunks/s ceiling at batch=64 for all-MiniLM-L6-v2).
124
- # Draining many files at once with a semaphore puts their chunks in flight
125
- # together → the embedder coalesces them into full batches → MPS climbs toward
126
- # the ceiling. Measured on Shopizer (1167 files / 3475 chunks): full init drops
127
- # from ~46.7s (serial) to ~36.0s (32) / ~34.3s (64), with identical row output.
128
- #
129
- # This stays inside ONE component, so the earlier mount_each→app_main win is
130
- # preserved: still exactly ONE merge_insert per table at commit. Memoization
131
- # (``@coco.fn(memo=True)``) and the lock-guarded tick counter are safe under
132
- # concurrency; ``parse_java`` uses a per-thread tree-sitter Parser (already
133
- # routed via ``asyncio.to_thread``) and ``splitter.split`` is synchronous so the
134
- # event loop cannot reenter it.
135
- #
136
- # Default 64 is sized to MATCH the embedder's hardcoded max_batch_size=64 (the
137
- # decorator above; not a constructor arg, so not raisable from the flow): ~64
138
- # files in flight reliably fills a 64-chunk batch and saturates MPS. Going higher
139
- # buys nothing — the batch is already capped — and lower underfills it. Memory
140
- # is NOT the limiting factor here: cocoindex buffers ALL staged rows until the
141
- # single final merge_insert regardless of concurrency, so peak RSS is set by
142
- # total chunk count (the commit buffer), not by how many files process at once.
143
- # Set to ``1`` for the old serial behavior; raise/lower only if you have also
144
- # changed the effective batch size or are constraining the commit buffer itself.
145
- _FILE_CONCURRENCY = max(
146
- 1,
147
- int(os.environ.get("JAVA_CODEBASE_RAG_FILE_CONCURRENCY", "64") or "64"),
148
- )
149
-
150
- # Thread-safe counter: cocoindex may call process_*_file concurrently
151
- # (mount_each parallelism is implementation-defined). A module-level lock guards
152
- # both the counter and the emission so two threads never interleave a tick.
153
- _vectors_done_lock = threading.Lock()
154
- _vectors_done_count = 0
155
-
156
-
157
- def _emit_vectors_progress(
158
- *,
159
- done: int | None = None,
160
- total: int | None = None,
161
- status: str = "running",
162
- elapsed_s: float | None = None,
163
- ) -> None:
164
- """Emit one ``JCIRAG_PROGRESS kind=vectors …`` line to stderr (flushed).
165
-
166
- Field order is fixed (kind, done, total, status, elapsed_s) so the parser
167
- and tests can pin substrings. Omitted fields are simply absent.
168
- """
169
- fields = ["kind=vectors"]
170
- if done is not None:
171
- fields.append(f"done={done}")
172
- if total is not None:
173
- fields.append(f"total={total}")
174
- fields.append(f"status={status}")
175
- if elapsed_s is not None:
176
- fields.append(f"elapsed_s={elapsed_s:.2f}")
177
- print("JCIRAG_PROGRESS " + " ".join(fields), file=sys.stderr, flush=True)
178
-
179
-
180
- def _tick_vectors_done() -> None:
181
- """Increment the shared per-file counter and emit a throttled ``done=k`` tick.
182
-
183
- Called once per successfully-processed file (after the ignore / empty
184
- early-returns). The tick is emitted every ``_VECTORS_TICK_EVERY`` files so
185
- stderr volume stays bounded on huge trees; the parent clamps to total on
186
- the terminal event, so the exact tick cadence is not load-bearing.
187
- """
188
- global _vectors_done_count
189
- with _vectors_done_lock:
190
- _vectors_done_count += 1
191
- n = _vectors_done_count
192
- if n % _VECTORS_TICK_EVERY != 0:
193
- return
194
- # Emit under the lock: the docstring above promises the lock guards both
195
- # the counter AND the emission, so two concurrent ticks can't emit their
196
- # ``done=N`` lines out of order. Contention is negligible (fires every
197
- # ~25 files).
198
- _emit_vectors_progress(done=n, status="running")
199
-
200
-
201
- def _approximate_vectors_total(project_root: Path) -> int:
202
- """Reproduce the matchers' include globs + LayeredIgnore for an approximate total.
203
-
204
- The flow applies two filtering layers: (1) ``PatternFilePathMatcher``
205
- excludes at walk time via ``LayeredIgnore.cocoindex_excluded_patterns()``,
206
- then (2) ``LayeredIgnore.is_ignored()`` plus an early-return for empty /
207
- undecodable files inside each ``process_*_file``. Files that early-return
208
- never tick, so this pre-walk OVERSTATES the total by the ignored / empty
209
- count. The parent clamps the bar to 100% on the terminal ``status=done``
210
- event, so the over-count cannot stall the bar.
211
-
212
- Mirrors the three ``localfs.walk_dir`` matchers in ``app_main``:
213
- - ``**/*.java``
214
- - ``**/src/main/resources/db/migration/*.sql``
215
- - ``**/src/main/resources/application*.yml`` and ``.yaml``
216
- """
217
- ignore = LayeredIgnore(project_root)
218
- excluded = ignore.cocoindex_excluded_patterns()
219
-
220
- def _excluded(rel_posix: str) -> bool:
221
- return any(fnmatch(rel_posix, pat) for pat in excluded)
222
-
223
- total = 0
224
- for dirpath, dirnames, filenames in os.walk(project_root):
225
- # Prune the same universal nuisance dirs as iter_java_source_files /
226
- # cocoindex walk. (build-output pruning is matcher-dependent in the
227
- # real walk; for an APPROXIMATE total this cheap prune is sufficient
228
- # — the clamp absorbs any residual divergence.)
229
- dirnames[:] = [
230
- d for d in dirnames if d not in (".git", ".hg", ".svn", "node_modules", ".venv", "venv")
231
- ]
232
- for fn in filenames:
233
- full = Path(dirpath) / fn
234
- try:
235
- rel = full.resolve().relative_to(project_root).as_posix()
236
- except ValueError:
237
- continue
238
- if _excluded(rel):
239
- continue
240
- # Java: **/*.java
241
- if fn.endswith(".java"):
242
- if not ignore.is_ignored(full):
243
- total += 1
244
- continue
245
- # SQL: **/src/main/resources/db/migration/*.sql
246
- if fn.endswith(".sql") and "/db/migration/" in rel:
247
- if not ignore.is_ignored(full):
248
- total += 1
249
- continue
250
- # YAML: **/src/main/resources/application*.yml / .yaml
251
- # NOTE: ``fn`` is the bare filename (e.g. ``application-cloud.yml``), so
252
- # the prefix predicate must be ``fn.startswith("application")`` —
253
- # ``"/application" in fn`` was always False (no leading slash in a bare
254
- # name) and under-counted every application YAML, driving the pre-walk
255
- # total below the actual done count. The ``rel``-based
256
- # ``"/src/main/resources/"`` gate stays (full path component).
257
- if fn.endswith((".yml", ".yaml")) and fn.startswith("application") and "/src/main/resources/" in rel:
258
- if not ignore.is_ignored(full):
259
- total += 1
260
- return total
261
-
262
-
263
- @dataclass
264
- class JavaLanceChunk:
265
- id: str
266
- filename: str
267
- language: str
268
- text: str
269
- range_start: int
270
- range_end: int
271
- start: dict[str, Any]
272
- end: dict[str, Any]
273
- embedding: Annotated[npt.NDArray[np.float32], EMBEDDER]
274
- package: str
275
- module: str
276
- microservice: str
277
- primary_type_fqn: str
278
- primary_type_kind: str
279
- role: str
280
- # Native PyArrow lists: without the LanceType override CocoIndex would JSON-encode
281
- # `list[str]` into a STRING column, which caller code then iterates character-by-character.
282
- capabilities: Annotated[list[str], LanceType(pa.list_(pa.string()))]
283
- annotations_on_type: Annotated[list[str], LanceType(pa.list_(pa.string()))]
284
- symbols: Annotated[list[str], LanceType(pa.list_(pa.string()))]
285
- ontology_version: int
286
- # Generated source detection: populated per-file, not per-chunk
287
- generated: bool
288
- generated_by: str | None
289
-
290
-
291
- @dataclass
292
- class SqlLanceChunk:
293
- id: str
294
- filename: str
295
- text: str
296
- range_start: int
297
- range_end: int
298
- start: dict[str, Any]
299
- end: dict[str, Any]
300
- embedding: Annotated[npt.NDArray[np.float32], EMBEDDER]
301
-
302
-
303
- @dataclass
304
- class YamlLanceChunk:
305
- id: str
306
- filename: str
307
- text: str
308
- range_start: int
309
- range_end: int
310
- start: dict[str, Any]
311
- end: dict[str, Any]
312
- embedding: Annotated[npt.NDArray[np.float32], EMBEDDER]
313
-
314
-
315
- @coco.lifespan
316
- async def coco_lifespan(builder: coco.EnvironmentBuilder) -> AsyncIterator[None]:
317
- idx_raw = os.environ.get("JAVA_CODEBASE_RAG_INDEX_DIR", "").strip()
318
- if idx_raw and not idx_raw.startswith(("s3://", "gs://", "az://")):
319
- index_dir = Path(idx_raw).expanduser().resolve()
320
- else:
321
- index_dir = (Path(".").resolve() / ".java-codebase-rag").resolve()
322
- index_dir.mkdir(parents=True, exist_ok=True)
323
- builder.settings.db_path = index_dir / "cocoindex.db"
324
-
325
- env_root = os.environ.get("JAVA_CODEBASE_RAG_SOURCE_ROOT", "").strip()
326
- if env_root:
327
- root = Path(env_root).expanduser().resolve()
328
- else:
329
- root = Path(".").resolve()
330
- builder.provide(PROJECT_ROOT, root)
331
-
332
- # Default to Apple Metal (MPS) when available: ~1.7x faster encode on
333
- # all-MiniLM-L6-v2 (measured), and the win grows with repo size since
334
- # embedding dominates on large trees. torch is already on the import path
335
- # here (sentence-transformers pulls it), so the availability check is free
336
- # in this child process — and it keeps the CLI parent (config.py) from ever
337
- # paying a torch import. Operators force CPU with SBERT_DEVICE=cpu.
338
- device = os.environ.get("SBERT_DEVICE") or None
339
- if device is None:
340
- try:
341
- import torch # noqa: WPS433 (local import: avoid parent-import cost)
342
- if torch.backends.mps.is_available():
343
- device = "mps"
344
- except Exception:
345
- pass
346
- embedder = SentenceTransformerEmbedder(
347
- resolved_sbert_model_for_process_env(SBERT_MODEL),
348
- device=device,
349
- trust_remote_code=True,
350
- )
351
- builder.provide(EMBEDDER, embedder)
352
- builder.provide(IGNORE, LayeredIgnore(root))
353
-
354
- uri = str(index_dir)
355
-
356
- @asynccontextmanager
357
- async def _lance_cm() -> AsyncIterator[Any]:
358
- conn = await lancedb.connect_async(uri)
359
- try:
360
- yield conn
361
- finally:
362
- conn.close()
363
-
364
- await builder.provide_async_with(LANCE_DB, _lance_cm())
365
- yield
366
-
367
-
368
- def _parse_and_enrich_java(
369
- content_bytes: bytes,
370
- chunks: list[Any],
371
- rel: str,
372
- project_root: Path,
373
- ) -> tuple[list[Any], Any]:
374
- """Parse one Java file and enrich every chunk, off the event loop.
375
-
376
- Returns a tuple of (enrichments, ast) where enrichments is a list of
377
- :class:`graph_enrich.ChunkEnrichment` aligned 1:1 with ``chunks``, and ast
378
- is the parsed :class:`JavaFileAst`. Intended to run via ``asyncio.to_thread``
379
- from ``process_java_file`` (vectors perf lever #2): while the worker thread
380
- parses + enriches, the event loop is free to drive other files and keep the
381
- embedder's batching queue fed.
382
-
383
- Thread-safety: ``parse_java`` uses a per-thread tree-sitter ``Parser``
384
- (see ``ast_java._parser``), so it is safe to call concurrently from these
385
- worker threads — including the transitive ``parse_java`` that ``enrich_chunk``
386
- triggers via ``collect_annotation_meta_chain`` → ``_collect_annotation_decl_index``.
387
- ``enrich_chunk`` is otherwise pure-Python over the now-immutable AST; its
388
- ``lru_cache`` reads are thread-safe under the GIL.
389
- """
390
- ast = parse_java(content_bytes)
391
- enrichments = [
392
- enrich_chunk(
393
- ast,
394
- chunk_start_byte=ch.start.byte_offset,
395
- chunk_end_byte=ch.end.byte_offset,
396
- file_path=rel,
397
- project_root=project_root,
398
- )
399
- for ch in chunks
400
- ]
401
- return enrichments, ast
402
-
403
-
404
- @coco.fn(memo=True)
405
- async def process_java_file(
406
- file: localfs.File,
407
- table: lancedb.TableTarget[JavaLanceChunk],
408
- ) -> None:
409
- embedder = coco.use_context(EMBEDDER)
410
- project_root = coco.use_context(PROJECT_ROOT)
411
- ignore = coco.use_context(IGNORE)
412
- if ignore.is_ignored((project_root / file.file_path.path).resolve()):
413
- return
414
- try:
415
- content = await file.read_text()
416
- except UnicodeDecodeError:
417
- return
418
- if not content.strip():
419
- return
420
-
421
- _tick_vectors_done()
422
-
423
- language = detect_code_language(filename=file.file_path.path.name) or "text"
424
- cs, mn, ov = JAVA_CHUNK
425
- # ``splitter.split`` stays inline: the module-level ``RecursiveSplitter``
426
- # shares one Rust object, so keeping split on the event loop preserves its
427
- # existing single-threaded access (no new cross-file concurrency hazard).
428
- chunks = splitter.split(
429
- content,
430
- cs,
431
- min_chunk_size=mn,
432
- chunk_overlap=ov,
433
- language=language,
434
- )
435
- rel = file.file_path.path.as_posix()
436
- content_bytes = content.encode("utf-8", errors="replace")
437
-
438
- # (vectors perf lever #2) parse + enrich off the event loop so the loop can
439
- # keep the embedder's batching queue fed while this file is being parsed.
440
- # parse_java is thread-safe (per-thread tree-sitter Parser in ast_java).
441
- enrichments, ast = await asyncio.to_thread(
442
- _parse_and_enrich_java, content_bytes, chunks, rel, project_root
443
- )
444
-
445
- # Compute generated source detection once per file (uses the AST and content_bytes)
446
- generated_config = load_generated_detection(project_root)
447
- generated, generated_by = classify_java_file(
448
- content_bytes, ast, config=generated_config, project_root=project_root
449
- )
450
-
451
- # (vectors perf lever #1) embed all chunks concurrently so the batched
452
- # embedder groups them into one ``model.encode(...)`` (max_batch_size=64)
453
- # instead of N serial batch-of-1 calls. Dominant win for ``increment``
454
- # (few changed files → little cross-file concurrency → otherwise no batching).
455
- embeddings = await asyncio.gather(*(embedder.embed(ch.text) for ch in chunks))
456
-
457
- for ch, enrich, emb in zip(chunks, enrichments, embeddings):
458
- rs, re = chunk_key_range(ch)
459
- table.declare_row(
460
- row=JavaLanceChunk(
461
- id=str(uuid.uuid4()),
462
- filename=rel,
463
- language=language,
464
- text=ch.text,
465
- range_start=rs,
466
- range_end=re,
467
- start=position_to_json(ch.start),
468
- end=position_to_json(ch.end),
469
- embedding=emb,
470
- package=enrich.package,
471
- module=enrich.module,
472
- microservice=enrich.microservice,
473
- primary_type_fqn=enrich.primary_type_fqn,
474
- primary_type_kind=enrich.primary_type_kind,
475
- role=enrich.role,
476
- capabilities=list(enrich.capabilities),
477
- annotations_on_type=enrich.annotations_on_type,
478
- symbols=enrich.symbols,
479
- ontology_version=ONTOLOGY_VERSION,
480
- generated=generated,
481
- generated_by=generated_by,
482
- )
483
- )
484
-
485
-
486
- @coco.fn(memo=True)
487
- async def process_sql_file(
488
- file: localfs.File,
489
- table: lancedb.TableTarget[SqlLanceChunk],
490
- ) -> None:
491
- embedder = coco.use_context(EMBEDDER)
492
- project_root = coco.use_context(PROJECT_ROOT)
493
- ignore = coco.use_context(IGNORE)
494
- if ignore.is_ignored((project_root / file.file_path.path).resolve()):
495
- return
496
- try:
497
- content = await file.read_text()
498
- except UnicodeDecodeError:
499
- return
500
- if not content.strip():
501
- return
502
-
503
- _tick_vectors_done()
504
-
505
- language = "sql"
506
- cs, mn, ov = SQL_CHUNK
507
- chunks = splitter.split(
508
- content,
509
- cs,
510
- min_chunk_size=mn,
511
- chunk_overlap=ov,
512
- language=language,
513
- )
514
- rel = file.file_path.path.as_posix()
515
-
516
- # (vectors perf lever #1) embed chunks concurrently → batched encode.
517
- embeddings = await asyncio.gather(*(embedder.embed(ch.text) for ch in chunks))
518
-
519
- for ch, emb in zip(chunks, embeddings):
520
- rs, re = chunk_key_range(ch)
521
- table.declare_row(
522
- row=SqlLanceChunk(
523
- id=str(uuid.uuid4()),
524
- filename=rel,
525
- text=ch.text,
526
- range_start=rs,
527
- range_end=re,
528
- start=position_to_json(ch.start),
529
- end=position_to_json(ch.end),
530
- embedding=emb,
531
- )
532
- )
533
-
534
-
535
- @coco.fn(memo=True)
536
- async def process_yaml_file(
537
- file: localfs.File,
538
- table: lancedb.TableTarget[YamlLanceChunk],
539
- ) -> None:
540
- embedder = coco.use_context(EMBEDDER)
541
- project_root = coco.use_context(PROJECT_ROOT)
542
- ignore = coco.use_context(IGNORE)
543
- if ignore.is_ignored((project_root / file.file_path.path).resolve()):
544
- return
545
- try:
546
- content = await file.read_text()
547
- except UnicodeDecodeError:
548
- return
549
- if not content.strip():
550
- return
551
-
552
- _tick_vectors_done()
553
-
554
- ext = file.file_path.path.suffix.lower()
555
- language = "yaml" if ext in (".yml", ".yaml") else "text"
556
- cs, mn, ov = YAML_CHUNK
557
- chunks = splitter.split(
558
- content,
559
- cs,
560
- min_chunk_size=mn,
561
- chunk_overlap=ov,
562
- language=language,
563
- )
564
- rel = file.file_path.path.as_posix()
565
-
566
- # (vectors perf lever #1) embed chunks concurrently → batched encode.
567
- embeddings = await asyncio.gather(*(embedder.embed(ch.text) for ch in chunks))
568
-
569
- for ch, emb in zip(chunks, embeddings):
570
- rs, re = chunk_key_range(ch)
571
- table.declare_row(
572
- row=YamlLanceChunk(
573
- id=str(uuid.uuid4()),
574
- filename=rel,
575
- text=ch.text,
576
- range_start=rs,
577
- range_end=re,
578
- start=position_to_json(ch.start),
579
- end=position_to_json(ch.end),
580
- embedding=emb,
581
- )
582
- )
583
-
584
-
585
- async def _drain_files_concurrently(
586
- files: Any, process_fn: Any, table: Any, sem: asyncio.Semaphore
587
- ) -> None:
588
- """Run ``process_fn(file, table)`` over every file with bounded concurrency.
589
-
590
- Replaces the serial ``async for … await process_*_file`` loop so the
591
- embedder's batching layer sees many files' chunks in flight at once (see
592
- ``_FILE_CONCURRENCY``). Materializes the async iterable up front — file
593
- handles are lightweight and cocoindex already realized the collection when
594
- the walker mounted, so this is not a second walk. An empty collection is a
595
- no-op (e.g. SQL/YAML tables on a repo with none).
596
- """
597
- items = [f async for _, f in files.items()]
598
- if not items:
599
- return
600
-
601
- async def _one(_file: Any) -> None:
602
- async with sem:
603
- await process_fn(_file, table)
604
-
605
- await asyncio.gather(*(_one(f) for f in items))
606
-
607
-
608
- @coco.fn
609
- async def app_main() -> None:
610
- java_schema = await lancedb.TableSchema.from_class(
611
- JavaLanceChunk,
612
- primary_key=["id"],
613
- )
614
- java_table = await lancedb.mount_table_target(
615
- LANCE_DB,
616
- LANCE_TABLE_NAMES[0],
617
- java_schema,
618
- )
619
-
620
- sql_schema = await lancedb.TableSchema.from_class(
621
- SqlLanceChunk,
622
- primary_key=["id"],
623
- )
624
- sql_table = await lancedb.mount_table_target(
625
- LANCE_DB,
626
- LANCE_TABLE_NAMES[1],
627
- sql_schema,
628
- )
629
-
630
- yaml_schema = await lancedb.TableSchema.from_class(
631
- YamlLanceChunk,
632
- primary_key=["id"],
633
- )
634
- yaml_table = await lancedb.mount_table_target(
635
- LANCE_DB,
636
- LANCE_TABLE_NAMES[2],
637
- yaml_schema,
638
- )
639
-
640
- project_root = coco.use_context(PROJECT_ROOT)
641
- # Warm per-project enrichment caches ONCE on the event-loop thread, BEFORE
642
- # coco.mount_each fans files into worker threads. collect_annotation_meta_chain
643
- # and load_brownfield_overrides are lru_cached per (resolved) project root;
644
- # without warming, the first wave of concurrent process_java_file worker
645
- # threads each cold-miss and redundantly walk+parse the ENTIRE project (a
646
- # thundering herd that would offset the embedding-batching win on large
647
- # repos — perf lever #2 made enrich concurrent). With warming, every worker
648
- # hits a populated cache (lru_cache reads are thread-safe). Key derivation
649
- # mirrors enrich_chunk exactly so the warmed entries are the ones workers hit.
650
- try:
651
- load_brownfield_overrides(project_root)
652
- try:
653
- prs = str(Path(project_root).resolve())
654
- except OSError:
655
- prs = str(project_root)
656
- collect_annotation_meta_chain(prs)
657
- except Exception:
658
- # Warm-up must never break indexing — a failure just means workers
659
- # cold-miss lazily (the pre-warming behavior). Swallow and continue.
660
- pass
661
- _ignore = LayeredIgnore(project_root)
662
- _walk_excludes = _ignore.cocoindex_excluded_patterns()
663
- # Emit ONE approximate total so the parent's renderer can show a determinate
664
- # bar (clamps to 100% on the terminal vectors event the parent emits on
665
- # cocoindex exit). Approximate — ignored / empty files over-state it; see
666
- # ``_approximate_vectors_total``. ``--full-reprocess`` only: on incremental
667
- # catch-up the @coco.fn(memo=True) cache skips unchanged files, so no total
668
- # is knowable up front → the parent renders indeterminate from the absence.
669
- try:
670
- total = _approximate_vectors_total(project_root)
671
- if total > 0:
672
- _emit_vectors_progress(total=total, status="running")
673
- except Exception:
674
- # The pre-walk must never break indexing — a failure here just means
675
- # the parent falls back to indeterminate. Swallow and continue.
676
- pass
677
- java_files = localfs.walk_dir(
678
- PROJECT_ROOT,
679
- recursive=True,
680
- path_matcher=PatternFilePathMatcher(
681
- included_patterns=["**/*.java"],
682
- excluded_patterns=_walk_excludes,
683
- ),
684
- )
685
- sql_files = localfs.walk_dir(
686
- PROJECT_ROOT,
687
- recursive=True,
688
- path_matcher=PatternFilePathMatcher(
689
- included_patterns=["**/src/main/resources/db/migration/*.sql"],
690
- excluded_patterns=_walk_excludes,
691
- ),
692
- )
693
- yaml_files = localfs.walk_dir(
694
- PROJECT_ROOT,
695
- recursive=True,
696
- path_matcher=PatternFilePathMatcher(
697
- included_patterns=[
698
- "**/src/main/resources/application*.yml",
699
- "**/src/main/resources/application*.yaml",
700
- ],
701
- excluded_patterns=_walk_excludes,
702
- ),
703
- )
704
-
705
- # PERF: declare all rows in ONE component (app_main) instead of one
706
- # component per file via coco.mount_each. cocoindex flushes target writes
707
- # once per processing component, and all declare_row calls inside a
708
- # component batch into a single Lance merge_insert (see _RowHandler.
709
- # _apply_actions). mount_each created one component PER FILE → ~1167
710
- # merge_insert transactions (one fragment + manifest commit each) → ~91s
711
- # of kernel I/O on a 1167-file repo. The single-component loop collapses
712
- # that to ONE merge_insert per table. cocoindex does not yet batch across
713
- # mount_each components natively (open issue cocoindex#2219), so the loop
714
- # is the supported workaround. process_*_file stay @coco.fn(memo=True), so
715
- # unchanged files still skip re-embedding on incremental; _RowHandler.
716
- # reconcile skips rows whose fingerprint is unchanged → increment carries
717
- # only changed rows in its single merge_insert.
718
- #
719
- # PERF (concurrency): drain files with a bounded semaphore instead of a
720
- # serial ``async for … await``. See ``_FILE_CONCURRENCY`` — this is what
721
- # lets the embedder's batching layer fill real batches (embedding dominates
722
- # init cost, and serial files starve it). One shared semaphore bounds total
723
- # in-flight work; tables are drained in order (java dominates, sql/yaml are
724
- # usually near-empty).
725
- _sem = asyncio.Semaphore(_FILE_CONCURRENCY)
726
- await _drain_files_concurrently(java_files, process_java_file, java_table, _sem)
727
- await _drain_files_concurrently(sql_files, process_sql_file, sql_table, _sem)
728
- await _drain_files_concurrently(yaml_files, process_yaml_file, yaml_table, _sem)
729
-
730
-
731
- app = coco.App(
732
- coco.AppConfig(name="JavaCodeIndexLance"),
733
- app_main,
734
- )