java-codebase-rag 0.11.2__py3-none-any.whl → 0.12.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. java_codebase_rag/_deprecation.py +103 -0
  2. java_codebase_rag/_version.py +2 -2
  3. java_codebase_rag/ast/ast_java.py +22 -0
  4. java_codebase_rag/ast/ast_kotlin.py +1794 -0
  5. java_codebase_rag/ast/chunk_heuristics.py +26 -5
  6. java_codebase_rag/ast/language.py +117 -0
  7. java_codebase_rag/cli.py +17 -17
  8. java_codebase_rag/cli_dispatch.py +251 -0
  9. java_codebase_rag/config.py +8 -8
  10. java_codebase_rag/eval/runner.py +3 -3
  11. java_codebase_rag/graph/build_ast_graph.py +130 -8
  12. java_codebase_rag/graph/graph_enrich.py +8 -5
  13. java_codebase_rag/graph/ladybug_queries.py +1 -1
  14. java_codebase_rag/graph/path_filtering.py +39 -7
  15. java_codebase_rag/index/java_index_flow_lancedb.py +160 -15
  16. java_codebase_rag/install_data/agents/explorer-rag-cli.md +6 -4
  17. java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +4 -4
  18. java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +4 -4
  19. java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +5 -5
  20. java_codebase_rag/installer.py +15 -15
  21. java_codebase_rag/jrag.py +25 -11
  22. java_codebase_rag/lance_optimize.py +7 -7
  23. java_codebase_rag/mcp/mcp_v2.py +2 -2
  24. java_codebase_rag/mcp/server.py +6 -4
  25. java_codebase_rag/pipeline.py +4 -4
  26. java_codebase_rag/progress.py +1 -1
  27. java_codebase_rag/search/search_lexical.py +1 -1
  28. java_codebase_rag/search/search_scoring.py +19 -5
  29. java_codebase_rag/watch/lock.py +1 -1
  30. java_codebase_rag/watch/watcher.py +45 -21
  31. {java_codebase_rag-0.11.2.dist-info → java_codebase_rag-0.12.0.dist-info}/METADATA +31 -22
  32. {java_codebase_rag-0.11.2.dist-info → java_codebase_rag-0.12.0.dist-info}/RECORD +36 -32
  33. java_codebase_rag-0.12.0.dist-info/entry_points.txt +5 -0
  34. java_codebase_rag-0.11.2.dist-info/entry_points.txt +0 -4
  35. {java_codebase_rag-0.11.2.dist-info → java_codebase_rag-0.12.0.dist-info}/WHEEL +0 -0
  36. {java_codebase_rag-0.11.2.dist-info → java_codebase_rag-0.12.0.dist-info}/licenses/LICENSE +0 -0
  37. {java_codebase_rag-0.11.2.dist-info → java_codebase_rag-0.12.0.dist-info}/top_level.txt +0 -0
@@ -50,8 +50,24 @@ from java_codebase_rag.ast.ast_java import (
50
50
  TypeDecl,
51
51
  injection_annotation_names,
52
52
  lombok_required_args_annotations,
53
- parse_java,
54
53
  )
54
+ from java_codebase_rag.ast.language import backend_for
55
+
56
+ # Kotlin multifile-facade merge (Task 9). Conditionally imported: on installs
57
+ # without the ``tree-sitter-kotlin`` grammar wheel (Intel-Mac graph-only image),
58
+ # ``ast_kotlin`` raises ImportError at module load. ``merge_multifile_facades``
59
+ # is a no-op on Java ASTs (they have no facades), so absent grammar → merge is
60
+ # skipped and Java indexing is byte-identical. Used in pass1 AFTER parsing all
61
+ # files but BEFORE ``_register_type`` (unmerged ``@file:JvmMultifileClass``
62
+ # facades share one FQN → ``tables.types[fqn] = entry`` would overwrite and one
63
+ # file's top-level functions would silently vanish from resolution).
64
+ try: # pragma: no cover - branch depends on whether the wheel is installed
65
+ from java_codebase_rag.ast.ast_kotlin import (
66
+ merge_multifile_facades as _merge_multifile_facades,
67
+ )
68
+ except ImportError: # grammar wheel absent — no .kt files to merge.
69
+ _merge_multifile_facades = None # type: ignore[assignment]
70
+
55
71
  from java_codebase_rag.graph.graph_enrich import (
56
72
  _load_config_cross_service_resolution,
57
73
  classify_java_file,
@@ -67,7 +83,7 @@ from java_codebase_rag.graph.graph_enrich import (
67
83
  resolve_routes_for_method,
68
84
  symbol_id,
69
85
  )
70
- from java_codebase_rag.graph.path_filtering import LayeredIgnore, iter_java_source_files
86
+ from java_codebase_rag.graph.path_filtering import LayeredIgnore, iter_source_files
71
87
  from java_codebase_rag.search.search_scoring import SYMBOL_FTS_INDEX as _SYMBOL_FTS_INDEX, _split_identifier
72
88
  from java_codebase_rag.graph.java_ontology import (
73
89
  CLIENT_KIND_FEIGN_METHOD,
@@ -184,6 +200,60 @@ _JAVA_LANG_SIMPLE = frozenset({
184
200
  })
185
201
 
186
202
 
203
+ # Kotlin default-import simple names → deterministic stdlib FQN (Task 13).
204
+ # Kotlin implicitly imports kotlin.*, kotlin.collections.*, kotlin.sequences.*,
205
+ # kotlin.ranges.*, kotlin.text.*, kotlin.comparisons.*, kotlin.annotation.*,
206
+ # kotlin.reflect.*, kotlin.jvm.*, kotlin.io.*, kotlin.math.*, kotlin.contracts.*.
207
+ # Types from these (e.g. ``List``, ``Map``, ``Sequence``, ``Pair``) referenced as
208
+ # supertypes would otherwise collapse to bare-name phantoms (``fqn="List"``). This
209
+ # map gives the phantom fallback a deterministic FQN, mirroring ``_JAVA_LANG_SIMPLE``
210
+ # for Java. Consulted ONLY when ``ast.language == "kotlin"`` (Kotlin-gated), so the
211
+ # Java resolution path is byte-identical. Curated to the commonly-referenced stdlib
212
+ # types; unknown simples keep falling through to the bare-name guess.
213
+ _KOTLIN_DEFAULT_SIMPLE: dict[str, str] = {
214
+ # kotlin.*
215
+ "Any": "kotlin.Any", "Unit": "kotlin.Unit", "Nothing": "kotlin.Nothing",
216
+ "Int": "kotlin.Int", "Long": "kotlin.Long", "Short": "kotlin.Short",
217
+ "Byte": "kotlin.Byte", "Double": "kotlin.Double", "Float": "kotlin.Float",
218
+ "Boolean": "kotlin.Boolean", "Char": "kotlin.Char", "String": "kotlin.String",
219
+ "Array": "kotlin.Array", "Pair": "kotlin.Pair", "Triple": "kotlin.Triple",
220
+ "Result": "kotlin.Result", "Enum": "kotlin.Enum", "Annotation": "kotlin.Annotation",
221
+ "Throwable": "kotlin.Throwable", "Exception": "kotlin.Exception",
222
+ "Error": "kotlin.Error", "Lazy": "kotlin.Lazy",
223
+ # kotlin.collections.*
224
+ "List": "kotlin.collections.List", "MutableList": "kotlin.collections.MutableList",
225
+ "Set": "kotlin.collections.Set", "MutableSet": "kotlin.collections.MutableSet",
226
+ "Map": "kotlin.collections.Map", "MutableMap": "kotlin.collections.MutableMap",
227
+ "Collection": "kotlin.collections.Collection",
228
+ "MutableCollection": "kotlin.collections.MutableCollection",
229
+ "Iterable": "kotlin.collections.Iterable",
230
+ "MutableIterable": "kotlin.collections.MutableIterable",
231
+ "ArrayList": "kotlin.collections.ArrayList", "HashMap": "kotlin.collections.HashMap",
232
+ "HashSet": "kotlin.collections.HashSet",
233
+ "LinkedHashMap": "kotlin.collections.LinkedHashMap",
234
+ "LinkedHashSet": "kotlin.collections.LinkedHashSet",
235
+ "Grouping": "kotlin.collections.Grouping",
236
+ "Iterator": "kotlin.collections.Iterator",
237
+ "MutableIterator": "kotlin.collections.MutableIterator",
238
+ "ListIterator": "kotlin.collections.ListIterator",
239
+ "MutableListIterator": "kotlin.collections.MutableListIterator",
240
+ "Comparator": "kotlin.Comparator",
241
+ # kotlin.sequences.*
242
+ "Sequence": "kotlin.sequences.Sequence",
243
+ # kotlin.ranges.*
244
+ "IntRange": "kotlin.ranges.IntRange", "LongRange": "kotlin.ranges.LongRange",
245
+ "CharRange": "kotlin.ranges.CharRange", "ClosedRange": "kotlin.ranges.ClosedRange",
246
+ "OpenEndRange": "kotlin.ranges.OpenEndRange",
247
+ # kotlin.text.*
248
+ "Regex": "kotlin.text.Regex", "Appendable": "kotlin.text.Appendable",
249
+ "MatchResult": "kotlin.text.MatchResult",
250
+ # kotlin.reflect.*
251
+ "KClass": "kotlin.reflect.KClass", "KCallable": "kotlin.reflect.KCallable",
252
+ "KProperty": "kotlin.reflect.KProperty", "KFunction": "kotlin.reflect.KFunction",
253
+ "KType": "kotlin.reflect.KType", "KParameter": "kotlin.reflect.KParameter",
254
+ }
255
+
256
+
187
257
  # ---------- dataclasses ----------
188
258
 
189
259
 
@@ -470,6 +540,10 @@ class GraphTables:
470
540
  type_role_by_node_id: dict[str, str] = field(default_factory=dict)
471
541
  # Populated in pass 1 (classify_java_file) and _load_existing_types for incremental rebuilds.
472
542
  type_generated_by_node_id: dict[str, tuple[bool, str | None]] = field(default_factory=dict)
543
+ # Per-build dedup for the same-FQN cross-file collision warning in
544
+ # ``_register_type`` (one warning per colliding FQN per build — avoids
545
+ # spam when >2 files claim one FQN). See ``_register_type``.
546
+ _warned_fqn_collisions: set[str] = field(default_factory=set)
473
547
 
474
548
 
475
549
  @dataclass
@@ -532,7 +606,7 @@ class FileHashTracker:
532
606
  current_files: set[str] = set()
533
607
  # Resolve source_root to handle symlinks
534
608
  source_root_resolved = source_root.resolve()
535
- for abs_path in iter_java_source_files(source_root, ignore=ignore):
609
+ for abs_path in iter_source_files(source_root, ignore=ignore):
536
610
  # Resolve the absolute path and compute relative path
537
611
  abs_path_resolved = abs_path.resolve()
538
612
  try:
@@ -929,7 +1003,7 @@ def _write_nodes_merge(
929
1003
  _write_nodes_impl(conn, tables, project_root=project_root, meta_chain=meta_chain)
930
1004
 
931
1005
 
932
- # ---------- file walk (see `path_filtering.iter_java_source_files`) ----------
1006
+ # ---------- file walk (see `path_filtering.iter_source_files`) ----------
933
1007
 
934
1008
 
935
1009
  # ---------- pass 1 ----------
@@ -967,6 +1041,24 @@ def _register_type(
967
1041
  outer_fqn=outer_fqn,
968
1042
  node_id=node_id,
969
1043
  )
1044
+ # Warn on a same-FQN collision across DISTINCT files (the Kotlin+Java
1045
+ # mixed-repo case, e.g. ``com.example.Foo`` in both ``Foo.java`` and
1046
+ # ``Foo.kt``). Registration stays last-wins (silent before Task 16); this
1047
+ # only surfaces the collision. Deduped per FQN per build so a pathological
1048
+ # N-way collision warns once, not N-1 times. A same-file re-register
1049
+ # (incremental re-parse) does not warn.
1050
+ existing = tables.types.get(decl.fqn)
1051
+ if (
1052
+ existing is not None
1053
+ and existing.file_path != file_path
1054
+ and decl.fqn not in tables._warned_fqn_collisions
1055
+ ):
1056
+ tables._warned_fqn_collisions.add(decl.fqn)
1057
+ log.warning(
1058
+ "same-FQN type registered from two distinct files (last wins): "
1059
+ "%s in %s and %s",
1060
+ decl.fqn, existing.file_path, file_path,
1061
+ )
970
1062
  tables.types[decl.fqn] = entry
971
1063
  tables.by_simple_name.setdefault(decl.name, []).append(entry)
972
1064
  tables.by_package.setdefault(package, []).append(entry)
@@ -1028,7 +1120,7 @@ def pass1_parse(
1028
1120
  removed = removed_files if removed_files is not None else set()
1029
1121
  pass1_total = len(scope_files - removed)
1030
1122
  else:
1031
- pass1_total = sum(1 for _ in iter_java_source_files(root, ignore=ignore))
1123
+ pass1_total = sum(1 for _ in iter_source_files(root, ignore=ignore))
1032
1124
  _emit_graph_progress(
1033
1125
  {"pass": "1/6", "done": 0, "total": pass1_total, "status": "running"},
1034
1126
  verbose=verbose,
@@ -1043,7 +1135,11 @@ def pass1_parse(
1043
1135
  with _VerbosePassHeartbeats("[graph] pass 1", verbose=verbose):
1044
1136
  if verbose and slow_sec > 0:
1045
1137
  time.sleep(slow_sec)
1046
- for p in iter_java_source_files(root, ignore=ignore):
1138
+ # Per-file generated classification, carried through to the registration
1139
+ # loop below (registration is deferred until after the multifile-facade
1140
+ # merge so merged facades win their FQN slot in tables.types).
1141
+ file_meta: dict[str, tuple[str, str, bool, str | None]] = {}
1142
+ for p in iter_source_files(root, ignore=ignore):
1047
1143
  # Skip files not in scope (if scope is provided)
1048
1144
  try:
1049
1145
  rel = p.resolve().relative_to(root.resolve()).as_posix()
@@ -1064,8 +1160,11 @@ def pass1_parse(
1064
1160
  continue
1065
1161
  if not content.strip():
1066
1162
  continue
1163
+ backend = backend_for(rel)
1164
+ if backend is None:
1165
+ continue
1067
1166
  try:
1068
- ast = parse_java(content, filename=rel, verbose=verbose)
1167
+ ast = backend.parse(content, filename=rel, verbose=verbose)
1069
1168
  except Exception:
1070
1169
  tables.parse_errors += 1
1071
1170
  continue
@@ -1081,6 +1180,7 @@ def pass1_parse(
1081
1180
  file_generated, file_generated_by = classify_java_file(
1082
1181
  content, ast, config=generated_config, project_root=root
1083
1182
  )
1183
+ file_meta[rel] = (module, microservice, file_generated, file_generated_by)
1084
1184
 
1085
1185
  # file node
1086
1186
  file_id = symbol_id("file", rel, rel, 0)
@@ -1090,6 +1190,24 @@ def pass1_parse(
1090
1190
  if ast.package and ast.package not in tables.packages:
1091
1191
  tables.packages[ast.package] = symbol_id("package", ast.package, "", 0)
1092
1192
 
1193
+ # Merge Kotlin ``@file:JvmMultifileClass`` facades BEFORE type
1194
+ # registration (Task 9 / Task 11). Multiple ``.kt`` files sharing
1195
+ # ``@file:JvmName("X")`` + ``@file:JvmMultifileClass()`` compile into ONE
1196
+ # JVM class ``pkg.X``; the per-file parse emits one facade per file, all
1197
+ # with FQN ``pkg.X``, so registering them per-file would collide in
1198
+ # ``tables.types[fqn]`` (last-write-wins) and silently drop every other
1199
+ # file's top-level functions from resolution. The merge concatenates the
1200
+ # members onto ONE retained facade and strips the duplicates. It is a
1201
+ # no-op on Java ASTs (no facades) and on single-file modules, so Java
1202
+ # registration output is byte-identical (same files, same walk order).
1203
+ if _merge_multifile_facades is not None and asts:
1204
+ _merge_multifile_facades(list(asts.values()))
1205
+
1206
+ # Register types AFTER the merge so a merged facade wins its FQN slot.
1207
+ # Iteration order is the dict's insertion order = walk order, identical
1208
+ # to the former per-file registration, so Java output is unchanged.
1209
+ for rel, ast in asts.items():
1210
+ module, microservice, file_generated, file_generated_by = file_meta[rel]
1093
1211
  for t in ast.top_level_types:
1094
1212
  _register_type(
1095
1213
  tables, t, file_path=rel,
@@ -1188,6 +1306,10 @@ def _phantom_target(
1188
1306
  guess_fqn = ast.explicit_imports[bare]
1189
1307
  elif bare in _JAVA_LANG_SIMPLE:
1190
1308
  guess_fqn = f"java.lang.{bare}"
1309
+ elif ast.language == "kotlin" and bare in _KOTLIN_DEFAULT_SIMPLE:
1310
+ # Kotlin default-import stdlib type (kotlin.collections.List etc.) —
1311
+ # Kotlin-gated so Java resolution is byte-identical. See Task 13.
1312
+ guess_fqn = _KOTLIN_DEFAULT_SIMPLE[bare]
1191
1313
  elif ast.wildcard_imports:
1192
1314
  # Pick first wildcard as a hint (imperfect but useful for display).
1193
1315
  guess_fqn = f"{ast.wildcard_imports[0]}.{bare}"
@@ -4170,7 +4292,7 @@ def _init_hash_tracker(source_root: Path, ladybug_path: Path) -> int:
4170
4292
  ignore = LayeredIgnore(source_root)
4171
4293
  all_files: set[str] = set()
4172
4294
  source_root_resolved = source_root.resolve()
4173
- for p in iter_java_source_files(source_root, ignore=ignore):
4295
+ for p in iter_source_files(source_root, ignore=ignore):
4174
4296
  p_resolved = p.resolve()
4175
4297
  try:
4176
4298
  rel_path = p_resolved.relative_to(source_root_resolved).as_posix()
@@ -38,11 +38,11 @@ from java_codebase_rag.ast.ast_java import (
38
38
  CODEBASE_PRODUCER_ANNOTATIONS,
39
39
  infer_capabilities_for_type,
40
40
  infer_role_for_type,
41
- parse_java,
42
41
  ROLE_ANNOTATIONS,
43
42
  _METHOD_ANN_TO_CAPABILITY,
44
43
  _TYPE_ANN_TO_CAPABILITY,
45
44
  )
45
+ from java_codebase_rag.ast.language import backend_for
46
46
  from java_codebase_rag.graph.java_ontology import (
47
47
  CLIENT_KIND_REST_TEMPLATE,
48
48
  VALID_CAPABILITIES,
@@ -52,7 +52,7 @@ from java_codebase_rag.graph.java_ontology import (
52
52
  VALID_ROUTE_FRAMEWORKS,
53
53
  VALID_ROUTE_KINDS,
54
54
  )
55
- from java_codebase_rag.graph.path_filtering import LayeredIgnore, iter_java_source_files
55
+ from java_codebase_rag.graph.path_filtering import LayeredIgnore, iter_source_files
56
56
 
57
57
  __all__ = [
58
58
  "AnnotationDecl",
@@ -387,7 +387,7 @@ def _collect_annotation_decl_index(project_root_str: str) -> dict[str, Annotatio
387
387
  return {}
388
388
  ignore = LayeredIgnore(root)
389
389
  decls: dict[str, AnnotationDecl] = {}
390
- for p in sorted(iter_java_source_files(root, ignore=ignore), key=str):
390
+ for p in sorted(iter_source_files(root, ignore=ignore), key=str):
391
391
  try:
392
392
  content = p.read_bytes()
393
393
  except OSError as exc:
@@ -398,8 +398,11 @@ def _collect_annotation_decl_index(project_root_str: str) -> dict[str, Annotatio
398
398
  continue
399
399
  if not content.strip():
400
400
  continue
401
+ backend = backend_for(p)
402
+ if backend is None:
403
+ continue
401
404
  try:
402
- jast = parse_java(content)
405
+ jast = backend.parse(content, filename=str(p))
403
406
  except Exception as exc:
404
407
  print(
405
408
  f"[lancedb-mcp] parse error in {p}: {exc}",
@@ -1728,7 +1731,7 @@ def classify_java_file(
1728
1731
 
1729
1732
  Args:
1730
1733
  source: Raw file bytes (required for header-banner detection).
1731
- ast: Parsed Java AST (from ast_java.parse_java_ast).
1734
+ ast: Parsed Java AST (from ast_java.parse_java).
1732
1735
  config: Optional detection config (defaults to empty config).
1733
1736
  project_root: Optional project root for loading default config.
1734
1737
 
@@ -380,7 +380,7 @@ class LadybugGraph:
380
380
  f"Graph ontology version {graph_version} is older than the "
381
381
  f"required version {_ONTOLOGY_VERSION}. "
382
382
  "Rebuild the graph: `python build_ast_graph.py --source-root <repo>`, "
383
- "or run `java-codebase-rag reprocess --source-root <repo>` for a full "
383
+ "or run `jrag reprocess --source-root <repo>` for a full "
384
384
  "Lance+Ladybug re-index."
385
385
  )
386
386
  cls._instance = instance
@@ -23,6 +23,8 @@ from typing import overload
23
23
 
24
24
  from pathspec import GitIgnoreSpec
25
25
 
26
+ from java_codebase_rag.ast.language import LANG_BACKENDS
27
+
26
28
  # Pruning for LocalFile sources: skip VCS, build outputs, dependency trees, and
27
29
  # test sources (we currently index prod Java only to keep the semantic index clean).
28
30
  # Also avoids EMFILE under default ulimits when the engine traverses in parallel.
@@ -427,26 +429,33 @@ class LayeredIgnore:
427
429
 
428
430
 
429
431
  @overload
430
- def iter_java_source_files(root: Path, exclude_globs: list[str]) -> Iterator[Path]: ...
432
+ def iter_source_files(root: Path, exclude_globs: list[str]) -> Iterator[Path]: ...
431
433
 
432
434
 
433
435
  @overload
434
- def iter_java_source_files(root: Path, *, ignore: LayeredIgnore) -> Iterator[Path]: ...
436
+ def iter_source_files(root: Path, *, ignore: LayeredIgnore) -> Iterator[Path]: ...
435
437
 
436
438
 
437
- def iter_java_source_files(
439
+ def iter_source_files(
438
440
  root: Path,
439
441
  exclude_globs: list[str] | None = None,
440
442
  *,
441
443
  ignore: LayeredIgnore | None = None,
442
444
  ) -> Iterator[Path]:
443
- """Walk ``root`` for ``*.java``, honouring prunes and layered ignore rules."""
445
+ """Walk ``root`` for source files of any registered language backend.
446
+
447
+ Yields files whose suffix is claimed by at least one backend in
448
+ :data:`~java_codebase_rag.ast.language.LANG_BACKENDS` (``.java`` always;
449
+ ``.kt`` when the ``tree-sitter-kotlin`` grammar imports). Pruning
450
+ (``UNCONDITIONAL_PRUNE_DIRS`` + ``_is_build_output_dir``) and the layered
451
+ ignore rules are unchanged from the former Java-only walk.
452
+ """
444
453
  if exclude_globs is not None and ignore is not None:
445
454
  raise TypeError("pass either exclude_globs or ignore=, not both")
446
455
  if exclude_globs is not None:
447
456
  warnings.warn(
448
- "iter_java_source_files(root, exclude_globs) is deprecated; "
449
- "use iter_java_source_files(root, ignore=LayeredIgnore(root, ...)).",
457
+ "iter_source_files(root, exclude_globs) is deprecated; "
458
+ "use iter_source_files(root, ignore=LayeredIgnore(root, ...)).",
450
459
  DeprecationWarning,
451
460
  stacklevel=2,
452
461
  )
@@ -455,6 +464,11 @@ def iter_java_source_files(
455
464
  ignore_ctx = ignore
456
465
  else:
457
466
  ignore_ctx = LayeredIgnore(root)
467
+ # Union of suffixes claimed by every registered backend (``.java`` always;
468
+ # ``.kt`` when the Kotlin grammar imports). Computed once per call.
469
+ known_suffixes: set[str] = set()
470
+ for backend in LANG_BACKENDS.values():
471
+ known_suffixes.update(backend.suffixes)
458
472
  root = root.resolve()
459
473
  for dirpath, dirnames, filenames in os.walk(root):
460
474
  # Universal nuisance dirs (VCS, IDE, deps) are pruned unconditionally.
@@ -469,9 +483,27 @@ def iter_java_source_files(
469
483
  and not _is_build_output_dir(dirpath, d)
470
484
  ]
471
485
  for fn in filenames:
472
- if not fn.endswith(".java"):
486
+ # Literal byte-identity match (mirrors the legacy ``endswith``
487
+ # behaviour): ``Path(".java").suffix`` is ``""``, so a dotfile named
488
+ # ``.java`` would be dropped by a suffix-in-set check. ``endswith``
489
+ # over the registered suffixes restores the exact old semantics.
490
+ if not any(fn.endswith(s) for s in known_suffixes):
473
491
  continue
474
492
  p = Path(dirpath) / fn
475
493
  if ignore_ctx.is_ignored(p):
476
494
  continue
477
495
  yield p
496
+
497
+
498
+ def iter_java_source_files(
499
+ root: Path,
500
+ exclude_globs: list[str] | None = None,
501
+ *,
502
+ ignore: LayeredIgnore | None = None,
503
+ ) -> Iterator[Path]:
504
+ """Deprecated alias for :func:`iter_source_files`.
505
+
506
+ Kept so any import site not migrated to the new name continues to work; new
507
+ call sites should use :func:`iter_source_files`.
508
+ """
509
+ return iter_source_files(root, exclude_globs, ignore=ignore)
@@ -6,7 +6,7 @@ LanceDB requires a single primary key per table; each chunk gets a UUID `id`.
6
6
  Environment:
7
7
  JAVA_CODEBASE_RAG_INDEX_DIR — Lance tables + LadybugDB + cocoindex state (default: ./.java-codebase-rag)
8
8
  JAVA_CODEBASE_RAG_SOURCE_ROOT — Java repo root for indexing (optional; else cocoindex cwd)
9
- SBERT_MODEL / SBERT_DEVICE — embedding (optional; YAML also supported via java-codebase-rag CLI)
9
+ SBERT_MODEL / SBERT_DEVICE — embedding (optional; YAML also supported via jrag CLI)
10
10
 
11
11
  Dependencies:
12
12
  pip install "cocoindex[lancedb]" sentence-transformers
@@ -50,7 +50,8 @@ from java_codebase_rag.index.java_index_v1_common import (
50
50
  position_to_json,
51
51
  )
52
52
  from java_codebase_rag.graph.path_filtering import LayeredIgnore
53
- from java_codebase_rag.ast.ast_java import ONTOLOGY_VERSION, parse_java
53
+ from java_codebase_rag.ast.ast_java import ONTOLOGY_VERSION
54
+ from java_codebase_rag.ast.language import LANG_BACKENDS, backend_for
54
55
  from java_codebase_rag.graph.graph_enrich import (
55
56
  classify_java_file,
56
57
  collect_annotation_meta_chain,
@@ -115,6 +116,21 @@ splitter = RecursiveSplitter()
115
116
  # parent clamps to total on the terminal event anyway).
116
117
  _VECTORS_TICK_EVERY = 25
117
118
 
119
+ # Suffixes that index into the ``JavaLanceChunk`` table — the registered
120
+ # language backends (``.java`` always; ``.kt`` when the Kotlin grammar imports).
121
+ # Derived from ``LANG_BACKENDS`` so this never drifts from what the graph builder
122
+ # parses (mirrors the watcher's ``INDEXED_SUFFIXES``). On a grammar-absent
123
+ # install this is just ``(".java",)`` and ``.kt`` files are skipped cleanly.
124
+ _INDEXED_SOURCE_SUFFIXES: tuple[str, ...] = tuple(
125
+ suffix for backend in LANG_BACKENDS.values() for suffix in backend.suffixes
126
+ )
127
+ # True iff some registered backend claims ``.kt`` (i.e. ``tree-sitter-kotlin``
128
+ # imported). Gates the ``.kt`` cocoindex matcher + ``process_kotlin_file`` drain
129
+ # in ``app_main`` so a grammar-absent install skips ``.kt`` by construction
130
+ # instead of crashing inside ``_parse_and_enrich_java`` (backend-for-``.kt``
131
+ # returns ``None`` → ``classify_java_file`` dereferences ``ast.all_types``).
132
+ _KOTLIN_REGISTERED: bool = ".kt" in _INDEXED_SOURCE_SUFFIXES
133
+
118
134
  # Bounded concurrency for the per-file drain in app_main. cocoindex's embedder
119
135
  # is ``@coco.fn.as_async(batching=True, runner=GPU, max_batch_size=64)`` — but
120
136
  # its batching layer only coalesces calls that are in flight SIMULTANEOUSLY. A
@@ -129,9 +145,9 @@ _VECTORS_TICK_EVERY = 25
129
145
  # This stays inside ONE component, so the earlier mount_each→app_main win is
130
146
  # preserved: still exactly ONE merge_insert per table at commit. Memoization
131
147
  # (``@coco.fn(memo=True)``) and the lock-guarded tick counter are safe under
132
- # concurrency; ``parse_java`` uses a per-thread tree-sitter Parser (already
133
- # routed via ``asyncio.to_thread``) and ``splitter.split`` is synchronous so the
134
- # event loop cannot reenter it.
148
+ # concurrency; the backend parse (``parse_java`` / ``parse_kotlin``) uses a
149
+ # per-thread tree-sitter Parser (already routed via ``asyncio.to_thread``)
150
+ # and ``splitter.split`` is synchronous so the event loop cannot reenter it.
135
151
  #
136
152
  # Default 64 is sized to MATCH the embedder's hardcoded max_batch_size=64 (the
137
153
  # decorator above; not a constructor arg, so not raisable from the flow): ~64
@@ -209,8 +225,8 @@ def _approximate_vectors_total(project_root: Path) -> int:
209
225
  count. The parent clamps the bar to 100% on the terminal ``status=done``
210
226
  event, so the over-count cannot stall the bar.
211
227
 
212
- Mirrors the three ``localfs.walk_dir`` matchers in ``app_main``:
213
- - ``**/*.java``
228
+ Mirrors the ``localfs.walk_dir`` matchers in ``app_main``:
229
+ - ``**/*.java`` and ``**/*.kt`` (registered language suffixes)
214
230
  - ``**/src/main/resources/db/migration/*.sql``
215
231
  - ``**/src/main/resources/application*.yml`` and ``.yaml``
216
232
  """
@@ -222,7 +238,7 @@ def _approximate_vectors_total(project_root: Path) -> int:
222
238
 
223
239
  total = 0
224
240
  for dirpath, dirnames, filenames in os.walk(project_root):
225
- # Prune the same universal nuisance dirs as iter_java_source_files /
241
+ # Prune the same universal nuisance dirs as iter_source_files /
226
242
  # cocoindex walk. (build-output pruning is matcher-dependent in the
227
243
  # real walk; for an APPROXIMATE total this cheap prune is sufficient
228
244
  # — the clamp absorbs any residual divergence.)
@@ -237,8 +253,12 @@ def _approximate_vectors_total(project_root: Path) -> int:
237
253
  continue
238
254
  if _excluded(rel):
239
255
  continue
240
- # Java: **/*.java
241
- if fn.endswith(".java"):
256
+ # Java + Kotlin: the registered source-language suffixes (see
257
+ # ``_INDEXED_SOURCE_SUFFIXES`` / ``LANG_BACKENDS`` — ``.java`` always,
258
+ # ``.kt`` when the Kotlin grammar imports). Both index into the same
259
+ # ``JavaLanceChunk`` table via ``process_java_file`` /
260
+ # ``process_kotlin_file``.
261
+ if fn.endswith(_INDEXED_SOURCE_SUFFIXES):
242
262
  if not ignore.is_ignored(full):
243
263
  total += 1
244
264
  continue
@@ -380,14 +400,23 @@ def _parse_and_enrich_java(
380
400
  parses + enriches, the event loop is free to drive other files and keep the
381
401
  embedder's batching queue fed.
382
402
 
383
- Thread-safety: ``parse_java`` uses a per-thread tree-sitter ``Parser``
384
- (see ``ast_java._parser``), so it is safe to call concurrently from these
385
- worker threads — including the transitive ``parse_java`` that ``enrich_chunk``
403
+ Thread-safety: the backend parse (``parse_java`` / ``parse_kotlin``) uses
404
+ a per-thread tree-sitter ``Parser`` (see ``ast_java._parser`` /
405
+ ``ast_kotlin._parser``), so it is safe to call concurrently from these
406
+ worker threads — including the transitive re-parse that ``enrich_chunk``
386
407
  triggers via ``collect_annotation_meta_chain`` → ``_collect_annotation_decl_index``.
387
408
  ``enrich_chunk`` is otherwise pure-Python over the now-immutable AST; its
388
409
  ``lru_cache`` reads are thread-safe under the GIL.
389
410
  """
390
- ast = parse_java(content_bytes)
411
+ backend = backend_for(rel)
412
+ if backend is None:
413
+ # Defensive: ``app_main`` registers the ``.kt`` matcher + kotlin drain
414
+ # only when ``_KOTLIN_REGISTERED`` (registry-derived), and ``.java`` is
415
+ # always registered — so by construction this is unreachable for every
416
+ # suffix the flow yields. Kept to honor the dispatch contract (a
417
+ # grammar-absent install never yields ``.kt`` here).
418
+ return [], None
419
+ ast = backend.parse(content_bytes, filename=rel)
391
420
  enrichments = [
392
421
  enrich_chunk(
393
422
  ast,
@@ -437,7 +466,8 @@ async def process_java_file(
437
466
 
438
467
  # (vectors perf lever #2) parse + enrich off the event loop so the loop can
439
468
  # keep the embedder's batching queue fed while this file is being parsed.
440
- # parse_java is thread-safe (per-thread tree-sitter Parser in ast_java).
469
+ # The backend parse (parse_java / parse_kotlin) is thread-safe (per-thread
470
+ # tree-sitter Parser in ast_java / ast_kotlin).
441
471
  enrichments, ast = await asyncio.to_thread(
442
472
  _parse_and_enrich_java, content_bytes, chunks, rel, project_root
443
473
  )
@@ -483,6 +513,98 @@ async def process_java_file(
483
513
  )
484
514
 
485
515
 
516
+ @coco.fn(memo=True)
517
+ async def process_kotlin_file(
518
+ file: localfs.File,
519
+ table: lancedb.TableTarget[JavaLanceChunk],
520
+ ) -> None:
521
+ """Index one ``.kt`` file into the SAME ``JavaLanceChunk`` table as Java.
522
+
523
+ Mirrors ``process_java_file``'s enrichment path (``enrich_chunk`` /
524
+ ``classify_java_file``) but parsing dispatches through ``backend_for(rel)``
525
+ (= ``parse_kotlin``) inside the shared ``_parse_and_enrich_java`` helper,
526
+ which is already language-agnostic. The chunk ``language`` field is set to
527
+ ``"kotlin"`` (``detect_code_language`` returns ``"kotlin"`` for ``.kt``).
528
+ The chunk schema (``primary_type_kind`` / ``role`` / ``capabilities``) is
529
+ language-agnostic, so no new column is needed.
530
+
531
+ Multifile-facade merge is NOT wired here: each ``process_*_file`` parses ONE
532
+ file independently (concurrent per-file drain), so a cross-file pre-pass is
533
+ awkward inside cocoindex's dataflow. The merge runs in ``build_ast_graph``
534
+ pass1 instead — the only site that registers facade TypeDecls into
535
+ ``tables.types`` (where unmerged facades would collide). Chunk enrichment
536
+ (``enrich_chunk``) uses the per-file AST only, so it is merge-independent.
537
+ """
538
+ embedder = coco.use_context(EMBEDDER)
539
+ project_root = coco.use_context(PROJECT_ROOT)
540
+ ignore = coco.use_context(IGNORE)
541
+ if ignore.is_ignored((project_root / file.file_path.path).resolve()):
542
+ return
543
+ try:
544
+ content = await file.read_text()
545
+ except UnicodeDecodeError:
546
+ return
547
+ if not content.strip():
548
+ return
549
+
550
+ _tick_vectors_done()
551
+
552
+ language = detect_code_language(filename=file.file_path.path.name) or "text"
553
+ cs, mn, ov = JAVA_CHUNK
554
+ chunks = splitter.split(
555
+ content,
556
+ cs,
557
+ min_chunk_size=mn,
558
+ chunk_overlap=ov,
559
+ language=language,
560
+ )
561
+ rel = file.file_path.path.as_posix()
562
+ content_bytes = content.encode("utf-8", errors="replace")
563
+
564
+ # ``_parse_and_enrich_java`` dispatches via ``backend_for(rel)`` → parse_kotlin
565
+ # for ``.kt``; the helper and ``enrich_chunk`` are language-agnostic. Run off
566
+ # the event loop so the embedder batching queue stays fed (vectors perf #2).
567
+ enrichments, ast = await asyncio.to_thread(
568
+ _parse_and_enrich_java, content_bytes, chunks, rel, project_root
569
+ )
570
+
571
+ generated_config = load_generated_detection(project_root)
572
+ generated, generated_by = classify_java_file(
573
+ content_bytes, ast, config=generated_config, project_root=project_root
574
+ )
575
+
576
+ # Embed all chunks concurrently → batched encode (vectors perf #1).
577
+ embeddings = await asyncio.gather(*(embedder.embed(ch.text) for ch in chunks))
578
+
579
+ for ch, enrich, emb in zip(chunks, enrichments, embeddings):
580
+ rs, re = chunk_key_range(ch)
581
+ table.declare_row(
582
+ row=JavaLanceChunk(
583
+ id=str(uuid.uuid4()),
584
+ filename=rel,
585
+ language=language,
586
+ text=ch.text,
587
+ range_start=rs,
588
+ range_end=re,
589
+ start=position_to_json(ch.start),
590
+ end=position_to_json(ch.end),
591
+ embedding=emb,
592
+ package=enrich.package,
593
+ module=enrich.module,
594
+ microservice=enrich.microservice,
595
+ primary_type_fqn=enrich.primary_type_fqn,
596
+ primary_type_kind=enrich.primary_type_kind,
597
+ role=enrich.role,
598
+ capabilities=list(enrich.capabilities),
599
+ annotations_on_type=enrich.annotations_on_type,
600
+ symbols=enrich.symbols,
601
+ ontology_version=ONTOLOGY_VERSION,
602
+ generated=generated,
603
+ generated_by=generated_by,
604
+ )
605
+ )
606
+
607
+
486
608
  @coco.fn(memo=True)
487
609
  async def process_sql_file(
488
610
  file: localfs.File,
@@ -682,6 +804,18 @@ async def app_main() -> None:
682
804
  excluded_patterns=_walk_excludes,
683
805
  ),
684
806
  )
807
+ kotlin_files = (
808
+ localfs.walk_dir(
809
+ PROJECT_ROOT,
810
+ recursive=True,
811
+ path_matcher=PatternFilePathMatcher(
812
+ included_patterns=["**/*.kt"],
813
+ excluded_patterns=_walk_excludes,
814
+ ),
815
+ )
816
+ if _KOTLIN_REGISTERED
817
+ else None
818
+ )
685
819
  sql_files = localfs.walk_dir(
686
820
  PROJECT_ROOT,
687
821
  recursive=True,
@@ -724,6 +858,17 @@ async def app_main() -> None:
724
858
  # usually near-empty).
725
859
  _sem = asyncio.Semaphore(_FILE_CONCURRENCY)
726
860
  await _drain_files_concurrently(java_files, process_java_file, java_table, _sem)
861
+ # Kotlin drains into the SAME ``java_table`` (JavaLanceChunk) — the chunk
862
+ # schema is language-agnostic and the ``language`` column distinguishes rows.
863
+ # Gated on ``_KOTLIN_REGISTERED`` (registry-derived): on a grammar-absent
864
+ # install ``.kt`` has no backend, so the matcher + drain are skipped entirely
865
+ # — no wasted read/chunk/embed, and no crash in ``_parse_and_enrich_java``
866
+ # (``backend_for(.kt)`` returns ``None`` → ``classify_java_file`` would
867
+ # dereference ``ast.all_types`` on ``None``).
868
+ if _KOTLIN_REGISTERED:
869
+ await _drain_files_concurrently(
870
+ kotlin_files, process_kotlin_file, java_table, _sem
871
+ )
727
872
  await _drain_files_concurrently(sql_files, process_sql_file, sql_table, _sem)
728
873
  await _drain_files_concurrently(yaml_files, process_yaml_file, yaml_table, _sem)
729
874