java-codebase-rag 0.9.4__py3-none-any.whl → 0.9.6__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. java_codebase_rag/absence/__init__.py +0 -0
  2. java_codebase_rag/absence/absence_diagnosis.py +700 -0
  3. java_codebase_rag/absence/absence_types.py +124 -0
  4. java_codebase_rag/absence/absence_vocab.py +455 -0
  5. java_codebase_rag/analysis/__init__.py +0 -0
  6. pr_analysis.py → java_codebase_rag/analysis/pr_analysis.py +1 -1
  7. resolve_service.py → java_codebase_rag/analysis/resolve_service.py +73 -6
  8. java_codebase_rag/ast/__init__.py +0 -0
  9. ast_java.py → java_codebase_rag/ast/ast_java.py +5 -5
  10. java_codebase_rag/cli.py +13 -18
  11. java_codebase_rag/config.py +116 -0
  12. java_codebase_rag/graph/__init__.py +0 -0
  13. build_ast_graph.py → java_codebase_rag/graph/build_ast_graph.py +89 -11
  14. graph_enrich.py → java_codebase_rag/graph/graph_enrich.py +248 -3
  15. graph_types.py → java_codebase_rag/graph/graph_types.py +6 -2
  16. java_ontology.py → java_codebase_rag/graph/java_ontology.py +1 -1
  17. ladybug_queries.py → java_codebase_rag/graph/ladybug_queries.py +6 -6
  18. java_codebase_rag/index/__init__.py +0 -0
  19. java_index_flow_lancedb.py → java_codebase_rag/index/java_index_flow_lancedb.py +30 -10
  20. java_codebase_rag/install_data/__init__.py +0 -0
  21. java_codebase_rag/jrag.py +71 -16
  22. java_codebase_rag/jrag_envelope.py +13 -4
  23. java_codebase_rag/jrag_hints.py +1 -1
  24. java_codebase_rag/jrag_render.py +67 -3
  25. java_codebase_rag/mcp/__init__.py +0 -0
  26. mcp_hints.py → java_codebase_rag/mcp/mcp_hints.py +1 -1
  27. mcp_v2.py → java_codebase_rag/mcp/mcp_v2.py +280 -81
  28. server.py → java_codebase_rag/mcp/server.py +138 -54
  29. java_codebase_rag/pipeline.py +26 -7
  30. java_codebase_rag/search/__init__.py +0 -0
  31. search_lancedb.py → java_codebase_rag/search/search_lancedb.py +53 -314
  32. java_codebase_rag/search/search_lexical.py +329 -0
  33. java_codebase_rag/search/search_scoring.py +338 -0
  34. {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/METADATA +2 -2
  35. java_codebase_rag-0.9.6.dist-info/RECORD +57 -0
  36. {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/entry_points.txt +1 -1
  37. java_codebase_rag-0.9.6.dist-info/top_level.txt +1 -0
  38. java_codebase_rag-0.9.4.dist-info/RECORD +0 -44
  39. java_codebase_rag-0.9.4.dist-info/top_level.txt +0 -19
  40. /brownfield_events.py → /java_codebase_rag/ast/brownfield_events.py +0 -0
  41. /chunk_heuristics.py → /java_codebase_rag/ast/chunk_heuristics.py +0 -0
  42. /path_filtering.py → /java_codebase_rag/graph/path_filtering.py +0 -0
  43. /java_index_v1_common.py → /java_codebase_rag/index/java_index_v1_common.py +0 -0
  44. /index_common.py → /java_codebase_rag/search/index_common.py +0 -0
  45. {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/WHEEL +0 -0
  46. {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/licenses/LICENSE +0 -0
@@ -41,7 +41,7 @@ from pathlib import Path
41
41
  import ladybug
42
42
  import pyarrow as pa
43
43
 
44
- from ast_java import (
44
+ from java_codebase_rag.ast.ast_java import (
45
45
  ONTOLOGY_VERSION,
46
46
  CallSite,
47
47
  JavaFileAst,
@@ -52,10 +52,12 @@ from ast_java import (
52
52
  lombok_required_args_annotations,
53
53
  parse_java,
54
54
  )
55
- from graph_enrich import (
55
+ from java_codebase_rag.graph.graph_enrich import (
56
56
  _load_config_cross_service_resolution,
57
+ classify_java_file,
57
58
  collect_annotation_meta_chain,
58
59
  load_brownfield_overrides,
60
+ load_generated_detection,
59
61
  microservice_for_path,
60
62
  module_for_path,
61
63
  phantom_id,
@@ -65,8 +67,8 @@ from graph_enrich import (
65
67
  resolve_routes_for_method,
66
68
  symbol_id,
67
69
  )
68
- from path_filtering import LayeredIgnore, iter_java_source_files
69
- from java_ontology import (
70
+ from java_codebase_rag.graph.path_filtering import LayeredIgnore, iter_java_source_files
71
+ from java_codebase_rag.graph.java_ontology import (
70
72
  CLIENT_KIND_FEIGN_METHOD,
71
73
  CLIENT_KIND_REST_TEMPLATE,
72
74
  VALID_CLIENT_KINDS,
@@ -465,6 +467,8 @@ class GraphTables:
465
467
  cross_service_resolution: str = "auto"
466
468
  # Populated in _write_nodes (same overrides + meta_chain as Symbol.role).
467
469
  type_role_by_node_id: dict[str, str] = field(default_factory=dict)
470
+ # Populated in pass 1 (classify_java_file) and _load_existing_types for incremental rebuilds.
471
+ type_generated_by_node_id: dict[str, tuple[bool, str | None]] = field(default_factory=dict)
468
472
 
469
473
 
470
474
  @dataclass
@@ -598,7 +602,7 @@ def _load_existing_types(conn: ladybug.Connection, tables: GraphTables, exclude_
598
602
  query = f"""
599
603
  MATCH (s:Symbol)
600
604
  {where}
601
- RETURN s.kind, s.fqn, s.name, s.filename, s.module, s.microservice, s.id, s.role
605
+ RETURN s.kind, s.fqn, s.name, s.filename, s.module, s.microservice, s.id, s.role, s.generated, s.generated_by
602
606
  """
603
607
  result = conn.execute(query, params)
604
608
  while result.has_next():
@@ -608,6 +612,8 @@ def _load_existing_types(conn: ladybug.Connection, tables: GraphTables, exclude_
608
612
  microservice = row[5] if len(row) > 5 else ""
609
613
  node_id = row[6] if len(row) > 6 else ""
610
614
  role = row[7] if len(row) > 7 else ""
615
+ generated = row[8] if len(row) > 8 else False
616
+ generated_by = row[9] if len(row) > 9 else ""
611
617
 
612
618
  decl = TypeDecl(name, kind, fqn)
613
619
  package = fqn[: -(len(name) + 1)] if fqn.endswith("." + name) else ""
@@ -629,6 +635,8 @@ def _load_existing_types(conn: ladybug.Connection, tables: GraphTables, exclude_
629
635
  # the default during node staging (issue #352 divergence #2).
630
636
  if role:
631
637
  tables.type_role_by_node_id[node_id] = role
638
+ # Seed the persisted generated/generated_by so stubs retain their values
639
+ tables.type_generated_by_node_id[node_id] = (bool(generated), generated_by)
632
640
 
633
641
 
634
642
  def _load_existing_members(conn: ladybug.Connection, tables: GraphTables, exclude_files: set[str] | None = None) -> None:
@@ -1061,6 +1069,12 @@ def pass1_parse(
1061
1069
  microservice = microservice_for_path(str(p), root)
1062
1070
  asts[rel] = ast
1063
1071
 
1072
+ # Classify the file once (generated or not, and which tool generated it)
1073
+ generated_config = load_generated_detection(str(root))
1074
+ file_generated, file_generated_by = classify_java_file(
1075
+ content, ast, config=generated_config, project_root=root
1076
+ )
1077
+
1064
1078
  # file node
1065
1079
  file_id = symbol_id("file", rel, rel, 0)
1066
1080
  tables.files[rel] = file_id
@@ -1075,6 +1089,12 @@ def pass1_parse(
1075
1089
  module=module, microservice=microservice, outer_fqn=None,
1076
1090
  )
1077
1091
 
1092
+ # Seed generated/generated_by for all types in this file (including nested)
1093
+ for t in ast.all_types:
1094
+ if t.fqn in tables.types:
1095
+ node_id = tables.types[t.fqn].node_id
1096
+ tables.type_generated_by_node_id[node_id] = (file_generated, file_generated_by)
1097
+
1078
1098
  if verbose:
1079
1099
  elapsed = time.time() - t0
1080
1100
  _emit_graph_progress(
@@ -2906,7 +2926,8 @@ _SCHEMA_NODE = (
2906
2926
  "filename STRING, start_line INT64, end_line INT64, "
2907
2927
  "start_byte INT64, end_byte INT64, "
2908
2928
  "modifiers STRING[], annotations STRING[], capabilities STRING[], "
2909
- "role STRING, signature STRING, parent_id STRING, resolved BOOLEAN"
2929
+ "role STRING, signature STRING, parent_id STRING, resolved BOOLEAN, "
2930
+ "generated BOOLEAN, generated_by STRING"
2910
2931
  ")"
2911
2932
  )
2912
2933
 
@@ -3088,6 +3109,7 @@ def _node_row(**kwargs) -> dict:
3088
3109
  "start_byte": 0, "end_byte": 0,
3089
3110
  "modifiers": [], "annotations": [], "capabilities": [],
3090
3111
  "role": "OTHER", "signature": "", "parent_id": "", "resolved": True,
3112
+ "generated": False, "generated_by": None,
3091
3113
  }
3092
3114
  base.update(kwargs)
3093
3115
  return base
@@ -3138,7 +3160,8 @@ def _existing_node_ids(conn: ladybug.Connection) -> set[str]:
3138
3160
  _NODE_COLUMNS = [
3139
3161
  "id", "kind", "name", "fqn", "package", "module", "microservice",
3140
3162
  "filename", "start_line", "end_line", "start_byte", "end_byte",
3141
- "modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved"
3163
+ "modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved",
3164
+ "generated", "generated_by"
3142
3165
  ]
3143
3166
 
3144
3167
  # Type declaration kinds. Tuple (not set) so the rendered SQL `IN` clause is
@@ -3161,7 +3184,8 @@ _SET_SYMBOL_BY_ID = (
3161
3184
  "n.start_byte = $start_byte, n.end_byte = $end_byte, "
3162
3185
  "n.modifiers = $modifiers, n.annotations = $annotations, "
3163
3186
  "n.capabilities = $capabilities, n.role = $role, "
3164
- "n.signature = $signature, n.parent_id = $parent_id, n.resolved = $resolved"
3187
+ "n.signature = $signature, n.parent_id = $parent_id, n.resolved = $resolved, "
3188
+ "n.generated = $generated, n.generated_by = $generated_by"
3165
3189
  )
3166
3190
 
3167
3191
  # Refresh every mutable Route field on an existing Route node by id. Mirrors the
@@ -3240,6 +3264,9 @@ def _write_nodes_impl(
3240
3264
  overrides=overrides,
3241
3265
  meta_chain=mch,
3242
3266
  )
3267
+ # Read generated/generated_by from pass-1 classification or stub persistence
3268
+ generated, generated_by = tables.type_generated_by_node_id.get(entry.node_id, (False, None))
3269
+
3243
3270
  if entry.loaded_from_db:
3244
3271
  stub_ids.add(entry.node_id)
3245
3272
  # Out-of-scope stub: its annotation-less decl collapses role to the
@@ -3250,6 +3277,7 @@ def _write_nodes_impl(
3250
3277
  # capabilities placeholder never reaches the graph.
3251
3278
  role = tables.type_role_by_node_id.get(entry.node_id, role)
3252
3279
  capabilities = []
3280
+ # For stubs, trust the persisted generated/generated_by (seeded by _load_existing_types)
3253
3281
  else:
3254
3282
  tables.type_role_by_node_id[entry.node_id] = role
3255
3283
  rows.append(_node_row(
@@ -3265,6 +3293,8 @@ def _write_nodes_impl(
3265
3293
  role=role,
3266
3294
  signature="",
3267
3295
  parent_id=tables.types[entry.outer_fqn].node_id if entry.outer_fqn and entry.outer_fqn in tables.types else "",
3296
+ generated=generated,
3297
+ generated_by=generated_by,
3268
3298
  ))
3269
3299
  # members (methods / constructors)
3270
3300
  for m in tables.members:
@@ -3774,7 +3804,7 @@ def incremental_rebuild(
3774
3804
  Returns IncrementalResult with statistics about the rebuild.
3775
3805
  Falls back to full rebuild if:
3776
3806
  - No previous graph exists
3777
- - Ontology version < 17 (missing source_file on edges)
3807
+ - Ontology version < ONTOLOGY_VERSION (stale schema; rebuild for current columns)
3778
3808
  - Crash marker exists (previous incremental run failed)
3779
3809
  - Dependent expansion exceeds expansion_cap
3780
3810
  """
@@ -3812,9 +3842,9 @@ def incremental_rebuild(
3812
3842
  if meta_result.has_next():
3813
3843
  row = meta_result.get_next()
3814
3844
  version = row[0] if row else 0
3815
- if version < 17:
3845
+ if version < ONTOLOGY_VERSION:
3816
3846
  if verbose:
3817
- _verbose_stderr_line(f"[increment] ontology version {version} < 17; falling back to full rebuild")
3847
+ _verbose_stderr_line(f"[increment] ontology version {version} < {ONTOLOGY_VERSION}; falling back to full rebuild")
3818
3848
  conn.close()
3819
3849
  db.close()
3820
3850
  del conn, db
@@ -4208,9 +4238,57 @@ def write_ladybug(
4208
4238
  _write_meta(conn, tables, source_root)
4209
4239
  conn.close()
4210
4240
  db.close()
4241
+
4242
+ # Build vocabulary index (best-effort, failure doesn't fail the graph build)
4243
+ _try_build_vocabulary_index(db_path, source_root, verbose)
4211
4244
  _init_hash_tracker(source_root, db_path)
4212
4245
 
4213
4246
 
4247
+ def _try_build_vocabulary_index(db_path: Path, source_root: Path, verbose: bool) -> None:
4248
+ """Build and save the vocabulary index as a sidecar (best-effort).
4249
+
4250
+ This is called after write_ladybug() completes. A build failure must not
4251
+ fail the graph build, so this is wrapped in try/except and logged.
4252
+
4253
+ Args:
4254
+ db_path: Path to the LadybugDB database file
4255
+ source_root: Source repository root
4256
+ verbose: Whether to emit verbose progress
4257
+ """
4258
+ try:
4259
+ from java_codebase_rag.absence.absence_vocab import VocabularyIndex, VOCAB_INDEX_FILENAME
4260
+ from java_codebase_rag.graph.ladybug_queries import LadybugGraph
4261
+
4262
+ t0 = time.time()
4263
+ if verbose:
4264
+ _verbose_stderr_line("[vocab] building vocabulary index")
4265
+
4266
+ # Open graph for reading
4267
+ graph = LadybugGraph.get(str(db_path))
4268
+
4269
+ # Read q from env var set by ResolvedOperatorConfig.subprocess_env()
4270
+ raw_q = os.environ.get("JAVA_CODEBASE_RAG_ABSENCE_NGRAM_Q", "3").strip()
4271
+ try:
4272
+ q = int(raw_q) if raw_q else 3
4273
+ except ValueError:
4274
+ q = 3 # Invalid env value falls back to default
4275
+ # Build index with configured q (or default 3)
4276
+ index = VocabularyIndex.build(graph, q=q)
4277
+
4278
+ # Save to sidecar next to the graph db
4279
+ sidecar_path = Path(db_path).parent / VOCAB_INDEX_FILENAME
4280
+ index.save(sidecar_path, ontology_version=ONTOLOGY_VERSION)
4281
+
4282
+ if verbose:
4283
+ _verbose_stderr_line(f"[vocab] index built with {index.symbol_count} symbols in {time.time() - t0:.2f}s")
4284
+
4285
+ except Exception as e:
4286
+ # Log but don't fail - graph build is the primary concern
4287
+ log.warning(f"Vocabulary index build failed (non-critical): {e}")
4288
+ if verbose:
4289
+ _verbose_stderr_line(f"[vocab] build failed (graph still written): {e}")
4290
+
4291
+
4214
4292
  # ---------- CLI ----------
4215
4293
 
4216
4294
 
@@ -19,12 +19,13 @@ Two location concepts are tracked per file:
19
19
  from __future__ import annotations
20
20
 
21
21
  import hashlib
22
+ import re
22
23
  import sys
23
24
  from dataclasses import dataclass, field, replace
24
25
  from functools import lru_cache
25
26
  from pathlib import Path
26
27
  from typing import Any, TypeVar
27
- from ast_java import (
28
+ from java_codebase_rag.ast.ast_java import (
28
29
  AnnotationRef,
29
30
  JavaFileAst,
30
31
  MethodDecl,
@@ -42,7 +43,7 @@ from ast_java import (
42
43
  _METHOD_ANN_TO_CAPABILITY,
43
44
  _TYPE_ANN_TO_CAPABILITY,
44
45
  )
45
- from java_ontology import (
46
+ from java_codebase_rag.graph.java_ontology import (
46
47
  CLIENT_KIND_REST_TEMPLATE,
47
48
  VALID_CAPABILITIES,
48
49
  VALID_CLIENT_KINDS,
@@ -51,7 +52,7 @@ from java_ontology import (
51
52
  VALID_ROUTE_FRAMEWORKS,
52
53
  VALID_ROUTE_KINDS,
53
54
  )
54
- from path_filtering import LayeredIgnore, iter_java_source_files
55
+ from java_codebase_rag.graph.path_filtering import LayeredIgnore, iter_java_source_files
55
56
 
56
57
  __all__ = [
57
58
  "AnnotationDecl",
@@ -141,6 +142,87 @@ def _load_config_microservice_roots(project_root_str: str) -> tuple[str, ...]:
141
142
  return ()
142
143
 
143
144
 
145
+ @lru_cache(maxsize=64)
146
+ def load_generated_detection(project_root_str: str | None) -> GeneratedDetectionConfig:
147
+ """Read `generated_detection` from `.java-codebase-rag.yml` at project_root.
148
+
149
+ Cached per project_root to avoid re-reading on every chunk. Returns empty
150
+ config when section absent or project_root is None. Malformed entries
151
+ (wrong types, non-string values) are dropped with a stderr warning.
152
+ """
153
+ if project_root_str is None:
154
+ return GeneratedDetectionConfig()
155
+
156
+ root = Path(project_root_str)
157
+ for name in CONFIG_FILENAMES:
158
+ candidate = root / name
159
+ if not candidate.is_file():
160
+ continue
161
+ try:
162
+ import yaml # PyYAML; already a transitive dep of cocoindex
163
+ except ImportError:
164
+ return GeneratedDetectionConfig()
165
+ try:
166
+ data = yaml.safe_load(candidate.read_text(encoding="utf-8"))
167
+ except Exception:
168
+ return GeneratedDetectionConfig()
169
+ if not isinstance(data, dict):
170
+ return GeneratedDetectionConfig()
171
+
172
+ raw = data.get("generated_detection")
173
+ if raw is None:
174
+ return GeneratedDetectionConfig()
175
+
176
+ if not isinstance(raw, dict):
177
+ import sys
178
+ print("[warn] generated_detection must be a dict; skipping",
179
+ file=sys.stderr)
180
+ return GeneratedDetectionConfig()
181
+
182
+ result = GeneratedDetectionConfig()
183
+
184
+ # Spec table: (config_key, field_name, type_converter, is_list_type)
185
+ # is_list_type: True = keep as list, False = convert to set
186
+ spec_table = [
187
+ ("header_patterns", "header_patterns", lambda x: x, True),
188
+ ("annotation_patterns", "annotation_patterns", lambda x: x, True),
189
+ ("force_fqns", "force_fqns", set, False),
190
+ ("exclude_fqns", "exclude_fqns", set, False),
191
+ ]
192
+
193
+ for config_key, field_name, type_conv, is_list_type in spec_table:
194
+ value = raw.get(config_key)
195
+ if value is not None:
196
+ if isinstance(value, list):
197
+ if is_list_type:
198
+ valid = [s for s in value if isinstance(s, str)]
199
+ else:
200
+ valid = {s for s in value if isinstance(s, str)}
201
+
202
+ if len(valid) != len(value):
203
+ import sys
204
+ print(f"[warn] generated_detection.{config_key}: "
205
+ "non-string entries dropped", file=sys.stderr)
206
+
207
+ # Update result with converted value
208
+ kwargs = {field_name: type_conv(valid)}
209
+ result = GeneratedDetectionConfig(
210
+ header_patterns=kwargs.get("header_patterns", result.header_patterns),
211
+ annotation_patterns=kwargs.get("annotation_patterns", result.annotation_patterns),
212
+ force_fqns=kwargs.get("force_fqns", result.force_fqns),
213
+ exclude_fqns=kwargs.get("exclude_fqns", result.exclude_fqns)
214
+ )
215
+ else:
216
+ import sys
217
+ print(f"[warn] generated_detection.{config_key}: "
218
+ "must be a list; skipping", file=sys.stderr)
219
+
220
+ return result
221
+
222
+ # No config file found → return empty config
223
+ return GeneratedDetectionConfig()
224
+
225
+
144
226
  @lru_cache(maxsize=64)
145
227
  def _load_config_cross_service_resolution(project_root_str: str) -> str:
146
228
  """Read `cross_service_resolution` from `.java-codebase-rag.yml` at project_root.
@@ -238,6 +320,19 @@ class BrownfieldOverrides:
238
320
  fqn_to_async_producer_hint: dict[str, AsyncProducerHint] = field(default_factory=dict)
239
321
 
240
322
 
323
+ @dataclass(frozen=True)
324
+ class GeneratedDetectionConfig:
325
+ """Config for generated-source detection.
326
+
327
+ Mirrors brownfield override pattern: frozen dataclass with
328
+ field(default_factory=...) for mutable defaults.
329
+ """
330
+ header_patterns: list[str] = field(default_factory=list)
331
+ annotation_patterns: list[str] = field(default_factory=list)
332
+ force_fqns: set[str] = field(default_factory=set)
333
+ exclude_fqns: set[str] = field(default_factory=set)
334
+
335
+
241
336
  def _meta_builtins() -> frozenset[str]:
242
337
  return (
243
338
  frozenset(ROLE_ANNOTATIONS)
@@ -1544,6 +1639,156 @@ def microservice_for_path(
1544
1639
  return ""
1545
1640
 
1546
1641
 
1642
+ # ---------- generated-source detection ----------
1643
+
1644
+
1645
+ # Built-in generator markers (v1 set). Verified against real generator output.
1646
+ # Annotation FQNs that mark generated code (any annotation with these FQNs
1647
+ # or simple names is considered a marker).
1648
+ _GENERATED_ANNOTATION_FQNS = {
1649
+ "javax.annotation.processing.Generated", # Standard Java (pre-Jakarta)
1650
+ "jakarta.annotation.processing.Generated", # Jakarta EE
1651
+ "org.immutables.value.Generated", # Immutables
1652
+ "lombok.Generated", # Lombok
1653
+ "com.squareup.javapoet.Generated", # JavaPoet
1654
+ }
1655
+
1656
+ # Header patterns for generators that emit banners (checked against first 4KB).
1657
+ # Patterns are compiled as case-insensitive regexes.
1658
+ _GENERATED_HEADER_PATTERNS = {
1659
+ re.compile(r"This file was generated by the OpenAPI Generator", re.IGNORECASE): "openapi",
1660
+ re.compile(r"Generated by the protocol buffer compiler", re.IGNORECASE): "protobuf",
1661
+ re.compile(r"This file was generated by jsonschema2pojo", re.IGNORECASE): "jsonschema2pojo",
1662
+ re.compile(r"generated by wsimport", re.IGNORECASE): "wsimport", # JAX-WS wsimport
1663
+ re.compile(r"WARNING: DO NOT EDIT.*generated by MapStruct", re.IGNORECASE): "mapstruct",
1664
+ }
1665
+
1666
+ # @Generated(value="...") patterns that identify the generator family.
1667
+ # These are matched against annotation arguments["value"] or arguments["comments"].
1668
+ _GENERATED_VALUE_PATTERNS = {
1669
+ re.compile(r"org\.openapitools\.codegen\."): "openapi",
1670
+ re.compile(r"org\.mapstruct\.ap\.MappingProcessor"): "mapstruct",
1671
+ re.compile(r"com\.google\.auto\.value\.processor\.AutoValueProcessor"): "autovalue",
1672
+ re.compile(r"org\.jooq\."): "jooq",
1673
+ re.compile(r"com\.querydsl\."): "querydsl",
1674
+ re.compile(r"org\.immutables\."): "immutables",
1675
+ }
1676
+
1677
+
1678
+ def _infer_family_from_annotation(annotation: AnnotationRef) -> str | None:
1679
+ """Infer generator family from @Generated annotation arguments.
1680
+
1681
+ Returns lowercased family slug or None if no match.
1682
+ """
1683
+ # Check value/comments arguments for generator identifiers FIRST
1684
+ # (handles javax.annotation.processing.Generated(value="org.mapstruct.ap.MappingProcessor"))
1685
+ value = annotation.arguments.get("value", "")
1686
+ comments = annotation.arguments.get("comments", "")
1687
+
1688
+ for pattern, family in _GENERATED_VALUE_PATTERNS.items():
1689
+ if pattern.search(value) or pattern.search(comments):
1690
+ return family
1691
+
1692
+ # Check annotation FQN for families identifiable by FQN itself
1693
+ if annotation.qualified in _GENERATED_ANNOTATION_FQNS:
1694
+ # Extract family from qualified name if possible
1695
+ if "lombok.Generated" in annotation.qualified:
1696
+ return "lombok"
1697
+ if "immutables" in annotation.qualified:
1698
+ return "immutables"
1699
+ # Generic javax/jakarta or JavaPoet -> unknown family
1700
+ return None
1701
+
1702
+ return None
1703
+
1704
+
1705
+ def _check_header_banners(header_text: str) -> str | None:
1706
+ """Check header (first 4KB) for generator banners.
1707
+
1708
+ Args:
1709
+ header_text: Decoded header text (first 4KB of source file).
1710
+
1711
+ Returns family slug if matched, None otherwise.
1712
+ """
1713
+ for pattern, family in _GENERATED_HEADER_PATTERNS.items():
1714
+ if pattern.search(header_text):
1715
+ return family
1716
+
1717
+ return None
1718
+
1719
+
1720
+ def classify_java_file(
1721
+ source: bytes,
1722
+ ast: "JavaFileAst",
1723
+ *,
1724
+ config: GeneratedDetectionConfig | None = None,
1725
+ project_root: str | Path | None = None,
1726
+ ) -> tuple[bool, str | None]:
1727
+ """Classify whether a Java source file is generated.
1728
+
1729
+ Args:
1730
+ source: Raw file bytes (required for header-banner detection).
1731
+ ast: Parsed Java AST (from ast_java.parse_java_ast).
1732
+ config: Optional detection config (defaults to empty config).
1733
+ project_root: Optional project root for loading default config.
1734
+
1735
+ Returns:
1736
+ (generated, generated_by) tuple:
1737
+ - generated: True if file is detected as generated code.
1738
+ - generated_by: Lowercased family slug (openapi, jsonschema2pojo,
1739
+ protobuf, mapstruct, wsimport, querydsl, jooq, immutables,
1740
+ autovalue, lombok) or None if family unknown.
1741
+ """
1742
+ # Load config if not provided
1743
+ if config is None:
1744
+ config = load_generated_detection(
1745
+ str(project_root) if project_root is not None else None
1746
+ )
1747
+
1748
+ # Decode header once for banner detection (first 4KB)
1749
+ HEADER_PREFIX_SIZE = 4096
1750
+ header_prefix = source[:HEADER_PREFIX_SIZE].decode("utf-8", errors="ignore")
1751
+
1752
+ # Collect all type FQNs for config-based checks
1753
+ all_type_fqns = {t.fqn for t in ast.all_types}
1754
+
1755
+ # Priority 1: exclude_fqns (override, wins even with markers)
1756
+ if all_type_fqns & config.exclude_fqns:
1757
+ return False, None
1758
+
1759
+ # Priority 2: force_fqns (forced generated, no markers needed)
1760
+ if all_type_fqns & config.force_fqns:
1761
+ return True, None
1762
+
1763
+ # Priority 3: annotation-based detection
1764
+ for typ in ast.all_types:
1765
+ for ann in typ.annotations:
1766
+ # Check simple name "Generated" first
1767
+ if ann.name == "Generated":
1768
+ family = _infer_family_from_annotation(ann)
1769
+ if family is not None:
1770
+ return True, family
1771
+ # Generic @Generated without identifiable value
1772
+ return True, None
1773
+
1774
+ # Check configured annotation patterns
1775
+ for pattern in config.annotation_patterns:
1776
+ if re.search(pattern, ann.qualified) or re.search(pattern, ann.name):
1777
+ return True, None
1778
+
1779
+ # Priority 4: header-banner detection
1780
+ family = _check_header_banners(header_prefix)
1781
+ if family is not None:
1782
+ return True, family
1783
+
1784
+ # Check configured header patterns
1785
+ for pattern in config.header_patterns:
1786
+ if re.search(pattern, header_prefix, re.IGNORECASE):
1787
+ return True, None
1788
+
1789
+ return False, None
1790
+
1791
+
1547
1792
  def detect_microservice_from_path(cwd: Path, source_root: Path) -> str | None:
1548
1793
  """Detect microservice from cwd for query-time auto-scope.
1549
1794
 
@@ -10,8 +10,8 @@ from typing import Any, Literal
10
10
 
11
11
  from pydantic import BaseModel
12
12
 
13
- from ladybug_queries import LadybugGraph
14
- from mcp_hints import generate_hints
13
+ from java_codebase_rag.graph.ladybug_queries import LadybugGraph
14
+ from java_codebase_rag.mcp.mcp_hints import generate_hints
15
15
 
16
16
  __all__ = [
17
17
  "NodeRef",
@@ -34,6 +34,8 @@ class NodeRef(BaseModel):
34
34
  microservice: str | None = None
35
35
  module: str | None = None
36
36
  role: str | None = None
37
+ generated: bool | None = None
38
+ generated_by: str | None = None
37
39
 
38
40
 
39
41
  class StructuredHint(BaseModel):
@@ -126,6 +128,8 @@ def _node_ref_from_row(kind: Literal["symbol", "route", "client", "producer"], r
126
128
  microservice=str(row.get("microservice") or "") or None,
127
129
  module=str(row.get("module") or "") or None,
128
130
  role=role,
131
+ generated=bool(row.get("generated")) if row.get("generated") is not None else None,
132
+ generated_by=str(row.get("generated_by")) if row.get("generated_by") else None,
129
133
  )
130
134
 
131
135
 
@@ -7,7 +7,7 @@ from __future__ import annotations
7
7
  from dataclasses import dataclass
8
8
  from typing import Literal
9
9
 
10
- from ast_java import (
10
+ from java_codebase_rag.ast.ast_java import (
11
11
  ROLE_ANNOTATIONS,
12
12
  _INJECTED_TYPES_TO_CAPABILITY,
13
13
  _METHOD_ANN_TO_CAPABILITY,
@@ -6,7 +6,7 @@ so the MCP server can serialize them without further mapping.
6
6
  The Ladybug database is opened read-only and cached per-process. This module is
7
7
  intentionally dependency-light: nothing here imports LanceDB or sentence-transformers.
8
8
 
9
- Cypher pitfalls (see also ``AGENTS.md``): avoid ``label(e) IN $list`` in ``WHERE`` for
9
+ Cypher pitfalls (see also ``CLAUDE.md``): avoid ``label(e) IN $list`` in ``WHERE`` for
10
10
  relationship-type filters; use OR of ``label(e) = $param`` with bound parameters.
11
11
  Typed unions ``-[e:A|B]-`` require every ``RETURN`` column on ``e`` to exist on all
12
12
  listed rel types, or the binder may fail.
@@ -24,7 +24,7 @@ from typing import Any, Literal
24
24
 
25
25
  import ladybug
26
26
 
27
- from ast_java import ONTOLOGY_VERSION as _ONTOLOGY_VERSION
27
+ from java_codebase_rag.ast.ast_java import ONTOLOGY_VERSION as _ONTOLOGY_VERSION
28
28
 
29
29
  log = logging.getLogger(__name__)
30
30
 
@@ -207,7 +207,7 @@ def _symbol_return_for(alias: str) -> str:
207
207
  f"{alias}.modifiers AS modifiers, {alias}.annotations AS annotations, "
208
208
  f"{alias}.capabilities AS capabilities, "
209
209
  f"{alias}.role AS role, {alias}.signature AS signature, "
210
- f"{alias}.parent_id AS parent_id, {alias}.resolved AS resolved"
210
+ f"{alias}.parent_id AS parent_id, {alias}.resolved AS resolved, {alias}.generated AS generated, {alias}.generated_by AS generated_by"
211
211
  )
212
212
 
213
213
 
@@ -296,7 +296,7 @@ def _row_to_symbol(row: dict[str, Any]) -> SymbolHit:
296
296
  _SYM_COLS = (
297
297
  "id", "kind", "name", "fqn", "package", "module", "microservice",
298
298
  "filename", "start_line", "end_line", "start_byte", "end_byte",
299
- "modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved",
299
+ "modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved", "generated", "generated_by",
300
300
  )
301
301
 
302
302
 
@@ -1107,14 +1107,14 @@ class LadybugGraph:
1107
1107
  f"s.{c} AS s_{c}" for c in (
1108
1108
  "id", "kind", "name", "fqn", "package", "module", "microservice",
1109
1109
  "filename", "start_line", "end_line", "start_byte", "end_byte",
1110
- "modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved",
1110
+ "modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved", "generated", "generated_by",
1111
1111
  )
1112
1112
  )
1113
1113
  t_proj = ", ".join(
1114
1114
  f"t.{c} AS t_{c}" for c in (
1115
1115
  "id", "kind", "name", "fqn", "package", "module", "microservice",
1116
1116
  "filename", "start_line", "end_line", "start_byte", "end_byte",
1117
- "modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved",
1117
+ "modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved", "generated", "generated_by",
1118
1118
  )
1119
1119
  )
1120
1120
  q = (
File without changes