java-codebase-rag 0.9.4__py3-none-any.whl → 0.9.6__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- java_codebase_rag/absence/__init__.py +0 -0
- java_codebase_rag/absence/absence_diagnosis.py +700 -0
- java_codebase_rag/absence/absence_types.py +124 -0
- java_codebase_rag/absence/absence_vocab.py +455 -0
- java_codebase_rag/analysis/__init__.py +0 -0
- pr_analysis.py → java_codebase_rag/analysis/pr_analysis.py +1 -1
- resolve_service.py → java_codebase_rag/analysis/resolve_service.py +73 -6
- java_codebase_rag/ast/__init__.py +0 -0
- ast_java.py → java_codebase_rag/ast/ast_java.py +5 -5
- java_codebase_rag/cli.py +13 -18
- java_codebase_rag/config.py +116 -0
- java_codebase_rag/graph/__init__.py +0 -0
- build_ast_graph.py → java_codebase_rag/graph/build_ast_graph.py +89 -11
- graph_enrich.py → java_codebase_rag/graph/graph_enrich.py +248 -3
- graph_types.py → java_codebase_rag/graph/graph_types.py +6 -2
- java_ontology.py → java_codebase_rag/graph/java_ontology.py +1 -1
- ladybug_queries.py → java_codebase_rag/graph/ladybug_queries.py +6 -6
- java_codebase_rag/index/__init__.py +0 -0
- java_index_flow_lancedb.py → java_codebase_rag/index/java_index_flow_lancedb.py +30 -10
- java_codebase_rag/install_data/__init__.py +0 -0
- java_codebase_rag/jrag.py +71 -16
- java_codebase_rag/jrag_envelope.py +13 -4
- java_codebase_rag/jrag_hints.py +1 -1
- java_codebase_rag/jrag_render.py +67 -3
- java_codebase_rag/mcp/__init__.py +0 -0
- mcp_hints.py → java_codebase_rag/mcp/mcp_hints.py +1 -1
- mcp_v2.py → java_codebase_rag/mcp/mcp_v2.py +280 -81
- server.py → java_codebase_rag/mcp/server.py +138 -54
- java_codebase_rag/pipeline.py +26 -7
- java_codebase_rag/search/__init__.py +0 -0
- search_lancedb.py → java_codebase_rag/search/search_lancedb.py +53 -314
- java_codebase_rag/search/search_lexical.py +329 -0
- java_codebase_rag/search/search_scoring.py +338 -0
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/METADATA +2 -2
- java_codebase_rag-0.9.6.dist-info/RECORD +57 -0
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/entry_points.txt +1 -1
- java_codebase_rag-0.9.6.dist-info/top_level.txt +1 -0
- java_codebase_rag-0.9.4.dist-info/RECORD +0 -44
- java_codebase_rag-0.9.4.dist-info/top_level.txt +0 -19
- /brownfield_events.py → /java_codebase_rag/ast/brownfield_events.py +0 -0
- /chunk_heuristics.py → /java_codebase_rag/ast/chunk_heuristics.py +0 -0
- /path_filtering.py → /java_codebase_rag/graph/path_filtering.py +0 -0
- /java_index_v1_common.py → /java_codebase_rag/index/java_index_v1_common.py +0 -0
- /index_common.py → /java_codebase_rag/search/index_common.py +0 -0
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/WHEEL +0 -0
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/licenses/LICENSE +0 -0
|
@@ -41,7 +41,7 @@ from pathlib import Path
|
|
|
41
41
|
import ladybug
|
|
42
42
|
import pyarrow as pa
|
|
43
43
|
|
|
44
|
-
from ast_java import (
|
|
44
|
+
from java_codebase_rag.ast.ast_java import (
|
|
45
45
|
ONTOLOGY_VERSION,
|
|
46
46
|
CallSite,
|
|
47
47
|
JavaFileAst,
|
|
@@ -52,10 +52,12 @@ from ast_java import (
|
|
|
52
52
|
lombok_required_args_annotations,
|
|
53
53
|
parse_java,
|
|
54
54
|
)
|
|
55
|
-
from graph_enrich import (
|
|
55
|
+
from java_codebase_rag.graph.graph_enrich import (
|
|
56
56
|
_load_config_cross_service_resolution,
|
|
57
|
+
classify_java_file,
|
|
57
58
|
collect_annotation_meta_chain,
|
|
58
59
|
load_brownfield_overrides,
|
|
60
|
+
load_generated_detection,
|
|
59
61
|
microservice_for_path,
|
|
60
62
|
module_for_path,
|
|
61
63
|
phantom_id,
|
|
@@ -65,8 +67,8 @@ from graph_enrich import (
|
|
|
65
67
|
resolve_routes_for_method,
|
|
66
68
|
symbol_id,
|
|
67
69
|
)
|
|
68
|
-
from path_filtering import LayeredIgnore, iter_java_source_files
|
|
69
|
-
from java_ontology import (
|
|
70
|
+
from java_codebase_rag.graph.path_filtering import LayeredIgnore, iter_java_source_files
|
|
71
|
+
from java_codebase_rag.graph.java_ontology import (
|
|
70
72
|
CLIENT_KIND_FEIGN_METHOD,
|
|
71
73
|
CLIENT_KIND_REST_TEMPLATE,
|
|
72
74
|
VALID_CLIENT_KINDS,
|
|
@@ -465,6 +467,8 @@ class GraphTables:
|
|
|
465
467
|
cross_service_resolution: str = "auto"
|
|
466
468
|
# Populated in _write_nodes (same overrides + meta_chain as Symbol.role).
|
|
467
469
|
type_role_by_node_id: dict[str, str] = field(default_factory=dict)
|
|
470
|
+
# Populated in pass 1 (classify_java_file) and _load_existing_types for incremental rebuilds.
|
|
471
|
+
type_generated_by_node_id: dict[str, tuple[bool, str | None]] = field(default_factory=dict)
|
|
468
472
|
|
|
469
473
|
|
|
470
474
|
@dataclass
|
|
@@ -598,7 +602,7 @@ def _load_existing_types(conn: ladybug.Connection, tables: GraphTables, exclude_
|
|
|
598
602
|
query = f"""
|
|
599
603
|
MATCH (s:Symbol)
|
|
600
604
|
{where}
|
|
601
|
-
RETURN s.kind, s.fqn, s.name, s.filename, s.module, s.microservice, s.id, s.role
|
|
605
|
+
RETURN s.kind, s.fqn, s.name, s.filename, s.module, s.microservice, s.id, s.role, s.generated, s.generated_by
|
|
602
606
|
"""
|
|
603
607
|
result = conn.execute(query, params)
|
|
604
608
|
while result.has_next():
|
|
@@ -608,6 +612,8 @@ def _load_existing_types(conn: ladybug.Connection, tables: GraphTables, exclude_
|
|
|
608
612
|
microservice = row[5] if len(row) > 5 else ""
|
|
609
613
|
node_id = row[6] if len(row) > 6 else ""
|
|
610
614
|
role = row[7] if len(row) > 7 else ""
|
|
615
|
+
generated = row[8] if len(row) > 8 else False
|
|
616
|
+
generated_by = row[9] if len(row) > 9 else ""
|
|
611
617
|
|
|
612
618
|
decl = TypeDecl(name, kind, fqn)
|
|
613
619
|
package = fqn[: -(len(name) + 1)] if fqn.endswith("." + name) else ""
|
|
@@ -629,6 +635,8 @@ def _load_existing_types(conn: ladybug.Connection, tables: GraphTables, exclude_
|
|
|
629
635
|
# the default during node staging (issue #352 divergence #2).
|
|
630
636
|
if role:
|
|
631
637
|
tables.type_role_by_node_id[node_id] = role
|
|
638
|
+
# Seed the persisted generated/generated_by so stubs retain their values
|
|
639
|
+
tables.type_generated_by_node_id[node_id] = (bool(generated), generated_by)
|
|
632
640
|
|
|
633
641
|
|
|
634
642
|
def _load_existing_members(conn: ladybug.Connection, tables: GraphTables, exclude_files: set[str] | None = None) -> None:
|
|
@@ -1061,6 +1069,12 @@ def pass1_parse(
|
|
|
1061
1069
|
microservice = microservice_for_path(str(p), root)
|
|
1062
1070
|
asts[rel] = ast
|
|
1063
1071
|
|
|
1072
|
+
# Classify the file once (generated or not, and which tool generated it)
|
|
1073
|
+
generated_config = load_generated_detection(str(root))
|
|
1074
|
+
file_generated, file_generated_by = classify_java_file(
|
|
1075
|
+
content, ast, config=generated_config, project_root=root
|
|
1076
|
+
)
|
|
1077
|
+
|
|
1064
1078
|
# file node
|
|
1065
1079
|
file_id = symbol_id("file", rel, rel, 0)
|
|
1066
1080
|
tables.files[rel] = file_id
|
|
@@ -1075,6 +1089,12 @@ def pass1_parse(
|
|
|
1075
1089
|
module=module, microservice=microservice, outer_fqn=None,
|
|
1076
1090
|
)
|
|
1077
1091
|
|
|
1092
|
+
# Seed generated/generated_by for all types in this file (including nested)
|
|
1093
|
+
for t in ast.all_types:
|
|
1094
|
+
if t.fqn in tables.types:
|
|
1095
|
+
node_id = tables.types[t.fqn].node_id
|
|
1096
|
+
tables.type_generated_by_node_id[node_id] = (file_generated, file_generated_by)
|
|
1097
|
+
|
|
1078
1098
|
if verbose:
|
|
1079
1099
|
elapsed = time.time() - t0
|
|
1080
1100
|
_emit_graph_progress(
|
|
@@ -2906,7 +2926,8 @@ _SCHEMA_NODE = (
|
|
|
2906
2926
|
"filename STRING, start_line INT64, end_line INT64, "
|
|
2907
2927
|
"start_byte INT64, end_byte INT64, "
|
|
2908
2928
|
"modifiers STRING[], annotations STRING[], capabilities STRING[], "
|
|
2909
|
-
"role STRING, signature STRING, parent_id STRING, resolved BOOLEAN"
|
|
2929
|
+
"role STRING, signature STRING, parent_id STRING, resolved BOOLEAN, "
|
|
2930
|
+
"generated BOOLEAN, generated_by STRING"
|
|
2910
2931
|
")"
|
|
2911
2932
|
)
|
|
2912
2933
|
|
|
@@ -3088,6 +3109,7 @@ def _node_row(**kwargs) -> dict:
|
|
|
3088
3109
|
"start_byte": 0, "end_byte": 0,
|
|
3089
3110
|
"modifiers": [], "annotations": [], "capabilities": [],
|
|
3090
3111
|
"role": "OTHER", "signature": "", "parent_id": "", "resolved": True,
|
|
3112
|
+
"generated": False, "generated_by": None,
|
|
3091
3113
|
}
|
|
3092
3114
|
base.update(kwargs)
|
|
3093
3115
|
return base
|
|
@@ -3138,7 +3160,8 @@ def _existing_node_ids(conn: ladybug.Connection) -> set[str]:
|
|
|
3138
3160
|
_NODE_COLUMNS = [
|
|
3139
3161
|
"id", "kind", "name", "fqn", "package", "module", "microservice",
|
|
3140
3162
|
"filename", "start_line", "end_line", "start_byte", "end_byte",
|
|
3141
|
-
"modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved"
|
|
3163
|
+
"modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved",
|
|
3164
|
+
"generated", "generated_by"
|
|
3142
3165
|
]
|
|
3143
3166
|
|
|
3144
3167
|
# Type declaration kinds. Tuple (not set) so the rendered SQL `IN` clause is
|
|
@@ -3161,7 +3184,8 @@ _SET_SYMBOL_BY_ID = (
|
|
|
3161
3184
|
"n.start_byte = $start_byte, n.end_byte = $end_byte, "
|
|
3162
3185
|
"n.modifiers = $modifiers, n.annotations = $annotations, "
|
|
3163
3186
|
"n.capabilities = $capabilities, n.role = $role, "
|
|
3164
|
-
"n.signature = $signature, n.parent_id = $parent_id, n.resolved = $resolved"
|
|
3187
|
+
"n.signature = $signature, n.parent_id = $parent_id, n.resolved = $resolved, "
|
|
3188
|
+
"n.generated = $generated, n.generated_by = $generated_by"
|
|
3165
3189
|
)
|
|
3166
3190
|
|
|
3167
3191
|
# Refresh every mutable Route field on an existing Route node by id. Mirrors the
|
|
@@ -3240,6 +3264,9 @@ def _write_nodes_impl(
|
|
|
3240
3264
|
overrides=overrides,
|
|
3241
3265
|
meta_chain=mch,
|
|
3242
3266
|
)
|
|
3267
|
+
# Read generated/generated_by from pass-1 classification or stub persistence
|
|
3268
|
+
generated, generated_by = tables.type_generated_by_node_id.get(entry.node_id, (False, None))
|
|
3269
|
+
|
|
3243
3270
|
if entry.loaded_from_db:
|
|
3244
3271
|
stub_ids.add(entry.node_id)
|
|
3245
3272
|
# Out-of-scope stub: its annotation-less decl collapses role to the
|
|
@@ -3250,6 +3277,7 @@ def _write_nodes_impl(
|
|
|
3250
3277
|
# capabilities placeholder never reaches the graph.
|
|
3251
3278
|
role = tables.type_role_by_node_id.get(entry.node_id, role)
|
|
3252
3279
|
capabilities = []
|
|
3280
|
+
# For stubs, trust the persisted generated/generated_by (seeded by _load_existing_types)
|
|
3253
3281
|
else:
|
|
3254
3282
|
tables.type_role_by_node_id[entry.node_id] = role
|
|
3255
3283
|
rows.append(_node_row(
|
|
@@ -3265,6 +3293,8 @@ def _write_nodes_impl(
|
|
|
3265
3293
|
role=role,
|
|
3266
3294
|
signature="",
|
|
3267
3295
|
parent_id=tables.types[entry.outer_fqn].node_id if entry.outer_fqn and entry.outer_fqn in tables.types else "",
|
|
3296
|
+
generated=generated,
|
|
3297
|
+
generated_by=generated_by,
|
|
3268
3298
|
))
|
|
3269
3299
|
# members (methods / constructors)
|
|
3270
3300
|
for m in tables.members:
|
|
@@ -3774,7 +3804,7 @@ def incremental_rebuild(
|
|
|
3774
3804
|
Returns IncrementalResult with statistics about the rebuild.
|
|
3775
3805
|
Falls back to full rebuild if:
|
|
3776
3806
|
- No previous graph exists
|
|
3777
|
-
- Ontology version <
|
|
3807
|
+
- Ontology version < ONTOLOGY_VERSION (stale schema; rebuild for current columns)
|
|
3778
3808
|
- Crash marker exists (previous incremental run failed)
|
|
3779
3809
|
- Dependent expansion exceeds expansion_cap
|
|
3780
3810
|
"""
|
|
@@ -3812,9 +3842,9 @@ def incremental_rebuild(
|
|
|
3812
3842
|
if meta_result.has_next():
|
|
3813
3843
|
row = meta_result.get_next()
|
|
3814
3844
|
version = row[0] if row else 0
|
|
3815
|
-
if version <
|
|
3845
|
+
if version < ONTOLOGY_VERSION:
|
|
3816
3846
|
if verbose:
|
|
3817
|
-
_verbose_stderr_line(f"[increment] ontology version {version} <
|
|
3847
|
+
_verbose_stderr_line(f"[increment] ontology version {version} < {ONTOLOGY_VERSION}; falling back to full rebuild")
|
|
3818
3848
|
conn.close()
|
|
3819
3849
|
db.close()
|
|
3820
3850
|
del conn, db
|
|
@@ -4208,9 +4238,57 @@ def write_ladybug(
|
|
|
4208
4238
|
_write_meta(conn, tables, source_root)
|
|
4209
4239
|
conn.close()
|
|
4210
4240
|
db.close()
|
|
4241
|
+
|
|
4242
|
+
# Build vocabulary index (best-effort, failure doesn't fail the graph build)
|
|
4243
|
+
_try_build_vocabulary_index(db_path, source_root, verbose)
|
|
4211
4244
|
_init_hash_tracker(source_root, db_path)
|
|
4212
4245
|
|
|
4213
4246
|
|
|
4247
|
+
def _try_build_vocabulary_index(db_path: Path, source_root: Path, verbose: bool) -> None:
|
|
4248
|
+
"""Build and save the vocabulary index as a sidecar (best-effort).
|
|
4249
|
+
|
|
4250
|
+
This is called after write_ladybug() completes. A build failure must not
|
|
4251
|
+
fail the graph build, so this is wrapped in try/except and logged.
|
|
4252
|
+
|
|
4253
|
+
Args:
|
|
4254
|
+
db_path: Path to the LadybugDB database file
|
|
4255
|
+
source_root: Source repository root
|
|
4256
|
+
verbose: Whether to emit verbose progress
|
|
4257
|
+
"""
|
|
4258
|
+
try:
|
|
4259
|
+
from java_codebase_rag.absence.absence_vocab import VocabularyIndex, VOCAB_INDEX_FILENAME
|
|
4260
|
+
from java_codebase_rag.graph.ladybug_queries import LadybugGraph
|
|
4261
|
+
|
|
4262
|
+
t0 = time.time()
|
|
4263
|
+
if verbose:
|
|
4264
|
+
_verbose_stderr_line("[vocab] building vocabulary index")
|
|
4265
|
+
|
|
4266
|
+
# Open graph for reading
|
|
4267
|
+
graph = LadybugGraph.get(str(db_path))
|
|
4268
|
+
|
|
4269
|
+
# Read q from env var set by ResolvedOperatorConfig.subprocess_env()
|
|
4270
|
+
raw_q = os.environ.get("JAVA_CODEBASE_RAG_ABSENCE_NGRAM_Q", "3").strip()
|
|
4271
|
+
try:
|
|
4272
|
+
q = int(raw_q) if raw_q else 3
|
|
4273
|
+
except ValueError:
|
|
4274
|
+
q = 3 # Invalid env value falls back to default
|
|
4275
|
+
# Build index with configured q (or default 3)
|
|
4276
|
+
index = VocabularyIndex.build(graph, q=q)
|
|
4277
|
+
|
|
4278
|
+
# Save to sidecar next to the graph db
|
|
4279
|
+
sidecar_path = Path(db_path).parent / VOCAB_INDEX_FILENAME
|
|
4280
|
+
index.save(sidecar_path, ontology_version=ONTOLOGY_VERSION)
|
|
4281
|
+
|
|
4282
|
+
if verbose:
|
|
4283
|
+
_verbose_stderr_line(f"[vocab] index built with {index.symbol_count} symbols in {time.time() - t0:.2f}s")
|
|
4284
|
+
|
|
4285
|
+
except Exception as e:
|
|
4286
|
+
# Log but don't fail - graph build is the primary concern
|
|
4287
|
+
log.warning(f"Vocabulary index build failed (non-critical): {e}")
|
|
4288
|
+
if verbose:
|
|
4289
|
+
_verbose_stderr_line(f"[vocab] build failed (graph still written): {e}")
|
|
4290
|
+
|
|
4291
|
+
|
|
4214
4292
|
# ---------- CLI ----------
|
|
4215
4293
|
|
|
4216
4294
|
|
|
@@ -19,12 +19,13 @@ Two location concepts are tracked per file:
|
|
|
19
19
|
from __future__ import annotations
|
|
20
20
|
|
|
21
21
|
import hashlib
|
|
22
|
+
import re
|
|
22
23
|
import sys
|
|
23
24
|
from dataclasses import dataclass, field, replace
|
|
24
25
|
from functools import lru_cache
|
|
25
26
|
from pathlib import Path
|
|
26
27
|
from typing import Any, TypeVar
|
|
27
|
-
from ast_java import (
|
|
28
|
+
from java_codebase_rag.ast.ast_java import (
|
|
28
29
|
AnnotationRef,
|
|
29
30
|
JavaFileAst,
|
|
30
31
|
MethodDecl,
|
|
@@ -42,7 +43,7 @@ from ast_java import (
|
|
|
42
43
|
_METHOD_ANN_TO_CAPABILITY,
|
|
43
44
|
_TYPE_ANN_TO_CAPABILITY,
|
|
44
45
|
)
|
|
45
|
-
from java_ontology import (
|
|
46
|
+
from java_codebase_rag.graph.java_ontology import (
|
|
46
47
|
CLIENT_KIND_REST_TEMPLATE,
|
|
47
48
|
VALID_CAPABILITIES,
|
|
48
49
|
VALID_CLIENT_KINDS,
|
|
@@ -51,7 +52,7 @@ from java_ontology import (
|
|
|
51
52
|
VALID_ROUTE_FRAMEWORKS,
|
|
52
53
|
VALID_ROUTE_KINDS,
|
|
53
54
|
)
|
|
54
|
-
from path_filtering import LayeredIgnore, iter_java_source_files
|
|
55
|
+
from java_codebase_rag.graph.path_filtering import LayeredIgnore, iter_java_source_files
|
|
55
56
|
|
|
56
57
|
__all__ = [
|
|
57
58
|
"AnnotationDecl",
|
|
@@ -141,6 +142,87 @@ def _load_config_microservice_roots(project_root_str: str) -> tuple[str, ...]:
|
|
|
141
142
|
return ()
|
|
142
143
|
|
|
143
144
|
|
|
145
|
+
@lru_cache(maxsize=64)
|
|
146
|
+
def load_generated_detection(project_root_str: str | None) -> GeneratedDetectionConfig:
|
|
147
|
+
"""Read `generated_detection` from `.java-codebase-rag.yml` at project_root.
|
|
148
|
+
|
|
149
|
+
Cached per project_root to avoid re-reading on every chunk. Returns empty
|
|
150
|
+
config when section absent or project_root is None. Malformed entries
|
|
151
|
+
(wrong types, non-string values) are dropped with a stderr warning.
|
|
152
|
+
"""
|
|
153
|
+
if project_root_str is None:
|
|
154
|
+
return GeneratedDetectionConfig()
|
|
155
|
+
|
|
156
|
+
root = Path(project_root_str)
|
|
157
|
+
for name in CONFIG_FILENAMES:
|
|
158
|
+
candidate = root / name
|
|
159
|
+
if not candidate.is_file():
|
|
160
|
+
continue
|
|
161
|
+
try:
|
|
162
|
+
import yaml # PyYAML; already a transitive dep of cocoindex
|
|
163
|
+
except ImportError:
|
|
164
|
+
return GeneratedDetectionConfig()
|
|
165
|
+
try:
|
|
166
|
+
data = yaml.safe_load(candidate.read_text(encoding="utf-8"))
|
|
167
|
+
except Exception:
|
|
168
|
+
return GeneratedDetectionConfig()
|
|
169
|
+
if not isinstance(data, dict):
|
|
170
|
+
return GeneratedDetectionConfig()
|
|
171
|
+
|
|
172
|
+
raw = data.get("generated_detection")
|
|
173
|
+
if raw is None:
|
|
174
|
+
return GeneratedDetectionConfig()
|
|
175
|
+
|
|
176
|
+
if not isinstance(raw, dict):
|
|
177
|
+
import sys
|
|
178
|
+
print("[warn] generated_detection must be a dict; skipping",
|
|
179
|
+
file=sys.stderr)
|
|
180
|
+
return GeneratedDetectionConfig()
|
|
181
|
+
|
|
182
|
+
result = GeneratedDetectionConfig()
|
|
183
|
+
|
|
184
|
+
# Spec table: (config_key, field_name, type_converter, is_list_type)
|
|
185
|
+
# is_list_type: True = keep as list, False = convert to set
|
|
186
|
+
spec_table = [
|
|
187
|
+
("header_patterns", "header_patterns", lambda x: x, True),
|
|
188
|
+
("annotation_patterns", "annotation_patterns", lambda x: x, True),
|
|
189
|
+
("force_fqns", "force_fqns", set, False),
|
|
190
|
+
("exclude_fqns", "exclude_fqns", set, False),
|
|
191
|
+
]
|
|
192
|
+
|
|
193
|
+
for config_key, field_name, type_conv, is_list_type in spec_table:
|
|
194
|
+
value = raw.get(config_key)
|
|
195
|
+
if value is not None:
|
|
196
|
+
if isinstance(value, list):
|
|
197
|
+
if is_list_type:
|
|
198
|
+
valid = [s for s in value if isinstance(s, str)]
|
|
199
|
+
else:
|
|
200
|
+
valid = {s for s in value if isinstance(s, str)}
|
|
201
|
+
|
|
202
|
+
if len(valid) != len(value):
|
|
203
|
+
import sys
|
|
204
|
+
print(f"[warn] generated_detection.{config_key}: "
|
|
205
|
+
"non-string entries dropped", file=sys.stderr)
|
|
206
|
+
|
|
207
|
+
# Update result with converted value
|
|
208
|
+
kwargs = {field_name: type_conv(valid)}
|
|
209
|
+
result = GeneratedDetectionConfig(
|
|
210
|
+
header_patterns=kwargs.get("header_patterns", result.header_patterns),
|
|
211
|
+
annotation_patterns=kwargs.get("annotation_patterns", result.annotation_patterns),
|
|
212
|
+
force_fqns=kwargs.get("force_fqns", result.force_fqns),
|
|
213
|
+
exclude_fqns=kwargs.get("exclude_fqns", result.exclude_fqns)
|
|
214
|
+
)
|
|
215
|
+
else:
|
|
216
|
+
import sys
|
|
217
|
+
print(f"[warn] generated_detection.{config_key}: "
|
|
218
|
+
"must be a list; skipping", file=sys.stderr)
|
|
219
|
+
|
|
220
|
+
return result
|
|
221
|
+
|
|
222
|
+
# No config file found → return empty config
|
|
223
|
+
return GeneratedDetectionConfig()
|
|
224
|
+
|
|
225
|
+
|
|
144
226
|
@lru_cache(maxsize=64)
|
|
145
227
|
def _load_config_cross_service_resolution(project_root_str: str) -> str:
|
|
146
228
|
"""Read `cross_service_resolution` from `.java-codebase-rag.yml` at project_root.
|
|
@@ -238,6 +320,19 @@ class BrownfieldOverrides:
|
|
|
238
320
|
fqn_to_async_producer_hint: dict[str, AsyncProducerHint] = field(default_factory=dict)
|
|
239
321
|
|
|
240
322
|
|
|
323
|
+
@dataclass(frozen=True)
|
|
324
|
+
class GeneratedDetectionConfig:
|
|
325
|
+
"""Config for generated-source detection.
|
|
326
|
+
|
|
327
|
+
Mirrors brownfield override pattern: frozen dataclass with
|
|
328
|
+
field(default_factory=...) for mutable defaults.
|
|
329
|
+
"""
|
|
330
|
+
header_patterns: list[str] = field(default_factory=list)
|
|
331
|
+
annotation_patterns: list[str] = field(default_factory=list)
|
|
332
|
+
force_fqns: set[str] = field(default_factory=set)
|
|
333
|
+
exclude_fqns: set[str] = field(default_factory=set)
|
|
334
|
+
|
|
335
|
+
|
|
241
336
|
def _meta_builtins() -> frozenset[str]:
|
|
242
337
|
return (
|
|
243
338
|
frozenset(ROLE_ANNOTATIONS)
|
|
@@ -1544,6 +1639,156 @@ def microservice_for_path(
|
|
|
1544
1639
|
return ""
|
|
1545
1640
|
|
|
1546
1641
|
|
|
1642
|
+
# ---------- generated-source detection ----------
|
|
1643
|
+
|
|
1644
|
+
|
|
1645
|
+
# Built-in generator markers (v1 set). Verified against real generator output.
|
|
1646
|
+
# Annotation FQNs that mark generated code (any annotation with these FQNs
|
|
1647
|
+
# or simple names is considered a marker).
|
|
1648
|
+
_GENERATED_ANNOTATION_FQNS = {
|
|
1649
|
+
"javax.annotation.processing.Generated", # Standard Java (pre-Jakarta)
|
|
1650
|
+
"jakarta.annotation.processing.Generated", # Jakarta EE
|
|
1651
|
+
"org.immutables.value.Generated", # Immutables
|
|
1652
|
+
"lombok.Generated", # Lombok
|
|
1653
|
+
"com.squareup.javapoet.Generated", # JavaPoet
|
|
1654
|
+
}
|
|
1655
|
+
|
|
1656
|
+
# Header patterns for generators that emit banners (checked against first 4KB).
|
|
1657
|
+
# Patterns are compiled as case-insensitive regexes.
|
|
1658
|
+
_GENERATED_HEADER_PATTERNS = {
|
|
1659
|
+
re.compile(r"This file was generated by the OpenAPI Generator", re.IGNORECASE): "openapi",
|
|
1660
|
+
re.compile(r"Generated by the protocol buffer compiler", re.IGNORECASE): "protobuf",
|
|
1661
|
+
re.compile(r"This file was generated by jsonschema2pojo", re.IGNORECASE): "jsonschema2pojo",
|
|
1662
|
+
re.compile(r"generated by wsimport", re.IGNORECASE): "wsimport", # JAX-WS wsimport
|
|
1663
|
+
re.compile(r"WARNING: DO NOT EDIT.*generated by MapStruct", re.IGNORECASE): "mapstruct",
|
|
1664
|
+
}
|
|
1665
|
+
|
|
1666
|
+
# @Generated(value="...") patterns that identify the generator family.
|
|
1667
|
+
# These are matched against annotation arguments["value"] or arguments["comments"].
|
|
1668
|
+
_GENERATED_VALUE_PATTERNS = {
|
|
1669
|
+
re.compile(r"org\.openapitools\.codegen\."): "openapi",
|
|
1670
|
+
re.compile(r"org\.mapstruct\.ap\.MappingProcessor"): "mapstruct",
|
|
1671
|
+
re.compile(r"com\.google\.auto\.value\.processor\.AutoValueProcessor"): "autovalue",
|
|
1672
|
+
re.compile(r"org\.jooq\."): "jooq",
|
|
1673
|
+
re.compile(r"com\.querydsl\."): "querydsl",
|
|
1674
|
+
re.compile(r"org\.immutables\."): "immutables",
|
|
1675
|
+
}
|
|
1676
|
+
|
|
1677
|
+
|
|
1678
|
+
def _infer_family_from_annotation(annotation: AnnotationRef) -> str | None:
|
|
1679
|
+
"""Infer generator family from @Generated annotation arguments.
|
|
1680
|
+
|
|
1681
|
+
Returns lowercased family slug or None if no match.
|
|
1682
|
+
"""
|
|
1683
|
+
# Check value/comments arguments for generator identifiers FIRST
|
|
1684
|
+
# (handles javax.annotation.processing.Generated(value="org.mapstruct.ap.MappingProcessor"))
|
|
1685
|
+
value = annotation.arguments.get("value", "")
|
|
1686
|
+
comments = annotation.arguments.get("comments", "")
|
|
1687
|
+
|
|
1688
|
+
for pattern, family in _GENERATED_VALUE_PATTERNS.items():
|
|
1689
|
+
if pattern.search(value) or pattern.search(comments):
|
|
1690
|
+
return family
|
|
1691
|
+
|
|
1692
|
+
# Check annotation FQN for families identifiable by FQN itself
|
|
1693
|
+
if annotation.qualified in _GENERATED_ANNOTATION_FQNS:
|
|
1694
|
+
# Extract family from qualified name if possible
|
|
1695
|
+
if "lombok.Generated" in annotation.qualified:
|
|
1696
|
+
return "lombok"
|
|
1697
|
+
if "immutables" in annotation.qualified:
|
|
1698
|
+
return "immutables"
|
|
1699
|
+
# Generic javax/jakarta or JavaPoet -> unknown family
|
|
1700
|
+
return None
|
|
1701
|
+
|
|
1702
|
+
return None
|
|
1703
|
+
|
|
1704
|
+
|
|
1705
|
+
def _check_header_banners(header_text: str) -> str | None:
|
|
1706
|
+
"""Check header (first 4KB) for generator banners.
|
|
1707
|
+
|
|
1708
|
+
Args:
|
|
1709
|
+
header_text: Decoded header text (first 4KB of source file).
|
|
1710
|
+
|
|
1711
|
+
Returns family slug if matched, None otherwise.
|
|
1712
|
+
"""
|
|
1713
|
+
for pattern, family in _GENERATED_HEADER_PATTERNS.items():
|
|
1714
|
+
if pattern.search(header_text):
|
|
1715
|
+
return family
|
|
1716
|
+
|
|
1717
|
+
return None
|
|
1718
|
+
|
|
1719
|
+
|
|
1720
|
+
def classify_java_file(
|
|
1721
|
+
source: bytes,
|
|
1722
|
+
ast: "JavaFileAst",
|
|
1723
|
+
*,
|
|
1724
|
+
config: GeneratedDetectionConfig | None = None,
|
|
1725
|
+
project_root: str | Path | None = None,
|
|
1726
|
+
) -> tuple[bool, str | None]:
|
|
1727
|
+
"""Classify whether a Java source file is generated.
|
|
1728
|
+
|
|
1729
|
+
Args:
|
|
1730
|
+
source: Raw file bytes (required for header-banner detection).
|
|
1731
|
+
ast: Parsed Java AST (from ast_java.parse_java_ast).
|
|
1732
|
+
config: Optional detection config (defaults to empty config).
|
|
1733
|
+
project_root: Optional project root for loading default config.
|
|
1734
|
+
|
|
1735
|
+
Returns:
|
|
1736
|
+
(generated, generated_by) tuple:
|
|
1737
|
+
- generated: True if file is detected as generated code.
|
|
1738
|
+
- generated_by: Lowercased family slug (openapi, jsonschema2pojo,
|
|
1739
|
+
protobuf, mapstruct, wsimport, querydsl, jooq, immutables,
|
|
1740
|
+
autovalue, lombok) or None if family unknown.
|
|
1741
|
+
"""
|
|
1742
|
+
# Load config if not provided
|
|
1743
|
+
if config is None:
|
|
1744
|
+
config = load_generated_detection(
|
|
1745
|
+
str(project_root) if project_root is not None else None
|
|
1746
|
+
)
|
|
1747
|
+
|
|
1748
|
+
# Decode header once for banner detection (first 4KB)
|
|
1749
|
+
HEADER_PREFIX_SIZE = 4096
|
|
1750
|
+
header_prefix = source[:HEADER_PREFIX_SIZE].decode("utf-8", errors="ignore")
|
|
1751
|
+
|
|
1752
|
+
# Collect all type FQNs for config-based checks
|
|
1753
|
+
all_type_fqns = {t.fqn for t in ast.all_types}
|
|
1754
|
+
|
|
1755
|
+
# Priority 1: exclude_fqns (override, wins even with markers)
|
|
1756
|
+
if all_type_fqns & config.exclude_fqns:
|
|
1757
|
+
return False, None
|
|
1758
|
+
|
|
1759
|
+
# Priority 2: force_fqns (forced generated, no markers needed)
|
|
1760
|
+
if all_type_fqns & config.force_fqns:
|
|
1761
|
+
return True, None
|
|
1762
|
+
|
|
1763
|
+
# Priority 3: annotation-based detection
|
|
1764
|
+
for typ in ast.all_types:
|
|
1765
|
+
for ann in typ.annotations:
|
|
1766
|
+
# Check simple name "Generated" first
|
|
1767
|
+
if ann.name == "Generated":
|
|
1768
|
+
family = _infer_family_from_annotation(ann)
|
|
1769
|
+
if family is not None:
|
|
1770
|
+
return True, family
|
|
1771
|
+
# Generic @Generated without identifiable value
|
|
1772
|
+
return True, None
|
|
1773
|
+
|
|
1774
|
+
# Check configured annotation patterns
|
|
1775
|
+
for pattern in config.annotation_patterns:
|
|
1776
|
+
if re.search(pattern, ann.qualified) or re.search(pattern, ann.name):
|
|
1777
|
+
return True, None
|
|
1778
|
+
|
|
1779
|
+
# Priority 4: header-banner detection
|
|
1780
|
+
family = _check_header_banners(header_prefix)
|
|
1781
|
+
if family is not None:
|
|
1782
|
+
return True, family
|
|
1783
|
+
|
|
1784
|
+
# Check configured header patterns
|
|
1785
|
+
for pattern in config.header_patterns:
|
|
1786
|
+
if re.search(pattern, header_prefix, re.IGNORECASE):
|
|
1787
|
+
return True, None
|
|
1788
|
+
|
|
1789
|
+
return False, None
|
|
1790
|
+
|
|
1791
|
+
|
|
1547
1792
|
def detect_microservice_from_path(cwd: Path, source_root: Path) -> str | None:
|
|
1548
1793
|
"""Detect microservice from cwd for query-time auto-scope.
|
|
1549
1794
|
|
|
@@ -10,8 +10,8 @@ from typing import Any, Literal
|
|
|
10
10
|
|
|
11
11
|
from pydantic import BaseModel
|
|
12
12
|
|
|
13
|
-
from ladybug_queries import LadybugGraph
|
|
14
|
-
from mcp_hints import generate_hints
|
|
13
|
+
from java_codebase_rag.graph.ladybug_queries import LadybugGraph
|
|
14
|
+
from java_codebase_rag.mcp.mcp_hints import generate_hints
|
|
15
15
|
|
|
16
16
|
__all__ = [
|
|
17
17
|
"NodeRef",
|
|
@@ -34,6 +34,8 @@ class NodeRef(BaseModel):
|
|
|
34
34
|
microservice: str | None = None
|
|
35
35
|
module: str | None = None
|
|
36
36
|
role: str | None = None
|
|
37
|
+
generated: bool | None = None
|
|
38
|
+
generated_by: str | None = None
|
|
37
39
|
|
|
38
40
|
|
|
39
41
|
class StructuredHint(BaseModel):
|
|
@@ -126,6 +128,8 @@ def _node_ref_from_row(kind: Literal["symbol", "route", "client", "producer"], r
|
|
|
126
128
|
microservice=str(row.get("microservice") or "") or None,
|
|
127
129
|
module=str(row.get("module") or "") or None,
|
|
128
130
|
role=role,
|
|
131
|
+
generated=bool(row.get("generated")) if row.get("generated") is not None else None,
|
|
132
|
+
generated_by=str(row.get("generated_by")) if row.get("generated_by") else None,
|
|
129
133
|
)
|
|
130
134
|
|
|
131
135
|
|
|
@@ -6,7 +6,7 @@ so the MCP server can serialize them without further mapping.
|
|
|
6
6
|
The Ladybug database is opened read-only and cached per-process. This module is
|
|
7
7
|
intentionally dependency-light: nothing here imports LanceDB or sentence-transformers.
|
|
8
8
|
|
|
9
|
-
Cypher pitfalls (see also ``
|
|
9
|
+
Cypher pitfalls (see also ``CLAUDE.md``): avoid ``label(e) IN $list`` in ``WHERE`` for
|
|
10
10
|
relationship-type filters; use OR of ``label(e) = $param`` with bound parameters.
|
|
11
11
|
Typed unions ``-[e:A|B]-`` require every ``RETURN`` column on ``e`` to exist on all
|
|
12
12
|
listed rel types, or the binder may fail.
|
|
@@ -24,7 +24,7 @@ from typing import Any, Literal
|
|
|
24
24
|
|
|
25
25
|
import ladybug
|
|
26
26
|
|
|
27
|
-
from ast_java import ONTOLOGY_VERSION as _ONTOLOGY_VERSION
|
|
27
|
+
from java_codebase_rag.ast.ast_java import ONTOLOGY_VERSION as _ONTOLOGY_VERSION
|
|
28
28
|
|
|
29
29
|
log = logging.getLogger(__name__)
|
|
30
30
|
|
|
@@ -207,7 +207,7 @@ def _symbol_return_for(alias: str) -> str:
|
|
|
207
207
|
f"{alias}.modifiers AS modifiers, {alias}.annotations AS annotations, "
|
|
208
208
|
f"{alias}.capabilities AS capabilities, "
|
|
209
209
|
f"{alias}.role AS role, {alias}.signature AS signature, "
|
|
210
|
-
f"{alias}.parent_id AS parent_id, {alias}.resolved AS resolved"
|
|
210
|
+
f"{alias}.parent_id AS parent_id, {alias}.resolved AS resolved, {alias}.generated AS generated, {alias}.generated_by AS generated_by"
|
|
211
211
|
)
|
|
212
212
|
|
|
213
213
|
|
|
@@ -296,7 +296,7 @@ def _row_to_symbol(row: dict[str, Any]) -> SymbolHit:
|
|
|
296
296
|
_SYM_COLS = (
|
|
297
297
|
"id", "kind", "name", "fqn", "package", "module", "microservice",
|
|
298
298
|
"filename", "start_line", "end_line", "start_byte", "end_byte",
|
|
299
|
-
"modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved",
|
|
299
|
+
"modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved", "generated", "generated_by",
|
|
300
300
|
)
|
|
301
301
|
|
|
302
302
|
|
|
@@ -1107,14 +1107,14 @@ class LadybugGraph:
|
|
|
1107
1107
|
f"s.{c} AS s_{c}" for c in (
|
|
1108
1108
|
"id", "kind", "name", "fqn", "package", "module", "microservice",
|
|
1109
1109
|
"filename", "start_line", "end_line", "start_byte", "end_byte",
|
|
1110
|
-
"modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved",
|
|
1110
|
+
"modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved", "generated", "generated_by",
|
|
1111
1111
|
)
|
|
1112
1112
|
)
|
|
1113
1113
|
t_proj = ", ".join(
|
|
1114
1114
|
f"t.{c} AS t_{c}" for c in (
|
|
1115
1115
|
"id", "kind", "name", "fqn", "package", "module", "microservice",
|
|
1116
1116
|
"filename", "start_line", "end_line", "start_byte", "end_byte",
|
|
1117
|
-
"modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved",
|
|
1117
|
+
"modifiers", "annotations", "capabilities", "role", "signature", "parent_id", "resolved", "generated", "generated_by",
|
|
1118
1118
|
)
|
|
1119
1119
|
)
|
|
1120
1120
|
q = (
|
|
File without changes
|