java-codebase-rag 0.9.4__py3-none-any.whl → 0.9.6__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- java_codebase_rag/absence/__init__.py +0 -0
- java_codebase_rag/absence/absence_diagnosis.py +700 -0
- java_codebase_rag/absence/absence_types.py +124 -0
- java_codebase_rag/absence/absence_vocab.py +455 -0
- java_codebase_rag/analysis/__init__.py +0 -0
- pr_analysis.py → java_codebase_rag/analysis/pr_analysis.py +1 -1
- resolve_service.py → java_codebase_rag/analysis/resolve_service.py +73 -6
- java_codebase_rag/ast/__init__.py +0 -0
- ast_java.py → java_codebase_rag/ast/ast_java.py +5 -5
- java_codebase_rag/cli.py +13 -18
- java_codebase_rag/config.py +116 -0
- java_codebase_rag/graph/__init__.py +0 -0
- build_ast_graph.py → java_codebase_rag/graph/build_ast_graph.py +89 -11
- graph_enrich.py → java_codebase_rag/graph/graph_enrich.py +248 -3
- graph_types.py → java_codebase_rag/graph/graph_types.py +6 -2
- java_ontology.py → java_codebase_rag/graph/java_ontology.py +1 -1
- ladybug_queries.py → java_codebase_rag/graph/ladybug_queries.py +6 -6
- java_codebase_rag/index/__init__.py +0 -0
- java_index_flow_lancedb.py → java_codebase_rag/index/java_index_flow_lancedb.py +30 -10
- java_codebase_rag/install_data/__init__.py +0 -0
- java_codebase_rag/jrag.py +71 -16
- java_codebase_rag/jrag_envelope.py +13 -4
- java_codebase_rag/jrag_hints.py +1 -1
- java_codebase_rag/jrag_render.py +67 -3
- java_codebase_rag/mcp/__init__.py +0 -0
- mcp_hints.py → java_codebase_rag/mcp/mcp_hints.py +1 -1
- mcp_v2.py → java_codebase_rag/mcp/mcp_v2.py +280 -81
- server.py → java_codebase_rag/mcp/server.py +138 -54
- java_codebase_rag/pipeline.py +26 -7
- java_codebase_rag/search/__init__.py +0 -0
- search_lancedb.py → java_codebase_rag/search/search_lancedb.py +53 -314
- java_codebase_rag/search/search_lexical.py +329 -0
- java_codebase_rag/search/search_scoring.py +338 -0
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/METADATA +2 -2
- java_codebase_rag-0.9.6.dist-info/RECORD +57 -0
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/entry_points.txt +1 -1
- java_codebase_rag-0.9.6.dist-info/top_level.txt +1 -0
- java_codebase_rag-0.9.4.dist-info/RECORD +0 -44
- java_codebase_rag-0.9.4.dist-info/top_level.txt +0 -19
- /brownfield_events.py → /java_codebase_rag/ast/brownfield_events.py +0 -0
- /chunk_heuristics.py → /java_codebase_rag/ast/chunk_heuristics.py +0 -0
- /path_filtering.py → /java_codebase_rag/graph/path_filtering.py +0 -0
- /java_index_v1_common.py → /java_codebase_rag/index/java_index_v1_common.py +0 -0
- /index_common.py → /java_codebase_rag/search/index_common.py +0 -0
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/WHEEL +0 -0
- {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/licenses/LICENSE +0 -0
|
@@ -41,7 +41,7 @@ from cocoindex.resources.file import PatternFilePathMatcher
|
|
|
41
41
|
|
|
42
42
|
from java_codebase_rag.config import resolved_sbert_model_for_process_env
|
|
43
43
|
from java_codebase_rag.lance_optimize import LANCE_TABLE_NAMES
|
|
44
|
-
from java_index_v1_common import (
|
|
44
|
+
from java_codebase_rag.index.java_index_v1_common import (
|
|
45
45
|
JAVA_CHUNK,
|
|
46
46
|
SBERT_MODEL,
|
|
47
47
|
SQL_CHUNK,
|
|
@@ -49,9 +49,15 @@ from java_index_v1_common import (
|
|
|
49
49
|
chunk_key_range,
|
|
50
50
|
position_to_json,
|
|
51
51
|
)
|
|
52
|
-
from path_filtering import LayeredIgnore
|
|
53
|
-
from ast_java import ONTOLOGY_VERSION, parse_java
|
|
54
|
-
from graph_enrich import
|
|
52
|
+
from java_codebase_rag.graph.path_filtering import LayeredIgnore
|
|
53
|
+
from java_codebase_rag.ast.ast_java import ONTOLOGY_VERSION, parse_java
|
|
54
|
+
from java_codebase_rag.graph.graph_enrich import (
|
|
55
|
+
classify_java_file,
|
|
56
|
+
collect_annotation_meta_chain,
|
|
57
|
+
enrich_chunk,
|
|
58
|
+
load_brownfield_overrides,
|
|
59
|
+
load_generated_detection,
|
|
60
|
+
)
|
|
55
61
|
|
|
56
62
|
# Older cocoindex (e.g. 1.0.0a43) uses ``tracked=False``; newer releases renamed
|
|
57
63
|
# the flag to ``detect_change`` (default False) and reject ``tracked``.
|
|
@@ -280,6 +286,9 @@ class JavaLanceChunk:
|
|
|
280
286
|
annotations_on_type: Annotated[list[str], LanceType(pa.list_(pa.string()))]
|
|
281
287
|
symbols: Annotated[list[str], LanceType(pa.list_(pa.string()))]
|
|
282
288
|
ontology_version: int
|
|
289
|
+
# Generated source detection: populated per-file, not per-chunk
|
|
290
|
+
generated: bool
|
|
291
|
+
generated_by: str | None
|
|
283
292
|
|
|
284
293
|
|
|
285
294
|
@dataclass
|
|
@@ -364,12 +373,13 @@ def _parse_and_enrich_java(
|
|
|
364
373
|
chunks: list[Any],
|
|
365
374
|
rel: str,
|
|
366
375
|
project_root: Path,
|
|
367
|
-
) -> list[Any]:
|
|
376
|
+
) -> tuple[list[Any], Any]:
|
|
368
377
|
"""Parse one Java file and enrich every chunk, off the event loop.
|
|
369
378
|
|
|
370
|
-
Returns a
|
|
371
|
-
|
|
372
|
-
|
|
379
|
+
Returns a tuple of (enrichments, ast) where enrichments is a list of
|
|
380
|
+
:class:`graph_enrich.ChunkEnrichment` aligned 1:1 with ``chunks``, and ast
|
|
381
|
+
is the parsed :class:`JavaFileAst`. Intended to run via ``asyncio.to_thread``
|
|
382
|
+
from ``process_java_file`` (vectors perf lever #2): while the worker thread
|
|
373
383
|
parses + enriches, the event loop is free to drive other files and keep the
|
|
374
384
|
embedder's batching queue fed.
|
|
375
385
|
|
|
@@ -381,7 +391,7 @@ def _parse_and_enrich_java(
|
|
|
381
391
|
``lru_cache`` reads are thread-safe under the GIL.
|
|
382
392
|
"""
|
|
383
393
|
ast = parse_java(content_bytes)
|
|
384
|
-
|
|
394
|
+
enrichments = [
|
|
385
395
|
enrich_chunk(
|
|
386
396
|
ast,
|
|
387
397
|
chunk_start_byte=ch.start.byte_offset,
|
|
@@ -391,6 +401,7 @@ def _parse_and_enrich_java(
|
|
|
391
401
|
)
|
|
392
402
|
for ch in chunks
|
|
393
403
|
]
|
|
404
|
+
return enrichments, ast
|
|
394
405
|
|
|
395
406
|
|
|
396
407
|
@coco.fn(memo=True)
|
|
@@ -430,9 +441,16 @@ async def process_java_file(
|
|
|
430
441
|
# (vectors perf lever #2) parse + enrich off the event loop so the loop can
|
|
431
442
|
# keep the embedder's batching queue fed while this file is being parsed.
|
|
432
443
|
# parse_java is thread-safe (per-thread tree-sitter Parser in ast_java).
|
|
433
|
-
enrichments = await asyncio.to_thread(
|
|
444
|
+
enrichments, ast = await asyncio.to_thread(
|
|
434
445
|
_parse_and_enrich_java, content_bytes, chunks, rel, project_root
|
|
435
446
|
)
|
|
447
|
+
|
|
448
|
+
# Compute generated source detection once per file (uses the AST and content_bytes)
|
|
449
|
+
generated_config = load_generated_detection(project_root)
|
|
450
|
+
generated, generated_by = classify_java_file(
|
|
451
|
+
content_bytes, ast, config=generated_config, project_root=project_root
|
|
452
|
+
)
|
|
453
|
+
|
|
436
454
|
# (vectors perf lever #1) embed all chunks concurrently so the batched
|
|
437
455
|
# embedder groups them into one ``model.encode(...)`` (max_batch_size=64)
|
|
438
456
|
# instead of N serial batch-of-1 calls. Dominant win for ``increment``
|
|
@@ -462,6 +480,8 @@ async def process_java_file(
|
|
|
462
480
|
annotations_on_type=enrich.annotations_on_type,
|
|
463
481
|
symbols=enrich.symbols,
|
|
464
482
|
ontology_version=ONTOLOGY_VERSION,
|
|
483
|
+
generated=generated,
|
|
484
|
+
generated_by=generated_by,
|
|
465
485
|
)
|
|
466
486
|
)
|
|
467
487
|
|
|
File without changes
|
java_codebase_rag/jrag.py
CHANGED
|
@@ -134,7 +134,7 @@ def _apply_auto_scope(args: argparse.Namespace, cfg, graph) -> None:
|
|
|
134
134
|
source_root = cfg.source_root if cfg.source_root else None
|
|
135
135
|
if not source_root:
|
|
136
136
|
return
|
|
137
|
-
from graph_enrich import detect_microservice_from_path
|
|
137
|
+
from java_codebase_rag.graph.graph_enrich import detect_microservice_from_path
|
|
138
138
|
|
|
139
139
|
candidate = detect_microservice_from_path(Path.cwd(), Path(source_root))
|
|
140
140
|
if not candidate:
|
|
@@ -1163,6 +1163,20 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
1163
1163
|
)
|
|
1164
1164
|
search.set_defaults(handler=_cmd_search, auto_scope=True)
|
|
1165
1165
|
|
|
1166
|
+
# ---- vocab-index subparser (PR-ABS-1) ----
|
|
1167
|
+
vocab_index = subparsers.add_parser(
|
|
1168
|
+
"vocab-index",
|
|
1169
|
+
help="Rebuild the vocabulary index (absence diagnosis).",
|
|
1170
|
+
parents=[_core_parser()],
|
|
1171
|
+
description=(
|
|
1172
|
+
"Rebuild the vocabulary index sidecar from the current Ladybug graph. "
|
|
1173
|
+
"The index is a search-optimized projection of Symbol nodes used for "
|
|
1174
|
+
"did-you-mean suggestions and external membership checks in absence "
|
|
1175
|
+
"diagnosis. Printed on success: symbol count and sidecar path."
|
|
1176
|
+
),
|
|
1177
|
+
)
|
|
1178
|
+
vocab_index.set_defaults(handler=_cmd_vocab_index, detail="full")
|
|
1179
|
+
|
|
1166
1180
|
return parser
|
|
1167
1181
|
|
|
1168
1182
|
|
|
@@ -1203,7 +1217,7 @@ def _load_graph(cfg): # type: ignore[no-untyped-def]
|
|
|
1203
1217
|
* ontology-mismatch (``RuntimeError`` from ``LadybugGraph.get``) ->
|
|
1204
1218
|
``_IndexStale`` (caught in ``main`` -> envelope with a rebuild hint).
|
|
1205
1219
|
"""
|
|
1206
|
-
from ladybug_queries import LadybugGraph
|
|
1220
|
+
from java_codebase_rag.graph.ladybug_queries import LadybugGraph
|
|
1207
1221
|
|
|
1208
1222
|
ladybug_path = str(cfg.ladybug_path)
|
|
1209
1223
|
if not LadybugGraph.exists(ladybug_path):
|
|
@@ -1217,6 +1231,40 @@ def _load_graph(cfg): # type: ignore[no-untyped-def]
|
|
|
1217
1231
|
raise _IndexStale(str(exc)) from exc
|
|
1218
1232
|
|
|
1219
1233
|
|
|
1234
|
+
def _cmd_vocab_index(args: argparse.Namespace) -> int:
|
|
1235
|
+
"""Rebuild the vocabulary index sidecar from the Ladybug graph."""
|
|
1236
|
+
from java_codebase_rag.ast.ast_java import ONTOLOGY_VERSION
|
|
1237
|
+
from java_codebase_rag.absence.absence_vocab import VocabularyIndex, VOCAB_INDEX_FILENAME
|
|
1238
|
+
|
|
1239
|
+
cfg = _resolve_cfg(args)
|
|
1240
|
+
try:
|
|
1241
|
+
graph = _load_graph(cfg)
|
|
1242
|
+
except (_IndexNotFound, _IndexStale) as exc:
|
|
1243
|
+
print(f"[error] {exc}", file=sys.stderr)
|
|
1244
|
+
return 2
|
|
1245
|
+
|
|
1246
|
+
# Build vocabulary index
|
|
1247
|
+
try:
|
|
1248
|
+
index = VocabularyIndex.build(graph, q=cfg.absence_ngram_q)
|
|
1249
|
+
except Exception as e:
|
|
1250
|
+
print(f"[error] Vocabulary index build failed: {e}", file=sys.stderr)
|
|
1251
|
+
return 1
|
|
1252
|
+
|
|
1253
|
+
# Save to sidecar
|
|
1254
|
+
sidecar_path = cfg.ladybug_path.parent / VOCAB_INDEX_FILENAME
|
|
1255
|
+
try:
|
|
1256
|
+
index.save(sidecar_path, ontology_version=ONTOLOGY_VERSION)
|
|
1257
|
+
except Exception as e:
|
|
1258
|
+
print(f"[error] Failed to save vocabulary index: {e}", file=sys.stderr)
|
|
1259
|
+
return 1
|
|
1260
|
+
|
|
1261
|
+
# Print success message (simple format for admin command)
|
|
1262
|
+
print(f"Vocabulary index rebuilt successfully:")
|
|
1263
|
+
print(f" Symbol count: {index.symbol_count}")
|
|
1264
|
+
print(f" Sidecar path: {sidecar_path}")
|
|
1265
|
+
return 0
|
|
1266
|
+
|
|
1267
|
+
|
|
1220
1268
|
def _cmd_status(args: argparse.Namespace) -> int:
|
|
1221
1269
|
from java_codebase_rag.jrag_envelope import Envelope
|
|
1222
1270
|
from java_codebase_rag.jrag_render import render
|
|
@@ -1517,7 +1565,7 @@ def _build_node_filter_or_error(filter_dict: dict):
|
|
|
1517
1565
|
A bad enum (e.g. ``--role FOO``) should be a user-facing validation error,
|
|
1518
1566
|
not an internal crash. Returns ``(node_filter, None)`` on success.
|
|
1519
1567
|
"""
|
|
1520
|
-
import mcp_v2
|
|
1568
|
+
from java_codebase_rag.mcp import mcp_v2
|
|
1521
1569
|
|
|
1522
1570
|
from java_codebase_rag.jrag_envelope import Envelope
|
|
1523
1571
|
from pydantic import ValidationError
|
|
@@ -1543,7 +1591,7 @@ def _cmd_find_filter_mode(
|
|
|
1543
1591
|
limit: int,
|
|
1544
1592
|
) -> int:
|
|
1545
1593
|
"""Find filter mode: build NodeFilter and call find_v2."""
|
|
1546
|
-
import mcp_v2
|
|
1594
|
+
from java_codebase_rag.mcp import mcp_v2
|
|
1547
1595
|
|
|
1548
1596
|
from java_codebase_rag.jrag_envelope import Envelope, next_actions_hook, normalize_enum, to_envelope_rows
|
|
1549
1597
|
from java_codebase_rag.jrag_render import render
|
|
@@ -1626,7 +1674,7 @@ def _cmd_find_filter_mode(
|
|
|
1626
1674
|
|
|
1627
1675
|
|
|
1628
1676
|
def _cmd_inspect(args: argparse.Namespace) -> int:
|
|
1629
|
-
import mcp_v2
|
|
1677
|
+
from java_codebase_rag.mcp import mcp_v2
|
|
1630
1678
|
|
|
1631
1679
|
from java_codebase_rag.jrag_envelope import Envelope, next_actions_hook, resolve_query
|
|
1632
1680
|
from java_codebase_rag.jrag_render import render
|
|
@@ -2526,7 +2574,7 @@ def _cmd_callees(args: argparse.Namespace) -> int:
|
|
|
2526
2574
|
# Producer root -> ASYNC_CALLS out (Producer -> :Route, the kafka_topic
|
|
2527
2575
|
# Route this producer publishes to — NOT a :Producer node).
|
|
2528
2576
|
if node.kind in ("client", "producer"):
|
|
2529
|
-
import mcp_v2
|
|
2577
|
+
from java_codebase_rag.mcp import mcp_v2
|
|
2530
2578
|
|
|
2531
2579
|
edge_types = ["HTTP_CALLS"] if node.kind == "client" else ["ASYNC_CALLS"]
|
|
2532
2580
|
out = mcp_v2.neighbors_v2(
|
|
@@ -2662,7 +2710,7 @@ def _cmd_callees(args: argparse.Namespace) -> int:
|
|
|
2662
2710
|
|
|
2663
2711
|
|
|
2664
2712
|
def _cmd_hierarchy(args: argparse.Namespace) -> int:
|
|
2665
|
-
import mcp_v2
|
|
2713
|
+
from java_codebase_rag.mcp import mcp_v2
|
|
2666
2714
|
|
|
2667
2715
|
cfg, graph, rc = _load_graph_or_error(args)
|
|
2668
2716
|
if rc:
|
|
@@ -2815,7 +2863,7 @@ def _cmd_subclasses(args: argparse.Namespace) -> int:
|
|
|
2815
2863
|
|
|
2816
2864
|
|
|
2817
2865
|
def _cmd_overrides(args: argparse.Namespace) -> int:
|
|
2818
|
-
import mcp_v2
|
|
2866
|
+
from java_codebase_rag.mcp import mcp_v2
|
|
2819
2867
|
|
|
2820
2868
|
cfg, graph, rc = _load_graph_or_error(args)
|
|
2821
2869
|
if rc:
|
|
@@ -2868,7 +2916,7 @@ def _cmd_overrides(args: argparse.Namespace) -> int:
|
|
|
2868
2916
|
|
|
2869
2917
|
|
|
2870
2918
|
def _cmd_overridden_by(args: argparse.Namespace) -> int:
|
|
2871
|
-
import mcp_v2
|
|
2919
|
+
from java_codebase_rag.mcp import mcp_v2
|
|
2872
2920
|
|
|
2873
2921
|
cfg, graph, rc = _load_graph_or_error(args)
|
|
2874
2922
|
if rc:
|
|
@@ -3176,7 +3224,7 @@ def _cmd_flow(args: argparse.Namespace) -> int:
|
|
|
3176
3224
|
|
|
3177
3225
|
|
|
3178
3226
|
def _cmd_dependencies(args: argparse.Namespace) -> int:
|
|
3179
|
-
import mcp_v2
|
|
3227
|
+
from java_codebase_rag.mcp import mcp_v2
|
|
3180
3228
|
|
|
3181
3229
|
cfg, graph, rc = _load_graph_or_error(args)
|
|
3182
3230
|
if rc:
|
|
@@ -3511,7 +3559,7 @@ def _cmd_outline(args: argparse.Namespace) -> int:
|
|
|
3511
3559
|
start_line<1). ``--limit`` caps the entry count (the file's symbol table
|
|
3512
3560
|
is otherwise unbounded); ``truncated`` is set when more entries exist.
|
|
3513
3561
|
"""
|
|
3514
|
-
from ladybug_queries import find_symbols_in_file_range
|
|
3562
|
+
from java_codebase_rag.graph.ladybug_queries import find_symbols_in_file_range
|
|
3515
3563
|
|
|
3516
3564
|
from java_codebase_rag.jrag_envelope import Envelope, mark_truncated, next_actions_hook
|
|
3517
3565
|
from java_codebase_rag.jrag_render import render
|
|
@@ -3581,8 +3629,8 @@ def _cmd_imports(args: argparse.Namespace) -> int:
|
|
|
3581
3629
|
a node per import: resolved graph Symbol when resolve_v2 hits (status=one),
|
|
3582
3630
|
or an unresolved placeholder carrying the raw FQN otherwise.
|
|
3583
3631
|
"""
|
|
3584
|
-
from ast_java import parse_java
|
|
3585
|
-
from resolve_service import resolve_v2
|
|
3632
|
+
from java_codebase_rag.ast.ast_java import parse_java
|
|
3633
|
+
from java_codebase_rag.analysis.resolve_service import resolve_v2
|
|
3586
3634
|
|
|
3587
3635
|
from java_codebase_rag.jrag_envelope import Envelope, next_actions_hook
|
|
3588
3636
|
from java_codebase_rag.jrag_render import render
|
|
@@ -4147,7 +4195,7 @@ def _zero_result_guidance(args: argparse.Namespace, graph) -> str | None:
|
|
|
4147
4195
|
empty (truly no matches for this query), or the probe itself errored
|
|
4148
4196
|
(non-fatal — the empty result still renders).
|
|
4149
4197
|
"""
|
|
4150
|
-
import mcp_v2
|
|
4198
|
+
from java_codebase_rag.mcp import mcp_v2
|
|
4151
4199
|
from collections import Counter
|
|
4152
4200
|
|
|
4153
4201
|
from java_codebase_rag.jrag_envelope import normalize_enum
|
|
@@ -4202,7 +4250,7 @@ def _cmd_search(args: argparse.Namespace) -> int:
|
|
|
4202
4250
|
truncation, and renders. --fuzzy is rejected IN-HANDLER (not argparse-exit)
|
|
4203
4251
|
so the error carries the canonical envelope shape.
|
|
4204
4252
|
"""
|
|
4205
|
-
import mcp_v2
|
|
4253
|
+
from java_codebase_rag.mcp import mcp_v2
|
|
4206
4254
|
|
|
4207
4255
|
from java_codebase_rag.jrag_envelope import Envelope, mark_truncated, next_actions_hook, normalize_enum
|
|
4208
4256
|
from java_codebase_rag.jrag_render import render
|
|
@@ -4321,13 +4369,20 @@ def _cmd_search(args: argparse.Namespace) -> int:
|
|
|
4321
4369
|
d["kind"] = "search_hit"
|
|
4322
4370
|
# Add explain token when --explain is set
|
|
4323
4371
|
if args.explain:
|
|
4324
|
-
|
|
4372
|
+
# search_lancedb is unimportable on graph-only (macOS Intel) installs —
|
|
4373
|
+
# lancedb/sentence-transformers are excluded by the PEP 508 markers in
|
|
4374
|
+
# pyproject.toml, and this module imports them at module top. Importing the
|
|
4375
|
+
# explain renderer from there would crash `jrag search ... --explain` on the
|
|
4376
|
+
# exact Intel install that runs the lexical path. search_scoring is
|
|
4377
|
+
# dependency-free and always installed, so the explain import works everywhere.
|
|
4378
|
+
from java_codebase_rag.search.search_scoring import explain_score_components
|
|
4325
4379
|
comps = d.get("score_components")
|
|
4326
4380
|
d["explain"] = explain_score_components(
|
|
4327
4381
|
comps,
|
|
4328
4382
|
role=d.get("role"),
|
|
4329
4383
|
hybrid=bool(args.hybrid),
|
|
4330
4384
|
graph_expanded=False,
|
|
4385
|
+
lexical=bool(getattr(out, "lexical_mode", False)),
|
|
4331
4386
|
)
|
|
4332
4387
|
hit_dicts.append(d)
|
|
4333
4388
|
|
|
@@ -18,7 +18,8 @@ import re
|
|
|
18
18
|
from dataclasses import dataclass, field
|
|
19
19
|
from typing import Any, Literal
|
|
20
20
|
|
|
21
|
-
from
|
|
21
|
+
from java_codebase_rag.absence.absence_types import AbsenceDiagnosis
|
|
22
|
+
from java_codebase_rag.graph.graph_types import NodeRef
|
|
22
23
|
|
|
23
24
|
__all__ = [
|
|
24
25
|
"Envelope",
|
|
@@ -118,6 +119,9 @@ class Envelope:
|
|
|
118
119
|
# zero in-repo callers. Distinguishes the *correct* empty result ("external
|
|
119
120
|
# entrypoint — no in-repo callers") from a bug-looking bare "0 callers".
|
|
120
121
|
is_external_entrypoint: bool = False
|
|
122
|
+
# Absence diagnosis explaining why a result is empty (PR-ABS-4). Carried
|
|
123
|
+
# from MCP outputs and rendered in CLI text/JSON. None on ok/ambiguous.
|
|
124
|
+
absence: AbsenceDiagnosis | None = None
|
|
121
125
|
|
|
122
126
|
def to_dict(self) -> dict[str, Any]:
|
|
123
127
|
"""Serialize to a JSON-ready dict, omitting empty optionals.
|
|
@@ -151,6 +155,8 @@ class Envelope:
|
|
|
151
155
|
out["message"] = self.message
|
|
152
156
|
if self.is_external_entrypoint:
|
|
153
157
|
out["is_external_entrypoint"] = True
|
|
158
|
+
if self.absence is not None:
|
|
159
|
+
out["absence"] = self.absence.model_dump()
|
|
154
160
|
return out
|
|
155
161
|
|
|
156
162
|
def to_json(self) -> str:
|
|
@@ -237,6 +243,8 @@ class Envelope:
|
|
|
237
243
|
out["message"] = self.message
|
|
238
244
|
if self.is_external_entrypoint:
|
|
239
245
|
out["is_external_entrypoint"] = True
|
|
246
|
+
if self.absence is not None:
|
|
247
|
+
out["absence"] = self.absence.model_dump()
|
|
240
248
|
return out
|
|
241
249
|
|
|
242
250
|
@staticmethod
|
|
@@ -524,10 +532,10 @@ def resolve_query(
|
|
|
524
532
|
means ``--service`` disambiguates which microservice's route is selected.
|
|
525
533
|
"""
|
|
526
534
|
# Lazy imports — keeps build_parser() / `jrag --help` free of resolve/ladybug.
|
|
527
|
-
from resolve_service import resolve_v2
|
|
535
|
+
from java_codebase_rag.analysis.resolve_service import resolve_v2
|
|
528
536
|
|
|
529
537
|
if graph is None:
|
|
530
|
-
from ladybug_queries import LadybugGraph
|
|
538
|
+
from java_codebase_rag.graph.ladybug_queries import LadybugGraph
|
|
531
539
|
|
|
532
540
|
graph = LadybugGraph.get(str(cfg.ladybug_path))
|
|
533
541
|
|
|
@@ -610,7 +618,7 @@ def resolve_query(
|
|
|
610
618
|
# the agent-facing CLI).
|
|
611
619
|
if "jrag search" not in raw_msg:
|
|
612
620
|
raw_msg = f"{raw_msg} Use `jrag search <query>` for ranked fuzzy lookup."
|
|
613
|
-
return None, Envelope(status="not_found", message=raw_msg)
|
|
621
|
+
return None, Envelope(status="not_found", message=raw_msg, absence=out.absence)
|
|
614
622
|
|
|
615
623
|
|
|
616
624
|
# Listing breadcrumbs (root is None): 1–2 template hints pointing at the natural
|
|
@@ -1082,4 +1090,5 @@ def project_envelope(envelope: Envelope, detail: str) -> Envelope:
|
|
|
1082
1090
|
file_location=envelope.file_location,
|
|
1083
1091
|
message=envelope.message,
|
|
1084
1092
|
is_external_entrypoint=envelope.is_external_entrypoint,
|
|
1093
|
+
absence=envelope.absence,
|
|
1085
1094
|
)
|
java_codebase_rag/jrag_hints.py
CHANGED
|
@@ -143,7 +143,7 @@ def next_actions(
|
|
|
143
143
|
# EDGE_SCHEMA is the canonical label set; we use it to skip labels we don't
|
|
144
144
|
# recognize (avoids emitting hints for spurious / future edge types the
|
|
145
145
|
# command map doesn't cover).
|
|
146
|
-
from java_ontology import EDGE_SCHEMA
|
|
146
|
+
from java_codebase_rag.graph.java_ontology import EDGE_SCHEMA
|
|
147
147
|
|
|
148
148
|
# Known virtual labels not in EDGE_SCHEMA (describe-time rollup constructs).
|
|
149
149
|
_VIRTUAL_LABELS = frozenset({"OVERRIDDEN_BY"})
|
java_codebase_rag/jrag_render.py
CHANGED
|
@@ -12,6 +12,7 @@ from __future__ import annotations
|
|
|
12
12
|
|
|
13
13
|
from typing import Any
|
|
14
14
|
|
|
15
|
+
from java_codebase_rag.absence.absence_types import AbsenceDiagnosis
|
|
15
16
|
from java_codebase_rag.jrag_envelope import Envelope, project_envelope, simple_name
|
|
16
17
|
|
|
17
18
|
__all__ = ["render", "tiered_name", "display_name"]
|
|
@@ -32,6 +33,23 @@ _CALLS_FAMILY_EDGES = frozenset({"CALLS", "HTTP_CALLS", "ASYNC_CALLS"})
|
|
|
32
33
|
# none) and are left untagged.
|
|
33
34
|
_ROUTE_KIND_TAGS: dict[str, str] = {"kafka_topic": "kafka", "http_endpoint": "http"}
|
|
34
35
|
|
|
36
|
+
# Absence verdict → human-readable label, shared by the not-found / listing /
|
|
37
|
+
# traversal empty-result renderers. ``AbsenceVerdict`` is a closed Literal of
|
|
38
|
+
# these four values.
|
|
39
|
+
_ABSENCE_VERDICT_TEXT: dict[str, str] = {
|
|
40
|
+
"not_in_project": "not in project",
|
|
41
|
+
"external_dependency": "external dependency",
|
|
42
|
+
"refine_query": "refine your query",
|
|
43
|
+
"correct_empty": "correct empty",
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _verdict_line(absence: AbsenceDiagnosis) -> str | None:
|
|
48
|
+
"""A ``Verdict: <label>`` line for an absence diagnosis, or ``None`` if the
|
|
49
|
+
verdict is not one of the known values."""
|
|
50
|
+
text = _ABSENCE_VERDICT_TEXT.get(absence.verdict)
|
|
51
|
+
return f"Verdict: {text}" if text else None
|
|
52
|
+
|
|
35
53
|
# Identity keys already represented in a listing line (display_name + @service +
|
|
36
54
|
# kind tag). At ``--detail full`` the per-row kv-block skips these (they are in
|
|
37
55
|
# the header line) and renders every OTHER key, so full listing == per-row
|
|
@@ -226,7 +244,35 @@ def _render_error(envelope: Envelope) -> str:
|
|
|
226
244
|
|
|
227
245
|
def _render_not_found(envelope: Envelope) -> str:
|
|
228
246
|
msg = envelope.message or "not found"
|
|
229
|
-
|
|
247
|
+
base = f"not found: {msg}"
|
|
248
|
+
|
|
249
|
+
# If absence diagnosis is present, append verdict + message (+ did-you-mean)
|
|
250
|
+
if envelope.absence is not None:
|
|
251
|
+
lines = [base]
|
|
252
|
+
# Verdict line (human-readable label)
|
|
253
|
+
vline = _verdict_line(envelope.absence)
|
|
254
|
+
if vline:
|
|
255
|
+
lines.append(vline)
|
|
256
|
+
|
|
257
|
+
# Per-cause explanation message — surfaces the diagnosis's authored help
|
|
258
|
+
# (external-identity context, filter-relaxation suggestions, etc.).
|
|
259
|
+
if envelope.absence.message:
|
|
260
|
+
lines.append(envelope.absence.message)
|
|
261
|
+
|
|
262
|
+
# Add did-you-mean line if closest_symbols is non-empty
|
|
263
|
+
if envelope.absence.closest_symbols:
|
|
264
|
+
symbols = [s.fqn for s in envelope.absence.closest_symbols]
|
|
265
|
+
if len(symbols) == 1:
|
|
266
|
+
lines.append(f"Did you mean: {symbols[0]}?")
|
|
267
|
+
elif len(symbols) == 2:
|
|
268
|
+
lines.append(f"Did you mean: {symbols[0]} or {symbols[1]}?")
|
|
269
|
+
else:
|
|
270
|
+
joined = ", ".join(symbols[:-1]) + f", or {symbols[-1]}"
|
|
271
|
+
lines.append(f"Did you mean: {joined}?")
|
|
272
|
+
|
|
273
|
+
return "\n".join(lines)
|
|
274
|
+
|
|
275
|
+
return base
|
|
230
276
|
|
|
231
277
|
|
|
232
278
|
def _render_listing(envelope: Envelope, *, noun: str, detail: str = "normal") -> str:
|
|
@@ -295,7 +341,12 @@ def _render_listing(envelope: Envelope, *, noun: str, detail: str = "normal") ->
|
|
|
295
341
|
if rest:
|
|
296
342
|
lines.extend(_render_inspect_block(rest, 1))
|
|
297
343
|
if not lines:
|
|
298
|
-
|
|
344
|
+
# Handle absence diagnosis (PR-ABS-4)
|
|
345
|
+
if envelope.absence is not None:
|
|
346
|
+
vline = _verdict_line(envelope.absence)
|
|
347
|
+
lines.append(vline if vline else f"0 {noun}".rstrip())
|
|
348
|
+
else:
|
|
349
|
+
lines.append(f"0 {noun}".rstrip())
|
|
299
350
|
# Listing breadcrumbs (Phase 2): <=2 `next:` hint lines when the listing
|
|
300
351
|
# command emitted agent_next_actions (routes/clients/producers/topics).
|
|
301
352
|
lines.extend(_next_action_lines(envelope))
|
|
@@ -413,10 +464,23 @@ def _render_traversal(envelope: Envelope, *, noun: str, detail: str = "normal")
|
|
|
413
464
|
root_node = envelope.nodes.get(root_id, {})
|
|
414
465
|
root_fqn = str(root_node.get("fqn") or "").strip()
|
|
415
466
|
root_svc = str(root_node.get("microservice") or "").strip()
|
|
416
|
-
|
|
467
|
+
|
|
468
|
+
# Handle absence diagnosis (PR-ABS-4)
|
|
469
|
+
if envelope.absence is not None:
|
|
470
|
+
absence = envelope.absence
|
|
471
|
+
if absence.verdict == "correct_empty":
|
|
472
|
+
# Same text as the is_external_entrypoint case.
|
|
473
|
+
parts = ["external entrypoint — no in-repo callers"]
|
|
474
|
+
else:
|
|
475
|
+
vline = _verdict_line(absence)
|
|
476
|
+
if vline:
|
|
477
|
+
lines.append(vline)
|
|
478
|
+
parts = [f"0 {noun}".rstrip()]
|
|
479
|
+
elif envelope.is_external_entrypoint:
|
|
417
480
|
parts = ["external entrypoint — no in-repo callers"]
|
|
418
481
|
else:
|
|
419
482
|
parts = [f"0 {noun}".rstrip()]
|
|
483
|
+
|
|
420
484
|
if root_fqn:
|
|
421
485
|
parts.append(root_fqn)
|
|
422
486
|
if root_svc:
|
|
File without changes
|
|
@@ -13,7 +13,7 @@ from __future__ import annotations
|
|
|
13
13
|
import json
|
|
14
14
|
from typing import Any, Literal, NamedTuple
|
|
15
15
|
|
|
16
|
-
from java_ontology import EDGE_SCHEMA, FUZZY_STRATEGY_SET
|
|
16
|
+
from java_codebase_rag.graph.java_ontology import EDGE_SCHEMA, FUZZY_STRATEGY_SET
|
|
17
17
|
|
|
18
18
|
# Normative schema description (propose §3.1) — imported by ``mcp_v2`` for Field(description=...).
|
|
19
19
|
MCP_HINTS_STRUCTURED_FIELD_DESCRIPTION = (
|