java-codebase-rag 0.9.4__py3-none-any.whl → 0.9.6__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. java_codebase_rag/absence/__init__.py +0 -0
  2. java_codebase_rag/absence/absence_diagnosis.py +700 -0
  3. java_codebase_rag/absence/absence_types.py +124 -0
  4. java_codebase_rag/absence/absence_vocab.py +455 -0
  5. java_codebase_rag/analysis/__init__.py +0 -0
  6. pr_analysis.py → java_codebase_rag/analysis/pr_analysis.py +1 -1
  7. resolve_service.py → java_codebase_rag/analysis/resolve_service.py +73 -6
  8. java_codebase_rag/ast/__init__.py +0 -0
  9. ast_java.py → java_codebase_rag/ast/ast_java.py +5 -5
  10. java_codebase_rag/cli.py +13 -18
  11. java_codebase_rag/config.py +116 -0
  12. java_codebase_rag/graph/__init__.py +0 -0
  13. build_ast_graph.py → java_codebase_rag/graph/build_ast_graph.py +89 -11
  14. graph_enrich.py → java_codebase_rag/graph/graph_enrich.py +248 -3
  15. graph_types.py → java_codebase_rag/graph/graph_types.py +6 -2
  16. java_ontology.py → java_codebase_rag/graph/java_ontology.py +1 -1
  17. ladybug_queries.py → java_codebase_rag/graph/ladybug_queries.py +6 -6
  18. java_codebase_rag/index/__init__.py +0 -0
  19. java_index_flow_lancedb.py → java_codebase_rag/index/java_index_flow_lancedb.py +30 -10
  20. java_codebase_rag/install_data/__init__.py +0 -0
  21. java_codebase_rag/jrag.py +71 -16
  22. java_codebase_rag/jrag_envelope.py +13 -4
  23. java_codebase_rag/jrag_hints.py +1 -1
  24. java_codebase_rag/jrag_render.py +67 -3
  25. java_codebase_rag/mcp/__init__.py +0 -0
  26. mcp_hints.py → java_codebase_rag/mcp/mcp_hints.py +1 -1
  27. mcp_v2.py → java_codebase_rag/mcp/mcp_v2.py +280 -81
  28. server.py → java_codebase_rag/mcp/server.py +138 -54
  29. java_codebase_rag/pipeline.py +26 -7
  30. java_codebase_rag/search/__init__.py +0 -0
  31. search_lancedb.py → java_codebase_rag/search/search_lancedb.py +53 -314
  32. java_codebase_rag/search/search_lexical.py +329 -0
  33. java_codebase_rag/search/search_scoring.py +338 -0
  34. {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/METADATA +2 -2
  35. java_codebase_rag-0.9.6.dist-info/RECORD +57 -0
  36. {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/entry_points.txt +1 -1
  37. java_codebase_rag-0.9.6.dist-info/top_level.txt +1 -0
  38. java_codebase_rag-0.9.4.dist-info/RECORD +0 -44
  39. java_codebase_rag-0.9.4.dist-info/top_level.txt +0 -19
  40. /brownfield_events.py → /java_codebase_rag/ast/brownfield_events.py +0 -0
  41. /chunk_heuristics.py → /java_codebase_rag/ast/chunk_heuristics.py +0 -0
  42. /path_filtering.py → /java_codebase_rag/graph/path_filtering.py +0 -0
  43. /java_index_v1_common.py → /java_codebase_rag/index/java_index_v1_common.py +0 -0
  44. /index_common.py → /java_codebase_rag/search/index_common.py +0 -0
  45. {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/WHEEL +0 -0
  46. {java_codebase_rag-0.9.4.dist-info → java_codebase_rag-0.9.6.dist-info}/licenses/LICENSE +0 -0
@@ -41,7 +41,7 @@ from cocoindex.resources.file import PatternFilePathMatcher
41
41
 
42
42
  from java_codebase_rag.config import resolved_sbert_model_for_process_env
43
43
  from java_codebase_rag.lance_optimize import LANCE_TABLE_NAMES
44
- from java_index_v1_common import (
44
+ from java_codebase_rag.index.java_index_v1_common import (
45
45
  JAVA_CHUNK,
46
46
  SBERT_MODEL,
47
47
  SQL_CHUNK,
@@ -49,9 +49,15 @@ from java_index_v1_common import (
49
49
  chunk_key_range,
50
50
  position_to_json,
51
51
  )
52
- from path_filtering import LayeredIgnore
53
- from ast_java import ONTOLOGY_VERSION, parse_java
54
- from graph_enrich import collect_annotation_meta_chain, enrich_chunk, load_brownfield_overrides
52
+ from java_codebase_rag.graph.path_filtering import LayeredIgnore
53
+ from java_codebase_rag.ast.ast_java import ONTOLOGY_VERSION, parse_java
54
+ from java_codebase_rag.graph.graph_enrich import (
55
+ classify_java_file,
56
+ collect_annotation_meta_chain,
57
+ enrich_chunk,
58
+ load_brownfield_overrides,
59
+ load_generated_detection,
60
+ )
55
61
 
56
62
  # Older cocoindex (e.g. 1.0.0a43) uses ``tracked=False``; newer releases renamed
57
63
  # the flag to ``detect_change`` (default False) and reject ``tracked``.
@@ -280,6 +286,9 @@ class JavaLanceChunk:
280
286
  annotations_on_type: Annotated[list[str], LanceType(pa.list_(pa.string()))]
281
287
  symbols: Annotated[list[str], LanceType(pa.list_(pa.string()))]
282
288
  ontology_version: int
289
+ # Generated source detection: populated per-file, not per-chunk
290
+ generated: bool
291
+ generated_by: str | None
283
292
 
284
293
 
285
294
  @dataclass
@@ -364,12 +373,13 @@ def _parse_and_enrich_java(
364
373
  chunks: list[Any],
365
374
  rel: str,
366
375
  project_root: Path,
367
- ) -> list[Any]:
376
+ ) -> tuple[list[Any], Any]:
368
377
  """Parse one Java file and enrich every chunk, off the event loop.
369
378
 
370
- Returns a list of :class:`graph_enrich.ChunkEnrichment` aligned 1:1 with
371
- ``chunks``. Intended to run via ``asyncio.to_thread`` from
372
- ``process_java_file`` (vectors perf lever #2): while the worker thread
379
+ Returns a tuple of (enrichments, ast) where enrichments is a list of
380
+ :class:`graph_enrich.ChunkEnrichment` aligned 1:1 with ``chunks``, and ast
381
+ is the parsed :class:`JavaFileAst`. Intended to run via ``asyncio.to_thread``
382
+ from ``process_java_file`` (vectors perf lever #2): while the worker thread
373
383
  parses + enriches, the event loop is free to drive other files and keep the
374
384
  embedder's batching queue fed.
375
385
 
@@ -381,7 +391,7 @@ def _parse_and_enrich_java(
381
391
  ``lru_cache`` reads are thread-safe under the GIL.
382
392
  """
383
393
  ast = parse_java(content_bytes)
384
- return [
394
+ enrichments = [
385
395
  enrich_chunk(
386
396
  ast,
387
397
  chunk_start_byte=ch.start.byte_offset,
@@ -391,6 +401,7 @@ def _parse_and_enrich_java(
391
401
  )
392
402
  for ch in chunks
393
403
  ]
404
+ return enrichments, ast
394
405
 
395
406
 
396
407
  @coco.fn(memo=True)
@@ -430,9 +441,16 @@ async def process_java_file(
430
441
  # (vectors perf lever #2) parse + enrich off the event loop so the loop can
431
442
  # keep the embedder's batching queue fed while this file is being parsed.
432
443
  # parse_java is thread-safe (per-thread tree-sitter Parser in ast_java).
433
- enrichments = await asyncio.to_thread(
444
+ enrichments, ast = await asyncio.to_thread(
434
445
  _parse_and_enrich_java, content_bytes, chunks, rel, project_root
435
446
  )
447
+
448
+ # Compute generated source detection once per file (uses the AST and content_bytes)
449
+ generated_config = load_generated_detection(project_root)
450
+ generated, generated_by = classify_java_file(
451
+ content_bytes, ast, config=generated_config, project_root=project_root
452
+ )
453
+
436
454
  # (vectors perf lever #1) embed all chunks concurrently so the batched
437
455
  # embedder groups them into one ``model.encode(...)`` (max_batch_size=64)
438
456
  # instead of N serial batch-of-1 calls. Dominant win for ``increment``
@@ -462,6 +480,8 @@ async def process_java_file(
462
480
  annotations_on_type=enrich.annotations_on_type,
463
481
  symbols=enrich.symbols,
464
482
  ontology_version=ONTOLOGY_VERSION,
483
+ generated=generated,
484
+ generated_by=generated_by,
465
485
  )
466
486
  )
467
487
 
File without changes
java_codebase_rag/jrag.py CHANGED
@@ -134,7 +134,7 @@ def _apply_auto_scope(args: argparse.Namespace, cfg, graph) -> None:
134
134
  source_root = cfg.source_root if cfg.source_root else None
135
135
  if not source_root:
136
136
  return
137
- from graph_enrich import detect_microservice_from_path
137
+ from java_codebase_rag.graph.graph_enrich import detect_microservice_from_path
138
138
 
139
139
  candidate = detect_microservice_from_path(Path.cwd(), Path(source_root))
140
140
  if not candidate:
@@ -1163,6 +1163,20 @@ def build_parser() -> argparse.ArgumentParser:
1163
1163
  )
1164
1164
  search.set_defaults(handler=_cmd_search, auto_scope=True)
1165
1165
 
1166
+ # ---- vocab-index subparser (PR-ABS-1) ----
1167
+ vocab_index = subparsers.add_parser(
1168
+ "vocab-index",
1169
+ help="Rebuild the vocabulary index (absence diagnosis).",
1170
+ parents=[_core_parser()],
1171
+ description=(
1172
+ "Rebuild the vocabulary index sidecar from the current Ladybug graph. "
1173
+ "The index is a search-optimized projection of Symbol nodes used for "
1174
+ "did-you-mean suggestions and external membership checks in absence "
1175
+ "diagnosis. Printed on success: symbol count and sidecar path."
1176
+ ),
1177
+ )
1178
+ vocab_index.set_defaults(handler=_cmd_vocab_index, detail="full")
1179
+
1166
1180
  return parser
1167
1181
 
1168
1182
 
@@ -1203,7 +1217,7 @@ def _load_graph(cfg): # type: ignore[no-untyped-def]
1203
1217
  * ontology-mismatch (``RuntimeError`` from ``LadybugGraph.get``) ->
1204
1218
  ``_IndexStale`` (caught in ``main`` -> envelope with a rebuild hint).
1205
1219
  """
1206
- from ladybug_queries import LadybugGraph
1220
+ from java_codebase_rag.graph.ladybug_queries import LadybugGraph
1207
1221
 
1208
1222
  ladybug_path = str(cfg.ladybug_path)
1209
1223
  if not LadybugGraph.exists(ladybug_path):
@@ -1217,6 +1231,40 @@ def _load_graph(cfg): # type: ignore[no-untyped-def]
1217
1231
  raise _IndexStale(str(exc)) from exc
1218
1232
 
1219
1233
 
1234
+ def _cmd_vocab_index(args: argparse.Namespace) -> int:
1235
+ """Rebuild the vocabulary index sidecar from the Ladybug graph."""
1236
+ from java_codebase_rag.ast.ast_java import ONTOLOGY_VERSION
1237
+ from java_codebase_rag.absence.absence_vocab import VocabularyIndex, VOCAB_INDEX_FILENAME
1238
+
1239
+ cfg = _resolve_cfg(args)
1240
+ try:
1241
+ graph = _load_graph(cfg)
1242
+ except (_IndexNotFound, _IndexStale) as exc:
1243
+ print(f"[error] {exc}", file=sys.stderr)
1244
+ return 2
1245
+
1246
+ # Build vocabulary index
1247
+ try:
1248
+ index = VocabularyIndex.build(graph, q=cfg.absence_ngram_q)
1249
+ except Exception as e:
1250
+ print(f"[error] Vocabulary index build failed: {e}", file=sys.stderr)
1251
+ return 1
1252
+
1253
+ # Save to sidecar
1254
+ sidecar_path = cfg.ladybug_path.parent / VOCAB_INDEX_FILENAME
1255
+ try:
1256
+ index.save(sidecar_path, ontology_version=ONTOLOGY_VERSION)
1257
+ except Exception as e:
1258
+ print(f"[error] Failed to save vocabulary index: {e}", file=sys.stderr)
1259
+ return 1
1260
+
1261
+ # Print success message (simple format for admin command)
1262
+ print(f"Vocabulary index rebuilt successfully:")
1263
+ print(f" Symbol count: {index.symbol_count}")
1264
+ print(f" Sidecar path: {sidecar_path}")
1265
+ return 0
1266
+
1267
+
1220
1268
  def _cmd_status(args: argparse.Namespace) -> int:
1221
1269
  from java_codebase_rag.jrag_envelope import Envelope
1222
1270
  from java_codebase_rag.jrag_render import render
@@ -1517,7 +1565,7 @@ def _build_node_filter_or_error(filter_dict: dict):
1517
1565
  A bad enum (e.g. ``--role FOO``) should be a user-facing validation error,
1518
1566
  not an internal crash. Returns ``(node_filter, None)`` on success.
1519
1567
  """
1520
- import mcp_v2
1568
+ from java_codebase_rag.mcp import mcp_v2
1521
1569
 
1522
1570
  from java_codebase_rag.jrag_envelope import Envelope
1523
1571
  from pydantic import ValidationError
@@ -1543,7 +1591,7 @@ def _cmd_find_filter_mode(
1543
1591
  limit: int,
1544
1592
  ) -> int:
1545
1593
  """Find filter mode: build NodeFilter and call find_v2."""
1546
- import mcp_v2
1594
+ from java_codebase_rag.mcp import mcp_v2
1547
1595
 
1548
1596
  from java_codebase_rag.jrag_envelope import Envelope, next_actions_hook, normalize_enum, to_envelope_rows
1549
1597
  from java_codebase_rag.jrag_render import render
@@ -1626,7 +1674,7 @@ def _cmd_find_filter_mode(
1626
1674
 
1627
1675
 
1628
1676
  def _cmd_inspect(args: argparse.Namespace) -> int:
1629
- import mcp_v2
1677
+ from java_codebase_rag.mcp import mcp_v2
1630
1678
 
1631
1679
  from java_codebase_rag.jrag_envelope import Envelope, next_actions_hook, resolve_query
1632
1680
  from java_codebase_rag.jrag_render import render
@@ -2526,7 +2574,7 @@ def _cmd_callees(args: argparse.Namespace) -> int:
2526
2574
  # Producer root -> ASYNC_CALLS out (Producer -> :Route, the kafka_topic
2527
2575
  # Route this producer publishes to — NOT a :Producer node).
2528
2576
  if node.kind in ("client", "producer"):
2529
- import mcp_v2
2577
+ from java_codebase_rag.mcp import mcp_v2
2530
2578
 
2531
2579
  edge_types = ["HTTP_CALLS"] if node.kind == "client" else ["ASYNC_CALLS"]
2532
2580
  out = mcp_v2.neighbors_v2(
@@ -2662,7 +2710,7 @@ def _cmd_callees(args: argparse.Namespace) -> int:
2662
2710
 
2663
2711
 
2664
2712
  def _cmd_hierarchy(args: argparse.Namespace) -> int:
2665
- import mcp_v2
2713
+ from java_codebase_rag.mcp import mcp_v2
2666
2714
 
2667
2715
  cfg, graph, rc = _load_graph_or_error(args)
2668
2716
  if rc:
@@ -2815,7 +2863,7 @@ def _cmd_subclasses(args: argparse.Namespace) -> int:
2815
2863
 
2816
2864
 
2817
2865
  def _cmd_overrides(args: argparse.Namespace) -> int:
2818
- import mcp_v2
2866
+ from java_codebase_rag.mcp import mcp_v2
2819
2867
 
2820
2868
  cfg, graph, rc = _load_graph_or_error(args)
2821
2869
  if rc:
@@ -2868,7 +2916,7 @@ def _cmd_overrides(args: argparse.Namespace) -> int:
2868
2916
 
2869
2917
 
2870
2918
  def _cmd_overridden_by(args: argparse.Namespace) -> int:
2871
- import mcp_v2
2919
+ from java_codebase_rag.mcp import mcp_v2
2872
2920
 
2873
2921
  cfg, graph, rc = _load_graph_or_error(args)
2874
2922
  if rc:
@@ -3176,7 +3224,7 @@ def _cmd_flow(args: argparse.Namespace) -> int:
3176
3224
 
3177
3225
 
3178
3226
  def _cmd_dependencies(args: argparse.Namespace) -> int:
3179
- import mcp_v2
3227
+ from java_codebase_rag.mcp import mcp_v2
3180
3228
 
3181
3229
  cfg, graph, rc = _load_graph_or_error(args)
3182
3230
  if rc:
@@ -3511,7 +3559,7 @@ def _cmd_outline(args: argparse.Namespace) -> int:
3511
3559
  start_line<1). ``--limit`` caps the entry count (the file's symbol table
3512
3560
  is otherwise unbounded); ``truncated`` is set when more entries exist.
3513
3561
  """
3514
- from ladybug_queries import find_symbols_in_file_range
3562
+ from java_codebase_rag.graph.ladybug_queries import find_symbols_in_file_range
3515
3563
 
3516
3564
  from java_codebase_rag.jrag_envelope import Envelope, mark_truncated, next_actions_hook
3517
3565
  from java_codebase_rag.jrag_render import render
@@ -3581,8 +3629,8 @@ def _cmd_imports(args: argparse.Namespace) -> int:
3581
3629
  a node per import: resolved graph Symbol when resolve_v2 hits (status=one),
3582
3630
  or an unresolved placeholder carrying the raw FQN otherwise.
3583
3631
  """
3584
- from ast_java import parse_java
3585
- from resolve_service import resolve_v2
3632
+ from java_codebase_rag.ast.ast_java import parse_java
3633
+ from java_codebase_rag.analysis.resolve_service import resolve_v2
3586
3634
 
3587
3635
  from java_codebase_rag.jrag_envelope import Envelope, next_actions_hook
3588
3636
  from java_codebase_rag.jrag_render import render
@@ -4147,7 +4195,7 @@ def _zero_result_guidance(args: argparse.Namespace, graph) -> str | None:
4147
4195
  empty (truly no matches for this query), or the probe itself errored
4148
4196
  (non-fatal — the empty result still renders).
4149
4197
  """
4150
- import mcp_v2
4198
+ from java_codebase_rag.mcp import mcp_v2
4151
4199
  from collections import Counter
4152
4200
 
4153
4201
  from java_codebase_rag.jrag_envelope import normalize_enum
@@ -4202,7 +4250,7 @@ def _cmd_search(args: argparse.Namespace) -> int:
4202
4250
  truncation, and renders. --fuzzy is rejected IN-HANDLER (not argparse-exit)
4203
4251
  so the error carries the canonical envelope shape.
4204
4252
  """
4205
- import mcp_v2
4253
+ from java_codebase_rag.mcp import mcp_v2
4206
4254
 
4207
4255
  from java_codebase_rag.jrag_envelope import Envelope, mark_truncated, next_actions_hook, normalize_enum
4208
4256
  from java_codebase_rag.jrag_render import render
@@ -4321,13 +4369,20 @@ def _cmd_search(args: argparse.Namespace) -> int:
4321
4369
  d["kind"] = "search_hit"
4322
4370
  # Add explain token when --explain is set
4323
4371
  if args.explain:
4324
- from search_lancedb import explain_score_components
4372
+ # search_lancedb is unimportable on graph-only (macOS Intel) installs —
4373
+ # lancedb/sentence-transformers are excluded by the PEP 508 markers in
4374
+ # pyproject.toml, and this module imports them at module top. Importing the
4375
+ # explain renderer from there would crash `jrag search ... --explain` on the
4376
+ # exact Intel install that runs the lexical path. search_scoring is
4377
+ # dependency-free and always installed, so the explain import works everywhere.
4378
+ from java_codebase_rag.search.search_scoring import explain_score_components
4325
4379
  comps = d.get("score_components")
4326
4380
  d["explain"] = explain_score_components(
4327
4381
  comps,
4328
4382
  role=d.get("role"),
4329
4383
  hybrid=bool(args.hybrid),
4330
4384
  graph_expanded=False,
4385
+ lexical=bool(getattr(out, "lexical_mode", False)),
4331
4386
  )
4332
4387
  hit_dicts.append(d)
4333
4388
 
@@ -18,7 +18,8 @@ import re
18
18
  from dataclasses import dataclass, field
19
19
  from typing import Any, Literal
20
20
 
21
- from graph_types import NodeRef
21
+ from java_codebase_rag.absence.absence_types import AbsenceDiagnosis
22
+ from java_codebase_rag.graph.graph_types import NodeRef
22
23
 
23
24
  __all__ = [
24
25
  "Envelope",
@@ -118,6 +119,9 @@ class Envelope:
118
119
  # zero in-repo callers. Distinguishes the *correct* empty result ("external
119
120
  # entrypoint — no in-repo callers") from a bug-looking bare "0 callers".
120
121
  is_external_entrypoint: bool = False
122
+ # Absence diagnosis explaining why a result is empty (PR-ABS-4). Carried
123
+ # from MCP outputs and rendered in CLI text/JSON. None on ok/ambiguous.
124
+ absence: AbsenceDiagnosis | None = None
121
125
 
122
126
  def to_dict(self) -> dict[str, Any]:
123
127
  """Serialize to a JSON-ready dict, omitting empty optionals.
@@ -151,6 +155,8 @@ class Envelope:
151
155
  out["message"] = self.message
152
156
  if self.is_external_entrypoint:
153
157
  out["is_external_entrypoint"] = True
158
+ if self.absence is not None:
159
+ out["absence"] = self.absence.model_dump()
154
160
  return out
155
161
 
156
162
  def to_json(self) -> str:
@@ -237,6 +243,8 @@ class Envelope:
237
243
  out["message"] = self.message
238
244
  if self.is_external_entrypoint:
239
245
  out["is_external_entrypoint"] = True
246
+ if self.absence is not None:
247
+ out["absence"] = self.absence.model_dump()
240
248
  return out
241
249
 
242
250
  @staticmethod
@@ -524,10 +532,10 @@ def resolve_query(
524
532
  means ``--service`` disambiguates which microservice's route is selected.
525
533
  """
526
534
  # Lazy imports — keeps build_parser() / `jrag --help` free of resolve/ladybug.
527
- from resolve_service import resolve_v2
535
+ from java_codebase_rag.analysis.resolve_service import resolve_v2
528
536
 
529
537
  if graph is None:
530
- from ladybug_queries import LadybugGraph
538
+ from java_codebase_rag.graph.ladybug_queries import LadybugGraph
531
539
 
532
540
  graph = LadybugGraph.get(str(cfg.ladybug_path))
533
541
 
@@ -610,7 +618,7 @@ def resolve_query(
610
618
  # the agent-facing CLI).
611
619
  if "jrag search" not in raw_msg:
612
620
  raw_msg = f"{raw_msg} Use `jrag search <query>` for ranked fuzzy lookup."
613
- return None, Envelope(status="not_found", message=raw_msg)
621
+ return None, Envelope(status="not_found", message=raw_msg, absence=out.absence)
614
622
 
615
623
 
616
624
  # Listing breadcrumbs (root is None): 1–2 template hints pointing at the natural
@@ -1082,4 +1090,5 @@ def project_envelope(envelope: Envelope, detail: str) -> Envelope:
1082
1090
  file_location=envelope.file_location,
1083
1091
  message=envelope.message,
1084
1092
  is_external_entrypoint=envelope.is_external_entrypoint,
1093
+ absence=envelope.absence,
1085
1094
  )
@@ -143,7 +143,7 @@ def next_actions(
143
143
  # EDGE_SCHEMA is the canonical label set; we use it to skip labels we don't
144
144
  # recognize (avoids emitting hints for spurious / future edge types the
145
145
  # command map doesn't cover).
146
- from java_ontology import EDGE_SCHEMA
146
+ from java_codebase_rag.graph.java_ontology import EDGE_SCHEMA
147
147
 
148
148
  # Known virtual labels not in EDGE_SCHEMA (describe-time rollup constructs).
149
149
  _VIRTUAL_LABELS = frozenset({"OVERRIDDEN_BY"})
@@ -12,6 +12,7 @@ from __future__ import annotations
12
12
 
13
13
  from typing import Any
14
14
 
15
+ from java_codebase_rag.absence.absence_types import AbsenceDiagnosis
15
16
  from java_codebase_rag.jrag_envelope import Envelope, project_envelope, simple_name
16
17
 
17
18
  __all__ = ["render", "tiered_name", "display_name"]
@@ -32,6 +33,23 @@ _CALLS_FAMILY_EDGES = frozenset({"CALLS", "HTTP_CALLS", "ASYNC_CALLS"})
32
33
  # none) and are left untagged.
33
34
  _ROUTE_KIND_TAGS: dict[str, str] = {"kafka_topic": "kafka", "http_endpoint": "http"}
34
35
 
36
+ # Absence verdict → human-readable label, shared by the not-found / listing /
37
+ # traversal empty-result renderers. ``AbsenceVerdict`` is a closed Literal of
38
+ # these four values.
39
+ _ABSENCE_VERDICT_TEXT: dict[str, str] = {
40
+ "not_in_project": "not in project",
41
+ "external_dependency": "external dependency",
42
+ "refine_query": "refine your query",
43
+ "correct_empty": "correct empty",
44
+ }
45
+
46
+
47
+ def _verdict_line(absence: AbsenceDiagnosis) -> str | None:
48
+ """A ``Verdict: <label>`` line for an absence diagnosis, or ``None`` if the
49
+ verdict is not one of the known values."""
50
+ text = _ABSENCE_VERDICT_TEXT.get(absence.verdict)
51
+ return f"Verdict: {text}" if text else None
52
+
35
53
  # Identity keys already represented in a listing line (display_name + @service +
36
54
  # kind tag). At ``--detail full`` the per-row kv-block skips these (they are in
37
55
  # the header line) and renders every OTHER key, so full listing == per-row
@@ -226,7 +244,35 @@ def _render_error(envelope: Envelope) -> str:
226
244
 
227
245
  def _render_not_found(envelope: Envelope) -> str:
228
246
  msg = envelope.message or "not found"
229
- return f"not found: {msg}"
247
+ base = f"not found: {msg}"
248
+
249
+ # If absence diagnosis is present, append verdict + message (+ did-you-mean)
250
+ if envelope.absence is not None:
251
+ lines = [base]
252
+ # Verdict line (human-readable label)
253
+ vline = _verdict_line(envelope.absence)
254
+ if vline:
255
+ lines.append(vline)
256
+
257
+ # Per-cause explanation message — surfaces the diagnosis's authored help
258
+ # (external-identity context, filter-relaxation suggestions, etc.).
259
+ if envelope.absence.message:
260
+ lines.append(envelope.absence.message)
261
+
262
+ # Add did-you-mean line if closest_symbols is non-empty
263
+ if envelope.absence.closest_symbols:
264
+ symbols = [s.fqn for s in envelope.absence.closest_symbols]
265
+ if len(symbols) == 1:
266
+ lines.append(f"Did you mean: {symbols[0]}?")
267
+ elif len(symbols) == 2:
268
+ lines.append(f"Did you mean: {symbols[0]} or {symbols[1]}?")
269
+ else:
270
+ joined = ", ".join(symbols[:-1]) + f", or {symbols[-1]}"
271
+ lines.append(f"Did you mean: {joined}?")
272
+
273
+ return "\n".join(lines)
274
+
275
+ return base
230
276
 
231
277
 
232
278
  def _render_listing(envelope: Envelope, *, noun: str, detail: str = "normal") -> str:
@@ -295,7 +341,12 @@ def _render_listing(envelope: Envelope, *, noun: str, detail: str = "normal") ->
295
341
  if rest:
296
342
  lines.extend(_render_inspect_block(rest, 1))
297
343
  if not lines:
298
- lines.append(f"0 {noun}".rstrip())
344
+ # Handle absence diagnosis (PR-ABS-4)
345
+ if envelope.absence is not None:
346
+ vline = _verdict_line(envelope.absence)
347
+ lines.append(vline if vline else f"0 {noun}".rstrip())
348
+ else:
349
+ lines.append(f"0 {noun}".rstrip())
299
350
  # Listing breadcrumbs (Phase 2): <=2 `next:` hint lines when the listing
300
351
  # command emitted agent_next_actions (routes/clients/producers/topics).
301
352
  lines.extend(_next_action_lines(envelope))
@@ -413,10 +464,23 @@ def _render_traversal(envelope: Envelope, *, noun: str, detail: str = "normal")
413
464
  root_node = envelope.nodes.get(root_id, {})
414
465
  root_fqn = str(root_node.get("fqn") or "").strip()
415
466
  root_svc = str(root_node.get("microservice") or "").strip()
416
- if envelope.is_external_entrypoint:
467
+
468
+ # Handle absence diagnosis (PR-ABS-4)
469
+ if envelope.absence is not None:
470
+ absence = envelope.absence
471
+ if absence.verdict == "correct_empty":
472
+ # Same text as the is_external_entrypoint case.
473
+ parts = ["external entrypoint — no in-repo callers"]
474
+ else:
475
+ vline = _verdict_line(absence)
476
+ if vline:
477
+ lines.append(vline)
478
+ parts = [f"0 {noun}".rstrip()]
479
+ elif envelope.is_external_entrypoint:
417
480
  parts = ["external entrypoint — no in-repo callers"]
418
481
  else:
419
482
  parts = [f"0 {noun}".rstrip()]
483
+
420
484
  if root_fqn:
421
485
  parts.append(root_fqn)
422
486
  if root_svc:
File without changes
@@ -13,7 +13,7 @@ from __future__ import annotations
13
13
  import json
14
14
  from typing import Any, Literal, NamedTuple
15
15
 
16
- from java_ontology import EDGE_SCHEMA, FUZZY_STRATEGY_SET
16
+ from java_codebase_rag.graph.java_ontology import EDGE_SCHEMA, FUZZY_STRATEGY_SET
17
17
 
18
18
  # Normative schema description (propose §3.1) — imported by ``mcp_v2`` for Field(description=...).
19
19
  MCP_HINTS_STRUCTURED_FIELD_DESCRIPTION = (