java-codebase-rag 0.11.2__py3-none-any.whl → 0.12.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. java_codebase_rag-0.12.1.dist-info/METADATA +35 -0
  2. java_codebase_rag-0.12.1.dist-info/RECORD +4 -0
  3. {java_codebase_rag-0.11.2.dist-info → java_codebase_rag-0.12.1.dist-info}/WHEEL +1 -1
  4. java_codebase_rag/_fdlimit.py +0 -56
  5. java_codebase_rag/_stdio.py +0 -32
  6. java_codebase_rag/_version.py +0 -35
  7. java_codebase_rag/absence/__init__.py +0 -0
  8. java_codebase_rag/absence/absence_diagnosis.py +0 -700
  9. java_codebase_rag/absence/absence_types.py +0 -124
  10. java_codebase_rag/absence/absence_vocab.py +0 -460
  11. java_codebase_rag/analysis/__init__.py +0 -0
  12. java_codebase_rag/analysis/pr_analysis.py +0 -563
  13. java_codebase_rag/analysis/resolve_service.py +0 -740
  14. java_codebase_rag/ast/__init__.py +0 -0
  15. java_codebase_rag/ast/ast_java.py +0 -2825
  16. java_codebase_rag/ast/brownfield_events.py +0 -58
  17. java_codebase_rag/ast/chunk_heuristics.py +0 -62
  18. java_codebase_rag/cli.py +0 -1215
  19. java_codebase_rag/cli_format.py +0 -85
  20. java_codebase_rag/cli_progress.py +0 -94
  21. java_codebase_rag/config.py +0 -833
  22. java_codebase_rag/eval/__init__.py +0 -1
  23. java_codebase_rag/eval/ground_truth.py +0 -100
  24. java_codebase_rag/eval/metrics.py +0 -107
  25. java_codebase_rag/eval/runner.py +0 -556
  26. java_codebase_rag/graph/__init__.py +0 -0
  27. java_codebase_rag/graph/build_ast_graph.py +0 -4471
  28. java_codebase_rag/graph/graph_enrich.py +0 -1937
  29. java_codebase_rag/graph/graph_types.py +0 -224
  30. java_codebase_rag/graph/java_ontology.py +0 -465
  31. java_codebase_rag/graph/ladybug_queries.py +0 -2213
  32. java_codebase_rag/graph/path_filtering.py +0 -477
  33. java_codebase_rag/index/__init__.py +0 -0
  34. java_codebase_rag/index/java_index_flow_lancedb.py +0 -734
  35. java_codebase_rag/index/java_index_v1_common.py +0 -33
  36. java_codebase_rag/install_data/__init__.py +0 -0
  37. java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -108
  38. java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
  39. java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
  40. java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
  41. java_codebase_rag/installer.py +0 -2188
  42. java_codebase_rag/jrag.py +0 -4531
  43. java_codebase_rag/jrag_envelope.py +0 -1107
  44. java_codebase_rag/jrag_hints.py +0 -204
  45. java_codebase_rag/jrag_render.py +0 -926
  46. java_codebase_rag/lance_optimize.py +0 -264
  47. java_codebase_rag/mcp/__init__.py +0 -0
  48. java_codebase_rag/mcp/mcp_hints.py +0 -932
  49. java_codebase_rag/mcp/mcp_v2.py +0 -1916
  50. java_codebase_rag/mcp/server.py +0 -884
  51. java_codebase_rag/pipeline.py +0 -531
  52. java_codebase_rag/progress.py +0 -570
  53. java_codebase_rag/read_payloads.py +0 -781
  54. java_codebase_rag/search/__init__.py +0 -0
  55. java_codebase_rag/search/index_common.py +0 -10
  56. java_codebase_rag/search/search_lancedb.py +0 -1296
  57. java_codebase_rag/search/search_lexical.py +0 -449
  58. java_codebase_rag/search/search_scoring.py +0 -523
  59. java_codebase_rag/watch/__init__.py +0 -0
  60. java_codebase_rag/watch/client.py +0 -230
  61. java_codebase_rag/watch/daemon.py +0 -396
  62. java_codebase_rag/watch/lock.py +0 -201
  63. java_codebase_rag/watch/paths.py +0 -76
  64. java_codebase_rag/watch/protocol.py +0 -122
  65. java_codebase_rag/watch/server.py +0 -273
  66. java_codebase_rag/watch/warm.py +0 -105
  67. java_codebase_rag/watch/watcher.py +0 -370
  68. java_codebase_rag-0.11.2.dist-info/METADATA +0 -331
  69. java_codebase_rag-0.11.2.dist-info/RECORD +0 -71
  70. java_codebase_rag-0.11.2.dist-info/entry_points.txt +0 -4
  71. java_codebase_rag-0.11.2.dist-info/licenses/LICENSE +0 -21
  72. java_codebase_rag-0.11.2.dist-info/top_level.txt +0 -1
  73. /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.1.dist-info/top_level.txt +0 -0
@@ -1 +0,0 @@
1
- # Eval package for IR metrics
@@ -1,100 +0,0 @@
1
- """Eval ground-truth — Tier-A auto generator + Tier-B file loader.
2
-
3
- Tier-A derives labeled queries deterministically from indexed symbols (no
4
- manual labeling). Tier-B loads hand-curated labeled queries from YAML/JSON.
5
- """
6
-
7
- from __future__ import annotations
8
-
9
- import json
10
- from dataclasses import dataclass
11
- from pathlib import Path
12
- from typing import Iterable, Protocol
13
-
14
- import yaml
15
-
16
- from java_codebase_rag.search.search_scoring import _split_identifier
17
-
18
-
19
- class SymbolLike(Protocol):
20
- """Structural type for symbols — duck-typed .fqn / .name."""
21
-
22
- fqn: str
23
- name: str
24
-
25
-
26
- @dataclass(frozen=True)
27
- class LabeledQuery:
28
- """A labeled retrieval query and its set of relevant Symbol FQNs."""
29
-
30
- query: str
31
- relevant: frozenset[str]
32
- tier: str
33
-
34
-
35
- def build_tier_a(symbols: Iterable[SymbolLike]) -> list[LabeledQuery]:
36
- """Auto-generate labeled queries from each symbol's simple name.
37
-
38
- For each symbol two query strings are derived from its simple name:
39
- 1. The original simple name verbatim (e.g. ``"DistributionChunkService"``)
40
- — matches identifier-joined index text.
41
- 2. A space-joined lowercase token form (e.g. ``"distribution chunk service"``)
42
- produced via ``search_scoring._split_identifier`` so tokenization parity
43
- with the FTS index holds.
44
-
45
- Symbols whose simple name splits to fewer than 2 tokens or is shorter than
46
- 3 characters are skipped (noise). Output is deterministic, sorted by
47
- ``(query, fqn)``; all entries carry ``tier="A"``.
48
- """
49
- out: list[LabeledQuery] = []
50
- for sym in symbols:
51
- name: str = sym.name
52
- if len(name) < 3:
53
- continue
54
- tokens = _split_identifier(name)
55
- if len(tokens) < 2:
56
- continue
57
- fqn: str = sym.fqn
58
- relevant = frozenset({fqn})
59
- # 1. identifier-joined form = ORIGINAL simple name (preserve case).
60
- out.append(LabeledQuery(name, relevant, "A"))
61
- # 2. space-joined lowercase token form.
62
- out.append(LabeledQuery(" ".join(tokens), relevant, "A"))
63
- out.sort(key=lambda q: (q.query, next(iter(q.relevant))))
64
- return out
65
-
66
-
67
- def load_tier_b(path: str | Path) -> list[LabeledQuery]:
68
- """Load hand-curated Tier-B labeled queries from a YAML (``.yaml``/``.yml``)
69
- or JSON (``.json``) file.
70
-
71
- Schema: a list of ``{query: str, relevant: [str, ...]}`` objects.
72
-
73
- Raises:
74
- FileNotFoundError: if the path does not exist (the runner checks
75
- existence before calling, treating absence as "Tier-B disabled").
76
- """
77
- p = Path(path)
78
- if not p.exists():
79
- raise FileNotFoundError(f"Tier-B ground-truth file not found: {p}")
80
-
81
- suffix = p.suffix.lower()
82
- raw = p.read_text()
83
- if suffix in (".yaml", ".yml"):
84
- data = yaml.safe_load(raw)
85
- elif suffix == ".json":
86
- data = json.loads(raw)
87
- else:
88
- # Fall back to YAML (superset of JSON) for unknown extensions.
89
- data = yaml.safe_load(raw)
90
-
91
- out: list[LabeledQuery] = []
92
- for entry in data or []:
93
- out.append(
94
- LabeledQuery(
95
- query=str(entry["query"]),
96
- relevant=frozenset(entry.get("relevant", []) or []),
97
- tier="B",
98
- )
99
- )
100
- return out
@@ -1,107 +0,0 @@
1
- """IR evaluation metrics — pure functions, stdlib only.
2
-
3
- Functions take `retrieved: list[str]` (ordered list of retrieved FQN ids)
4
- and `relevant: set[str]` (ground-truth relevant set).
5
- """
6
-
7
- from __future__ import annotations
8
-
9
-
10
- def recall_at_k(retrieved: list[str], relevant: set[str], k: int) -> float:
11
- """Fraction of relevant documents appearing in retrieved[:k].
12
-
13
- Args:
14
- retrieved: Ordered list of retrieved document IDs.
15
- relevant: Set of ground-truth relevant document IDs.
16
- k: Cut-off rank (1-indexed).
17
-
18
- Returns:
19
- Recall@k in [0.0, 1.0]. Returns 0.0 if relevant is empty.
20
- """
21
- if not relevant:
22
- return 0.0
23
-
24
- retrieved_at_k = set(retrieved[:k])
25
- relevant_retrieved = retrieved_at_k.intersection(relevant)
26
-
27
- return len(relevant_retrieved) / len(relevant)
28
-
29
-
30
- def precision_at_k(retrieved: list[str], relevant: set[str], k: int) -> float:
31
- """Precision at cut-off k: |retrieved[:k] ∩ relevant| / k.
32
-
33
- Args:
34
- retrieved: Ordered list of retrieved document IDs.
35
- relevant: Set of ground-truth relevant document IDs.
36
- k: Cut-off rank (1-indexed).
37
-
38
- Returns:
39
- Precision@k in [0.0, 1.0]. Returns 0.0 if k == 0.
40
- """
41
- if k == 0:
42
- return 0.0
43
-
44
- retrieved_at_k = set(retrieved[:k])
45
- relevant_retrieved = retrieved_at_k.intersection(relevant)
46
-
47
- return len(relevant_retrieved) / k
48
-
49
-
50
- def reciprocal_rank(retrieved: list[str], relevant: set[str]) -> float:
51
- """Reciprocal rank: 1.0 / rank of first retrieved relevant document.
52
-
53
- Args:
54
- retrieved: Ordered list of retrieved document IDs.
55
- relevant: Set of ground-truth relevant document IDs.
56
-
57
- Returns:
58
- Reciprocal rank in [0.0, 1.0]. Returns 0.0 if no relevant document
59
- is retrieved. Ranks are 1-indexed (first hit = 1.0).
60
- """
61
- for rank, doc_id in enumerate(retrieved, start=1):
62
- if doc_id in relevant:
63
- return 1.0 / rank
64
-
65
- return 0.0
66
-
67
-
68
- def mean(values: list[float]) -> float:
69
- """Arithmetic mean of a list of floats.
70
-
71
- Args:
72
- values: List of float values.
73
-
74
- Returns:
75
- Arithmetic mean. Returns 0.0 for empty list.
76
- """
77
- if not values:
78
- return 0.0
79
-
80
- return sum(values) / len(values)
81
-
82
-
83
- def aggregate(per_query: list[dict]) -> dict[str, float]:
84
- """Aggregate per-query metrics into means across all queries.
85
-
86
- Takes a list of per-query metric dicts and computes the mean of each
87
- metric across all queries. All dicts must have the same keys.
88
-
89
- Args:
90
- per_query: List of dicts, each containing metric names as keys
91
- and float values (e.g., {"recall@10": 1.0, "mrr": 0.5, ...}).
92
-
93
- Returns:
94
- Dict with same keys as input, containing mean of each metric
95
- across all queries (e.g., {"recall@10": 0.42, "mrr": 0.55, ...}).
96
- """
97
- if not per_query:
98
- return {}
99
-
100
- metric_names = list(per_query[0].keys())
101
- result: dict[str, float] = {}
102
-
103
- for metric in metric_names:
104
- values = [query_metrics[metric] for query_metrics in per_query]
105
- result[metric] = mean(values)
106
-
107
- return result