java-codebase-rag 0.12.0__py3-none-any.whl → 0.12.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. java_codebase_rag-0.12.2.dist-info/METADATA +35 -0
  2. java_codebase_rag-0.12.2.dist-info/RECORD +4 -0
  3. {java_codebase_rag-0.12.0.dist-info → java_codebase_rag-0.12.2.dist-info}/WHEEL +1 -1
  4. java_codebase_rag/_deprecation.py +0 -103
  5. java_codebase_rag/_fdlimit.py +0 -56
  6. java_codebase_rag/_stdio.py +0 -32
  7. java_codebase_rag/_version.py +0 -35
  8. java_codebase_rag/absence/__init__.py +0 -0
  9. java_codebase_rag/absence/absence_diagnosis.py +0 -700
  10. java_codebase_rag/absence/absence_types.py +0 -124
  11. java_codebase_rag/absence/absence_vocab.py +0 -460
  12. java_codebase_rag/analysis/__init__.py +0 -0
  13. java_codebase_rag/analysis/pr_analysis.py +0 -563
  14. java_codebase_rag/analysis/resolve_service.py +0 -740
  15. java_codebase_rag/ast/__init__.py +0 -0
  16. java_codebase_rag/ast/ast_java.py +0 -2847
  17. java_codebase_rag/ast/ast_kotlin.py +0 -1794
  18. java_codebase_rag/ast/brownfield_events.py +0 -58
  19. java_codebase_rag/ast/chunk_heuristics.py +0 -83
  20. java_codebase_rag/ast/language.py +0 -117
  21. java_codebase_rag/cli.py +0 -1215
  22. java_codebase_rag/cli_dispatch.py +0 -251
  23. java_codebase_rag/cli_format.py +0 -85
  24. java_codebase_rag/cli_progress.py +0 -94
  25. java_codebase_rag/config.py +0 -833
  26. java_codebase_rag/eval/__init__.py +0 -1
  27. java_codebase_rag/eval/ground_truth.py +0 -100
  28. java_codebase_rag/eval/metrics.py +0 -107
  29. java_codebase_rag/eval/runner.py +0 -556
  30. java_codebase_rag/graph/__init__.py +0 -0
  31. java_codebase_rag/graph/build_ast_graph.py +0 -4593
  32. java_codebase_rag/graph/graph_enrich.py +0 -1940
  33. java_codebase_rag/graph/graph_types.py +0 -224
  34. java_codebase_rag/graph/java_ontology.py +0 -465
  35. java_codebase_rag/graph/ladybug_queries.py +0 -2213
  36. java_codebase_rag/graph/path_filtering.py +0 -509
  37. java_codebase_rag/index/__init__.py +0 -0
  38. java_codebase_rag/index/java_index_flow_lancedb.py +0 -879
  39. java_codebase_rag/index/java_index_v1_common.py +0 -33
  40. java_codebase_rag/install_data/__init__.py +0 -0
  41. java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -110
  42. java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
  43. java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
  44. java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
  45. java_codebase_rag/installer.py +0 -2188
  46. java_codebase_rag/jrag.py +0 -4545
  47. java_codebase_rag/jrag_envelope.py +0 -1107
  48. java_codebase_rag/jrag_hints.py +0 -204
  49. java_codebase_rag/jrag_render.py +0 -926
  50. java_codebase_rag/lance_optimize.py +0 -264
  51. java_codebase_rag/mcp/__init__.py +0 -0
  52. java_codebase_rag/mcp/mcp_hints.py +0 -932
  53. java_codebase_rag/mcp/mcp_v2.py +0 -1916
  54. java_codebase_rag/mcp/server.py +0 -886
  55. java_codebase_rag/pipeline.py +0 -531
  56. java_codebase_rag/progress.py +0 -570
  57. java_codebase_rag/read_payloads.py +0 -781
  58. java_codebase_rag/search/__init__.py +0 -0
  59. java_codebase_rag/search/index_common.py +0 -10
  60. java_codebase_rag/search/search_lancedb.py +0 -1296
  61. java_codebase_rag/search/search_lexical.py +0 -449
  62. java_codebase_rag/search/search_scoring.py +0 -537
  63. java_codebase_rag/watch/__init__.py +0 -0
  64. java_codebase_rag/watch/client.py +0 -230
  65. java_codebase_rag/watch/daemon.py +0 -396
  66. java_codebase_rag/watch/lock.py +0 -201
  67. java_codebase_rag/watch/paths.py +0 -76
  68. java_codebase_rag/watch/protocol.py +0 -122
  69. java_codebase_rag/watch/server.py +0 -273
  70. java_codebase_rag/watch/warm.py +0 -105
  71. java_codebase_rag/watch/watcher.py +0 -394
  72. java_codebase_rag-0.12.0.dist-info/METADATA +0 -340
  73. java_codebase_rag-0.12.0.dist-info/RECORD +0 -75
  74. java_codebase_rag-0.12.0.dist-info/entry_points.txt +0 -5
  75. java_codebase_rag-0.12.0.dist-info/licenses/LICENSE +0 -21
  76. java_codebase_rag-0.12.0.dist-info/top_level.txt +0 -1
  77. /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.2.dist-info/top_level.txt +0 -0
@@ -1 +0,0 @@
1
- # Eval package for IR metrics
@@ -1,100 +0,0 @@
1
- """Eval ground-truth — Tier-A auto generator + Tier-B file loader.
2
-
3
- Tier-A derives labeled queries deterministically from indexed symbols (no
4
- manual labeling). Tier-B loads hand-curated labeled queries from YAML/JSON.
5
- """
6
-
7
- from __future__ import annotations
8
-
9
- import json
10
- from dataclasses import dataclass
11
- from pathlib import Path
12
- from typing import Iterable, Protocol
13
-
14
- import yaml
15
-
16
- from java_codebase_rag.search.search_scoring import _split_identifier
17
-
18
-
19
- class SymbolLike(Protocol):
20
- """Structural type for symbols — duck-typed .fqn / .name."""
21
-
22
- fqn: str
23
- name: str
24
-
25
-
26
- @dataclass(frozen=True)
27
- class LabeledQuery:
28
- """A labeled retrieval query and its set of relevant Symbol FQNs."""
29
-
30
- query: str
31
- relevant: frozenset[str]
32
- tier: str
33
-
34
-
35
- def build_tier_a(symbols: Iterable[SymbolLike]) -> list[LabeledQuery]:
36
- """Auto-generate labeled queries from each symbol's simple name.
37
-
38
- For each symbol two query strings are derived from its simple name:
39
- 1. The original simple name verbatim (e.g. ``"DistributionChunkService"``)
40
- — matches identifier-joined index text.
41
- 2. A space-joined lowercase token form (e.g. ``"distribution chunk service"``)
42
- produced via ``search_scoring._split_identifier`` so tokenization parity
43
- with the FTS index holds.
44
-
45
- Symbols whose simple name splits to fewer than 2 tokens or is shorter than
46
- 3 characters are skipped (noise). Output is deterministic, sorted by
47
- ``(query, fqn)``; all entries carry ``tier="A"``.
48
- """
49
- out: list[LabeledQuery] = []
50
- for sym in symbols:
51
- name: str = sym.name
52
- if len(name) < 3:
53
- continue
54
- tokens = _split_identifier(name)
55
- if len(tokens) < 2:
56
- continue
57
- fqn: str = sym.fqn
58
- relevant = frozenset({fqn})
59
- # 1. identifier-joined form = ORIGINAL simple name (preserve case).
60
- out.append(LabeledQuery(name, relevant, "A"))
61
- # 2. space-joined lowercase token form.
62
- out.append(LabeledQuery(" ".join(tokens), relevant, "A"))
63
- out.sort(key=lambda q: (q.query, next(iter(q.relevant))))
64
- return out
65
-
66
-
67
- def load_tier_b(path: str | Path) -> list[LabeledQuery]:
68
- """Load hand-curated Tier-B labeled queries from a YAML (``.yaml``/``.yml``)
69
- or JSON (``.json``) file.
70
-
71
- Schema: a list of ``{query: str, relevant: [str, ...]}`` objects.
72
-
73
- Raises:
74
- FileNotFoundError: if the path does not exist (the runner checks
75
- existence before calling, treating absence as "Tier-B disabled").
76
- """
77
- p = Path(path)
78
- if not p.exists():
79
- raise FileNotFoundError(f"Tier-B ground-truth file not found: {p}")
80
-
81
- suffix = p.suffix.lower()
82
- raw = p.read_text()
83
- if suffix in (".yaml", ".yml"):
84
- data = yaml.safe_load(raw)
85
- elif suffix == ".json":
86
- data = json.loads(raw)
87
- else:
88
- # Fall back to YAML (superset of JSON) for unknown extensions.
89
- data = yaml.safe_load(raw)
90
-
91
- out: list[LabeledQuery] = []
92
- for entry in data or []:
93
- out.append(
94
- LabeledQuery(
95
- query=str(entry["query"]),
96
- relevant=frozenset(entry.get("relevant", []) or []),
97
- tier="B",
98
- )
99
- )
100
- return out
@@ -1,107 +0,0 @@
1
- """IR evaluation metrics — pure functions, stdlib only.
2
-
3
- Functions take `retrieved: list[str]` (ordered list of retrieved FQN ids)
4
- and `relevant: set[str]` (ground-truth relevant set).
5
- """
6
-
7
- from __future__ import annotations
8
-
9
-
10
- def recall_at_k(retrieved: list[str], relevant: set[str], k: int) -> float:
11
- """Fraction of relevant documents appearing in retrieved[:k].
12
-
13
- Args:
14
- retrieved: Ordered list of retrieved document IDs.
15
- relevant: Set of ground-truth relevant document IDs.
16
- k: Cut-off rank (1-indexed).
17
-
18
- Returns:
19
- Recall@k in [0.0, 1.0]. Returns 0.0 if relevant is empty.
20
- """
21
- if not relevant:
22
- return 0.0
23
-
24
- retrieved_at_k = set(retrieved[:k])
25
- relevant_retrieved = retrieved_at_k.intersection(relevant)
26
-
27
- return len(relevant_retrieved) / len(relevant)
28
-
29
-
30
- def precision_at_k(retrieved: list[str], relevant: set[str], k: int) -> float:
31
- """Precision at cut-off k: |retrieved[:k] ∩ relevant| / k.
32
-
33
- Args:
34
- retrieved: Ordered list of retrieved document IDs.
35
- relevant: Set of ground-truth relevant document IDs.
36
- k: Cut-off rank (1-indexed).
37
-
38
- Returns:
39
- Precision@k in [0.0, 1.0]. Returns 0.0 if k == 0.
40
- """
41
- if k == 0:
42
- return 0.0
43
-
44
- retrieved_at_k = set(retrieved[:k])
45
- relevant_retrieved = retrieved_at_k.intersection(relevant)
46
-
47
- return len(relevant_retrieved) / k
48
-
49
-
50
- def reciprocal_rank(retrieved: list[str], relevant: set[str]) -> float:
51
- """Reciprocal rank: 1.0 / rank of first retrieved relevant document.
52
-
53
- Args:
54
- retrieved: Ordered list of retrieved document IDs.
55
- relevant: Set of ground-truth relevant document IDs.
56
-
57
- Returns:
58
- Reciprocal rank in [0.0, 1.0]. Returns 0.0 if no relevant document
59
- is retrieved. Ranks are 1-indexed (first hit = 1.0).
60
- """
61
- for rank, doc_id in enumerate(retrieved, start=1):
62
- if doc_id in relevant:
63
- return 1.0 / rank
64
-
65
- return 0.0
66
-
67
-
68
- def mean(values: list[float]) -> float:
69
- """Arithmetic mean of a list of floats.
70
-
71
- Args:
72
- values: List of float values.
73
-
74
- Returns:
75
- Arithmetic mean. Returns 0.0 for empty list.
76
- """
77
- if not values:
78
- return 0.0
79
-
80
- return sum(values) / len(values)
81
-
82
-
83
- def aggregate(per_query: list[dict]) -> dict[str, float]:
84
- """Aggregate per-query metrics into means across all queries.
85
-
86
- Takes a list of per-query metric dicts and computes the mean of each
87
- metric across all queries. All dicts must have the same keys.
88
-
89
- Args:
90
- per_query: List of dicts, each containing metric names as keys
91
- and float values (e.g., {"recall@10": 1.0, "mrr": 0.5, ...}).
92
-
93
- Returns:
94
- Dict with same keys as input, containing mean of each metric
95
- across all queries (e.g., {"recall@10": 0.42, "mrr": 0.55, ...}).
96
- """
97
- if not per_query:
98
- return {}
99
-
100
- metric_names = list(per_query[0].keys())
101
- result: dict[str, float] = {}
102
-
103
- for metric in metric_names:
104
- values = [query_metrics[metric] for query_metrics in per_query]
105
- result[metric] = mean(values)
106
-
107
- return result