java-codebase-rag 0.12.0__py3-none-any.whl → 0.12.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. java_codebase_rag-0.12.2.dist-info/METADATA +35 -0
  2. java_codebase_rag-0.12.2.dist-info/RECORD +4 -0
  3. {java_codebase_rag-0.12.0.dist-info → java_codebase_rag-0.12.2.dist-info}/WHEEL +1 -1
  4. java_codebase_rag/_deprecation.py +0 -103
  5. java_codebase_rag/_fdlimit.py +0 -56
  6. java_codebase_rag/_stdio.py +0 -32
  7. java_codebase_rag/_version.py +0 -35
  8. java_codebase_rag/absence/__init__.py +0 -0
  9. java_codebase_rag/absence/absence_diagnosis.py +0 -700
  10. java_codebase_rag/absence/absence_types.py +0 -124
  11. java_codebase_rag/absence/absence_vocab.py +0 -460
  12. java_codebase_rag/analysis/__init__.py +0 -0
  13. java_codebase_rag/analysis/pr_analysis.py +0 -563
  14. java_codebase_rag/analysis/resolve_service.py +0 -740
  15. java_codebase_rag/ast/__init__.py +0 -0
  16. java_codebase_rag/ast/ast_java.py +0 -2847
  17. java_codebase_rag/ast/ast_kotlin.py +0 -1794
  18. java_codebase_rag/ast/brownfield_events.py +0 -58
  19. java_codebase_rag/ast/chunk_heuristics.py +0 -83
  20. java_codebase_rag/ast/language.py +0 -117
  21. java_codebase_rag/cli.py +0 -1215
  22. java_codebase_rag/cli_dispatch.py +0 -251
  23. java_codebase_rag/cli_format.py +0 -85
  24. java_codebase_rag/cli_progress.py +0 -94
  25. java_codebase_rag/config.py +0 -833
  26. java_codebase_rag/eval/__init__.py +0 -1
  27. java_codebase_rag/eval/ground_truth.py +0 -100
  28. java_codebase_rag/eval/metrics.py +0 -107
  29. java_codebase_rag/eval/runner.py +0 -556
  30. java_codebase_rag/graph/__init__.py +0 -0
  31. java_codebase_rag/graph/build_ast_graph.py +0 -4593
  32. java_codebase_rag/graph/graph_enrich.py +0 -1940
  33. java_codebase_rag/graph/graph_types.py +0 -224
  34. java_codebase_rag/graph/java_ontology.py +0 -465
  35. java_codebase_rag/graph/ladybug_queries.py +0 -2213
  36. java_codebase_rag/graph/path_filtering.py +0 -509
  37. java_codebase_rag/index/__init__.py +0 -0
  38. java_codebase_rag/index/java_index_flow_lancedb.py +0 -879
  39. java_codebase_rag/index/java_index_v1_common.py +0 -33
  40. java_codebase_rag/install_data/__init__.py +0 -0
  41. java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -110
  42. java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
  43. java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
  44. java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
  45. java_codebase_rag/installer.py +0 -2188
  46. java_codebase_rag/jrag.py +0 -4545
  47. java_codebase_rag/jrag_envelope.py +0 -1107
  48. java_codebase_rag/jrag_hints.py +0 -204
  49. java_codebase_rag/jrag_render.py +0 -926
  50. java_codebase_rag/lance_optimize.py +0 -264
  51. java_codebase_rag/mcp/__init__.py +0 -0
  52. java_codebase_rag/mcp/mcp_hints.py +0 -932
  53. java_codebase_rag/mcp/mcp_v2.py +0 -1916
  54. java_codebase_rag/mcp/server.py +0 -886
  55. java_codebase_rag/pipeline.py +0 -531
  56. java_codebase_rag/progress.py +0 -570
  57. java_codebase_rag/read_payloads.py +0 -781
  58. java_codebase_rag/search/__init__.py +0 -0
  59. java_codebase_rag/search/index_common.py +0 -10
  60. java_codebase_rag/search/search_lancedb.py +0 -1296
  61. java_codebase_rag/search/search_lexical.py +0 -449
  62. java_codebase_rag/search/search_scoring.py +0 -537
  63. java_codebase_rag/watch/__init__.py +0 -0
  64. java_codebase_rag/watch/client.py +0 -230
  65. java_codebase_rag/watch/daemon.py +0 -396
  66. java_codebase_rag/watch/lock.py +0 -201
  67. java_codebase_rag/watch/paths.py +0 -76
  68. java_codebase_rag/watch/protocol.py +0 -122
  69. java_codebase_rag/watch/server.py +0 -273
  70. java_codebase_rag/watch/warm.py +0 -105
  71. java_codebase_rag/watch/watcher.py +0 -394
  72. java_codebase_rag-0.12.0.dist-info/METADATA +0 -340
  73. java_codebase_rag-0.12.0.dist-info/RECORD +0 -75
  74. java_codebase_rag-0.12.0.dist-info/entry_points.txt +0 -5
  75. java_codebase_rag-0.12.0.dist-info/licenses/LICENSE +0 -21
  76. java_codebase_rag-0.12.0.dist-info/top_level.txt +0 -1
  77. /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.2.dist-info/top_level.txt +0 -0
@@ -1,264 +0,0 @@
1
- """Serialized post-flow LanceDB optimize with commit-conflict retry.
2
-
3
- Historically (cocoindex 1.0.7) this existed because cocoindex scheduled
4
- ``table.optimize()`` (a LanceDB **Rewrite**/compaction) as a *background*
5
- ``asyncio`` task that raced concurrent ``table.delete()`` (**Delete**)
6
- transactions — LanceDB does not allow a Rewrite to commit concurrently with a
7
- Delete (upstream lancedb#1504), surfacing as a flood of::
8
-
9
- RuntimeError: lance error: Retryable commit conflict for version N: \
10
- This Rewrite transaction was preempted by concurrent transaction Delete ...
11
-
12
- cocoindex >=1.0.15 made optimize **inline and stats-driven** (only compacts
13
- when small fragments accumulate, inside the merge_insert commit path), so the
14
- race is gone and the flow no longer disables anything. This module still runs a
15
- *single*, serialized optimize after the flow returns (exit 0 → no concurrent
16
- writers) as a clean final compaction + scalar/FTS index build, retrying the rare
17
- residual commit conflict that two internal compaction passes can still produce.
18
- """
19
- from __future__ import annotations
20
-
21
- import asyncio
22
- import sys
23
- import time
24
- from pathlib import Path
25
- from typing import Callable, Literal
26
-
27
- # Mirrors ``ProgressStatus`` in ``progress.py``; kept local (rather than imported)
28
- # so this module never pays the ``rich`` cost at import time — see
29
- # ``_make_optimize_event``.
30
- _OptimizeStatus = Literal["running", "done", "failed"]
31
-
32
- # Single source of truth for the three Lance table names created by the flow.
33
- # Keep in sync with ``search_lancedb.TABLES`` (the values there mirror these).
34
- LANCE_TABLE_NAMES: tuple[str, ...] = (
35
- "javacodeindex_java_code",
36
- "sqlschemaindex_sql_schema",
37
- "yamlconfigindex_yaml_config",
38
- )
39
-
40
-
41
- def _make_optimize_event(
42
- *,
43
- status: _OptimizeStatus,
44
- elapsed_s: float | None = None,
45
- ):
46
- """Build a ``ProgressEvent(kind="optimize", …)`` lazily (progress is parent-side).
47
-
48
- ``lance_optimize`` runs in-process in the parent (called by
49
- ``pipeline._maybe_run_serialized_optimize`` and
50
- ``server.run_refresh_pipeline``); it routes progress to the renderer via the
51
- in-process ``on_progress`` callback — NOT via stderr (which would corrupt
52
- the Live region). The import is local so the flow (which imports
53
- ``LANCE_TABLE_NAMES`` at definition time) never pays the ``rich`` cost.
54
- """
55
- from java_codebase_rag.progress import ProgressEvent
56
-
57
- return ProgressEvent(
58
- kind="optimize",
59
- phase=None,
60
- pass_=None,
61
- done=None,
62
- total=None,
63
- status=status,
64
- elapsed_s=elapsed_s,
65
- )
66
-
67
- # Commit conflicts are transient; a handful of exponential-backoff retries is
68
- # enough because, post-flow, there are no concurrent writers — only successive
69
- # optimize/compaction passes within this single serialized call can still
70
- # transiently preempt one another.
71
- _MAX_ATTEMPTS = 6
72
- _BASE_BACKOFF_S = 0.1
73
-
74
- # Substrings identifying the retryable Lance commit-conflict error. LanceDB
75
- # wraps the underlying lance error text into the raised ``RuntimeError`` str,
76
- # so a substring match is the robust detector (no dedicated exception type).
77
- _RETRYABLE_MARKERS = (
78
- "Retryable commit conflict",
79
- "preempted by concurrent transaction",
80
- )
81
-
82
-
83
- def _is_retryable(exc: BaseException) -> bool:
84
- text = str(exc)
85
- return any(marker in text for marker in _RETRYABLE_MARKERS)
86
-
87
-
88
- async def _list_table_names(db: object) -> set[str]:
89
- """Existing table names across LanceDB API variants (``list_tables`` ≥ ``table_names``)."""
90
- if hasattr(db, "list_tables"):
91
- response = await db.list_tables()
92
- return set(getattr(response, "tables", response))
93
- return set(await db.table_names())
94
-
95
-
96
- async def optimize_lance_tables(
97
- index_dir: Path,
98
- *,
99
- quiet: bool = False,
100
- on_progress: Callable | None = None,
101
- ) -> dict[str, str]:
102
- """Optimize all known Lance tables under *index_dir*, serially, with retry.
103
-
104
- Runs ``table.optimize()`` for each name in :data:`LANCE_TABLE_NAMES` that
105
- exists in the DB. Retryable commit conflicts are retried with exponential
106
- backoff; any other exception (or an exhausted retry budget) is captured
107
- per-table in the returned dict and logged to **stderr** — never stdout,
108
- since this is callable from stdio-MCP / JSON-stdout contexts.
109
-
110
- Args:
111
- index_dir: directory holding the Lance tables (the flow's LanceDB URI).
112
- quiet: when True, suppress the per-table success/skip info lines on
113
- stderr (errors are always logged).
114
- on_progress: optional in-process progress callback (the parent's
115
- renderer ``on_progress``). When given, emits
116
- ``ProgressEvent(kind="optimize", status="running")`` on entry and a
117
- terminal ``status="done"``/``"failed"`` event on exit (covers BOTH
118
- call sites: ``pipeline._maybe_run_serialized_optimize`` and
119
- ``server.run_refresh_pipeline``). In-process only — NEVER prints to
120
- stderr (that would corrupt the Live region).
121
-
122
- Returns:
123
- Mapping of table name → status. Values are ``"ok"``, ``"skipped"``
124
- (table absent — e.g. a repo with no SQL/YAML), or ``"error: <text>"``.
125
- """
126
- # Lazy import: the flow imports this module for LANCE_TABLE_NAMES and must
127
- # not pay the lancedb import cost at flow-definition time.
128
- import lancedb
129
-
130
- if on_progress is not None:
131
- on_progress(_make_optimize_event(status="running"))
132
- t0 = time.perf_counter()
133
- results: dict[str, str] = {}
134
- failed = False
135
- try:
136
- db = await lancedb.connect_async(str(index_dir))
137
- try:
138
- try:
139
- existing = await _list_table_names(db)
140
- except Exception as exc:
141
- print(
142
- f"jrag: optimize: failed to list tables in "
143
- f"{index_dir}: {exc}",
144
- file=sys.stderr,
145
- )
146
- failed = True
147
- return {name: f"error: list failed: {exc}" for name in LANCE_TABLE_NAMES}
148
-
149
- for name in LANCE_TABLE_NAMES:
150
- if name not in existing:
151
- results[name] = "skipped"
152
- if not quiet:
153
- print(
154
- f"jrag: optimize: {name} absent, skipped",
155
- file=sys.stderr,
156
- )
157
- continue
158
- try:
159
- table = await db.open_table(name)
160
- except Exception as exc:
161
- results[name] = f"error: open failed: {exc}"
162
- failed = True
163
- print(
164
- f"jrag: optimize: {name} open failed: {exc}",
165
- file=sys.stderr,
166
- )
167
- continue
168
-
169
- last_exc: BaseException | None = None
170
- for attempt in range(_MAX_ATTEMPTS):
171
- try:
172
- await table.optimize()
173
- last_exc = None
174
- break
175
- except Exception as exc:
176
- last_exc = exc
177
- if _is_retryable(exc) and attempt < _MAX_ATTEMPTS - 1:
178
- await asyncio.sleep(_BASE_BACKOFF_S * (2**attempt))
179
- continue
180
- # Non-retryable, or retries exhausted: stop the loop and
181
- # surface below — do not swallow silently.
182
- break
183
-
184
- if last_exc is None:
185
- results[name] = "ok"
186
- # Best-effort BTREE scalar index on the primary key ("id").
187
- # cocoindex's merge_insert defaults to use_index=True but
188
- # never creates a scalar PK index itself (declaring
189
- # primary_key in the schema does NOT auto-build a lance
190
- # index), so without this every merge_insert — increment
191
- # included — is a forced full scan of the PK column,
192
- # O(existing rows). On a large repo that scan dominates
193
- # increment wall-clock; with the index present the join does
194
- # lookups (~O(batch*log N)). Failure is non-fatal (the table
195
- # is still correct, just un-indexed) and never alters the
196
- # "ok" status, mirroring the FTS block below. ``replace=True``
197
- # keeps it idempotent across runs; table.optimize() above
198
- # maintains it on subsequent runs.
199
- try:
200
- from lancedb.index import BTree
201
- await table.create_index("id", config=BTree(), replace=True)
202
- except Exception as exc:
203
- low = str(exc).lower()
204
- if not any(
205
- w in low for w in ("exist", "duplicate", "already", "same name")
206
- ) and not quiet:
207
- print(
208
- f"jrag: optimize: {name} id-index skipped: {exc}",
209
- file=sys.stderr,
210
- )
211
- # Best-effort FTS index at index time (PR-SEARCH-3) so hybrid
212
- # search works on all tables (java/sql/yaml) without a
213
- # first-query race. Failure is non-fatal — the lazy
214
- # ensure_text_fts_index in search_lancedb.py is the runtime
215
- # fallback — so it never alters the "ok" optimize status; we
216
- # only log the skip when verbose.
217
- try:
218
- from lancedb.index import FTS
219
- await table.create_index("text", config=FTS(), replace=True)
220
- except Exception as exc:
221
- low = str(exc).lower()
222
- if not any(
223
- w in low for w in ("exist", "duplicate", "already", "same name")
224
- ) and not quiet:
225
- print(
226
- f"jrag: optimize: {name} fts skipped: {exc}",
227
- file=sys.stderr,
228
- )
229
- if not quiet:
230
- print(
231
- f"jrag: optimize: {name} ok",
232
- file=sys.stderr,
233
- )
234
- else:
235
- results[name] = f"error: {last_exc}"
236
- failed = True
237
- print(
238
- f"jrag: optimize: {name} failed: {last_exc}",
239
- file=sys.stderr,
240
- )
241
- finally:
242
- # ``AsyncConnection.close`` is a *sync* method in lancedb 0.30.x.
243
- db.close()
244
- return results
245
- except Exception:
246
- # An unexpected exception (e.g. ``connect_async`` raised, or a table-
247
- # independent failure) must still flip the terminal event to failed so
248
- # the renderer's task doesn't render a green check on a crash. Re-raise
249
- # after marking — the caller (``_maybe_run_serialized_optimize`` /
250
- # ``run_refresh_pipeline``) treats optimize failure as non-fatal and
251
- # logs it, but the renderer must reflect the truth.
252
- failed = True
253
- raise
254
- finally:
255
- # Always emit a terminal optimize event so the renderer's task never
256
- # hangs at "running" — even on exception (the parent treats a failed
257
- # optimize as non-fatal: the index is still searchable un-compacted).
258
- if on_progress is not None:
259
- on_progress(
260
- _make_optimize_event(
261
- status="failed" if failed else "done",
262
- elapsed_s=time.perf_counter() - t0,
263
- )
264
- )
File without changes