java-codebase-rag 0.12.0__py3-none-any.whl → 0.12.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- java_codebase_rag-0.12.2.dist-info/METADATA +35 -0
- java_codebase_rag-0.12.2.dist-info/RECORD +4 -0
- {java_codebase_rag-0.12.0.dist-info → java_codebase_rag-0.12.2.dist-info}/WHEEL +1 -1
- java_codebase_rag/_deprecation.py +0 -103
- java_codebase_rag/_fdlimit.py +0 -56
- java_codebase_rag/_stdio.py +0 -32
- java_codebase_rag/_version.py +0 -35
- java_codebase_rag/absence/__init__.py +0 -0
- java_codebase_rag/absence/absence_diagnosis.py +0 -700
- java_codebase_rag/absence/absence_types.py +0 -124
- java_codebase_rag/absence/absence_vocab.py +0 -460
- java_codebase_rag/analysis/__init__.py +0 -0
- java_codebase_rag/analysis/pr_analysis.py +0 -563
- java_codebase_rag/analysis/resolve_service.py +0 -740
- java_codebase_rag/ast/__init__.py +0 -0
- java_codebase_rag/ast/ast_java.py +0 -2847
- java_codebase_rag/ast/ast_kotlin.py +0 -1794
- java_codebase_rag/ast/brownfield_events.py +0 -58
- java_codebase_rag/ast/chunk_heuristics.py +0 -83
- java_codebase_rag/ast/language.py +0 -117
- java_codebase_rag/cli.py +0 -1215
- java_codebase_rag/cli_dispatch.py +0 -251
- java_codebase_rag/cli_format.py +0 -85
- java_codebase_rag/cli_progress.py +0 -94
- java_codebase_rag/config.py +0 -833
- java_codebase_rag/eval/__init__.py +0 -1
- java_codebase_rag/eval/ground_truth.py +0 -100
- java_codebase_rag/eval/metrics.py +0 -107
- java_codebase_rag/eval/runner.py +0 -556
- java_codebase_rag/graph/__init__.py +0 -0
- java_codebase_rag/graph/build_ast_graph.py +0 -4593
- java_codebase_rag/graph/graph_enrich.py +0 -1940
- java_codebase_rag/graph/graph_types.py +0 -224
- java_codebase_rag/graph/java_ontology.py +0 -465
- java_codebase_rag/graph/ladybug_queries.py +0 -2213
- java_codebase_rag/graph/path_filtering.py +0 -509
- java_codebase_rag/index/__init__.py +0 -0
- java_codebase_rag/index/java_index_flow_lancedb.py +0 -879
- java_codebase_rag/index/java_index_v1_common.py +0 -33
- java_codebase_rag/install_data/__init__.py +0 -0
- java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -110
- java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
- java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
- java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
- java_codebase_rag/installer.py +0 -2188
- java_codebase_rag/jrag.py +0 -4545
- java_codebase_rag/jrag_envelope.py +0 -1107
- java_codebase_rag/jrag_hints.py +0 -204
- java_codebase_rag/jrag_render.py +0 -926
- java_codebase_rag/lance_optimize.py +0 -264
- java_codebase_rag/mcp/__init__.py +0 -0
- java_codebase_rag/mcp/mcp_hints.py +0 -932
- java_codebase_rag/mcp/mcp_v2.py +0 -1916
- java_codebase_rag/mcp/server.py +0 -886
- java_codebase_rag/pipeline.py +0 -531
- java_codebase_rag/progress.py +0 -570
- java_codebase_rag/read_payloads.py +0 -781
- java_codebase_rag/search/__init__.py +0 -0
- java_codebase_rag/search/index_common.py +0 -10
- java_codebase_rag/search/search_lancedb.py +0 -1296
- java_codebase_rag/search/search_lexical.py +0 -449
- java_codebase_rag/search/search_scoring.py +0 -537
- java_codebase_rag/watch/__init__.py +0 -0
- java_codebase_rag/watch/client.py +0 -230
- java_codebase_rag/watch/daemon.py +0 -396
- java_codebase_rag/watch/lock.py +0 -201
- java_codebase_rag/watch/paths.py +0 -76
- java_codebase_rag/watch/protocol.py +0 -122
- java_codebase_rag/watch/server.py +0 -273
- java_codebase_rag/watch/warm.py +0 -105
- java_codebase_rag/watch/watcher.py +0 -394
- java_codebase_rag-0.12.0.dist-info/METADATA +0 -340
- java_codebase_rag-0.12.0.dist-info/RECORD +0 -75
- java_codebase_rag-0.12.0.dist-info/entry_points.txt +0 -5
- java_codebase_rag-0.12.0.dist-info/licenses/LICENSE +0 -21
- java_codebase_rag-0.12.0.dist-info/top_level.txt +0 -1
- /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.2.dist-info/top_level.txt +0 -0
|
@@ -1,563 +0,0 @@
|
|
|
1
|
-
"""Unified-diff → symbol mapping and PR-style risk scoring (B4 / PR-B).
|
|
2
|
-
|
|
3
|
-
Uses the `unidiff` library for parsing. Graph-resident symbols only; newly
|
|
4
|
-
added Java members are not modelled — see `notes` on the returned report.
|
|
5
|
-
"""
|
|
6
|
-
from __future__ import annotations
|
|
7
|
-
|
|
8
|
-
import re
|
|
9
|
-
from dataclasses import asdict, dataclass
|
|
10
|
-
from typing import Any
|
|
11
|
-
|
|
12
|
-
from unidiff import PatchSet
|
|
13
|
-
from unidiff.errors import UnidiffParseError
|
|
14
|
-
|
|
15
|
-
from java_codebase_rag.graph.ladybug_queries import SymbolHit, find_symbols_in_file_range, _row_to_symbol
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
@dataclass
|
|
19
|
-
class DiffHunk:
|
|
20
|
-
"""One unified-diff hunk in the *new* file coordinate system."""
|
|
21
|
-
|
|
22
|
-
target_path: str
|
|
23
|
-
source_path: str
|
|
24
|
-
target_line_start: int # inclusive, 1-based; 0 when the hunk has no new-file lines
|
|
25
|
-
target_line_end: int # inclusive
|
|
26
|
-
source_line_start: int
|
|
27
|
-
source_line_end: int
|
|
28
|
-
source_length: int = 0
|
|
29
|
-
target_length: int = 0
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
@dataclass
|
|
33
|
-
class ChangedSymbol:
|
|
34
|
-
symbol_id: str
|
|
35
|
-
fqn: str
|
|
36
|
-
kind: str # 'method' | 'type' | 'field'
|
|
37
|
-
change_type: str # 'added' | 'removed' | 'modified'
|
|
38
|
-
file: str
|
|
39
|
-
hunk_lines: list[int]
|
|
40
|
-
cross_service_callers_count: int = 0
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
@dataclass
|
|
44
|
-
class PrRiskReport:
|
|
45
|
-
changed_symbols: list[ChangedSymbol]
|
|
46
|
-
blast_radius_total: int
|
|
47
|
-
blast_radius_by_symbol: dict[str, int]
|
|
48
|
-
cross_service_callers: int
|
|
49
|
-
routes_touched: list[str]
|
|
50
|
-
risk_score: float
|
|
51
|
-
risk_band: str
|
|
52
|
-
notes: list[str]
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
_BINARY_DIFF_LINE = re.compile(r"^Binary files .+ differ\s*$")
|
|
56
|
-
# Heuristic: new Java method/ctor-looking line. Covers annotations, method-level
|
|
57
|
-
# generics, `default` interface methods, and return types with spaces (e.g.
|
|
58
|
-
# `Map<String, String> m(`). Misses multi-line signatures, some compact record
|
|
59
|
-
# forms, and unusual annotations; `_notes_for_unindexed_additions` is best-effort.
|
|
60
|
-
_DECL_ADD = re.compile(
|
|
61
|
-
r"^\+\s*"
|
|
62
|
-
r"(?:(?:@[\w.]+\([^)]*\))\s+)*"
|
|
63
|
-
r"(?:<[^>]+>\s+)?"
|
|
64
|
-
r"(?:(?:public|private|protected|default|static|final|synchronized|abstract|native)\s+)*"
|
|
65
|
-
r"(.+?)\s+(\w+)\s*\(",
|
|
66
|
-
)
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
def _strip_ab_prefix(path: str) -> str:
|
|
70
|
-
p = path.strip()
|
|
71
|
-
if p.startswith(("a/", "b/")):
|
|
72
|
-
return p[2:]
|
|
73
|
-
return p
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
def _hunk_ranges(h: Any) -> tuple[tuple[int, int], tuple[int, int]]:
|
|
77
|
-
"""Return ((src_start, src_end inclusive), (tgt_start, tgt_end inclusive))."""
|
|
78
|
-
src_len = int(getattr(h, "source_length", 0) or 0)
|
|
79
|
-
tgt_len = int(getattr(h, "target_length", 0) or 0)
|
|
80
|
-
src_start = int(getattr(h, "source_start", 0) or 0)
|
|
81
|
-
tgt_start = int(getattr(h, "target_start", 0) or 0)
|
|
82
|
-
if src_len <= 0:
|
|
83
|
-
src_start, src_end = 0, 0
|
|
84
|
-
else:
|
|
85
|
-
src_end = src_start + src_len - 1
|
|
86
|
-
if tgt_len <= 0:
|
|
87
|
-
tgt_start, tgt_end = 0, 0
|
|
88
|
-
else:
|
|
89
|
-
tgt_end = tgt_start + tgt_len - 1
|
|
90
|
-
return (src_start, src_end), (tgt_start, tgt_end)
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
def parse_unified_diff(diff_text: str) -> list[DiffHunk]:
|
|
94
|
-
"""Parse `diff_text` into logical hunks (non-binary, non-rename files only)."""
|
|
95
|
-
if not (diff_text or "").strip():
|
|
96
|
-
return []
|
|
97
|
-
try:
|
|
98
|
-
patches = PatchSet(diff_text.splitlines(keepends=True))
|
|
99
|
-
except UnidiffParseError:
|
|
100
|
-
return []
|
|
101
|
-
out: list[DiffHunk] = []
|
|
102
|
-
for pf in patches:
|
|
103
|
-
if getattr(pf, "is_rename", False):
|
|
104
|
-
continue
|
|
105
|
-
tgt = _strip_ab_prefix(str(pf.path or ""))
|
|
106
|
-
src = _strip_ab_prefix(str(getattr(pf, "source_file", "") or pf.path or ""))
|
|
107
|
-
if not tgt:
|
|
108
|
-
continue
|
|
109
|
-
for h in pf:
|
|
110
|
-
(s0, s1), (t0, t1) = _hunk_ranges(h)
|
|
111
|
-
sl = int(getattr(h, "source_length", 0) or 0)
|
|
112
|
-
tl = int(getattr(h, "target_length", 0) or 0)
|
|
113
|
-
out.append(
|
|
114
|
-
DiffHunk(
|
|
115
|
-
target_path=tgt,
|
|
116
|
-
source_path=src,
|
|
117
|
-
target_line_start=t0,
|
|
118
|
-
target_line_end=t1,
|
|
119
|
-
source_line_start=s0,
|
|
120
|
-
source_line_end=s1,
|
|
121
|
-
source_length=sl,
|
|
122
|
-
target_length=tl,
|
|
123
|
-
)
|
|
124
|
-
)
|
|
125
|
-
return out
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
def collect_diff_file_notes(diff_text: str) -> list[str]:
|
|
129
|
-
"""Collect human-readable notes for binary diffs and renames (no crash)."""
|
|
130
|
-
notes: list[str] = []
|
|
131
|
-
if not (diff_text or "").strip():
|
|
132
|
-
return notes
|
|
133
|
-
for line in diff_text.splitlines():
|
|
134
|
-
if _BINARY_DIFF_LINE.match(line):
|
|
135
|
-
notes.append(f"skipped binary diff: {line.strip()}")
|
|
136
|
-
try:
|
|
137
|
-
patches = PatchSet(diff_text.splitlines(keepends=True))
|
|
138
|
-
except UnidiffParseError:
|
|
139
|
-
notes.append("diff text could not be fully parsed as a unified patch")
|
|
140
|
-
return notes
|
|
141
|
-
for pf in patches:
|
|
142
|
-
if getattr(pf, "is_rename", False):
|
|
143
|
-
a = _strip_ab_prefix(str(getattr(pf, "source_file", "") or ""))
|
|
144
|
-
b = _strip_ab_prefix(str(pf.path or ""))
|
|
145
|
-
notes.append(f"rename (symbols not mapped): {a} -> {b}")
|
|
146
|
-
return notes
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
def _resolve_graph_filename(
|
|
150
|
-
graph: Any,
|
|
151
|
-
path: str,
|
|
152
|
-
*,
|
|
153
|
-
ambiguity_notes: list[str] | None = None,
|
|
154
|
-
) -> str | None:
|
|
155
|
-
"""Map a diff path to `Symbol.filename` values stored in LadybugDB."""
|
|
156
|
-
variants = {_strip_ab_prefix(path)}
|
|
157
|
-
for v in list(variants):
|
|
158
|
-
if v.startswith("./"):
|
|
159
|
-
variants.add(v[2:])
|
|
160
|
-
for candidate in variants:
|
|
161
|
-
if not candidate:
|
|
162
|
-
continue
|
|
163
|
-
rows = graph._rows(
|
|
164
|
-
"MATCH (s:Symbol) WHERE s.filename = $fn RETURN s.filename AS fn LIMIT 1",
|
|
165
|
-
{"fn": candidate},
|
|
166
|
-
)
|
|
167
|
-
if rows and rows[0].get("fn"):
|
|
168
|
-
return str(rows[0]["fn"])
|
|
169
|
-
tail = path.strip().split("/")[-1]
|
|
170
|
-
if tail:
|
|
171
|
-
rows = graph._rows(
|
|
172
|
-
"MATCH (s:Symbol) WHERE s.filename ENDS WITH $tail "
|
|
173
|
-
"RETURN DISTINCT s.filename AS fn LIMIT 8",
|
|
174
|
-
{"tail": "/" + tail},
|
|
175
|
-
)
|
|
176
|
-
n = len(rows)
|
|
177
|
-
if n > 1 and ambiguity_notes is not None:
|
|
178
|
-
fns = [str(r.get("fn") or "") for r in rows if r.get("fn")]
|
|
179
|
-
ambiguity_notes.append(
|
|
180
|
-
f"ambiguous filename tail {tail!r} ({n} graph paths); "
|
|
181
|
-
f"ENDS WITH resolution skipped ({', '.join(fns[:4])}"
|
|
182
|
-
f"{'…' if len(fns) > 4 else ''})",
|
|
183
|
-
)
|
|
184
|
-
if n == 1 and rows[0].get("fn"):
|
|
185
|
-
return str(rows[0]["fn"])
|
|
186
|
-
return None
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
def _symbol_to_changed(
|
|
190
|
-
sym: SymbolHit,
|
|
191
|
-
*,
|
|
192
|
-
change_type: str,
|
|
193
|
-
lines: list[int],
|
|
194
|
-
) -> ChangedSymbol:
|
|
195
|
-
kind = sym.kind
|
|
196
|
-
if kind in ("class", "interface", "enum", "record", "annotation"):
|
|
197
|
-
mapped_kind = "type"
|
|
198
|
-
elif kind == "field":
|
|
199
|
-
mapped_kind = "field"
|
|
200
|
-
elif kind == "constructor":
|
|
201
|
-
mapped_kind = "method"
|
|
202
|
-
else:
|
|
203
|
-
mapped_kind = "method"
|
|
204
|
-
uniq = sorted({int(x) for x in lines if int(x) > 0})
|
|
205
|
-
return ChangedSymbol(
|
|
206
|
-
symbol_id=sym.id,
|
|
207
|
-
fqn=sym.fqn,
|
|
208
|
-
kind=mapped_kind,
|
|
209
|
-
change_type=change_type,
|
|
210
|
-
file=sym.filename,
|
|
211
|
-
hunk_lines=uniq,
|
|
212
|
-
)
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
def _decl_added_lines_for_file(diff_text: str, resolved_filename: str) -> int:
|
|
216
|
-
"""Count `+` lines in the diff that look like Java member declarations for one file."""
|
|
217
|
-
lines = diff_text.splitlines()
|
|
218
|
-
in_file = False
|
|
219
|
-
n = 0
|
|
220
|
-
for line in lines:
|
|
221
|
-
if line.startswith("+++ "):
|
|
222
|
-
rest = line[4:].strip()
|
|
223
|
-
if rest.startswith("b/"):
|
|
224
|
-
rest = rest[2:]
|
|
225
|
-
in_file = rest.endswith(resolved_filename) or resolved_filename.endswith(rest)
|
|
226
|
-
continue
|
|
227
|
-
if not in_file:
|
|
228
|
-
continue
|
|
229
|
-
if _DECL_ADD.match(line):
|
|
230
|
-
n += 1
|
|
231
|
-
return n
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
def _notes_for_unindexed_additions(
|
|
235
|
-
graph: Any,
|
|
236
|
-
diff_text: str,
|
|
237
|
-
changed: list[ChangedSymbol],
|
|
238
|
-
hunks: list[DiffHunk],
|
|
239
|
-
) -> list[str]:
|
|
240
|
-
"""Heuristic: added declaration lines vs indexed methods touched on the same file."""
|
|
241
|
-
notes: list[str] = []
|
|
242
|
-
if not diff_text.strip():
|
|
243
|
-
return notes
|
|
244
|
-
for h in hunks:
|
|
245
|
-
tgt_fn = _resolve_graph_filename(graph, h.target_path)
|
|
246
|
-
if not tgt_fn or h.target_line_start <= 0:
|
|
247
|
-
continue
|
|
248
|
-
decls = _decl_added_lines_for_file(diff_text, tgt_fn)
|
|
249
|
-
if decls <= 0:
|
|
250
|
-
continue
|
|
251
|
-
methods_here = [c for c in changed if c.kind == "method" and c.file == tgt_fn]
|
|
252
|
-
if decls > len(methods_here):
|
|
253
|
-
extra = decls - len(methods_here)
|
|
254
|
-
notes.append(
|
|
255
|
-
f"{extra} new method(s) not yet indexed; risk underestimated",
|
|
256
|
-
)
|
|
257
|
-
return notes
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
def map_hunks_to_symbols(
|
|
261
|
-
graph: Any,
|
|
262
|
-
hunks: list[DiffHunk],
|
|
263
|
-
*,
|
|
264
|
-
path_ambiguity_notes: list[str] | None = None,
|
|
265
|
-
) -> list[ChangedSymbol]:
|
|
266
|
-
"""Map diff hunks to overlapping `Symbol` rows (graph-resident only)."""
|
|
267
|
-
by_id: dict[str, ChangedSymbol] = {}
|
|
268
|
-
|
|
269
|
-
def merge(sym: ChangedSymbol) -> None:
|
|
270
|
-
existing = by_id.get(sym.symbol_id)
|
|
271
|
-
if existing is None:
|
|
272
|
-
by_id[sym.symbol_id] = sym
|
|
273
|
-
else:
|
|
274
|
-
if existing.change_type == "modified" or sym.change_type == "modified":
|
|
275
|
-
ct = "modified"
|
|
276
|
-
elif existing.change_type == "removed" or sym.change_type == "removed":
|
|
277
|
-
ct = "removed"
|
|
278
|
-
else:
|
|
279
|
-
ct = sym.change_type
|
|
280
|
-
merged_lines = sorted(set(existing.hunk_lines + sym.hunk_lines))
|
|
281
|
-
by_id[sym.symbol_id] = ChangedSymbol(
|
|
282
|
-
symbol_id=existing.symbol_id,
|
|
283
|
-
fqn=existing.fqn,
|
|
284
|
-
kind=existing.kind,
|
|
285
|
-
change_type=ct,
|
|
286
|
-
file=existing.file,
|
|
287
|
-
hunk_lines=merged_lines,
|
|
288
|
-
)
|
|
289
|
-
|
|
290
|
-
for h in hunks:
|
|
291
|
-
tgt_fn = _resolve_graph_filename(
|
|
292
|
-
graph, h.target_path, ambiguity_notes=path_ambiguity_notes,
|
|
293
|
-
)
|
|
294
|
-
src_fn = (
|
|
295
|
-
_resolve_graph_filename(
|
|
296
|
-
graph, h.source_path, ambiguity_notes=path_ambiguity_notes,
|
|
297
|
-
)
|
|
298
|
-
if h.source_path
|
|
299
|
-
else tgt_fn
|
|
300
|
-
)
|
|
301
|
-
if not tgt_fn and not src_fn:
|
|
302
|
-
continue
|
|
303
|
-
|
|
304
|
-
minus_only = h.target_length == 0 and h.source_length > 0
|
|
305
|
-
|
|
306
|
-
# Removed lines on old file (process before modified so mixed hunks prefer modified)
|
|
307
|
-
if h.source_line_start > 0 and h.source_line_end >= h.source_line_start and src_fn:
|
|
308
|
-
rows = find_symbols_in_file_range(
|
|
309
|
-
graph,
|
|
310
|
-
filename=src_fn,
|
|
311
|
-
start_line=h.source_line_start,
|
|
312
|
-
end_line=h.source_line_end,
|
|
313
|
-
)
|
|
314
|
-
for sym in rows:
|
|
315
|
-
if sym.kind == "file":
|
|
316
|
-
continue
|
|
317
|
-
overlap = list(range(
|
|
318
|
-
max(h.source_line_start, sym.start_line),
|
|
319
|
-
min(h.source_line_end, sym.end_line) + 1,
|
|
320
|
-
))
|
|
321
|
-
if minus_only:
|
|
322
|
-
merge(_symbol_to_changed(sym, change_type="removed", lines=overlap))
|
|
323
|
-
|
|
324
|
-
# Modified / added lines on new file
|
|
325
|
-
if h.target_line_start > 0 and h.target_line_end >= h.target_line_start and tgt_fn:
|
|
326
|
-
rows = find_symbols_in_file_range(
|
|
327
|
-
graph,
|
|
328
|
-
filename=tgt_fn,
|
|
329
|
-
start_line=h.target_line_start,
|
|
330
|
-
end_line=h.target_line_end,
|
|
331
|
-
)
|
|
332
|
-
for sym in rows:
|
|
333
|
-
if sym.kind == "file":
|
|
334
|
-
continue
|
|
335
|
-
merge(_symbol_to_changed(sym, change_type="modified", lines=list(range(
|
|
336
|
-
max(h.target_line_start, sym.start_line),
|
|
337
|
-
min(h.target_line_end, sym.end_line) + 1,
|
|
338
|
-
))))
|
|
339
|
-
|
|
340
|
-
return list(by_id.values())
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
def _impact_needle_for_changed(_graph: Any, fqn: str, mapped_kind: str) -> str:
|
|
344
|
-
"""Pick the `impact_analysis` needle: type FQN for members, else the symbol FQN."""
|
|
345
|
-
if mapped_kind in ("method", "field", "constructor"):
|
|
346
|
-
if "#" in fqn:
|
|
347
|
-
return fqn.split("#", 1)[0]
|
|
348
|
-
return fqn
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
def _is_public_interface_method(graph: Any, sym: SymbolHit) -> bool:
|
|
352
|
-
if sym.kind != "method":
|
|
353
|
-
return False
|
|
354
|
-
if "private" in (sym.modifiers or []):
|
|
355
|
-
return False
|
|
356
|
-
type_fqn = sym.fqn.split("#", 1)[0] if "#" in sym.fqn else sym.fqn
|
|
357
|
-
rows = graph._rows(
|
|
358
|
-
"MATCH (t:Symbol) WHERE t.fqn = $f AND t.kind = 'interface' RETURN t.id LIMIT 1",
|
|
359
|
-
{"f": type_fqn},
|
|
360
|
-
)
|
|
361
|
-
return bool(rows)
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
def _route_ids_for_symbol(graph: Any, symbol_id: str) -> list[str]:
|
|
365
|
-
# Note: LadybugDB rejects `ORDER BY r.id` together with `RETURN DISTINCT r.id` (binder loses `r`).
|
|
366
|
-
q = (
|
|
367
|
-
"MATCH (s:Symbol)-[e:EXPOSES]->(r:Route) WHERE s.id = $sid "
|
|
368
|
-
"RETURN r.id AS id ORDER BY id"
|
|
369
|
-
)
|
|
370
|
-
seen: set[str] = set()
|
|
371
|
-
out: list[str] = []
|
|
372
|
-
for row in graph._rows(q, {"sid": symbol_id}):
|
|
373
|
-
rid = str(row.get("id") or "")
|
|
374
|
-
if rid and rid not in seen:
|
|
375
|
-
seen.add(rid)
|
|
376
|
-
out.append(rid)
|
|
377
|
-
return out
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
def _route_natural_id(graph: Any, rid: str) -> str:
|
|
381
|
-
"""Map a raw Route node id to its agent-facing natural identifier.
|
|
382
|
-
|
|
383
|
-
Mirrors the envelope contract (``METHOD path`` for HTTP endpoints,
|
|
384
|
-
``topic:<name>`` for kafka topics surfaced as :Route) so ``routes_touched``
|
|
385
|
-
in the PR risk report is readable instead of leaking raw graph ids like
|
|
386
|
-
``r:970ffaa960a4f65d``. Falls back to ``rid`` only if the route vanished.
|
|
387
|
-
"""
|
|
388
|
-
rows = graph._rows(
|
|
389
|
-
"MATCH (r:Route {id: $rid}) "
|
|
390
|
-
"RETURN r.method AS method, r.path_template AS path_template, "
|
|
391
|
-
"r.path AS path, r.topic AS topic LIMIT 1",
|
|
392
|
-
{"rid": rid},
|
|
393
|
-
)
|
|
394
|
-
if not rows:
|
|
395
|
-
return rid
|
|
396
|
-
r = rows[0]
|
|
397
|
-
method = str(r.get("method") or "")
|
|
398
|
-
path = str(r.get("path_template") or r.get("path") or "")
|
|
399
|
-
if method or path:
|
|
400
|
-
return f"{method} {path}".strip()
|
|
401
|
-
topic = str(r.get("topic") or "")
|
|
402
|
-
return f"topic:{topic}" if topic else rid
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
def compute_risk(graph: Any, changed: list[ChangedSymbol]) -> PrRiskReport:
|
|
406
|
-
"""Aggregate blast radius, routes, cross-service callers, and v1 risk score.
|
|
407
|
-
|
|
408
|
-
Risk score stays in [0, 1]. Cross-service route callers add a bounded
|
|
409
|
-
bump (up to +1.0) after normalization so they influence rank while
|
|
410
|
-
preserving the public scalar contract.
|
|
411
|
-
"""
|
|
412
|
-
blast_by: dict[str, int] = {}
|
|
413
|
-
blast_total = 0
|
|
414
|
-
routes: list[str] = []
|
|
415
|
-
cross_total = 0
|
|
416
|
-
|
|
417
|
-
sym_cols = (
|
|
418
|
-
"id", "kind", "name", "fqn", "package", "module", "microservice",
|
|
419
|
-
"filename", "start_line", "end_line", "start_byte", "end_byte",
|
|
420
|
-
"modifiers", "annotations", "capabilities", "role", "signature",
|
|
421
|
-
"parent_id", "resolved",
|
|
422
|
-
)
|
|
423
|
-
_sym_return = ", ".join(f"s.{c} AS {c}" for c in sym_cols)
|
|
424
|
-
|
|
425
|
-
iface_hit = 0.0
|
|
426
|
-
enriched_changed: list[ChangedSymbol] = []
|
|
427
|
-
for cs in changed:
|
|
428
|
-
sym_row = graph._rows(
|
|
429
|
-
"MATCH (s:Symbol) WHERE s.id = $id RETURN " + _sym_return,
|
|
430
|
-
{"id": cs.symbol_id},
|
|
431
|
-
)
|
|
432
|
-
if not sym_row:
|
|
433
|
-
continue
|
|
434
|
-
row0 = sym_row[0]
|
|
435
|
-
if iface_hit < 1.0:
|
|
436
|
-
sym = _row_to_symbol(row0)
|
|
437
|
-
if _is_public_interface_method(graph, sym):
|
|
438
|
-
iface_hit = 1.0
|
|
439
|
-
fqn = str(row0.get("fqn") or cs.fqn)
|
|
440
|
-
needle = _impact_needle_for_changed(graph, fqn, cs.kind)
|
|
441
|
-
ia = graph.impact_analysis(needle, depth=2, limit=400)
|
|
442
|
-
n = len(ia)
|
|
443
|
-
# Key blast radius by the symbol's FQN (agent-facing identifier), not
|
|
444
|
-
# its raw graph id — the report is operator-facing JSON.
|
|
445
|
-
blast_by[fqn or cs.symbol_id] = n
|
|
446
|
-
blast_total += n
|
|
447
|
-
|
|
448
|
-
for e in graph.find_callers(cs.fqn, depth=2, limit=400):
|
|
449
|
-
if (
|
|
450
|
-
e.src.microservice
|
|
451
|
-
and e.dst.microservice
|
|
452
|
-
and e.src.microservice != e.dst.microservice
|
|
453
|
-
):
|
|
454
|
-
cross_total += 1
|
|
455
|
-
|
|
456
|
-
cs_cross_service = 0
|
|
457
|
-
route_ids = _route_ids_for_symbol(graph, cs.symbol_id)
|
|
458
|
-
for rid in route_ids:
|
|
459
|
-
# Record the route by its natural identifier (METHOD path /
|
|
460
|
-
# topic:name), not the raw graph id.
|
|
461
|
-
label = _route_natural_id(graph, rid)
|
|
462
|
-
if label and label not in routes:
|
|
463
|
-
routes.append(label)
|
|
464
|
-
callers = graph._rows(
|
|
465
|
-
"MATCH (s:Symbol)-[:DECLARES_CLIENT]->(c:Client)-[e:HTTP_CALLS]->(r:Route {id: $rid}) "
|
|
466
|
-
"WHERE e.match = 'cross_service' "
|
|
467
|
-
"RETURN c.id AS id LIMIT 500",
|
|
468
|
-
{"rid": rid},
|
|
469
|
-
)
|
|
470
|
-
callers += graph._rows(
|
|
471
|
-
"MATCH (s:Symbol)-[:DECLARES_PRODUCER]->(p:Producer)-[e:ASYNC_CALLS]->(r:Route {id: $rid}) "
|
|
472
|
-
"WHERE e.match = 'cross_service' "
|
|
473
|
-
"RETURN p.id AS id LIMIT 500",
|
|
474
|
-
{"rid": rid},
|
|
475
|
-
)
|
|
476
|
-
cs_cross_service += len(callers)
|
|
477
|
-
enriched_changed.append(
|
|
478
|
-
ChangedSymbol(
|
|
479
|
-
symbol_id=cs.symbol_id,
|
|
480
|
-
fqn=cs.fqn,
|
|
481
|
-
kind=cs.kind,
|
|
482
|
-
change_type=cs.change_type,
|
|
483
|
-
file=cs.file,
|
|
484
|
-
hunk_lines=list(cs.hunk_lines),
|
|
485
|
-
cross_service_callers_count=cs_cross_service,
|
|
486
|
-
),
|
|
487
|
-
)
|
|
488
|
-
|
|
489
|
-
def _normalize(x: float, ceiling: float) -> float:
|
|
490
|
-
if ceiling <= 0:
|
|
491
|
-
return 0.0
|
|
492
|
-
return min(float(x), ceiling) / ceiling
|
|
493
|
-
|
|
494
|
-
# v1 risk weights / ceilings (PR-B §1.2): intentionally simple baselines;
|
|
495
|
-
# these constants are expected to be tuned after real-world use — do not treat as stable.
|
|
496
|
-
w_blast, cap_blast = 0.4, 100.0
|
|
497
|
-
w_cross, cap_cross = 0.3, 20.0
|
|
498
|
-
w_iface = 0.2
|
|
499
|
-
w_routes, cap_routes = 0.1, 5.0
|
|
500
|
-
|
|
501
|
-
raw = (
|
|
502
|
-
w_blast * _normalize(float(blast_total), cap_blast)
|
|
503
|
-
+ w_cross * _normalize(float(cross_total), cap_cross)
|
|
504
|
-
+ w_iface * iface_hit
|
|
505
|
-
+ w_routes * _normalize(float(len(routes)), cap_routes)
|
|
506
|
-
)
|
|
507
|
-
cross_service_bonus = min(
|
|
508
|
-
5.0,
|
|
509
|
-
float(sum(c.cross_service_callers_count for c in enriched_changed)),
|
|
510
|
-
)
|
|
511
|
-
score = max(0.0, min(1.0, raw + (cross_service_bonus / 5.0)))
|
|
512
|
-
if score < 0.3:
|
|
513
|
-
band = "low"
|
|
514
|
-
elif score < 0.7:
|
|
515
|
-
band = "medium"
|
|
516
|
-
else:
|
|
517
|
-
band = "high"
|
|
518
|
-
|
|
519
|
-
return PrRiskReport(
|
|
520
|
-
changed_symbols=list(enriched_changed),
|
|
521
|
-
blast_radius_total=blast_total,
|
|
522
|
-
blast_radius_by_symbol=blast_by,
|
|
523
|
-
cross_service_callers=cross_total,
|
|
524
|
-
routes_touched=routes,
|
|
525
|
-
risk_score=score,
|
|
526
|
-
risk_band=band,
|
|
527
|
-
notes=[],
|
|
528
|
-
)
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
def pr_report_to_dict(rep: PrRiskReport) -> dict[str, Any]:
|
|
532
|
-
return {
|
|
533
|
-
"changed_symbols": [asdict(c) for c in rep.changed_symbols],
|
|
534
|
-
"blast_radius_total": rep.blast_radius_total,
|
|
535
|
-
"blast_radius_by_symbol": dict(rep.blast_radius_by_symbol),
|
|
536
|
-
"cross_service_callers": rep.cross_service_callers,
|
|
537
|
-
"routes_touched": list(rep.routes_touched),
|
|
538
|
-
"risk_score": rep.risk_score,
|
|
539
|
-
"risk_band": rep.risk_band,
|
|
540
|
-
"notes": list(rep.notes),
|
|
541
|
-
}
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
def analyze_pr_pipeline(graph: Any, diff_unified: str) -> PrRiskReport:
|
|
545
|
-
"""Full PR-B pipeline: parse → notes → map → risk."""
|
|
546
|
-
notes = collect_diff_file_notes(diff_unified)
|
|
547
|
-
hunks = parse_unified_diff(diff_unified)
|
|
548
|
-
path_amb: list[str] = []
|
|
549
|
-
changed = map_hunks_to_symbols(graph, hunks, path_ambiguity_notes=path_amb)
|
|
550
|
-
notes.extend(path_amb)
|
|
551
|
-
notes.extend(_notes_for_unindexed_additions(graph, diff_unified, changed, hunks))
|
|
552
|
-
rep = compute_risk(graph, changed)
|
|
553
|
-
merged = list(dict.fromkeys([*notes, *rep.notes]))
|
|
554
|
-
return PrRiskReport(
|
|
555
|
-
changed_symbols=rep.changed_symbols,
|
|
556
|
-
blast_radius_total=rep.blast_radius_total,
|
|
557
|
-
blast_radius_by_symbol=rep.blast_radius_by_symbol,
|
|
558
|
-
cross_service_callers=rep.cross_service_callers,
|
|
559
|
-
routes_touched=rep.routes_touched,
|
|
560
|
-
risk_score=rep.risk_score,
|
|
561
|
-
risk_band=rep.risk_band,
|
|
562
|
-
notes=merged,
|
|
563
|
-
)
|