java-codebase-rag 0.12.0__py3-none-any.whl → 0.12.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. java_codebase_rag-0.12.2.dist-info/METADATA +35 -0
  2. java_codebase_rag-0.12.2.dist-info/RECORD +4 -0
  3. {java_codebase_rag-0.12.0.dist-info → java_codebase_rag-0.12.2.dist-info}/WHEEL +1 -1
  4. java_codebase_rag/_deprecation.py +0 -103
  5. java_codebase_rag/_fdlimit.py +0 -56
  6. java_codebase_rag/_stdio.py +0 -32
  7. java_codebase_rag/_version.py +0 -35
  8. java_codebase_rag/absence/__init__.py +0 -0
  9. java_codebase_rag/absence/absence_diagnosis.py +0 -700
  10. java_codebase_rag/absence/absence_types.py +0 -124
  11. java_codebase_rag/absence/absence_vocab.py +0 -460
  12. java_codebase_rag/analysis/__init__.py +0 -0
  13. java_codebase_rag/analysis/pr_analysis.py +0 -563
  14. java_codebase_rag/analysis/resolve_service.py +0 -740
  15. java_codebase_rag/ast/__init__.py +0 -0
  16. java_codebase_rag/ast/ast_java.py +0 -2847
  17. java_codebase_rag/ast/ast_kotlin.py +0 -1794
  18. java_codebase_rag/ast/brownfield_events.py +0 -58
  19. java_codebase_rag/ast/chunk_heuristics.py +0 -83
  20. java_codebase_rag/ast/language.py +0 -117
  21. java_codebase_rag/cli.py +0 -1215
  22. java_codebase_rag/cli_dispatch.py +0 -251
  23. java_codebase_rag/cli_format.py +0 -85
  24. java_codebase_rag/cli_progress.py +0 -94
  25. java_codebase_rag/config.py +0 -833
  26. java_codebase_rag/eval/__init__.py +0 -1
  27. java_codebase_rag/eval/ground_truth.py +0 -100
  28. java_codebase_rag/eval/metrics.py +0 -107
  29. java_codebase_rag/eval/runner.py +0 -556
  30. java_codebase_rag/graph/__init__.py +0 -0
  31. java_codebase_rag/graph/build_ast_graph.py +0 -4593
  32. java_codebase_rag/graph/graph_enrich.py +0 -1940
  33. java_codebase_rag/graph/graph_types.py +0 -224
  34. java_codebase_rag/graph/java_ontology.py +0 -465
  35. java_codebase_rag/graph/ladybug_queries.py +0 -2213
  36. java_codebase_rag/graph/path_filtering.py +0 -509
  37. java_codebase_rag/index/__init__.py +0 -0
  38. java_codebase_rag/index/java_index_flow_lancedb.py +0 -879
  39. java_codebase_rag/index/java_index_v1_common.py +0 -33
  40. java_codebase_rag/install_data/__init__.py +0 -0
  41. java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -110
  42. java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
  43. java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
  44. java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
  45. java_codebase_rag/installer.py +0 -2188
  46. java_codebase_rag/jrag.py +0 -4545
  47. java_codebase_rag/jrag_envelope.py +0 -1107
  48. java_codebase_rag/jrag_hints.py +0 -204
  49. java_codebase_rag/jrag_render.py +0 -926
  50. java_codebase_rag/lance_optimize.py +0 -264
  51. java_codebase_rag/mcp/__init__.py +0 -0
  52. java_codebase_rag/mcp/mcp_hints.py +0 -932
  53. java_codebase_rag/mcp/mcp_v2.py +0 -1916
  54. java_codebase_rag/mcp/server.py +0 -886
  55. java_codebase_rag/pipeline.py +0 -531
  56. java_codebase_rag/progress.py +0 -570
  57. java_codebase_rag/read_payloads.py +0 -781
  58. java_codebase_rag/search/__init__.py +0 -0
  59. java_codebase_rag/search/index_common.py +0 -10
  60. java_codebase_rag/search/search_lancedb.py +0 -1296
  61. java_codebase_rag/search/search_lexical.py +0 -449
  62. java_codebase_rag/search/search_scoring.py +0 -537
  63. java_codebase_rag/watch/__init__.py +0 -0
  64. java_codebase_rag/watch/client.py +0 -230
  65. java_codebase_rag/watch/daemon.py +0 -396
  66. java_codebase_rag/watch/lock.py +0 -201
  67. java_codebase_rag/watch/paths.py +0 -76
  68. java_codebase_rag/watch/protocol.py +0 -122
  69. java_codebase_rag/watch/server.py +0 -273
  70. java_codebase_rag/watch/warm.py +0 -105
  71. java_codebase_rag/watch/watcher.py +0 -394
  72. java_codebase_rag-0.12.0.dist-info/METADATA +0 -340
  73. java_codebase_rag-0.12.0.dist-info/RECORD +0 -75
  74. java_codebase_rag-0.12.0.dist-info/entry_points.txt +0 -5
  75. java_codebase_rag-0.12.0.dist-info/licenses/LICENSE +0 -21
  76. java_codebase_rag-0.12.0.dist-info/top_level.txt +0 -1
  77. /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.2.dist-info/top_level.txt +0 -0
@@ -1,563 +0,0 @@
1
- """Unified-diff → symbol mapping and PR-style risk scoring (B4 / PR-B).
2
-
3
- Uses the `unidiff` library for parsing. Graph-resident symbols only; newly
4
- added Java members are not modelled — see `notes` on the returned report.
5
- """
6
- from __future__ import annotations
7
-
8
- import re
9
- from dataclasses import asdict, dataclass
10
- from typing import Any
11
-
12
- from unidiff import PatchSet
13
- from unidiff.errors import UnidiffParseError
14
-
15
- from java_codebase_rag.graph.ladybug_queries import SymbolHit, find_symbols_in_file_range, _row_to_symbol
16
-
17
-
18
- @dataclass
19
- class DiffHunk:
20
- """One unified-diff hunk in the *new* file coordinate system."""
21
-
22
- target_path: str
23
- source_path: str
24
- target_line_start: int # inclusive, 1-based; 0 when the hunk has no new-file lines
25
- target_line_end: int # inclusive
26
- source_line_start: int
27
- source_line_end: int
28
- source_length: int = 0
29
- target_length: int = 0
30
-
31
-
32
- @dataclass
33
- class ChangedSymbol:
34
- symbol_id: str
35
- fqn: str
36
- kind: str # 'method' | 'type' | 'field'
37
- change_type: str # 'added' | 'removed' | 'modified'
38
- file: str
39
- hunk_lines: list[int]
40
- cross_service_callers_count: int = 0
41
-
42
-
43
- @dataclass
44
- class PrRiskReport:
45
- changed_symbols: list[ChangedSymbol]
46
- blast_radius_total: int
47
- blast_radius_by_symbol: dict[str, int]
48
- cross_service_callers: int
49
- routes_touched: list[str]
50
- risk_score: float
51
- risk_band: str
52
- notes: list[str]
53
-
54
-
55
- _BINARY_DIFF_LINE = re.compile(r"^Binary files .+ differ\s*$")
56
- # Heuristic: new Java method/ctor-looking line. Covers annotations, method-level
57
- # generics, `default` interface methods, and return types with spaces (e.g.
58
- # `Map<String, String> m(`). Misses multi-line signatures, some compact record
59
- # forms, and unusual annotations; `_notes_for_unindexed_additions` is best-effort.
60
- _DECL_ADD = re.compile(
61
- r"^\+\s*"
62
- r"(?:(?:@[\w.]+\([^)]*\))\s+)*"
63
- r"(?:<[^>]+>\s+)?"
64
- r"(?:(?:public|private|protected|default|static|final|synchronized|abstract|native)\s+)*"
65
- r"(.+?)\s+(\w+)\s*\(",
66
- )
67
-
68
-
69
- def _strip_ab_prefix(path: str) -> str:
70
- p = path.strip()
71
- if p.startswith(("a/", "b/")):
72
- return p[2:]
73
- return p
74
-
75
-
76
- def _hunk_ranges(h: Any) -> tuple[tuple[int, int], tuple[int, int]]:
77
- """Return ((src_start, src_end inclusive), (tgt_start, tgt_end inclusive))."""
78
- src_len = int(getattr(h, "source_length", 0) or 0)
79
- tgt_len = int(getattr(h, "target_length", 0) or 0)
80
- src_start = int(getattr(h, "source_start", 0) or 0)
81
- tgt_start = int(getattr(h, "target_start", 0) or 0)
82
- if src_len <= 0:
83
- src_start, src_end = 0, 0
84
- else:
85
- src_end = src_start + src_len - 1
86
- if tgt_len <= 0:
87
- tgt_start, tgt_end = 0, 0
88
- else:
89
- tgt_end = tgt_start + tgt_len - 1
90
- return (src_start, src_end), (tgt_start, tgt_end)
91
-
92
-
93
- def parse_unified_diff(diff_text: str) -> list[DiffHunk]:
94
- """Parse `diff_text` into logical hunks (non-binary, non-rename files only)."""
95
- if not (diff_text or "").strip():
96
- return []
97
- try:
98
- patches = PatchSet(diff_text.splitlines(keepends=True))
99
- except UnidiffParseError:
100
- return []
101
- out: list[DiffHunk] = []
102
- for pf in patches:
103
- if getattr(pf, "is_rename", False):
104
- continue
105
- tgt = _strip_ab_prefix(str(pf.path or ""))
106
- src = _strip_ab_prefix(str(getattr(pf, "source_file", "") or pf.path or ""))
107
- if not tgt:
108
- continue
109
- for h in pf:
110
- (s0, s1), (t0, t1) = _hunk_ranges(h)
111
- sl = int(getattr(h, "source_length", 0) or 0)
112
- tl = int(getattr(h, "target_length", 0) or 0)
113
- out.append(
114
- DiffHunk(
115
- target_path=tgt,
116
- source_path=src,
117
- target_line_start=t0,
118
- target_line_end=t1,
119
- source_line_start=s0,
120
- source_line_end=s1,
121
- source_length=sl,
122
- target_length=tl,
123
- )
124
- )
125
- return out
126
-
127
-
128
- def collect_diff_file_notes(diff_text: str) -> list[str]:
129
- """Collect human-readable notes for binary diffs and renames (no crash)."""
130
- notes: list[str] = []
131
- if not (diff_text or "").strip():
132
- return notes
133
- for line in diff_text.splitlines():
134
- if _BINARY_DIFF_LINE.match(line):
135
- notes.append(f"skipped binary diff: {line.strip()}")
136
- try:
137
- patches = PatchSet(diff_text.splitlines(keepends=True))
138
- except UnidiffParseError:
139
- notes.append("diff text could not be fully parsed as a unified patch")
140
- return notes
141
- for pf in patches:
142
- if getattr(pf, "is_rename", False):
143
- a = _strip_ab_prefix(str(getattr(pf, "source_file", "") or ""))
144
- b = _strip_ab_prefix(str(pf.path or ""))
145
- notes.append(f"rename (symbols not mapped): {a} -> {b}")
146
- return notes
147
-
148
-
149
- def _resolve_graph_filename(
150
- graph: Any,
151
- path: str,
152
- *,
153
- ambiguity_notes: list[str] | None = None,
154
- ) -> str | None:
155
- """Map a diff path to `Symbol.filename` values stored in LadybugDB."""
156
- variants = {_strip_ab_prefix(path)}
157
- for v in list(variants):
158
- if v.startswith("./"):
159
- variants.add(v[2:])
160
- for candidate in variants:
161
- if not candidate:
162
- continue
163
- rows = graph._rows(
164
- "MATCH (s:Symbol) WHERE s.filename = $fn RETURN s.filename AS fn LIMIT 1",
165
- {"fn": candidate},
166
- )
167
- if rows and rows[0].get("fn"):
168
- return str(rows[0]["fn"])
169
- tail = path.strip().split("/")[-1]
170
- if tail:
171
- rows = graph._rows(
172
- "MATCH (s:Symbol) WHERE s.filename ENDS WITH $tail "
173
- "RETURN DISTINCT s.filename AS fn LIMIT 8",
174
- {"tail": "/" + tail},
175
- )
176
- n = len(rows)
177
- if n > 1 and ambiguity_notes is not None:
178
- fns = [str(r.get("fn") or "") for r in rows if r.get("fn")]
179
- ambiguity_notes.append(
180
- f"ambiguous filename tail {tail!r} ({n} graph paths); "
181
- f"ENDS WITH resolution skipped ({', '.join(fns[:4])}"
182
- f"{'…' if len(fns) > 4 else ''})",
183
- )
184
- if n == 1 and rows[0].get("fn"):
185
- return str(rows[0]["fn"])
186
- return None
187
-
188
-
189
- def _symbol_to_changed(
190
- sym: SymbolHit,
191
- *,
192
- change_type: str,
193
- lines: list[int],
194
- ) -> ChangedSymbol:
195
- kind = sym.kind
196
- if kind in ("class", "interface", "enum", "record", "annotation"):
197
- mapped_kind = "type"
198
- elif kind == "field":
199
- mapped_kind = "field"
200
- elif kind == "constructor":
201
- mapped_kind = "method"
202
- else:
203
- mapped_kind = "method"
204
- uniq = sorted({int(x) for x in lines if int(x) > 0})
205
- return ChangedSymbol(
206
- symbol_id=sym.id,
207
- fqn=sym.fqn,
208
- kind=mapped_kind,
209
- change_type=change_type,
210
- file=sym.filename,
211
- hunk_lines=uniq,
212
- )
213
-
214
-
215
- def _decl_added_lines_for_file(diff_text: str, resolved_filename: str) -> int:
216
- """Count `+` lines in the diff that look like Java member declarations for one file."""
217
- lines = diff_text.splitlines()
218
- in_file = False
219
- n = 0
220
- for line in lines:
221
- if line.startswith("+++ "):
222
- rest = line[4:].strip()
223
- if rest.startswith("b/"):
224
- rest = rest[2:]
225
- in_file = rest.endswith(resolved_filename) or resolved_filename.endswith(rest)
226
- continue
227
- if not in_file:
228
- continue
229
- if _DECL_ADD.match(line):
230
- n += 1
231
- return n
232
-
233
-
234
- def _notes_for_unindexed_additions(
235
- graph: Any,
236
- diff_text: str,
237
- changed: list[ChangedSymbol],
238
- hunks: list[DiffHunk],
239
- ) -> list[str]:
240
- """Heuristic: added declaration lines vs indexed methods touched on the same file."""
241
- notes: list[str] = []
242
- if not diff_text.strip():
243
- return notes
244
- for h in hunks:
245
- tgt_fn = _resolve_graph_filename(graph, h.target_path)
246
- if not tgt_fn or h.target_line_start <= 0:
247
- continue
248
- decls = _decl_added_lines_for_file(diff_text, tgt_fn)
249
- if decls <= 0:
250
- continue
251
- methods_here = [c for c in changed if c.kind == "method" and c.file == tgt_fn]
252
- if decls > len(methods_here):
253
- extra = decls - len(methods_here)
254
- notes.append(
255
- f"{extra} new method(s) not yet indexed; risk underestimated",
256
- )
257
- return notes
258
-
259
-
260
- def map_hunks_to_symbols(
261
- graph: Any,
262
- hunks: list[DiffHunk],
263
- *,
264
- path_ambiguity_notes: list[str] | None = None,
265
- ) -> list[ChangedSymbol]:
266
- """Map diff hunks to overlapping `Symbol` rows (graph-resident only)."""
267
- by_id: dict[str, ChangedSymbol] = {}
268
-
269
- def merge(sym: ChangedSymbol) -> None:
270
- existing = by_id.get(sym.symbol_id)
271
- if existing is None:
272
- by_id[sym.symbol_id] = sym
273
- else:
274
- if existing.change_type == "modified" or sym.change_type == "modified":
275
- ct = "modified"
276
- elif existing.change_type == "removed" or sym.change_type == "removed":
277
- ct = "removed"
278
- else:
279
- ct = sym.change_type
280
- merged_lines = sorted(set(existing.hunk_lines + sym.hunk_lines))
281
- by_id[sym.symbol_id] = ChangedSymbol(
282
- symbol_id=existing.symbol_id,
283
- fqn=existing.fqn,
284
- kind=existing.kind,
285
- change_type=ct,
286
- file=existing.file,
287
- hunk_lines=merged_lines,
288
- )
289
-
290
- for h in hunks:
291
- tgt_fn = _resolve_graph_filename(
292
- graph, h.target_path, ambiguity_notes=path_ambiguity_notes,
293
- )
294
- src_fn = (
295
- _resolve_graph_filename(
296
- graph, h.source_path, ambiguity_notes=path_ambiguity_notes,
297
- )
298
- if h.source_path
299
- else tgt_fn
300
- )
301
- if not tgt_fn and not src_fn:
302
- continue
303
-
304
- minus_only = h.target_length == 0 and h.source_length > 0
305
-
306
- # Removed lines on old file (process before modified so mixed hunks prefer modified)
307
- if h.source_line_start > 0 and h.source_line_end >= h.source_line_start and src_fn:
308
- rows = find_symbols_in_file_range(
309
- graph,
310
- filename=src_fn,
311
- start_line=h.source_line_start,
312
- end_line=h.source_line_end,
313
- )
314
- for sym in rows:
315
- if sym.kind == "file":
316
- continue
317
- overlap = list(range(
318
- max(h.source_line_start, sym.start_line),
319
- min(h.source_line_end, sym.end_line) + 1,
320
- ))
321
- if minus_only:
322
- merge(_symbol_to_changed(sym, change_type="removed", lines=overlap))
323
-
324
- # Modified / added lines on new file
325
- if h.target_line_start > 0 and h.target_line_end >= h.target_line_start and tgt_fn:
326
- rows = find_symbols_in_file_range(
327
- graph,
328
- filename=tgt_fn,
329
- start_line=h.target_line_start,
330
- end_line=h.target_line_end,
331
- )
332
- for sym in rows:
333
- if sym.kind == "file":
334
- continue
335
- merge(_symbol_to_changed(sym, change_type="modified", lines=list(range(
336
- max(h.target_line_start, sym.start_line),
337
- min(h.target_line_end, sym.end_line) + 1,
338
- ))))
339
-
340
- return list(by_id.values())
341
-
342
-
343
- def _impact_needle_for_changed(_graph: Any, fqn: str, mapped_kind: str) -> str:
344
- """Pick the `impact_analysis` needle: type FQN for members, else the symbol FQN."""
345
- if mapped_kind in ("method", "field", "constructor"):
346
- if "#" in fqn:
347
- return fqn.split("#", 1)[0]
348
- return fqn
349
-
350
-
351
- def _is_public_interface_method(graph: Any, sym: SymbolHit) -> bool:
352
- if sym.kind != "method":
353
- return False
354
- if "private" in (sym.modifiers or []):
355
- return False
356
- type_fqn = sym.fqn.split("#", 1)[0] if "#" in sym.fqn else sym.fqn
357
- rows = graph._rows(
358
- "MATCH (t:Symbol) WHERE t.fqn = $f AND t.kind = 'interface' RETURN t.id LIMIT 1",
359
- {"f": type_fqn},
360
- )
361
- return bool(rows)
362
-
363
-
364
- def _route_ids_for_symbol(graph: Any, symbol_id: str) -> list[str]:
365
- # Note: LadybugDB rejects `ORDER BY r.id` together with `RETURN DISTINCT r.id` (binder loses `r`).
366
- q = (
367
- "MATCH (s:Symbol)-[e:EXPOSES]->(r:Route) WHERE s.id = $sid "
368
- "RETURN r.id AS id ORDER BY id"
369
- )
370
- seen: set[str] = set()
371
- out: list[str] = []
372
- for row in graph._rows(q, {"sid": symbol_id}):
373
- rid = str(row.get("id") or "")
374
- if rid and rid not in seen:
375
- seen.add(rid)
376
- out.append(rid)
377
- return out
378
-
379
-
380
- def _route_natural_id(graph: Any, rid: str) -> str:
381
- """Map a raw Route node id to its agent-facing natural identifier.
382
-
383
- Mirrors the envelope contract (``METHOD path`` for HTTP endpoints,
384
- ``topic:<name>`` for kafka topics surfaced as :Route) so ``routes_touched``
385
- in the PR risk report is readable instead of leaking raw graph ids like
386
- ``r:970ffaa960a4f65d``. Falls back to ``rid`` only if the route vanished.
387
- """
388
- rows = graph._rows(
389
- "MATCH (r:Route {id: $rid}) "
390
- "RETURN r.method AS method, r.path_template AS path_template, "
391
- "r.path AS path, r.topic AS topic LIMIT 1",
392
- {"rid": rid},
393
- )
394
- if not rows:
395
- return rid
396
- r = rows[0]
397
- method = str(r.get("method") or "")
398
- path = str(r.get("path_template") or r.get("path") or "")
399
- if method or path:
400
- return f"{method} {path}".strip()
401
- topic = str(r.get("topic") or "")
402
- return f"topic:{topic}" if topic else rid
403
-
404
-
405
- def compute_risk(graph: Any, changed: list[ChangedSymbol]) -> PrRiskReport:
406
- """Aggregate blast radius, routes, cross-service callers, and v1 risk score.
407
-
408
- Risk score stays in [0, 1]. Cross-service route callers add a bounded
409
- bump (up to +1.0) after normalization so they influence rank while
410
- preserving the public scalar contract.
411
- """
412
- blast_by: dict[str, int] = {}
413
- blast_total = 0
414
- routes: list[str] = []
415
- cross_total = 0
416
-
417
- sym_cols = (
418
- "id", "kind", "name", "fqn", "package", "module", "microservice",
419
- "filename", "start_line", "end_line", "start_byte", "end_byte",
420
- "modifiers", "annotations", "capabilities", "role", "signature",
421
- "parent_id", "resolved",
422
- )
423
- _sym_return = ", ".join(f"s.{c} AS {c}" for c in sym_cols)
424
-
425
- iface_hit = 0.0
426
- enriched_changed: list[ChangedSymbol] = []
427
- for cs in changed:
428
- sym_row = graph._rows(
429
- "MATCH (s:Symbol) WHERE s.id = $id RETURN " + _sym_return,
430
- {"id": cs.symbol_id},
431
- )
432
- if not sym_row:
433
- continue
434
- row0 = sym_row[0]
435
- if iface_hit < 1.0:
436
- sym = _row_to_symbol(row0)
437
- if _is_public_interface_method(graph, sym):
438
- iface_hit = 1.0
439
- fqn = str(row0.get("fqn") or cs.fqn)
440
- needle = _impact_needle_for_changed(graph, fqn, cs.kind)
441
- ia = graph.impact_analysis(needle, depth=2, limit=400)
442
- n = len(ia)
443
- # Key blast radius by the symbol's FQN (agent-facing identifier), not
444
- # its raw graph id — the report is operator-facing JSON.
445
- blast_by[fqn or cs.symbol_id] = n
446
- blast_total += n
447
-
448
- for e in graph.find_callers(cs.fqn, depth=2, limit=400):
449
- if (
450
- e.src.microservice
451
- and e.dst.microservice
452
- and e.src.microservice != e.dst.microservice
453
- ):
454
- cross_total += 1
455
-
456
- cs_cross_service = 0
457
- route_ids = _route_ids_for_symbol(graph, cs.symbol_id)
458
- for rid in route_ids:
459
- # Record the route by its natural identifier (METHOD path /
460
- # topic:name), not the raw graph id.
461
- label = _route_natural_id(graph, rid)
462
- if label and label not in routes:
463
- routes.append(label)
464
- callers = graph._rows(
465
- "MATCH (s:Symbol)-[:DECLARES_CLIENT]->(c:Client)-[e:HTTP_CALLS]->(r:Route {id: $rid}) "
466
- "WHERE e.match = 'cross_service' "
467
- "RETURN c.id AS id LIMIT 500",
468
- {"rid": rid},
469
- )
470
- callers += graph._rows(
471
- "MATCH (s:Symbol)-[:DECLARES_PRODUCER]->(p:Producer)-[e:ASYNC_CALLS]->(r:Route {id: $rid}) "
472
- "WHERE e.match = 'cross_service' "
473
- "RETURN p.id AS id LIMIT 500",
474
- {"rid": rid},
475
- )
476
- cs_cross_service += len(callers)
477
- enriched_changed.append(
478
- ChangedSymbol(
479
- symbol_id=cs.symbol_id,
480
- fqn=cs.fqn,
481
- kind=cs.kind,
482
- change_type=cs.change_type,
483
- file=cs.file,
484
- hunk_lines=list(cs.hunk_lines),
485
- cross_service_callers_count=cs_cross_service,
486
- ),
487
- )
488
-
489
- def _normalize(x: float, ceiling: float) -> float:
490
- if ceiling <= 0:
491
- return 0.0
492
- return min(float(x), ceiling) / ceiling
493
-
494
- # v1 risk weights / ceilings (PR-B §1.2): intentionally simple baselines;
495
- # these constants are expected to be tuned after real-world use — do not treat as stable.
496
- w_blast, cap_blast = 0.4, 100.0
497
- w_cross, cap_cross = 0.3, 20.0
498
- w_iface = 0.2
499
- w_routes, cap_routes = 0.1, 5.0
500
-
501
- raw = (
502
- w_blast * _normalize(float(blast_total), cap_blast)
503
- + w_cross * _normalize(float(cross_total), cap_cross)
504
- + w_iface * iface_hit
505
- + w_routes * _normalize(float(len(routes)), cap_routes)
506
- )
507
- cross_service_bonus = min(
508
- 5.0,
509
- float(sum(c.cross_service_callers_count for c in enriched_changed)),
510
- )
511
- score = max(0.0, min(1.0, raw + (cross_service_bonus / 5.0)))
512
- if score < 0.3:
513
- band = "low"
514
- elif score < 0.7:
515
- band = "medium"
516
- else:
517
- band = "high"
518
-
519
- return PrRiskReport(
520
- changed_symbols=list(enriched_changed),
521
- blast_radius_total=blast_total,
522
- blast_radius_by_symbol=blast_by,
523
- cross_service_callers=cross_total,
524
- routes_touched=routes,
525
- risk_score=score,
526
- risk_band=band,
527
- notes=[],
528
- )
529
-
530
-
531
- def pr_report_to_dict(rep: PrRiskReport) -> dict[str, Any]:
532
- return {
533
- "changed_symbols": [asdict(c) for c in rep.changed_symbols],
534
- "blast_radius_total": rep.blast_radius_total,
535
- "blast_radius_by_symbol": dict(rep.blast_radius_by_symbol),
536
- "cross_service_callers": rep.cross_service_callers,
537
- "routes_touched": list(rep.routes_touched),
538
- "risk_score": rep.risk_score,
539
- "risk_band": rep.risk_band,
540
- "notes": list(rep.notes),
541
- }
542
-
543
-
544
- def analyze_pr_pipeline(graph: Any, diff_unified: str) -> PrRiskReport:
545
- """Full PR-B pipeline: parse → notes → map → risk."""
546
- notes = collect_diff_file_notes(diff_unified)
547
- hunks = parse_unified_diff(diff_unified)
548
- path_amb: list[str] = []
549
- changed = map_hunks_to_symbols(graph, hunks, path_ambiguity_notes=path_amb)
550
- notes.extend(path_amb)
551
- notes.extend(_notes_for_unindexed_additions(graph, diff_unified, changed, hunks))
552
- rep = compute_risk(graph, changed)
553
- merged = list(dict.fromkeys([*notes, *rep.notes]))
554
- return PrRiskReport(
555
- changed_symbols=rep.changed_symbols,
556
- blast_radius_total=rep.blast_radius_total,
557
- blast_radius_by_symbol=rep.blast_radius_by_symbol,
558
- cross_service_callers=rep.cross_service_callers,
559
- routes_touched=rep.routes_touched,
560
- risk_score=rep.risk_score,
561
- risk_band=rep.risk_band,
562
- notes=merged,
563
- )