java-codebase-rag 0.11.2__py3-none-any.whl → 0.12.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. java_codebase_rag-0.12.1.dist-info/METADATA +35 -0
  2. java_codebase_rag-0.12.1.dist-info/RECORD +4 -0
  3. {java_codebase_rag-0.11.2.dist-info → java_codebase_rag-0.12.1.dist-info}/WHEEL +1 -1
  4. java_codebase_rag/_fdlimit.py +0 -56
  5. java_codebase_rag/_stdio.py +0 -32
  6. java_codebase_rag/_version.py +0 -35
  7. java_codebase_rag/absence/__init__.py +0 -0
  8. java_codebase_rag/absence/absence_diagnosis.py +0 -700
  9. java_codebase_rag/absence/absence_types.py +0 -124
  10. java_codebase_rag/absence/absence_vocab.py +0 -460
  11. java_codebase_rag/analysis/__init__.py +0 -0
  12. java_codebase_rag/analysis/pr_analysis.py +0 -563
  13. java_codebase_rag/analysis/resolve_service.py +0 -740
  14. java_codebase_rag/ast/__init__.py +0 -0
  15. java_codebase_rag/ast/ast_java.py +0 -2825
  16. java_codebase_rag/ast/brownfield_events.py +0 -58
  17. java_codebase_rag/ast/chunk_heuristics.py +0 -62
  18. java_codebase_rag/cli.py +0 -1215
  19. java_codebase_rag/cli_format.py +0 -85
  20. java_codebase_rag/cli_progress.py +0 -94
  21. java_codebase_rag/config.py +0 -833
  22. java_codebase_rag/eval/__init__.py +0 -1
  23. java_codebase_rag/eval/ground_truth.py +0 -100
  24. java_codebase_rag/eval/metrics.py +0 -107
  25. java_codebase_rag/eval/runner.py +0 -556
  26. java_codebase_rag/graph/__init__.py +0 -0
  27. java_codebase_rag/graph/build_ast_graph.py +0 -4471
  28. java_codebase_rag/graph/graph_enrich.py +0 -1937
  29. java_codebase_rag/graph/graph_types.py +0 -224
  30. java_codebase_rag/graph/java_ontology.py +0 -465
  31. java_codebase_rag/graph/ladybug_queries.py +0 -2213
  32. java_codebase_rag/graph/path_filtering.py +0 -477
  33. java_codebase_rag/index/__init__.py +0 -0
  34. java_codebase_rag/index/java_index_flow_lancedb.py +0 -734
  35. java_codebase_rag/index/java_index_v1_common.py +0 -33
  36. java_codebase_rag/install_data/__init__.py +0 -0
  37. java_codebase_rag/install_data/agents/explorer-rag-cli.md +0 -108
  38. java_codebase_rag/install_data/agents/explorer-rag-enhanced.md +0 -152
  39. java_codebase_rag/install_data/skills/explore-codebase/SKILL.md +0 -165
  40. java_codebase_rag/install_data/skills/explore-codebase-cli/SKILL.md +0 -107
  41. java_codebase_rag/installer.py +0 -2188
  42. java_codebase_rag/jrag.py +0 -4531
  43. java_codebase_rag/jrag_envelope.py +0 -1107
  44. java_codebase_rag/jrag_hints.py +0 -204
  45. java_codebase_rag/jrag_render.py +0 -926
  46. java_codebase_rag/lance_optimize.py +0 -264
  47. java_codebase_rag/mcp/__init__.py +0 -0
  48. java_codebase_rag/mcp/mcp_hints.py +0 -932
  49. java_codebase_rag/mcp/mcp_v2.py +0 -1916
  50. java_codebase_rag/mcp/server.py +0 -884
  51. java_codebase_rag/pipeline.py +0 -531
  52. java_codebase_rag/progress.py +0 -570
  53. java_codebase_rag/read_payloads.py +0 -781
  54. java_codebase_rag/search/__init__.py +0 -0
  55. java_codebase_rag/search/index_common.py +0 -10
  56. java_codebase_rag/search/search_lancedb.py +0 -1296
  57. java_codebase_rag/search/search_lexical.py +0 -449
  58. java_codebase_rag/search/search_scoring.py +0 -523
  59. java_codebase_rag/watch/__init__.py +0 -0
  60. java_codebase_rag/watch/client.py +0 -230
  61. java_codebase_rag/watch/daemon.py +0 -396
  62. java_codebase_rag/watch/lock.py +0 -201
  63. java_codebase_rag/watch/paths.py +0 -76
  64. java_codebase_rag/watch/protocol.py +0 -122
  65. java_codebase_rag/watch/server.py +0 -273
  66. java_codebase_rag/watch/warm.py +0 -105
  67. java_codebase_rag/watch/watcher.py +0 -370
  68. java_codebase_rag-0.11.2.dist-info/METADATA +0 -331
  69. java_codebase_rag-0.11.2.dist-info/RECORD +0 -71
  70. java_codebase_rag-0.11.2.dist-info/entry_points.txt +0 -4
  71. java_codebase_rag-0.11.2.dist-info/licenses/LICENSE +0 -21
  72. java_codebase_rag-0.11.2.dist-info/top_level.txt +0 -1
  73. /java_codebase_rag/__init__.py → /java_codebase_rag-0.12.1.dist-info/top_level.txt +0 -0
@@ -1,563 +0,0 @@
1
- """Unified-diff → symbol mapping and PR-style risk scoring (B4 / PR-B).
2
-
3
- Uses the `unidiff` library for parsing. Graph-resident symbols only; newly
4
- added Java members are not modelled — see `notes` on the returned report.
5
- """
6
- from __future__ import annotations
7
-
8
- import re
9
- from dataclasses import asdict, dataclass
10
- from typing import Any
11
-
12
- from unidiff import PatchSet
13
- from unidiff.errors import UnidiffParseError
14
-
15
- from java_codebase_rag.graph.ladybug_queries import SymbolHit, find_symbols_in_file_range, _row_to_symbol
16
-
17
-
18
- @dataclass
19
- class DiffHunk:
20
- """One unified-diff hunk in the *new* file coordinate system."""
21
-
22
- target_path: str
23
- source_path: str
24
- target_line_start: int # inclusive, 1-based; 0 when the hunk has no new-file lines
25
- target_line_end: int # inclusive
26
- source_line_start: int
27
- source_line_end: int
28
- source_length: int = 0
29
- target_length: int = 0
30
-
31
-
32
- @dataclass
33
- class ChangedSymbol:
34
- symbol_id: str
35
- fqn: str
36
- kind: str # 'method' | 'type' | 'field'
37
- change_type: str # 'added' | 'removed' | 'modified'
38
- file: str
39
- hunk_lines: list[int]
40
- cross_service_callers_count: int = 0
41
-
42
-
43
- @dataclass
44
- class PrRiskReport:
45
- changed_symbols: list[ChangedSymbol]
46
- blast_radius_total: int
47
- blast_radius_by_symbol: dict[str, int]
48
- cross_service_callers: int
49
- routes_touched: list[str]
50
- risk_score: float
51
- risk_band: str
52
- notes: list[str]
53
-
54
-
55
- _BINARY_DIFF_LINE = re.compile(r"^Binary files .+ differ\s*$")
56
- # Heuristic: new Java method/ctor-looking line. Covers annotations, method-level
57
- # generics, `default` interface methods, and return types with spaces (e.g.
58
- # `Map<String, String> m(`). Misses multi-line signatures, some compact record
59
- # forms, and unusual annotations; `_notes_for_unindexed_additions` is best-effort.
60
- _DECL_ADD = re.compile(
61
- r"^\+\s*"
62
- r"(?:(?:@[\w.]+\([^)]*\))\s+)*"
63
- r"(?:<[^>]+>\s+)?"
64
- r"(?:(?:public|private|protected|default|static|final|synchronized|abstract|native)\s+)*"
65
- r"(.+?)\s+(\w+)\s*\(",
66
- )
67
-
68
-
69
- def _strip_ab_prefix(path: str) -> str:
70
- p = path.strip()
71
- if p.startswith(("a/", "b/")):
72
- return p[2:]
73
- return p
74
-
75
-
76
- def _hunk_ranges(h: Any) -> tuple[tuple[int, int], tuple[int, int]]:
77
- """Return ((src_start, src_end inclusive), (tgt_start, tgt_end inclusive))."""
78
- src_len = int(getattr(h, "source_length", 0) or 0)
79
- tgt_len = int(getattr(h, "target_length", 0) or 0)
80
- src_start = int(getattr(h, "source_start", 0) or 0)
81
- tgt_start = int(getattr(h, "target_start", 0) or 0)
82
- if src_len <= 0:
83
- src_start, src_end = 0, 0
84
- else:
85
- src_end = src_start + src_len - 1
86
- if tgt_len <= 0:
87
- tgt_start, tgt_end = 0, 0
88
- else:
89
- tgt_end = tgt_start + tgt_len - 1
90
- return (src_start, src_end), (tgt_start, tgt_end)
91
-
92
-
93
- def parse_unified_diff(diff_text: str) -> list[DiffHunk]:
94
- """Parse `diff_text` into logical hunks (non-binary, non-rename files only)."""
95
- if not (diff_text or "").strip():
96
- return []
97
- try:
98
- patches = PatchSet(diff_text.splitlines(keepends=True))
99
- except UnidiffParseError:
100
- return []
101
- out: list[DiffHunk] = []
102
- for pf in patches:
103
- if getattr(pf, "is_rename", False):
104
- continue
105
- tgt = _strip_ab_prefix(str(pf.path or ""))
106
- src = _strip_ab_prefix(str(getattr(pf, "source_file", "") or pf.path or ""))
107
- if not tgt:
108
- continue
109
- for h in pf:
110
- (s0, s1), (t0, t1) = _hunk_ranges(h)
111
- sl = int(getattr(h, "source_length", 0) or 0)
112
- tl = int(getattr(h, "target_length", 0) or 0)
113
- out.append(
114
- DiffHunk(
115
- target_path=tgt,
116
- source_path=src,
117
- target_line_start=t0,
118
- target_line_end=t1,
119
- source_line_start=s0,
120
- source_line_end=s1,
121
- source_length=sl,
122
- target_length=tl,
123
- )
124
- )
125
- return out
126
-
127
-
128
- def collect_diff_file_notes(diff_text: str) -> list[str]:
129
- """Collect human-readable notes for binary diffs and renames (no crash)."""
130
- notes: list[str] = []
131
- if not (diff_text or "").strip():
132
- return notes
133
- for line in diff_text.splitlines():
134
- if _BINARY_DIFF_LINE.match(line):
135
- notes.append(f"skipped binary diff: {line.strip()}")
136
- try:
137
- patches = PatchSet(diff_text.splitlines(keepends=True))
138
- except UnidiffParseError:
139
- notes.append("diff text could not be fully parsed as a unified patch")
140
- return notes
141
- for pf in patches:
142
- if getattr(pf, "is_rename", False):
143
- a = _strip_ab_prefix(str(getattr(pf, "source_file", "") or ""))
144
- b = _strip_ab_prefix(str(pf.path or ""))
145
- notes.append(f"rename (symbols not mapped): {a} -> {b}")
146
- return notes
147
-
148
-
149
- def _resolve_graph_filename(
150
- graph: Any,
151
- path: str,
152
- *,
153
- ambiguity_notes: list[str] | None = None,
154
- ) -> str | None:
155
- """Map a diff path to `Symbol.filename` values stored in LadybugDB."""
156
- variants = {_strip_ab_prefix(path)}
157
- for v in list(variants):
158
- if v.startswith("./"):
159
- variants.add(v[2:])
160
- for candidate in variants:
161
- if not candidate:
162
- continue
163
- rows = graph._rows(
164
- "MATCH (s:Symbol) WHERE s.filename = $fn RETURN s.filename AS fn LIMIT 1",
165
- {"fn": candidate},
166
- )
167
- if rows and rows[0].get("fn"):
168
- return str(rows[0]["fn"])
169
- tail = path.strip().split("/")[-1]
170
- if tail:
171
- rows = graph._rows(
172
- "MATCH (s:Symbol) WHERE s.filename ENDS WITH $tail "
173
- "RETURN DISTINCT s.filename AS fn LIMIT 8",
174
- {"tail": "/" + tail},
175
- )
176
- n = len(rows)
177
- if n > 1 and ambiguity_notes is not None:
178
- fns = [str(r.get("fn") or "") for r in rows if r.get("fn")]
179
- ambiguity_notes.append(
180
- f"ambiguous filename tail {tail!r} ({n} graph paths); "
181
- f"ENDS WITH resolution skipped ({', '.join(fns[:4])}"
182
- f"{'…' if len(fns) > 4 else ''})",
183
- )
184
- if n == 1 and rows[0].get("fn"):
185
- return str(rows[0]["fn"])
186
- return None
187
-
188
-
189
- def _symbol_to_changed(
190
- sym: SymbolHit,
191
- *,
192
- change_type: str,
193
- lines: list[int],
194
- ) -> ChangedSymbol:
195
- kind = sym.kind
196
- if kind in ("class", "interface", "enum", "record", "annotation"):
197
- mapped_kind = "type"
198
- elif kind == "field":
199
- mapped_kind = "field"
200
- elif kind == "constructor":
201
- mapped_kind = "method"
202
- else:
203
- mapped_kind = "method"
204
- uniq = sorted({int(x) for x in lines if int(x) > 0})
205
- return ChangedSymbol(
206
- symbol_id=sym.id,
207
- fqn=sym.fqn,
208
- kind=mapped_kind,
209
- change_type=change_type,
210
- file=sym.filename,
211
- hunk_lines=uniq,
212
- )
213
-
214
-
215
- def _decl_added_lines_for_file(diff_text: str, resolved_filename: str) -> int:
216
- """Count `+` lines in the diff that look like Java member declarations for one file."""
217
- lines = diff_text.splitlines()
218
- in_file = False
219
- n = 0
220
- for line in lines:
221
- if line.startswith("+++ "):
222
- rest = line[4:].strip()
223
- if rest.startswith("b/"):
224
- rest = rest[2:]
225
- in_file = rest.endswith(resolved_filename) or resolved_filename.endswith(rest)
226
- continue
227
- if not in_file:
228
- continue
229
- if _DECL_ADD.match(line):
230
- n += 1
231
- return n
232
-
233
-
234
- def _notes_for_unindexed_additions(
235
- graph: Any,
236
- diff_text: str,
237
- changed: list[ChangedSymbol],
238
- hunks: list[DiffHunk],
239
- ) -> list[str]:
240
- """Heuristic: added declaration lines vs indexed methods touched on the same file."""
241
- notes: list[str] = []
242
- if not diff_text.strip():
243
- return notes
244
- for h in hunks:
245
- tgt_fn = _resolve_graph_filename(graph, h.target_path)
246
- if not tgt_fn or h.target_line_start <= 0:
247
- continue
248
- decls = _decl_added_lines_for_file(diff_text, tgt_fn)
249
- if decls <= 0:
250
- continue
251
- methods_here = [c for c in changed if c.kind == "method" and c.file == tgt_fn]
252
- if decls > len(methods_here):
253
- extra = decls - len(methods_here)
254
- notes.append(
255
- f"{extra} new method(s) not yet indexed; risk underestimated",
256
- )
257
- return notes
258
-
259
-
260
- def map_hunks_to_symbols(
261
- graph: Any,
262
- hunks: list[DiffHunk],
263
- *,
264
- path_ambiguity_notes: list[str] | None = None,
265
- ) -> list[ChangedSymbol]:
266
- """Map diff hunks to overlapping `Symbol` rows (graph-resident only)."""
267
- by_id: dict[str, ChangedSymbol] = {}
268
-
269
- def merge(sym: ChangedSymbol) -> None:
270
- existing = by_id.get(sym.symbol_id)
271
- if existing is None:
272
- by_id[sym.symbol_id] = sym
273
- else:
274
- if existing.change_type == "modified" or sym.change_type == "modified":
275
- ct = "modified"
276
- elif existing.change_type == "removed" or sym.change_type == "removed":
277
- ct = "removed"
278
- else:
279
- ct = sym.change_type
280
- merged_lines = sorted(set(existing.hunk_lines + sym.hunk_lines))
281
- by_id[sym.symbol_id] = ChangedSymbol(
282
- symbol_id=existing.symbol_id,
283
- fqn=existing.fqn,
284
- kind=existing.kind,
285
- change_type=ct,
286
- file=existing.file,
287
- hunk_lines=merged_lines,
288
- )
289
-
290
- for h in hunks:
291
- tgt_fn = _resolve_graph_filename(
292
- graph, h.target_path, ambiguity_notes=path_ambiguity_notes,
293
- )
294
- src_fn = (
295
- _resolve_graph_filename(
296
- graph, h.source_path, ambiguity_notes=path_ambiguity_notes,
297
- )
298
- if h.source_path
299
- else tgt_fn
300
- )
301
- if not tgt_fn and not src_fn:
302
- continue
303
-
304
- minus_only = h.target_length == 0 and h.source_length > 0
305
-
306
- # Removed lines on old file (process before modified so mixed hunks prefer modified)
307
- if h.source_line_start > 0 and h.source_line_end >= h.source_line_start and src_fn:
308
- rows = find_symbols_in_file_range(
309
- graph,
310
- filename=src_fn,
311
- start_line=h.source_line_start,
312
- end_line=h.source_line_end,
313
- )
314
- for sym in rows:
315
- if sym.kind == "file":
316
- continue
317
- overlap = list(range(
318
- max(h.source_line_start, sym.start_line),
319
- min(h.source_line_end, sym.end_line) + 1,
320
- ))
321
- if minus_only:
322
- merge(_symbol_to_changed(sym, change_type="removed", lines=overlap))
323
-
324
- # Modified / added lines on new file
325
- if h.target_line_start > 0 and h.target_line_end >= h.target_line_start and tgt_fn:
326
- rows = find_symbols_in_file_range(
327
- graph,
328
- filename=tgt_fn,
329
- start_line=h.target_line_start,
330
- end_line=h.target_line_end,
331
- )
332
- for sym in rows:
333
- if sym.kind == "file":
334
- continue
335
- merge(_symbol_to_changed(sym, change_type="modified", lines=list(range(
336
- max(h.target_line_start, sym.start_line),
337
- min(h.target_line_end, sym.end_line) + 1,
338
- ))))
339
-
340
- return list(by_id.values())
341
-
342
-
343
- def _impact_needle_for_changed(_graph: Any, fqn: str, mapped_kind: str) -> str:
344
- """Pick the `impact_analysis` needle: type FQN for members, else the symbol FQN."""
345
- if mapped_kind in ("method", "field", "constructor"):
346
- if "#" in fqn:
347
- return fqn.split("#", 1)[0]
348
- return fqn
349
-
350
-
351
- def _is_public_interface_method(graph: Any, sym: SymbolHit) -> bool:
352
- if sym.kind != "method":
353
- return False
354
- if "private" in (sym.modifiers or []):
355
- return False
356
- type_fqn = sym.fqn.split("#", 1)[0] if "#" in sym.fqn else sym.fqn
357
- rows = graph._rows(
358
- "MATCH (t:Symbol) WHERE t.fqn = $f AND t.kind = 'interface' RETURN t.id LIMIT 1",
359
- {"f": type_fqn},
360
- )
361
- return bool(rows)
362
-
363
-
364
- def _route_ids_for_symbol(graph: Any, symbol_id: str) -> list[str]:
365
- # Note: LadybugDB rejects `ORDER BY r.id` together with `RETURN DISTINCT r.id` (binder loses `r`).
366
- q = (
367
- "MATCH (s:Symbol)-[e:EXPOSES]->(r:Route) WHERE s.id = $sid "
368
- "RETURN r.id AS id ORDER BY id"
369
- )
370
- seen: set[str] = set()
371
- out: list[str] = []
372
- for row in graph._rows(q, {"sid": symbol_id}):
373
- rid = str(row.get("id") or "")
374
- if rid and rid not in seen:
375
- seen.add(rid)
376
- out.append(rid)
377
- return out
378
-
379
-
380
- def _route_natural_id(graph: Any, rid: str) -> str:
381
- """Map a raw Route node id to its agent-facing natural identifier.
382
-
383
- Mirrors the envelope contract (``METHOD path`` for HTTP endpoints,
384
- ``topic:<name>`` for kafka topics surfaced as :Route) so ``routes_touched``
385
- in the PR risk report is readable instead of leaking raw graph ids like
386
- ``r:970ffaa960a4f65d``. Falls back to ``rid`` only if the route vanished.
387
- """
388
- rows = graph._rows(
389
- "MATCH (r:Route {id: $rid}) "
390
- "RETURN r.method AS method, r.path_template AS path_template, "
391
- "r.path AS path, r.topic AS topic LIMIT 1",
392
- {"rid": rid},
393
- )
394
- if not rows:
395
- return rid
396
- r = rows[0]
397
- method = str(r.get("method") or "")
398
- path = str(r.get("path_template") or r.get("path") or "")
399
- if method or path:
400
- return f"{method} {path}".strip()
401
- topic = str(r.get("topic") or "")
402
- return f"topic:{topic}" if topic else rid
403
-
404
-
405
- def compute_risk(graph: Any, changed: list[ChangedSymbol]) -> PrRiskReport:
406
- """Aggregate blast radius, routes, cross-service callers, and v1 risk score.
407
-
408
- Risk score stays in [0, 1]. Cross-service route callers add a bounded
409
- bump (up to +1.0) after normalization so they influence rank while
410
- preserving the public scalar contract.
411
- """
412
- blast_by: dict[str, int] = {}
413
- blast_total = 0
414
- routes: list[str] = []
415
- cross_total = 0
416
-
417
- sym_cols = (
418
- "id", "kind", "name", "fqn", "package", "module", "microservice",
419
- "filename", "start_line", "end_line", "start_byte", "end_byte",
420
- "modifiers", "annotations", "capabilities", "role", "signature",
421
- "parent_id", "resolved",
422
- )
423
- _sym_return = ", ".join(f"s.{c} AS {c}" for c in sym_cols)
424
-
425
- iface_hit = 0.0
426
- enriched_changed: list[ChangedSymbol] = []
427
- for cs in changed:
428
- sym_row = graph._rows(
429
- "MATCH (s:Symbol) WHERE s.id = $id RETURN " + _sym_return,
430
- {"id": cs.symbol_id},
431
- )
432
- if not sym_row:
433
- continue
434
- row0 = sym_row[0]
435
- if iface_hit < 1.0:
436
- sym = _row_to_symbol(row0)
437
- if _is_public_interface_method(graph, sym):
438
- iface_hit = 1.0
439
- fqn = str(row0.get("fqn") or cs.fqn)
440
- needle = _impact_needle_for_changed(graph, fqn, cs.kind)
441
- ia = graph.impact_analysis(needle, depth=2, limit=400)
442
- n = len(ia)
443
- # Key blast radius by the symbol's FQN (agent-facing identifier), not
444
- # its raw graph id — the report is operator-facing JSON.
445
- blast_by[fqn or cs.symbol_id] = n
446
- blast_total += n
447
-
448
- for e in graph.find_callers(cs.fqn, depth=2, limit=400):
449
- if (
450
- e.src.microservice
451
- and e.dst.microservice
452
- and e.src.microservice != e.dst.microservice
453
- ):
454
- cross_total += 1
455
-
456
- cs_cross_service = 0
457
- route_ids = _route_ids_for_symbol(graph, cs.symbol_id)
458
- for rid in route_ids:
459
- # Record the route by its natural identifier (METHOD path /
460
- # topic:name), not the raw graph id.
461
- label = _route_natural_id(graph, rid)
462
- if label and label not in routes:
463
- routes.append(label)
464
- callers = graph._rows(
465
- "MATCH (s:Symbol)-[:DECLARES_CLIENT]->(c:Client)-[e:HTTP_CALLS]->(r:Route {id: $rid}) "
466
- "WHERE e.match = 'cross_service' "
467
- "RETURN c.id AS id LIMIT 500",
468
- {"rid": rid},
469
- )
470
- callers += graph._rows(
471
- "MATCH (s:Symbol)-[:DECLARES_PRODUCER]->(p:Producer)-[e:ASYNC_CALLS]->(r:Route {id: $rid}) "
472
- "WHERE e.match = 'cross_service' "
473
- "RETURN p.id AS id LIMIT 500",
474
- {"rid": rid},
475
- )
476
- cs_cross_service += len(callers)
477
- enriched_changed.append(
478
- ChangedSymbol(
479
- symbol_id=cs.symbol_id,
480
- fqn=cs.fqn,
481
- kind=cs.kind,
482
- change_type=cs.change_type,
483
- file=cs.file,
484
- hunk_lines=list(cs.hunk_lines),
485
- cross_service_callers_count=cs_cross_service,
486
- ),
487
- )
488
-
489
- def _normalize(x: float, ceiling: float) -> float:
490
- if ceiling <= 0:
491
- return 0.0
492
- return min(float(x), ceiling) / ceiling
493
-
494
- # v1 risk weights / ceilings (PR-B §1.2): intentionally simple baselines;
495
- # these constants are expected to be tuned after real-world use — do not treat as stable.
496
- w_blast, cap_blast = 0.4, 100.0
497
- w_cross, cap_cross = 0.3, 20.0
498
- w_iface = 0.2
499
- w_routes, cap_routes = 0.1, 5.0
500
-
501
- raw = (
502
- w_blast * _normalize(float(blast_total), cap_blast)
503
- + w_cross * _normalize(float(cross_total), cap_cross)
504
- + w_iface * iface_hit
505
- + w_routes * _normalize(float(len(routes)), cap_routes)
506
- )
507
- cross_service_bonus = min(
508
- 5.0,
509
- float(sum(c.cross_service_callers_count for c in enriched_changed)),
510
- )
511
- score = max(0.0, min(1.0, raw + (cross_service_bonus / 5.0)))
512
- if score < 0.3:
513
- band = "low"
514
- elif score < 0.7:
515
- band = "medium"
516
- else:
517
- band = "high"
518
-
519
- return PrRiskReport(
520
- changed_symbols=list(enriched_changed),
521
- blast_radius_total=blast_total,
522
- blast_radius_by_symbol=blast_by,
523
- cross_service_callers=cross_total,
524
- routes_touched=routes,
525
- risk_score=score,
526
- risk_band=band,
527
- notes=[],
528
- )
529
-
530
-
531
- def pr_report_to_dict(rep: PrRiskReport) -> dict[str, Any]:
532
- return {
533
- "changed_symbols": [asdict(c) for c in rep.changed_symbols],
534
- "blast_radius_total": rep.blast_radius_total,
535
- "blast_radius_by_symbol": dict(rep.blast_radius_by_symbol),
536
- "cross_service_callers": rep.cross_service_callers,
537
- "routes_touched": list(rep.routes_touched),
538
- "risk_score": rep.risk_score,
539
- "risk_band": rep.risk_band,
540
- "notes": list(rep.notes),
541
- }
542
-
543
-
544
- def analyze_pr_pipeline(graph: Any, diff_unified: str) -> PrRiskReport:
545
- """Full PR-B pipeline: parse → notes → map → risk."""
546
- notes = collect_diff_file_notes(diff_unified)
547
- hunks = parse_unified_diff(diff_unified)
548
- path_amb: list[str] = []
549
- changed = map_hunks_to_symbols(graph, hunks, path_ambiguity_notes=path_amb)
550
- notes.extend(path_amb)
551
- notes.extend(_notes_for_unindexed_additions(graph, diff_unified, changed, hunks))
552
- rep = compute_risk(graph, changed)
553
- merged = list(dict.fromkeys([*notes, *rep.notes]))
554
- return PrRiskReport(
555
- changed_symbols=rep.changed_symbols,
556
- blast_radius_total=rep.blast_radius_total,
557
- blast_radius_by_symbol=rep.blast_radius_by_symbol,
558
- cross_service_callers=rep.cross_service_callers,
559
- routes_touched=rep.routes_touched,
560
- risk_score=rep.risk_score,
561
- risk_band=rep.risk_band,
562
- notes=merged,
563
- )