splitagent 0.0.3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. splitagent/__init__.py +8 -0
  2. splitagent/__main__.py +6 -0
  3. splitagent/agents/__init__.py +10 -0
  4. splitagent/agents/base.py +477 -0
  5. splitagent/agents/blue.py +57 -0
  6. splitagent/agents/chat.py +60 -0
  7. splitagent/agents/prompts.py +462 -0
  8. splitagent/agents/red.py +75 -0
  9. splitagent/cli.py +701 -0
  10. splitagent/config.py +697 -0
  11. splitagent/core/__init__.py +19 -0
  12. splitagent/core/bus.py +62 -0
  13. splitagent/core/context.py +587 -0
  14. splitagent/core/context_manager.py +381 -0
  15. splitagent/core/engine.py +424 -0
  16. splitagent/core/models.py +310 -0
  17. splitagent/core/proc.py +73 -0
  18. splitagent/core/sandbox.py +184 -0
  19. splitagent/core/toolbox.py +520 -0
  20. splitagent/core/workspace.py +420 -0
  21. splitagent/desktop/__init__.py +7 -0
  22. splitagent/desktop/api.py +525 -0
  23. splitagent/desktop/app.py +1131 -0
  24. splitagent/desktop/web/app.js +3067 -0
  25. splitagent/desktop/web/assets/Inter.ttf +0 -0
  26. splitagent/desktop/web/assets/JetBrainsMonoNerdFontMono-Regular.woff2 +0 -0
  27. splitagent/desktop/web/index.html +760 -0
  28. splitagent/desktop/web/styles.css +1612 -0
  29. splitagent/errors.py +27 -0
  30. splitagent/llm/__init__.py +8 -0
  31. splitagent/llm/client.py +488 -0
  32. splitagent/llm/types.py +172 -0
  33. splitagent/report/__init__.py +9 -0
  34. splitagent/report/cvss.py +93 -0
  35. splitagent/report/generator.py +733 -0
  36. splitagent/tools/__init__.py +8 -0
  37. splitagent/tools/base.py +135 -0
  38. splitagent/tools/defense.py +475 -0
  39. splitagent/tools/exploit.py +318 -0
  40. splitagent/tools/http_pool.py +109 -0
  41. splitagent/tools/knowledge.py +376 -0
  42. splitagent/tools/recon.py +182 -0
  43. splitagent/tools/registry.py +62 -0
  44. splitagent/tools/validate.py +908 -0
  45. splitagent/tools/web.py +386 -0
  46. splitagent/tools/workspace_tools.py +411 -0
  47. splitagent/ui/__init__.py +5 -0
  48. splitagent/ui/app.py +389 -0
  49. splitagent/ui/stream.py +234 -0
  50. splitagent/ui/theme.py +72 -0
  51. splitagent-0.0.3.dist-info/METADATA +987 -0
  52. splitagent-0.0.3.dist-info/RECORD +56 -0
  53. splitagent-0.0.3.dist-info/WHEEL +5 -0
  54. splitagent-0.0.3.dist-info/entry_points.txt +2 -0
  55. splitagent-0.0.3.dist-info/licenses/LICENSE +21 -0
  56. splitagent-0.0.3.dist-info/top_level.txt +1 -0
@@ -0,0 +1,19 @@
1
+ """Core orchestration primitives."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from splitagent.core.bus import Event, EventBus
6
+ from splitagent.core.context import SharedContext
7
+ from splitagent.core.engine import Engine
8
+ from splitagent.core.models import Finding, Mitigation, Round, SessionState
9
+
10
+ __all__ = [
11
+ "Engine",
12
+ "Event",
13
+ "EventBus",
14
+ "Finding",
15
+ "Mitigation",
16
+ "Round",
17
+ "SessionState",
18
+ "SharedContext",
19
+ ]
splitagent/core/bus.py ADDED
@@ -0,0 +1,62 @@
1
+ """A tiny async event bus used to stream agent activity to the UI."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ import inspect
7
+ from collections.abc import Awaitable, Callable
8
+ from dataclasses import dataclass, field
9
+ from datetime import datetime, timezone
10
+ from typing import Any
11
+
12
+ Listener = Callable[["Event"], Any] | Callable[["Event"], Awaitable[Any]]
13
+
14
+
15
+ def _now() -> str:
16
+ return datetime.now(timezone.utc).isoformat(timespec="milliseconds")
17
+
18
+
19
+ @dataclass
20
+ class Event:
21
+ """A single unit of agent activity."""
22
+
23
+ type: str
24
+ # types: session.start | round.start | phase.start | agent.text |
25
+ # agent.tool_call | agent.tool_result | finding | mitigation |
26
+ # phase.end | round.end | session.end | log | error
27
+ agent: str = "core"
28
+ data: dict[str, Any] = field(default_factory=dict)
29
+ ts: str = field(default_factory=_now)
30
+
31
+ def summary(self) -> str:
32
+ return self.data.get("text") or self.data.get("title") or self.type
33
+
34
+
35
+ class EventBus:
36
+ """Dispatches events to any number of sync or async listeners."""
37
+
38
+ def __init__(self) -> None:
39
+ self._listeners: list[Listener] = []
40
+ self._lock = asyncio.Lock()
41
+
42
+ def subscribe(self, listener: Listener) -> Callable[[], None]:
43
+ self._listeners.append(listener)
44
+
45
+ def unsubscribe() -> None:
46
+ if listener in self._listeners:
47
+ self._listeners.remove(listener)
48
+
49
+ return unsubscribe
50
+
51
+ async def emit(self, event: Event) -> None:
52
+ for listener in list(self._listeners):
53
+ try:
54
+ result = listener(event)
55
+ if inspect.isawaitable(result):
56
+ await result
57
+ except Exception: # pragma: no cover - a broken UI must not stop the run
58
+ continue
59
+
60
+ async def emit_many(self, events: list[Event]) -> None:
61
+ for event in events:
62
+ await self.emit(event)
@@ -0,0 +1,587 @@
1
+ """Encrypted shared context.
2
+
3
+ The context is the single source of truth the Red and Blue agents read from
4
+ and write to. It is serialised to JSON and encrypted at rest with a Fernet
5
+ key stored in the global config directory, so audited evidence and secrets
6
+ never touch disk in clear text.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ import re
13
+ from pathlib import Path
14
+ from typing import Any
15
+
16
+ from splitagent.config import ensure_home, global_key_path, sessions_dir
17
+ from splitagent.core.bus import Event, EventBus
18
+ from splitagent.core.context_manager import ContextPolicy
19
+ from splitagent.core.models import (
20
+ Finding,
21
+ Mitigation,
22
+ Round,
23
+ SessionState,
24
+ normalize_severity,
25
+ )
26
+ from splitagent.errors import ConfigError
27
+
28
+
29
+ def _load_fernet() -> Any:
30
+ try:
31
+ from cryptography.fernet import Fernet
32
+ except ImportError: # pragma: no cover - dependency is declared
33
+ return None
34
+ return Fernet
35
+
36
+
37
+ def load_or_create_key(path: Path | None = None) -> bytes | None:
38
+ """Return the Fernet key, creating it on first use. ``None`` if unavailable."""
39
+ Fernet = _load_fernet()
40
+ if Fernet is None:
41
+ return None
42
+ path = path or global_key_path()
43
+ if path.exists():
44
+ return path.read_bytes().strip()
45
+ ensure_home()
46
+ key = Fernet.generate_key()
47
+ path.write_bytes(key)
48
+ try:
49
+ import os
50
+
51
+ os.chmod(path, 0o600)
52
+ except OSError: # pragma: no cover
53
+ pass
54
+ return key
55
+
56
+
57
+ class SharedContext:
58
+ """Thread-safe-ish wrapper around :class:`SessionState` with persistence."""
59
+
60
+ def __init__(
61
+ self,
62
+ state: SessionState,
63
+ bus: EventBus | None = None,
64
+ directory: Path | None = None,
65
+ ) -> None:
66
+ self.state = state
67
+ self.bus = bus or EventBus()
68
+ self.directory = directory or sessions_dir()
69
+ self._fernet = _load_fernet()
70
+ self._key: bytes | None = None
71
+ # Context window policy shared by every agent in this session.
72
+ self.policy = ContextPolicy()
73
+ self.checkpoints: list[dict[str, Any]] = []
74
+ # The agents' own working directory (set by the engine).
75
+ self.workspace: Any = None
76
+
77
+ # -- constructors ------------------------------------------------------ #
78
+ @classmethod
79
+ def create(
80
+ cls,
81
+ target: str,
82
+ target_kind: str = "web",
83
+ scope: list[str] | None = None,
84
+ name: str = "splitagent",
85
+ model: str = "",
86
+ provider: str = "",
87
+ bus: EventBus | None = None,
88
+ directory: Path | None = None,
89
+ ) -> SharedContext:
90
+ state = SessionState(
91
+ name=name,
92
+ target=target,
93
+ target_kind=target_kind,
94
+ scope=list(scope or []),
95
+ model=model,
96
+ provider=provider,
97
+ )
98
+ return cls(state, bus=bus, directory=directory)
99
+
100
+ # -- crypto ------------------------------------------------------------ #
101
+ def _ensure_key(self) -> bytes | None:
102
+ if self._fernet is None:
103
+ return None
104
+ if self._key is None:
105
+ self._key = load_or_create_key()
106
+ return self._key
107
+
108
+ def encrypt(self, payload: bytes) -> bytes:
109
+ key = self._ensure_key()
110
+ if key is None or self._fernet is None:
111
+ raise ConfigError(
112
+ "cryptography is required to persist encrypted sessions. "
113
+ "Install it with `pip install cryptography`."
114
+ )
115
+ return self._fernet(key).encrypt(payload)
116
+
117
+ def decrypt(self, token: bytes) -> bytes:
118
+ key = self._ensure_key()
119
+ if key is None or self._fernet is None:
120
+ raise ConfigError("cryptography is unavailable; cannot decrypt session.")
121
+ return self._fernet(key).decrypt(token)
122
+
123
+ # -- persistence ------------------------------------------------------- #
124
+ def path_for(self, session_id: str | None = None) -> Path:
125
+ return self.directory / f"{session_id or self.state.id}.session.enc"
126
+
127
+ def save(self) -> Path:
128
+ self.directory.mkdir(parents=True, exist_ok=True)
129
+ payload = json.dumps(self.state.to_dict(), ensure_ascii=False).encode("utf-8")
130
+ token = self.encrypt(payload)
131
+ path = self.path_for()
132
+ path.write_bytes(token)
133
+ return path
134
+
135
+ @classmethod
136
+ def load(
137
+ cls, session_id: str, bus: EventBus | None = None, directory: Path | None = None
138
+ ) -> SharedContext:
139
+ directory = directory or sessions_dir()
140
+ if not re.fullmatch(r"[A-Za-z0-9_-]+", session_id or ""):
141
+ raise ConfigError(f"Invalid session id: {session_id!r}")
142
+ path = directory / f"{session_id}.session.enc"
143
+ if not path.exists():
144
+ raise ConfigError(f"Session '{session_id}' not found at {path}")
145
+ ctx = cls(SessionState(), bus=bus, directory=directory)
146
+ data = json.loads(ctx.decrypt(path.read_bytes()).decode("utf-8"))
147
+ ctx.state = SessionState.from_dict(data)
148
+ return ctx
149
+
150
+ @classmethod
151
+ def load_path(cls, path: Path, bus: EventBus | None = None) -> SharedContext:
152
+ ctx = cls(SessionState(), bus=bus, directory=path.parent)
153
+ data = json.loads(ctx.decrypt(path.read_bytes()).decode("utf-8"))
154
+ ctx.state = SessionState.from_dict(data)
155
+ return ctx
156
+
157
+ def export_json(self, path: Path) -> Path:
158
+ path.parent.mkdir(parents=True, exist_ok=True)
159
+ path.write_text(
160
+ json.dumps(self.state.to_dict(), indent=2, ensure_ascii=False),
161
+ encoding="utf-8",
162
+ )
163
+ return path
164
+
165
+ # -- mutators ---------------------------------------------------------- #
166
+ def find_duplicate(self, title: str, endpoint: str, category: str = "") -> Finding | None:
167
+ """Return an existing finding that describes the same issue.
168
+
169
+ The Red Agent re-tests the same surface every round, so without this
170
+ the report fills with near-identical entries. Observed in a real run:
171
+ phpinfo disclosure recorded three times, XSS twice, TRACE twice, all
172
+ with different wording but the same endpoint and category.
173
+
174
+ Two findings clash when they share an endpoint **and** either the same
175
+ category or a strongly overlapping title. Different issues on the same
176
+ URL (an XSS and a path traversal) keep their separate entries.
177
+ """
178
+ norm_ep = _normalise_endpoint(endpoint)
179
+ key = _finding_key(title, endpoint)
180
+ cat = (category or "").strip().lower()
181
+ # "general" is the absence of a category, not a shared one: treating it
182
+ # as a match would merge every uncategorised finding on the same URL.
183
+ category_matches = bool(cat and cat != "general")
184
+ for existing in self.state.findings:
185
+ if _normalise_endpoint(existing.endpoint) != norm_ep:
186
+ continue
187
+ existing_cat = existing.category.strip().lower()
188
+ if category_matches and existing_cat == cat:
189
+ return existing
190
+ if _finding_key(existing.title, existing.endpoint) == key:
191
+ return existing
192
+ if _titles_overlap(title, existing.title):
193
+ return existing
194
+ return None
195
+
196
+ async def add_finding(self, data: dict[str, Any], agent: str = "red") -> Finding:
197
+ score = _as_float(data.get("cvss_score"))
198
+ severity = normalize_severity(data.get("severity"), score)
199
+ title = str(data.get("title", "Untitled finding"))
200
+ endpoint = str(data.get("endpoint", ""))
201
+ category = str(data.get("category", "general"))
202
+ duplicate = (
203
+ self.find_duplicate(title, endpoint, category) if data.get("dedupe", True) else None
204
+ )
205
+ if duplicate is not None:
206
+ # Keep the strongest score and freshest evidence, but do not add a
207
+ # second entry to the report.
208
+ if score is not None and score > duplicate.cvss_score:
209
+ duplicate.cvss_score = score
210
+ duplicate.cvss_vector = str(data.get("cvss_vector", duplicate.cvss_vector))
211
+ duplicate.severity = normalize_severity(severity, score)
212
+ evidence = str(data.get("evidence", ""))
213
+ if evidence and evidence not in duplicate.evidence:
214
+ duplicate.evidence = (duplicate.evidence + "\n\n" + evidence)[:8000]
215
+ await self.bus.emit(
216
+ Event(
217
+ type="finding.merged",
218
+ agent=agent,
219
+ data={"id": duplicate.id, "title": duplicate.title},
220
+ )
221
+ )
222
+ return duplicate
223
+ finding = Finding(
224
+ title=title,
225
+ category=category,
226
+ description=str(data.get("description", "")),
227
+ severity=severity,
228
+ cvss_vector=str(data.get("cvss_vector", "")),
229
+ cvss_score=score if score is not None else 0.0,
230
+ target=str(data.get("target", self.state.target)),
231
+ endpoint=endpoint,
232
+ evidence=str(data.get("evidence", ""))[:8000],
233
+ recommendation=str(data.get("recommendation", "")),
234
+ cwe=str(data.get("cwe", "")),
235
+ owasp=str(data.get("owasp", "")),
236
+ impact=str(data.get("impact", "")),
237
+ reproduction=str(data.get("reproduction", "")),
238
+ references=list(data.get("references", []) or []),
239
+ confidence=str(data.get("confidence", "medium")),
240
+ discovered_by=agent,
241
+ round=int(data.get("round", self._current_round_index())),
242
+ )
243
+ self.state.add_finding(finding)
244
+ await self.bus.emit(
245
+ Event(
246
+ type="finding",
247
+ agent=agent,
248
+ data={"id": finding.id, "title": finding.title, "severity": severity},
249
+ )
250
+ )
251
+ return finding
252
+
253
+ async def add_mitigation(self, data: dict[str, Any], agent: str = "blue") -> Mitigation:
254
+ finding_id = str(data.get("finding_id", ""))
255
+ mitigation = Mitigation(
256
+ finding_id=finding_id,
257
+ title=str(data.get("title", "Mitigation")),
258
+ kind=str(data.get("kind", "config")),
259
+ description=str(data.get("description", "")),
260
+ content=str(data.get("content", "")),
261
+ rationale=str(data.get("rationale", "")),
262
+ round=int(data.get("round", self._current_round_index())),
263
+ )
264
+ self.state.add_mitigation(mitigation)
265
+ finding = self.state.get_finding(finding_id)
266
+ if finding is not None:
267
+ finding.status = "mitigated"
268
+ await self.bus.emit(
269
+ Event(
270
+ type="mitigation",
271
+ agent=agent,
272
+ data={
273
+ "id": mitigation.id,
274
+ "finding_id": finding_id,
275
+ "title": mitigation.title,
276
+ "kind": mitigation.kind,
277
+ },
278
+ )
279
+ )
280
+ if finding is not None:
281
+ await self.bus.emit(
282
+ Event(
283
+ type="finding.updated",
284
+ agent=agent,
285
+ data={"id": finding.id, "status": finding.status},
286
+ )
287
+ )
288
+ return mitigation
289
+
290
+ async def add_note(self, note: str) -> None:
291
+ self.state.notes.append(note)
292
+ await self.bus.emit(Event(type="log", agent="core", data={"text": note}))
293
+
294
+ async def add_checkpoint(self, agent: str, summary: str, removed: int, preserved: int) -> None:
295
+ """Record a compaction checkpoint for audit and replay."""
296
+ from datetime import datetime, timezone
297
+
298
+ checkpoint = {
299
+ "agent": agent,
300
+ "round": self._current_round_index(),
301
+ "removed_tokens": removed,
302
+ "preserved_tokens": preserved,
303
+ "summary": summary,
304
+ "created_at": datetime.now(timezone.utc).isoformat(timespec="seconds"),
305
+ }
306
+ self.checkpoints.append(checkpoint)
307
+ self.state.checkpoints.append(checkpoint)
308
+ await self.bus.emit(
309
+ Event(
310
+ type="checkpoint",
311
+ agent=agent,
312
+ data={
313
+ "removed": removed,
314
+ "preserved": preserved,
315
+ "summary": summary[:4000],
316
+ },
317
+ )
318
+ )
319
+
320
+ async def set_todos(self, todos: list[dict[str, Any]]) -> None:
321
+ """Replace the session task list (OpenCode's ``todowrite`` pattern)."""
322
+ cleaned: list[dict[str, str]] = []
323
+ for item in todos or []:
324
+ if not isinstance(item, dict):
325
+ continue
326
+ content = str(item.get("content", "")).strip()
327
+ if not content:
328
+ continue
329
+ status = str(item.get("status", "pending")).lower()
330
+ if status not in ("pending", "in_progress", "completed", "cancelled"):
331
+ status = "pending"
332
+ priority = str(item.get("priority", "medium")).lower()
333
+ if priority not in ("high", "medium", "low"):
334
+ priority = "medium"
335
+ cleaned.append({"content": content, "status": status, "priority": priority})
336
+ self.state.todos = cleaned
337
+ await self.bus.emit(Event(type="todos", agent="core", data={"todos": cleaned}))
338
+
339
+ def todo_summary(self) -> str:
340
+ if not self.state.todos:
341
+ return "(no task list yet)"
342
+ return "\n".join(
343
+ f"- [{item['status']}] {item['content']} ({item['priority']})"
344
+ for item in self.state.todos
345
+ )
346
+
347
+ async def add_trace(self, agent: str, trace: list[dict[str, Any]]) -> None:
348
+ """Persist a full reasoning/tool trace so the run can be replayed."""
349
+ if not trace:
350
+ return
351
+ entries = self.state.traces.setdefault(agent, [])
352
+ entries.append(
353
+ {
354
+ "round": self._current_round_index(),
355
+ "entries": trace,
356
+ }
357
+ )
358
+ # Bound the persisted trace so long engagements stay loadable.
359
+ if len(entries) > 8:
360
+ del entries[: len(entries) - 8]
361
+
362
+ # -- rounds ------------------------------------------------------------ #
363
+ def _current_round_index(self) -> int:
364
+ latest = self.state.latest_round()
365
+ return latest.index if latest else 1
366
+
367
+ def begin_round(self, index: int) -> Round:
368
+ round_ = Round(index=index)
369
+ self.state.rounds.append(round_)
370
+ return round_
371
+
372
+ def end_round(self, index: int, red_summary: str, blue_summary: str) -> None:
373
+ from datetime import datetime, timezone
374
+
375
+ for round_ in self.state.rounds:
376
+ if round_.index == index:
377
+ round_.red_summary = red_summary
378
+ round_.blue_summary = blue_summary
379
+ round_.ended_at = datetime.now(timezone.utc).isoformat(timespec="seconds")
380
+ break
381
+
382
+ # -- views for prompts ------------------------------------------------- #
383
+ def findings_digest(self, limit: int = 40) -> str:
384
+ if not self.state.findings:
385
+ return "(no findings yet)"
386
+ lines: list[str] = []
387
+ for finding in self.state.findings[-limit:]:
388
+ lines.append(
389
+ f"- [{finding.id}] {finding.severity.upper()} "
390
+ f"({finding.cvss_score:.1f}) {finding.title} @ "
391
+ f"{finding.endpoint or finding.target} [{finding.status}]"
392
+ )
393
+ return "\n".join(lines)
394
+
395
+ def mitigations_digest(self, limit: int = 40) -> str:
396
+ if not self.state.mitigations:
397
+ return "(no mitigations yet)"
398
+ lines = [
399
+ f"- [{m.id}] {m.kind}: {m.title} -> finding {m.finding_id or 'n/a'} ({m.status})"
400
+ for m in self.state.mitigations[-limit:]
401
+ ]
402
+ return "\n".join(lines)
403
+
404
+ def open_findings(self) -> list[Finding]:
405
+ return [f for f in self.state.findings if f.status == "open"]
406
+
407
+ def summary_dict(self) -> dict[str, Any]:
408
+ return {
409
+ "session": self.state.id,
410
+ "target": self.state.target,
411
+ "rounds": len(self.state.rounds),
412
+ "findings": len(self.state.findings),
413
+ "mitigations": len(self.state.mitigations),
414
+ "severity": self.state.severity_counts(),
415
+ "resilience": self.state.resilience_score(),
416
+ }
417
+
418
+
419
+ _NOISE = (
420
+ "blue",
421
+ "red",
422
+ "agent",
423
+ "review",
424
+ "confirmed",
425
+ "re-check",
426
+ "recheck",
427
+ "round",
428
+ "duplicate",
429
+ "again",
430
+ "still",
431
+ )
432
+
433
+ # Filler that carries no identifying meaning; ignored when comparing titles.
434
+ _FILLER = frozenset(
435
+ [
436
+ "a",
437
+ "an",
438
+ "and",
439
+ "are",
440
+ "as",
441
+ "at",
442
+ "be",
443
+ "been",
444
+ "both",
445
+ "by",
446
+ "for",
447
+ "from",
448
+ "has",
449
+ "have",
450
+ "in",
451
+ "into",
452
+ "is",
453
+ "it",
454
+ "its",
455
+ "of",
456
+ "on",
457
+ "or",
458
+ "over",
459
+ "the",
460
+ "their",
461
+ "there",
462
+ "these",
463
+ "this",
464
+ "those",
465
+ "to",
466
+ "unauthenticated",
467
+ "via",
468
+ "was",
469
+ "were",
470
+ "with",
471
+ "enabled",
472
+ "exposed",
473
+ "disclosed",
474
+ "served",
475
+ "mounted",
476
+ "method",
477
+ "methods",
478
+ "stack",
479
+ "stacks",
480
+ "web",
481
+ "root",
482
+ "end-of-life",
483
+ "outdated",
484
+ "still",
485
+ "missing",
486
+ ]
487
+ )
488
+
489
+
490
+ def _normalise_endpoint(endpoint: str) -> str:
491
+ """Collapse endpoint spellings that refer to the same surface.
492
+
493
+ ``/`` keeps its identity (never collapse to empty), ``/a/`` becomes
494
+ ``/a``, and the wording agents wrap around ports is stripped so
495
+ ``/ and /dav/`` and ``/dav/`` agree.
496
+ """
497
+ import re
498
+
499
+ value = (endpoint or "").strip().lower()
500
+ value = value.split("?")[0]
501
+ # Drop the "port 80" / "ports 32768 and 8080" prose agents add.
502
+ value = re.sub(r"\bports?\s+[\d\s,and]*", "", value)
503
+ value = value.replace(" and ", ", ")
504
+ parts = []
505
+ for chunk in value.split(","):
506
+ chunk = chunk.strip().rstrip("/")
507
+ if chunk:
508
+ parts.append(chunk)
509
+ if not parts:
510
+ return "/"
511
+ return ",".join(sorted(set(parts)))
512
+
513
+
514
+ def _finding_key(title: str, endpoint: str) -> tuple[str, str]:
515
+ """Normalise a finding into a stable identity for duplicate detection."""
516
+ import re
517
+
518
+ words = re.findall(r"[a-z0-9]+", (title or "").lower())
519
+ # Drop agent bookkeeping words so "XSS in page param — Blue review"
520
+ # collapses onto "XSS in page param".
521
+ kept = [w for w in words if w not in _NOISE]
522
+ stem = " ".join(kept[:10])
523
+ return stem, _normalise_endpoint(endpoint)
524
+
525
+
526
+ def _significant(title: str) -> tuple[set[str], set[str]]:
527
+ """Split a title into (identifying tokens, acronyms/identifiers)."""
528
+ import re
529
+
530
+ words = [w for w in re.findall(r"[a-z0-9]+", (title or "").lower()) if w not in _NOISE]
531
+ identity = {w for w in words if w not in _FILLER and len(w) > 2}
532
+ # Decisive = product names and identifiers: a CVE id, a version, a file or
533
+ # service name. Acronyms only count from four letters up, so the "xss"
534
+ # inside "index" and "page" cannot masquerade as a shared identifier.
535
+ decisive = {
536
+ w
537
+ for w in identity
538
+ if any(ch.isdigit() for ch in w)
539
+ or (
540
+ len(w) >= 4 and w in {"webdav", "phpinfo", "trace", "server", "status", "dav", "vsftpd"}
541
+ )
542
+ }
543
+ return identity, decisive
544
+
545
+
546
+ def _titles_overlap(a: str, b: str, threshold: float = 0.5) -> bool:
547
+ """True when two titles describe the same issue in different words.
548
+
549
+ The signal is the *shared identifying vocabulary*: two phrasings of the
550
+ same phpinfo/TRACE/WebDAV/headers issue share distinctive words even when
551
+ the surrounding prose differs. "Missing CSP header" vs "Missing HSTS
552
+ header" share only the generic filler, so they stay separate.
553
+
554
+ Decisive tokens (acronyms, versions, identifiers) break ties: a pair that
555
+ agrees on one of those is a merge even if the rest is reworded.
556
+ """
557
+ left, left_dec = _significant(a)
558
+ right, right_dec = _significant(b)
559
+ if not left or not right:
560
+ return False
561
+ shared = left & right
562
+ shared_dec = left_dec & right_dec
563
+ # An acronym/identifier both sides name is decisive on its own.
564
+ if shared_dec:
565
+ return True
566
+ if not shared:
567
+ return False
568
+ # Distinguishing words are those only one side has. If each side names its
569
+ # own specific thing (CSP vs HSTS, xss vs disclosure), they are different
570
+ # findings even though the surrounding wording matches.
571
+ left_only = left - right
572
+ right_only = right - left
573
+ if left_only and right_only:
574
+ # Shared vocabulary must clearly dominate for a merge.
575
+ ratio = len(shared) / max(len(left), len(right))
576
+ if ratio < 0.6:
577
+ return False
578
+ return len(shared) / max(1, min(len(left), len(right))) >= threshold
579
+
580
+
581
+ def _as_float(value: Any) -> float | None:
582
+ if value is None or value == "":
583
+ return None
584
+ try:
585
+ return float(value)
586
+ except (TypeError, ValueError):
587
+ return None