splitagent 0.0.3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. splitagent/__init__.py +8 -0
  2. splitagent/__main__.py +6 -0
  3. splitagent/agents/__init__.py +10 -0
  4. splitagent/agents/base.py +477 -0
  5. splitagent/agents/blue.py +57 -0
  6. splitagent/agents/chat.py +60 -0
  7. splitagent/agents/prompts.py +462 -0
  8. splitagent/agents/red.py +75 -0
  9. splitagent/cli.py +701 -0
  10. splitagent/config.py +697 -0
  11. splitagent/core/__init__.py +19 -0
  12. splitagent/core/bus.py +62 -0
  13. splitagent/core/context.py +587 -0
  14. splitagent/core/context_manager.py +381 -0
  15. splitagent/core/engine.py +424 -0
  16. splitagent/core/models.py +310 -0
  17. splitagent/core/proc.py +73 -0
  18. splitagent/core/sandbox.py +184 -0
  19. splitagent/core/toolbox.py +520 -0
  20. splitagent/core/workspace.py +420 -0
  21. splitagent/desktop/__init__.py +7 -0
  22. splitagent/desktop/api.py +525 -0
  23. splitagent/desktop/app.py +1131 -0
  24. splitagent/desktop/web/app.js +3067 -0
  25. splitagent/desktop/web/assets/Inter.ttf +0 -0
  26. splitagent/desktop/web/assets/JetBrainsMonoNerdFontMono-Regular.woff2 +0 -0
  27. splitagent/desktop/web/index.html +760 -0
  28. splitagent/desktop/web/styles.css +1612 -0
  29. splitagent/errors.py +27 -0
  30. splitagent/llm/__init__.py +8 -0
  31. splitagent/llm/client.py +488 -0
  32. splitagent/llm/types.py +172 -0
  33. splitagent/report/__init__.py +9 -0
  34. splitagent/report/cvss.py +93 -0
  35. splitagent/report/generator.py +733 -0
  36. splitagent/tools/__init__.py +8 -0
  37. splitagent/tools/base.py +135 -0
  38. splitagent/tools/defense.py +475 -0
  39. splitagent/tools/exploit.py +318 -0
  40. splitagent/tools/http_pool.py +109 -0
  41. splitagent/tools/knowledge.py +376 -0
  42. splitagent/tools/recon.py +182 -0
  43. splitagent/tools/registry.py +62 -0
  44. splitagent/tools/validate.py +908 -0
  45. splitagent/tools/web.py +386 -0
  46. splitagent/tools/workspace_tools.py +411 -0
  47. splitagent/ui/__init__.py +5 -0
  48. splitagent/ui/app.py +389 -0
  49. splitagent/ui/stream.py +234 -0
  50. splitagent/ui/theme.py +72 -0
  51. splitagent-0.0.3.dist-info/METADATA +987 -0
  52. splitagent-0.0.3.dist-info/RECORD +56 -0
  53. splitagent-0.0.3.dist-info/WHEEL +5 -0
  54. splitagent-0.0.3.dist-info/entry_points.txt +2 -0
  55. splitagent-0.0.3.dist-info/licenses/LICENSE +21 -0
  56. splitagent-0.0.3.dist-info/top_level.txt +1 -0
@@ -0,0 +1,386 @@
1
+ """Web/HTTP reconnaissance and configuration-audit tools."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ import re
7
+ import time
8
+ from typing import Any
9
+ from urllib.parse import urljoin, urlparse
10
+
11
+ import httpx
12
+
13
+ from splitagent.tools.base import Tool, ToolContext
14
+ from splitagent.tools.http_pool import get_client, request_headers
15
+
16
+ SECURITY_HEADERS = {
17
+ "strict-transport-security": "HSTS not set: TLS downgrade/stripping possible.",
18
+ "content-security-policy": "No CSP: reflected content can execute scripts.",
19
+ "x-content-type-options": "MIME sniffing not disabled.",
20
+ "x-frame-options": "Clickjacking protection missing.",
21
+ "referrer-policy": "Referrer leakage possible.",
22
+ "permissions-policy": "Browser features not restricted.",
23
+ }
24
+
25
+ INFO_HEADERS = ("server", "x-powered-by", "x-aspnet-version", "x-generator")
26
+
27
+ TECH_SIGNATURES = {
28
+ "wp-content": "WordPress",
29
+ "wp-includes": "WordPress",
30
+ "joomla": "Joomla",
31
+ "drupal": "Drupal",
32
+ "react": "React",
33
+ "vue": "Vue.js",
34
+ "angular": "Angular",
35
+ "jquery": "jQuery",
36
+ "bootstrap": "Bootstrap",
37
+ "django": "Django",
38
+ "laravel": "Laravel",
39
+ "php": "PHP",
40
+ "express": "Express",
41
+ "nginx": "Nginx",
42
+ "apache": "Apache",
43
+ "tomcat": "Tomcat",
44
+ "spring": "Spring",
45
+ }
46
+
47
+ DEFAULT_PATHS = [
48
+ "/robots.txt",
49
+ "/sitemap.xml",
50
+ "/security.txt",
51
+ "/.well-known/security.txt",
52
+ "/.git/HEAD",
53
+ "/.env",
54
+ "/.svn/entries",
55
+ "/backup.zip",
56
+ "/admin",
57
+ "/admin/login",
58
+ "/api",
59
+ "/api/v1",
60
+ "/swagger.json",
61
+ "/openapi.json",
62
+ "/actuator",
63
+ "/actuator/health",
64
+ "/debug",
65
+ "/phpinfo.php",
66
+ "/server-status",
67
+ "/graphql",
68
+ ]
69
+
70
+
71
+ _REDIRECT_CODES = {301, 302, 303, 307, 308}
72
+ _MAX_REDIRECTS = 5
73
+
74
+
75
+ async def _request(
76
+ ctx: ToolContext,
77
+ url: str,
78
+ method: str = "GET",
79
+ headers: dict[str, str] | None = None,
80
+ params: dict[str, Any] | None = None,
81
+ data: Any = None,
82
+ follow: bool = True,
83
+ max_body: int = 8000,
84
+ ) -> dict[str, Any]:
85
+ ctx.check_scope(url)
86
+ started = time.perf_counter()
87
+ merged_headers = ctx.auth_headers()
88
+ merged_headers.update(headers or {})
89
+ # Never let httpx follow redirects on its own: a single in-scope URL can
90
+ # 302 to an out-of-scope host. We follow manually and re-check the scope of
91
+ # every hop.
92
+ client = await get_client(follow=False, verify=False, timeout=15.0, headers=merged_headers)
93
+ # The client already carries merged_headers as defaults; sending them again
94
+ # would duplicate every header.
95
+ response = await client.request(
96
+ method.upper(),
97
+ url,
98
+ headers=request_headers(client, headers),
99
+ params=params,
100
+ data=data,
101
+ )
102
+ if follow:
103
+ hops = 0
104
+ while response.status_code in _REDIRECT_CODES and hops < _MAX_REDIRECTS:
105
+ location = response.headers.get("location")
106
+ if not location:
107
+ break
108
+ next_url = urljoin(str(response.url), location)
109
+ ctx.check_scope(next_url) # raises ScopeError for an off-scope hop
110
+ hops += 1
111
+ response = await client.get(
112
+ next_url,
113
+ headers=request_headers(client, headers),
114
+ )
115
+ elapsed = round((time.perf_counter() - started) * 1000, 1)
116
+ body = response.text[:max_body]
117
+ return {
118
+ "url": str(response.url),
119
+ "status": response.status_code,
120
+ "reason": response.reason_phrase,
121
+ "headers": {k.lower(): v for k, v in response.headers.items()},
122
+ "content_type": response.headers.get("content-type", ""),
123
+ "elapsed_ms": elapsed,
124
+ "body_length": len(response.content),
125
+ "body": body,
126
+ "truncated": len(response.content) > max_body,
127
+ }
128
+
129
+
130
+ async def http_request(
131
+ ctx: ToolContext,
132
+ url: str,
133
+ method: str = "GET",
134
+ headers: dict[str, str] | None = None,
135
+ ) -> dict[str, Any]:
136
+ return await _request(ctx, url, method=method, headers=headers)
137
+
138
+
139
+ async def fetch(
140
+ ctx: ToolContext,
141
+ url: str,
142
+ method: str = "GET",
143
+ headers: dict[str, str] | None = None,
144
+ params: dict[str, Any] | None = None,
145
+ data: Any = None,
146
+ follow: bool = True,
147
+ ) -> dict[str, Any]:
148
+ return await _request(
149
+ ctx, url, method=method, headers=headers, params=params, data=data, follow=follow
150
+ )
151
+
152
+
153
+ async def headers_audit(ctx: ToolContext, url: str) -> dict[str, Any]:
154
+ response = await _request(ctx, url, method="GET")
155
+ headers = response["headers"]
156
+ missing = [
157
+ {"header": name, "impact": impact}
158
+ for name, impact in SECURITY_HEADERS.items()
159
+ if name not in headers
160
+ ]
161
+ exposed = {name: headers[name] for name in INFO_HEADERS if name in headers}
162
+ cookies = []
163
+ for header_name, value in headers.items():
164
+ if header_name == "set-cookie":
165
+ cookies.append(value)
166
+ cookie_flags = {
167
+ "secure": any("secure" in c.lower() for c in cookies),
168
+ "httponly": any("httponly" in c.lower() for c in cookies),
169
+ "samesite": any("samesite" in c.lower() for c in cookies),
170
+ }
171
+ return {
172
+ "url": response["url"],
173
+ "status": response["status"],
174
+ "missing_security_headers": missing,
175
+ "information_disclosure": exposed,
176
+ "cookie_count": len(cookies),
177
+ "cookie_flags": cookie_flags if cookies else {},
178
+ }
179
+
180
+
181
+ async def probe_paths(
182
+ ctx: ToolContext,
183
+ url: str,
184
+ paths: list[str] | None = None,
185
+ concurrency: int = 12,
186
+ ) -> dict[str, Any]:
187
+ ctx.check_scope(url)
188
+ candidates = paths or DEFAULT_PATHS
189
+ semaphore = asyncio.Semaphore(concurrency)
190
+ found: list[dict[str, Any]] = []
191
+ client = await get_client(follow=False, verify=False, timeout=10.0, headers=ctx.auth_headers())
192
+
193
+ async def worker(path: str) -> None:
194
+ target = urljoin(url.rstrip("/") + "/", path.lstrip("/"))
195
+ async with semaphore:
196
+ try:
197
+ response = await client.get(target)
198
+ except httpx.HTTPError:
199
+ return
200
+ interesting = response.status_code not in (404, 410) or (response.status_code == 403)
201
+ if not interesting:
202
+ return
203
+ snippet = response.text[:300]
204
+ found.append(
205
+ {
206
+ "path": path,
207
+ "url": target,
208
+ "status": response.status_code,
209
+ "length": len(response.content),
210
+ "snippet": snippet,
211
+ }
212
+ )
213
+
214
+ await asyncio.gather(*(worker(p) for p in candidates))
215
+ found.sort(key=lambda item: item["path"])
216
+ return {"base": url, "found": found}
217
+
218
+
219
+ async def crawl(
220
+ ctx: ToolContext, url: str, max_pages: int = 20, same_host: bool = True
221
+ ) -> dict[str, Any]:
222
+ ctx.check_scope(url)
223
+ base_host = urlparse(url).hostname
224
+ seen: set[str] = set()
225
+ queue = [url]
226
+ pages: list[dict[str, Any]] = []
227
+ forms: list[dict[str, Any]] = []
228
+ link_re = re.compile(r"""href=["']([^"'#]+)["']""", re.IGNORECASE)
229
+ form_re = re.compile(r"<form[^>]*>(.*?)</form>", re.IGNORECASE | re.DOTALL)
230
+ action_re = re.compile(r"""action=["']([^"']*)["']""", re.IGNORECASE)
231
+ input_re = re.compile(r"""name=["']([^"']+)["']""", re.IGNORECASE)
232
+
233
+ # follow=False: each URL is scope-checked below, and redirects must not be
234
+ # followed silently to a host outside the scope.
235
+ client = await get_client(follow=False, verify=False, timeout=10.0, headers=ctx.auth_headers())
236
+ while queue and len(pages) < max_pages:
237
+ current = queue.pop(0)
238
+ if current in seen:
239
+ continue
240
+ seen.add(current)
241
+ try:
242
+ ctx.check_scope(current)
243
+ except Exception:
244
+ continue
245
+ try:
246
+ response = await client.get(current)
247
+ except httpx.HTTPError:
248
+ continue
249
+ body = response.text[:60000]
250
+ pages.append(
251
+ {
252
+ "url": str(response.url),
253
+ "status": response.status_code,
254
+ "title": _extract_title(body),
255
+ }
256
+ )
257
+ for match in form_re.finditer(body):
258
+ form_html = match.group(1)
259
+ action = action_re.search(match.group(0))
260
+ inputs = input_re.findall(form_html)
261
+ forms.append(
262
+ {
263
+ "page": str(response.url),
264
+ "action": action.group(1) if action else "",
265
+ "method": "post" if 'method="post"' in match.group(0).lower() else "get",
266
+ "inputs": inputs,
267
+ }
268
+ )
269
+ for link in link_re.findall(body):
270
+ absolute = urljoin(str(response.url), link)
271
+ parsed = urlparse(absolute)
272
+ if parsed.scheme not in ("http", "https"):
273
+ continue
274
+ if same_host and parsed.hostname != base_host:
275
+ continue
276
+ if absolute not in seen and absolute not in queue:
277
+ queue.append(absolute)
278
+
279
+ return {
280
+ "base": url,
281
+ "pages": pages,
282
+ "forms": forms,
283
+ "page_count": len(pages),
284
+ "form_count": len(forms),
285
+ }
286
+
287
+
288
+ def fingerprint(ctx: ToolContext, url: str = "") -> dict[str, Any]:
289
+ return {"url": url or ctx.target.url, "note": "use headers_audit/fetch to fingerprint"}
290
+
291
+
292
+ def detect_technologies(body: str, headers: dict[str, str]) -> list[str]:
293
+ haystack = body.lower()
294
+ found = {name for signature, name in TECH_SIGNATURES.items() if signature in haystack}
295
+ header_blob = " ".join(f"{k}:{v}" for k, v in headers.items()).lower()
296
+ for signature, name in TECH_SIGNATURES.items():
297
+ if signature in header_blob:
298
+ found.add(name)
299
+ return sorted(found)
300
+
301
+
302
+ def _extract_title(body: str) -> str:
303
+ match = re.search(r"<title[^>]*>(.*?)</title>", body, re.IGNORECASE | re.DOTALL)
304
+ return re.sub(r"\s+", " ", match.group(1)).strip()[:120] if match else ""
305
+
306
+
307
+ def red_web_tools(ctx: ToolContext) -> list[Tool]:
308
+ async def _fetch(
309
+ url: str,
310
+ method: str = "GET",
311
+ headers: dict[str, str] | None = None,
312
+ params: dict[str, Any] | None = None,
313
+ data: Any = None,
314
+ ) -> dict[str, Any]:
315
+ return await fetch(ctx, url, method=method, headers=headers, params=params, data=data)
316
+
317
+ return [
318
+ Tool(
319
+ name="http_request",
320
+ description=(
321
+ "Send an arbitrary HTTP request to the target and inspect the "
322
+ "status, response headers, timing and body slice."
323
+ ),
324
+ parameters={
325
+ "type": "object",
326
+ "properties": {
327
+ "url": {"type": "string"},
328
+ "method": {"type": "string", "default": "GET"},
329
+ "headers": {"type": "object"},
330
+ "params": {"type": "object"},
331
+ "data": {"type": "object"},
332
+ },
333
+ "required": ["url"],
334
+ },
335
+ func=_fetch,
336
+ scope="red",
337
+ ),
338
+ Tool(
339
+ name="audit_security_headers",
340
+ description=(
341
+ "Audit HTTP security headers, information disclosure and cookie "
342
+ "flags for a URL. Produces concrete, evidence-backed weaknesses."
343
+ ),
344
+ parameters={
345
+ "type": "object",
346
+ "properties": {"url": {"type": "string"}},
347
+ "required": ["url"],
348
+ },
349
+ func=lambda url: headers_audit(ctx, url),
350
+ scope="red",
351
+ ),
352
+ Tool(
353
+ name="probe_paths",
354
+ description=(
355
+ "Probe a curated list of sensitive paths (.git, .env, admin, "
356
+ "swagger, actuator...) and report anything that is not a 404."
357
+ ),
358
+ parameters={
359
+ "type": "object",
360
+ "properties": {
361
+ "url": {"type": "string"},
362
+ "paths": {"type": "array", "items": {"type": "string"}},
363
+ },
364
+ "required": ["url"],
365
+ },
366
+ func=lambda url, paths=None: probe_paths(ctx, url, paths),
367
+ scope="red",
368
+ ),
369
+ Tool(
370
+ name="crawl",
371
+ description=(
372
+ "Crawl the target up to max_pages, returning discovered URLs, "
373
+ "titles and HTML forms (useful to find input points)."
374
+ ),
375
+ parameters={
376
+ "type": "object",
377
+ "properties": {
378
+ "url": {"type": "string"},
379
+ "max_pages": {"type": "integer", "default": 20},
380
+ },
381
+ "required": ["url"],
382
+ },
383
+ func=lambda url, max_pages=20: crawl(ctx, url, max_pages=max_pages),
384
+ scope="red",
385
+ ),
386
+ ]