splitagent 0.0.3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- splitagent/__init__.py +8 -0
- splitagent/__main__.py +6 -0
- splitagent/agents/__init__.py +10 -0
- splitagent/agents/base.py +477 -0
- splitagent/agents/blue.py +57 -0
- splitagent/agents/chat.py +60 -0
- splitagent/agents/prompts.py +462 -0
- splitagent/agents/red.py +75 -0
- splitagent/cli.py +701 -0
- splitagent/config.py +697 -0
- splitagent/core/__init__.py +19 -0
- splitagent/core/bus.py +62 -0
- splitagent/core/context.py +587 -0
- splitagent/core/context_manager.py +381 -0
- splitagent/core/engine.py +424 -0
- splitagent/core/models.py +310 -0
- splitagent/core/proc.py +73 -0
- splitagent/core/sandbox.py +184 -0
- splitagent/core/toolbox.py +520 -0
- splitagent/core/workspace.py +420 -0
- splitagent/desktop/__init__.py +7 -0
- splitagent/desktop/api.py +525 -0
- splitagent/desktop/app.py +1131 -0
- splitagent/desktop/web/app.js +3067 -0
- splitagent/desktop/web/assets/Inter.ttf +0 -0
- splitagent/desktop/web/assets/JetBrainsMonoNerdFontMono-Regular.woff2 +0 -0
- splitagent/desktop/web/index.html +760 -0
- splitagent/desktop/web/styles.css +1612 -0
- splitagent/errors.py +27 -0
- splitagent/llm/__init__.py +8 -0
- splitagent/llm/client.py +488 -0
- splitagent/llm/types.py +172 -0
- splitagent/report/__init__.py +9 -0
- splitagent/report/cvss.py +93 -0
- splitagent/report/generator.py +733 -0
- splitagent/tools/__init__.py +8 -0
- splitagent/tools/base.py +135 -0
- splitagent/tools/defense.py +475 -0
- splitagent/tools/exploit.py +318 -0
- splitagent/tools/http_pool.py +109 -0
- splitagent/tools/knowledge.py +376 -0
- splitagent/tools/recon.py +182 -0
- splitagent/tools/registry.py +62 -0
- splitagent/tools/validate.py +908 -0
- splitagent/tools/web.py +386 -0
- splitagent/tools/workspace_tools.py +411 -0
- splitagent/ui/__init__.py +5 -0
- splitagent/ui/app.py +389 -0
- splitagent/ui/stream.py +234 -0
- splitagent/ui/theme.py +72 -0
- splitagent-0.0.3.dist-info/METADATA +987 -0
- splitagent-0.0.3.dist-info/RECORD +56 -0
- splitagent-0.0.3.dist-info/WHEEL +5 -0
- splitagent-0.0.3.dist-info/entry_points.txt +2 -0
- splitagent-0.0.3.dist-info/licenses/LICENSE +21 -0
- splitagent-0.0.3.dist-info/top_level.txt +1 -0
splitagent/tools/web.py
ADDED
|
@@ -0,0 +1,386 @@
|
|
|
1
|
+
"""Web/HTTP reconnaissance and configuration-audit tools."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import re
|
|
7
|
+
import time
|
|
8
|
+
from typing import Any
|
|
9
|
+
from urllib.parse import urljoin, urlparse
|
|
10
|
+
|
|
11
|
+
import httpx
|
|
12
|
+
|
|
13
|
+
from splitagent.tools.base import Tool, ToolContext
|
|
14
|
+
from splitagent.tools.http_pool import get_client, request_headers
|
|
15
|
+
|
|
16
|
+
SECURITY_HEADERS = {
|
|
17
|
+
"strict-transport-security": "HSTS not set: TLS downgrade/stripping possible.",
|
|
18
|
+
"content-security-policy": "No CSP: reflected content can execute scripts.",
|
|
19
|
+
"x-content-type-options": "MIME sniffing not disabled.",
|
|
20
|
+
"x-frame-options": "Clickjacking protection missing.",
|
|
21
|
+
"referrer-policy": "Referrer leakage possible.",
|
|
22
|
+
"permissions-policy": "Browser features not restricted.",
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
INFO_HEADERS = ("server", "x-powered-by", "x-aspnet-version", "x-generator")
|
|
26
|
+
|
|
27
|
+
TECH_SIGNATURES = {
|
|
28
|
+
"wp-content": "WordPress",
|
|
29
|
+
"wp-includes": "WordPress",
|
|
30
|
+
"joomla": "Joomla",
|
|
31
|
+
"drupal": "Drupal",
|
|
32
|
+
"react": "React",
|
|
33
|
+
"vue": "Vue.js",
|
|
34
|
+
"angular": "Angular",
|
|
35
|
+
"jquery": "jQuery",
|
|
36
|
+
"bootstrap": "Bootstrap",
|
|
37
|
+
"django": "Django",
|
|
38
|
+
"laravel": "Laravel",
|
|
39
|
+
"php": "PHP",
|
|
40
|
+
"express": "Express",
|
|
41
|
+
"nginx": "Nginx",
|
|
42
|
+
"apache": "Apache",
|
|
43
|
+
"tomcat": "Tomcat",
|
|
44
|
+
"spring": "Spring",
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
DEFAULT_PATHS = [
|
|
48
|
+
"/robots.txt",
|
|
49
|
+
"/sitemap.xml",
|
|
50
|
+
"/security.txt",
|
|
51
|
+
"/.well-known/security.txt",
|
|
52
|
+
"/.git/HEAD",
|
|
53
|
+
"/.env",
|
|
54
|
+
"/.svn/entries",
|
|
55
|
+
"/backup.zip",
|
|
56
|
+
"/admin",
|
|
57
|
+
"/admin/login",
|
|
58
|
+
"/api",
|
|
59
|
+
"/api/v1",
|
|
60
|
+
"/swagger.json",
|
|
61
|
+
"/openapi.json",
|
|
62
|
+
"/actuator",
|
|
63
|
+
"/actuator/health",
|
|
64
|
+
"/debug",
|
|
65
|
+
"/phpinfo.php",
|
|
66
|
+
"/server-status",
|
|
67
|
+
"/graphql",
|
|
68
|
+
]
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
_REDIRECT_CODES = {301, 302, 303, 307, 308}
|
|
72
|
+
_MAX_REDIRECTS = 5
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
async def _request(
|
|
76
|
+
ctx: ToolContext,
|
|
77
|
+
url: str,
|
|
78
|
+
method: str = "GET",
|
|
79
|
+
headers: dict[str, str] | None = None,
|
|
80
|
+
params: dict[str, Any] | None = None,
|
|
81
|
+
data: Any = None,
|
|
82
|
+
follow: bool = True,
|
|
83
|
+
max_body: int = 8000,
|
|
84
|
+
) -> dict[str, Any]:
|
|
85
|
+
ctx.check_scope(url)
|
|
86
|
+
started = time.perf_counter()
|
|
87
|
+
merged_headers = ctx.auth_headers()
|
|
88
|
+
merged_headers.update(headers or {})
|
|
89
|
+
# Never let httpx follow redirects on its own: a single in-scope URL can
|
|
90
|
+
# 302 to an out-of-scope host. We follow manually and re-check the scope of
|
|
91
|
+
# every hop.
|
|
92
|
+
client = await get_client(follow=False, verify=False, timeout=15.0, headers=merged_headers)
|
|
93
|
+
# The client already carries merged_headers as defaults; sending them again
|
|
94
|
+
# would duplicate every header.
|
|
95
|
+
response = await client.request(
|
|
96
|
+
method.upper(),
|
|
97
|
+
url,
|
|
98
|
+
headers=request_headers(client, headers),
|
|
99
|
+
params=params,
|
|
100
|
+
data=data,
|
|
101
|
+
)
|
|
102
|
+
if follow:
|
|
103
|
+
hops = 0
|
|
104
|
+
while response.status_code in _REDIRECT_CODES and hops < _MAX_REDIRECTS:
|
|
105
|
+
location = response.headers.get("location")
|
|
106
|
+
if not location:
|
|
107
|
+
break
|
|
108
|
+
next_url = urljoin(str(response.url), location)
|
|
109
|
+
ctx.check_scope(next_url) # raises ScopeError for an off-scope hop
|
|
110
|
+
hops += 1
|
|
111
|
+
response = await client.get(
|
|
112
|
+
next_url,
|
|
113
|
+
headers=request_headers(client, headers),
|
|
114
|
+
)
|
|
115
|
+
elapsed = round((time.perf_counter() - started) * 1000, 1)
|
|
116
|
+
body = response.text[:max_body]
|
|
117
|
+
return {
|
|
118
|
+
"url": str(response.url),
|
|
119
|
+
"status": response.status_code,
|
|
120
|
+
"reason": response.reason_phrase,
|
|
121
|
+
"headers": {k.lower(): v for k, v in response.headers.items()},
|
|
122
|
+
"content_type": response.headers.get("content-type", ""),
|
|
123
|
+
"elapsed_ms": elapsed,
|
|
124
|
+
"body_length": len(response.content),
|
|
125
|
+
"body": body,
|
|
126
|
+
"truncated": len(response.content) > max_body,
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
async def http_request(
|
|
131
|
+
ctx: ToolContext,
|
|
132
|
+
url: str,
|
|
133
|
+
method: str = "GET",
|
|
134
|
+
headers: dict[str, str] | None = None,
|
|
135
|
+
) -> dict[str, Any]:
|
|
136
|
+
return await _request(ctx, url, method=method, headers=headers)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
async def fetch(
|
|
140
|
+
ctx: ToolContext,
|
|
141
|
+
url: str,
|
|
142
|
+
method: str = "GET",
|
|
143
|
+
headers: dict[str, str] | None = None,
|
|
144
|
+
params: dict[str, Any] | None = None,
|
|
145
|
+
data: Any = None,
|
|
146
|
+
follow: bool = True,
|
|
147
|
+
) -> dict[str, Any]:
|
|
148
|
+
return await _request(
|
|
149
|
+
ctx, url, method=method, headers=headers, params=params, data=data, follow=follow
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
async def headers_audit(ctx: ToolContext, url: str) -> dict[str, Any]:
|
|
154
|
+
response = await _request(ctx, url, method="GET")
|
|
155
|
+
headers = response["headers"]
|
|
156
|
+
missing = [
|
|
157
|
+
{"header": name, "impact": impact}
|
|
158
|
+
for name, impact in SECURITY_HEADERS.items()
|
|
159
|
+
if name not in headers
|
|
160
|
+
]
|
|
161
|
+
exposed = {name: headers[name] for name in INFO_HEADERS if name in headers}
|
|
162
|
+
cookies = []
|
|
163
|
+
for header_name, value in headers.items():
|
|
164
|
+
if header_name == "set-cookie":
|
|
165
|
+
cookies.append(value)
|
|
166
|
+
cookie_flags = {
|
|
167
|
+
"secure": any("secure" in c.lower() for c in cookies),
|
|
168
|
+
"httponly": any("httponly" in c.lower() for c in cookies),
|
|
169
|
+
"samesite": any("samesite" in c.lower() for c in cookies),
|
|
170
|
+
}
|
|
171
|
+
return {
|
|
172
|
+
"url": response["url"],
|
|
173
|
+
"status": response["status"],
|
|
174
|
+
"missing_security_headers": missing,
|
|
175
|
+
"information_disclosure": exposed,
|
|
176
|
+
"cookie_count": len(cookies),
|
|
177
|
+
"cookie_flags": cookie_flags if cookies else {},
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
async def probe_paths(
|
|
182
|
+
ctx: ToolContext,
|
|
183
|
+
url: str,
|
|
184
|
+
paths: list[str] | None = None,
|
|
185
|
+
concurrency: int = 12,
|
|
186
|
+
) -> dict[str, Any]:
|
|
187
|
+
ctx.check_scope(url)
|
|
188
|
+
candidates = paths or DEFAULT_PATHS
|
|
189
|
+
semaphore = asyncio.Semaphore(concurrency)
|
|
190
|
+
found: list[dict[str, Any]] = []
|
|
191
|
+
client = await get_client(follow=False, verify=False, timeout=10.0, headers=ctx.auth_headers())
|
|
192
|
+
|
|
193
|
+
async def worker(path: str) -> None:
|
|
194
|
+
target = urljoin(url.rstrip("/") + "/", path.lstrip("/"))
|
|
195
|
+
async with semaphore:
|
|
196
|
+
try:
|
|
197
|
+
response = await client.get(target)
|
|
198
|
+
except httpx.HTTPError:
|
|
199
|
+
return
|
|
200
|
+
interesting = response.status_code not in (404, 410) or (response.status_code == 403)
|
|
201
|
+
if not interesting:
|
|
202
|
+
return
|
|
203
|
+
snippet = response.text[:300]
|
|
204
|
+
found.append(
|
|
205
|
+
{
|
|
206
|
+
"path": path,
|
|
207
|
+
"url": target,
|
|
208
|
+
"status": response.status_code,
|
|
209
|
+
"length": len(response.content),
|
|
210
|
+
"snippet": snippet,
|
|
211
|
+
}
|
|
212
|
+
)
|
|
213
|
+
|
|
214
|
+
await asyncio.gather(*(worker(p) for p in candidates))
|
|
215
|
+
found.sort(key=lambda item: item["path"])
|
|
216
|
+
return {"base": url, "found": found}
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
async def crawl(
|
|
220
|
+
ctx: ToolContext, url: str, max_pages: int = 20, same_host: bool = True
|
|
221
|
+
) -> dict[str, Any]:
|
|
222
|
+
ctx.check_scope(url)
|
|
223
|
+
base_host = urlparse(url).hostname
|
|
224
|
+
seen: set[str] = set()
|
|
225
|
+
queue = [url]
|
|
226
|
+
pages: list[dict[str, Any]] = []
|
|
227
|
+
forms: list[dict[str, Any]] = []
|
|
228
|
+
link_re = re.compile(r"""href=["']([^"'#]+)["']""", re.IGNORECASE)
|
|
229
|
+
form_re = re.compile(r"<form[^>]*>(.*?)</form>", re.IGNORECASE | re.DOTALL)
|
|
230
|
+
action_re = re.compile(r"""action=["']([^"']*)["']""", re.IGNORECASE)
|
|
231
|
+
input_re = re.compile(r"""name=["']([^"']+)["']""", re.IGNORECASE)
|
|
232
|
+
|
|
233
|
+
# follow=False: each URL is scope-checked below, and redirects must not be
|
|
234
|
+
# followed silently to a host outside the scope.
|
|
235
|
+
client = await get_client(follow=False, verify=False, timeout=10.0, headers=ctx.auth_headers())
|
|
236
|
+
while queue and len(pages) < max_pages:
|
|
237
|
+
current = queue.pop(0)
|
|
238
|
+
if current in seen:
|
|
239
|
+
continue
|
|
240
|
+
seen.add(current)
|
|
241
|
+
try:
|
|
242
|
+
ctx.check_scope(current)
|
|
243
|
+
except Exception:
|
|
244
|
+
continue
|
|
245
|
+
try:
|
|
246
|
+
response = await client.get(current)
|
|
247
|
+
except httpx.HTTPError:
|
|
248
|
+
continue
|
|
249
|
+
body = response.text[:60000]
|
|
250
|
+
pages.append(
|
|
251
|
+
{
|
|
252
|
+
"url": str(response.url),
|
|
253
|
+
"status": response.status_code,
|
|
254
|
+
"title": _extract_title(body),
|
|
255
|
+
}
|
|
256
|
+
)
|
|
257
|
+
for match in form_re.finditer(body):
|
|
258
|
+
form_html = match.group(1)
|
|
259
|
+
action = action_re.search(match.group(0))
|
|
260
|
+
inputs = input_re.findall(form_html)
|
|
261
|
+
forms.append(
|
|
262
|
+
{
|
|
263
|
+
"page": str(response.url),
|
|
264
|
+
"action": action.group(1) if action else "",
|
|
265
|
+
"method": "post" if 'method="post"' in match.group(0).lower() else "get",
|
|
266
|
+
"inputs": inputs,
|
|
267
|
+
}
|
|
268
|
+
)
|
|
269
|
+
for link in link_re.findall(body):
|
|
270
|
+
absolute = urljoin(str(response.url), link)
|
|
271
|
+
parsed = urlparse(absolute)
|
|
272
|
+
if parsed.scheme not in ("http", "https"):
|
|
273
|
+
continue
|
|
274
|
+
if same_host and parsed.hostname != base_host:
|
|
275
|
+
continue
|
|
276
|
+
if absolute not in seen and absolute not in queue:
|
|
277
|
+
queue.append(absolute)
|
|
278
|
+
|
|
279
|
+
return {
|
|
280
|
+
"base": url,
|
|
281
|
+
"pages": pages,
|
|
282
|
+
"forms": forms,
|
|
283
|
+
"page_count": len(pages),
|
|
284
|
+
"form_count": len(forms),
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def fingerprint(ctx: ToolContext, url: str = "") -> dict[str, Any]:
|
|
289
|
+
return {"url": url or ctx.target.url, "note": "use headers_audit/fetch to fingerprint"}
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def detect_technologies(body: str, headers: dict[str, str]) -> list[str]:
|
|
293
|
+
haystack = body.lower()
|
|
294
|
+
found = {name for signature, name in TECH_SIGNATURES.items() if signature in haystack}
|
|
295
|
+
header_blob = " ".join(f"{k}:{v}" for k, v in headers.items()).lower()
|
|
296
|
+
for signature, name in TECH_SIGNATURES.items():
|
|
297
|
+
if signature in header_blob:
|
|
298
|
+
found.add(name)
|
|
299
|
+
return sorted(found)
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def _extract_title(body: str) -> str:
|
|
303
|
+
match = re.search(r"<title[^>]*>(.*?)</title>", body, re.IGNORECASE | re.DOTALL)
|
|
304
|
+
return re.sub(r"\s+", " ", match.group(1)).strip()[:120] if match else ""
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def red_web_tools(ctx: ToolContext) -> list[Tool]:
|
|
308
|
+
async def _fetch(
|
|
309
|
+
url: str,
|
|
310
|
+
method: str = "GET",
|
|
311
|
+
headers: dict[str, str] | None = None,
|
|
312
|
+
params: dict[str, Any] | None = None,
|
|
313
|
+
data: Any = None,
|
|
314
|
+
) -> dict[str, Any]:
|
|
315
|
+
return await fetch(ctx, url, method=method, headers=headers, params=params, data=data)
|
|
316
|
+
|
|
317
|
+
return [
|
|
318
|
+
Tool(
|
|
319
|
+
name="http_request",
|
|
320
|
+
description=(
|
|
321
|
+
"Send an arbitrary HTTP request to the target and inspect the "
|
|
322
|
+
"status, response headers, timing and body slice."
|
|
323
|
+
),
|
|
324
|
+
parameters={
|
|
325
|
+
"type": "object",
|
|
326
|
+
"properties": {
|
|
327
|
+
"url": {"type": "string"},
|
|
328
|
+
"method": {"type": "string", "default": "GET"},
|
|
329
|
+
"headers": {"type": "object"},
|
|
330
|
+
"params": {"type": "object"},
|
|
331
|
+
"data": {"type": "object"},
|
|
332
|
+
},
|
|
333
|
+
"required": ["url"],
|
|
334
|
+
},
|
|
335
|
+
func=_fetch,
|
|
336
|
+
scope="red",
|
|
337
|
+
),
|
|
338
|
+
Tool(
|
|
339
|
+
name="audit_security_headers",
|
|
340
|
+
description=(
|
|
341
|
+
"Audit HTTP security headers, information disclosure and cookie "
|
|
342
|
+
"flags for a URL. Produces concrete, evidence-backed weaknesses."
|
|
343
|
+
),
|
|
344
|
+
parameters={
|
|
345
|
+
"type": "object",
|
|
346
|
+
"properties": {"url": {"type": "string"}},
|
|
347
|
+
"required": ["url"],
|
|
348
|
+
},
|
|
349
|
+
func=lambda url: headers_audit(ctx, url),
|
|
350
|
+
scope="red",
|
|
351
|
+
),
|
|
352
|
+
Tool(
|
|
353
|
+
name="probe_paths",
|
|
354
|
+
description=(
|
|
355
|
+
"Probe a curated list of sensitive paths (.git, .env, admin, "
|
|
356
|
+
"swagger, actuator...) and report anything that is not a 404."
|
|
357
|
+
),
|
|
358
|
+
parameters={
|
|
359
|
+
"type": "object",
|
|
360
|
+
"properties": {
|
|
361
|
+
"url": {"type": "string"},
|
|
362
|
+
"paths": {"type": "array", "items": {"type": "string"}},
|
|
363
|
+
},
|
|
364
|
+
"required": ["url"],
|
|
365
|
+
},
|
|
366
|
+
func=lambda url, paths=None: probe_paths(ctx, url, paths),
|
|
367
|
+
scope="red",
|
|
368
|
+
),
|
|
369
|
+
Tool(
|
|
370
|
+
name="crawl",
|
|
371
|
+
description=(
|
|
372
|
+
"Crawl the target up to max_pages, returning discovered URLs, "
|
|
373
|
+
"titles and HTML forms (useful to find input points)."
|
|
374
|
+
),
|
|
375
|
+
parameters={
|
|
376
|
+
"type": "object",
|
|
377
|
+
"properties": {
|
|
378
|
+
"url": {"type": "string"},
|
|
379
|
+
"max_pages": {"type": "integer", "default": 20},
|
|
380
|
+
},
|
|
381
|
+
"required": ["url"],
|
|
382
|
+
},
|
|
383
|
+
func=lambda url, max_pages=20: crawl(ctx, url, max_pages=max_pages),
|
|
384
|
+
scope="red",
|
|
385
|
+
),
|
|
386
|
+
]
|