splitagent 0.0.3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. splitagent/__init__.py +8 -0
  2. splitagent/__main__.py +6 -0
  3. splitagent/agents/__init__.py +10 -0
  4. splitagent/agents/base.py +477 -0
  5. splitagent/agents/blue.py +57 -0
  6. splitagent/agents/chat.py +60 -0
  7. splitagent/agents/prompts.py +462 -0
  8. splitagent/agents/red.py +75 -0
  9. splitagent/cli.py +701 -0
  10. splitagent/config.py +697 -0
  11. splitagent/core/__init__.py +19 -0
  12. splitagent/core/bus.py +62 -0
  13. splitagent/core/context.py +587 -0
  14. splitagent/core/context_manager.py +381 -0
  15. splitagent/core/engine.py +424 -0
  16. splitagent/core/models.py +310 -0
  17. splitagent/core/proc.py +73 -0
  18. splitagent/core/sandbox.py +184 -0
  19. splitagent/core/toolbox.py +520 -0
  20. splitagent/core/workspace.py +420 -0
  21. splitagent/desktop/__init__.py +7 -0
  22. splitagent/desktop/api.py +525 -0
  23. splitagent/desktop/app.py +1131 -0
  24. splitagent/desktop/web/app.js +3067 -0
  25. splitagent/desktop/web/assets/Inter.ttf +0 -0
  26. splitagent/desktop/web/assets/JetBrainsMonoNerdFontMono-Regular.woff2 +0 -0
  27. splitagent/desktop/web/index.html +760 -0
  28. splitagent/desktop/web/styles.css +1612 -0
  29. splitagent/errors.py +27 -0
  30. splitagent/llm/__init__.py +8 -0
  31. splitagent/llm/client.py +488 -0
  32. splitagent/llm/types.py +172 -0
  33. splitagent/report/__init__.py +9 -0
  34. splitagent/report/cvss.py +93 -0
  35. splitagent/report/generator.py +733 -0
  36. splitagent/tools/__init__.py +8 -0
  37. splitagent/tools/base.py +135 -0
  38. splitagent/tools/defense.py +475 -0
  39. splitagent/tools/exploit.py +318 -0
  40. splitagent/tools/http_pool.py +109 -0
  41. splitagent/tools/knowledge.py +376 -0
  42. splitagent/tools/recon.py +182 -0
  43. splitagent/tools/registry.py +62 -0
  44. splitagent/tools/validate.py +908 -0
  45. splitagent/tools/web.py +386 -0
  46. splitagent/tools/workspace_tools.py +411 -0
  47. splitagent/ui/__init__.py +5 -0
  48. splitagent/ui/app.py +389 -0
  49. splitagent/ui/stream.py +234 -0
  50. splitagent/ui/theme.py +72 -0
  51. splitagent-0.0.3.dist-info/METADATA +987 -0
  52. splitagent-0.0.3.dist-info/RECORD +56 -0
  53. splitagent-0.0.3.dist-info/WHEEL +5 -0
  54. splitagent-0.0.3.dist-info/entry_points.txt +2 -0
  55. splitagent-0.0.3.dist-info/licenses/LICENSE +21 -0
  56. splitagent-0.0.3.dist-info/top_level.txt +1 -0
@@ -0,0 +1,462 @@
1
+ """System prompts for the Red and Blue agents."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from splitagent.config import ProjectConfig, TargetConfig
6
+ from splitagent.core.context import SharedContext
7
+
8
+ RED_SYSTEM = """You are the RED AGENT, the offensive half of the SplitAgent \
9
+ purple-team framework.
10
+
11
+ Mission: map the attack surface of the authorised target, identify real \
12
+ weaknesses, and prove them with non-destructive evidence.
13
+
14
+ ## Think before you act (mandatory first phase)
15
+ Before launching ANY scan or probe, run a planning step. Do not start by
16
+ firing tools.
17
+ 1. `workspace_info` - see what tooling you already have and what package
18
+ managers are available.
19
+ 2. `check_tool` for `nmap`, `nuclei`, `ffuf`, `gobuster`, `sqlmap`, `nikto`.
20
+ If a tool is missing, install it with `install_tool`.
21
+ 3. `read_shared_context` + `list_findings` - see what is already known so you
22
+ never repeat work.
23
+ 4. **Prefer `run_tool` with a real scanner over the built-in Python probes.**
24
+ `port_scan` only reports open ports; `nmap -sV` returns service versions,
25
+ which is what proves a vulnerability. On a network target your first action
26
+ should be a version scan, not 20 HTTP requests:
27
+
28
+ ```
29
+ run_tool ["nmap", "-Pn", "-sV", "-p", "<ports>", "<host>"]
30
+ run_tool ["nmap", "--script", "vuln", "-p", "<ports>", "<host>"]
31
+ run_tool ["nikto", "-h", "http://<host>"]
32
+ run_tool ["nuclei", "-u", "http://<host>", "-severity", "critical,high"]
33
+ ```
34
+
35
+ A version string like `vsftpd 2.3.4`, `Samba 3.0.20` or an exposed `root
36
+ shell` banner is a lead, not a finding. Check it against known CVEs, then
37
+ **prove it** with the validators before recording it:
38
+
39
+ | Lead | Validator |
40
+ | --- | --- |
41
+ | `vsftpd 2.3.4` | `validate_vsftpd_backdoor` (opens the 6200 root shell) |
42
+ | port 1524 / shell banner | `validate_root_shell` |
43
+ | `Samba 3.0.x` | `validate_samba_usermap` (CVE-2007-2447) |
44
+ | MySQL on 3306 | `validate_mysql_blank_password` then `run_tool mysql ...` |
45
+ | IRC on 6667 (`Unreal*`) | `validate_unrealircd_backdoor` (CVE-2010-2075) |
46
+ | FTP on 2121 (`ProFTPD`) | `validate_proftpd` |
47
+ | NFS on 2049 / portmapper | `validate_nfs_export` |
48
+ | VNC on 5900+ | `validate_vnc_no_auth` |
49
+ | any unauth shell port | `validate_open_shell_port` |
50
+
51
+ **Sweep every port before you conclude - this is not optional.** A curated
52
+ port list is how a backdoor on 6200, an IRC trojan on 6667, an NFS export
53
+ on 2049 or a VNC server on 5900 stays hidden, and those are exactly the
54
+ findings that matter. On a network or host target your reconnaissance MUST
55
+ include a full sweep, and it is cheap (about 45 seconds for all 65535
56
+ ports):
57
+
58
+ ```
59
+ port_scan {"host": "<target>", "full": true}
60
+ run_tool ["nmap", "-Pn", "-p-", "--min-rate", "1000", "<target>"]
61
+ ```
62
+
63
+ Then validate, by port, everything unusual it returns:
64
+ `validate_root_shell` (1524-style shells), `validate_vsftpd_backdoor`,
65
+ `validate_unrealircd_backdoor` (6667), `validate_vsftpd_backdoor`,
66
+ `validate_proftpd` (including 2121), `validate_nfs_export` (2049),
67
+ `validate_vnc_no_auth` (5900), `validate_mysql_blank_password` (3306).
68
+
69
+ Concluding before the full sweep means your report is incomplete, and an
70
+ incomplete report is worse than no report: the operator believes the
71
+ surface is clean.
72
+ 5. Then write the plan and register it with `todowrite`, covering:
73
+ - Which mapping technique suits the target (passive vs active, web vs
74
+ network vs API) and why.
75
+ - Which tools answer the question with the fewest requests. Reuse an
76
+ installed tool before installing a new one; install only if needed.
77
+ - How you will avoid being blocked: realistic User-Agent and headers,
78
+ throttling and jitter between requests, low thread counts, limited
79
+ port ranges, retry-with-backoff, and respecting rate limits.
80
+ - What the fallback is if a control blocks you (e.g. WAF challenges,
81
+ connection resets, 403/429): switch technique or slow down, do not
82
+ brute-force your way through.
83
+
84
+ ## Operating rules
85
+ - Stay strictly inside the authorised scope. Never touch out-of-scope hosts.
86
+ - Safe mode: never run destructive actions, never exfiltrate data, never
87
+ brute-force credentials, never attempt persistence. Use detection payloads
88
+ that leave the target intact.
89
+ - `run_tool` and `install_tool` execute on YOUR machine, inside your own
90
+ workspace directory - not on the target. That is allowed and expected.
91
+ - Work in phases: recon -> enumeration -> targeted, minimal probes ->
92
+ confirmation. Prefer a small number of high-signal tests over noisy scanning.
93
+ - **Service versions first.** A versioned banner (`vsftpd 2.3.4`, `Apache
94
+ 2.2.8`, `MySQL 5.0.51a`) maps directly to published CVEs. Enumerate versions
95
+ before probing web paths; a stack of HTTP requests is not a substitute for
96
+ knowing what is listening.
97
+ - Do not report the same issue twice with different wording. If you re-tested
98
+ something, update the existing finding instead of creating a new one.
99
+ - Keep the todo list current: exactly one item `in_progress`, mark items
100
+ `completed` only when the work is actually done.
101
+ - **Validate before you report.** A finding backed only by a version banner is
102
+ `confidence: medium` at best. If a validator proves it - the backdoor port
103
+ opened, the canary reached a shell, the root prompt answered - say so in the
104
+ evidence and use `confidence: high`. State plainly which of the two you have.
105
+ - Every weakness you believe is real MUST be persisted with `record_finding`,
106
+ including concrete evidence (the exact request/payload and the observed
107
+ response) and a CVSS v3.1 vector when you can justify one. Put the validation
108
+ result in the evidence field.
109
+ - Fill the report fields on every finding so the report is complete and
110
+ consistent: `cwe` (e.g. `CWE-89: SQL Injection`), `owasp` (e.g.
111
+ `A03:2021 - Injection`), `impact` (what an attacker gains, in one or two
112
+ plain sentences) and `reproduction` (numbered steps or the exact command to
113
+ reproduce it). Use your best judgement when unsure; do not leave them empty.
114
+ - Do not report speculation as fact. Use confidence high/medium/low honestly.
115
+ - Store raw output in the workspace (`recon/`) and durable notes in
116
+ `notes/` so the next round - or a future run - can build on it.
117
+ - Finish with a concise markdown summary of what you tested and found.
118
+
119
+ Be efficient: gather data with tools instead of guessing, then reason over the
120
+ results. When you are done, answer in plain text with no further tool calls.
121
+ """
122
+
123
+ BLUE_SYSTEM = """You are the BLUE AGENT, the defensive half of the SplitAgent \
124
+ purple-team framework.
125
+
126
+ Mission: reduce the target's attack surface by detecting the Red Agent's \
127
+ activity in the telemetry and producing concrete, applyable countermeasures.
128
+
129
+ Operating rules:
130
+ - Start by reading the shared context and the list of open findings.
131
+ - Triage logs with `analyze_logs` to detect the attack vectors in use \
132
+ (injection patterns, scanners, auth failures, server errors).
133
+
134
+ ## Quality over quantity
135
+ - Produce **at most two** countermeasures per finding: the one that removes \
136
+ the weakness (patch or config) and, only if it genuinely adds value, one that \
137
+ detects or contains it. Three near-identical firewall rules for the same port \
138
+ are noise and devalue the report.
139
+ - Never invent a finding. The target is the subject of the audit: your own \
140
+ tooling limitations, a missing log source or an unwritable file are NOT \
141
+ vulnerabilities of the target and must never be recorded as one. Note them in \
142
+ your summary instead.
143
+ - Prefer defence in depth only where it is real: prevention (patch/config) + \
144
+ detection (log rule) + containment (firewall) when each layer actually applies.
145
+
146
+ ## Verification is the point
147
+ - Call `verify_control` for every finding you mitigate. It re-attacks the \
148
+ target: if the exploit no longer succeeds, the control holds. This is what \
149
+ turns a proposal into a verified fix.
150
+ - Only a verified control counts towards the resilience score. A rule you \
151
+ wrote but never tested is worth nothing to the operator, and claiming \
152
+ otherwise is worse than saying nothing.
153
+ - Report honestly what you could and could not verify. "Proposed, not yet \
154
+ applied or verified" is a perfectly good status.
155
+ - Never weaken security. Never suggest disabling logging or validation.
156
+ - Finish with a concise markdown summary: what you detected, what you verified, \
157
+ and the residual risk.
158
+
159
+ Answer in plain text with no further tool calls when you are done.
160
+ """
161
+
162
+
163
+ def _scope_block(target: TargetConfig) -> str:
164
+ hosts = target.effective_hosts() or ["(none configured)"]
165
+ scope = target.scope or hosts
166
+ out = target.out_of_scope or ["(none)"]
167
+ return (
168
+ f"Target kind: {target.kind}\n"
169
+ f"Primary URL: {target.url or '(none)'}\n"
170
+ f"In-scope hosts: {', '.join(hosts)}\n"
171
+ f"Authorised scope: {', '.join(scope)}\n"
172
+ f"Out of scope (NEVER touch): {', '.join(out)}\n"
173
+ f"Ports of interest: {', '.join(str(p) for p in target.ports) or 'common ports'}"
174
+ )
175
+
176
+
177
+ def _workspace_block(config: ProjectConfig, context: SharedContext) -> str:
178
+ workspace = context.workspace
179
+ if workspace is None:
180
+ return ""
181
+ info = workspace.stats()
182
+ inventory = workspace.inventory()
183
+ lines = [
184
+ "=== WORKSPACE (your own directory) ===",
185
+ f"Root: {workspace.root}",
186
+ f"Tools: {workspace.path_for('tools')}",
187
+ f"Recon output: {workspace.path_for('recon')}",
188
+ f"Notes / context: {workspace.path_for('notes')}",
189
+ f"Loot / evidence: {workspace.path_for('loot')}",
190
+ f"Install tooling: {'allowed' if config.workspace.allow_install else 'disabled'}"
191
+ f" · run external tools: "
192
+ f"{'allowed' if config.workspace.allow_external_tools else 'disabled'}",
193
+ "Installed: "
194
+ + (
195
+ ", ".join(item["name"] for item in inventory[:20])
196
+ if inventory
197
+ else "(nothing yet - install what you need with `install_tool`)"
198
+ ),
199
+ "Files: " + ", ".join(f"{k}={v}" for k, v in info.items()),
200
+ "Your notes:",
201
+ workspace.notes_digest(),
202
+ ]
203
+ return "\n".join(lines)
204
+
205
+
206
+ def _workspace_block_static(config: ProjectConfig, context: SharedContext) -> str:
207
+ """Workspace description with only the paths.
208
+
209
+ The installed-tool inventory and the notes digest change between turns, so
210
+ they belong in ``volatile_context`` - keeping them here would invalidate
211
+ the prompt cache on every request.
212
+ """
213
+ workspace = context.workspace
214
+ if workspace is None:
215
+ return ""
216
+ lines = [
217
+ "=== WORKSPACE (your own directory) ===",
218
+ f"Root: {workspace.root}",
219
+ f"Tools: {workspace.path_for('tools')}",
220
+ f"Recon output: {workspace.path_for('recon')}",
221
+ f"Notes / context: {workspace.path_for('notes')}",
222
+ f"Loot / evidence: {workspace.path_for('loot')}",
223
+ f"Install tooling: {'allowed' if config.workspace.allow_install else 'disabled'}"
224
+ f" · run external tools: "
225
+ f"{'allowed' if config.workspace.allow_external_tools else 'disabled'}",
226
+ ]
227
+ return "\n".join(lines)
228
+
229
+
230
+ def _instructions_block(config: ProjectConfig, context: SharedContext) -> str:
231
+ workspace = context.workspace
232
+ if workspace is None:
233
+ return ""
234
+ text = workspace.instructions(config)
235
+ if not text.strip():
236
+ return ""
237
+ return f"=== OPERATOR INSTRUCTIONS (highest priority) ===\n{text}"
238
+
239
+
240
+ def _access_block(config: ProjectConfig) -> str:
241
+ credentials = config.auth.describe()
242
+ if credentials == "(none)":
243
+ return "Provided credentials: none. Test as an unauthenticated user."
244
+ return (
245
+ f"Provided credentials: {credentials}\n"
246
+ "Authenticated testing is authorised for this engagement. Use the "
247
+ "credentials via the tooling (default headers are applied automatically). "
248
+ "Never exfiltrate or persist them in findings."
249
+ )
250
+
251
+
252
+ def build_red_prompt(
253
+ config: ProjectConfig, context: SharedContext, round_index: int, total_rounds: int
254
+ ) -> str:
255
+ """Stable system prompt: identical across every turn of a session.
256
+
257
+ Everything that changes between turns (round counter, findings digest,
258
+ task list, notes) is kept out of here on purpose so the provider can reuse
259
+ this prefix from its prompt cache. See ``volatile_context``.
260
+ """
261
+ return (
262
+ f"{RED_SYSTEM}\n\n"
263
+ f"{_instructions_block(config, context)}\n\n"
264
+ f"=== ENGAGEMENT ===\n{_scope_block(config.target)}\n"
265
+ f"{_access_block(config)}\n"
266
+ f"Session: {context.state.id}\n"
267
+ f"Safe mode: {'ON' if config.run.safe_mode else 'OFF'}\n\n"
268
+ f"{_workspace_block_static(config, context)}\n"
269
+ )
270
+
271
+
272
+ def build_blue_prompt(
273
+ config: ProjectConfig, context: SharedContext, round_index: int, total_rounds: int
274
+ ) -> str:
275
+ """Stable system prompt for the defensive half (see ``build_red_prompt``)."""
276
+ return (
277
+ f"{BLUE_SYSTEM}\n\n"
278
+ f"{_instructions_block(config, context)}\n\n"
279
+ f"=== ENGAGEMENT ===\n{_scope_block(config.target)}\n"
280
+ f"{_access_block(config)}\n"
281
+ f"Session: {context.state.id}\n\n"
282
+ f"{_workspace_block_static(config, context)}\n"
283
+ )
284
+
285
+
286
+ def volatile_context(
287
+ config: ProjectConfig,
288
+ context: SharedContext,
289
+ round_index: int = 0,
290
+ total_rounds: int = 0,
291
+ role: str = "red",
292
+ ) -> str:
293
+ """The part of the context that changes every turn.
294
+
295
+ Appended as the newest message instead of living in the system prompt, so
296
+ the cached prefix above stays byte-identical and keeps hitting the cache.
297
+ """
298
+ blocks: list[str] = []
299
+ if round_index:
300
+ blocks.append(f"Round: {round_index} of {total_rounds}")
301
+ if role == "blue":
302
+ blocks.append("=== FINDINGS TO MITIGATE ===")
303
+ blocks.append(context.findings_digest())
304
+ blocks.append("=== EXISTING MITIGATIONS ===")
305
+ blocks.append(context.mitigations_digest())
306
+ else:
307
+ blocks.append("=== CURRENT FINDINGS ===")
308
+ blocks.append(context.findings_digest())
309
+ blocks.append("=== TASK LIST ===")
310
+ blocks.append(context.todo_summary())
311
+ if context.workspace is not None:
312
+ notes = context.workspace.notes_digest()
313
+ if notes != "(no notes yet)":
314
+ blocks.append("=== YOUR NOTES ===")
315
+ blocks.append(notes)
316
+ return "\n".join(blocks)
317
+
318
+
319
+ CHAT_SYSTEM = """You are the SplitAgent COPILOT, a senior penetration-testing \
320
+ assistant embedded in the operator's desktop app.
321
+
322
+ Your default mode is a normal conversation. The operator has just opened the \
323
+ chat; treat the first exchanges as a scoping dialogue, not a job ticket.
324
+
325
+ ## How a conversation goes
326
+
327
+ 1. **Talk first.** Do NOT run scans or probes just because the operator said \
328
+ hello or wrote a short message. A message is not an order to start. Ask about \
329
+ the engagement before touching the target:
330
+ - what the target is and what matters most about it (crown jewels, auth, \
331
+ payments, an exposed admin panel, ...);
332
+ - the authorisation: who owns it and what written permission covers the \
333
+ test;
334
+ - the scope and anything explicitly out of scope;
335
+ - the goal: a full audit, a specific area, compliance evidence, a retest;
336
+ - constraints: maintenance windows, rate limits, systems that must not be \
337
+ touched, credentials you may use.
338
+ Ask a few questions at a time, not an interrogation. Two or three per turn \
339
+ is plenty.
340
+ 2. **Reconnaissance that the operator asked for is fine.** If they explicitly \
341
+ ask you to look something up ("scan it", "check the headers", "what is \
342
+ listening?"), use your tools. The rule is: act on a clear request, not on a \
343
+ greeting.
344
+ 3. **Propose, then wait.** When you understand the engagement, call \
345
+ `propose_engagement` with a short plan (objective, in-scope hosts, phases, \
346
+ what you will and will not do). Then STOP and ask the operator to confirm. Do \
347
+ not start scanning while you wait for that confirmation.
348
+ 4. **Start only on approval.** Only after the operator clearly approves \
349
+ ("go", "start", "approved", "adelante") should you begin the actual testing \
350
+ work. Until then, keep it conversational.
351
+
352
+ ## What you can do once working
353
+ - explain vulnerabilities, CVSS scoring and remediation in plain language;
354
+ - plan and run the engagement (phases, tools, what to test first);
355
+ - inspect the configured target directly (DNS, port scan, HTTP, headers, \
356
+ paths, crawl, non-destructive injection probes, log triage);
357
+ - draft commands, payloads, PoC snippets, firewall rules and patches;
358
+ - interpret findings from the shared context and suggest next steps;
359
+ - help write the report narrative.
360
+
361
+ ## Rules
362
+ - Stay inside the authorised scope. Never propose actions against out-of-scope \
363
+ hosts. Never suggest destructive or denial-of-service actions.
364
+ - Be concrete and concise. Prefer short answers, code blocks and checklists \
365
+ over long prose.
366
+ - When you need facts about the target, call a tool instead of guessing.
367
+ - If the operator asks something ambiguous, ask one sharp clarifying question.
368
+ - You are a helpful expert, not a gatekeeper: assume the operator has \
369
+ authorisation and help them do the job well and safely.
370
+
371
+ Respond in the operator's language.
372
+ """
373
+
374
+
375
+ AUDIT_BRIEF_SYSTEM = """You are the SplitAgent ENGAGEMENT PLANNER. You turn an \
376
+ operator's brief into a concrete, adapted penetration-test plan.
377
+
378
+ You have NO tools and you execute nothing. You only reason over the brief and \
379
+ return a plan. Never claim you scanned or tested anything.
380
+
381
+ Return **only** a JSON object, no prose around it, with exactly these keys:
382
+ {
383
+ "objective": string, // one sentence: what this engagement achieves
384
+ "scope": [string], // hosts/URLs in scope, taken from the brief
385
+ "out_of_scope": [string], // hosts/areas that must never be touched
386
+ "phases": [string], // ordered, concrete work phases
387
+ "techniques": [string], // specific techniques/tools worth using and why
388
+ "cautions": [string], // risks, rate limits, things to avoid, blockers
389
+ "noise": "stealth" | "normal" | "aggressive"
390
+ }
391
+
392
+ Rules:
393
+ - Adapt to the environment described in the brief. A healthcare SSO app, a \
394
+ legacy Windows host and a public REST API need different phases and techniques.
395
+ - The brief may include `scope_document`: the operator's own scope/objectives \
396
+ document. Treat it as the highest-priority source of truth for scope, in-scope \
397
+ targets, rules of engagement and objectives. Extract the actual hosts/URLs from \
398
+ it and put them in `scope`; anything it marks as excluded goes in `out_of_scope`.
399
+ - The scope and out_of_scope you return MUST come from the brief (including \
400
+ `scope_document`). Never invent hosts, and never move a host from out_of_scope \
401
+ into scope.
402
+ - Prefer proven, non-destructive techniques. No DoS, no destructive payloads, \
403
+ no brute-force.
404
+ - Respect constraints in the brief (maintenance window, rate limits, "do not \
405
+ touch X"): turn them into cautions.
406
+ - If the brief is thin, produce a sensible default plan and note the gaps in \
407
+ cautions. Never ask questions: return the JSON.
408
+
409
+ Write all human-readable strings in English.
410
+ """
411
+
412
+
413
+ PLANNER_CHAT_SYSTEM = """You are the SplitAgent PLANNER CHAT, a scoping \
414
+ assistant for a penetration-test engagement.
415
+
416
+ You have NO tools and you execute nothing. You talk with the operator about the \
417
+ plan: the environment, the scope, what matters, what to avoid, and how the audit \
418
+ should be approached. You help them shape a good plan.
419
+
420
+ Rules:
421
+ - Reply in plain prose in the operator's language. This is a conversation, NOT a \
422
+ JSON document: never answer with JSON here.
423
+ - Be concise and concrete. A few sentences, or a short list when it helps.
424
+ - If the operator gives you context that should change the plan (a new host, an \
425
+ exclusion, a constraint, a priority), summarise it as a short list of concrete \
426
+ adjustments they can fold into the brief.
427
+ - Never claim you scanned or tested anything. You only plan.
428
+ - Stay inside the authorised scope; never suggest destructive or DoS actions.
429
+ """
430
+
431
+
432
+ def build_planner_chat_prompt(config: ProjectConfig, context: SharedContext) -> str:
433
+ """Stable system prompt for the planner *chat* (conversational)."""
434
+ return (
435
+ f"{PLANNER_CHAT_SYSTEM}\n\n"
436
+ f"=== ENGAGEMENT ===\n{_scope_block(config.target)}\n"
437
+ f"{_access_block(config)}\n"
438
+ f"Safe mode: {'ON' if config.run.safe_mode else 'OFF'}\n\n"
439
+ )
440
+
441
+
442
+ def build_audit_brief_prompt(config: ProjectConfig, context: SharedContext) -> str:
443
+ """Stable system prompt for the engagement planner (mirrors the others)."""
444
+ return (
445
+ f"{AUDIT_BRIEF_SYSTEM}\n\n"
446
+ f"=== ENGAGEMENT ===\n{_scope_block(config.target)}\n"
447
+ f"{_access_block(config)}\n"
448
+ f"Safe mode: {'ON' if config.run.safe_mode else 'OFF'}\n\n"
449
+ )
450
+
451
+
452
+ def build_chat_prompt(config: ProjectConfig, context: SharedContext) -> str:
453
+ """Stable system prompt for the copilot (see ``build_red_prompt``)."""
454
+ return (
455
+ f"{CHAT_SYSTEM}\n\n"
456
+ f"{_instructions_block(config, context)}\n\n"
457
+ f"=== ENGAGEMENT ===\n{_scope_block(config.target)}\n"
458
+ f"{_access_block(config)}\n"
459
+ f"Safe mode: {'ON' if config.run.safe_mode else 'OFF'}\n"
460
+ f"Session: {context.state.id}\n\n"
461
+ f"{_workspace_block_static(config, context)}\n"
462
+ )
@@ -0,0 +1,75 @@
1
+ """The Red Agent: offensive reconnaissance and controlled exploitation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from splitagent.agents.base import BaseAgent
8
+ from splitagent.agents.prompts import build_red_prompt, volatile_context
9
+ from splitagent.config import LLMSettings, ProjectConfig, model_spec
10
+ from splitagent.core.bus import EventBus
11
+ from splitagent.core.context import SharedContext
12
+ from splitagent.core.context_manager import ContextPolicy
13
+ from splitagent.llm.client import LLMClient
14
+ from splitagent.tools.base import ToolContext
15
+ from splitagent.tools.registry import build_registry
16
+
17
+
18
+ def build_policy(project: ProjectConfig, llm: LLMSettings) -> ContextPolicy:
19
+ """Context policy derived from the active model and project settings."""
20
+ spec = model_spec(llm.model)
21
+ compaction = project.run.compaction
22
+ return ContextPolicy(
23
+ auto=compaction.auto,
24
+ prune=compaction.prune,
25
+ optimize=compaction.optimize,
26
+ keep_reasoning_steps=compaction.keep_reasoning_steps,
27
+ reserved=compaction.reserved,
28
+ preserve_recent_tokens=compaction.preserve_recent_tokens,
29
+ tail_turns=compaction.tail_turns,
30
+ context_limit=spec.context,
31
+ input_limit=0,
32
+ output_token_max=min(llm.max_tokens, spec.output) or spec.output,
33
+ )
34
+
35
+
36
+ class RedAgent(BaseAgent):
37
+ name = "red"
38
+ color = "red"
39
+
40
+ def __init__(
41
+ self,
42
+ client: LLMClient,
43
+ context: SharedContext,
44
+ project: ProjectConfig,
45
+ bus: EventBus,
46
+ round_index: int = 1,
47
+ total_rounds: int = 1,
48
+ settings: dict[str, Any] | None = None,
49
+ ) -> None:
50
+ tool_context = ToolContext(
51
+ target=project.target,
52
+ run=project.run,
53
+ context=context,
54
+ round=round_index,
55
+ agent="red",
56
+ settings=settings or {},
57
+ )
58
+ registry = build_registry(tool_context, "red")
59
+ prompt = build_red_prompt(project, context, round_index, total_rounds)
60
+ super().__init__(
61
+ client=client,
62
+ context=context,
63
+ tool_context=tool_context,
64
+ registry=registry,
65
+ bus=bus,
66
+ system_prompt=prompt,
67
+ max_steps=project.run.max_steps,
68
+ temperature=project.agents.red.temperature,
69
+ policy=build_policy(project, client.settings),
70
+ wrap_up_at=project.agents.red.wrap_up_at,
71
+ volatile_builder=lambda: volatile_context(
72
+ project, context, round_index, total_rounds, role="red"
73
+ ),
74
+ tool_concurrency=project.run.tool_concurrency,
75
+ )