splitagent 0.0.3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. splitagent/__init__.py +8 -0
  2. splitagent/__main__.py +6 -0
  3. splitagent/agents/__init__.py +10 -0
  4. splitagent/agents/base.py +477 -0
  5. splitagent/agents/blue.py +57 -0
  6. splitagent/agents/chat.py +60 -0
  7. splitagent/agents/prompts.py +462 -0
  8. splitagent/agents/red.py +75 -0
  9. splitagent/cli.py +701 -0
  10. splitagent/config.py +697 -0
  11. splitagent/core/__init__.py +19 -0
  12. splitagent/core/bus.py +62 -0
  13. splitagent/core/context.py +587 -0
  14. splitagent/core/context_manager.py +381 -0
  15. splitagent/core/engine.py +424 -0
  16. splitagent/core/models.py +310 -0
  17. splitagent/core/proc.py +73 -0
  18. splitagent/core/sandbox.py +184 -0
  19. splitagent/core/toolbox.py +520 -0
  20. splitagent/core/workspace.py +420 -0
  21. splitagent/desktop/__init__.py +7 -0
  22. splitagent/desktop/api.py +525 -0
  23. splitagent/desktop/app.py +1131 -0
  24. splitagent/desktop/web/app.js +3067 -0
  25. splitagent/desktop/web/assets/Inter.ttf +0 -0
  26. splitagent/desktop/web/assets/JetBrainsMonoNerdFontMono-Regular.woff2 +0 -0
  27. splitagent/desktop/web/index.html +760 -0
  28. splitagent/desktop/web/styles.css +1612 -0
  29. splitagent/errors.py +27 -0
  30. splitagent/llm/__init__.py +8 -0
  31. splitagent/llm/client.py +488 -0
  32. splitagent/llm/types.py +172 -0
  33. splitagent/report/__init__.py +9 -0
  34. splitagent/report/cvss.py +93 -0
  35. splitagent/report/generator.py +733 -0
  36. splitagent/tools/__init__.py +8 -0
  37. splitagent/tools/base.py +135 -0
  38. splitagent/tools/defense.py +475 -0
  39. splitagent/tools/exploit.py +318 -0
  40. splitagent/tools/http_pool.py +109 -0
  41. splitagent/tools/knowledge.py +376 -0
  42. splitagent/tools/recon.py +182 -0
  43. splitagent/tools/registry.py +62 -0
  44. splitagent/tools/validate.py +908 -0
  45. splitagent/tools/web.py +386 -0
  46. splitagent/tools/workspace_tools.py +411 -0
  47. splitagent/ui/__init__.py +5 -0
  48. splitagent/ui/app.py +389 -0
  49. splitagent/ui/stream.py +234 -0
  50. splitagent/ui/theme.py +72 -0
  51. splitagent-0.0.3.dist-info/METADATA +987 -0
  52. splitagent-0.0.3.dist-info/RECORD +56 -0
  53. splitagent-0.0.3.dist-info/WHEEL +5 -0
  54. splitagent-0.0.3.dist-info/entry_points.txt +2 -0
  55. splitagent-0.0.3.dist-info/licenses/LICENSE +21 -0
  56. splitagent-0.0.3.dist-info/top_level.txt +1 -0
@@ -0,0 +1,381 @@
1
+ """Context window management: usable budget, pruning and compaction.
2
+
3
+ Ported from OpenCode's session model (``session/overflow.ts`` and
4
+ ``session/compaction.ts``):
5
+
6
+ * **usable()** - how many prompt tokens fit, reserving room for the reply.
7
+ * **is_overflow()** - whether the last turn exceeded that budget.
8
+ * **prune()** - walk backwards protecting the most recent
9
+ ``PRUNE_PROTECT`` tokens of tool output, then blank the output of older
10
+ tool results, which frees context without losing conversation.
11
+ * **compact()** - keep a tail of recent turns within a token budget and
12
+ summarise everything before it into a single checkpoint message.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from dataclasses import dataclass
18
+ from typing import Any
19
+
20
+ from splitagent.llm.types import ChatMessage, estimate_tokens
21
+
22
+ # --- OpenCode constants ------------------------------------------------- #
23
+ COMPACTION_BUFFER = 20_000
24
+ PRUNE_MINIMUM = 20_000
25
+ PRUNE_PROTECT = 40_000
26
+ MIN_PRESERVE_RECENT_TOKENS = 2_000
27
+ MAX_PRESERVE_RECENT_TOKENS = 15_000
28
+ TOOL_OUTPUT_MAX_CHARS = 2_000
29
+ PRUNE_PROTECTED_TOOLS = ("skill",)
30
+
31
+ # --- Efficiency defaults (ported from OpenCode's per-step handling) -------- #
32
+ # Reasoning is thought, not context: OpenCode drops the reasoning of every
33
+ # step that already produced a tool result and keeps it only on the final
34
+ # answer. Replaying it every turn is the single biggest source of waste.
35
+ KEEP_REASONING_STEPS = 1
36
+ # Results from these tools are state snapshots; only the newest one matters.
37
+ STATEFUL_TOOLS = (
38
+ "workspace_info",
39
+ "read_shared_context",
40
+ "list_findings",
41
+ "todowrite",
42
+ "check_tool",
43
+ )
44
+ # Output of these tools is already bounded by the tool itself.
45
+ COMPACT_TOOLS = ("todowrite", "record_finding", "record_mitigation")
46
+
47
+
48
+ @dataclass
49
+ class ContextPolicy:
50
+ """Tunable knobs, mapped 1:1 onto OpenCode's ``compaction`` config."""
51
+
52
+ auto: bool = True
53
+ prune: bool = True
54
+ reserved: int | None = None
55
+ preserve_recent_tokens: int | None = None
56
+ tail_turns: int | None = None
57
+ context_limit: int = 128_000
58
+ input_limit: int = 0
59
+ output_token_max: int = 32_000
60
+ # Aggressive per-request reductions (default on).
61
+ optimize: bool = True
62
+ keep_reasoning_steps: int = KEEP_REASONING_STEPS
63
+
64
+
65
+ @dataclass
66
+ class CompactionResult:
67
+ compacted: bool = False
68
+ removed: int = 0
69
+ preserved: int = 0
70
+ summary: str = ""
71
+ reason: str = ""
72
+
73
+
74
+ @dataclass
75
+ class PruneResult:
76
+ pruned: int = 0
77
+ protected: int = 0
78
+ parts: int = 0
79
+
80
+
81
+ def max_output_tokens(policy: ContextPolicy) -> int:
82
+ return max(0, policy.output_token_max)
83
+
84
+
85
+ def usable(policy: ContextPolicy) -> int:
86
+ """Prompt budget: context window minus the room reserved for the reply."""
87
+ if policy.context_limit <= 0:
88
+ return 0
89
+ if policy.input_limit:
90
+ reserved = policy.reserved
91
+ if reserved is None:
92
+ reserved = min(COMPACTION_BUFFER, max_output_tokens(policy))
93
+ return max(0, policy.input_limit - reserved)
94
+ # Without an explicit reserved budget, leave room for the longest reply
95
+ # the model can produce (OpenCode caps that reservation with a buffer).
96
+ reserved = policy.reserved
97
+ if reserved is None:
98
+ reserved = max_output_tokens(policy)
99
+ return max(0, policy.context_limit - reserved)
100
+
101
+
102
+ def is_overflow(policy: ContextPolicy, usage: dict[str, Any]) -> bool:
103
+ """True when the last turn consumed more than the usable budget."""
104
+ if not policy.auto or policy.context_limit <= 0:
105
+ return False
106
+ total = usage.get("total_tokens")
107
+ if not total:
108
+ total = int(usage.get("prompt_tokens") or 0) + int(usage.get("completion_tokens") or 0)
109
+ return int(total) >= usable(policy)
110
+
111
+
112
+ def messages_tokens(messages: list[ChatMessage]) -> int:
113
+ return sum(message.token_estimate() for message in messages)
114
+
115
+
116
+ # --------------------------------------------------------------------------- #
117
+ # Per-request projection: what actually gets sent to the model
118
+ # --------------------------------------------------------------------------- #
119
+ def strip_reasoning(messages: list[ChatMessage], keep_last: int = KEEP_REASONING_STEPS) -> int:
120
+ """Drop reasoning from all but the most recent assistant turns.
121
+
122
+ Reasoning is scaffolding: once a step has produced tool calls the model has
123
+ already acted on it, so replaying it wastes the whole context window.
124
+ Returns the number of tokens saved.
125
+ """
126
+ saved = 0
127
+ indices = [
128
+ index
129
+ for index, message in enumerate(messages)
130
+ if message.role == "assistant" and message.reasoning
131
+ ]
132
+ keep = set(indices[-keep_last:]) if keep_last > 0 else set()
133
+ for index in indices:
134
+ if index in keep:
135
+ continue
136
+ saved += estimate_tokens(messages[index].reasoning)
137
+ messages[index].reasoning = ""
138
+ return saved
139
+
140
+
141
+ def supersede_stateful_results(messages: list[ChatMessage]) -> int:
142
+ """Blank older snapshots from tools whose newest result supersedes them.
143
+
144
+ ``workspace_info``, ``read_shared_context`` or ``list_findings`` return the
145
+ full current state; only the latest call is meaningful. Earlier copies are
146
+ replaced by a marker so the conversation stays coherent.
147
+ """
148
+ latest: dict[str, int] = {}
149
+ for index, message in enumerate(messages):
150
+ if message.role == "tool" and message.name in STATEFUL_TOOLS:
151
+ latest[message.name] = index
152
+ saved = 0
153
+ for index, message in enumerate(messages):
154
+ if message.role != "tool" or message.name not in STATEFUL_TOOLS:
155
+ continue
156
+ if message.compacted or latest.get(message.name) == index:
157
+ continue
158
+ saved += estimate_tokens(message.content)
159
+ message.content = f"[superseded by a newer {message.name} call]"
160
+ message.compacted = True
161
+ return saved
162
+
163
+
164
+ def dedupe_tool_calls(messages: list[ChatMessage]) -> int:
165
+ """Collapse consecutive identical tool calls from separate steps."""
166
+ saved = 0
167
+ seen: set[tuple[str, str]] = set()
168
+ for message in messages:
169
+ for call in message.tool_calls:
170
+ signature = (call.name, call.arguments or "{}")
171
+ if signature in seen:
172
+ saved += estimate_tokens(call.arguments) + estimate_tokens(call.name)
173
+ else:
174
+ seen.add(signature)
175
+ return saved
176
+
177
+
178
+ def optimize(
179
+ messages: list[ChatMessage], policy: ContextPolicy
180
+ ) -> tuple[list[ChatMessage], dict[str, int]]:
181
+ """Apply the cheap, lossless-ish reductions before every request.
182
+
183
+ Order matters: reasoning first (largest win, no information loss for
184
+ decisions already taken), then superseded snapshots, then pruning of old
185
+ tool output. Never touches the system prompt or the latest turn.
186
+ """
187
+ if not policy.optimize:
188
+ return messages, {"reasoning": 0, "stateful": 0, "pruned": 0}
189
+ stats = {"reasoning": 0, "stateful": 0, "pruned": 0}
190
+ stats["reasoning"] = strip_reasoning(messages, policy.keep_reasoning_steps)
191
+ stats["stateful"] = supersede_stateful_results(messages)
192
+ prune_result = prune(messages, policy)
193
+ stats["pruned"] = prune_result.pruned
194
+ return messages, stats
195
+
196
+
197
+ def protect_budget(policy: ContextPolicy) -> int:
198
+ if policy.preserve_recent_tokens is not None:
199
+ return policy.preserve_recent_tokens
200
+ return min(
201
+ MAX_PRESERVE_RECENT_TOKENS,
202
+ max(MIN_PRESERVE_RECENT_TOKENS, int(usable(policy) * 0.25)),
203
+ )
204
+
205
+
206
+ def prune(messages: list[ChatMessage], policy: ContextPolicy) -> PruneResult:
207
+ """Erase the output of old tool results, protecting recent ones.
208
+
209
+ Walks backwards accumulating tool-output tokens; once more than
210
+ ``PRUNE_PROTECT`` tokens have been seen, every older tool result is
211
+ blanked (it stays in the transcript for auditing).
212
+ """
213
+ if not policy.prune:
214
+ return PruneResult()
215
+ pruned = 0
216
+ protected: list[tuple[ChatMessage, int]] = []
217
+ cleared: list[tuple[ChatMessage, str, int]] = []
218
+ for message in reversed(messages):
219
+ if message.role != "tool" or message.compacted:
220
+ continue
221
+ if message.pinned:
222
+ continue
223
+ size = estimate_tokens(message.content)
224
+ if not protected:
225
+ # Always keep the most recent tool result, exactly like OpenCode.
226
+ protected.append((message, size))
227
+ continue
228
+ protected_tokens = sum(item[1] for item in protected)
229
+ if protected_tokens + size <= PRUNE_PROTECT:
230
+ protected.append((message, size))
231
+ continue
232
+ cleared.append((message, message.content, size))
233
+ pruned += size
234
+ protected_tokens = sum(item[1] for item in protected)
235
+ if pruned < PRUNE_MINIMUM:
236
+ # Not worth mutating the transcript: leave every message untouched.
237
+ return PruneResult(pruned=0, protected=protected_tokens, parts=0)
238
+ for message, _, _ in cleared:
239
+ message.compacted = True
240
+ message.content = "[Old tool result content cleared to save context]"
241
+ return PruneResult(pruned=pruned, protected=protected_tokens, parts=len(cleared))
242
+
243
+
244
+ def _turn_starts(messages: list[ChatMessage]) -> list[int]:
245
+ """Indexes of user turns, used as turn boundaries."""
246
+ return [i for i, message in enumerate(messages) if message.role == "user"]
247
+
248
+
249
+ def _summarize(messages: list[ChatMessage], max_chars: int = 12_000) -> str:
250
+ """Build a terse, structured checkpoint from the head of the transcript."""
251
+ lines: list[str] = []
252
+ for message in messages:
253
+ if message.role == "tool" or message.compacted:
254
+ continue
255
+ if message.role == "assistant" and message.tool_calls:
256
+ calls = ", ".join(call.name for call in message.tool_calls)
257
+ if calls:
258
+ lines.append(f"- red/blue tool calls: {calls}")
259
+ text = (message.content or "").strip()
260
+ if not text:
261
+ continue
262
+ label = {
263
+ "system": "system",
264
+ "user": "task",
265
+ "assistant": "agent",
266
+ }.get(message.role, message.role)
267
+ snippet = " ".join(text.split())[:400]
268
+ lines.append(f"- {label}: {snippet}")
269
+ summary = "\n".join(lines)
270
+ if len(summary) > max_chars:
271
+ summary = summary[:max_chars] + "\n- ...(older history omitted)"
272
+ return summary
273
+
274
+
275
+ def compact(
276
+ messages: list[ChatMessage], policy: ContextPolicy
277
+ ) -> tuple[list[ChatMessage], CompactionResult]:
278
+ """Return a compacted copy of ``messages``.
279
+
280
+ The head is replaced by a single checkpoint message; the tail (recent
281
+ turns) is kept verbatim within ``protect_budget``. The system message is
282
+ always preserved.
283
+ """
284
+ if not policy.auto:
285
+ return messages, CompactionResult(reason="auto compaction disabled")
286
+
287
+ system = [m for m in messages if m.role == "system"]
288
+ body = [m for m in messages if m.role != "system"]
289
+ if len(body) <= 2:
290
+ return messages, CompactionResult(reason="nothing to compact")
291
+
292
+ budget = protect_budget(policy)
293
+ starts = _turn_starts(body)
294
+ if not starts:
295
+ return messages, CompactionResult(reason="no user turns")
296
+
297
+ # Anchor the tail at a user-turn boundary within the budget.
298
+ tail_start = starts[-1]
299
+ total = 0
300
+ for index in range(len(starts) - 1, -1, -1):
301
+ start = starts[index]
302
+ end = starts[index + 1] if index + 1 < len(starts) else len(body)
303
+ size = messages_tokens(body[start:end])
304
+ if total + size <= budget or index == len(starts) - 1:
305
+ total += size
306
+ tail_start = start
307
+ continue
308
+ break
309
+
310
+ head = body[:tail_start]
311
+ if not head:
312
+ return messages, CompactionResult(reason="tail already fits the budget")
313
+
314
+ summary = _summarize(head)
315
+ checkpoint = ChatMessage(
316
+ role="user",
317
+ pinned=True,
318
+ content=(
319
+ "[Context checkpoint] Earlier work in this engagement was summarised "
320
+ "to stay within the model context window. Details omitted here are "
321
+ "still available in the full transcript and the session file.\n\n"
322
+ f"{summary}"
323
+ ),
324
+ )
325
+ removed = messages_tokens(head)
326
+ compacted_messages = [*system, checkpoint, *body[tail_start:]]
327
+ return compacted_messages, CompactionResult(
328
+ compacted=True,
329
+ removed=removed,
330
+ preserved=messages_tokens(body[tail_start:]),
331
+ summary=summary,
332
+ reason="context exceeded the usable budget",
333
+ )
334
+
335
+
336
+ def prune_tool_output(text: str, max_chars: int = TOOL_OUTPUT_MAX_CHARS) -> str:
337
+ """Bound a tool result before storing it in the transcript."""
338
+ if len(text) <= max_chars:
339
+ return text
340
+ return f"{text[:max_chars]}\n[tool output truncated for context: {len(text) - max_chars} chars omitted]"
341
+
342
+
343
+ def context_report(
344
+ messages: list[ChatMessage], policy: ContextPolicy, usage: dict[str, Any] | None = None
345
+ ) -> dict[str, Any]:
346
+ budget = usable(policy)
347
+ used = messages_tokens(messages)
348
+ # Effective size after the per-request projection, which is what the
349
+ # provider actually bills for.
350
+ projected = used
351
+ if policy.optimize:
352
+ scratch = [
353
+ ChatMessage(
354
+ role=m.role,
355
+ content=m.content,
356
+ reasoning=m.reasoning,
357
+ tool_calls=m.tool_calls,
358
+ name=m.name,
359
+ compacted=m.compacted,
360
+ pinned=m.pinned,
361
+ )
362
+ for m in messages
363
+ ]
364
+ strip_reasoning(scratch, policy.keep_reasoning_steps)
365
+ supersede_stateful_results(scratch)
366
+ projected = messages_tokens(scratch)
367
+ return {
368
+ "context_limit": policy.context_limit,
369
+ "usable": budget,
370
+ "estimated": used,
371
+ "projected": projected,
372
+ "saved": max(0, used - projected),
373
+ "saving_percent": round((1 - projected / used) * 100, 1) if used else 0.0,
374
+ "percent": round((projected / budget) * 100, 1) if budget else 0.0,
375
+ "overflow": is_overflow(policy, usage or {}),
376
+ "prune": policy.prune,
377
+ "auto_compact": policy.auto,
378
+ "optimize": policy.optimize,
379
+ "protect_budget": protect_budget(policy),
380
+ "messages": len(messages),
381
+ }