splitagent 0.0.3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- splitagent/__init__.py +8 -0
- splitagent/__main__.py +6 -0
- splitagent/agents/__init__.py +10 -0
- splitagent/agents/base.py +477 -0
- splitagent/agents/blue.py +57 -0
- splitagent/agents/chat.py +60 -0
- splitagent/agents/prompts.py +462 -0
- splitagent/agents/red.py +75 -0
- splitagent/cli.py +701 -0
- splitagent/config.py +697 -0
- splitagent/core/__init__.py +19 -0
- splitagent/core/bus.py +62 -0
- splitagent/core/context.py +587 -0
- splitagent/core/context_manager.py +381 -0
- splitagent/core/engine.py +424 -0
- splitagent/core/models.py +310 -0
- splitagent/core/proc.py +73 -0
- splitagent/core/sandbox.py +184 -0
- splitagent/core/toolbox.py +520 -0
- splitagent/core/workspace.py +420 -0
- splitagent/desktop/__init__.py +7 -0
- splitagent/desktop/api.py +525 -0
- splitagent/desktop/app.py +1131 -0
- splitagent/desktop/web/app.js +3067 -0
- splitagent/desktop/web/assets/Inter.ttf +0 -0
- splitagent/desktop/web/assets/JetBrainsMonoNerdFontMono-Regular.woff2 +0 -0
- splitagent/desktop/web/index.html +760 -0
- splitagent/desktop/web/styles.css +1612 -0
- splitagent/errors.py +27 -0
- splitagent/llm/__init__.py +8 -0
- splitagent/llm/client.py +488 -0
- splitagent/llm/types.py +172 -0
- splitagent/report/__init__.py +9 -0
- splitagent/report/cvss.py +93 -0
- splitagent/report/generator.py +733 -0
- splitagent/tools/__init__.py +8 -0
- splitagent/tools/base.py +135 -0
- splitagent/tools/defense.py +475 -0
- splitagent/tools/exploit.py +318 -0
- splitagent/tools/http_pool.py +109 -0
- splitagent/tools/knowledge.py +376 -0
- splitagent/tools/recon.py +182 -0
- splitagent/tools/registry.py +62 -0
- splitagent/tools/validate.py +908 -0
- splitagent/tools/web.py +386 -0
- splitagent/tools/workspace_tools.py +411 -0
- splitagent/ui/__init__.py +5 -0
- splitagent/ui/app.py +389 -0
- splitagent/ui/stream.py +234 -0
- splitagent/ui/theme.py +72 -0
- splitagent-0.0.3.dist-info/METADATA +987 -0
- splitagent-0.0.3.dist-info/RECORD +56 -0
- splitagent-0.0.3.dist-info/WHEEL +5 -0
- splitagent-0.0.3.dist-info/entry_points.txt +2 -0
- splitagent-0.0.3.dist-info/licenses/LICENSE +21 -0
- splitagent-0.0.3.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,381 @@
|
|
|
1
|
+
"""Context window management: usable budget, pruning and compaction.
|
|
2
|
+
|
|
3
|
+
Ported from OpenCode's session model (``session/overflow.ts`` and
|
|
4
|
+
``session/compaction.ts``):
|
|
5
|
+
|
|
6
|
+
* **usable()** - how many prompt tokens fit, reserving room for the reply.
|
|
7
|
+
* **is_overflow()** - whether the last turn exceeded that budget.
|
|
8
|
+
* **prune()** - walk backwards protecting the most recent
|
|
9
|
+
``PRUNE_PROTECT`` tokens of tool output, then blank the output of older
|
|
10
|
+
tool results, which frees context without losing conversation.
|
|
11
|
+
* **compact()** - keep a tail of recent turns within a token budget and
|
|
12
|
+
summarise everything before it into a single checkpoint message.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
from splitagent.llm.types import ChatMessage, estimate_tokens
|
|
21
|
+
|
|
22
|
+
# --- OpenCode constants ------------------------------------------------- #
|
|
23
|
+
COMPACTION_BUFFER = 20_000
|
|
24
|
+
PRUNE_MINIMUM = 20_000
|
|
25
|
+
PRUNE_PROTECT = 40_000
|
|
26
|
+
MIN_PRESERVE_RECENT_TOKENS = 2_000
|
|
27
|
+
MAX_PRESERVE_RECENT_TOKENS = 15_000
|
|
28
|
+
TOOL_OUTPUT_MAX_CHARS = 2_000
|
|
29
|
+
PRUNE_PROTECTED_TOOLS = ("skill",)
|
|
30
|
+
|
|
31
|
+
# --- Efficiency defaults (ported from OpenCode's per-step handling) -------- #
|
|
32
|
+
# Reasoning is thought, not context: OpenCode drops the reasoning of every
|
|
33
|
+
# step that already produced a tool result and keeps it only on the final
|
|
34
|
+
# answer. Replaying it every turn is the single biggest source of waste.
|
|
35
|
+
KEEP_REASONING_STEPS = 1
|
|
36
|
+
# Results from these tools are state snapshots; only the newest one matters.
|
|
37
|
+
STATEFUL_TOOLS = (
|
|
38
|
+
"workspace_info",
|
|
39
|
+
"read_shared_context",
|
|
40
|
+
"list_findings",
|
|
41
|
+
"todowrite",
|
|
42
|
+
"check_tool",
|
|
43
|
+
)
|
|
44
|
+
# Output of these tools is already bounded by the tool itself.
|
|
45
|
+
COMPACT_TOOLS = ("todowrite", "record_finding", "record_mitigation")
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass
|
|
49
|
+
class ContextPolicy:
|
|
50
|
+
"""Tunable knobs, mapped 1:1 onto OpenCode's ``compaction`` config."""
|
|
51
|
+
|
|
52
|
+
auto: bool = True
|
|
53
|
+
prune: bool = True
|
|
54
|
+
reserved: int | None = None
|
|
55
|
+
preserve_recent_tokens: int | None = None
|
|
56
|
+
tail_turns: int | None = None
|
|
57
|
+
context_limit: int = 128_000
|
|
58
|
+
input_limit: int = 0
|
|
59
|
+
output_token_max: int = 32_000
|
|
60
|
+
# Aggressive per-request reductions (default on).
|
|
61
|
+
optimize: bool = True
|
|
62
|
+
keep_reasoning_steps: int = KEEP_REASONING_STEPS
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
@dataclass
|
|
66
|
+
class CompactionResult:
|
|
67
|
+
compacted: bool = False
|
|
68
|
+
removed: int = 0
|
|
69
|
+
preserved: int = 0
|
|
70
|
+
summary: str = ""
|
|
71
|
+
reason: str = ""
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
@dataclass
|
|
75
|
+
class PruneResult:
|
|
76
|
+
pruned: int = 0
|
|
77
|
+
protected: int = 0
|
|
78
|
+
parts: int = 0
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def max_output_tokens(policy: ContextPolicy) -> int:
|
|
82
|
+
return max(0, policy.output_token_max)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def usable(policy: ContextPolicy) -> int:
|
|
86
|
+
"""Prompt budget: context window minus the room reserved for the reply."""
|
|
87
|
+
if policy.context_limit <= 0:
|
|
88
|
+
return 0
|
|
89
|
+
if policy.input_limit:
|
|
90
|
+
reserved = policy.reserved
|
|
91
|
+
if reserved is None:
|
|
92
|
+
reserved = min(COMPACTION_BUFFER, max_output_tokens(policy))
|
|
93
|
+
return max(0, policy.input_limit - reserved)
|
|
94
|
+
# Without an explicit reserved budget, leave room for the longest reply
|
|
95
|
+
# the model can produce (OpenCode caps that reservation with a buffer).
|
|
96
|
+
reserved = policy.reserved
|
|
97
|
+
if reserved is None:
|
|
98
|
+
reserved = max_output_tokens(policy)
|
|
99
|
+
return max(0, policy.context_limit - reserved)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def is_overflow(policy: ContextPolicy, usage: dict[str, Any]) -> bool:
|
|
103
|
+
"""True when the last turn consumed more than the usable budget."""
|
|
104
|
+
if not policy.auto or policy.context_limit <= 0:
|
|
105
|
+
return False
|
|
106
|
+
total = usage.get("total_tokens")
|
|
107
|
+
if not total:
|
|
108
|
+
total = int(usage.get("prompt_tokens") or 0) + int(usage.get("completion_tokens") or 0)
|
|
109
|
+
return int(total) >= usable(policy)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def messages_tokens(messages: list[ChatMessage]) -> int:
|
|
113
|
+
return sum(message.token_estimate() for message in messages)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
# --------------------------------------------------------------------------- #
|
|
117
|
+
# Per-request projection: what actually gets sent to the model
|
|
118
|
+
# --------------------------------------------------------------------------- #
|
|
119
|
+
def strip_reasoning(messages: list[ChatMessage], keep_last: int = KEEP_REASONING_STEPS) -> int:
|
|
120
|
+
"""Drop reasoning from all but the most recent assistant turns.
|
|
121
|
+
|
|
122
|
+
Reasoning is scaffolding: once a step has produced tool calls the model has
|
|
123
|
+
already acted on it, so replaying it wastes the whole context window.
|
|
124
|
+
Returns the number of tokens saved.
|
|
125
|
+
"""
|
|
126
|
+
saved = 0
|
|
127
|
+
indices = [
|
|
128
|
+
index
|
|
129
|
+
for index, message in enumerate(messages)
|
|
130
|
+
if message.role == "assistant" and message.reasoning
|
|
131
|
+
]
|
|
132
|
+
keep = set(indices[-keep_last:]) if keep_last > 0 else set()
|
|
133
|
+
for index in indices:
|
|
134
|
+
if index in keep:
|
|
135
|
+
continue
|
|
136
|
+
saved += estimate_tokens(messages[index].reasoning)
|
|
137
|
+
messages[index].reasoning = ""
|
|
138
|
+
return saved
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def supersede_stateful_results(messages: list[ChatMessage]) -> int:
|
|
142
|
+
"""Blank older snapshots from tools whose newest result supersedes them.
|
|
143
|
+
|
|
144
|
+
``workspace_info``, ``read_shared_context`` or ``list_findings`` return the
|
|
145
|
+
full current state; only the latest call is meaningful. Earlier copies are
|
|
146
|
+
replaced by a marker so the conversation stays coherent.
|
|
147
|
+
"""
|
|
148
|
+
latest: dict[str, int] = {}
|
|
149
|
+
for index, message in enumerate(messages):
|
|
150
|
+
if message.role == "tool" and message.name in STATEFUL_TOOLS:
|
|
151
|
+
latest[message.name] = index
|
|
152
|
+
saved = 0
|
|
153
|
+
for index, message in enumerate(messages):
|
|
154
|
+
if message.role != "tool" or message.name not in STATEFUL_TOOLS:
|
|
155
|
+
continue
|
|
156
|
+
if message.compacted or latest.get(message.name) == index:
|
|
157
|
+
continue
|
|
158
|
+
saved += estimate_tokens(message.content)
|
|
159
|
+
message.content = f"[superseded by a newer {message.name} call]"
|
|
160
|
+
message.compacted = True
|
|
161
|
+
return saved
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def dedupe_tool_calls(messages: list[ChatMessage]) -> int:
|
|
165
|
+
"""Collapse consecutive identical tool calls from separate steps."""
|
|
166
|
+
saved = 0
|
|
167
|
+
seen: set[tuple[str, str]] = set()
|
|
168
|
+
for message in messages:
|
|
169
|
+
for call in message.tool_calls:
|
|
170
|
+
signature = (call.name, call.arguments or "{}")
|
|
171
|
+
if signature in seen:
|
|
172
|
+
saved += estimate_tokens(call.arguments) + estimate_tokens(call.name)
|
|
173
|
+
else:
|
|
174
|
+
seen.add(signature)
|
|
175
|
+
return saved
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def optimize(
|
|
179
|
+
messages: list[ChatMessage], policy: ContextPolicy
|
|
180
|
+
) -> tuple[list[ChatMessage], dict[str, int]]:
|
|
181
|
+
"""Apply the cheap, lossless-ish reductions before every request.
|
|
182
|
+
|
|
183
|
+
Order matters: reasoning first (largest win, no information loss for
|
|
184
|
+
decisions already taken), then superseded snapshots, then pruning of old
|
|
185
|
+
tool output. Never touches the system prompt or the latest turn.
|
|
186
|
+
"""
|
|
187
|
+
if not policy.optimize:
|
|
188
|
+
return messages, {"reasoning": 0, "stateful": 0, "pruned": 0}
|
|
189
|
+
stats = {"reasoning": 0, "stateful": 0, "pruned": 0}
|
|
190
|
+
stats["reasoning"] = strip_reasoning(messages, policy.keep_reasoning_steps)
|
|
191
|
+
stats["stateful"] = supersede_stateful_results(messages)
|
|
192
|
+
prune_result = prune(messages, policy)
|
|
193
|
+
stats["pruned"] = prune_result.pruned
|
|
194
|
+
return messages, stats
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def protect_budget(policy: ContextPolicy) -> int:
|
|
198
|
+
if policy.preserve_recent_tokens is not None:
|
|
199
|
+
return policy.preserve_recent_tokens
|
|
200
|
+
return min(
|
|
201
|
+
MAX_PRESERVE_RECENT_TOKENS,
|
|
202
|
+
max(MIN_PRESERVE_RECENT_TOKENS, int(usable(policy) * 0.25)),
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def prune(messages: list[ChatMessage], policy: ContextPolicy) -> PruneResult:
|
|
207
|
+
"""Erase the output of old tool results, protecting recent ones.
|
|
208
|
+
|
|
209
|
+
Walks backwards accumulating tool-output tokens; once more than
|
|
210
|
+
``PRUNE_PROTECT`` tokens have been seen, every older tool result is
|
|
211
|
+
blanked (it stays in the transcript for auditing).
|
|
212
|
+
"""
|
|
213
|
+
if not policy.prune:
|
|
214
|
+
return PruneResult()
|
|
215
|
+
pruned = 0
|
|
216
|
+
protected: list[tuple[ChatMessage, int]] = []
|
|
217
|
+
cleared: list[tuple[ChatMessage, str, int]] = []
|
|
218
|
+
for message in reversed(messages):
|
|
219
|
+
if message.role != "tool" or message.compacted:
|
|
220
|
+
continue
|
|
221
|
+
if message.pinned:
|
|
222
|
+
continue
|
|
223
|
+
size = estimate_tokens(message.content)
|
|
224
|
+
if not protected:
|
|
225
|
+
# Always keep the most recent tool result, exactly like OpenCode.
|
|
226
|
+
protected.append((message, size))
|
|
227
|
+
continue
|
|
228
|
+
protected_tokens = sum(item[1] for item in protected)
|
|
229
|
+
if protected_tokens + size <= PRUNE_PROTECT:
|
|
230
|
+
protected.append((message, size))
|
|
231
|
+
continue
|
|
232
|
+
cleared.append((message, message.content, size))
|
|
233
|
+
pruned += size
|
|
234
|
+
protected_tokens = sum(item[1] for item in protected)
|
|
235
|
+
if pruned < PRUNE_MINIMUM:
|
|
236
|
+
# Not worth mutating the transcript: leave every message untouched.
|
|
237
|
+
return PruneResult(pruned=0, protected=protected_tokens, parts=0)
|
|
238
|
+
for message, _, _ in cleared:
|
|
239
|
+
message.compacted = True
|
|
240
|
+
message.content = "[Old tool result content cleared to save context]"
|
|
241
|
+
return PruneResult(pruned=pruned, protected=protected_tokens, parts=len(cleared))
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _turn_starts(messages: list[ChatMessage]) -> list[int]:
|
|
245
|
+
"""Indexes of user turns, used as turn boundaries."""
|
|
246
|
+
return [i for i, message in enumerate(messages) if message.role == "user"]
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _summarize(messages: list[ChatMessage], max_chars: int = 12_000) -> str:
|
|
250
|
+
"""Build a terse, structured checkpoint from the head of the transcript."""
|
|
251
|
+
lines: list[str] = []
|
|
252
|
+
for message in messages:
|
|
253
|
+
if message.role == "tool" or message.compacted:
|
|
254
|
+
continue
|
|
255
|
+
if message.role == "assistant" and message.tool_calls:
|
|
256
|
+
calls = ", ".join(call.name for call in message.tool_calls)
|
|
257
|
+
if calls:
|
|
258
|
+
lines.append(f"- red/blue tool calls: {calls}")
|
|
259
|
+
text = (message.content or "").strip()
|
|
260
|
+
if not text:
|
|
261
|
+
continue
|
|
262
|
+
label = {
|
|
263
|
+
"system": "system",
|
|
264
|
+
"user": "task",
|
|
265
|
+
"assistant": "agent",
|
|
266
|
+
}.get(message.role, message.role)
|
|
267
|
+
snippet = " ".join(text.split())[:400]
|
|
268
|
+
lines.append(f"- {label}: {snippet}")
|
|
269
|
+
summary = "\n".join(lines)
|
|
270
|
+
if len(summary) > max_chars:
|
|
271
|
+
summary = summary[:max_chars] + "\n- ...(older history omitted)"
|
|
272
|
+
return summary
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def compact(
|
|
276
|
+
messages: list[ChatMessage], policy: ContextPolicy
|
|
277
|
+
) -> tuple[list[ChatMessage], CompactionResult]:
|
|
278
|
+
"""Return a compacted copy of ``messages``.
|
|
279
|
+
|
|
280
|
+
The head is replaced by a single checkpoint message; the tail (recent
|
|
281
|
+
turns) is kept verbatim within ``protect_budget``. The system message is
|
|
282
|
+
always preserved.
|
|
283
|
+
"""
|
|
284
|
+
if not policy.auto:
|
|
285
|
+
return messages, CompactionResult(reason="auto compaction disabled")
|
|
286
|
+
|
|
287
|
+
system = [m for m in messages if m.role == "system"]
|
|
288
|
+
body = [m for m in messages if m.role != "system"]
|
|
289
|
+
if len(body) <= 2:
|
|
290
|
+
return messages, CompactionResult(reason="nothing to compact")
|
|
291
|
+
|
|
292
|
+
budget = protect_budget(policy)
|
|
293
|
+
starts = _turn_starts(body)
|
|
294
|
+
if not starts:
|
|
295
|
+
return messages, CompactionResult(reason="no user turns")
|
|
296
|
+
|
|
297
|
+
# Anchor the tail at a user-turn boundary within the budget.
|
|
298
|
+
tail_start = starts[-1]
|
|
299
|
+
total = 0
|
|
300
|
+
for index in range(len(starts) - 1, -1, -1):
|
|
301
|
+
start = starts[index]
|
|
302
|
+
end = starts[index + 1] if index + 1 < len(starts) else len(body)
|
|
303
|
+
size = messages_tokens(body[start:end])
|
|
304
|
+
if total + size <= budget or index == len(starts) - 1:
|
|
305
|
+
total += size
|
|
306
|
+
tail_start = start
|
|
307
|
+
continue
|
|
308
|
+
break
|
|
309
|
+
|
|
310
|
+
head = body[:tail_start]
|
|
311
|
+
if not head:
|
|
312
|
+
return messages, CompactionResult(reason="tail already fits the budget")
|
|
313
|
+
|
|
314
|
+
summary = _summarize(head)
|
|
315
|
+
checkpoint = ChatMessage(
|
|
316
|
+
role="user",
|
|
317
|
+
pinned=True,
|
|
318
|
+
content=(
|
|
319
|
+
"[Context checkpoint] Earlier work in this engagement was summarised "
|
|
320
|
+
"to stay within the model context window. Details omitted here are "
|
|
321
|
+
"still available in the full transcript and the session file.\n\n"
|
|
322
|
+
f"{summary}"
|
|
323
|
+
),
|
|
324
|
+
)
|
|
325
|
+
removed = messages_tokens(head)
|
|
326
|
+
compacted_messages = [*system, checkpoint, *body[tail_start:]]
|
|
327
|
+
return compacted_messages, CompactionResult(
|
|
328
|
+
compacted=True,
|
|
329
|
+
removed=removed,
|
|
330
|
+
preserved=messages_tokens(body[tail_start:]),
|
|
331
|
+
summary=summary,
|
|
332
|
+
reason="context exceeded the usable budget",
|
|
333
|
+
)
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def prune_tool_output(text: str, max_chars: int = TOOL_OUTPUT_MAX_CHARS) -> str:
|
|
337
|
+
"""Bound a tool result before storing it in the transcript."""
|
|
338
|
+
if len(text) <= max_chars:
|
|
339
|
+
return text
|
|
340
|
+
return f"{text[:max_chars]}\n[tool output truncated for context: {len(text) - max_chars} chars omitted]"
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def context_report(
|
|
344
|
+
messages: list[ChatMessage], policy: ContextPolicy, usage: dict[str, Any] | None = None
|
|
345
|
+
) -> dict[str, Any]:
|
|
346
|
+
budget = usable(policy)
|
|
347
|
+
used = messages_tokens(messages)
|
|
348
|
+
# Effective size after the per-request projection, which is what the
|
|
349
|
+
# provider actually bills for.
|
|
350
|
+
projected = used
|
|
351
|
+
if policy.optimize:
|
|
352
|
+
scratch = [
|
|
353
|
+
ChatMessage(
|
|
354
|
+
role=m.role,
|
|
355
|
+
content=m.content,
|
|
356
|
+
reasoning=m.reasoning,
|
|
357
|
+
tool_calls=m.tool_calls,
|
|
358
|
+
name=m.name,
|
|
359
|
+
compacted=m.compacted,
|
|
360
|
+
pinned=m.pinned,
|
|
361
|
+
)
|
|
362
|
+
for m in messages
|
|
363
|
+
]
|
|
364
|
+
strip_reasoning(scratch, policy.keep_reasoning_steps)
|
|
365
|
+
supersede_stateful_results(scratch)
|
|
366
|
+
projected = messages_tokens(scratch)
|
|
367
|
+
return {
|
|
368
|
+
"context_limit": policy.context_limit,
|
|
369
|
+
"usable": budget,
|
|
370
|
+
"estimated": used,
|
|
371
|
+
"projected": projected,
|
|
372
|
+
"saved": max(0, used - projected),
|
|
373
|
+
"saving_percent": round((1 - projected / used) * 100, 1) if used else 0.0,
|
|
374
|
+
"percent": round((projected / budget) * 100, 1) if budget else 0.0,
|
|
375
|
+
"overflow": is_overflow(policy, usage or {}),
|
|
376
|
+
"prune": policy.prune,
|
|
377
|
+
"auto_compact": policy.auto,
|
|
378
|
+
"optimize": policy.optimize,
|
|
379
|
+
"protect_budget": protect_budget(policy),
|
|
380
|
+
"messages": len(messages),
|
|
381
|
+
}
|