loopbrake 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- loopbrake-0.1.0/.gitignore +7 -0
- loopbrake-0.1.0/LICENSE +21 -0
- loopbrake-0.1.0/PKG-INFO +255 -0
- loopbrake-0.1.0/README.md +231 -0
- loopbrake-0.1.0/pyproject.toml +52 -0
- loopbrake-0.1.0/src/loopbrake/__init__.py +8 -0
- loopbrake-0.1.0/src/loopbrake/agent_sdk.py +84 -0
- loopbrake-0.1.0/src/loopbrake/brake.py +123 -0
- loopbrake-0.1.0/src/loopbrake/calibration.py +107 -0
- loopbrake-0.1.0/src/loopbrake/cli.py +111 -0
- loopbrake-0.1.0/src/loopbrake/conformal.py +21 -0
- loopbrake-0.1.0/src/loopbrake/records.py +103 -0
- loopbrake-0.1.0/src/loopbrake/signals.py +226 -0
- loopbrake-0.1.0/src/loopbrake/traces.py +186 -0
loopbrake-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Sahil Selokar
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
loopbrake-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: loopbrake
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Stops stuck AI agent runs, with a guaranteed limit on stopping good ones.
|
|
5
|
+
Project-URL: Homepage, https://github.com/SahilSelokar/LoopBrake
|
|
6
|
+
Project-URL: Source, https://github.com/SahilSelokar/LoopBrake
|
|
7
|
+
Project-URL: Results, https://github.com/SahilSelokar/LoopBrake/tree/main/eval/results
|
|
8
|
+
Author: Sahil Selokar
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: agent loops,ai agents,claude code,conformal prediction,llm,stop controller
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
20
|
+
Requires-Python: >=3.11
|
|
21
|
+
Provides-Extra: agent-sdk
|
|
22
|
+
Requires-Dist: claude-agent-sdk; extra == 'agent-sdk'
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
|
|
25
|
+
<div align="center">
|
|
26
|
+
|
|
27
|
+
# LoopBrake
|
|
28
|
+
|
|
29
|
+
**Brakes for stuck AI agents.**
|
|
30
|
+
|
|
31
|
+
Stop agent runs that are going in circles, with a guaranteed limit on how often a good run gets stopped.
|
|
32
|
+
|
|
33
|
+
[](#status)
|
|
34
|
+
[](pyproject.toml)
|
|
35
|
+
[](pyproject.toml)
|
|
36
|
+
[](LICENSE)
|
|
37
|
+
|
|
38
|
+
</div>
|
|
39
|
+
|
|
40
|
+
---
|
|
41
|
+
|
|
42
|
+
## The problem
|
|
43
|
+
|
|
44
|
+
AI agents get stuck. They run the same command again and again, re-read the same file, or hit the
|
|
45
|
+
same error, and they keep spending tokens until a step or budget limit finally stops them.
|
|
46
|
+
|
|
47
|
+
The usual fixes are blunt: a fixed step limit, or a rule like "stop after the same action 3 times".
|
|
48
|
+
Set them tight and you kill runs that would have succeeded. Set them loose and you pay for every loop.
|
|
49
|
+
|
|
50
|
+
## The idea
|
|
51
|
+
|
|
52
|
+
LoopBrake watches a run one step at a time and gives each step a **stuck score**. When the score
|
|
53
|
+
crosses a **stop line**, it stops the run and says why.
|
|
54
|
+
|
|
55
|
+
The stop line is not guessed. It is set from **your own past successful runs**, so that at most a
|
|
56
|
+
chosen share of good runs, for example 5 in 100, would ever cross it. That is a statistical
|
|
57
|
+
guarantee (split-conformal calibration), not a tuned target.
|
|
58
|
+
|
|
59
|
+
**In v1 the stuck score is simply how long the run has gone on.** Two experiments (below) tested
|
|
60
|
+
smarter scores, and neither beat it on agents they were not tuned on. The stuck signals (repeats,
|
|
61
|
+
nothing new, same error again) still run, but only to explain why a stopped run looked stuck.
|
|
62
|
+
|
|
63
|
+
```mermaid
|
|
64
|
+
flowchart LR
|
|
65
|
+
A[Agent takes a step] --> B[Stuck score]
|
|
66
|
+
B --> C{Above the stop line?}
|
|
67
|
+
C -- no --> A
|
|
68
|
+
C -- yes --> D[Stop the run and say why]
|
|
69
|
+
E[(Your past successful runs)] -. set the stop line .-> C
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
## What a stop looks like
|
|
73
|
+
|
|
74
|
+
A real run from the public SWE-bench data (GPT-5-mini), replayed through LoopBrake. The agent kept
|
|
75
|
+
searching the same files for a function and found nothing new:
|
|
76
|
+
|
|
77
|
+
```text
|
|
78
|
+
swe-gpt5mini · matplotlib__matplotlib-25775 · stopped at step 27 of 72 · saved 2,023,115 tokens (84%)
|
|
79
|
+
Reason: repeating in 5 of last 5 steps (similar to step 19); nothing new in 2 of last 5 steps (0 of 21 output lines new)
|
|
80
|
+
|
|
81
|
+
23 sed -n '760,1080p' lib/matplotlib/backend_bases.py
|
|
82
|
+
24 grep -n "def set_antialiased" -n lib/matplotlib/backend_bases.py || true
|
|
83
|
+
25 sed -n '892,932p' lib/matplotlib/backend_bases.py
|
|
84
|
+
26 sed -n '1,220p' lib/matplotlib/patches.py
|
|
85
|
+
27 grep -n "set_antialiased" -n lib -R || true
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
More in [eval/results/kill-stories.md](eval/results/kill-stories.md).
|
|
89
|
+
|
|
90
|
+
## Use it
|
|
91
|
+
|
|
92
|
+
**Install.** `pip install loopbrake`, or `uv add loopbrake`. Until the first PyPI release, use
|
|
93
|
+
`pip install git+https://github.com/SahilSelokar/LoopBrake`. No other packages are needed.
|
|
94
|
+
|
|
95
|
+
**1. Set your stop line from your own past runs.** It needs at least 19 successful runs (at α 5%);
|
|
96
|
+
with fewer, LoopBrake only watches.
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
loopbrake calibrate ~/.claude/projects/<your-project>/ --project my-agent # your Claude Code history
|
|
100
|
+
loopbrake calibrate my_runs.jsonl --project my-agent # or recorded runs
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
**2. Add it to your agent loop.**
|
|
104
|
+
|
|
105
|
+
```python
|
|
106
|
+
import loopbrake
|
|
107
|
+
|
|
108
|
+
with loopbrake.start(project="my-agent") as brake:
|
|
109
|
+
for action, result in my_agent_steps(): # your loop
|
|
110
|
+
decision = brake.step(action, result)
|
|
111
|
+
if decision.stop:
|
|
112
|
+
print(decision.reason); break
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
**3. See how it's doing.**
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
loopbrake status --project my-agent # runs watched, stops, and stops you marked as mistakes
|
|
119
|
+
loopbrake feedback <run-id> --mistaken # tell LoopBrake a stop was wrong
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
**Claude's agent toolkit:** `pip install "loopbrake[agent-sdk]"`, then
|
|
123
|
+
`ClaudeAgentOptions(hooks=loopbrake.agent_sdk.hooks(project="my-agent"))`.
|
|
124
|
+
|
|
125
|
+
Everything stays on your machine, in `~/.loopbrake`. LoopBrake never uses the network. If something
|
|
126
|
+
inside it fails, it switches to watching only. It never crashes or stops your agent because of its
|
|
127
|
+
own problem.
|
|
128
|
+
|
|
129
|
+
## Phase 1 results
|
|
130
|
+
|
|
131
|
+
Before building the product, we tested the idea offline. We replayed **2,979 recorded runs** from
|
|
132
|
+
six agent groups, on coding tasks
|
|
133
|
+
([SWE-bench Verified](https://github.com/SWE-bench/experiments)) and customer-service tasks
|
|
134
|
+
([τ-bench](https://github.com/sierra-research/tau-bench)).
|
|
135
|
+
|
|
136
|
+
| Finding | Result |
|
|
137
|
+
|---|---|
|
|
138
|
+
| The guarantee held: good runs stopped by the chosen method, on every test group | **at most 4.9%** (limit 5%) |
|
|
139
|
+
| The naive rule "stop at 20 steps or 3 repeats" stops good runs | **99.3%** (Devstral), **29.1%** (GPT-5-mini) |
|
|
140
|
+
| Tokens saved by a calibrated step limit (GPT-5-mini, 100 past runs) | **11.2%** |
|
|
141
|
+
| Tokens saved by the stuck signals (same setting) | **12.9%** |
|
|
142
|
+
| Extra saving from stuck signals over the step limit, with 95% interval | SWE-bench **+1.5%** [−9.7%, +13.2%]<br>τ-bench **+0.9%** [−3.1%, +5.0%] |
|
|
143
|
+
|
|
144
|
+
**Verdict: NO-GO.** The cheap stuck signals do not clearly beat a calibrated step limit. Following
|
|
145
|
+
the plan set in advance, a second experiment tested a progress judge (below).
|
|
146
|
+
For comparison, [FailFast](https://arxiv.org/abs/2608.03222), a trained monitor, reports 14.6–20.4%
|
|
147
|
+
saved at 5% of good runs stopped. Its threshold was set on the same data it reports on, though, so
|
|
148
|
+
that 5% is a target, not a guarantee.
|
|
149
|
+
|
|
150
|
+
Full report: [eval/results/results.md](eval/results/results.md)
|
|
151
|
+
|
|
152
|
+
## Progress judge: also NO-GO
|
|
153
|
+
|
|
154
|
+
Counting repeats is not the same as understanding a step, so the second experiment asked a fast
|
|
155
|
+
hosted decision model, [Jev](https://docs.typesafe.ai) (`jev-1.13.0`), about every step: *did this
|
|
156
|
+
move the agent closer to finishing?* and *what kind of step was it?* Savings are counted **net**:
|
|
157
|
+
every token the judge reads is subtracted. The method was chosen on one agent and committed
|
|
158
|
+
(`0827ece`) before any other agent was judged.
|
|
159
|
+
|
|
160
|
+
| Dataset | Extra net saving over the step limit, with 95% interval |
|
|
161
|
+
|---|---|
|
|
162
|
+
| SWE-bench (GPT-5-mini) | **−6.3%** [−18.9%, +1.9%] |
|
|
163
|
+
| τ-bench (4 groups) | **−4.2%** [−8.2%, −1.5%] |
|
|
164
|
+
|
|
165
|
+
What we learned:
|
|
166
|
+
|
|
167
|
+
- **A threshold tuned on one agent did not travel.** The chosen method asked the judge only when the
|
|
168
|
+
cheap score passed a level set on Devstral's long runs. On the other agents that level was almost
|
|
169
|
+
never reached (0–0.3% of steps), so the method rarely stopped anything.
|
|
170
|
+
- **A judge on every step is expensive.** On short customer-service tasks it used about as many
|
|
171
|
+
tokens as the agent itself.
|
|
172
|
+
- **Even ignoring its cost, the judge did not spot stuck runs better than counting steps**: for
|
|
173
|
+
example 7.3% vs 9.3% of tokens saved on GPT-5-mini.
|
|
174
|
+
- The guarantee held on every group. The whole experiment took 73,812 judgments, for about $3.91.
|
|
175
|
+
|
|
176
|
+
Full report: [eval/results/judge/results.md](eval/results/judge/results.md)
|
|
177
|
+
|
|
178
|
+
## How the results are kept honest
|
|
179
|
+
|
|
180
|
+
- **Chosen in advance.** The method was picked using one agent's runs and committed *before* any
|
|
181
|
+
test runs were scored (commits `8b8eaf3` and, for the judge, `0827ece`).
|
|
182
|
+
- **Mistakes stay visible.** The first results were committed exactly as they came out, including a
|
|
183
|
+
flaw in how runs were split (`e655b96`). The fix is a separate, documented commit (`76e8cdc`), with
|
|
184
|
+
before-and-after numbers in [CORRECTIONS.md](eval/results/CORRECTIONS.md).
|
|
185
|
+
- **Tested on unseen runs.** The stop line is always checked on runs it was not set from.
|
|
186
|
+
- **Reproducible.** Every number above comes from `eval/results/` and comes out identical on every run.
|
|
187
|
+
|
|
188
|
+
## How the guarantee works
|
|
189
|
+
|
|
190
|
+
1. Each step gets a stuck score. A run's score is the highest score it reaches.
|
|
191
|
+
2. Take *n* past successful runs and sort their scores. The stop line is the *k*-th smallest score,
|
|
192
|
+
where *k* = ⌈(*n* + 1)(1 − α)⌉ and α is the share of good runs you accept losing (for example 5%).
|
|
193
|
+
3. A new successful run then crosses the stop line with probability at most α.
|
|
194
|
+
|
|
195
|
+
With α = 5%, LoopBrake needs at least 19 past successful runs. With fewer, it only watches and
|
|
196
|
+
never stops anything.
|
|
197
|
+
|
|
198
|
+
The guarantee holds when new runs look like the past ones (same agent, same kind of tasks), and when
|
|
199
|
+
the past runs come from different tasks. We learned the second condition the hard way: see
|
|
200
|
+
[CORRECTIONS.md](eval/results/CORRECTIONS.md).
|
|
201
|
+
|
|
202
|
+
## Reproduce
|
|
203
|
+
|
|
204
|
+
Requires [uv](https://docs.astral.sh/uv/) and Python 3.11+.
|
|
205
|
+
|
|
206
|
+
```bash
|
|
207
|
+
git clone https://github.com/SahilSelokar/LoopBrake && cd LoopBrake
|
|
208
|
+
uv run pytest # the test suite
|
|
209
|
+
uv run python eval/fetch.py # one-time download, about 450 MB, into ~/.loopbrake/data
|
|
210
|
+
uv run python eval/run.py --final # about 1 minute; writes eval/results/
|
|
211
|
+
```
|
|
212
|
+
|
|
213
|
+
The progress judge needs a [Typesafe](https://typesafe.ai) API key in `TYPESAFE_API_KEY` (or in
|
|
214
|
+
`~/.loopbrake/typesafe_key`). Judging every public step costs about $4:
|
|
215
|
+
|
|
216
|
+
```bash
|
|
217
|
+
uv run python eval/tasks.py # task texts for the judge
|
|
218
|
+
uv run python eval/judge.py --group swe-devstral # one group at a time; resumable
|
|
219
|
+
uv run python eval/judge_eval.py --final # reads stored answers only; writes eval/results/judge/
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
## Repository layout
|
|
223
|
+
|
|
224
|
+
```text
|
|
225
|
+
src/loopbrake/ scoring core: stuck signals, stop-line rule, run readers (standard library only)
|
|
226
|
+
eval/ the experiments: fetch.py downloads the data, run.py replays runs, judge.py asks the
|
|
227
|
+
progress judge, judge_eval.py scores its answers; results/ holds the published numbers
|
|
228
|
+
specs/ design: constitution, roadmap, and the spec, plan, research and tasks of each experiment
|
|
229
|
+
tests/ tests, including a check of the guarantee on simulated data
|
|
230
|
+
liveness.py the original naive rule, kept as the baseline
|
|
231
|
+
```
|
|
232
|
+
|
|
233
|
+
## Roadmap
|
|
234
|
+
|
|
235
|
+
| Phase | What | Status |
|
|
236
|
+
|---|---|---|
|
|
237
|
+
| 1 | **Experiment**: does it work on real runs? | Done: NO-GO for cheap signals |
|
|
238
|
+
| 1b | **Progress judge**: a hosted decision model judges whether each step moved the run forward | Done: NO-GO |
|
|
239
|
+
| 2 | **Python package**: `pip install loopbrake`; a stop line on run length, with a guarantee and a readable reason | Built (v0.1.0); first PyPI release pending |
|
|
240
|
+
| 3 | **Claude Code plugin**: stop stuck sessions live, calibrated on your own history | Planned |
|
|
241
|
+
| 4 | **Observability**: live dashboard, plus export to Datadog, Grafana and others via OpenTelemetry | Planned |
|
|
242
|
+
| 5 | **Launch**: a demo agent, the public release and a video | Planned |
|
|
243
|
+
|
|
244
|
+
The full plan is in [specs/roadmap.md](specs/roadmap.md).
|
|
245
|
+
|
|
246
|
+
## Status
|
|
247
|
+
|
|
248
|
+
v0.1.0 is built and tested. The first PyPI release comes once publishing is set up; until then,
|
|
249
|
+
install from GitHub (see "Use it"). Its stop rule is a stop line on run length, set from your own past
|
|
250
|
+
successful runs, with a guaranteed limit on stopping good runs.
|
|
251
|
+
|
|
252
|
+
## License
|
|
253
|
+
|
|
254
|
+
[MIT](LICENSE). One test fixture is a trimmed τ-bench run, used under τ-bench's own MIT license
|
|
255
|
+
(see [tests/fixtures/NOTICE.md](tests/fixtures/NOTICE.md)).
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
<div align="center">
|
|
2
|
+
|
|
3
|
+
# LoopBrake
|
|
4
|
+
|
|
5
|
+
**Brakes for stuck AI agents.**
|
|
6
|
+
|
|
7
|
+
Stop agent runs that are going in circles, with a guaranteed limit on how often a good run gets stopped.
|
|
8
|
+
|
|
9
|
+
[](#status)
|
|
10
|
+
[](pyproject.toml)
|
|
11
|
+
[](pyproject.toml)
|
|
12
|
+
[](LICENSE)
|
|
13
|
+
|
|
14
|
+
</div>
|
|
15
|
+
|
|
16
|
+
---
|
|
17
|
+
|
|
18
|
+
## The problem
|
|
19
|
+
|
|
20
|
+
AI agents get stuck. They run the same command again and again, re-read the same file, or hit the
|
|
21
|
+
same error, and they keep spending tokens until a step or budget limit finally stops them.
|
|
22
|
+
|
|
23
|
+
The usual fixes are blunt: a fixed step limit, or a rule like "stop after the same action 3 times".
|
|
24
|
+
Set them tight and you kill runs that would have succeeded. Set them loose and you pay for every loop.
|
|
25
|
+
|
|
26
|
+
## The idea
|
|
27
|
+
|
|
28
|
+
LoopBrake watches a run one step at a time and gives each step a **stuck score**. When the score
|
|
29
|
+
crosses a **stop line**, it stops the run and says why.
|
|
30
|
+
|
|
31
|
+
The stop line is not guessed. It is set from **your own past successful runs**, so that at most a
|
|
32
|
+
chosen share of good runs, for example 5 in 100, would ever cross it. That is a statistical
|
|
33
|
+
guarantee (split-conformal calibration), not a tuned target.
|
|
34
|
+
|
|
35
|
+
**In v1 the stuck score is simply how long the run has gone on.** Two experiments (below) tested
|
|
36
|
+
smarter scores, and neither beat it on agents they were not tuned on. The stuck signals (repeats,
|
|
37
|
+
nothing new, same error again) still run, but only to explain why a stopped run looked stuck.
|
|
38
|
+
|
|
39
|
+
```mermaid
|
|
40
|
+
flowchart LR
|
|
41
|
+
A[Agent takes a step] --> B[Stuck score]
|
|
42
|
+
B --> C{Above the stop line?}
|
|
43
|
+
C -- no --> A
|
|
44
|
+
C -- yes --> D[Stop the run and say why]
|
|
45
|
+
E[(Your past successful runs)] -. set the stop line .-> C
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
## What a stop looks like
|
|
49
|
+
|
|
50
|
+
A real run from the public SWE-bench data (GPT-5-mini), replayed through LoopBrake. The agent kept
|
|
51
|
+
searching the same files for a function and found nothing new:
|
|
52
|
+
|
|
53
|
+
```text
|
|
54
|
+
swe-gpt5mini · matplotlib__matplotlib-25775 · stopped at step 27 of 72 · saved 2,023,115 tokens (84%)
|
|
55
|
+
Reason: repeating in 5 of last 5 steps (similar to step 19); nothing new in 2 of last 5 steps (0 of 21 output lines new)
|
|
56
|
+
|
|
57
|
+
23 sed -n '760,1080p' lib/matplotlib/backend_bases.py
|
|
58
|
+
24 grep -n "def set_antialiased" -n lib/matplotlib/backend_bases.py || true
|
|
59
|
+
25 sed -n '892,932p' lib/matplotlib/backend_bases.py
|
|
60
|
+
26 sed -n '1,220p' lib/matplotlib/patches.py
|
|
61
|
+
27 grep -n "set_antialiased" -n lib -R || true
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
More in [eval/results/kill-stories.md](eval/results/kill-stories.md).
|
|
65
|
+
|
|
66
|
+
## Use it
|
|
67
|
+
|
|
68
|
+
**Install.** `pip install loopbrake`, or `uv add loopbrake`. Until the first PyPI release, use
|
|
69
|
+
`pip install git+https://github.com/SahilSelokar/LoopBrake`. No other packages are needed.
|
|
70
|
+
|
|
71
|
+
**1. Set your stop line from your own past runs.** It needs at least 19 successful runs (at α 5%);
|
|
72
|
+
with fewer, LoopBrake only watches.
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
loopbrake calibrate ~/.claude/projects/<your-project>/ --project my-agent # your Claude Code history
|
|
76
|
+
loopbrake calibrate my_runs.jsonl --project my-agent # or recorded runs
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
**2. Add it to your agent loop.**
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
import loopbrake
|
|
83
|
+
|
|
84
|
+
with loopbrake.start(project="my-agent") as brake:
|
|
85
|
+
for action, result in my_agent_steps(): # your loop
|
|
86
|
+
decision = brake.step(action, result)
|
|
87
|
+
if decision.stop:
|
|
88
|
+
print(decision.reason); break
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
**3. See how it's doing.**
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
loopbrake status --project my-agent # runs watched, stops, and stops you marked as mistakes
|
|
95
|
+
loopbrake feedback <run-id> --mistaken # tell LoopBrake a stop was wrong
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
**Claude's agent toolkit:** `pip install "loopbrake[agent-sdk]"`, then
|
|
99
|
+
`ClaudeAgentOptions(hooks=loopbrake.agent_sdk.hooks(project="my-agent"))`.
|
|
100
|
+
|
|
101
|
+
Everything stays on your machine, in `~/.loopbrake`. LoopBrake never uses the network. If something
|
|
102
|
+
inside it fails, it switches to watching only. It never crashes or stops your agent because of its
|
|
103
|
+
own problem.
|
|
104
|
+
|
|
105
|
+
## Phase 1 results
|
|
106
|
+
|
|
107
|
+
Before building the product, we tested the idea offline. We replayed **2,979 recorded runs** from
|
|
108
|
+
six agent groups, on coding tasks
|
|
109
|
+
([SWE-bench Verified](https://github.com/SWE-bench/experiments)) and customer-service tasks
|
|
110
|
+
([τ-bench](https://github.com/sierra-research/tau-bench)).
|
|
111
|
+
|
|
112
|
+
| Finding | Result |
|
|
113
|
+
|---|---|
|
|
114
|
+
| The guarantee held: good runs stopped by the chosen method, on every test group | **at most 4.9%** (limit 5%) |
|
|
115
|
+
| The naive rule "stop at 20 steps or 3 repeats" stops good runs | **99.3%** (Devstral), **29.1%** (GPT-5-mini) |
|
|
116
|
+
| Tokens saved by a calibrated step limit (GPT-5-mini, 100 past runs) | **11.2%** |
|
|
117
|
+
| Tokens saved by the stuck signals (same setting) | **12.9%** |
|
|
118
|
+
| Extra saving from stuck signals over the step limit, with 95% interval | SWE-bench **+1.5%** [−9.7%, +13.2%]<br>τ-bench **+0.9%** [−3.1%, +5.0%] |
|
|
119
|
+
|
|
120
|
+
**Verdict: NO-GO.** The cheap stuck signals do not clearly beat a calibrated step limit. Following
|
|
121
|
+
the plan set in advance, a second experiment tested a progress judge (below).
|
|
122
|
+
For comparison, [FailFast](https://arxiv.org/abs/2608.03222), a trained monitor, reports 14.6–20.4%
|
|
123
|
+
saved at 5% of good runs stopped. Its threshold was set on the same data it reports on, though, so
|
|
124
|
+
that 5% is a target, not a guarantee.
|
|
125
|
+
|
|
126
|
+
Full report: [eval/results/results.md](eval/results/results.md)
|
|
127
|
+
|
|
128
|
+
## Progress judge: also NO-GO
|
|
129
|
+
|
|
130
|
+
Counting repeats is not the same as understanding a step, so the second experiment asked a fast
|
|
131
|
+
hosted decision model, [Jev](https://docs.typesafe.ai) (`jev-1.13.0`), about every step: *did this
|
|
132
|
+
move the agent closer to finishing?* and *what kind of step was it?* Savings are counted **net**:
|
|
133
|
+
every token the judge reads is subtracted. The method was chosen on one agent and committed
|
|
134
|
+
(`0827ece`) before any other agent was judged.
|
|
135
|
+
|
|
136
|
+
| Dataset | Extra net saving over the step limit, with 95% interval |
|
|
137
|
+
|---|---|
|
|
138
|
+
| SWE-bench (GPT-5-mini) | **−6.3%** [−18.9%, +1.9%] |
|
|
139
|
+
| τ-bench (4 groups) | **−4.2%** [−8.2%, −1.5%] |
|
|
140
|
+
|
|
141
|
+
What we learned:
|
|
142
|
+
|
|
143
|
+
- **A threshold tuned on one agent did not travel.** The chosen method asked the judge only when the
|
|
144
|
+
cheap score passed a level set on Devstral's long runs. On the other agents that level was almost
|
|
145
|
+
never reached (0–0.3% of steps), so the method rarely stopped anything.
|
|
146
|
+
- **A judge on every step is expensive.** On short customer-service tasks it used about as many
|
|
147
|
+
tokens as the agent itself.
|
|
148
|
+
- **Even ignoring its cost, the judge did not spot stuck runs better than counting steps**: for
|
|
149
|
+
example 7.3% vs 9.3% of tokens saved on GPT-5-mini.
|
|
150
|
+
- The guarantee held on every group. The whole experiment took 73,812 judgments, for about $3.91.
|
|
151
|
+
|
|
152
|
+
Full report: [eval/results/judge/results.md](eval/results/judge/results.md)
|
|
153
|
+
|
|
154
|
+
## How the results are kept honest
|
|
155
|
+
|
|
156
|
+
- **Chosen in advance.** The method was picked using one agent's runs and committed *before* any
|
|
157
|
+
test runs were scored (commits `8b8eaf3` and, for the judge, `0827ece`).
|
|
158
|
+
- **Mistakes stay visible.** The first results were committed exactly as they came out, including a
|
|
159
|
+
flaw in how runs were split (`e655b96`). The fix is a separate, documented commit (`76e8cdc`), with
|
|
160
|
+
before-and-after numbers in [CORRECTIONS.md](eval/results/CORRECTIONS.md).
|
|
161
|
+
- **Tested on unseen runs.** The stop line is always checked on runs it was not set from.
|
|
162
|
+
- **Reproducible.** Every number above comes from `eval/results/` and comes out identical on every run.
|
|
163
|
+
|
|
164
|
+
## How the guarantee works
|
|
165
|
+
|
|
166
|
+
1. Each step gets a stuck score. A run's score is the highest score it reaches.
|
|
167
|
+
2. Take *n* past successful runs and sort their scores. The stop line is the *k*-th smallest score,
|
|
168
|
+
where *k* = ⌈(*n* + 1)(1 − α)⌉ and α is the share of good runs you accept losing (for example 5%).
|
|
169
|
+
3. A new successful run then crosses the stop line with probability at most α.
|
|
170
|
+
|
|
171
|
+
With α = 5%, LoopBrake needs at least 19 past successful runs. With fewer, it only watches and
|
|
172
|
+
never stops anything.
|
|
173
|
+
|
|
174
|
+
The guarantee holds when new runs look like the past ones (same agent, same kind of tasks), and when
|
|
175
|
+
the past runs come from different tasks. We learned the second condition the hard way: see
|
|
176
|
+
[CORRECTIONS.md](eval/results/CORRECTIONS.md).
|
|
177
|
+
|
|
178
|
+
## Reproduce
|
|
179
|
+
|
|
180
|
+
Requires [uv](https://docs.astral.sh/uv/) and Python 3.11+.
|
|
181
|
+
|
|
182
|
+
```bash
|
|
183
|
+
git clone https://github.com/SahilSelokar/LoopBrake && cd LoopBrake
|
|
184
|
+
uv run pytest # the test suite
|
|
185
|
+
uv run python eval/fetch.py # one-time download, about 450 MB, into ~/.loopbrake/data
|
|
186
|
+
uv run python eval/run.py --final # about 1 minute; writes eval/results/
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
The progress judge needs a [Typesafe](https://typesafe.ai) API key in `TYPESAFE_API_KEY` (or in
|
|
190
|
+
`~/.loopbrake/typesafe_key`). Judging every public step costs about $4:
|
|
191
|
+
|
|
192
|
+
```bash
|
|
193
|
+
uv run python eval/tasks.py # task texts for the judge
|
|
194
|
+
uv run python eval/judge.py --group swe-devstral # one group at a time; resumable
|
|
195
|
+
uv run python eval/judge_eval.py --final # reads stored answers only; writes eval/results/judge/
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
## Repository layout
|
|
199
|
+
|
|
200
|
+
```text
|
|
201
|
+
src/loopbrake/ scoring core: stuck signals, stop-line rule, run readers (standard library only)
|
|
202
|
+
eval/ the experiments: fetch.py downloads the data, run.py replays runs, judge.py asks the
|
|
203
|
+
progress judge, judge_eval.py scores its answers; results/ holds the published numbers
|
|
204
|
+
specs/ design: constitution, roadmap, and the spec, plan, research and tasks of each experiment
|
|
205
|
+
tests/ tests, including a check of the guarantee on simulated data
|
|
206
|
+
liveness.py the original naive rule, kept as the baseline
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
## Roadmap
|
|
210
|
+
|
|
211
|
+
| Phase | What | Status |
|
|
212
|
+
|---|---|---|
|
|
213
|
+
| 1 | **Experiment**: does it work on real runs? | Done: NO-GO for cheap signals |
|
|
214
|
+
| 1b | **Progress judge**: a hosted decision model judges whether each step moved the run forward | Done: NO-GO |
|
|
215
|
+
| 2 | **Python package**: `pip install loopbrake`; a stop line on run length, with a guarantee and a readable reason | Built (v0.1.0); first PyPI release pending |
|
|
216
|
+
| 3 | **Claude Code plugin**: stop stuck sessions live, calibrated on your own history | Planned |
|
|
217
|
+
| 4 | **Observability**: live dashboard, plus export to Datadog, Grafana and others via OpenTelemetry | Planned |
|
|
218
|
+
| 5 | **Launch**: a demo agent, the public release and a video | Planned |
|
|
219
|
+
|
|
220
|
+
The full plan is in [specs/roadmap.md](specs/roadmap.md).
|
|
221
|
+
|
|
222
|
+
## Status
|
|
223
|
+
|
|
224
|
+
v0.1.0 is built and tested. The first PyPI release comes once publishing is set up; until then,
|
|
225
|
+
install from GitHub (see "Use it"). Its stop rule is a stop line on run length, set from your own past
|
|
226
|
+
successful runs, with a guaranteed limit on stopping good runs.
|
|
227
|
+
|
|
228
|
+
## License
|
|
229
|
+
|
|
230
|
+
[MIT](LICENSE). One test fixture is a trimmed τ-bench run, used under τ-bench's own MIT license
|
|
231
|
+
(see [tests/fixtures/NOTICE.md](tests/fixtures/NOTICE.md)).
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "loopbrake"
|
|
3
|
+
dynamic = ["version"]
|
|
4
|
+
description = "Stops stuck AI agent runs, with a guaranteed limit on stopping good ones."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.11"
|
|
7
|
+
license = "MIT"
|
|
8
|
+
license-files = ["LICENSE"]
|
|
9
|
+
authors = [{ name = "Sahil Selokar" }]
|
|
10
|
+
keywords = ["ai agents", "llm", "stop controller", "conformal prediction", "claude code", "agent loops"]
|
|
11
|
+
classifiers = [
|
|
12
|
+
"Development Status :: 3 - Alpha",
|
|
13
|
+
"Intended Audience :: Developers",
|
|
14
|
+
"Operating System :: OS Independent",
|
|
15
|
+
"Programming Language :: Python :: 3",
|
|
16
|
+
"Programming Language :: Python :: 3.11",
|
|
17
|
+
"Programming Language :: Python :: 3.12",
|
|
18
|
+
"Programming Language :: Python :: 3.13",
|
|
19
|
+
"Topic :: Software Development :: Libraries",
|
|
20
|
+
]
|
|
21
|
+
dependencies = []
|
|
22
|
+
|
|
23
|
+
[project.optional-dependencies]
|
|
24
|
+
agent-sdk = ["claude-agent-sdk"]
|
|
25
|
+
|
|
26
|
+
[project.scripts]
|
|
27
|
+
loopbrake = "loopbrake.cli:main"
|
|
28
|
+
|
|
29
|
+
[project.urls]
|
|
30
|
+
Homepage = "https://github.com/SahilSelokar/LoopBrake"
|
|
31
|
+
Source = "https://github.com/SahilSelokar/LoopBrake"
|
|
32
|
+
Results = "https://github.com/SahilSelokar/LoopBrake/tree/main/eval/results"
|
|
33
|
+
|
|
34
|
+
[dependency-groups]
|
|
35
|
+
dev = ["pytest"]
|
|
36
|
+
|
|
37
|
+
[build-system]
|
|
38
|
+
requires = ["hatchling"]
|
|
39
|
+
build-backend = "hatchling.build"
|
|
40
|
+
|
|
41
|
+
[tool.hatch.version]
|
|
42
|
+
path = "src/loopbrake/__init__.py"
|
|
43
|
+
|
|
44
|
+
[tool.hatch.build.targets.wheel]
|
|
45
|
+
packages = ["src/loopbrake"]
|
|
46
|
+
|
|
47
|
+
[tool.hatch.build.targets.sdist]
|
|
48
|
+
include = ["src/", "README.md", "LICENSE", "pyproject.toml"]
|
|
49
|
+
|
|
50
|
+
[tool.pytest.ini_options]
|
|
51
|
+
testpaths = ["tests"]
|
|
52
|
+
pythonpath = ["."]
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""LoopBrake: stops stuck AI agent runs, with a guaranteed limit on stopping good ones."""
|
|
2
|
+
|
|
3
|
+
__version__ = "0.1.0"
|
|
4
|
+
|
|
5
|
+
from loopbrake.brake import Brake, Decision, start # noqa: E402
|
|
6
|
+
from loopbrake.calibration import calibrate # noqa: E402
|
|
7
|
+
|
|
8
|
+
__all__ = ["Brake", "Decision", "start", "calibrate", "__version__"]
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""Adapter for Claude's agent toolkit (`claude-agent-sdk`). Contract: specs/003-core-package/contracts/agent-sdk.md.
|
|
2
|
+
|
|
3
|
+
Thin by design (constitution Principle V): it turns toolkit hook events into brake.step() calls and the
|
|
4
|
+
brake's decision into the toolkit's stop reply. All stop logic lives in loopbrake.brake.
|
|
5
|
+
"""
|
|
6
|
+
import json
|
|
7
|
+
import warnings
|
|
8
|
+
|
|
9
|
+
from loopbrake.brake import start
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _text(value):
|
|
13
|
+
"""A tool's output as text: strings as they are, content blocks joined, anything else as JSON."""
|
|
14
|
+
if value is None:
|
|
15
|
+
return ""
|
|
16
|
+
if isinstance(value, str):
|
|
17
|
+
return value
|
|
18
|
+
if isinstance(value, list) and all(isinstance(b, dict) and "text" in b for b in value):
|
|
19
|
+
return "\n".join(b["text"] for b in value)
|
|
20
|
+
return json.dumps(value, sort_keys=True, ensure_ascii=False, default=str)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class Hooks:
|
|
24
|
+
"""One brake per session; a new user prompt starts a new run (as Claude Code turns are calibrated)."""
|
|
25
|
+
|
|
26
|
+
def __init__(self, project="default", home=None):
|
|
27
|
+
self.project, self.home, self.brakes, self._warned = project, home, {}, False
|
|
28
|
+
|
|
29
|
+
def _brake(self, session):
|
|
30
|
+
b = self.brakes.get(session)
|
|
31
|
+
if b is None or b.ended:
|
|
32
|
+
b = self.brakes[session] = start(self.project, session=session, home=self.home)
|
|
33
|
+
return b
|
|
34
|
+
|
|
35
|
+
async def user_prompt_submit(self, input_data, tool_use_id=None, context=None):
|
|
36
|
+
try:
|
|
37
|
+
session = input_data.get("session_id") or "default"
|
|
38
|
+
old = self.brakes.pop(session, None)
|
|
39
|
+
if old:
|
|
40
|
+
old.end()
|
|
41
|
+
self.brakes[session] = start(self.project, session=session, home=self.home)
|
|
42
|
+
except Exception as e:
|
|
43
|
+
self._fail(e)
|
|
44
|
+
return {}
|
|
45
|
+
|
|
46
|
+
async def post_tool_use(self, input_data, tool_use_id=None, context=None):
|
|
47
|
+
try:
|
|
48
|
+
tool = input_data.get("tool_name") or "tool"
|
|
49
|
+
action = f"{tool} {json.dumps(input_data.get('tool_input') or {}, sort_keys=True, ensure_ascii=False)}"
|
|
50
|
+
decision = self._brake(input_data.get("session_id") or "default").step(
|
|
51
|
+
action, _text(input_data.get("tool_response")), tool=tool)
|
|
52
|
+
if decision.stop:
|
|
53
|
+
return {"continue_": False, "stopReason": decision.reason}
|
|
54
|
+
except Exception as e:
|
|
55
|
+
self._fail(e)
|
|
56
|
+
return {}
|
|
57
|
+
|
|
58
|
+
async def stop(self, input_data, tool_use_id=None, context=None):
|
|
59
|
+
try:
|
|
60
|
+
b = self.brakes.pop(input_data.get("session_id") or "default", None)
|
|
61
|
+
if b:
|
|
62
|
+
b.end()
|
|
63
|
+
except Exception as e:
|
|
64
|
+
self._fail(e)
|
|
65
|
+
return {}
|
|
66
|
+
|
|
67
|
+
def _fail(self, e):
|
|
68
|
+
if not self._warned:
|
|
69
|
+
self._warned = True
|
|
70
|
+
warnings.warn(f"loopbrake: agent hook error ({e!r}); the agent continues unbraked", RuntimeWarning, stacklevel=2)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def hooks(project="default", *, home=None):
|
|
74
|
+
"""The `hooks=` value for ClaudeAgentOptions: UserPromptSubmit, PostToolUse and Stop."""
|
|
75
|
+
try:
|
|
76
|
+
from claude_agent_sdk import HookMatcher
|
|
77
|
+
except ImportError as e:
|
|
78
|
+
raise ImportError('loopbrake.agent_sdk needs the agent toolkit: pip install "loopbrake[agent-sdk]"') from e
|
|
79
|
+
h = Hooks(project, home)
|
|
80
|
+
return {
|
|
81
|
+
"UserPromptSubmit": [HookMatcher(hooks=[h.user_prompt_submit])],
|
|
82
|
+
"PostToolUse": [HookMatcher(hooks=[h.post_tool_use])],
|
|
83
|
+
"Stop": [HookMatcher(hooks=[h.stop])],
|
|
84
|
+
}
|