loopbrake 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- loopbrake/__init__.py +8 -0
- loopbrake/agent_sdk.py +84 -0
- loopbrake/brake.py +123 -0
- loopbrake/calibration.py +107 -0
- loopbrake/cli.py +111 -0
- loopbrake/conformal.py +21 -0
- loopbrake/records.py +103 -0
- loopbrake/signals.py +226 -0
- loopbrake/traces.py +186 -0
- loopbrake-0.1.0.dist-info/METADATA +255 -0
- loopbrake-0.1.0.dist-info/RECORD +14 -0
- loopbrake-0.1.0.dist-info/WHEEL +4 -0
- loopbrake-0.1.0.dist-info/entry_points.txt +2 -0
- loopbrake-0.1.0.dist-info/licenses/LICENSE +21 -0
loopbrake/__init__.py
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""LoopBrake: stops stuck AI agent runs, with a guaranteed limit on stopping good ones."""
|
|
2
|
+
|
|
3
|
+
__version__ = "0.1.0"
|
|
4
|
+
|
|
5
|
+
from loopbrake.brake import Brake, Decision, start # noqa: E402
|
|
6
|
+
from loopbrake.calibration import calibrate # noqa: E402
|
|
7
|
+
|
|
8
|
+
__all__ = ["Brake", "Decision", "start", "calibrate", "__version__"]
|
loopbrake/agent_sdk.py
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""Adapter for Claude's agent toolkit (`claude-agent-sdk`). Contract: specs/003-core-package/contracts/agent-sdk.md.
|
|
2
|
+
|
|
3
|
+
Thin by design (constitution Principle V): it turns toolkit hook events into brake.step() calls and the
|
|
4
|
+
brake's decision into the toolkit's stop reply. All stop logic lives in loopbrake.brake.
|
|
5
|
+
"""
|
|
6
|
+
import json
|
|
7
|
+
import warnings
|
|
8
|
+
|
|
9
|
+
from loopbrake.brake import start
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _text(value):
|
|
13
|
+
"""A tool's output as text: strings as they are, content blocks joined, anything else as JSON."""
|
|
14
|
+
if value is None:
|
|
15
|
+
return ""
|
|
16
|
+
if isinstance(value, str):
|
|
17
|
+
return value
|
|
18
|
+
if isinstance(value, list) and all(isinstance(b, dict) and "text" in b for b in value):
|
|
19
|
+
return "\n".join(b["text"] for b in value)
|
|
20
|
+
return json.dumps(value, sort_keys=True, ensure_ascii=False, default=str)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class Hooks:
|
|
24
|
+
"""One brake per session; a new user prompt starts a new run (as Claude Code turns are calibrated)."""
|
|
25
|
+
|
|
26
|
+
def __init__(self, project="default", home=None):
|
|
27
|
+
self.project, self.home, self.brakes, self._warned = project, home, {}, False
|
|
28
|
+
|
|
29
|
+
def _brake(self, session):
|
|
30
|
+
b = self.brakes.get(session)
|
|
31
|
+
if b is None or b.ended:
|
|
32
|
+
b = self.brakes[session] = start(self.project, session=session, home=self.home)
|
|
33
|
+
return b
|
|
34
|
+
|
|
35
|
+
async def user_prompt_submit(self, input_data, tool_use_id=None, context=None):
|
|
36
|
+
try:
|
|
37
|
+
session = input_data.get("session_id") or "default"
|
|
38
|
+
old = self.brakes.pop(session, None)
|
|
39
|
+
if old:
|
|
40
|
+
old.end()
|
|
41
|
+
self.brakes[session] = start(self.project, session=session, home=self.home)
|
|
42
|
+
except Exception as e:
|
|
43
|
+
self._fail(e)
|
|
44
|
+
return {}
|
|
45
|
+
|
|
46
|
+
async def post_tool_use(self, input_data, tool_use_id=None, context=None):
|
|
47
|
+
try:
|
|
48
|
+
tool = input_data.get("tool_name") or "tool"
|
|
49
|
+
action = f"{tool} {json.dumps(input_data.get('tool_input') or {}, sort_keys=True, ensure_ascii=False)}"
|
|
50
|
+
decision = self._brake(input_data.get("session_id") or "default").step(
|
|
51
|
+
action, _text(input_data.get("tool_response")), tool=tool)
|
|
52
|
+
if decision.stop:
|
|
53
|
+
return {"continue_": False, "stopReason": decision.reason}
|
|
54
|
+
except Exception as e:
|
|
55
|
+
self._fail(e)
|
|
56
|
+
return {}
|
|
57
|
+
|
|
58
|
+
async def stop(self, input_data, tool_use_id=None, context=None):
|
|
59
|
+
try:
|
|
60
|
+
b = self.brakes.pop(input_data.get("session_id") or "default", None)
|
|
61
|
+
if b:
|
|
62
|
+
b.end()
|
|
63
|
+
except Exception as e:
|
|
64
|
+
self._fail(e)
|
|
65
|
+
return {}
|
|
66
|
+
|
|
67
|
+
def _fail(self, e):
|
|
68
|
+
if not self._warned:
|
|
69
|
+
self._warned = True
|
|
70
|
+
warnings.warn(f"loopbrake: agent hook error ({e!r}); the agent continues unbraked", RuntimeWarning, stacklevel=2)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def hooks(project="default", *, home=None):
|
|
74
|
+
"""The `hooks=` value for ClaudeAgentOptions: UserPromptSubmit, PostToolUse and Stop."""
|
|
75
|
+
try:
|
|
76
|
+
from claude_agent_sdk import HookMatcher
|
|
77
|
+
except ImportError as e:
|
|
78
|
+
raise ImportError('loopbrake.agent_sdk needs the agent toolkit: pip install "loopbrake[agent-sdk]"') from e
|
|
79
|
+
h = Hooks(project, home)
|
|
80
|
+
return {
|
|
81
|
+
"UserPromptSubmit": [HookMatcher(hooks=[h.user_prompt_submit])],
|
|
82
|
+
"PostToolUse": [HookMatcher(hooks=[h.post_tool_use])],
|
|
83
|
+
"Stop": [HookMatcher(hooks=[h.stop])],
|
|
84
|
+
}
|
loopbrake/brake.py
ADDED
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""The brake: one per run. Stops a run once it goes past the stop line set from your own past
|
|
2
|
+
successful runs, and says why. Contract: specs/003-core-package/contracts/python-api.md.
|
|
3
|
+
|
|
4
|
+
v1's stop rule is the calibrated step budget (constitution 2.2.0). The stuck signals only explain.
|
|
5
|
+
"""
|
|
6
|
+
import uuid
|
|
7
|
+
import warnings
|
|
8
|
+
from typing import NamedTuple
|
|
9
|
+
|
|
10
|
+
from loopbrake import calibration, records
|
|
11
|
+
from loopbrake.signals import Step, method
|
|
12
|
+
|
|
13
|
+
EXCERPT = 200
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class Decision(NamedTuple):
|
|
17
|
+
stop: bool
|
|
18
|
+
step: int
|
|
19
|
+
reason: str
|
|
20
|
+
watch_only: bool
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _decide(steps, stop_line):
|
|
24
|
+
"""The certified rule, unchanged from the experiments: stop when the `steps` score passes the line."""
|
|
25
|
+
# ponytail: rescores every step so far (O(t)) to call the certified scorer literally; about 0.1 ms
|
|
26
|
+
# at 250 steps. A running count is the upgrade, with tests/test_replay_data.py guarding agreement.
|
|
27
|
+
return stop_line is not None and method("steps")(steps)[-1][0] > stop_line
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class Brake:
|
|
31
|
+
def __init__(self, project, session, run, home, calibration, record=True):
|
|
32
|
+
self.project, self.session, self.run, self.home = project, session, run, home
|
|
33
|
+
self.calibration = calibration
|
|
34
|
+
line = calibration.get("stop_line") if calibration and not calibration.get("watch_only") else None
|
|
35
|
+
self.stop_line = line
|
|
36
|
+
self.watch_only = line is None
|
|
37
|
+
self.steps, self.stopped, self.reason, self.ended, self.tokens = [], False, "", False, None
|
|
38
|
+
self._warned = False
|
|
39
|
+
self._writer = records.RunWriter(home / "runs" / f"{session}.jsonl")
|
|
40
|
+
self._writer.on = record
|
|
41
|
+
cal = calibration or {}
|
|
42
|
+
self._record("run_start", project=project, calibration={
|
|
43
|
+
"method": "steps", "alpha": cal.get("alpha"), "n": cal.get("n"), "k": cal.get("k"),
|
|
44
|
+
"stop_line": self.stop_line, "watch_only": self.watch_only})
|
|
45
|
+
|
|
46
|
+
# ---- public ----
|
|
47
|
+
|
|
48
|
+
def step(self, action, result="", *, tool=None, tokens=None, error=None):
|
|
49
|
+
"""Report one step. Returns a Decision; once it says stop, it keeps saying stop."""
|
|
50
|
+
try:
|
|
51
|
+
return self._step(action, result, tool, tokens, error)
|
|
52
|
+
except Exception as e: # never hurt the host agent (spec FR-005)
|
|
53
|
+
self._fail(e)
|
|
54
|
+
return Decision(self.stopped, len(self.steps), self.reason, self.watch_only)
|
|
55
|
+
|
|
56
|
+
def end(self, status=None):
|
|
57
|
+
try:
|
|
58
|
+
if self.ended:
|
|
59
|
+
return
|
|
60
|
+
self.ended = True
|
|
61
|
+
status = status or ("stopped" if self.stopped else "finished")
|
|
62
|
+
self._record("run_end", status=status, steps=len(self.steps), tokens=self.tokens)
|
|
63
|
+
except Exception as e:
|
|
64
|
+
self._fail(e)
|
|
65
|
+
|
|
66
|
+
def __enter__(self):
|
|
67
|
+
return self
|
|
68
|
+
|
|
69
|
+
def __exit__(self, exc_type, exc, tb):
|
|
70
|
+
self.end("stopped" if self.stopped else "interrupted" if exc_type else "finished")
|
|
71
|
+
return False
|
|
72
|
+
|
|
73
|
+
# ---- internals ----
|
|
74
|
+
|
|
75
|
+
def _step(self, action, result, tool, tokens, error):
|
|
76
|
+
action = str(action)
|
|
77
|
+
self.steps.append(Step(action, "" if result is None else str(result), error, int(tokens or 0)))
|
|
78
|
+
if tokens is not None:
|
|
79
|
+
self.tokens = (self.tokens or 0) + int(tokens)
|
|
80
|
+
t = len(self.steps)
|
|
81
|
+
self._record("step", step=t, tool=tool, action_excerpt=action[:EXCERPT], tokens=tokens, error=error)
|
|
82
|
+
if self.stopped:
|
|
83
|
+
return Decision(True, t, self.reason, False)
|
|
84
|
+
if _decide(self.steps, self.stop_line):
|
|
85
|
+
self.stopped = True
|
|
86
|
+
self.reason = self._explain(t)
|
|
87
|
+
self._record("stop", step=t, stop_line=self.stop_line, reason=self.reason)
|
|
88
|
+
return Decision(True, t, self.reason, False)
|
|
89
|
+
return Decision(False, t, "", self.watch_only)
|
|
90
|
+
|
|
91
|
+
def _explain(self, t):
|
|
92
|
+
cal = self.calibration or {}
|
|
93
|
+
reason = f"stopped at step {t}: past the stop line of {self.stop_line} steps"
|
|
94
|
+
if cal.get("n") and cal.get("alpha"):
|
|
95
|
+
reason += f" set from your {cal['n']} past successful runs (α {cal['alpha']:.0%})"
|
|
96
|
+
seen = method("max", lam=0.9)(self.steps)[-1][1] # explains only; never decides
|
|
97
|
+
return f"{reason}; {seen}" if seen else reason
|
|
98
|
+
|
|
99
|
+
def _record(self, event, **fields):
|
|
100
|
+
self._writer.write({"event": event, "session": self.session, "run": self.run} | fields)
|
|
101
|
+
|
|
102
|
+
def _fail(self, e):
|
|
103
|
+
self.stop_line, self.watch_only = None, True
|
|
104
|
+
if not self._warned:
|
|
105
|
+
self._warned = True
|
|
106
|
+
warnings.warn(f"loopbrake: internal error ({e!r}); watching only for the rest of this run", RuntimeWarning, stacklevel=3)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def replay(run, calibration):
|
|
110
|
+
"""Feed a recorded run through a brake, writing no records. Returns the stop step, or None."""
|
|
111
|
+
b = Brake("replay", "replay", run.run, records.home(), calibration, record=False)
|
|
112
|
+
for s in run.steps:
|
|
113
|
+
if b.step(s.action, s.observation, tokens=s.tokens, error=s.error).stop:
|
|
114
|
+
return len(b.steps)
|
|
115
|
+
return None
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def start(project="default", *, session=None, run=None, home=None):
|
|
119
|
+
"""Start one run. Missing or damaged calibration means watch-only, never an exception."""
|
|
120
|
+
if not records.valid_project(project):
|
|
121
|
+
raise ValueError(f"project names may use letters, digits, '.', '_' and '-' (got {project!r})")
|
|
122
|
+
h = records.home(home)
|
|
123
|
+
return Brake(project, session or uuid.uuid4().hex, run or uuid.uuid4().hex[:12], h, calibration.load(project, h))
|
loopbrake/calibration.py
ADDED
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"""Calibration: set the stop line from your own past successful runs (constitution Principle I).
|
|
2
|
+
|
|
3
|
+
Contract: specs/003-core-package/contracts/python-api.md and records.md.
|
|
4
|
+
"""
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import math
|
|
8
|
+
import warnings
|
|
9
|
+
from datetime import date
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from loopbrake import __version__, records
|
|
13
|
+
from loopbrake.conformal import rank, threshold
|
|
14
|
+
from loopbrake.traces import claude_code_turns, read_runs
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def path(project, home):
|
|
18
|
+
return home / "calibration" / f"{project}.json"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def load(project, home):
|
|
22
|
+
"""The calibration record for a project, or None if there is none or it is damaged (watch-only)."""
|
|
23
|
+
p = path(project, home)
|
|
24
|
+
if not p.exists():
|
|
25
|
+
return None
|
|
26
|
+
try:
|
|
27
|
+
rec = json.loads(p.read_text())
|
|
28
|
+
line = rec["stop_line"]
|
|
29
|
+
ok = (rec["method"] == "steps" and isinstance(rec["watch_only"], bool) and isinstance(rec["n"], int)
|
|
30
|
+
and 0 < rec["alpha"] < 1 and (line is None or (isinstance(line, int) and line >= 0)))
|
|
31
|
+
if not ok:
|
|
32
|
+
raise ValueError("unexpected values")
|
|
33
|
+
return rec
|
|
34
|
+
except Exception as e:
|
|
35
|
+
warnings.warn(f"loopbrake: calibration file {p} is damaged ({e}); watching only", RuntimeWarning, stacklevel=3)
|
|
36
|
+
return None
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def runs_needed(alpha):
|
|
40
|
+
"""The fewest successful runs that give a stop line at this alpha (fewer means watch-only)."""
|
|
41
|
+
n = 1
|
|
42
|
+
while rank(n, alpha) > n:
|
|
43
|
+
n += 1
|
|
44
|
+
return n
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _from_runs_file(src, exclude):
|
|
48
|
+
"""One successful run per task (the first by run id), skipping excluded ids."""
|
|
49
|
+
runs, _ = read_runs(src)
|
|
50
|
+
first, seen, excluded = {}, 0, 0
|
|
51
|
+
for r in sorted(runs, key=lambda r: r.run):
|
|
52
|
+
if not r.steps:
|
|
53
|
+
continue
|
|
54
|
+
seen += 1
|
|
55
|
+
if r.run in exclude:
|
|
56
|
+
excluded += 1
|
|
57
|
+
continue
|
|
58
|
+
if r.success and r.task not in first:
|
|
59
|
+
first[r.task] = len(r.steps)
|
|
60
|
+
source = {"kind": "runs-file", "sha256": hashlib.sha256(src.read_bytes()).hexdigest(), "runs_seen": seen, "excluded": excluded}
|
|
61
|
+
return list(first.values()), source
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _from_claude_code(folder, exclude):
|
|
65
|
+
"""Every turn is its own task; success follows the experiments' rule (not interrupted, not excluded)."""
|
|
66
|
+
files = sorted(folder.glob("*.jsonl"))
|
|
67
|
+
if not files:
|
|
68
|
+
raise FileNotFoundError(f"no Claude Code session files (*.jsonl) in {folder}")
|
|
69
|
+
lengths, seen, excluded = [], 0, 0
|
|
70
|
+
for f in files:
|
|
71
|
+
for r in claude_code_turns(f, exclude)[0]:
|
|
72
|
+
if not r.steps:
|
|
73
|
+
continue
|
|
74
|
+
seen += 1
|
|
75
|
+
excluded += r.run in exclude
|
|
76
|
+
if r.success:
|
|
77
|
+
lengths.append(len(r.steps))
|
|
78
|
+
listing = "\n".join(f"{f.name}\t{f.stat().st_size}" for f in files) # names and sizes only, never content
|
|
79
|
+
source = {"kind": "claude-code", "sha256": hashlib.sha256(listing.encode()).hexdigest(), "runs_seen": seen, "excluded": excluded}
|
|
80
|
+
return lengths, source
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def calibrate(source, *, project="default", alpha=0.05, home=None):
|
|
84
|
+
"""Set a project's stop line from past runs and save it. Returns the calibration record.
|
|
85
|
+
|
|
86
|
+
`source`: a runs file (common format) or a Claude Code project folder. The record holds counts and a
|
|
87
|
+
fingerprint only, never text from the source (spec FR-007).
|
|
88
|
+
"""
|
|
89
|
+
if not records.valid_project(project):
|
|
90
|
+
raise ValueError(f"project names may use letters, digits, '.', '_' and '-' (got {project!r})")
|
|
91
|
+
h = records.home(home)
|
|
92
|
+
src = Path(source).expanduser()
|
|
93
|
+
if not src.exists():
|
|
94
|
+
raise FileNotFoundError(f"no such file or folder: {src}")
|
|
95
|
+
exclude = records.read_exclude(h)
|
|
96
|
+
lengths, src_info = _from_claude_code(src, exclude) if src.is_dir() else _from_runs_file(src, exclude)
|
|
97
|
+
line = threshold(lengths, alpha)
|
|
98
|
+
rec = {
|
|
99
|
+
"v": 1, "project": project, "method": "steps", "alpha": alpha,
|
|
100
|
+
"n": len(lengths), "k": rank(len(lengths), alpha),
|
|
101
|
+
"stop_line": None if line == math.inf else int(line), "watch_only": line == math.inf,
|
|
102
|
+
"source": src_info, "created": date.today().isoformat(), "version": __version__,
|
|
103
|
+
}
|
|
104
|
+
p = path(project, h)
|
|
105
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
106
|
+
p.write_text(json.dumps(rec, indent=1) + "\n")
|
|
107
|
+
return rec
|
loopbrake/cli.py
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
"""The `loopbrake` command. Contract: specs/003-core-package/contracts/cli.md."""
|
|
2
|
+
import argparse
|
|
3
|
+
import os
|
|
4
|
+
import sys
|
|
5
|
+
|
|
6
|
+
from loopbrake import __version__, calibration, records
|
|
7
|
+
from loopbrake.brake import replay
|
|
8
|
+
from loopbrake.traces import read_runs
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _calibrate(args):
|
|
12
|
+
rec = calibration.calibrate(args.source, project=args.project, alpha=args.alpha)
|
|
13
|
+
if rec["watch_only"]:
|
|
14
|
+
more = calibration.runs_needed(args.alpha) - rec["n"]
|
|
15
|
+
print(f"watch-only: {rec['n']} successful runs found; need {more} more for α {args.alpha:.0%}")
|
|
16
|
+
else:
|
|
17
|
+
print(f"stop line: {rec['stop_line']} steps (from {rec['n']} successful runs, k = {rec['k']}, α {args.alpha:.0%})")
|
|
18
|
+
print(f"saved: {calibration.path(args.project, records.home())}")
|
|
19
|
+
return 0
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _status(args):
|
|
23
|
+
h = records.home()
|
|
24
|
+
projects = [args.project] if args.project else sorted(p.stem for p in (h / "calibration").glob("*.json")) or ["default"]
|
|
25
|
+
for project in projects:
|
|
26
|
+
rec = calibration.load(project, h)
|
|
27
|
+
if rec is None:
|
|
28
|
+
line = "no calibration (watch-only)"
|
|
29
|
+
elif rec["watch_only"]:
|
|
30
|
+
line = f"watch-only (only {rec['n']} successful runs)"
|
|
31
|
+
else:
|
|
32
|
+
line = f"stop line: {rec['stop_line']} steps (from {rec['n']} successful runs, α {rec['alpha']:.0%}, {rec['source']['kind']}, {rec['created']})"
|
|
33
|
+
st = records.status(h, project)
|
|
34
|
+
print(f"project {project}: {line}")
|
|
35
|
+
print(f" runs watched: {st['watched']} (of {st['runs']} recorded)")
|
|
36
|
+
print(f" runs stopped: {st['stopped']}")
|
|
37
|
+
print(f" mistaken stops: {st['mistaken']} of an allowance of {st['allowance']:.1f}")
|
|
38
|
+
return 0
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _feedback(args):
|
|
42
|
+
verdict = "mistaken_stop" if args.mistaken else "exclude"
|
|
43
|
+
try:
|
|
44
|
+
records.add_feedback(records.home(), args.run, verdict)
|
|
45
|
+
except LookupError as e:
|
|
46
|
+
print(f"loopbrake: {e}", file=sys.stderr)
|
|
47
|
+
return 1
|
|
48
|
+
print(f"recorded: {args.run} {'marked as a mistaken stop' if args.mistaken else 'left out of future calibration'}")
|
|
49
|
+
return 0
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _replay(args):
|
|
53
|
+
if args.stop_line is not None:
|
|
54
|
+
cal = {"stop_line": args.stop_line, "watch_only": False}
|
|
55
|
+
else:
|
|
56
|
+
cal = calibration.load(args.project, records.home())
|
|
57
|
+
if cal is None or cal["watch_only"]:
|
|
58
|
+
print(f"loopbrake: project {args.project} has no stop line yet; calibrate it first", file=sys.stderr)
|
|
59
|
+
return 2
|
|
60
|
+
runs, _ = read_runs(args.runs_file)
|
|
61
|
+
stopped_ok = stopped_failed = 0
|
|
62
|
+
for r in runs:
|
|
63
|
+
if not r.steps:
|
|
64
|
+
continue
|
|
65
|
+
step = replay(r, cal)
|
|
66
|
+
if step is not None:
|
|
67
|
+
stopped_ok += r.success
|
|
68
|
+
stopped_failed += not r.success
|
|
69
|
+
print(f"{r.run}\t{'-' if step is None else f'stop at {step}'}\t{'success' if r.success else 'failed'}")
|
|
70
|
+
print(f"runs {sum(1 for r in runs if r.steps)}, stopped {stopped_ok + stopped_failed} ({stopped_ok} successful, {stopped_failed} failed)")
|
|
71
|
+
return 0
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def main(argv=None):
|
|
75
|
+
ap = argparse.ArgumentParser(prog="loopbrake", description="Stops stuck AI agent runs, with a guaranteed limit on stopping good ones.")
|
|
76
|
+
ap.add_argument("--version", action="store_true", help="print the version")
|
|
77
|
+
sub = ap.add_subparsers(dest="command")
|
|
78
|
+
c = sub.add_parser("calibrate", help="set a stop line from past runs")
|
|
79
|
+
c.add_argument("source", help="a runs file, or a Claude Code project folder")
|
|
80
|
+
c.add_argument("--project", default="default")
|
|
81
|
+
c.add_argument("--alpha", type=float, default=0.05, help="highest share of good runs you accept stopping (default 0.05)")
|
|
82
|
+
s = sub.add_parser("status", help="stop line, runs watched and stopped, mistaken stops")
|
|
83
|
+
s.add_argument("--project")
|
|
84
|
+
f = sub.add_parser("feedback", help="mark a stop as a mistake, or leave a run out of calibration")
|
|
85
|
+
f.add_argument("run")
|
|
86
|
+
g = f.add_mutually_exclusive_group(required=True)
|
|
87
|
+
g.add_argument("--mistaken", action="store_true")
|
|
88
|
+
g.add_argument("--exclude", action="store_true")
|
|
89
|
+
r = sub.add_parser("replay", help="show where recorded runs would stop (writes nothing)")
|
|
90
|
+
r.add_argument("runs_file")
|
|
91
|
+
g2 = r.add_mutually_exclusive_group(required=True)
|
|
92
|
+
g2.add_argument("--project")
|
|
93
|
+
g2.add_argument("--stop-line", type=int)
|
|
94
|
+
args = ap.parse_args(argv)
|
|
95
|
+
if args.version:
|
|
96
|
+
print(f"loopbrake {__version__}")
|
|
97
|
+
return 0
|
|
98
|
+
if not args.command:
|
|
99
|
+
ap.print_help()
|
|
100
|
+
return 0
|
|
101
|
+
try:
|
|
102
|
+
return {"calibrate": _calibrate, "status": _status, "feedback": _feedback, "replay": _replay}[args.command](args)
|
|
103
|
+
except (OSError, ValueError, KeyError) as e:
|
|
104
|
+
if os.environ.get("LOOPBRAKE_DEBUG") == "1":
|
|
105
|
+
raise
|
|
106
|
+
print(f"loopbrake: {e}", file=sys.stderr)
|
|
107
|
+
return 2
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
if __name__ == "__main__":
|
|
111
|
+
sys.exit(main())
|
loopbrake/conformal.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""The stop-line rule (constitution Principle I).
|
|
2
|
+
|
|
3
|
+
Given the scores of n past successful runs, pick the stop line so that a new successful run
|
|
4
|
+
from the same agent and task mix goes over it with probability at most alpha.
|
|
5
|
+
"""
|
|
6
|
+
import math
|
|
7
|
+
from fractions import Fraction
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def rank(n, alpha):
|
|
11
|
+
"""Which calibration score becomes the stop line: the k-th smallest, k = ceil((n + 1)(1 - alpha)).
|
|
12
|
+
|
|
13
|
+
Exact fractions, so rounding can never change k.
|
|
14
|
+
"""
|
|
15
|
+
return math.ceil((n + 1) * (1 - Fraction(str(alpha))))
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def threshold(run_scores, alpha):
|
|
19
|
+
"""The stop line. math.inf when there are too few runs, so nothing is ever stopped."""
|
|
20
|
+
k = rank(len(run_scores), alpha)
|
|
21
|
+
return math.inf if k > len(run_scores) else sorted(run_scores)[k - 1]
|
loopbrake/records.py
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""Local records: where LoopBrake keeps run records and the exclude list, and how status is read back.
|
|
2
|
+
|
|
3
|
+
Format: specs/003-core-package/contracts/records.md. Nothing here touches the network (Principle VI).
|
|
4
|
+
"""
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import re
|
|
8
|
+
import warnings
|
|
9
|
+
from datetime import datetime, timezone
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
_PROJECT = re.compile(r"[A-Za-z0-9._-]{1,64}")
|
|
13
|
+
VERDICTS = ("mistaken_stop", "exclude")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def home(path=None):
|
|
17
|
+
"""The LoopBrake folder: `path`, else $LOOPBRAKE_HOME, else ~/.loopbrake."""
|
|
18
|
+
return Path(path or os.environ.get("LOOPBRAKE_HOME") or Path.home() / ".loopbrake")
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def valid_project(name):
|
|
22
|
+
"""Project names become file names, so only plain characters are allowed."""
|
|
23
|
+
return isinstance(name, str) and bool(_PROJECT.fullmatch(name)) and name not in (".", "..")
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _now():
|
|
27
|
+
return datetime.now(timezone.utc).isoformat(timespec="seconds")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class RunWriter:
|
|
31
|
+
"""Appends one event per line, readable only by the owner. If a write fails, recording turns off
|
|
32
|
+
for this writer; it never raises, so it can't hurt the agent being watched (spec FR-005)."""
|
|
33
|
+
|
|
34
|
+
def __init__(self, path):
|
|
35
|
+
self.path, self.on = Path(path), True
|
|
36
|
+
|
|
37
|
+
def write(self, event):
|
|
38
|
+
if not self.on:
|
|
39
|
+
return
|
|
40
|
+
line = (json.dumps({"v": 1, "ts": _now()} | event, ensure_ascii=False) + "\n").encode()
|
|
41
|
+
try:
|
|
42
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
43
|
+
fd = os.open(self.path, os.O_WRONLY | os.O_APPEND | os.O_CREAT, 0o600)
|
|
44
|
+
try:
|
|
45
|
+
os.write(fd, line) # one write per line, so parallel runs don't interleave
|
|
46
|
+
finally:
|
|
47
|
+
os.close(fd)
|
|
48
|
+
except OSError as e:
|
|
49
|
+
self.on = False
|
|
50
|
+
warnings.warn(f"loopbrake: can't write run records ({e}); recording is off for this run", RuntimeWarning, stacklevel=3)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _files(h):
|
|
54
|
+
return sorted((h / "runs").glob("*.jsonl")) if (h / "runs").exists() else []
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def read_events(h):
|
|
58
|
+
for path in _files(h):
|
|
59
|
+
for line in path.read_text(encoding="utf-8").splitlines():
|
|
60
|
+
try:
|
|
61
|
+
yield json.loads(line) | {"_file": path}
|
|
62
|
+
except json.JSONDecodeError:
|
|
63
|
+
continue
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def status(h, project=None):
|
|
67
|
+
"""Runs, runs watched with a stop line, stops, stops marked mistaken, and the allowance (α × watched)."""
|
|
68
|
+
events = list(read_events(h))
|
|
69
|
+
starts = {e["run"]: e for e in events if e.get("event") == "run_start" and (project is None or e.get("project") == project)}
|
|
70
|
+
watched = {r: e for r, e in starts.items() if not (e.get("calibration") or {}).get("watch_only", True)}
|
|
71
|
+
stopped = {e["run"] for e in events if e.get("event") == "stop" and e.get("run") in starts}
|
|
72
|
+
mistaken = {e["run"] for e in events if e.get("event") == "feedback" and e.get("verdict") == "mistaken_stop" and e.get("run") in starts}
|
|
73
|
+
return {
|
|
74
|
+
"runs": len(starts),
|
|
75
|
+
"watched": len(watched),
|
|
76
|
+
"stopped": len(stopped),
|
|
77
|
+
"mistaken": len(mistaken),
|
|
78
|
+
"allowance": sum(e["calibration"].get("alpha", 0) for e in watched.values()),
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def add_feedback(h, run, verdict):
|
|
83
|
+
"""Record that a stop was a mistake, or that a run should be left out of future calibration."""
|
|
84
|
+
if verdict not in VERDICTS:
|
|
85
|
+
raise ValueError(f"verdict must be one of {VERDICTS}")
|
|
86
|
+
start = next((e for e in read_events(h) if e.get("event") == "run_start" and e.get("run") == run), None)
|
|
87
|
+
if start is None:
|
|
88
|
+
raise LookupError(f"no run {run!r} in {h / 'runs'}")
|
|
89
|
+
RunWriter(start["_file"]).write({"event": "feedback", "session": start.get("session"), "run": run, "verdict": verdict})
|
|
90
|
+
if verdict == "exclude":
|
|
91
|
+
add_exclude(h, run)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def read_exclude(h):
|
|
95
|
+
path = h / "exclude.txt"
|
|
96
|
+
return set(path.read_text().split()) if path.exists() else set()
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def add_exclude(h, run_id):
|
|
100
|
+
if run_id not in read_exclude(h):
|
|
101
|
+
h.mkdir(parents=True, exist_ok=True)
|
|
102
|
+
with open(h / "exclude.txt", "a", encoding="utf-8") as f:
|
|
103
|
+
f.write(run_id + "\n")
|
loopbrake/signals.py
ADDED
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
"""Stuck scores. A method turns a run's steps into one (score, reason) pair per step.
|
|
2
|
+
|
|
3
|
+
Rules (constitution Principle II): a scorer looks only at the steps it is given, always gives
|
|
4
|
+
the same output for the same input, and keeps nothing between calls.
|
|
5
|
+
"""
|
|
6
|
+
import re
|
|
7
|
+
from collections import defaultdict
|
|
8
|
+
from functools import partial
|
|
9
|
+
from typing import NamedTuple
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class Step(NamedTuple):
|
|
13
|
+
action: str # what the agent did, e.g. 'bash {"command": "ls"}'
|
|
14
|
+
observation: str # what came back; may be empty
|
|
15
|
+
error: bool | None # True/False when the source says so, None when unknown
|
|
16
|
+
tokens: int # tokens billed for the model call that produced this step
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _short(text, n=80):
|
|
20
|
+
text = " ".join(text.split())
|
|
21
|
+
return text if len(text) <= n else text[: n - 1] + "…"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
# ---------- simple methods (research R5) ----------
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _fixed(steps):
|
|
28
|
+
"""The original liveness.py rule: stop on the 3rd identical action, or at step 21."""
|
|
29
|
+
seen = defaultdict(list)
|
|
30
|
+
out = []
|
|
31
|
+
for t, s in enumerate(steps, 1):
|
|
32
|
+
seen[s.action].append(t)
|
|
33
|
+
count = len(seen[s.action])
|
|
34
|
+
if t > 20:
|
|
35
|
+
reason = "step limit of 20 reached"
|
|
36
|
+
elif count >= 3:
|
|
37
|
+
reason = f"same action seen {count} times (steps {', '.join(map(str, seen[s.action][-3:]))})"
|
|
38
|
+
else:
|
|
39
|
+
reason = ""
|
|
40
|
+
out.append((max(count / 3, t / 21), reason))
|
|
41
|
+
return out
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _exact(steps):
|
|
45
|
+
"""Most times any single action has been repeated so far."""
|
|
46
|
+
seen = defaultdict(int)
|
|
47
|
+
top, top_action, out = 0, "", []
|
|
48
|
+
for s in steps:
|
|
49
|
+
seen[s.action] += 1
|
|
50
|
+
if seen[s.action] > top:
|
|
51
|
+
top, top_action = seen[s.action], s.action
|
|
52
|
+
out.append((top, f"same action {top} times: {_short(top_action)}"))
|
|
53
|
+
return out
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _steps(steps):
|
|
57
|
+
"""How long the run has gone on (FailFast's "Duration" control)."""
|
|
58
|
+
return [(t, f"step {t}") for t in range(1, len(steps) + 1)]
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
_BASELINE_SCORERS = {"fixed": _fixed, "exact": _exact, "steps": _steps}
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
# ---------- stuck signals (research R6): each gives a value from 0 to 1 per step ----------
|
|
65
|
+
|
|
66
|
+
WINDOW = 10 # a repeat is checked against this many previous actions
|
|
67
|
+
RECENT = 5 # reasons describe this many latest steps
|
|
68
|
+
_WORD = re.compile(r"[a-z0-9_]+")
|
|
69
|
+
_DIGITS = re.compile(r"\d+")
|
|
70
|
+
_ERROR_WORDS = re.compile(r"error|exception|traceback|not found|no such file", re.IGNORECASE)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _norm(line):
|
|
74
|
+
"""Collapse spaces and turn every number into 0, so timestamps and counters don't look new."""
|
|
75
|
+
return _DIGITS.sub("0", " ".join(line.split()))
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _fuzzy(steps):
|
|
79
|
+
"""How closely this action repeats one of the previous 10: overlap of their word sets (Jaccard).
|
|
80
|
+
|
|
81
|
+
Digits are kept, so paging through a file (lines 100-200, then 200-300) is not a repeat.
|
|
82
|
+
"""
|
|
83
|
+
words = [set(_WORD.findall(s.action.lower())) for s in steps]
|
|
84
|
+
out = []
|
|
85
|
+
for i, w in enumerate(words):
|
|
86
|
+
best, where = 0.0, None
|
|
87
|
+
for j in range(max(0, i - WINDOW), i):
|
|
88
|
+
union = w | words[j]
|
|
89
|
+
sim = len(w & words[j]) / len(union) if union else 0.0
|
|
90
|
+
if sim > 0 and sim >= best: # on a tie, point at the latest step
|
|
91
|
+
best, where = sim, j
|
|
92
|
+
detail = "" if where is None else f"{'same command as' if best == 1 else 'similar to'} step {where + 1}"
|
|
93
|
+
out.append((best, detail))
|
|
94
|
+
return out
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _stale(steps):
|
|
98
|
+
"""Share of this step's output lines the run has already seen. Empty output scores 0:
|
|
99
|
+
many useful actions (like editing a file) print nothing."""
|
|
100
|
+
seen, out = set(), []
|
|
101
|
+
for s in steps:
|
|
102
|
+
lines = [x for x in map(_norm, s.observation.splitlines()) if x]
|
|
103
|
+
if not lines:
|
|
104
|
+
out.append((0.0, ""))
|
|
105
|
+
continue
|
|
106
|
+
new = sum(1 for x in lines if x not in seen)
|
|
107
|
+
seen.update(lines)
|
|
108
|
+
out.append((1 - new / len(lines), f"{new} of {len(lines)} output lines new"))
|
|
109
|
+
return out
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _errors(steps):
|
|
113
|
+
"""1 when this step fails with an error the run has already hit (same last line, numbers masked)."""
|
|
114
|
+
first, out = {}, []
|
|
115
|
+
for t, s in enumerate(steps, 1):
|
|
116
|
+
last = next((x for x in reversed(s.observation.splitlines()) if x.strip()), "")
|
|
117
|
+
failed = s.error if s.error is not None else bool(_ERROR_WORDS.search(last))
|
|
118
|
+
if not failed:
|
|
119
|
+
out.append((0.0, ""))
|
|
120
|
+
continue
|
|
121
|
+
sig = _norm(last)[:200]
|
|
122
|
+
if sig in first:
|
|
123
|
+
out.append((1.0, f"as step {first[sig]}: {_short(sig, 60)}"))
|
|
124
|
+
else:
|
|
125
|
+
first[sig] = t
|
|
126
|
+
out.append((0.0, ""))
|
|
127
|
+
return out
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
_SIGNAL_FNS = {"fuzzy": _fuzzy, "stale": _stale, "errors": _errors}
|
|
131
|
+
_LABELS = {"fuzzy": "repeating", "stale": "nothing new", "errors": "same error again"}
|
|
132
|
+
_ALL = ("fuzzy", "stale", "errors")
|
|
133
|
+
_PARTS = {"fuzzy": ("fuzzy",), "stale": ("stale",), "errors": ("errors",), "mean": _ALL, "max": _ALL, "loop": ("fuzzy", "stale")}
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _combine(name, vals):
|
|
137
|
+
if name == "mean":
|
|
138
|
+
return sum(vals.values()) / len(vals)
|
|
139
|
+
if name == "max":
|
|
140
|
+
return max(vals.values())
|
|
141
|
+
if name == "loop": # repeating AND learning nothing
|
|
142
|
+
return vals["fuzzy"] * vals["stale"]
|
|
143
|
+
return vals[name]
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _reason(per, parts, i):
|
|
147
|
+
lo = max(0, i - RECENT + 1)
|
|
148
|
+
bits = []
|
|
149
|
+
for p in parts:
|
|
150
|
+
hits = [j for j in range(lo, i + 1) if per[p][j][0] >= 0.5]
|
|
151
|
+
if hits:
|
|
152
|
+
bits.append(f"{_LABELS[p]} in {len(hits)} of last {i - lo + 1} steps ({per[p][hits[-1]][1]})")
|
|
153
|
+
return "; ".join(bits)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _signal_scorer(name, lam, steps):
|
|
157
|
+
"""Running score S_t = lam * S_(t-1) + u_t: one odd step fades, steady stuckness adds up."""
|
|
158
|
+
# ponytail: recomputes every signal from all steps on each call, so a live run costs O(t^2) overall.
|
|
159
|
+
# Fine for the 250-step runs seen here; Phase 2's Brake can keep running state instead, with the
|
|
160
|
+
# "never looks ahead" test checking both give the same scores.
|
|
161
|
+
parts = _PARTS[name]
|
|
162
|
+
per = {p: _SIGNAL_FNS[p](steps) for p in parts}
|
|
163
|
+
out, score = [], 0.0
|
|
164
|
+
for i in range(len(steps)):
|
|
165
|
+
score = lam * score + _combine(name, {p: per[p][i][0] for p in parts})
|
|
166
|
+
out.append((score, _reason(per, parts, i)))
|
|
167
|
+
return out
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
BASELINES = tuple(_BASELINE_SCORERS)
|
|
171
|
+
SIGNALS = tuple(_PARTS)
|
|
172
|
+
METHODS = BASELINES + SIGNALS
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def method(name, lam=None):
|
|
176
|
+
"""The scorer for a method: scorer(steps) -> [(score, reason), ...], one per step."""
|
|
177
|
+
if name in BASELINES:
|
|
178
|
+
if lam is not None:
|
|
179
|
+
raise ValueError(f"{name} takes no lam")
|
|
180
|
+
return _BASELINE_SCORERS[name]
|
|
181
|
+
if name in SIGNALS:
|
|
182
|
+
if lam is None:
|
|
183
|
+
raise ValueError(f"{name} needs lam")
|
|
184
|
+
return partial(_signal_scorer, name, float(lam))
|
|
185
|
+
raise ValueError(f"unknown method {name!r}")
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
# ---------- judged scores (feature 002, research R6) ----------
|
|
189
|
+
# A judge says how likely each step was to make progress (p, 0 to 1). These scorers turn those
|
|
190
|
+
# answers into a stuck score. They stay pure: the same steps and answers always give the same scores.
|
|
191
|
+
|
|
192
|
+
JUDGED = ("judge", "judge_steps", "judge_max")
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _judged_scorer(name, lam, steps, progress, kinds=None):
|
|
196
|
+
"""progress[i] is P(progress) for step i, or None for "no opinion", which leaves the score as it was.
|
|
197
|
+
kinds[i] is an optional (kind, p) pair; it only feeds the reason text."""
|
|
198
|
+
cheap = [max(v for v, _ in vals) for vals in zip(*(_SIGNAL_FNS[s](steps) for s in _ALL))] if name == "judge_max" else None
|
|
199
|
+
out, score = [], 0.0
|
|
200
|
+
for i, p in enumerate(progress):
|
|
201
|
+
if p is not None:
|
|
202
|
+
if name == "judge":
|
|
203
|
+
u = 1 - p
|
|
204
|
+
elif name == "judge_steps":
|
|
205
|
+
u = (1 + (1 - p)) / 2 # every step costs some time; an unproductive one costs more
|
|
206
|
+
else:
|
|
207
|
+
u = max(1 - p, cheap[i])
|
|
208
|
+
score = lam * score + u
|
|
209
|
+
lo = max(0, i - RECENT + 1)
|
|
210
|
+
stuck = sum(1 for q in progress[lo : i + 1] if q is not None and q < 0.5)
|
|
211
|
+
reason = ""
|
|
212
|
+
if stuck:
|
|
213
|
+
reason = f"judge: no progress in {stuck} of last {i - lo + 1} steps"
|
|
214
|
+
if kinds and kinds[i]:
|
|
215
|
+
reason += f" ({kinds[i][0]}, {kinds[i][1]:.2f})"
|
|
216
|
+
out.append((score, reason))
|
|
217
|
+
return out
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def judged(name, lam):
|
|
221
|
+
"""Scorer for a judged method: scorer(steps, progress, kinds=None) -> [(score, reason), ...]."""
|
|
222
|
+
if name not in JUDGED:
|
|
223
|
+
raise ValueError(f"unknown judged method {name!r}")
|
|
224
|
+
if lam is None:
|
|
225
|
+
raise ValueError(f"{name} needs lam")
|
|
226
|
+
return partial(_judged_scorer, name, float(lam))
|
loopbrake/traces.py
ADDED
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
"""Read and write runs. A run is one agent attempt at one task, as a list of steps.
|
|
2
|
+
|
|
3
|
+
File format: specs/001-offline-eval/contracts/normalized-runs.md (JSON Lines, one run per line).
|
|
4
|
+
"""
|
|
5
|
+
import json
|
|
6
|
+
from collections import Counter
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import NamedTuple
|
|
9
|
+
|
|
10
|
+
from loopbrake.signals import Step
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class Run(NamedTuple):
|
|
14
|
+
group: str
|
|
15
|
+
dataset: str
|
|
16
|
+
task: str
|
|
17
|
+
run: str
|
|
18
|
+
success: bool
|
|
19
|
+
exit: str | None
|
|
20
|
+
tokens_measured: bool
|
|
21
|
+
steps: tuple # of Step, in order, never empty
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def write_runs(path, runs):
|
|
25
|
+
with open(path, "w", encoding="utf-8") as f:
|
|
26
|
+
for r in runs:
|
|
27
|
+
d = r._asdict()
|
|
28
|
+
d["steps"] = [s._asdict() for s in r.steps]
|
|
29
|
+
f.write(json.dumps(d, ensure_ascii=False) + "\n")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def read_runs(path):
|
|
33
|
+
"""Returns (runs, skipped). A bad line is skipped and counted by reason; it never raises."""
|
|
34
|
+
runs, skipped = [], Counter()
|
|
35
|
+
with open(path, encoding="utf-8") as f:
|
|
36
|
+
for line in f:
|
|
37
|
+
if not line.strip():
|
|
38
|
+
continue
|
|
39
|
+
try:
|
|
40
|
+
d = json.loads(line)
|
|
41
|
+
except json.JSONDecodeError:
|
|
42
|
+
skipped["bad json"] += 1
|
|
43
|
+
continue
|
|
44
|
+
problem = _problem(d)
|
|
45
|
+
if problem:
|
|
46
|
+
skipped[problem] += 1
|
|
47
|
+
continue
|
|
48
|
+
steps = tuple(Step(s["action"], s["observation"], s["error"], s["tokens"]) for s in d["steps"])
|
|
49
|
+
runs.append(Run(d["group"], d["dataset"], d["task"], d["run"], d["success"], d["exit"], d["tokens_measured"], steps))
|
|
50
|
+
return runs, skipped
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _problem(d):
|
|
54
|
+
if not isinstance(d, dict) or not all(isinstance(d.get(k), str) for k in ("group", "dataset", "task", "run")):
|
|
55
|
+
return "bad field"
|
|
56
|
+
if not isinstance(d.get("success"), bool) or not isinstance(d.get("tokens_measured"), bool):
|
|
57
|
+
return "bad field"
|
|
58
|
+
if d.get("exit") is not None and not isinstance(d["exit"], str):
|
|
59
|
+
return "bad field"
|
|
60
|
+
steps = d.get("steps")
|
|
61
|
+
if not isinstance(steps, list) or not steps:
|
|
62
|
+
return "no steps"
|
|
63
|
+
for s in steps:
|
|
64
|
+
if not isinstance(s, dict) or not isinstance(s.get("observation"), str):
|
|
65
|
+
return "bad field"
|
|
66
|
+
if not isinstance(s.get("action"), str) or not s["action"]:
|
|
67
|
+
return "empty action"
|
|
68
|
+
tokens = s.get("tokens")
|
|
69
|
+
if not isinstance(tokens, int) or isinstance(tokens, bool) or tokens < 0:
|
|
70
|
+
return "bad tokens"
|
|
71
|
+
if "error" not in s or (s["error"] is not None and not isinstance(s["error"], bool)):
|
|
72
|
+
return "bad error flag" # isinstance, because 1 == True would sneak past an `in` check
|
|
73
|
+
return None
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
# ---------- Claude Code session transcripts (research R10) ----------
|
|
77
|
+
# The format is internal to Claude Code and can change between releases, so this is the only
|
|
78
|
+
# function that reads it, and tests/fixtures/claude_code_session.jsonl pins what it expects.
|
|
79
|
+
|
|
80
|
+
INTERRUPTED = "[Request interrupted by user"
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _text(content):
|
|
84
|
+
if isinstance(content, str):
|
|
85
|
+
return content
|
|
86
|
+
if isinstance(content, list):
|
|
87
|
+
return "\n".join(b.get("text", "") for b in content if isinstance(b, dict) and b.get("type") == "text")
|
|
88
|
+
return ""
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _usage_tokens(usage):
|
|
92
|
+
"""Every token the call read or wrote, cached or not: what the turn really consumed."""
|
|
93
|
+
keys = ("input_tokens", "cache_creation_input_tokens", "cache_read_input_tokens", "output_tokens")
|
|
94
|
+
return sum(usage.get(k) or 0 for k in keys) if isinstance(usage, dict) else 0
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class _Turn:
|
|
98
|
+
def __init__(self, turn_id):
|
|
99
|
+
self.id, self.steps, self.by_call = turn_id, [], {}
|
|
100
|
+
self.interrupted, self.pending = False, 0
|
|
101
|
+
self.msg, self.msg_tokens, self.msg_charged = None, 0, True
|
|
102
|
+
|
|
103
|
+
def add_tokens(self, n): # to the previous step, or held for the first one
|
|
104
|
+
if self.steps:
|
|
105
|
+
self.steps[-1]["tokens"] += n
|
|
106
|
+
else:
|
|
107
|
+
self.pending += n
|
|
108
|
+
|
|
109
|
+
def close_message(self): # a response that called no tool: its tokens go to the step before it
|
|
110
|
+
if self.msg is not None and not self.msg_charged:
|
|
111
|
+
self.add_tokens(self.msg_tokens)
|
|
112
|
+
self.msg_charged = True
|
|
113
|
+
|
|
114
|
+
def run(self, group, exclude):
|
|
115
|
+
self.close_message()
|
|
116
|
+
steps = tuple(Step(s["action"], s["observation"], s["error"], s["tokens"]) for s in self.steps)
|
|
117
|
+
exit = "interrupted" if self.interrupted else None
|
|
118
|
+
return Run(group, "claude-code-local", self.id, self.id, not self.interrupted and self.id not in exclude, exit, True, steps)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def claude_code_turns(transcript, exclude=()):
|
|
122
|
+
"""Split one Claude Code session transcript into turns, one Run per user prompt.
|
|
123
|
+
|
|
124
|
+
Returns (runs, skipped). A turn succeeds when it was not interrupted and is not in `exclude`
|
|
125
|
+
(turn ids: the uuid of the prompt record). The run's group is the project folder name; callers
|
|
126
|
+
rename it before anything leaves the machine (constitution Principle VI).
|
|
127
|
+
"""
|
|
128
|
+
group = Path(transcript).parent.name
|
|
129
|
+
runs, skipped, seen_msgs = [], Counter(), set()
|
|
130
|
+
turn = None
|
|
131
|
+
with open(transcript, encoding="utf-8") as f:
|
|
132
|
+
for line in f:
|
|
133
|
+
try:
|
|
134
|
+
rec = json.loads(line)
|
|
135
|
+
except json.JSONDecodeError:
|
|
136
|
+
skipped["bad json"] += 1
|
|
137
|
+
continue
|
|
138
|
+
if rec.get("isSidechain"):
|
|
139
|
+
skipped["sidechain"] += 1
|
|
140
|
+
continue
|
|
141
|
+
kind, msg = rec.get("type"), rec.get("message")
|
|
142
|
+
if kind not in ("user", "assistant") or not isinstance(msg, dict):
|
|
143
|
+
skipped["not a message"] += 1
|
|
144
|
+
continue
|
|
145
|
+
content = msg.get("content")
|
|
146
|
+
if kind == "user":
|
|
147
|
+
if rec.get("isMeta"):
|
|
148
|
+
continue
|
|
149
|
+
results = [b for b in content if isinstance(b, dict) and b.get("type") == "tool_result"] if isinstance(content, list) else []
|
|
150
|
+
if results:
|
|
151
|
+
for b in results:
|
|
152
|
+
step = turn.by_call.get(b.get("tool_use_id")) if turn else None
|
|
153
|
+
if step is not None:
|
|
154
|
+
step["observation"] = _text(b.get("content"))
|
|
155
|
+
step["error"] = bool(b.get("is_error", False))
|
|
156
|
+
continue
|
|
157
|
+
text = _text(content)
|
|
158
|
+
if text.startswith(INTERRUPTED):
|
|
159
|
+
if turn:
|
|
160
|
+
turn.interrupted = True
|
|
161
|
+
continue
|
|
162
|
+
if text.startswith("<local-command"):
|
|
163
|
+
continue
|
|
164
|
+
if turn:
|
|
165
|
+
runs.append(turn.run(group, exclude))
|
|
166
|
+
turn = _Turn(rec.get("uuid") or f"turn-{len(runs) + 1}")
|
|
167
|
+
continue
|
|
168
|
+
if turn is None:
|
|
169
|
+
continue # assistant output before any prompt
|
|
170
|
+
mid = msg.get("id")
|
|
171
|
+
if mid != turn.msg:
|
|
172
|
+
turn.close_message()
|
|
173
|
+
turn.msg, turn.msg_charged = mid, False
|
|
174
|
+
turn.msg_tokens = 0 if mid in seen_msgs else _usage_tokens(msg.get("usage"))
|
|
175
|
+
seen_msgs.add(mid) # one response is split across records that repeat its usage
|
|
176
|
+
for b in content if isinstance(content, list) else []:
|
|
177
|
+
if isinstance(b, dict) and b.get("type") == "tool_use":
|
|
178
|
+
tokens = 0 if turn.msg_charged else turn.msg_tokens + turn.pending
|
|
179
|
+
turn.msg_charged, turn.pending = True, 0
|
|
180
|
+
step = {"action": f"{b.get('name')} {json.dumps(b.get('input') or {}, sort_keys=True, ensure_ascii=False)}",
|
|
181
|
+
"observation": "", "error": None, "tokens": tokens}
|
|
182
|
+
turn.steps.append(step)
|
|
183
|
+
turn.by_call[b.get("id")] = step
|
|
184
|
+
if turn:
|
|
185
|
+
runs.append(turn.run(group, exclude))
|
|
186
|
+
return runs, skipped
|
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: loopbrake
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Stops stuck AI agent runs, with a guaranteed limit on stopping good ones.
|
|
5
|
+
Project-URL: Homepage, https://github.com/SahilSelokar/LoopBrake
|
|
6
|
+
Project-URL: Source, https://github.com/SahilSelokar/LoopBrake
|
|
7
|
+
Project-URL: Results, https://github.com/SahilSelokar/LoopBrake/tree/main/eval/results
|
|
8
|
+
Author: Sahil Selokar
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: agent loops,ai agents,claude code,conformal prediction,llm,stop controller
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
20
|
+
Requires-Python: >=3.11
|
|
21
|
+
Provides-Extra: agent-sdk
|
|
22
|
+
Requires-Dist: claude-agent-sdk; extra == 'agent-sdk'
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
|
|
25
|
+
<div align="center">
|
|
26
|
+
|
|
27
|
+
# LoopBrake
|
|
28
|
+
|
|
29
|
+
**Brakes for stuck AI agents.**
|
|
30
|
+
|
|
31
|
+
Stop agent runs that are going in circles, with a guaranteed limit on how often a good run gets stopped.
|
|
32
|
+
|
|
33
|
+
[](#status)
|
|
34
|
+
[](pyproject.toml)
|
|
35
|
+
[](pyproject.toml)
|
|
36
|
+
[](LICENSE)
|
|
37
|
+
|
|
38
|
+
</div>
|
|
39
|
+
|
|
40
|
+
---
|
|
41
|
+
|
|
42
|
+
## The problem
|
|
43
|
+
|
|
44
|
+
AI agents get stuck. They run the same command again and again, re-read the same file, or hit the
|
|
45
|
+
same error, and they keep spending tokens until a step or budget limit finally stops them.
|
|
46
|
+
|
|
47
|
+
The usual fixes are blunt: a fixed step limit, or a rule like "stop after the same action 3 times".
|
|
48
|
+
Set them tight and you kill runs that would have succeeded. Set them loose and you pay for every loop.
|
|
49
|
+
|
|
50
|
+
## The idea
|
|
51
|
+
|
|
52
|
+
LoopBrake watches a run one step at a time and gives each step a **stuck score**. When the score
|
|
53
|
+
crosses a **stop line**, it stops the run and says why.
|
|
54
|
+
|
|
55
|
+
The stop line is not guessed. It is set from **your own past successful runs**, so that at most a
|
|
56
|
+
chosen share of good runs, for example 5 in 100, would ever cross it. That is a statistical
|
|
57
|
+
guarantee (split-conformal calibration), not a tuned target.
|
|
58
|
+
|
|
59
|
+
**In v1 the stuck score is simply how long the run has gone on.** Two experiments (below) tested
|
|
60
|
+
smarter scores, and neither beat it on agents they were not tuned on. The stuck signals (repeats,
|
|
61
|
+
nothing new, same error again) still run, but only to explain why a stopped run looked stuck.
|
|
62
|
+
|
|
63
|
+
```mermaid
|
|
64
|
+
flowchart LR
|
|
65
|
+
A[Agent takes a step] --> B[Stuck score]
|
|
66
|
+
B --> C{Above the stop line?}
|
|
67
|
+
C -- no --> A
|
|
68
|
+
C -- yes --> D[Stop the run and say why]
|
|
69
|
+
E[(Your past successful runs)] -. set the stop line .-> C
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
## What a stop looks like
|
|
73
|
+
|
|
74
|
+
A real run from the public SWE-bench data (GPT-5-mini), replayed through LoopBrake. The agent kept
|
|
75
|
+
searching the same files for a function and found nothing new:
|
|
76
|
+
|
|
77
|
+
```text
|
|
78
|
+
swe-gpt5mini · matplotlib__matplotlib-25775 · stopped at step 27 of 72 · saved 2,023,115 tokens (84%)
|
|
79
|
+
Reason: repeating in 5 of last 5 steps (similar to step 19); nothing new in 2 of last 5 steps (0 of 21 output lines new)
|
|
80
|
+
|
|
81
|
+
23 sed -n '760,1080p' lib/matplotlib/backend_bases.py
|
|
82
|
+
24 grep -n "def set_antialiased" -n lib/matplotlib/backend_bases.py || true
|
|
83
|
+
25 sed -n '892,932p' lib/matplotlib/backend_bases.py
|
|
84
|
+
26 sed -n '1,220p' lib/matplotlib/patches.py
|
|
85
|
+
27 grep -n "set_antialiased" -n lib -R || true
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
More in [eval/results/kill-stories.md](eval/results/kill-stories.md).
|
|
89
|
+
|
|
90
|
+
## Use it
|
|
91
|
+
|
|
92
|
+
**Install.** `pip install loopbrake`, or `uv add loopbrake`. Until the first PyPI release, use
|
|
93
|
+
`pip install git+https://github.com/SahilSelokar/LoopBrake`. No other packages are needed.
|
|
94
|
+
|
|
95
|
+
**1. Set your stop line from your own past runs.** It needs at least 19 successful runs (at α 5%);
|
|
96
|
+
with fewer, LoopBrake only watches.
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
loopbrake calibrate ~/.claude/projects/<your-project>/ --project my-agent # your Claude Code history
|
|
100
|
+
loopbrake calibrate my_runs.jsonl --project my-agent # or recorded runs
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
**2. Add it to your agent loop.**
|
|
104
|
+
|
|
105
|
+
```python
|
|
106
|
+
import loopbrake
|
|
107
|
+
|
|
108
|
+
with loopbrake.start(project="my-agent") as brake:
|
|
109
|
+
for action, result in my_agent_steps(): # your loop
|
|
110
|
+
decision = brake.step(action, result)
|
|
111
|
+
if decision.stop:
|
|
112
|
+
print(decision.reason); break
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
**3. See how it's doing.**
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
loopbrake status --project my-agent # runs watched, stops, and stops you marked as mistakes
|
|
119
|
+
loopbrake feedback <run-id> --mistaken # tell LoopBrake a stop was wrong
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
**Claude's agent toolkit:** `pip install "loopbrake[agent-sdk]"`, then
|
|
123
|
+
`ClaudeAgentOptions(hooks=loopbrake.agent_sdk.hooks(project="my-agent"))`.
|
|
124
|
+
|
|
125
|
+
Everything stays on your machine, in `~/.loopbrake`. LoopBrake never uses the network. If something
|
|
126
|
+
inside it fails, it switches to watching only. It never crashes or stops your agent because of its
|
|
127
|
+
own problem.
|
|
128
|
+
|
|
129
|
+
## Phase 1 results
|
|
130
|
+
|
|
131
|
+
Before building the product, we tested the idea offline. We replayed **2,979 recorded runs** from
|
|
132
|
+
six agent groups, on coding tasks
|
|
133
|
+
([SWE-bench Verified](https://github.com/SWE-bench/experiments)) and customer-service tasks
|
|
134
|
+
([τ-bench](https://github.com/sierra-research/tau-bench)).
|
|
135
|
+
|
|
136
|
+
| Finding | Result |
|
|
137
|
+
|---|---|
|
|
138
|
+
| The guarantee held: good runs stopped by the chosen method, on every test group | **at most 4.9%** (limit 5%) |
|
|
139
|
+
| The naive rule "stop at 20 steps or 3 repeats" stops good runs | **99.3%** (Devstral), **29.1%** (GPT-5-mini) |
|
|
140
|
+
| Tokens saved by a calibrated step limit (GPT-5-mini, 100 past runs) | **11.2%** |
|
|
141
|
+
| Tokens saved by the stuck signals (same setting) | **12.9%** |
|
|
142
|
+
| Extra saving from stuck signals over the step limit, with 95% interval | SWE-bench **+1.5%** [−9.7%, +13.2%]<br>τ-bench **+0.9%** [−3.1%, +5.0%] |
|
|
143
|
+
|
|
144
|
+
**Verdict: NO-GO.** The cheap stuck signals do not clearly beat a calibrated step limit. Following
|
|
145
|
+
the plan set in advance, a second experiment tested a progress judge (below).
|
|
146
|
+
For comparison, [FailFast](https://arxiv.org/abs/2608.03222), a trained monitor, reports 14.6–20.4%
|
|
147
|
+
saved at 5% of good runs stopped. Its threshold was set on the same data it reports on, though, so
|
|
148
|
+
that 5% is a target, not a guarantee.
|
|
149
|
+
|
|
150
|
+
Full report: [eval/results/results.md](eval/results/results.md)
|
|
151
|
+
|
|
152
|
+
## Progress judge: also NO-GO
|
|
153
|
+
|
|
154
|
+
Counting repeats is not the same as understanding a step, so the second experiment asked a fast
|
|
155
|
+
hosted decision model, [Jev](https://docs.typesafe.ai) (`jev-1.13.0`), about every step: *did this
|
|
156
|
+
move the agent closer to finishing?* and *what kind of step was it?* Savings are counted **net**:
|
|
157
|
+
every token the judge reads is subtracted. The method was chosen on one agent and committed
|
|
158
|
+
(`0827ece`) before any other agent was judged.
|
|
159
|
+
|
|
160
|
+
| Dataset | Extra net saving over the step limit, with 95% interval |
|
|
161
|
+
|---|---|
|
|
162
|
+
| SWE-bench (GPT-5-mini) | **−6.3%** [−18.9%, +1.9%] |
|
|
163
|
+
| τ-bench (4 groups) | **−4.2%** [−8.2%, −1.5%] |
|
|
164
|
+
|
|
165
|
+
What we learned:
|
|
166
|
+
|
|
167
|
+
- **A threshold tuned on one agent did not travel.** The chosen method asked the judge only when the
|
|
168
|
+
cheap score passed a level set on Devstral's long runs. On the other agents that level was almost
|
|
169
|
+
never reached (0–0.3% of steps), so the method rarely stopped anything.
|
|
170
|
+
- **A judge on every step is expensive.** On short customer-service tasks it used about as many
|
|
171
|
+
tokens as the agent itself.
|
|
172
|
+
- **Even ignoring its cost, the judge did not spot stuck runs better than counting steps**: for
|
|
173
|
+
example 7.3% vs 9.3% of tokens saved on GPT-5-mini.
|
|
174
|
+
- The guarantee held on every group. The whole experiment took 73,812 judgments, for about $3.91.
|
|
175
|
+
|
|
176
|
+
Full report: [eval/results/judge/results.md](eval/results/judge/results.md)
|
|
177
|
+
|
|
178
|
+
## How the results are kept honest
|
|
179
|
+
|
|
180
|
+
- **Chosen in advance.** The method was picked using one agent's runs and committed *before* any
|
|
181
|
+
test runs were scored (commits `8b8eaf3` and, for the judge, `0827ece`).
|
|
182
|
+
- **Mistakes stay visible.** The first results were committed exactly as they came out, including a
|
|
183
|
+
flaw in how runs were split (`e655b96`). The fix is a separate, documented commit (`76e8cdc`), with
|
|
184
|
+
before-and-after numbers in [CORRECTIONS.md](eval/results/CORRECTIONS.md).
|
|
185
|
+
- **Tested on unseen runs.** The stop line is always checked on runs it was not set from.
|
|
186
|
+
- **Reproducible.** Every number above comes from `eval/results/` and comes out identical on every run.
|
|
187
|
+
|
|
188
|
+
## How the guarantee works
|
|
189
|
+
|
|
190
|
+
1. Each step gets a stuck score. A run's score is the highest score it reaches.
|
|
191
|
+
2. Take *n* past successful runs and sort their scores. The stop line is the *k*-th smallest score,
|
|
192
|
+
where *k* = ⌈(*n* + 1)(1 − α)⌉ and α is the share of good runs you accept losing (for example 5%).
|
|
193
|
+
3. A new successful run then crosses the stop line with probability at most α.
|
|
194
|
+
|
|
195
|
+
With α = 5%, LoopBrake needs at least 19 past successful runs. With fewer, it only watches and
|
|
196
|
+
never stops anything.
|
|
197
|
+
|
|
198
|
+
The guarantee holds when new runs look like the past ones (same agent, same kind of tasks), and when
|
|
199
|
+
the past runs come from different tasks. We learned the second condition the hard way: see
|
|
200
|
+
[CORRECTIONS.md](eval/results/CORRECTIONS.md).
|
|
201
|
+
|
|
202
|
+
## Reproduce
|
|
203
|
+
|
|
204
|
+
Requires [uv](https://docs.astral.sh/uv/) and Python 3.11+.
|
|
205
|
+
|
|
206
|
+
```bash
|
|
207
|
+
git clone https://github.com/SahilSelokar/LoopBrake && cd LoopBrake
|
|
208
|
+
uv run pytest # the test suite
|
|
209
|
+
uv run python eval/fetch.py # one-time download, about 450 MB, into ~/.loopbrake/data
|
|
210
|
+
uv run python eval/run.py --final # about 1 minute; writes eval/results/
|
|
211
|
+
```
|
|
212
|
+
|
|
213
|
+
The progress judge needs a [Typesafe](https://typesafe.ai) API key in `TYPESAFE_API_KEY` (or in
|
|
214
|
+
`~/.loopbrake/typesafe_key`). Judging every public step costs about $4:
|
|
215
|
+
|
|
216
|
+
```bash
|
|
217
|
+
uv run python eval/tasks.py # task texts for the judge
|
|
218
|
+
uv run python eval/judge.py --group swe-devstral # one group at a time; resumable
|
|
219
|
+
uv run python eval/judge_eval.py --final # reads stored answers only; writes eval/results/judge/
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
## Repository layout
|
|
223
|
+
|
|
224
|
+
```text
|
|
225
|
+
src/loopbrake/ scoring core: stuck signals, stop-line rule, run readers (standard library only)
|
|
226
|
+
eval/ the experiments: fetch.py downloads the data, run.py replays runs, judge.py asks the
|
|
227
|
+
progress judge, judge_eval.py scores its answers; results/ holds the published numbers
|
|
228
|
+
specs/ design: constitution, roadmap, and the spec, plan, research and tasks of each experiment
|
|
229
|
+
tests/ tests, including a check of the guarantee on simulated data
|
|
230
|
+
liveness.py the original naive rule, kept as the baseline
|
|
231
|
+
```
|
|
232
|
+
|
|
233
|
+
## Roadmap
|
|
234
|
+
|
|
235
|
+
| Phase | What | Status |
|
|
236
|
+
|---|---|---|
|
|
237
|
+
| 1 | **Experiment**: does it work on real runs? | Done: NO-GO for cheap signals |
|
|
238
|
+
| 1b | **Progress judge**: a hosted decision model judges whether each step moved the run forward | Done: NO-GO |
|
|
239
|
+
| 2 | **Python package**: `pip install loopbrake`; a stop line on run length, with a guarantee and a readable reason | Built (v0.1.0); first PyPI release pending |
|
|
240
|
+
| 3 | **Claude Code plugin**: stop stuck sessions live, calibrated on your own history | Planned |
|
|
241
|
+
| 4 | **Observability**: live dashboard, plus export to Datadog, Grafana and others via OpenTelemetry | Planned |
|
|
242
|
+
| 5 | **Launch**: a demo agent, the public release and a video | Planned |
|
|
243
|
+
|
|
244
|
+
The full plan is in [specs/roadmap.md](specs/roadmap.md).
|
|
245
|
+
|
|
246
|
+
## Status
|
|
247
|
+
|
|
248
|
+
v0.1.0 is built and tested. The first PyPI release comes once publishing is set up; until then,
|
|
249
|
+
install from GitHub (see "Use it"). Its stop rule is a stop line on run length, set from your own past
|
|
250
|
+
successful runs, with a guaranteed limit on stopping good runs.
|
|
251
|
+
|
|
252
|
+
## License
|
|
253
|
+
|
|
254
|
+
[MIT](LICENSE). One test fixture is a trimmed τ-bench run, used under τ-bench's own MIT license
|
|
255
|
+
(see [tests/fixtures/NOTICE.md](tests/fixtures/NOTICE.md)).
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
loopbrake/__init__.py,sha256=z8u1dzoZoj7ga-0G6rQkU7SCdytA-7ZJYaH7lgM4Oug,308
|
|
2
|
+
loopbrake/agent_sdk.py,sha256=P8MJHWh13fK_ollf1YmsdHG_T8wCssu4WcWJNvnExxI,3379
|
|
3
|
+
loopbrake/brake.py,sha256=kdqqVmv0tAsakUOXMWnbc8C5_5vOgk28V08G9a4FkiI,5408
|
|
4
|
+
loopbrake/calibration.py,sha256=yvnQvKX9eeZFbC_lSNmbMPFSoTTZdDIUC5rCBU0uReg,4244
|
|
5
|
+
loopbrake/cli.py,sha256=JlsdMvzjjYw4BjQiQfxvu8NFcs8gEv2XbcKOpqmHUg8,4819
|
|
6
|
+
loopbrake/conformal.py,sha256=DY-73iNiIVk-_XpQi-A3yDrxpZ_D8ce9_Iw8VT289eM,752
|
|
7
|
+
loopbrake/records.py,sha256=RrNXQ3bO25D8ZiOjWIP2TSQ4aQCZbti-n01mZ4btNxs,4039
|
|
8
|
+
loopbrake/signals.py,sha256=r7DFUUDUR_D-QQc5vk6F_jfBzvB5rXPMj_7vB8ie8vM,8519
|
|
9
|
+
loopbrake/traces.py,sha256=bGGfrR9tPLpR0H86tDH2MdyUK2EJjV-3MyfCYqrQtgU,7837
|
|
10
|
+
loopbrake-0.1.0.dist-info/METADATA,sha256=j_rQOUkizzgStVrU53o_LEqHncRYXLpy3egaTbzmxeE,11964
|
|
11
|
+
loopbrake-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
12
|
+
loopbrake-0.1.0.dist-info/entry_points.txt,sha256=nHCcVuPCF8zXEf9raOwfRLiyqz7S_lp7qrdODRqDgFM,49
|
|
13
|
+
loopbrake-0.1.0.dist-info/licenses/LICENSE,sha256=fkXmARFvnpYeFBsjGBm6VWDxGO0IiF0UuStAscn_U6s,1070
|
|
14
|
+
loopbrake-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Sahil Selokar
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|