swearbench 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
swearbench/__init__.py ADDED
@@ -0,0 +1 @@
1
+ __version__ = "0.2.1"
swearbench/__main__.py ADDED
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from .cli import main
4
+
5
+ sys.exit(main())
swearbench/card.py ADDED
@@ -0,0 +1,66 @@
1
+ """Leaderboard card, styled to match chart.svg. Model names and numbers only; never quotes."""
2
+ from __future__ import annotations
3
+
4
+ from html import escape
5
+
6
+ from .chart import FAMILIES, PAD, STYLE, _text_w, family, pretty
7
+
8
+ MODE_LABEL = {
9
+ "fabrication": "lies", "overreach": "overreach", "giving_up": "gave up", "regression": "breaks things",
10
+ "ignored": "ignores me", "repeat": "repeats", "incomplete": "half-done", "insult": "insults",
11
+ "profanity": "swearing", "shouting": "CAPS", "sarcasm": "sarcasm", "slow": "slow",
12
+ }
13
+
14
+ W = 760
15
+ TOP = 120
16
+ ROW_H = 40
17
+ RANK_END = PAD + 16
18
+ NAME_X = RANK_END + 14
19
+ BAR_X = NAME_X + 160
20
+ BAR_W = 180
21
+ SCORE_END = BAR_X + BAR_W + 40
22
+ STATS_X = SCORE_END + 24
23
+
24
+
25
+ def svg(res, rows=8):
26
+ board = res["board"][:rows]
27
+ h = TOP + ROW_H * max(len(board), 1) + PAD + 20
28
+ hi = max([s["score"] for s in board] + [1])
29
+ o = [f'<svg xmlns="http://www.w3.org/2000/svg" class="c" width="{W}" height="{h}" viewBox="0 0 {W} {h}" '
30
+ 'role="img" aria-label="SwearBench leaderboard">',
31
+ f"<style>{STYLE}</style>",
32
+ f'<rect width="{W}" height="{h}" rx="14" fill="var(--surface)"/>',
33
+ f'<text x="{PAD}" y="{PAD + 18}" font-size="20" font-weight="700" fill="var(--ink)">SwearBench</text>']
34
+
35
+ lx, ly = PAD, PAD + 52
36
+ for i, (name, _) in enumerate(FAMILIES):
37
+ o.append(f'<circle cx="{lx + 5}" cy="{ly - 4}" r="5" fill="var(--s{i + 1})"/>'
38
+ f'<text x="{lx + 15}" y="{ly}" font-size="12" fill="var(--ink2)">{name}</text>')
39
+ lx += 15 + _text_w(name) + 24
40
+ o.append(f'<text x="{W - PAD}" y="{ly}" font-size="12" fill="var(--muted)" text-anchor="end">'
41
+ f'{res["n_reactions"]} reactions</text>')
42
+
43
+ for i, s in enumerate(board):
44
+ y = TOP + i * ROW_H
45
+ mid = y + ROW_H / 2
46
+ base = mid + 4.5
47
+ col = f"var(--s{family(s['model']) + 1})"
48
+ top_mode = max(s["modes"], key=s["modes"].get) if s["modes"] else None
49
+ o += [f'<line x1="{PAD}" x2="{W - PAD}" y1="{y}" y2="{y}" stroke="var(--grid)" stroke-width="1"/>',
50
+ f'<text x="{RANK_END}" y="{base}" font-size="12" fill="var(--muted)" text-anchor="end">{i + 1}</text>',
51
+ f'<text x="{NAME_X}" y="{base}" font-size="13" font-weight="600" fill="var(--ink)">'
52
+ f'{escape(pretty(s["model"]))}</text>',
53
+ f'<rect x="{BAR_X}" y="{mid - 3}" width="{BAR_W}" height="6" rx="3" fill="var(--grid)"/>',
54
+ f'<rect x="{BAR_X}" y="{mid - 3}" width="{max(6, BAR_W * max(s["score"], 0) / hi):.0f}" height="6" '
55
+ f'rx="3" fill="{col}"/>',
56
+ f'<text x="{SCORE_END}" y="{base}" font-size="13" font-weight="700" fill="var(--ink)" '
57
+ f'text-anchor="end" font-variant-numeric="tabular-nums">{s["score"]:.0f}</text>',
58
+ f'<text x="{STATS_X}" y="{base}" font-size="12" fill="var(--ink2)">'
59
+ f'ships {s["outcome"]:.0f}% · {s["angry_pct"]:.0f}% angry</text>',
60
+ f'<text x="{W - PAD}" y="{base}" font-size="12" fill="var(--muted)" text-anchor="end">'
61
+ f'{MODE_LABEL.get(top_mode, "")}</text>']
62
+ end = TOP + ROW_H * len(board)
63
+ o.append(f'<line x1="{PAD}" x2="{W - PAD}" y1="{end}" y2="{end}" stroke="var(--axis)" stroke-width="1"/>')
64
+ o.append(f'<text x="{PAD}" y="{h - PAD}" font-size="11" fill="var(--muted)">github.com/AlenHay/swearbench</text>')
65
+ o.append("</svg>")
66
+ return "\n".join(o)
swearbench/chart.py ADDED
@@ -0,0 +1,134 @@
1
+ """Swearing vs. result: one dot per model. x = rage per 100 turns, y = share of sessions that land."""
2
+ from __future__ import annotations
3
+
4
+ import math
5
+ import re
6
+ from html import escape
7
+
8
+ FAMILIES = [("Claude", ("claude",)), ("GPT", ("gpt", "o1", "o3", "o4", "codex")), ("Other", ())]
9
+
10
+ STYLE = """
11
+ .c{--surface:#fcfcfb;--ink:#0b0b0b;--ink2:#52514e;--muted:#898781;--grid:#e1e0d9;--axis:#c3c2b7;
12
+ --s1:#2a78d6;--s2:#eb6834;--s3:#1baf7a}
13
+ @media (prefers-color-scheme:dark){.c{--surface:#1a1a19;--ink:#fff;--ink2:#c3c2b7;--grid:#2c2c2a;--axis:#383835;
14
+ --s1:#3987e5;--s2:#d95926;--s3:#199e70}}
15
+ text{font-family:ui-sans-serif,system-ui,-apple-system,"Segoe UI",sans-serif}
16
+ """
17
+
18
+
19
+ def pretty(model):
20
+ """claude-opus-5-5 -> Opus 5.5, gpt-5.6-sol -> GPT-5.6 Sol, gemini-3.8-flash-high -> Gemini 3.8 Flash."""
21
+ m = re.sub(r"-\d{8}$", "", model.split("/")[-1])
22
+ if m.startswith("claude-"):
23
+ parts = m[7:].split("-")
24
+ return f"{parts[0].title()} {'.'.join(parts[1:])}"
25
+ if m.startswith("gpt-"):
26
+ ver, *rest = m[4:].split("-")
27
+ return " ".join([f"GPT-{ver}"] + [r.title() for r in rest])
28
+ words = [w for w in m.split("-") if w not in ("high", "low", "medium", "free", "preview")]
29
+ return " ".join(w if any(c.isdigit() for c in w) else w.title() for w in words)
30
+
31
+
32
+ def family(model):
33
+ for i, (_, prefixes) in enumerate(FAMILIES[:-1]):
34
+ if model.startswith(prefixes):
35
+ return i
36
+ return len(FAMILIES) - 1
37
+
38
+
39
+ def _ticks(lo, hi, n=5):
40
+ step = 10 ** math.floor(math.log10((hi - lo) / n or 1))
41
+ for m in (1, 2, 2.5, 5, 10):
42
+ if (hi - lo) / (step * m) <= n:
43
+ step *= m
44
+ break
45
+ start = math.floor(lo / step) * step
46
+ return [start + i * step for i in range(int((hi - start) / step) + 2) if start + i * step <= hi + 1e-9]
47
+
48
+
49
+ PAD = 32
50
+
51
+
52
+ def _text_w(text, size=12):
53
+ return len(text) * size * 0.56
54
+
55
+
56
+ def svg(res):
57
+ pts = res["board"]
58
+ W, H = 760, 540
59
+ L, R, T, B = PAD + 48, PAD, 120, 76
60
+ pw, ph = W - L - R, H - T - B
61
+ xs = [p["rage100"] for p in pts] or [0, 1]
62
+ x_lo, x_hi = math.floor(min(xs) * 0.9 / 10) * 10, math.ceil(max(xs) * 1.05 / 10) * 10
63
+ x_lo, x_hi = (x_lo, x_hi) if x_hi > x_lo else (0, 100)
64
+ X = lambda v: L + (v - x_lo) / (x_hi - x_lo) * pw # noqa: E731
65
+ Y = lambda v: T + ph - v / 100 * ph # noqa: E731
66
+ max_n = max([p["sessions"] for p in pts] + [1])
67
+
68
+ o = [f'<svg xmlns="http://www.w3.org/2000/svg" class="c" width="{W}" height="{H}" viewBox="0 0 {W} {H}" '
69
+ 'role="img" aria-labelledby="t d">',
70
+ f"<style>{STYLE}</style>",
71
+ '<title id="t">SwearBench: how much swearing gets you what result</title>',
72
+ '<desc id="d">' + escape("; ".join(f'{p["model"]}: rage {p["rage100"]:.0f} per 100 turns, '
73
+ f'ships {p["outcome"]:.0f}%' for p in pts)) + "</desc>",
74
+ f'<rect width="{W}" height="{H}" rx="14" fill="var(--surface)"/>',
75
+ f'<text x="{PAD}" y="{PAD + 18}" font-size="20" font-weight="700" fill="var(--ink)">'
76
+ 'Swearing vs. result</text>']
77
+
78
+ lx, ly = PAD, PAD + 52
79
+ for i, (name, _) in enumerate(FAMILIES):
80
+ o.append(f'<circle cx="{lx + 5}" cy="{ly - 4}" r="5" fill="var(--s{i + 1})"/>'
81
+ f'<text x="{lx + 15}" y="{ly}" font-size="12" fill="var(--ink2)">{name}</text>')
82
+ lx += 15 + _text_w(name) + 24
83
+
84
+ for v in _ticks(x_lo, x_hi):
85
+ o.append(f'<line x1="{X(v):.1f}" x2="{X(v):.1f}" y1="{T}" y2="{T + ph}" stroke="var(--grid)" stroke-width="1"/>')
86
+ o.append(f'<text x="{X(v):.1f}" y="{T + ph + 22}" font-size="11" fill="var(--muted)" text-anchor="middle">{v:g}</text>')
87
+ for v in range(0, 101, 20):
88
+ o.append(f'<line x1="{L}" x2="{L + pw}" y1="{Y(v):.1f}" y2="{Y(v):.1f}" stroke="var(--grid)" stroke-width="1"/>')
89
+ o.append(f'<text x="{L - 12}" y="{Y(v) + 4:.1f}" font-size="11" fill="var(--muted)" text-anchor="end">{v}%</text>')
90
+ o.append(f'<line x1="{L}" x2="{L + pw}" y1="{T + ph}" y2="{T + ph}" stroke="var(--axis)" stroke-width="1"/>')
91
+ o.append(f'<text x="{L + pw / 2}" y="{H - PAD}" font-size="12" fill="var(--ink2)" text-anchor="middle">'
92
+ 'Swearing →</text>')
93
+ o.append(f'<text transform="translate({PAD + 4} {T + ph / 2}) rotate(-90)" font-size="12" fill="var(--ink2)" '
94
+ 'text-anchor="middle">Shipped →</text>')
95
+
96
+ dots = []
97
+ for p in sorted(pts, key=lambda p: -p["sessions"]):
98
+ cx, cy = X(p["rage100"]), Y(p["outcome"])
99
+ r = 5 + 9 * math.sqrt(p["sessions"] / max_n)
100
+ col = f"var(--s{family(p['model']) + 1})"
101
+ lo, hi = p.get("outcome_ci", (p["outcome"], p["outcome"]))
102
+ tip = escape(f'{p["model"]}\nrage {p["rage100"]:.0f}/100 turns · {p["angry_pct"]:.0f}% angry msgs\n'
103
+ f'ships {p["outcome"]:.0f}% of {p["decided"]} decided sessions (90% CI {lo:.0f}–{hi:.0f}%)\n'
104
+ f'SwearBench {p["score"]:.1f}')
105
+ o.append(f'<g><title>{tip}</title>'
106
+ f'<line x1="{cx:.1f}" x2="{cx:.1f}" y1="{Y(hi):.1f}" y2="{Y(lo):.1f}" stroke="{col}" stroke-width="2" '
107
+ 'stroke-opacity="0.45" stroke-linecap="round"/>'
108
+ f'<circle cx="{cx:.1f}" cy="{cy:.1f}" r="{r + 8:.1f}" fill="transparent"/>'
109
+ f'<circle cx="{cx:.1f}" cy="{cy:.1f}" r="{r:.1f}" fill="{col}" stroke="var(--surface)" stroke-width="2"/></g>')
110
+ dots.append((p, cx, cy, r))
111
+
112
+ boxes = [(cx - r, cy - r, cx + r, cy + r) for _, cx, cy, r in dots]
113
+ for p, cx, cy, r in dots:
114
+ label = pretty(p["model"])
115
+ w = _text_w(label)
116
+ best = None
117
+ for dy in (0, -18, 18, -32, 32, -46, 46):
118
+ for side in (1, -1):
119
+ x0 = cx + r + 8 if side > 0 else cx - r - 8 - w
120
+ y = cy + 4 + dy
121
+ box = (x0, y - 11, x0 + w, y + 3)
122
+ if not (L + 2 <= box[0] and box[2] <= L + pw - 2 and T + 28 <= box[1] and box[3] <= T + ph - 20):
123
+ continue
124
+ overlap = sum(max(0, min(box[2], b[2]) - max(box[0], b[0])) * max(0, min(box[3], b[3]) - max(box[1], b[1]))
125
+ for b in boxes)
126
+ if best is None or overlap < best[0]:
127
+ best = (overlap, x0, y, box)
128
+ if best and best[0] == 0:
129
+ break
130
+ _, x0, y, box = best or (0, cx + r + 6, cy + 4, (cx + r + 6, cy - 7, cx + r + 6 + w, cy + 7))
131
+ boxes.append(box)
132
+ o.append(f'<text x="{x0:.1f}" y="{y:.1f}" font-size="12" fill="var(--ink)">{escape(label)}</text>')
133
+ o.append("</svg>")
134
+ return "\n".join(o)
swearbench/cli.py ADDED
@@ -0,0 +1,77 @@
1
+ """swearbench: rank AI coding models by how much you got mad at them."""
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import json
6
+ import os
7
+ import sys
8
+
9
+ from . import card, chart, judge, score, sources
10
+
11
+
12
+ def main(argv=None):
13
+ ap = argparse.ArgumentParser(prog="swearbench", description=__doc__)
14
+ ap.add_argument("--judge", choices=["claude-cli", "codex-cli", "anthropic", "openai", "command"],
15
+ help="who labels your messages (default: first available of claude, codex, ANTHROPIC_API_KEY, OPENAI_API_KEY)")
16
+ ap.add_argument("--judge-model", help="model id for the judge backend")
17
+ ap.add_argument("--judge-cmd", help="with --judge command: shell command that reads a prompt on stdin and prints the reply")
18
+ ap.add_argument("--since", help="only reactions on/after this date (YYYY-MM-DD)")
19
+ ap.add_argument("--min-reactions", type=int, default=40, help="reactions a model needs to be ranked (default 40)")
20
+ ap.add_argument("--exclude", action="append", default=[], help="model id to leave out (repeatable)")
21
+ ap.add_argument("--no-t3", action="store_true", help="ignore T3 Code's database even if present")
22
+ ap.add_argument("--no-quotes", action="store_true", help="leave the hall of shame out of report.md")
23
+ ap.add_argument("--out", default="swearbench-out", help="output directory (default ./swearbench-out)")
24
+ ap.add_argument("--workers", type=int, default=8)
25
+ ap.add_argument("--dry-run", action="store_true", help="show what was found and what would be judged, then stop")
26
+ ap.add_argument("-y", "--yes", action="store_true", help="don't ask before sending messages to the judge")
27
+ a = ap.parse_args(argv)
28
+
29
+ corpus = sources.collect(use_t3=not a.no_t3, exclude_dirs=["swearbench"])
30
+ if a.since:
31
+ corpus.reactions = [r for r in corpus.reactions if r.ts >= a.since]
32
+ corpus.interrupts = [i for i in corpus.interrupts if i.ts >= a.since]
33
+ print("Found: " + ("; ".join(corpus.found) or "no logs"), file=sys.stderr)
34
+ if not corpus.reactions:
35
+ print("No human messages found. Supported: Claude Code, Codex, OpenCode, T3 Code.", file=sys.stderr)
36
+ return 1
37
+
38
+ backend = a.judge or judge.detect_backend()
39
+ if not backend:
40
+ print("No judge available: install claude or codex CLI, or set ANTHROPIC_API_KEY / OPENAI_API_KEY, "
41
+ "or use --judge command --judge-cmd '...'.", file=sys.stderr)
42
+ return 1
43
+ model = a.judge_model or judge.DEFAULT_MODELS.get(backend)
44
+ cached = judge.load_cache()
45
+ todo = len({judge.key(r) for r in corpus.reactions} - set(cached))
46
+ print(f"{len(corpus.reactions)} messages from you; {todo} not judged yet. Judge: {backend}"
47
+ f"{' / ' + model if model else ''}.", file=sys.stderr)
48
+ if a.dry_run:
49
+ return 0
50
+ if todo and not a.yes:
51
+ if input(f"Send {todo} of your messages (with the AI reply each one answers) to the judge? [y/N] ").lower() != "y":
52
+ return 1
53
+
54
+ labels = judge.label(corpus.reactions, backend, model, a.judge_cmd, workers=a.workers)
55
+ res = score.build(corpus, labels, a.min_reactions, set(a.exclude))
56
+ family = (model or backend).split("-")[0]
57
+ if any(s["model"].startswith(family) for s in res["board"]):
58
+ print(f"Note: the judge ({model or backend}) is from a family being ranked. "
59
+ "Try a different --judge to check it isn't playing favourites.", file=sys.stderr)
60
+
61
+ os.makedirs(a.out, exist_ok=True)
62
+ report = score.markdown(res, quotes=not a.no_quotes)
63
+ with open(os.path.join(a.out, "report.md"), "w") as f:
64
+ f.write(report)
65
+ with open(os.path.join(a.out, "card.svg"), "w") as f:
66
+ f.write(card.svg(res))
67
+ with open(os.path.join(a.out, "chart.svg"), "w") as f:
68
+ f.write(chart.svg(res))
69
+ with open(os.path.join(a.out, "results.json"), "w") as f:
70
+ json.dump({k: v for k, v in res.items()}, f, indent=1, default=str)
71
+ print(report.split("\n## ")[0])
72
+ print(f"Wrote {a.out}/report.md, card.svg, chart.svg, results.json", file=sys.stderr)
73
+ return 0
74
+
75
+
76
+ if __name__ == "__main__":
77
+ sys.exit(main())
swearbench/judge.py ADDED
@@ -0,0 +1,149 @@
1
+ """LLM judge: labels each reaction with anger, anger modes and satisfaction. Results are cached on disk."""
2
+ from __future__ import annotations
3
+
4
+ import hashlib
5
+ import json
6
+ import os
7
+ import shlex
8
+ import shutil
9
+ import subprocess
10
+ import sys
11
+ import tempfile
12
+ import urllib.request
13
+ from concurrent.futures import ThreadPoolExecutor
14
+
15
+ RUBRIC_VERSION = "2"
16
+ CACHE = os.path.join(os.environ.get("XDG_CACHE_HOME", os.path.expanduser("~/.cache")), "swearbench", "labels.jsonl")
17
+
18
+ RUBRIC = """You are labelling a developer's chat messages to AI coding agents, to measure how frustrated
19
+ the developer was with the AI. Each item has `prev` (tail of the AI's last reply) and `msg` (the
20
+ developer's next message). Judge `msg` as a reaction to the AI's work.
21
+
22
+ Return ONLY a JSON array, one object per item, same order:
23
+ {"id": str,
24
+ "agent_authored": bool, // msg is clearly written by another AI/agent (long structured brief, "You are...", tool-dispatch prose), not a human
25
+ "anger": 0-4, // 0 calm/neutral, 1 mild annoyance, 2 clearly irritated, 3 angry, 4 furious
26
+ "target": "model"|"external"|"self"|"none", // who the frustration is aimed at; tools/vendors/infra/third parties = external
27
+ "modes": [..], // zero or more, only when target=="model":
28
+ // "profanity" swearing directed at the work or the model
29
+ // "insult" calling the model stupid/useless/lazy/idiot etc.
30
+ // "shouting" caps, !!!, ??? for emphasis
31
+ // "repeat" had to repeat an instruction / "I already told you" / "again"
32
+ // "ignored" model ignored or violated an explicit instruction or rule
33
+ // "fabrication" model claimed success/facts that were false, hallucinated, lied
34
+ // "incomplete" lazy, half-done, stopped early, left TODOs, didn't verify
35
+ // "regression" model broke something that worked
36
+ // "overreach" did unrequested/destructive things, scope creep, touched what it shouldn't
37
+ // "slow" too slow, too many questions, too verbose, wasted time
38
+ // "sarcasm" sarcastic/passive-aggressive phrasing
39
+ // "giving_up" abandons the model/approach, "I'll do it myself", "forget it", threatens to switch
40
+ // "taste" subjective critique while iterating on look/feel/wording ("looks lame", "too much text");
41
+ // the work does what was asked, the developer just doesn't like it yet
42
+ "blames_earlier": bool, // msg complains that something done BEFORE the AI's last reply (earlier merged/shipped work,
43
+ // a previous session) is broken, missing or wrong, rather than the last reply itself
44
+ "satisfaction": -2..2, // -2 rejects the work, -1 wants fixes, 0 neutral/new task, 1 accepts, 2 explicit praise/delight
45
+ "quote": str // <=80 char excerpt that best shows the anger (or "" if anger==0)
46
+ }
47
+ Profanity used positively ("fucking cool") is not anger. Terse instructions are not anger. A new
48
+ unrelated task after a reply implies mild acceptance (satisfaction 0 or 1), not anger.
49
+
50
+ ITEMS:
51
+ """
52
+
53
+ DEFAULT_MODELS = {"claude-cli": "claude-sonnet-5-5", "codex-cli": None, "anthropic": "claude-sonnet-5-5", "openai": "gpt-5.5"}
54
+
55
+
56
+ def key(r):
57
+ return hashlib.sha1(f"{RUBRIC_VERSION}\0{r.prev_reply[-500:]}\0{r.text[:1500]}".encode()).hexdigest()
58
+
59
+
60
+ def load_cache():
61
+ out = {}
62
+ if os.path.exists(CACHE):
63
+ with open(CACHE) as f:
64
+ for line in f:
65
+ try:
66
+ d = json.loads(line)
67
+ out[d["key"]] = d
68
+ except (ValueError, KeyError):
69
+ pass
70
+ return out
71
+
72
+
73
+ def detect_backend():
74
+ if shutil.which("claude"):
75
+ return "claude-cli"
76
+ if shutil.which("codex"):
77
+ return "codex-cli"
78
+ if os.environ.get("ANTHROPIC_API_KEY"):
79
+ return "anthropic"
80
+ if os.environ.get("OPENAI_API_KEY"):
81
+ return "openai"
82
+ return None
83
+
84
+
85
+ def _post(url, headers, body):
86
+ req = urllib.request.Request(url, json.dumps(body).encode(), {"content-type": "application/json", **headers})
87
+ with urllib.request.urlopen(req, timeout=600) as r:
88
+ return json.loads(r.read())
89
+
90
+
91
+ def call(backend, model, cmd, prompt):
92
+ if backend == "command":
93
+ return subprocess.run(shlex.split(cmd), input=prompt, capture_output=True, text=True, timeout=900).stdout
94
+ if backend == "claude-cli":
95
+ args = ["claude", "-p", "--tools", ""] + (["--model", model] if model else [])
96
+ return subprocess.run(args, input=prompt, capture_output=True, text=True, timeout=900).stdout
97
+ if backend == "codex-cli":
98
+ with tempfile.NamedTemporaryFile("r", suffix=".txt") as out:
99
+ args = ["codex", "exec", "--skip-git-repo-check", "--sandbox", "read-only", "-o", out.name]
100
+ subprocess.run(args + (["-m", model] if model else []) + ["-"], input=prompt,
101
+ capture_output=True, text=True, timeout=900)
102
+ return out.read()
103
+ if backend == "anthropic":
104
+ r = _post("https://api.anthropic.com/v1/messages",
105
+ {"x-api-key": os.environ["ANTHROPIC_API_KEY"], "anthropic-version": "2023-06-01"},
106
+ {"model": model, "max_tokens": 16000, "messages": [{"role": "user", "content": prompt}]})
107
+ return "".join(b.get("text", "") for b in r["content"])
108
+ if backend == "openai":
109
+ r = _post("https://api.openai.com/v1/chat/completions",
110
+ {"authorization": "Bearer " + os.environ["OPENAI_API_KEY"]},
111
+ {"model": model, "messages": [{"role": "user", "content": prompt}]})
112
+ return r["choices"][0]["message"]["content"]
113
+ raise ValueError(f"unknown judge backend {backend}")
114
+
115
+
116
+ def _label_batch(batch, backend, model, cmd):
117
+ items = [{"id": str(i), "prev": r.prev_reply[-500:], "msg": r.text[:1500]} for i, r in enumerate(batch)]
118
+ prompt = RUBRIC + json.dumps(items, ensure_ascii=False)
119
+ err = "unparseable output"
120
+ for _ in range(3):
121
+ try:
122
+ txt = call(backend, model, cmd, prompt) or ""
123
+ arr = json.loads(txt[txt.index("["):txt.rindex("]") + 1])
124
+ got = {str(a.get("id")): a for a in arr if isinstance(a, dict)}
125
+ if len(got) >= len(batch) * 0.9:
126
+ return [(key(r), got[str(i)]) for i, r in enumerate(batch) if str(i) in got]
127
+ except Exception as e: # noqa: BLE001
128
+ err = e
129
+ print(f" judge batch failed after 3 tries: {err}", file=sys.stderr)
130
+ return []
131
+
132
+
133
+ def label(reactions, backend, model, cmd=None, batch=30, workers=8):
134
+ cache = load_cache()
135
+ todo = [r for r in {key(r): r for r in reactions}.values() if key(r) not in cache]
136
+ if todo:
137
+ os.makedirs(os.path.dirname(CACHE), exist_ok=True)
138
+ batches = [todo[i:i + batch] for i in range(0, len(todo), batch)]
139
+ print(f"Judging {len(todo)} new messages in {len(batches)} batches with {backend}"
140
+ f"{' / ' + model if model else ''} ...", file=sys.stderr)
141
+ with open(CACHE, "a") as f, ThreadPoolExecutor(workers) as ex:
142
+ for n, res in enumerate(ex.map(lambda b: _label_batch(b, backend, model, cmd), batches), 1):
143
+ for k, lab in res:
144
+ lab = {**lab, "key": k, "judge": f"{backend}:{model or 'default'}"}
145
+ cache[k] = lab
146
+ f.write(json.dumps(lab, ensure_ascii=False) + "\n")
147
+ f.flush()
148
+ print(f" {n}/{len(batches)}", file=sys.stderr)
149
+ return {r.id: cache[key(r)] for r in reactions if key(r) in cache}
swearbench/score.py ADDED
@@ -0,0 +1,233 @@
1
+ """Turn labels into leaderboards."""
2
+ from __future__ import annotations
3
+
4
+ import random
5
+ import re
6
+ from collections import Counter, defaultdict
7
+
8
+ MODE_WEIGHT = {
9
+ "fabrication": 3, "overreach": 3, "giving_up": 3, "regression": 2.5,
10
+ "ignored": 2, "repeat": 2, "incomplete": 1.5, "insult": 1.5,
11
+ "profanity": 1, "shouting": 0.5, "sarcasm": 0.5, "slow": 0.5, "taste": 0.5,
12
+ }
13
+ BREACH = {"fabrication", "overreach", "giving_up", "regression", "ignored", "repeat", "incomplete"}
14
+ TASTE_DISCOUNT = 0.25
15
+ FRICTION_K = 0.25
16
+ REGRET_WINDOW_DAYS = 7
17
+ INTERRUPT_RAGE = 1.5
18
+ SWEAR = re.compile(r"\b(fuck\w*|shit\w*|wtf|damn\w*|crap\w*|bullshit|stupid|idiot\w*|dumb\w*|moron\w*|"
19
+ r"bitch\w*|asshole|sucks?)\b", re.I)
20
+
21
+
22
+ def rage(lab):
23
+ """Anger squared, so one blowup outweighs several grumbles, plus how you got mad.
24
+ Critique of taste while iterating, with no breach of trust, counts a quarter."""
25
+ if lab.get("agent_authored") or lab.get("target") != "model" or lab.get("anger", 0) < 1:
26
+ return 0.0
27
+ modes = set(lab.get("modes", []))
28
+ r = lab["anger"] ** 2 + sum(MODE_WEIGHT.get(m, 0) for m in modes)
29
+ return r * TASTE_DISCOUNT if "taste" in modes and not modes & BREACH else r
30
+
31
+
32
+ def blame_shares(reactions, labels):
33
+ """For a complaint about earlier work, split it across the models that worked in the same workspace
34
+ in the week before, outside the current session, by how many turns each had there."""
35
+ import datetime as dt
36
+
37
+ def t(r):
38
+ return dt.datetime.fromisoformat(r.ts.replace("Z", "+00:00"))
39
+
40
+ by_ws = defaultdict(list)
41
+ for r in reactions:
42
+ if r.workspace and r.ts:
43
+ by_ws[r.workspace].append(r)
44
+ for rs in by_ws.values():
45
+ rs.sort(key=lambda r: r.ts)
46
+ shares = {}
47
+ window = dt.timedelta(days=REGRET_WINDOW_DAYS)
48
+ for r in reactions:
49
+ lab = labels.get(r.id)
50
+ if not lab or not lab.get("blames_earlier") or rage(lab) == 0 or not r.workspace:
51
+ continue
52
+ now = t(r)
53
+ prior = Counter(o.model for o in by_ws[r.workspace]
54
+ if o.session != r.session and o.ts < r.ts and now - t(o) <= window and o.model)
55
+ total = sum(prior.values())
56
+ if total:
57
+ shares[r.id] = {m: n / total for m, n in prior.items()}
58
+ return shares
59
+
60
+
61
+ def _own(r, l, shares):
62
+ return rage(l) * shares[r.id].get(r.model, 0.0) if r.id in shares else rage(l)
63
+
64
+
65
+ def _friction(rows, n_int, shares, inbound_per_turn):
66
+ n = len(rows)
67
+ total = sum(_own(r, l, shares) for r, l in rows) + inbound_per_turn * n + INTERRUPT_RAGE * n_int
68
+ clean = sum(1 for _, l in rows if rage(l) == 0 and l.get("satisfaction", 0) >= 0)
69
+ sat = sum(l.get("satisfaction", 0) for _, l in rows) / n
70
+ rage100 = 100 * total / n
71
+ clean_pct = 100 * clean / n
72
+ return {
73
+ "n": n, "friction": 100 - FRICTION_K * rage100 + 10 * sat, "clean_pct": clean_pct, "rage100": rage100,
74
+ "angry_pct": 100 * sum(rage(l) > 0 for _, l in rows) / n, "sat": sat,
75
+ "int100": 100 * n_int / n,
76
+ "swears100": 100 * sum(len(SWEAR.findall(r.text)) for r, _ in rows) / n,
77
+ }
78
+
79
+
80
+ def _verdict(rows):
81
+ """True = you accepted the work, False = you rejected it, None = it stopped or moved elsewhere without a verdict."""
82
+ last = rows[-1][1]
83
+ if rage(last) > 0 or last.get("satisfaction", 0) < 0:
84
+ return False
85
+ if last.get("satisfaction", 0) >= 1:
86
+ return True
87
+ return None
88
+
89
+
90
+ def _stats(sessions, n_int, merged, shares, inbound_per_turn):
91
+ rows = [x for s in sessions for x in s]
92
+ st = _friction(rows, n_int, shares, inbound_per_turn)
93
+ st["sessions"] = len(sessions)
94
+ verdicts = [v for v in map(_verdict, sessions) if v is not None]
95
+ st["decided"] = len(verdicts)
96
+ st["outcome"] = 100 * sum(verdicts) / len(verdicts) if verdicts else 0.0
97
+ tracked = [s for s in sessions if merged["since"] and s[0][0].ts >= merged["since"]]
98
+ st["merged_n"] = len(tracked)
99
+ st["merged_pct"] = 100 * sum(s[0][0].session in merged["ids"] for s in tracked) / len(tracked) if tracked else None
100
+ st["score"] = (st["friction"] + st["outcome"]) / 2
101
+ return st
102
+
103
+
104
+ def _ci(sessions, n_int, merged, shares, inbound_per_turn, field, k=300):
105
+ rng = random.Random(0)
106
+ xs = sorted(_stats(rng.choices(sessions, k=len(sessions)), n_int, merged, shares, inbound_per_turn)[field]
107
+ for _ in range(k))
108
+ return xs[int(k * .05)], xs[int(k * .95)]
109
+
110
+
111
+ def modes_share(rows):
112
+ c = Counter()
113
+ for _, l in rows:
114
+ if rage(l) > 0:
115
+ for m in l.get("modes", []):
116
+ c[m] += MODE_WEIGHT.get(m, 0)
117
+ total = sum(c.values()) or 1
118
+ return {m: c[m] / total for m in MODE_WEIGHT if c[m]}
119
+
120
+
121
+ def build(corpus, labels, min_reactions=40, exclude=()):
122
+ per = defaultdict(list)
123
+ for r in corpus.reactions:
124
+ lab = labels.get(r.id)
125
+ if lab and not lab.get("agent_authored") and r.model and r.model not in exclude:
126
+ per[r.model].append((r, lab))
127
+ ints = Counter(i.model for i in corpus.interrupts)
128
+ shares = blame_shares(corpus.reactions, labels)
129
+ inbound = Counter()
130
+ for r in corpus.reactions:
131
+ if r.id in shares:
132
+ for m, share in shares[r.id].items():
133
+ if m != r.model:
134
+ inbound[m] += rage(labels[r.id]) * share
135
+
136
+ board = []
137
+ for model, rows in per.items():
138
+ if len(rows) < min_reactions:
139
+ continue
140
+ by_session = defaultdict(list)
141
+ for r, l in sorted(rows, key=lambda x: x[0].ts):
142
+ by_session[r.session].append((r, l))
143
+ sessions = list(by_session.values())
144
+ merged = {"ids": corpus.merged, "since": corpus.pr_tracking_since}
145
+ inb = inbound[model] / len(rows)
146
+ s = _stats(sessions, ints[model], merged, shares, inb)
147
+ s["model"], s["modes"] = model, modes_share(rows)
148
+ s["regret_in"], s["regret_out"] = inbound[model], sum(
149
+ rage(l) * (1 - shares[r.id].get(model, 0.0)) for r, l in rows if r.id in shares)
150
+ s["blowups"] = sum(1 for _, l in rows if rage(l) > 0 and l.get("anger", 0) >= 4)
151
+ s["ci"] = _ci(sessions, ints[model], merged, shares, inb, "score")
152
+ s["outcome_ci"] = _ci(sessions, ints[model], merged, shares, inb, "outcome")
153
+ s["worst"] = [l for _, l in sorted(rows, key=lambda x: -rage(x[1]))[:3] if rage(l) > 0]
154
+ board.append(s)
155
+ board.sort(key=lambda s: -s["score"])
156
+
157
+ tokens = []
158
+ for model, rows in per.items():
159
+ you = corpus.usage.tokens.get(model, {}).get("you")
160
+ start = corpus.usage.first_ts.get(model)
161
+ if not you or not you[1] or not start:
162
+ continue
163
+ win = [(r, l) for r, l in rows if r.ts >= start]
164
+ if len(win) < min_reactions:
165
+ continue
166
+ n_int = sum(1 for i in corpus.interrupts if i.model == model and i.ts >= start)
167
+ out_m = you[1] / 1e6
168
+ tokens.append({
169
+ "model": model, "out_m": out_m, "agents_out_m": corpus.usage.tokens[model].get("agents", [0, 0, 0])[1] / 1e6,
170
+ "total_b": sum(you) / 1e9, "n": len(win),
171
+ "rage_per_m": (sum(_own(r, l, shares) for r, l in win) + INTERRUPT_RAGE * n_int) / out_m,
172
+ "angry_per_m": sum(rage(l) > 0 for _, l in win) / out_m, "out_per_reaction": you[1] / len(win),
173
+ })
174
+ tokens.sort(key=lambda t: t["rage_per_m"])
175
+
176
+ sent = set(per)
177
+ agent_only = sorted(m for m, o in corpus.usage.tokens.items() if m not in sent)
178
+ small = sorted((m, len(r)) for m, r in per.items() if len(r) < min_reactions)
179
+ return {"board": board, "tokens": tokens, "agent_only": agent_only, "small": small,
180
+ "n_reactions": sum(len(r) for r in per.values()), "n_interrupts": len(corpus.interrupts)}
181
+
182
+
183
+ def _merged_cell(s):
184
+ return "—" if s["merged_pct"] is None else f"{s['merged_pct']:.0f}% of {s['merged_n']}"
185
+
186
+
187
+ def markdown(res, quotes=True):
188
+ o = ["# SwearBench\n",
189
+ f"{res['n_reactions']} of your reactions to AI replies, {res['n_interrupts']} interrupts.\n",
190
+ "| # | Model | SwearBench ↑ | 90% CI | Ships | Friction | Decided / sessions | Merged PR* | Reactions | Clean turns | "
191
+ "Rage /100 turns | Angry msgs | 4/4 blowups | Regret charged in / passed on | Interrupts /100 | Swears /100 | "
192
+ "Avg satisfaction |",
193
+ "|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|"]
194
+ for i, s in enumerate(res["board"], 1):
195
+ o.append(f"| {i} | {s['model']} | **{s['score']:.1f}** | {s['ci'][0]:.0f}–{s['ci'][1]:.0f} | {s['outcome']:.0f}% | {s['friction']:.1f} | "
196
+ f"{s['decided']} / {s['sessions']} | "
197
+ f"{_merged_cell(s)} | {s['n']} | "
198
+ f"{s['clean_pct']:.0f}% | {s['rage100']:.1f} | {s['angry_pct']:.1f}% | {s['blowups']} | "
199
+ f"{s['regret_in']:.0f} / {s['regret_out']:.0f} | {s['int100']:.1f} | "
200
+ f"{s['swears100']:.1f} | {s['sat']:+.2f} |")
201
+ if res["tokens"]:
202
+ o += ["\n## Per token of work\n",
203
+ "Output tokens the model produced in sessions you drove; reactions counted from the first logged token on.\n",
204
+ "| # | Model | Your output tok | Agent-run output tok | Your total tok | Reactions | Rage per 1M out tok ↓ | "
205
+ "Angry msgs per 1M | Out tok per reaction |",
206
+ "|---|---|---|---|---|---|---|---|---|"]
207
+ for i, t in enumerate(res["tokens"], 1):
208
+ o.append(f"| {i} | {t['model']} | {t['out_m']:.1f}M | {t['agents_out_m']:.1f}M | {t['total_b']:.2f}B | {t['n']} | "
209
+ f"**{t['rage_per_m']:.1f}** | {t['angry_per_m']:.1f} | {t['out_per_reaction'] / 1000:.1f}k |")
210
+ o += ["\n## How you get mad at each model (share of rage points)\n",
211
+ "| Model | " + " | ".join(MODE_WEIGHT) + " |", "|---|" + "---|" * len(MODE_WEIGHT)]
212
+ for s in res["board"]:
213
+ o.append(f"| {s['model']} | " + " | ".join(f"{100 * s['modes'][m]:.0f}%" if m in s["modes"] else "·"
214
+ for m in MODE_WEIGHT) + " |")
215
+ if quotes:
216
+ o.append("\n## Hall of shame\n\n> Contains your own words. Review before sharing.\n")
217
+ for s in res["board"]:
218
+ o.append(f"**{s['model']}**")
219
+ o += [f"- [{l['anger']}/4, {', '.join(l.get('modes', [])) or '—'}] “{l.get('quote', '')}”" for l in s["worst"]]
220
+ o.append("")
221
+ if res["small"]:
222
+ o.append("Too few reactions to rank: " + ", ".join(f"{m} ({n})" for m, n in res["small"]))
223
+ if res["agent_only"]:
224
+ o.append("\nOnly ever ran as subagents / headless runs (never ranked): " + ", ".join(res["agent_only"]))
225
+ o.append(f"\nSwearBench = (Friction + Ships) / 2. Friction = 100 − {FRICTION_K} × rage per 100 turns + 10 × "
226
+ "avg satisfaction; rage = anger² + mode weights, only when aimed at the model, a quarter for taste-only "
227
+ "critique; each interrupt adds 1.5. Complaints about earlier work are charged to the models that worked "
228
+ f"in that workspace in the {REGRET_WINDOW_DAYS} days before. "
229
+ "Ships = accepted ÷ (accepted + rejected) sessions; sessions that stop or are handed off without a verdict "
230
+ "are left out, so limits and model switches don't count as failures. "
231
+ "*Merged PR is informational (T3 Code only, sessions after it started tracking PRs) and not in the score. "
232
+ "Intervals bootstrap over sessions. See chart.svg for swearing vs. result.")
233
+ return "\n".join(o) + "\n"
swearbench/sources.py ADDED
@@ -0,0 +1,317 @@
1
+ """Readers for local agent logs. Each yields human reactions, interrupts and token usage."""
2
+ from __future__ import annotations
3
+
4
+ import glob
5
+ import json
6
+ import os
7
+ import sqlite3
8
+ from collections import defaultdict
9
+ from dataclasses import dataclass, field
10
+
11
+ HOME = os.path.expanduser("~")
12
+
13
+
14
+ @dataclass
15
+ class Reaction:
16
+ id: str
17
+ source: str
18
+ session: str
19
+ ts: str
20
+ text: str
21
+ prev_reply: str
22
+ model: str
23
+ workspace: str = ""
24
+
25
+
26
+ @dataclass
27
+ class Interrupt:
28
+ source: str
29
+ session: str
30
+ ts: str
31
+ model: str
32
+
33
+
34
+ @dataclass
35
+ class Usage:
36
+ """Per model and origin ("you" = sessions a human drove, "agents" = subagents/headless runs)."""
37
+ tokens: dict = field(default_factory=lambda: defaultdict(lambda: defaultdict(lambda: [0, 0, 0])))
38
+ first_ts: dict = field(default_factory=dict)
39
+
40
+ def add(self, model, origin, ts, inp, out, cache):
41
+ model = norm(model)
42
+ t = self.tokens[model][origin]
43
+ t[0] += inp
44
+ t[1] += out
45
+ t[2] += cache
46
+ if origin == "you" and ts and (model not in self.first_ts or ts < self.first_ts[model]):
47
+ self.first_ts[model] = ts
48
+
49
+
50
+ @dataclass
51
+ class Corpus:
52
+ reactions: list = field(default_factory=list)
53
+ interrupts: list = field(default_factory=list)
54
+ usage: Usage = field(default_factory=Usage)
55
+ found: list = field(default_factory=list)
56
+ merged: set = field(default_factory=set)
57
+ pr_tracking_since: str | None = None
58
+
59
+
60
+ def norm(model):
61
+ if not model:
62
+ return None
63
+ model = model.strip().lower()
64
+ if model.startswith("claude-"):
65
+ model = model.replace(".", "-")
66
+ return model
67
+
68
+
69
+ def workspace_of(path):
70
+ """Repo-level key for a working directory; worktrees of one repo share it."""
71
+ if not path:
72
+ return ""
73
+ parts = path.rstrip("/").split("/")
74
+ if "worktrees" in parts and parts.index("worktrees") + 1 < len(parts):
75
+ return parts[parts.index("worktrees") + 1]
76
+ return parts[-1]
77
+
78
+
79
+ def ms_to_iso(ms):
80
+ import datetime
81
+ return datetime.datetime.fromtimestamp(ms / 1000, datetime.timezone.utc).isoformat().replace("+00:00", "Z")
82
+
83
+
84
+ def _jsonl(path):
85
+ with open(path, errors="replace") as f:
86
+ for line in f:
87
+ try:
88
+ yield json.loads(line)
89
+ except ValueError:
90
+ continue
91
+
92
+
93
+ def _ro(path):
94
+ return sqlite3.connect(f"file:{path}?mode=ro", uri=True)
95
+
96
+
97
+ # --- Claude Code -----------------------------------------------------------------------------
98
+
99
+ HEADLESS_ENTRYPOINTS = {"sdk-cli", "sdk-py"}
100
+
101
+
102
+ def _claude_text(content):
103
+ if isinstance(content, str):
104
+ return content
105
+ if any(isinstance(p, dict) and p.get("type") == "tool_result" for p in content):
106
+ return None
107
+ parts = [p.get("text", "") for p in content if isinstance(p, dict) and p.get("type") == "text"]
108
+ return "\n".join(parts) if parts else None
109
+
110
+
111
+ def read_claude(corpus, root, skip_session=lambda entrypoint: False, exclude_dirs=()):
112
+ files = glob.glob(os.path.join(root, "**", "*.jsonl"), recursive=True)
113
+ if not files:
114
+ return
115
+ corpus.found.append(f"Claude Code: {len(files)} transcripts")
116
+ for path in files:
117
+ excluded = any(d in path for d in exclude_dirs)
118
+ subagent = "/subagents/" in path
119
+ entrypoint = None
120
+ model = None
121
+ reply = ""
122
+ cwd = ""
123
+ usage = {}
124
+ pending = []
125
+ for d in _jsonl(path):
126
+ kind = d.get("type")
127
+ msg = d.get("message") or {}
128
+ if kind == "assistant":
129
+ if msg.get("model") and msg["model"] != "<synthetic>":
130
+ model = msg["model"]
131
+ if msg.get("usage"):
132
+ usage[msg.get("id") or d.get("uuid")] = (model, msg["usage"], d.get("isSidechain"), d.get("timestamp"))
133
+ text = _claude_text(msg.get("content") or [])
134
+ if text and text.strip():
135
+ reply = text
136
+ elif kind == "user" and not d.get("isSidechain") and not d.get("isMeta"):
137
+ entrypoint = entrypoint or d.get("entrypoint")
138
+ cwd = cwd or d.get("cwd", "")
139
+ text = _claude_text(msg.get("content") or "")
140
+ if not text or d.get("promptSource") == "system":
141
+ continue
142
+ text = text.strip()
143
+ if text.startswith("[Request interrupted by user"):
144
+ if model:
145
+ corpus.interrupts.append(Interrupt("claude", path, d.get("timestamp", ""), norm(model)))
146
+ continue
147
+ if text.startswith(("<", "Caveat:")) or not model:
148
+ continue
149
+ pending.append(Reaction(d.get("uuid", ""), "claude", path, d.get("timestamp", ""), text,
150
+ reply[-900:], norm(model), workspace_of(cwd)))
151
+ reply = ""
152
+ headless = entrypoint in HEADLESS_ENTRYPOINTS
153
+ human = not (subagent or headless or excluded)
154
+ if human and not skip_session(entrypoint):
155
+ corpus.reactions += pending
156
+ for model_id, u, side, ts in usage.values():
157
+ if excluded:
158
+ continue
159
+ origin = "you" if human and not side else "agents"
160
+ corpus.usage.add(model_id, origin, ts, u.get("input_tokens", 0), u.get("output_tokens", 0),
161
+ u.get("cache_read_input_tokens", 0) + u.get("cache_creation_input_tokens", 0))
162
+
163
+
164
+ # --- Codex -------------------------------------------------------------------------------------
165
+
166
+ def read_codex(corpus, root, skip_originator=lambda originator: False):
167
+ files = glob.glob(os.path.join(root, "**", "*.jsonl"), recursive=True)
168
+ if not files:
169
+ return
170
+ corpus.found.append(f"Codex: {len(files)} sessions")
171
+ for path in files:
172
+ human, skip, model, reply, cwd = False, False, None, "", ""
173
+ for d in _jsonl(path):
174
+ p = d.get("payload") or {}
175
+ kind = d.get("type")
176
+ if kind == "session_meta":
177
+ source = p.get("source")
178
+ human = isinstance(source, str) and source != "exec" and p.get("originator") != "codex_exec"
179
+ skip = skip_originator(p.get("originator") or "")
180
+ cwd = p.get("cwd") or ""
181
+ elif kind == "turn_context":
182
+ model = p.get("model") or model
183
+ elif kind == "event_msg":
184
+ t = p.get("type")
185
+ if t == "agent_message":
186
+ reply = p.get("message") or reply
187
+ elif t == "user_message" and human and not skip and model:
188
+ text = (p.get("message") or "").strip()
189
+ if text and not text.startswith("<"):
190
+ corpus.reactions.append(Reaction(f"{path}:{d.get('timestamp')}", "codex", path,
191
+ d.get("timestamp", ""), text, reply[-900:], norm(model),
192
+ workspace_of(cwd)))
193
+ reply = ""
194
+ elif t == "turn_aborted" and human and not skip and model:
195
+ corpus.interrupts.append(Interrupt("codex", path, d.get("timestamp", ""), norm(model)))
196
+ elif t == "token_count" and model and (p.get("info") or {}).get("last_token_usage"):
197
+ u = p["info"]["last_token_usage"]
198
+ cached = u.get("cached_input_tokens", 0)
199
+ corpus.usage.add(model, "you" if human else "agents", d.get("timestamp"),
200
+ u.get("input_tokens", 0) - cached, u.get("output_tokens", 0), cached)
201
+
202
+
203
+ # --- OpenCode ----------------------------------------------------------------------------------
204
+
205
+ def read_opencode(corpus, db, skip_sessions=frozenset()):
206
+ if not os.path.exists(db):
207
+ return
208
+ c = _ro(db)
209
+ try:
210
+ sessions = dict(c.execute("select id, parent_id from session"))
211
+ dirs = dict(c.execute("select id, directory from session"))
212
+ except sqlite3.Error:
213
+ return
214
+ corpus.found.append(f"OpenCode: {len(sessions)} sessions")
215
+ texts = defaultdict(list)
216
+ for mid, data in c.execute("select message_id, data from part order by time_created"):
217
+ p = json.loads(data)
218
+ if p.get("type") == "text" and not p.get("synthetic"):
219
+ texts[mid].append(p.get("text", ""))
220
+ state = {}
221
+ for mid, sid, created, data in c.execute(
222
+ "select id, session_id, time_created, data from message order by session_id, time_created"):
223
+ d = json.loads(data)
224
+ model, reply = state.get(sid, (None, ""))
225
+ human = sessions.get(sid) is None
226
+ ts = ms_to_iso(created)
227
+ text = "\n".join(texts.get(mid, [])).strip()
228
+ if d.get("role") == "assistant":
229
+ model = "opencode/" + d.get("modelID", "?")
230
+ t = d.get("tokens") or {}
231
+ corpus.usage.add(model, "you" if human else "agents", ts, t.get("input", 0),
232
+ t.get("output", 0) + t.get("reasoning", 0), (t.get("cache") or {}).get("read", 0))
233
+ if (d.get("error") or {}).get("name") == "MessageAbortedError" and human and sid not in skip_sessions:
234
+ corpus.interrupts.append(Interrupt("opencode", sid, ts, norm(model)))
235
+ state[sid] = (model, text or reply)
236
+ elif d.get("role") == "user":
237
+ if human and model and text and sid not in skip_sessions:
238
+ corpus.reactions.append(Reaction(mid, "opencode", sid, ts, text, reply[-900:], norm(model),
239
+ workspace_of(dirs.get(sid))))
240
+ state[sid] = (model, "")
241
+
242
+
243
+ # --- T3 Code (optional; gives exact per-turn model attribution across providers) ----------------
244
+
245
+ def t3_db():
246
+ base = os.path.join(HOME, ".config", "t3", "userdata")
247
+ for name in ("statev2.sqlite", "state.sqlite"):
248
+ path = os.path.join(base, name)
249
+ if os.path.exists(path):
250
+ return path
251
+ return None
252
+
253
+
254
+ def t3_provider_sessions(db):
255
+ out = defaultdict(set)
256
+ for provider, cursor in _ro(db).execute("select provider_name, resume_cursor_json from provider_session_runtime"):
257
+ cur = json.loads(cursor or "{}")
258
+ out[provider].add(cur.get("sessionId") or cur.get("threadId"))
259
+ return out
260
+
261
+
262
+ def read_t3(corpus, db):
263
+ c = _ro(db)
264
+ children = {r[0] for r in c.execute("select child_thread_id from orchestration_v2_projection_subagents "
265
+ "where child_thread_id is not null")} if _has(c, "orchestration_v2_projection_subagents") else set()
266
+ turn_model = {}
267
+ for mid, payload in c.execute("select json_extract(payload_json,'$.messageId'), payload_json from orchestration_events "
268
+ "where event_type='thread.turn-start-requested'"):
269
+ turn_model[mid] = (json.loads(payload).get("modelSelection") or {}).get("model")
270
+ project = dict(c.execute("select thread_id, project_id from projection_threads"))
271
+ by_thread = defaultdict(list)
272
+ for mid, tid, role, text, ts in c.execute("select message_id, thread_id, role, text, created_at from projection_thread_messages "
273
+ "where role in ('user','assistant') order by created_at"):
274
+ if tid not in children:
275
+ by_thread[tid].append((mid, role, text or "", ts))
276
+ corpus.found.append(f"T3 Code: {len(by_thread)} threads")
277
+ sent = defaultdict(list)
278
+ for tid, msgs in by_thread.items():
279
+ current, reply = None, ""
280
+ for mid, role, text, ts in msgs:
281
+ if role == "assistant":
282
+ reply = text if text.strip() else reply
283
+ continue
284
+ if current and text.strip():
285
+ corpus.reactions.append(Reaction(mid, "t3", tid, ts, text.strip(), reply[-900:], norm(current),
286
+ project.get(tid) or ""))
287
+ current = turn_model.get(mid) or current
288
+ sent[tid].append((ts, current))
289
+ reply = ""
290
+ if _has(c, "projection_thread_pull_requests"):
291
+ corpus.merged |= {tid for tid, state in c.execute(
292
+ "select thread_id, json_extract(snapshot_json,'$.state') from projection_thread_pull_requests") if state == "merged"}
293
+ corpus.pr_tracking_since = c.execute("select min(linked_at) from projection_thread_pull_requests").fetchone()[0]
294
+ for tid, ts in c.execute("select json_extract(payload_json,'$.threadId'), occurred_at from orchestration_events "
295
+ "where event_type='thread.turn-interrupt-requested'"):
296
+ before = [m for t, m in sent.get(tid, []) if t <= ts]
297
+ if before and before[-1]:
298
+ corpus.interrupts.append(Interrupt("t3", tid, ts, norm(before[-1])))
299
+
300
+
301
+ def _has(c, table):
302
+ return c.execute("select 1 from sqlite_master where name=?", (table,)).fetchone() is not None
303
+
304
+
305
+ def collect(use_t3=True, exclude_dirs=()):
306
+ corpus = Corpus()
307
+ db = t3_db() if use_t3 else None
308
+ t3_sessions = t3_provider_sessions(db) if db else {}
309
+ if db:
310
+ read_t3(corpus, db)
311
+ read_claude(corpus, os.path.join(HOME, ".claude", "projects"),
312
+ skip_session=(lambda ep: ep == "sdk-ts") if db else (lambda ep: False), exclude_dirs=exclude_dirs)
313
+ read_codex(corpus, os.path.join(HOME, ".codex", "sessions"),
314
+ skip_originator=(lambda o: "t3" in o.lower()) if db else (lambda o: False))
315
+ read_opencode(corpus, os.path.join(HOME, ".local", "share", "opencode", "opencode.db"),
316
+ skip_sessions=frozenset(t3_sessions.get("opencode", ())))
317
+ return corpus
@@ -0,0 +1,130 @@
1
+ Metadata-Version: 2.5
2
+ Name: swearbench
3
+ Version: 0.2.1
4
+ Summary: Rank AI coding models by how much they made you swear and how often their work shipped, from your own local chat logs.
5
+ License-Expression: MIT
6
+ License-File: LICENSE
7
+ Requires-Python: >=3.10
8
+ Description-Content-Type: text/markdown
9
+
10
+ # SwearBench
11
+
12
+ **Which AI coding model made you swear the least, and still shipped?**
13
+
14
+ Public benchmarks measure what models can do. SwearBench measures how they made *you* feel: it reads your
15
+ own local agent logs, finds every message you sent in reaction to a model's reply, has an LLM judge label how
16
+ mad you were and why, checks whether the work ended up accepted, and ranks the models.
17
+
18
+ Example from the author's own logs (~4,600 messages, July–October 2026):
19
+
20
+ ![swearing vs. result](docs/example-chart.svg)
21
+
22
+ ![leaderboard card](docs/example-card.svg)
23
+
24
+ ## Run it
25
+
26
+ ```sh
27
+ uvx swearbench
28
+ ```
29
+
30
+ (or `pipx run swearbench`, or `pip install swearbench`)
31
+
32
+ It prints a leaderboard and writes `swearbench-out/report.md`, `chart.svg` (swearing vs. result), `card.svg` and `results.json`.
33
+ Before anything leaves your machine it tells you how many messages it will send to the judge and asks.
34
+
35
+ | Flag | |
36
+ |---|---|
37
+ | `--dry-run` | show what logs were found and how many messages would be judged |
38
+ | `--judge claude-cli\|codex-cli\|anthropic\|openai\|command` | who labels your messages (default: first of `claude`, `codex`, `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`) |
39
+ | `--judge-model ID` | model for the judge |
40
+ | `--judge command --judge-cmd "ollama run qwen3"` | any command that reads a prompt on stdin and prints the reply; keeps everything local |
41
+ | `--since 2026-09-01` | only recent reactions |
42
+ | `--exclude MODEL` | drop a model from the ranking |
43
+ | `--no-quotes` | leave the hall of shame out of the report |
44
+
45
+ Labels are cached in `~/.cache/swearbench/`, so re-runs only judge new messages.
46
+
47
+ ## Where it looks
48
+
49
+ | Tool | Location |
50
+ |---|---|
51
+ | Claude Code | `~/.claude/projects/**/*.jsonl` |
52
+ | Codex CLI | `~/.codex/sessions/**/*.jsonl` |
53
+ | OpenCode | `~/.local/share/opencode/opencode.db` |
54
+ | T3 Code | `~/.config/t3/userdata/state*.sqlite` (used for exact per-turn model attribution when present) |
55
+
56
+ Only messages **you** typed count. Subagent transcripts, headless runs (`claude -p`, `codex exec`) and messages
57
+ the judge flags as agent-written are skipped. Models that only ever ran as subagents are listed, never ranked.
58
+
59
+ ## How it scores
60
+
61
+ Every message you send is charged to the model whose reply you were answering. The judge labels it:
62
+
63
+ - **anger** 0–4, and **who it's aimed at**: the model, or something external (swearing at your cloud
64
+ provider is not the model's fault; "this looks fucking cool" is not anger)
65
+ - **how** you got mad, weighted by how bad it is:
66
+
67
+ | Mode | Weight | | Mode | Weight |
68
+ |---|---|---|---|---|
69
+ | fabrication (claimed success that wasn't) | 3 | | incomplete / lazy | 1.5 |
70
+ | overreach (did things you didn't ask) | 3 | | insult | 1.5 |
71
+ | giving up on it | 3 | | profanity | 1 |
72
+ | regression (broke what worked) | 2.5 | | shouting | 0.5 |
73
+ | ignored an instruction | 2 | | sarcasm | 0.5 |
74
+ | made you repeat yourself | 2 | | slow / verbose | 0.5 |
75
+ | | | | taste (critique while iterating on looks) | 0.5 |
76
+
77
+ - **satisfaction** −2 (rejects the work) to +2 (praise)
78
+ - whether it **blames earlier work** (something shipped before the last reply is broken or missing)
79
+
80
+ Rage for a message is built to match how it felt, not how often it happened:
81
+
82
+ - **Severity beats frequency.** Rage = anger² + mode weights, so one 4/4 blowup (16) outweighs four
83
+ 1/4 grumbles (4). The report counts 4/4 blowups per model.
84
+ - **Taste isn't failure.** "That looks lame" while iterating on a design, with no broken rule, lie,
85
+ regression or ignored instruction, counts a quarter.
86
+ - **Regret goes to whoever caused it.** A complaint about earlier work ("why did X disappear") is charged to
87
+ the models that worked in the same repo in the week before, split by how many turns each had there, not
88
+ to the model that happens to be fixing it.
89
+ - Each interrupt adds 1.5.
90
+
91
+ The headline is half friction, half result:
92
+
93
+ ```
94
+ Friction = 100 − 0.25 × rage per 100 turns + 10 × average satisfaction
95
+ Ships = accepted ÷ (accepted + rejected) sessions
96
+ SwearBench = (Friction + Ships) / 2
97
+ ```
98
+
99
+ Friction deliberately ignores how *often* you were annoyed (a model you use for lots of quick "merge it"
100
+ turns would look calm by volume alone); it only counts how much rage piled up per turn.
101
+ A session's verdict is your last reaction in it. Sessions that stop or move to another model without a
102
+ verdict (usage limits, "pick up the work" in a new thread) are left out of Ships rather than counted as
103
+ failures; switching away in anger still counts as a rejection.
104
+ Friction alone rewards a model that is pleasant but never finishes; Ships alone ignores what it cost you
105
+ to get there. The chart plots the two against each other.
106
+
107
+ Intervals are a 90% bootstrap over sessions; models with fewer than 40 reactions aren't ranked.
108
+ With T3 Code, the report also shows the share of sessions that ended in a merged PR. It is informational
109
+ only, since T3 records PRs only from when it started tracking them.
110
+
111
+ **Per token of work.** If the logs carry token usage, the report adds a second ranking: rage per million output
112
+ tokens the model produced in sessions you drove. A model that does twice the work per message gets credit for it.
113
+ Subagent token use is shown separately.
114
+
115
+ ## Caveats
116
+
117
+ - n = 1. It measures you, your tasks and your mood as much as the models. Models used in different months
118
+ did different work; the report tells you counts, not causes.
119
+ - The judge is a model too. If it belongs to a family being ranked, SwearBench says so; re-run with another
120
+ `--judge` and compare.
121
+ - Regret attribution is a heuristic: it blames whoever worked in the same repo during the previous week,
122
+ not the session that actually introduced the problem.
123
+ - The weights are opinions. Rage, taste and severity weights live at the top of `score.py`; change them
124
+ and re-run, labels are cached.
125
+ - Deleted or rotated logs mean missing data, especially for the per-token view.
126
+ - `report.md` quotes your own messages. Read it before you share it. The card and chart have no quotes.
127
+
128
+ ## License
129
+
130
+ MIT
@@ -0,0 +1,13 @@
1
+ swearbench/__init__.py,sha256=HfjVOrpTnmZ-xVFCYSVmX50EXaBQeJteUHG-PD6iQs8,22
2
+ swearbench/__main__.py,sha256=E6Gls0DNz8GQK2K-kOUIx8cYhgANW_CH54VKrfCfs14,52
3
+ swearbench/card.py,sha256=xysCWH-6cRQ9fm71GIEKpnrLGp58e5rFShpQyDpaS08,3412
4
+ swearbench/chart.py,sha256=EfyUCD8xmyt_vzeJDz-lWkDotW2yJcKVQrIWN3b5T-s,6595
5
+ swearbench/cli.py,sha256=VzigcBt5oGTFVYKCQaXaoMJWaalHK-DUU8ku7Dbh0Bg,4108
6
+ swearbench/judge.py,sha256=FxrpUHlj5LXJExWW3FJ8v4kKQuBz-yTbcfU6R0Ik3i8,7452
7
+ swearbench/score.py,sha256=PJpV_yRiI2nfen0VcM7xS0pYGjYPqK0-fKdAshZ93sc,11518
8
+ swearbench/sources.py,sha256=LJN1jBkFksvZja-h12Y2UxjmpdJuXVmlJWL6JCqpINI,13645
9
+ swearbench-0.2.1.dist-info/METADATA,sha256=a-Sgjw_t4X6u89boms860vp9___6M7wXmVgvK9k4EfE,6293
10
+ swearbench-0.2.1.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
11
+ swearbench-0.2.1.dist-info/entry_points.txt,sha256=qrmKvKi9_IF4klWqvGhS4o6RU6kt_CgCaahl67rev1s,51
12
+ swearbench-0.2.1.dist-info/licenses/LICENSE,sha256=_OBJk5LyJLaZeO_rY8OZO6IKaDvmb0Wgf9Wc0PI_1Fg,1064
13
+ swearbench-0.2.1.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ swearbench = swearbench.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 AlenHay
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.