swearbench 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- swearbench/__init__.py +1 -0
- swearbench/__main__.py +5 -0
- swearbench/card.py +66 -0
- swearbench/chart.py +134 -0
- swearbench/cli.py +77 -0
- swearbench/judge.py +149 -0
- swearbench/score.py +233 -0
- swearbench/sources.py +317 -0
- swearbench-0.2.1.dist-info/METADATA +130 -0
- swearbench-0.2.1.dist-info/RECORD +13 -0
- swearbench-0.2.1.dist-info/WHEEL +4 -0
- swearbench-0.2.1.dist-info/entry_points.txt +2 -0
- swearbench-0.2.1.dist-info/licenses/LICENSE +21 -0
swearbench/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.2.1"
|
swearbench/__main__.py
ADDED
swearbench/card.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""Leaderboard card, styled to match chart.svg. Model names and numbers only; never quotes."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from html import escape
|
|
5
|
+
|
|
6
|
+
from .chart import FAMILIES, PAD, STYLE, _text_w, family, pretty
|
|
7
|
+
|
|
8
|
+
MODE_LABEL = {
|
|
9
|
+
"fabrication": "lies", "overreach": "overreach", "giving_up": "gave up", "regression": "breaks things",
|
|
10
|
+
"ignored": "ignores me", "repeat": "repeats", "incomplete": "half-done", "insult": "insults",
|
|
11
|
+
"profanity": "swearing", "shouting": "CAPS", "sarcasm": "sarcasm", "slow": "slow",
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
W = 760
|
|
15
|
+
TOP = 120
|
|
16
|
+
ROW_H = 40
|
|
17
|
+
RANK_END = PAD + 16
|
|
18
|
+
NAME_X = RANK_END + 14
|
|
19
|
+
BAR_X = NAME_X + 160
|
|
20
|
+
BAR_W = 180
|
|
21
|
+
SCORE_END = BAR_X + BAR_W + 40
|
|
22
|
+
STATS_X = SCORE_END + 24
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def svg(res, rows=8):
|
|
26
|
+
board = res["board"][:rows]
|
|
27
|
+
h = TOP + ROW_H * max(len(board), 1) + PAD + 20
|
|
28
|
+
hi = max([s["score"] for s in board] + [1])
|
|
29
|
+
o = [f'<svg xmlns="http://www.w3.org/2000/svg" class="c" width="{W}" height="{h}" viewBox="0 0 {W} {h}" '
|
|
30
|
+
'role="img" aria-label="SwearBench leaderboard">',
|
|
31
|
+
f"<style>{STYLE}</style>",
|
|
32
|
+
f'<rect width="{W}" height="{h}" rx="14" fill="var(--surface)"/>',
|
|
33
|
+
f'<text x="{PAD}" y="{PAD + 18}" font-size="20" font-weight="700" fill="var(--ink)">SwearBench</text>']
|
|
34
|
+
|
|
35
|
+
lx, ly = PAD, PAD + 52
|
|
36
|
+
for i, (name, _) in enumerate(FAMILIES):
|
|
37
|
+
o.append(f'<circle cx="{lx + 5}" cy="{ly - 4}" r="5" fill="var(--s{i + 1})"/>'
|
|
38
|
+
f'<text x="{lx + 15}" y="{ly}" font-size="12" fill="var(--ink2)">{name}</text>')
|
|
39
|
+
lx += 15 + _text_w(name) + 24
|
|
40
|
+
o.append(f'<text x="{W - PAD}" y="{ly}" font-size="12" fill="var(--muted)" text-anchor="end">'
|
|
41
|
+
f'{res["n_reactions"]} reactions</text>')
|
|
42
|
+
|
|
43
|
+
for i, s in enumerate(board):
|
|
44
|
+
y = TOP + i * ROW_H
|
|
45
|
+
mid = y + ROW_H / 2
|
|
46
|
+
base = mid + 4.5
|
|
47
|
+
col = f"var(--s{family(s['model']) + 1})"
|
|
48
|
+
top_mode = max(s["modes"], key=s["modes"].get) if s["modes"] else None
|
|
49
|
+
o += [f'<line x1="{PAD}" x2="{W - PAD}" y1="{y}" y2="{y}" stroke="var(--grid)" stroke-width="1"/>',
|
|
50
|
+
f'<text x="{RANK_END}" y="{base}" font-size="12" fill="var(--muted)" text-anchor="end">{i + 1}</text>',
|
|
51
|
+
f'<text x="{NAME_X}" y="{base}" font-size="13" font-weight="600" fill="var(--ink)">'
|
|
52
|
+
f'{escape(pretty(s["model"]))}</text>',
|
|
53
|
+
f'<rect x="{BAR_X}" y="{mid - 3}" width="{BAR_W}" height="6" rx="3" fill="var(--grid)"/>',
|
|
54
|
+
f'<rect x="{BAR_X}" y="{mid - 3}" width="{max(6, BAR_W * max(s["score"], 0) / hi):.0f}" height="6" '
|
|
55
|
+
f'rx="3" fill="{col}"/>',
|
|
56
|
+
f'<text x="{SCORE_END}" y="{base}" font-size="13" font-weight="700" fill="var(--ink)" '
|
|
57
|
+
f'text-anchor="end" font-variant-numeric="tabular-nums">{s["score"]:.0f}</text>',
|
|
58
|
+
f'<text x="{STATS_X}" y="{base}" font-size="12" fill="var(--ink2)">'
|
|
59
|
+
f'ships {s["outcome"]:.0f}% · {s["angry_pct"]:.0f}% angry</text>',
|
|
60
|
+
f'<text x="{W - PAD}" y="{base}" font-size="12" fill="var(--muted)" text-anchor="end">'
|
|
61
|
+
f'{MODE_LABEL.get(top_mode, "")}</text>']
|
|
62
|
+
end = TOP + ROW_H * len(board)
|
|
63
|
+
o.append(f'<line x1="{PAD}" x2="{W - PAD}" y1="{end}" y2="{end}" stroke="var(--axis)" stroke-width="1"/>')
|
|
64
|
+
o.append(f'<text x="{PAD}" y="{h - PAD}" font-size="11" fill="var(--muted)">github.com/AlenHay/swearbench</text>')
|
|
65
|
+
o.append("</svg>")
|
|
66
|
+
return "\n".join(o)
|
swearbench/chart.py
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
"""Swearing vs. result: one dot per model. x = rage per 100 turns, y = share of sessions that land."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import math
|
|
5
|
+
import re
|
|
6
|
+
from html import escape
|
|
7
|
+
|
|
8
|
+
FAMILIES = [("Claude", ("claude",)), ("GPT", ("gpt", "o1", "o3", "o4", "codex")), ("Other", ())]
|
|
9
|
+
|
|
10
|
+
STYLE = """
|
|
11
|
+
.c{--surface:#fcfcfb;--ink:#0b0b0b;--ink2:#52514e;--muted:#898781;--grid:#e1e0d9;--axis:#c3c2b7;
|
|
12
|
+
--s1:#2a78d6;--s2:#eb6834;--s3:#1baf7a}
|
|
13
|
+
@media (prefers-color-scheme:dark){.c{--surface:#1a1a19;--ink:#fff;--ink2:#c3c2b7;--grid:#2c2c2a;--axis:#383835;
|
|
14
|
+
--s1:#3987e5;--s2:#d95926;--s3:#199e70}}
|
|
15
|
+
text{font-family:ui-sans-serif,system-ui,-apple-system,"Segoe UI",sans-serif}
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def pretty(model):
|
|
20
|
+
"""claude-opus-5-5 -> Opus 5.5, gpt-5.6-sol -> GPT-5.6 Sol, gemini-3.8-flash-high -> Gemini 3.8 Flash."""
|
|
21
|
+
m = re.sub(r"-\d{8}$", "", model.split("/")[-1])
|
|
22
|
+
if m.startswith("claude-"):
|
|
23
|
+
parts = m[7:].split("-")
|
|
24
|
+
return f"{parts[0].title()} {'.'.join(parts[1:])}"
|
|
25
|
+
if m.startswith("gpt-"):
|
|
26
|
+
ver, *rest = m[4:].split("-")
|
|
27
|
+
return " ".join([f"GPT-{ver}"] + [r.title() for r in rest])
|
|
28
|
+
words = [w for w in m.split("-") if w not in ("high", "low", "medium", "free", "preview")]
|
|
29
|
+
return " ".join(w if any(c.isdigit() for c in w) else w.title() for w in words)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def family(model):
|
|
33
|
+
for i, (_, prefixes) in enumerate(FAMILIES[:-1]):
|
|
34
|
+
if model.startswith(prefixes):
|
|
35
|
+
return i
|
|
36
|
+
return len(FAMILIES) - 1
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _ticks(lo, hi, n=5):
|
|
40
|
+
step = 10 ** math.floor(math.log10((hi - lo) / n or 1))
|
|
41
|
+
for m in (1, 2, 2.5, 5, 10):
|
|
42
|
+
if (hi - lo) / (step * m) <= n:
|
|
43
|
+
step *= m
|
|
44
|
+
break
|
|
45
|
+
start = math.floor(lo / step) * step
|
|
46
|
+
return [start + i * step for i in range(int((hi - start) / step) + 2) if start + i * step <= hi + 1e-9]
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
PAD = 32
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _text_w(text, size=12):
|
|
53
|
+
return len(text) * size * 0.56
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def svg(res):
|
|
57
|
+
pts = res["board"]
|
|
58
|
+
W, H = 760, 540
|
|
59
|
+
L, R, T, B = PAD + 48, PAD, 120, 76
|
|
60
|
+
pw, ph = W - L - R, H - T - B
|
|
61
|
+
xs = [p["rage100"] for p in pts] or [0, 1]
|
|
62
|
+
x_lo, x_hi = math.floor(min(xs) * 0.9 / 10) * 10, math.ceil(max(xs) * 1.05 / 10) * 10
|
|
63
|
+
x_lo, x_hi = (x_lo, x_hi) if x_hi > x_lo else (0, 100)
|
|
64
|
+
X = lambda v: L + (v - x_lo) / (x_hi - x_lo) * pw # noqa: E731
|
|
65
|
+
Y = lambda v: T + ph - v / 100 * ph # noqa: E731
|
|
66
|
+
max_n = max([p["sessions"] for p in pts] + [1])
|
|
67
|
+
|
|
68
|
+
o = [f'<svg xmlns="http://www.w3.org/2000/svg" class="c" width="{W}" height="{H}" viewBox="0 0 {W} {H}" '
|
|
69
|
+
'role="img" aria-labelledby="t d">',
|
|
70
|
+
f"<style>{STYLE}</style>",
|
|
71
|
+
'<title id="t">SwearBench: how much swearing gets you what result</title>',
|
|
72
|
+
'<desc id="d">' + escape("; ".join(f'{p["model"]}: rage {p["rage100"]:.0f} per 100 turns, '
|
|
73
|
+
f'ships {p["outcome"]:.0f}%' for p in pts)) + "</desc>",
|
|
74
|
+
f'<rect width="{W}" height="{H}" rx="14" fill="var(--surface)"/>',
|
|
75
|
+
f'<text x="{PAD}" y="{PAD + 18}" font-size="20" font-weight="700" fill="var(--ink)">'
|
|
76
|
+
'Swearing vs. result</text>']
|
|
77
|
+
|
|
78
|
+
lx, ly = PAD, PAD + 52
|
|
79
|
+
for i, (name, _) in enumerate(FAMILIES):
|
|
80
|
+
o.append(f'<circle cx="{lx + 5}" cy="{ly - 4}" r="5" fill="var(--s{i + 1})"/>'
|
|
81
|
+
f'<text x="{lx + 15}" y="{ly}" font-size="12" fill="var(--ink2)">{name}</text>')
|
|
82
|
+
lx += 15 + _text_w(name) + 24
|
|
83
|
+
|
|
84
|
+
for v in _ticks(x_lo, x_hi):
|
|
85
|
+
o.append(f'<line x1="{X(v):.1f}" x2="{X(v):.1f}" y1="{T}" y2="{T + ph}" stroke="var(--grid)" stroke-width="1"/>')
|
|
86
|
+
o.append(f'<text x="{X(v):.1f}" y="{T + ph + 22}" font-size="11" fill="var(--muted)" text-anchor="middle">{v:g}</text>')
|
|
87
|
+
for v in range(0, 101, 20):
|
|
88
|
+
o.append(f'<line x1="{L}" x2="{L + pw}" y1="{Y(v):.1f}" y2="{Y(v):.1f}" stroke="var(--grid)" stroke-width="1"/>')
|
|
89
|
+
o.append(f'<text x="{L - 12}" y="{Y(v) + 4:.1f}" font-size="11" fill="var(--muted)" text-anchor="end">{v}%</text>')
|
|
90
|
+
o.append(f'<line x1="{L}" x2="{L + pw}" y1="{T + ph}" y2="{T + ph}" stroke="var(--axis)" stroke-width="1"/>')
|
|
91
|
+
o.append(f'<text x="{L + pw / 2}" y="{H - PAD}" font-size="12" fill="var(--ink2)" text-anchor="middle">'
|
|
92
|
+
'Swearing →</text>')
|
|
93
|
+
o.append(f'<text transform="translate({PAD + 4} {T + ph / 2}) rotate(-90)" font-size="12" fill="var(--ink2)" '
|
|
94
|
+
'text-anchor="middle">Shipped →</text>')
|
|
95
|
+
|
|
96
|
+
dots = []
|
|
97
|
+
for p in sorted(pts, key=lambda p: -p["sessions"]):
|
|
98
|
+
cx, cy = X(p["rage100"]), Y(p["outcome"])
|
|
99
|
+
r = 5 + 9 * math.sqrt(p["sessions"] / max_n)
|
|
100
|
+
col = f"var(--s{family(p['model']) + 1})"
|
|
101
|
+
lo, hi = p.get("outcome_ci", (p["outcome"], p["outcome"]))
|
|
102
|
+
tip = escape(f'{p["model"]}\nrage {p["rage100"]:.0f}/100 turns · {p["angry_pct"]:.0f}% angry msgs\n'
|
|
103
|
+
f'ships {p["outcome"]:.0f}% of {p["decided"]} decided sessions (90% CI {lo:.0f}–{hi:.0f}%)\n'
|
|
104
|
+
f'SwearBench {p["score"]:.1f}')
|
|
105
|
+
o.append(f'<g><title>{tip}</title>'
|
|
106
|
+
f'<line x1="{cx:.1f}" x2="{cx:.1f}" y1="{Y(hi):.1f}" y2="{Y(lo):.1f}" stroke="{col}" stroke-width="2" '
|
|
107
|
+
'stroke-opacity="0.45" stroke-linecap="round"/>'
|
|
108
|
+
f'<circle cx="{cx:.1f}" cy="{cy:.1f}" r="{r + 8:.1f}" fill="transparent"/>'
|
|
109
|
+
f'<circle cx="{cx:.1f}" cy="{cy:.1f}" r="{r:.1f}" fill="{col}" stroke="var(--surface)" stroke-width="2"/></g>')
|
|
110
|
+
dots.append((p, cx, cy, r))
|
|
111
|
+
|
|
112
|
+
boxes = [(cx - r, cy - r, cx + r, cy + r) for _, cx, cy, r in dots]
|
|
113
|
+
for p, cx, cy, r in dots:
|
|
114
|
+
label = pretty(p["model"])
|
|
115
|
+
w = _text_w(label)
|
|
116
|
+
best = None
|
|
117
|
+
for dy in (0, -18, 18, -32, 32, -46, 46):
|
|
118
|
+
for side in (1, -1):
|
|
119
|
+
x0 = cx + r + 8 if side > 0 else cx - r - 8 - w
|
|
120
|
+
y = cy + 4 + dy
|
|
121
|
+
box = (x0, y - 11, x0 + w, y + 3)
|
|
122
|
+
if not (L + 2 <= box[0] and box[2] <= L + pw - 2 and T + 28 <= box[1] and box[3] <= T + ph - 20):
|
|
123
|
+
continue
|
|
124
|
+
overlap = sum(max(0, min(box[2], b[2]) - max(box[0], b[0])) * max(0, min(box[3], b[3]) - max(box[1], b[1]))
|
|
125
|
+
for b in boxes)
|
|
126
|
+
if best is None or overlap < best[0]:
|
|
127
|
+
best = (overlap, x0, y, box)
|
|
128
|
+
if best and best[0] == 0:
|
|
129
|
+
break
|
|
130
|
+
_, x0, y, box = best or (0, cx + r + 6, cy + 4, (cx + r + 6, cy - 7, cx + r + 6 + w, cy + 7))
|
|
131
|
+
boxes.append(box)
|
|
132
|
+
o.append(f'<text x="{x0:.1f}" y="{y:.1f}" font-size="12" fill="var(--ink)">{escape(label)}</text>')
|
|
133
|
+
o.append("</svg>")
|
|
134
|
+
return "\n".join(o)
|
swearbench/cli.py
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""swearbench: rank AI coding models by how much you got mad at them."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import sys
|
|
8
|
+
|
|
9
|
+
from . import card, chart, judge, score, sources
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def main(argv=None):
|
|
13
|
+
ap = argparse.ArgumentParser(prog="swearbench", description=__doc__)
|
|
14
|
+
ap.add_argument("--judge", choices=["claude-cli", "codex-cli", "anthropic", "openai", "command"],
|
|
15
|
+
help="who labels your messages (default: first available of claude, codex, ANTHROPIC_API_KEY, OPENAI_API_KEY)")
|
|
16
|
+
ap.add_argument("--judge-model", help="model id for the judge backend")
|
|
17
|
+
ap.add_argument("--judge-cmd", help="with --judge command: shell command that reads a prompt on stdin and prints the reply")
|
|
18
|
+
ap.add_argument("--since", help="only reactions on/after this date (YYYY-MM-DD)")
|
|
19
|
+
ap.add_argument("--min-reactions", type=int, default=40, help="reactions a model needs to be ranked (default 40)")
|
|
20
|
+
ap.add_argument("--exclude", action="append", default=[], help="model id to leave out (repeatable)")
|
|
21
|
+
ap.add_argument("--no-t3", action="store_true", help="ignore T3 Code's database even if present")
|
|
22
|
+
ap.add_argument("--no-quotes", action="store_true", help="leave the hall of shame out of report.md")
|
|
23
|
+
ap.add_argument("--out", default="swearbench-out", help="output directory (default ./swearbench-out)")
|
|
24
|
+
ap.add_argument("--workers", type=int, default=8)
|
|
25
|
+
ap.add_argument("--dry-run", action="store_true", help="show what was found and what would be judged, then stop")
|
|
26
|
+
ap.add_argument("-y", "--yes", action="store_true", help="don't ask before sending messages to the judge")
|
|
27
|
+
a = ap.parse_args(argv)
|
|
28
|
+
|
|
29
|
+
corpus = sources.collect(use_t3=not a.no_t3, exclude_dirs=["swearbench"])
|
|
30
|
+
if a.since:
|
|
31
|
+
corpus.reactions = [r for r in corpus.reactions if r.ts >= a.since]
|
|
32
|
+
corpus.interrupts = [i for i in corpus.interrupts if i.ts >= a.since]
|
|
33
|
+
print("Found: " + ("; ".join(corpus.found) or "no logs"), file=sys.stderr)
|
|
34
|
+
if not corpus.reactions:
|
|
35
|
+
print("No human messages found. Supported: Claude Code, Codex, OpenCode, T3 Code.", file=sys.stderr)
|
|
36
|
+
return 1
|
|
37
|
+
|
|
38
|
+
backend = a.judge or judge.detect_backend()
|
|
39
|
+
if not backend:
|
|
40
|
+
print("No judge available: install claude or codex CLI, or set ANTHROPIC_API_KEY / OPENAI_API_KEY, "
|
|
41
|
+
"or use --judge command --judge-cmd '...'.", file=sys.stderr)
|
|
42
|
+
return 1
|
|
43
|
+
model = a.judge_model or judge.DEFAULT_MODELS.get(backend)
|
|
44
|
+
cached = judge.load_cache()
|
|
45
|
+
todo = len({judge.key(r) for r in corpus.reactions} - set(cached))
|
|
46
|
+
print(f"{len(corpus.reactions)} messages from you; {todo} not judged yet. Judge: {backend}"
|
|
47
|
+
f"{' / ' + model if model else ''}.", file=sys.stderr)
|
|
48
|
+
if a.dry_run:
|
|
49
|
+
return 0
|
|
50
|
+
if todo and not a.yes:
|
|
51
|
+
if input(f"Send {todo} of your messages (with the AI reply each one answers) to the judge? [y/N] ").lower() != "y":
|
|
52
|
+
return 1
|
|
53
|
+
|
|
54
|
+
labels = judge.label(corpus.reactions, backend, model, a.judge_cmd, workers=a.workers)
|
|
55
|
+
res = score.build(corpus, labels, a.min_reactions, set(a.exclude))
|
|
56
|
+
family = (model or backend).split("-")[0]
|
|
57
|
+
if any(s["model"].startswith(family) for s in res["board"]):
|
|
58
|
+
print(f"Note: the judge ({model or backend}) is from a family being ranked. "
|
|
59
|
+
"Try a different --judge to check it isn't playing favourites.", file=sys.stderr)
|
|
60
|
+
|
|
61
|
+
os.makedirs(a.out, exist_ok=True)
|
|
62
|
+
report = score.markdown(res, quotes=not a.no_quotes)
|
|
63
|
+
with open(os.path.join(a.out, "report.md"), "w") as f:
|
|
64
|
+
f.write(report)
|
|
65
|
+
with open(os.path.join(a.out, "card.svg"), "w") as f:
|
|
66
|
+
f.write(card.svg(res))
|
|
67
|
+
with open(os.path.join(a.out, "chart.svg"), "w") as f:
|
|
68
|
+
f.write(chart.svg(res))
|
|
69
|
+
with open(os.path.join(a.out, "results.json"), "w") as f:
|
|
70
|
+
json.dump({k: v for k, v in res.items()}, f, indent=1, default=str)
|
|
71
|
+
print(report.split("\n## ")[0])
|
|
72
|
+
print(f"Wrote {a.out}/report.md, card.svg, chart.svg, results.json", file=sys.stderr)
|
|
73
|
+
return 0
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
if __name__ == "__main__":
|
|
77
|
+
sys.exit(main())
|
swearbench/judge.py
ADDED
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""LLM judge: labels each reaction with anger, anger modes and satisfaction. Results are cached on disk."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import hashlib
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import shlex
|
|
8
|
+
import shutil
|
|
9
|
+
import subprocess
|
|
10
|
+
import sys
|
|
11
|
+
import tempfile
|
|
12
|
+
import urllib.request
|
|
13
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
14
|
+
|
|
15
|
+
RUBRIC_VERSION = "2"
|
|
16
|
+
CACHE = os.path.join(os.environ.get("XDG_CACHE_HOME", os.path.expanduser("~/.cache")), "swearbench", "labels.jsonl")
|
|
17
|
+
|
|
18
|
+
RUBRIC = """You are labelling a developer's chat messages to AI coding agents, to measure how frustrated
|
|
19
|
+
the developer was with the AI. Each item has `prev` (tail of the AI's last reply) and `msg` (the
|
|
20
|
+
developer's next message). Judge `msg` as a reaction to the AI's work.
|
|
21
|
+
|
|
22
|
+
Return ONLY a JSON array, one object per item, same order:
|
|
23
|
+
{"id": str,
|
|
24
|
+
"agent_authored": bool, // msg is clearly written by another AI/agent (long structured brief, "You are...", tool-dispatch prose), not a human
|
|
25
|
+
"anger": 0-4, // 0 calm/neutral, 1 mild annoyance, 2 clearly irritated, 3 angry, 4 furious
|
|
26
|
+
"target": "model"|"external"|"self"|"none", // who the frustration is aimed at; tools/vendors/infra/third parties = external
|
|
27
|
+
"modes": [..], // zero or more, only when target=="model":
|
|
28
|
+
// "profanity" swearing directed at the work or the model
|
|
29
|
+
// "insult" calling the model stupid/useless/lazy/idiot etc.
|
|
30
|
+
// "shouting" caps, !!!, ??? for emphasis
|
|
31
|
+
// "repeat" had to repeat an instruction / "I already told you" / "again"
|
|
32
|
+
// "ignored" model ignored or violated an explicit instruction or rule
|
|
33
|
+
// "fabrication" model claimed success/facts that were false, hallucinated, lied
|
|
34
|
+
// "incomplete" lazy, half-done, stopped early, left TODOs, didn't verify
|
|
35
|
+
// "regression" model broke something that worked
|
|
36
|
+
// "overreach" did unrequested/destructive things, scope creep, touched what it shouldn't
|
|
37
|
+
// "slow" too slow, too many questions, too verbose, wasted time
|
|
38
|
+
// "sarcasm" sarcastic/passive-aggressive phrasing
|
|
39
|
+
// "giving_up" abandons the model/approach, "I'll do it myself", "forget it", threatens to switch
|
|
40
|
+
// "taste" subjective critique while iterating on look/feel/wording ("looks lame", "too much text");
|
|
41
|
+
// the work does what was asked, the developer just doesn't like it yet
|
|
42
|
+
"blames_earlier": bool, // msg complains that something done BEFORE the AI's last reply (earlier merged/shipped work,
|
|
43
|
+
// a previous session) is broken, missing or wrong, rather than the last reply itself
|
|
44
|
+
"satisfaction": -2..2, // -2 rejects the work, -1 wants fixes, 0 neutral/new task, 1 accepts, 2 explicit praise/delight
|
|
45
|
+
"quote": str // <=80 char excerpt that best shows the anger (or "" if anger==0)
|
|
46
|
+
}
|
|
47
|
+
Profanity used positively ("fucking cool") is not anger. Terse instructions are not anger. A new
|
|
48
|
+
unrelated task after a reply implies mild acceptance (satisfaction 0 or 1), not anger.
|
|
49
|
+
|
|
50
|
+
ITEMS:
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
DEFAULT_MODELS = {"claude-cli": "claude-sonnet-5-5", "codex-cli": None, "anthropic": "claude-sonnet-5-5", "openai": "gpt-5.5"}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def key(r):
|
|
57
|
+
return hashlib.sha1(f"{RUBRIC_VERSION}\0{r.prev_reply[-500:]}\0{r.text[:1500]}".encode()).hexdigest()
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def load_cache():
|
|
61
|
+
out = {}
|
|
62
|
+
if os.path.exists(CACHE):
|
|
63
|
+
with open(CACHE) as f:
|
|
64
|
+
for line in f:
|
|
65
|
+
try:
|
|
66
|
+
d = json.loads(line)
|
|
67
|
+
out[d["key"]] = d
|
|
68
|
+
except (ValueError, KeyError):
|
|
69
|
+
pass
|
|
70
|
+
return out
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def detect_backend():
|
|
74
|
+
if shutil.which("claude"):
|
|
75
|
+
return "claude-cli"
|
|
76
|
+
if shutil.which("codex"):
|
|
77
|
+
return "codex-cli"
|
|
78
|
+
if os.environ.get("ANTHROPIC_API_KEY"):
|
|
79
|
+
return "anthropic"
|
|
80
|
+
if os.environ.get("OPENAI_API_KEY"):
|
|
81
|
+
return "openai"
|
|
82
|
+
return None
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _post(url, headers, body):
|
|
86
|
+
req = urllib.request.Request(url, json.dumps(body).encode(), {"content-type": "application/json", **headers})
|
|
87
|
+
with urllib.request.urlopen(req, timeout=600) as r:
|
|
88
|
+
return json.loads(r.read())
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def call(backend, model, cmd, prompt):
|
|
92
|
+
if backend == "command":
|
|
93
|
+
return subprocess.run(shlex.split(cmd), input=prompt, capture_output=True, text=True, timeout=900).stdout
|
|
94
|
+
if backend == "claude-cli":
|
|
95
|
+
args = ["claude", "-p", "--tools", ""] + (["--model", model] if model else [])
|
|
96
|
+
return subprocess.run(args, input=prompt, capture_output=True, text=True, timeout=900).stdout
|
|
97
|
+
if backend == "codex-cli":
|
|
98
|
+
with tempfile.NamedTemporaryFile("r", suffix=".txt") as out:
|
|
99
|
+
args = ["codex", "exec", "--skip-git-repo-check", "--sandbox", "read-only", "-o", out.name]
|
|
100
|
+
subprocess.run(args + (["-m", model] if model else []) + ["-"], input=prompt,
|
|
101
|
+
capture_output=True, text=True, timeout=900)
|
|
102
|
+
return out.read()
|
|
103
|
+
if backend == "anthropic":
|
|
104
|
+
r = _post("https://api.anthropic.com/v1/messages",
|
|
105
|
+
{"x-api-key": os.environ["ANTHROPIC_API_KEY"], "anthropic-version": "2023-06-01"},
|
|
106
|
+
{"model": model, "max_tokens": 16000, "messages": [{"role": "user", "content": prompt}]})
|
|
107
|
+
return "".join(b.get("text", "") for b in r["content"])
|
|
108
|
+
if backend == "openai":
|
|
109
|
+
r = _post("https://api.openai.com/v1/chat/completions",
|
|
110
|
+
{"authorization": "Bearer " + os.environ["OPENAI_API_KEY"]},
|
|
111
|
+
{"model": model, "messages": [{"role": "user", "content": prompt}]})
|
|
112
|
+
return r["choices"][0]["message"]["content"]
|
|
113
|
+
raise ValueError(f"unknown judge backend {backend}")
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _label_batch(batch, backend, model, cmd):
|
|
117
|
+
items = [{"id": str(i), "prev": r.prev_reply[-500:], "msg": r.text[:1500]} for i, r in enumerate(batch)]
|
|
118
|
+
prompt = RUBRIC + json.dumps(items, ensure_ascii=False)
|
|
119
|
+
err = "unparseable output"
|
|
120
|
+
for _ in range(3):
|
|
121
|
+
try:
|
|
122
|
+
txt = call(backend, model, cmd, prompt) or ""
|
|
123
|
+
arr = json.loads(txt[txt.index("["):txt.rindex("]") + 1])
|
|
124
|
+
got = {str(a.get("id")): a for a in arr if isinstance(a, dict)}
|
|
125
|
+
if len(got) >= len(batch) * 0.9:
|
|
126
|
+
return [(key(r), got[str(i)]) for i, r in enumerate(batch) if str(i) in got]
|
|
127
|
+
except Exception as e: # noqa: BLE001
|
|
128
|
+
err = e
|
|
129
|
+
print(f" judge batch failed after 3 tries: {err}", file=sys.stderr)
|
|
130
|
+
return []
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def label(reactions, backend, model, cmd=None, batch=30, workers=8):
|
|
134
|
+
cache = load_cache()
|
|
135
|
+
todo = [r for r in {key(r): r for r in reactions}.values() if key(r) not in cache]
|
|
136
|
+
if todo:
|
|
137
|
+
os.makedirs(os.path.dirname(CACHE), exist_ok=True)
|
|
138
|
+
batches = [todo[i:i + batch] for i in range(0, len(todo), batch)]
|
|
139
|
+
print(f"Judging {len(todo)} new messages in {len(batches)} batches with {backend}"
|
|
140
|
+
f"{' / ' + model if model else ''} ...", file=sys.stderr)
|
|
141
|
+
with open(CACHE, "a") as f, ThreadPoolExecutor(workers) as ex:
|
|
142
|
+
for n, res in enumerate(ex.map(lambda b: _label_batch(b, backend, model, cmd), batches), 1):
|
|
143
|
+
for k, lab in res:
|
|
144
|
+
lab = {**lab, "key": k, "judge": f"{backend}:{model or 'default'}"}
|
|
145
|
+
cache[k] = lab
|
|
146
|
+
f.write(json.dumps(lab, ensure_ascii=False) + "\n")
|
|
147
|
+
f.flush()
|
|
148
|
+
print(f" {n}/{len(batches)}", file=sys.stderr)
|
|
149
|
+
return {r.id: cache[key(r)] for r in reactions if key(r) in cache}
|
swearbench/score.py
ADDED
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
"""Turn labels into leaderboards."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import random
|
|
5
|
+
import re
|
|
6
|
+
from collections import Counter, defaultdict
|
|
7
|
+
|
|
8
|
+
MODE_WEIGHT = {
|
|
9
|
+
"fabrication": 3, "overreach": 3, "giving_up": 3, "regression": 2.5,
|
|
10
|
+
"ignored": 2, "repeat": 2, "incomplete": 1.5, "insult": 1.5,
|
|
11
|
+
"profanity": 1, "shouting": 0.5, "sarcasm": 0.5, "slow": 0.5, "taste": 0.5,
|
|
12
|
+
}
|
|
13
|
+
BREACH = {"fabrication", "overreach", "giving_up", "regression", "ignored", "repeat", "incomplete"}
|
|
14
|
+
TASTE_DISCOUNT = 0.25
|
|
15
|
+
FRICTION_K = 0.25
|
|
16
|
+
REGRET_WINDOW_DAYS = 7
|
|
17
|
+
INTERRUPT_RAGE = 1.5
|
|
18
|
+
SWEAR = re.compile(r"\b(fuck\w*|shit\w*|wtf|damn\w*|crap\w*|bullshit|stupid|idiot\w*|dumb\w*|moron\w*|"
|
|
19
|
+
r"bitch\w*|asshole|sucks?)\b", re.I)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def rage(lab):
|
|
23
|
+
"""Anger squared, so one blowup outweighs several grumbles, plus how you got mad.
|
|
24
|
+
Critique of taste while iterating, with no breach of trust, counts a quarter."""
|
|
25
|
+
if lab.get("agent_authored") or lab.get("target") != "model" or lab.get("anger", 0) < 1:
|
|
26
|
+
return 0.0
|
|
27
|
+
modes = set(lab.get("modes", []))
|
|
28
|
+
r = lab["anger"] ** 2 + sum(MODE_WEIGHT.get(m, 0) for m in modes)
|
|
29
|
+
return r * TASTE_DISCOUNT if "taste" in modes and not modes & BREACH else r
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def blame_shares(reactions, labels):
|
|
33
|
+
"""For a complaint about earlier work, split it across the models that worked in the same workspace
|
|
34
|
+
in the week before, outside the current session, by how many turns each had there."""
|
|
35
|
+
import datetime as dt
|
|
36
|
+
|
|
37
|
+
def t(r):
|
|
38
|
+
return dt.datetime.fromisoformat(r.ts.replace("Z", "+00:00"))
|
|
39
|
+
|
|
40
|
+
by_ws = defaultdict(list)
|
|
41
|
+
for r in reactions:
|
|
42
|
+
if r.workspace and r.ts:
|
|
43
|
+
by_ws[r.workspace].append(r)
|
|
44
|
+
for rs in by_ws.values():
|
|
45
|
+
rs.sort(key=lambda r: r.ts)
|
|
46
|
+
shares = {}
|
|
47
|
+
window = dt.timedelta(days=REGRET_WINDOW_DAYS)
|
|
48
|
+
for r in reactions:
|
|
49
|
+
lab = labels.get(r.id)
|
|
50
|
+
if not lab or not lab.get("blames_earlier") or rage(lab) == 0 or not r.workspace:
|
|
51
|
+
continue
|
|
52
|
+
now = t(r)
|
|
53
|
+
prior = Counter(o.model for o in by_ws[r.workspace]
|
|
54
|
+
if o.session != r.session and o.ts < r.ts and now - t(o) <= window and o.model)
|
|
55
|
+
total = sum(prior.values())
|
|
56
|
+
if total:
|
|
57
|
+
shares[r.id] = {m: n / total for m, n in prior.items()}
|
|
58
|
+
return shares
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _own(r, l, shares):
|
|
62
|
+
return rage(l) * shares[r.id].get(r.model, 0.0) if r.id in shares else rage(l)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _friction(rows, n_int, shares, inbound_per_turn):
|
|
66
|
+
n = len(rows)
|
|
67
|
+
total = sum(_own(r, l, shares) for r, l in rows) + inbound_per_turn * n + INTERRUPT_RAGE * n_int
|
|
68
|
+
clean = sum(1 for _, l in rows if rage(l) == 0 and l.get("satisfaction", 0) >= 0)
|
|
69
|
+
sat = sum(l.get("satisfaction", 0) for _, l in rows) / n
|
|
70
|
+
rage100 = 100 * total / n
|
|
71
|
+
clean_pct = 100 * clean / n
|
|
72
|
+
return {
|
|
73
|
+
"n": n, "friction": 100 - FRICTION_K * rage100 + 10 * sat, "clean_pct": clean_pct, "rage100": rage100,
|
|
74
|
+
"angry_pct": 100 * sum(rage(l) > 0 for _, l in rows) / n, "sat": sat,
|
|
75
|
+
"int100": 100 * n_int / n,
|
|
76
|
+
"swears100": 100 * sum(len(SWEAR.findall(r.text)) for r, _ in rows) / n,
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _verdict(rows):
|
|
81
|
+
"""True = you accepted the work, False = you rejected it, None = it stopped or moved elsewhere without a verdict."""
|
|
82
|
+
last = rows[-1][1]
|
|
83
|
+
if rage(last) > 0 or last.get("satisfaction", 0) < 0:
|
|
84
|
+
return False
|
|
85
|
+
if last.get("satisfaction", 0) >= 1:
|
|
86
|
+
return True
|
|
87
|
+
return None
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _stats(sessions, n_int, merged, shares, inbound_per_turn):
|
|
91
|
+
rows = [x for s in sessions for x in s]
|
|
92
|
+
st = _friction(rows, n_int, shares, inbound_per_turn)
|
|
93
|
+
st["sessions"] = len(sessions)
|
|
94
|
+
verdicts = [v for v in map(_verdict, sessions) if v is not None]
|
|
95
|
+
st["decided"] = len(verdicts)
|
|
96
|
+
st["outcome"] = 100 * sum(verdicts) / len(verdicts) if verdicts else 0.0
|
|
97
|
+
tracked = [s for s in sessions if merged["since"] and s[0][0].ts >= merged["since"]]
|
|
98
|
+
st["merged_n"] = len(tracked)
|
|
99
|
+
st["merged_pct"] = 100 * sum(s[0][0].session in merged["ids"] for s in tracked) / len(tracked) if tracked else None
|
|
100
|
+
st["score"] = (st["friction"] + st["outcome"]) / 2
|
|
101
|
+
return st
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _ci(sessions, n_int, merged, shares, inbound_per_turn, field, k=300):
|
|
105
|
+
rng = random.Random(0)
|
|
106
|
+
xs = sorted(_stats(rng.choices(sessions, k=len(sessions)), n_int, merged, shares, inbound_per_turn)[field]
|
|
107
|
+
for _ in range(k))
|
|
108
|
+
return xs[int(k * .05)], xs[int(k * .95)]
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def modes_share(rows):
|
|
112
|
+
c = Counter()
|
|
113
|
+
for _, l in rows:
|
|
114
|
+
if rage(l) > 0:
|
|
115
|
+
for m in l.get("modes", []):
|
|
116
|
+
c[m] += MODE_WEIGHT.get(m, 0)
|
|
117
|
+
total = sum(c.values()) or 1
|
|
118
|
+
return {m: c[m] / total for m in MODE_WEIGHT if c[m]}
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def build(corpus, labels, min_reactions=40, exclude=()):
|
|
122
|
+
per = defaultdict(list)
|
|
123
|
+
for r in corpus.reactions:
|
|
124
|
+
lab = labels.get(r.id)
|
|
125
|
+
if lab and not lab.get("agent_authored") and r.model and r.model not in exclude:
|
|
126
|
+
per[r.model].append((r, lab))
|
|
127
|
+
ints = Counter(i.model for i in corpus.interrupts)
|
|
128
|
+
shares = blame_shares(corpus.reactions, labels)
|
|
129
|
+
inbound = Counter()
|
|
130
|
+
for r in corpus.reactions:
|
|
131
|
+
if r.id in shares:
|
|
132
|
+
for m, share in shares[r.id].items():
|
|
133
|
+
if m != r.model:
|
|
134
|
+
inbound[m] += rage(labels[r.id]) * share
|
|
135
|
+
|
|
136
|
+
board = []
|
|
137
|
+
for model, rows in per.items():
|
|
138
|
+
if len(rows) < min_reactions:
|
|
139
|
+
continue
|
|
140
|
+
by_session = defaultdict(list)
|
|
141
|
+
for r, l in sorted(rows, key=lambda x: x[0].ts):
|
|
142
|
+
by_session[r.session].append((r, l))
|
|
143
|
+
sessions = list(by_session.values())
|
|
144
|
+
merged = {"ids": corpus.merged, "since": corpus.pr_tracking_since}
|
|
145
|
+
inb = inbound[model] / len(rows)
|
|
146
|
+
s = _stats(sessions, ints[model], merged, shares, inb)
|
|
147
|
+
s["model"], s["modes"] = model, modes_share(rows)
|
|
148
|
+
s["regret_in"], s["regret_out"] = inbound[model], sum(
|
|
149
|
+
rage(l) * (1 - shares[r.id].get(model, 0.0)) for r, l in rows if r.id in shares)
|
|
150
|
+
s["blowups"] = sum(1 for _, l in rows if rage(l) > 0 and l.get("anger", 0) >= 4)
|
|
151
|
+
s["ci"] = _ci(sessions, ints[model], merged, shares, inb, "score")
|
|
152
|
+
s["outcome_ci"] = _ci(sessions, ints[model], merged, shares, inb, "outcome")
|
|
153
|
+
s["worst"] = [l for _, l in sorted(rows, key=lambda x: -rage(x[1]))[:3] if rage(l) > 0]
|
|
154
|
+
board.append(s)
|
|
155
|
+
board.sort(key=lambda s: -s["score"])
|
|
156
|
+
|
|
157
|
+
tokens = []
|
|
158
|
+
for model, rows in per.items():
|
|
159
|
+
you = corpus.usage.tokens.get(model, {}).get("you")
|
|
160
|
+
start = corpus.usage.first_ts.get(model)
|
|
161
|
+
if not you or not you[1] or not start:
|
|
162
|
+
continue
|
|
163
|
+
win = [(r, l) for r, l in rows if r.ts >= start]
|
|
164
|
+
if len(win) < min_reactions:
|
|
165
|
+
continue
|
|
166
|
+
n_int = sum(1 for i in corpus.interrupts if i.model == model and i.ts >= start)
|
|
167
|
+
out_m = you[1] / 1e6
|
|
168
|
+
tokens.append({
|
|
169
|
+
"model": model, "out_m": out_m, "agents_out_m": corpus.usage.tokens[model].get("agents", [0, 0, 0])[1] / 1e6,
|
|
170
|
+
"total_b": sum(you) / 1e9, "n": len(win),
|
|
171
|
+
"rage_per_m": (sum(_own(r, l, shares) for r, l in win) + INTERRUPT_RAGE * n_int) / out_m,
|
|
172
|
+
"angry_per_m": sum(rage(l) > 0 for _, l in win) / out_m, "out_per_reaction": you[1] / len(win),
|
|
173
|
+
})
|
|
174
|
+
tokens.sort(key=lambda t: t["rage_per_m"])
|
|
175
|
+
|
|
176
|
+
sent = set(per)
|
|
177
|
+
agent_only = sorted(m for m, o in corpus.usage.tokens.items() if m not in sent)
|
|
178
|
+
small = sorted((m, len(r)) for m, r in per.items() if len(r) < min_reactions)
|
|
179
|
+
return {"board": board, "tokens": tokens, "agent_only": agent_only, "small": small,
|
|
180
|
+
"n_reactions": sum(len(r) for r in per.values()), "n_interrupts": len(corpus.interrupts)}
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _merged_cell(s):
|
|
184
|
+
return "—" if s["merged_pct"] is None else f"{s['merged_pct']:.0f}% of {s['merged_n']}"
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def markdown(res, quotes=True):
|
|
188
|
+
o = ["# SwearBench\n",
|
|
189
|
+
f"{res['n_reactions']} of your reactions to AI replies, {res['n_interrupts']} interrupts.\n",
|
|
190
|
+
"| # | Model | SwearBench ↑ | 90% CI | Ships | Friction | Decided / sessions | Merged PR* | Reactions | Clean turns | "
|
|
191
|
+
"Rage /100 turns | Angry msgs | 4/4 blowups | Regret charged in / passed on | Interrupts /100 | Swears /100 | "
|
|
192
|
+
"Avg satisfaction |",
|
|
193
|
+
"|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|"]
|
|
194
|
+
for i, s in enumerate(res["board"], 1):
|
|
195
|
+
o.append(f"| {i} | {s['model']} | **{s['score']:.1f}** | {s['ci'][0]:.0f}–{s['ci'][1]:.0f} | {s['outcome']:.0f}% | {s['friction']:.1f} | "
|
|
196
|
+
f"{s['decided']} / {s['sessions']} | "
|
|
197
|
+
f"{_merged_cell(s)} | {s['n']} | "
|
|
198
|
+
f"{s['clean_pct']:.0f}% | {s['rage100']:.1f} | {s['angry_pct']:.1f}% | {s['blowups']} | "
|
|
199
|
+
f"{s['regret_in']:.0f} / {s['regret_out']:.0f} | {s['int100']:.1f} | "
|
|
200
|
+
f"{s['swears100']:.1f} | {s['sat']:+.2f} |")
|
|
201
|
+
if res["tokens"]:
|
|
202
|
+
o += ["\n## Per token of work\n",
|
|
203
|
+
"Output tokens the model produced in sessions you drove; reactions counted from the first logged token on.\n",
|
|
204
|
+
"| # | Model | Your output tok | Agent-run output tok | Your total tok | Reactions | Rage per 1M out tok ↓ | "
|
|
205
|
+
"Angry msgs per 1M | Out tok per reaction |",
|
|
206
|
+
"|---|---|---|---|---|---|---|---|---|"]
|
|
207
|
+
for i, t in enumerate(res["tokens"], 1):
|
|
208
|
+
o.append(f"| {i} | {t['model']} | {t['out_m']:.1f}M | {t['agents_out_m']:.1f}M | {t['total_b']:.2f}B | {t['n']} | "
|
|
209
|
+
f"**{t['rage_per_m']:.1f}** | {t['angry_per_m']:.1f} | {t['out_per_reaction'] / 1000:.1f}k |")
|
|
210
|
+
o += ["\n## How you get mad at each model (share of rage points)\n",
|
|
211
|
+
"| Model | " + " | ".join(MODE_WEIGHT) + " |", "|---|" + "---|" * len(MODE_WEIGHT)]
|
|
212
|
+
for s in res["board"]:
|
|
213
|
+
o.append(f"| {s['model']} | " + " | ".join(f"{100 * s['modes'][m]:.0f}%" if m in s["modes"] else "·"
|
|
214
|
+
for m in MODE_WEIGHT) + " |")
|
|
215
|
+
if quotes:
|
|
216
|
+
o.append("\n## Hall of shame\n\n> Contains your own words. Review before sharing.\n")
|
|
217
|
+
for s in res["board"]:
|
|
218
|
+
o.append(f"**{s['model']}**")
|
|
219
|
+
o += [f"- [{l['anger']}/4, {', '.join(l.get('modes', [])) or '—'}] “{l.get('quote', '')}”" for l in s["worst"]]
|
|
220
|
+
o.append("")
|
|
221
|
+
if res["small"]:
|
|
222
|
+
o.append("Too few reactions to rank: " + ", ".join(f"{m} ({n})" for m, n in res["small"]))
|
|
223
|
+
if res["agent_only"]:
|
|
224
|
+
o.append("\nOnly ever ran as subagents / headless runs (never ranked): " + ", ".join(res["agent_only"]))
|
|
225
|
+
o.append(f"\nSwearBench = (Friction + Ships) / 2. Friction = 100 − {FRICTION_K} × rage per 100 turns + 10 × "
|
|
226
|
+
"avg satisfaction; rage = anger² + mode weights, only when aimed at the model, a quarter for taste-only "
|
|
227
|
+
"critique; each interrupt adds 1.5. Complaints about earlier work are charged to the models that worked "
|
|
228
|
+
f"in that workspace in the {REGRET_WINDOW_DAYS} days before. "
|
|
229
|
+
"Ships = accepted ÷ (accepted + rejected) sessions; sessions that stop or are handed off without a verdict "
|
|
230
|
+
"are left out, so limits and model switches don't count as failures. "
|
|
231
|
+
"*Merged PR is informational (T3 Code only, sessions after it started tracking PRs) and not in the score. "
|
|
232
|
+
"Intervals bootstrap over sessions. See chart.svg for swearing vs. result.")
|
|
233
|
+
return "\n".join(o) + "\n"
|
swearbench/sources.py
ADDED
|
@@ -0,0 +1,317 @@
|
|
|
1
|
+
"""Readers for local agent logs. Each yields human reactions, interrupts and token usage."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import glob
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import sqlite3
|
|
8
|
+
from collections import defaultdict
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
|
|
11
|
+
HOME = os.path.expanduser("~")
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass
|
|
15
|
+
class Reaction:
|
|
16
|
+
id: str
|
|
17
|
+
source: str
|
|
18
|
+
session: str
|
|
19
|
+
ts: str
|
|
20
|
+
text: str
|
|
21
|
+
prev_reply: str
|
|
22
|
+
model: str
|
|
23
|
+
workspace: str = ""
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class Interrupt:
|
|
28
|
+
source: str
|
|
29
|
+
session: str
|
|
30
|
+
ts: str
|
|
31
|
+
model: str
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class Usage:
|
|
36
|
+
"""Per model and origin ("you" = sessions a human drove, "agents" = subagents/headless runs)."""
|
|
37
|
+
tokens: dict = field(default_factory=lambda: defaultdict(lambda: defaultdict(lambda: [0, 0, 0])))
|
|
38
|
+
first_ts: dict = field(default_factory=dict)
|
|
39
|
+
|
|
40
|
+
def add(self, model, origin, ts, inp, out, cache):
|
|
41
|
+
model = norm(model)
|
|
42
|
+
t = self.tokens[model][origin]
|
|
43
|
+
t[0] += inp
|
|
44
|
+
t[1] += out
|
|
45
|
+
t[2] += cache
|
|
46
|
+
if origin == "you" and ts and (model not in self.first_ts or ts < self.first_ts[model]):
|
|
47
|
+
self.first_ts[model] = ts
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass
|
|
51
|
+
class Corpus:
|
|
52
|
+
reactions: list = field(default_factory=list)
|
|
53
|
+
interrupts: list = field(default_factory=list)
|
|
54
|
+
usage: Usage = field(default_factory=Usage)
|
|
55
|
+
found: list = field(default_factory=list)
|
|
56
|
+
merged: set = field(default_factory=set)
|
|
57
|
+
pr_tracking_since: str | None = None
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def norm(model):
|
|
61
|
+
if not model:
|
|
62
|
+
return None
|
|
63
|
+
model = model.strip().lower()
|
|
64
|
+
if model.startswith("claude-"):
|
|
65
|
+
model = model.replace(".", "-")
|
|
66
|
+
return model
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def workspace_of(path):
|
|
70
|
+
"""Repo-level key for a working directory; worktrees of one repo share it."""
|
|
71
|
+
if not path:
|
|
72
|
+
return ""
|
|
73
|
+
parts = path.rstrip("/").split("/")
|
|
74
|
+
if "worktrees" in parts and parts.index("worktrees") + 1 < len(parts):
|
|
75
|
+
return parts[parts.index("worktrees") + 1]
|
|
76
|
+
return parts[-1]
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def ms_to_iso(ms):
|
|
80
|
+
import datetime
|
|
81
|
+
return datetime.datetime.fromtimestamp(ms / 1000, datetime.timezone.utc).isoformat().replace("+00:00", "Z")
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _jsonl(path):
|
|
85
|
+
with open(path, errors="replace") as f:
|
|
86
|
+
for line in f:
|
|
87
|
+
try:
|
|
88
|
+
yield json.loads(line)
|
|
89
|
+
except ValueError:
|
|
90
|
+
continue
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _ro(path):
|
|
94
|
+
return sqlite3.connect(f"file:{path}?mode=ro", uri=True)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
# --- Claude Code -----------------------------------------------------------------------------
|
|
98
|
+
|
|
99
|
+
HEADLESS_ENTRYPOINTS = {"sdk-cli", "sdk-py"}
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _claude_text(content):
|
|
103
|
+
if isinstance(content, str):
|
|
104
|
+
return content
|
|
105
|
+
if any(isinstance(p, dict) and p.get("type") == "tool_result" for p in content):
|
|
106
|
+
return None
|
|
107
|
+
parts = [p.get("text", "") for p in content if isinstance(p, dict) and p.get("type") == "text"]
|
|
108
|
+
return "\n".join(parts) if parts else None
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def read_claude(corpus, root, skip_session=lambda entrypoint: False, exclude_dirs=()):
|
|
112
|
+
files = glob.glob(os.path.join(root, "**", "*.jsonl"), recursive=True)
|
|
113
|
+
if not files:
|
|
114
|
+
return
|
|
115
|
+
corpus.found.append(f"Claude Code: {len(files)} transcripts")
|
|
116
|
+
for path in files:
|
|
117
|
+
excluded = any(d in path for d in exclude_dirs)
|
|
118
|
+
subagent = "/subagents/" in path
|
|
119
|
+
entrypoint = None
|
|
120
|
+
model = None
|
|
121
|
+
reply = ""
|
|
122
|
+
cwd = ""
|
|
123
|
+
usage = {}
|
|
124
|
+
pending = []
|
|
125
|
+
for d in _jsonl(path):
|
|
126
|
+
kind = d.get("type")
|
|
127
|
+
msg = d.get("message") or {}
|
|
128
|
+
if kind == "assistant":
|
|
129
|
+
if msg.get("model") and msg["model"] != "<synthetic>":
|
|
130
|
+
model = msg["model"]
|
|
131
|
+
if msg.get("usage"):
|
|
132
|
+
usage[msg.get("id") or d.get("uuid")] = (model, msg["usage"], d.get("isSidechain"), d.get("timestamp"))
|
|
133
|
+
text = _claude_text(msg.get("content") or [])
|
|
134
|
+
if text and text.strip():
|
|
135
|
+
reply = text
|
|
136
|
+
elif kind == "user" and not d.get("isSidechain") and not d.get("isMeta"):
|
|
137
|
+
entrypoint = entrypoint or d.get("entrypoint")
|
|
138
|
+
cwd = cwd or d.get("cwd", "")
|
|
139
|
+
text = _claude_text(msg.get("content") or "")
|
|
140
|
+
if not text or d.get("promptSource") == "system":
|
|
141
|
+
continue
|
|
142
|
+
text = text.strip()
|
|
143
|
+
if text.startswith("[Request interrupted by user"):
|
|
144
|
+
if model:
|
|
145
|
+
corpus.interrupts.append(Interrupt("claude", path, d.get("timestamp", ""), norm(model)))
|
|
146
|
+
continue
|
|
147
|
+
if text.startswith(("<", "Caveat:")) or not model:
|
|
148
|
+
continue
|
|
149
|
+
pending.append(Reaction(d.get("uuid", ""), "claude", path, d.get("timestamp", ""), text,
|
|
150
|
+
reply[-900:], norm(model), workspace_of(cwd)))
|
|
151
|
+
reply = ""
|
|
152
|
+
headless = entrypoint in HEADLESS_ENTRYPOINTS
|
|
153
|
+
human = not (subagent or headless or excluded)
|
|
154
|
+
if human and not skip_session(entrypoint):
|
|
155
|
+
corpus.reactions += pending
|
|
156
|
+
for model_id, u, side, ts in usage.values():
|
|
157
|
+
if excluded:
|
|
158
|
+
continue
|
|
159
|
+
origin = "you" if human and not side else "agents"
|
|
160
|
+
corpus.usage.add(model_id, origin, ts, u.get("input_tokens", 0), u.get("output_tokens", 0),
|
|
161
|
+
u.get("cache_read_input_tokens", 0) + u.get("cache_creation_input_tokens", 0))
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
# --- Codex -------------------------------------------------------------------------------------
|
|
165
|
+
|
|
166
|
+
def read_codex(corpus, root, skip_originator=lambda originator: False):
|
|
167
|
+
files = glob.glob(os.path.join(root, "**", "*.jsonl"), recursive=True)
|
|
168
|
+
if not files:
|
|
169
|
+
return
|
|
170
|
+
corpus.found.append(f"Codex: {len(files)} sessions")
|
|
171
|
+
for path in files:
|
|
172
|
+
human, skip, model, reply, cwd = False, False, None, "", ""
|
|
173
|
+
for d in _jsonl(path):
|
|
174
|
+
p = d.get("payload") or {}
|
|
175
|
+
kind = d.get("type")
|
|
176
|
+
if kind == "session_meta":
|
|
177
|
+
source = p.get("source")
|
|
178
|
+
human = isinstance(source, str) and source != "exec" and p.get("originator") != "codex_exec"
|
|
179
|
+
skip = skip_originator(p.get("originator") or "")
|
|
180
|
+
cwd = p.get("cwd") or ""
|
|
181
|
+
elif kind == "turn_context":
|
|
182
|
+
model = p.get("model") or model
|
|
183
|
+
elif kind == "event_msg":
|
|
184
|
+
t = p.get("type")
|
|
185
|
+
if t == "agent_message":
|
|
186
|
+
reply = p.get("message") or reply
|
|
187
|
+
elif t == "user_message" and human and not skip and model:
|
|
188
|
+
text = (p.get("message") or "").strip()
|
|
189
|
+
if text and not text.startswith("<"):
|
|
190
|
+
corpus.reactions.append(Reaction(f"{path}:{d.get('timestamp')}", "codex", path,
|
|
191
|
+
d.get("timestamp", ""), text, reply[-900:], norm(model),
|
|
192
|
+
workspace_of(cwd)))
|
|
193
|
+
reply = ""
|
|
194
|
+
elif t == "turn_aborted" and human and not skip and model:
|
|
195
|
+
corpus.interrupts.append(Interrupt("codex", path, d.get("timestamp", ""), norm(model)))
|
|
196
|
+
elif t == "token_count" and model and (p.get("info") or {}).get("last_token_usage"):
|
|
197
|
+
u = p["info"]["last_token_usage"]
|
|
198
|
+
cached = u.get("cached_input_tokens", 0)
|
|
199
|
+
corpus.usage.add(model, "you" if human else "agents", d.get("timestamp"),
|
|
200
|
+
u.get("input_tokens", 0) - cached, u.get("output_tokens", 0), cached)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
# --- OpenCode ----------------------------------------------------------------------------------
|
|
204
|
+
|
|
205
|
+
def read_opencode(corpus, db, skip_sessions=frozenset()):
|
|
206
|
+
if not os.path.exists(db):
|
|
207
|
+
return
|
|
208
|
+
c = _ro(db)
|
|
209
|
+
try:
|
|
210
|
+
sessions = dict(c.execute("select id, parent_id from session"))
|
|
211
|
+
dirs = dict(c.execute("select id, directory from session"))
|
|
212
|
+
except sqlite3.Error:
|
|
213
|
+
return
|
|
214
|
+
corpus.found.append(f"OpenCode: {len(sessions)} sessions")
|
|
215
|
+
texts = defaultdict(list)
|
|
216
|
+
for mid, data in c.execute("select message_id, data from part order by time_created"):
|
|
217
|
+
p = json.loads(data)
|
|
218
|
+
if p.get("type") == "text" and not p.get("synthetic"):
|
|
219
|
+
texts[mid].append(p.get("text", ""))
|
|
220
|
+
state = {}
|
|
221
|
+
for mid, sid, created, data in c.execute(
|
|
222
|
+
"select id, session_id, time_created, data from message order by session_id, time_created"):
|
|
223
|
+
d = json.loads(data)
|
|
224
|
+
model, reply = state.get(sid, (None, ""))
|
|
225
|
+
human = sessions.get(sid) is None
|
|
226
|
+
ts = ms_to_iso(created)
|
|
227
|
+
text = "\n".join(texts.get(mid, [])).strip()
|
|
228
|
+
if d.get("role") == "assistant":
|
|
229
|
+
model = "opencode/" + d.get("modelID", "?")
|
|
230
|
+
t = d.get("tokens") or {}
|
|
231
|
+
corpus.usage.add(model, "you" if human else "agents", ts, t.get("input", 0),
|
|
232
|
+
t.get("output", 0) + t.get("reasoning", 0), (t.get("cache") or {}).get("read", 0))
|
|
233
|
+
if (d.get("error") or {}).get("name") == "MessageAbortedError" and human and sid not in skip_sessions:
|
|
234
|
+
corpus.interrupts.append(Interrupt("opencode", sid, ts, norm(model)))
|
|
235
|
+
state[sid] = (model, text or reply)
|
|
236
|
+
elif d.get("role") == "user":
|
|
237
|
+
if human and model and text and sid not in skip_sessions:
|
|
238
|
+
corpus.reactions.append(Reaction(mid, "opencode", sid, ts, text, reply[-900:], norm(model),
|
|
239
|
+
workspace_of(dirs.get(sid))))
|
|
240
|
+
state[sid] = (model, "")
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
# --- T3 Code (optional; gives exact per-turn model attribution across providers) ----------------
|
|
244
|
+
|
|
245
|
+
def t3_db():
|
|
246
|
+
base = os.path.join(HOME, ".config", "t3", "userdata")
|
|
247
|
+
for name in ("statev2.sqlite", "state.sqlite"):
|
|
248
|
+
path = os.path.join(base, name)
|
|
249
|
+
if os.path.exists(path):
|
|
250
|
+
return path
|
|
251
|
+
return None
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def t3_provider_sessions(db):
|
|
255
|
+
out = defaultdict(set)
|
|
256
|
+
for provider, cursor in _ro(db).execute("select provider_name, resume_cursor_json from provider_session_runtime"):
|
|
257
|
+
cur = json.loads(cursor or "{}")
|
|
258
|
+
out[provider].add(cur.get("sessionId") or cur.get("threadId"))
|
|
259
|
+
return out
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def read_t3(corpus, db):
|
|
263
|
+
c = _ro(db)
|
|
264
|
+
children = {r[0] for r in c.execute("select child_thread_id from orchestration_v2_projection_subagents "
|
|
265
|
+
"where child_thread_id is not null")} if _has(c, "orchestration_v2_projection_subagents") else set()
|
|
266
|
+
turn_model = {}
|
|
267
|
+
for mid, payload in c.execute("select json_extract(payload_json,'$.messageId'), payload_json from orchestration_events "
|
|
268
|
+
"where event_type='thread.turn-start-requested'"):
|
|
269
|
+
turn_model[mid] = (json.loads(payload).get("modelSelection") or {}).get("model")
|
|
270
|
+
project = dict(c.execute("select thread_id, project_id from projection_threads"))
|
|
271
|
+
by_thread = defaultdict(list)
|
|
272
|
+
for mid, tid, role, text, ts in c.execute("select message_id, thread_id, role, text, created_at from projection_thread_messages "
|
|
273
|
+
"where role in ('user','assistant') order by created_at"):
|
|
274
|
+
if tid not in children:
|
|
275
|
+
by_thread[tid].append((mid, role, text or "", ts))
|
|
276
|
+
corpus.found.append(f"T3 Code: {len(by_thread)} threads")
|
|
277
|
+
sent = defaultdict(list)
|
|
278
|
+
for tid, msgs in by_thread.items():
|
|
279
|
+
current, reply = None, ""
|
|
280
|
+
for mid, role, text, ts in msgs:
|
|
281
|
+
if role == "assistant":
|
|
282
|
+
reply = text if text.strip() else reply
|
|
283
|
+
continue
|
|
284
|
+
if current and text.strip():
|
|
285
|
+
corpus.reactions.append(Reaction(mid, "t3", tid, ts, text.strip(), reply[-900:], norm(current),
|
|
286
|
+
project.get(tid) or ""))
|
|
287
|
+
current = turn_model.get(mid) or current
|
|
288
|
+
sent[tid].append((ts, current))
|
|
289
|
+
reply = ""
|
|
290
|
+
if _has(c, "projection_thread_pull_requests"):
|
|
291
|
+
corpus.merged |= {tid for tid, state in c.execute(
|
|
292
|
+
"select thread_id, json_extract(snapshot_json,'$.state') from projection_thread_pull_requests") if state == "merged"}
|
|
293
|
+
corpus.pr_tracking_since = c.execute("select min(linked_at) from projection_thread_pull_requests").fetchone()[0]
|
|
294
|
+
for tid, ts in c.execute("select json_extract(payload_json,'$.threadId'), occurred_at from orchestration_events "
|
|
295
|
+
"where event_type='thread.turn-interrupt-requested'"):
|
|
296
|
+
before = [m for t, m in sent.get(tid, []) if t <= ts]
|
|
297
|
+
if before and before[-1]:
|
|
298
|
+
corpus.interrupts.append(Interrupt("t3", tid, ts, norm(before[-1])))
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def _has(c, table):
|
|
302
|
+
return c.execute("select 1 from sqlite_master where name=?", (table,)).fetchone() is not None
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def collect(use_t3=True, exclude_dirs=()):
|
|
306
|
+
corpus = Corpus()
|
|
307
|
+
db = t3_db() if use_t3 else None
|
|
308
|
+
t3_sessions = t3_provider_sessions(db) if db else {}
|
|
309
|
+
if db:
|
|
310
|
+
read_t3(corpus, db)
|
|
311
|
+
read_claude(corpus, os.path.join(HOME, ".claude", "projects"),
|
|
312
|
+
skip_session=(lambda ep: ep == "sdk-ts") if db else (lambda ep: False), exclude_dirs=exclude_dirs)
|
|
313
|
+
read_codex(corpus, os.path.join(HOME, ".codex", "sessions"),
|
|
314
|
+
skip_originator=(lambda o: "t3" in o.lower()) if db else (lambda o: False))
|
|
315
|
+
read_opencode(corpus, os.path.join(HOME, ".local", "share", "opencode", "opencode.db"),
|
|
316
|
+
skip_sessions=frozenset(t3_sessions.get("opencode", ())))
|
|
317
|
+
return corpus
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: swearbench
|
|
3
|
+
Version: 0.2.1
|
|
4
|
+
Summary: Rank AI coding models by how much they made you swear and how often their work shipped, from your own local chat logs.
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
License-File: LICENSE
|
|
7
|
+
Requires-Python: >=3.10
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
|
|
10
|
+
# SwearBench
|
|
11
|
+
|
|
12
|
+
**Which AI coding model made you swear the least, and still shipped?**
|
|
13
|
+
|
|
14
|
+
Public benchmarks measure what models can do. SwearBench measures how they made *you* feel: it reads your
|
|
15
|
+
own local agent logs, finds every message you sent in reaction to a model's reply, has an LLM judge label how
|
|
16
|
+
mad you were and why, checks whether the work ended up accepted, and ranks the models.
|
|
17
|
+
|
|
18
|
+
Example from the author's own logs (~4,600 messages, July–October 2026):
|
|
19
|
+
|
|
20
|
+

|
|
21
|
+
|
|
22
|
+

|
|
23
|
+
|
|
24
|
+
## Run it
|
|
25
|
+
|
|
26
|
+
```sh
|
|
27
|
+
uvx swearbench
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
(or `pipx run swearbench`, or `pip install swearbench`)
|
|
31
|
+
|
|
32
|
+
It prints a leaderboard and writes `swearbench-out/report.md`, `chart.svg` (swearing vs. result), `card.svg` and `results.json`.
|
|
33
|
+
Before anything leaves your machine it tells you how many messages it will send to the judge and asks.
|
|
34
|
+
|
|
35
|
+
| Flag | |
|
|
36
|
+
|---|---|
|
|
37
|
+
| `--dry-run` | show what logs were found and how many messages would be judged |
|
|
38
|
+
| `--judge claude-cli\|codex-cli\|anthropic\|openai\|command` | who labels your messages (default: first of `claude`, `codex`, `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`) |
|
|
39
|
+
| `--judge-model ID` | model for the judge |
|
|
40
|
+
| `--judge command --judge-cmd "ollama run qwen3"` | any command that reads a prompt on stdin and prints the reply; keeps everything local |
|
|
41
|
+
| `--since 2026-09-01` | only recent reactions |
|
|
42
|
+
| `--exclude MODEL` | drop a model from the ranking |
|
|
43
|
+
| `--no-quotes` | leave the hall of shame out of the report |
|
|
44
|
+
|
|
45
|
+
Labels are cached in `~/.cache/swearbench/`, so re-runs only judge new messages.
|
|
46
|
+
|
|
47
|
+
## Where it looks
|
|
48
|
+
|
|
49
|
+
| Tool | Location |
|
|
50
|
+
|---|---|
|
|
51
|
+
| Claude Code | `~/.claude/projects/**/*.jsonl` |
|
|
52
|
+
| Codex CLI | `~/.codex/sessions/**/*.jsonl` |
|
|
53
|
+
| OpenCode | `~/.local/share/opencode/opencode.db` |
|
|
54
|
+
| T3 Code | `~/.config/t3/userdata/state*.sqlite` (used for exact per-turn model attribution when present) |
|
|
55
|
+
|
|
56
|
+
Only messages **you** typed count. Subagent transcripts, headless runs (`claude -p`, `codex exec`) and messages
|
|
57
|
+
the judge flags as agent-written are skipped. Models that only ever ran as subagents are listed, never ranked.
|
|
58
|
+
|
|
59
|
+
## How it scores
|
|
60
|
+
|
|
61
|
+
Every message you send is charged to the model whose reply you were answering. The judge labels it:
|
|
62
|
+
|
|
63
|
+
- **anger** 0–4, and **who it's aimed at**: the model, or something external (swearing at your cloud
|
|
64
|
+
provider is not the model's fault; "this looks fucking cool" is not anger)
|
|
65
|
+
- **how** you got mad, weighted by how bad it is:
|
|
66
|
+
|
|
67
|
+
| Mode | Weight | | Mode | Weight |
|
|
68
|
+
|---|---|---|---|---|
|
|
69
|
+
| fabrication (claimed success that wasn't) | 3 | | incomplete / lazy | 1.5 |
|
|
70
|
+
| overreach (did things you didn't ask) | 3 | | insult | 1.5 |
|
|
71
|
+
| giving up on it | 3 | | profanity | 1 |
|
|
72
|
+
| regression (broke what worked) | 2.5 | | shouting | 0.5 |
|
|
73
|
+
| ignored an instruction | 2 | | sarcasm | 0.5 |
|
|
74
|
+
| made you repeat yourself | 2 | | slow / verbose | 0.5 |
|
|
75
|
+
| | | | taste (critique while iterating on looks) | 0.5 |
|
|
76
|
+
|
|
77
|
+
- **satisfaction** −2 (rejects the work) to +2 (praise)
|
|
78
|
+
- whether it **blames earlier work** (something shipped before the last reply is broken or missing)
|
|
79
|
+
|
|
80
|
+
Rage for a message is built to match how it felt, not how often it happened:
|
|
81
|
+
|
|
82
|
+
- **Severity beats frequency.** Rage = anger² + mode weights, so one 4/4 blowup (16) outweighs four
|
|
83
|
+
1/4 grumbles (4). The report counts 4/4 blowups per model.
|
|
84
|
+
- **Taste isn't failure.** "That looks lame" while iterating on a design, with no broken rule, lie,
|
|
85
|
+
regression or ignored instruction, counts a quarter.
|
|
86
|
+
- **Regret goes to whoever caused it.** A complaint about earlier work ("why did X disappear") is charged to
|
|
87
|
+
the models that worked in the same repo in the week before, split by how many turns each had there, not
|
|
88
|
+
to the model that happens to be fixing it.
|
|
89
|
+
- Each interrupt adds 1.5.
|
|
90
|
+
|
|
91
|
+
The headline is half friction, half result:
|
|
92
|
+
|
|
93
|
+
```
|
|
94
|
+
Friction = 100 − 0.25 × rage per 100 turns + 10 × average satisfaction
|
|
95
|
+
Ships = accepted ÷ (accepted + rejected) sessions
|
|
96
|
+
SwearBench = (Friction + Ships) / 2
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
Friction deliberately ignores how *often* you were annoyed (a model you use for lots of quick "merge it"
|
|
100
|
+
turns would look calm by volume alone); it only counts how much rage piled up per turn.
|
|
101
|
+
A session's verdict is your last reaction in it. Sessions that stop or move to another model without a
|
|
102
|
+
verdict (usage limits, "pick up the work" in a new thread) are left out of Ships rather than counted as
|
|
103
|
+
failures; switching away in anger still counts as a rejection.
|
|
104
|
+
Friction alone rewards a model that is pleasant but never finishes; Ships alone ignores what it cost you
|
|
105
|
+
to get there. The chart plots the two against each other.
|
|
106
|
+
|
|
107
|
+
Intervals are a 90% bootstrap over sessions; models with fewer than 40 reactions aren't ranked.
|
|
108
|
+
With T3 Code, the report also shows the share of sessions that ended in a merged PR. It is informational
|
|
109
|
+
only, since T3 records PRs only from when it started tracking them.
|
|
110
|
+
|
|
111
|
+
**Per token of work.** If the logs carry token usage, the report adds a second ranking: rage per million output
|
|
112
|
+
tokens the model produced in sessions you drove. A model that does twice the work per message gets credit for it.
|
|
113
|
+
Subagent token use is shown separately.
|
|
114
|
+
|
|
115
|
+
## Caveats
|
|
116
|
+
|
|
117
|
+
- n = 1. It measures you, your tasks and your mood as much as the models. Models used in different months
|
|
118
|
+
did different work; the report tells you counts, not causes.
|
|
119
|
+
- The judge is a model too. If it belongs to a family being ranked, SwearBench says so; re-run with another
|
|
120
|
+
`--judge` and compare.
|
|
121
|
+
- Regret attribution is a heuristic: it blames whoever worked in the same repo during the previous week,
|
|
122
|
+
not the session that actually introduced the problem.
|
|
123
|
+
- The weights are opinions. Rage, taste and severity weights live at the top of `score.py`; change them
|
|
124
|
+
and re-run, labels are cached.
|
|
125
|
+
- Deleted or rotated logs mean missing data, especially for the per-token view.
|
|
126
|
+
- `report.md` quotes your own messages. Read it before you share it. The card and chart have no quotes.
|
|
127
|
+
|
|
128
|
+
## License
|
|
129
|
+
|
|
130
|
+
MIT
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
swearbench/__init__.py,sha256=HfjVOrpTnmZ-xVFCYSVmX50EXaBQeJteUHG-PD6iQs8,22
|
|
2
|
+
swearbench/__main__.py,sha256=E6Gls0DNz8GQK2K-kOUIx8cYhgANW_CH54VKrfCfs14,52
|
|
3
|
+
swearbench/card.py,sha256=xysCWH-6cRQ9fm71GIEKpnrLGp58e5rFShpQyDpaS08,3412
|
|
4
|
+
swearbench/chart.py,sha256=EfyUCD8xmyt_vzeJDz-lWkDotW2yJcKVQrIWN3b5T-s,6595
|
|
5
|
+
swearbench/cli.py,sha256=VzigcBt5oGTFVYKCQaXaoMJWaalHK-DUU8ku7Dbh0Bg,4108
|
|
6
|
+
swearbench/judge.py,sha256=FxrpUHlj5LXJExWW3FJ8v4kKQuBz-yTbcfU6R0Ik3i8,7452
|
|
7
|
+
swearbench/score.py,sha256=PJpV_yRiI2nfen0VcM7xS0pYGjYPqK0-fKdAshZ93sc,11518
|
|
8
|
+
swearbench/sources.py,sha256=LJN1jBkFksvZja-h12Y2UxjmpdJuXVmlJWL6JCqpINI,13645
|
|
9
|
+
swearbench-0.2.1.dist-info/METADATA,sha256=a-Sgjw_t4X6u89boms860vp9___6M7wXmVgvK9k4EfE,6293
|
|
10
|
+
swearbench-0.2.1.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
11
|
+
swearbench-0.2.1.dist-info/entry_points.txt,sha256=qrmKvKi9_IF4klWqvGhS4o6RU6kt_CgCaahl67rev1s,51
|
|
12
|
+
swearbench-0.2.1.dist-info/licenses/LICENSE,sha256=_OBJk5LyJLaZeO_rY8OZO6IKaDvmb0Wgf9Wc0PI_1Fg,1064
|
|
13
|
+
swearbench-0.2.1.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 AlenHay
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|