pairjudge 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pairjudge-0.1.0/LICENSE +21 -0
- pairjudge-0.1.0/PKG-INFO +179 -0
- pairjudge-0.1.0/README.md +142 -0
- pairjudge-0.1.0/pyproject.toml +55 -0
- pairjudge-0.1.0/setup.cfg +4 -0
- pairjudge-0.1.0/src/pairjudge/__init__.py +53 -0
- pairjudge-0.1.0/src/pairjudge/data.py +166 -0
- pairjudge-0.1.0/src/pairjudge/judge.py +185 -0
- pairjudge-0.1.0/src/pairjudge/packing.py +267 -0
- pairjudge-0.1.0/src/pairjudge/pseudo_label.py +76 -0
- pairjudge-0.1.0/src/pairjudge/training.py +275 -0
- pairjudge-0.1.0/src/pairjudge.egg-info/PKG-INFO +179 -0
- pairjudge-0.1.0/src/pairjudge.egg-info/SOURCES.txt +18 -0
- pairjudge-0.1.0/src/pairjudge.egg-info/dependency_links.txt +1 -0
- pairjudge-0.1.0/src/pairjudge.egg-info/requires.txt +21 -0
- pairjudge-0.1.0/src/pairjudge.egg-info/top_level.txt +1 -0
- pairjudge-0.1.0/tests/test_data.py +136 -0
- pairjudge-0.1.0/tests/test_judge.py +114 -0
- pairjudge-0.1.0/tests/test_packing.py +198 -0
- pairjudge-0.1.0/tests/test_training.py +78 -0
pairjudge-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2024 Daoyuan Li
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
pairjudge-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pairjudge
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Train and serve pairwise LLM judges (A/B/tie) with budget-aware multi-turn packing and position-bias correction
|
|
5
|
+
Author-email: Daoyuan Li <lidaoyuan2816@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/DaoyuanLi2816/pairjudge
|
|
8
|
+
Project-URL: Issues, https://github.com/DaoyuanLi2816/pairjudge/issues
|
|
9
|
+
Keywords: llm-as-judge,reward-model,preference-learning,rlhf,chatbot-arena
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
15
|
+
Requires-Python: >=3.9
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Requires-Dist: numpy
|
|
19
|
+
Requires-Dist: pandas
|
|
20
|
+
Requires-Dist: pyyaml
|
|
21
|
+
Provides-Extra: judge
|
|
22
|
+
Requires-Dist: torch; extra == "judge"
|
|
23
|
+
Requires-Dist: transformers>=4.46; extra == "judge"
|
|
24
|
+
Provides-Extra: train
|
|
25
|
+
Requires-Dist: torch; extra == "train"
|
|
26
|
+
Requires-Dist: transformers>=4.46; extra == "train"
|
|
27
|
+
Requires-Dist: peft; extra == "train"
|
|
28
|
+
Requires-Dist: datasets; extra == "train"
|
|
29
|
+
Requires-Dist: accelerate; extra == "train"
|
|
30
|
+
Requires-Dist: scikit-learn; extra == "train"
|
|
31
|
+
Provides-Extra: test
|
|
32
|
+
Requires-Dist: pytest; extra == "test"
|
|
33
|
+
Requires-Dist: torch; extra == "test"
|
|
34
|
+
Requires-Dist: transformers>=4.46; extra == "test"
|
|
35
|
+
Requires-Dist: scikit-learn; extra == "test"
|
|
36
|
+
Dynamic: license-file
|
|
37
|
+
|
|
38
|
+
# pairjudge
|
|
39
|
+
|
|
40
|
+
**Train and serve pairwise LLM judges (A wins / B wins / tie) — with budget-aware multi-turn packing, position-bias correction, and pseudo-label distillation.**
|
|
41
|
+
|
|
42
|
+
[](https://github.com/DaoyuanLi2816/pairjudge/actions)
|
|
43
|
+
[](LICENSE)
|
|
44
|
+

|
|
45
|
+
[](https://www.kaggle.com/competitions/lmsys-chatbot-arena/leaderboard)
|
|
46
|
+
|
|
47
|
+
`pairjudge` is the generalized core of the **4th-place (gold medal) solution** to Kaggle's [LMSYS — Chatbot Arena Human Preference Predictions](https://www.kaggle.com/competitions/lmsys-chatbot-arena/overview) (1,849 teams), extracted into a small, tested library you can run on **your own preference data with any Hugging Face backbone**. The exact competition artifacts are preserved untouched in [`competition/`](competition/README.md), and a golden test pins the library's default behavior to the medal-winning code **byte for byte**.
|
|
48
|
+
|
|
49
|
+
Use it when you need a model that answers: *given a prompt and two candidate responses, which one would a human prefer — or is it a tie?* That model is the engine behind response reranking, A/B evaluation of fine-tunes, RLHF/RLAIF reward signals, and arena-style leaderboards.
|
|
50
|
+
|
|
51
|
+
## Why not just an off-the-shelf reward model?
|
|
52
|
+
|
|
53
|
+
Three problems show up the moment you train a pairwise judge on real conversations, and they are exactly what this library packages:
|
|
54
|
+
|
|
55
|
+
**1. Truncation silently destroys the comparison.**
|
|
56
|
+
A judge input holds a multi-turn conversation *plus two responses per turn*. With naive left- or right-truncation, long inputs routinely lose response B (or the prompt) entirely — the judge then learns position artifacts instead of preferences. `PairPacker` packs rounds greedily and, when the budget runs out, truncates the final round *proportionally* (default 20% prompt / 40% response A / 40% response B), marks every cut with an explicit ellipsis, and drops rounds that can't be shown honestly. Guarantee: never exceeds `max_length`, and every retained round shows all three fields.
|
|
57
|
+
|
|
58
|
+
**2. Pairwise judges have position bias.**
|
|
59
|
+
Swap A and B and a naive judge changes its verdict on a measurable fraction of pairs. `PairwiseJudge.predict_proba(swap_debias=True)` scores each pair in both orders and averages in the original frame — order-invariant by construction. `position_flip_rate()` measures how biased your judge is before you decide to pay the 2x compute.
|
|
60
|
+
|
|
61
|
+
**3. Human preference labels are scarce and noisy.**
|
|
62
|
+
The medal recipe is a two-phase semi-supervised loop: train on human labels → pseudo-label a large unlabeled pool with **full probability distributions** → retrain with soft-label KL distillation (`label_mode: soft`). Ties are a first-class third category throughout — real human preference data is full of them, and scalar Bradley–Terry reward models (e.g. TRL's `RewardTrainer`, `num_labels=1`) cannot represent them.
|
|
63
|
+
|
|
64
|
+
## Install
|
|
65
|
+
|
|
66
|
+
```bash
|
|
67
|
+
pip install -e . # core: packing + data loaders (no torch needed)
|
|
68
|
+
pip install -e .[judge] # + inference (torch, transformers)
|
|
69
|
+
pip install -e .[train] # + LoRA fine-tuning (peft, datasets, accelerate)
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
## 60 seconds
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
from pairjudge import PairPacker, PackerConfig, from_pairs
|
|
76
|
+
|
|
77
|
+
# 1. Pack pairwise conversations into a token budget — any HF tokenizer.
|
|
78
|
+
from transformers import AutoTokenizer
|
|
79
|
+
tok = AutoTokenizer.from_pretrained("Qwen/Qwen2.5-0.5B-Instruct")
|
|
80
|
+
packer = PairPacker(tok, PackerConfig(max_length=2048))
|
|
81
|
+
packed = packer.pack(
|
|
82
|
+
prompts=["Explain quantum entanglement to a 10-year-old."],
|
|
83
|
+
responses_a=["Imagine two magic coins..."],
|
|
84
|
+
responses_b=["Quantum entanglement is a physical phenomenon..."],
|
|
85
|
+
)
|
|
86
|
+
packed.input_ids # <= 2048 tokens, prompt + BOTH responses guaranteed visible
|
|
87
|
+
packed.truncated # False — everything fit
|
|
88
|
+
|
|
89
|
+
# 2. Judge a pair with a trained model, position-bias-free.
|
|
90
|
+
from pairjudge import PairwiseJudge
|
|
91
|
+
judge = PairwiseJudge.from_pretrained("path/to/your/judge")
|
|
92
|
+
df = from_pairs(
|
|
93
|
+
prompts=["Explain quantum entanglement to a 10-year-old."],
|
|
94
|
+
responses_a=["Imagine two magic coins..."],
|
|
95
|
+
responses_b=["Quantum entanglement is a physical phenomenon..."],
|
|
96
|
+
)
|
|
97
|
+
judge.predict_proba(df, swap_debias=True) # [[p_a_wins, p_b_wins, p_tie]]
|
|
98
|
+
judge.position_flip_rate(df) # how order-sensitive is my judge?
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
## Train your own judge
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
# Small judge on one consumer GPU (Qwen2.5-0.5B, ungated):
|
|
105
|
+
python -m pairjudge.training --cfg examples/configs/quickstart.yaml
|
|
106
|
+
|
|
107
|
+
# The competition setup (gemma-2-9b-it, 4x A100):
|
|
108
|
+
python -m pairjudge.training --cfg examples/configs/reproduce_competition.yaml
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
Input is either an Arena-format CSV (the Kaggle competition schema) or a parquet with canonical columns — `prompt` / `response_a` / `response_b` as per-round string lists plus one-hot (or soft) `winner_*` columns. `pairjudge.data` ships loaders for Arena CSVs and UltraFeedback-style chosen/rejected data, plus `from_pairs()` for plain Python lists.
|
|
112
|
+
|
|
113
|
+
The full two-phase distillation loop:
|
|
114
|
+
|
|
115
|
+
```bash
|
|
116
|
+
# Phase 1: train on human labels
|
|
117
|
+
python -m pairjudge.training --cfg phase1.yaml # label_mode: hard
|
|
118
|
+
|
|
119
|
+
# Pseudo-label an unlabeled pool with the phase-1 judge (soft labels)
|
|
120
|
+
python -m pairjudge.pseudo_label \
|
|
121
|
+
--model ./output/judge/merged \
|
|
122
|
+
--data pool.parquet --out pool_pl.parquet --swap-debias
|
|
123
|
+
|
|
124
|
+
# Phase 2: retrain from scratch on human + soft labels with KL loss
|
|
125
|
+
python -m pairjudge.training --cfg phase2.yaml # label_mode: soft
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
In the competition, this loop (88k human-labeled + 30k pseudo-labeled UltraFeedback conversations) was a decisive part of the gap between a good model and a gold-medal one.
|
|
129
|
+
|
|
130
|
+
## Inference guardrails
|
|
131
|
+
|
|
132
|
+
Two degenerate cases are worth handling outside the model — on competition data this was worth a measurable amount of log-loss:
|
|
133
|
+
|
|
134
|
+
```python
|
|
135
|
+
from pairjudge import empty_and_identical_masks
|
|
136
|
+
|
|
137
|
+
a_empty, b_empty, identical = empty_and_identical_masks(raw_df)
|
|
138
|
+
proba[a_empty] = [0.04, 0.88, 0.08] # empty response loses — but never bet 1.0
|
|
139
|
+
proba[b_empty] = [0.88, 0.04, 0.08] # labels are noisy; log-loss punishes overconfidence
|
|
140
|
+
proba[identical] = [0.06, 0.06, 0.88] # identical responses are a tie
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
## How it relates to TRL's `RewardTrainer`
|
|
144
|
+
|
|
145
|
+
| | TRL `RewardTrainer` | `pairjudge` |
|
|
146
|
+
|---|---|---|
|
|
147
|
+
| Output | scalar reward (`num_labels=1`) | 3-class distribution (A / B / **tie**) |
|
|
148
|
+
| Loss | Bradley–Terry (logsigmoid of reward gap) | CE on human labels, KL on soft pseudo-labels |
|
|
149
|
+
| Ties | not representable | first-class |
|
|
150
|
+
| Multi-turn pair truncation | generic | proportional, all-fields-guaranteed |
|
|
151
|
+
| Position bias | n/a at inference (scores singletons) | swap-debias averaging + flip-rate diagnostic |
|
|
152
|
+
|
|
153
|
+
If you need a scalar reward for PPO-style RLHF, use TRL. If you need a *judge* that compares two concrete responses — for evaluation, reranking, data labeling, or arena prediction — and your data has ties, this is the recipe that placed 4th of 1,849 on exactly that task.
|
|
154
|
+
|
|
155
|
+
## Provenance & validation
|
|
156
|
+
|
|
157
|
+
- The competition scripts, configs, inference notebook and certificate are preserved verbatim in [`competition/`](competition/README.md), including the full original write-up.
|
|
158
|
+
- `tests/test_packing.py::TestCompetitionEquivalence` fuzzes 1,500 conversations against a verbatim copy of the competition tokenizer ([`tests/reference_impl.py`](tests/reference_impl.py)) and asserts byte-identical output with default settings — the library *is* the medal-winning code, not a reimplementation of it.
|
|
159
|
+
- Final leaderboard: **4th / 1,849** ([gold medal](https://www.kaggle.com/certification/competitions/distiller/lmsys-chatbot-arena), $20,000 prize).
|
|
160
|
+
|
|
161
|
+
## Citation
|
|
162
|
+
|
|
163
|
+
```bibtex
|
|
164
|
+
@misc{li2024pairjudge,
|
|
165
|
+
author = {Daoyuan Li},
|
|
166
|
+
title = {pairjudge: pairwise LLM judges with budget-aware packing and position-bias correction},
|
|
167
|
+
year = {2024},
|
|
168
|
+
url = {https://github.com/DaoyuanLi2816/pairjudge},
|
|
169
|
+
note = {Generalized from the 4th-place solution, Kaggle LMSYS Chatbot Arena Human Preference Predictions}
|
|
170
|
+
}
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
## License
|
|
174
|
+
|
|
175
|
+
MIT — see [LICENSE](LICENSE).
|
|
176
|
+
|
|
177
|
+
## Author
|
|
178
|
+
|
|
179
|
+
Daoyuan Li — [Kaggle (distiller)](https://www.kaggle.com/distiller) · lidaoyuan2816@gmail.com
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
# pairjudge
|
|
2
|
+
|
|
3
|
+
**Train and serve pairwise LLM judges (A wins / B wins / tie) — with budget-aware multi-turn packing, position-bias correction, and pseudo-label distillation.**
|
|
4
|
+
|
|
5
|
+
[](https://github.com/DaoyuanLi2816/pairjudge/actions)
|
|
6
|
+
[](LICENSE)
|
|
7
|
+

|
|
8
|
+
[](https://www.kaggle.com/competitions/lmsys-chatbot-arena/leaderboard)
|
|
9
|
+
|
|
10
|
+
`pairjudge` is the generalized core of the **4th-place (gold medal) solution** to Kaggle's [LMSYS — Chatbot Arena Human Preference Predictions](https://www.kaggle.com/competitions/lmsys-chatbot-arena/overview) (1,849 teams), extracted into a small, tested library you can run on **your own preference data with any Hugging Face backbone**. The exact competition artifacts are preserved untouched in [`competition/`](competition/README.md), and a golden test pins the library's default behavior to the medal-winning code **byte for byte**.
|
|
11
|
+
|
|
12
|
+
Use it when you need a model that answers: *given a prompt and two candidate responses, which one would a human prefer — or is it a tie?* That model is the engine behind response reranking, A/B evaluation of fine-tunes, RLHF/RLAIF reward signals, and arena-style leaderboards.
|
|
13
|
+
|
|
14
|
+
## Why not just an off-the-shelf reward model?
|
|
15
|
+
|
|
16
|
+
Three problems show up the moment you train a pairwise judge on real conversations, and they are exactly what this library packages:
|
|
17
|
+
|
|
18
|
+
**1. Truncation silently destroys the comparison.**
|
|
19
|
+
A judge input holds a multi-turn conversation *plus two responses per turn*. With naive left- or right-truncation, long inputs routinely lose response B (or the prompt) entirely — the judge then learns position artifacts instead of preferences. `PairPacker` packs rounds greedily and, when the budget runs out, truncates the final round *proportionally* (default 20% prompt / 40% response A / 40% response B), marks every cut with an explicit ellipsis, and drops rounds that can't be shown honestly. Guarantee: never exceeds `max_length`, and every retained round shows all three fields.
|
|
20
|
+
|
|
21
|
+
**2. Pairwise judges have position bias.**
|
|
22
|
+
Swap A and B and a naive judge changes its verdict on a measurable fraction of pairs. `PairwiseJudge.predict_proba(swap_debias=True)` scores each pair in both orders and averages in the original frame — order-invariant by construction. `position_flip_rate()` measures how biased your judge is before you decide to pay the 2x compute.
|
|
23
|
+
|
|
24
|
+
**3. Human preference labels are scarce and noisy.**
|
|
25
|
+
The medal recipe is a two-phase semi-supervised loop: train on human labels → pseudo-label a large unlabeled pool with **full probability distributions** → retrain with soft-label KL distillation (`label_mode: soft`). Ties are a first-class third category throughout — real human preference data is full of them, and scalar Bradley–Terry reward models (e.g. TRL's `RewardTrainer`, `num_labels=1`) cannot represent them.
|
|
26
|
+
|
|
27
|
+
## Install
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
pip install -e . # core: packing + data loaders (no torch needed)
|
|
31
|
+
pip install -e .[judge] # + inference (torch, transformers)
|
|
32
|
+
pip install -e .[train] # + LoRA fine-tuning (peft, datasets, accelerate)
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
## 60 seconds
|
|
36
|
+
|
|
37
|
+
```python
|
|
38
|
+
from pairjudge import PairPacker, PackerConfig, from_pairs
|
|
39
|
+
|
|
40
|
+
# 1. Pack pairwise conversations into a token budget — any HF tokenizer.
|
|
41
|
+
from transformers import AutoTokenizer
|
|
42
|
+
tok = AutoTokenizer.from_pretrained("Qwen/Qwen2.5-0.5B-Instruct")
|
|
43
|
+
packer = PairPacker(tok, PackerConfig(max_length=2048))
|
|
44
|
+
packed = packer.pack(
|
|
45
|
+
prompts=["Explain quantum entanglement to a 10-year-old."],
|
|
46
|
+
responses_a=["Imagine two magic coins..."],
|
|
47
|
+
responses_b=["Quantum entanglement is a physical phenomenon..."],
|
|
48
|
+
)
|
|
49
|
+
packed.input_ids # <= 2048 tokens, prompt + BOTH responses guaranteed visible
|
|
50
|
+
packed.truncated # False — everything fit
|
|
51
|
+
|
|
52
|
+
# 2. Judge a pair with a trained model, position-bias-free.
|
|
53
|
+
from pairjudge import PairwiseJudge
|
|
54
|
+
judge = PairwiseJudge.from_pretrained("path/to/your/judge")
|
|
55
|
+
df = from_pairs(
|
|
56
|
+
prompts=["Explain quantum entanglement to a 10-year-old."],
|
|
57
|
+
responses_a=["Imagine two magic coins..."],
|
|
58
|
+
responses_b=["Quantum entanglement is a physical phenomenon..."],
|
|
59
|
+
)
|
|
60
|
+
judge.predict_proba(df, swap_debias=True) # [[p_a_wins, p_b_wins, p_tie]]
|
|
61
|
+
judge.position_flip_rate(df) # how order-sensitive is my judge?
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
## Train your own judge
|
|
65
|
+
|
|
66
|
+
```bash
|
|
67
|
+
# Small judge on one consumer GPU (Qwen2.5-0.5B, ungated):
|
|
68
|
+
python -m pairjudge.training --cfg examples/configs/quickstart.yaml
|
|
69
|
+
|
|
70
|
+
# The competition setup (gemma-2-9b-it, 4x A100):
|
|
71
|
+
python -m pairjudge.training --cfg examples/configs/reproduce_competition.yaml
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
Input is either an Arena-format CSV (the Kaggle competition schema) or a parquet with canonical columns — `prompt` / `response_a` / `response_b` as per-round string lists plus one-hot (or soft) `winner_*` columns. `pairjudge.data` ships loaders for Arena CSVs and UltraFeedback-style chosen/rejected data, plus `from_pairs()` for plain Python lists.
|
|
75
|
+
|
|
76
|
+
The full two-phase distillation loop:
|
|
77
|
+
|
|
78
|
+
```bash
|
|
79
|
+
# Phase 1: train on human labels
|
|
80
|
+
python -m pairjudge.training --cfg phase1.yaml # label_mode: hard
|
|
81
|
+
|
|
82
|
+
# Pseudo-label an unlabeled pool with the phase-1 judge (soft labels)
|
|
83
|
+
python -m pairjudge.pseudo_label \
|
|
84
|
+
--model ./output/judge/merged \
|
|
85
|
+
--data pool.parquet --out pool_pl.parquet --swap-debias
|
|
86
|
+
|
|
87
|
+
# Phase 2: retrain from scratch on human + soft labels with KL loss
|
|
88
|
+
python -m pairjudge.training --cfg phase2.yaml # label_mode: soft
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
In the competition, this loop (88k human-labeled + 30k pseudo-labeled UltraFeedback conversations) was a decisive part of the gap between a good model and a gold-medal one.
|
|
92
|
+
|
|
93
|
+
## Inference guardrails
|
|
94
|
+
|
|
95
|
+
Two degenerate cases are worth handling outside the model — on competition data this was worth a measurable amount of log-loss:
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
from pairjudge import empty_and_identical_masks
|
|
99
|
+
|
|
100
|
+
a_empty, b_empty, identical = empty_and_identical_masks(raw_df)
|
|
101
|
+
proba[a_empty] = [0.04, 0.88, 0.08] # empty response loses — but never bet 1.0
|
|
102
|
+
proba[b_empty] = [0.88, 0.04, 0.08] # labels are noisy; log-loss punishes overconfidence
|
|
103
|
+
proba[identical] = [0.06, 0.06, 0.88] # identical responses are a tie
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
## How it relates to TRL's `RewardTrainer`
|
|
107
|
+
|
|
108
|
+
| | TRL `RewardTrainer` | `pairjudge` |
|
|
109
|
+
|---|---|---|
|
|
110
|
+
| Output | scalar reward (`num_labels=1`) | 3-class distribution (A / B / **tie**) |
|
|
111
|
+
| Loss | Bradley–Terry (logsigmoid of reward gap) | CE on human labels, KL on soft pseudo-labels |
|
|
112
|
+
| Ties | not representable | first-class |
|
|
113
|
+
| Multi-turn pair truncation | generic | proportional, all-fields-guaranteed |
|
|
114
|
+
| Position bias | n/a at inference (scores singletons) | swap-debias averaging + flip-rate diagnostic |
|
|
115
|
+
|
|
116
|
+
If you need a scalar reward for PPO-style RLHF, use TRL. If you need a *judge* that compares two concrete responses — for evaluation, reranking, data labeling, or arena prediction — and your data has ties, this is the recipe that placed 4th of 1,849 on exactly that task.
|
|
117
|
+
|
|
118
|
+
## Provenance & validation
|
|
119
|
+
|
|
120
|
+
- The competition scripts, configs, inference notebook and certificate are preserved verbatim in [`competition/`](competition/README.md), including the full original write-up.
|
|
121
|
+
- `tests/test_packing.py::TestCompetitionEquivalence` fuzzes 1,500 conversations against a verbatim copy of the competition tokenizer ([`tests/reference_impl.py`](tests/reference_impl.py)) and asserts byte-identical output with default settings — the library *is* the medal-winning code, not a reimplementation of it.
|
|
122
|
+
- Final leaderboard: **4th / 1,849** ([gold medal](https://www.kaggle.com/certification/competitions/distiller/lmsys-chatbot-arena), $20,000 prize).
|
|
123
|
+
|
|
124
|
+
## Citation
|
|
125
|
+
|
|
126
|
+
```bibtex
|
|
127
|
+
@misc{li2024pairjudge,
|
|
128
|
+
author = {Daoyuan Li},
|
|
129
|
+
title = {pairjudge: pairwise LLM judges with budget-aware packing and position-bias correction},
|
|
130
|
+
year = {2024},
|
|
131
|
+
url = {https://github.com/DaoyuanLi2816/pairjudge},
|
|
132
|
+
note = {Generalized from the 4th-place solution, Kaggle LMSYS Chatbot Arena Human Preference Predictions}
|
|
133
|
+
}
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
## License
|
|
137
|
+
|
|
138
|
+
MIT — see [LICENSE](LICENSE).
|
|
139
|
+
|
|
140
|
+
## Author
|
|
141
|
+
|
|
142
|
+
Daoyuan Li — [Kaggle (distiller)](https://www.kaggle.com/distiller) · lidaoyuan2816@gmail.com
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=64"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "pairjudge"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Train and serve pairwise LLM judges (A/B/tie) with budget-aware multi-turn packing and position-bias correction"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "Daoyuan Li", email = "lidaoyuan2816@gmail.com" }]
|
|
13
|
+
keywords = [
|
|
14
|
+
"llm-as-judge",
|
|
15
|
+
"reward-model",
|
|
16
|
+
"preference-learning",
|
|
17
|
+
"rlhf",
|
|
18
|
+
"chatbot-arena",
|
|
19
|
+
]
|
|
20
|
+
classifiers = [
|
|
21
|
+
"Development Status :: 4 - Beta",
|
|
22
|
+
"Intended Audience :: Science/Research",
|
|
23
|
+
"License :: OSI Approved :: MIT License",
|
|
24
|
+
"Programming Language :: Python :: 3",
|
|
25
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
26
|
+
]
|
|
27
|
+
dependencies = [
|
|
28
|
+
"numpy",
|
|
29
|
+
"pandas",
|
|
30
|
+
"pyyaml",
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
[project.optional-dependencies]
|
|
34
|
+
# Inference: load a trained judge and predict preferences.
|
|
35
|
+
judge = ["torch", "transformers>=4.46"]
|
|
36
|
+
# Training: fine-tune a judge with LoRA on your own preference data.
|
|
37
|
+
train = [
|
|
38
|
+
"torch",
|
|
39
|
+
"transformers>=4.46",
|
|
40
|
+
"peft",
|
|
41
|
+
"datasets",
|
|
42
|
+
"accelerate",
|
|
43
|
+
"scikit-learn",
|
|
44
|
+
]
|
|
45
|
+
test = ["pytest", "torch", "transformers>=4.46", "scikit-learn"]
|
|
46
|
+
|
|
47
|
+
[project.urls]
|
|
48
|
+
Homepage = "https://github.com/DaoyuanLi2816/pairjudge"
|
|
49
|
+
Issues = "https://github.com/DaoyuanLi2816/pairjudge/issues"
|
|
50
|
+
|
|
51
|
+
[tool.setuptools.packages.find]
|
|
52
|
+
where = ["src"]
|
|
53
|
+
|
|
54
|
+
[tool.pytest.ini_options]
|
|
55
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""pairjudge — train and serve pairwise LLM judges (A/B/tie).
|
|
2
|
+
|
|
3
|
+
Extracted and generalized from the 4th-place (gold medal, 4/1849) solution
|
|
4
|
+
to the Kaggle competition "LMSYS — Chatbot Arena Human Preference
|
|
5
|
+
Predictions". The packing defaults are byte-for-byte equivalent to the
|
|
6
|
+
competition tokenization (golden-tested).
|
|
7
|
+
|
|
8
|
+
Core pieces:
|
|
9
|
+
|
|
10
|
+
- :class:`pairjudge.PairPacker` — budget-aware packing of multi-turn
|
|
11
|
+
(prompt, response A, response B) conversations. No heavy dependencies.
|
|
12
|
+
- :class:`pairjudge.PairwiseJudge` — inference with optional swap-based
|
|
13
|
+
position-bias correction (requires ``pairjudge[judge]``).
|
|
14
|
+
- :mod:`pairjudge.training` — LoRA fine-tuning with hard (CE) or soft (KL
|
|
15
|
+
distillation) labels (requires ``pairjudge[train]``).
|
|
16
|
+
- :mod:`pairjudge.data` — loaders normalizing Arena/UltraFeedback-style data
|
|
17
|
+
to one canonical schema.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from .data import (
|
|
21
|
+
EMPTY_RESPONSE_PATTERNS,
|
|
22
|
+
empty_and_identical_masks,
|
|
23
|
+
from_pairs,
|
|
24
|
+
load_arena_csv,
|
|
25
|
+
load_ultrafeedback,
|
|
26
|
+
)
|
|
27
|
+
from .packing import PackedExample, PackerConfig, PairPacker, hard_label
|
|
28
|
+
|
|
29
|
+
__version__ = "0.1.0"
|
|
30
|
+
|
|
31
|
+
__all__ = [
|
|
32
|
+
"PairPacker",
|
|
33
|
+
"PackerConfig",
|
|
34
|
+
"PackedExample",
|
|
35
|
+
"hard_label",
|
|
36
|
+
"load_arena_csv",
|
|
37
|
+
"load_ultrafeedback",
|
|
38
|
+
"from_pairs",
|
|
39
|
+
"empty_and_identical_masks",
|
|
40
|
+
"EMPTY_RESPONSE_PATTERNS",
|
|
41
|
+
"PairwiseJudge",
|
|
42
|
+
"swap_average",
|
|
43
|
+
"__version__",
|
|
44
|
+
]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def __getattr__(name):
|
|
48
|
+
# Lazy import: PairwiseJudge needs torch/transformers, which are optional.
|
|
49
|
+
if name in ("PairwiseJudge", "swap_average"):
|
|
50
|
+
from . import judge
|
|
51
|
+
|
|
52
|
+
return getattr(judge, name)
|
|
53
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
"""Loaders that normalize preference data into one canonical schema.
|
|
2
|
+
|
|
3
|
+
Canonical columns (one row = one conversation):
|
|
4
|
+
|
|
5
|
+
- ``id``: str
|
|
6
|
+
- ``prompt``: list[str] — user prompt per round
|
|
7
|
+
- ``response_a`` / ``response_b``: list[str] — candidate responses per round
|
|
8
|
+
- ``winner_model_a`` / ``winner_model_b`` / ``winner_tie``: float — one-hot
|
|
9
|
+
for human labels, arbitrary distribution for pseudo-labels
|
|
10
|
+
|
|
11
|
+
Everything downstream (packing, training, judging) consumes this schema, so
|
|
12
|
+
supporting a new dataset means writing one loader function.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
import random
|
|
19
|
+
from typing import Iterable, Optional, Sequence
|
|
20
|
+
|
|
21
|
+
import pandas as pd
|
|
22
|
+
|
|
23
|
+
#: String encodings of an empty response as they appear in Chatbot Arena
|
|
24
|
+
#: exports. A response equal to any of these is "empty" for labeling and
|
|
25
|
+
#: guardrail purposes.
|
|
26
|
+
EMPTY_RESPONSE_PATTERNS = ('[null]', '[]', '[ ]', '[ ]', '[""]', '["",""]')
|
|
27
|
+
|
|
28
|
+
_LIST_COLUMNS = ("prompt", "response_a", "response_b")
|
|
29
|
+
_WINNER_COLUMNS = ("winner_model_a", "winner_model_b", "winner_tie")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _is_empty(series: pd.Series) -> pd.Series:
|
|
33
|
+
return series.isin(EMPTY_RESPONSE_PATTERNS)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def load_arena_csv(
|
|
37
|
+
path_or_df,
|
|
38
|
+
drop_identical: bool = True,
|
|
39
|
+
relabel_empty: bool = True,
|
|
40
|
+
) -> pd.DataFrame:
|
|
41
|
+
"""Load a Chatbot-Arena-style CSV (LMSYS competition format).
|
|
42
|
+
|
|
43
|
+
Expects JSON-encoded list columns ``prompt``/``response_a``/``response_b``
|
|
44
|
+
and one-hot winner columns.
|
|
45
|
+
|
|
46
|
+
Cleaning rules (from the gold-medal solution):
|
|
47
|
+
|
|
48
|
+
- rows where *both* responses are empty are dropped (no signal);
|
|
49
|
+
- rows where the two responses are byte-identical are dropped when
|
|
50
|
+
``drop_identical`` — the label is noise there, and inference handles
|
|
51
|
+
that case with a guardrail instead (:func:`empty_and_identical_masks`);
|
|
52
|
+
- rows where exactly one response is empty are relabeled so the non-empty
|
|
53
|
+
side wins when ``relabel_empty`` — annotators almost always prefer *any*
|
|
54
|
+
answer over a blank one, and the few contrary labels are noise.
|
|
55
|
+
"""
|
|
56
|
+
df = path_or_df if isinstance(path_or_df, pd.DataFrame) else pd.read_csv(path_or_df, encoding="utf-8")
|
|
57
|
+
df = df.copy()
|
|
58
|
+
|
|
59
|
+
a_empty, b_empty = _is_empty(df["response_a"]), _is_empty(df["response_b"])
|
|
60
|
+
df = df[~(a_empty & b_empty)]
|
|
61
|
+
if drop_identical:
|
|
62
|
+
df = df[~(df["response_a"] == df["response_b"])]
|
|
63
|
+
|
|
64
|
+
if relabel_empty:
|
|
65
|
+
a_empty, b_empty = _is_empty(df["response_a"]), _is_empty(df["response_b"])
|
|
66
|
+
df.loc[a_empty, list(_WINNER_COLUMNS)] = [0.0, 1.0, 0.0]
|
|
67
|
+
df.loc[b_empty, list(_WINNER_COLUMNS)] = [1.0, 0.0, 0.0]
|
|
68
|
+
|
|
69
|
+
for col in _LIST_COLUMNS:
|
|
70
|
+
df[col] = df[col].apply(json.loads)
|
|
71
|
+
|
|
72
|
+
df = df.reset_index(drop=True)
|
|
73
|
+
df["id"] = df["id"].astype(str)
|
|
74
|
+
return df
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def load_ultrafeedback(
|
|
78
|
+
path_or_df,
|
|
79
|
+
seed: int = 42,
|
|
80
|
+
dedup_by_prompt: bool = True,
|
|
81
|
+
) -> pd.DataFrame:
|
|
82
|
+
"""Convert an UltraFeedback-style chosen/rejected dataset to the canonical schema.
|
|
83
|
+
|
|
84
|
+
Each record has ``prompt`` (str) and ``chosen``/``rejected`` conversations
|
|
85
|
+
(list of ``{"role", "content"}`` messages, assistant reply at index 1).
|
|
86
|
+
The chosen/rejected pair is assigned to A/B *uniformly at random* (seeded)
|
|
87
|
+
so the resulting dataset is free of position bias by construction.
|
|
88
|
+
"""
|
|
89
|
+
df = path_or_df if isinstance(path_or_df, pd.DataFrame) else pd.read_parquet(path_or_df)
|
|
90
|
+
rng = random.Random(seed)
|
|
91
|
+
|
|
92
|
+
records = []
|
|
93
|
+
for _, row in df.iterrows():
|
|
94
|
+
chosen = [row["chosen"][1]["content"]]
|
|
95
|
+
rejected = [row["rejected"][1]["content"]]
|
|
96
|
+
if rng.random() > 0.5:
|
|
97
|
+
response_a, response_b, winner = chosen, rejected, "a"
|
|
98
|
+
else:
|
|
99
|
+
response_a, response_b, winner = rejected, chosen, "b"
|
|
100
|
+
records.append(
|
|
101
|
+
{
|
|
102
|
+
"prompt": [row["prompt"]],
|
|
103
|
+
"response_a": response_a,
|
|
104
|
+
"response_b": response_b,
|
|
105
|
+
"winner_model_a": 1.0 if winner == "a" else 0.0,
|
|
106
|
+
"winner_model_b": 1.0 if winner == "b" else 0.0,
|
|
107
|
+
"winner_tie": 0.0,
|
|
108
|
+
}
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
out = pd.DataFrame(records)
|
|
112
|
+
if dedup_by_prompt:
|
|
113
|
+
out["_key"] = out["prompt"].apply(lambda x: x[0])
|
|
114
|
+
out = out.drop_duplicates(subset=["_key"], ignore_index=True)
|
|
115
|
+
out = out.drop(columns=["_key"])
|
|
116
|
+
out["id"] = out.index.astype(str)
|
|
117
|
+
return out
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def from_pairs(
|
|
121
|
+
prompts: Sequence[str],
|
|
122
|
+
responses_a: Sequence[str],
|
|
123
|
+
responses_b: Sequence[str],
|
|
124
|
+
winners: Optional[Iterable[str]] = None,
|
|
125
|
+
) -> pd.DataFrame:
|
|
126
|
+
"""Build a canonical dataframe from flat single-turn pairs.
|
|
127
|
+
|
|
128
|
+
``winners`` entries are ``"a"``, ``"b"`` or ``"tie"``; omit for unlabeled
|
|
129
|
+
data (e.g. inference or pseudo-labeling inputs).
|
|
130
|
+
"""
|
|
131
|
+
df = pd.DataFrame(
|
|
132
|
+
{
|
|
133
|
+
"prompt": [[p] for p in prompts],
|
|
134
|
+
"response_a": [[r] for r in responses_a],
|
|
135
|
+
"response_b": [[r] for r in responses_b],
|
|
136
|
+
}
|
|
137
|
+
)
|
|
138
|
+
if winners is not None:
|
|
139
|
+
winners = list(winners)
|
|
140
|
+
bad = sorted({w for w in winners} - {"a", "b", "tie"})
|
|
141
|
+
if bad:
|
|
142
|
+
raise ValueError(f"winners must be 'a', 'b' or 'tie', got {bad}")
|
|
143
|
+
df["winner_model_a"] = [1.0 if w == "a" else 0.0 for w in winners]
|
|
144
|
+
df["winner_model_b"] = [1.0 if w == "b" else 0.0 for w in winners]
|
|
145
|
+
df["winner_tie"] = [1.0 if w == "tie" else 0.0 for w in winners]
|
|
146
|
+
df["id"] = df.index.astype(str)
|
|
147
|
+
return df
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def empty_and_identical_masks(df: pd.DataFrame):
|
|
151
|
+
"""Inference guardrail masks for degenerate pairs.
|
|
152
|
+
|
|
153
|
+
Returns boolean Series ``(a_empty, b_empty, identical)`` over raw (still
|
|
154
|
+
JSON-encoded) response columns. A judge should not be trusted on these
|
|
155
|
+
rows: an empty response loses against a non-empty one, and identical
|
|
156
|
+
responses are a tie. Overriding the model's prediction with fixed,
|
|
157
|
+
*calibrated* probabilities (not 0/1 — labels are noisy and log-loss
|
|
158
|
+
punishes overconfidence) is worth a measurable amount of log-loss; the
|
|
159
|
+
gold-medal solution used ``[0.04, 0.88, 0.08]`` for empty-vs-non-empty
|
|
160
|
+
and ``[0.06, 0.06, 0.88]`` for identical pairs.
|
|
161
|
+
"""
|
|
162
|
+
return (
|
|
163
|
+
_is_empty(df["response_a"]),
|
|
164
|
+
_is_empty(df["response_b"]),
|
|
165
|
+
df["response_a"] == df["response_b"],
|
|
166
|
+
)
|