pairjudge 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2024 Daoyuan Li
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,179 @@
1
+ Metadata-Version: 2.4
2
+ Name: pairjudge
3
+ Version: 0.1.0
4
+ Summary: Train and serve pairwise LLM judges (A/B/tie) with budget-aware multi-turn packing and position-bias correction
5
+ Author-email: Daoyuan Li <lidaoyuan2816@gmail.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/DaoyuanLi2816/pairjudge
8
+ Project-URL: Issues, https://github.com/DaoyuanLi2816/pairjudge/issues
9
+ Keywords: llm-as-judge,reward-model,preference-learning,rlhf,chatbot-arena
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Intended Audience :: Science/Research
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
15
+ Requires-Python: >=3.9
16
+ Description-Content-Type: text/markdown
17
+ License-File: LICENSE
18
+ Requires-Dist: numpy
19
+ Requires-Dist: pandas
20
+ Requires-Dist: pyyaml
21
+ Provides-Extra: judge
22
+ Requires-Dist: torch; extra == "judge"
23
+ Requires-Dist: transformers>=4.46; extra == "judge"
24
+ Provides-Extra: train
25
+ Requires-Dist: torch; extra == "train"
26
+ Requires-Dist: transformers>=4.46; extra == "train"
27
+ Requires-Dist: peft; extra == "train"
28
+ Requires-Dist: datasets; extra == "train"
29
+ Requires-Dist: accelerate; extra == "train"
30
+ Requires-Dist: scikit-learn; extra == "train"
31
+ Provides-Extra: test
32
+ Requires-Dist: pytest; extra == "test"
33
+ Requires-Dist: torch; extra == "test"
34
+ Requires-Dist: transformers>=4.46; extra == "test"
35
+ Requires-Dist: scikit-learn; extra == "test"
36
+ Dynamic: license-file
37
+
38
+ # pairjudge
39
+
40
+ **Train and serve pairwise LLM judges (A wins / B wins / tie) — with budget-aware multi-turn packing, position-bias correction, and pseudo-label distillation.**
41
+
42
+ [![CI](https://github.com/DaoyuanLi2816/pairjudge/actions/workflows/ci.yml/badge.svg)](https://github.com/DaoyuanLi2816/pairjudge/actions)
43
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)
44
+ ![Python](https://img.shields.io/badge/python-3.9%2B-blue)
45
+ [![Kaggle Gold](https://img.shields.io/badge/Kaggle-Gold%20%C2%B7%204th%20of%201849-FFD700)](https://www.kaggle.com/competitions/lmsys-chatbot-arena/leaderboard)
46
+
47
+ `pairjudge` is the generalized core of the **4th-place (gold medal) solution** to Kaggle's [LMSYS — Chatbot Arena Human Preference Predictions](https://www.kaggle.com/competitions/lmsys-chatbot-arena/overview) (1,849 teams), extracted into a small, tested library you can run on **your own preference data with any Hugging Face backbone**. The exact competition artifacts are preserved untouched in [`competition/`](competition/README.md), and a golden test pins the library's default behavior to the medal-winning code **byte for byte**.
48
+
49
+ Use it when you need a model that answers: *given a prompt and two candidate responses, which one would a human prefer — or is it a tie?* That model is the engine behind response reranking, A/B evaluation of fine-tunes, RLHF/RLAIF reward signals, and arena-style leaderboards.
50
+
51
+ ## Why not just an off-the-shelf reward model?
52
+
53
+ Three problems show up the moment you train a pairwise judge on real conversations, and they are exactly what this library packages:
54
+
55
+ **1. Truncation silently destroys the comparison.**
56
+ A judge input holds a multi-turn conversation *plus two responses per turn*. With naive left- or right-truncation, long inputs routinely lose response B (or the prompt) entirely — the judge then learns position artifacts instead of preferences. `PairPacker` packs rounds greedily and, when the budget runs out, truncates the final round *proportionally* (default 20% prompt / 40% response A / 40% response B), marks every cut with an explicit ellipsis, and drops rounds that can't be shown honestly. Guarantee: never exceeds `max_length`, and every retained round shows all three fields.
57
+
58
+ **2. Pairwise judges have position bias.**
59
+ Swap A and B and a naive judge changes its verdict on a measurable fraction of pairs. `PairwiseJudge.predict_proba(swap_debias=True)` scores each pair in both orders and averages in the original frame — order-invariant by construction. `position_flip_rate()` measures how biased your judge is before you decide to pay the 2x compute.
60
+
61
+ **3. Human preference labels are scarce and noisy.**
62
+ The medal recipe is a two-phase semi-supervised loop: train on human labels → pseudo-label a large unlabeled pool with **full probability distributions** → retrain with soft-label KL distillation (`label_mode: soft`). Ties are a first-class third category throughout — real human preference data is full of them, and scalar Bradley–Terry reward models (e.g. TRL's `RewardTrainer`, `num_labels=1`) cannot represent them.
63
+
64
+ ## Install
65
+
66
+ ```bash
67
+ pip install -e . # core: packing + data loaders (no torch needed)
68
+ pip install -e .[judge] # + inference (torch, transformers)
69
+ pip install -e .[train] # + LoRA fine-tuning (peft, datasets, accelerate)
70
+ ```
71
+
72
+ ## 60 seconds
73
+
74
+ ```python
75
+ from pairjudge import PairPacker, PackerConfig, from_pairs
76
+
77
+ # 1. Pack pairwise conversations into a token budget — any HF tokenizer.
78
+ from transformers import AutoTokenizer
79
+ tok = AutoTokenizer.from_pretrained("Qwen/Qwen2.5-0.5B-Instruct")
80
+ packer = PairPacker(tok, PackerConfig(max_length=2048))
81
+ packed = packer.pack(
82
+ prompts=["Explain quantum entanglement to a 10-year-old."],
83
+ responses_a=["Imagine two magic coins..."],
84
+ responses_b=["Quantum entanglement is a physical phenomenon..."],
85
+ )
86
+ packed.input_ids # <= 2048 tokens, prompt + BOTH responses guaranteed visible
87
+ packed.truncated # False — everything fit
88
+
89
+ # 2. Judge a pair with a trained model, position-bias-free.
90
+ from pairjudge import PairwiseJudge
91
+ judge = PairwiseJudge.from_pretrained("path/to/your/judge")
92
+ df = from_pairs(
93
+ prompts=["Explain quantum entanglement to a 10-year-old."],
94
+ responses_a=["Imagine two magic coins..."],
95
+ responses_b=["Quantum entanglement is a physical phenomenon..."],
96
+ )
97
+ judge.predict_proba(df, swap_debias=True) # [[p_a_wins, p_b_wins, p_tie]]
98
+ judge.position_flip_rate(df) # how order-sensitive is my judge?
99
+ ```
100
+
101
+ ## Train your own judge
102
+
103
+ ```bash
104
+ # Small judge on one consumer GPU (Qwen2.5-0.5B, ungated):
105
+ python -m pairjudge.training --cfg examples/configs/quickstart.yaml
106
+
107
+ # The competition setup (gemma-2-9b-it, 4x A100):
108
+ python -m pairjudge.training --cfg examples/configs/reproduce_competition.yaml
109
+ ```
110
+
111
+ Input is either an Arena-format CSV (the Kaggle competition schema) or a parquet with canonical columns — `prompt` / `response_a` / `response_b` as per-round string lists plus one-hot (or soft) `winner_*` columns. `pairjudge.data` ships loaders for Arena CSVs and UltraFeedback-style chosen/rejected data, plus `from_pairs()` for plain Python lists.
112
+
113
+ The full two-phase distillation loop:
114
+
115
+ ```bash
116
+ # Phase 1: train on human labels
117
+ python -m pairjudge.training --cfg phase1.yaml # label_mode: hard
118
+
119
+ # Pseudo-label an unlabeled pool with the phase-1 judge (soft labels)
120
+ python -m pairjudge.pseudo_label \
121
+ --model ./output/judge/merged \
122
+ --data pool.parquet --out pool_pl.parquet --swap-debias
123
+
124
+ # Phase 2: retrain from scratch on human + soft labels with KL loss
125
+ python -m pairjudge.training --cfg phase2.yaml # label_mode: soft
126
+ ```
127
+
128
+ In the competition, this loop (88k human-labeled + 30k pseudo-labeled UltraFeedback conversations) was a decisive part of the gap between a good model and a gold-medal one.
129
+
130
+ ## Inference guardrails
131
+
132
+ Two degenerate cases are worth handling outside the model — on competition data this was worth a measurable amount of log-loss:
133
+
134
+ ```python
135
+ from pairjudge import empty_and_identical_masks
136
+
137
+ a_empty, b_empty, identical = empty_and_identical_masks(raw_df)
138
+ proba[a_empty] = [0.04, 0.88, 0.08] # empty response loses — but never bet 1.0
139
+ proba[b_empty] = [0.88, 0.04, 0.08] # labels are noisy; log-loss punishes overconfidence
140
+ proba[identical] = [0.06, 0.06, 0.88] # identical responses are a tie
141
+ ```
142
+
143
+ ## How it relates to TRL's `RewardTrainer`
144
+
145
+ | | TRL `RewardTrainer` | `pairjudge` |
146
+ |---|---|---|
147
+ | Output | scalar reward (`num_labels=1`) | 3-class distribution (A / B / **tie**) |
148
+ | Loss | Bradley–Terry (logsigmoid of reward gap) | CE on human labels, KL on soft pseudo-labels |
149
+ | Ties | not representable | first-class |
150
+ | Multi-turn pair truncation | generic | proportional, all-fields-guaranteed |
151
+ | Position bias | n/a at inference (scores singletons) | swap-debias averaging + flip-rate diagnostic |
152
+
153
+ If you need a scalar reward for PPO-style RLHF, use TRL. If you need a *judge* that compares two concrete responses — for evaluation, reranking, data labeling, or arena prediction — and your data has ties, this is the recipe that placed 4th of 1,849 on exactly that task.
154
+
155
+ ## Provenance & validation
156
+
157
+ - The competition scripts, configs, inference notebook and certificate are preserved verbatim in [`competition/`](competition/README.md), including the full original write-up.
158
+ - `tests/test_packing.py::TestCompetitionEquivalence` fuzzes 1,500 conversations against a verbatim copy of the competition tokenizer ([`tests/reference_impl.py`](tests/reference_impl.py)) and asserts byte-identical output with default settings — the library *is* the medal-winning code, not a reimplementation of it.
159
+ - Final leaderboard: **4th / 1,849** ([gold medal](https://www.kaggle.com/certification/competitions/distiller/lmsys-chatbot-arena), $20,000 prize).
160
+
161
+ ## Citation
162
+
163
+ ```bibtex
164
+ @misc{li2024pairjudge,
165
+ author = {Daoyuan Li},
166
+ title = {pairjudge: pairwise LLM judges with budget-aware packing and position-bias correction},
167
+ year = {2024},
168
+ url = {https://github.com/DaoyuanLi2816/pairjudge},
169
+ note = {Generalized from the 4th-place solution, Kaggle LMSYS Chatbot Arena Human Preference Predictions}
170
+ }
171
+ ```
172
+
173
+ ## License
174
+
175
+ MIT — see [LICENSE](LICENSE).
176
+
177
+ ## Author
178
+
179
+ Daoyuan Li — [Kaggle (distiller)](https://www.kaggle.com/distiller) · lidaoyuan2816@gmail.com
@@ -0,0 +1,142 @@
1
+ # pairjudge
2
+
3
+ **Train and serve pairwise LLM judges (A wins / B wins / tie) — with budget-aware multi-turn packing, position-bias correction, and pseudo-label distillation.**
4
+
5
+ [![CI](https://github.com/DaoyuanLi2816/pairjudge/actions/workflows/ci.yml/badge.svg)](https://github.com/DaoyuanLi2816/pairjudge/actions)
6
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)
7
+ ![Python](https://img.shields.io/badge/python-3.9%2B-blue)
8
+ [![Kaggle Gold](https://img.shields.io/badge/Kaggle-Gold%20%C2%B7%204th%20of%201849-FFD700)](https://www.kaggle.com/competitions/lmsys-chatbot-arena/leaderboard)
9
+
10
+ `pairjudge` is the generalized core of the **4th-place (gold medal) solution** to Kaggle's [LMSYS — Chatbot Arena Human Preference Predictions](https://www.kaggle.com/competitions/lmsys-chatbot-arena/overview) (1,849 teams), extracted into a small, tested library you can run on **your own preference data with any Hugging Face backbone**. The exact competition artifacts are preserved untouched in [`competition/`](competition/README.md), and a golden test pins the library's default behavior to the medal-winning code **byte for byte**.
11
+
12
+ Use it when you need a model that answers: *given a prompt and two candidate responses, which one would a human prefer — or is it a tie?* That model is the engine behind response reranking, A/B evaluation of fine-tunes, RLHF/RLAIF reward signals, and arena-style leaderboards.
13
+
14
+ ## Why not just an off-the-shelf reward model?
15
+
16
+ Three problems show up the moment you train a pairwise judge on real conversations, and they are exactly what this library packages:
17
+
18
+ **1. Truncation silently destroys the comparison.**
19
+ A judge input holds a multi-turn conversation *plus two responses per turn*. With naive left- or right-truncation, long inputs routinely lose response B (or the prompt) entirely — the judge then learns position artifacts instead of preferences. `PairPacker` packs rounds greedily and, when the budget runs out, truncates the final round *proportionally* (default 20% prompt / 40% response A / 40% response B), marks every cut with an explicit ellipsis, and drops rounds that can't be shown honestly. Guarantee: never exceeds `max_length`, and every retained round shows all three fields.
20
+
21
+ **2. Pairwise judges have position bias.**
22
+ Swap A and B and a naive judge changes its verdict on a measurable fraction of pairs. `PairwiseJudge.predict_proba(swap_debias=True)` scores each pair in both orders and averages in the original frame — order-invariant by construction. `position_flip_rate()` measures how biased your judge is before you decide to pay the 2x compute.
23
+
24
+ **3. Human preference labels are scarce and noisy.**
25
+ The medal recipe is a two-phase semi-supervised loop: train on human labels → pseudo-label a large unlabeled pool with **full probability distributions** → retrain with soft-label KL distillation (`label_mode: soft`). Ties are a first-class third category throughout — real human preference data is full of them, and scalar Bradley–Terry reward models (e.g. TRL's `RewardTrainer`, `num_labels=1`) cannot represent them.
26
+
27
+ ## Install
28
+
29
+ ```bash
30
+ pip install -e . # core: packing + data loaders (no torch needed)
31
+ pip install -e .[judge] # + inference (torch, transformers)
32
+ pip install -e .[train] # + LoRA fine-tuning (peft, datasets, accelerate)
33
+ ```
34
+
35
+ ## 60 seconds
36
+
37
+ ```python
38
+ from pairjudge import PairPacker, PackerConfig, from_pairs
39
+
40
+ # 1. Pack pairwise conversations into a token budget — any HF tokenizer.
41
+ from transformers import AutoTokenizer
42
+ tok = AutoTokenizer.from_pretrained("Qwen/Qwen2.5-0.5B-Instruct")
43
+ packer = PairPacker(tok, PackerConfig(max_length=2048))
44
+ packed = packer.pack(
45
+ prompts=["Explain quantum entanglement to a 10-year-old."],
46
+ responses_a=["Imagine two magic coins..."],
47
+ responses_b=["Quantum entanglement is a physical phenomenon..."],
48
+ )
49
+ packed.input_ids # <= 2048 tokens, prompt + BOTH responses guaranteed visible
50
+ packed.truncated # False — everything fit
51
+
52
+ # 2. Judge a pair with a trained model, position-bias-free.
53
+ from pairjudge import PairwiseJudge
54
+ judge = PairwiseJudge.from_pretrained("path/to/your/judge")
55
+ df = from_pairs(
56
+ prompts=["Explain quantum entanglement to a 10-year-old."],
57
+ responses_a=["Imagine two magic coins..."],
58
+ responses_b=["Quantum entanglement is a physical phenomenon..."],
59
+ )
60
+ judge.predict_proba(df, swap_debias=True) # [[p_a_wins, p_b_wins, p_tie]]
61
+ judge.position_flip_rate(df) # how order-sensitive is my judge?
62
+ ```
63
+
64
+ ## Train your own judge
65
+
66
+ ```bash
67
+ # Small judge on one consumer GPU (Qwen2.5-0.5B, ungated):
68
+ python -m pairjudge.training --cfg examples/configs/quickstart.yaml
69
+
70
+ # The competition setup (gemma-2-9b-it, 4x A100):
71
+ python -m pairjudge.training --cfg examples/configs/reproduce_competition.yaml
72
+ ```
73
+
74
+ Input is either an Arena-format CSV (the Kaggle competition schema) or a parquet with canonical columns — `prompt` / `response_a` / `response_b` as per-round string lists plus one-hot (or soft) `winner_*` columns. `pairjudge.data` ships loaders for Arena CSVs and UltraFeedback-style chosen/rejected data, plus `from_pairs()` for plain Python lists.
75
+
76
+ The full two-phase distillation loop:
77
+
78
+ ```bash
79
+ # Phase 1: train on human labels
80
+ python -m pairjudge.training --cfg phase1.yaml # label_mode: hard
81
+
82
+ # Pseudo-label an unlabeled pool with the phase-1 judge (soft labels)
83
+ python -m pairjudge.pseudo_label \
84
+ --model ./output/judge/merged \
85
+ --data pool.parquet --out pool_pl.parquet --swap-debias
86
+
87
+ # Phase 2: retrain from scratch on human + soft labels with KL loss
88
+ python -m pairjudge.training --cfg phase2.yaml # label_mode: soft
89
+ ```
90
+
91
+ In the competition, this loop (88k human-labeled + 30k pseudo-labeled UltraFeedback conversations) was a decisive part of the gap between a good model and a gold-medal one.
92
+
93
+ ## Inference guardrails
94
+
95
+ Two degenerate cases are worth handling outside the model — on competition data this was worth a measurable amount of log-loss:
96
+
97
+ ```python
98
+ from pairjudge import empty_and_identical_masks
99
+
100
+ a_empty, b_empty, identical = empty_and_identical_masks(raw_df)
101
+ proba[a_empty] = [0.04, 0.88, 0.08] # empty response loses — but never bet 1.0
102
+ proba[b_empty] = [0.88, 0.04, 0.08] # labels are noisy; log-loss punishes overconfidence
103
+ proba[identical] = [0.06, 0.06, 0.88] # identical responses are a tie
104
+ ```
105
+
106
+ ## How it relates to TRL's `RewardTrainer`
107
+
108
+ | | TRL `RewardTrainer` | `pairjudge` |
109
+ |---|---|---|
110
+ | Output | scalar reward (`num_labels=1`) | 3-class distribution (A / B / **tie**) |
111
+ | Loss | Bradley–Terry (logsigmoid of reward gap) | CE on human labels, KL on soft pseudo-labels |
112
+ | Ties | not representable | first-class |
113
+ | Multi-turn pair truncation | generic | proportional, all-fields-guaranteed |
114
+ | Position bias | n/a at inference (scores singletons) | swap-debias averaging + flip-rate diagnostic |
115
+
116
+ If you need a scalar reward for PPO-style RLHF, use TRL. If you need a *judge* that compares two concrete responses — for evaluation, reranking, data labeling, or arena prediction — and your data has ties, this is the recipe that placed 4th of 1,849 on exactly that task.
117
+
118
+ ## Provenance & validation
119
+
120
+ - The competition scripts, configs, inference notebook and certificate are preserved verbatim in [`competition/`](competition/README.md), including the full original write-up.
121
+ - `tests/test_packing.py::TestCompetitionEquivalence` fuzzes 1,500 conversations against a verbatim copy of the competition tokenizer ([`tests/reference_impl.py`](tests/reference_impl.py)) and asserts byte-identical output with default settings — the library *is* the medal-winning code, not a reimplementation of it.
122
+ - Final leaderboard: **4th / 1,849** ([gold medal](https://www.kaggle.com/certification/competitions/distiller/lmsys-chatbot-arena), $20,000 prize).
123
+
124
+ ## Citation
125
+
126
+ ```bibtex
127
+ @misc{li2024pairjudge,
128
+ author = {Daoyuan Li},
129
+ title = {pairjudge: pairwise LLM judges with budget-aware packing and position-bias correction},
130
+ year = {2024},
131
+ url = {https://github.com/DaoyuanLi2816/pairjudge},
132
+ note = {Generalized from the 4th-place solution, Kaggle LMSYS Chatbot Arena Human Preference Predictions}
133
+ }
134
+ ```
135
+
136
+ ## License
137
+
138
+ MIT — see [LICENSE](LICENSE).
139
+
140
+ ## Author
141
+
142
+ Daoyuan Li — [Kaggle (distiller)](https://www.kaggle.com/distiller) · lidaoyuan2816@gmail.com
@@ -0,0 +1,55 @@
1
+ [build-system]
2
+ requires = ["setuptools>=64"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "pairjudge"
7
+ version = "0.1.0"
8
+ description = "Train and serve pairwise LLM judges (A/B/tie) with budget-aware multi-turn packing and position-bias correction"
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = { text = "MIT" }
12
+ authors = [{ name = "Daoyuan Li", email = "lidaoyuan2816@gmail.com" }]
13
+ keywords = [
14
+ "llm-as-judge",
15
+ "reward-model",
16
+ "preference-learning",
17
+ "rlhf",
18
+ "chatbot-arena",
19
+ ]
20
+ classifiers = [
21
+ "Development Status :: 4 - Beta",
22
+ "Intended Audience :: Science/Research",
23
+ "License :: OSI Approved :: MIT License",
24
+ "Programming Language :: Python :: 3",
25
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
26
+ ]
27
+ dependencies = [
28
+ "numpy",
29
+ "pandas",
30
+ "pyyaml",
31
+ ]
32
+
33
+ [project.optional-dependencies]
34
+ # Inference: load a trained judge and predict preferences.
35
+ judge = ["torch", "transformers>=4.46"]
36
+ # Training: fine-tune a judge with LoRA on your own preference data.
37
+ train = [
38
+ "torch",
39
+ "transformers>=4.46",
40
+ "peft",
41
+ "datasets",
42
+ "accelerate",
43
+ "scikit-learn",
44
+ ]
45
+ test = ["pytest", "torch", "transformers>=4.46", "scikit-learn"]
46
+
47
+ [project.urls]
48
+ Homepage = "https://github.com/DaoyuanLi2816/pairjudge"
49
+ Issues = "https://github.com/DaoyuanLi2816/pairjudge/issues"
50
+
51
+ [tool.setuptools.packages.find]
52
+ where = ["src"]
53
+
54
+ [tool.pytest.ini_options]
55
+ testpaths = ["tests"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,53 @@
1
+ """pairjudge — train and serve pairwise LLM judges (A/B/tie).
2
+
3
+ Extracted and generalized from the 4th-place (gold medal, 4/1849) solution
4
+ to the Kaggle competition "LMSYS — Chatbot Arena Human Preference
5
+ Predictions". The packing defaults are byte-for-byte equivalent to the
6
+ competition tokenization (golden-tested).
7
+
8
+ Core pieces:
9
+
10
+ - :class:`pairjudge.PairPacker` — budget-aware packing of multi-turn
11
+ (prompt, response A, response B) conversations. No heavy dependencies.
12
+ - :class:`pairjudge.PairwiseJudge` — inference with optional swap-based
13
+ position-bias correction (requires ``pairjudge[judge]``).
14
+ - :mod:`pairjudge.training` — LoRA fine-tuning with hard (CE) or soft (KL
15
+ distillation) labels (requires ``pairjudge[train]``).
16
+ - :mod:`pairjudge.data` — loaders normalizing Arena/UltraFeedback-style data
17
+ to one canonical schema.
18
+ """
19
+
20
+ from .data import (
21
+ EMPTY_RESPONSE_PATTERNS,
22
+ empty_and_identical_masks,
23
+ from_pairs,
24
+ load_arena_csv,
25
+ load_ultrafeedback,
26
+ )
27
+ from .packing import PackedExample, PackerConfig, PairPacker, hard_label
28
+
29
+ __version__ = "0.1.0"
30
+
31
+ __all__ = [
32
+ "PairPacker",
33
+ "PackerConfig",
34
+ "PackedExample",
35
+ "hard_label",
36
+ "load_arena_csv",
37
+ "load_ultrafeedback",
38
+ "from_pairs",
39
+ "empty_and_identical_masks",
40
+ "EMPTY_RESPONSE_PATTERNS",
41
+ "PairwiseJudge",
42
+ "swap_average",
43
+ "__version__",
44
+ ]
45
+
46
+
47
+ def __getattr__(name):
48
+ # Lazy import: PairwiseJudge needs torch/transformers, which are optional.
49
+ if name in ("PairwiseJudge", "swap_average"):
50
+ from . import judge
51
+
52
+ return getattr(judge, name)
53
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
@@ -0,0 +1,166 @@
1
+ """Loaders that normalize preference data into one canonical schema.
2
+
3
+ Canonical columns (one row = one conversation):
4
+
5
+ - ``id``: str
6
+ - ``prompt``: list[str] — user prompt per round
7
+ - ``response_a`` / ``response_b``: list[str] — candidate responses per round
8
+ - ``winner_model_a`` / ``winner_model_b`` / ``winner_tie``: float — one-hot
9
+ for human labels, arbitrary distribution for pseudo-labels
10
+
11
+ Everything downstream (packing, training, judging) consumes this schema, so
12
+ supporting a new dataset means writing one loader function.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import json
18
+ import random
19
+ from typing import Iterable, Optional, Sequence
20
+
21
+ import pandas as pd
22
+
23
+ #: String encodings of an empty response as they appear in Chatbot Arena
24
+ #: exports. A response equal to any of these is "empty" for labeling and
25
+ #: guardrail purposes.
26
+ EMPTY_RESPONSE_PATTERNS = ('[null]', '[]', '[ ]', '[ ]', '[""]', '["",""]')
27
+
28
+ _LIST_COLUMNS = ("prompt", "response_a", "response_b")
29
+ _WINNER_COLUMNS = ("winner_model_a", "winner_model_b", "winner_tie")
30
+
31
+
32
+ def _is_empty(series: pd.Series) -> pd.Series:
33
+ return series.isin(EMPTY_RESPONSE_PATTERNS)
34
+
35
+
36
+ def load_arena_csv(
37
+ path_or_df,
38
+ drop_identical: bool = True,
39
+ relabel_empty: bool = True,
40
+ ) -> pd.DataFrame:
41
+ """Load a Chatbot-Arena-style CSV (LMSYS competition format).
42
+
43
+ Expects JSON-encoded list columns ``prompt``/``response_a``/``response_b``
44
+ and one-hot winner columns.
45
+
46
+ Cleaning rules (from the gold-medal solution):
47
+
48
+ - rows where *both* responses are empty are dropped (no signal);
49
+ - rows where the two responses are byte-identical are dropped when
50
+ ``drop_identical`` — the label is noise there, and inference handles
51
+ that case with a guardrail instead (:func:`empty_and_identical_masks`);
52
+ - rows where exactly one response is empty are relabeled so the non-empty
53
+ side wins when ``relabel_empty`` — annotators almost always prefer *any*
54
+ answer over a blank one, and the few contrary labels are noise.
55
+ """
56
+ df = path_or_df if isinstance(path_or_df, pd.DataFrame) else pd.read_csv(path_or_df, encoding="utf-8")
57
+ df = df.copy()
58
+
59
+ a_empty, b_empty = _is_empty(df["response_a"]), _is_empty(df["response_b"])
60
+ df = df[~(a_empty & b_empty)]
61
+ if drop_identical:
62
+ df = df[~(df["response_a"] == df["response_b"])]
63
+
64
+ if relabel_empty:
65
+ a_empty, b_empty = _is_empty(df["response_a"]), _is_empty(df["response_b"])
66
+ df.loc[a_empty, list(_WINNER_COLUMNS)] = [0.0, 1.0, 0.0]
67
+ df.loc[b_empty, list(_WINNER_COLUMNS)] = [1.0, 0.0, 0.0]
68
+
69
+ for col in _LIST_COLUMNS:
70
+ df[col] = df[col].apply(json.loads)
71
+
72
+ df = df.reset_index(drop=True)
73
+ df["id"] = df["id"].astype(str)
74
+ return df
75
+
76
+
77
+ def load_ultrafeedback(
78
+ path_or_df,
79
+ seed: int = 42,
80
+ dedup_by_prompt: bool = True,
81
+ ) -> pd.DataFrame:
82
+ """Convert an UltraFeedback-style chosen/rejected dataset to the canonical schema.
83
+
84
+ Each record has ``prompt`` (str) and ``chosen``/``rejected`` conversations
85
+ (list of ``{"role", "content"}`` messages, assistant reply at index 1).
86
+ The chosen/rejected pair is assigned to A/B *uniformly at random* (seeded)
87
+ so the resulting dataset is free of position bias by construction.
88
+ """
89
+ df = path_or_df if isinstance(path_or_df, pd.DataFrame) else pd.read_parquet(path_or_df)
90
+ rng = random.Random(seed)
91
+
92
+ records = []
93
+ for _, row in df.iterrows():
94
+ chosen = [row["chosen"][1]["content"]]
95
+ rejected = [row["rejected"][1]["content"]]
96
+ if rng.random() > 0.5:
97
+ response_a, response_b, winner = chosen, rejected, "a"
98
+ else:
99
+ response_a, response_b, winner = rejected, chosen, "b"
100
+ records.append(
101
+ {
102
+ "prompt": [row["prompt"]],
103
+ "response_a": response_a,
104
+ "response_b": response_b,
105
+ "winner_model_a": 1.0 if winner == "a" else 0.0,
106
+ "winner_model_b": 1.0 if winner == "b" else 0.0,
107
+ "winner_tie": 0.0,
108
+ }
109
+ )
110
+
111
+ out = pd.DataFrame(records)
112
+ if dedup_by_prompt:
113
+ out["_key"] = out["prompt"].apply(lambda x: x[0])
114
+ out = out.drop_duplicates(subset=["_key"], ignore_index=True)
115
+ out = out.drop(columns=["_key"])
116
+ out["id"] = out.index.astype(str)
117
+ return out
118
+
119
+
120
+ def from_pairs(
121
+ prompts: Sequence[str],
122
+ responses_a: Sequence[str],
123
+ responses_b: Sequence[str],
124
+ winners: Optional[Iterable[str]] = None,
125
+ ) -> pd.DataFrame:
126
+ """Build a canonical dataframe from flat single-turn pairs.
127
+
128
+ ``winners`` entries are ``"a"``, ``"b"`` or ``"tie"``; omit for unlabeled
129
+ data (e.g. inference or pseudo-labeling inputs).
130
+ """
131
+ df = pd.DataFrame(
132
+ {
133
+ "prompt": [[p] for p in prompts],
134
+ "response_a": [[r] for r in responses_a],
135
+ "response_b": [[r] for r in responses_b],
136
+ }
137
+ )
138
+ if winners is not None:
139
+ winners = list(winners)
140
+ bad = sorted({w for w in winners} - {"a", "b", "tie"})
141
+ if bad:
142
+ raise ValueError(f"winners must be 'a', 'b' or 'tie', got {bad}")
143
+ df["winner_model_a"] = [1.0 if w == "a" else 0.0 for w in winners]
144
+ df["winner_model_b"] = [1.0 if w == "b" else 0.0 for w in winners]
145
+ df["winner_tie"] = [1.0 if w == "tie" else 0.0 for w in winners]
146
+ df["id"] = df.index.astype(str)
147
+ return df
148
+
149
+
150
+ def empty_and_identical_masks(df: pd.DataFrame):
151
+ """Inference guardrail masks for degenerate pairs.
152
+
153
+ Returns boolean Series ``(a_empty, b_empty, identical)`` over raw (still
154
+ JSON-encoded) response columns. A judge should not be trusted on these
155
+ rows: an empty response loses against a non-empty one, and identical
156
+ responses are a tie. Overriding the model's prediction with fixed,
157
+ *calibrated* probabilities (not 0/1 — labels are noisy and log-loss
158
+ punishes overconfidence) is worth a measurable amount of log-loss; the
159
+ gold-medal solution used ``[0.04, 0.88, 0.08]`` for empty-vs-non-empty
160
+ and ``[0.06, 0.06, 0.88]`` for identical pairs.
161
+ """
162
+ return (
163
+ _is_empty(df["response_a"]),
164
+ _is_empty(df["response_b"]),
165
+ df["response_a"] == df["response_b"],
166
+ )