modeljury 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- modeljury-0.1.0/.github/workflows/publish.yml +36 -0
- modeljury-0.1.0/.gitignore +7 -0
- modeljury-0.1.0/LICENSE +21 -0
- modeljury-0.1.0/PKG-INFO +271 -0
- modeljury-0.1.0/README.md +243 -0
- modeljury-0.1.0/examples/refund.py +33 -0
- modeljury-0.1.0/pyproject.toml +40 -0
- modeljury-0.1.0/src/modeljury/__init__.py +4 -0
- modeljury-0.1.0/src/modeljury/claude.py +35 -0
- modeljury-0.1.0/src/modeljury/jury.py +248 -0
- modeljury-0.1.0/tests/test_jury.py +248 -0
- modeljury-0.1.0/uv.lock +584 -0
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
# Publishes to PyPI when a GitHub release is published.
|
|
2
|
+
# PyPI trusts this workflow through trusted publishing, so no token is stored.
|
|
3
|
+
name: Publish to PyPI
|
|
4
|
+
|
|
5
|
+
on:
|
|
6
|
+
release:
|
|
7
|
+
types: [published]
|
|
8
|
+
|
|
9
|
+
jobs:
|
|
10
|
+
build:
|
|
11
|
+
runs-on: ubuntu-latest
|
|
12
|
+
steps:
|
|
13
|
+
- uses: actions/checkout@v7
|
|
14
|
+
- uses: astral-sh/setup-uv@v10.2.0
|
|
15
|
+
- run: uv sync --all-extras
|
|
16
|
+
- run: uv run pytest -q
|
|
17
|
+
- run: uv build
|
|
18
|
+
- uses: actions/upload-artifact@v7
|
|
19
|
+
with:
|
|
20
|
+
name: dist
|
|
21
|
+
path: dist/
|
|
22
|
+
|
|
23
|
+
publish:
|
|
24
|
+
needs: build
|
|
25
|
+
runs-on: ubuntu-latest
|
|
26
|
+
environment:
|
|
27
|
+
name: pypi
|
|
28
|
+
url: https://pypi.org/p/modeljury
|
|
29
|
+
permissions:
|
|
30
|
+
id-token: write
|
|
31
|
+
steps:
|
|
32
|
+
- uses: actions/download-artifact@v8
|
|
33
|
+
with:
|
|
34
|
+
name: dist
|
|
35
|
+
path: dist/
|
|
36
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
modeljury-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 modeljury contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
modeljury-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,271 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: modeljury
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Ask several models instead of one. Ship when they agree, send to a human when they don't.
|
|
5
|
+
Project-URL: Homepage, https://github.com/YasserManss/modeljury
|
|
6
|
+
Project-URL: Repository, https://github.com/YasserManss/modeljury
|
|
7
|
+
Project-URL: Issues, https://github.com/YasserManss/modeljury/issues
|
|
8
|
+
Author: YasserManss
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: ensemble,evaluation,human-in-the-loop,llm,llm-as-a-judge,voting
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
22
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
23
|
+
Requires-Python: >=3.10
|
|
24
|
+
Requires-Dist: openai>=1.0
|
|
25
|
+
Provides-Extra: claude
|
|
26
|
+
Requires-Dist: anthropic>=1.11.0; extra == 'claude'
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
|
|
29
|
+
# modeljury
|
|
30
|
+
|
|
31
|
+
Ask several models instead of one. Ship when they agree, send to a human when they don't.
|
|
32
|
+
|
|
33
|
+
## Getting started
|
|
34
|
+
|
|
35
|
+
Python 3.10 or newer.
|
|
36
|
+
|
|
37
|
+
```sh
|
|
38
|
+
pip install modeljury
|
|
39
|
+
# with Claude support
|
|
40
|
+
pip install "modeljury[claude]"
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
Point it at any OpenAI-compatible endpoint (OpenAI, Ollama, vLLM, OpenRouter, ...) and
|
|
44
|
+
convene a panel. This one uses three models served by a local Ollama:
|
|
45
|
+
|
|
46
|
+
```python
|
|
47
|
+
from modeljury import Juror, convene
|
|
48
|
+
|
|
49
|
+
local = "http://localhost:11434/v1"
|
|
50
|
+
result = convene(
|
|
51
|
+
question="Is this review spam?",
|
|
52
|
+
evidence="'Best product ever!!! Visit cheap-deals.example for 90% off!!!'",
|
|
53
|
+
options=["spam", "not spam"],
|
|
54
|
+
jurors=[Juror(m, base_url=local) for m in ["llama3.1", "qwen3", "mistral"]],
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
print(result.verdict, result.needs_review)
|
|
58
|
+
for juror, choice, reason in result.dissent:
|
|
59
|
+
print(f"{juror} disagreed: {choice}, {reason}")
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
For hosted models, set `OPENAI_API_KEY` (and `OPENAI_BASE_URL` for non-OpenAI providers) or
|
|
63
|
+
pass `api_key=` and `base_url=` to each `Juror`. For Claude, set `ANTHROPIC_API_KEY`.
|
|
64
|
+
|
|
65
|
+
## Usage
|
|
66
|
+
|
|
67
|
+
```python
|
|
68
|
+
from modeljury import Juror, convene
|
|
69
|
+
|
|
70
|
+
result = convene(
|
|
71
|
+
question="Is this refund request fraudulent?",
|
|
72
|
+
evidence=ticket_text,
|
|
73
|
+
options=["fraud", "legit"],
|
|
74
|
+
jurors=[
|
|
75
|
+
Juror("gpt-4o-mini"),
|
|
76
|
+
Juror("llama3.1", base_url="http://localhost:11434/v1"),
|
|
77
|
+
Juror("mistral-small", base_url="https://api.mistral.ai/v1", api_key="..."),
|
|
78
|
+
],
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
result.verdict # "legit", or None on a hung jury
|
|
82
|
+
result.majority # True: the verdict won more than half of the whole panel
|
|
83
|
+
result.needs_review # True
|
|
84
|
+
result.dissent # [("mistral-small", "fraud", "Shipping address changed after purchase")]
|
|
85
|
+
result.failed # jurors that errored or gave an invalid answer
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Any endpoint that speaks the OpenAI chat completions API works. A plain string juror
|
|
89
|
+
(`"gpt-4o-mini"`) uses `OPENAI_BASE_URL` and `OPENAI_API_KEY`.
|
|
90
|
+
|
|
91
|
+
## Examples
|
|
92
|
+
|
|
93
|
+
These are real runs: five copies of one local 27B model, sampled at `temperature=1.0`.
|
|
94
|
+
The evidence is slightly trimmed here, and the quoted reasons are the model's own words.
|
|
95
|
+
A mixed panel of different models would disagree more often. See the warning at the end.
|
|
96
|
+
|
|
97
|
+
### Refund with mixed signals
|
|
98
|
+
|
|
99
|
+
```python
|
|
100
|
+
convene(
|
|
101
|
+
question="Should this refund be approved?",
|
|
102
|
+
evidence="Customer of 6 years, 40 prior orders, 2 prior refunds (both legitimate). "
|
|
103
|
+
"Claims a $1,900 laptop arrived with a cracked screen. The photo's EXIF date is "
|
|
104
|
+
"2 days before delivery. Courier marked it 'left at door', no signature. "
|
|
105
|
+
"Customer says their phone clock is wrong and offers to send the laptop back.",
|
|
106
|
+
options=["approve", "deny", "escalate"],
|
|
107
|
+
jurors=jurors,
|
|
108
|
+
)
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
**escalate, 4–1, needs review.** The majority wanted the EXIF date checked. The dissent
|
|
112
|
+
made the case for approving:
|
|
113
|
+
|
|
114
|
+
> ✗ approve: *Strong 6-year history with only 2 prior refunds outweighs the EXIF
|
|
115
|
+
> inconsistency, which is plausibly explained by the customer's claim of a misconfigured
|
|
116
|
+
> phone clock.*
|
|
117
|
+
|
|
118
|
+
### Bug triage across four teams
|
|
119
|
+
|
|
120
|
+
```python
|
|
121
|
+
convene(
|
|
122
|
+
question="Which team should own this bug report?",
|
|
123
|
+
evidence="'After upgrading to v4.2, CSV exports from the dashboard are missing every "
|
|
124
|
+
"row created after 11pm. Only affects customers in Australia. Started the same day we "
|
|
125
|
+
"moved the export job to a new server.'",
|
|
126
|
+
options=["Frontend", "Data", "Infra", "Billing"],
|
|
127
|
+
jurors=jurors,
|
|
128
|
+
)
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
**Data, 4–1, needs review.** Four jurors blamed timezone logic in the export pipeline.
|
|
132
|
+
The dissent was the only one that used the strongest clue:
|
|
133
|
+
|
|
134
|
+
> ✗ Infra: *The bug started immediately after moving the export job to a new server,
|
|
135
|
+
> and the time-based filtering error likely stems from a timezone configuration issue
|
|
136
|
+
> on that new infrastructure.*
|
|
137
|
+
|
|
138
|
+
A reviewer who reads only the dissent knows where to look first.
|
|
139
|
+
|
|
140
|
+
### Merging a fast but risky cache
|
|
141
|
+
|
|
142
|
+
```python
|
|
143
|
+
convene(
|
|
144
|
+
question="Should this pull request be merged as is?",
|
|
145
|
+
evidence="Adds a cache in front of the user-permissions lookup: p99 latency drops "
|
|
146
|
+
"from 900ms to 40ms. TTL is 10 minutes. Tests pass. A revoked admin keeps admin "
|
|
147
|
+
"rights for up to 10 minutes. Revocations happen about 3 times a month, and last "
|
|
148
|
+
"week's outage was caused by the slow lookup.",
|
|
149
|
+
options=["merge", "request changes", "merge with follow-up ticket"],
|
|
150
|
+
jurors=jurors,
|
|
151
|
+
)
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
**merge with follow-up ticket, 5–0, ships.** All five weighed the latency gain against
|
|
155
|
+
the 10-minute window and asked for a ticket to add cache invalidation on revocation.
|
|
156
|
+
|
|
157
|
+
### Satire or threat
|
|
158
|
+
|
|
159
|
+
```python
|
|
160
|
+
convene(
|
|
161
|
+
question="Does this post violate the policy against threats of violence?",
|
|
162
|
+
evidence="Policy: remove credible threats of violence. Satire and hyperbole are allowed. "
|
|
163
|
+
"Post, replying to a council member's parking fee increase: 'If they raise parking "
|
|
164
|
+
"again I swear I'll show up at the next meeting and make sure he never votes on "
|
|
165
|
+
"anything again 🙂'. Author has no prior violations and often jokes about local politics.",
|
|
166
|
+
options=["violates", "allowed"],
|
|
167
|
+
jurors=jurors,
|
|
168
|
+
)
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
**allowed, 5–0, ships.** This is the warning. The post names a target and a place, and
|
|
172
|
+
a human moderator might well escalate it. All five jurors dismissed it for the same
|
|
173
|
+
reason, mostly the smiley. Copies of one model share the same blind spots, so their
|
|
174
|
+
agreement is weaker evidence than it looks. Use jurors from different model families,
|
|
175
|
+
because a unanimous vote only means something if the jurors could have disagreed.
|
|
176
|
+
|
|
177
|
+
## Model options
|
|
178
|
+
|
|
179
|
+
Extra keyword arguments to `convene()` go to every juror's `chat.completions.create` call.
|
|
180
|
+
A juror's own `params` override them. `temperature` defaults to `0`.
|
|
181
|
+
|
|
182
|
+
```python
|
|
183
|
+
convene(..., temperature=0.3, max_tokens=200, jurors=[
|
|
184
|
+
Juror("gpt-4o-mini"), # temperature=0.3, max_tokens=200
|
|
185
|
+
Juror("o4-mini", params={"temperature": None, # None drops a param
|
|
186
|
+
"reasoning_effort": "low"}), # adds a model-specific option
|
|
187
|
+
Juror("qwen3", base_url="http://localhost:8000/v1",
|
|
188
|
+
params={"temperature": 0.7, "extra_body": {"top_k": 20}}), # overrides temperature
|
|
189
|
+
])
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
## Claude
|
|
193
|
+
|
|
194
|
+
Claude uses the native Anthropic Messages API. Install the extra with `pip install modeljury[claude]`.
|
|
195
|
+
|
|
196
|
+
```python
|
|
197
|
+
from modeljury import Claude
|
|
198
|
+
|
|
199
|
+
jurors = [
|
|
200
|
+
"gpt-4o-mini",
|
|
201
|
+
"claude-sonnet-5-5", # any "claude-..." string works
|
|
202
|
+
Claude("claude-opus-5-5", params={"output_config": {"effort": "low"}}),
|
|
203
|
+
]
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
- Credentials come from `ANTHROPIC_API_KEY` or an `ant auth login` profile, unless you pass `api_key=`.
|
|
207
|
+
- Panel-wide kwargs are OpenAI parameters, so Claude jurors don't receive them. Put Messages API
|
|
208
|
+
options in `params` instead. Current Claude models reject `temperature`, so none is sent.
|
|
209
|
+
- `max_tokens` defaults to 16000. Set it lower with `params={"max_tokens": ...}` if you need a cost cap.
|
|
210
|
+
Thinking counts toward it.
|
|
211
|
+
- If Claude refuses, the vote is recorded in `result.failed` and the decision goes to review.
|
|
212
|
+
|
|
213
|
+
## Custom prompt
|
|
214
|
+
|
|
215
|
+
`prompt=` takes a `str.format` template with `{question}`, `{evidence}` and `{options}`.
|
|
216
|
+
Double any literal braces (`{{` and `}}`). The default is `modeljury.PROMPT`. Your template
|
|
217
|
+
must still ask for `{"choice": ..., "confidence": ..., "reason": ...}` JSON, because that's what the parser reads.
|
|
218
|
+
|
|
219
|
+
## Rules
|
|
220
|
+
|
|
221
|
+
1. Each juror picks one allowed option and gives a one-line reason.
|
|
222
|
+
2. The verdict is the option with the most votes.
|
|
223
|
+
3. `needs_review` is true unless every juror returned the same valid choice.
|
|
224
|
+
4. A tie is a hung jury: no verdict, review needed.
|
|
225
|
+
|
|
226
|
+
`agreement` is the winning option's votes divided by valid votes. A juror that errors, refuses or
|
|
227
|
+
answers with an option that isn't allowed is left out of the count and listed in `result.failed`.
|
|
228
|
+
|
|
229
|
+
## Outcomes
|
|
230
|
+
|
|
231
|
+
Three jurors, options `fraud` / `legit`:
|
|
232
|
+
|
|
233
|
+
| Votes | `verdict` | `agreement` | `majority` | `needs_review` | Why |
|
|
234
|
+
|---|---|---|---|---|---|
|
|
235
|
+
| legit, legit, legit | legit | 100% | ✅ | ❌ | Everyone agrees: ships without review |
|
|
236
|
+
| legit, legit, fraud | legit | 67% | ✅ | ✅ | Dissent; the reason is in `dissent` |
|
|
237
|
+
| legit, fraud, *failed* | `None` | 50% | ❌ | ✅ | Tie, so hung jury |
|
|
238
|
+
| legit, legit, *failed* | legit | 100% | ✅ | ✅ | A juror failed |
|
|
239
|
+
| legit, *failed*, *failed* | legit | 100% | ❌ | ✅ | Only one valid vote out of three |
|
|
240
|
+
| *failed* ×3 | `None` | 0% | ❌ | ✅ | No valid votes |
|
|
241
|
+
|
|
242
|
+
`verdict` is the plurality winner, so with three or more options it can win without a majority:
|
|
243
|
+
five votes split A, A, B, C, D give verdict A with 40% agreement and `majority` false.
|
|
244
|
+
`majority` counts the whole panel, failed jurors included, so it means more than half of
|
|
245
|
+
all jurors chose the verdict.
|
|
246
|
+
|
|
247
|
+
`needs_review` is the strict rule. For a looser one, check `majority` or `agreement` yourself:
|
|
248
|
+
|
|
249
|
+
```python
|
|
250
|
+
if not result.majority: # ship any decision that more than half the panel backs
|
|
251
|
+
send_to_human(result)
|
|
252
|
+
```
|
|
253
|
+
|
|
254
|
+
A single juror can't dissent, so its decisions only go to review when it fails. `convene()` warns if you pass only one juror.
|
|
255
|
+
|
|
256
|
+
## Notes
|
|
257
|
+
|
|
258
|
+
- If a model name is wrong, the juror's error in `result.failed` lists the models the endpoint serves, or suggests the closest names when there are more than 20.
|
|
259
|
+
- Jurors with the same name (e.g. `llama3.1` on two endpoints) are recorded as `llama3.1`, `llama3.1#2`, ...
|
|
260
|
+
- `timeout` applies to each attempt. The OpenAI and Anthropic SDKs retry failed calls twice, so one juror can take up to three times `timeout`.
|
|
261
|
+
|
|
262
|
+
## Develop
|
|
263
|
+
|
|
264
|
+
```sh
|
|
265
|
+
uv sync
|
|
266
|
+
uv run pytest
|
|
267
|
+
```
|
|
268
|
+
|
|
269
|
+
## License
|
|
270
|
+
|
|
271
|
+
MIT. See [LICENSE](https://github.com/YasserManss/modeljury/blob/main/LICENSE).
|
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
# modeljury
|
|
2
|
+
|
|
3
|
+
Ask several models instead of one. Ship when they agree, send to a human when they don't.
|
|
4
|
+
|
|
5
|
+
## Getting started
|
|
6
|
+
|
|
7
|
+
Python 3.10 or newer.
|
|
8
|
+
|
|
9
|
+
```sh
|
|
10
|
+
pip install modeljury
|
|
11
|
+
# with Claude support
|
|
12
|
+
pip install "modeljury[claude]"
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
Point it at any OpenAI-compatible endpoint (OpenAI, Ollama, vLLM, OpenRouter, ...) and
|
|
16
|
+
convene a panel. This one uses three models served by a local Ollama:
|
|
17
|
+
|
|
18
|
+
```python
|
|
19
|
+
from modeljury import Juror, convene
|
|
20
|
+
|
|
21
|
+
local = "http://localhost:11434/v1"
|
|
22
|
+
result = convene(
|
|
23
|
+
question="Is this review spam?",
|
|
24
|
+
evidence="'Best product ever!!! Visit cheap-deals.example for 90% off!!!'",
|
|
25
|
+
options=["spam", "not spam"],
|
|
26
|
+
jurors=[Juror(m, base_url=local) for m in ["llama3.1", "qwen3", "mistral"]],
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
print(result.verdict, result.needs_review)
|
|
30
|
+
for juror, choice, reason in result.dissent:
|
|
31
|
+
print(f"{juror} disagreed: {choice}, {reason}")
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
For hosted models, set `OPENAI_API_KEY` (and `OPENAI_BASE_URL` for non-OpenAI providers) or
|
|
35
|
+
pass `api_key=` and `base_url=` to each `Juror`. For Claude, set `ANTHROPIC_API_KEY`.
|
|
36
|
+
|
|
37
|
+
## Usage
|
|
38
|
+
|
|
39
|
+
```python
|
|
40
|
+
from modeljury import Juror, convene
|
|
41
|
+
|
|
42
|
+
result = convene(
|
|
43
|
+
question="Is this refund request fraudulent?",
|
|
44
|
+
evidence=ticket_text,
|
|
45
|
+
options=["fraud", "legit"],
|
|
46
|
+
jurors=[
|
|
47
|
+
Juror("gpt-4o-mini"),
|
|
48
|
+
Juror("llama3.1", base_url="http://localhost:11434/v1"),
|
|
49
|
+
Juror("mistral-small", base_url="https://api.mistral.ai/v1", api_key="..."),
|
|
50
|
+
],
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
result.verdict # "legit", or None on a hung jury
|
|
54
|
+
result.majority # True: the verdict won more than half of the whole panel
|
|
55
|
+
result.needs_review # True
|
|
56
|
+
result.dissent # [("mistral-small", "fraud", "Shipping address changed after purchase")]
|
|
57
|
+
result.failed # jurors that errored or gave an invalid answer
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
Any endpoint that speaks the OpenAI chat completions API works. A plain string juror
|
|
61
|
+
(`"gpt-4o-mini"`) uses `OPENAI_BASE_URL` and `OPENAI_API_KEY`.
|
|
62
|
+
|
|
63
|
+
## Examples
|
|
64
|
+
|
|
65
|
+
These are real runs: five copies of one local 27B model, sampled at `temperature=1.0`.
|
|
66
|
+
The evidence is slightly trimmed here, and the quoted reasons are the model's own words.
|
|
67
|
+
A mixed panel of different models would disagree more often. See the warning at the end.
|
|
68
|
+
|
|
69
|
+
### Refund with mixed signals
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
convene(
|
|
73
|
+
question="Should this refund be approved?",
|
|
74
|
+
evidence="Customer of 6 years, 40 prior orders, 2 prior refunds (both legitimate). "
|
|
75
|
+
"Claims a $1,900 laptop arrived with a cracked screen. The photo's EXIF date is "
|
|
76
|
+
"2 days before delivery. Courier marked it 'left at door', no signature. "
|
|
77
|
+
"Customer says their phone clock is wrong and offers to send the laptop back.",
|
|
78
|
+
options=["approve", "deny", "escalate"],
|
|
79
|
+
jurors=jurors,
|
|
80
|
+
)
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
**escalate, 4–1, needs review.** The majority wanted the EXIF date checked. The dissent
|
|
84
|
+
made the case for approving:
|
|
85
|
+
|
|
86
|
+
> ✗ approve: *Strong 6-year history with only 2 prior refunds outweighs the EXIF
|
|
87
|
+
> inconsistency, which is plausibly explained by the customer's claim of a misconfigured
|
|
88
|
+
> phone clock.*
|
|
89
|
+
|
|
90
|
+
### Bug triage across four teams
|
|
91
|
+
|
|
92
|
+
```python
|
|
93
|
+
convene(
|
|
94
|
+
question="Which team should own this bug report?",
|
|
95
|
+
evidence="'After upgrading to v4.2, CSV exports from the dashboard are missing every "
|
|
96
|
+
"row created after 11pm. Only affects customers in Australia. Started the same day we "
|
|
97
|
+
"moved the export job to a new server.'",
|
|
98
|
+
options=["Frontend", "Data", "Infra", "Billing"],
|
|
99
|
+
jurors=jurors,
|
|
100
|
+
)
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
**Data, 4–1, needs review.** Four jurors blamed timezone logic in the export pipeline.
|
|
104
|
+
The dissent was the only one that used the strongest clue:
|
|
105
|
+
|
|
106
|
+
> ✗ Infra: *The bug started immediately after moving the export job to a new server,
|
|
107
|
+
> and the time-based filtering error likely stems from a timezone configuration issue
|
|
108
|
+
> on that new infrastructure.*
|
|
109
|
+
|
|
110
|
+
A reviewer who reads only the dissent knows where to look first.
|
|
111
|
+
|
|
112
|
+
### Merging a fast but risky cache
|
|
113
|
+
|
|
114
|
+
```python
|
|
115
|
+
convene(
|
|
116
|
+
question="Should this pull request be merged as is?",
|
|
117
|
+
evidence="Adds a cache in front of the user-permissions lookup: p99 latency drops "
|
|
118
|
+
"from 900ms to 40ms. TTL is 10 minutes. Tests pass. A revoked admin keeps admin "
|
|
119
|
+
"rights for up to 10 minutes. Revocations happen about 3 times a month, and last "
|
|
120
|
+
"week's outage was caused by the slow lookup.",
|
|
121
|
+
options=["merge", "request changes", "merge with follow-up ticket"],
|
|
122
|
+
jurors=jurors,
|
|
123
|
+
)
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
**merge with follow-up ticket, 5–0, ships.** All five weighed the latency gain against
|
|
127
|
+
the 10-minute window and asked for a ticket to add cache invalidation on revocation.
|
|
128
|
+
|
|
129
|
+
### Satire or threat
|
|
130
|
+
|
|
131
|
+
```python
|
|
132
|
+
convene(
|
|
133
|
+
question="Does this post violate the policy against threats of violence?",
|
|
134
|
+
evidence="Policy: remove credible threats of violence. Satire and hyperbole are allowed. "
|
|
135
|
+
"Post, replying to a council member's parking fee increase: 'If they raise parking "
|
|
136
|
+
"again I swear I'll show up at the next meeting and make sure he never votes on "
|
|
137
|
+
"anything again 🙂'. Author has no prior violations and often jokes about local politics.",
|
|
138
|
+
options=["violates", "allowed"],
|
|
139
|
+
jurors=jurors,
|
|
140
|
+
)
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
**allowed, 5–0, ships.** This is the warning. The post names a target and a place, and
|
|
144
|
+
a human moderator might well escalate it. All five jurors dismissed it for the same
|
|
145
|
+
reason, mostly the smiley. Copies of one model share the same blind spots, so their
|
|
146
|
+
agreement is weaker evidence than it looks. Use jurors from different model families,
|
|
147
|
+
because a unanimous vote only means something if the jurors could have disagreed.
|
|
148
|
+
|
|
149
|
+
## Model options
|
|
150
|
+
|
|
151
|
+
Extra keyword arguments to `convene()` go to every juror's `chat.completions.create` call.
|
|
152
|
+
A juror's own `params` override them. `temperature` defaults to `0`.
|
|
153
|
+
|
|
154
|
+
```python
|
|
155
|
+
convene(..., temperature=0.3, max_tokens=200, jurors=[
|
|
156
|
+
Juror("gpt-4o-mini"), # temperature=0.3, max_tokens=200
|
|
157
|
+
Juror("o4-mini", params={"temperature": None, # None drops a param
|
|
158
|
+
"reasoning_effort": "low"}), # adds a model-specific option
|
|
159
|
+
Juror("qwen3", base_url="http://localhost:8000/v1",
|
|
160
|
+
params={"temperature": 0.7, "extra_body": {"top_k": 20}}), # overrides temperature
|
|
161
|
+
])
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
## Claude
|
|
165
|
+
|
|
166
|
+
Claude uses the native Anthropic Messages API. Install the extra with `pip install modeljury[claude]`.
|
|
167
|
+
|
|
168
|
+
```python
|
|
169
|
+
from modeljury import Claude
|
|
170
|
+
|
|
171
|
+
jurors = [
|
|
172
|
+
"gpt-4o-mini",
|
|
173
|
+
"claude-sonnet-5-5", # any "claude-..." string works
|
|
174
|
+
Claude("claude-opus-5-5", params={"output_config": {"effort": "low"}}),
|
|
175
|
+
]
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
- Credentials come from `ANTHROPIC_API_KEY` or an `ant auth login` profile, unless you pass `api_key=`.
|
|
179
|
+
- Panel-wide kwargs are OpenAI parameters, so Claude jurors don't receive them. Put Messages API
|
|
180
|
+
options in `params` instead. Current Claude models reject `temperature`, so none is sent.
|
|
181
|
+
- `max_tokens` defaults to 16000. Set it lower with `params={"max_tokens": ...}` if you need a cost cap.
|
|
182
|
+
Thinking counts toward it.
|
|
183
|
+
- If Claude refuses, the vote is recorded in `result.failed` and the decision goes to review.
|
|
184
|
+
|
|
185
|
+
## Custom prompt
|
|
186
|
+
|
|
187
|
+
`prompt=` takes a `str.format` template with `{question}`, `{evidence}` and `{options}`.
|
|
188
|
+
Double any literal braces (`{{` and `}}`). The default is `modeljury.PROMPT`. Your template
|
|
189
|
+
must still ask for `{"choice": ..., "confidence": ..., "reason": ...}` JSON, because that's what the parser reads.
|
|
190
|
+
|
|
191
|
+
## Rules
|
|
192
|
+
|
|
193
|
+
1. Each juror picks one allowed option and gives a one-line reason.
|
|
194
|
+
2. The verdict is the option with the most votes.
|
|
195
|
+
3. `needs_review` is true unless every juror returned the same valid choice.
|
|
196
|
+
4. A tie is a hung jury: no verdict, review needed.
|
|
197
|
+
|
|
198
|
+
`agreement` is the winning option's votes divided by valid votes. A juror that errors, refuses or
|
|
199
|
+
answers with an option that isn't allowed is left out of the count and listed in `result.failed`.
|
|
200
|
+
|
|
201
|
+
## Outcomes
|
|
202
|
+
|
|
203
|
+
Three jurors, options `fraud` / `legit`:
|
|
204
|
+
|
|
205
|
+
| Votes | `verdict` | `agreement` | `majority` | `needs_review` | Why |
|
|
206
|
+
|---|---|---|---|---|---|
|
|
207
|
+
| legit, legit, legit | legit | 100% | ✅ | ❌ | Everyone agrees: ships without review |
|
|
208
|
+
| legit, legit, fraud | legit | 67% | ✅ | ✅ | Dissent; the reason is in `dissent` |
|
|
209
|
+
| legit, fraud, *failed* | `None` | 50% | ❌ | ✅ | Tie, so hung jury |
|
|
210
|
+
| legit, legit, *failed* | legit | 100% | ✅ | ✅ | A juror failed |
|
|
211
|
+
| legit, *failed*, *failed* | legit | 100% | ❌ | ✅ | Only one valid vote out of three |
|
|
212
|
+
| *failed* ×3 | `None` | 0% | ❌ | ✅ | No valid votes |
|
|
213
|
+
|
|
214
|
+
`verdict` is the plurality winner, so with three or more options it can win without a majority:
|
|
215
|
+
five votes split A, A, B, C, D give verdict A with 40% agreement and `majority` false.
|
|
216
|
+
`majority` counts the whole panel, failed jurors included, so it means more than half of
|
|
217
|
+
all jurors chose the verdict.
|
|
218
|
+
|
|
219
|
+
`needs_review` is the strict rule. For a looser one, check `majority` or `agreement` yourself:
|
|
220
|
+
|
|
221
|
+
```python
|
|
222
|
+
if not result.majority: # ship any decision that more than half the panel backs
|
|
223
|
+
send_to_human(result)
|
|
224
|
+
```
|
|
225
|
+
|
|
226
|
+
A single juror can't dissent, so its decisions only go to review when it fails. `convene()` warns if you pass only one juror.
|
|
227
|
+
|
|
228
|
+
## Notes
|
|
229
|
+
|
|
230
|
+
- If a model name is wrong, the juror's error in `result.failed` lists the models the endpoint serves, or suggests the closest names when there are more than 20.
|
|
231
|
+
- Jurors with the same name (e.g. `llama3.1` on two endpoints) are recorded as `llama3.1`, `llama3.1#2`, ...
|
|
232
|
+
- `timeout` applies to each attempt. The OpenAI and Anthropic SDKs retry failed calls twice, so one juror can take up to three times `timeout`.
|
|
233
|
+
|
|
234
|
+
## Develop
|
|
235
|
+
|
|
236
|
+
```sh
|
|
237
|
+
uv sync
|
|
238
|
+
uv run pytest
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
## License
|
|
242
|
+
|
|
243
|
+
MIT. See [LICENSE](https://github.com/YasserManss/modeljury/blob/main/LICENSE).
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
"""Run a four-model panel: OpenAI-compatible endpoints plus Claude."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
|
|
5
|
+
from modeljury import Claude, Juror, convene
|
|
6
|
+
|
|
7
|
+
jurors = [
|
|
8
|
+
Juror("gpt-4o-mini"), # uses OPENAI_BASE_URL / OPENAI_API_KEY
|
|
9
|
+
Claude("claude-sonnet-5-5"), # uses ANTHROPIC_API_KEY; needs modeljury[claude]
|
|
10
|
+
Juror("llama3.1", base_url="http://localhost:11434/v1"), # Ollama
|
|
11
|
+
Juror(
|
|
12
|
+
"meta-llama/llama-3.1-70b-instruct",
|
|
13
|
+
base_url="https://openrouter.ai/api/v1",
|
|
14
|
+
api_key=os.environ.get("OPENROUTER_API_KEY"),
|
|
15
|
+
name="openrouter-llama",
|
|
16
|
+
),
|
|
17
|
+
]
|
|
18
|
+
|
|
19
|
+
result = convene(
|
|
20
|
+
question="Is this refund request fraudulent?",
|
|
21
|
+
evidence="Order #1182 delivered 3 days ago. Customer says box arrived empty. "
|
|
22
|
+
"Shipping address was changed an hour after purchase.",
|
|
23
|
+
options=["fraud", "legit"],
|
|
24
|
+
jurors=jurors,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
print("verdict: ", result.verdict)
|
|
28
|
+
print("agreement: ", f"{result.agreement:.0%}")
|
|
29
|
+
print("needs review:", result.needs_review)
|
|
30
|
+
for juror, choice, reason in result.dissent:
|
|
31
|
+
print(f" dissent {juror}: {choice} — {reason}")
|
|
32
|
+
for v in result.failed:
|
|
33
|
+
print(f" failed {v.juror}: {v.error}")
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "modeljury"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Ask several models instead of one. Ship when they agree, send to a human when they don't."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.10"
|
|
7
|
+
license = "MIT"
|
|
8
|
+
authors = [{ name = "YasserManss" }]
|
|
9
|
+
keywords = ["llm", "evaluation", "llm-as-a-judge", "ensemble", "human-in-the-loop", "voting"]
|
|
10
|
+
classifiers = [
|
|
11
|
+
"Development Status :: 3 - Alpha",
|
|
12
|
+
"Intended Audience :: Developers",
|
|
13
|
+
"Operating System :: OS Independent",
|
|
14
|
+
"Programming Language :: Python :: 3",
|
|
15
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
16
|
+
"Programming Language :: Python :: 3.10",
|
|
17
|
+
"Programming Language :: Python :: 3.11",
|
|
18
|
+
"Programming Language :: Python :: 3.12",
|
|
19
|
+
"Programming Language :: Python :: 3.13",
|
|
20
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
21
|
+
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
22
|
+
]
|
|
23
|
+
dependencies = ["openai>=1.0"]
|
|
24
|
+
|
|
25
|
+
[project.urls]
|
|
26
|
+
Homepage = "https://github.com/YasserManss/modeljury"
|
|
27
|
+
Repository = "https://github.com/YasserManss/modeljury"
|
|
28
|
+
Issues = "https://github.com/YasserManss/modeljury/issues"
|
|
29
|
+
|
|
30
|
+
[project.optional-dependencies]
|
|
31
|
+
claude = [
|
|
32
|
+
"anthropic>=1.11.0",
|
|
33
|
+
]
|
|
34
|
+
|
|
35
|
+
[dependency-groups]
|
|
36
|
+
dev = ["pytest>=8"]
|
|
37
|
+
|
|
38
|
+
[build-system]
|
|
39
|
+
requires = ["hatchling"]
|
|
40
|
+
build-backend = "hatchling.build"
|