modeljury 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,36 @@
1
+ # Publishes to PyPI when a GitHub release is published.
2
+ # PyPI trusts this workflow through trusted publishing, so no token is stored.
3
+ name: Publish to PyPI
4
+
5
+ on:
6
+ release:
7
+ types: [published]
8
+
9
+ jobs:
10
+ build:
11
+ runs-on: ubuntu-latest
12
+ steps:
13
+ - uses: actions/checkout@v7
14
+ - uses: astral-sh/setup-uv@v10.2.0
15
+ - run: uv sync --all-extras
16
+ - run: uv run pytest -q
17
+ - run: uv build
18
+ - uses: actions/upload-artifact@v7
19
+ with:
20
+ name: dist
21
+ path: dist/
22
+
23
+ publish:
24
+ needs: build
25
+ runs-on: ubuntu-latest
26
+ environment:
27
+ name: pypi
28
+ url: https://pypi.org/p/modeljury
29
+ permissions:
30
+ id-token: write
31
+ steps:
32
+ - uses: actions/download-artifact@v8
33
+ with:
34
+ name: dist
35
+ path: dist/
36
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,7 @@
1
+ .venv/
2
+ __pycache__/
3
+ *.pyc
4
+ .pytest_cache/
5
+ dist/
6
+ build/
7
+ *.egg-info/
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 modeljury contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,271 @@
1
+ Metadata-Version: 2.5
2
+ Name: modeljury
3
+ Version: 0.1.0
4
+ Summary: Ask several models instead of one. Ship when they agree, send to a human when they don't.
5
+ Project-URL: Homepage, https://github.com/YasserManss/modeljury
6
+ Project-URL: Repository, https://github.com/YasserManss/modeljury
7
+ Project-URL: Issues, https://github.com/YasserManss/modeljury/issues
8
+ Author: YasserManss
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: ensemble,evaluation,human-in-the-loop,llm,llm-as-a-judge,voting
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3 :: Only
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
22
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
23
+ Requires-Python: >=3.10
24
+ Requires-Dist: openai>=1.0
25
+ Provides-Extra: claude
26
+ Requires-Dist: anthropic>=1.11.0; extra == 'claude'
27
+ Description-Content-Type: text/markdown
28
+
29
+ # modeljury
30
+
31
+ Ask several models instead of one. Ship when they agree, send to a human when they don't.
32
+
33
+ ## Getting started
34
+
35
+ Python 3.10 or newer.
36
+
37
+ ```sh
38
+ pip install modeljury
39
+ # with Claude support
40
+ pip install "modeljury[claude]"
41
+ ```
42
+
43
+ Point it at any OpenAI-compatible endpoint (OpenAI, Ollama, vLLM, OpenRouter, ...) and
44
+ convene a panel. This one uses three models served by a local Ollama:
45
+
46
+ ```python
47
+ from modeljury import Juror, convene
48
+
49
+ local = "http://localhost:11434/v1"
50
+ result = convene(
51
+ question="Is this review spam?",
52
+ evidence="'Best product ever!!! Visit cheap-deals.example for 90% off!!!'",
53
+ options=["spam", "not spam"],
54
+ jurors=[Juror(m, base_url=local) for m in ["llama3.1", "qwen3", "mistral"]],
55
+ )
56
+
57
+ print(result.verdict, result.needs_review)
58
+ for juror, choice, reason in result.dissent:
59
+ print(f"{juror} disagreed: {choice}, {reason}")
60
+ ```
61
+
62
+ For hosted models, set `OPENAI_API_KEY` (and `OPENAI_BASE_URL` for non-OpenAI providers) or
63
+ pass `api_key=` and `base_url=` to each `Juror`. For Claude, set `ANTHROPIC_API_KEY`.
64
+
65
+ ## Usage
66
+
67
+ ```python
68
+ from modeljury import Juror, convene
69
+
70
+ result = convene(
71
+ question="Is this refund request fraudulent?",
72
+ evidence=ticket_text,
73
+ options=["fraud", "legit"],
74
+ jurors=[
75
+ Juror("gpt-4o-mini"),
76
+ Juror("llama3.1", base_url="http://localhost:11434/v1"),
77
+ Juror("mistral-small", base_url="https://api.mistral.ai/v1", api_key="..."),
78
+ ],
79
+ )
80
+
81
+ result.verdict # "legit", or None on a hung jury
82
+ result.majority # True: the verdict won more than half of the whole panel
83
+ result.needs_review # True
84
+ result.dissent # [("mistral-small", "fraud", "Shipping address changed after purchase")]
85
+ result.failed # jurors that errored or gave an invalid answer
86
+ ```
87
+
88
+ Any endpoint that speaks the OpenAI chat completions API works. A plain string juror
89
+ (`"gpt-4o-mini"`) uses `OPENAI_BASE_URL` and `OPENAI_API_KEY`.
90
+
91
+ ## Examples
92
+
93
+ These are real runs: five copies of one local 27B model, sampled at `temperature=1.0`.
94
+ The evidence is slightly trimmed here, and the quoted reasons are the model's own words.
95
+ A mixed panel of different models would disagree more often. See the warning at the end.
96
+
97
+ ### Refund with mixed signals
98
+
99
+ ```python
100
+ convene(
101
+ question="Should this refund be approved?",
102
+ evidence="Customer of 6 years, 40 prior orders, 2 prior refunds (both legitimate). "
103
+ "Claims a $1,900 laptop arrived with a cracked screen. The photo's EXIF date is "
104
+ "2 days before delivery. Courier marked it 'left at door', no signature. "
105
+ "Customer says their phone clock is wrong and offers to send the laptop back.",
106
+ options=["approve", "deny", "escalate"],
107
+ jurors=jurors,
108
+ )
109
+ ```
110
+
111
+ **escalate, 4–1, needs review.** The majority wanted the EXIF date checked. The dissent
112
+ made the case for approving:
113
+
114
+ > ✗ approve: *Strong 6-year history with only 2 prior refunds outweighs the EXIF
115
+ > inconsistency, which is plausibly explained by the customer's claim of a misconfigured
116
+ > phone clock.*
117
+
118
+ ### Bug triage across four teams
119
+
120
+ ```python
121
+ convene(
122
+ question="Which team should own this bug report?",
123
+ evidence="'After upgrading to v4.2, CSV exports from the dashboard are missing every "
124
+ "row created after 11pm. Only affects customers in Australia. Started the same day we "
125
+ "moved the export job to a new server.'",
126
+ options=["Frontend", "Data", "Infra", "Billing"],
127
+ jurors=jurors,
128
+ )
129
+ ```
130
+
131
+ **Data, 4–1, needs review.** Four jurors blamed timezone logic in the export pipeline.
132
+ The dissent was the only one that used the strongest clue:
133
+
134
+ > ✗ Infra: *The bug started immediately after moving the export job to a new server,
135
+ > and the time-based filtering error likely stems from a timezone configuration issue
136
+ > on that new infrastructure.*
137
+
138
+ A reviewer who reads only the dissent knows where to look first.
139
+
140
+ ### Merging a fast but risky cache
141
+
142
+ ```python
143
+ convene(
144
+ question="Should this pull request be merged as is?",
145
+ evidence="Adds a cache in front of the user-permissions lookup: p99 latency drops "
146
+ "from 900ms to 40ms. TTL is 10 minutes. Tests pass. A revoked admin keeps admin "
147
+ "rights for up to 10 minutes. Revocations happen about 3 times a month, and last "
148
+ "week's outage was caused by the slow lookup.",
149
+ options=["merge", "request changes", "merge with follow-up ticket"],
150
+ jurors=jurors,
151
+ )
152
+ ```
153
+
154
+ **merge with follow-up ticket, 5–0, ships.** All five weighed the latency gain against
155
+ the 10-minute window and asked for a ticket to add cache invalidation on revocation.
156
+
157
+ ### Satire or threat
158
+
159
+ ```python
160
+ convene(
161
+ question="Does this post violate the policy against threats of violence?",
162
+ evidence="Policy: remove credible threats of violence. Satire and hyperbole are allowed. "
163
+ "Post, replying to a council member's parking fee increase: 'If they raise parking "
164
+ "again I swear I'll show up at the next meeting and make sure he never votes on "
165
+ "anything again 🙂'. Author has no prior violations and often jokes about local politics.",
166
+ options=["violates", "allowed"],
167
+ jurors=jurors,
168
+ )
169
+ ```
170
+
171
+ **allowed, 5–0, ships.** This is the warning. The post names a target and a place, and
172
+ a human moderator might well escalate it. All five jurors dismissed it for the same
173
+ reason, mostly the smiley. Copies of one model share the same blind spots, so their
174
+ agreement is weaker evidence than it looks. Use jurors from different model families,
175
+ because a unanimous vote only means something if the jurors could have disagreed.
176
+
177
+ ## Model options
178
+
179
+ Extra keyword arguments to `convene()` go to every juror's `chat.completions.create` call.
180
+ A juror's own `params` override them. `temperature` defaults to `0`.
181
+
182
+ ```python
183
+ convene(..., temperature=0.3, max_tokens=200, jurors=[
184
+ Juror("gpt-4o-mini"), # temperature=0.3, max_tokens=200
185
+ Juror("o4-mini", params={"temperature": None, # None drops a param
186
+ "reasoning_effort": "low"}), # adds a model-specific option
187
+ Juror("qwen3", base_url="http://localhost:8000/v1",
188
+ params={"temperature": 0.7, "extra_body": {"top_k": 20}}), # overrides temperature
189
+ ])
190
+ ```
191
+
192
+ ## Claude
193
+
194
+ Claude uses the native Anthropic Messages API. Install the extra with `pip install modeljury[claude]`.
195
+
196
+ ```python
197
+ from modeljury import Claude
198
+
199
+ jurors = [
200
+ "gpt-4o-mini",
201
+ "claude-sonnet-5-5", # any "claude-..." string works
202
+ Claude("claude-opus-5-5", params={"output_config": {"effort": "low"}}),
203
+ ]
204
+ ```
205
+
206
+ - Credentials come from `ANTHROPIC_API_KEY` or an `ant auth login` profile, unless you pass `api_key=`.
207
+ - Panel-wide kwargs are OpenAI parameters, so Claude jurors don't receive them. Put Messages API
208
+ options in `params` instead. Current Claude models reject `temperature`, so none is sent.
209
+ - `max_tokens` defaults to 16000. Set it lower with `params={"max_tokens": ...}` if you need a cost cap.
210
+ Thinking counts toward it.
211
+ - If Claude refuses, the vote is recorded in `result.failed` and the decision goes to review.
212
+
213
+ ## Custom prompt
214
+
215
+ `prompt=` takes a `str.format` template with `{question}`, `{evidence}` and `{options}`.
216
+ Double any literal braces (`{{` and `}}`). The default is `modeljury.PROMPT`. Your template
217
+ must still ask for `{"choice": ..., "confidence": ..., "reason": ...}` JSON, because that's what the parser reads.
218
+
219
+ ## Rules
220
+
221
+ 1. Each juror picks one allowed option and gives a one-line reason.
222
+ 2. The verdict is the option with the most votes.
223
+ 3. `needs_review` is true unless every juror returned the same valid choice.
224
+ 4. A tie is a hung jury: no verdict, review needed.
225
+
226
+ `agreement` is the winning option's votes divided by valid votes. A juror that errors, refuses or
227
+ answers with an option that isn't allowed is left out of the count and listed in `result.failed`.
228
+
229
+ ## Outcomes
230
+
231
+ Three jurors, options `fraud` / `legit`:
232
+
233
+ | Votes | `verdict` | `agreement` | `majority` | `needs_review` | Why |
234
+ |---|---|---|---|---|---|
235
+ | legit, legit, legit | legit | 100% | ✅ | ❌ | Everyone agrees: ships without review |
236
+ | legit, legit, fraud | legit | 67% | ✅ | ✅ | Dissent; the reason is in `dissent` |
237
+ | legit, fraud, *failed* | `None` | 50% | ❌ | ✅ | Tie, so hung jury |
238
+ | legit, legit, *failed* | legit | 100% | ✅ | ✅ | A juror failed |
239
+ | legit, *failed*, *failed* | legit | 100% | ❌ | ✅ | Only one valid vote out of three |
240
+ | *failed* ×3 | `None` | 0% | ❌ | ✅ | No valid votes |
241
+
242
+ `verdict` is the plurality winner, so with three or more options it can win without a majority:
243
+ five votes split A, A, B, C, D give verdict A with 40% agreement and `majority` false.
244
+ `majority` counts the whole panel, failed jurors included, so it means more than half of
245
+ all jurors chose the verdict.
246
+
247
+ `needs_review` is the strict rule. For a looser one, check `majority` or `agreement` yourself:
248
+
249
+ ```python
250
+ if not result.majority: # ship any decision that more than half the panel backs
251
+ send_to_human(result)
252
+ ```
253
+
254
+ A single juror can't dissent, so its decisions only go to review when it fails. `convene()` warns if you pass only one juror.
255
+
256
+ ## Notes
257
+
258
+ - If a model name is wrong, the juror's error in `result.failed` lists the models the endpoint serves, or suggests the closest names when there are more than 20.
259
+ - Jurors with the same name (e.g. `llama3.1` on two endpoints) are recorded as `llama3.1`, `llama3.1#2`, ...
260
+ - `timeout` applies to each attempt. The OpenAI and Anthropic SDKs retry failed calls twice, so one juror can take up to three times `timeout`.
261
+
262
+ ## Develop
263
+
264
+ ```sh
265
+ uv sync
266
+ uv run pytest
267
+ ```
268
+
269
+ ## License
270
+
271
+ MIT. See [LICENSE](https://github.com/YasserManss/modeljury/blob/main/LICENSE).
@@ -0,0 +1,243 @@
1
+ # modeljury
2
+
3
+ Ask several models instead of one. Ship when they agree, send to a human when they don't.
4
+
5
+ ## Getting started
6
+
7
+ Python 3.10 or newer.
8
+
9
+ ```sh
10
+ pip install modeljury
11
+ # with Claude support
12
+ pip install "modeljury[claude]"
13
+ ```
14
+
15
+ Point it at any OpenAI-compatible endpoint (OpenAI, Ollama, vLLM, OpenRouter, ...) and
16
+ convene a panel. This one uses three models served by a local Ollama:
17
+
18
+ ```python
19
+ from modeljury import Juror, convene
20
+
21
+ local = "http://localhost:11434/v1"
22
+ result = convene(
23
+ question="Is this review spam?",
24
+ evidence="'Best product ever!!! Visit cheap-deals.example for 90% off!!!'",
25
+ options=["spam", "not spam"],
26
+ jurors=[Juror(m, base_url=local) for m in ["llama3.1", "qwen3", "mistral"]],
27
+ )
28
+
29
+ print(result.verdict, result.needs_review)
30
+ for juror, choice, reason in result.dissent:
31
+ print(f"{juror} disagreed: {choice}, {reason}")
32
+ ```
33
+
34
+ For hosted models, set `OPENAI_API_KEY` (and `OPENAI_BASE_URL` for non-OpenAI providers) or
35
+ pass `api_key=` and `base_url=` to each `Juror`. For Claude, set `ANTHROPIC_API_KEY`.
36
+
37
+ ## Usage
38
+
39
+ ```python
40
+ from modeljury import Juror, convene
41
+
42
+ result = convene(
43
+ question="Is this refund request fraudulent?",
44
+ evidence=ticket_text,
45
+ options=["fraud", "legit"],
46
+ jurors=[
47
+ Juror("gpt-4o-mini"),
48
+ Juror("llama3.1", base_url="http://localhost:11434/v1"),
49
+ Juror("mistral-small", base_url="https://api.mistral.ai/v1", api_key="..."),
50
+ ],
51
+ )
52
+
53
+ result.verdict # "legit", or None on a hung jury
54
+ result.majority # True: the verdict won more than half of the whole panel
55
+ result.needs_review # True
56
+ result.dissent # [("mistral-small", "fraud", "Shipping address changed after purchase")]
57
+ result.failed # jurors that errored or gave an invalid answer
58
+ ```
59
+
60
+ Any endpoint that speaks the OpenAI chat completions API works. A plain string juror
61
+ (`"gpt-4o-mini"`) uses `OPENAI_BASE_URL` and `OPENAI_API_KEY`.
62
+
63
+ ## Examples
64
+
65
+ These are real runs: five copies of one local 27B model, sampled at `temperature=1.0`.
66
+ The evidence is slightly trimmed here, and the quoted reasons are the model's own words.
67
+ A mixed panel of different models would disagree more often. See the warning at the end.
68
+
69
+ ### Refund with mixed signals
70
+
71
+ ```python
72
+ convene(
73
+ question="Should this refund be approved?",
74
+ evidence="Customer of 6 years, 40 prior orders, 2 prior refunds (both legitimate). "
75
+ "Claims a $1,900 laptop arrived with a cracked screen. The photo's EXIF date is "
76
+ "2 days before delivery. Courier marked it 'left at door', no signature. "
77
+ "Customer says their phone clock is wrong and offers to send the laptop back.",
78
+ options=["approve", "deny", "escalate"],
79
+ jurors=jurors,
80
+ )
81
+ ```
82
+
83
+ **escalate, 4–1, needs review.** The majority wanted the EXIF date checked. The dissent
84
+ made the case for approving:
85
+
86
+ > ✗ approve: *Strong 6-year history with only 2 prior refunds outweighs the EXIF
87
+ > inconsistency, which is plausibly explained by the customer's claim of a misconfigured
88
+ > phone clock.*
89
+
90
+ ### Bug triage across four teams
91
+
92
+ ```python
93
+ convene(
94
+ question="Which team should own this bug report?",
95
+ evidence="'After upgrading to v4.2, CSV exports from the dashboard are missing every "
96
+ "row created after 11pm. Only affects customers in Australia. Started the same day we "
97
+ "moved the export job to a new server.'",
98
+ options=["Frontend", "Data", "Infra", "Billing"],
99
+ jurors=jurors,
100
+ )
101
+ ```
102
+
103
+ **Data, 4–1, needs review.** Four jurors blamed timezone logic in the export pipeline.
104
+ The dissent was the only one that used the strongest clue:
105
+
106
+ > ✗ Infra: *The bug started immediately after moving the export job to a new server,
107
+ > and the time-based filtering error likely stems from a timezone configuration issue
108
+ > on that new infrastructure.*
109
+
110
+ A reviewer who reads only the dissent knows where to look first.
111
+
112
+ ### Merging a fast but risky cache
113
+
114
+ ```python
115
+ convene(
116
+ question="Should this pull request be merged as is?",
117
+ evidence="Adds a cache in front of the user-permissions lookup: p99 latency drops "
118
+ "from 900ms to 40ms. TTL is 10 minutes. Tests pass. A revoked admin keeps admin "
119
+ "rights for up to 10 minutes. Revocations happen about 3 times a month, and last "
120
+ "week's outage was caused by the slow lookup.",
121
+ options=["merge", "request changes", "merge with follow-up ticket"],
122
+ jurors=jurors,
123
+ )
124
+ ```
125
+
126
+ **merge with follow-up ticket, 5–0, ships.** All five weighed the latency gain against
127
+ the 10-minute window and asked for a ticket to add cache invalidation on revocation.
128
+
129
+ ### Satire or threat
130
+
131
+ ```python
132
+ convene(
133
+ question="Does this post violate the policy against threats of violence?",
134
+ evidence="Policy: remove credible threats of violence. Satire and hyperbole are allowed. "
135
+ "Post, replying to a council member's parking fee increase: 'If they raise parking "
136
+ "again I swear I'll show up at the next meeting and make sure he never votes on "
137
+ "anything again 🙂'. Author has no prior violations and often jokes about local politics.",
138
+ options=["violates", "allowed"],
139
+ jurors=jurors,
140
+ )
141
+ ```
142
+
143
+ **allowed, 5–0, ships.** This is the warning. The post names a target and a place, and
144
+ a human moderator might well escalate it. All five jurors dismissed it for the same
145
+ reason, mostly the smiley. Copies of one model share the same blind spots, so their
146
+ agreement is weaker evidence than it looks. Use jurors from different model families,
147
+ because a unanimous vote only means something if the jurors could have disagreed.
148
+
149
+ ## Model options
150
+
151
+ Extra keyword arguments to `convene()` go to every juror's `chat.completions.create` call.
152
+ A juror's own `params` override them. `temperature` defaults to `0`.
153
+
154
+ ```python
155
+ convene(..., temperature=0.3, max_tokens=200, jurors=[
156
+ Juror("gpt-4o-mini"), # temperature=0.3, max_tokens=200
157
+ Juror("o4-mini", params={"temperature": None, # None drops a param
158
+ "reasoning_effort": "low"}), # adds a model-specific option
159
+ Juror("qwen3", base_url="http://localhost:8000/v1",
160
+ params={"temperature": 0.7, "extra_body": {"top_k": 20}}), # overrides temperature
161
+ ])
162
+ ```
163
+
164
+ ## Claude
165
+
166
+ Claude uses the native Anthropic Messages API. Install the extra with `pip install modeljury[claude]`.
167
+
168
+ ```python
169
+ from modeljury import Claude
170
+
171
+ jurors = [
172
+ "gpt-4o-mini",
173
+ "claude-sonnet-5-5", # any "claude-..." string works
174
+ Claude("claude-opus-5-5", params={"output_config": {"effort": "low"}}),
175
+ ]
176
+ ```
177
+
178
+ - Credentials come from `ANTHROPIC_API_KEY` or an `ant auth login` profile, unless you pass `api_key=`.
179
+ - Panel-wide kwargs are OpenAI parameters, so Claude jurors don't receive them. Put Messages API
180
+ options in `params` instead. Current Claude models reject `temperature`, so none is sent.
181
+ - `max_tokens` defaults to 16000. Set it lower with `params={"max_tokens": ...}` if you need a cost cap.
182
+ Thinking counts toward it.
183
+ - If Claude refuses, the vote is recorded in `result.failed` and the decision goes to review.
184
+
185
+ ## Custom prompt
186
+
187
+ `prompt=` takes a `str.format` template with `{question}`, `{evidence}` and `{options}`.
188
+ Double any literal braces (`{{` and `}}`). The default is `modeljury.PROMPT`. Your template
189
+ must still ask for `{"choice": ..., "confidence": ..., "reason": ...}` JSON, because that's what the parser reads.
190
+
191
+ ## Rules
192
+
193
+ 1. Each juror picks one allowed option and gives a one-line reason.
194
+ 2. The verdict is the option with the most votes.
195
+ 3. `needs_review` is true unless every juror returned the same valid choice.
196
+ 4. A tie is a hung jury: no verdict, review needed.
197
+
198
+ `agreement` is the winning option's votes divided by valid votes. A juror that errors, refuses or
199
+ answers with an option that isn't allowed is left out of the count and listed in `result.failed`.
200
+
201
+ ## Outcomes
202
+
203
+ Three jurors, options `fraud` / `legit`:
204
+
205
+ | Votes | `verdict` | `agreement` | `majority` | `needs_review` | Why |
206
+ |---|---|---|---|---|---|
207
+ | legit, legit, legit | legit | 100% | ✅ | ❌ | Everyone agrees: ships without review |
208
+ | legit, legit, fraud | legit | 67% | ✅ | ✅ | Dissent; the reason is in `dissent` |
209
+ | legit, fraud, *failed* | `None` | 50% | ❌ | ✅ | Tie, so hung jury |
210
+ | legit, legit, *failed* | legit | 100% | ✅ | ✅ | A juror failed |
211
+ | legit, *failed*, *failed* | legit | 100% | ❌ | ✅ | Only one valid vote out of three |
212
+ | *failed* ×3 | `None` | 0% | ❌ | ✅ | No valid votes |
213
+
214
+ `verdict` is the plurality winner, so with three or more options it can win without a majority:
215
+ five votes split A, A, B, C, D give verdict A with 40% agreement and `majority` false.
216
+ `majority` counts the whole panel, failed jurors included, so it means more than half of
217
+ all jurors chose the verdict.
218
+
219
+ `needs_review` is the strict rule. For a looser one, check `majority` or `agreement` yourself:
220
+
221
+ ```python
222
+ if not result.majority: # ship any decision that more than half the panel backs
223
+ send_to_human(result)
224
+ ```
225
+
226
+ A single juror can't dissent, so its decisions only go to review when it fails. `convene()` warns if you pass only one juror.
227
+
228
+ ## Notes
229
+
230
+ - If a model name is wrong, the juror's error in `result.failed` lists the models the endpoint serves, or suggests the closest names when there are more than 20.
231
+ - Jurors with the same name (e.g. `llama3.1` on two endpoints) are recorded as `llama3.1`, `llama3.1#2`, ...
232
+ - `timeout` applies to each attempt. The OpenAI and Anthropic SDKs retry failed calls twice, so one juror can take up to three times `timeout`.
233
+
234
+ ## Develop
235
+
236
+ ```sh
237
+ uv sync
238
+ uv run pytest
239
+ ```
240
+
241
+ ## License
242
+
243
+ MIT. See [LICENSE](https://github.com/YasserManss/modeljury/blob/main/LICENSE).
@@ -0,0 +1,33 @@
1
+ """Run a four-model panel: OpenAI-compatible endpoints plus Claude."""
2
+
3
+ import os
4
+
5
+ from modeljury import Claude, Juror, convene
6
+
7
+ jurors = [
8
+ Juror("gpt-4o-mini"), # uses OPENAI_BASE_URL / OPENAI_API_KEY
9
+ Claude("claude-sonnet-5-5"), # uses ANTHROPIC_API_KEY; needs modeljury[claude]
10
+ Juror("llama3.1", base_url="http://localhost:11434/v1"), # Ollama
11
+ Juror(
12
+ "meta-llama/llama-3.1-70b-instruct",
13
+ base_url="https://openrouter.ai/api/v1",
14
+ api_key=os.environ.get("OPENROUTER_API_KEY"),
15
+ name="openrouter-llama",
16
+ ),
17
+ ]
18
+
19
+ result = convene(
20
+ question="Is this refund request fraudulent?",
21
+ evidence="Order #1182 delivered 3 days ago. Customer says box arrived empty. "
22
+ "Shipping address was changed an hour after purchase.",
23
+ options=["fraud", "legit"],
24
+ jurors=jurors,
25
+ )
26
+
27
+ print("verdict: ", result.verdict)
28
+ print("agreement: ", f"{result.agreement:.0%}")
29
+ print("needs review:", result.needs_review)
30
+ for juror, choice, reason in result.dissent:
31
+ print(f" dissent {juror}: {choice} — {reason}")
32
+ for v in result.failed:
33
+ print(f" failed {v.juror}: {v.error}")
@@ -0,0 +1,40 @@
1
+ [project]
2
+ name = "modeljury"
3
+ version = "0.1.0"
4
+ description = "Ask several models instead of one. Ship when they agree, send to a human when they don't."
5
+ readme = "README.md"
6
+ requires-python = ">=3.10"
7
+ license = "MIT"
8
+ authors = [{ name = "YasserManss" }]
9
+ keywords = ["llm", "evaluation", "llm-as-a-judge", "ensemble", "human-in-the-loop", "voting"]
10
+ classifiers = [
11
+ "Development Status :: 3 - Alpha",
12
+ "Intended Audience :: Developers",
13
+ "Operating System :: OS Independent",
14
+ "Programming Language :: Python :: 3",
15
+ "Programming Language :: Python :: 3 :: Only",
16
+ "Programming Language :: Python :: 3.10",
17
+ "Programming Language :: Python :: 3.11",
18
+ "Programming Language :: Python :: 3.12",
19
+ "Programming Language :: Python :: 3.13",
20
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
21
+ "Topic :: Software Development :: Libraries :: Python Modules",
22
+ ]
23
+ dependencies = ["openai>=1.0"]
24
+
25
+ [project.urls]
26
+ Homepage = "https://github.com/YasserManss/modeljury"
27
+ Repository = "https://github.com/YasserManss/modeljury"
28
+ Issues = "https://github.com/YasserManss/modeljury/issues"
29
+
30
+ [project.optional-dependencies]
31
+ claude = [
32
+ "anthropic>=1.11.0",
33
+ ]
34
+
35
+ [dependency-groups]
36
+ dev = ["pytest>=8"]
37
+
38
+ [build-system]
39
+ requires = ["hatchling"]
40
+ build-backend = "hatchling.build"
@@ -0,0 +1,4 @@
1
+ from .claude import Claude
2
+ from .jury import PROMPT, Juror, Result, Vote, convene
3
+
4
+ __all__ = ["PROMPT", "Claude", "Juror", "Result", "Vote", "convene"]