agentjury 0.4.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentjury-0.4.4/LICENSE +21 -0
- agentjury-0.4.4/PKG-INFO +260 -0
- agentjury-0.4.4/README.md +222 -0
- agentjury-0.4.4/agentjury/__init__.py +22 -0
- agentjury-0.4.4/agentjury/aggregate.py +120 -0
- agentjury-0.4.4/agentjury/cli.py +312 -0
- agentjury-0.4.4/agentjury/judges/__init__.py +14 -0
- agentjury-0.4.4/agentjury/judges/anthropic_judge.py +75 -0
- agentjury-0.4.4/agentjury/judges/base.py +268 -0
- agentjury-0.4.4/agentjury/judges/fake.py +42 -0
- agentjury-0.4.4/agentjury/judges/openai_judge.py +37 -0
- agentjury-0.4.4/agentjury/panel.py +61 -0
- agentjury-0.4.4/agentjury/protocol.py +230 -0
- agentjury-0.4.4/agentjury.egg-info/PKG-INFO +260 -0
- agentjury-0.4.4/agentjury.egg-info/SOURCES.txt +25 -0
- agentjury-0.4.4/agentjury.egg-info/dependency_links.txt +1 -0
- agentjury-0.4.4/agentjury.egg-info/entry_points.txt +2 -0
- agentjury-0.4.4/agentjury.egg-info/requires.txt +17 -0
- agentjury-0.4.4/agentjury.egg-info/top_level.txt +1 -0
- agentjury-0.4.4/pyproject.toml +55 -0
- agentjury-0.4.4/setup.cfg +4 -0
- agentjury-0.4.4/tests/test_adversarial_live.py +37 -0
- agentjury-0.4.4/tests/test_aggregate.py +272 -0
- agentjury-0.4.4/tests/test_cli_adjudicate.py +145 -0
- agentjury-0.4.4/tests/test_hermes_plugin.py +241 -0
- agentjury-0.4.4/tests/test_parse.py +38 -0
- agentjury-0.4.4/tests/test_prompts.py +53 -0
agentjury-0.4.4/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 RASHID HASSAN AL-DERHAM
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
agentjury-0.4.4/PKG-INFO
ADDED
|
@@ -0,0 +1,260 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: agentjury
|
|
3
|
+
Version: 0.4.4
|
|
4
|
+
Summary: Peer review for AI agents using independent model juries and deterministic aggregation.
|
|
5
|
+
Author: Rashid Al-Derham
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/madad-rashid/AgentJury
|
|
8
|
+
Project-URL: Repository, https://github.com/madad-rashid/AgentJury
|
|
9
|
+
Project-URL: Issues, https://github.com/madad-rashid/AgentJury/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/madad-rashid/AgentJury/releases
|
|
11
|
+
Keywords: ai-agents,llm,llm-evaluation,agent-evaluation,multi-agent,peer-review
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
20
|
+
Classifier: Topic :: Software Development :: Testing
|
|
21
|
+
Requires-Python: >=3.11
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Requires-Dist: pydantic>=2.0
|
|
25
|
+
Requires-Dist: python-dotenv>=1.0
|
|
26
|
+
Provides-Extra: openai
|
|
27
|
+
Requires-Dist: openai>=1.0; extra == "openai"
|
|
28
|
+
Provides-Extra: anthropic
|
|
29
|
+
Requires-Dist: anthropic>=0.30; extra == "anthropic"
|
|
30
|
+
Provides-Extra: all
|
|
31
|
+
Requires-Dist: openai>=1.0; extra == "all"
|
|
32
|
+
Requires-Dist: anthropic>=0.30; extra == "all"
|
|
33
|
+
Provides-Extra: dev
|
|
34
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
35
|
+
Requires-Dist: build>=1.2; extra == "dev"
|
|
36
|
+
Requires-Dist: twine>=5.0; extra == "dev"
|
|
37
|
+
Dynamic: license-file
|
|
38
|
+
|
|
39
|
+
# AgentJury
|
|
40
|
+
|
|
41
|
+
[](https://github.com/madad-rashid/AgentJury/actions/workflows/tests.yml)
|
|
42
|
+
[](https://www.python.org/)
|
|
43
|
+
[](LICENSE)
|
|
44
|
+
|
|
45
|
+
**Peer review for AI agents.**
|
|
46
|
+
|
|
47
|
+
Your agent says the task is finished. AgentJury asks independent, blind AI reviewers whether the work is good enough before you trust it.
|
|
48
|
+
|
|
49
|
+
Each reviewer votes ▲ approve, ▼ revise, or – abstain. AgentJury combines those opinions with deterministic rules. No final LLM gets a deciding vote.
|
|
50
|
+
|
|
51
|
+
```text
|
|
52
|
+
Controlled Institutional Private-Credit Pilot.md +43
|
|
53
|
+
▲4 ▼1 score 8.7 consensus 80% verified
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
AgentJury is framework-independent. The first live integration is Hermes, and the core protocol works with any system that can build a `ReviewRequest`.
|
|
57
|
+
|
|
58
|
+
## Looking for testers
|
|
59
|
+
|
|
60
|
+
AgentJury is in public alpha. I am looking for developers running real agent workflows who are willing to test the jury on completed tasks and report where it fails.
|
|
61
|
+
|
|
62
|
+
Useful feedback includes:
|
|
63
|
+
|
|
64
|
+
- the framework or agent you used
|
|
65
|
+
- the reviewer panel and models
|
|
66
|
+
- the verdict, latency, and approximate cost
|
|
67
|
+
- reviewer disagreements or false findings
|
|
68
|
+
- installation friction and integration problems
|
|
69
|
+
|
|
70
|
+
Open an issue at <https://github.com/madad-rashid/AgentJury/issues>. Please do not post proprietary task content or API keys.
|
|
71
|
+
|
|
72
|
+
## Quick start
|
|
73
|
+
|
|
74
|
+
Install AgentJury from PyPI:
|
|
75
|
+
|
|
76
|
+
```bash
|
|
77
|
+
pip install "agentjury[all]"
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
Set `OPENAI_API_KEY` and `ANTHROPIC_API_KEY` in your environment or a local `.env` file, then review an agent output:
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
agentjury review task.md output.md
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
Example:
|
|
87
|
+
|
|
88
|
+
```text
|
|
89
|
+
▲2 ▼1 score 7.0 consensus 67% diversity 67% jury 3/3 verified
|
|
90
|
+
jury confidence index 35% (heuristic, not a probability)
|
|
91
|
+
|
|
92
|
+
▲ 8 accuracy/openai Sourced figure, drivers accurately characterized.
|
|
93
|
+
▼ 5 critic/anthropic Citation has no year or report; one claim is unsupported.
|
|
94
|
+
▲ 8 executive/openai Concise and decision-ready.
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Choose your own panel:
|
|
98
|
+
|
|
99
|
+
```bash
|
|
100
|
+
agentjury review task.md output.md \
|
|
101
|
+
--panel accuracy:openai,critic:anthropic,evidence:anthropic,executive:openai
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Run `agentjury roles` to see the built-in roles. Every verdict is saved to `.agentjury/verdicts/`.
|
|
105
|
+
|
|
106
|
+
## Architecture
|
|
107
|
+
|
|
108
|
+
```mermaid
|
|
109
|
+
flowchart LR
|
|
110
|
+
A[Agent or framework] --> R[ReviewRequest]
|
|
111
|
+
R --> O[OpenAI judge]
|
|
112
|
+
R --> C[Anthropic judge]
|
|
113
|
+
R --> X[Local or custom judge]
|
|
114
|
+
O --> G[Deterministic aggregator]
|
|
115
|
+
C --> G
|
|
116
|
+
X --> G
|
|
117
|
+
G --> V[Verdict]
|
|
118
|
+
V --> H[Human adjudication]
|
|
119
|
+
H --> P[(Future reviewer reputation)]
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
AgentJury separates generation from verification. Reviewers see the task and output, but never see one another's votes before submitting their own.
|
|
123
|
+
|
|
124
|
+
## Design principles
|
|
125
|
+
|
|
126
|
+
- **Blind review.** Judges do not see other reviewers' opinions before voting.
|
|
127
|
+
- **Deterministic aggregation.** No model acts as a final arbiter.
|
|
128
|
+
- **Provider diversity.** A multi-provider jury cannot verify work from one provider's judges alone.
|
|
129
|
+
- **Strict quorum.** Failed calls and abstentions do not silently become approval.
|
|
130
|
+
- **No unilateral block.** A single reviewer cannot block a task by itself.
|
|
131
|
+
- **Auditable identity.** Requests, runs, reviews, reviewer configurations, and findings each have stable IDs.
|
|
132
|
+
- **Human adjudication.** Individual findings can be graded so reviewer reliability can later be measured from evidence rather than assumed.
|
|
133
|
+
- **Framework independence.** AgentJury reviews work produced elsewhere. It is not another agent framework.
|
|
134
|
+
|
|
135
|
+
## How a verdict is reached
|
|
136
|
+
|
|
137
|
+
Judges vote ▲ approve, ▼ revise, or – abstain. Abstentions are recorded but never counted as approval, and they count against quorum.
|
|
138
|
+
|
|
139
|
+
No single judge can block. `blocked` requires blocking findings from two different providers. One blocking finding downgrades the result to `needs_revision`.
|
|
140
|
+
|
|
141
|
+
A panel needs a quorum of voters, by default a strict majority of requested judges:
|
|
142
|
+
|
|
143
|
+
```text
|
|
144
|
+
1→1, 2→2, 3→2, 4→3, 5→3, 6→4
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
A panel built from several providers must also hear from at least two of them. Otherwise the status is `insufficient_jury` and the votes are informational only.
|
|
148
|
+
|
|
149
|
+
Each judge call has a timeout, one retry on provider error, and one repair round-trip if the reply is not valid JSON. A failed judge is recorded as an error and the rest of the panel continues.
|
|
150
|
+
|
|
151
|
+
Exit codes:
|
|
152
|
+
|
|
153
|
+
```text
|
|
154
|
+
0 verified
|
|
155
|
+
1 needs_revision
|
|
156
|
+
2 blocked
|
|
157
|
+
3 insufficient_jury
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
The confidence figure is a heuristic index, not a calibrated probability. The plan is to calibrate it against human adjudication once enough real data exists.
|
|
161
|
+
|
|
162
|
+
Judges treat everything they review as untrusted data. Instructions hidden inside an agent output are treated as content, not reviewer instructions. `tests/test_adversarial_live.py` attacks the jury with `examples/injected_output.md`; run it with `AGENTJURY_LIVE=1`.
|
|
163
|
+
|
|
164
|
+
## Adjudication
|
|
165
|
+
|
|
166
|
+
Reputation is designed to come from human grading of individual findings, not from treating an entire review as one correct or incorrect event.
|
|
167
|
+
|
|
168
|
+
```bash
|
|
169
|
+
agentjury verdicts --dir <where-verdicts-live>
|
|
170
|
+
agentjury adjudicate 9a9a900dc86b --judge critic/anthropic \
|
|
171
|
+
--finding 1 wrong --finding 2 wrong --finding 3 correct \
|
|
172
|
+
--verdict disagree --note "figure is in the cited source"
|
|
173
|
+
agentjury adjudicate 9a9a900dc86b --producer-verdict correct
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
Findings are numbered as displayed. Grades are written back into the verdict JSON as current state and appended as events to `adjudications.jsonl` in the same folder. The event log records who changed which finding, from what to what, when, and why.
|
|
177
|
+
|
|
178
|
+
Set `AGENTJURY_VERDICT_DIR` to avoid repeating `--dir`.
|
|
179
|
+
|
|
180
|
+
Identity hierarchy:
|
|
181
|
+
|
|
182
|
+
- `request_id`: the work being evaluated
|
|
183
|
+
- `run_id`: one jury execution of that request
|
|
184
|
+
- `review_id`: one judge's opinion
|
|
185
|
+
- `config_id`: the reviewer configuration used for future reputation measurement, including provider, model, role, prompt hash, and relevant parameters
|
|
186
|
+
- `finding.id`: one specific issue raised by a reviewer
|
|
187
|
+
|
|
188
|
+
Verdicts are saved as `<request_id>-<run_id>.json`.
|
|
189
|
+
|
|
190
|
+
## Custom roles
|
|
191
|
+
|
|
192
|
+
Give a jury domain expertise with a JSON file of `{"role_name": "description"}`:
|
|
193
|
+
|
|
194
|
+
```bash
|
|
195
|
+
agentjury review task.md output.md \
|
|
196
|
+
--roles examples/roles.json \
|
|
197
|
+
--panel accuracy:openai,domain_expert:anthropic,executive:openai
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
## Integrations
|
|
201
|
+
|
|
202
|
+
### Hermes Agent
|
|
203
|
+
|
|
204
|
+
`integrations/hermes/` contains the first live integration. It reviews substantial Hermes responses in the background, saves verdicts, writes verdict metadata into markdown frontmatter, and feeds major findings back on the next relevant turn.
|
|
205
|
+
|
|
206
|
+
See [integrations/hermes/README.md](integrations/hermes/README.md) for installation and configuration.
|
|
207
|
+
|
|
208
|
+
Adapters for other agent frameworks are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
209
|
+
|
|
210
|
+
## Protocol
|
|
211
|
+
|
|
212
|
+
Current schema: **0.5**.
|
|
213
|
+
|
|
214
|
+
Print the schemas with:
|
|
215
|
+
|
|
216
|
+
```bash
|
|
217
|
+
agentjury schema request
|
|
218
|
+
agentjury schema verdict
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
The main objects are:
|
|
222
|
+
|
|
223
|
+
- `ReviewRequest`: task, agent output, optional context and artifacts, task type, domain, and producer metadata
|
|
224
|
+
- `Review`: one independent judge opinion with vote, score, reason, findings, IDs, reviewer configuration, telemetry, and adjudication slots
|
|
225
|
+
- `Verdict`: deterministic aggregate with votes, score, consensus, diversity, confidence index, status, and the underlying reviews
|
|
226
|
+
|
|
227
|
+
Every field needed by the planned reputation system is recorded from the first review. Reputation weighting is not active yet.
|
|
228
|
+
|
|
229
|
+
## Development
|
|
230
|
+
|
|
231
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md) for local setup, tests, judge-provider adapters, framework integrations, and pull requests.
|
|
232
|
+
|
|
233
|
+
See [docs/PUBLISHING.md](docs/PUBLISHING.md) for the release and PyPI checklist.
|
|
234
|
+
|
|
235
|
+
## Status
|
|
236
|
+
|
|
237
|
+
Public alpha. The core aggregation rules are intentionally stable while real verdicts are collected through the Hermes integration and direct CLI use.
|
|
238
|
+
|
|
239
|
+
The next research step is reviewer reputation by task type using human-adjudicated findings, followed by diversity weighting from observed disagreement patterns.
|
|
240
|
+
|
|
241
|
+
## Roadmap
|
|
242
|
+
|
|
243
|
+
- [x] Protocol schema
|
|
244
|
+
- [x] Judge interface with OpenAI and Anthropic adapters
|
|
245
|
+
- [x] Deterministic aggregator
|
|
246
|
+
- [x] CLI: `agentjury review task.md output.md`
|
|
247
|
+
- [x] Hermes integration
|
|
248
|
+
- [x] Review-event schema with telemetry and adjudication slots
|
|
249
|
+
- [x] Quorum, non-unilateral blocking, prompt-injection defence, custom roles
|
|
250
|
+
- [x] Abstain vote, provider floor, retry, repair, timeouts, CI
|
|
251
|
+
- [x] Human finding-level adjudication and append-only adjudication history
|
|
252
|
+
- [x] PyPI release
|
|
253
|
+
- [ ] Additional judge providers and local-model adapter
|
|
254
|
+
- [ ] Reviewer reputation by task type, weighted by human agreement over time
|
|
255
|
+
- [ ] Jury diversity weighting from historical disagreement
|
|
256
|
+
- [ ] Calibrated confidence from observed outcomes
|
|
257
|
+
|
|
258
|
+
## License
|
|
259
|
+
|
|
260
|
+
MIT
|
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
# AgentJury
|
|
2
|
+
|
|
3
|
+
[](https://github.com/madad-rashid/AgentJury/actions/workflows/tests.yml)
|
|
4
|
+
[](https://www.python.org/)
|
|
5
|
+
[](LICENSE)
|
|
6
|
+
|
|
7
|
+
**Peer review for AI agents.**
|
|
8
|
+
|
|
9
|
+
Your agent says the task is finished. AgentJury asks independent, blind AI reviewers whether the work is good enough before you trust it.
|
|
10
|
+
|
|
11
|
+
Each reviewer votes ▲ approve, ▼ revise, or – abstain. AgentJury combines those opinions with deterministic rules. No final LLM gets a deciding vote.
|
|
12
|
+
|
|
13
|
+
```text
|
|
14
|
+
Controlled Institutional Private-Credit Pilot.md +43
|
|
15
|
+
▲4 ▼1 score 8.7 consensus 80% verified
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
AgentJury is framework-independent. The first live integration is Hermes, and the core protocol works with any system that can build a `ReviewRequest`.
|
|
19
|
+
|
|
20
|
+
## Looking for testers
|
|
21
|
+
|
|
22
|
+
AgentJury is in public alpha. I am looking for developers running real agent workflows who are willing to test the jury on completed tasks and report where it fails.
|
|
23
|
+
|
|
24
|
+
Useful feedback includes:
|
|
25
|
+
|
|
26
|
+
- the framework or agent you used
|
|
27
|
+
- the reviewer panel and models
|
|
28
|
+
- the verdict, latency, and approximate cost
|
|
29
|
+
- reviewer disagreements or false findings
|
|
30
|
+
- installation friction and integration problems
|
|
31
|
+
|
|
32
|
+
Open an issue at <https://github.com/madad-rashid/AgentJury/issues>. Please do not post proprietary task content or API keys.
|
|
33
|
+
|
|
34
|
+
## Quick start
|
|
35
|
+
|
|
36
|
+
Install AgentJury from PyPI:
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
pip install "agentjury[all]"
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Set `OPENAI_API_KEY` and `ANTHROPIC_API_KEY` in your environment or a local `.env` file, then review an agent output:
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
agentjury review task.md output.md
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
Example:
|
|
49
|
+
|
|
50
|
+
```text
|
|
51
|
+
▲2 ▼1 score 7.0 consensus 67% diversity 67% jury 3/3 verified
|
|
52
|
+
jury confidence index 35% (heuristic, not a probability)
|
|
53
|
+
|
|
54
|
+
▲ 8 accuracy/openai Sourced figure, drivers accurately characterized.
|
|
55
|
+
▼ 5 critic/anthropic Citation has no year or report; one claim is unsupported.
|
|
56
|
+
▲ 8 executive/openai Concise and decision-ready.
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Choose your own panel:
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
agentjury review task.md output.md \
|
|
63
|
+
--panel accuracy:openai,critic:anthropic,evidence:anthropic,executive:openai
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Run `agentjury roles` to see the built-in roles. Every verdict is saved to `.agentjury/verdicts/`.
|
|
67
|
+
|
|
68
|
+
## Architecture
|
|
69
|
+
|
|
70
|
+
```mermaid
|
|
71
|
+
flowchart LR
|
|
72
|
+
A[Agent or framework] --> R[ReviewRequest]
|
|
73
|
+
R --> O[OpenAI judge]
|
|
74
|
+
R --> C[Anthropic judge]
|
|
75
|
+
R --> X[Local or custom judge]
|
|
76
|
+
O --> G[Deterministic aggregator]
|
|
77
|
+
C --> G
|
|
78
|
+
X --> G
|
|
79
|
+
G --> V[Verdict]
|
|
80
|
+
V --> H[Human adjudication]
|
|
81
|
+
H --> P[(Future reviewer reputation)]
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
AgentJury separates generation from verification. Reviewers see the task and output, but never see one another's votes before submitting their own.
|
|
85
|
+
|
|
86
|
+
## Design principles
|
|
87
|
+
|
|
88
|
+
- **Blind review.** Judges do not see other reviewers' opinions before voting.
|
|
89
|
+
- **Deterministic aggregation.** No model acts as a final arbiter.
|
|
90
|
+
- **Provider diversity.** A multi-provider jury cannot verify work from one provider's judges alone.
|
|
91
|
+
- **Strict quorum.** Failed calls and abstentions do not silently become approval.
|
|
92
|
+
- **No unilateral block.** A single reviewer cannot block a task by itself.
|
|
93
|
+
- **Auditable identity.** Requests, runs, reviews, reviewer configurations, and findings each have stable IDs.
|
|
94
|
+
- **Human adjudication.** Individual findings can be graded so reviewer reliability can later be measured from evidence rather than assumed.
|
|
95
|
+
- **Framework independence.** AgentJury reviews work produced elsewhere. It is not another agent framework.
|
|
96
|
+
|
|
97
|
+
## How a verdict is reached
|
|
98
|
+
|
|
99
|
+
Judges vote ▲ approve, ▼ revise, or – abstain. Abstentions are recorded but never counted as approval, and they count against quorum.
|
|
100
|
+
|
|
101
|
+
No single judge can block. `blocked` requires blocking findings from two different providers. One blocking finding downgrades the result to `needs_revision`.
|
|
102
|
+
|
|
103
|
+
A panel needs a quorum of voters, by default a strict majority of requested judges:
|
|
104
|
+
|
|
105
|
+
```text
|
|
106
|
+
1→1, 2→2, 3→2, 4→3, 5→3, 6→4
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
A panel built from several providers must also hear from at least two of them. Otherwise the status is `insufficient_jury` and the votes are informational only.
|
|
110
|
+
|
|
111
|
+
Each judge call has a timeout, one retry on provider error, and one repair round-trip if the reply is not valid JSON. A failed judge is recorded as an error and the rest of the panel continues.
|
|
112
|
+
|
|
113
|
+
Exit codes:
|
|
114
|
+
|
|
115
|
+
```text
|
|
116
|
+
0 verified
|
|
117
|
+
1 needs_revision
|
|
118
|
+
2 blocked
|
|
119
|
+
3 insufficient_jury
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
The confidence figure is a heuristic index, not a calibrated probability. The plan is to calibrate it against human adjudication once enough real data exists.
|
|
123
|
+
|
|
124
|
+
Judges treat everything they review as untrusted data. Instructions hidden inside an agent output are treated as content, not reviewer instructions. `tests/test_adversarial_live.py` attacks the jury with `examples/injected_output.md`; run it with `AGENTJURY_LIVE=1`.
|
|
125
|
+
|
|
126
|
+
## Adjudication
|
|
127
|
+
|
|
128
|
+
Reputation is designed to come from human grading of individual findings, not from treating an entire review as one correct or incorrect event.
|
|
129
|
+
|
|
130
|
+
```bash
|
|
131
|
+
agentjury verdicts --dir <where-verdicts-live>
|
|
132
|
+
agentjury adjudicate 9a9a900dc86b --judge critic/anthropic \
|
|
133
|
+
--finding 1 wrong --finding 2 wrong --finding 3 correct \
|
|
134
|
+
--verdict disagree --note "figure is in the cited source"
|
|
135
|
+
agentjury adjudicate 9a9a900dc86b --producer-verdict correct
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
Findings are numbered as displayed. Grades are written back into the verdict JSON as current state and appended as events to `adjudications.jsonl` in the same folder. The event log records who changed which finding, from what to what, when, and why.
|
|
139
|
+
|
|
140
|
+
Set `AGENTJURY_VERDICT_DIR` to avoid repeating `--dir`.
|
|
141
|
+
|
|
142
|
+
Identity hierarchy:
|
|
143
|
+
|
|
144
|
+
- `request_id`: the work being evaluated
|
|
145
|
+
- `run_id`: one jury execution of that request
|
|
146
|
+
- `review_id`: one judge's opinion
|
|
147
|
+
- `config_id`: the reviewer configuration used for future reputation measurement, including provider, model, role, prompt hash, and relevant parameters
|
|
148
|
+
- `finding.id`: one specific issue raised by a reviewer
|
|
149
|
+
|
|
150
|
+
Verdicts are saved as `<request_id>-<run_id>.json`.
|
|
151
|
+
|
|
152
|
+
## Custom roles
|
|
153
|
+
|
|
154
|
+
Give a jury domain expertise with a JSON file of `{"role_name": "description"}`:
|
|
155
|
+
|
|
156
|
+
```bash
|
|
157
|
+
agentjury review task.md output.md \
|
|
158
|
+
--roles examples/roles.json \
|
|
159
|
+
--panel accuracy:openai,domain_expert:anthropic,executive:openai
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
## Integrations
|
|
163
|
+
|
|
164
|
+
### Hermes Agent
|
|
165
|
+
|
|
166
|
+
`integrations/hermes/` contains the first live integration. It reviews substantial Hermes responses in the background, saves verdicts, writes verdict metadata into markdown frontmatter, and feeds major findings back on the next relevant turn.
|
|
167
|
+
|
|
168
|
+
See [integrations/hermes/README.md](integrations/hermes/README.md) for installation and configuration.
|
|
169
|
+
|
|
170
|
+
Adapters for other agent frameworks are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
171
|
+
|
|
172
|
+
## Protocol
|
|
173
|
+
|
|
174
|
+
Current schema: **0.5**.
|
|
175
|
+
|
|
176
|
+
Print the schemas with:
|
|
177
|
+
|
|
178
|
+
```bash
|
|
179
|
+
agentjury schema request
|
|
180
|
+
agentjury schema verdict
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
The main objects are:
|
|
184
|
+
|
|
185
|
+
- `ReviewRequest`: task, agent output, optional context and artifacts, task type, domain, and producer metadata
|
|
186
|
+
- `Review`: one independent judge opinion with vote, score, reason, findings, IDs, reviewer configuration, telemetry, and adjudication slots
|
|
187
|
+
- `Verdict`: deterministic aggregate with votes, score, consensus, diversity, confidence index, status, and the underlying reviews
|
|
188
|
+
|
|
189
|
+
Every field needed by the planned reputation system is recorded from the first review. Reputation weighting is not active yet.
|
|
190
|
+
|
|
191
|
+
## Development
|
|
192
|
+
|
|
193
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md) for local setup, tests, judge-provider adapters, framework integrations, and pull requests.
|
|
194
|
+
|
|
195
|
+
See [docs/PUBLISHING.md](docs/PUBLISHING.md) for the release and PyPI checklist.
|
|
196
|
+
|
|
197
|
+
## Status
|
|
198
|
+
|
|
199
|
+
Public alpha. The core aggregation rules are intentionally stable while real verdicts are collected through the Hermes integration and direct CLI use.
|
|
200
|
+
|
|
201
|
+
The next research step is reviewer reputation by task type using human-adjudicated findings, followed by diversity weighting from observed disagreement patterns.
|
|
202
|
+
|
|
203
|
+
## Roadmap
|
|
204
|
+
|
|
205
|
+
- [x] Protocol schema
|
|
206
|
+
- [x] Judge interface with OpenAI and Anthropic adapters
|
|
207
|
+
- [x] Deterministic aggregator
|
|
208
|
+
- [x] CLI: `agentjury review task.md output.md`
|
|
209
|
+
- [x] Hermes integration
|
|
210
|
+
- [x] Review-event schema with telemetry and adjudication slots
|
|
211
|
+
- [x] Quorum, non-unilateral blocking, prompt-injection defence, custom roles
|
|
212
|
+
- [x] Abstain vote, provider floor, retry, repair, timeouts, CI
|
|
213
|
+
- [x] Human finding-level adjudication and append-only adjudication history
|
|
214
|
+
- [x] PyPI release
|
|
215
|
+
- [ ] Additional judge providers and local-model adapter
|
|
216
|
+
- [ ] Reviewer reputation by task type, weighted by human agreement over time
|
|
217
|
+
- [ ] Jury diversity weighting from historical disagreement
|
|
218
|
+
- [ ] Calibrated confidence from observed outcomes
|
|
219
|
+
|
|
220
|
+
## License
|
|
221
|
+
|
|
222
|
+
MIT
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""AgentJury: peer review for AI agents."""
|
|
2
|
+
|
|
3
|
+
from .aggregate import aggregate
|
|
4
|
+
from .panel import Panel
|
|
5
|
+
from .protocol import (
|
|
6
|
+
SCHEMA_VERSION,
|
|
7
|
+
Artifact,
|
|
8
|
+
Finding,
|
|
9
|
+
HumanReview,
|
|
10
|
+
Producer,
|
|
11
|
+
Review,
|
|
12
|
+
ReviewRequest,
|
|
13
|
+
Verdict,
|
|
14
|
+
Vote,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
__version__ = "0.4.4"
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"SCHEMA_VERSION", "Artifact", "Finding", "HumanReview", "Producer",
|
|
21
|
+
"Review", "ReviewRequest", "Verdict", "Vote", "Panel", "aggregate",
|
|
22
|
+
]
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Turn a list of independent Reviews into one Verdict.
|
|
3
|
+
|
|
4
|
+
No LLM calls here. The rules are written down so anyone can predict the verdict:
|
|
5
|
+
|
|
6
|
+
voters responding judges who did not abstain
|
|
7
|
+
up / down count of approve / revise votes among voters
|
|
8
|
+
score mean of voters' scores
|
|
9
|
+
consensus share of voters who sided with the majority vote
|
|
10
|
+
diversity distinct providers among voters / voters
|
|
11
|
+
confidence heuristic index: consensus, discounted for small panels, wide
|
|
12
|
+
score spread, and low diversity. NOT a calibrated probability.
|
|
13
|
+
|
|
14
|
+
status
|
|
15
|
+
insufficient_jury fewer than `quorum` judges voted, OR the panel was built
|
|
16
|
+
from two or more providers but only one provider's judges
|
|
17
|
+
voted. Votes are reported; no verdict is reached.
|
|
18
|
+
blocked blocking findings from at least two providers (or from at
|
|
19
|
+
least two judges when the panel has only one provider).
|
|
20
|
+
No single judge can block on its own.
|
|
21
|
+
needs_revision majority revise, a tie, or exactly one blocking source.
|
|
22
|
+
verified majority approve and no blocking finding.
|
|
23
|
+
|
|
24
|
+
quorum default is a strict majority of requested judges:
|
|
25
|
+
1->1, 2->2, 3->2, 4->3, 5->3, 6->4
|
|
26
|
+
|
|
27
|
+
Reviewer reputation will later weight these votes. For now every voter counts once.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
from .protocol import Review, ReviewRequest, Verdict, Vote
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def default_quorum(requested: int) -> int:
|
|
36
|
+
"""Strict majority of the requested panel."""
|
|
37
|
+
return max(1, requested // 2 + 1)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _empty(request: ReviewRequest, reviews, errors, requested, quorum, panel_id) -> Verdict:
|
|
41
|
+
return Verdict(
|
|
42
|
+
request_id=request.request_id, panel_id=panel_id,
|
|
43
|
+
requested=requested, responded=len(reviews), abstained=len(reviews), quorum=quorum,
|
|
44
|
+
task_type=request.task_type, domain=request.domain, producer=request.producer,
|
|
45
|
+
up=0, down=0, score=0.0, consensus=0.0, diversity=0.0, confidence=0.0,
|
|
46
|
+
status="insufficient_jury", reviews=reviews, errors=errors or [],
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def aggregate(
|
|
51
|
+
request: ReviewRequest,
|
|
52
|
+
reviews: list[Review],
|
|
53
|
+
errors: list[str] | None = None,
|
|
54
|
+
*,
|
|
55
|
+
requested: int | None = None,
|
|
56
|
+
quorum: int | None = None,
|
|
57
|
+
panel_id: str | None = None,
|
|
58
|
+
requested_providers: int | None = None,
|
|
59
|
+
) -> Verdict:
|
|
60
|
+
requested = requested if requested is not None else len(reviews)
|
|
61
|
+
quorum = quorum if quorum is not None else default_quorum(requested)
|
|
62
|
+
|
|
63
|
+
voters = [r for r in reviews if r.vote != Vote.ABSTAIN]
|
|
64
|
+
abstained = len(reviews) - len(voters)
|
|
65
|
+
if not voters:
|
|
66
|
+
return _empty(request, reviews, errors, requested, quorum, panel_id)
|
|
67
|
+
|
|
68
|
+
n = len(voters)
|
|
69
|
+
up = sum(1 for r in voters if r.vote == Vote.APPROVE)
|
|
70
|
+
down = n - up
|
|
71
|
+
scores = [r.score for r in voters]
|
|
72
|
+
score = sum(scores) / n
|
|
73
|
+
|
|
74
|
+
consensus = max(up, down) / n
|
|
75
|
+
providers = {r.provider for r in voters}
|
|
76
|
+
diversity = len(providers) / n
|
|
77
|
+
|
|
78
|
+
panel_factor = n / (n + 1) # 1 voter -> 0.5, 3 -> 0.75, 5 -> 0.83
|
|
79
|
+
spread_factor = 1 - (max(scores) - min(scores)) / 10 # identical scores -> 1.0
|
|
80
|
+
diversity_factor = 0.5 + 0.5 * diversity # all one provider (n=3) -> 0.67; all distinct -> 1.0
|
|
81
|
+
confidence = round(consensus * panel_factor * spread_factor * diversity_factor, 3)
|
|
82
|
+
|
|
83
|
+
blocking_reviews = [r for r in voters if r.blocking]
|
|
84
|
+
blocking_providers = {r.provider for r in blocking_reviews}
|
|
85
|
+
independent_blocks = len(blocking_providers) >= 2 or (len(providers) == 1 and len(blocking_reviews) >= 2)
|
|
86
|
+
|
|
87
|
+
# A multi-provider panel that only heard from one provider has lost the
|
|
88
|
+
# independence it was built for, so it cannot reach a verdict.
|
|
89
|
+
wanted_providers = requested_providers if requested_providers is not None else len(providers)
|
|
90
|
+
provider_floor = min(2, wanted_providers)
|
|
91
|
+
|
|
92
|
+
if n < quorum or len(providers) < provider_floor:
|
|
93
|
+
status = "insufficient_jury"
|
|
94
|
+
elif independent_blocks:
|
|
95
|
+
status = "blocked"
|
|
96
|
+
elif blocking_reviews or up <= down:
|
|
97
|
+
status = "needs_revision"
|
|
98
|
+
else:
|
|
99
|
+
status = "verified"
|
|
100
|
+
|
|
101
|
+
return Verdict(
|
|
102
|
+
request_id=request.request_id,
|
|
103
|
+
panel_id=panel_id,
|
|
104
|
+
requested=requested,
|
|
105
|
+
responded=len(reviews),
|
|
106
|
+
abstained=abstained,
|
|
107
|
+
quorum=quorum,
|
|
108
|
+
task_type=request.task_type,
|
|
109
|
+
domain=request.domain,
|
|
110
|
+
producer=request.producer,
|
|
111
|
+
up=up,
|
|
112
|
+
down=down,
|
|
113
|
+
score=round(score, 2),
|
|
114
|
+
consensus=round(consensus, 3),
|
|
115
|
+
diversity=round(diversity, 3),
|
|
116
|
+
confidence=confidence,
|
|
117
|
+
status=status,
|
|
118
|
+
reviews=reviews,
|
|
119
|
+
errors=errors or [],
|
|
120
|
+
)
|