agentjury 0.4.4__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentjury/__init__.py +22 -0
- agentjury/aggregate.py +120 -0
- agentjury/cli.py +312 -0
- agentjury/judges/__init__.py +14 -0
- agentjury/judges/anthropic_judge.py +75 -0
- agentjury/judges/base.py +268 -0
- agentjury/judges/fake.py +42 -0
- agentjury/judges/openai_judge.py +37 -0
- agentjury/panel.py +61 -0
- agentjury/protocol.py +230 -0
- agentjury-0.4.4.dist-info/METADATA +260 -0
- agentjury-0.4.4.dist-info/RECORD +16 -0
- agentjury-0.4.4.dist-info/WHEEL +5 -0
- agentjury-0.4.4.dist-info/entry_points.txt +2 -0
- agentjury-0.4.4.dist-info/licenses/LICENSE +21 -0
- agentjury-0.4.4.dist-info/top_level.txt +1 -0
agentjury/__init__.py
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""AgentJury: peer review for AI agents."""
|
|
2
|
+
|
|
3
|
+
from .aggregate import aggregate
|
|
4
|
+
from .panel import Panel
|
|
5
|
+
from .protocol import (
|
|
6
|
+
SCHEMA_VERSION,
|
|
7
|
+
Artifact,
|
|
8
|
+
Finding,
|
|
9
|
+
HumanReview,
|
|
10
|
+
Producer,
|
|
11
|
+
Review,
|
|
12
|
+
ReviewRequest,
|
|
13
|
+
Verdict,
|
|
14
|
+
Vote,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
__version__ = "0.4.4"
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"SCHEMA_VERSION", "Artifact", "Finding", "HumanReview", "Producer",
|
|
21
|
+
"Review", "ReviewRequest", "Verdict", "Vote", "Panel", "aggregate",
|
|
22
|
+
]
|
agentjury/aggregate.py
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Turn a list of independent Reviews into one Verdict.
|
|
3
|
+
|
|
4
|
+
No LLM calls here. The rules are written down so anyone can predict the verdict:
|
|
5
|
+
|
|
6
|
+
voters responding judges who did not abstain
|
|
7
|
+
up / down count of approve / revise votes among voters
|
|
8
|
+
score mean of voters' scores
|
|
9
|
+
consensus share of voters who sided with the majority vote
|
|
10
|
+
diversity distinct providers among voters / voters
|
|
11
|
+
confidence heuristic index: consensus, discounted for small panels, wide
|
|
12
|
+
score spread, and low diversity. NOT a calibrated probability.
|
|
13
|
+
|
|
14
|
+
status
|
|
15
|
+
insufficient_jury fewer than `quorum` judges voted, OR the panel was built
|
|
16
|
+
from two or more providers but only one provider's judges
|
|
17
|
+
voted. Votes are reported; no verdict is reached.
|
|
18
|
+
blocked blocking findings from at least two providers (or from at
|
|
19
|
+
least two judges when the panel has only one provider).
|
|
20
|
+
No single judge can block on its own.
|
|
21
|
+
needs_revision majority revise, a tie, or exactly one blocking source.
|
|
22
|
+
verified majority approve and no blocking finding.
|
|
23
|
+
|
|
24
|
+
quorum default is a strict majority of requested judges:
|
|
25
|
+
1->1, 2->2, 3->2, 4->3, 5->3, 6->4
|
|
26
|
+
|
|
27
|
+
Reviewer reputation will later weight these votes. For now every voter counts once.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
from .protocol import Review, ReviewRequest, Verdict, Vote
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def default_quorum(requested: int) -> int:
|
|
36
|
+
"""Strict majority of the requested panel."""
|
|
37
|
+
return max(1, requested // 2 + 1)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _empty(request: ReviewRequest, reviews, errors, requested, quorum, panel_id) -> Verdict:
|
|
41
|
+
return Verdict(
|
|
42
|
+
request_id=request.request_id, panel_id=panel_id,
|
|
43
|
+
requested=requested, responded=len(reviews), abstained=len(reviews), quorum=quorum,
|
|
44
|
+
task_type=request.task_type, domain=request.domain, producer=request.producer,
|
|
45
|
+
up=0, down=0, score=0.0, consensus=0.0, diversity=0.0, confidence=0.0,
|
|
46
|
+
status="insufficient_jury", reviews=reviews, errors=errors or [],
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def aggregate(
|
|
51
|
+
request: ReviewRequest,
|
|
52
|
+
reviews: list[Review],
|
|
53
|
+
errors: list[str] | None = None,
|
|
54
|
+
*,
|
|
55
|
+
requested: int | None = None,
|
|
56
|
+
quorum: int | None = None,
|
|
57
|
+
panel_id: str | None = None,
|
|
58
|
+
requested_providers: int | None = None,
|
|
59
|
+
) -> Verdict:
|
|
60
|
+
requested = requested if requested is not None else len(reviews)
|
|
61
|
+
quorum = quorum if quorum is not None else default_quorum(requested)
|
|
62
|
+
|
|
63
|
+
voters = [r for r in reviews if r.vote != Vote.ABSTAIN]
|
|
64
|
+
abstained = len(reviews) - len(voters)
|
|
65
|
+
if not voters:
|
|
66
|
+
return _empty(request, reviews, errors, requested, quorum, panel_id)
|
|
67
|
+
|
|
68
|
+
n = len(voters)
|
|
69
|
+
up = sum(1 for r in voters if r.vote == Vote.APPROVE)
|
|
70
|
+
down = n - up
|
|
71
|
+
scores = [r.score for r in voters]
|
|
72
|
+
score = sum(scores) / n
|
|
73
|
+
|
|
74
|
+
consensus = max(up, down) / n
|
|
75
|
+
providers = {r.provider for r in voters}
|
|
76
|
+
diversity = len(providers) / n
|
|
77
|
+
|
|
78
|
+
panel_factor = n / (n + 1) # 1 voter -> 0.5, 3 -> 0.75, 5 -> 0.83
|
|
79
|
+
spread_factor = 1 - (max(scores) - min(scores)) / 10 # identical scores -> 1.0
|
|
80
|
+
diversity_factor = 0.5 + 0.5 * diversity # all one provider (n=3) -> 0.67; all distinct -> 1.0
|
|
81
|
+
confidence = round(consensus * panel_factor * spread_factor * diversity_factor, 3)
|
|
82
|
+
|
|
83
|
+
blocking_reviews = [r for r in voters if r.blocking]
|
|
84
|
+
blocking_providers = {r.provider for r in blocking_reviews}
|
|
85
|
+
independent_blocks = len(blocking_providers) >= 2 or (len(providers) == 1 and len(blocking_reviews) >= 2)
|
|
86
|
+
|
|
87
|
+
# A multi-provider panel that only heard from one provider has lost the
|
|
88
|
+
# independence it was built for, so it cannot reach a verdict.
|
|
89
|
+
wanted_providers = requested_providers if requested_providers is not None else len(providers)
|
|
90
|
+
provider_floor = min(2, wanted_providers)
|
|
91
|
+
|
|
92
|
+
if n < quorum or len(providers) < provider_floor:
|
|
93
|
+
status = "insufficient_jury"
|
|
94
|
+
elif independent_blocks:
|
|
95
|
+
status = "blocked"
|
|
96
|
+
elif blocking_reviews or up <= down:
|
|
97
|
+
status = "needs_revision"
|
|
98
|
+
else:
|
|
99
|
+
status = "verified"
|
|
100
|
+
|
|
101
|
+
return Verdict(
|
|
102
|
+
request_id=request.request_id,
|
|
103
|
+
panel_id=panel_id,
|
|
104
|
+
requested=requested,
|
|
105
|
+
responded=len(reviews),
|
|
106
|
+
abstained=abstained,
|
|
107
|
+
quorum=quorum,
|
|
108
|
+
task_type=request.task_type,
|
|
109
|
+
domain=request.domain,
|
|
110
|
+
producer=request.producer,
|
|
111
|
+
up=up,
|
|
112
|
+
down=down,
|
|
113
|
+
score=round(score, 2),
|
|
114
|
+
consensus=round(consensus, 3),
|
|
115
|
+
diversity=round(diversity, 3),
|
|
116
|
+
confidence=confidence,
|
|
117
|
+
status=status,
|
|
118
|
+
reviews=reviews,
|
|
119
|
+
errors=errors or [],
|
|
120
|
+
)
|
agentjury/cli.py
ADDED
|
@@ -0,0 +1,312 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Command-line interface.
|
|
3
|
+
|
|
4
|
+
agentjury review TASK OUTPUT [--panel SPEC] [--roles FILE] [--quorum N] [--task-type T] [--domain D] [--json]
|
|
5
|
+
agentjury roles [--roles FILE]
|
|
6
|
+
agentjury schema [request|verdict]
|
|
7
|
+
agentjury verdicts [--dir DIR] [-n N]
|
|
8
|
+
agentjury adjudicate ID [--judge J] [--finding N LABEL]... [--verdict agree|partial|disagree]
|
|
9
|
+
[--producer-verdict correct|flawed] [--note TEXT] [--dir DIR]
|
|
10
|
+
|
|
11
|
+
Verdicts are read from --dir, else $AGENTJURY_VERDICT_DIR, else .agentjury/verdicts.
|
|
12
|
+
|
|
13
|
+
Exit codes: 0 verified, 1 needs_revision, 2 blocked, 3 insufficient_jury.
|
|
14
|
+
|
|
15
|
+
TASK and OUTPUT are files (or "-" to read OUTPUT from stdin).
|
|
16
|
+
PANEL is a comma-separated list of role:provider pairs, for example
|
|
17
|
+
accuracy:openai,critic:anthropic,executive:openai
|
|
18
|
+
Every verdict is saved to .agentjury/verdicts/<request_id>.json so that
|
|
19
|
+
reviews accumulate over time.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import argparse
|
|
25
|
+
import json
|
|
26
|
+
import os
|
|
27
|
+
import sys
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
|
|
30
|
+
from dotenv import load_dotenv
|
|
31
|
+
|
|
32
|
+
from .judges import ROLES, anthropic_judge, load_roles, openai_judge
|
|
33
|
+
from .judges.base import Judge
|
|
34
|
+
from .panel import Panel
|
|
35
|
+
from .protocol import HumanReview, Producer, ReviewRequest, Verdict
|
|
36
|
+
|
|
37
|
+
DEFAULT_PANEL = "accuracy:openai,critic:anthropic,executive:openai"
|
|
38
|
+
VERDICT_DIR = Path(".agentjury") / "verdicts"
|
|
39
|
+
|
|
40
|
+
PROVIDERS = {
|
|
41
|
+
"openai": openai_judge,
|
|
42
|
+
"anthropic": anthropic_judge,
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
SEVERITY_MARK = {"minor": "-", "major": "!", "blocking": "X"}
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
# Exit codes, so shell scripts and CI can branch without parsing output.
|
|
49
|
+
EXIT = {"verified": 0, "needs_revision": 1, "blocked": 2, "insufficient_jury": 3}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def build_panel(spec: str, quorum: int | None = None) -> Panel:
|
|
53
|
+
judges: list[Judge] = []
|
|
54
|
+
for item in spec.split(","):
|
|
55
|
+
item = item.strip()
|
|
56
|
+
if not item:
|
|
57
|
+
continue
|
|
58
|
+
try:
|
|
59
|
+
role, provider = item.split(":")
|
|
60
|
+
except ValueError:
|
|
61
|
+
sys.exit(f"Bad panel entry {item!r}. Use role:provider, e.g. critic:anthropic")
|
|
62
|
+
if role not in ROLES:
|
|
63
|
+
sys.exit(f"Unknown role {role!r}. Known roles: {', '.join(sorted(ROLES))}")
|
|
64
|
+
if provider not in PROVIDERS:
|
|
65
|
+
sys.exit(f"Unknown provider {provider!r}. Known providers: {', '.join(PROVIDERS)}")
|
|
66
|
+
judges.append(PROVIDERS[provider](role))
|
|
67
|
+
return Panel(judges, quorum=quorum)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def read(path: str) -> str:
|
|
71
|
+
if path == "-":
|
|
72
|
+
return sys.stdin.read()
|
|
73
|
+
return Path(path).read_text(encoding="utf-8")
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def save(verdict: Verdict) -> Path:
|
|
77
|
+
VERDICT_DIR.mkdir(parents=True, exist_ok=True)
|
|
78
|
+
out = VERDICT_DIR / verdict.filename
|
|
79
|
+
out.write_text(verdict.model_dump_json(indent=2), encoding="utf-8")
|
|
80
|
+
return out
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def print_verdict(verdict: Verdict) -> None:
|
|
84
|
+
print(verdict.render())
|
|
85
|
+
print(f"jury confidence index {verdict.confidence:.0%} (heuristic, not a probability)")
|
|
86
|
+
if verdict.status == "insufficient_jury":
|
|
87
|
+
voters = verdict.responded - verdict.abstained
|
|
88
|
+
providers = len({r.provider for r in verdict.reviews if r.vote != "abstain"})
|
|
89
|
+
print(f"Insufficient jury: {voters} of {verdict.requested} judges voted (quorum {verdict.quorum}), "
|
|
90
|
+
f"from {providers} provider(s). No verdict.")
|
|
91
|
+
print()
|
|
92
|
+
for r in verdict.reviews:
|
|
93
|
+
arrow = {"approve": "▲", "revise": "▼", "abstain": "–"}[r.vote]
|
|
94
|
+
meta = f"{r.latency_ms / 1000:.1f}s" if r.latency_ms is not None else ""
|
|
95
|
+
print(f"{arrow} {r.score:>2.0f} {r.judge:<22} {r.reason} [{meta}]")
|
|
96
|
+
for f in r.findings:
|
|
97
|
+
print(f" {SEVERITY_MARK[f.severity]} {f.text}")
|
|
98
|
+
for e in verdict.errors:
|
|
99
|
+
print(f"! {e}")
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def cmd_review(args: argparse.Namespace) -> int:
|
|
103
|
+
if args.roles:
|
|
104
|
+
load_roles(args.roles)
|
|
105
|
+
request = ReviewRequest(
|
|
106
|
+
task=read(args.task),
|
|
107
|
+
output=read(args.output),
|
|
108
|
+
context=read(args.context) if args.context else None,
|
|
109
|
+
task_type=args.task_type,
|
|
110
|
+
domain=args.domain,
|
|
111
|
+
producer=Producer(
|
|
112
|
+
agent=args.agent,
|
|
113
|
+
framework=args.framework,
|
|
114
|
+
provider=args.producer_provider,
|
|
115
|
+
model=args.producer_model,
|
|
116
|
+
),
|
|
117
|
+
)
|
|
118
|
+
verdict = build_panel(args.panel, quorum=args.quorum).review(request)
|
|
119
|
+
|
|
120
|
+
if args.json:
|
|
121
|
+
print(verdict.model_dump_json(indent=2))
|
|
122
|
+
else:
|
|
123
|
+
print_verdict(verdict)
|
|
124
|
+
|
|
125
|
+
if not args.no_save:
|
|
126
|
+
path = save(verdict)
|
|
127
|
+
if not args.json:
|
|
128
|
+
print(f"\nsaved {path}")
|
|
129
|
+
|
|
130
|
+
return EXIT[verdict.status]
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def verdict_dir(args: argparse.Namespace) -> Path:
|
|
134
|
+
return Path(getattr(args, "dir", None) or os.environ.get("AGENTJURY_VERDICT_DIR") or VERDICT_DIR)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def load_verdict(args: argparse.Namespace) -> tuple[Path, Verdict]:
|
|
138
|
+
ref = args.request_id
|
|
139
|
+
path = Path(ref)
|
|
140
|
+
if not path.is_file():
|
|
141
|
+
d = verdict_dir(args)
|
|
142
|
+
candidates = sorted(f for f in d.glob("*.json") if ref in f.stem) if d.is_dir() else []
|
|
143
|
+
if len(candidates) != 1:
|
|
144
|
+
hint = f"{len(candidates)} matches" if candidates else "no match"
|
|
145
|
+
sys.exit(f"Cannot find verdict {ref!r} in {d} ({hint}). Use `agentjury verdicts --dir {d}` to list.")
|
|
146
|
+
path = candidates[0]
|
|
147
|
+
return path, Verdict.model_validate_json(path.read_text(encoding="utf-8"))
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def cmd_verdicts(args: argparse.Namespace) -> int:
|
|
151
|
+
d = verdict_dir(args)
|
|
152
|
+
files = sorted(d.glob("*.json"), key=lambda f: f.stat().st_mtime, reverse=True) if d.is_dir() else []
|
|
153
|
+
if not files:
|
|
154
|
+
print(f"No verdicts in {d}")
|
|
155
|
+
return 0
|
|
156
|
+
print(f"{d}\n")
|
|
157
|
+
for f in files[: args.n]:
|
|
158
|
+
v = Verdict.model_validate_json(f.read_text(encoding="utf-8"))
|
|
159
|
+
graded = sum(1 for r in v.reviews for fi in r.findings if fi.adjudication)
|
|
160
|
+
total = sum(len(r.findings) for r in v.reviews)
|
|
161
|
+
mark = f" [adjudicated {graded}/{total}]" if graded else ""
|
|
162
|
+
prod = f" producer:{v.human_verdict}" if v.human_verdict else ""
|
|
163
|
+
print(f"run {v.run_id} req {v.request_id} {v.created_at:%Y-%m-%d %H:%M} {v.render()}{mark}{prod}")
|
|
164
|
+
return 0
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
ADJUDICATION_LOG = "adjudications.jsonl"
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _adjudicator() -> str:
|
|
171
|
+
import getpass
|
|
172
|
+
return os.environ.get("AGENTJURY_ADJUDICATOR") or getpass.getuser()
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def log_event(directory: Path, event: dict) -> None:
|
|
176
|
+
"""Append-only audit trail. The verdict JSON holds current state; this holds history."""
|
|
177
|
+
from uuid import uuid4
|
|
178
|
+
event = {"event_id": uuid4().hex[:12], **event}
|
|
179
|
+
with (directory / ADJUDICATION_LOG).open("a", encoding="utf-8") as f:
|
|
180
|
+
f.write(json.dumps(event, default=str) + "\n")
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def cmd_adjudicate(args: argparse.Namespace) -> int:
|
|
184
|
+
from datetime import datetime, timezone
|
|
185
|
+
|
|
186
|
+
path, v = load_verdict(args)
|
|
187
|
+
now = datetime.now(timezone.utc)
|
|
188
|
+
who = _adjudicator()
|
|
189
|
+
base = {"at": now.isoformat(timespec="seconds"), "adjudicator": who,
|
|
190
|
+
"run_id": v.run_id, "request_id": v.request_id, "note": args.note}
|
|
191
|
+
changed: list[str] = []
|
|
192
|
+
|
|
193
|
+
if args.finding or args.verdict:
|
|
194
|
+
if not args.judge:
|
|
195
|
+
sys.exit("--judge is required when grading findings or a review. Judges: " +
|
|
196
|
+
", ".join(r.judge for r in v.reviews))
|
|
197
|
+
matches = [r for r in v.reviews if r.judge == args.judge or r.review_id == args.judge
|
|
198
|
+
or r.role == args.judge]
|
|
199
|
+
if len(matches) != 1:
|
|
200
|
+
sys.exit(f"--judge {args.judge!r} matched {len(matches)} reviews. Judges: " +
|
|
201
|
+
", ".join(r.judge for r in v.reviews))
|
|
202
|
+
review = matches[0]
|
|
203
|
+
for ref, label in args.finding or []:
|
|
204
|
+
if label not in ("correct", "partially_correct", "wrong"):
|
|
205
|
+
sys.exit(f"Finding label must be correct, partially_correct, or wrong; got {label!r}.")
|
|
206
|
+
target = None
|
|
207
|
+
if ref.isdigit() and 1 <= int(ref) <= len(review.findings):
|
|
208
|
+
target = review.findings[int(ref) - 1]
|
|
209
|
+
else:
|
|
210
|
+
target = next((f for f in review.findings if f.id == ref), None)
|
|
211
|
+
if target is None:
|
|
212
|
+
sys.exit(f"{review.judge} has no finding {ref!r} (it has {len(review.findings)}).")
|
|
213
|
+
log_event(path.parent, {**base, "kind": "finding", "review_id": review.review_id,
|
|
214
|
+
"judge": review.judge, "config_id": review.config_id,
|
|
215
|
+
"finding_id": target.id, "old": target.adjudication, "new": label})
|
|
216
|
+
target.adjudication = label
|
|
217
|
+
target.adjudicated_at = now
|
|
218
|
+
changed.append(f"{review.judge} finding {ref}: {label}")
|
|
219
|
+
if args.verdict:
|
|
220
|
+
old = review.human_review.verdict if review.human_review else None
|
|
221
|
+
log_event(path.parent, {**base, "kind": "review", "review_id": review.review_id,
|
|
222
|
+
"judge": review.judge, "config_id": review.config_id,
|
|
223
|
+
"old": old, "new": args.verdict})
|
|
224
|
+
review.human_review = HumanReview(verdict=args.verdict, note=args.note, reviewed_at=now)
|
|
225
|
+
changed.append(f"{review.judge} review: {args.verdict}")
|
|
226
|
+
|
|
227
|
+
if args.producer_verdict:
|
|
228
|
+
log_event(path.parent, {**base, "kind": "producer", "old": v.human_verdict, "new": args.producer_verdict})
|
|
229
|
+
v.human_verdict = args.producer_verdict
|
|
230
|
+
v.human_note = args.note
|
|
231
|
+
v.adjudicated_at = now
|
|
232
|
+
changed.append(f"producer output: {args.producer_verdict}")
|
|
233
|
+
|
|
234
|
+
if not changed:
|
|
235
|
+
sys.exit("Nothing to record. Give --finding, --verdict, or --producer-verdict.")
|
|
236
|
+
|
|
237
|
+
path.write_text(v.model_dump_json(indent=2), encoding="utf-8")
|
|
238
|
+
print(f"run {v.run_id} {path}")
|
|
239
|
+
for c in changed:
|
|
240
|
+
print(f" {c}")
|
|
241
|
+
print(f" logged by {who} -> {path.parent / ADJUDICATION_LOG}")
|
|
242
|
+
return 0
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def cmd_roles(args: argparse.Namespace) -> int:
|
|
246
|
+
if args.roles:
|
|
247
|
+
load_roles(args.roles)
|
|
248
|
+
for name, desc in ROLES.items():
|
|
249
|
+
print(f"{name:<10} {desc}")
|
|
250
|
+
return 0
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def cmd_schema(args: argparse.Namespace) -> int:
|
|
254
|
+
model = {"request": ReviewRequest, "verdict": Verdict}[args.object]
|
|
255
|
+
print(json.dumps(model.model_json_schema(), indent=2))
|
|
256
|
+
return 0
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def main(argv: list[str] | None = None) -> int:
|
|
260
|
+
load_dotenv()
|
|
261
|
+
parser = argparse.ArgumentParser(prog="agentjury", description="Peer review for AI agent output.")
|
|
262
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
263
|
+
|
|
264
|
+
p = sub.add_parser("review", help="Review an agent's output with a panel of judges.")
|
|
265
|
+
p.add_argument("task", help="File containing the task the agent was given.")
|
|
266
|
+
p.add_argument("output", help="File containing the agent's output, or - for stdin.")
|
|
267
|
+
p.add_argument("--context", help="File with background the judges should know.")
|
|
268
|
+
p.add_argument("--panel", default=os.environ.get("AGENTJURY_PANEL", DEFAULT_PANEL),
|
|
269
|
+
help=f"role:provider pairs, comma-separated (default: {DEFAULT_PANEL})")
|
|
270
|
+
p.add_argument("--roles", default=os.environ.get("AGENTJURY_ROLES"),
|
|
271
|
+
help="JSON file of extra roles {name: description}, e.g. a domain expert.")
|
|
272
|
+
p.add_argument("--quorum", type=int, help="Minimum judges that must respond (default: majority).")
|
|
273
|
+
p.add_argument("--task-type", help="Kind of work, e.g. financial_analysis, code_review, summary.")
|
|
274
|
+
p.add_argument("--domain", help="Subject area, e.g. private_credit, python.")
|
|
275
|
+
p.add_argument("--agent", help="Name of the agent that did the work.")
|
|
276
|
+
p.add_argument("--framework", help="Framework the agent runs on, e.g. hermes.")
|
|
277
|
+
p.add_argument("--producer-provider", help="Provider of the model that did the work, e.g. anthropic.")
|
|
278
|
+
p.add_argument("--producer-model", help="Model that did the work, e.g. claude-fable-5-1.")
|
|
279
|
+
p.add_argument("--json", action="store_true", help="Print the full verdict as JSON.")
|
|
280
|
+
p.add_argument("--no-save", action="store_true", help="Do not write the verdict to .agentjury/.")
|
|
281
|
+
p.set_defaults(func=cmd_review)
|
|
282
|
+
|
|
283
|
+
r = sub.add_parser("roles", help="List available judge roles.")
|
|
284
|
+
r.add_argument("--roles", default=os.environ.get("AGENTJURY_ROLES"), help="JSON file of extra roles.")
|
|
285
|
+
r.set_defaults(func=cmd_roles)
|
|
286
|
+
|
|
287
|
+
vl = sub.add_parser("verdicts", help="List saved verdicts, newest first.")
|
|
288
|
+
vl.add_argument("--dir", help="Verdict directory (default: $AGENTJURY_VERDICT_DIR or .agentjury/verdicts).")
|
|
289
|
+
vl.add_argument("-n", type=int, default=20, help="How many to show.")
|
|
290
|
+
vl.set_defaults(func=cmd_verdicts)
|
|
291
|
+
|
|
292
|
+
a = sub.add_parser("adjudicate", help="Record a human judgement on a saved verdict.")
|
|
293
|
+
a.add_argument("request_id", metavar="ID", help="run_id or request_id (or unique fragment), or a path to the JSON file.")
|
|
294
|
+
a.add_argument("--dir", help="Verdict directory (default: $AGENTJURY_VERDICT_DIR or .agentjury/verdicts).")
|
|
295
|
+
a.add_argument("--judge", help="Which review to grade: judge name (critic/anthropic), role, or review_id.")
|
|
296
|
+
a.add_argument("--finding", nargs=2, action="append", metavar=("N", "LABEL"),
|
|
297
|
+
help="Grade finding N (1-based, as shown) as correct, partially_correct, or wrong. Repeatable.")
|
|
298
|
+
a.add_argument("--verdict", choices=["agree", "partial", "disagree"], help="Your overall view of that judge's review.")
|
|
299
|
+
a.add_argument("--producer-verdict", choices=["correct", "flawed"], help="Your view of the agent's output itself.")
|
|
300
|
+
a.add_argument("--note", help="Free-text reason, stored with the review and/or producer verdict.")
|
|
301
|
+
a.set_defaults(func=cmd_adjudicate)
|
|
302
|
+
|
|
303
|
+
s = sub.add_parser("schema", help="Print the JSON schema for the protocol objects.")
|
|
304
|
+
s.add_argument("object", choices=["request", "verdict"], nargs="?", default="verdict")
|
|
305
|
+
s.set_defaults(func=cmd_schema)
|
|
306
|
+
|
|
307
|
+
args = parser.parse_args(argv)
|
|
308
|
+
return args.func(args)
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
if __name__ == "__main__":
|
|
312
|
+
sys.exit(main())
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
from .base import ROLES, RUBRIC_VERSION, Completion, Judge, JudgeOpinion, load_roles, parse_opinion, register_roles
|
|
2
|
+
from .fake import FakeJudge
|
|
3
|
+
|
|
4
|
+
__all__ = ["ROLES", "RUBRIC_VERSION", "Completion", "Judge", "JudgeOpinion", "load_roles", "parse_opinion", "register_roles", "FakeJudge"]
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def openai_judge(role: str, model: str | None = None):
|
|
8
|
+
from .openai_judge import OpenAIJudge
|
|
9
|
+
return OpenAIJudge(role, model) if model else OpenAIJudge(role)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def anthropic_judge(role: str, model: str | None = None):
|
|
13
|
+
from .anthropic_judge import AnthropicJudge
|
|
14
|
+
return AnthropicJudge(role, model) if model else AnthropicJudge(role)
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Judge backed by an Anthropic model.
|
|
3
|
+
|
|
4
|
+
Requires ANTHROPIC_API_KEY in the environment.
|
|
5
|
+
|
|
6
|
+
Workspace: if the key spans multiple workspaces, Anthropic requires an
|
|
7
|
+
`anthropic-workspace-id` header; set ANTHROPIC_WORKSPACE_ID and it is sent.
|
|
8
|
+
|
|
9
|
+
Thinking: Claude Sonnet 5 and later reason before answering by default, and
|
|
10
|
+
that reasoning counts against max_tokens. We give the model 4096 tokens and
|
|
11
|
+
default effort to "medium", which is plenty for a short JSON review.
|
|
12
|
+
ANTHROPIC_EFFORT low | medium | high (default medium)
|
|
13
|
+
ANTHROPIC_THINKING adaptive | disabled (default adaptive; Fable/Mythos reject disabled)
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import os
|
|
19
|
+
|
|
20
|
+
from .base import Completion, Judge
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class AnthropicJudge(Judge):
|
|
24
|
+
provider = "anthropic"
|
|
25
|
+
|
|
26
|
+
def __init__(self, role: str, model: str = "claude-sonnet-5", max_tokens: int = 4096,
|
|
27
|
+
effort: str | None = None, thinking: str | None = None, timeout: float = 90.0):
|
|
28
|
+
super().__init__(role, model, timeout=timeout)
|
|
29
|
+
try:
|
|
30
|
+
from anthropic import Anthropic
|
|
31
|
+
except ImportError as exc: # optional dependency
|
|
32
|
+
raise ImportError(
|
|
33
|
+
'The anthropic package is not installed. Run: pip install "agentjury[anthropic]"'
|
|
34
|
+
) from exc
|
|
35
|
+
|
|
36
|
+
headers = {}
|
|
37
|
+
workspace_id = os.environ.get("ANTHROPIC_WORKSPACE_ID")
|
|
38
|
+
if workspace_id:
|
|
39
|
+
headers["anthropic-workspace-id"] = workspace_id
|
|
40
|
+
|
|
41
|
+
# max_retries=0: AgentJury owns the retry policy, not the SDK.
|
|
42
|
+
self._client = Anthropic(default_headers=headers, timeout=timeout, max_retries=0)
|
|
43
|
+
self.max_tokens = max_tokens
|
|
44
|
+
self.effort = effort or os.environ.get("ANTHROPIC_EFFORT", "medium")
|
|
45
|
+
self.thinking = thinking or os.environ.get("ANTHROPIC_THINKING", "adaptive")
|
|
46
|
+
self.params = {"max_tokens": max_tokens, "effort": self.effort, "thinking": self.thinking}
|
|
47
|
+
|
|
48
|
+
def complete(self, system: str, user: str) -> Completion:
|
|
49
|
+
kwargs = dict(
|
|
50
|
+
model=self.model,
|
|
51
|
+
max_tokens=self.max_tokens,
|
|
52
|
+
system=system,
|
|
53
|
+
messages=[{"role": "user", "content": user}],
|
|
54
|
+
output_config={"effort": self.effort},
|
|
55
|
+
)
|
|
56
|
+
if self.thinking == "disabled":
|
|
57
|
+
kwargs["thinking"] = {"type": "disabled"}
|
|
58
|
+
|
|
59
|
+
response = self._client.messages.create(**kwargs)
|
|
60
|
+
text = "".join(block.text for block in response.content if block.type == "text")
|
|
61
|
+
|
|
62
|
+
if not text.strip():
|
|
63
|
+
kinds = [block.type for block in response.content]
|
|
64
|
+
raise RuntimeError(
|
|
65
|
+
f"{self.model} returned no text (stop_reason={response.stop_reason}, blocks={kinds}). "
|
|
66
|
+
f"If stop_reason is max_tokens, raise max_tokens or lower ANTHROPIC_EFFORT."
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
usage = response.usage
|
|
70
|
+
return Completion(
|
|
71
|
+
text=text,
|
|
72
|
+
tokens_in=getattr(usage, "input_tokens", None),
|
|
73
|
+
tokens_out=getattr(usage, "output_tokens", None),
|
|
74
|
+
response_id=getattr(response, "id", None),
|
|
75
|
+
)
|