agentjury 0.4.4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
agentjury/__init__.py ADDED
@@ -0,0 +1,22 @@
1
+ """AgentJury: peer review for AI agents."""
2
+
3
+ from .aggregate import aggregate
4
+ from .panel import Panel
5
+ from .protocol import (
6
+ SCHEMA_VERSION,
7
+ Artifact,
8
+ Finding,
9
+ HumanReview,
10
+ Producer,
11
+ Review,
12
+ ReviewRequest,
13
+ Verdict,
14
+ Vote,
15
+ )
16
+
17
+ __version__ = "0.4.4"
18
+
19
+ __all__ = [
20
+ "SCHEMA_VERSION", "Artifact", "Finding", "HumanReview", "Producer",
21
+ "Review", "ReviewRequest", "Verdict", "Vote", "Panel", "aggregate",
22
+ ]
agentjury/aggregate.py ADDED
@@ -0,0 +1,120 @@
1
+ """
2
+ Turn a list of independent Reviews into one Verdict.
3
+
4
+ No LLM calls here. The rules are written down so anyone can predict the verdict:
5
+
6
+ voters responding judges who did not abstain
7
+ up / down count of approve / revise votes among voters
8
+ score mean of voters' scores
9
+ consensus share of voters who sided with the majority vote
10
+ diversity distinct providers among voters / voters
11
+ confidence heuristic index: consensus, discounted for small panels, wide
12
+ score spread, and low diversity. NOT a calibrated probability.
13
+
14
+ status
15
+ insufficient_jury fewer than `quorum` judges voted, OR the panel was built
16
+ from two or more providers but only one provider's judges
17
+ voted. Votes are reported; no verdict is reached.
18
+ blocked blocking findings from at least two providers (or from at
19
+ least two judges when the panel has only one provider).
20
+ No single judge can block on its own.
21
+ needs_revision majority revise, a tie, or exactly one blocking source.
22
+ verified majority approve and no blocking finding.
23
+
24
+ quorum default is a strict majority of requested judges:
25
+ 1->1, 2->2, 3->2, 4->3, 5->3, 6->4
26
+
27
+ Reviewer reputation will later weight these votes. For now every voter counts once.
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ from .protocol import Review, ReviewRequest, Verdict, Vote
33
+
34
+
35
+ def default_quorum(requested: int) -> int:
36
+ """Strict majority of the requested panel."""
37
+ return max(1, requested // 2 + 1)
38
+
39
+
40
+ def _empty(request: ReviewRequest, reviews, errors, requested, quorum, panel_id) -> Verdict:
41
+ return Verdict(
42
+ request_id=request.request_id, panel_id=panel_id,
43
+ requested=requested, responded=len(reviews), abstained=len(reviews), quorum=quorum,
44
+ task_type=request.task_type, domain=request.domain, producer=request.producer,
45
+ up=0, down=0, score=0.0, consensus=0.0, diversity=0.0, confidence=0.0,
46
+ status="insufficient_jury", reviews=reviews, errors=errors or [],
47
+ )
48
+
49
+
50
+ def aggregate(
51
+ request: ReviewRequest,
52
+ reviews: list[Review],
53
+ errors: list[str] | None = None,
54
+ *,
55
+ requested: int | None = None,
56
+ quorum: int | None = None,
57
+ panel_id: str | None = None,
58
+ requested_providers: int | None = None,
59
+ ) -> Verdict:
60
+ requested = requested if requested is not None else len(reviews)
61
+ quorum = quorum if quorum is not None else default_quorum(requested)
62
+
63
+ voters = [r for r in reviews if r.vote != Vote.ABSTAIN]
64
+ abstained = len(reviews) - len(voters)
65
+ if not voters:
66
+ return _empty(request, reviews, errors, requested, quorum, panel_id)
67
+
68
+ n = len(voters)
69
+ up = sum(1 for r in voters if r.vote == Vote.APPROVE)
70
+ down = n - up
71
+ scores = [r.score for r in voters]
72
+ score = sum(scores) / n
73
+
74
+ consensus = max(up, down) / n
75
+ providers = {r.provider for r in voters}
76
+ diversity = len(providers) / n
77
+
78
+ panel_factor = n / (n + 1) # 1 voter -> 0.5, 3 -> 0.75, 5 -> 0.83
79
+ spread_factor = 1 - (max(scores) - min(scores)) / 10 # identical scores -> 1.0
80
+ diversity_factor = 0.5 + 0.5 * diversity # all one provider (n=3) -> 0.67; all distinct -> 1.0
81
+ confidence = round(consensus * panel_factor * spread_factor * diversity_factor, 3)
82
+
83
+ blocking_reviews = [r for r in voters if r.blocking]
84
+ blocking_providers = {r.provider for r in blocking_reviews}
85
+ independent_blocks = len(blocking_providers) >= 2 or (len(providers) == 1 and len(blocking_reviews) >= 2)
86
+
87
+ # A multi-provider panel that only heard from one provider has lost the
88
+ # independence it was built for, so it cannot reach a verdict.
89
+ wanted_providers = requested_providers if requested_providers is not None else len(providers)
90
+ provider_floor = min(2, wanted_providers)
91
+
92
+ if n < quorum or len(providers) < provider_floor:
93
+ status = "insufficient_jury"
94
+ elif independent_blocks:
95
+ status = "blocked"
96
+ elif blocking_reviews or up <= down:
97
+ status = "needs_revision"
98
+ else:
99
+ status = "verified"
100
+
101
+ return Verdict(
102
+ request_id=request.request_id,
103
+ panel_id=panel_id,
104
+ requested=requested,
105
+ responded=len(reviews),
106
+ abstained=abstained,
107
+ quorum=quorum,
108
+ task_type=request.task_type,
109
+ domain=request.domain,
110
+ producer=request.producer,
111
+ up=up,
112
+ down=down,
113
+ score=round(score, 2),
114
+ consensus=round(consensus, 3),
115
+ diversity=round(diversity, 3),
116
+ confidence=confidence,
117
+ status=status,
118
+ reviews=reviews,
119
+ errors=errors or [],
120
+ )
agentjury/cli.py ADDED
@@ -0,0 +1,312 @@
1
+ """
2
+ Command-line interface.
3
+
4
+ agentjury review TASK OUTPUT [--panel SPEC] [--roles FILE] [--quorum N] [--task-type T] [--domain D] [--json]
5
+ agentjury roles [--roles FILE]
6
+ agentjury schema [request|verdict]
7
+ agentjury verdicts [--dir DIR] [-n N]
8
+ agentjury adjudicate ID [--judge J] [--finding N LABEL]... [--verdict agree|partial|disagree]
9
+ [--producer-verdict correct|flawed] [--note TEXT] [--dir DIR]
10
+
11
+ Verdicts are read from --dir, else $AGENTJURY_VERDICT_DIR, else .agentjury/verdicts.
12
+
13
+ Exit codes: 0 verified, 1 needs_revision, 2 blocked, 3 insufficient_jury.
14
+
15
+ TASK and OUTPUT are files (or "-" to read OUTPUT from stdin).
16
+ PANEL is a comma-separated list of role:provider pairs, for example
17
+ accuracy:openai,critic:anthropic,executive:openai
18
+ Every verdict is saved to .agentjury/verdicts/<request_id>.json so that
19
+ reviews accumulate over time.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import argparse
25
+ import json
26
+ import os
27
+ import sys
28
+ from pathlib import Path
29
+
30
+ from dotenv import load_dotenv
31
+
32
+ from .judges import ROLES, anthropic_judge, load_roles, openai_judge
33
+ from .judges.base import Judge
34
+ from .panel import Panel
35
+ from .protocol import HumanReview, Producer, ReviewRequest, Verdict
36
+
37
+ DEFAULT_PANEL = "accuracy:openai,critic:anthropic,executive:openai"
38
+ VERDICT_DIR = Path(".agentjury") / "verdicts"
39
+
40
+ PROVIDERS = {
41
+ "openai": openai_judge,
42
+ "anthropic": anthropic_judge,
43
+ }
44
+
45
+ SEVERITY_MARK = {"minor": "-", "major": "!", "blocking": "X"}
46
+
47
+
48
+ # Exit codes, so shell scripts and CI can branch without parsing output.
49
+ EXIT = {"verified": 0, "needs_revision": 1, "blocked": 2, "insufficient_jury": 3}
50
+
51
+
52
+ def build_panel(spec: str, quorum: int | None = None) -> Panel:
53
+ judges: list[Judge] = []
54
+ for item in spec.split(","):
55
+ item = item.strip()
56
+ if not item:
57
+ continue
58
+ try:
59
+ role, provider = item.split(":")
60
+ except ValueError:
61
+ sys.exit(f"Bad panel entry {item!r}. Use role:provider, e.g. critic:anthropic")
62
+ if role not in ROLES:
63
+ sys.exit(f"Unknown role {role!r}. Known roles: {', '.join(sorted(ROLES))}")
64
+ if provider not in PROVIDERS:
65
+ sys.exit(f"Unknown provider {provider!r}. Known providers: {', '.join(PROVIDERS)}")
66
+ judges.append(PROVIDERS[provider](role))
67
+ return Panel(judges, quorum=quorum)
68
+
69
+
70
+ def read(path: str) -> str:
71
+ if path == "-":
72
+ return sys.stdin.read()
73
+ return Path(path).read_text(encoding="utf-8")
74
+
75
+
76
+ def save(verdict: Verdict) -> Path:
77
+ VERDICT_DIR.mkdir(parents=True, exist_ok=True)
78
+ out = VERDICT_DIR / verdict.filename
79
+ out.write_text(verdict.model_dump_json(indent=2), encoding="utf-8")
80
+ return out
81
+
82
+
83
+ def print_verdict(verdict: Verdict) -> None:
84
+ print(verdict.render())
85
+ print(f"jury confidence index {verdict.confidence:.0%} (heuristic, not a probability)")
86
+ if verdict.status == "insufficient_jury":
87
+ voters = verdict.responded - verdict.abstained
88
+ providers = len({r.provider for r in verdict.reviews if r.vote != "abstain"})
89
+ print(f"Insufficient jury: {voters} of {verdict.requested} judges voted (quorum {verdict.quorum}), "
90
+ f"from {providers} provider(s). No verdict.")
91
+ print()
92
+ for r in verdict.reviews:
93
+ arrow = {"approve": "▲", "revise": "▼", "abstain": "–"}[r.vote]
94
+ meta = f"{r.latency_ms / 1000:.1f}s" if r.latency_ms is not None else ""
95
+ print(f"{arrow} {r.score:>2.0f} {r.judge:<22} {r.reason} [{meta}]")
96
+ for f in r.findings:
97
+ print(f" {SEVERITY_MARK[f.severity]} {f.text}")
98
+ for e in verdict.errors:
99
+ print(f"! {e}")
100
+
101
+
102
+ def cmd_review(args: argparse.Namespace) -> int:
103
+ if args.roles:
104
+ load_roles(args.roles)
105
+ request = ReviewRequest(
106
+ task=read(args.task),
107
+ output=read(args.output),
108
+ context=read(args.context) if args.context else None,
109
+ task_type=args.task_type,
110
+ domain=args.domain,
111
+ producer=Producer(
112
+ agent=args.agent,
113
+ framework=args.framework,
114
+ provider=args.producer_provider,
115
+ model=args.producer_model,
116
+ ),
117
+ )
118
+ verdict = build_panel(args.panel, quorum=args.quorum).review(request)
119
+
120
+ if args.json:
121
+ print(verdict.model_dump_json(indent=2))
122
+ else:
123
+ print_verdict(verdict)
124
+
125
+ if not args.no_save:
126
+ path = save(verdict)
127
+ if not args.json:
128
+ print(f"\nsaved {path}")
129
+
130
+ return EXIT[verdict.status]
131
+
132
+
133
+ def verdict_dir(args: argparse.Namespace) -> Path:
134
+ return Path(getattr(args, "dir", None) or os.environ.get("AGENTJURY_VERDICT_DIR") or VERDICT_DIR)
135
+
136
+
137
+ def load_verdict(args: argparse.Namespace) -> tuple[Path, Verdict]:
138
+ ref = args.request_id
139
+ path = Path(ref)
140
+ if not path.is_file():
141
+ d = verdict_dir(args)
142
+ candidates = sorted(f for f in d.glob("*.json") if ref in f.stem) if d.is_dir() else []
143
+ if len(candidates) != 1:
144
+ hint = f"{len(candidates)} matches" if candidates else "no match"
145
+ sys.exit(f"Cannot find verdict {ref!r} in {d} ({hint}). Use `agentjury verdicts --dir {d}` to list.")
146
+ path = candidates[0]
147
+ return path, Verdict.model_validate_json(path.read_text(encoding="utf-8"))
148
+
149
+
150
+ def cmd_verdicts(args: argparse.Namespace) -> int:
151
+ d = verdict_dir(args)
152
+ files = sorted(d.glob("*.json"), key=lambda f: f.stat().st_mtime, reverse=True) if d.is_dir() else []
153
+ if not files:
154
+ print(f"No verdicts in {d}")
155
+ return 0
156
+ print(f"{d}\n")
157
+ for f in files[: args.n]:
158
+ v = Verdict.model_validate_json(f.read_text(encoding="utf-8"))
159
+ graded = sum(1 for r in v.reviews for fi in r.findings if fi.adjudication)
160
+ total = sum(len(r.findings) for r in v.reviews)
161
+ mark = f" [adjudicated {graded}/{total}]" if graded else ""
162
+ prod = f" producer:{v.human_verdict}" if v.human_verdict else ""
163
+ print(f"run {v.run_id} req {v.request_id} {v.created_at:%Y-%m-%d %H:%M} {v.render()}{mark}{prod}")
164
+ return 0
165
+
166
+
167
+ ADJUDICATION_LOG = "adjudications.jsonl"
168
+
169
+
170
+ def _adjudicator() -> str:
171
+ import getpass
172
+ return os.environ.get("AGENTJURY_ADJUDICATOR") or getpass.getuser()
173
+
174
+
175
+ def log_event(directory: Path, event: dict) -> None:
176
+ """Append-only audit trail. The verdict JSON holds current state; this holds history."""
177
+ from uuid import uuid4
178
+ event = {"event_id": uuid4().hex[:12], **event}
179
+ with (directory / ADJUDICATION_LOG).open("a", encoding="utf-8") as f:
180
+ f.write(json.dumps(event, default=str) + "\n")
181
+
182
+
183
+ def cmd_adjudicate(args: argparse.Namespace) -> int:
184
+ from datetime import datetime, timezone
185
+
186
+ path, v = load_verdict(args)
187
+ now = datetime.now(timezone.utc)
188
+ who = _adjudicator()
189
+ base = {"at": now.isoformat(timespec="seconds"), "adjudicator": who,
190
+ "run_id": v.run_id, "request_id": v.request_id, "note": args.note}
191
+ changed: list[str] = []
192
+
193
+ if args.finding or args.verdict:
194
+ if not args.judge:
195
+ sys.exit("--judge is required when grading findings or a review. Judges: " +
196
+ ", ".join(r.judge for r in v.reviews))
197
+ matches = [r for r in v.reviews if r.judge == args.judge or r.review_id == args.judge
198
+ or r.role == args.judge]
199
+ if len(matches) != 1:
200
+ sys.exit(f"--judge {args.judge!r} matched {len(matches)} reviews. Judges: " +
201
+ ", ".join(r.judge for r in v.reviews))
202
+ review = matches[0]
203
+ for ref, label in args.finding or []:
204
+ if label not in ("correct", "partially_correct", "wrong"):
205
+ sys.exit(f"Finding label must be correct, partially_correct, or wrong; got {label!r}.")
206
+ target = None
207
+ if ref.isdigit() and 1 <= int(ref) <= len(review.findings):
208
+ target = review.findings[int(ref) - 1]
209
+ else:
210
+ target = next((f for f in review.findings if f.id == ref), None)
211
+ if target is None:
212
+ sys.exit(f"{review.judge} has no finding {ref!r} (it has {len(review.findings)}).")
213
+ log_event(path.parent, {**base, "kind": "finding", "review_id": review.review_id,
214
+ "judge": review.judge, "config_id": review.config_id,
215
+ "finding_id": target.id, "old": target.adjudication, "new": label})
216
+ target.adjudication = label
217
+ target.adjudicated_at = now
218
+ changed.append(f"{review.judge} finding {ref}: {label}")
219
+ if args.verdict:
220
+ old = review.human_review.verdict if review.human_review else None
221
+ log_event(path.parent, {**base, "kind": "review", "review_id": review.review_id,
222
+ "judge": review.judge, "config_id": review.config_id,
223
+ "old": old, "new": args.verdict})
224
+ review.human_review = HumanReview(verdict=args.verdict, note=args.note, reviewed_at=now)
225
+ changed.append(f"{review.judge} review: {args.verdict}")
226
+
227
+ if args.producer_verdict:
228
+ log_event(path.parent, {**base, "kind": "producer", "old": v.human_verdict, "new": args.producer_verdict})
229
+ v.human_verdict = args.producer_verdict
230
+ v.human_note = args.note
231
+ v.adjudicated_at = now
232
+ changed.append(f"producer output: {args.producer_verdict}")
233
+
234
+ if not changed:
235
+ sys.exit("Nothing to record. Give --finding, --verdict, or --producer-verdict.")
236
+
237
+ path.write_text(v.model_dump_json(indent=2), encoding="utf-8")
238
+ print(f"run {v.run_id} {path}")
239
+ for c in changed:
240
+ print(f" {c}")
241
+ print(f" logged by {who} -> {path.parent / ADJUDICATION_LOG}")
242
+ return 0
243
+
244
+
245
+ def cmd_roles(args: argparse.Namespace) -> int:
246
+ if args.roles:
247
+ load_roles(args.roles)
248
+ for name, desc in ROLES.items():
249
+ print(f"{name:<10} {desc}")
250
+ return 0
251
+
252
+
253
+ def cmd_schema(args: argparse.Namespace) -> int:
254
+ model = {"request": ReviewRequest, "verdict": Verdict}[args.object]
255
+ print(json.dumps(model.model_json_schema(), indent=2))
256
+ return 0
257
+
258
+
259
+ def main(argv: list[str] | None = None) -> int:
260
+ load_dotenv()
261
+ parser = argparse.ArgumentParser(prog="agentjury", description="Peer review for AI agent output.")
262
+ sub = parser.add_subparsers(dest="command", required=True)
263
+
264
+ p = sub.add_parser("review", help="Review an agent's output with a panel of judges.")
265
+ p.add_argument("task", help="File containing the task the agent was given.")
266
+ p.add_argument("output", help="File containing the agent's output, or - for stdin.")
267
+ p.add_argument("--context", help="File with background the judges should know.")
268
+ p.add_argument("--panel", default=os.environ.get("AGENTJURY_PANEL", DEFAULT_PANEL),
269
+ help=f"role:provider pairs, comma-separated (default: {DEFAULT_PANEL})")
270
+ p.add_argument("--roles", default=os.environ.get("AGENTJURY_ROLES"),
271
+ help="JSON file of extra roles {name: description}, e.g. a domain expert.")
272
+ p.add_argument("--quorum", type=int, help="Minimum judges that must respond (default: majority).")
273
+ p.add_argument("--task-type", help="Kind of work, e.g. financial_analysis, code_review, summary.")
274
+ p.add_argument("--domain", help="Subject area, e.g. private_credit, python.")
275
+ p.add_argument("--agent", help="Name of the agent that did the work.")
276
+ p.add_argument("--framework", help="Framework the agent runs on, e.g. hermes.")
277
+ p.add_argument("--producer-provider", help="Provider of the model that did the work, e.g. anthropic.")
278
+ p.add_argument("--producer-model", help="Model that did the work, e.g. claude-fable-5-1.")
279
+ p.add_argument("--json", action="store_true", help="Print the full verdict as JSON.")
280
+ p.add_argument("--no-save", action="store_true", help="Do not write the verdict to .agentjury/.")
281
+ p.set_defaults(func=cmd_review)
282
+
283
+ r = sub.add_parser("roles", help="List available judge roles.")
284
+ r.add_argument("--roles", default=os.environ.get("AGENTJURY_ROLES"), help="JSON file of extra roles.")
285
+ r.set_defaults(func=cmd_roles)
286
+
287
+ vl = sub.add_parser("verdicts", help="List saved verdicts, newest first.")
288
+ vl.add_argument("--dir", help="Verdict directory (default: $AGENTJURY_VERDICT_DIR or .agentjury/verdicts).")
289
+ vl.add_argument("-n", type=int, default=20, help="How many to show.")
290
+ vl.set_defaults(func=cmd_verdicts)
291
+
292
+ a = sub.add_parser("adjudicate", help="Record a human judgement on a saved verdict.")
293
+ a.add_argument("request_id", metavar="ID", help="run_id or request_id (or unique fragment), or a path to the JSON file.")
294
+ a.add_argument("--dir", help="Verdict directory (default: $AGENTJURY_VERDICT_DIR or .agentjury/verdicts).")
295
+ a.add_argument("--judge", help="Which review to grade: judge name (critic/anthropic), role, or review_id.")
296
+ a.add_argument("--finding", nargs=2, action="append", metavar=("N", "LABEL"),
297
+ help="Grade finding N (1-based, as shown) as correct, partially_correct, or wrong. Repeatable.")
298
+ a.add_argument("--verdict", choices=["agree", "partial", "disagree"], help="Your overall view of that judge's review.")
299
+ a.add_argument("--producer-verdict", choices=["correct", "flawed"], help="Your view of the agent's output itself.")
300
+ a.add_argument("--note", help="Free-text reason, stored with the review and/or producer verdict.")
301
+ a.set_defaults(func=cmd_adjudicate)
302
+
303
+ s = sub.add_parser("schema", help="Print the JSON schema for the protocol objects.")
304
+ s.add_argument("object", choices=["request", "verdict"], nargs="?", default="verdict")
305
+ s.set_defaults(func=cmd_schema)
306
+
307
+ args = parser.parse_args(argv)
308
+ return args.func(args)
309
+
310
+
311
+ if __name__ == "__main__":
312
+ sys.exit(main())
@@ -0,0 +1,14 @@
1
+ from .base import ROLES, RUBRIC_VERSION, Completion, Judge, JudgeOpinion, load_roles, parse_opinion, register_roles
2
+ from .fake import FakeJudge
3
+
4
+ __all__ = ["ROLES", "RUBRIC_VERSION", "Completion", "Judge", "JudgeOpinion", "load_roles", "parse_opinion", "register_roles", "FakeJudge"]
5
+
6
+
7
+ def openai_judge(role: str, model: str | None = None):
8
+ from .openai_judge import OpenAIJudge
9
+ return OpenAIJudge(role, model) if model else OpenAIJudge(role)
10
+
11
+
12
+ def anthropic_judge(role: str, model: str | None = None):
13
+ from .anthropic_judge import AnthropicJudge
14
+ return AnthropicJudge(role, model) if model else AnthropicJudge(role)
@@ -0,0 +1,75 @@
1
+ """
2
+ Judge backed by an Anthropic model.
3
+
4
+ Requires ANTHROPIC_API_KEY in the environment.
5
+
6
+ Workspace: if the key spans multiple workspaces, Anthropic requires an
7
+ `anthropic-workspace-id` header; set ANTHROPIC_WORKSPACE_ID and it is sent.
8
+
9
+ Thinking: Claude Sonnet 5 and later reason before answering by default, and
10
+ that reasoning counts against max_tokens. We give the model 4096 tokens and
11
+ default effort to "medium", which is plenty for a short JSON review.
12
+ ANTHROPIC_EFFORT low | medium | high (default medium)
13
+ ANTHROPIC_THINKING adaptive | disabled (default adaptive; Fable/Mythos reject disabled)
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import os
19
+
20
+ from .base import Completion, Judge
21
+
22
+
23
+ class AnthropicJudge(Judge):
24
+ provider = "anthropic"
25
+
26
+ def __init__(self, role: str, model: str = "claude-sonnet-5", max_tokens: int = 4096,
27
+ effort: str | None = None, thinking: str | None = None, timeout: float = 90.0):
28
+ super().__init__(role, model, timeout=timeout)
29
+ try:
30
+ from anthropic import Anthropic
31
+ except ImportError as exc: # optional dependency
32
+ raise ImportError(
33
+ 'The anthropic package is not installed. Run: pip install "agentjury[anthropic]"'
34
+ ) from exc
35
+
36
+ headers = {}
37
+ workspace_id = os.environ.get("ANTHROPIC_WORKSPACE_ID")
38
+ if workspace_id:
39
+ headers["anthropic-workspace-id"] = workspace_id
40
+
41
+ # max_retries=0: AgentJury owns the retry policy, not the SDK.
42
+ self._client = Anthropic(default_headers=headers, timeout=timeout, max_retries=0)
43
+ self.max_tokens = max_tokens
44
+ self.effort = effort or os.environ.get("ANTHROPIC_EFFORT", "medium")
45
+ self.thinking = thinking or os.environ.get("ANTHROPIC_THINKING", "adaptive")
46
+ self.params = {"max_tokens": max_tokens, "effort": self.effort, "thinking": self.thinking}
47
+
48
+ def complete(self, system: str, user: str) -> Completion:
49
+ kwargs = dict(
50
+ model=self.model,
51
+ max_tokens=self.max_tokens,
52
+ system=system,
53
+ messages=[{"role": "user", "content": user}],
54
+ output_config={"effort": self.effort},
55
+ )
56
+ if self.thinking == "disabled":
57
+ kwargs["thinking"] = {"type": "disabled"}
58
+
59
+ response = self._client.messages.create(**kwargs)
60
+ text = "".join(block.text for block in response.content if block.type == "text")
61
+
62
+ if not text.strip():
63
+ kinds = [block.type for block in response.content]
64
+ raise RuntimeError(
65
+ f"{self.model} returned no text (stop_reason={response.stop_reason}, blocks={kinds}). "
66
+ f"If stop_reason is max_tokens, raise max_tokens or lower ANTHROPIC_EFFORT."
67
+ )
68
+
69
+ usage = response.usage
70
+ return Completion(
71
+ text=text,
72
+ tokens_in=getattr(usage, "input_tokens", None),
73
+ tokens_out=getattr(usage, "output_tokens", None),
74
+ response_id=getattr(response, "id", None),
75
+ )