answer-engine-benchmark 0.1.1__tar.gz → 0.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {answer_engine_benchmark-0.1.1/src/answer_engine_benchmark.egg-info → answer_engine_benchmark-0.1.2}/PKG-INFO +34 -2
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/README.md +33 -1
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/pyproject.toml +1 -1
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark/__init__.py +1 -1
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark/cli.py +7 -4
- answer_engine_benchmark-0.1.2/src/answer_engine_benchmark/engines.py +489 -0
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark/runner.py +2 -0
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark/summary.py +3 -0
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2/src/answer_engine_benchmark.egg-info}/PKG-INFO +34 -2
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark.egg-info/SOURCES.txt +1 -0
- answer_engine_benchmark-0.1.2/tests/test_claude_cli.py +213 -0
- answer_engine_benchmark-0.1.1/src/answer_engine_benchmark/engines.py +0 -240
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/LICENSE +0 -0
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/setup.cfg +0 -0
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark/__main__.py +0 -0
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark/questions.py +0 -0
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark/scoring.py +0 -0
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark.egg-info/dependency_links.txt +0 -0
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark.egg-info/entry_points.txt +0 -0
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark.egg-info/requires.txt +0 -0
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark.egg-info/top_level.txt +0 -0
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/tests/test_engines.py +0 -0
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/tests/test_questions.py +0 -0
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/tests/test_run_and_report.py +0 -0
- {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/tests/test_scoring.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: answer-engine-benchmark
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.2
|
|
4
4
|
Summary: Ask ChatGPT, Gemini, Perplexity and Claude your buyers' questions, several times each, and count who they cite.
|
|
5
5
|
Author: Synapse Research Ltd
|
|
6
6
|
License-Expression: MIT
|
|
@@ -99,11 +99,31 @@ Gemini answer would look like it cited Google.
|
|
|
99
99
|
| `gemini` | `GEMINI_API_KEY` | `gemini-3.5-flash` | Google Search grounding |
|
|
100
100
|
| `perplexity` | `PERPLEXITY_API_KEY` | `perplexity/sonar` | Responses API `web_search` tool |
|
|
101
101
|
| `claude` | `ANTHROPIC_API_KEY` | `claude-sonnet-5` | Messages API web search tool |
|
|
102
|
+
| `claude-cli` | none, uses your `claude` login | whatever the CLI picks | Claude Code's `WebSearch` and `WebFetch` tools |
|
|
102
103
|
|
|
103
104
|
Keys are read from the environment only. They are sent in request headers and
|
|
104
105
|
removed from any error message before it is written, so they never reach the
|
|
105
106
|
results file. An engine with no key is skipped.
|
|
106
107
|
|
|
108
|
+
`claude-cli` is for people who pay for a Claude subscription and have no API
|
|
109
|
+
key. It runs the [Claude Code](https://docs.claude.com/en/docs/claude-code) CLI
|
|
110
|
+
(`claude -p`) once per call, with only the web search and fetch tools switched
|
|
111
|
+
on. It never joins a run on its own: name it with `--engine claude-cli`. If
|
|
112
|
+
`claude` is not on your PATH, the run stops before the first call and says so.
|
|
113
|
+
Log in once by running `claude`. On macOS, where the login sits in the Keychain,
|
|
114
|
+
run `claude setup-token` and export `CLAUDE_CODE_OAUTH_TOKEN` instead. Its rows
|
|
115
|
+
record a cost of 0 and `"subscription": true`. The calls use up your plan's
|
|
116
|
+
usage limits and never show on an API bill. A call gets 600 seconds by
|
|
117
|
+
default. Change that with `--timeout`, which works for every engine (the API
|
|
118
|
+
engines default to 180). `--model claude-cli=sonnet` is passed on to the CLI.
|
|
119
|
+
Its sources are the search results and fetched pages from the tool calls, plus
|
|
120
|
+
any link in the answer text.
|
|
121
|
+
|
|
122
|
+
The CLI's answers are not the API's answers. Claude Code adds its own system
|
|
123
|
+
prompt. The model and search limits come from your account and plan. Two people
|
|
124
|
+
can get different results from the same question file. Say which one you
|
|
125
|
+
used when you publish numbers.
|
|
126
|
+
|
|
107
127
|
Change a model with `--model claude=claude-opus-5`. Pick the models your
|
|
108
128
|
buyers actually use in the apps, or say in the report which ones you used.
|
|
109
129
|
|
|
@@ -148,6 +168,17 @@ API. The CLI passed the operator's own git identity to the model, and many brand
|
|
|
148
168
|
answers then told the reader the company was probably their own. The API
|
|
149
169
|
adapters here send only the question and a one-line instruction to cite sources.
|
|
150
170
|
|
|
171
|
+
The `claude-cli` engine guards against the same leak. Each call runs in a new,
|
|
172
|
+
empty temporary folder with no git repository above it. HOME and the config
|
|
173
|
+
folder are temporary too, and git is told to read no config at all. Only an
|
|
174
|
+
allow-list of environment variables gets through (PATH, locale, proxy and CA
|
|
175
|
+
settings), so `GIT_*`, `USER` and any `CLAUDE*` or `ANTHROPIC*` variables are
|
|
176
|
+
dropped. Your `CLAUDE.md` files, memory, settings, hooks and MCP servers are
|
|
177
|
+
never loaded. Only your login is copied in, and if the CLI refreshes it during
|
|
178
|
+
the call, the new one is written back. What cannot be hidden is the account:
|
|
179
|
+
the answers still come from your Claude login, on your plan, so results depend
|
|
180
|
+
on the local account.
|
|
181
|
+
|
|
151
182
|
## Example
|
|
152
183
|
|
|
153
184
|
`examples/synapse-launch-week-2026-09.md` is a real run on one company's own
|
|
@@ -173,7 +204,8 @@ pip install -e ".[test]"
|
|
|
173
204
|
pytest
|
|
174
205
|
```
|
|
175
206
|
|
|
176
|
-
|
|
207
|
+
46 tests. They mock every API and stand in a fake `claude` for the CLI, so they
|
|
208
|
+
spend nothing and need no keys or login.
|
|
177
209
|
|
|
178
210
|
## Licence
|
|
179
211
|
|
|
@@ -76,11 +76,31 @@ Gemini answer would look like it cited Google.
|
|
|
76
76
|
| `gemini` | `GEMINI_API_KEY` | `gemini-3.5-flash` | Google Search grounding |
|
|
77
77
|
| `perplexity` | `PERPLEXITY_API_KEY` | `perplexity/sonar` | Responses API `web_search` tool |
|
|
78
78
|
| `claude` | `ANTHROPIC_API_KEY` | `claude-sonnet-5` | Messages API web search tool |
|
|
79
|
+
| `claude-cli` | none, uses your `claude` login | whatever the CLI picks | Claude Code's `WebSearch` and `WebFetch` tools |
|
|
79
80
|
|
|
80
81
|
Keys are read from the environment only. They are sent in request headers and
|
|
81
82
|
removed from any error message before it is written, so they never reach the
|
|
82
83
|
results file. An engine with no key is skipped.
|
|
83
84
|
|
|
85
|
+
`claude-cli` is for people who pay for a Claude subscription and have no API
|
|
86
|
+
key. It runs the [Claude Code](https://docs.claude.com/en/docs/claude-code) CLI
|
|
87
|
+
(`claude -p`) once per call, with only the web search and fetch tools switched
|
|
88
|
+
on. It never joins a run on its own: name it with `--engine claude-cli`. If
|
|
89
|
+
`claude` is not on your PATH, the run stops before the first call and says so.
|
|
90
|
+
Log in once by running `claude`. On macOS, where the login sits in the Keychain,
|
|
91
|
+
run `claude setup-token` and export `CLAUDE_CODE_OAUTH_TOKEN` instead. Its rows
|
|
92
|
+
record a cost of 0 and `"subscription": true`. The calls use up your plan's
|
|
93
|
+
usage limits and never show on an API bill. A call gets 600 seconds by
|
|
94
|
+
default. Change that with `--timeout`, which works for every engine (the API
|
|
95
|
+
engines default to 180). `--model claude-cli=sonnet` is passed on to the CLI.
|
|
96
|
+
Its sources are the search results and fetched pages from the tool calls, plus
|
|
97
|
+
any link in the answer text.
|
|
98
|
+
|
|
99
|
+
The CLI's answers are not the API's answers. Claude Code adds its own system
|
|
100
|
+
prompt. The model and search limits come from your account and plan. Two people
|
|
101
|
+
can get different results from the same question file. Say which one you
|
|
102
|
+
used when you publish numbers.
|
|
103
|
+
|
|
84
104
|
Change a model with `--model claude=claude-opus-5`. Pick the models your
|
|
85
105
|
buyers actually use in the apps, or say in the report which ones you used.
|
|
86
106
|
|
|
@@ -125,6 +145,17 @@ API. The CLI passed the operator's own git identity to the model, and many brand
|
|
|
125
145
|
answers then told the reader the company was probably their own. The API
|
|
126
146
|
adapters here send only the question and a one-line instruction to cite sources.
|
|
127
147
|
|
|
148
|
+
The `claude-cli` engine guards against the same leak. Each call runs in a new,
|
|
149
|
+
empty temporary folder with no git repository above it. HOME and the config
|
|
150
|
+
folder are temporary too, and git is told to read no config at all. Only an
|
|
151
|
+
allow-list of environment variables gets through (PATH, locale, proxy and CA
|
|
152
|
+
settings), so `GIT_*`, `USER` and any `CLAUDE*` or `ANTHROPIC*` variables are
|
|
153
|
+
dropped. Your `CLAUDE.md` files, memory, settings, hooks and MCP servers are
|
|
154
|
+
never loaded. Only your login is copied in, and if the CLI refreshes it during
|
|
155
|
+
the call, the new one is written back. What cannot be hidden is the account:
|
|
156
|
+
the answers still come from your Claude login, on your plan, so results depend
|
|
157
|
+
on the local account.
|
|
158
|
+
|
|
128
159
|
## Example
|
|
129
160
|
|
|
130
161
|
`examples/synapse-launch-week-2026-09.md` is a real run on one company's own
|
|
@@ -150,7 +181,8 @@ pip install -e ".[test]"
|
|
|
150
181
|
pytest
|
|
151
182
|
```
|
|
152
183
|
|
|
153
|
-
|
|
184
|
+
46 tests. They mock every API and stand in a fake `claude` for the CLI, so they
|
|
185
|
+
spend nothing and need no keys or login.
|
|
154
186
|
|
|
155
187
|
## Licence
|
|
156
188
|
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "answer-engine-benchmark"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.2"
|
|
8
8
|
description = "Ask ChatGPT, Gemini, Perplexity and Claude your buyers' questions, several times each, and count who they cite."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
{answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark/cli.py
RENAMED
|
@@ -35,6 +35,8 @@ def _engine_flags(p: argparse.ArgumentParser) -> None:
|
|
|
35
35
|
p.add_argument("--engine", action="append", choices=sorted(ENGINES), help="only these engines (repeatable)")
|
|
36
36
|
p.add_argument("--model", action="append", type=_kv, default=[], metavar="ENGINE=MODEL",
|
|
37
37
|
help="override a model, e.g. --model claude=claude-opus-5")
|
|
38
|
+
p.add_argument("--timeout", type=float, metavar="SECONDS",
|
|
39
|
+
help="give up on one call after this long (default 180, claude-cli 600)")
|
|
38
40
|
p.add_argument("--runs", type=int, default=3, help="times to ask each question on each engine (default 3)")
|
|
39
41
|
p.add_argument("--max-calls", type=int, default=500,
|
|
40
42
|
help="refuse to start a run that needs more API calls than this (default 500)")
|
|
@@ -111,19 +113,20 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
111
113
|
return 0
|
|
112
114
|
|
|
113
115
|
q = load(args.questions, dict(args.set), allow_unfilled=getattr(args, "allow_unfilled", False))
|
|
114
|
-
engines = available(only=args.engine, models=dict(args.model))
|
|
116
|
+
engines = available(only=args.engine, models=dict(args.model), timeout=args.timeout)
|
|
115
117
|
if args.cmd == "check":
|
|
116
118
|
print(f"ok: {count(q)} questions in {len(q['groups'])} groups; ours = {q['ours']}")
|
|
117
119
|
missing = [f"{n} ({e.env_key})" for n, e in ENGINES.items()
|
|
118
|
-
if n not in engines and (not args.engine or n in args.engine)]
|
|
120
|
+
if e.env_key and n not in engines and (not args.engine or n in args.engine)]
|
|
119
121
|
if missing:
|
|
120
122
|
print("no key set for: " + ", ".join(missing))
|
|
121
123
|
_plan(q, engines, args.runs, args.max_calls)
|
|
122
124
|
return 0
|
|
123
125
|
|
|
124
126
|
if not engines:
|
|
125
|
-
keys = ", ".join(e.env_key for e in ENGINES.values())
|
|
126
|
-
print(f"aeb: no engine has a key. Set one or more of: {keys}
|
|
127
|
+
keys = ", ".join(e.env_key for e in ENGINES.values() if e.env_key)
|
|
128
|
+
print(f"aeb: no engine has a key. Set one or more of: {keys}, "
|
|
129
|
+
"or use --engine claude-cli", file=sys.stderr)
|
|
127
130
|
return 2
|
|
128
131
|
|
|
129
132
|
if args.cmd == "run":
|
|
@@ -0,0 +1,489 @@
|
|
|
1
|
+
"""One adapter per answer engine, over plain HTTPS so no vendor SDK is needed.
|
|
2
|
+
|
|
3
|
+
Every adapter takes a question and returns:
|
|
4
|
+
|
|
5
|
+
{"text": str, "urls": [str], "model": str, "usage": dict, "cost_usd": float | None}
|
|
6
|
+
|
|
7
|
+
`urls` holds every source the engine attached to the answer, in the order it gave
|
|
8
|
+
them, because the position of your first citation is scored. Keys are read from
|
|
9
|
+
the environment only and are never written to a log or a results file.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import os
|
|
16
|
+
import re
|
|
17
|
+
import shutil
|
|
18
|
+
import signal
|
|
19
|
+
import subprocess
|
|
20
|
+
import tempfile
|
|
21
|
+
from collections.abc import Callable
|
|
22
|
+
from dataclasses import dataclass
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
|
|
25
|
+
import requests
|
|
26
|
+
|
|
27
|
+
from .scoring import URL_RX
|
|
28
|
+
|
|
29
|
+
TIMEOUT = 180
|
|
30
|
+
SYSTEM_HINT = "Answer as you would for any user. Cite your sources with links."
|
|
31
|
+
|
|
32
|
+
# Public list prices in USD per 1M tokens (input, output), checked 2026-09. Vendors change
|
|
33
|
+
# these, so a run's cost is an estimate unless the API reports it (Perplexity does).
|
|
34
|
+
PRICES: dict[str, tuple[float, float]] = {
|
|
35
|
+
"gpt-5-mini": (0.25, 2.00),
|
|
36
|
+
"gemini-3.5-flash": (0.30, 2.50),
|
|
37
|
+
"claude-opus-5": (5.00, 25.00),
|
|
38
|
+
"claude-sonnet-5": (2.00, 10.00),
|
|
39
|
+
"perplexity/sonar": (1.00, 1.00),
|
|
40
|
+
}
|
|
41
|
+
# Claude's web search tool is billed per search on top of tokens: $10 per 1,000.
|
|
42
|
+
ANTHROPIC_SEARCH_USD = 0.01
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def token_cost(model: str, inp: int, out: int) -> float | None:
|
|
46
|
+
p = PRICES.get(model)
|
|
47
|
+
if p is None:
|
|
48
|
+
# Dated snapshots such as gpt-5-mini-2025-08-07 price like their base name.
|
|
49
|
+
p = next((v for k, v in PRICES.items() if model.startswith(k + "-")), None)
|
|
50
|
+
return round((inp * p[0] + out * p[1]) / 1e6, 6) if p else None
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
# ----------------------------------------------------------------------------- OpenAI ----
|
|
54
|
+
def openai_answer(question: str, *, key: str, model: str = "gpt-5-mini", timeout: float = TIMEOUT) -> dict:
|
|
55
|
+
body = {
|
|
56
|
+
"model": model,
|
|
57
|
+
"tools": [{"type": "web_search"}],
|
|
58
|
+
"tool_choice": "auto",
|
|
59
|
+
"input": question,
|
|
60
|
+
"instructions": SYSTEM_HINT,
|
|
61
|
+
}
|
|
62
|
+
r = requests.post("https://api.openai.com/v1/responses",
|
|
63
|
+
headers={"Authorization": f"Bearer {key}"}, json=body, timeout=timeout)
|
|
64
|
+
r.raise_for_status()
|
|
65
|
+
d = r.json()
|
|
66
|
+
text, urls = [], []
|
|
67
|
+
for item in d.get("output", []) or []:
|
|
68
|
+
if item.get("type") != "message":
|
|
69
|
+
continue
|
|
70
|
+
for c in item.get("content", []) or []:
|
|
71
|
+
if c.get("type") == "output_text":
|
|
72
|
+
text.append(c.get("text", ""))
|
|
73
|
+
urls += [a["url"] for a in c.get("annotations", []) or []
|
|
74
|
+
if a.get("type") == "url_citation" and a.get("url")]
|
|
75
|
+
u = d.get("usage", {}) or {}
|
|
76
|
+
return {
|
|
77
|
+
"text": "\n".join(text),
|
|
78
|
+
"urls": urls,
|
|
79
|
+
"model": d.get("model", model),
|
|
80
|
+
"usage": u,
|
|
81
|
+
"cost_usd": token_cost(d.get("model", model), u.get("input_tokens", 0), u.get("output_tokens", 0)),
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
# ----------------------------------------------------------------------------- Gemini ----
|
|
86
|
+
def gemini_answer(question: str, *, key: str, model: str = "gemini-3.5-flash", timeout: float = TIMEOUT) -> dict:
|
|
87
|
+
body = {"contents": [{"parts": [{"text": question}]}], "tools": [{"google_search": {}}]}
|
|
88
|
+
# The key goes in a header, not the URL. In the query string it ends up in the text of
|
|
89
|
+
# any HTTPError, and from there in the results file.
|
|
90
|
+
r = requests.post(
|
|
91
|
+
f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent",
|
|
92
|
+
headers={"x-goog-api-key": key}, json=body, timeout=timeout,
|
|
93
|
+
)
|
|
94
|
+
r.raise_for_status()
|
|
95
|
+
d = r.json()
|
|
96
|
+
cand = (d.get("candidates") or [{}])[0]
|
|
97
|
+
text = "\n".join(p.get("text", "") for p in (cand.get("content") or {}).get("parts", []) or [])
|
|
98
|
+
gm = cand.get("groundingMetadata") or {}
|
|
99
|
+
urls = [(ch.get("web") or {}).get("uri") for ch in gm.get("groundingChunks", []) or []]
|
|
100
|
+
u = d.get("usageMetadata", {}) or {}
|
|
101
|
+
return {
|
|
102
|
+
"text": text,
|
|
103
|
+
# These are vertexaisearch.cloud.google.com redirect URLs; resolve.py turns them
|
|
104
|
+
# into the real sources before scoring.
|
|
105
|
+
"urls": [x for x in urls if x],
|
|
106
|
+
"model": model,
|
|
107
|
+
"usage": u,
|
|
108
|
+
"cost_usd": token_cost(model, u.get("promptTokenCount", 0), u.get("candidatesTokenCount", 0)),
|
|
109
|
+
"grounding_queries": gm.get("webSearchQueries", []),
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
# ----------------------------------------------------------------------------- Claude ----
|
|
114
|
+
def anthropic_answer(question: str, *, key: str, model: str = "claude-sonnet-5", timeout: float = TIMEOUT) -> dict:
|
|
115
|
+
"""Claude through the Messages API with the server-side web search tool.
|
|
116
|
+
|
|
117
|
+
No model fallback is configured on purpose: an answer from a different model would be
|
|
118
|
+
scored as this one. A refusal is recorded as an error instead.
|
|
119
|
+
"""
|
|
120
|
+
messages: list[dict] = [{"role": "user", "content": question}]
|
|
121
|
+
headers = {"x-api-key": key, "anthropic-version": "2023-06-01"}
|
|
122
|
+
text: list[str] = []
|
|
123
|
+
urls: list[str] = []
|
|
124
|
+
usage = {"input_tokens": 0, "output_tokens": 0, "web_search_requests": 0}
|
|
125
|
+
served = model
|
|
126
|
+
# A long server-tool turn can stop with `pause_turn`. Sending the partial turn back
|
|
127
|
+
# lets it continue. Three rounds is plenty for a single question.
|
|
128
|
+
for _ in range(3):
|
|
129
|
+
body = {
|
|
130
|
+
"model": model,
|
|
131
|
+
"max_tokens": 16000,
|
|
132
|
+
"system": SYSTEM_HINT,
|
|
133
|
+
"messages": messages,
|
|
134
|
+
"tools": [{"type": "web_search_20260209", "name": "web_search", "max_uses": 5}],
|
|
135
|
+
}
|
|
136
|
+
r = requests.post("https://api.anthropic.com/v1/messages", headers=headers, json=body,
|
|
137
|
+
timeout=timeout)
|
|
138
|
+
r.raise_for_status()
|
|
139
|
+
d = r.json()
|
|
140
|
+
served = d.get("model", served)
|
|
141
|
+
u = d.get("usage", {}) or {}
|
|
142
|
+
usage["input_tokens"] += u.get("input_tokens", 0)
|
|
143
|
+
usage["output_tokens"] += u.get("output_tokens", 0)
|
|
144
|
+
usage["web_search_requests"] += (u.get("server_tool_use") or {}).get("web_search_requests", 0)
|
|
145
|
+
for b in d.get("content", []) or []:
|
|
146
|
+
if b.get("type") == "text":
|
|
147
|
+
text.append(b.get("text", ""))
|
|
148
|
+
urls += [c["url"] for c in b.get("citations", []) or [] if c.get("url")]
|
|
149
|
+
elif b.get("type") == "web_search_tool_result" and isinstance(b.get("content"), list):
|
|
150
|
+
urls += [x["url"] for x in b["content"] if isinstance(x, dict) and x.get("url")]
|
|
151
|
+
stop = d.get("stop_reason")
|
|
152
|
+
if stop == "refusal":
|
|
153
|
+
raise RuntimeError(f"refusal: {(d.get('stop_details') or {}).get('category')}")
|
|
154
|
+
if stop != "pause_turn":
|
|
155
|
+
break
|
|
156
|
+
messages = [messages[0], {"role": "assistant", "content": d.get("content", [])}]
|
|
157
|
+
cost = token_cost(served, usage["input_tokens"], usage["output_tokens"])
|
|
158
|
+
if cost is not None:
|
|
159
|
+
cost = round(cost + usage["web_search_requests"] * ANTHROPIC_SEARCH_USD, 6)
|
|
160
|
+
return {"text": "".join(text), "urls": urls, "model": served, "usage": usage, "cost_usd": cost}
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
# ------------------------------------------------------------------------- Perplexity ----
|
|
164
|
+
def perplexity_answer(question: str, *, key: str, model: str = "perplexity/sonar", timeout: float = TIMEOUT) -> dict:
|
|
165
|
+
"""Perplexity's Responses API. Web search is OFF unless the tool is passed.
|
|
166
|
+
|
|
167
|
+
Without `tools=[{"type": "web_search"}]` the call still succeeds and returns a fluent
|
|
168
|
+
answer naming real companies, with zero citations. For a benchmark scored on
|
|
169
|
+
citations that is the worst failure there is: no error, and no data.
|
|
170
|
+
"""
|
|
171
|
+
body = {"model": model, "input": question, "tools": [{"type": "web_search"}]}
|
|
172
|
+
r = requests.post("https://api.perplexity.ai/v1/responses",
|
|
173
|
+
headers={"Authorization": f"Bearer {key}"}, json=body, timeout=timeout)
|
|
174
|
+
r.raise_for_status()
|
|
175
|
+
d = r.json()
|
|
176
|
+
text, urls = "", []
|
|
177
|
+
for item in d.get("output", []) or []:
|
|
178
|
+
if item.get("type") == "search_results":
|
|
179
|
+
urls += [x.get("url") for x in item.get("results") or [] if x.get("url")]
|
|
180
|
+
for c in item.get("content", []) or []:
|
|
181
|
+
text += c.get("text", "")
|
|
182
|
+
urls += [a.get("url") for a in c.get("annotations") or [] if a.get("url")]
|
|
183
|
+
u = d.get("usage", {}) or {}
|
|
184
|
+
reported = (u.get("cost") or {}).get("total_cost")
|
|
185
|
+
return {
|
|
186
|
+
"text": text,
|
|
187
|
+
"urls": list(dict.fromkeys(urls)),
|
|
188
|
+
"model": model,
|
|
189
|
+
"usage": u,
|
|
190
|
+
"cost_usd": round(float(reported), 6) if reported is not None
|
|
191
|
+
else token_cost(model, u.get("input_tokens", 0), u.get("output_tokens", 0)),
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
# ------------------------------------------------------------------------- Claude CLI ----
|
|
196
|
+
# Claude through the locally logged-in Claude Code CLI (`claude -p`), for people with a Claude
|
|
197
|
+
# subscription and no API key. The CLI is a coding agent that normally reads a lot about the
|
|
198
|
+
# person running it: git identity, CLAUDE.md files, memory, settings, hooks, MCP servers. In an
|
|
199
|
+
# earlier run it passed the operator's git username to the model, and brand answers then told
|
|
200
|
+
# the reader the company was probably their own. So every call runs as a stranger: HOME, the
|
|
201
|
+
# config dir and the working directory are fresh empty temp dirs, git reads no config, and the
|
|
202
|
+
# environment is an allow-list. Only the login credentials are copied in.
|
|
203
|
+
CLAUDE_CLI_TIMEOUT = 600
|
|
204
|
+
# Everything else is dropped: USER, EMAIL, every GIT_* variable, the calling session's CLAUDE*
|
|
205
|
+
# and ANTHROPIC* variables, SSH agents and so on.
|
|
206
|
+
_CLI_ENV_KEEP = ("PATH", "LANG", "LANGUAGE", "LC_ALL", "TZ", "TMPDIR",
|
|
207
|
+
"HTTP_PROXY", "HTTPS_PROXY", "NO_PROXY", "http_proxy", "https_proxy", "no_proxy",
|
|
208
|
+
"SSL_CERT_FILE", "SSL_CERT_DIR", "NODE_EXTRA_CA_CERTS")
|
|
209
|
+
# A long-lived token from `claude setup-token`. When set, no credentials file is copied.
|
|
210
|
+
_CLI_TOKEN_VAR = "CLAUDE_CODE_OAUTH_TOKEN"
|
|
211
|
+
_CLI_TOOLS = "WebSearch,WebFetch"
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
class EngineUnavailable(ValueError):
|
|
215
|
+
"""An engine was asked for but cannot run on this machine."""
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def claude_exe(env: dict | None = None) -> str:
|
|
219
|
+
env = os.environ if env is None else env
|
|
220
|
+
exe = shutil.which("claude", path=env.get("PATH"))
|
|
221
|
+
if not exe:
|
|
222
|
+
raise EngineUnavailable("claude-cli needs the Claude Code CLI: no `claude` on PATH. "
|
|
223
|
+
"Install it and run `claude` once to log in.")
|
|
224
|
+
return exe
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _cli_config_dir(env: dict) -> Path:
|
|
228
|
+
return Path(env.get("CLAUDE_CONFIG_DIR") or Path(env.get("HOME") or Path.home()) / ".claude")
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def cli_env(home: str, src: dict | None = None) -> dict:
|
|
232
|
+
"""The environment of someone who has never met the operator."""
|
|
233
|
+
src = os.environ if src is None else src
|
|
234
|
+
env = {k: src[k] for k in (*_CLI_ENV_KEEP, _CLI_TOKEN_VAR) if src.get(k)}
|
|
235
|
+
env.update({
|
|
236
|
+
"HOME": home,
|
|
237
|
+
"CLAUDE_CONFIG_DIR": os.path.join(home, ".claude"),
|
|
238
|
+
"XDG_CONFIG_HOME": os.path.join(home, ".config"),
|
|
239
|
+
"XDG_DATA_HOME": os.path.join(home, ".local", "share"),
|
|
240
|
+
"XDG_CACHE_HOME": os.path.join(home, ".cache"),
|
|
241
|
+
"XDG_STATE_HOME": os.path.join(home, ".local", "state"),
|
|
242
|
+
"GIT_CONFIG_GLOBAL": os.devnull,
|
|
243
|
+
"GIT_CONFIG_NOSYSTEM": "1",
|
|
244
|
+
"DISABLE_AUTOUPDATER": "1",
|
|
245
|
+
"USER": "user",
|
|
246
|
+
"LOGNAME": "user",
|
|
247
|
+
})
|
|
248
|
+
return env
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _stage_auth(cfg_dir: str, src: dict) -> Path | None:
|
|
252
|
+
"""Copy only the login credentials into the empty config dir. Returns the original's path,
|
|
253
|
+
or None when the token variable carries the login instead."""
|
|
254
|
+
os.makedirs(cfg_dir, mode=0o700, exist_ok=True)
|
|
255
|
+
if src.get(_CLI_TOKEN_VAR):
|
|
256
|
+
return None
|
|
257
|
+
real = _cli_config_dir(src) / ".credentials.json"
|
|
258
|
+
if not real.is_file():
|
|
259
|
+
# macOS keeps the login in the Keychain, where it cannot be copied.
|
|
260
|
+
raise EngineUnavailable(f"claude-cli: no login found at {real}. Run `claude setup-token` and "
|
|
261
|
+
f"export {_CLI_TOKEN_VAR}, or run `claude` once to log in.")
|
|
262
|
+
fd = os.open(os.path.join(cfg_dir, ".credentials.json"), os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
|
|
263
|
+
with os.fdopen(fd, "wb") as f:
|
|
264
|
+
f.write(real.read_bytes())
|
|
265
|
+
return real
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def _expires_at(raw: bytes) -> int:
|
|
269
|
+
try:
|
|
270
|
+
return int(json.loads(raw)["claudeAiOauth"]["expiresAt"])
|
|
271
|
+
except (ValueError, KeyError, TypeError):
|
|
272
|
+
return 0
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def _write_back_auth(real: Path, staged: str) -> None:
|
|
276
|
+
"""If the CLI refreshed its login inside the temp dir, hand the new one back. The refresh
|
|
277
|
+
token rotates, so keeping the old one could log the operator's own sessions out."""
|
|
278
|
+
try:
|
|
279
|
+
new, old = Path(staged).read_bytes(), real.read_bytes()
|
|
280
|
+
except OSError:
|
|
281
|
+
return
|
|
282
|
+
if new == old or _expires_at(new) <= _expires_at(old):
|
|
283
|
+
return
|
|
284
|
+
tmp = real.with_name(real.name + ".aeb.tmp")
|
|
285
|
+
fd = os.open(tmp, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
|
|
286
|
+
with os.fdopen(fd, "wb") as f:
|
|
287
|
+
f.write(new)
|
|
288
|
+
os.replace(tmp, real)
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _run_cli(cmd: list[str], prompt: str, *, cwd: str, env: dict, timeout: float) -> subprocess.CompletedProcess:
|
|
292
|
+
# The prompt goes in on stdin, so a question that starts with "-" is never read as a flag.
|
|
293
|
+
# Own process group, so a timeout kills the CLI and anything it started.
|
|
294
|
+
p = subprocess.Popen(cmd, cwd=cwd, env=env, stdin=subprocess.PIPE, stdout=subprocess.PIPE,
|
|
295
|
+
stderr=subprocess.PIPE, text=True, start_new_session=True)
|
|
296
|
+
try:
|
|
297
|
+
out, err = p.communicate(prompt, timeout=timeout)
|
|
298
|
+
except subprocess.TimeoutExpired:
|
|
299
|
+
try:
|
|
300
|
+
os.killpg(p.pid, signal.SIGKILL)
|
|
301
|
+
except OSError:
|
|
302
|
+
p.kill()
|
|
303
|
+
p.communicate()
|
|
304
|
+
raise TimeoutError(f"claude-cli: no answer after {timeout:g}s") from None
|
|
305
|
+
return subprocess.CompletedProcess(cmd, p.returncode, out, err)
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def _tool_text(content) -> str:
|
|
309
|
+
if isinstance(content, str):
|
|
310
|
+
return content
|
|
311
|
+
if isinstance(content, list):
|
|
312
|
+
return "".join(c.get("text", "") for c in content if isinstance(c, dict))
|
|
313
|
+
return ""
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
def _search_links(text: str) -> list[str]:
|
|
317
|
+
"""WebSearch results arrive as text with a `Links: [{"title": ..., "url": ...}]` list."""
|
|
318
|
+
i = text.find("Links: [")
|
|
319
|
+
if i < 0:
|
|
320
|
+
return []
|
|
321
|
+
try:
|
|
322
|
+
links, _ = json.JSONDecoder().raw_decode(text, i + len("Links: "))
|
|
323
|
+
except ValueError:
|
|
324
|
+
return []
|
|
325
|
+
return [x["url"] for x in links if isinstance(x, dict) and isinstance(x.get("url"), str)]
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def parse_cli_stream(stdout: str) -> dict:
|
|
329
|
+
"""Read `--output-format stream-json`. Sources come from the tool stream: the CLI's web
|
|
330
|
+
tools run client-side, so the usage counter for server-side searches stays at 0."""
|
|
331
|
+
calls: dict[str, tuple[str, dict]] = {}
|
|
332
|
+
tools: list[str] = []
|
|
333
|
+
urls: list[str] = []
|
|
334
|
+
final: dict = {}
|
|
335
|
+
for line in stdout.splitlines():
|
|
336
|
+
try:
|
|
337
|
+
ev = json.loads(line)
|
|
338
|
+
except ValueError:
|
|
339
|
+
continue
|
|
340
|
+
if not isinstance(ev, dict):
|
|
341
|
+
continue
|
|
342
|
+
blocks = (ev.get("message") or {}).get("content") or []
|
|
343
|
+
if ev.get("type") == "assistant":
|
|
344
|
+
for b in blocks:
|
|
345
|
+
if isinstance(b, dict) and b.get("type") == "tool_use" and b.get("name"):
|
|
346
|
+
tools.append(b["name"])
|
|
347
|
+
calls[b.get("id", "")] = (b["name"], b.get("input") or {})
|
|
348
|
+
elif ev.get("type") == "user":
|
|
349
|
+
for b in blocks:
|
|
350
|
+
if not isinstance(b, dict) or b.get("type") != "tool_result" or b.get("is_error"):
|
|
351
|
+
continue
|
|
352
|
+
name, inp = calls.get(b.get("tool_use_id", ""), ("", {}))
|
|
353
|
+
if name == "WebSearch":
|
|
354
|
+
urls += _search_links(_tool_text(b.get("content")))
|
|
355
|
+
elif name == "WebFetch" and isinstance(inp.get("url"), str):
|
|
356
|
+
urls.append(inp["url"])
|
|
357
|
+
elif ev.get("type") == "result":
|
|
358
|
+
final = ev
|
|
359
|
+
if not final:
|
|
360
|
+
raise RuntimeError("claude-cli: no result in the CLI output")
|
|
361
|
+
if final.get("is_error") or final.get("subtype") != "success":
|
|
362
|
+
raise RuntimeError(f"claude-cli: {final.get('subtype')}: {str(final.get('result', ''))[:200]}")
|
|
363
|
+
text = final.get("result") or ""
|
|
364
|
+
mu = final.get("modelUsage") or {}
|
|
365
|
+
# The CLI may use a small model for side jobs; the answer comes from the one that wrote most.
|
|
366
|
+
model = max(mu, key=lambda m: (mu[m] or {}).get("outputTokens", 0)) if mu else ""
|
|
367
|
+
return {
|
|
368
|
+
"text": text,
|
|
369
|
+
"urls": list(dict.fromkeys(urls + URL_RX.findall(text))),
|
|
370
|
+
"model": model,
|
|
371
|
+
"usage": {
|
|
372
|
+
"turns": final.get("num_turns"),
|
|
373
|
+
"tools": tools,
|
|
374
|
+
"web_search_requests": tools.count("WebSearch"),
|
|
375
|
+
"web_fetch_requests": tools.count("WebFetch"),
|
|
376
|
+
"duration_ms": final.get("duration_ms"),
|
|
377
|
+
# What the CLI says the call would cost at API prices. Not charged on a subscription.
|
|
378
|
+
"api_equivalent_usd": final.get("total_cost_usd"),
|
|
379
|
+
},
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def claude_cli_answer(question: str, *, key: str = "", model: str = "default",
|
|
384
|
+
timeout: float = CLAUDE_CLI_TIMEOUT, env: dict | None = None) -> dict:
|
|
385
|
+
"""Claude through the local Claude Code CLI, with only WebSearch and WebFetch available.
|
|
386
|
+
|
|
387
|
+
Uses the operator's Claude subscription, not API credit, so `cost_usd` is 0 and the row
|
|
388
|
+
is flagged `subscription`. `key` is unused. `model="default"` lets the CLI pick.
|
|
389
|
+
"""
|
|
390
|
+
src = os.environ if env is None else env
|
|
391
|
+
exe = claude_exe(src) # from the caller's PATH, before the environment is replaced
|
|
392
|
+
with tempfile.TemporaryDirectory(prefix="aeb-home-") as home, \
|
|
393
|
+
tempfile.TemporaryDirectory(prefix="aeb-cwd-") as cwd:
|
|
394
|
+
cenv = cli_env(home, src)
|
|
395
|
+
cenv["GIT_CEILING_DIRECTORIES"] = os.path.dirname(cwd) # never find a repo above the temp dir
|
|
396
|
+
real = _stage_auth(cenv["CLAUDE_CONFIG_DIR"], src)
|
|
397
|
+
mcp = os.path.join(home, "mcp.json")
|
|
398
|
+
with open(mcp, "w", encoding="utf-8") as f:
|
|
399
|
+
f.write('{"mcpServers": {}}')
|
|
400
|
+
cmd = [exe, "-p", "--output-format", "stream-json", "--verbose",
|
|
401
|
+
"--append-system-prompt", SYSTEM_HINT, "--max-turns", "12",
|
|
402
|
+
"--tools", _CLI_TOOLS, "--allowedTools", _CLI_TOOLS,
|
|
403
|
+
"--strict-mcp-config", "--mcp-config", mcp, "--setting-sources", "user"]
|
|
404
|
+
if model and model != "default":
|
|
405
|
+
cmd += ["--model", model]
|
|
406
|
+
try:
|
|
407
|
+
r = _run_cli(cmd, question, cwd=cwd, env=cenv, timeout=timeout)
|
|
408
|
+
finally:
|
|
409
|
+
if real is not None:
|
|
410
|
+
_write_back_auth(real, os.path.join(cenv["CLAUDE_CONFIG_DIR"], ".credentials.json"))
|
|
411
|
+
try:
|
|
412
|
+
out = parse_cli_stream(r.stdout or "")
|
|
413
|
+
except RuntimeError as e:
|
|
414
|
+
detail = (r.stderr or "").strip()[:200]
|
|
415
|
+
raise RuntimeError(f"{e} (exit {r.returncode}){': ' + detail if detail else ''}") from None
|
|
416
|
+
if r.returncode != 0:
|
|
417
|
+
raise RuntimeError(f"claude-cli exit {r.returncode}: {(r.stderr or '').strip()[:200]}")
|
|
418
|
+
out["model"] = out["model"] or model
|
|
419
|
+
out["cost_usd"] = 0.0
|
|
420
|
+
out["subscription"] = True
|
|
421
|
+
return out
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
@dataclass(frozen=True)
|
|
425
|
+
class Engine:
|
|
426
|
+
name: str
|
|
427
|
+
env_key: str # empty: the engine needs no key and runs only when asked for by name
|
|
428
|
+
fn: Callable[..., dict]
|
|
429
|
+
default_model: str
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
ENGINES: dict[str, Engine] = {
|
|
433
|
+
"openai": Engine("openai", "OPENAI_API_KEY", openai_answer, "gpt-5-mini"),
|
|
434
|
+
"gemini": Engine("gemini", "GEMINI_API_KEY", gemini_answer, "gemini-3.5-flash"),
|
|
435
|
+
"claude": Engine("claude", "ANTHROPIC_API_KEY", anthropic_answer, "claude-sonnet-5"),
|
|
436
|
+
"perplexity": Engine("perplexity", "PERPLEXITY_API_KEY", perplexity_answer, "perplexity/sonar"),
|
|
437
|
+
"claude-cli": Engine("claude-cli", "", claude_cli_answer, "default"),
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
@dataclass(frozen=True)
|
|
442
|
+
class Bound:
|
|
443
|
+
"""An engine with its key and model resolved. `key` is never printed or stored."""
|
|
444
|
+
|
|
445
|
+
name: str
|
|
446
|
+
model: str
|
|
447
|
+
key: str
|
|
448
|
+
fn: Callable[..., dict]
|
|
449
|
+
timeout: float | None = None # None: the adapter's own default (180 s, claude-cli 600 s)
|
|
450
|
+
|
|
451
|
+
def ask(self, question: str) -> dict:
|
|
452
|
+
if self.timeout is None:
|
|
453
|
+
return self.fn(question, key=self.key, model=self.model)
|
|
454
|
+
return self.fn(question, key=self.key, model=self.model, timeout=self.timeout)
|
|
455
|
+
|
|
456
|
+
def __repr__(self) -> str: # keep the key out of tracebacks and debug output
|
|
457
|
+
return f"Bound(name={self.name!r}, model={self.model!r})"
|
|
458
|
+
|
|
459
|
+
|
|
460
|
+
def available(env: dict | None = None, *, only: list[str] | None = None,
|
|
461
|
+
models: dict[str, str] | None = None, timeout: float | None = None) -> dict[str, Bound]:
|
|
462
|
+
"""Engines whose key is set, plus keyless engines named in `only`. `models` overrides a
|
|
463
|
+
default model per engine and `timeout` the per-call limit in seconds."""
|
|
464
|
+
env = os.environ if env is None else env
|
|
465
|
+
models = models or {}
|
|
466
|
+
out = {}
|
|
467
|
+
for name, e in ENGINES.items():
|
|
468
|
+
if only and name not in only:
|
|
469
|
+
continue
|
|
470
|
+
if e.env_key:
|
|
471
|
+
key = env.get(e.env_key, "")
|
|
472
|
+
if not key:
|
|
473
|
+
continue
|
|
474
|
+
elif only:
|
|
475
|
+
claude_exe(env) # fail now, with a clear message, rather than on every call
|
|
476
|
+
key = ""
|
|
477
|
+
else:
|
|
478
|
+
continue
|
|
479
|
+
out[name] = Bound(name, models.get(name, e.default_model), key, e.fn, timeout)
|
|
480
|
+
return out
|
|
481
|
+
|
|
482
|
+
|
|
483
|
+
def redact(message: str, secrets: list[str]) -> str:
|
|
484
|
+
"""Remove every key from an error message before it is logged."""
|
|
485
|
+
for s in secrets:
|
|
486
|
+
if s and len(s) >= 6:
|
|
487
|
+
message = message.replace(s, "[redacted]")
|
|
488
|
+
# Belt and braces for key-shaped strings that were not in the list.
|
|
489
|
+
return re.sub(r"(key=)[^&\s]+", r"\1[redacted]", message)
|
|
@@ -50,6 +50,8 @@ def run(q: dict, engines: dict[str, Bound], *, runs: int = 3, out: Path,
|
|
|
50
50
|
"urls": ans.get("urls", []),
|
|
51
51
|
"usage": ans.get("usage", {}),
|
|
52
52
|
"cost_usd": ans.get("cost_usd"),
|
|
53
|
+
# True when the call used a subscription (claude-cli), so cost_usd is 0.
|
|
54
|
+
"subscription": bool(ans.get("subscription")),
|
|
53
55
|
**score(ans, q["ours"], q["brand_terms"], stale_markers=q["stale_markers"]),
|
|
54
56
|
}
|
|
55
57
|
f.write(json.dumps(rec, ensure_ascii=False) + "\n")
|
|
@@ -85,6 +85,9 @@ def render(records: list[dict], ours: list[str], *, anonymise: bool = False, top
|
|
|
85
85
|
lines.append(f"| {k} | {b['answered']} | {b['errors']} | {_pct(b['mention_rate'])} | "
|
|
86
86
|
f"{b['cited']} of {b['answered']} ({_pct(b['citation_rate'])}) | "
|
|
87
87
|
f"{b['avg_position'] or '-'} | {b['cost']:.4f} |")
|
|
88
|
+
subs = sorted({r["engine"] for r in records if r.get("subscription")})
|
|
89
|
+
if subs:
|
|
90
|
+
lines += ["", f"{', '.join(subs)} ran on a subscription, not API credit, so its cost shows as 0."]
|
|
88
91
|
|
|
89
92
|
engines = sorted({r["engine"] for r in records})
|
|
90
93
|
groups = list(dict.fromkeys(r["group"] for r in records))
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: answer-engine-benchmark
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.2
|
|
4
4
|
Summary: Ask ChatGPT, Gemini, Perplexity and Claude your buyers' questions, several times each, and count who they cite.
|
|
5
5
|
Author: Synapse Research Ltd
|
|
6
6
|
License-Expression: MIT
|
|
@@ -99,11 +99,31 @@ Gemini answer would look like it cited Google.
|
|
|
99
99
|
| `gemini` | `GEMINI_API_KEY` | `gemini-3.5-flash` | Google Search grounding |
|
|
100
100
|
| `perplexity` | `PERPLEXITY_API_KEY` | `perplexity/sonar` | Responses API `web_search` tool |
|
|
101
101
|
| `claude` | `ANTHROPIC_API_KEY` | `claude-sonnet-5` | Messages API web search tool |
|
|
102
|
+
| `claude-cli` | none, uses your `claude` login | whatever the CLI picks | Claude Code's `WebSearch` and `WebFetch` tools |
|
|
102
103
|
|
|
103
104
|
Keys are read from the environment only. They are sent in request headers and
|
|
104
105
|
removed from any error message before it is written, so they never reach the
|
|
105
106
|
results file. An engine with no key is skipped.
|
|
106
107
|
|
|
108
|
+
`claude-cli` is for people who pay for a Claude subscription and have no API
|
|
109
|
+
key. It runs the [Claude Code](https://docs.claude.com/en/docs/claude-code) CLI
|
|
110
|
+
(`claude -p`) once per call, with only the web search and fetch tools switched
|
|
111
|
+
on. It never joins a run on its own: name it with `--engine claude-cli`. If
|
|
112
|
+
`claude` is not on your PATH, the run stops before the first call and says so.
|
|
113
|
+
Log in once by running `claude`. On macOS, where the login sits in the Keychain,
|
|
114
|
+
run `claude setup-token` and export `CLAUDE_CODE_OAUTH_TOKEN` instead. Its rows
|
|
115
|
+
record a cost of 0 and `"subscription": true`. The calls use up your plan's
|
|
116
|
+
usage limits and never show on an API bill. A call gets 600 seconds by
|
|
117
|
+
default. Change that with `--timeout`, which works for every engine (the API
|
|
118
|
+
engines default to 180). `--model claude-cli=sonnet` is passed on to the CLI.
|
|
119
|
+
Its sources are the search results and fetched pages from the tool calls, plus
|
|
120
|
+
any link in the answer text.
|
|
121
|
+
|
|
122
|
+
The CLI's answers are not the API's answers. Claude Code adds its own system
|
|
123
|
+
prompt. The model and search limits come from your account and plan. Two people
|
|
124
|
+
can get different results from the same question file. Say which one you
|
|
125
|
+
used when you publish numbers.
|
|
126
|
+
|
|
107
127
|
Change a model with `--model claude=claude-opus-5`. Pick the models your
|
|
108
128
|
buyers actually use in the apps, or say in the report which ones you used.
|
|
109
129
|
|
|
@@ -148,6 +168,17 @@ API. The CLI passed the operator's own git identity to the model, and many brand
|
|
|
148
168
|
answers then told the reader the company was probably their own. The API
|
|
149
169
|
adapters here send only the question and a one-line instruction to cite sources.
|
|
150
170
|
|
|
171
|
+
The `claude-cli` engine guards against the same leak. Each call runs in a new,
|
|
172
|
+
empty temporary folder with no git repository above it. HOME and the config
|
|
173
|
+
folder are temporary too, and git is told to read no config at all. Only an
|
|
174
|
+
allow-list of environment variables gets through (PATH, locale, proxy and CA
|
|
175
|
+
settings), so `GIT_*`, `USER` and any `CLAUDE*` or `ANTHROPIC*` variables are
|
|
176
|
+
dropped. Your `CLAUDE.md` files, memory, settings, hooks and MCP servers are
|
|
177
|
+
never loaded. Only your login is copied in, and if the CLI refreshes it during
|
|
178
|
+
the call, the new one is written back. What cannot be hidden is the account:
|
|
179
|
+
the answers still come from your Claude login, on your plan, so results depend
|
|
180
|
+
on the local account.
|
|
181
|
+
|
|
151
182
|
## Example
|
|
152
183
|
|
|
153
184
|
`examples/synapse-launch-week-2026-09.md` is a real run on one company's own
|
|
@@ -173,7 +204,8 @@ pip install -e ".[test]"
|
|
|
173
204
|
pytest
|
|
174
205
|
```
|
|
175
206
|
|
|
176
|
-
|
|
207
|
+
46 tests. They mock every API and stand in a fake `claude` for the CLI, so they
|
|
208
|
+
spend nothing and need no keys or login.
|
|
177
209
|
|
|
178
210
|
## Licence
|
|
179
211
|
|
|
@@ -15,6 +15,7 @@ src/answer_engine_benchmark.egg-info/dependency_links.txt
|
|
|
15
15
|
src/answer_engine_benchmark.egg-info/entry_points.txt
|
|
16
16
|
src/answer_engine_benchmark.egg-info/requires.txt
|
|
17
17
|
src/answer_engine_benchmark.egg-info/top_level.txt
|
|
18
|
+
tests/test_claude_cli.py
|
|
18
19
|
tests/test_engines.py
|
|
19
20
|
tests/test_questions.py
|
|
20
21
|
tests/test_run_and_report.py
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""claude-cli engine tests. A fake `claude` script stands in for the real CLI, so nothing
|
|
2
|
+
here needs a login or makes a network call."""
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
import stat
|
|
7
|
+
import sys
|
|
8
|
+
import textwrap
|
|
9
|
+
import time
|
|
10
|
+
|
|
11
|
+
import pytest
|
|
12
|
+
|
|
13
|
+
from answer_engine_benchmark import engines as E
|
|
14
|
+
from answer_engine_benchmark.cli import main
|
|
15
|
+
from answer_engine_benchmark.runner import run
|
|
16
|
+
|
|
17
|
+
SEARCH_RESULT = ('Web search results for query: "acme"\n\nLinks: [{"title": "A [1]", "url": "https://s.test/1"},'
|
|
18
|
+
'{"title": "B", "url": "https://acme.test/about"}]\n\nMore text')
|
|
19
|
+
|
|
20
|
+
STREAM = [
|
|
21
|
+
{"type": "system", "subtype": "init", "tools": ["WebFetch", "WebSearch"]},
|
|
22
|
+
{"type": "assistant", "message": {"content": [
|
|
23
|
+
{"type": "tool_use", "id": "t1", "name": "WebSearch", "input": {"query": "acme"}}]}},
|
|
24
|
+
{"type": "user", "message": {"content": [{"type": "tool_result", "tool_use_id": "t1", "content": SEARCH_RESULT}]}},
|
|
25
|
+
{"type": "assistant", "message": {"content": [
|
|
26
|
+
{"type": "tool_use", "id": "t2", "name": "WebFetch", "input": {"url": "https://down.test/"}},
|
|
27
|
+
{"type": "tool_use", "id": "t3", "name": "WebFetch", "input": {"url": "https://f.test/page"}}]}},
|
|
28
|
+
{"type": "user", "message": {"content": [
|
|
29
|
+
{"type": "tool_result", "tool_use_id": "t2", "content": "ENOTFOUND", "is_error": True},
|
|
30
|
+
{"type": "tool_result", "tool_use_id": "t3", "content": [{"type": "text", "text": "page"}]}]}},
|
|
31
|
+
{"type": "result", "subtype": "success", "is_error": False, "num_turns": 3, "duration_ms": 1200,
|
|
32
|
+
"result": "Acme makes widgets ([About](https://acme.test/about)). See https://t.test/x.",
|
|
33
|
+
"total_cost_usd": 0.0421,
|
|
34
|
+
"modelUsage": {"claude-haiku-5": {"outputTokens": 40}, "claude-opus-5": {"outputTokens": 900}}},
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def make_fake(tmp_path, *, stream=None, exit_code=0, stderr="", sleep=0.0):
|
|
39
|
+
"""Put a fake `claude` on a fresh PATH. It records what it was given, then replies."""
|
|
40
|
+
bindir = tmp_path / "bin"
|
|
41
|
+
bindir.mkdir(exist_ok=True)
|
|
42
|
+
record = tmp_path / "record.json"
|
|
43
|
+
out = "\n".join(json.dumps(e) for e in (STREAM if stream is None else stream))
|
|
44
|
+
script = bindir / "claude"
|
|
45
|
+
script.write_text(textwrap.dedent(f"""\
|
|
46
|
+
#!{sys.executable}
|
|
47
|
+
import json, os, sys, time
|
|
48
|
+
cfg = os.environ.get("CLAUDE_CONFIG_DIR", "")
|
|
49
|
+
cred = os.path.join(cfg, ".credentials.json")
|
|
50
|
+
json.dump({{"argv": sys.argv[1:], "stdin": sys.stdin.read(), "cwd": os.getcwd(),
|
|
51
|
+
"cwd_files": os.listdir("."), "env": dict(os.environ),
|
|
52
|
+
"cred": open(cred).read() if os.path.exists(cred) else None}},
|
|
53
|
+
open({str(record)!r}, "w"))
|
|
54
|
+
time.sleep({sleep})
|
|
55
|
+
sys.stdout.write({out!r})
|
|
56
|
+
sys.stderr.write({stderr!r})
|
|
57
|
+
sys.exit({exit_code})
|
|
58
|
+
"""))
|
|
59
|
+
script.chmod(script.stat().st_mode | stat.S_IXUSR)
|
|
60
|
+
return bindir, record
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@pytest.fixture
|
|
64
|
+
def user_home(tmp_path):
|
|
65
|
+
"""A pretend operator: logged in, with a git identity and a CLAUDE.md."""
|
|
66
|
+
home = tmp_path / "home"
|
|
67
|
+
(home / ".claude").mkdir(parents=True)
|
|
68
|
+
(home / ".claude" / ".credentials.json").write_text('{"claudeAiOauth": {"expiresAt": 100}}')
|
|
69
|
+
(home / ".claude" / "CLAUDE.md").write_text("The user owns Acme.")
|
|
70
|
+
(home / ".gitconfig").write_text("[user]\n name = operator-handle\n")
|
|
71
|
+
return home
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def env_for(bindir, home, **extra):
|
|
75
|
+
return {"PATH": f"{bindir}{os.pathsep}/usr/bin{os.pathsep}/bin", "HOME": str(home), "USER": "operator-handle",
|
|
76
|
+
"GIT_AUTHOR_NAME": "operator-handle", "GIT_DIR": "/somewhere/.git", "ANTHROPIC_API_KEY": "sk-ant-x",
|
|
77
|
+
"CLAUDECODE": "1", "LANG": "C.UTF-8", **extra}
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def test_answer_text_sources_and_zero_cost(tmp_path, user_home):
|
|
81
|
+
bindir, _ = make_fake(tmp_path)
|
|
82
|
+
out = E.claude_cli_answer("Who makes widgets?", env=env_for(bindir, user_home))
|
|
83
|
+
assert out["text"].startswith("Acme makes widgets")
|
|
84
|
+
# Search results, then pages fetched without error, then links in the text; no repeats.
|
|
85
|
+
assert out["urls"] == ["https://s.test/1", "https://acme.test/about", "https://f.test/page", "https://t.test/x."]
|
|
86
|
+
assert out["model"] == "claude-opus-5"
|
|
87
|
+
assert out["cost_usd"] == 0.0 and out["subscription"] is True
|
|
88
|
+
assert out["usage"]["web_search_requests"] == 1 and out["usage"]["web_fetch_requests"] == 2
|
|
89
|
+
assert out["usage"]["api_equivalent_usd"] == 0.0421
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def test_runs_as_a_stranger(tmp_path, user_home):
|
|
93
|
+
bindir, record = make_fake(tmp_path)
|
|
94
|
+
E.claude_cli_answer("-rf what is acme?", env=env_for(bindir, user_home))
|
|
95
|
+
rec = json.loads(record.read_text())
|
|
96
|
+
env = rec["env"]
|
|
97
|
+
# Fresh empty working dir and HOME, outside any repo, and gone afterwards.
|
|
98
|
+
assert rec["cwd_files"] == [] and not os.path.exists(rec["cwd"])
|
|
99
|
+
assert env["HOME"] != str(user_home) and not os.path.exists(env["HOME"])
|
|
100
|
+
assert env["CLAUDE_CONFIG_DIR"].startswith(env["HOME"])
|
|
101
|
+
assert env["GIT_CONFIG_GLOBAL"] == os.devnull and env["GIT_CONFIG_NOSYSTEM"] == "1"
|
|
102
|
+
assert env["GIT_CEILING_DIRECTORIES"] == os.path.dirname(rec["cwd"])
|
|
103
|
+
assert not [k for k in env if k.startswith("GIT_") and k not in
|
|
104
|
+
("GIT_CONFIG_GLOBAL", "GIT_CONFIG_NOSYSTEM", "GIT_CEILING_DIRECTORIES")]
|
|
105
|
+
assert "operator-handle" not in json.dumps(env)
|
|
106
|
+
assert "ANTHROPIC_API_KEY" not in env and "CLAUDECODE" not in env
|
|
107
|
+
# Only the login was copied in; the operator's CLAUDE.md was not.
|
|
108
|
+
assert json.loads(rec["cred"]) == {"claudeAiOauth": {"expiresAt": 100}}
|
|
109
|
+
# The question goes in on stdin, so a leading "-" cannot become a flag.
|
|
110
|
+
assert rec["stdin"] == "-rf what is acme?" and "-rf what is acme?" not in rec["argv"]
|
|
111
|
+
argv = rec["argv"]
|
|
112
|
+
assert argv[:3] == ["-p", "--output-format", "stream-json"]
|
|
113
|
+
assert argv[argv.index("--tools") + 1] == "WebSearch,WebFetch"
|
|
114
|
+
assert argv[argv.index("--allowedTools") + 1] == "WebSearch,WebFetch"
|
|
115
|
+
assert "--strict-mcp-config" in argv and "--model" not in argv
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def test_model_override_is_passed(tmp_path, user_home):
|
|
119
|
+
bindir, record = make_fake(tmp_path)
|
|
120
|
+
E.claude_cli_answer("q", model="sonnet", env=env_for(bindir, user_home))
|
|
121
|
+
argv = json.loads(record.read_text())["argv"]
|
|
122
|
+
assert argv[argv.index("--model") + 1] == "sonnet"
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def test_token_variable_replaces_the_credentials_file(tmp_path):
|
|
126
|
+
bindir, record = make_fake(tmp_path)
|
|
127
|
+
E.claude_cli_answer("q", env=env_for(bindir, tmp_path / "nobody", CLAUDE_CODE_OAUTH_TOKEN="tok"))
|
|
128
|
+
rec = json.loads(record.read_text())
|
|
129
|
+
assert rec["env"]["CLAUDE_CODE_OAUTH_TOKEN"] == "tok" and rec["cred"] is None
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def test_no_login_is_a_clear_error(tmp_path):
|
|
133
|
+
bindir, _ = make_fake(tmp_path)
|
|
134
|
+
with pytest.raises(E.EngineUnavailable, match="setup-token"):
|
|
135
|
+
E.claude_cli_answer("q", env=env_for(bindir, tmp_path / "nobody"))
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def test_refreshed_login_is_written_back(tmp_path, user_home):
|
|
139
|
+
bindir, _ = make_fake(tmp_path)
|
|
140
|
+
script = bindir / "claude"
|
|
141
|
+
# The fake refreshes its login the way the real CLI does when the token is near expiry.
|
|
142
|
+
script.write_text(script.read_text().replace(
|
|
143
|
+
"time.sleep(", 'open(cred, "w").write(\'{"claudeAiOauth": {"expiresAt": 200}}\'); time.sleep('))
|
|
144
|
+
E.claude_cli_answer("q", env=env_for(bindir, user_home))
|
|
145
|
+
assert json.loads((user_home / ".claude" / ".credentials.json").read_text())["claudeAiOauth"]["expiresAt"] == 200
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def test_timeout(tmp_path, user_home):
|
|
149
|
+
bindir, _ = make_fake(tmp_path, sleep=30)
|
|
150
|
+
t0 = time.time()
|
|
151
|
+
with pytest.raises(TimeoutError, match="after 1s"):
|
|
152
|
+
E.claude_cli_answer("q", timeout=1, env=env_for(bindir, user_home))
|
|
153
|
+
assert time.time() - t0 < 10
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def test_cli_error_result_and_bad_exit(tmp_path, user_home):
|
|
157
|
+
bindir, _ = make_fake(tmp_path, stream=[{"type": "result", "subtype": "error_max_turns", "is_error": True}],
|
|
158
|
+
exit_code=1)
|
|
159
|
+
with pytest.raises(RuntimeError, match="error_max_turns"):
|
|
160
|
+
E.claude_cli_answer("q", env=env_for(bindir, user_home))
|
|
161
|
+
bindir, _ = make_fake(tmp_path, stream=[], exit_code=1, stderr="Invalid API key")
|
|
162
|
+
with pytest.raises(RuntimeError, match="Invalid API key"):
|
|
163
|
+
E.claude_cli_answer("q", env=env_for(bindir, user_home))
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def test_missing_cli_is_a_clear_error(tmp_path):
|
|
167
|
+
env = {"PATH": str(tmp_path)}
|
|
168
|
+
with pytest.raises(E.EngineUnavailable, match="no `claude` on PATH"):
|
|
169
|
+
E.available(env, only=["claude-cli"])
|
|
170
|
+
with pytest.raises(E.EngineUnavailable):
|
|
171
|
+
E.claude_cli_answer("q", env=env)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def test_selected_only_by_name(tmp_path):
|
|
175
|
+
bindir, _ = make_fake(tmp_path)
|
|
176
|
+
env = {"PATH": str(bindir), "OPENAI_API_KEY": "sk-1"}
|
|
177
|
+
assert list(E.available(env)) == ["openai"] # never joins a run on its own
|
|
178
|
+
got = E.available(env, only=["claude-cli"], timeout=42)
|
|
179
|
+
assert list(got) == ["claude-cli"] and got["claude-cli"].timeout == 42
|
|
180
|
+
assert E.available(env, only=["claude-cli"])["claude-cli"].timeout is None # the adapter's default
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def test_bound_passes_timeout_only_when_set():
|
|
184
|
+
seen = []
|
|
185
|
+
|
|
186
|
+
def old_style(question, *, key, model): # an adapter written before --timeout existed
|
|
187
|
+
seen.append(model)
|
|
188
|
+
return {}
|
|
189
|
+
|
|
190
|
+
E.Bound("x", "m", "", old_style).ask("q")
|
|
191
|
+
assert seen == ["m"]
|
|
192
|
+
got = {}
|
|
193
|
+
E.Bound("x", "m", "", lambda q, **kw: got.update(kw) or {}, timeout=5).ask("q")
|
|
194
|
+
assert got["timeout"] == 5
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def test_run_records_subscription_flag(tmp_path, user_home, monkeypatch):
|
|
198
|
+
bindir, _ = make_fake(tmp_path)
|
|
199
|
+
for k, v in env_for(bindir, user_home).items():
|
|
200
|
+
monkeypatch.setenv(k, v)
|
|
201
|
+
q = {"ours": ["acme.test"], "brand_terms": ["Acme"], "stale_markers": [], "groups": {"brand": ["What is Acme?"]}}
|
|
202
|
+
eng = E.available(only=["claude-cli"])
|
|
203
|
+
recs = run(q, eng, runs=1, out=tmp_path / "out" / "answers.jsonl", sleep=0)
|
|
204
|
+
assert recs[0]["subscription"] is True and recs[0]["cost_usd"] == 0.0
|
|
205
|
+
assert recs[0]["cited"] and recs[0]["engine"] == "claude-cli"
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def test_cli_reports_a_missing_claude(tmp_path, monkeypatch, capsys):
|
|
209
|
+
qf = tmp_path / "q.yaml"
|
|
210
|
+
qf.write_text("ours: [acme.test]\nbrand_terms: [Acme]\ngroups:\n brand:\n - What is Acme?\n")
|
|
211
|
+
monkeypatch.setenv("PATH", str(tmp_path))
|
|
212
|
+
assert main(["run", str(qf), "--engine", "claude-cli", "--out", str(tmp_path / "o")]) == 2
|
|
213
|
+
assert "no `claude` on PATH" in capsys.readouterr().err
|
|
@@ -1,240 +0,0 @@
|
|
|
1
|
-
"""One adapter per answer engine, over plain HTTPS so no vendor SDK is needed.
|
|
2
|
-
|
|
3
|
-
Every adapter takes a question and returns:
|
|
4
|
-
|
|
5
|
-
{"text": str, "urls": [str], "model": str, "usage": dict, "cost_usd": float | None}
|
|
6
|
-
|
|
7
|
-
`urls` holds every source the engine attached to the answer, in the order it gave
|
|
8
|
-
them, because the position of your first citation is scored. Keys are read from
|
|
9
|
-
the environment only and are never written to a log or a results file.
|
|
10
|
-
"""
|
|
11
|
-
|
|
12
|
-
from __future__ import annotations
|
|
13
|
-
|
|
14
|
-
import os
|
|
15
|
-
import re
|
|
16
|
-
from collections.abc import Callable
|
|
17
|
-
from dataclasses import dataclass
|
|
18
|
-
|
|
19
|
-
import requests
|
|
20
|
-
|
|
21
|
-
TIMEOUT = 180
|
|
22
|
-
SYSTEM_HINT = "Answer as you would for any user. Cite your sources with links."
|
|
23
|
-
|
|
24
|
-
# Public list prices in USD per 1M tokens (input, output), checked 2026-09. Vendors change
|
|
25
|
-
# these, so a run's cost is an estimate unless the API reports it (Perplexity does).
|
|
26
|
-
PRICES: dict[str, tuple[float, float]] = {
|
|
27
|
-
"gpt-5-mini": (0.25, 2.00),
|
|
28
|
-
"gemini-3.5-flash": (0.30, 2.50),
|
|
29
|
-
"claude-opus-5": (5.00, 25.00),
|
|
30
|
-
"claude-sonnet-5": (2.00, 10.00),
|
|
31
|
-
"perplexity/sonar": (1.00, 1.00),
|
|
32
|
-
}
|
|
33
|
-
# Claude's web search tool is billed per search on top of tokens: $10 per 1,000.
|
|
34
|
-
ANTHROPIC_SEARCH_USD = 0.01
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
def token_cost(model: str, inp: int, out: int) -> float | None:
|
|
38
|
-
p = PRICES.get(model)
|
|
39
|
-
if p is None:
|
|
40
|
-
# Dated snapshots such as gpt-5-mini-2025-08-07 price like their base name.
|
|
41
|
-
p = next((v for k, v in PRICES.items() if model.startswith(k + "-")), None)
|
|
42
|
-
return round((inp * p[0] + out * p[1]) / 1e6, 6) if p else None
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
# ----------------------------------------------------------------------------- OpenAI ----
|
|
46
|
-
def openai_answer(question: str, *, key: str, model: str = "gpt-5-mini") -> dict:
|
|
47
|
-
body = {
|
|
48
|
-
"model": model,
|
|
49
|
-
"tools": [{"type": "web_search"}],
|
|
50
|
-
"tool_choice": "auto",
|
|
51
|
-
"input": question,
|
|
52
|
-
"instructions": SYSTEM_HINT,
|
|
53
|
-
}
|
|
54
|
-
r = requests.post("https://api.openai.com/v1/responses",
|
|
55
|
-
headers={"Authorization": f"Bearer {key}"}, json=body, timeout=TIMEOUT)
|
|
56
|
-
r.raise_for_status()
|
|
57
|
-
d = r.json()
|
|
58
|
-
text, urls = [], []
|
|
59
|
-
for item in d.get("output", []) or []:
|
|
60
|
-
if item.get("type") != "message":
|
|
61
|
-
continue
|
|
62
|
-
for c in item.get("content", []) or []:
|
|
63
|
-
if c.get("type") == "output_text":
|
|
64
|
-
text.append(c.get("text", ""))
|
|
65
|
-
urls += [a["url"] for a in c.get("annotations", []) or []
|
|
66
|
-
if a.get("type") == "url_citation" and a.get("url")]
|
|
67
|
-
u = d.get("usage", {}) or {}
|
|
68
|
-
return {
|
|
69
|
-
"text": "\n".join(text),
|
|
70
|
-
"urls": urls,
|
|
71
|
-
"model": d.get("model", model),
|
|
72
|
-
"usage": u,
|
|
73
|
-
"cost_usd": token_cost(d.get("model", model), u.get("input_tokens", 0), u.get("output_tokens", 0)),
|
|
74
|
-
}
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
# ----------------------------------------------------------------------------- Gemini ----
|
|
78
|
-
def gemini_answer(question: str, *, key: str, model: str = "gemini-3.5-flash") -> dict:
|
|
79
|
-
body = {"contents": [{"parts": [{"text": question}]}], "tools": [{"google_search": {}}]}
|
|
80
|
-
# The key goes in a header, not the URL. In the query string it ends up in the text of
|
|
81
|
-
# any HTTPError, and from there in the results file.
|
|
82
|
-
r = requests.post(
|
|
83
|
-
f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent",
|
|
84
|
-
headers={"x-goog-api-key": key}, json=body, timeout=TIMEOUT,
|
|
85
|
-
)
|
|
86
|
-
r.raise_for_status()
|
|
87
|
-
d = r.json()
|
|
88
|
-
cand = (d.get("candidates") or [{}])[0]
|
|
89
|
-
text = "\n".join(p.get("text", "") for p in (cand.get("content") or {}).get("parts", []) or [])
|
|
90
|
-
gm = cand.get("groundingMetadata") or {}
|
|
91
|
-
urls = [(ch.get("web") or {}).get("uri") for ch in gm.get("groundingChunks", []) or []]
|
|
92
|
-
u = d.get("usageMetadata", {}) or {}
|
|
93
|
-
return {
|
|
94
|
-
"text": text,
|
|
95
|
-
# These are vertexaisearch.cloud.google.com redirect URLs; resolve.py turns them
|
|
96
|
-
# into the real sources before scoring.
|
|
97
|
-
"urls": [x for x in urls if x],
|
|
98
|
-
"model": model,
|
|
99
|
-
"usage": u,
|
|
100
|
-
"cost_usd": token_cost(model, u.get("promptTokenCount", 0), u.get("candidatesTokenCount", 0)),
|
|
101
|
-
"grounding_queries": gm.get("webSearchQueries", []),
|
|
102
|
-
}
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
# ----------------------------------------------------------------------------- Claude ----
|
|
106
|
-
def anthropic_answer(question: str, *, key: str, model: str = "claude-sonnet-5") -> dict:
|
|
107
|
-
"""Claude through the Messages API with the server-side web search tool.
|
|
108
|
-
|
|
109
|
-
No model fallback is configured on purpose: an answer from a different model would be
|
|
110
|
-
scored as this one. A refusal is recorded as an error instead.
|
|
111
|
-
"""
|
|
112
|
-
messages: list[dict] = [{"role": "user", "content": question}]
|
|
113
|
-
headers = {"x-api-key": key, "anthropic-version": "2023-06-01"}
|
|
114
|
-
text: list[str] = []
|
|
115
|
-
urls: list[str] = []
|
|
116
|
-
usage = {"input_tokens": 0, "output_tokens": 0, "web_search_requests": 0}
|
|
117
|
-
served = model
|
|
118
|
-
# A long server-tool turn can stop with `pause_turn`. Sending the partial turn back
|
|
119
|
-
# lets it continue. Three rounds is plenty for a single question.
|
|
120
|
-
for _ in range(3):
|
|
121
|
-
body = {
|
|
122
|
-
"model": model,
|
|
123
|
-
"max_tokens": 16000,
|
|
124
|
-
"system": SYSTEM_HINT,
|
|
125
|
-
"messages": messages,
|
|
126
|
-
"tools": [{"type": "web_search_20260209", "name": "web_search", "max_uses": 5}],
|
|
127
|
-
}
|
|
128
|
-
r = requests.post("https://api.anthropic.com/v1/messages", headers=headers, json=body,
|
|
129
|
-
timeout=TIMEOUT)
|
|
130
|
-
r.raise_for_status()
|
|
131
|
-
d = r.json()
|
|
132
|
-
served = d.get("model", served)
|
|
133
|
-
u = d.get("usage", {}) or {}
|
|
134
|
-
usage["input_tokens"] += u.get("input_tokens", 0)
|
|
135
|
-
usage["output_tokens"] += u.get("output_tokens", 0)
|
|
136
|
-
usage["web_search_requests"] += (u.get("server_tool_use") or {}).get("web_search_requests", 0)
|
|
137
|
-
for b in d.get("content", []) or []:
|
|
138
|
-
if b.get("type") == "text":
|
|
139
|
-
text.append(b.get("text", ""))
|
|
140
|
-
urls += [c["url"] for c in b.get("citations", []) or [] if c.get("url")]
|
|
141
|
-
elif b.get("type") == "web_search_tool_result" and isinstance(b.get("content"), list):
|
|
142
|
-
urls += [x["url"] for x in b["content"] if isinstance(x, dict) and x.get("url")]
|
|
143
|
-
stop = d.get("stop_reason")
|
|
144
|
-
if stop == "refusal":
|
|
145
|
-
raise RuntimeError(f"refusal: {(d.get('stop_details') or {}).get('category')}")
|
|
146
|
-
if stop != "pause_turn":
|
|
147
|
-
break
|
|
148
|
-
messages = [messages[0], {"role": "assistant", "content": d.get("content", [])}]
|
|
149
|
-
cost = token_cost(served, usage["input_tokens"], usage["output_tokens"])
|
|
150
|
-
if cost is not None:
|
|
151
|
-
cost = round(cost + usage["web_search_requests"] * ANTHROPIC_SEARCH_USD, 6)
|
|
152
|
-
return {"text": "".join(text), "urls": urls, "model": served, "usage": usage, "cost_usd": cost}
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
# ------------------------------------------------------------------------- Perplexity ----
|
|
156
|
-
def perplexity_answer(question: str, *, key: str, model: str = "perplexity/sonar") -> dict:
|
|
157
|
-
"""Perplexity's Responses API. Web search is OFF unless the tool is passed.
|
|
158
|
-
|
|
159
|
-
Without `tools=[{"type": "web_search"}]` the call still succeeds and returns a fluent
|
|
160
|
-
answer naming real companies, with zero citations. For a benchmark scored on
|
|
161
|
-
citations that is the worst failure there is: no error, and no data.
|
|
162
|
-
"""
|
|
163
|
-
body = {"model": model, "input": question, "tools": [{"type": "web_search"}]}
|
|
164
|
-
r = requests.post("https://api.perplexity.ai/v1/responses",
|
|
165
|
-
headers={"Authorization": f"Bearer {key}"}, json=body, timeout=TIMEOUT)
|
|
166
|
-
r.raise_for_status()
|
|
167
|
-
d = r.json()
|
|
168
|
-
text, urls = "", []
|
|
169
|
-
for item in d.get("output", []) or []:
|
|
170
|
-
if item.get("type") == "search_results":
|
|
171
|
-
urls += [x.get("url") for x in item.get("results") or [] if x.get("url")]
|
|
172
|
-
for c in item.get("content", []) or []:
|
|
173
|
-
text += c.get("text", "")
|
|
174
|
-
urls += [a.get("url") for a in c.get("annotations") or [] if a.get("url")]
|
|
175
|
-
u = d.get("usage", {}) or {}
|
|
176
|
-
reported = (u.get("cost") or {}).get("total_cost")
|
|
177
|
-
return {
|
|
178
|
-
"text": text,
|
|
179
|
-
"urls": list(dict.fromkeys(urls)),
|
|
180
|
-
"model": model,
|
|
181
|
-
"usage": u,
|
|
182
|
-
"cost_usd": round(float(reported), 6) if reported is not None
|
|
183
|
-
else token_cost(model, u.get("input_tokens", 0), u.get("output_tokens", 0)),
|
|
184
|
-
}
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
@dataclass(frozen=True)
|
|
188
|
-
class Engine:
|
|
189
|
-
name: str
|
|
190
|
-
env_key: str
|
|
191
|
-
fn: Callable[..., dict]
|
|
192
|
-
default_model: str
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
ENGINES: dict[str, Engine] = {
|
|
196
|
-
"openai": Engine("openai", "OPENAI_API_KEY", openai_answer, "gpt-5-mini"),
|
|
197
|
-
"gemini": Engine("gemini", "GEMINI_API_KEY", gemini_answer, "gemini-3.5-flash"),
|
|
198
|
-
"claude": Engine("claude", "ANTHROPIC_API_KEY", anthropic_answer, "claude-sonnet-5"),
|
|
199
|
-
"perplexity": Engine("perplexity", "PERPLEXITY_API_KEY", perplexity_answer, "perplexity/sonar"),
|
|
200
|
-
}
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
@dataclass(frozen=True)
|
|
204
|
-
class Bound:
|
|
205
|
-
"""An engine with its key and model resolved. `key` is never printed or stored."""
|
|
206
|
-
|
|
207
|
-
name: str
|
|
208
|
-
model: str
|
|
209
|
-
key: str
|
|
210
|
-
fn: Callable[..., dict]
|
|
211
|
-
|
|
212
|
-
def ask(self, question: str) -> dict:
|
|
213
|
-
return self.fn(question, key=self.key, model=self.model)
|
|
214
|
-
|
|
215
|
-
def __repr__(self) -> str: # keep the key out of tracebacks and debug output
|
|
216
|
-
return f"Bound(name={self.name!r}, model={self.model!r})"
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
def available(env: dict | None = None, *, only: list[str] | None = None,
|
|
220
|
-
models: dict[str, str] | None = None) -> dict[str, Bound]:
|
|
221
|
-
"""Engines whose key is set. `models` overrides a default model per engine."""
|
|
222
|
-
env = os.environ if env is None else env
|
|
223
|
-
models = models or {}
|
|
224
|
-
out = {}
|
|
225
|
-
for name, e in ENGINES.items():
|
|
226
|
-
if only and name not in only:
|
|
227
|
-
continue
|
|
228
|
-
key = env.get(e.env_key, "")
|
|
229
|
-
if key:
|
|
230
|
-
out[name] = Bound(name, models.get(name, e.default_model), key, e.fn)
|
|
231
|
-
return out
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
def redact(message: str, secrets: list[str]) -> str:
|
|
235
|
-
"""Remove every key from an error message before it is logged."""
|
|
236
|
-
for s in secrets:
|
|
237
|
-
if s and len(s) >= 6:
|
|
238
|
-
message = message.replace(s, "[redacted]")
|
|
239
|
-
# Belt and braces for key-shaped strings that were not in the list.
|
|
240
|
-
return re.sub(r"(key=)[^&\s]+", r"\1[redacted]", message)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/tests/test_run_and_report.py
RENAMED
|
File without changes
|
|
File without changes
|