answer-engine-benchmark 0.1.1__tar.gz → 0.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (25) hide show
  1. {answer_engine_benchmark-0.1.1/src/answer_engine_benchmark.egg-info → answer_engine_benchmark-0.1.2}/PKG-INFO +34 -2
  2. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/README.md +33 -1
  3. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/pyproject.toml +1 -1
  4. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark/__init__.py +1 -1
  5. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark/cli.py +7 -4
  6. answer_engine_benchmark-0.1.2/src/answer_engine_benchmark/engines.py +489 -0
  7. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark/runner.py +2 -0
  8. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark/summary.py +3 -0
  9. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2/src/answer_engine_benchmark.egg-info}/PKG-INFO +34 -2
  10. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark.egg-info/SOURCES.txt +1 -0
  11. answer_engine_benchmark-0.1.2/tests/test_claude_cli.py +213 -0
  12. answer_engine_benchmark-0.1.1/src/answer_engine_benchmark/engines.py +0 -240
  13. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/LICENSE +0 -0
  14. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/setup.cfg +0 -0
  15. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark/__main__.py +0 -0
  16. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark/questions.py +0 -0
  17. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark/scoring.py +0 -0
  18. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark.egg-info/dependency_links.txt +0 -0
  19. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark.egg-info/entry_points.txt +0 -0
  20. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark.egg-info/requires.txt +0 -0
  21. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/src/answer_engine_benchmark.egg-info/top_level.txt +0 -0
  22. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/tests/test_engines.py +0 -0
  23. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/tests/test_questions.py +0 -0
  24. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/tests/test_run_and_report.py +0 -0
  25. {answer_engine_benchmark-0.1.1 → answer_engine_benchmark-0.1.2}/tests/test_scoring.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: answer-engine-benchmark
3
- Version: 0.1.1
3
+ Version: 0.1.2
4
4
  Summary: Ask ChatGPT, Gemini, Perplexity and Claude your buyers' questions, several times each, and count who they cite.
5
5
  Author: Synapse Research Ltd
6
6
  License-Expression: MIT
@@ -99,11 +99,31 @@ Gemini answer would look like it cited Google.
99
99
  | `gemini` | `GEMINI_API_KEY` | `gemini-3.5-flash` | Google Search grounding |
100
100
  | `perplexity` | `PERPLEXITY_API_KEY` | `perplexity/sonar` | Responses API `web_search` tool |
101
101
  | `claude` | `ANTHROPIC_API_KEY` | `claude-sonnet-5` | Messages API web search tool |
102
+ | `claude-cli` | none, uses your `claude` login | whatever the CLI picks | Claude Code's `WebSearch` and `WebFetch` tools |
102
103
 
103
104
  Keys are read from the environment only. They are sent in request headers and
104
105
  removed from any error message before it is written, so they never reach the
105
106
  results file. An engine with no key is skipped.
106
107
 
108
+ `claude-cli` is for people who pay for a Claude subscription and have no API
109
+ key. It runs the [Claude Code](https://docs.claude.com/en/docs/claude-code) CLI
110
+ (`claude -p`) once per call, with only the web search and fetch tools switched
111
+ on. It never joins a run on its own: name it with `--engine claude-cli`. If
112
+ `claude` is not on your PATH, the run stops before the first call and says so.
113
+ Log in once by running `claude`. On macOS, where the login sits in the Keychain,
114
+ run `claude setup-token` and export `CLAUDE_CODE_OAUTH_TOKEN` instead. Its rows
115
+ record a cost of 0 and `"subscription": true`. The calls use up your plan's
116
+ usage limits and never show on an API bill. A call gets 600 seconds by
117
+ default. Change that with `--timeout`, which works for every engine (the API
118
+ engines default to 180). `--model claude-cli=sonnet` is passed on to the CLI.
119
+ Its sources are the search results and fetched pages from the tool calls, plus
120
+ any link in the answer text.
121
+
122
+ The CLI's answers are not the API's answers. Claude Code adds its own system
123
+ prompt. The model and search limits come from your account and plan. Two people
124
+ can get different results from the same question file. Say which one you
125
+ used when you publish numbers.
126
+
107
127
  Change a model with `--model claude=claude-opus-5`. Pick the models your
108
128
  buyers actually use in the apps, or say in the report which ones you used.
109
129
 
@@ -148,6 +168,17 @@ API. The CLI passed the operator's own git identity to the model, and many brand
148
168
  answers then told the reader the company was probably their own. The API
149
169
  adapters here send only the question and a one-line instruction to cite sources.
150
170
 
171
+ The `claude-cli` engine guards against the same leak. Each call runs in a new,
172
+ empty temporary folder with no git repository above it. HOME and the config
173
+ folder are temporary too, and git is told to read no config at all. Only an
174
+ allow-list of environment variables gets through (PATH, locale, proxy and CA
175
+ settings), so `GIT_*`, `USER` and any `CLAUDE*` or `ANTHROPIC*` variables are
176
+ dropped. Your `CLAUDE.md` files, memory, settings, hooks and MCP servers are
177
+ never loaded. Only your login is copied in, and if the CLI refreshes it during
178
+ the call, the new one is written back. What cannot be hidden is the account:
179
+ the answers still come from your Claude login, on your plan, so results depend
180
+ on the local account.
181
+
151
182
  ## Example
152
183
 
153
184
  `examples/synapse-launch-week-2026-09.md` is a real run on one company's own
@@ -173,7 +204,8 @@ pip install -e ".[test]"
173
204
  pytest
174
205
  ```
175
206
 
176
- 33 tests. They mock every API, so they spend nothing and need no keys.
207
+ 46 tests. They mock every API and stand in a fake `claude` for the CLI, so they
208
+ spend nothing and need no keys or login.
177
209
 
178
210
  ## Licence
179
211
 
@@ -76,11 +76,31 @@ Gemini answer would look like it cited Google.
76
76
  | `gemini` | `GEMINI_API_KEY` | `gemini-3.5-flash` | Google Search grounding |
77
77
  | `perplexity` | `PERPLEXITY_API_KEY` | `perplexity/sonar` | Responses API `web_search` tool |
78
78
  | `claude` | `ANTHROPIC_API_KEY` | `claude-sonnet-5` | Messages API web search tool |
79
+ | `claude-cli` | none, uses your `claude` login | whatever the CLI picks | Claude Code's `WebSearch` and `WebFetch` tools |
79
80
 
80
81
  Keys are read from the environment only. They are sent in request headers and
81
82
  removed from any error message before it is written, so they never reach the
82
83
  results file. An engine with no key is skipped.
83
84
 
85
+ `claude-cli` is for people who pay for a Claude subscription and have no API
86
+ key. It runs the [Claude Code](https://docs.claude.com/en/docs/claude-code) CLI
87
+ (`claude -p`) once per call, with only the web search and fetch tools switched
88
+ on. It never joins a run on its own: name it with `--engine claude-cli`. If
89
+ `claude` is not on your PATH, the run stops before the first call and says so.
90
+ Log in once by running `claude`. On macOS, where the login sits in the Keychain,
91
+ run `claude setup-token` and export `CLAUDE_CODE_OAUTH_TOKEN` instead. Its rows
92
+ record a cost of 0 and `"subscription": true`. The calls use up your plan's
93
+ usage limits and never show on an API bill. A call gets 600 seconds by
94
+ default. Change that with `--timeout`, which works for every engine (the API
95
+ engines default to 180). `--model claude-cli=sonnet` is passed on to the CLI.
96
+ Its sources are the search results and fetched pages from the tool calls, plus
97
+ any link in the answer text.
98
+
99
+ The CLI's answers are not the API's answers. Claude Code adds its own system
100
+ prompt. The model and search limits come from your account and plan. Two people
101
+ can get different results from the same question file. Say which one you
102
+ used when you publish numbers.
103
+
84
104
  Change a model with `--model claude=claude-opus-5`. Pick the models your
85
105
  buyers actually use in the apps, or say in the report which ones you used.
86
106
 
@@ -125,6 +145,17 @@ API. The CLI passed the operator's own git identity to the model, and many brand
125
145
  answers then told the reader the company was probably their own. The API
126
146
  adapters here send only the question and a one-line instruction to cite sources.
127
147
 
148
+ The `claude-cli` engine guards against the same leak. Each call runs in a new,
149
+ empty temporary folder with no git repository above it. HOME and the config
150
+ folder are temporary too, and git is told to read no config at all. Only an
151
+ allow-list of environment variables gets through (PATH, locale, proxy and CA
152
+ settings), so `GIT_*`, `USER` and any `CLAUDE*` or `ANTHROPIC*` variables are
153
+ dropped. Your `CLAUDE.md` files, memory, settings, hooks and MCP servers are
154
+ never loaded. Only your login is copied in, and if the CLI refreshes it during
155
+ the call, the new one is written back. What cannot be hidden is the account:
156
+ the answers still come from your Claude login, on your plan, so results depend
157
+ on the local account.
158
+
128
159
  ## Example
129
160
 
130
161
  `examples/synapse-launch-week-2026-09.md` is a real run on one company's own
@@ -150,7 +181,8 @@ pip install -e ".[test]"
150
181
  pytest
151
182
  ```
152
183
 
153
- 33 tests. They mock every API, so they spend nothing and need no keys.
184
+ 46 tests. They mock every API and stand in a fake `claude` for the CLI, so they
185
+ spend nothing and need no keys or login.
154
186
 
155
187
  ## Licence
156
188
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "answer-engine-benchmark"
7
- version = "0.1.1"
7
+ version = "0.1.2"
8
8
  description = "Ask ChatGPT, Gemini, Perplexity and Claude your buyers' questions, several times each, and count who they cite."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -1,3 +1,3 @@
1
1
  """answer-engine-benchmark: ask AI answer engines your buyers' questions and count who they cite."""
2
2
 
3
- __version__ = "0.1.1"
3
+ __version__ = "0.1.2"
@@ -35,6 +35,8 @@ def _engine_flags(p: argparse.ArgumentParser) -> None:
35
35
  p.add_argument("--engine", action="append", choices=sorted(ENGINES), help="only these engines (repeatable)")
36
36
  p.add_argument("--model", action="append", type=_kv, default=[], metavar="ENGINE=MODEL",
37
37
  help="override a model, e.g. --model claude=claude-opus-5")
38
+ p.add_argument("--timeout", type=float, metavar="SECONDS",
39
+ help="give up on one call after this long (default 180, claude-cli 600)")
38
40
  p.add_argument("--runs", type=int, default=3, help="times to ask each question on each engine (default 3)")
39
41
  p.add_argument("--max-calls", type=int, default=500,
40
42
  help="refuse to start a run that needs more API calls than this (default 500)")
@@ -111,19 +113,20 @@ def main(argv: list[str] | None = None) -> int:
111
113
  return 0
112
114
 
113
115
  q = load(args.questions, dict(args.set), allow_unfilled=getattr(args, "allow_unfilled", False))
114
- engines = available(only=args.engine, models=dict(args.model))
116
+ engines = available(only=args.engine, models=dict(args.model), timeout=args.timeout)
115
117
  if args.cmd == "check":
116
118
  print(f"ok: {count(q)} questions in {len(q['groups'])} groups; ours = {q['ours']}")
117
119
  missing = [f"{n} ({e.env_key})" for n, e in ENGINES.items()
118
- if n not in engines and (not args.engine or n in args.engine)]
120
+ if e.env_key and n not in engines and (not args.engine or n in args.engine)]
119
121
  if missing:
120
122
  print("no key set for: " + ", ".join(missing))
121
123
  _plan(q, engines, args.runs, args.max_calls)
122
124
  return 0
123
125
 
124
126
  if not engines:
125
- keys = ", ".join(e.env_key for e in ENGINES.values())
126
- print(f"aeb: no engine has a key. Set one or more of: {keys}", file=sys.stderr)
127
+ keys = ", ".join(e.env_key for e in ENGINES.values() if e.env_key)
128
+ print(f"aeb: no engine has a key. Set one or more of: {keys}, "
129
+ "or use --engine claude-cli", file=sys.stderr)
127
130
  return 2
128
131
 
129
132
  if args.cmd == "run":
@@ -0,0 +1,489 @@
1
+ """One adapter per answer engine, over plain HTTPS so no vendor SDK is needed.
2
+
3
+ Every adapter takes a question and returns:
4
+
5
+ {"text": str, "urls": [str], "model": str, "usage": dict, "cost_usd": float | None}
6
+
7
+ `urls` holds every source the engine attached to the answer, in the order it gave
8
+ them, because the position of your first citation is scored. Keys are read from
9
+ the environment only and are never written to a log or a results file.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import json
15
+ import os
16
+ import re
17
+ import shutil
18
+ import signal
19
+ import subprocess
20
+ import tempfile
21
+ from collections.abc import Callable
22
+ from dataclasses import dataclass
23
+ from pathlib import Path
24
+
25
+ import requests
26
+
27
+ from .scoring import URL_RX
28
+
29
+ TIMEOUT = 180
30
+ SYSTEM_HINT = "Answer as you would for any user. Cite your sources with links."
31
+
32
+ # Public list prices in USD per 1M tokens (input, output), checked 2026-09. Vendors change
33
+ # these, so a run's cost is an estimate unless the API reports it (Perplexity does).
34
+ PRICES: dict[str, tuple[float, float]] = {
35
+ "gpt-5-mini": (0.25, 2.00),
36
+ "gemini-3.5-flash": (0.30, 2.50),
37
+ "claude-opus-5": (5.00, 25.00),
38
+ "claude-sonnet-5": (2.00, 10.00),
39
+ "perplexity/sonar": (1.00, 1.00),
40
+ }
41
+ # Claude's web search tool is billed per search on top of tokens: $10 per 1,000.
42
+ ANTHROPIC_SEARCH_USD = 0.01
43
+
44
+
45
+ def token_cost(model: str, inp: int, out: int) -> float | None:
46
+ p = PRICES.get(model)
47
+ if p is None:
48
+ # Dated snapshots such as gpt-5-mini-2025-08-07 price like their base name.
49
+ p = next((v for k, v in PRICES.items() if model.startswith(k + "-")), None)
50
+ return round((inp * p[0] + out * p[1]) / 1e6, 6) if p else None
51
+
52
+
53
+ # ----------------------------------------------------------------------------- OpenAI ----
54
+ def openai_answer(question: str, *, key: str, model: str = "gpt-5-mini", timeout: float = TIMEOUT) -> dict:
55
+ body = {
56
+ "model": model,
57
+ "tools": [{"type": "web_search"}],
58
+ "tool_choice": "auto",
59
+ "input": question,
60
+ "instructions": SYSTEM_HINT,
61
+ }
62
+ r = requests.post("https://api.openai.com/v1/responses",
63
+ headers={"Authorization": f"Bearer {key}"}, json=body, timeout=timeout)
64
+ r.raise_for_status()
65
+ d = r.json()
66
+ text, urls = [], []
67
+ for item in d.get("output", []) or []:
68
+ if item.get("type") != "message":
69
+ continue
70
+ for c in item.get("content", []) or []:
71
+ if c.get("type") == "output_text":
72
+ text.append(c.get("text", ""))
73
+ urls += [a["url"] for a in c.get("annotations", []) or []
74
+ if a.get("type") == "url_citation" and a.get("url")]
75
+ u = d.get("usage", {}) or {}
76
+ return {
77
+ "text": "\n".join(text),
78
+ "urls": urls,
79
+ "model": d.get("model", model),
80
+ "usage": u,
81
+ "cost_usd": token_cost(d.get("model", model), u.get("input_tokens", 0), u.get("output_tokens", 0)),
82
+ }
83
+
84
+
85
+ # ----------------------------------------------------------------------------- Gemini ----
86
+ def gemini_answer(question: str, *, key: str, model: str = "gemini-3.5-flash", timeout: float = TIMEOUT) -> dict:
87
+ body = {"contents": [{"parts": [{"text": question}]}], "tools": [{"google_search": {}}]}
88
+ # The key goes in a header, not the URL. In the query string it ends up in the text of
89
+ # any HTTPError, and from there in the results file.
90
+ r = requests.post(
91
+ f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent",
92
+ headers={"x-goog-api-key": key}, json=body, timeout=timeout,
93
+ )
94
+ r.raise_for_status()
95
+ d = r.json()
96
+ cand = (d.get("candidates") or [{}])[0]
97
+ text = "\n".join(p.get("text", "") for p in (cand.get("content") or {}).get("parts", []) or [])
98
+ gm = cand.get("groundingMetadata") or {}
99
+ urls = [(ch.get("web") or {}).get("uri") for ch in gm.get("groundingChunks", []) or []]
100
+ u = d.get("usageMetadata", {}) or {}
101
+ return {
102
+ "text": text,
103
+ # These are vertexaisearch.cloud.google.com redirect URLs; resolve.py turns them
104
+ # into the real sources before scoring.
105
+ "urls": [x for x in urls if x],
106
+ "model": model,
107
+ "usage": u,
108
+ "cost_usd": token_cost(model, u.get("promptTokenCount", 0), u.get("candidatesTokenCount", 0)),
109
+ "grounding_queries": gm.get("webSearchQueries", []),
110
+ }
111
+
112
+
113
+ # ----------------------------------------------------------------------------- Claude ----
114
+ def anthropic_answer(question: str, *, key: str, model: str = "claude-sonnet-5", timeout: float = TIMEOUT) -> dict:
115
+ """Claude through the Messages API with the server-side web search tool.
116
+
117
+ No model fallback is configured on purpose: an answer from a different model would be
118
+ scored as this one. A refusal is recorded as an error instead.
119
+ """
120
+ messages: list[dict] = [{"role": "user", "content": question}]
121
+ headers = {"x-api-key": key, "anthropic-version": "2023-06-01"}
122
+ text: list[str] = []
123
+ urls: list[str] = []
124
+ usage = {"input_tokens": 0, "output_tokens": 0, "web_search_requests": 0}
125
+ served = model
126
+ # A long server-tool turn can stop with `pause_turn`. Sending the partial turn back
127
+ # lets it continue. Three rounds is plenty for a single question.
128
+ for _ in range(3):
129
+ body = {
130
+ "model": model,
131
+ "max_tokens": 16000,
132
+ "system": SYSTEM_HINT,
133
+ "messages": messages,
134
+ "tools": [{"type": "web_search_20260209", "name": "web_search", "max_uses": 5}],
135
+ }
136
+ r = requests.post("https://api.anthropic.com/v1/messages", headers=headers, json=body,
137
+ timeout=timeout)
138
+ r.raise_for_status()
139
+ d = r.json()
140
+ served = d.get("model", served)
141
+ u = d.get("usage", {}) or {}
142
+ usage["input_tokens"] += u.get("input_tokens", 0)
143
+ usage["output_tokens"] += u.get("output_tokens", 0)
144
+ usage["web_search_requests"] += (u.get("server_tool_use") or {}).get("web_search_requests", 0)
145
+ for b in d.get("content", []) or []:
146
+ if b.get("type") == "text":
147
+ text.append(b.get("text", ""))
148
+ urls += [c["url"] for c in b.get("citations", []) or [] if c.get("url")]
149
+ elif b.get("type") == "web_search_tool_result" and isinstance(b.get("content"), list):
150
+ urls += [x["url"] for x in b["content"] if isinstance(x, dict) and x.get("url")]
151
+ stop = d.get("stop_reason")
152
+ if stop == "refusal":
153
+ raise RuntimeError(f"refusal: {(d.get('stop_details') or {}).get('category')}")
154
+ if stop != "pause_turn":
155
+ break
156
+ messages = [messages[0], {"role": "assistant", "content": d.get("content", [])}]
157
+ cost = token_cost(served, usage["input_tokens"], usage["output_tokens"])
158
+ if cost is not None:
159
+ cost = round(cost + usage["web_search_requests"] * ANTHROPIC_SEARCH_USD, 6)
160
+ return {"text": "".join(text), "urls": urls, "model": served, "usage": usage, "cost_usd": cost}
161
+
162
+
163
+ # ------------------------------------------------------------------------- Perplexity ----
164
+ def perplexity_answer(question: str, *, key: str, model: str = "perplexity/sonar", timeout: float = TIMEOUT) -> dict:
165
+ """Perplexity's Responses API. Web search is OFF unless the tool is passed.
166
+
167
+ Without `tools=[{"type": "web_search"}]` the call still succeeds and returns a fluent
168
+ answer naming real companies, with zero citations. For a benchmark scored on
169
+ citations that is the worst failure there is: no error, and no data.
170
+ """
171
+ body = {"model": model, "input": question, "tools": [{"type": "web_search"}]}
172
+ r = requests.post("https://api.perplexity.ai/v1/responses",
173
+ headers={"Authorization": f"Bearer {key}"}, json=body, timeout=timeout)
174
+ r.raise_for_status()
175
+ d = r.json()
176
+ text, urls = "", []
177
+ for item in d.get("output", []) or []:
178
+ if item.get("type") == "search_results":
179
+ urls += [x.get("url") for x in item.get("results") or [] if x.get("url")]
180
+ for c in item.get("content", []) or []:
181
+ text += c.get("text", "")
182
+ urls += [a.get("url") for a in c.get("annotations") or [] if a.get("url")]
183
+ u = d.get("usage", {}) or {}
184
+ reported = (u.get("cost") or {}).get("total_cost")
185
+ return {
186
+ "text": text,
187
+ "urls": list(dict.fromkeys(urls)),
188
+ "model": model,
189
+ "usage": u,
190
+ "cost_usd": round(float(reported), 6) if reported is not None
191
+ else token_cost(model, u.get("input_tokens", 0), u.get("output_tokens", 0)),
192
+ }
193
+
194
+
195
+ # ------------------------------------------------------------------------- Claude CLI ----
196
+ # Claude through the locally logged-in Claude Code CLI (`claude -p`), for people with a Claude
197
+ # subscription and no API key. The CLI is a coding agent that normally reads a lot about the
198
+ # person running it: git identity, CLAUDE.md files, memory, settings, hooks, MCP servers. In an
199
+ # earlier run it passed the operator's git username to the model, and brand answers then told
200
+ # the reader the company was probably their own. So every call runs as a stranger: HOME, the
201
+ # config dir and the working directory are fresh empty temp dirs, git reads no config, and the
202
+ # environment is an allow-list. Only the login credentials are copied in.
203
+ CLAUDE_CLI_TIMEOUT = 600
204
+ # Everything else is dropped: USER, EMAIL, every GIT_* variable, the calling session's CLAUDE*
205
+ # and ANTHROPIC* variables, SSH agents and so on.
206
+ _CLI_ENV_KEEP = ("PATH", "LANG", "LANGUAGE", "LC_ALL", "TZ", "TMPDIR",
207
+ "HTTP_PROXY", "HTTPS_PROXY", "NO_PROXY", "http_proxy", "https_proxy", "no_proxy",
208
+ "SSL_CERT_FILE", "SSL_CERT_DIR", "NODE_EXTRA_CA_CERTS")
209
+ # A long-lived token from `claude setup-token`. When set, no credentials file is copied.
210
+ _CLI_TOKEN_VAR = "CLAUDE_CODE_OAUTH_TOKEN"
211
+ _CLI_TOOLS = "WebSearch,WebFetch"
212
+
213
+
214
+ class EngineUnavailable(ValueError):
215
+ """An engine was asked for but cannot run on this machine."""
216
+
217
+
218
+ def claude_exe(env: dict | None = None) -> str:
219
+ env = os.environ if env is None else env
220
+ exe = shutil.which("claude", path=env.get("PATH"))
221
+ if not exe:
222
+ raise EngineUnavailable("claude-cli needs the Claude Code CLI: no `claude` on PATH. "
223
+ "Install it and run `claude` once to log in.")
224
+ return exe
225
+
226
+
227
+ def _cli_config_dir(env: dict) -> Path:
228
+ return Path(env.get("CLAUDE_CONFIG_DIR") or Path(env.get("HOME") or Path.home()) / ".claude")
229
+
230
+
231
+ def cli_env(home: str, src: dict | None = None) -> dict:
232
+ """The environment of someone who has never met the operator."""
233
+ src = os.environ if src is None else src
234
+ env = {k: src[k] for k in (*_CLI_ENV_KEEP, _CLI_TOKEN_VAR) if src.get(k)}
235
+ env.update({
236
+ "HOME": home,
237
+ "CLAUDE_CONFIG_DIR": os.path.join(home, ".claude"),
238
+ "XDG_CONFIG_HOME": os.path.join(home, ".config"),
239
+ "XDG_DATA_HOME": os.path.join(home, ".local", "share"),
240
+ "XDG_CACHE_HOME": os.path.join(home, ".cache"),
241
+ "XDG_STATE_HOME": os.path.join(home, ".local", "state"),
242
+ "GIT_CONFIG_GLOBAL": os.devnull,
243
+ "GIT_CONFIG_NOSYSTEM": "1",
244
+ "DISABLE_AUTOUPDATER": "1",
245
+ "USER": "user",
246
+ "LOGNAME": "user",
247
+ })
248
+ return env
249
+
250
+
251
+ def _stage_auth(cfg_dir: str, src: dict) -> Path | None:
252
+ """Copy only the login credentials into the empty config dir. Returns the original's path,
253
+ or None when the token variable carries the login instead."""
254
+ os.makedirs(cfg_dir, mode=0o700, exist_ok=True)
255
+ if src.get(_CLI_TOKEN_VAR):
256
+ return None
257
+ real = _cli_config_dir(src) / ".credentials.json"
258
+ if not real.is_file():
259
+ # macOS keeps the login in the Keychain, where it cannot be copied.
260
+ raise EngineUnavailable(f"claude-cli: no login found at {real}. Run `claude setup-token` and "
261
+ f"export {_CLI_TOKEN_VAR}, or run `claude` once to log in.")
262
+ fd = os.open(os.path.join(cfg_dir, ".credentials.json"), os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
263
+ with os.fdopen(fd, "wb") as f:
264
+ f.write(real.read_bytes())
265
+ return real
266
+
267
+
268
+ def _expires_at(raw: bytes) -> int:
269
+ try:
270
+ return int(json.loads(raw)["claudeAiOauth"]["expiresAt"])
271
+ except (ValueError, KeyError, TypeError):
272
+ return 0
273
+
274
+
275
+ def _write_back_auth(real: Path, staged: str) -> None:
276
+ """If the CLI refreshed its login inside the temp dir, hand the new one back. The refresh
277
+ token rotates, so keeping the old one could log the operator's own sessions out."""
278
+ try:
279
+ new, old = Path(staged).read_bytes(), real.read_bytes()
280
+ except OSError:
281
+ return
282
+ if new == old or _expires_at(new) <= _expires_at(old):
283
+ return
284
+ tmp = real.with_name(real.name + ".aeb.tmp")
285
+ fd = os.open(tmp, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
286
+ with os.fdopen(fd, "wb") as f:
287
+ f.write(new)
288
+ os.replace(tmp, real)
289
+
290
+
291
+ def _run_cli(cmd: list[str], prompt: str, *, cwd: str, env: dict, timeout: float) -> subprocess.CompletedProcess:
292
+ # The prompt goes in on stdin, so a question that starts with "-" is never read as a flag.
293
+ # Own process group, so a timeout kills the CLI and anything it started.
294
+ p = subprocess.Popen(cmd, cwd=cwd, env=env, stdin=subprocess.PIPE, stdout=subprocess.PIPE,
295
+ stderr=subprocess.PIPE, text=True, start_new_session=True)
296
+ try:
297
+ out, err = p.communicate(prompt, timeout=timeout)
298
+ except subprocess.TimeoutExpired:
299
+ try:
300
+ os.killpg(p.pid, signal.SIGKILL)
301
+ except OSError:
302
+ p.kill()
303
+ p.communicate()
304
+ raise TimeoutError(f"claude-cli: no answer after {timeout:g}s") from None
305
+ return subprocess.CompletedProcess(cmd, p.returncode, out, err)
306
+
307
+
308
+ def _tool_text(content) -> str:
309
+ if isinstance(content, str):
310
+ return content
311
+ if isinstance(content, list):
312
+ return "".join(c.get("text", "") for c in content if isinstance(c, dict))
313
+ return ""
314
+
315
+
316
+ def _search_links(text: str) -> list[str]:
317
+ """WebSearch results arrive as text with a `Links: [{"title": ..., "url": ...}]` list."""
318
+ i = text.find("Links: [")
319
+ if i < 0:
320
+ return []
321
+ try:
322
+ links, _ = json.JSONDecoder().raw_decode(text, i + len("Links: "))
323
+ except ValueError:
324
+ return []
325
+ return [x["url"] for x in links if isinstance(x, dict) and isinstance(x.get("url"), str)]
326
+
327
+
328
+ def parse_cli_stream(stdout: str) -> dict:
329
+ """Read `--output-format stream-json`. Sources come from the tool stream: the CLI's web
330
+ tools run client-side, so the usage counter for server-side searches stays at 0."""
331
+ calls: dict[str, tuple[str, dict]] = {}
332
+ tools: list[str] = []
333
+ urls: list[str] = []
334
+ final: dict = {}
335
+ for line in stdout.splitlines():
336
+ try:
337
+ ev = json.loads(line)
338
+ except ValueError:
339
+ continue
340
+ if not isinstance(ev, dict):
341
+ continue
342
+ blocks = (ev.get("message") or {}).get("content") or []
343
+ if ev.get("type") == "assistant":
344
+ for b in blocks:
345
+ if isinstance(b, dict) and b.get("type") == "tool_use" and b.get("name"):
346
+ tools.append(b["name"])
347
+ calls[b.get("id", "")] = (b["name"], b.get("input") or {})
348
+ elif ev.get("type") == "user":
349
+ for b in blocks:
350
+ if not isinstance(b, dict) or b.get("type") != "tool_result" or b.get("is_error"):
351
+ continue
352
+ name, inp = calls.get(b.get("tool_use_id", ""), ("", {}))
353
+ if name == "WebSearch":
354
+ urls += _search_links(_tool_text(b.get("content")))
355
+ elif name == "WebFetch" and isinstance(inp.get("url"), str):
356
+ urls.append(inp["url"])
357
+ elif ev.get("type") == "result":
358
+ final = ev
359
+ if not final:
360
+ raise RuntimeError("claude-cli: no result in the CLI output")
361
+ if final.get("is_error") or final.get("subtype") != "success":
362
+ raise RuntimeError(f"claude-cli: {final.get('subtype')}: {str(final.get('result', ''))[:200]}")
363
+ text = final.get("result") or ""
364
+ mu = final.get("modelUsage") or {}
365
+ # The CLI may use a small model for side jobs; the answer comes from the one that wrote most.
366
+ model = max(mu, key=lambda m: (mu[m] or {}).get("outputTokens", 0)) if mu else ""
367
+ return {
368
+ "text": text,
369
+ "urls": list(dict.fromkeys(urls + URL_RX.findall(text))),
370
+ "model": model,
371
+ "usage": {
372
+ "turns": final.get("num_turns"),
373
+ "tools": tools,
374
+ "web_search_requests": tools.count("WebSearch"),
375
+ "web_fetch_requests": tools.count("WebFetch"),
376
+ "duration_ms": final.get("duration_ms"),
377
+ # What the CLI says the call would cost at API prices. Not charged on a subscription.
378
+ "api_equivalent_usd": final.get("total_cost_usd"),
379
+ },
380
+ }
381
+
382
+
383
+ def claude_cli_answer(question: str, *, key: str = "", model: str = "default",
384
+ timeout: float = CLAUDE_CLI_TIMEOUT, env: dict | None = None) -> dict:
385
+ """Claude through the local Claude Code CLI, with only WebSearch and WebFetch available.
386
+
387
+ Uses the operator's Claude subscription, not API credit, so `cost_usd` is 0 and the row
388
+ is flagged `subscription`. `key` is unused. `model="default"` lets the CLI pick.
389
+ """
390
+ src = os.environ if env is None else env
391
+ exe = claude_exe(src) # from the caller's PATH, before the environment is replaced
392
+ with tempfile.TemporaryDirectory(prefix="aeb-home-") as home, \
393
+ tempfile.TemporaryDirectory(prefix="aeb-cwd-") as cwd:
394
+ cenv = cli_env(home, src)
395
+ cenv["GIT_CEILING_DIRECTORIES"] = os.path.dirname(cwd) # never find a repo above the temp dir
396
+ real = _stage_auth(cenv["CLAUDE_CONFIG_DIR"], src)
397
+ mcp = os.path.join(home, "mcp.json")
398
+ with open(mcp, "w", encoding="utf-8") as f:
399
+ f.write('{"mcpServers": {}}')
400
+ cmd = [exe, "-p", "--output-format", "stream-json", "--verbose",
401
+ "--append-system-prompt", SYSTEM_HINT, "--max-turns", "12",
402
+ "--tools", _CLI_TOOLS, "--allowedTools", _CLI_TOOLS,
403
+ "--strict-mcp-config", "--mcp-config", mcp, "--setting-sources", "user"]
404
+ if model and model != "default":
405
+ cmd += ["--model", model]
406
+ try:
407
+ r = _run_cli(cmd, question, cwd=cwd, env=cenv, timeout=timeout)
408
+ finally:
409
+ if real is not None:
410
+ _write_back_auth(real, os.path.join(cenv["CLAUDE_CONFIG_DIR"], ".credentials.json"))
411
+ try:
412
+ out = parse_cli_stream(r.stdout or "")
413
+ except RuntimeError as e:
414
+ detail = (r.stderr or "").strip()[:200]
415
+ raise RuntimeError(f"{e} (exit {r.returncode}){': ' + detail if detail else ''}") from None
416
+ if r.returncode != 0:
417
+ raise RuntimeError(f"claude-cli exit {r.returncode}: {(r.stderr or '').strip()[:200]}")
418
+ out["model"] = out["model"] or model
419
+ out["cost_usd"] = 0.0
420
+ out["subscription"] = True
421
+ return out
422
+
423
+
424
+ @dataclass(frozen=True)
425
+ class Engine:
426
+ name: str
427
+ env_key: str # empty: the engine needs no key and runs only when asked for by name
428
+ fn: Callable[..., dict]
429
+ default_model: str
430
+
431
+
432
+ ENGINES: dict[str, Engine] = {
433
+ "openai": Engine("openai", "OPENAI_API_KEY", openai_answer, "gpt-5-mini"),
434
+ "gemini": Engine("gemini", "GEMINI_API_KEY", gemini_answer, "gemini-3.5-flash"),
435
+ "claude": Engine("claude", "ANTHROPIC_API_KEY", anthropic_answer, "claude-sonnet-5"),
436
+ "perplexity": Engine("perplexity", "PERPLEXITY_API_KEY", perplexity_answer, "perplexity/sonar"),
437
+ "claude-cli": Engine("claude-cli", "", claude_cli_answer, "default"),
438
+ }
439
+
440
+
441
+ @dataclass(frozen=True)
442
+ class Bound:
443
+ """An engine with its key and model resolved. `key` is never printed or stored."""
444
+
445
+ name: str
446
+ model: str
447
+ key: str
448
+ fn: Callable[..., dict]
449
+ timeout: float | None = None # None: the adapter's own default (180 s, claude-cli 600 s)
450
+
451
+ def ask(self, question: str) -> dict:
452
+ if self.timeout is None:
453
+ return self.fn(question, key=self.key, model=self.model)
454
+ return self.fn(question, key=self.key, model=self.model, timeout=self.timeout)
455
+
456
+ def __repr__(self) -> str: # keep the key out of tracebacks and debug output
457
+ return f"Bound(name={self.name!r}, model={self.model!r})"
458
+
459
+
460
+ def available(env: dict | None = None, *, only: list[str] | None = None,
461
+ models: dict[str, str] | None = None, timeout: float | None = None) -> dict[str, Bound]:
462
+ """Engines whose key is set, plus keyless engines named in `only`. `models` overrides a
463
+ default model per engine and `timeout` the per-call limit in seconds."""
464
+ env = os.environ if env is None else env
465
+ models = models or {}
466
+ out = {}
467
+ for name, e in ENGINES.items():
468
+ if only and name not in only:
469
+ continue
470
+ if e.env_key:
471
+ key = env.get(e.env_key, "")
472
+ if not key:
473
+ continue
474
+ elif only:
475
+ claude_exe(env) # fail now, with a clear message, rather than on every call
476
+ key = ""
477
+ else:
478
+ continue
479
+ out[name] = Bound(name, models.get(name, e.default_model), key, e.fn, timeout)
480
+ return out
481
+
482
+
483
+ def redact(message: str, secrets: list[str]) -> str:
484
+ """Remove every key from an error message before it is logged."""
485
+ for s in secrets:
486
+ if s and len(s) >= 6:
487
+ message = message.replace(s, "[redacted]")
488
+ # Belt and braces for key-shaped strings that were not in the list.
489
+ return re.sub(r"(key=)[^&\s]+", r"\1[redacted]", message)
@@ -50,6 +50,8 @@ def run(q: dict, engines: dict[str, Bound], *, runs: int = 3, out: Path,
50
50
  "urls": ans.get("urls", []),
51
51
  "usage": ans.get("usage", {}),
52
52
  "cost_usd": ans.get("cost_usd"),
53
+ # True when the call used a subscription (claude-cli), so cost_usd is 0.
54
+ "subscription": bool(ans.get("subscription")),
53
55
  **score(ans, q["ours"], q["brand_terms"], stale_markers=q["stale_markers"]),
54
56
  }
55
57
  f.write(json.dumps(rec, ensure_ascii=False) + "\n")
@@ -85,6 +85,9 @@ def render(records: list[dict], ours: list[str], *, anonymise: bool = False, top
85
85
  lines.append(f"| {k} | {b['answered']} | {b['errors']} | {_pct(b['mention_rate'])} | "
86
86
  f"{b['cited']} of {b['answered']} ({_pct(b['citation_rate'])}) | "
87
87
  f"{b['avg_position'] or '-'} | {b['cost']:.4f} |")
88
+ subs = sorted({r["engine"] for r in records if r.get("subscription")})
89
+ if subs:
90
+ lines += ["", f"{', '.join(subs)} ran on a subscription, not API credit, so its cost shows as 0."]
88
91
 
89
92
  engines = sorted({r["engine"] for r in records})
90
93
  groups = list(dict.fromkeys(r["group"] for r in records))
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: answer-engine-benchmark
3
- Version: 0.1.1
3
+ Version: 0.1.2
4
4
  Summary: Ask ChatGPT, Gemini, Perplexity and Claude your buyers' questions, several times each, and count who they cite.
5
5
  Author: Synapse Research Ltd
6
6
  License-Expression: MIT
@@ -99,11 +99,31 @@ Gemini answer would look like it cited Google.
99
99
  | `gemini` | `GEMINI_API_KEY` | `gemini-3.5-flash` | Google Search grounding |
100
100
  | `perplexity` | `PERPLEXITY_API_KEY` | `perplexity/sonar` | Responses API `web_search` tool |
101
101
  | `claude` | `ANTHROPIC_API_KEY` | `claude-sonnet-5` | Messages API web search tool |
102
+ | `claude-cli` | none, uses your `claude` login | whatever the CLI picks | Claude Code's `WebSearch` and `WebFetch` tools |
102
103
 
103
104
  Keys are read from the environment only. They are sent in request headers and
104
105
  removed from any error message before it is written, so they never reach the
105
106
  results file. An engine with no key is skipped.
106
107
 
108
+ `claude-cli` is for people who pay for a Claude subscription and have no API
109
+ key. It runs the [Claude Code](https://docs.claude.com/en/docs/claude-code) CLI
110
+ (`claude -p`) once per call, with only the web search and fetch tools switched
111
+ on. It never joins a run on its own: name it with `--engine claude-cli`. If
112
+ `claude` is not on your PATH, the run stops before the first call and says so.
113
+ Log in once by running `claude`. On macOS, where the login sits in the Keychain,
114
+ run `claude setup-token` and export `CLAUDE_CODE_OAUTH_TOKEN` instead. Its rows
115
+ record a cost of 0 and `"subscription": true`. The calls use up your plan's
116
+ usage limits and never show on an API bill. A call gets 600 seconds by
117
+ default. Change that with `--timeout`, which works for every engine (the API
118
+ engines default to 180). `--model claude-cli=sonnet` is passed on to the CLI.
119
+ Its sources are the search results and fetched pages from the tool calls, plus
120
+ any link in the answer text.
121
+
122
+ The CLI's answers are not the API's answers. Claude Code adds its own system
123
+ prompt. The model and search limits come from your account and plan. Two people
124
+ can get different results from the same question file. Say which one you
125
+ used when you publish numbers.
126
+
107
127
  Change a model with `--model claude=claude-opus-5`. Pick the models your
108
128
  buyers actually use in the apps, or say in the report which ones you used.
109
129
 
@@ -148,6 +168,17 @@ API. The CLI passed the operator's own git identity to the model, and many brand
148
168
  answers then told the reader the company was probably their own. The API
149
169
  adapters here send only the question and a one-line instruction to cite sources.
150
170
 
171
+ The `claude-cli` engine guards against the same leak. Each call runs in a new,
172
+ empty temporary folder with no git repository above it. HOME and the config
173
+ folder are temporary too, and git is told to read no config at all. Only an
174
+ allow-list of environment variables gets through (PATH, locale, proxy and CA
175
+ settings), so `GIT_*`, `USER` and any `CLAUDE*` or `ANTHROPIC*` variables are
176
+ dropped. Your `CLAUDE.md` files, memory, settings, hooks and MCP servers are
177
+ never loaded. Only your login is copied in, and if the CLI refreshes it during
178
+ the call, the new one is written back. What cannot be hidden is the account:
179
+ the answers still come from your Claude login, on your plan, so results depend
180
+ on the local account.
181
+
151
182
  ## Example
152
183
 
153
184
  `examples/synapse-launch-week-2026-09.md` is a real run on one company's own
@@ -173,7 +204,8 @@ pip install -e ".[test]"
173
204
  pytest
174
205
  ```
175
206
 
176
- 33 tests. They mock every API, so they spend nothing and need no keys.
207
+ 46 tests. They mock every API and stand in a fake `claude` for the CLI, so they
208
+ spend nothing and need no keys or login.
177
209
 
178
210
  ## Licence
179
211
 
@@ -15,6 +15,7 @@ src/answer_engine_benchmark.egg-info/dependency_links.txt
15
15
  src/answer_engine_benchmark.egg-info/entry_points.txt
16
16
  src/answer_engine_benchmark.egg-info/requires.txt
17
17
  src/answer_engine_benchmark.egg-info/top_level.txt
18
+ tests/test_claude_cli.py
18
19
  tests/test_engines.py
19
20
  tests/test_questions.py
20
21
  tests/test_run_and_report.py
@@ -0,0 +1,213 @@
1
+ """claude-cli engine tests. A fake `claude` script stands in for the real CLI, so nothing
2
+ here needs a login or makes a network call."""
3
+
4
+ import json
5
+ import os
6
+ import stat
7
+ import sys
8
+ import textwrap
9
+ import time
10
+
11
+ import pytest
12
+
13
+ from answer_engine_benchmark import engines as E
14
+ from answer_engine_benchmark.cli import main
15
+ from answer_engine_benchmark.runner import run
16
+
17
+ SEARCH_RESULT = ('Web search results for query: "acme"\n\nLinks: [{"title": "A [1]", "url": "https://s.test/1"},'
18
+ '{"title": "B", "url": "https://acme.test/about"}]\n\nMore text')
19
+
20
+ STREAM = [
21
+ {"type": "system", "subtype": "init", "tools": ["WebFetch", "WebSearch"]},
22
+ {"type": "assistant", "message": {"content": [
23
+ {"type": "tool_use", "id": "t1", "name": "WebSearch", "input": {"query": "acme"}}]}},
24
+ {"type": "user", "message": {"content": [{"type": "tool_result", "tool_use_id": "t1", "content": SEARCH_RESULT}]}},
25
+ {"type": "assistant", "message": {"content": [
26
+ {"type": "tool_use", "id": "t2", "name": "WebFetch", "input": {"url": "https://down.test/"}},
27
+ {"type": "tool_use", "id": "t3", "name": "WebFetch", "input": {"url": "https://f.test/page"}}]}},
28
+ {"type": "user", "message": {"content": [
29
+ {"type": "tool_result", "tool_use_id": "t2", "content": "ENOTFOUND", "is_error": True},
30
+ {"type": "tool_result", "tool_use_id": "t3", "content": [{"type": "text", "text": "page"}]}]}},
31
+ {"type": "result", "subtype": "success", "is_error": False, "num_turns": 3, "duration_ms": 1200,
32
+ "result": "Acme makes widgets ([About](https://acme.test/about)). See https://t.test/x.",
33
+ "total_cost_usd": 0.0421,
34
+ "modelUsage": {"claude-haiku-5": {"outputTokens": 40}, "claude-opus-5": {"outputTokens": 900}}},
35
+ ]
36
+
37
+
38
+ def make_fake(tmp_path, *, stream=None, exit_code=0, stderr="", sleep=0.0):
39
+ """Put a fake `claude` on a fresh PATH. It records what it was given, then replies."""
40
+ bindir = tmp_path / "bin"
41
+ bindir.mkdir(exist_ok=True)
42
+ record = tmp_path / "record.json"
43
+ out = "\n".join(json.dumps(e) for e in (STREAM if stream is None else stream))
44
+ script = bindir / "claude"
45
+ script.write_text(textwrap.dedent(f"""\
46
+ #!{sys.executable}
47
+ import json, os, sys, time
48
+ cfg = os.environ.get("CLAUDE_CONFIG_DIR", "")
49
+ cred = os.path.join(cfg, ".credentials.json")
50
+ json.dump({{"argv": sys.argv[1:], "stdin": sys.stdin.read(), "cwd": os.getcwd(),
51
+ "cwd_files": os.listdir("."), "env": dict(os.environ),
52
+ "cred": open(cred).read() if os.path.exists(cred) else None}},
53
+ open({str(record)!r}, "w"))
54
+ time.sleep({sleep})
55
+ sys.stdout.write({out!r})
56
+ sys.stderr.write({stderr!r})
57
+ sys.exit({exit_code})
58
+ """))
59
+ script.chmod(script.stat().st_mode | stat.S_IXUSR)
60
+ return bindir, record
61
+
62
+
63
+ @pytest.fixture
64
+ def user_home(tmp_path):
65
+ """A pretend operator: logged in, with a git identity and a CLAUDE.md."""
66
+ home = tmp_path / "home"
67
+ (home / ".claude").mkdir(parents=True)
68
+ (home / ".claude" / ".credentials.json").write_text('{"claudeAiOauth": {"expiresAt": 100}}')
69
+ (home / ".claude" / "CLAUDE.md").write_text("The user owns Acme.")
70
+ (home / ".gitconfig").write_text("[user]\n name = operator-handle\n")
71
+ return home
72
+
73
+
74
+ def env_for(bindir, home, **extra):
75
+ return {"PATH": f"{bindir}{os.pathsep}/usr/bin{os.pathsep}/bin", "HOME": str(home), "USER": "operator-handle",
76
+ "GIT_AUTHOR_NAME": "operator-handle", "GIT_DIR": "/somewhere/.git", "ANTHROPIC_API_KEY": "sk-ant-x",
77
+ "CLAUDECODE": "1", "LANG": "C.UTF-8", **extra}
78
+
79
+
80
+ def test_answer_text_sources_and_zero_cost(tmp_path, user_home):
81
+ bindir, _ = make_fake(tmp_path)
82
+ out = E.claude_cli_answer("Who makes widgets?", env=env_for(bindir, user_home))
83
+ assert out["text"].startswith("Acme makes widgets")
84
+ # Search results, then pages fetched without error, then links in the text; no repeats.
85
+ assert out["urls"] == ["https://s.test/1", "https://acme.test/about", "https://f.test/page", "https://t.test/x."]
86
+ assert out["model"] == "claude-opus-5"
87
+ assert out["cost_usd"] == 0.0 and out["subscription"] is True
88
+ assert out["usage"]["web_search_requests"] == 1 and out["usage"]["web_fetch_requests"] == 2
89
+ assert out["usage"]["api_equivalent_usd"] == 0.0421
90
+
91
+
92
+ def test_runs_as_a_stranger(tmp_path, user_home):
93
+ bindir, record = make_fake(tmp_path)
94
+ E.claude_cli_answer("-rf what is acme?", env=env_for(bindir, user_home))
95
+ rec = json.loads(record.read_text())
96
+ env = rec["env"]
97
+ # Fresh empty working dir and HOME, outside any repo, and gone afterwards.
98
+ assert rec["cwd_files"] == [] and not os.path.exists(rec["cwd"])
99
+ assert env["HOME"] != str(user_home) and not os.path.exists(env["HOME"])
100
+ assert env["CLAUDE_CONFIG_DIR"].startswith(env["HOME"])
101
+ assert env["GIT_CONFIG_GLOBAL"] == os.devnull and env["GIT_CONFIG_NOSYSTEM"] == "1"
102
+ assert env["GIT_CEILING_DIRECTORIES"] == os.path.dirname(rec["cwd"])
103
+ assert not [k for k in env if k.startswith("GIT_") and k not in
104
+ ("GIT_CONFIG_GLOBAL", "GIT_CONFIG_NOSYSTEM", "GIT_CEILING_DIRECTORIES")]
105
+ assert "operator-handle" not in json.dumps(env)
106
+ assert "ANTHROPIC_API_KEY" not in env and "CLAUDECODE" not in env
107
+ # Only the login was copied in; the operator's CLAUDE.md was not.
108
+ assert json.loads(rec["cred"]) == {"claudeAiOauth": {"expiresAt": 100}}
109
+ # The question goes in on stdin, so a leading "-" cannot become a flag.
110
+ assert rec["stdin"] == "-rf what is acme?" and "-rf what is acme?" not in rec["argv"]
111
+ argv = rec["argv"]
112
+ assert argv[:3] == ["-p", "--output-format", "stream-json"]
113
+ assert argv[argv.index("--tools") + 1] == "WebSearch,WebFetch"
114
+ assert argv[argv.index("--allowedTools") + 1] == "WebSearch,WebFetch"
115
+ assert "--strict-mcp-config" in argv and "--model" not in argv
116
+
117
+
118
+ def test_model_override_is_passed(tmp_path, user_home):
119
+ bindir, record = make_fake(tmp_path)
120
+ E.claude_cli_answer("q", model="sonnet", env=env_for(bindir, user_home))
121
+ argv = json.loads(record.read_text())["argv"]
122
+ assert argv[argv.index("--model") + 1] == "sonnet"
123
+
124
+
125
+ def test_token_variable_replaces_the_credentials_file(tmp_path):
126
+ bindir, record = make_fake(tmp_path)
127
+ E.claude_cli_answer("q", env=env_for(bindir, tmp_path / "nobody", CLAUDE_CODE_OAUTH_TOKEN="tok"))
128
+ rec = json.loads(record.read_text())
129
+ assert rec["env"]["CLAUDE_CODE_OAUTH_TOKEN"] == "tok" and rec["cred"] is None
130
+
131
+
132
+ def test_no_login_is_a_clear_error(tmp_path):
133
+ bindir, _ = make_fake(tmp_path)
134
+ with pytest.raises(E.EngineUnavailable, match="setup-token"):
135
+ E.claude_cli_answer("q", env=env_for(bindir, tmp_path / "nobody"))
136
+
137
+
138
+ def test_refreshed_login_is_written_back(tmp_path, user_home):
139
+ bindir, _ = make_fake(tmp_path)
140
+ script = bindir / "claude"
141
+ # The fake refreshes its login the way the real CLI does when the token is near expiry.
142
+ script.write_text(script.read_text().replace(
143
+ "time.sleep(", 'open(cred, "w").write(\'{"claudeAiOauth": {"expiresAt": 200}}\'); time.sleep('))
144
+ E.claude_cli_answer("q", env=env_for(bindir, user_home))
145
+ assert json.loads((user_home / ".claude" / ".credentials.json").read_text())["claudeAiOauth"]["expiresAt"] == 200
146
+
147
+
148
+ def test_timeout(tmp_path, user_home):
149
+ bindir, _ = make_fake(tmp_path, sleep=30)
150
+ t0 = time.time()
151
+ with pytest.raises(TimeoutError, match="after 1s"):
152
+ E.claude_cli_answer("q", timeout=1, env=env_for(bindir, user_home))
153
+ assert time.time() - t0 < 10
154
+
155
+
156
+ def test_cli_error_result_and_bad_exit(tmp_path, user_home):
157
+ bindir, _ = make_fake(tmp_path, stream=[{"type": "result", "subtype": "error_max_turns", "is_error": True}],
158
+ exit_code=1)
159
+ with pytest.raises(RuntimeError, match="error_max_turns"):
160
+ E.claude_cli_answer("q", env=env_for(bindir, user_home))
161
+ bindir, _ = make_fake(tmp_path, stream=[], exit_code=1, stderr="Invalid API key")
162
+ with pytest.raises(RuntimeError, match="Invalid API key"):
163
+ E.claude_cli_answer("q", env=env_for(bindir, user_home))
164
+
165
+
166
+ def test_missing_cli_is_a_clear_error(tmp_path):
167
+ env = {"PATH": str(tmp_path)}
168
+ with pytest.raises(E.EngineUnavailable, match="no `claude` on PATH"):
169
+ E.available(env, only=["claude-cli"])
170
+ with pytest.raises(E.EngineUnavailable):
171
+ E.claude_cli_answer("q", env=env)
172
+
173
+
174
+ def test_selected_only_by_name(tmp_path):
175
+ bindir, _ = make_fake(tmp_path)
176
+ env = {"PATH": str(bindir), "OPENAI_API_KEY": "sk-1"}
177
+ assert list(E.available(env)) == ["openai"] # never joins a run on its own
178
+ got = E.available(env, only=["claude-cli"], timeout=42)
179
+ assert list(got) == ["claude-cli"] and got["claude-cli"].timeout == 42
180
+ assert E.available(env, only=["claude-cli"])["claude-cli"].timeout is None # the adapter's default
181
+
182
+
183
+ def test_bound_passes_timeout_only_when_set():
184
+ seen = []
185
+
186
+ def old_style(question, *, key, model): # an adapter written before --timeout existed
187
+ seen.append(model)
188
+ return {}
189
+
190
+ E.Bound("x", "m", "", old_style).ask("q")
191
+ assert seen == ["m"]
192
+ got = {}
193
+ E.Bound("x", "m", "", lambda q, **kw: got.update(kw) or {}, timeout=5).ask("q")
194
+ assert got["timeout"] == 5
195
+
196
+
197
+ def test_run_records_subscription_flag(tmp_path, user_home, monkeypatch):
198
+ bindir, _ = make_fake(tmp_path)
199
+ for k, v in env_for(bindir, user_home).items():
200
+ monkeypatch.setenv(k, v)
201
+ q = {"ours": ["acme.test"], "brand_terms": ["Acme"], "stale_markers": [], "groups": {"brand": ["What is Acme?"]}}
202
+ eng = E.available(only=["claude-cli"])
203
+ recs = run(q, eng, runs=1, out=tmp_path / "out" / "answers.jsonl", sleep=0)
204
+ assert recs[0]["subscription"] is True and recs[0]["cost_usd"] == 0.0
205
+ assert recs[0]["cited"] and recs[0]["engine"] == "claude-cli"
206
+
207
+
208
+ def test_cli_reports_a_missing_claude(tmp_path, monkeypatch, capsys):
209
+ qf = tmp_path / "q.yaml"
210
+ qf.write_text("ours: [acme.test]\nbrand_terms: [Acme]\ngroups:\n brand:\n - What is Acme?\n")
211
+ monkeypatch.setenv("PATH", str(tmp_path))
212
+ assert main(["run", str(qf), "--engine", "claude-cli", "--out", str(tmp_path / "o")]) == 2
213
+ assert "no `claude` on PATH" in capsys.readouterr().err
@@ -1,240 +0,0 @@
1
- """One adapter per answer engine, over plain HTTPS so no vendor SDK is needed.
2
-
3
- Every adapter takes a question and returns:
4
-
5
- {"text": str, "urls": [str], "model": str, "usage": dict, "cost_usd": float | None}
6
-
7
- `urls` holds every source the engine attached to the answer, in the order it gave
8
- them, because the position of your first citation is scored. Keys are read from
9
- the environment only and are never written to a log or a results file.
10
- """
11
-
12
- from __future__ import annotations
13
-
14
- import os
15
- import re
16
- from collections.abc import Callable
17
- from dataclasses import dataclass
18
-
19
- import requests
20
-
21
- TIMEOUT = 180
22
- SYSTEM_HINT = "Answer as you would for any user. Cite your sources with links."
23
-
24
- # Public list prices in USD per 1M tokens (input, output), checked 2026-09. Vendors change
25
- # these, so a run's cost is an estimate unless the API reports it (Perplexity does).
26
- PRICES: dict[str, tuple[float, float]] = {
27
- "gpt-5-mini": (0.25, 2.00),
28
- "gemini-3.5-flash": (0.30, 2.50),
29
- "claude-opus-5": (5.00, 25.00),
30
- "claude-sonnet-5": (2.00, 10.00),
31
- "perplexity/sonar": (1.00, 1.00),
32
- }
33
- # Claude's web search tool is billed per search on top of tokens: $10 per 1,000.
34
- ANTHROPIC_SEARCH_USD = 0.01
35
-
36
-
37
- def token_cost(model: str, inp: int, out: int) -> float | None:
38
- p = PRICES.get(model)
39
- if p is None:
40
- # Dated snapshots such as gpt-5-mini-2025-08-07 price like their base name.
41
- p = next((v for k, v in PRICES.items() if model.startswith(k + "-")), None)
42
- return round((inp * p[0] + out * p[1]) / 1e6, 6) if p else None
43
-
44
-
45
- # ----------------------------------------------------------------------------- OpenAI ----
46
- def openai_answer(question: str, *, key: str, model: str = "gpt-5-mini") -> dict:
47
- body = {
48
- "model": model,
49
- "tools": [{"type": "web_search"}],
50
- "tool_choice": "auto",
51
- "input": question,
52
- "instructions": SYSTEM_HINT,
53
- }
54
- r = requests.post("https://api.openai.com/v1/responses",
55
- headers={"Authorization": f"Bearer {key}"}, json=body, timeout=TIMEOUT)
56
- r.raise_for_status()
57
- d = r.json()
58
- text, urls = [], []
59
- for item in d.get("output", []) or []:
60
- if item.get("type") != "message":
61
- continue
62
- for c in item.get("content", []) or []:
63
- if c.get("type") == "output_text":
64
- text.append(c.get("text", ""))
65
- urls += [a["url"] for a in c.get("annotations", []) or []
66
- if a.get("type") == "url_citation" and a.get("url")]
67
- u = d.get("usage", {}) or {}
68
- return {
69
- "text": "\n".join(text),
70
- "urls": urls,
71
- "model": d.get("model", model),
72
- "usage": u,
73
- "cost_usd": token_cost(d.get("model", model), u.get("input_tokens", 0), u.get("output_tokens", 0)),
74
- }
75
-
76
-
77
- # ----------------------------------------------------------------------------- Gemini ----
78
- def gemini_answer(question: str, *, key: str, model: str = "gemini-3.5-flash") -> dict:
79
- body = {"contents": [{"parts": [{"text": question}]}], "tools": [{"google_search": {}}]}
80
- # The key goes in a header, not the URL. In the query string it ends up in the text of
81
- # any HTTPError, and from there in the results file.
82
- r = requests.post(
83
- f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent",
84
- headers={"x-goog-api-key": key}, json=body, timeout=TIMEOUT,
85
- )
86
- r.raise_for_status()
87
- d = r.json()
88
- cand = (d.get("candidates") or [{}])[0]
89
- text = "\n".join(p.get("text", "") for p in (cand.get("content") or {}).get("parts", []) or [])
90
- gm = cand.get("groundingMetadata") or {}
91
- urls = [(ch.get("web") or {}).get("uri") for ch in gm.get("groundingChunks", []) or []]
92
- u = d.get("usageMetadata", {}) or {}
93
- return {
94
- "text": text,
95
- # These are vertexaisearch.cloud.google.com redirect URLs; resolve.py turns them
96
- # into the real sources before scoring.
97
- "urls": [x for x in urls if x],
98
- "model": model,
99
- "usage": u,
100
- "cost_usd": token_cost(model, u.get("promptTokenCount", 0), u.get("candidatesTokenCount", 0)),
101
- "grounding_queries": gm.get("webSearchQueries", []),
102
- }
103
-
104
-
105
- # ----------------------------------------------------------------------------- Claude ----
106
- def anthropic_answer(question: str, *, key: str, model: str = "claude-sonnet-5") -> dict:
107
- """Claude through the Messages API with the server-side web search tool.
108
-
109
- No model fallback is configured on purpose: an answer from a different model would be
110
- scored as this one. A refusal is recorded as an error instead.
111
- """
112
- messages: list[dict] = [{"role": "user", "content": question}]
113
- headers = {"x-api-key": key, "anthropic-version": "2023-06-01"}
114
- text: list[str] = []
115
- urls: list[str] = []
116
- usage = {"input_tokens": 0, "output_tokens": 0, "web_search_requests": 0}
117
- served = model
118
- # A long server-tool turn can stop with `pause_turn`. Sending the partial turn back
119
- # lets it continue. Three rounds is plenty for a single question.
120
- for _ in range(3):
121
- body = {
122
- "model": model,
123
- "max_tokens": 16000,
124
- "system": SYSTEM_HINT,
125
- "messages": messages,
126
- "tools": [{"type": "web_search_20260209", "name": "web_search", "max_uses": 5}],
127
- }
128
- r = requests.post("https://api.anthropic.com/v1/messages", headers=headers, json=body,
129
- timeout=TIMEOUT)
130
- r.raise_for_status()
131
- d = r.json()
132
- served = d.get("model", served)
133
- u = d.get("usage", {}) or {}
134
- usage["input_tokens"] += u.get("input_tokens", 0)
135
- usage["output_tokens"] += u.get("output_tokens", 0)
136
- usage["web_search_requests"] += (u.get("server_tool_use") or {}).get("web_search_requests", 0)
137
- for b in d.get("content", []) or []:
138
- if b.get("type") == "text":
139
- text.append(b.get("text", ""))
140
- urls += [c["url"] for c in b.get("citations", []) or [] if c.get("url")]
141
- elif b.get("type") == "web_search_tool_result" and isinstance(b.get("content"), list):
142
- urls += [x["url"] for x in b["content"] if isinstance(x, dict) and x.get("url")]
143
- stop = d.get("stop_reason")
144
- if stop == "refusal":
145
- raise RuntimeError(f"refusal: {(d.get('stop_details') or {}).get('category')}")
146
- if stop != "pause_turn":
147
- break
148
- messages = [messages[0], {"role": "assistant", "content": d.get("content", [])}]
149
- cost = token_cost(served, usage["input_tokens"], usage["output_tokens"])
150
- if cost is not None:
151
- cost = round(cost + usage["web_search_requests"] * ANTHROPIC_SEARCH_USD, 6)
152
- return {"text": "".join(text), "urls": urls, "model": served, "usage": usage, "cost_usd": cost}
153
-
154
-
155
- # ------------------------------------------------------------------------- Perplexity ----
156
- def perplexity_answer(question: str, *, key: str, model: str = "perplexity/sonar") -> dict:
157
- """Perplexity's Responses API. Web search is OFF unless the tool is passed.
158
-
159
- Without `tools=[{"type": "web_search"}]` the call still succeeds and returns a fluent
160
- answer naming real companies, with zero citations. For a benchmark scored on
161
- citations that is the worst failure there is: no error, and no data.
162
- """
163
- body = {"model": model, "input": question, "tools": [{"type": "web_search"}]}
164
- r = requests.post("https://api.perplexity.ai/v1/responses",
165
- headers={"Authorization": f"Bearer {key}"}, json=body, timeout=TIMEOUT)
166
- r.raise_for_status()
167
- d = r.json()
168
- text, urls = "", []
169
- for item in d.get("output", []) or []:
170
- if item.get("type") == "search_results":
171
- urls += [x.get("url") for x in item.get("results") or [] if x.get("url")]
172
- for c in item.get("content", []) or []:
173
- text += c.get("text", "")
174
- urls += [a.get("url") for a in c.get("annotations") or [] if a.get("url")]
175
- u = d.get("usage", {}) or {}
176
- reported = (u.get("cost") or {}).get("total_cost")
177
- return {
178
- "text": text,
179
- "urls": list(dict.fromkeys(urls)),
180
- "model": model,
181
- "usage": u,
182
- "cost_usd": round(float(reported), 6) if reported is not None
183
- else token_cost(model, u.get("input_tokens", 0), u.get("output_tokens", 0)),
184
- }
185
-
186
-
187
- @dataclass(frozen=True)
188
- class Engine:
189
- name: str
190
- env_key: str
191
- fn: Callable[..., dict]
192
- default_model: str
193
-
194
-
195
- ENGINES: dict[str, Engine] = {
196
- "openai": Engine("openai", "OPENAI_API_KEY", openai_answer, "gpt-5-mini"),
197
- "gemini": Engine("gemini", "GEMINI_API_KEY", gemini_answer, "gemini-3.5-flash"),
198
- "claude": Engine("claude", "ANTHROPIC_API_KEY", anthropic_answer, "claude-sonnet-5"),
199
- "perplexity": Engine("perplexity", "PERPLEXITY_API_KEY", perplexity_answer, "perplexity/sonar"),
200
- }
201
-
202
-
203
- @dataclass(frozen=True)
204
- class Bound:
205
- """An engine with its key and model resolved. `key` is never printed or stored."""
206
-
207
- name: str
208
- model: str
209
- key: str
210
- fn: Callable[..., dict]
211
-
212
- def ask(self, question: str) -> dict:
213
- return self.fn(question, key=self.key, model=self.model)
214
-
215
- def __repr__(self) -> str: # keep the key out of tracebacks and debug output
216
- return f"Bound(name={self.name!r}, model={self.model!r})"
217
-
218
-
219
- def available(env: dict | None = None, *, only: list[str] | None = None,
220
- models: dict[str, str] | None = None) -> dict[str, Bound]:
221
- """Engines whose key is set. `models` overrides a default model per engine."""
222
- env = os.environ if env is None else env
223
- models = models or {}
224
- out = {}
225
- for name, e in ENGINES.items():
226
- if only and name not in only:
227
- continue
228
- key = env.get(e.env_key, "")
229
- if key:
230
- out[name] = Bound(name, models.get(name, e.default_model), key, e.fn)
231
- return out
232
-
233
-
234
- def redact(message: str, secrets: list[str]) -> str:
235
- """Remove every key from an error message before it is logged."""
236
- for s in secrets:
237
- if s and len(s) >= 6:
238
- message = message.replace(s, "[redacted]")
239
- # Belt and braces for key-shaped strings that were not in the list.
240
- return re.sub(r"(key=)[^&\s]+", r"\1[redacted]", message)