answer-engine-benchmark 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- answer_engine_benchmark/__init__.py +3 -0
- answer_engine_benchmark/__main__.py +3 -0
- answer_engine_benchmark/cli.py +168 -0
- answer_engine_benchmark/engines.py +240 -0
- answer_engine_benchmark/questions.py +66 -0
- answer_engine_benchmark/runner.py +76 -0
- answer_engine_benchmark/scoring.py +131 -0
- answer_engine_benchmark/summary.py +170 -0
- answer_engine_benchmark-0.1.0.dist-info/METADATA +180 -0
- answer_engine_benchmark-0.1.0.dist-info/RECORD +14 -0
- answer_engine_benchmark-0.1.0.dist-info/WHEEL +5 -0
- answer_engine_benchmark-0.1.0.dist-info/entry_points.txt +2 -0
- answer_engine_benchmark-0.1.0.dist-info/licenses/LICENSE +21 -0
- answer_engine_benchmark-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
"""aeb: ask AI answer engines your buyers' questions and count who they cite.
|
|
2
|
+
|
|
3
|
+
Exit codes: 0 done, 1 a run finished but every call failed, 2 bad input.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import argparse
|
|
9
|
+
import sys
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from . import __version__
|
|
13
|
+
from .engines import ENGINES, available
|
|
14
|
+
from .questions import QuestionError, count, load
|
|
15
|
+
from .runner import read_jsonl, rescore, run
|
|
16
|
+
from .summary import compare, pick_questions, render, render_compare
|
|
17
|
+
|
|
18
|
+
DOCS = "https://synapsereality.io/open-source/answer-engine-benchmark/"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _kv(raw: str) -> tuple[str, str]:
|
|
22
|
+
k, sep, v = raw.partition("=")
|
|
23
|
+
if not sep or not k.strip():
|
|
24
|
+
raise argparse.ArgumentTypeError(f"expected name=value, got {raw!r}")
|
|
25
|
+
return k.strip(), v.strip()
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _common(p: argparse.ArgumentParser) -> None:
|
|
29
|
+
p.add_argument("questions", help="question file (YAML)")
|
|
30
|
+
p.add_argument("--set", action="append", type=_kv, default=[], metavar="NAME=VALUE",
|
|
31
|
+
help="fill a {placeholder}, e.g. --set brand='Acme Ltd' (repeatable)")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _engine_flags(p: argparse.ArgumentParser) -> None:
|
|
35
|
+
p.add_argument("--engine", action="append", choices=sorted(ENGINES), help="only these engines (repeatable)")
|
|
36
|
+
p.add_argument("--model", action="append", type=_kv, default=[], metavar="ENGINE=MODEL",
|
|
37
|
+
help="override a model, e.g. --model claude=claude-opus-5")
|
|
38
|
+
p.add_argument("--runs", type=int, default=3, help="times to ask each question on each engine (default 3)")
|
|
39
|
+
p.add_argument("--max-calls", type=int, default=500,
|
|
40
|
+
help="refuse to start a run that needs more API calls than this (default 500)")
|
|
41
|
+
p.add_argument("--sleep", type=float, default=1.0, help="seconds between calls (default 1)")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
45
|
+
ap = argparse.ArgumentParser(prog="aeb", description=__doc__, epilog=f"Docs: {DOCS}",
|
|
46
|
+
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
47
|
+
ap.add_argument("--version", action="version", version=f"aeb {__version__}")
|
|
48
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
49
|
+
|
|
50
|
+
p = sub.add_parser("check", help="validate a question file and show what a run would cost in calls")
|
|
51
|
+
_common(p)
|
|
52
|
+
_engine_flags(p)
|
|
53
|
+
p.add_argument("--allow-unfilled", action="store_true", help="accept the template's example values")
|
|
54
|
+
|
|
55
|
+
p = sub.add_parser("run", help="ask every question on every engine and write answers.jsonl + report.md")
|
|
56
|
+
_common(p)
|
|
57
|
+
_engine_flags(p)
|
|
58
|
+
p.add_argument("--out", required=True, type=Path, help="output folder")
|
|
59
|
+
p.add_argument("--group", action="append", help="only these question groups (repeatable)")
|
|
60
|
+
p.add_argument("--dry-run", action="store_true", help="first question of each group, 1 run: a cheap smoke test")
|
|
61
|
+
p.add_argument("--anonymise", action="store_true", help="label other domains Source A, B, ... in report.md")
|
|
62
|
+
|
|
63
|
+
p = sub.add_parser("report", help="rebuild report.md from answers.jsonl (spends nothing)")
|
|
64
|
+
p.add_argument("answers", type=Path, help="answers.jsonl")
|
|
65
|
+
p.add_argument("questions", nargs="?", help="question file: re-score with its ours/brand_terms first")
|
|
66
|
+
p.add_argument("--set", action="append", type=_kv, default=[], metavar="NAME=VALUE")
|
|
67
|
+
p.add_argument("--anonymise", action="store_true")
|
|
68
|
+
p.add_argument("--out", type=Path, help="write here instead of stdout")
|
|
69
|
+
|
|
70
|
+
p = sub.add_parser("noise", help="re-ask a sample of questions and compare with the baseline")
|
|
71
|
+
_common(p)
|
|
72
|
+
_engine_flags(p)
|
|
73
|
+
p.add_argument("--baseline", required=True, type=Path, help="answers.jsonl from the first run")
|
|
74
|
+
p.add_argument("--out", required=True, type=Path, help="output folder for the repeat run")
|
|
75
|
+
p.add_argument("--sample", type=int, default=5, help="questions to re-ask, spread across groups (default 5)")
|
|
76
|
+
p.add_argument("--tolerance", type=int, default=1, help="allowed difference in citations (default 1)")
|
|
77
|
+
return ap
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _plan(q: dict, engines: dict, runs: int, max_calls: int) -> int | None:
|
|
81
|
+
calls = count(q) * len(engines) * runs
|
|
82
|
+
names = ", ".join(f"{n} ({e.model})" for n, e in engines.items()) or "none"
|
|
83
|
+
print(f"{count(q)} questions x {len(engines)} engines x {runs} runs = {calls} calls. Engines: {names}")
|
|
84
|
+
if calls > max_calls:
|
|
85
|
+
print(f"aeb: {calls} calls is over --max-calls {max_calls}. Raise it if you mean it.", file=sys.stderr)
|
|
86
|
+
return None
|
|
87
|
+
return calls
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _progress(rec: dict) -> None:
|
|
91
|
+
flag = "ERR " if rec["error"] else "CITE" if rec["cited"] else "ment" if rec["mentioned"] else " - "
|
|
92
|
+
cost = rec["cost_usd"] or 0
|
|
93
|
+
print(f"{flag} {rec['engine']:<10} {rec['group']:<11} pos={rec['position'] or '-':<3} ${cost:.4f} "
|
|
94
|
+
f"{rec['question'][:70]} {rec['error'][:80]}", flush=True)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def main(argv: list[str] | None = None) -> int:
|
|
98
|
+
args = build_parser().parse_args(argv)
|
|
99
|
+
try:
|
|
100
|
+
if args.cmd == "report":
|
|
101
|
+
recs = read_jsonl(args.answers)
|
|
102
|
+
ours = []
|
|
103
|
+
if args.questions:
|
|
104
|
+
q = load(args.questions, dict(args.set), allow_unfilled=True)
|
|
105
|
+
recs, ours = rescore(recs, q), q["ours"]
|
|
106
|
+
text = render(recs, ours, anonymise=args.anonymise)
|
|
107
|
+
if args.out:
|
|
108
|
+
args.out.write_text(text, encoding="utf-8")
|
|
109
|
+
else:
|
|
110
|
+
print(text, end="")
|
|
111
|
+
return 0
|
|
112
|
+
|
|
113
|
+
q = load(args.questions, dict(args.set), allow_unfilled=getattr(args, "allow_unfilled", False))
|
|
114
|
+
engines = available(only=args.engine, models=dict(args.model))
|
|
115
|
+
if args.cmd == "check":
|
|
116
|
+
print(f"ok: {count(q)} questions in {len(q['groups'])} groups; ours = {q['ours']}")
|
|
117
|
+
missing = [f"{n} ({e.env_key})" for n, e in ENGINES.items()
|
|
118
|
+
if n not in engines and (not args.engine or n in args.engine)]
|
|
119
|
+
if missing:
|
|
120
|
+
print("no key set for: " + ", ".join(missing))
|
|
121
|
+
_plan(q, engines, args.runs, args.max_calls)
|
|
122
|
+
return 0
|
|
123
|
+
|
|
124
|
+
if not engines:
|
|
125
|
+
keys = ", ".join(e.env_key for e in ENGINES.values())
|
|
126
|
+
print(f"aeb: no engine has a key. Set one or more of: {keys}", file=sys.stderr)
|
|
127
|
+
return 2
|
|
128
|
+
|
|
129
|
+
if args.cmd == "run":
|
|
130
|
+
if args.group:
|
|
131
|
+
q["groups"] = {g: v for g, v in q["groups"].items() if g in args.group}
|
|
132
|
+
if args.dry_run:
|
|
133
|
+
q["groups"] = {g: v[:1] for g, v in q["groups"].items()}
|
|
134
|
+
args.runs = 1
|
|
135
|
+
if _plan(q, engines, args.runs, args.max_calls) is None:
|
|
136
|
+
return 2
|
|
137
|
+
answers = args.out / "answers.jsonl"
|
|
138
|
+
recs = run(q, engines, runs=args.runs, out=answers, on_result=_progress, sleep=args.sleep)
|
|
139
|
+
everything = read_jsonl(answers)
|
|
140
|
+
(args.out / "report.md").write_text(render(everything, q["ours"], anonymise=args.anonymise),
|
|
141
|
+
encoding="utf-8")
|
|
142
|
+
cost = sum(r.get("cost_usd") or 0 for r in recs)
|
|
143
|
+
print(f"wrote {answers} (+{len(recs)} answers) and report.md. This run cost about ${cost:.3f}")
|
|
144
|
+
return 1 if recs and all(r["error"] for r in recs) else 0
|
|
145
|
+
|
|
146
|
+
if args.cmd == "noise":
|
|
147
|
+
baseline = read_jsonl(args.baseline)
|
|
148
|
+
picked = pick_questions(q, args.sample)
|
|
149
|
+
groups: dict[str, list[str]] = {}
|
|
150
|
+
for g, question in picked:
|
|
151
|
+
groups.setdefault(g, []).append(question)
|
|
152
|
+
q["groups"] = groups
|
|
153
|
+
if _plan(q, engines, args.runs, args.max_calls) is None:
|
|
154
|
+
return 2
|
|
155
|
+
answers = args.out / "repeat.jsonl"
|
|
156
|
+
run(q, engines, runs=args.runs, out=answers, on_result=_progress, sleep=args.sleep, label="repeat")
|
|
157
|
+
rows = compare(baseline, read_jsonl(answers), tolerance=args.tolerance)
|
|
158
|
+
text = render_compare(rows, args.tolerance)
|
|
159
|
+
(args.out / "noise.md").write_text(text, encoding="utf-8")
|
|
160
|
+
print(text, end="")
|
|
161
|
+
return 0
|
|
162
|
+
except QuestionError as e:
|
|
163
|
+
print(f"aeb: {e}", file=sys.stderr)
|
|
164
|
+
return 2
|
|
165
|
+
except (OSError, ValueError) as e:
|
|
166
|
+
print(f"aeb: {e}", file=sys.stderr)
|
|
167
|
+
return 2
|
|
168
|
+
return 2
|
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
"""One adapter per answer engine, over plain HTTPS so no vendor SDK is needed.
|
|
2
|
+
|
|
3
|
+
Every adapter takes a question and returns:
|
|
4
|
+
|
|
5
|
+
{"text": str, "urls": [str], "model": str, "usage": dict, "cost_usd": float | None}
|
|
6
|
+
|
|
7
|
+
`urls` holds every source the engine attached to the answer, in the order it gave
|
|
8
|
+
them, because the position of your first citation is scored. Keys are read from
|
|
9
|
+
the environment only and are never written to a log or a results file.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import os
|
|
15
|
+
import re
|
|
16
|
+
from collections.abc import Callable
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
|
|
19
|
+
import requests
|
|
20
|
+
|
|
21
|
+
TIMEOUT = 180
|
|
22
|
+
SYSTEM_HINT = "Answer as you would for any user. Cite your sources with links."
|
|
23
|
+
|
|
24
|
+
# Public list prices in USD per 1M tokens (input, output), checked 2026-09. Vendors change
|
|
25
|
+
# these, so a run's cost is an estimate unless the API reports it (Perplexity does).
|
|
26
|
+
PRICES: dict[str, tuple[float, float]] = {
|
|
27
|
+
"gpt-5-mini": (0.25, 2.00),
|
|
28
|
+
"gemini-3.5-flash": (0.30, 2.50),
|
|
29
|
+
"claude-opus-5": (5.00, 25.00),
|
|
30
|
+
"claude-sonnet-5": (2.00, 10.00),
|
|
31
|
+
"perplexity/sonar": (1.00, 1.00),
|
|
32
|
+
}
|
|
33
|
+
# Claude's web search tool is billed per search on top of tokens: $10 per 1,000.
|
|
34
|
+
ANTHROPIC_SEARCH_USD = 0.01
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def token_cost(model: str, inp: int, out: int) -> float | None:
|
|
38
|
+
p = PRICES.get(model)
|
|
39
|
+
if p is None:
|
|
40
|
+
# Dated snapshots such as gpt-5-mini-2025-08-07 price like their base name.
|
|
41
|
+
p = next((v for k, v in PRICES.items() if model.startswith(k + "-")), None)
|
|
42
|
+
return round((inp * p[0] + out * p[1]) / 1e6, 6) if p else None
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
# ----------------------------------------------------------------------------- OpenAI ----
|
|
46
|
+
def openai_answer(question: str, *, key: str, model: str = "gpt-5-mini") -> dict:
|
|
47
|
+
body = {
|
|
48
|
+
"model": model,
|
|
49
|
+
"tools": [{"type": "web_search"}],
|
|
50
|
+
"tool_choice": "auto",
|
|
51
|
+
"input": question,
|
|
52
|
+
"instructions": SYSTEM_HINT,
|
|
53
|
+
}
|
|
54
|
+
r = requests.post("https://api.openai.com/v1/responses",
|
|
55
|
+
headers={"Authorization": f"Bearer {key}"}, json=body, timeout=TIMEOUT)
|
|
56
|
+
r.raise_for_status()
|
|
57
|
+
d = r.json()
|
|
58
|
+
text, urls = [], []
|
|
59
|
+
for item in d.get("output", []) or []:
|
|
60
|
+
if item.get("type") != "message":
|
|
61
|
+
continue
|
|
62
|
+
for c in item.get("content", []) or []:
|
|
63
|
+
if c.get("type") == "output_text":
|
|
64
|
+
text.append(c.get("text", ""))
|
|
65
|
+
urls += [a["url"] for a in c.get("annotations", []) or []
|
|
66
|
+
if a.get("type") == "url_citation" and a.get("url")]
|
|
67
|
+
u = d.get("usage", {}) or {}
|
|
68
|
+
return {
|
|
69
|
+
"text": "\n".join(text),
|
|
70
|
+
"urls": urls,
|
|
71
|
+
"model": d.get("model", model),
|
|
72
|
+
"usage": u,
|
|
73
|
+
"cost_usd": token_cost(d.get("model", model), u.get("input_tokens", 0), u.get("output_tokens", 0)),
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# ----------------------------------------------------------------------------- Gemini ----
|
|
78
|
+
def gemini_answer(question: str, *, key: str, model: str = "gemini-3.5-flash") -> dict:
|
|
79
|
+
body = {"contents": [{"parts": [{"text": question}]}], "tools": [{"google_search": {}}]}
|
|
80
|
+
# The key goes in a header, not the URL. In the query string it ends up in the text of
|
|
81
|
+
# any HTTPError, and from there in the results file.
|
|
82
|
+
r = requests.post(
|
|
83
|
+
f"https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent",
|
|
84
|
+
headers={"x-goog-api-key": key}, json=body, timeout=TIMEOUT,
|
|
85
|
+
)
|
|
86
|
+
r.raise_for_status()
|
|
87
|
+
d = r.json()
|
|
88
|
+
cand = (d.get("candidates") or [{}])[0]
|
|
89
|
+
text = "\n".join(p.get("text", "") for p in (cand.get("content") or {}).get("parts", []) or [])
|
|
90
|
+
gm = cand.get("groundingMetadata") or {}
|
|
91
|
+
urls = [(ch.get("web") or {}).get("uri") for ch in gm.get("groundingChunks", []) or []]
|
|
92
|
+
u = d.get("usageMetadata", {}) or {}
|
|
93
|
+
return {
|
|
94
|
+
"text": text,
|
|
95
|
+
# These are vertexaisearch.cloud.google.com redirect URLs; resolve.py turns them
|
|
96
|
+
# into the real sources before scoring.
|
|
97
|
+
"urls": [x for x in urls if x],
|
|
98
|
+
"model": model,
|
|
99
|
+
"usage": u,
|
|
100
|
+
"cost_usd": token_cost(model, u.get("promptTokenCount", 0), u.get("candidatesTokenCount", 0)),
|
|
101
|
+
"grounding_queries": gm.get("webSearchQueries", []),
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
# ----------------------------------------------------------------------------- Claude ----
|
|
106
|
+
def anthropic_answer(question: str, *, key: str, model: str = "claude-sonnet-5") -> dict:
|
|
107
|
+
"""Claude through the Messages API with the server-side web search tool.
|
|
108
|
+
|
|
109
|
+
No model fallback is configured on purpose: an answer from a different model would be
|
|
110
|
+
scored as this one. A refusal is recorded as an error instead.
|
|
111
|
+
"""
|
|
112
|
+
messages: list[dict] = [{"role": "user", "content": question}]
|
|
113
|
+
headers = {"x-api-key": key, "anthropic-version": "2023-06-01"}
|
|
114
|
+
text: list[str] = []
|
|
115
|
+
urls: list[str] = []
|
|
116
|
+
usage = {"input_tokens": 0, "output_tokens": 0, "web_search_requests": 0}
|
|
117
|
+
served = model
|
|
118
|
+
# A long server-tool turn can stop with `pause_turn`. Sending the partial turn back
|
|
119
|
+
# lets it continue. Three rounds is plenty for a single question.
|
|
120
|
+
for _ in range(3):
|
|
121
|
+
body = {
|
|
122
|
+
"model": model,
|
|
123
|
+
"max_tokens": 16000,
|
|
124
|
+
"system": SYSTEM_HINT,
|
|
125
|
+
"messages": messages,
|
|
126
|
+
"tools": [{"type": "web_search_20260209", "name": "web_search", "max_uses": 5}],
|
|
127
|
+
}
|
|
128
|
+
r = requests.post("https://api.anthropic.com/v1/messages", headers=headers, json=body,
|
|
129
|
+
timeout=TIMEOUT)
|
|
130
|
+
r.raise_for_status()
|
|
131
|
+
d = r.json()
|
|
132
|
+
served = d.get("model", served)
|
|
133
|
+
u = d.get("usage", {}) or {}
|
|
134
|
+
usage["input_tokens"] += u.get("input_tokens", 0)
|
|
135
|
+
usage["output_tokens"] += u.get("output_tokens", 0)
|
|
136
|
+
usage["web_search_requests"] += (u.get("server_tool_use") or {}).get("web_search_requests", 0)
|
|
137
|
+
for b in d.get("content", []) or []:
|
|
138
|
+
if b.get("type") == "text":
|
|
139
|
+
text.append(b.get("text", ""))
|
|
140
|
+
urls += [c["url"] for c in b.get("citations", []) or [] if c.get("url")]
|
|
141
|
+
elif b.get("type") == "web_search_tool_result" and isinstance(b.get("content"), list):
|
|
142
|
+
urls += [x["url"] for x in b["content"] if isinstance(x, dict) and x.get("url")]
|
|
143
|
+
stop = d.get("stop_reason")
|
|
144
|
+
if stop == "refusal":
|
|
145
|
+
raise RuntimeError(f"refusal: {(d.get('stop_details') or {}).get('category')}")
|
|
146
|
+
if stop != "pause_turn":
|
|
147
|
+
break
|
|
148
|
+
messages = [messages[0], {"role": "assistant", "content": d.get("content", [])}]
|
|
149
|
+
cost = token_cost(served, usage["input_tokens"], usage["output_tokens"])
|
|
150
|
+
if cost is not None:
|
|
151
|
+
cost = round(cost + usage["web_search_requests"] * ANTHROPIC_SEARCH_USD, 6)
|
|
152
|
+
return {"text": "".join(text), "urls": urls, "model": served, "usage": usage, "cost_usd": cost}
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
# ------------------------------------------------------------------------- Perplexity ----
|
|
156
|
+
def perplexity_answer(question: str, *, key: str, model: str = "perplexity/sonar") -> dict:
|
|
157
|
+
"""Perplexity's Responses API. Web search is OFF unless the tool is passed.
|
|
158
|
+
|
|
159
|
+
Without `tools=[{"type": "web_search"}]` the call still succeeds and returns a fluent
|
|
160
|
+
answer naming real companies, with zero citations. For a benchmark scored on
|
|
161
|
+
citations that is the worst failure there is: no error, and no data.
|
|
162
|
+
"""
|
|
163
|
+
body = {"model": model, "input": question, "tools": [{"type": "web_search"}]}
|
|
164
|
+
r = requests.post("https://api.perplexity.ai/v1/responses",
|
|
165
|
+
headers={"Authorization": f"Bearer {key}"}, json=body, timeout=TIMEOUT)
|
|
166
|
+
r.raise_for_status()
|
|
167
|
+
d = r.json()
|
|
168
|
+
text, urls = "", []
|
|
169
|
+
for item in d.get("output", []) or []:
|
|
170
|
+
if item.get("type") == "search_results":
|
|
171
|
+
urls += [x.get("url") for x in item.get("results") or [] if x.get("url")]
|
|
172
|
+
for c in item.get("content", []) or []:
|
|
173
|
+
text += c.get("text", "")
|
|
174
|
+
urls += [a.get("url") for a in c.get("annotations") or [] if a.get("url")]
|
|
175
|
+
u = d.get("usage", {}) or {}
|
|
176
|
+
reported = (u.get("cost") or {}).get("total_cost")
|
|
177
|
+
return {
|
|
178
|
+
"text": text,
|
|
179
|
+
"urls": list(dict.fromkeys(urls)),
|
|
180
|
+
"model": model,
|
|
181
|
+
"usage": u,
|
|
182
|
+
"cost_usd": round(float(reported), 6) if reported is not None
|
|
183
|
+
else token_cost(model, u.get("input_tokens", 0), u.get("output_tokens", 0)),
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
@dataclass(frozen=True)
|
|
188
|
+
class Engine:
|
|
189
|
+
name: str
|
|
190
|
+
env_key: str
|
|
191
|
+
fn: Callable[..., dict]
|
|
192
|
+
default_model: str
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
ENGINES: dict[str, Engine] = {
|
|
196
|
+
"openai": Engine("openai", "OPENAI_API_KEY", openai_answer, "gpt-5-mini"),
|
|
197
|
+
"gemini": Engine("gemini", "GEMINI_API_KEY", gemini_answer, "gemini-3.5-flash"),
|
|
198
|
+
"claude": Engine("claude", "ANTHROPIC_API_KEY", anthropic_answer, "claude-sonnet-5"),
|
|
199
|
+
"perplexity": Engine("perplexity", "PERPLEXITY_API_KEY", perplexity_answer, "perplexity/sonar"),
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
@dataclass(frozen=True)
|
|
204
|
+
class Bound:
|
|
205
|
+
"""An engine with its key and model resolved. `key` is never printed or stored."""
|
|
206
|
+
|
|
207
|
+
name: str
|
|
208
|
+
model: str
|
|
209
|
+
key: str
|
|
210
|
+
fn: Callable[..., dict]
|
|
211
|
+
|
|
212
|
+
def ask(self, question: str) -> dict:
|
|
213
|
+
return self.fn(question, key=self.key, model=self.model)
|
|
214
|
+
|
|
215
|
+
def __repr__(self) -> str: # keep the key out of tracebacks and debug output
|
|
216
|
+
return f"Bound(name={self.name!r}, model={self.model!r})"
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def available(env: dict | None = None, *, only: list[str] | None = None,
|
|
220
|
+
models: dict[str, str] | None = None) -> dict[str, Bound]:
|
|
221
|
+
"""Engines whose key is set. `models` overrides a default model per engine."""
|
|
222
|
+
env = os.environ if env is None else env
|
|
223
|
+
models = models or {}
|
|
224
|
+
out = {}
|
|
225
|
+
for name, e in ENGINES.items():
|
|
226
|
+
if only and name not in only:
|
|
227
|
+
continue
|
|
228
|
+
key = env.get(e.env_key, "")
|
|
229
|
+
if key:
|
|
230
|
+
out[name] = Bound(name, models.get(name, e.default_model), key, e.fn)
|
|
231
|
+
return out
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def redact(message: str, secrets: list[str]) -> str:
|
|
235
|
+
"""Remove every key from an error message before it is logged."""
|
|
236
|
+
for s in secrets:
|
|
237
|
+
if s and len(s) >= 6:
|
|
238
|
+
message = message.replace(s, "[redacted]")
|
|
239
|
+
# Belt and braces for key-shaped strings that were not in the list.
|
|
240
|
+
return re.sub(r"(key=)[^&\s]+", r"\1[redacted]", message)
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""Question files: YAML with placeholders you fill with your own brand."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
import yaml
|
|
9
|
+
|
|
10
|
+
PLACEHOLDER = re.compile(r"\{([a-z_]+)\}")
|
|
11
|
+
# Values the shipped template uses so it parses. A run refuses them.
|
|
12
|
+
UNFILLED = {"your brand", "example.com", "your category", "your audience", "your city"}
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class QuestionError(ValueError):
|
|
16
|
+
pass
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _fill(value, vars_: dict[str, str]):
|
|
20
|
+
if isinstance(value, str):
|
|
21
|
+
def sub(m: re.Match) -> str:
|
|
22
|
+
k = m.group(1)
|
|
23
|
+
if k not in vars_:
|
|
24
|
+
raise QuestionError(f"placeholder {{{k}}} has no value; add it under vars: or pass --set {k}=...")
|
|
25
|
+
return str(vars_[k])
|
|
26
|
+
return PLACEHOLDER.sub(sub, value)
|
|
27
|
+
if isinstance(value, list):
|
|
28
|
+
return [_fill(v, vars_) for v in value]
|
|
29
|
+
if isinstance(value, dict):
|
|
30
|
+
return {k: _fill(v, vars_) for k, v in value.items()}
|
|
31
|
+
return value
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def load(path: str | Path, overrides: dict[str, str] | None = None, *, allow_unfilled: bool = False) -> dict:
|
|
35
|
+
"""Read a question file and fill its placeholders.
|
|
36
|
+
|
|
37
|
+
Values come from the file's `vars:` block, then from `overrides` (the --set flags).
|
|
38
|
+
Returns {"ours", "brand_terms", "stale_markers", "groups": {name: [question]}}.
|
|
39
|
+
"""
|
|
40
|
+
raw = yaml.safe_load(Path(path).read_text(encoding="utf-8")) or {}
|
|
41
|
+
vars_ = {str(k): str(v) for k, v in (raw.get("vars") or {}).items()}
|
|
42
|
+
vars_.update(overrides or {})
|
|
43
|
+
if not allow_unfilled:
|
|
44
|
+
left = sorted(k for k, v in vars_.items() if v.strip().lower() in UNFILLED)
|
|
45
|
+
if left:
|
|
46
|
+
raise QuestionError(
|
|
47
|
+
"these vars still hold the template's example values: " + ", ".join(left)
|
|
48
|
+
+ ". Set them in the file or with --set name=value.")
|
|
49
|
+
q = _fill({k: raw.get(k) for k in ("ours", "brand_terms", "stale_markers", "groups")}, vars_)
|
|
50
|
+
if not q.get("ours"):
|
|
51
|
+
raise QuestionError("`ours` is empty: list the domains (or domain/path) that count as your citation")
|
|
52
|
+
if not q.get("brand_terms"):
|
|
53
|
+
raise QuestionError("`brand_terms` is empty: list the names that count as a mention")
|
|
54
|
+
groups = q.get("groups") or {}
|
|
55
|
+
if not isinstance(groups, dict) or not any(groups.values()):
|
|
56
|
+
raise QuestionError("`groups` has no questions")
|
|
57
|
+
for g, qs in groups.items():
|
|
58
|
+
if not isinstance(qs, list) or not all(isinstance(x, str) and x.strip() for x in qs):
|
|
59
|
+
raise QuestionError(f"group {g!r} must be a list of question strings")
|
|
60
|
+
q["stale_markers"] = q.get("stale_markers") or []
|
|
61
|
+
q["groups"] = {g: list(qs) for g, qs in groups.items()}
|
|
62
|
+
return q
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def count(q: dict) -> int:
|
|
66
|
+
return sum(len(v) for v in q["groups"].values())
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""Ask every question on every engine, N times, and append each answer to a JSONL file."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import time
|
|
7
|
+
from collections.abc import Callable
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from .engines import Bound, redact
|
|
11
|
+
from .scoring import Resolver, score
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def ask_once(engine: Bound, question: str) -> tuple[dict, str]:
|
|
15
|
+
try:
|
|
16
|
+
return engine.ask(question), ""
|
|
17
|
+
except Exception as e: # noqa: BLE001 - one failed call is a row, not a crash
|
|
18
|
+
msg = redact(f"{type(e).__name__}: {e}", [engine.key])[:300]
|
|
19
|
+
return {"text": "", "urls": [], "model": engine.model, "usage": {}, "cost_usd": 0.0}, msg
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def run(q: dict, engines: dict[str, Bound], *, runs: int = 3, out: Path,
|
|
23
|
+
on_result: Callable[[dict], None] | None = None, sleep: float = 1.0,
|
|
24
|
+
resolver: Resolver | None = None, label: str = "baseline") -> list[dict]:
|
|
25
|
+
"""Every question x engine x run. Each record is written and flushed as it lands, so
|
|
26
|
+
an interrupted run keeps what it paid for."""
|
|
27
|
+
resolver = resolver or Resolver(out.parent / "redirect_cache.json")
|
|
28
|
+
records = []
|
|
29
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
30
|
+
with out.open("a", encoding="utf-8") as f:
|
|
31
|
+
for group, qs in q["groups"].items():
|
|
32
|
+
for question in qs:
|
|
33
|
+
for name, eng in engines.items():
|
|
34
|
+
for i in range(runs):
|
|
35
|
+
t0 = time.time()
|
|
36
|
+
ans, err = ask_once(eng, question)
|
|
37
|
+
resolver.warm(ans.get("urls", []))
|
|
38
|
+
ans["urls"] = [resolver(u) for u in ans.get("urls", [])]
|
|
39
|
+
rec = {
|
|
40
|
+
"ts": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
|
41
|
+
"label": label,
|
|
42
|
+
"group": group,
|
|
43
|
+
"question": question,
|
|
44
|
+
"engine": name,
|
|
45
|
+
"model": ans.get("model") or eng.model,
|
|
46
|
+
"run": i + 1,
|
|
47
|
+
"seconds": round(time.time() - t0, 1),
|
|
48
|
+
"error": err,
|
|
49
|
+
"answer": ans.get("text", ""),
|
|
50
|
+
"urls": ans.get("urls", []),
|
|
51
|
+
"usage": ans.get("usage", {}),
|
|
52
|
+
"cost_usd": ans.get("cost_usd"),
|
|
53
|
+
**score(ans, q["ours"], q["brand_terms"], stale_markers=q["stale_markers"]),
|
|
54
|
+
}
|
|
55
|
+
f.write(json.dumps(rec, ensure_ascii=False) + "\n")
|
|
56
|
+
f.flush()
|
|
57
|
+
records.append(rec)
|
|
58
|
+
if on_result:
|
|
59
|
+
on_result(rec)
|
|
60
|
+
if sleep:
|
|
61
|
+
time.sleep(sleep)
|
|
62
|
+
resolver.save()
|
|
63
|
+
return records
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def read_jsonl(path: Path) -> list[dict]:
|
|
67
|
+
return [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines() if line.strip()]
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def rescore(records: list[dict], q: dict) -> list[dict]:
|
|
71
|
+
"""Score stored answers again, for example after changing `ours` or `stale_markers`.
|
|
72
|
+
Spends nothing."""
|
|
73
|
+
for r in records:
|
|
74
|
+
r.update(score({"text": r.get("answer", ""), "urls": r.get("urls", [])},
|
|
75
|
+
q["ours"], q["brand_terms"], stale_markers=q["stale_markers"]))
|
|
76
|
+
return records
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""The citation rule and the rest of the scoring.
|
|
2
|
+
|
|
3
|
+
An answer CITES you when one of the URLs it gives, in the text or in its source
|
|
4
|
+
list, is on one of your properties (see `is_ours`). It MENTIONS you when one of
|
|
5
|
+
your brand terms appears in the text. Only aggregates over repeated runs mean
|
|
6
|
+
anything: the same question asked twice can come back with different sources.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
import re
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from urllib.parse import urlsplit
|
|
15
|
+
|
|
16
|
+
import requests
|
|
17
|
+
|
|
18
|
+
URL_RX = re.compile(r"https?://[^\s\)\]\"'>]+")
|
|
19
|
+
# Only matches within this many characters of a brand term count as being about you.
|
|
20
|
+
PROXIMITY = 200
|
|
21
|
+
|
|
22
|
+
# Gemini's grounding sources are proxy URLs on this host that redirect to the real page.
|
|
23
|
+
# Scored as they are, every Gemini citation looks like a citation of Google.
|
|
24
|
+
REDIRECT_HOSTS = ("vertexaisearch.cloud.google.com",)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class Resolver:
|
|
28
|
+
"""Follows citation-proxy redirects to the real URL, with a JSON cache on disk."""
|
|
29
|
+
|
|
30
|
+
def __init__(self, cache_file: Path | None = None, timeout: int = 20) -> None:
|
|
31
|
+
self.cache_file = cache_file
|
|
32
|
+
self.timeout = timeout
|
|
33
|
+
self.cache: dict[str, str] = {}
|
|
34
|
+
if cache_file and cache_file.exists():
|
|
35
|
+
try:
|
|
36
|
+
self.cache = json.loads(cache_file.read_text())
|
|
37
|
+
except (OSError, ValueError):
|
|
38
|
+
self.cache = {}
|
|
39
|
+
|
|
40
|
+
def __call__(self, url: str) -> str:
|
|
41
|
+
if not any(h in url for h in REDIRECT_HOSTS):
|
|
42
|
+
return url
|
|
43
|
+
if url not in self.cache:
|
|
44
|
+
try:
|
|
45
|
+
self.cache[url] = requests.head(url, allow_redirects=True, timeout=self.timeout).url or url
|
|
46
|
+
except requests.RequestException:
|
|
47
|
+
return url # not cached, so a later run tries again
|
|
48
|
+
return self.cache[url]
|
|
49
|
+
|
|
50
|
+
def warm(self, urls: list[str], workers: int = 16) -> None:
|
|
51
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
52
|
+
|
|
53
|
+
todo = [u for u in dict.fromkeys(urls) if any(h in u for h in REDIRECT_HOSTS) and u not in self.cache]
|
|
54
|
+
if todo:
|
|
55
|
+
with ThreadPoolExecutor(max_workers=workers) as ex:
|
|
56
|
+
list(ex.map(self, todo))
|
|
57
|
+
|
|
58
|
+
def save(self) -> None:
|
|
59
|
+
if self.cache_file:
|
|
60
|
+
self.cache_file.parent.mkdir(parents=True, exist_ok=True)
|
|
61
|
+
self.cache_file.write_text(json.dumps(self.cache))
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def domain(url: str) -> str:
|
|
65
|
+
host = (urlsplit(url).hostname or "").lower()
|
|
66
|
+
return host.removeprefix("www.")
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def is_ours(url: str, ours: list[str]) -> bool:
|
|
70
|
+
"""Is this URL one of your properties?
|
|
71
|
+
|
|
72
|
+
A bare domain (`example.com`) matches that host and its subdomains. A
|
|
73
|
+
path-qualified entry (`github.com/example`) matches only up to a path
|
|
74
|
+
boundary, so `github.com/example/repo` is yours and `github.com/example-fan`
|
|
75
|
+
is not. Reducing entries to their host would make every GitHub citation yours.
|
|
76
|
+
"""
|
|
77
|
+
host = domain(url)
|
|
78
|
+
rest = url.split("://", 1)[-1].lower()
|
|
79
|
+
rest = rest.removeprefix("www.")
|
|
80
|
+
for o in ours:
|
|
81
|
+
o = o.lower().strip().rstrip("/")
|
|
82
|
+
if o.startswith(("http://", "https://")):
|
|
83
|
+
o = o.split("://", 1)[1]
|
|
84
|
+
o = o.removeprefix("www.")
|
|
85
|
+
if "/" in o:
|
|
86
|
+
if rest.startswith(o) and (len(rest) == len(o) or rest[len(o)] in "/?#"):
|
|
87
|
+
return True
|
|
88
|
+
elif host == o or host.endswith("." + o):
|
|
89
|
+
return True
|
|
90
|
+
return False
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _near(text: str, terms: list[str], needles: list[str]) -> list[str]:
|
|
94
|
+
"""Needles that appear within PROXIMITY characters of any term."""
|
|
95
|
+
low = text.lower()
|
|
96
|
+
spans = [m.start() for t in terms if t for m in re.finditer(re.escape(t.lower()), low)]
|
|
97
|
+
hits = []
|
|
98
|
+
for n in needles:
|
|
99
|
+
for m in re.finditer(re.escape(n.lower()), low):
|
|
100
|
+
if any(abs(m.start() - s) <= PROXIMITY for s in spans):
|
|
101
|
+
hits.append(n)
|
|
102
|
+
break
|
|
103
|
+
return hits
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def score(answer: dict, ours: list[str], brand_terms: list[str], *,
|
|
107
|
+
stale_markers: list[str] | None = None, resolve=None) -> dict:
|
|
108
|
+
"""Score one answer. `resolve` maps a URL to its real destination (see Resolver)."""
|
|
109
|
+
resolve = resolve or (lambda u: u)
|
|
110
|
+
text = answer.get("text", "") or ""
|
|
111
|
+
cited = list(dict.fromkeys((answer.get("urls") or []) + URL_RX.findall(text)))
|
|
112
|
+
cited = [resolve(u.rstrip(".,;")) for u in cited]
|
|
113
|
+
domains = list(dict.fromkeys(d for d in (domain(u) for u in cited) if d))
|
|
114
|
+
our_urls = [u for u in cited if is_ours(u, ours)]
|
|
115
|
+
# Position: rank of your first source among the distinct sources cited. A source is
|
|
116
|
+
# a domain, except that your URLs on a shared host (github.com/you) count as a
|
|
117
|
+
# source of their own, so someone else's repo cited first still ranks ahead of you.
|
|
118
|
+
sources = list(dict.fromkeys(
|
|
119
|
+
("ours:" if is_ours(u, ours) else "") + domain(u) for u in cited if domain(u)))
|
|
120
|
+
position = next((i + 1 for i, s in enumerate(sources) if s.startswith("ours:")), None)
|
|
121
|
+
mentioned = any(t and t.lower() in text.lower() for t in brand_terms)
|
|
122
|
+
stale = _near(text, brand_terms, stale_markers or []) if mentioned else []
|
|
123
|
+
return {
|
|
124
|
+
"mentioned": mentioned,
|
|
125
|
+
"cited": bool(our_urls),
|
|
126
|
+
"position": position,
|
|
127
|
+
"our_urls": our_urls[:5],
|
|
128
|
+
"domains_cited": domains,
|
|
129
|
+
"n_domains": len(domains),
|
|
130
|
+
"stale": stale,
|
|
131
|
+
}
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
"""Aggregates, share of voice, the Markdown report, and the noise check."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections import defaultdict
|
|
6
|
+
|
|
7
|
+
from .scoring import is_ours
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _bucket() -> dict:
|
|
11
|
+
return {"runs": 0, "errors": 0, "mentioned": 0, "cited": 0, "stale": 0, "positions": [], "cost": 0.0}
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def summarise(records: list[dict]) -> dict:
|
|
15
|
+
"""Rates are over answered runs: a failed call is counted as an error, never as a miss."""
|
|
16
|
+
slices: dict[str, dict] = defaultdict(_bucket)
|
|
17
|
+
for r in records:
|
|
18
|
+
for key in ("all", f"engine:{r['engine']}", f"group:{r['group']}", f"{r['group']} x {r['engine']}"):
|
|
19
|
+
b = slices[key]
|
|
20
|
+
b["runs"] += 1
|
|
21
|
+
b["cost"] += r.get("cost_usd") or 0
|
|
22
|
+
if r.get("error"):
|
|
23
|
+
b["errors"] += 1
|
|
24
|
+
continue
|
|
25
|
+
b["mentioned"] += bool(r.get("mentioned"))
|
|
26
|
+
b["cited"] += bool(r.get("cited"))
|
|
27
|
+
b["stale"] += bool(r.get("stale"))
|
|
28
|
+
if r.get("position"):
|
|
29
|
+
b["positions"].append(r["position"])
|
|
30
|
+
out = {}
|
|
31
|
+
for key, b in slices.items():
|
|
32
|
+
n = b["runs"] - b["errors"]
|
|
33
|
+
pos = b.pop("positions")
|
|
34
|
+
out[key] = {
|
|
35
|
+
**b,
|
|
36
|
+
"answered": n,
|
|
37
|
+
"mention_rate": round(b["mentioned"] / n, 4) if n else None,
|
|
38
|
+
"citation_rate": round(b["cited"] / n, 4) if n else None,
|
|
39
|
+
"avg_position": round(sum(pos) / len(pos), 2) if pos else None,
|
|
40
|
+
"cost": round(b["cost"], 4),
|
|
41
|
+
}
|
|
42
|
+
return out
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def share_of_voice(records: list[dict], *, groups: list[str] | None = None) -> list[tuple[str, int]]:
|
|
46
|
+
"""Domains by the number of answers citing them at least once."""
|
|
47
|
+
counts: dict[str, int] = defaultdict(int)
|
|
48
|
+
for r in records:
|
|
49
|
+
if r.get("error") or (groups and r["group"] not in groups):
|
|
50
|
+
continue
|
|
51
|
+
for d in set(r.get("domains_cited") or []):
|
|
52
|
+
counts[d] += 1
|
|
53
|
+
return sorted(counts.items(), key=lambda kv: (-kv[1], kv[0]))
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _label(i: int) -> str:
|
|
57
|
+
s = ""
|
|
58
|
+
i += 1
|
|
59
|
+
while i:
|
|
60
|
+
i, rem = divmod(i - 1, 26)
|
|
61
|
+
s = chr(65 + rem) + s
|
|
62
|
+
return s
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _pct(x: float | None) -> str:
|
|
66
|
+
return "n/a" if x is None else f"{x:.0%}"
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def render(records: list[dict], ours: list[str], *, anonymise: bool = False, top: int = 15) -> str:
|
|
70
|
+
"""A Markdown report. With `anonymise`, every domain that is not yours becomes
|
|
71
|
+
"Source A", "Source B" and so on, which is what you want before sharing it."""
|
|
72
|
+
s = summarise(records)
|
|
73
|
+
questions = {r["question"] for r in records}
|
|
74
|
+
lines = [
|
|
75
|
+
f"# Answer engine benchmark: {len(records)} answers, {len(questions)} questions",
|
|
76
|
+
"",
|
|
77
|
+
"Cited = the answer links to one of your domains. Rates leave out failed calls.",
|
|
78
|
+
"",
|
|
79
|
+
"| slice | answers | errors | mentioned | cited | avg position when cited | cost USD |",
|
|
80
|
+
"|---|---|---|---|---|---|---|",
|
|
81
|
+
]
|
|
82
|
+
order = ["all"] + sorted(k for k in s if k.startswith("engine:")) + sorted(k for k in s if k.startswith("group:"))
|
|
83
|
+
for k in order:
|
|
84
|
+
b = s[k]
|
|
85
|
+
lines.append(f"| {k} | {b['answered']} | {b['errors']} | {_pct(b['mention_rate'])} | "
|
|
86
|
+
f"{b['cited']} of {b['answered']} ({_pct(b['citation_rate'])}) | "
|
|
87
|
+
f"{b['avg_position'] or '-'} | {b['cost']:.4f} |")
|
|
88
|
+
|
|
89
|
+
engines = sorted({r["engine"] for r in records})
|
|
90
|
+
groups = list(dict.fromkeys(r["group"] for r in records))
|
|
91
|
+
lines += ["", "## Cited, by group and engine", "", "| group | " + " | ".join(engines) + " |",
|
|
92
|
+
"|---|" + "---|" * len(engines)]
|
|
93
|
+
for g in groups:
|
|
94
|
+
cells = []
|
|
95
|
+
for e in engines:
|
|
96
|
+
b = s.get(f"{g} x {e}")
|
|
97
|
+
cells.append(f"{b['cited']} of {b['answered']}" if b else "-")
|
|
98
|
+
lines.append(f"| {g} | " + " | ".join(cells) + " |")
|
|
99
|
+
|
|
100
|
+
sov = share_of_voice(records)
|
|
101
|
+
names: dict[str, str] = {}
|
|
102
|
+
lines += ["", "## Sources cited most", "", "| source | answers citing it |", "|---|---|"]
|
|
103
|
+
for i, (d, n) in enumerate(sov[:top]):
|
|
104
|
+
mine = is_ours("https://" + d, ours)
|
|
105
|
+
if mine:
|
|
106
|
+
label = f"**{d}** (yours)"
|
|
107
|
+
elif anonymise:
|
|
108
|
+
label = names.setdefault(d, f"Source {_label(len(names))}")
|
|
109
|
+
else:
|
|
110
|
+
label = d
|
|
111
|
+
lines.append(f"| {label} | {n} |")
|
|
112
|
+
|
|
113
|
+
never = sorted(questions - {r["question"] for r in records if r.get("cited")})
|
|
114
|
+
lines += ["", "## Questions where you were never cited", ""] + [f"- {x}" for x in never or ["none"]]
|
|
115
|
+
stale = [(r["engine"], r["question"], r["stale"]) for r in records if r.get("stale")]
|
|
116
|
+
lines += ["", "## Answers with a stale or wrong marker near your name", ""]
|
|
117
|
+
lines += [f"- {e}: {q} ({', '.join(m)})" for e, q, m in stale[:30]] or ["- none"]
|
|
118
|
+
return "\n".join(lines) + "\n"
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
# ------------------------------------------------------------------------ noise check ----
|
|
122
|
+
def pick_questions(q: dict, n: int) -> list[tuple[str, str]]:
|
|
123
|
+
"""n questions spread across groups, round robin, in file order. Deterministic, so a
|
|
124
|
+
re-run of the noise check asks the same ones."""
|
|
125
|
+
pools = {g: list(qs) for g, qs in q["groups"].items()}
|
|
126
|
+
picked: list[tuple[str, str]] = []
|
|
127
|
+
while len(picked) < n and any(pools.values()):
|
|
128
|
+
for g in list(pools):
|
|
129
|
+
if pools[g] and len(picked) < n:
|
|
130
|
+
picked.append((g, pools[g].pop(0)))
|
|
131
|
+
return picked
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def compare(baseline: list[dict], repeat: list[dict], *, tolerance: int = 1) -> list[dict]:
|
|
135
|
+
"""Per question and engine: citations in the repeat against the baseline.
|
|
136
|
+
|
|
137
|
+
When the two have a different number of answered runs, the baseline count is scaled
|
|
138
|
+
to the repeat's run count. A pair is `within` when the difference is at most
|
|
139
|
+
`tolerance` citations.
|
|
140
|
+
"""
|
|
141
|
+
def tally(recs: list[dict]) -> dict[tuple[str, str], tuple[int, int]]:
|
|
142
|
+
t: dict[tuple[str, str], list[int]] = defaultdict(lambda: [0, 0])
|
|
143
|
+
for r in recs:
|
|
144
|
+
if r.get("error"):
|
|
145
|
+
continue
|
|
146
|
+
k = (r["question"], r["engine"])
|
|
147
|
+
t[k][0] += bool(r.get("cited"))
|
|
148
|
+
t[k][1] += 1
|
|
149
|
+
return {k: (v[0], v[1]) for k, v in t.items()}
|
|
150
|
+
|
|
151
|
+
base, rep = tally(baseline), tally(repeat)
|
|
152
|
+
rows = []
|
|
153
|
+
for k in sorted(rep):
|
|
154
|
+
if k not in base:
|
|
155
|
+
continue
|
|
156
|
+
(bc, bn), (rc, rn) = base[k], rep[k]
|
|
157
|
+
expected = bc * rn / bn if bn else 0
|
|
158
|
+
delta = rc - expected
|
|
159
|
+
rows.append({"question": k[0], "engine": k[1], "baseline": f"{bc} of {bn}", "repeat": f"{rc} of {rn}",
|
|
160
|
+
"delta": round(delta, 2), "within": abs(delta) <= tolerance})
|
|
161
|
+
return rows
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def render_compare(rows: list[dict], tolerance: int) -> str:
|
|
165
|
+
ok = sum(r["within"] for r in rows)
|
|
166
|
+
lines = [f"# Noise check: {ok} of {len(rows)} question-engine pairs within +/-{tolerance} citation", "",
|
|
167
|
+
"| question | engine | baseline | repeat | delta | within |", "|---|---|---|---|---|---|"]
|
|
168
|
+
lines += [f"| {r['question']} | {r['engine']} | {r['baseline']} | {r['repeat']} | {r['delta']:+g} | "
|
|
169
|
+
f"{'yes' if r['within'] else 'NO'} |" for r in rows]
|
|
170
|
+
return "\n".join(lines) + "\n"
|
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: answer-engine-benchmark
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Ask ChatGPT, Gemini, Perplexity and Claude your buyers' questions, several times each, and count who they cite.
|
|
5
|
+
Author: Synapse Research Ltd
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://synapsereality.io/open-source/answer-engine-benchmark/
|
|
8
|
+
Project-URL: Documentation, https://synapsereality.io/open-source/answer-engine-benchmark/
|
|
9
|
+
Project-URL: Repository, https://github.com/synapsereality/answer-engine-benchmark
|
|
10
|
+
Project-URL: Issues, https://github.com/synapsereality/answer-engine-benchmark/issues
|
|
11
|
+
Keywords: aeo,geo,answer-engine-optimization,llm,citations,benchmark,seo
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Internet :: WWW/HTTP :: Indexing/Search
|
|
15
|
+
Requires-Python: >=3.10
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Requires-Dist: requests>=2.28
|
|
19
|
+
Requires-Dist: PyYAML>=6
|
|
20
|
+
Provides-Extra: test
|
|
21
|
+
Requires-Dist: pytest>=8; extra == "test"
|
|
22
|
+
Dynamic: license-file
|
|
23
|
+
|
|
24
|
+
# answer-engine-benchmark
|
|
25
|
+
|
|
26
|
+
Asks ChatGPT, Gemini, Perplexity and Claude the questions your buyers ask, with
|
|
27
|
+
web search on, several times each. Then it counts how often each answer cites
|
|
28
|
+
your site, what it cites instead, and what it says about you. Every answer is
|
|
29
|
+
kept in a JSONL file, so you can check any number in the report against the
|
|
30
|
+
text behind it.
|
|
31
|
+
|
|
32
|
+
Docs: https://synapsereality.io/open-source/answer-engine-benchmark/
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
pip install answer-engine-benchmark
|
|
36
|
+
export OPENAI_API_KEY=... GEMINI_API_KEY=... PERPLEXITY_API_KEY=... ANTHROPIC_API_KEY=...
|
|
37
|
+
aeb run questions/template.yaml --out runs/first \
|
|
38
|
+
--set brand="Acme Analytics" --set domain=acme.example \
|
|
39
|
+
--set category="invoice software" --set audience="small accounting firms"
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## Why repeat every question
|
|
43
|
+
|
|
44
|
+
The same question can come back with different sources a minute later. One
|
|
45
|
+
answer is one sample. The default is 3 runs per question per engine, and every
|
|
46
|
+
rate in the report is over those runs. `aeb noise` re-asks a sample of questions
|
|
47
|
+
later and tells you whether a change you see is bigger than the noise.
|
|
48
|
+
|
|
49
|
+
## The question set
|
|
50
|
+
|
|
51
|
+
`questions/template.yaml` holds 24 questions in four groups:
|
|
52
|
+
|
|
53
|
+
| group | the buyer is | your name in the question |
|
|
54
|
+
|---|---|---|
|
|
55
|
+
| discovery | looking for a provider | no |
|
|
56
|
+
| comparison | learning how to choose | no |
|
|
57
|
+
| brand | asking about you | yes |
|
|
58
|
+
| trust | checking you are safe to buy from | yes |
|
|
59
|
+
|
|
60
|
+
Discovery and comparison tell you whether engines find you when nobody asked
|
|
61
|
+
for you. Brand and trust tell you what they say when somebody does.
|
|
62
|
+
|
|
63
|
+
Fill the four `vars` in the file, or pass them with `--set`. A run won't start
|
|
64
|
+
while any of them still holds its example value. Add your own questions under
|
|
65
|
+
any group, or new groups. Write them the way a buyer types, and never make a
|
|
66
|
+
competitor the subject of a question.
|
|
67
|
+
|
|
68
|
+
The file also sets:
|
|
69
|
+
|
|
70
|
+
- `ours`: domains that count as your citation. `acme.example` covers its
|
|
71
|
+
subdomains. `github.com/acme` covers only paths under it, so a citation of
|
|
72
|
+
someone else's GitHub repo is not yours.
|
|
73
|
+
- `brand_terms`: names that count as a mention.
|
|
74
|
+
- `stale_markers`: phrases that describe you wrongly or out of date, such as an
|
|
75
|
+
old product or a wrong founding year. One is flagged only within 200
|
|
76
|
+
characters of a brand term, so another company "founded in 2015" doesn't count
|
|
77
|
+
against you.
|
|
78
|
+
|
|
79
|
+
## How an answer is scored
|
|
80
|
+
|
|
81
|
+
**Cited** means one of the answer's URLs is yours. The URLs are the sources the
|
|
82
|
+
engine attached plus any link in the text. **Mentioned** means a brand term
|
|
83
|
+
appears in the text. An answer can mention you without citing you, and that
|
|
84
|
+
difference is worth watching. **Position** is the rank of your first source
|
|
85
|
+
among the distinct sites the answer cited.
|
|
86
|
+
|
|
87
|
+
Failed calls are recorded as errors and left out of every rate. A rate limit
|
|
88
|
+
never shows up as "not cited".
|
|
89
|
+
|
|
90
|
+
Gemini returns its sources as Google redirect links. They are resolved to the
|
|
91
|
+
real pages before scoring, with a cache next to the results. Otherwise every
|
|
92
|
+
Gemini answer would look like it cited Google.
|
|
93
|
+
|
|
94
|
+
## Engines
|
|
95
|
+
|
|
96
|
+
| engine | key | default model | how it searches |
|
|
97
|
+
|---|---|---|---|
|
|
98
|
+
| `openai` | `OPENAI_API_KEY` | `gpt-5-mini` | Responses API `web_search` tool |
|
|
99
|
+
| `gemini` | `GEMINI_API_KEY` | `gemini-3.5-flash` | Google Search grounding |
|
|
100
|
+
| `perplexity` | `PERPLEXITY_API_KEY` | `perplexity/sonar` | Responses API `web_search` tool |
|
|
101
|
+
| `claude` | `ANTHROPIC_API_KEY` | `claude-sonnet-5` | Messages API web search tool |
|
|
102
|
+
|
|
103
|
+
Keys are read from the environment only. They are sent in request headers and
|
|
104
|
+
removed from any error message before it is written, so they never reach the
|
|
105
|
+
results file. An engine with no key is skipped.
|
|
106
|
+
|
|
107
|
+
Change a model with `--model claude=claude-opus-5`. Pick the models your
|
|
108
|
+
buyers actually use in the apps, or say in the report which ones you used.
|
|
109
|
+
|
|
110
|
+
Perplexity's API doesn't search unless you ask it to. Without the search tool
|
|
111
|
+
it still answers, fluently, with no citations. This adapter always turns search
|
|
112
|
+
on.
|
|
113
|
+
|
|
114
|
+
## Commands
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
aeb check QUESTIONS [--set ...] # validate, show which keys are set and how many calls a run needs
|
|
118
|
+
aeb run QUESTIONS --out DIR [--set ...] # ask everything, write DIR/answers.jsonl and DIR/report.md
|
|
119
|
+
aeb run QUESTIONS --out DIR --dry-run # first question of each group, once: a cheap smoke test
|
|
120
|
+
aeb report DIR/answers.jsonl [QUESTIONS] # rebuild the report, re-scoring if you changed the question file
|
|
121
|
+
aeb noise QUESTIONS --baseline DIR/answers.jsonl --out DIR2 --sample 5
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
`--anonymise` on `run` and `report` replaces every domain that isn't yours with
|
|
125
|
+
"Source A", "Source B" and so on. Use it before you share a report outside your
|
|
126
|
+
company. `--max-calls` (default 500) stops a run that would make more calls
|
|
127
|
+
than you expected.
|
|
128
|
+
|
|
129
|
+
`answers.jsonl` gets one line per answer, appended as it arrives, so a stopped
|
|
130
|
+
run keeps what it already paid for. Each line holds the question, engine, model,
|
|
131
|
+
run number, the full answer text, every URL, token usage, cost, and the score
|
|
132
|
+
fields.
|
|
133
|
+
|
|
134
|
+
## What it costs
|
|
135
|
+
|
|
136
|
+
A full run of the template is 24 questions x 4 engines x 3 runs = 288 calls.
|
|
137
|
+
Our own run of 360 answers cost $1.39 in API fees without Claude (details in
|
|
138
|
+
`examples/`). Adding Claude costs more, because web search on the Claude API is
|
|
139
|
+
billed per search on top of tokens. Each
|
|
140
|
+
row carries its cost. Perplexity reports the real cost. The others are
|
|
141
|
+
estimated from list prices in `engines.py`, so check them against your bills.
|
|
142
|
+
|
|
143
|
+
## Measure as a stranger
|
|
144
|
+
|
|
145
|
+
Run the benchmark from an account and machine that has never been told who you
|
|
146
|
+
are. We learned this from a run through a coding assistant's CLI instead of an
|
|
147
|
+
API. The CLI passed the operator's own git identity to the model, and many brand
|
|
148
|
+
answers then told the reader the company was probably their own. The API
|
|
149
|
+
adapters here send only the question and a one-line instruction to cite sources.
|
|
150
|
+
|
|
151
|
+
## Example
|
|
152
|
+
|
|
153
|
+
`examples/synapse-launch-week-2026-09.md` is a real run on one company's own
|
|
154
|
+
site, with the published figures only. 105 of 156 answers that named the
|
|
155
|
+
company cited its site. Of the 204 that didn't name it, 1 did. It is an example
|
|
156
|
+
of the output, not part of the question set.
|
|
157
|
+
|
|
158
|
+
## Install from source
|
|
159
|
+
|
|
160
|
+
```bash
|
|
161
|
+
git clone https://github.com/synapsereality/answer-engine-benchmark
|
|
162
|
+
cd answer-engine-benchmark
|
|
163
|
+
pip install .
|
|
164
|
+
aeb --help
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
Python 3.10 or later. Needs `requests` and `PyYAML`.
|
|
168
|
+
|
|
169
|
+
## Tests
|
|
170
|
+
|
|
171
|
+
```bash
|
|
172
|
+
pip install -e ".[test]"
|
|
173
|
+
pytest
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
33 tests. They mock every API, so they spend nothing and need no keys.
|
|
177
|
+
|
|
178
|
+
## Licence
|
|
179
|
+
|
|
180
|
+
MIT. Made by [Synapse](https://synapsereality.io/open-source/answer-engine-benchmark/).
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
answer_engine_benchmark/__init__.py,sha256=NSAqFwcQhCATH4ZFhWJZn8w3qxys2NYBtxFn0ZbCAm4,124
|
|
2
|
+
answer_engine_benchmark/__main__.py,sha256=k1ocEWawweo1qCJWNFAAvyxz3tcY13dzvCenHszij30,48
|
|
3
|
+
answer_engine_benchmark/cli.py,sha256=UAhDHi9PzFQwM7b7HzAwI2hzYsH5K1IWzwuIIDehH6U,8238
|
|
4
|
+
answer_engine_benchmark/engines.py,sha256=45NtCAcsl1UmFH6yQrLsSRCw4CQFzC804Ctk0zWNKCY,10272
|
|
5
|
+
answer_engine_benchmark/questions.py,sha256=8f6d7KGMflNS-_Zqz0x-H2vcoVH_fNMEV2FLXiazaLE,2740
|
|
6
|
+
answer_engine_benchmark/runner.py,sha256=R6df-hgpw600b6xpJyYessT3-pYwX8vGyJ5LkfsLJzc,3389
|
|
7
|
+
answer_engine_benchmark/scoring.py,sha256=Z-3iItdQCdh8v4zOIKVNuoL34N0RYvK7uZsIRcRx4IQ,5328
|
|
8
|
+
answer_engine_benchmark/summary.py,sha256=5S3BajlbW99gsihtYOZBlkrlYMjkv6xuLkmHgWJnefo,7157
|
|
9
|
+
answer_engine_benchmark-0.1.0.dist-info/licenses/LICENSE,sha256=zcvyWWK63eEsOTtAaaeqm_Ng-i9qAwO2I-7sHNcti6c,1077
|
|
10
|
+
answer_engine_benchmark-0.1.0.dist-info/METADATA,sha256=icY0EjsqCGwlCUNyhRf84RaYDx-Cg7hCqn_HEha0-2k,7613
|
|
11
|
+
answer_engine_benchmark-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
12
|
+
answer_engine_benchmark-0.1.0.dist-info/entry_points.txt,sha256=47Vg0cYT6JAvUlhxVtqFECcerh2vtsvS_rr3QIR0Kh4,57
|
|
13
|
+
answer_engine_benchmark-0.1.0.dist-info/top_level.txt,sha256=09-xkykfQfaUL3HnAexMMjl2eXLRULX2C-2KESjzCMY,24
|
|
14
|
+
answer_engine_benchmark-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Synapse Research Ltd
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
answer_engine_benchmark
|