answer-engine-benchmark 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- answer_engine_benchmark-0.1.0/LICENSE +21 -0
- answer_engine_benchmark-0.1.0/PKG-INFO +180 -0
- answer_engine_benchmark-0.1.0/README.md +157 -0
- answer_engine_benchmark-0.1.0/pyproject.toml +43 -0
- answer_engine_benchmark-0.1.0/setup.cfg +4 -0
- answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/__init__.py +3 -0
- answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/__main__.py +3 -0
- answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/cli.py +168 -0
- answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/engines.py +240 -0
- answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/questions.py +66 -0
- answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/runner.py +76 -0
- answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/scoring.py +131 -0
- answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/summary.py +170 -0
- answer_engine_benchmark-0.1.0/src/answer_engine_benchmark.egg-info/PKG-INFO +180 -0
- answer_engine_benchmark-0.1.0/src/answer_engine_benchmark.egg-info/SOURCES.txt +21 -0
- answer_engine_benchmark-0.1.0/src/answer_engine_benchmark.egg-info/dependency_links.txt +1 -0
- answer_engine_benchmark-0.1.0/src/answer_engine_benchmark.egg-info/entry_points.txt +2 -0
- answer_engine_benchmark-0.1.0/src/answer_engine_benchmark.egg-info/requires.txt +5 -0
- answer_engine_benchmark-0.1.0/src/answer_engine_benchmark.egg-info/top_level.txt +1 -0
- answer_engine_benchmark-0.1.0/tests/test_engines.py +112 -0
- answer_engine_benchmark-0.1.0/tests/test_questions.py +53 -0
- answer_engine_benchmark-0.1.0/tests/test_run_and_report.py +110 -0
- answer_engine_benchmark-0.1.0/tests/test_scoring.py +82 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Synapse Research Ltd
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: answer-engine-benchmark
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Ask ChatGPT, Gemini, Perplexity and Claude your buyers' questions, several times each, and count who they cite.
|
|
5
|
+
Author: Synapse Research Ltd
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://synapsereality.io/open-source/answer-engine-benchmark/
|
|
8
|
+
Project-URL: Documentation, https://synapsereality.io/open-source/answer-engine-benchmark/
|
|
9
|
+
Project-URL: Repository, https://github.com/synapsereality/answer-engine-benchmark
|
|
10
|
+
Project-URL: Issues, https://github.com/synapsereality/answer-engine-benchmark/issues
|
|
11
|
+
Keywords: aeo,geo,answer-engine-optimization,llm,citations,benchmark,seo
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Internet :: WWW/HTTP :: Indexing/Search
|
|
15
|
+
Requires-Python: >=3.10
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Requires-Dist: requests>=2.28
|
|
19
|
+
Requires-Dist: PyYAML>=6
|
|
20
|
+
Provides-Extra: test
|
|
21
|
+
Requires-Dist: pytest>=8; extra == "test"
|
|
22
|
+
Dynamic: license-file
|
|
23
|
+
|
|
24
|
+
# answer-engine-benchmark
|
|
25
|
+
|
|
26
|
+
Asks ChatGPT, Gemini, Perplexity and Claude the questions your buyers ask, with
|
|
27
|
+
web search on, several times each. Then it counts how often each answer cites
|
|
28
|
+
your site, what it cites instead, and what it says about you. Every answer is
|
|
29
|
+
kept in a JSONL file, so you can check any number in the report against the
|
|
30
|
+
text behind it.
|
|
31
|
+
|
|
32
|
+
Docs: https://synapsereality.io/open-source/answer-engine-benchmark/
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
pip install answer-engine-benchmark
|
|
36
|
+
export OPENAI_API_KEY=... GEMINI_API_KEY=... PERPLEXITY_API_KEY=... ANTHROPIC_API_KEY=...
|
|
37
|
+
aeb run questions/template.yaml --out runs/first \
|
|
38
|
+
--set brand="Acme Analytics" --set domain=acme.example \
|
|
39
|
+
--set category="invoice software" --set audience="small accounting firms"
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## Why repeat every question
|
|
43
|
+
|
|
44
|
+
The same question can come back with different sources a minute later. One
|
|
45
|
+
answer is one sample. The default is 3 runs per question per engine, and every
|
|
46
|
+
rate in the report is over those runs. `aeb noise` re-asks a sample of questions
|
|
47
|
+
later and tells you whether a change you see is bigger than the noise.
|
|
48
|
+
|
|
49
|
+
## The question set
|
|
50
|
+
|
|
51
|
+
`questions/template.yaml` holds 24 questions in four groups:
|
|
52
|
+
|
|
53
|
+
| group | the buyer is | your name in the question |
|
|
54
|
+
|---|---|---|
|
|
55
|
+
| discovery | looking for a provider | no |
|
|
56
|
+
| comparison | learning how to choose | no |
|
|
57
|
+
| brand | asking about you | yes |
|
|
58
|
+
| trust | checking you are safe to buy from | yes |
|
|
59
|
+
|
|
60
|
+
Discovery and comparison tell you whether engines find you when nobody asked
|
|
61
|
+
for you. Brand and trust tell you what they say when somebody does.
|
|
62
|
+
|
|
63
|
+
Fill the four `vars` in the file, or pass them with `--set`. A run won't start
|
|
64
|
+
while any of them still holds its example value. Add your own questions under
|
|
65
|
+
any group, or new groups. Write them the way a buyer types, and never make a
|
|
66
|
+
competitor the subject of a question.
|
|
67
|
+
|
|
68
|
+
The file also sets:
|
|
69
|
+
|
|
70
|
+
- `ours`: domains that count as your citation. `acme.example` covers its
|
|
71
|
+
subdomains. `github.com/acme` covers only paths under it, so a citation of
|
|
72
|
+
someone else's GitHub repo is not yours.
|
|
73
|
+
- `brand_terms`: names that count as a mention.
|
|
74
|
+
- `stale_markers`: phrases that describe you wrongly or out of date, such as an
|
|
75
|
+
old product or a wrong founding year. One is flagged only within 200
|
|
76
|
+
characters of a brand term, so another company "founded in 2015" doesn't count
|
|
77
|
+
against you.
|
|
78
|
+
|
|
79
|
+
## How an answer is scored
|
|
80
|
+
|
|
81
|
+
**Cited** means one of the answer's URLs is yours. The URLs are the sources the
|
|
82
|
+
engine attached plus any link in the text. **Mentioned** means a brand term
|
|
83
|
+
appears in the text. An answer can mention you without citing you, and that
|
|
84
|
+
difference is worth watching. **Position** is the rank of your first source
|
|
85
|
+
among the distinct sites the answer cited.
|
|
86
|
+
|
|
87
|
+
Failed calls are recorded as errors and left out of every rate. A rate limit
|
|
88
|
+
never shows up as "not cited".
|
|
89
|
+
|
|
90
|
+
Gemini returns its sources as Google redirect links. They are resolved to the
|
|
91
|
+
real pages before scoring, with a cache next to the results. Otherwise every
|
|
92
|
+
Gemini answer would look like it cited Google.
|
|
93
|
+
|
|
94
|
+
## Engines
|
|
95
|
+
|
|
96
|
+
| engine | key | default model | how it searches |
|
|
97
|
+
|---|---|---|---|
|
|
98
|
+
| `openai` | `OPENAI_API_KEY` | `gpt-5-mini` | Responses API `web_search` tool |
|
|
99
|
+
| `gemini` | `GEMINI_API_KEY` | `gemini-3.5-flash` | Google Search grounding |
|
|
100
|
+
| `perplexity` | `PERPLEXITY_API_KEY` | `perplexity/sonar` | Responses API `web_search` tool |
|
|
101
|
+
| `claude` | `ANTHROPIC_API_KEY` | `claude-sonnet-5` | Messages API web search tool |
|
|
102
|
+
|
|
103
|
+
Keys are read from the environment only. They are sent in request headers and
|
|
104
|
+
removed from any error message before it is written, so they never reach the
|
|
105
|
+
results file. An engine with no key is skipped.
|
|
106
|
+
|
|
107
|
+
Change a model with `--model claude=claude-opus-5`. Pick the models your
|
|
108
|
+
buyers actually use in the apps, or say in the report which ones you used.
|
|
109
|
+
|
|
110
|
+
Perplexity's API doesn't search unless you ask it to. Without the search tool
|
|
111
|
+
it still answers, fluently, with no citations. This adapter always turns search
|
|
112
|
+
on.
|
|
113
|
+
|
|
114
|
+
## Commands
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
aeb check QUESTIONS [--set ...] # validate, show which keys are set and how many calls a run needs
|
|
118
|
+
aeb run QUESTIONS --out DIR [--set ...] # ask everything, write DIR/answers.jsonl and DIR/report.md
|
|
119
|
+
aeb run QUESTIONS --out DIR --dry-run # first question of each group, once: a cheap smoke test
|
|
120
|
+
aeb report DIR/answers.jsonl [QUESTIONS] # rebuild the report, re-scoring if you changed the question file
|
|
121
|
+
aeb noise QUESTIONS --baseline DIR/answers.jsonl --out DIR2 --sample 5
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
`--anonymise` on `run` and `report` replaces every domain that isn't yours with
|
|
125
|
+
"Source A", "Source B" and so on. Use it before you share a report outside your
|
|
126
|
+
company. `--max-calls` (default 500) stops a run that would make more calls
|
|
127
|
+
than you expected.
|
|
128
|
+
|
|
129
|
+
`answers.jsonl` gets one line per answer, appended as it arrives, so a stopped
|
|
130
|
+
run keeps what it already paid for. Each line holds the question, engine, model,
|
|
131
|
+
run number, the full answer text, every URL, token usage, cost, and the score
|
|
132
|
+
fields.
|
|
133
|
+
|
|
134
|
+
## What it costs
|
|
135
|
+
|
|
136
|
+
A full run of the template is 24 questions x 4 engines x 3 runs = 288 calls.
|
|
137
|
+
Our own run of 360 answers cost $1.39 in API fees without Claude (details in
|
|
138
|
+
`examples/`). Adding Claude costs more, because web search on the Claude API is
|
|
139
|
+
billed per search on top of tokens. Each
|
|
140
|
+
row carries its cost. Perplexity reports the real cost. The others are
|
|
141
|
+
estimated from list prices in `engines.py`, so check them against your bills.
|
|
142
|
+
|
|
143
|
+
## Measure as a stranger
|
|
144
|
+
|
|
145
|
+
Run the benchmark from an account and machine that has never been told who you
|
|
146
|
+
are. We learned this from a run through a coding assistant's CLI instead of an
|
|
147
|
+
API. The CLI passed the operator's own git identity to the model, and many brand
|
|
148
|
+
answers then told the reader the company was probably their own. The API
|
|
149
|
+
adapters here send only the question and a one-line instruction to cite sources.
|
|
150
|
+
|
|
151
|
+
## Example
|
|
152
|
+
|
|
153
|
+
`examples/synapse-launch-week-2026-09.md` is a real run on one company's own
|
|
154
|
+
site, with the published figures only. 105 of 156 answers that named the
|
|
155
|
+
company cited its site. Of the 204 that didn't name it, 1 did. It is an example
|
|
156
|
+
of the output, not part of the question set.
|
|
157
|
+
|
|
158
|
+
## Install from source
|
|
159
|
+
|
|
160
|
+
```bash
|
|
161
|
+
git clone https://github.com/synapsereality/answer-engine-benchmark
|
|
162
|
+
cd answer-engine-benchmark
|
|
163
|
+
pip install .
|
|
164
|
+
aeb --help
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
Python 3.10 or later. Needs `requests` and `PyYAML`.
|
|
168
|
+
|
|
169
|
+
## Tests
|
|
170
|
+
|
|
171
|
+
```bash
|
|
172
|
+
pip install -e ".[test]"
|
|
173
|
+
pytest
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
33 tests. They mock every API, so they spend nothing and need no keys.
|
|
177
|
+
|
|
178
|
+
## Licence
|
|
179
|
+
|
|
180
|
+
MIT. Made by [Synapse](https://synapsereality.io/open-source/answer-engine-benchmark/).
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
# answer-engine-benchmark
|
|
2
|
+
|
|
3
|
+
Asks ChatGPT, Gemini, Perplexity and Claude the questions your buyers ask, with
|
|
4
|
+
web search on, several times each. Then it counts how often each answer cites
|
|
5
|
+
your site, what it cites instead, and what it says about you. Every answer is
|
|
6
|
+
kept in a JSONL file, so you can check any number in the report against the
|
|
7
|
+
text behind it.
|
|
8
|
+
|
|
9
|
+
Docs: https://synapsereality.io/open-source/answer-engine-benchmark/
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
pip install answer-engine-benchmark
|
|
13
|
+
export OPENAI_API_KEY=... GEMINI_API_KEY=... PERPLEXITY_API_KEY=... ANTHROPIC_API_KEY=...
|
|
14
|
+
aeb run questions/template.yaml --out runs/first \
|
|
15
|
+
--set brand="Acme Analytics" --set domain=acme.example \
|
|
16
|
+
--set category="invoice software" --set audience="small accounting firms"
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
## Why repeat every question
|
|
20
|
+
|
|
21
|
+
The same question can come back with different sources a minute later. One
|
|
22
|
+
answer is one sample. The default is 3 runs per question per engine, and every
|
|
23
|
+
rate in the report is over those runs. `aeb noise` re-asks a sample of questions
|
|
24
|
+
later and tells you whether a change you see is bigger than the noise.
|
|
25
|
+
|
|
26
|
+
## The question set
|
|
27
|
+
|
|
28
|
+
`questions/template.yaml` holds 24 questions in four groups:
|
|
29
|
+
|
|
30
|
+
| group | the buyer is | your name in the question |
|
|
31
|
+
|---|---|---|
|
|
32
|
+
| discovery | looking for a provider | no |
|
|
33
|
+
| comparison | learning how to choose | no |
|
|
34
|
+
| brand | asking about you | yes |
|
|
35
|
+
| trust | checking you are safe to buy from | yes |
|
|
36
|
+
|
|
37
|
+
Discovery and comparison tell you whether engines find you when nobody asked
|
|
38
|
+
for you. Brand and trust tell you what they say when somebody does.
|
|
39
|
+
|
|
40
|
+
Fill the four `vars` in the file, or pass them with `--set`. A run won't start
|
|
41
|
+
while any of them still holds its example value. Add your own questions under
|
|
42
|
+
any group, or new groups. Write them the way a buyer types, and never make a
|
|
43
|
+
competitor the subject of a question.
|
|
44
|
+
|
|
45
|
+
The file also sets:
|
|
46
|
+
|
|
47
|
+
- `ours`: domains that count as your citation. `acme.example` covers its
|
|
48
|
+
subdomains. `github.com/acme` covers only paths under it, so a citation of
|
|
49
|
+
someone else's GitHub repo is not yours.
|
|
50
|
+
- `brand_terms`: names that count as a mention.
|
|
51
|
+
- `stale_markers`: phrases that describe you wrongly or out of date, such as an
|
|
52
|
+
old product or a wrong founding year. One is flagged only within 200
|
|
53
|
+
characters of a brand term, so another company "founded in 2015" doesn't count
|
|
54
|
+
against you.
|
|
55
|
+
|
|
56
|
+
## How an answer is scored
|
|
57
|
+
|
|
58
|
+
**Cited** means one of the answer's URLs is yours. The URLs are the sources the
|
|
59
|
+
engine attached plus any link in the text. **Mentioned** means a brand term
|
|
60
|
+
appears in the text. An answer can mention you without citing you, and that
|
|
61
|
+
difference is worth watching. **Position** is the rank of your first source
|
|
62
|
+
among the distinct sites the answer cited.
|
|
63
|
+
|
|
64
|
+
Failed calls are recorded as errors and left out of every rate. A rate limit
|
|
65
|
+
never shows up as "not cited".
|
|
66
|
+
|
|
67
|
+
Gemini returns its sources as Google redirect links. They are resolved to the
|
|
68
|
+
real pages before scoring, with a cache next to the results. Otherwise every
|
|
69
|
+
Gemini answer would look like it cited Google.
|
|
70
|
+
|
|
71
|
+
## Engines
|
|
72
|
+
|
|
73
|
+
| engine | key | default model | how it searches |
|
|
74
|
+
|---|---|---|---|
|
|
75
|
+
| `openai` | `OPENAI_API_KEY` | `gpt-5-mini` | Responses API `web_search` tool |
|
|
76
|
+
| `gemini` | `GEMINI_API_KEY` | `gemini-3.5-flash` | Google Search grounding |
|
|
77
|
+
| `perplexity` | `PERPLEXITY_API_KEY` | `perplexity/sonar` | Responses API `web_search` tool |
|
|
78
|
+
| `claude` | `ANTHROPIC_API_KEY` | `claude-sonnet-5` | Messages API web search tool |
|
|
79
|
+
|
|
80
|
+
Keys are read from the environment only. They are sent in request headers and
|
|
81
|
+
removed from any error message before it is written, so they never reach the
|
|
82
|
+
results file. An engine with no key is skipped.
|
|
83
|
+
|
|
84
|
+
Change a model with `--model claude=claude-opus-5`. Pick the models your
|
|
85
|
+
buyers actually use in the apps, or say in the report which ones you used.
|
|
86
|
+
|
|
87
|
+
Perplexity's API doesn't search unless you ask it to. Without the search tool
|
|
88
|
+
it still answers, fluently, with no citations. This adapter always turns search
|
|
89
|
+
on.
|
|
90
|
+
|
|
91
|
+
## Commands
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
aeb check QUESTIONS [--set ...] # validate, show which keys are set and how many calls a run needs
|
|
95
|
+
aeb run QUESTIONS --out DIR [--set ...] # ask everything, write DIR/answers.jsonl and DIR/report.md
|
|
96
|
+
aeb run QUESTIONS --out DIR --dry-run # first question of each group, once: a cheap smoke test
|
|
97
|
+
aeb report DIR/answers.jsonl [QUESTIONS] # rebuild the report, re-scoring if you changed the question file
|
|
98
|
+
aeb noise QUESTIONS --baseline DIR/answers.jsonl --out DIR2 --sample 5
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
`--anonymise` on `run` and `report` replaces every domain that isn't yours with
|
|
102
|
+
"Source A", "Source B" and so on. Use it before you share a report outside your
|
|
103
|
+
company. `--max-calls` (default 500) stops a run that would make more calls
|
|
104
|
+
than you expected.
|
|
105
|
+
|
|
106
|
+
`answers.jsonl` gets one line per answer, appended as it arrives, so a stopped
|
|
107
|
+
run keeps what it already paid for. Each line holds the question, engine, model,
|
|
108
|
+
run number, the full answer text, every URL, token usage, cost, and the score
|
|
109
|
+
fields.
|
|
110
|
+
|
|
111
|
+
## What it costs
|
|
112
|
+
|
|
113
|
+
A full run of the template is 24 questions x 4 engines x 3 runs = 288 calls.
|
|
114
|
+
Our own run of 360 answers cost $1.39 in API fees without Claude (details in
|
|
115
|
+
`examples/`). Adding Claude costs more, because web search on the Claude API is
|
|
116
|
+
billed per search on top of tokens. Each
|
|
117
|
+
row carries its cost. Perplexity reports the real cost. The others are
|
|
118
|
+
estimated from list prices in `engines.py`, so check them against your bills.
|
|
119
|
+
|
|
120
|
+
## Measure as a stranger
|
|
121
|
+
|
|
122
|
+
Run the benchmark from an account and machine that has never been told who you
|
|
123
|
+
are. We learned this from a run through a coding assistant's CLI instead of an
|
|
124
|
+
API. The CLI passed the operator's own git identity to the model, and many brand
|
|
125
|
+
answers then told the reader the company was probably their own. The API
|
|
126
|
+
adapters here send only the question and a one-line instruction to cite sources.
|
|
127
|
+
|
|
128
|
+
## Example
|
|
129
|
+
|
|
130
|
+
`examples/synapse-launch-week-2026-09.md` is a real run on one company's own
|
|
131
|
+
site, with the published figures only. 105 of 156 answers that named the
|
|
132
|
+
company cited its site. Of the 204 that didn't name it, 1 did. It is an example
|
|
133
|
+
of the output, not part of the question set.
|
|
134
|
+
|
|
135
|
+
## Install from source
|
|
136
|
+
|
|
137
|
+
```bash
|
|
138
|
+
git clone https://github.com/synapsereality/answer-engine-benchmark
|
|
139
|
+
cd answer-engine-benchmark
|
|
140
|
+
pip install .
|
|
141
|
+
aeb --help
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
Python 3.10 or later. Needs `requests` and `PyYAML`.
|
|
145
|
+
|
|
146
|
+
## Tests
|
|
147
|
+
|
|
148
|
+
```bash
|
|
149
|
+
pip install -e ".[test]"
|
|
150
|
+
pytest
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
33 tests. They mock every API, so they spend nothing and need no keys.
|
|
154
|
+
|
|
155
|
+
## Licence
|
|
156
|
+
|
|
157
|
+
MIT. Made by [Synapse](https://synapsereality.io/open-source/answer-engine-benchmark/).
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=69"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "answer-engine-benchmark"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Ask ChatGPT, Gemini, Perplexity and Claude your buyers' questions, several times each, and count who they cite."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
requires-python = ">=3.10"
|
|
13
|
+
authors = [{ name = "Synapse Research Ltd" }]
|
|
14
|
+
keywords = ["aeo", "geo", "answer-engine-optimization", "llm", "citations", "benchmark", "seo"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Environment :: Console",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Topic :: Internet :: WWW/HTTP :: Indexing/Search",
|
|
19
|
+
]
|
|
20
|
+
dependencies = ["requests>=2.28", "PyYAML>=6"]
|
|
21
|
+
|
|
22
|
+
[project.optional-dependencies]
|
|
23
|
+
test = ["pytest>=8"]
|
|
24
|
+
|
|
25
|
+
[project.urls]
|
|
26
|
+
Homepage = "https://synapsereality.io/open-source/answer-engine-benchmark/"
|
|
27
|
+
Documentation = "https://synapsereality.io/open-source/answer-engine-benchmark/"
|
|
28
|
+
Repository = "https://github.com/synapsereality/answer-engine-benchmark"
|
|
29
|
+
Issues = "https://github.com/synapsereality/answer-engine-benchmark/issues"
|
|
30
|
+
|
|
31
|
+
[project.scripts]
|
|
32
|
+
aeb = "answer_engine_benchmark.cli:main"
|
|
33
|
+
|
|
34
|
+
[tool.setuptools.packages.find]
|
|
35
|
+
where = ["src"]
|
|
36
|
+
|
|
37
|
+
[tool.pytest.ini_options]
|
|
38
|
+
pythonpath = ["src"]
|
|
39
|
+
testpaths = ["tests"]
|
|
40
|
+
|
|
41
|
+
[tool.ruff]
|
|
42
|
+
line-length = 120
|
|
43
|
+
target-version = "py310"
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
"""aeb: ask AI answer engines your buyers' questions and count who they cite.
|
|
2
|
+
|
|
3
|
+
Exit codes: 0 done, 1 a run finished but every call failed, 2 bad input.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import argparse
|
|
9
|
+
import sys
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from . import __version__
|
|
13
|
+
from .engines import ENGINES, available
|
|
14
|
+
from .questions import QuestionError, count, load
|
|
15
|
+
from .runner import read_jsonl, rescore, run
|
|
16
|
+
from .summary import compare, pick_questions, render, render_compare
|
|
17
|
+
|
|
18
|
+
DOCS = "https://synapsereality.io/open-source/answer-engine-benchmark/"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _kv(raw: str) -> tuple[str, str]:
|
|
22
|
+
k, sep, v = raw.partition("=")
|
|
23
|
+
if not sep or not k.strip():
|
|
24
|
+
raise argparse.ArgumentTypeError(f"expected name=value, got {raw!r}")
|
|
25
|
+
return k.strip(), v.strip()
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _common(p: argparse.ArgumentParser) -> None:
|
|
29
|
+
p.add_argument("questions", help="question file (YAML)")
|
|
30
|
+
p.add_argument("--set", action="append", type=_kv, default=[], metavar="NAME=VALUE",
|
|
31
|
+
help="fill a {placeholder}, e.g. --set brand='Acme Ltd' (repeatable)")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _engine_flags(p: argparse.ArgumentParser) -> None:
|
|
35
|
+
p.add_argument("--engine", action="append", choices=sorted(ENGINES), help="only these engines (repeatable)")
|
|
36
|
+
p.add_argument("--model", action="append", type=_kv, default=[], metavar="ENGINE=MODEL",
|
|
37
|
+
help="override a model, e.g. --model claude=claude-opus-5")
|
|
38
|
+
p.add_argument("--runs", type=int, default=3, help="times to ask each question on each engine (default 3)")
|
|
39
|
+
p.add_argument("--max-calls", type=int, default=500,
|
|
40
|
+
help="refuse to start a run that needs more API calls than this (default 500)")
|
|
41
|
+
p.add_argument("--sleep", type=float, default=1.0, help="seconds between calls (default 1)")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
45
|
+
ap = argparse.ArgumentParser(prog="aeb", description=__doc__, epilog=f"Docs: {DOCS}",
|
|
46
|
+
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
47
|
+
ap.add_argument("--version", action="version", version=f"aeb {__version__}")
|
|
48
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
49
|
+
|
|
50
|
+
p = sub.add_parser("check", help="validate a question file and show what a run would cost in calls")
|
|
51
|
+
_common(p)
|
|
52
|
+
_engine_flags(p)
|
|
53
|
+
p.add_argument("--allow-unfilled", action="store_true", help="accept the template's example values")
|
|
54
|
+
|
|
55
|
+
p = sub.add_parser("run", help="ask every question on every engine and write answers.jsonl + report.md")
|
|
56
|
+
_common(p)
|
|
57
|
+
_engine_flags(p)
|
|
58
|
+
p.add_argument("--out", required=True, type=Path, help="output folder")
|
|
59
|
+
p.add_argument("--group", action="append", help="only these question groups (repeatable)")
|
|
60
|
+
p.add_argument("--dry-run", action="store_true", help="first question of each group, 1 run: a cheap smoke test")
|
|
61
|
+
p.add_argument("--anonymise", action="store_true", help="label other domains Source A, B, ... in report.md")
|
|
62
|
+
|
|
63
|
+
p = sub.add_parser("report", help="rebuild report.md from answers.jsonl (spends nothing)")
|
|
64
|
+
p.add_argument("answers", type=Path, help="answers.jsonl")
|
|
65
|
+
p.add_argument("questions", nargs="?", help="question file: re-score with its ours/brand_terms first")
|
|
66
|
+
p.add_argument("--set", action="append", type=_kv, default=[], metavar="NAME=VALUE")
|
|
67
|
+
p.add_argument("--anonymise", action="store_true")
|
|
68
|
+
p.add_argument("--out", type=Path, help="write here instead of stdout")
|
|
69
|
+
|
|
70
|
+
p = sub.add_parser("noise", help="re-ask a sample of questions and compare with the baseline")
|
|
71
|
+
_common(p)
|
|
72
|
+
_engine_flags(p)
|
|
73
|
+
p.add_argument("--baseline", required=True, type=Path, help="answers.jsonl from the first run")
|
|
74
|
+
p.add_argument("--out", required=True, type=Path, help="output folder for the repeat run")
|
|
75
|
+
p.add_argument("--sample", type=int, default=5, help="questions to re-ask, spread across groups (default 5)")
|
|
76
|
+
p.add_argument("--tolerance", type=int, default=1, help="allowed difference in citations (default 1)")
|
|
77
|
+
return ap
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _plan(q: dict, engines: dict, runs: int, max_calls: int) -> int | None:
|
|
81
|
+
calls = count(q) * len(engines) * runs
|
|
82
|
+
names = ", ".join(f"{n} ({e.model})" for n, e in engines.items()) or "none"
|
|
83
|
+
print(f"{count(q)} questions x {len(engines)} engines x {runs} runs = {calls} calls. Engines: {names}")
|
|
84
|
+
if calls > max_calls:
|
|
85
|
+
print(f"aeb: {calls} calls is over --max-calls {max_calls}. Raise it if you mean it.", file=sys.stderr)
|
|
86
|
+
return None
|
|
87
|
+
return calls
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _progress(rec: dict) -> None:
|
|
91
|
+
flag = "ERR " if rec["error"] else "CITE" if rec["cited"] else "ment" if rec["mentioned"] else " - "
|
|
92
|
+
cost = rec["cost_usd"] or 0
|
|
93
|
+
print(f"{flag} {rec['engine']:<10} {rec['group']:<11} pos={rec['position'] or '-':<3} ${cost:.4f} "
|
|
94
|
+
f"{rec['question'][:70]} {rec['error'][:80]}", flush=True)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def main(argv: list[str] | None = None) -> int:
|
|
98
|
+
args = build_parser().parse_args(argv)
|
|
99
|
+
try:
|
|
100
|
+
if args.cmd == "report":
|
|
101
|
+
recs = read_jsonl(args.answers)
|
|
102
|
+
ours = []
|
|
103
|
+
if args.questions:
|
|
104
|
+
q = load(args.questions, dict(args.set), allow_unfilled=True)
|
|
105
|
+
recs, ours = rescore(recs, q), q["ours"]
|
|
106
|
+
text = render(recs, ours, anonymise=args.anonymise)
|
|
107
|
+
if args.out:
|
|
108
|
+
args.out.write_text(text, encoding="utf-8")
|
|
109
|
+
else:
|
|
110
|
+
print(text, end="")
|
|
111
|
+
return 0
|
|
112
|
+
|
|
113
|
+
q = load(args.questions, dict(args.set), allow_unfilled=getattr(args, "allow_unfilled", False))
|
|
114
|
+
engines = available(only=args.engine, models=dict(args.model))
|
|
115
|
+
if args.cmd == "check":
|
|
116
|
+
print(f"ok: {count(q)} questions in {len(q['groups'])} groups; ours = {q['ours']}")
|
|
117
|
+
missing = [f"{n} ({e.env_key})" for n, e in ENGINES.items()
|
|
118
|
+
if n not in engines and (not args.engine or n in args.engine)]
|
|
119
|
+
if missing:
|
|
120
|
+
print("no key set for: " + ", ".join(missing))
|
|
121
|
+
_plan(q, engines, args.runs, args.max_calls)
|
|
122
|
+
return 0
|
|
123
|
+
|
|
124
|
+
if not engines:
|
|
125
|
+
keys = ", ".join(e.env_key for e in ENGINES.values())
|
|
126
|
+
print(f"aeb: no engine has a key. Set one or more of: {keys}", file=sys.stderr)
|
|
127
|
+
return 2
|
|
128
|
+
|
|
129
|
+
if args.cmd == "run":
|
|
130
|
+
if args.group:
|
|
131
|
+
q["groups"] = {g: v for g, v in q["groups"].items() if g in args.group}
|
|
132
|
+
if args.dry_run:
|
|
133
|
+
q["groups"] = {g: v[:1] for g, v in q["groups"].items()}
|
|
134
|
+
args.runs = 1
|
|
135
|
+
if _plan(q, engines, args.runs, args.max_calls) is None:
|
|
136
|
+
return 2
|
|
137
|
+
answers = args.out / "answers.jsonl"
|
|
138
|
+
recs = run(q, engines, runs=args.runs, out=answers, on_result=_progress, sleep=args.sleep)
|
|
139
|
+
everything = read_jsonl(answers)
|
|
140
|
+
(args.out / "report.md").write_text(render(everything, q["ours"], anonymise=args.anonymise),
|
|
141
|
+
encoding="utf-8")
|
|
142
|
+
cost = sum(r.get("cost_usd") or 0 for r in recs)
|
|
143
|
+
print(f"wrote {answers} (+{len(recs)} answers) and report.md. This run cost about ${cost:.3f}")
|
|
144
|
+
return 1 if recs and all(r["error"] for r in recs) else 0
|
|
145
|
+
|
|
146
|
+
if args.cmd == "noise":
|
|
147
|
+
baseline = read_jsonl(args.baseline)
|
|
148
|
+
picked = pick_questions(q, args.sample)
|
|
149
|
+
groups: dict[str, list[str]] = {}
|
|
150
|
+
for g, question in picked:
|
|
151
|
+
groups.setdefault(g, []).append(question)
|
|
152
|
+
q["groups"] = groups
|
|
153
|
+
if _plan(q, engines, args.runs, args.max_calls) is None:
|
|
154
|
+
return 2
|
|
155
|
+
answers = args.out / "repeat.jsonl"
|
|
156
|
+
run(q, engines, runs=args.runs, out=answers, on_result=_progress, sleep=args.sleep, label="repeat")
|
|
157
|
+
rows = compare(baseline, read_jsonl(answers), tolerance=args.tolerance)
|
|
158
|
+
text = render_compare(rows, args.tolerance)
|
|
159
|
+
(args.out / "noise.md").write_text(text, encoding="utf-8")
|
|
160
|
+
print(text, end="")
|
|
161
|
+
return 0
|
|
162
|
+
except QuestionError as e:
|
|
163
|
+
print(f"aeb: {e}", file=sys.stderr)
|
|
164
|
+
return 2
|
|
165
|
+
except (OSError, ValueError) as e:
|
|
166
|
+
print(f"aeb: {e}", file=sys.stderr)
|
|
167
|
+
return 2
|
|
168
|
+
return 2
|