answer-engine-benchmark 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. answer_engine_benchmark-0.1.0/LICENSE +21 -0
  2. answer_engine_benchmark-0.1.0/PKG-INFO +180 -0
  3. answer_engine_benchmark-0.1.0/README.md +157 -0
  4. answer_engine_benchmark-0.1.0/pyproject.toml +43 -0
  5. answer_engine_benchmark-0.1.0/setup.cfg +4 -0
  6. answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/__init__.py +3 -0
  7. answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/__main__.py +3 -0
  8. answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/cli.py +168 -0
  9. answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/engines.py +240 -0
  10. answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/questions.py +66 -0
  11. answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/runner.py +76 -0
  12. answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/scoring.py +131 -0
  13. answer_engine_benchmark-0.1.0/src/answer_engine_benchmark/summary.py +170 -0
  14. answer_engine_benchmark-0.1.0/src/answer_engine_benchmark.egg-info/PKG-INFO +180 -0
  15. answer_engine_benchmark-0.1.0/src/answer_engine_benchmark.egg-info/SOURCES.txt +21 -0
  16. answer_engine_benchmark-0.1.0/src/answer_engine_benchmark.egg-info/dependency_links.txt +1 -0
  17. answer_engine_benchmark-0.1.0/src/answer_engine_benchmark.egg-info/entry_points.txt +2 -0
  18. answer_engine_benchmark-0.1.0/src/answer_engine_benchmark.egg-info/requires.txt +5 -0
  19. answer_engine_benchmark-0.1.0/src/answer_engine_benchmark.egg-info/top_level.txt +1 -0
  20. answer_engine_benchmark-0.1.0/tests/test_engines.py +112 -0
  21. answer_engine_benchmark-0.1.0/tests/test_questions.py +53 -0
  22. answer_engine_benchmark-0.1.0/tests/test_run_and_report.py +110 -0
  23. answer_engine_benchmark-0.1.0/tests/test_scoring.py +82 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Synapse Research Ltd
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,180 @@
1
+ Metadata-Version: 2.4
2
+ Name: answer-engine-benchmark
3
+ Version: 0.1.0
4
+ Summary: Ask ChatGPT, Gemini, Perplexity and Claude your buyers' questions, several times each, and count who they cite.
5
+ Author: Synapse Research Ltd
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://synapsereality.io/open-source/answer-engine-benchmark/
8
+ Project-URL: Documentation, https://synapsereality.io/open-source/answer-engine-benchmark/
9
+ Project-URL: Repository, https://github.com/synapsereality/answer-engine-benchmark
10
+ Project-URL: Issues, https://github.com/synapsereality/answer-engine-benchmark/issues
11
+ Keywords: aeo,geo,answer-engine-optimization,llm,citations,benchmark,seo
12
+ Classifier: Environment :: Console
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Topic :: Internet :: WWW/HTTP :: Indexing/Search
15
+ Requires-Python: >=3.10
16
+ Description-Content-Type: text/markdown
17
+ License-File: LICENSE
18
+ Requires-Dist: requests>=2.28
19
+ Requires-Dist: PyYAML>=6
20
+ Provides-Extra: test
21
+ Requires-Dist: pytest>=8; extra == "test"
22
+ Dynamic: license-file
23
+
24
+ # answer-engine-benchmark
25
+
26
+ Asks ChatGPT, Gemini, Perplexity and Claude the questions your buyers ask, with
27
+ web search on, several times each. Then it counts how often each answer cites
28
+ your site, what it cites instead, and what it says about you. Every answer is
29
+ kept in a JSONL file, so you can check any number in the report against the
30
+ text behind it.
31
+
32
+ Docs: https://synapsereality.io/open-source/answer-engine-benchmark/
33
+
34
+ ```bash
35
+ pip install answer-engine-benchmark
36
+ export OPENAI_API_KEY=... GEMINI_API_KEY=... PERPLEXITY_API_KEY=... ANTHROPIC_API_KEY=...
37
+ aeb run questions/template.yaml --out runs/first \
38
+ --set brand="Acme Analytics" --set domain=acme.example \
39
+ --set category="invoice software" --set audience="small accounting firms"
40
+ ```
41
+
42
+ ## Why repeat every question
43
+
44
+ The same question can come back with different sources a minute later. One
45
+ answer is one sample. The default is 3 runs per question per engine, and every
46
+ rate in the report is over those runs. `aeb noise` re-asks a sample of questions
47
+ later and tells you whether a change you see is bigger than the noise.
48
+
49
+ ## The question set
50
+
51
+ `questions/template.yaml` holds 24 questions in four groups:
52
+
53
+ | group | the buyer is | your name in the question |
54
+ |---|---|---|
55
+ | discovery | looking for a provider | no |
56
+ | comparison | learning how to choose | no |
57
+ | brand | asking about you | yes |
58
+ | trust | checking you are safe to buy from | yes |
59
+
60
+ Discovery and comparison tell you whether engines find you when nobody asked
61
+ for you. Brand and trust tell you what they say when somebody does.
62
+
63
+ Fill the four `vars` in the file, or pass them with `--set`. A run won't start
64
+ while any of them still holds its example value. Add your own questions under
65
+ any group, or new groups. Write them the way a buyer types, and never make a
66
+ competitor the subject of a question.
67
+
68
+ The file also sets:
69
+
70
+ - `ours`: domains that count as your citation. `acme.example` covers its
71
+ subdomains. `github.com/acme` covers only paths under it, so a citation of
72
+ someone else's GitHub repo is not yours.
73
+ - `brand_terms`: names that count as a mention.
74
+ - `stale_markers`: phrases that describe you wrongly or out of date, such as an
75
+ old product or a wrong founding year. One is flagged only within 200
76
+ characters of a brand term, so another company "founded in 2015" doesn't count
77
+ against you.
78
+
79
+ ## How an answer is scored
80
+
81
+ **Cited** means one of the answer's URLs is yours. The URLs are the sources the
82
+ engine attached plus any link in the text. **Mentioned** means a brand term
83
+ appears in the text. An answer can mention you without citing you, and that
84
+ difference is worth watching. **Position** is the rank of your first source
85
+ among the distinct sites the answer cited.
86
+
87
+ Failed calls are recorded as errors and left out of every rate. A rate limit
88
+ never shows up as "not cited".
89
+
90
+ Gemini returns its sources as Google redirect links. They are resolved to the
91
+ real pages before scoring, with a cache next to the results. Otherwise every
92
+ Gemini answer would look like it cited Google.
93
+
94
+ ## Engines
95
+
96
+ | engine | key | default model | how it searches |
97
+ |---|---|---|---|
98
+ | `openai` | `OPENAI_API_KEY` | `gpt-5-mini` | Responses API `web_search` tool |
99
+ | `gemini` | `GEMINI_API_KEY` | `gemini-3.5-flash` | Google Search grounding |
100
+ | `perplexity` | `PERPLEXITY_API_KEY` | `perplexity/sonar` | Responses API `web_search` tool |
101
+ | `claude` | `ANTHROPIC_API_KEY` | `claude-sonnet-5` | Messages API web search tool |
102
+
103
+ Keys are read from the environment only. They are sent in request headers and
104
+ removed from any error message before it is written, so they never reach the
105
+ results file. An engine with no key is skipped.
106
+
107
+ Change a model with `--model claude=claude-opus-5`. Pick the models your
108
+ buyers actually use in the apps, or say in the report which ones you used.
109
+
110
+ Perplexity's API doesn't search unless you ask it to. Without the search tool
111
+ it still answers, fluently, with no citations. This adapter always turns search
112
+ on.
113
+
114
+ ## Commands
115
+
116
+ ```bash
117
+ aeb check QUESTIONS [--set ...] # validate, show which keys are set and how many calls a run needs
118
+ aeb run QUESTIONS --out DIR [--set ...] # ask everything, write DIR/answers.jsonl and DIR/report.md
119
+ aeb run QUESTIONS --out DIR --dry-run # first question of each group, once: a cheap smoke test
120
+ aeb report DIR/answers.jsonl [QUESTIONS] # rebuild the report, re-scoring if you changed the question file
121
+ aeb noise QUESTIONS --baseline DIR/answers.jsonl --out DIR2 --sample 5
122
+ ```
123
+
124
+ `--anonymise` on `run` and `report` replaces every domain that isn't yours with
125
+ "Source A", "Source B" and so on. Use it before you share a report outside your
126
+ company. `--max-calls` (default 500) stops a run that would make more calls
127
+ than you expected.
128
+
129
+ `answers.jsonl` gets one line per answer, appended as it arrives, so a stopped
130
+ run keeps what it already paid for. Each line holds the question, engine, model,
131
+ run number, the full answer text, every URL, token usage, cost, and the score
132
+ fields.
133
+
134
+ ## What it costs
135
+
136
+ A full run of the template is 24 questions x 4 engines x 3 runs = 288 calls.
137
+ Our own run of 360 answers cost $1.39 in API fees without Claude (details in
138
+ `examples/`). Adding Claude costs more, because web search on the Claude API is
139
+ billed per search on top of tokens. Each
140
+ row carries its cost. Perplexity reports the real cost. The others are
141
+ estimated from list prices in `engines.py`, so check them against your bills.
142
+
143
+ ## Measure as a stranger
144
+
145
+ Run the benchmark from an account and machine that has never been told who you
146
+ are. We learned this from a run through a coding assistant's CLI instead of an
147
+ API. The CLI passed the operator's own git identity to the model, and many brand
148
+ answers then told the reader the company was probably their own. The API
149
+ adapters here send only the question and a one-line instruction to cite sources.
150
+
151
+ ## Example
152
+
153
+ `examples/synapse-launch-week-2026-09.md` is a real run on one company's own
154
+ site, with the published figures only. 105 of 156 answers that named the
155
+ company cited its site. Of the 204 that didn't name it, 1 did. It is an example
156
+ of the output, not part of the question set.
157
+
158
+ ## Install from source
159
+
160
+ ```bash
161
+ git clone https://github.com/synapsereality/answer-engine-benchmark
162
+ cd answer-engine-benchmark
163
+ pip install .
164
+ aeb --help
165
+ ```
166
+
167
+ Python 3.10 or later. Needs `requests` and `PyYAML`.
168
+
169
+ ## Tests
170
+
171
+ ```bash
172
+ pip install -e ".[test]"
173
+ pytest
174
+ ```
175
+
176
+ 33 tests. They mock every API, so they spend nothing and need no keys.
177
+
178
+ ## Licence
179
+
180
+ MIT. Made by [Synapse](https://synapsereality.io/open-source/answer-engine-benchmark/).
@@ -0,0 +1,157 @@
1
+ # answer-engine-benchmark
2
+
3
+ Asks ChatGPT, Gemini, Perplexity and Claude the questions your buyers ask, with
4
+ web search on, several times each. Then it counts how often each answer cites
5
+ your site, what it cites instead, and what it says about you. Every answer is
6
+ kept in a JSONL file, so you can check any number in the report against the
7
+ text behind it.
8
+
9
+ Docs: https://synapsereality.io/open-source/answer-engine-benchmark/
10
+
11
+ ```bash
12
+ pip install answer-engine-benchmark
13
+ export OPENAI_API_KEY=... GEMINI_API_KEY=... PERPLEXITY_API_KEY=... ANTHROPIC_API_KEY=...
14
+ aeb run questions/template.yaml --out runs/first \
15
+ --set brand="Acme Analytics" --set domain=acme.example \
16
+ --set category="invoice software" --set audience="small accounting firms"
17
+ ```
18
+
19
+ ## Why repeat every question
20
+
21
+ The same question can come back with different sources a minute later. One
22
+ answer is one sample. The default is 3 runs per question per engine, and every
23
+ rate in the report is over those runs. `aeb noise` re-asks a sample of questions
24
+ later and tells you whether a change you see is bigger than the noise.
25
+
26
+ ## The question set
27
+
28
+ `questions/template.yaml` holds 24 questions in four groups:
29
+
30
+ | group | the buyer is | your name in the question |
31
+ |---|---|---|
32
+ | discovery | looking for a provider | no |
33
+ | comparison | learning how to choose | no |
34
+ | brand | asking about you | yes |
35
+ | trust | checking you are safe to buy from | yes |
36
+
37
+ Discovery and comparison tell you whether engines find you when nobody asked
38
+ for you. Brand and trust tell you what they say when somebody does.
39
+
40
+ Fill the four `vars` in the file, or pass them with `--set`. A run won't start
41
+ while any of them still holds its example value. Add your own questions under
42
+ any group, or new groups. Write them the way a buyer types, and never make a
43
+ competitor the subject of a question.
44
+
45
+ The file also sets:
46
+
47
+ - `ours`: domains that count as your citation. `acme.example` covers its
48
+ subdomains. `github.com/acme` covers only paths under it, so a citation of
49
+ someone else's GitHub repo is not yours.
50
+ - `brand_terms`: names that count as a mention.
51
+ - `stale_markers`: phrases that describe you wrongly or out of date, such as an
52
+ old product or a wrong founding year. One is flagged only within 200
53
+ characters of a brand term, so another company "founded in 2015" doesn't count
54
+ against you.
55
+
56
+ ## How an answer is scored
57
+
58
+ **Cited** means one of the answer's URLs is yours. The URLs are the sources the
59
+ engine attached plus any link in the text. **Mentioned** means a brand term
60
+ appears in the text. An answer can mention you without citing you, and that
61
+ difference is worth watching. **Position** is the rank of your first source
62
+ among the distinct sites the answer cited.
63
+
64
+ Failed calls are recorded as errors and left out of every rate. A rate limit
65
+ never shows up as "not cited".
66
+
67
+ Gemini returns its sources as Google redirect links. They are resolved to the
68
+ real pages before scoring, with a cache next to the results. Otherwise every
69
+ Gemini answer would look like it cited Google.
70
+
71
+ ## Engines
72
+
73
+ | engine | key | default model | how it searches |
74
+ |---|---|---|---|
75
+ | `openai` | `OPENAI_API_KEY` | `gpt-5-mini` | Responses API `web_search` tool |
76
+ | `gemini` | `GEMINI_API_KEY` | `gemini-3.5-flash` | Google Search grounding |
77
+ | `perplexity` | `PERPLEXITY_API_KEY` | `perplexity/sonar` | Responses API `web_search` tool |
78
+ | `claude` | `ANTHROPIC_API_KEY` | `claude-sonnet-5` | Messages API web search tool |
79
+
80
+ Keys are read from the environment only. They are sent in request headers and
81
+ removed from any error message before it is written, so they never reach the
82
+ results file. An engine with no key is skipped.
83
+
84
+ Change a model with `--model claude=claude-opus-5`. Pick the models your
85
+ buyers actually use in the apps, or say in the report which ones you used.
86
+
87
+ Perplexity's API doesn't search unless you ask it to. Without the search tool
88
+ it still answers, fluently, with no citations. This adapter always turns search
89
+ on.
90
+
91
+ ## Commands
92
+
93
+ ```bash
94
+ aeb check QUESTIONS [--set ...] # validate, show which keys are set and how many calls a run needs
95
+ aeb run QUESTIONS --out DIR [--set ...] # ask everything, write DIR/answers.jsonl and DIR/report.md
96
+ aeb run QUESTIONS --out DIR --dry-run # first question of each group, once: a cheap smoke test
97
+ aeb report DIR/answers.jsonl [QUESTIONS] # rebuild the report, re-scoring if you changed the question file
98
+ aeb noise QUESTIONS --baseline DIR/answers.jsonl --out DIR2 --sample 5
99
+ ```
100
+
101
+ `--anonymise` on `run` and `report` replaces every domain that isn't yours with
102
+ "Source A", "Source B" and so on. Use it before you share a report outside your
103
+ company. `--max-calls` (default 500) stops a run that would make more calls
104
+ than you expected.
105
+
106
+ `answers.jsonl` gets one line per answer, appended as it arrives, so a stopped
107
+ run keeps what it already paid for. Each line holds the question, engine, model,
108
+ run number, the full answer text, every URL, token usage, cost, and the score
109
+ fields.
110
+
111
+ ## What it costs
112
+
113
+ A full run of the template is 24 questions x 4 engines x 3 runs = 288 calls.
114
+ Our own run of 360 answers cost $1.39 in API fees without Claude (details in
115
+ `examples/`). Adding Claude costs more, because web search on the Claude API is
116
+ billed per search on top of tokens. Each
117
+ row carries its cost. Perplexity reports the real cost. The others are
118
+ estimated from list prices in `engines.py`, so check them against your bills.
119
+
120
+ ## Measure as a stranger
121
+
122
+ Run the benchmark from an account and machine that has never been told who you
123
+ are. We learned this from a run through a coding assistant's CLI instead of an
124
+ API. The CLI passed the operator's own git identity to the model, and many brand
125
+ answers then told the reader the company was probably their own. The API
126
+ adapters here send only the question and a one-line instruction to cite sources.
127
+
128
+ ## Example
129
+
130
+ `examples/synapse-launch-week-2026-09.md` is a real run on one company's own
131
+ site, with the published figures only. 105 of 156 answers that named the
132
+ company cited its site. Of the 204 that didn't name it, 1 did. It is an example
133
+ of the output, not part of the question set.
134
+
135
+ ## Install from source
136
+
137
+ ```bash
138
+ git clone https://github.com/synapsereality/answer-engine-benchmark
139
+ cd answer-engine-benchmark
140
+ pip install .
141
+ aeb --help
142
+ ```
143
+
144
+ Python 3.10 or later. Needs `requests` and `PyYAML`.
145
+
146
+ ## Tests
147
+
148
+ ```bash
149
+ pip install -e ".[test]"
150
+ pytest
151
+ ```
152
+
153
+ 33 tests. They mock every API, so they spend nothing and need no keys.
154
+
155
+ ## Licence
156
+
157
+ MIT. Made by [Synapse](https://synapsereality.io/open-source/answer-engine-benchmark/).
@@ -0,0 +1,43 @@
1
+ [build-system]
2
+ requires = ["setuptools>=69"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "answer-engine-benchmark"
7
+ version = "0.1.0"
8
+ description = "Ask ChatGPT, Gemini, Perplexity and Claude your buyers' questions, several times each, and count who they cite."
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ license-files = ["LICENSE"]
12
+ requires-python = ">=3.10"
13
+ authors = [{ name = "Synapse Research Ltd" }]
14
+ keywords = ["aeo", "geo", "answer-engine-optimization", "llm", "citations", "benchmark", "seo"]
15
+ classifiers = [
16
+ "Environment :: Console",
17
+ "Programming Language :: Python :: 3",
18
+ "Topic :: Internet :: WWW/HTTP :: Indexing/Search",
19
+ ]
20
+ dependencies = ["requests>=2.28", "PyYAML>=6"]
21
+
22
+ [project.optional-dependencies]
23
+ test = ["pytest>=8"]
24
+
25
+ [project.urls]
26
+ Homepage = "https://synapsereality.io/open-source/answer-engine-benchmark/"
27
+ Documentation = "https://synapsereality.io/open-source/answer-engine-benchmark/"
28
+ Repository = "https://github.com/synapsereality/answer-engine-benchmark"
29
+ Issues = "https://github.com/synapsereality/answer-engine-benchmark/issues"
30
+
31
+ [project.scripts]
32
+ aeb = "answer_engine_benchmark.cli:main"
33
+
34
+ [tool.setuptools.packages.find]
35
+ where = ["src"]
36
+
37
+ [tool.pytest.ini_options]
38
+ pythonpath = ["src"]
39
+ testpaths = ["tests"]
40
+
41
+ [tool.ruff]
42
+ line-length = 120
43
+ target-version = "py310"
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,3 @@
1
+ """answer-engine-benchmark: ask AI answer engines your buyers' questions and count who they cite."""
2
+
3
+ __version__ = "0.1.0"
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ raise SystemExit(main())
@@ -0,0 +1,168 @@
1
+ """aeb: ask AI answer engines your buyers' questions and count who they cite.
2
+
3
+ Exit codes: 0 done, 1 a run finished but every call failed, 2 bad input.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import argparse
9
+ import sys
10
+ from pathlib import Path
11
+
12
+ from . import __version__
13
+ from .engines import ENGINES, available
14
+ from .questions import QuestionError, count, load
15
+ from .runner import read_jsonl, rescore, run
16
+ from .summary import compare, pick_questions, render, render_compare
17
+
18
+ DOCS = "https://synapsereality.io/open-source/answer-engine-benchmark/"
19
+
20
+
21
+ def _kv(raw: str) -> tuple[str, str]:
22
+ k, sep, v = raw.partition("=")
23
+ if not sep or not k.strip():
24
+ raise argparse.ArgumentTypeError(f"expected name=value, got {raw!r}")
25
+ return k.strip(), v.strip()
26
+
27
+
28
+ def _common(p: argparse.ArgumentParser) -> None:
29
+ p.add_argument("questions", help="question file (YAML)")
30
+ p.add_argument("--set", action="append", type=_kv, default=[], metavar="NAME=VALUE",
31
+ help="fill a {placeholder}, e.g. --set brand='Acme Ltd' (repeatable)")
32
+
33
+
34
+ def _engine_flags(p: argparse.ArgumentParser) -> None:
35
+ p.add_argument("--engine", action="append", choices=sorted(ENGINES), help="only these engines (repeatable)")
36
+ p.add_argument("--model", action="append", type=_kv, default=[], metavar="ENGINE=MODEL",
37
+ help="override a model, e.g. --model claude=claude-opus-5")
38
+ p.add_argument("--runs", type=int, default=3, help="times to ask each question on each engine (default 3)")
39
+ p.add_argument("--max-calls", type=int, default=500,
40
+ help="refuse to start a run that needs more API calls than this (default 500)")
41
+ p.add_argument("--sleep", type=float, default=1.0, help="seconds between calls (default 1)")
42
+
43
+
44
+ def build_parser() -> argparse.ArgumentParser:
45
+ ap = argparse.ArgumentParser(prog="aeb", description=__doc__, epilog=f"Docs: {DOCS}",
46
+ formatter_class=argparse.RawDescriptionHelpFormatter)
47
+ ap.add_argument("--version", action="version", version=f"aeb {__version__}")
48
+ sub = ap.add_subparsers(dest="cmd", required=True)
49
+
50
+ p = sub.add_parser("check", help="validate a question file and show what a run would cost in calls")
51
+ _common(p)
52
+ _engine_flags(p)
53
+ p.add_argument("--allow-unfilled", action="store_true", help="accept the template's example values")
54
+
55
+ p = sub.add_parser("run", help="ask every question on every engine and write answers.jsonl + report.md")
56
+ _common(p)
57
+ _engine_flags(p)
58
+ p.add_argument("--out", required=True, type=Path, help="output folder")
59
+ p.add_argument("--group", action="append", help="only these question groups (repeatable)")
60
+ p.add_argument("--dry-run", action="store_true", help="first question of each group, 1 run: a cheap smoke test")
61
+ p.add_argument("--anonymise", action="store_true", help="label other domains Source A, B, ... in report.md")
62
+
63
+ p = sub.add_parser("report", help="rebuild report.md from answers.jsonl (spends nothing)")
64
+ p.add_argument("answers", type=Path, help="answers.jsonl")
65
+ p.add_argument("questions", nargs="?", help="question file: re-score with its ours/brand_terms first")
66
+ p.add_argument("--set", action="append", type=_kv, default=[], metavar="NAME=VALUE")
67
+ p.add_argument("--anonymise", action="store_true")
68
+ p.add_argument("--out", type=Path, help="write here instead of stdout")
69
+
70
+ p = sub.add_parser("noise", help="re-ask a sample of questions and compare with the baseline")
71
+ _common(p)
72
+ _engine_flags(p)
73
+ p.add_argument("--baseline", required=True, type=Path, help="answers.jsonl from the first run")
74
+ p.add_argument("--out", required=True, type=Path, help="output folder for the repeat run")
75
+ p.add_argument("--sample", type=int, default=5, help="questions to re-ask, spread across groups (default 5)")
76
+ p.add_argument("--tolerance", type=int, default=1, help="allowed difference in citations (default 1)")
77
+ return ap
78
+
79
+
80
+ def _plan(q: dict, engines: dict, runs: int, max_calls: int) -> int | None:
81
+ calls = count(q) * len(engines) * runs
82
+ names = ", ".join(f"{n} ({e.model})" for n, e in engines.items()) or "none"
83
+ print(f"{count(q)} questions x {len(engines)} engines x {runs} runs = {calls} calls. Engines: {names}")
84
+ if calls > max_calls:
85
+ print(f"aeb: {calls} calls is over --max-calls {max_calls}. Raise it if you mean it.", file=sys.stderr)
86
+ return None
87
+ return calls
88
+
89
+
90
+ def _progress(rec: dict) -> None:
91
+ flag = "ERR " if rec["error"] else "CITE" if rec["cited"] else "ment" if rec["mentioned"] else " - "
92
+ cost = rec["cost_usd"] or 0
93
+ print(f"{flag} {rec['engine']:<10} {rec['group']:<11} pos={rec['position'] or '-':<3} ${cost:.4f} "
94
+ f"{rec['question'][:70]} {rec['error'][:80]}", flush=True)
95
+
96
+
97
+ def main(argv: list[str] | None = None) -> int:
98
+ args = build_parser().parse_args(argv)
99
+ try:
100
+ if args.cmd == "report":
101
+ recs = read_jsonl(args.answers)
102
+ ours = []
103
+ if args.questions:
104
+ q = load(args.questions, dict(args.set), allow_unfilled=True)
105
+ recs, ours = rescore(recs, q), q["ours"]
106
+ text = render(recs, ours, anonymise=args.anonymise)
107
+ if args.out:
108
+ args.out.write_text(text, encoding="utf-8")
109
+ else:
110
+ print(text, end="")
111
+ return 0
112
+
113
+ q = load(args.questions, dict(args.set), allow_unfilled=getattr(args, "allow_unfilled", False))
114
+ engines = available(only=args.engine, models=dict(args.model))
115
+ if args.cmd == "check":
116
+ print(f"ok: {count(q)} questions in {len(q['groups'])} groups; ours = {q['ours']}")
117
+ missing = [f"{n} ({e.env_key})" for n, e in ENGINES.items()
118
+ if n not in engines and (not args.engine or n in args.engine)]
119
+ if missing:
120
+ print("no key set for: " + ", ".join(missing))
121
+ _plan(q, engines, args.runs, args.max_calls)
122
+ return 0
123
+
124
+ if not engines:
125
+ keys = ", ".join(e.env_key for e in ENGINES.values())
126
+ print(f"aeb: no engine has a key. Set one or more of: {keys}", file=sys.stderr)
127
+ return 2
128
+
129
+ if args.cmd == "run":
130
+ if args.group:
131
+ q["groups"] = {g: v for g, v in q["groups"].items() if g in args.group}
132
+ if args.dry_run:
133
+ q["groups"] = {g: v[:1] for g, v in q["groups"].items()}
134
+ args.runs = 1
135
+ if _plan(q, engines, args.runs, args.max_calls) is None:
136
+ return 2
137
+ answers = args.out / "answers.jsonl"
138
+ recs = run(q, engines, runs=args.runs, out=answers, on_result=_progress, sleep=args.sleep)
139
+ everything = read_jsonl(answers)
140
+ (args.out / "report.md").write_text(render(everything, q["ours"], anonymise=args.anonymise),
141
+ encoding="utf-8")
142
+ cost = sum(r.get("cost_usd") or 0 for r in recs)
143
+ print(f"wrote {answers} (+{len(recs)} answers) and report.md. This run cost about ${cost:.3f}")
144
+ return 1 if recs and all(r["error"] for r in recs) else 0
145
+
146
+ if args.cmd == "noise":
147
+ baseline = read_jsonl(args.baseline)
148
+ picked = pick_questions(q, args.sample)
149
+ groups: dict[str, list[str]] = {}
150
+ for g, question in picked:
151
+ groups.setdefault(g, []).append(question)
152
+ q["groups"] = groups
153
+ if _plan(q, engines, args.runs, args.max_calls) is None:
154
+ return 2
155
+ answers = args.out / "repeat.jsonl"
156
+ run(q, engines, runs=args.runs, out=answers, on_result=_progress, sleep=args.sleep, label="repeat")
157
+ rows = compare(baseline, read_jsonl(answers), tolerance=args.tolerance)
158
+ text = render_compare(rows, args.tolerance)
159
+ (args.out / "noise.md").write_text(text, encoding="utf-8")
160
+ print(text, end="")
161
+ return 0
162
+ except QuestionError as e:
163
+ print(f"aeb: {e}", file=sys.stderr)
164
+ return 2
165
+ except (OSError, ValueError) as e:
166
+ print(f"aeb: {e}", file=sys.stderr)
167
+ return 2
168
+ return 2